@miraland-labs/conduit-bridge 0.16.108 → 0.16.110

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/execution.js CHANGED
@@ -25,7 +25,7 @@ import { ensureLandCommit } from "./ensure-land-commit.js";
25
25
  import { AGENT_NO_LAND_COMMIT_PREFIX, AgentNoLandCommitError, agentNoLandCommitMessage, requiresLandCommit } from "./land-contract.js";
26
26
  import { agentClaimsRepositoryWork, claimsVsGitMismatch, isLandTreeEmpty, reportClaimsRepositoryWork, } from "./land-git-reality.js";
27
27
  const execFileAsync = promisify(execFile);
28
- import { captureVerificationFailure, detectGateRanNothing, ensureTestEvidence, needsTestEvidence, runBoundedVerificationCommand, verificationEvidenceCommands, VerificationGateRanNothingError } from "./ensure-test-evidence.js";
28
+ import { boundedTail, captureVerificationFailure, detectGateRanNothing, ensureTestEvidence, needsTestEvidence, runBoundedVerificationCommand, verificationEvidenceCommands, VerificationGateRanNothingError } from "./ensure-test-evidence.js";
29
29
  import { ensureChangeEvidence } from "./ensure-change-evidence.js";
30
30
  import { ensureNormativeEvidence } from "./ensure-normative-evidence.js";
31
31
  import { changedPathsSince, compareWorktreeState, diffStatSince, worktreeStateFingerprint, worktreeWrittenPaths } from "./git-witness.js";
@@ -2872,8 +2872,44 @@ export async function verificationScopeDryRun(input) {
2872
2872
  : written.filter((path) => !input.changeScope.some((scope) => pathMatchesScope(path, scope)));
2873
2873
  return { ranNothing, redBase, outsideScope };
2874
2874
  }
2875
+ /** Prompt and failure lines keep a short tail; the investigation brief caps each failure at 1,000. */
2876
+ const DELIVERY_VERIFICATION_GATE_TAIL_CHARS = 800;
2877
+ /**
2878
+ * Run the T191 gate set in the verification worktree at the delivered commit: declared
2879
+ * `.conduit/verification` ∪ the commands the criteria name, else the discovered set. One at a
2880
+ * time, same bounded runner as the pre-agent dry run. Writes the gates made are discarded so the
2881
+ * agent still reads the delivered tree.
2882
+ */
2883
+ export async function runDeliveryVerificationGates(input) {
2884
+ const commands = verificationEvidenceCommands([...input.verificationCommands], input.declaredVerificationCommands ?? [], input.acceptance ?? []);
2885
+ const run = input.runCommand ?? runBoundedVerificationCommand;
2886
+ const results = [];
2887
+ for (const command of commands) {
2888
+ try {
2889
+ const result = await run(command, input.workspace);
2890
+ results.push({
2891
+ command,
2892
+ exitCode: result.code,
2893
+ stdoutTail: boundedTail(result.stdout, DELIVERY_VERIFICATION_GATE_TAIL_CHARS),
2894
+ stderrTail: boundedTail(result.stderr, DELIVERY_VERIFICATION_GATE_TAIL_CHARS),
2895
+ });
2896
+ }
2897
+ catch (error) {
2898
+ // A gate that could not run is a failed gate: record it so settlement still names it.
2899
+ const message = error instanceof Error ? error.message : String(error);
2900
+ results.push({
2901
+ command,
2902
+ exitCode: -1,
2903
+ stdoutTail: "",
2904
+ stderrTail: boundedTail(message, DELIVERY_VERIFICATION_GATE_TAIL_CHARS),
2905
+ });
2906
+ }
2907
+ }
2908
+ await discardWorktreeWrites(input.workspace);
2909
+ return results;
2910
+ }
2875
2911
  /** Return the worktree to its commit: tracked files restored, new files removed, ignored files kept. */
2876
- async function discardWorktreeWrites(worktree) {
2912
+ export async function discardWorktreeWrites(worktree) {
2877
2913
  const options = { timeout: 60_000, maxBuffer: 2_000_000 };
2878
2914
  await execFileAsync("git", ["-C", worktree, "checkout", "--", "."], options).catch(() => undefined);
2879
2915
  await execFileAsync("git", ["-C", worktree, "clean", "-fd"], options).catch(() => undefined);
@@ -75,6 +75,32 @@ export const failureSignalSchema = z.object({
75
75
  */
76
76
  model: z.string().max(200).optional(),
77
77
  });
78
+ /**
79
+ * The class–disposition pairs a classified failure may carry. Disposition is a function of the
80
+ * cause, not of class: the six canonical pairs plus `contract`/`hold` (the work package is the
81
+ * defect; a rework would spend repair on code that is not at fault) and `environment`/`retry`
82
+ * (the computer hiccupped; the next lane runs it again).
83
+ *
84
+ * Declared once so the envelope schema and both twins share one list. `ClassifiedFailure` stays a
85
+ * free pair; this list is the boundary check.
86
+ */
87
+ export const ADMITTED_FAILURE_PAIRS = [
88
+ ["transient", "retry"],
89
+ ["environment", "hold"],
90
+ ["environment", "retry"],
91
+ ["contract", "rework"],
92
+ ["contract", "hold"],
93
+ ["quality", "rework"],
94
+ ["authority", "decision"],
95
+ ["platform", "stop"],
96
+ ];
97
+ /**
98
+ * Whether this class and disposition are one of the admitted pairs. The envelope schema calls it
99
+ * so a supplied envelope cannot carry a pairing the builders never emit.
100
+ */
101
+ export function isAdmittedFailurePair(failureClass, disposition) {
102
+ return ADMITTED_FAILURE_PAIRS.some(([admittedClass, admittedDisposition]) => admittedClass === failureClass && admittedDisposition === disposition);
103
+ }
78
104
  /**
79
105
  * The causes Bridge names itself — before the agent starts, or while finishing its report — as one
80
106
  * table keyed by code, so Bridge sends the envelope and the control plane validates it (K3).
@@ -6,19 +6,22 @@
6
6
  * the worktree it reads is deleted whichever way the run ends.
7
7
  */
8
8
  import { z } from "zod";
9
- import { boundedTail } from "./ensure-test-evidence.js";
9
+ import { boundedTail, runBoundedVerificationCommand } from "./ensure-test-evidence.js";
10
10
  import { createAttemptWorktree, removeAttemptWorktree } from "./attempt-worktree.js";
11
11
  import { buildWorkspaceBrief, ensureCommitAvailable, normalizeRepositoryUrl } from "./brief.js";
12
12
  import { headCommitOrNull } from "./git.js";
13
13
  import { redactSecrets } from "./config.js";
14
- import { learnDriverFuel, resolveAssignmentModel } from "./execution.js";
14
+ import { learnDriverFuel, resolveAssignmentModel, changedPathsSince, runDeliveryVerificationGates, discardWorktreeWrites } from "./execution.js";
15
15
  import { DRIVERS, extractAgentReportJsonText, parseJsonObjectCandidate } from "./driver.js";
16
16
  import { pickDriverForClaim, resolveDriverFuel, supportsReadOnlyDiagnosis } from "./drivers.js";
17
- import { changedPathsSince } from "./execution.js";
18
17
  import { diffStatSince } from "./git-witness.js";
19
18
  import { switchManagedWorkspace } from "./on-shift-apply.js";
20
19
  import { execFile } from "node:child_process";
20
+ import { mkdtemp, rm, writeFile } from "node:fs/promises";
21
+ import { tmpdir } from "node:os";
22
+ import { join } from "node:path";
21
23
  import { promisify } from "node:util";
24
+ import { isRunnableVerificationCommand } from "./execution-class.js";
22
25
  const execFileAsync = promisify(execFile);
23
26
  const budgetSchema = z.object({
24
27
  max_files: z.number().int().min(1).max(24).optional().default(8),
@@ -36,6 +39,10 @@ export const investigationAssignmentSchema = z.object({
36
39
  /** T122: present on a delivery verification — the contract base the delivered head was built on. */
37
40
  base_commit: z.string().nullable().optional().default(null),
38
41
  source_attempt_id: z.string().uuid().nullable().optional().default(null),
42
+ /** Delivery verification: the task's acceptance strings. Absent or [] on a plain investigation. */
43
+ acceptance: z.array(z.string().trim().min(1).max(4_000)).max(20).optional().default([]),
44
+ /** Delivery verification: the brief's `[sN]` / `[kN]` lines. Absent or [] on a plain investigation. */
45
+ brief_keys: z.array(z.string().trim().min(1).max(4_000)).max(40).optional().default([]),
39
46
  });
40
47
  /**
41
48
  * Enforcement is a vendor's claim; verification is ours. Both run on every investigation — this only
@@ -66,9 +73,24 @@ const investigationBriefSchema = z.object({
66
73
  files_read: z.number().int().min(0).max(24).default(0),
67
74
  bytes_read: z.number().int().min(0).max(512 * 1024).default(0),
68
75
  commands_run: z.array(z.string().trim().min(1).max(200)).max(10).default([]),
76
+ /**
77
+ * K12: probe source the agent wrote. Bridge executes these; it never copies claimed results into
78
+ * step_witness. Unknown keys such as a claimed exit_code are stripped.
79
+ */
80
+ probes: z.array(z.object({
81
+ key: z.string().trim().min(1).max(40),
82
+ test: z.string().max(300).optional(),
83
+ lang: z.enum(["ts", "py", "sh"]).optional(),
84
+ source: z.string().max(8_000).optional(),
85
+ checks: z.string().max(500).optional().default(""),
86
+ })).max(20).default([]),
69
87
  });
70
88
  /** T122: a delivery verification is an investigation whose question carries this header. */
71
89
  export const DELIVERY_VERIFICATION_HEADER = "DELIVERY VERIFICATION";
90
+ /** Owner-verbatim: the lane writes probe source; Bridge executes. Shown only when BRIEF KEYS are present. */
91
+ export const DELIVERY_VERIFICATION_PROBE_INSTRUCTION = 'For every BRIEF KEY that states behaviour, add one entry to probes: either {"key":"[sN]","test":"<registered command that asserts it>","checks":"<one line: what it proves>"} or {"key":"[sN]","lang":"ts|py|sh","source":"<a short script that exits 0 only when the key holds at this commit and non-zero otherwise; it may import repository modules by absolute path from the working directory>","checks":"<one line>"}. You do not run probes; Bridge runs them after you finish. Keys that state no behaviour (wording, scope, docs) get no probe.';
92
+ const BRIEF_JSON_EXAMPLE = '{"summary":"one paragraph: what the change does and whether it meets the criteria","findings":["grounded fact with file/symbol"],"verification":["command — result"],"limitations":["what you could not settle"],"criteria":[{"criterion":"c1","assessment":"supported|unsupported|unclear","reason":"what you saw"}],"command_failures":["command that exited non-zero — the decisive output line"],"files_read":0,"bytes_read":0,"commands_run":["command you actually ran"]}';
93
+ const BRIEF_JSON_EXAMPLE_WITH_PROBES = '{"summary":"one paragraph: what the change does and whether it meets the criteria","findings":["grounded fact with file/symbol"],"verification":["command — result"],"limitations":["what you could not settle"],"criteria":[{"criterion":"c1","assessment":"supported|unsupported|unclear","reason":"what you saw"}],"command_failures":["command that exited non-zero — the decisive output line"],"probes":[],"files_read":0,"bytes_read":0,"commands_run":["command you actually ran"]}';
72
94
  export function isDeliveryVerification(assignment) {
73
95
  return assignment.question.trimStart().startsWith(DELIVERY_VERIFICATION_HEADER);
74
96
  }
@@ -95,7 +117,7 @@ export function buildInvestigationPrompt(assignment, commit, context) {
95
117
  }
96
118
  /**
97
119
  * T122: the Conductor's own check of a delivery, done the way a careful reviewer does it by hand —
98
- * run the registered commands at the delivered head, read the diff, judge every criterion. The
120
+ * read the gates Bridge already ran at the delivered head, read the diff, judge every criterion. The
99
121
  * verdict per criterion is the product; prose without verdicts is an unusable brief.
100
122
  */
101
123
  export function buildDeliveryVerificationPrompt(assignment, commit, context) {
@@ -109,27 +131,70 @@ export function buildDeliveryVerificationPrompt(assignment, commit, context) {
109
131
  const commands = context.verificationCommands.length
110
132
  ? `REGISTERED COMMANDS (the only commands your shell accepts)\n${context.verificationCommands.map((command) => `- ${command}`).join("\n")}`
111
133
  : "REGISTERED COMMANDS\n- none: this repository registers no verification command, so judge from the files only";
134
+ const keys = (assignment.brief_keys ?? []).map((key) => key.trim()).filter(Boolean);
135
+ const briefKeysBlock = keys.length ? `BRIEF KEYS\n${keys.join("\n")}` : "";
136
+ const probeBlock = keys.length ? DELIVERY_VERIFICATION_PROBE_INSTRUCTION : "";
112
137
  return [
113
138
  "You are Conductor's independent delivery verification. A coding agent delivered this commit against approved acceptance criteria; you check it the way a careful reviewer would, and you never take the agent's word for anything.",
114
139
  "You have repo_read and test_run only. Do not edit, create, delete, format, commit, branch, push, install, or change any file.",
115
140
  diff,
116
- "Run every REGISTERED COMMAND the criteria name, exactly as registered, and read its real output. Run nothing that is not registered. Never start a server, deploy, or call a production system.",
141
+ formatGatesRunBlock(context.gatesRun ?? []),
142
+ "Bridge already ran the gates at this commit. The GATES RUN block is the only gate fact — use those exit codes and output tails; do not choose which commands count as gates. You may still run a REGISTERED COMMAND to read more. Run nothing that is not registered. Never start a server, deploy, or call a production system.",
117
143
  // T132: three verifications ended at the time limit with no brief. Order the work so the brief
118
144
  // exists before the budget does not.
119
- "Order of work: run the registered commands first, one at a time and never two at once (a test suite that rewrites a source file and restores it races against itself when run concurrently, and a changed tree voids your run), then read the changed files that matter, then write the brief. Your time is limited: a brief with `unclear` verdicts for what you did not reach is the required outcome; a run that ends without a brief settles nothing.",
145
+ "Order of work: read the GATES RUN facts first, then the changed files that matter, then write the brief. You may run a registered command to read more, one at a time and never two at once (a test suite that rewrites a source file and restores it races against itself when run concurrently, and a changed tree voids your run). Your time is limited: a brief with `unclear` verdicts for what you did not reach is the required outcome; a run that ends without a brief settles nothing.",
120
146
  "Judge each ACCEPTANCE criterion by its key: supported only when you saw the code or the command output that meets it; unsupported when you saw it is not met, with the file or output that shows it; unclear when this commit and these commands cannot settle it. A criterion about wording in a pull request or a report is unclear here — you cannot see those.",
121
147
  "Do not quote secrets, credentials, tokens, or file bodies. Name paths, symbols, and command output lines instead.",
122
148
  "",
123
149
  `PINNED COMMIT ${commit}`,
124
150
  commands,
125
151
  assignment.question,
152
+ ...(briefKeysBlock ? [briefKeysBlock] : []),
153
+ ...(probeBlock ? [probeBlock] : []),
126
154
  `BUDGET\nAt most ${assignment.budget.max_files} files read in full; the change summary and command output are not counted.`,
127
155
  "",
128
156
  "Return only one fenced ```json object with exactly this shape:",
129
- '{"summary":"one paragraph: what the change does and whether it meets the criteria","findings":["grounded fact with file/symbol"],"verification":["command — result"],"limitations":["what you could not settle"],"criteria":[{"criterion":"c1","assessment":"supported|unsupported|unclear","reason":"what you saw"}],"command_failures":["command that exited non-zero — the decisive output line"],"files_read":0,"bytes_read":0,"commands_run":["command you actually ran"]}',
130
- "criteria has one entry per ACCEPTANCE key. Close the fence and add no prose after it.",
157
+ keys.length ? BRIEF_JSON_EXAMPLE_WITH_PROBES : BRIEF_JSON_EXAMPLE,
158
+ keys.length
159
+ ? "criteria has one entry per ACCEPTANCE key. probes has at most 20 entries, one per BRIEF KEY that states behaviour. Close the fence and add no prose after it."
160
+ : "criteria has one entry per ACCEPTANCE key. Close the fence and add no prose after it.",
131
161
  ].join("\n");
132
162
  }
163
+ function formatGatesRunBlock(gates) {
164
+ const header = "GATES RUN (Bridge ran these at the pinned commit before you started; these are the only gate facts)";
165
+ if (!gates.length)
166
+ return `${header}\n- none`;
167
+ return `${header}\n${gates.map((gate) => {
168
+ const tail = gate.stderrTail.trim() || gate.stdoutTail.trim();
169
+ return tail
170
+ ? `- ${gate.command} — exit ${gate.exitCode}\n ${tail.replace(/\n/g, "\n ")}`
171
+ : `- ${gate.command} — exit ${gate.exitCode}`;
172
+ }).join("\n")}`;
173
+ }
174
+ function gateFailureLine(tail) {
175
+ const lines = tail.split(/\r?\n/).map((line) => line.trim()).filter(Boolean);
176
+ const marked = lines.find((line) => /FAIL|Error|error|failed|✗/.test(line));
177
+ const chosen = marked ?? lines.at(-1) ?? tail;
178
+ return chosen.length > 1_000 ? chosen.slice(0, 1_000) : chosen;
179
+ }
180
+ function gateFailureFact(gate) {
181
+ const tail = gate.stderrTail.trim() || gate.stdoutTail.trim() || "exited non-zero";
182
+ return boundedTail(`${gate.command} — exit ${gate.exitCode}: ${gateFailureLine(tail)}`, 1_000);
183
+ }
184
+ function settleCommandsRun(gates, reported) {
185
+ const out = [];
186
+ const seen = new Set();
187
+ for (const command of [...gates.map((gate) => gate.command), ...reported]) {
188
+ const trimmed = command.trim();
189
+ if (!trimmed || seen.has(trimmed))
190
+ continue;
191
+ seen.add(trimmed);
192
+ out.push(trimmed);
193
+ if (out.length === 10)
194
+ break;
195
+ }
196
+ return out;
197
+ }
133
198
  /** One message for every unusable answer, so the retry decision reads it rather than guessing. */
134
199
  export const UNUSABLE_BRIEF = "Investigation returned no usable brief";
135
200
  /** T130: how much of an unusable reply travels with the failure, so the next run can be diagnosed. */
@@ -193,6 +258,146 @@ export async function verifyObservationOnly(input) {
193
258
  }
194
259
  return { ok: true };
195
260
  }
261
+ const PROBE_TIMEOUT_MS = 60_000;
262
+ const PROBE_MAX_BUFFER_BYTES = 2_000_000;
263
+ const PROBE_TAIL_CHARS = 1_000;
264
+ const PROBE_MAX = 20;
265
+ const PROBE_COMMAND_MAX = 300;
266
+ const PROBE_CHECKS_MAX = 500;
267
+ const PROBE_KEY_MAX = 40;
268
+ function probeEnv() {
269
+ const env = {};
270
+ if (process.env.PATH !== undefined)
271
+ env.PATH = process.env.PATH;
272
+ if (process.env.HOME !== undefined)
273
+ env.HOME = process.env.HOME;
274
+ return env;
275
+ }
276
+ function probeInterpreter(language) {
277
+ if (language === "ts")
278
+ return "node --import tsx";
279
+ if (language === "py")
280
+ return "python3";
281
+ return "bash";
282
+ }
283
+ function probeFilename(language) {
284
+ if (language === "ts")
285
+ return "probe.ts";
286
+ if (language === "py")
287
+ return "probe.py";
288
+ return "probe.sh";
289
+ }
290
+ function probeArgv(language, file) {
291
+ if (language === "ts")
292
+ return ["node", "--import", "tsx", file];
293
+ if (language === "py")
294
+ return ["python3", file];
295
+ return ["bash", file];
296
+ }
297
+ function clipWitnessField(value, max) {
298
+ return value.length > max ? value.slice(0, max) : value;
299
+ }
300
+ async function execProbe(argv, workspace) {
301
+ const bin = argv[0];
302
+ if (!bin)
303
+ throw new Error("Probe command is empty");
304
+ try {
305
+ const { stdout, stderr } = await execFileAsync(bin, argv.slice(1), {
306
+ cwd: workspace,
307
+ timeout: PROBE_TIMEOUT_MS,
308
+ maxBuffer: PROBE_MAX_BUFFER_BYTES,
309
+ env: probeEnv(),
310
+ });
311
+ return { stdout: String(stdout), stderr: String(stderr), code: 0 };
312
+ }
313
+ catch (error) {
314
+ const err = error;
315
+ if (typeof err.code === "string")
316
+ throw error;
317
+ return {
318
+ stdout: String(err.stdout ?? ""),
319
+ stderr: String(err.stderr || err.message || "probe failed"),
320
+ code: err.killed === true || typeof err.code !== "number" ? -1 : err.code,
321
+ };
322
+ }
323
+ }
324
+ function witnessRecord(input) {
325
+ return {
326
+ key: clipWitnessField(input.key, PROBE_KEY_MAX),
327
+ kind: input.kind,
328
+ command: clipWitnessField(input.command, PROBE_COMMAND_MAX),
329
+ exit_code: input.exitCode,
330
+ tail: boundedTail(`${input.stderr.trim()}\n${input.stdout.trim()}`.trim(), PROBE_TAIL_CHARS),
331
+ checks: clipWitnessField(input.checks, PROBE_CHECKS_MAX),
332
+ };
333
+ }
334
+ /**
335
+ * Run at most twenty agent-authored probes after the read-only proof. Records only what Bridge
336
+ * executed; agent-claimed exit codes never enter the witness.
337
+ */
338
+ export async function runDeliveryVerificationProbes(input) {
339
+ const selected = [...input.probes].slice(0, PROBE_MAX);
340
+ const witnesses = [];
341
+ for (const probe of selected) {
342
+ const checks = probe.checks ?? "";
343
+ const named = probe.test?.trim() ?? "";
344
+ const source = probe.source?.trim() ?? "";
345
+ if (named) {
346
+ if (!isRunnableVerificationCommand(named)) {
347
+ witnesses.push(witnessRecord({
348
+ key: probe.key, kind: "test", command: named, exitCode: -1, stdout: "", stderr: "not a registered command", checks,
349
+ }));
350
+ continue;
351
+ }
352
+ let exitCode = -1;
353
+ let stdout = "";
354
+ let stderr = "";
355
+ try {
356
+ const result = await (input.runCommand ?? runBoundedVerificationCommand)(named, input.workspace);
357
+ exitCode = result.code;
358
+ stdout = result.stdout;
359
+ stderr = result.stderr;
360
+ }
361
+ catch (error) {
362
+ stderr = error instanceof Error ? error.message : String(error);
363
+ exitCode = -1;
364
+ }
365
+ witnesses.push(witnessRecord({
366
+ key: probe.key, kind: "test", command: named, exitCode, stdout, stderr, checks,
367
+ }));
368
+ continue;
369
+ }
370
+ if (!source || !probe.lang)
371
+ continue;
372
+ const language = probe.lang;
373
+ const command = `${probeInterpreter(language)} ${probe.key}`;
374
+ let dir = null;
375
+ let exitCode = -1;
376
+ let stdout = "";
377
+ let stderr = "";
378
+ try {
379
+ dir = await mkdtemp(join(tmpdir(), "conduit-probe-"));
380
+ const file = join(dir, probeFilename(language));
381
+ await writeFile(file, source, { encoding: "utf8" });
382
+ const result = await execProbe(probeArgv(language, file), input.workspace);
383
+ exitCode = result.code;
384
+ stdout = result.stdout;
385
+ stderr = result.stderr;
386
+ }
387
+ catch (error) {
388
+ stderr = error instanceof Error ? error.message : String(error);
389
+ exitCode = -1;
390
+ }
391
+ finally {
392
+ if (dir)
393
+ await rm(dir, { recursive: true, force: true }).catch(() => undefined);
394
+ }
395
+ witnesses.push(witnessRecord({
396
+ key: probe.key, kind: "probe", command, exitCode, stdout, stderr, checks,
397
+ }));
398
+ }
399
+ return witnesses;
400
+ }
196
401
  async function settle(client, investigationId, body) {
197
402
  await client.request(`/runner/v1/investigations/${investigationId}/settle`, {
198
403
  method: "POST",
@@ -303,9 +508,20 @@ export async function executeNextInvestigation(client, config, workspace, brief,
303
508
  : undefined;
304
509
  // T126: the prompt names the commands of the repository this lane opened — not the control
305
510
  // plane's idea of them — and carries the diff stat Bridge computed, since the lane cannot run git.
511
+ const deliveryVerification = isDeliveryVerification(assignment);
512
+ const gatesRun = deliveryVerification
513
+ ? await runDeliveryVerificationGates({
514
+ workspace: attemptWorkspace,
515
+ verificationCommands: liveBrief?.verification ?? [],
516
+ declaredVerificationCommands: liveBrief?.declared_verification ?? [],
517
+ acceptance: assignment.acceptance ?? [],
518
+ runCommand: options.runVerificationCommand,
519
+ })
520
+ : [];
306
521
  const context = {
307
522
  verificationCommands: liveBrief?.verification ?? [],
308
523
  diffStat: assignment.base_commit ? await diffStatSince(attemptWorkspace, assignment.base_commit) : null,
524
+ gatesRun,
309
525
  };
310
526
  // T154: the lane resolves its model the way the authoring lane does. Without it Claude Code ran
311
527
  // at its CLI default, which Conduit's /v1 rejects as an unknown alias (hosted-1, 2026-09-10:
@@ -345,6 +561,15 @@ export async function executeNextInvestigation(client, config, workspace, brief,
345
561
  console.error(`Investigation ${assignment.id} discarded — ${check.error}: ${check.detail}`);
346
562
  return true;
347
563
  }
564
+ const stepWitness = deliveryVerification
565
+ ? await runDeliveryVerificationProbes({
566
+ workspace: attemptWorkspace,
567
+ probes: observed.probes ?? [],
568
+ runCommand: options.runProbeCommand,
569
+ })
570
+ : [];
571
+ if (deliveryVerification)
572
+ await discardWorktreeWrites(attemptWorkspace);
348
573
  await settle(client, assignment.id, {
349
574
  lease_token: leaseToken,
350
575
  status: "completed",
@@ -354,7 +579,10 @@ export async function executeNextInvestigation(client, config, workspace, brief,
354
579
  verification: observed.verification,
355
580
  limitations: observed.limitations,
356
581
  criteria: observed.criteria,
357
- command_failures: observed.command_failures,
582
+ command_failures: deliveryVerification
583
+ ? gatesRun.filter((gate) => gate.exitCode !== 0).map(gateFailureFact).slice(0, 10)
584
+ : observed.command_failures,
585
+ step_witness: stepWitness,
358
586
  witness: {
359
587
  repository_fingerprint: assignment.repository_fingerprint ?? liveRepository ?? "unknown",
360
588
  commit,
@@ -362,7 +590,9 @@ export async function executeNextInvestigation(client, config, workspace, brief,
362
590
  bounded_by: boundedByForDriver(driverId),
363
591
  files_read: observed.files_read,
364
592
  bytes_read: observed.bytes_read,
365
- commands_run: observed.commands_run,
593
+ commands_run: deliveryVerification
594
+ ? settleCommandsRun(gatesRun, observed.commands_run)
595
+ : observed.commands_run,
366
596
  },
367
597
  },
368
598
  });
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@miraland-labs/conduit-bridge",
3
- "version": "0.16.108",
3
+ "version": "0.16.110",
4
4
  "description": "Conduit Bridge CLI \u2014 join, connect, disconnect, multi-driver lanes, and run Claude Code / Codex / Cursor / OpenCode / Pi / Kiro / Antigravity / Grok Build agents for a Conduit organization",
5
5
  "type": "module",
6
6
  "bin": {