@miraland-labs/conduit-bridge 0.16.108 → 0.16.110
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/execution.js +38 -2
- package/dist/failure-signal.js +26 -0
- package/dist/investigation.js +240 -10
- package/package.json +1 -1
package/dist/execution.js
CHANGED
|
@@ -25,7 +25,7 @@ import { ensureLandCommit } from "./ensure-land-commit.js";
|
|
|
25
25
|
import { AGENT_NO_LAND_COMMIT_PREFIX, AgentNoLandCommitError, agentNoLandCommitMessage, requiresLandCommit } from "./land-contract.js";
|
|
26
26
|
import { agentClaimsRepositoryWork, claimsVsGitMismatch, isLandTreeEmpty, reportClaimsRepositoryWork, } from "./land-git-reality.js";
|
|
27
27
|
const execFileAsync = promisify(execFile);
|
|
28
|
-
import { captureVerificationFailure, detectGateRanNothing, ensureTestEvidence, needsTestEvidence, runBoundedVerificationCommand, verificationEvidenceCommands, VerificationGateRanNothingError } from "./ensure-test-evidence.js";
|
|
28
|
+
import { boundedTail, captureVerificationFailure, detectGateRanNothing, ensureTestEvidence, needsTestEvidence, runBoundedVerificationCommand, verificationEvidenceCommands, VerificationGateRanNothingError } from "./ensure-test-evidence.js";
|
|
29
29
|
import { ensureChangeEvidence } from "./ensure-change-evidence.js";
|
|
30
30
|
import { ensureNormativeEvidence } from "./ensure-normative-evidence.js";
|
|
31
31
|
import { changedPathsSince, compareWorktreeState, diffStatSince, worktreeStateFingerprint, worktreeWrittenPaths } from "./git-witness.js";
|
|
@@ -2872,8 +2872,44 @@ export async function verificationScopeDryRun(input) {
|
|
|
2872
2872
|
: written.filter((path) => !input.changeScope.some((scope) => pathMatchesScope(path, scope)));
|
|
2873
2873
|
return { ranNothing, redBase, outsideScope };
|
|
2874
2874
|
}
|
|
2875
|
+
/** Prompt and failure lines keep a short tail; the investigation brief caps each failure at 1,000. */
|
|
2876
|
+
const DELIVERY_VERIFICATION_GATE_TAIL_CHARS = 800;
|
|
2877
|
+
/**
|
|
2878
|
+
* Run the T191 gate set in the verification worktree at the delivered commit: declared
|
|
2879
|
+
* `.conduit/verification` ∪ the commands the criteria name, else the discovered set. One at a
|
|
2880
|
+
* time, same bounded runner as the pre-agent dry run. Writes the gates made are discarded so the
|
|
2881
|
+
* agent still reads the delivered tree.
|
|
2882
|
+
*/
|
|
2883
|
+
export async function runDeliveryVerificationGates(input) {
|
|
2884
|
+
const commands = verificationEvidenceCommands([...input.verificationCommands], input.declaredVerificationCommands ?? [], input.acceptance ?? []);
|
|
2885
|
+
const run = input.runCommand ?? runBoundedVerificationCommand;
|
|
2886
|
+
const results = [];
|
|
2887
|
+
for (const command of commands) {
|
|
2888
|
+
try {
|
|
2889
|
+
const result = await run(command, input.workspace);
|
|
2890
|
+
results.push({
|
|
2891
|
+
command,
|
|
2892
|
+
exitCode: result.code,
|
|
2893
|
+
stdoutTail: boundedTail(result.stdout, DELIVERY_VERIFICATION_GATE_TAIL_CHARS),
|
|
2894
|
+
stderrTail: boundedTail(result.stderr, DELIVERY_VERIFICATION_GATE_TAIL_CHARS),
|
|
2895
|
+
});
|
|
2896
|
+
}
|
|
2897
|
+
catch (error) {
|
|
2898
|
+
// A gate that could not run is a failed gate: record it so settlement still names it.
|
|
2899
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
2900
|
+
results.push({
|
|
2901
|
+
command,
|
|
2902
|
+
exitCode: -1,
|
|
2903
|
+
stdoutTail: "",
|
|
2904
|
+
stderrTail: boundedTail(message, DELIVERY_VERIFICATION_GATE_TAIL_CHARS),
|
|
2905
|
+
});
|
|
2906
|
+
}
|
|
2907
|
+
}
|
|
2908
|
+
await discardWorktreeWrites(input.workspace);
|
|
2909
|
+
return results;
|
|
2910
|
+
}
|
|
2875
2911
|
/** Return the worktree to its commit: tracked files restored, new files removed, ignored files kept. */
|
|
2876
|
-
async function discardWorktreeWrites(worktree) {
|
|
2912
|
+
export async function discardWorktreeWrites(worktree) {
|
|
2877
2913
|
const options = { timeout: 60_000, maxBuffer: 2_000_000 };
|
|
2878
2914
|
await execFileAsync("git", ["-C", worktree, "checkout", "--", "."], options).catch(() => undefined);
|
|
2879
2915
|
await execFileAsync("git", ["-C", worktree, "clean", "-fd"], options).catch(() => undefined);
|
package/dist/failure-signal.js
CHANGED
|
@@ -75,6 +75,32 @@ export const failureSignalSchema = z.object({
|
|
|
75
75
|
*/
|
|
76
76
|
model: z.string().max(200).optional(),
|
|
77
77
|
});
|
|
78
|
+
/**
|
|
79
|
+
* The class–disposition pairs a classified failure may carry. Disposition is a function of the
|
|
80
|
+
* cause, not of class: the six canonical pairs plus `contract`/`hold` (the work package is the
|
|
81
|
+
* defect; a rework would spend repair on code that is not at fault) and `environment`/`retry`
|
|
82
|
+
* (the computer hiccupped; the next lane runs it again).
|
|
83
|
+
*
|
|
84
|
+
* Declared once so the envelope schema and both twins share one list. `ClassifiedFailure` stays a
|
|
85
|
+
* free pair; this list is the boundary check.
|
|
86
|
+
*/
|
|
87
|
+
export const ADMITTED_FAILURE_PAIRS = [
|
|
88
|
+
["transient", "retry"],
|
|
89
|
+
["environment", "hold"],
|
|
90
|
+
["environment", "retry"],
|
|
91
|
+
["contract", "rework"],
|
|
92
|
+
["contract", "hold"],
|
|
93
|
+
["quality", "rework"],
|
|
94
|
+
["authority", "decision"],
|
|
95
|
+
["platform", "stop"],
|
|
96
|
+
];
|
|
97
|
+
/**
|
|
98
|
+
* Whether this class and disposition are one of the admitted pairs. The envelope schema calls it
|
|
99
|
+
* so a supplied envelope cannot carry a pairing the builders never emit.
|
|
100
|
+
*/
|
|
101
|
+
export function isAdmittedFailurePair(failureClass, disposition) {
|
|
102
|
+
return ADMITTED_FAILURE_PAIRS.some(([admittedClass, admittedDisposition]) => admittedClass === failureClass && admittedDisposition === disposition);
|
|
103
|
+
}
|
|
78
104
|
/**
|
|
79
105
|
* The causes Bridge names itself — before the agent starts, or while finishing its report — as one
|
|
80
106
|
* table keyed by code, so Bridge sends the envelope and the control plane validates it (K3).
|
package/dist/investigation.js
CHANGED
|
@@ -6,19 +6,22 @@
|
|
|
6
6
|
* the worktree it reads is deleted whichever way the run ends.
|
|
7
7
|
*/
|
|
8
8
|
import { z } from "zod";
|
|
9
|
-
import { boundedTail } from "./ensure-test-evidence.js";
|
|
9
|
+
import { boundedTail, runBoundedVerificationCommand } from "./ensure-test-evidence.js";
|
|
10
10
|
import { createAttemptWorktree, removeAttemptWorktree } from "./attempt-worktree.js";
|
|
11
11
|
import { buildWorkspaceBrief, ensureCommitAvailable, normalizeRepositoryUrl } from "./brief.js";
|
|
12
12
|
import { headCommitOrNull } from "./git.js";
|
|
13
13
|
import { redactSecrets } from "./config.js";
|
|
14
|
-
import { learnDriverFuel, resolveAssignmentModel } from "./execution.js";
|
|
14
|
+
import { learnDriverFuel, resolveAssignmentModel, changedPathsSince, runDeliveryVerificationGates, discardWorktreeWrites } from "./execution.js";
|
|
15
15
|
import { DRIVERS, extractAgentReportJsonText, parseJsonObjectCandidate } from "./driver.js";
|
|
16
16
|
import { pickDriverForClaim, resolveDriverFuel, supportsReadOnlyDiagnosis } from "./drivers.js";
|
|
17
|
-
import { changedPathsSince } from "./execution.js";
|
|
18
17
|
import { diffStatSince } from "./git-witness.js";
|
|
19
18
|
import { switchManagedWorkspace } from "./on-shift-apply.js";
|
|
20
19
|
import { execFile } from "node:child_process";
|
|
20
|
+
import { mkdtemp, rm, writeFile } from "node:fs/promises";
|
|
21
|
+
import { tmpdir } from "node:os";
|
|
22
|
+
import { join } from "node:path";
|
|
21
23
|
import { promisify } from "node:util";
|
|
24
|
+
import { isRunnableVerificationCommand } from "./execution-class.js";
|
|
22
25
|
const execFileAsync = promisify(execFile);
|
|
23
26
|
const budgetSchema = z.object({
|
|
24
27
|
max_files: z.number().int().min(1).max(24).optional().default(8),
|
|
@@ -36,6 +39,10 @@ export const investigationAssignmentSchema = z.object({
|
|
|
36
39
|
/** T122: present on a delivery verification — the contract base the delivered head was built on. */
|
|
37
40
|
base_commit: z.string().nullable().optional().default(null),
|
|
38
41
|
source_attempt_id: z.string().uuid().nullable().optional().default(null),
|
|
42
|
+
/** Delivery verification: the task's acceptance strings. Absent or [] on a plain investigation. */
|
|
43
|
+
acceptance: z.array(z.string().trim().min(1).max(4_000)).max(20).optional().default([]),
|
|
44
|
+
/** Delivery verification: the brief's `[sN]` / `[kN]` lines. Absent or [] on a plain investigation. */
|
|
45
|
+
brief_keys: z.array(z.string().trim().min(1).max(4_000)).max(40).optional().default([]),
|
|
39
46
|
});
|
|
40
47
|
/**
|
|
41
48
|
* Enforcement is a vendor's claim; verification is ours. Both run on every investigation — this only
|
|
@@ -66,9 +73,24 @@ const investigationBriefSchema = z.object({
|
|
|
66
73
|
files_read: z.number().int().min(0).max(24).default(0),
|
|
67
74
|
bytes_read: z.number().int().min(0).max(512 * 1024).default(0),
|
|
68
75
|
commands_run: z.array(z.string().trim().min(1).max(200)).max(10).default([]),
|
|
76
|
+
/**
|
|
77
|
+
* K12: probe source the agent wrote. Bridge executes these; it never copies claimed results into
|
|
78
|
+
* step_witness. Unknown keys such as a claimed exit_code are stripped.
|
|
79
|
+
*/
|
|
80
|
+
probes: z.array(z.object({
|
|
81
|
+
key: z.string().trim().min(1).max(40),
|
|
82
|
+
test: z.string().max(300).optional(),
|
|
83
|
+
lang: z.enum(["ts", "py", "sh"]).optional(),
|
|
84
|
+
source: z.string().max(8_000).optional(),
|
|
85
|
+
checks: z.string().max(500).optional().default(""),
|
|
86
|
+
})).max(20).default([]),
|
|
69
87
|
});
|
|
70
88
|
/** T122: a delivery verification is an investigation whose question carries this header. */
|
|
71
89
|
export const DELIVERY_VERIFICATION_HEADER = "DELIVERY VERIFICATION";
|
|
90
|
+
/** Owner-verbatim: the lane writes probe source; Bridge executes. Shown only when BRIEF KEYS are present. */
|
|
91
|
+
export const DELIVERY_VERIFICATION_PROBE_INSTRUCTION = 'For every BRIEF KEY that states behaviour, add one entry to probes: either {"key":"[sN]","test":"<registered command that asserts it>","checks":"<one line: what it proves>"} or {"key":"[sN]","lang":"ts|py|sh","source":"<a short script that exits 0 only when the key holds at this commit and non-zero otherwise; it may import repository modules by absolute path from the working directory>","checks":"<one line>"}. You do not run probes; Bridge runs them after you finish. Keys that state no behaviour (wording, scope, docs) get no probe.';
|
|
92
|
+
const BRIEF_JSON_EXAMPLE = '{"summary":"one paragraph: what the change does and whether it meets the criteria","findings":["grounded fact with file/symbol"],"verification":["command — result"],"limitations":["what you could not settle"],"criteria":[{"criterion":"c1","assessment":"supported|unsupported|unclear","reason":"what you saw"}],"command_failures":["command that exited non-zero — the decisive output line"],"files_read":0,"bytes_read":0,"commands_run":["command you actually ran"]}';
|
|
93
|
+
const BRIEF_JSON_EXAMPLE_WITH_PROBES = '{"summary":"one paragraph: what the change does and whether it meets the criteria","findings":["grounded fact with file/symbol"],"verification":["command — result"],"limitations":["what you could not settle"],"criteria":[{"criterion":"c1","assessment":"supported|unsupported|unclear","reason":"what you saw"}],"command_failures":["command that exited non-zero — the decisive output line"],"probes":[],"files_read":0,"bytes_read":0,"commands_run":["command you actually ran"]}';
|
|
72
94
|
export function isDeliveryVerification(assignment) {
|
|
73
95
|
return assignment.question.trimStart().startsWith(DELIVERY_VERIFICATION_HEADER);
|
|
74
96
|
}
|
|
@@ -95,7 +117,7 @@ export function buildInvestigationPrompt(assignment, commit, context) {
|
|
|
95
117
|
}
|
|
96
118
|
/**
|
|
97
119
|
* T122: the Conductor's own check of a delivery, done the way a careful reviewer does it by hand —
|
|
98
|
-
*
|
|
120
|
+
* read the gates Bridge already ran at the delivered head, read the diff, judge every criterion. The
|
|
99
121
|
* verdict per criterion is the product; prose without verdicts is an unusable brief.
|
|
100
122
|
*/
|
|
101
123
|
export function buildDeliveryVerificationPrompt(assignment, commit, context) {
|
|
@@ -109,27 +131,70 @@ export function buildDeliveryVerificationPrompt(assignment, commit, context) {
|
|
|
109
131
|
const commands = context.verificationCommands.length
|
|
110
132
|
? `REGISTERED COMMANDS (the only commands your shell accepts)\n${context.verificationCommands.map((command) => `- ${command}`).join("\n")}`
|
|
111
133
|
: "REGISTERED COMMANDS\n- none: this repository registers no verification command, so judge from the files only";
|
|
134
|
+
const keys = (assignment.brief_keys ?? []).map((key) => key.trim()).filter(Boolean);
|
|
135
|
+
const briefKeysBlock = keys.length ? `BRIEF KEYS\n${keys.join("\n")}` : "";
|
|
136
|
+
const probeBlock = keys.length ? DELIVERY_VERIFICATION_PROBE_INSTRUCTION : "";
|
|
112
137
|
return [
|
|
113
138
|
"You are Conductor's independent delivery verification. A coding agent delivered this commit against approved acceptance criteria; you check it the way a careful reviewer would, and you never take the agent's word for anything.",
|
|
114
139
|
"You have repo_read and test_run only. Do not edit, create, delete, format, commit, branch, push, install, or change any file.",
|
|
115
140
|
diff,
|
|
116
|
-
|
|
141
|
+
formatGatesRunBlock(context.gatesRun ?? []),
|
|
142
|
+
"Bridge already ran the gates at this commit. The GATES RUN block is the only gate fact — use those exit codes and output tails; do not choose which commands count as gates. You may still run a REGISTERED COMMAND to read more. Run nothing that is not registered. Never start a server, deploy, or call a production system.",
|
|
117
143
|
// T132: three verifications ended at the time limit with no brief. Order the work so the brief
|
|
118
144
|
// exists before the budget does not.
|
|
119
|
-
"Order of work:
|
|
145
|
+
"Order of work: read the GATES RUN facts first, then the changed files that matter, then write the brief. You may run a registered command to read more, one at a time and never two at once (a test suite that rewrites a source file and restores it races against itself when run concurrently, and a changed tree voids your run). Your time is limited: a brief with `unclear` verdicts for what you did not reach is the required outcome; a run that ends without a brief settles nothing.",
|
|
120
146
|
"Judge each ACCEPTANCE criterion by its key: supported only when you saw the code or the command output that meets it; unsupported when you saw it is not met, with the file or output that shows it; unclear when this commit and these commands cannot settle it. A criterion about wording in a pull request or a report is unclear here — you cannot see those.",
|
|
121
147
|
"Do not quote secrets, credentials, tokens, or file bodies. Name paths, symbols, and command output lines instead.",
|
|
122
148
|
"",
|
|
123
149
|
`PINNED COMMIT ${commit}`,
|
|
124
150
|
commands,
|
|
125
151
|
assignment.question,
|
|
152
|
+
...(briefKeysBlock ? [briefKeysBlock] : []),
|
|
153
|
+
...(probeBlock ? [probeBlock] : []),
|
|
126
154
|
`BUDGET\nAt most ${assignment.budget.max_files} files read in full; the change summary and command output are not counted.`,
|
|
127
155
|
"",
|
|
128
156
|
"Return only one fenced ```json object with exactly this shape:",
|
|
129
|
-
|
|
130
|
-
|
|
157
|
+
keys.length ? BRIEF_JSON_EXAMPLE_WITH_PROBES : BRIEF_JSON_EXAMPLE,
|
|
158
|
+
keys.length
|
|
159
|
+
? "criteria has one entry per ACCEPTANCE key. probes has at most 20 entries, one per BRIEF KEY that states behaviour. Close the fence and add no prose after it."
|
|
160
|
+
: "criteria has one entry per ACCEPTANCE key. Close the fence and add no prose after it.",
|
|
131
161
|
].join("\n");
|
|
132
162
|
}
|
|
163
|
+
function formatGatesRunBlock(gates) {
|
|
164
|
+
const header = "GATES RUN (Bridge ran these at the pinned commit before you started; these are the only gate facts)";
|
|
165
|
+
if (!gates.length)
|
|
166
|
+
return `${header}\n- none`;
|
|
167
|
+
return `${header}\n${gates.map((gate) => {
|
|
168
|
+
const tail = gate.stderrTail.trim() || gate.stdoutTail.trim();
|
|
169
|
+
return tail
|
|
170
|
+
? `- ${gate.command} — exit ${gate.exitCode}\n ${tail.replace(/\n/g, "\n ")}`
|
|
171
|
+
: `- ${gate.command} — exit ${gate.exitCode}`;
|
|
172
|
+
}).join("\n")}`;
|
|
173
|
+
}
|
|
174
|
+
function gateFailureLine(tail) {
|
|
175
|
+
const lines = tail.split(/\r?\n/).map((line) => line.trim()).filter(Boolean);
|
|
176
|
+
const marked = lines.find((line) => /FAIL|Error|error|failed|✗/.test(line));
|
|
177
|
+
const chosen = marked ?? lines.at(-1) ?? tail;
|
|
178
|
+
return chosen.length > 1_000 ? chosen.slice(0, 1_000) : chosen;
|
|
179
|
+
}
|
|
180
|
+
function gateFailureFact(gate) {
|
|
181
|
+
const tail = gate.stderrTail.trim() || gate.stdoutTail.trim() || "exited non-zero";
|
|
182
|
+
return boundedTail(`${gate.command} — exit ${gate.exitCode}: ${gateFailureLine(tail)}`, 1_000);
|
|
183
|
+
}
|
|
184
|
+
function settleCommandsRun(gates, reported) {
|
|
185
|
+
const out = [];
|
|
186
|
+
const seen = new Set();
|
|
187
|
+
for (const command of [...gates.map((gate) => gate.command), ...reported]) {
|
|
188
|
+
const trimmed = command.trim();
|
|
189
|
+
if (!trimmed || seen.has(trimmed))
|
|
190
|
+
continue;
|
|
191
|
+
seen.add(trimmed);
|
|
192
|
+
out.push(trimmed);
|
|
193
|
+
if (out.length === 10)
|
|
194
|
+
break;
|
|
195
|
+
}
|
|
196
|
+
return out;
|
|
197
|
+
}
|
|
133
198
|
/** One message for every unusable answer, so the retry decision reads it rather than guessing. */
|
|
134
199
|
export const UNUSABLE_BRIEF = "Investigation returned no usable brief";
|
|
135
200
|
/** T130: how much of an unusable reply travels with the failure, so the next run can be diagnosed. */
|
|
@@ -193,6 +258,146 @@ export async function verifyObservationOnly(input) {
|
|
|
193
258
|
}
|
|
194
259
|
return { ok: true };
|
|
195
260
|
}
|
|
261
|
+
const PROBE_TIMEOUT_MS = 60_000;
|
|
262
|
+
const PROBE_MAX_BUFFER_BYTES = 2_000_000;
|
|
263
|
+
const PROBE_TAIL_CHARS = 1_000;
|
|
264
|
+
const PROBE_MAX = 20;
|
|
265
|
+
const PROBE_COMMAND_MAX = 300;
|
|
266
|
+
const PROBE_CHECKS_MAX = 500;
|
|
267
|
+
const PROBE_KEY_MAX = 40;
|
|
268
|
+
function probeEnv() {
|
|
269
|
+
const env = {};
|
|
270
|
+
if (process.env.PATH !== undefined)
|
|
271
|
+
env.PATH = process.env.PATH;
|
|
272
|
+
if (process.env.HOME !== undefined)
|
|
273
|
+
env.HOME = process.env.HOME;
|
|
274
|
+
return env;
|
|
275
|
+
}
|
|
276
|
+
function probeInterpreter(language) {
|
|
277
|
+
if (language === "ts")
|
|
278
|
+
return "node --import tsx";
|
|
279
|
+
if (language === "py")
|
|
280
|
+
return "python3";
|
|
281
|
+
return "bash";
|
|
282
|
+
}
|
|
283
|
+
function probeFilename(language) {
|
|
284
|
+
if (language === "ts")
|
|
285
|
+
return "probe.ts";
|
|
286
|
+
if (language === "py")
|
|
287
|
+
return "probe.py";
|
|
288
|
+
return "probe.sh";
|
|
289
|
+
}
|
|
290
|
+
function probeArgv(language, file) {
|
|
291
|
+
if (language === "ts")
|
|
292
|
+
return ["node", "--import", "tsx", file];
|
|
293
|
+
if (language === "py")
|
|
294
|
+
return ["python3", file];
|
|
295
|
+
return ["bash", file];
|
|
296
|
+
}
|
|
297
|
+
function clipWitnessField(value, max) {
|
|
298
|
+
return value.length > max ? value.slice(0, max) : value;
|
|
299
|
+
}
|
|
300
|
+
async function execProbe(argv, workspace) {
|
|
301
|
+
const bin = argv[0];
|
|
302
|
+
if (!bin)
|
|
303
|
+
throw new Error("Probe command is empty");
|
|
304
|
+
try {
|
|
305
|
+
const { stdout, stderr } = await execFileAsync(bin, argv.slice(1), {
|
|
306
|
+
cwd: workspace,
|
|
307
|
+
timeout: PROBE_TIMEOUT_MS,
|
|
308
|
+
maxBuffer: PROBE_MAX_BUFFER_BYTES,
|
|
309
|
+
env: probeEnv(),
|
|
310
|
+
});
|
|
311
|
+
return { stdout: String(stdout), stderr: String(stderr), code: 0 };
|
|
312
|
+
}
|
|
313
|
+
catch (error) {
|
|
314
|
+
const err = error;
|
|
315
|
+
if (typeof err.code === "string")
|
|
316
|
+
throw error;
|
|
317
|
+
return {
|
|
318
|
+
stdout: String(err.stdout ?? ""),
|
|
319
|
+
stderr: String(err.stderr || err.message || "probe failed"),
|
|
320
|
+
code: err.killed === true || typeof err.code !== "number" ? -1 : err.code,
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
function witnessRecord(input) {
|
|
325
|
+
return {
|
|
326
|
+
key: clipWitnessField(input.key, PROBE_KEY_MAX),
|
|
327
|
+
kind: input.kind,
|
|
328
|
+
command: clipWitnessField(input.command, PROBE_COMMAND_MAX),
|
|
329
|
+
exit_code: input.exitCode,
|
|
330
|
+
tail: boundedTail(`${input.stderr.trim()}\n${input.stdout.trim()}`.trim(), PROBE_TAIL_CHARS),
|
|
331
|
+
checks: clipWitnessField(input.checks, PROBE_CHECKS_MAX),
|
|
332
|
+
};
|
|
333
|
+
}
|
|
334
|
+
/**
|
|
335
|
+
* Run at most twenty agent-authored probes after the read-only proof. Records only what Bridge
|
|
336
|
+
* executed; agent-claimed exit codes never enter the witness.
|
|
337
|
+
*/
|
|
338
|
+
export async function runDeliveryVerificationProbes(input) {
|
|
339
|
+
const selected = [...input.probes].slice(0, PROBE_MAX);
|
|
340
|
+
const witnesses = [];
|
|
341
|
+
for (const probe of selected) {
|
|
342
|
+
const checks = probe.checks ?? "";
|
|
343
|
+
const named = probe.test?.trim() ?? "";
|
|
344
|
+
const source = probe.source?.trim() ?? "";
|
|
345
|
+
if (named) {
|
|
346
|
+
if (!isRunnableVerificationCommand(named)) {
|
|
347
|
+
witnesses.push(witnessRecord({
|
|
348
|
+
key: probe.key, kind: "test", command: named, exitCode: -1, stdout: "", stderr: "not a registered command", checks,
|
|
349
|
+
}));
|
|
350
|
+
continue;
|
|
351
|
+
}
|
|
352
|
+
let exitCode = -1;
|
|
353
|
+
let stdout = "";
|
|
354
|
+
let stderr = "";
|
|
355
|
+
try {
|
|
356
|
+
const result = await (input.runCommand ?? runBoundedVerificationCommand)(named, input.workspace);
|
|
357
|
+
exitCode = result.code;
|
|
358
|
+
stdout = result.stdout;
|
|
359
|
+
stderr = result.stderr;
|
|
360
|
+
}
|
|
361
|
+
catch (error) {
|
|
362
|
+
stderr = error instanceof Error ? error.message : String(error);
|
|
363
|
+
exitCode = -1;
|
|
364
|
+
}
|
|
365
|
+
witnesses.push(witnessRecord({
|
|
366
|
+
key: probe.key, kind: "test", command: named, exitCode, stdout, stderr, checks,
|
|
367
|
+
}));
|
|
368
|
+
continue;
|
|
369
|
+
}
|
|
370
|
+
if (!source || !probe.lang)
|
|
371
|
+
continue;
|
|
372
|
+
const language = probe.lang;
|
|
373
|
+
const command = `${probeInterpreter(language)} ${probe.key}`;
|
|
374
|
+
let dir = null;
|
|
375
|
+
let exitCode = -1;
|
|
376
|
+
let stdout = "";
|
|
377
|
+
let stderr = "";
|
|
378
|
+
try {
|
|
379
|
+
dir = await mkdtemp(join(tmpdir(), "conduit-probe-"));
|
|
380
|
+
const file = join(dir, probeFilename(language));
|
|
381
|
+
await writeFile(file, source, { encoding: "utf8" });
|
|
382
|
+
const result = await execProbe(probeArgv(language, file), input.workspace);
|
|
383
|
+
exitCode = result.code;
|
|
384
|
+
stdout = result.stdout;
|
|
385
|
+
stderr = result.stderr;
|
|
386
|
+
}
|
|
387
|
+
catch (error) {
|
|
388
|
+
stderr = error instanceof Error ? error.message : String(error);
|
|
389
|
+
exitCode = -1;
|
|
390
|
+
}
|
|
391
|
+
finally {
|
|
392
|
+
if (dir)
|
|
393
|
+
await rm(dir, { recursive: true, force: true }).catch(() => undefined);
|
|
394
|
+
}
|
|
395
|
+
witnesses.push(witnessRecord({
|
|
396
|
+
key: probe.key, kind: "probe", command, exitCode, stdout, stderr, checks,
|
|
397
|
+
}));
|
|
398
|
+
}
|
|
399
|
+
return witnesses;
|
|
400
|
+
}
|
|
196
401
|
async function settle(client, investigationId, body) {
|
|
197
402
|
await client.request(`/runner/v1/investigations/${investigationId}/settle`, {
|
|
198
403
|
method: "POST",
|
|
@@ -303,9 +508,20 @@ export async function executeNextInvestigation(client, config, workspace, brief,
|
|
|
303
508
|
: undefined;
|
|
304
509
|
// T126: the prompt names the commands of the repository this lane opened — not the control
|
|
305
510
|
// plane's idea of them — and carries the diff stat Bridge computed, since the lane cannot run git.
|
|
511
|
+
const deliveryVerification = isDeliveryVerification(assignment);
|
|
512
|
+
const gatesRun = deliveryVerification
|
|
513
|
+
? await runDeliveryVerificationGates({
|
|
514
|
+
workspace: attemptWorkspace,
|
|
515
|
+
verificationCommands: liveBrief?.verification ?? [],
|
|
516
|
+
declaredVerificationCommands: liveBrief?.declared_verification ?? [],
|
|
517
|
+
acceptance: assignment.acceptance ?? [],
|
|
518
|
+
runCommand: options.runVerificationCommand,
|
|
519
|
+
})
|
|
520
|
+
: [];
|
|
306
521
|
const context = {
|
|
307
522
|
verificationCommands: liveBrief?.verification ?? [],
|
|
308
523
|
diffStat: assignment.base_commit ? await diffStatSince(attemptWorkspace, assignment.base_commit) : null,
|
|
524
|
+
gatesRun,
|
|
309
525
|
};
|
|
310
526
|
// T154: the lane resolves its model the way the authoring lane does. Without it Claude Code ran
|
|
311
527
|
// at its CLI default, which Conduit's /v1 rejects as an unknown alias (hosted-1, 2026-09-10:
|
|
@@ -345,6 +561,15 @@ export async function executeNextInvestigation(client, config, workspace, brief,
|
|
|
345
561
|
console.error(`Investigation ${assignment.id} discarded — ${check.error}: ${check.detail}`);
|
|
346
562
|
return true;
|
|
347
563
|
}
|
|
564
|
+
const stepWitness = deliveryVerification
|
|
565
|
+
? await runDeliveryVerificationProbes({
|
|
566
|
+
workspace: attemptWorkspace,
|
|
567
|
+
probes: observed.probes ?? [],
|
|
568
|
+
runCommand: options.runProbeCommand,
|
|
569
|
+
})
|
|
570
|
+
: [];
|
|
571
|
+
if (deliveryVerification)
|
|
572
|
+
await discardWorktreeWrites(attemptWorkspace);
|
|
348
573
|
await settle(client, assignment.id, {
|
|
349
574
|
lease_token: leaseToken,
|
|
350
575
|
status: "completed",
|
|
@@ -354,7 +579,10 @@ export async function executeNextInvestigation(client, config, workspace, brief,
|
|
|
354
579
|
verification: observed.verification,
|
|
355
580
|
limitations: observed.limitations,
|
|
356
581
|
criteria: observed.criteria,
|
|
357
|
-
command_failures:
|
|
582
|
+
command_failures: deliveryVerification
|
|
583
|
+
? gatesRun.filter((gate) => gate.exitCode !== 0).map(gateFailureFact).slice(0, 10)
|
|
584
|
+
: observed.command_failures,
|
|
585
|
+
step_witness: stepWitness,
|
|
358
586
|
witness: {
|
|
359
587
|
repository_fingerprint: assignment.repository_fingerprint ?? liveRepository ?? "unknown",
|
|
360
588
|
commit,
|
|
@@ -362,7 +590,9 @@ export async function executeNextInvestigation(client, config, workspace, brief,
|
|
|
362
590
|
bounded_by: boundedByForDriver(driverId),
|
|
363
591
|
files_read: observed.files_read,
|
|
364
592
|
bytes_read: observed.bytes_read,
|
|
365
|
-
commands_run:
|
|
593
|
+
commands_run: deliveryVerification
|
|
594
|
+
? settleCommandsRun(gatesRun, observed.commands_run)
|
|
595
|
+
: observed.commands_run,
|
|
366
596
|
},
|
|
367
597
|
},
|
|
368
598
|
});
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@miraland-labs/conduit-bridge",
|
|
3
|
-
"version": "0.16.
|
|
3
|
+
"version": "0.16.110",
|
|
4
4
|
"description": "Conduit Bridge CLI \u2014 join, connect, disconnect, multi-driver lanes, and run Claude Code / Codex / Cursor / OpenCode / Pi / Kiro / Antigravity / Grok Build agents for a Conduit organization",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|