@sagentlab/navarch-runtime 0.1.48 → 0.1.50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -0
- package/dist/cli.cjs +20 -5
- package/dist/delivery.cjs +154 -0
- package/dist/git-worktree.cjs +14 -0
- package/dist/session.cjs +82 -26
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -837,3 +837,13 @@ Each invocation rebuilds from Git blobs at the requested commit, avoiding stale
|
|
|
837
837
|
persistent indexes. Bounds are 2,000 files, 10 MB total source, 500 KB per file,
|
|
838
838
|
100 returned nodes and 300 edges. A later persistent index can optimize large
|
|
839
839
|
repositories without changing the query format.
|
|
840
|
+
|
|
841
|
+
### Pinning a connection model
|
|
842
|
+
|
|
843
|
+
`connect --agent codex --model gpt-6-astra` saves the model alongside the local
|
|
844
|
+
machine identity. `start` and `supervise` reuse that model, including after worker
|
|
845
|
+
restarts. Their optional `--model` flag overrides the saved choice. The pin wins
|
|
846
|
+
over the project's session model for the selected adapter only; switching agent
|
|
847
|
+
types does not reuse another adapter's pin. Reconnect without `--model` to return
|
|
848
|
+
to project-controlled model selection. Older identities without a model keep
|
|
849
|
+
using the control-plane execution policy.
|
package/dist/cli.cjs
CHANGED
|
@@ -76,6 +76,14 @@ function agentFromFlag(flags) {
|
|
|
76
76
|
}
|
|
77
77
|
return value;
|
|
78
78
|
}
|
|
79
|
+
/** Saved pins only apply to the adapter they were selected for. */
|
|
80
|
+
function modelFromFlags(flags, agentType, identity) {
|
|
81
|
+
const model = flags.model ?? (identity?.agent_type === agentType ? identity.model : undefined);
|
|
82
|
+
if (model !== undefined && (!model.trim() || model === "true" || /\s/.test(model))) {
|
|
83
|
+
throw new Error("--model requires a non-empty model reference without whitespace.");
|
|
84
|
+
}
|
|
85
|
+
return model;
|
|
86
|
+
}
|
|
79
87
|
function parseArgs(argv) {
|
|
80
88
|
const [command, ...rest] = argv;
|
|
81
89
|
const flags = {};
|
|
@@ -118,6 +126,7 @@ async function registerCommand(flags) {
|
|
|
118
126
|
.filter(Boolean);
|
|
119
127
|
const ownerZone = flags["owner-zone"] ?? config.ownerZone;
|
|
120
128
|
const agentType = agentFromFlag(flags) ?? config.agentType;
|
|
129
|
+
modelFromFlags(flags, agentType); // Validate before consuming an enrollment token.
|
|
121
130
|
const client = new api_cjs_1.NavarchApiClient({ baseUrl: apiBase });
|
|
122
131
|
const result = await client.registerMachine({
|
|
123
132
|
enrollment_token: enrollmentToken,
|
|
@@ -132,6 +141,7 @@ async function registerCommand(flags) {
|
|
|
132
141
|
name,
|
|
133
142
|
api_base: apiBase,
|
|
134
143
|
agent_type: agentType,
|
|
144
|
+
model: modelFromFlags(flags, agentType),
|
|
135
145
|
});
|
|
136
146
|
// Printed exactly once. Never logged or echoed again after this point.
|
|
137
147
|
console.log("Machine registered.");
|
|
@@ -166,6 +176,7 @@ async function connectCommand(flags) {
|
|
|
166
176
|
.map((s) => s.trim())
|
|
167
177
|
.filter(Boolean);
|
|
168
178
|
const agentType = agentFromFlag(flags) ?? config.agentType;
|
|
179
|
+
modelFromFlags(flags, agentType); // Validate before consuming an enrollment token.
|
|
169
180
|
const client = new api_cjs_1.NavarchApiClient({ baseUrl: apiBase });
|
|
170
181
|
const result = await client.connectMachine({
|
|
171
182
|
enrollment_token: enrollmentToken,
|
|
@@ -180,6 +191,7 @@ async function connectCommand(flags) {
|
|
|
180
191
|
name,
|
|
181
192
|
api_base: apiBase,
|
|
182
193
|
agent_type: agentType,
|
|
194
|
+
model: modelFromFlags(flags, agentType),
|
|
183
195
|
});
|
|
184
196
|
// Printed exactly once. Never logged or echoed again after this point.
|
|
185
197
|
console.log("Machine connected.");
|
|
@@ -195,7 +207,7 @@ async function startCommand(flags) {
|
|
|
195
207
|
// time, and finally the backwards-compatible Claude Code default.
|
|
196
208
|
const agentType = agentFromFlag(flags) ??
|
|
197
209
|
(process.env.NAVARCH_AGENT ? baseConfig.agentType : identity.agent_type ?? baseConfig.agentType);
|
|
198
|
-
const config = { ...baseConfig, agentType };
|
|
210
|
+
const config = { ...baseConfig, agentType, model: modelFromFlags(flags, agentType, identity) };
|
|
199
211
|
const api = new api_cjs_1.NavarchApiClient({ baseUrl: identity.api_base, token: identity.token });
|
|
200
212
|
const capacity = new capacity_cjs_1.CapacityTracker(config.maxSessions);
|
|
201
213
|
const activeSessionIds = new Set();
|
|
@@ -277,6 +289,9 @@ async function superviseCommand(flags) {
|
|
|
277
289
|
const agentType = agentFromFlag(flags) ??
|
|
278
290
|
(process.env.NAVARCH_AGENT ? config.agentType : identity.agent_type ?? config.agentType);
|
|
279
291
|
const workerArgs = ["--agent", agentType];
|
|
292
|
+
const model = modelFromFlags(flags, agentType, identity);
|
|
293
|
+
if (model)
|
|
294
|
+
workerArgs.push("--model", model);
|
|
280
295
|
// Pin the identity for this supervisor's lifetime. Without this snapshot, a
|
|
281
296
|
// replacement worker rereads machine.json after an automatic update and can
|
|
282
297
|
// silently become a different machine if another terminal reused the same
|
|
@@ -332,11 +347,11 @@ function helpText() {
|
|
|
332
347
|
|
|
333
348
|
Usage:
|
|
334
349
|
navarch-runtime register --token <enrollment-token> --name <machine-name> \\
|
|
335
|
-
[--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp] [--capabilities a,b] [--max-sessions N] [--owner-zone z] [--api-base url]
|
|
350
|
+
[--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp] [--model <model>] [--capabilities a,b] [--max-sessions N] [--owner-zone z] [--api-base url]
|
|
336
351
|
navarch-runtime connect --token <enrollment-token> --name <machine-name> \\
|
|
337
|
-
[--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp] [--project <project-id>] [--capabilities a,b] [--max-sessions N] [--api-base url]
|
|
338
|
-
navarch-runtime start [--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp]
|
|
339
|
-
navarch-runtime supervise [--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp]
|
|
352
|
+
[--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp] [--model <model>] [--project <project-id>] [--capabilities a,b] [--max-sessions N] [--api-base url]
|
|
353
|
+
navarch-runtime start [--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp] [--model <model>]
|
|
354
|
+
navarch-runtime supervise [--config-dir <path>] [--agent claude-code|codex|gemini|opencode|acp] [--model <model>]
|
|
340
355
|
navarch-runtime doctor [--config-dir <path>]
|
|
341
356
|
navarch-runtime code-graph --query <symbol-or-file> [--mode search|callers|callees|impact] [--commit HEAD] [--depth 1..5] [--repo <path>]
|
|
342
357
|
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
// Delivery verification for BYO sessions.
|
|
3
|
+
//
|
|
4
|
+
// `mapExitCondition()` decides a lease outcome from how the agent process
|
|
5
|
+
// ended: a zero exit with no error events is reported `completed`. That is
|
|
6
|
+
// evidence the process finished cleanly, not evidence the task was delivered.
|
|
7
|
+
// session.cts already looks up the head-branch pull request, but only to push
|
|
8
|
+
// onto `evidenceUrls` — the outcome was settled before the lookup ran, so a
|
|
9
|
+
// session that committed work and never opened (or never pushed) a pull
|
|
10
|
+
// request still closed its lease as a success.
|
|
11
|
+
//
|
|
12
|
+
// This module reads what the worktree actually holds after a turn and turns it
|
|
13
|
+
// into a verdict. The hosted executor runs the same rules from inside its
|
|
14
|
+
// sandbox (worker/hosted-agent/delivery.ts); the two differ only in how they
|
|
15
|
+
// reach the checkout, which is why the rules and their justifications live in
|
|
16
|
+
// both places rather than in a shared package the Worker cannot import.
|
|
17
|
+
//
|
|
18
|
+
// Direction of failure: only positive evidence of stranded work gates. An
|
|
19
|
+
// unreadable checkout, a failed pull-request lookup, or an unresolvable base
|
|
20
|
+
// ref all produce `unknown`, which behaves exactly as this code did before the
|
|
21
|
+
// gate existed. A verification mechanism that fails good sessions when its own
|
|
22
|
+
// dependency is unavailable is worse than no mechanism.
|
|
23
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
24
|
+
exports.probeWorktreeDelivery = probeWorktreeDelivery;
|
|
25
|
+
exports.evaluateDelivery = evaluateDelivery;
|
|
26
|
+
exports.renderDeliveryCorrectionPrompt = renderDeliveryCorrectionPrompt;
|
|
27
|
+
const sandbox_cjs_1 = require("./sandbox.cjs");
|
|
28
|
+
const GIT_TIMEOUT_MS = 30_000;
|
|
29
|
+
/**
|
|
30
|
+
* Reads the worktree's delivery state.
|
|
31
|
+
*
|
|
32
|
+
* Every field degrades independently: a git invocation that fails leaves its
|
|
33
|
+
* field "unknown" (-1, or `dirty: false`) rather than throwing, because a
|
|
34
|
+
* probe that cannot describe one property should still report the others.
|
|
35
|
+
* `dirty` is deliberately conservative — an unreadable `git status` reports a
|
|
36
|
+
* clean tree, so the gate under-fires rather than failing a good session.
|
|
37
|
+
*/
|
|
38
|
+
async function probeWorktreeDelivery(args) {
|
|
39
|
+
const runner = args.runner ?? sandbox_cjs_1.nodeCommandRunner;
|
|
40
|
+
const git = async (gitArgs) => {
|
|
41
|
+
try {
|
|
42
|
+
const result = await runner.run("git", ["-C", args.worktreePath, ...gitArgs], {
|
|
43
|
+
timeoutMs: GIT_TIMEOUT_MS,
|
|
44
|
+
});
|
|
45
|
+
return result.code === 0 ? result.stdout : null;
|
|
46
|
+
}
|
|
47
|
+
catch {
|
|
48
|
+
return null;
|
|
49
|
+
}
|
|
50
|
+
};
|
|
51
|
+
const count = (raw) => {
|
|
52
|
+
if (raw === null)
|
|
53
|
+
return -1;
|
|
54
|
+
const parsed = Number(raw.trim());
|
|
55
|
+
return Number.isInteger(parsed) && parsed >= 0 ? parsed : -1;
|
|
56
|
+
};
|
|
57
|
+
const [branchRaw, statusRaw, aheadRaw, unpushedRaw] = await Promise.all([
|
|
58
|
+
git(["rev-parse", "--abbrev-ref", "HEAD"]),
|
|
59
|
+
git(["status", "--porcelain"]),
|
|
60
|
+
args.baseRef ? git(["rev-list", "--count", `${args.baseRef}..HEAD`]) : Promise.resolve(null),
|
|
61
|
+
// Commits no origin ref carries. Independent of the base ref, so it still
|
|
62
|
+
// answers "was this pushed?" when the base cannot be resolved at all.
|
|
63
|
+
git(["rev-list", "--count", "HEAD", "--not", "--remotes=origin"]),
|
|
64
|
+
]);
|
|
65
|
+
const branch = branchRaw?.trim() ?? "";
|
|
66
|
+
return {
|
|
67
|
+
branch: branch && branch !== "HEAD" ? branch : null,
|
|
68
|
+
dirty: statusRaw !== null && statusRaw.trim().length > 0,
|
|
69
|
+
commitsAhead: count(aheadRaw),
|
|
70
|
+
unpushedCommits: count(unpushedRaw),
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Maps a probe to a verdict.
|
|
75
|
+
*
|
|
76
|
+
* Dirty files are stranded even if a PR carries the committed HEAD. Once
|
|
77
|
+
* the tree is clean, the SHA-verified PR lookup is stronger evidence than
|
|
78
|
+
* potentially stale remote-tracking refs. Otherwise ordering runs from most to least
|
|
79
|
+
* certain — uncommitted edits, then commits that were never pushed, then
|
|
80
|
+
* pushed commits with no pull request — so the reason a session is failed
|
|
81
|
+
* names the earliest point the delivery chain broke.
|
|
82
|
+
*/
|
|
83
|
+
function evaluateDelivery(probe) {
|
|
84
|
+
if (probe.dirty) {
|
|
85
|
+
return {
|
|
86
|
+
status: "incomplete",
|
|
87
|
+
reason: "the worktree still has uncommitted changes, so the work was never committed or pushed",
|
|
88
|
+
};
|
|
89
|
+
}
|
|
90
|
+
if (probe.pullRequestUrl)
|
|
91
|
+
return { status: "delivered", pullRequestUrl: probe.pullRequestUrl };
|
|
92
|
+
const hasWork = probe.commitsAhead > 0 || probe.unpushedCommits > 0;
|
|
93
|
+
if (!hasWork) {
|
|
94
|
+
// Both counts unknown: the checkout could not be read at all.
|
|
95
|
+
if (probe.commitsAhead < 0 && probe.unpushedCommits < 0) {
|
|
96
|
+
return { status: "unknown", reason: "the worktree's commit state could not be read" };
|
|
97
|
+
}
|
|
98
|
+
// A base that could not be resolved hides commits that a resumed task
|
|
99
|
+
// branch already carried, so only a confirmed zero means "no work".
|
|
100
|
+
if (probe.commitsAhead < 0 && probe.unpushedCommits === 0) {
|
|
101
|
+
return { status: "unknown", reason: "the branch's base ref could not be resolved" };
|
|
102
|
+
}
|
|
103
|
+
return { status: "no_changes" };
|
|
104
|
+
}
|
|
105
|
+
if (probe.unpushedCommits > 0) {
|
|
106
|
+
const plural = probe.unpushedCommits === 1 ? "commit" : "commits";
|
|
107
|
+
return {
|
|
108
|
+
status: "incomplete",
|
|
109
|
+
reason: `${probe.unpushedCommits} ${plural} on ${describeBranch(probe.branch)} ` +
|
|
110
|
+
"were never pushed, so the work exists only in this session's worktree",
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
// Pushed, but nothing carries it. Only gate when GitHub actually answered.
|
|
114
|
+
if (probe.pullRequestLookupFailed) {
|
|
115
|
+
return { status: "unknown", reason: "the pull request lookup could not run, so delivery cannot be confirmed" };
|
|
116
|
+
}
|
|
117
|
+
const plural = probe.commitsAhead === 1 ? "commit" : "commits";
|
|
118
|
+
return {
|
|
119
|
+
status: "incomplete",
|
|
120
|
+
reason: `${probe.commitsAhead} ${plural} were pushed to ${describeBranch(probe.branch)} ` +
|
|
121
|
+
"but no pull request was opened for that branch",
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
function describeBranch(branch) {
|
|
125
|
+
return branch ? `\`${branch}\`` : "a detached HEAD";
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Prompt for the turn that follows an `incomplete` verdict.
|
|
129
|
+
*
|
|
130
|
+
* session.cts already re-runs the agent in the same worktree for mid-flight
|
|
131
|
+
* human guidance and for remediable control-plane rejections; this is the same
|
|
132
|
+
* shape of turn for the same reason — the partial work on disk is the state
|
|
133
|
+
* worth continuing from, so the prompt points at it rather than restating the
|
|
134
|
+
* task.
|
|
135
|
+
*/
|
|
136
|
+
function renderDeliveryCorrectionPrompt(args) {
|
|
137
|
+
return [
|
|
138
|
+
args.originalPrompt.trim(),
|
|
139
|
+
"",
|
|
140
|
+
"## Delivery is incomplete",
|
|
141
|
+
"",
|
|
142
|
+
`Your previous turn ended without delivering the task: ${args.reason}.`,
|
|
143
|
+
"",
|
|
144
|
+
`This is delivery attempt ${args.attempt} of ${args.maxAttempts}. You are in the same worktree as before,`,
|
|
145
|
+
"so your earlier edits, commits, and branch are all still here — start with `git status` and",
|
|
146
|
+
"`git log --oneline` and continue from there rather than redoing the work.",
|
|
147
|
+
"",
|
|
148
|
+
"Finish the delivery chain: commit the work, push the branch, and open the pull request.",
|
|
149
|
+
"If part of the task genuinely cannot be completed, still deliver what does work as a pull request",
|
|
150
|
+
"and state plainly in the description what is missing and why.",
|
|
151
|
+
"",
|
|
152
|
+
"This session's worktree is deleted when the session ends. Work that is not pushed is lost.",
|
|
153
|
+
].join("\n");
|
|
154
|
+
}
|
package/dist/git-worktree.cjs
CHANGED
|
@@ -304,6 +304,20 @@ class GitWorktree {
|
|
|
304
304
|
this.taskOwnershipTrailer,
|
|
305
305
|
], false);
|
|
306
306
|
}
|
|
307
|
+
/**
|
|
308
|
+
* Remote-tracking ref a delivery is measured against: the repository's
|
|
309
|
+
* default branch. Public so completion can count the task's commits without
|
|
310
|
+
* repeating default-branch discovery, and null-safe so an unreachable remote
|
|
311
|
+
* leaves the delivery gate "unknown" rather than failing the session.
|
|
312
|
+
*/
|
|
313
|
+
async resolveDeliveryBase() {
|
|
314
|
+
try {
|
|
315
|
+
return await this.resolveStartRef();
|
|
316
|
+
}
|
|
317
|
+
catch {
|
|
318
|
+
return null;
|
|
319
|
+
}
|
|
320
|
+
}
|
|
307
321
|
/**
|
|
308
322
|
* Resolves the remote-tracking ref new session branches start from, or null
|
|
309
323
|
* when the remote has no branches at all (a freshly provisioned empty repo).
|
package/dist/session.cjs
CHANGED
|
@@ -21,6 +21,7 @@ const logger_cjs_1 = require("./logger.cjs");
|
|
|
21
21
|
const git_worktree_cjs_1 = require("./git-worktree.cjs");
|
|
22
22
|
const worktree_janitor_cjs_1 = require("./worktree-janitor.cjs");
|
|
23
23
|
const github_pr_cjs_1 = require("./github-pr.cjs");
|
|
24
|
+
const delivery_cjs_1 = require("./delivery.cjs");
|
|
24
25
|
const worktree_guard_cjs_1 = require("./worktree-guard.cjs");
|
|
25
26
|
const adapter_watchdog_cjs_1 = require("./adapter-watchdog.cjs");
|
|
26
27
|
const lease_heartbeat_cjs_1 = require("./lease-heartbeat.cjs");
|
|
@@ -29,7 +30,29 @@ const adapter_capacity_cjs_1 = require("./adapter-capacity.cjs");
|
|
|
29
30
|
const MCP_CONFIG_FILENAME = "mcp-config.json";
|
|
30
31
|
const GIT_CREDENTIAL_HELPER_FILENAME = "git-credential-navarch.cjs";
|
|
31
32
|
const COMPLETION_REMEDIATION_RETRIES = 2;
|
|
33
|
+
/**
|
|
34
|
+
* Turns the delivery gate (delivery.cjs) may spend on a session that finished
|
|
35
|
+
* cleanly but left work undelivered. Counted separately from the
|
|
36
|
+
* control-plane remediation retries above: the two failures are independent,
|
|
37
|
+
* and one must not consume the other's budget.
|
|
38
|
+
*/
|
|
39
|
+
const DELIVERY_CORRECTION_RETRIES = 1;
|
|
32
40
|
const log = (0, logger_cjs_1.createLogger)("session");
|
|
41
|
+
function resolveSessionExecution(config, claimed) {
|
|
42
|
+
const runtime = claimed.runtime ?? config.agentType;
|
|
43
|
+
const execution = claimed.context_bundle.execution ?? {
|
|
44
|
+
profile: claimed.task.execution_profile ?? "standard",
|
|
45
|
+
model: runtime === "codex" ? "gpt-6-astra"
|
|
46
|
+
: runtime === "gemini" ? "auto"
|
|
47
|
+
: runtime === "opencode" || runtime === "acp" ? "default"
|
|
48
|
+
: "claude-opus-5",
|
|
49
|
+
reasoning_effort: "medium",
|
|
50
|
+
};
|
|
51
|
+
return {
|
|
52
|
+
...execution,
|
|
53
|
+
...(runtime === config.agentType && config.model ? { model: config.model } : {}),
|
|
54
|
+
};
|
|
55
|
+
}
|
|
33
56
|
/**
|
|
34
57
|
* Runs one claimed task end to end (implementation-plan.md WP-07):
|
|
35
58
|
* 1. write the prompt file
|
|
@@ -62,19 +85,9 @@ async function runSession(deps, claimed, sessionId) {
|
|
|
62
85
|
// explicit pre-start signal. Otherwise it waits for expiry and bypasses
|
|
63
86
|
// the repeat/backoff policy entirely.
|
|
64
87
|
const { api, config } = deps;
|
|
65
|
-
const { lease_id: leaseId, task
|
|
88
|
+
const { lease_id: leaseId, task } = claimed;
|
|
66
89
|
const runtime = claimed.runtime ?? config.agentType;
|
|
67
|
-
const execution =
|
|
68
|
-
profile: task.execution_profile ?? "standard",
|
|
69
|
-
model: runtime === "codex"
|
|
70
|
-
? "gpt-6-astra"
|
|
71
|
-
: runtime === "gemini"
|
|
72
|
-
? "auto"
|
|
73
|
-
: runtime === "opencode" || runtime === "acp"
|
|
74
|
-
? "default"
|
|
75
|
-
: "claude-opus-5",
|
|
76
|
-
reasoning_effort: "medium",
|
|
77
|
-
};
|
|
90
|
+
const execution = resolveSessionExecution(config, claimed);
|
|
78
91
|
const crashDetail = describeCompletionError(err).slice(0, 1800);
|
|
79
92
|
const failureSummary = `Session crashed: ${crashDetail}`;
|
|
80
93
|
log.error(`session ${leaseId} threw before adapter start: ${crashDetail}`);
|
|
@@ -102,17 +115,7 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
|
|
|
102
115
|
const { api, config } = deps;
|
|
103
116
|
const { lease_id: leaseId, task, context_bundle: bundle } = claimed;
|
|
104
117
|
const runtime = claimed.runtime ?? config.agentType;
|
|
105
|
-
const execution =
|
|
106
|
-
profile: task.execution_profile ?? "standard",
|
|
107
|
-
model: runtime === "codex"
|
|
108
|
-
? "gpt-6-astra"
|
|
109
|
-
: runtime === "gemini"
|
|
110
|
-
? "auto"
|
|
111
|
-
: runtime === "opencode" || runtime === "acp"
|
|
112
|
-
? "default"
|
|
113
|
-
: "claude-opus-5",
|
|
114
|
-
reasoning_effort: "medium",
|
|
115
|
-
};
|
|
118
|
+
const execution = resolveSessionExecution(config, claimed);
|
|
116
119
|
// Isolation posture reported on every completion so `sessions` records what
|
|
117
120
|
// a run actually executed under (#660). Derived from config alone, so the
|
|
118
121
|
// pre-start failure paths below can stamp it too. Host mode records
|
|
@@ -417,6 +420,7 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
|
|
|
417
420
|
const maxRuntimeMs = task.max_runtime_ms ?? config.sessionTimeoutMs;
|
|
418
421
|
let nextPrompt = null;
|
|
419
422
|
let completionRemediationRetries = 0;
|
|
423
|
+
let deliveryCorrectionRetries = 0;
|
|
420
424
|
while (true) {
|
|
421
425
|
// Guidance can arrive while the worktree/sandbox is being prepared.
|
|
422
426
|
// It is already included in deliveredGuidance, so clear the pending
|
|
@@ -504,8 +508,10 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
|
|
|
504
508
|
// Preserve the original exact-branch lookup if HEAD cannot be read.
|
|
505
509
|
log.warn(`worktree HEAD lookup failed for ${leaseId}: ${String(err)}`);
|
|
506
510
|
}
|
|
511
|
+
let pullRequestUrl = null;
|
|
512
|
+
let pullRequestLookupFailed = false;
|
|
507
513
|
try {
|
|
508
|
-
|
|
514
|
+
pullRequestUrl = await (0, github_pr_cjs_1.findHeadBranchPullRequestUrl)({
|
|
509
515
|
repository: bundle.repository?.full_name ?? task.repo,
|
|
510
516
|
headBranch,
|
|
511
517
|
headSha,
|
|
@@ -514,14 +520,61 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
|
|
|
514
520
|
allowShaFallback: allowBranchRename,
|
|
515
521
|
githubToken,
|
|
516
522
|
});
|
|
517
|
-
if (
|
|
518
|
-
mapping.evidenceUrls.push(
|
|
523
|
+
if (pullRequestUrl)
|
|
524
|
+
mapping.evidenceUrls.push(pullRequestUrl);
|
|
519
525
|
}
|
|
520
526
|
catch (err) {
|
|
521
527
|
// Evidence discovery is best-effort: a GitHub outage or token scope
|
|
522
528
|
// mismatch must not turn an otherwise valid completion into a crash.
|
|
529
|
+
// The delivery gate below reads this flag and declines to gate rather
|
|
530
|
+
// than failing a session whose pull request it simply could not see.
|
|
531
|
+
pullRequestLookupFailed = true;
|
|
523
532
|
log.warn(`head-branch PR lookup failed for ${leaseId}: ${String(err)}`);
|
|
524
533
|
}
|
|
534
|
+
// -- delivery gate -----------------------------------------------------
|
|
535
|
+
// A clean exit is not delivery. Read what the worktree actually holds
|
|
536
|
+
// and, when work was started but never shipped, spend a correction turn
|
|
537
|
+
// in the same worktree before failing the lease. See delivery.cjs for
|
|
538
|
+
// why only positive evidence gates.
|
|
539
|
+
let deliveryFailureSummary = null;
|
|
540
|
+
if (mapping.leaseOutcome === "completed" && !leaseLost) {
|
|
541
|
+
let delivery;
|
|
542
|
+
try {
|
|
543
|
+
const probe = await (0, delivery_cjs_1.probeWorktreeDelivery)({
|
|
544
|
+
worktreePath: gitWorktree.worktreePath,
|
|
545
|
+
baseRef: await gitWorktree.resolveDeliveryBase(),
|
|
546
|
+
});
|
|
547
|
+
delivery = (0, delivery_cjs_1.evaluateDelivery)({ ...probe, pullRequestUrl, pullRequestLookupFailed });
|
|
548
|
+
}
|
|
549
|
+
catch (err) {
|
|
550
|
+
delivery = { status: "unknown", reason: `the delivery probe failed: ${String(err)}` };
|
|
551
|
+
}
|
|
552
|
+
if (delivery.status === "unknown") {
|
|
553
|
+
log.warn(`delivery for ${leaseId} could not be verified: ${delivery.reason}; completing anyway.`);
|
|
554
|
+
}
|
|
555
|
+
if (delivery.status === "incomplete") {
|
|
556
|
+
if (deliveryCorrectionRetries < DELIVERY_CORRECTION_RETRIES) {
|
|
557
|
+
deliveryCorrectionRetries += 1;
|
|
558
|
+
log.warn(`session ${leaseId} finished without delivering (${delivery.reason}); ` +
|
|
559
|
+
`restarting agent turn ${deliveryCorrectionRetries}/${DELIVERY_CORRECTION_RETRIES} in the same worktree.`);
|
|
560
|
+
nextPrompt = (0, delivery_cjs_1.renderDeliveryCorrectionPrompt)({
|
|
561
|
+
originalPrompt: runPrompt,
|
|
562
|
+
reason: delivery.reason,
|
|
563
|
+
attempt: deliveryCorrectionRetries,
|
|
564
|
+
maxAttempts: DELIVERY_CORRECTION_RETRIES,
|
|
565
|
+
});
|
|
566
|
+
continue;
|
|
567
|
+
}
|
|
568
|
+
deliveryFailureSummary =
|
|
569
|
+
`Session ended without delivering the task: ${delivery.reason}. ` +
|
|
570
|
+
`${deliveryCorrectionRetries} correction turn(s) did not recover it. ` +
|
|
571
|
+
"The session worktree is deleted at teardown, so the work is gone; the task needs another attempt.";
|
|
572
|
+
log.warn(`failing ${leaseId}: ${deliveryFailureSummary}`);
|
|
573
|
+
mapping.leaseOutcome = "failed";
|
|
574
|
+
mapping.exitStatus = "failed";
|
|
575
|
+
mapping.reportSummary = `${deliveryFailureSummary}\n\n---\n\n${mapping.reportSummary}`;
|
|
576
|
+
}
|
|
577
|
+
}
|
|
525
578
|
const knownSecrets = registry.list();
|
|
526
579
|
if (mapping.leaseOutcome === "failed") {
|
|
527
580
|
log.warn(`adapter failed for ${leaseId}: ${(0, redact_cjs_1.redactText)(mapping.reportSummary, knownSecrets)}`);
|
|
@@ -554,6 +607,9 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
|
|
|
554
607
|
? verificationFailureReport(task.task_type, redactedReport)
|
|
555
608
|
: redactedReport,
|
|
556
609
|
evidence_urls: mapping.evidenceUrls,
|
|
610
|
+
...(deliveryFailureSummary
|
|
611
|
+
? { failure_summary: (0, redact_cjs_1.redactText)(deliveryFailureSummary, knownSecrets) }
|
|
612
|
+
: {}),
|
|
557
613
|
cost: {
|
|
558
614
|
...(result.tokensIn !== undefined ? { tokens_in: result.tokensIn } : {}),
|
|
559
615
|
...(result.tokensOut !== undefined ? { tokens_out: result.tokensOut } : {}),
|
package/package.json
CHANGED