@deksden-com/dd-flow-cli 0.9.0-beta.47 → 0.9.0-beta.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/dist/build-info.json +3 -3
- package/dist/services/run-controller.js +49 -9
- package/dist/services/vnext-code.js +1 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,20 @@
|
|
|
1
1
|
# @deksden-com/dd-flow-cli
|
|
2
2
|
|
|
3
|
+
## 0.9.0-beta.49
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- Treat a completed CODE gate command that returns `code_gate_failed` as terminal
|
|
8
|
+
recovery evidence, so the coordinator creates the returned repair Work instead
|
|
9
|
+
of waiting indefinitely or duplicating the aggregate gate.
|
|
10
|
+
|
|
11
|
+
## 0.9.0-beta.48
|
|
12
|
+
|
|
13
|
+
### Patch Changes
|
|
14
|
+
|
|
15
|
+
- Recheck a controller lease after a transient runtime-registry write failure before
|
|
16
|
+
dispatching more work, while retaining the hard stop for an explicit lease loss.
|
|
17
|
+
|
|
3
18
|
## 0.9.0-beta.47
|
|
4
19
|
|
|
5
20
|
### Patch Changes
|
package/dist/build-info.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"cli_package": "@deksden-com/dd-flow-cli",
|
|
3
|
-
"cli_version": "0.9.0-beta.
|
|
4
|
-
"cli_commit": "
|
|
5
|
-
"built_at": "2026-09-
|
|
3
|
+
"cli_version": "0.9.0-beta.49",
|
|
4
|
+
"cli_commit": "111629d923b54e718a4bb1466416f5089c59bed2",
|
|
5
|
+
"built_at": "2026-09-11T08:14:51.272Z",
|
|
6
6
|
"built_with_canon": {
|
|
7
7
|
"version": "4.1.0",
|
|
8
8
|
"commit": "ef349bf47cba1c987468e51d73a0dbadbd48dc1f",
|
|
@@ -111,6 +111,50 @@ export async function launchRunController(context, input) {
|
|
|
111
111
|
}
|
|
112
112
|
return await spawnController(context, requireController(context, controllerId));
|
|
113
113
|
}
|
|
114
|
+
/**
|
|
115
|
+
* A registry write can be temporarily unavailable while another runtime client
|
|
116
|
+
* owns SQLite's writer lock. That is not evidence that this controller lost
|
|
117
|
+
* its lease. Productive dispatch still requires a successful recheck before
|
|
118
|
+
* proceeding, while an explicit negative renewal remains a hard ownership
|
|
119
|
+
* loss.
|
|
120
|
+
*/
|
|
121
|
+
export function controllerLeaseMonitor(heartbeat) {
|
|
122
|
+
let lost = false;
|
|
123
|
+
let unconfirmed = null;
|
|
124
|
+
const lostError = () => new AppError("controller_lease_lost", "Managed owner lease could not be confirmed", 1);
|
|
125
|
+
return {
|
|
126
|
+
tick() {
|
|
127
|
+
try {
|
|
128
|
+
if (!heartbeat())
|
|
129
|
+
lost = true;
|
|
130
|
+
else
|
|
131
|
+
unconfirmed = null;
|
|
132
|
+
}
|
|
133
|
+
catch (error) {
|
|
134
|
+
unconfirmed = error;
|
|
135
|
+
}
|
|
136
|
+
},
|
|
137
|
+
isConfirmed() { return !lost && unconfirmed === null; },
|
|
138
|
+
assertConfirmed() {
|
|
139
|
+
if (lost)
|
|
140
|
+
throw lostError();
|
|
141
|
+
if (unconfirmed === null)
|
|
142
|
+
return;
|
|
143
|
+
try {
|
|
144
|
+
if (!heartbeat()) {
|
|
145
|
+
lost = true;
|
|
146
|
+
throw lostError();
|
|
147
|
+
}
|
|
148
|
+
unconfirmed = null;
|
|
149
|
+
}
|
|
150
|
+
catch (error) {
|
|
151
|
+
if (error instanceof AppError)
|
|
152
|
+
throw error;
|
|
153
|
+
throw new AppError("controller_lease_unconfirmed", "Managed owner lease could not be refreshed", 1, { cause: error instanceof Error ? error.message : String(error) });
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
};
|
|
157
|
+
}
|
|
114
158
|
/** No initialization, provider calls or lease takeover on the observation path. */
|
|
115
159
|
export function runControllerStatus(context, input) {
|
|
116
160
|
if (input.after !== undefined && (!Number.isSafeInteger(input.after) || input.after < 0))
|
|
@@ -134,27 +178,23 @@ export async function serveRunController(context, input) {
|
|
|
134
178
|
if (!row.process_id || !row.process_lease_token)
|
|
135
179
|
throw new AppError("controller_owner_mismatch", "Managed owner has no resource lease", 1);
|
|
136
180
|
confirmManagedProcess(context, { id: row.process_id, leaseToken: row.process_lease_token, pid: process.pid, processGroupId: process.pid, ownerPid: process.pid });
|
|
137
|
-
|
|
181
|
+
const lease = controllerLeaseMonitor(() => heartbeatManagedProcess(context, { id: row.process_id, leaseToken: row.process_lease_token }));
|
|
138
182
|
const heartbeat = setInterval(() => {
|
|
139
183
|
try {
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
if (!leaseLost) {
|
|
184
|
+
lease.tick();
|
|
185
|
+
if (lease.isConfirmed()) {
|
|
143
186
|
const guard = recoveryGuard(context, row.project_id, row.run_id);
|
|
144
187
|
if ((guard?.generation ?? 0) === row.generation && !["draining", "sealed", "resuming"].includes(guard?.status ?? ""))
|
|
145
188
|
context.db.run("UPDATE merge_requests SET dispatch_lease_expires_at = ?, updated_at = ? WHERE project_id = ? AND run_id = ? AND dispatch_owner = ? AND status = 'dispatching'", [new Date(Date.now() + 120_000).toISOString(), context.now(), row.project_id, row.run_id, row.controller_id]);
|
|
146
189
|
}
|
|
147
190
|
}
|
|
148
|
-
catch {
|
|
149
|
-
leaseLost = true;
|
|
150
|
-
}
|
|
191
|
+
catch { /* The lease monitor distinguishes renewal uncertainty from loss. */ }
|
|
151
192
|
}, 30_000);
|
|
152
193
|
heartbeat.unref();
|
|
153
194
|
const manifest = JSON.parse(row.manifest_json);
|
|
154
195
|
const state = JSON.parse(row.state_json);
|
|
155
196
|
const assertPhysicalOwner = () => {
|
|
156
|
-
|
|
157
|
-
throw new AppError("controller_lease_lost", "Managed owner lease could not be confirmed", 1);
|
|
197
|
+
lease.assertConfirmed();
|
|
158
198
|
const current = requireController(context, row.controller_id);
|
|
159
199
|
if (current.owner_pid !== process.pid || current.owner_token !== row.owner_token || !["running", "capturing", "waiting_for_user", "waiting_for_context", "waiting_for_children", "waiting_for_capacity", "waiting_for_merge"].includes(current.status))
|
|
160
200
|
throw new AppError("controller_owner_mismatch", "Controller no longer owns productive dispatch", 1);
|
|
@@ -455,7 +455,7 @@ function coordinatorPrompt(context, input) {
|
|
|
455
455
|
"",
|
|
456
456
|
"<execution_commands>",
|
|
457
457
|
"Launch only entries listed in graph.ready. Use at most the known available slot count. Every registered CODE Work runs in a fresh child Session, including a serial dependency chain; the coordinator owns dispatch and the stage conclusion, not implementation Work. Every child starts with its exact start_command and receives its complete packet from dd-flow. After a Work finishes, use the graph returned by work finish to launch newly ready Work. To refresh the parent graph yourself use the exact command: " + `${flowCommand(context)} work ls --run ${input.run.id} --ready --project-root ${JSON.stringify(input.projectRoot)} --json`,
|
|
458
|
-
"A quiet child is still running until the harness reports its turn completed, failed, cancelled or explicitly needs attention. An elapsed nominal wait, silence, or no new artifact is not an unresponsive-worker failure. Never interrupt, replace, relaunch, or stage-block a still-running child for that reason, even if an external controller asks. Long work finish and stage finish commands emit check progress on stderr. After you issue the exact CODE stage finish command, wait for that same command to return: do not inspect its PID, start a second finish command, or infer failure from quiet output. Close a disposable child only after its Work is accepted or explicitly failed/cancelled and the harness reports the turn settled.",
|
|
458
|
+
"A quiet child is still running until the harness reports its turn completed, failed, cancelled or explicitly needs attention. An elapsed nominal wait, silence, or no new artifact is not an unresponsive-worker failure. Never interrupt, replace, relaunch, or stage-block a still-running child for that reason, even if an external controller asks. Long work finish and stage finish commands emit check progress on stderr. After you issue the exact CODE stage finish command, wait for that same command to return once: a completed command with a non-zero exit and structured `code_gate_failed` output is its terminal result, not a reason to keep waiting. Read that returned error and run its repair command; do not inspect its PID, start a second finish command, or infer failure from quiet output. Close a disposable child only after its Work is accepted or explicitly failed/cancelled and the harness reports the turn settled.",
|
|
459
459
|
`A repairable engine, harness, or environment failure is not a user question. Record it without finishing CODE: ${flowCommand(context)} stage block ${input.run.id} --stage code --work ${input.rootWork.work_id} --kind <engine|harness|environment> --code <stable-code> --summary-stdin --retryable --project-root ${JSON.stringify(input.projectRoot)} --json. Repair it externally, then run the exact unblock_command returned by dd-flow and continue this same stage.`,
|
|
460
460
|
`When every CODE and repair Work is completed, write ${path.join(input.root, "code-verification.json")} using the exact contract below. Mark passed only when all accepted requirements and current-gate acceptance criteria are implemented or explicitly evidenced; list every remaining issue in unresolved. Every evidence_refs item must already exist as a relative workspace path or run://${input.run.id}/ path. Do not claim a browser or other check receipt that was not retained. Then finish: ${finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "code-verification.json"))}`,
|
|
461
461
|
"If the aggregate gate fails, do not stop after `code_gate_failed`: that rejected finish does not create a repair Work. In the same coordinator Turn, use its returned repair command with the relevant completed origin Work IDs and a concise repair objective, then stop so the runner can dispatch the newly declared repair. Do not edit invisibly in the root orchestrator.",
|