@deksden-com/dd-flow-cli 0.9.0-beta.74 → 0.9.0-beta.75
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +8 -0
- package/dist/build-info.json +3 -3
- package/dist/harness-runtime/lib/dd-droid.mjs +2 -1
- package/dist/harness-runtime/lib/delegation-instructions.d.mts +10 -0
- package/dist/harness-runtime/lib/delegation-instructions.mjs +121 -0
- package/dist/services/controller-fanout.js +5 -9
- package/dist/services/vnext-code-review.js +1 -1
- package/dist/services/vnext-code.js +1 -1
- package/dist/services/vnext-plan-review.js +1 -1
- package/dist/services/work-registry.js +13 -3
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,13 @@
|
|
|
1
1
|
# @deksden-com/dd-flow-cli
|
|
2
2
|
|
|
3
|
+
## 0.9.0-beta.75
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- Use adapter-owned native delegation instructions for Codex and the other
|
|
8
|
+
supported harnesses, preserving independent child Work ownership and clearer
|
|
9
|
+
native-session diagnostics across fan-out and recovery.
|
|
10
|
+
|
|
3
11
|
## 0.9.0-beta.64
|
|
4
12
|
|
|
5
13
|
### Patch Changes
|
package/dist/build-info.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"cli_package": "@deksden-com/dd-flow-cli",
|
|
3
|
-
"cli_version": "0.9.0-beta.
|
|
4
|
-
"cli_commit": "
|
|
5
|
-
"built_at": "2026-09-
|
|
3
|
+
"cli_version": "0.9.0-beta.75",
|
|
4
|
+
"cli_commit": "60b05d220e32542b0e4cf38234595fdad51b8e80",
|
|
5
|
+
"built_at": "2026-09-17T10:16:26.661Z",
|
|
6
6
|
"built_with_canon": {
|
|
7
7
|
"version": "4.1.1",
|
|
8
8
|
"commit": "97f811d33c212ae3497020178b1ed825c7c3ebac",
|
|
@@ -16,6 +16,7 @@ import { DroidMetadataReader, readDroidRoutingLog } from "./droid-observation.mj
|
|
|
16
16
|
const metadataReaders = new Map();
|
|
17
17
|
function metadataReader(factory) { if (!metadataReaders.has(factory)) metadataReaders.set(factory, new DroidMetadataReader(factory)); return metadataReaders.get(factory); }
|
|
18
18
|
import { confirmDaemonProcess, finishDaemonProcess, heartbeatDaemonProcess, registerDaemonProcess, stopProcessGroup } from "./managed-daemon.mjs";
|
|
19
|
+
import { renderAdapterPolicy } from "./delegation-instructions.mjs";
|
|
19
20
|
|
|
20
21
|
const run = promisify(execFile);
|
|
21
22
|
export const DROID_CONTRACT = "dd-droid-harness@1";
|
|
@@ -262,7 +263,7 @@ export class DroidRuntime {
|
|
|
262
263
|
if (this.rootId) throw new DroidError("session_already_created", "This execution daemon already owns a root Session");
|
|
263
264
|
await this.start(); const sessionId = randomUUID(); this.rootId = sessionId; const requestId = randomUUID();
|
|
264
265
|
const binding = { operationId, saved: { operation: "session.create", provider_session_id: sessionId, request_id: requestId } }; await this.saveBinding(binding);
|
|
265
|
-
const result = await this.request("droid.initialize_session", { machineId: "dd-eval", cwd: this.config.cwd, sessionId, modelId: this.config.model, reasoningEffort: this.config.reasoning, interactionMode: "auto", autonomyLevel: "high", autoRejectPermissionRequests: true, disableBuiltinSkills: true, disabledToolIds: ["AskUser"], mcpServers: [], systemPrompt: { type: "preset", preset: "droid", append:
|
|
266
|
+
const result = await this.request("droid.initialize_session", { machineId: "dd-eval", cwd: this.config.cwd, sessionId, modelId: this.config.model, reasoningEffort: this.config.reasoning, interactionMode: "auto", autonomyLevel: "high", autoRejectPermissionRequests: true, disableBuiltinSkills: true, disabledToolIds: ["AskUser"], mcpServers: [], systemPrompt: { type: "preset", preset: "droid", append: `${renderAdapterPolicy({ harness: "droid-cli" })} Use the same worker for technical capacity markers and separately declared productive Work. Never delegate an already-bound coordinator or Stage Work through Task: execute it in this root Session. A productive Task must receive the runner's exact work start command and may execute only that child Work; otherwise it must stop. User clarification belongs to the root coordinator through dd-flow stage pause, not native AskUser. At a work_fanout boundary, stop when the launcher says runner dispatches and wait for the runner's next instruction. Never use Missions.` } }, requestId);
|
|
266
267
|
if (result.sessionId !== sessionId) throw new DroidError("session_identity_mismatch", "Droid initialized a different Session");
|
|
267
268
|
this.rootId = result.sessionId; this.loaded = true; this.observedProfile = await this.observeSettings(this.rootId, result.settings, "native.initialize_session"); this.workingState = "idle"; await this.save();
|
|
268
269
|
return await this.inspect(sessionId);
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
export type DelegationWork = { work_id: string; start_command: string; launch_policy?: string };
|
|
2
|
+
export type DelegationInput = { harness: string; stage: string; capacity: number; works?: DelegationWork[] };
|
|
3
|
+
export function delegationAdapter(harness: string): string | null;
|
|
4
|
+
export function delegationContract(harness: string): string;
|
|
5
|
+
export function workerDelegationTask(input: { workId: string; startCommand: string; launchPolicy?: string }): string;
|
|
6
|
+
export function renderDelegationInstructions(input: DelegationInput): string;
|
|
7
|
+
export function renderCapacityInstructions(input: { harness: string; maximum: number }): string;
|
|
8
|
+
export function renderWaitInstructions(input: { harness: string; stage: string }): string;
|
|
9
|
+
export function renderAdapterPolicy(input: { harness: string }): string;
|
|
10
|
+
export function renderRecoveryProbeInstructions(input: { harness: string; command: string }): string;
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
const ADAPTERS = {
|
|
2
|
+
codex: {
|
|
3
|
+
name: "Codex",
|
|
4
|
+
launch: "Call the native `spawn_agent` tool in its exposed namespace once per assignment. Use the schema available in this Session: v2 accepts `task_name`, `message`, and `fork_turns: \"none\"`; v1 accepts `message` and `fork_context: false` without task_name. Never mix these schemas. Use the assignment's task_name only with v2. Keep the current model and reasoning by omitting overrides. Do not create a root Session or run the child's command yourself.",
|
|
5
|
+
wait: "The spawn result is a handle, not completion. Use the exposed native wait_agent (or wait in older schemas) and completion messages. V2 waits for mailbox updates; v1 waits on the returned agent IDs using the arguments in its tool schema. An update is not necessarily final: wait for each child's terminal result. A wait timeout leaves the same child running: continue waiting without relaunch or interruption. Release a settled child only when a release tool is exposed and its Work result has been recorded."
|
|
6
|
+
},
|
|
7
|
+
zcode: {
|
|
8
|
+
name: "ZCode",
|
|
9
|
+
launch: "Use the native `Agent` tool for every assignment in this current session. Use `TaskOutput` (or the tool's native wait operation) to await the returned child; never call a shell, create a new root session, or ask a child to create descendants.",
|
|
10
|
+
wait: "Wait on the returned native Agent task handles with `TaskOutput` (or the native wait operation) until all direct children settle. A missing or quiet result is not permission to launch a replacement.",
|
|
11
|
+
},
|
|
12
|
+
droid: {
|
|
13
|
+
name: "Droid",
|
|
14
|
+
launch: "Use the native `Task` tool for every assignment with `subagent_type: \"dd-flow-worker\"` and no complexity override. Do not use Missions, a root Session, or a shell command as a delegation substitute.",
|
|
15
|
+
wait: "Wait for each native Task result. Do not interrupt or replace a quiet Task; the coordinator remains responsible for the Stage.",
|
|
16
|
+
},
|
|
17
|
+
grok: {
|
|
18
|
+
name: "Grok",
|
|
19
|
+
launch: "Use the native `spawn_subagent` tool for each assignment in this current session. Keep the returned child handles and wait with `get_command_or_subagent_output`; do not invoke external roots or shell-based substitutes.",
|
|
20
|
+
wait: "Use `get_command_or_subagent_output` for every returned child handle before continuing. Never infer completion from silence or replace an unavailable child.",
|
|
21
|
+
},
|
|
22
|
+
opencode: {
|
|
23
|
+
name: "OpenCode",
|
|
24
|
+
launch: "Use the model-facing native `task` tool with description, prompt, and an available subagent_type that can execute the assignment. Put the assignment message in prompt. Omit task_id to create a fresh child. `subtask` is an API message part, not the model tool name. Use foreground completion, or the exposed background completion mechanism when available. Do not create a new root Session or use an external runner.",
|
|
25
|
+
wait: "Await each native task result or its background completion notification. Retain its returned task handle; never synthesize a Session ID or relaunch a quiet child."
|
|
26
|
+
},
|
|
27
|
+
antigravity: {
|
|
28
|
+
name: "Antigravity",
|
|
29
|
+
launch: "Use the harness's native subagent action in this current conversation for each assignment. Do not create a root Session, use an external runner, or ask a child to create descendants; the adapter will observe the native `subagent_info.subagents` records.",
|
|
30
|
+
wait: "Wait for the native child result and let the adapter observe its conversation ID and terminal status. Never invent or synthesize IDs, infer completion from text, or replace a quiet child.",
|
|
31
|
+
}
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
export function delegationAdapter(harness) {
|
|
35
|
+
const aliases = { codex: "codex", "codex-desktop": "codex", zcode: "zcode", "zcode-acp": "zcode", droid: "droid", "droid-cli": "droid", grok: "grok", "grok-acp": "grok", opencode: "opencode", "opencode-server": "opencode", agy: "antigravity", antigravity: "antigravity", "antigravity-cli": "antigravity" };
|
|
36
|
+
return Object.hasOwn(aliases, harness) ? aliases[harness] : null;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function requireAdapter(harness) {
|
|
40
|
+
const key = delegationAdapter(harness);
|
|
41
|
+
if (!key) throw Object.assign(new Error(`Unsupported delegation adapter: ${harness}`), { code: "delegation_contract_unsupported" });
|
|
42
|
+
return ADAPTERS[key];
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export function delegationContract(harness) {
|
|
46
|
+
requireAdapter(harness);
|
|
47
|
+
return `dd-flow/delegation/${delegationAdapter(harness)}@1`;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export function workerDelegationTask({ workId, startCommand, launchPolicy }) {
|
|
51
|
+
if (!workId || !startCommand) throw new Error("ready fan-out Work lacks its exact start command");
|
|
52
|
+
const context = launchPolicy === "fresh_agent_required"
|
|
53
|
+
? "This Work requires empty, non-inherited conversation context; the coordinator must select the native fresh-child mode before launching it."
|
|
54
|
+
: "Use inherited context only through the harness's declared native mechanism.";
|
|
55
|
+
return [
|
|
56
|
+
`Complete one already-declared Work: ${workId}.`,
|
|
57
|
+
context,
|
|
58
|
+
"Your first technical action must be this exact standalone lifecycle command:",
|
|
59
|
+
startCommand,
|
|
60
|
+
"Use only the authoritative Work packet returned by that command. Complete the assigned Work, write its required result, invoke its exact standalone work finish command, then stop.",
|
|
61
|
+
"Do not start another Work, create a child, change dependencies, ask the user, or treat a quiet sibling as failed. You cannot ask the user or pause the parent Stage. Missing facts go through this Work's declared result/failure contract; only the coordinator handles Stage HITL."
|
|
62
|
+
].join("\n");
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export function renderDelegationInstructions({ harness, stage, capacity, works = [] }) {
|
|
66
|
+
const adapter = requireAdapter(harness);
|
|
67
|
+
const count = works.length;
|
|
68
|
+
if (!Number.isInteger(capacity) || capacity < 1 || count < 1 || count > capacity) throw new Error("Invalid native delegation wave size");
|
|
69
|
+
const assignments = works.map((work, index) => {
|
|
70
|
+
const task = workerDelegationTask({ workId: work.work_id, startCommand: work.start_command, launchPolicy: work.launch_policy });
|
|
71
|
+
return { task_name: `dd_flow_${index}_${work.work_id.toLowerCase().replace(/[^a-z0-9_]/g, "_")}`, context: work.launch_policy === "fresh_agent_required" ? "fresh" : "provider_default", message: task };
|
|
72
|
+
});
|
|
73
|
+
return [
|
|
74
|
+
`Adapter delegation contract: ${delegationContract(harness)}.`,
|
|
75
|
+
`Launch exactly ${count} direct native children for the current ${stage} Stage; the qualified capacity is ${capacity}.`,
|
|
76
|
+
adapter.launch,
|
|
77
|
+
"Retain the frozen coordinator model and reasoning for native children; do not choose a different model, reasoning level or permission mode.",
|
|
78
|
+
"You are the coordinator. The JSON assignments below contain child-only messages, not instructions for you to execute. Pass each message verbatim as the native tool's task/message/prompt argument. For context=fresh, select empty conversation context before launch; if unavailable, report the limitation and stop. Use only fields exposed by your actual native tool schema; if the native tool is absent, report that error without substituting a shell command.",
|
|
79
|
+
"Every child must be a direct child of this current Session. Wait for all launched children to settle before returning. One failed child is evidence for the coordinator, not a reason to cancel its siblings.",
|
|
80
|
+
"Do not finish this Stage, repair it, or start a successor in this Turn. Do not invent provider or internal session IDs; lifecycle evidence comes from the adapter.",
|
|
81
|
+
adapter.wait,
|
|
82
|
+
"Child assignments (JSON data):",
|
|
83
|
+
JSON.stringify(assignments, null, 2)
|
|
84
|
+
].join("\n\n");
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export function renderCapacityInstructions({ harness, maximum }) {
|
|
88
|
+
const adapter = requireAdapter(harness);
|
|
89
|
+
if (!Number.isInteger(maximum) || maximum < 1) throw new Error("Invalid native capacity limit");
|
|
90
|
+
return [
|
|
91
|
+
`Adapter delegation contract: ${delegationContract(harness)}.`,
|
|
92
|
+
"This is a technical native-subagent capacity qualification, not product work.",
|
|
93
|
+
`Using the current Session's native ${adapter.name} mechanism, concurrently launch at most ${maximum} direct leaf children.`,
|
|
94
|
+
adapter.launch,
|
|
95
|
+
"Give each child a distinct number and a task_name such as capacity_1 if required by the tool. Its entire task is: return that number, without tools, files, dd-flow commands, or descendants. Use empty conversation context. Do not retry, replace, or add children after a launch refusal.",
|
|
96
|
+
adapter.wait,
|
|
97
|
+
"Wait for every child actually launched to settle, then return a compact summary."
|
|
98
|
+
].join("\n");
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export function renderWaitInstructions({ harness, stage }) {
|
|
102
|
+
const adapter = requireAdapter(harness);
|
|
103
|
+
return [
|
|
104
|
+
`Native child Work for ${stage} is still running.`,
|
|
105
|
+
adapter.wait,
|
|
106
|
+
"Do not create a Session, Work, or additional child agent. Do not cancel siblings merely because one child failed, and do not finish this Stage."
|
|
107
|
+
].join("\n");
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export function renderAdapterPolicy({ harness }) {
|
|
111
|
+
const adapter = requireAdapter(harness);
|
|
112
|
+
return [adapter.launch, adapter.wait, "Lifecycle identity and completion must come from adapter observations; never synthesize provider or internal Session IDs."].join(" ");
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export function renderRecoveryProbeInstructions({ harness, command }) {
|
|
116
|
+
const adapter = requireAdapter(harness);
|
|
117
|
+
return ["Technical native child cancellation qualification, not product work.", adapter.launch,
|
|
118
|
+
"Launch exactly one fresh direct child, task_name recovery_probe if required. Pass this JSON string as its message/prompt:",
|
|
119
|
+
JSON.stringify(`Run exactly this read-only command: ${command}\nDo not access files, use the network, modify anything, or create children.`),
|
|
120
|
+
adapter.wait, "External operator control will interrupt this tree. Do not close it early or launch replacements."].join("\n");
|
|
121
|
+
}
|
|
@@ -3,6 +3,7 @@ import { getVnextFanoutStatus, observeVnextNativeChildren, reconcileVnextFanout,
|
|
|
3
3
|
import { getFlowRunVariables } from "./runs.js";
|
|
4
4
|
import { dispatchVnextPlanReview } from "./vnext-plan-review.js";
|
|
5
5
|
import { resolveExecutionPolicy } from "./execution-policy.js";
|
|
6
|
+
import { renderDelegationInstructions } from "../harness-runtime/lib/delegation-instructions.mjs";
|
|
6
7
|
/** Stage entry acknowledges the packet before the controller issues child commands. */
|
|
7
8
|
export function controllerStageEntryPrompt(stage, command, responseFile) {
|
|
8
9
|
return [
|
|
@@ -63,15 +64,10 @@ export async function nextControllerFanout(context, input) {
|
|
|
63
64
|
if (typeof capacity !== "number" || !Number.isInteger(capacity) || capacity < 1)
|
|
64
65
|
return { kind: "wait", reason: "qualified_capacity_required" };
|
|
65
66
|
const wave = choices.slice(0, capacity).map(choice => choice.work);
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
if (!work.work_id || !work.start_command)
|
|
71
|
-
throw new AppError("fanout_contract_invalid", "Ready Work omits its trusted start command", 1);
|
|
72
|
-
return `Assign one leaf child Work ${work.work_id}. ${work.launch_policy === "fresh_agent_required" ? "Use empty, non-inherited context when supported; a new Session id alone does not prove empty context." : "Inherit context only through the harness's declared mechanism."}\nIts first technical action must be this exact standalone command:\n${work.start_command}\nUse only the Work packet returned by it. Complete its declared result and exact work finish command, then stop. Do not create children or start another Work. Missing facts go through the Work's failure/result contract; only the coordinator handles Stage HITL.`;
|
|
73
|
-
})
|
|
74
|
-
].join("\n\n"));
|
|
67
|
+
for (const work of wave)
|
|
68
|
+
if (!work.work_id || !work.start_command)
|
|
69
|
+
throw new AppError("fanout_contract_invalid", "Ready Work omits its trusted start command", 1);
|
|
70
|
+
return prompt(renderDelegationInstructions({ harness: input.harness, stage: input.stage, capacity, works: wave }));
|
|
75
71
|
}
|
|
76
72
|
if (fanout.works.running)
|
|
77
73
|
return { kind: "wait", reason: "work_receipts_pending" };
|
|
@@ -277,7 +277,7 @@ export function assertCodeReviewCoverage(groups, reviewers) {
|
|
|
277
277
|
throw new AppError("reviewer_coverage_invalid", "CODE-REVIEW requires every expected reviewer group", 2, { missing_groups: missing });
|
|
278
278
|
}
|
|
279
279
|
export function reviewerTask(home, group) { return `Read-only independent CODE review for ${group.key}. Assess every assigned aspect exactly once: ${group.aspect_ids.join(", ")}. Read the accepted PLAN, ${path.join(home, "05-code", "stage-report.json")}, and any files needed under the bounded CODE evidence root ${path.join(home, "05-code")}. Verify that universal and exclusive constraints (only, any, all, never, whole) and stated exceptions were not narrowed. For every changed mutation guarded by membership, ownership, authorization or parent lifecycle state, trace the decision to the write boundary: the predicate must remain in the write statement or the guard and write must share one explicit transaction with the needed lock. A separate earlier read is not proof of current authority or lifecycle state; report a material finding when this invariant is broken. Evidence closes a claim only when it exercises the named failure mechanism; a sequential negative test does not prove a concurrency race, and proof limits cannot waive an accepted obligation. Report only material, evidenced defects: a violated obligation or rule, direct evidence, impact, and minimum required outcome. Do not report taste, cosmetics, or untargeted refactoring. Use local finding ids FIND-001, FIND-002, and so on; dd-flow adds the Work-qualified canonical reference. Return dd-flow/code-review-result@1.`; }
|
|
280
|
-
function orchestratorPrompt(context, input) { const template = read(path.join(input.run.workspace_root, ".memory-bank", "dd-flow", "vnext", "code-review.md")); const decision = path.join(input.root, "decision.json"); const checks = acceptedCodeChecks(context, input.run.project_id, input.run.id).map(({ id, purpose }) => ({ id, purpose })); return ["<stage_identity>", `- RUN: ${input.run.id}`, `- root Work: ${input.rootWork.work_id}`, `- stage: ${stage}`, `- mode: ${input.mode}`, "</stage_identity>", "", "<trusted_runtime_context>", `- project root: ${input.projectRoot}`, `- immutable write workspace: ${input.run.workspace_root}`, `- stage workspace: ${input.root}`, `- bounded CODE evidence root: ${path.join(input.run.run_root, "05-code")}`, "CODE is already semantically verified and all declared checks passed. Do not redo CODE verification; conduct independent quality review.", "</trusted_runtime_context>", "", "<review_groups>", ...input.groups.map((group) => `- ${group.key}: ${group.aspect_ids.join(", ")}`), "</review_groups>", "", "<execution_commands>", "Each reviewer Work must run in one fresh child session
|
|
280
|
+
function orchestratorPrompt(context, input) { const template = read(path.join(input.run.workspace_root, ".memory-bank", "dd-flow", "vnext", "code-review.md")); const decision = path.join(input.root, "decision.json"); const checks = acceptedCodeChecks(context, input.run.project_id, input.run.id).map(({ id, purpose }) => ({ id, purpose })); return ["<stage_identity>", `- RUN: ${input.run.id}`, `- root Work: ${input.rootWork.work_id}`, `- stage: ${stage}`, `- mode: ${input.mode}`, "</stage_identity>", "", "<trusted_runtime_context>", `- project root: ${input.projectRoot}`, `- immutable write workspace: ${input.run.workspace_root}`, `- stage workspace: ${input.root}`, `- bounded CODE evidence root: ${path.join(input.run.run_root, "05-code")}`, "CODE is already semantically verified and all declared checks passed. Do not redo CODE verification; conduct independent quality review.", "</trusted_runtime_context>", "", "<review_groups>", ...input.groups.map((group) => `- ${group.key}: ${group.aspect_ids.join(", ")}`), "</review_groups>", "", "<execution_commands>", "Do not choose a provider delegation tool from this stage prompt. The shared controller supplies the selected adapter's native tool, exact parameters, child task, and wait contract at the Work-graph boundary. Each reviewer Work must run in one fresh child session; the coordinator must never claim a reviewer Work itself. Start each ready Work with its exact start_command from the work graph. Reviewers are read-only and must not create subagents. Do not interrupt a quiet worker.", `Ready reviewer Works: ${flowCommand(context)} work ls --run ${input.run.id} --ready --project-root ${JSON.stringify(input.projectRoot)} --json`, `Finish only after every reviewer Work settles and you have written ${decision}: ${finishCommand(context, input.run.id, input.projectRoot, decision)}`, "Reviewer results use local FIND-NNN ids. Classify them by the canonical WRK-.../FIND-NNN finding_ref returned by dd-flow. Fix P0/P1. Fix bounded safe P2 by default; defer only a legitimate P2 with a named DEF. P3 is an observation or a reasoned rejection, not automatic repair/DEF. For every disposition fix, check_refs is mandatory: choose the one or more causal checks from the accepted list below that the repair must rerun. Do not guess a CHK id or copy every check. The first successful Finish freezes this decision and creates one repair Work when needed. Run it, then call the returned next.finish_command; do not reuse an earlier invocation-id. Review is not repeated.", "If the post-repair aggregate gate fails, do not stop after `code_review_gate_failed`: that rejected finish does not create a repair Work. In the same coordinator Turn, use the returned repair command with the failed receipt, a relevant completed origin Work ID, and a concise repair objective; then stop so the runner can dispatch the new repair. Do not edit invisibly in the root orchestrator.", `Accepted repair checks: ${JSON.stringify(checks)}`, "```json", JSON.stringify({ schema_id: "dd-flow/code-review-decision@3", summary: "Evidence-backed conclusion.", findings: [{ finding_ref: "WRK-001-review/FIND-001", disposition: "fix | defer | reject | duplicate", reason: "Why this classification is correct.", check_refs: ["CHK-CAUSAL-CHECK only when disposition is fix"], def_id: "DEF-0001 only for an allowed P2 deferral", duplicate_of: "canonical finding_ref only for duplicate" }] }, null, 2), "```", "</execution_commands>", "", "<stage_instructions>", template, "</stage_instructions>", ""].join("\n"); }
|
|
281
281
|
function canonicalReviewEvidence(decision, reviewers) {
|
|
282
282
|
const known = new Set(canonicalCodeFindings(reviewers).map((item) => item.finding_ref));
|
|
283
283
|
const items = new Map(decision.findings.map((item) => [item.finding_ref, item]));
|
|
@@ -464,7 +464,7 @@ function coordinatorPrompt(context, input) {
|
|
|
464
464
|
"</code_graph>",
|
|
465
465
|
"",
|
|
466
466
|
"<execution_commands>",
|
|
467
|
-
"
|
|
467
|
+
"Do not choose a provider delegation tool from this stage prompt. At each work_fanout boundary, stop at the Work-graph boundary; the shared controller supplies the selected adapter's native tool, exact parameters, child task, and wait contract. It launches only entries listed in graph.ready and uses at most the qualified capacity. Every registered CODE Work runs in a fresh child Session, including a serial dependency chain; the coordinator owns dispatch and the stage conclusion, not implementation Work. Every child starts with its exact start_command and receives its complete packet from dd-flow. After a Work finishes, the controller uses the graph returned by work finish to launch newly ready Work. To refresh the parent graph yourself use the exact command: " + `${flowCommand(context)} work ls --run ${input.run.id} --ready --project-root ${JSON.stringify(input.projectRoot)} --json`,
|
|
468
468
|
"A quiet child is still running until the harness reports its turn completed, failed, cancelled or explicitly needs attention. An elapsed nominal wait, silence, or no new artifact is not an unresponsive-worker failure. Never interrupt, replace, relaunch, or stage-block a still-running child for that reason, even if an external controller asks. Long work finish and stage finish commands emit check progress on stderr. After you issue the exact CODE stage finish command, wait for that same command to return once: a completed command with a non-zero exit and structured `code_gate_failed` output is its terminal result, not a reason to keep waiting. Read that returned error and run its repair command; do not inspect its PID, start a second finish command, or infer failure from quiet output. Close a disposable child only after its Work is accepted or explicitly failed/cancelled and the harness reports the turn settled.",
|
|
469
469
|
`A repairable engine, harness, or environment failure is not a user question. Record it without finishing CODE: ${flowCommand(context)} stage block ${input.run.id} --stage code --work ${input.rootWork.work_id} --kind <engine|harness|environment> --code <stable-code> --summary-stdin --retryable --project-root ${JSON.stringify(input.projectRoot)} --json. Repair it externally, then run the exact unblock_command returned by dd-flow and continue this same stage.`,
|
|
470
470
|
`When every CODE and repair Work is completed, write ${path.join(input.root, "code-verification.json")} using the exact contract below. Mark passed only when all accepted requirements and current-gate acceptance criteria are implemented or explicitly evidenced; list every remaining issue in unresolved. Every evidence_refs item must already exist as a relative workspace path or run://${input.run.id}/ path. Do not claim a browser or other check receipt that was not retained. Then finish: ${finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "code-verification.json"))}`,
|
|
@@ -275,7 +275,7 @@ function orchestratorPrompt(context, input) {
|
|
|
275
275
|
const capacity = getStageFanoutCapacity(context, { projectRoot: input.projectRoot, runId: input.run.id, stage });
|
|
276
276
|
const reviewerLaunch = capacity.source === "external_policy"
|
|
277
277
|
? `After dispatch, return at the Work-graph boundary. The shared runtime launches the queued reviewer Works as separate external Sessions, at most ${capacity.available_slots} in parallel under the RUN's frozen profiles. Do not launch native children, create provider roots, run a capacity probe, or record this external limit as native capacity. Reviewers are read-only leaf workers. Continue the semantic decision only after their Work receipts settle; do not substitute a missing result or relaunch a settled reviewer.`
|
|
278
|
-
: "After dispatch,
|
|
278
|
+
: "After dispatch, stop at the Work-graph boundary. The shared controller supplies the selected adapter's native tool, exact parameters, child task, and wait contract; do not choose a provider tool from this stage prompt or substitute a shell command. It launches at most the qualified capacity at once and starts unchanged queued Works only after the current wave settles. A launch rejected before it starts is not review evidence: do not create a replacement. Each reviewer must be a genuinely fresh harness child Session; the lifecycle adapter binds that observed Session, so do not bind or supply a Session ID manually. Reviewers are read-only and must not create children. As soon as a reviewer result is accepted, release that reviewer Session when the harness permits.";
|
|
279
279
|
return ["<stage_identity>", `- RUN: ${input.run.id}`, `- Work: ${input.workId}`, "- Stage: plan-review", `- Mode: ${input.effective}`, "</stage_identity>", "", "<trusted_runtime_context>", "These facts were collected by dd-flow. Trust them; do not repeat CLI, Git, compatibility, permission or schema discovery.", `- Project root: ${input.projectRoot}`, `- Stage workspace: ${input.root}`, `- PLAN revision: ${revision}`, `- PLAN report checksum: ${input.planChecksum}`, `- Generated CODE batch checksum: ${input.batchChecksum}`, "</trusted_runtime_context>", "", workspaceContract, "", "<review_groups>", ...input.groups.map((group) => `- ${group.key}: ${group.aspect_ids.join(", ")}`), "</review_groups>", "", "<execution_commands>", `Dispatch fresh reviewers: ${dispatchCommand(context, input.run.id, input.projectRoot)}`, `If dispatch reports qualified_capacity_required, stop. The external harness controller qualifies the selected profile outside this RUN and records the resulting integer with ${capacityRecordCommand(context, input.run.id, input.projectRoot, "<qualified-native-child-count>")}. PLAN-REVIEW never launches a capacity probe.`, reviewerLaunch, "Review the execution environment of every selected check as part of its proof: a reset/fixture process, service process and client process must share the intended data and configuration world. A runtime entrypoint that can break that invariant must be explicit in one Work's task and verification and ordered before its consumer. planned_write_areas may advertise likely overlap, but do not treat them as ownership; required_read alone is not a delivery plan.", "If the final decision needs user input with no reasonable default, run this exact one-command heredoc, replacing only its placeholder body. The heredoc is the permitted stdin form; do not use cat, a pipe, a temporary file or a second shell command:", "```sh", input.pauseCommandTemplate, "```", "Ask the returned user_message, stop, then resume this same PLAN-REVIEW Work. Do not write decision.json or finish first.", `When all reviewer results are complete and every user question is resolved, classify every material finding, fix accepted findings in this same PLAN-REVIEW Work, then write ${decision} and finish: ${finishCommand(context, input.run.id, input.projectRoot, decision)}`, "Reviewer findings use local FIND-NNN ids. dd-flow exposes each finding to this coordinator as WRK-.../FIND-NNN; use that canonical finding_ref in the decision.", "A completed reviewer result with needs_changes or blocked is evidence, not the stage outcome. Classify its material findings and apply accepted fixes in this one review pass; do not start a second review automatically. Only a missing, malformed or unfinished reviewer result blocks the stage. For an accepted correction, increment PLAN revision and update only plan.json and the relevant aspect map. Do not edit or list code-work-batch.json: the CLI validates final PLAN and regenerates it. If no material correction is needed, set correction.status=not_required. The CLI checks mechanical handoff coherence; it does not prove semantic correctness.", "```json", JSON.stringify({ schema_id: "dd-flow/plan-review-decision@3", outcome: "accepted | failed | cancelled", summary: "Concise evidence-backed final decision.", finding_decisions: [{ finding_ref: "WRK-001-review/FIND-001", decision: "accepted_fix | rejected | deferred_as_DEF | requires_user | duplicate", reason: "Why." }], correction: { status: "not_required | applied", previous_plan_revision: revision, changed_paths: [], summary: "No material correction was needed, or summarize the applied correction." } }, null, 2), "```", "</execution_commands>", "", "<stage_instructions>", template, "</stage_instructions>", ""].join("\n");
|
|
280
280
|
}
|
|
281
281
|
function reviewGroups(home, workspaceRoot) {
|
|
@@ -776,16 +776,26 @@ function bindSession(context, work, run, identity, now) {
|
|
|
776
776
|
// Work ancestry and provider-tree containment are different relations. An
|
|
777
777
|
// isolated worker is logically a child Work but physically a provider root.
|
|
778
778
|
const parentSession = inferredParentSession ?? identity.parentSessionId;
|
|
779
|
+
// Stored provider ancestry uses internal keys, just like hook_events.
|
|
780
|
+
// Resolve native IDs only when rendering diagnostics; changing storage here
|
|
781
|
+
// breaks existing bindings and usage ancestry joins.
|
|
779
782
|
const providerParentSession = identity.parentSessionId;
|
|
783
|
+
const identityDetails = (sessionKey = identity.sessionId, parentKey = parentSession) => ({
|
|
784
|
+
internal_session_key: sessionKey,
|
|
785
|
+
native_session_id: sessionKey === identity.sessionId ? identity.nativeSessionId : context.db.get("SELECT provider_session_id FROM sessions WHERE project_id = ? AND session_id = ?", [work.project_id, sessionKey])?.provider_session_id ?? null,
|
|
786
|
+
internal_parent_session_key: parentKey,
|
|
787
|
+
native_parent_session_id: parentKey ? context.db.get("SELECT provider_session_id FROM sessions WHERE project_id = ? AND session_id = ?", [work.project_id, parentKey])?.provider_session_id ?? null : null,
|
|
788
|
+
harness: identity.harness
|
|
789
|
+
});
|
|
780
790
|
if (work.parent_work_id && !parentSession)
|
|
781
791
|
throw new AppError("parent_session_required", "Child Work requires a confirmed parent Work/Session link", 1, { work_id: work.work_id, parent_work_id: work.parent_work_id });
|
|
782
792
|
if (work.launch_policy === "fresh_agent_required" && (identity.sessionId === parentSession || context.db.get("SELECT 1 FROM work_sessions ws JOIN works w ON w.work_id = ws.work_id WHERE w.project_id = ? AND w.run_id = ? AND ws.session_id = ? LIMIT 1", [work.project_id, work.run_id, identity.sessionId])))
|
|
783
|
-
throw new AppError("fresh_session_required", "This Work requires a fresh Session in this RUN", 1, { work_id: work.work_id,
|
|
793
|
+
throw new AppError("fresh_session_required", "This Work requires a fresh Session in this RUN", 1, { work_id: work.work_id, ...identityDetails() });
|
|
784
794
|
const existing = context.db.get("SELECT session_id, parent_session_id, provider_parent_session_id FROM sessions WHERE project_id = ? AND session_id = ?", [work.project_id, identity.sessionId]);
|
|
785
795
|
if (existing && existing.parent_session_id && parentSession && existing.parent_session_id !== parentSession)
|
|
786
|
-
throw new AppError("session_parent_conflict", "Observed Session already has a different immutable parent", 1, {
|
|
796
|
+
throw new AppError("session_parent_conflict", "Observed Session already has a different immutable parent", 1, { work_id: work.work_id, ...identityDetails(existing.session_id, existing.parent_session_id), conflicting_internal_parent_session_key: parentSession });
|
|
787
797
|
if (existing?.provider_parent_session_id && providerParentSession && existing.provider_parent_session_id !== providerParentSession)
|
|
788
|
-
throw new AppError("provider_session_parent_conflict", "Observed provider Session already has a different immutable provider parent", 1, { session_id:
|
|
798
|
+
throw new AppError("provider_session_parent_conflict", "Observed provider Session already has a different immutable provider parent", 1, { work_id: work.work_id, ...identityDetails(existing.session_id, existing.parent_session_id), conflicting_internal_provider_parent_session_key: providerParentSession, conflicting_native_parent_session_id: context.db.get("SELECT provider_session_id FROM sessions WHERE project_id = ? AND session_id = ?", [work.project_id, providerParentSession])?.provider_session_id ?? null });
|
|
789
799
|
context.db.run(`INSERT INTO sessions (session_id, project_id, harness, provider_session_id, provider_parent_session_id, agent_id, parent_session_id, provider, model, reasoning, mode, agent_type, project_root, flow_kind, status, run_id, protocol_id, worker_id, workspace_path, continuation_policy, current_stage, next_action, last_action_hash, continuation_count, stop_reason, transcript_path, cwd, metadata_json, coverage_units_json, created_at, updated_at, stopped_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'vnext', 'active', ?, NULL, ?, ?, 'go_router', 'work', NULL, NULL, 0, NULL, ?, ?, '{}', '[]', ?, ?, NULL) ON CONFLICT(session_id, project_id) DO UPDATE SET harness = excluded.harness, provider_session_id = COALESCE(excluded.provider_session_id, provider_session_id), provider_parent_session_id = COALESCE(excluded.provider_parent_session_id, provider_parent_session_id), agent_id = COALESCE(excluded.agent_id, agent_id), parent_session_id = COALESCE(excluded.parent_session_id, parent_session_id), provider = COALESCE(excluded.provider, provider), model = COALESCE(excluded.model, model), reasoning = COALESCE(excluded.reasoning, reasoning), mode = COALESCE(excluded.mode, mode), agent_type = COALESCE(excluded.agent_type, agent_type), flow_kind = excluded.flow_kind, run_id = excluded.run_id, worker_id = excluded.worker_id, workspace_path = excluded.workspace_path, transcript_path = COALESCE(excluded.transcript_path, transcript_path), cwd = excluded.cwd, updated_at = excluded.updated_at`, [identity.sessionId, work.project_id, identity.harness, identity.providerSessionId, providerParentSession, identity.agentId, existing?.parent_session_id ?? (parentSession === identity.sessionId ? null : parentSession), identity.provider, identity.model, identity.reasoning, identity.mode, identity.agentType, run.project_root, work.run_id, work.work_id, run.workspace_root, identity.transcriptPath, run.workspace_root, now, now]);
|
|
790
800
|
reactivateBoundSession(context, work.project_id, identity.sessionId, now);
|
|
791
801
|
}
|