@deksden-com/dd-flow-cli 0.9.0-beta.116 → 0.9.0-beta.118

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,17 @@
1
1
  # @deksden-com/dd-flow-cli
2
2
 
3
+ ## 0.9.0-beta.118
4
+
5
+ ### Patch Changes
6
+
7
+ - Admit the tested ZCode bridge fix for managed turns that remain active after a frozen event watermark, and allow 30 seconds for Antigravity's version check during doctor and daemon startup.
8
+
9
+ ## 0.9.0-beta.117
10
+
11
+ ### Patch Changes
12
+
13
+ - Preserve Grok child model facts, replay owned Codex tool events, and let CODE collect aggregate-check evidence before semantic acceptance.
14
+
3
15
  ## 0.9.0-beta.116
4
16
 
5
17
  ### Patch Changes
@@ -1,8 +1,8 @@
1
1
  {
2
2
  "cli_package": "@deksden-com/dd-flow-cli",
3
- "cli_version": "0.9.0-beta.116",
4
- "cli_commit": "c30f54c04930f51d7982c49369a96bca51eabe65",
5
- "built_at": "2026-09-27T20:17:11.827Z",
3
+ "cli_version": "0.9.0-beta.118",
4
+ "cli_commit": "19b1d443644710c4d17e02a7175a5e556e7d8f8e",
5
+ "built_at": "2026-09-28T12:22:30.459Z",
6
6
  "built_with_canon": {
7
7
  "version": "4.1.1",
8
8
  "commit": "678daa038287c948ada5b2d785a6dcc925c7b891",
@@ -11,7 +11,7 @@ import { agyWorkspaceHooks } from "./agy-hooks.mjs";
11
11
  import { durableDaemonDispatch } from "./daemon-operations.mjs";
12
12
  import path from "node:path";
13
13
  import { observeModel } from "./model-observations.mjs";
14
- import { AgyError, assertProfile, executable, observedProfile, runAgy, usageSnapshot } from "./dd-agy.mjs";
14
+ import { AGY_VERSION_TIMEOUT_MS, AgyError, assertProfile, executable, observedProfile, runAgy, usageSnapshot } from "./dd-agy.mjs";
15
15
  import { prepareRuntimeOwner, cleanupFailedStart, cleanupFailedDaemonStart, confirmDaemonProcess, confirmDaemonStopped, finishDaemonProcess, scheduleDaemonProcessHeartbeat, registerDaemonProcess, stopProcessGroup } from "./managed-daemon.mjs";
16
16
 
17
17
  const REQUEST_SCHEMA = "dd-agy/daemon-request@1", RESPONSE_SCHEMA = "dd-agy/daemon-response@1", STATE_SCHEMA = "dd-agy/daemon-state@1";
@@ -189,7 +189,7 @@ export async function startDaemon(options) {
189
189
  if (previous?.shutdown_state === "clean") await authorizeRetainedDaemonResume(paths.dir, previous, options.sessionId, config);
190
190
  if (previous?.shutdown_state === "running" && previous.active_tree) throw new DaemonError("invalid_harness_crash", "previous Antigravity daemon died with an active or unproven tree");
191
191
  await removeSocket(paths); await prepare(paths, config);
192
- const version = (await runAgy(config.bin, ["--version"], { cwd: config.cwd, env: sanitizedEnv(config), timeoutMs: 10_000 })).stdout.trim();
192
+ const version = (await runAgy(config.bin, ["--version"], { cwd: config.cwd, env: sanitizedEnv(config), timeoutMs: AGY_VERSION_TIMEOUT_MS })).stdout.trim();
193
193
  const resourceProcess = await registerDaemonProcess(config, { kind: "agy-daemon", owner: `agy:${path.basename(paths.dir)}`, operation: `agy-daemon:${path.basename(paths.dir)}`, stdout: paths.log, stderr: paths.log });
194
194
  const retained = retainedAgyState(previous, config.resumeSessionId);
195
195
  await writeJson(paths.state, { schema_id: STATE_SCHEMA, daemon_id: config.daemonId, shutdown_state: "starting", active_tree: false, unclaimed_activity: retained.unclaimed_activity ?? false, retained_tree_settlement: retained.retained_tree_settlement ?? null, versions: { agy: version }, config, sessions: [], session_observations: retained.session_observations ?? [], descendants: retained.descendants ?? [], tool_calls: retained.tool_calls ?? [], completed_tool_steps: retained.completed_tool_steps ?? [], tool_identity_complete: retained.tool_identity_complete !== false, turn_generation: retained.turn_generation ?? 0, last_result: retained.last_result ?? null, resource_process: resourceProcess });
@@ -6,6 +6,7 @@ import path from "node:path";
6
6
  import { checkObservedProfile } from "./model-observations.mjs";
7
7
 
8
8
  const DEFAULT_MODEL = "gemini-3.1-pro-high";
9
+ export const AGY_VERSION_TIMEOUT_MS = 30_000;
9
10
 
10
11
  export class AgyError extends Error {
11
12
  constructor(code, message, retryable = false, details) { super(message); this.code = code; this.retryable = retryable; this.details = details; }
@@ -50,7 +51,7 @@ export function assertProfile(requested, observed) { return checkObservedProfile
50
51
  export async function doctor(options = {}) {
51
52
  const bin = options.bin ?? process.env.DD_AGY_BIN ?? "agy", temporary = await mkdtemp(path.join(os.tmpdir(), "dd-agy-doctor-"));
52
53
  try {
53
- const version = (await runAgy(bin, ["--version"], { timeoutMs: 10_000 })).stdout.trim();
54
+ const version = (await runAgy(bin, ["--version"], { timeoutMs: AGY_VERSION_TIMEOUT_MS })).stdout.trim();
54
55
  const models = await runAgy(bin, [`--gemini_dir=${temporary}`, "--app_data_dir=runtime", "models"], { timeoutMs: options.timeoutMs ?? 30_000 });
55
56
  const modelIds = models.stdout.split(/\r?\n/).map(line => line.split(/\s+/)[0]).filter(Boolean);
56
57
  if (!modelIds.includes(options.model ?? DEFAULT_MODEL)) throw new AgyError("agy_model_unavailable", `required Antigravity model is unavailable: ${options.model ?? DEFAULT_MODEL}`);
@@ -11,6 +11,7 @@ import os from "node:os";
11
11
  import path from "node:path";
12
12
 
13
13
  import { AcpBridge } from "./dd-zcode.mjs";
14
+ import { observeModel, readModelObservations } from "./model-observations.mjs";
14
15
  import { cancelSessionWithBridge, createSessionWithBridge, doctor, forkSessionWithBridge, inspectSessionWithBridge, promptSessionWithBridge } from "./dd-grok.mjs";
15
16
  import { prepareRuntimeOwner, cleanupFailedDaemonStart, closeManagedBridge, confirmDaemonProcess, confirmDaemonStopped, finishDaemonProcess, scheduleDaemonProcessHeartbeat, registerDaemonProcess, startManagedBridge, stopProcessGroup } from "./managed-daemon.mjs";
16
17
 
@@ -201,16 +202,6 @@ export class Runtime {
201
202
  // A descendant is a managed native Session, not merely a liveness item.
202
203
  // Register it in the same registry used by inspect/hook.resolve so a
203
204
  // parentless inspection cannot downgrade it into a second root.
204
- this.sessions.set(id, {
205
- ...prior,
206
- provider_session_id: id,
207
- adapter_session_id: prior?.adapter_session_id ?? id,
208
- parent_provider_session_id: parent,
209
- root_provider_session_id: rootProviderSessionId,
210
- cwd,
211
- native_root_receipt: prior?.native_root_receipt ?? null
212
- });
213
- this.bridge.registerModelSession?.(id, id, this.state.config, parent);
214
205
  const descendant = this.descendants.get(id);
215
206
  const attempt = value.attempt_id;
216
207
  const retired = descendant?.retired_attempt_ids ?? [];
@@ -229,6 +220,16 @@ export class Runtime {
229
220
  return;
230
221
  }
231
222
  if (!newAttempt && terminal(descendant?.status) && terminal(status) && descendant.status !== status) throw new DaemonError("native_child_outcome_conflict", "Grok child has contradictory terminal outcomes", false, { session_id: id, previous: descendant.status, observed: status });
223
+ this.sessions.set(id, {
224
+ ...prior,
225
+ provider_session_id: id,
226
+ adapter_session_id: prior?.adapter_session_id ?? id,
227
+ parent_provider_session_id: parent,
228
+ root_provider_session_id: rootProviderSessionId,
229
+ cwd,
230
+ native_root_receipt: prior?.native_root_receipt ?? null
231
+ });
232
+ this.bridge.registerModelSession?.(id, id, this.state.config, parent);
232
233
  this.descendants.set(id, {
233
234
  provider_session_id: id,
234
235
  parent_provider_session_id: parent,
@@ -238,13 +239,20 @@ export class Runtime {
238
239
  ...(newAttempt && descendant?.attempt_id ? { retired_attempt_ids: [...retired, descendant.attempt_id] } : retired.length ? { retired_attempt_ids: retired } : {}),
239
240
  ...(!newAttempt && !terminal(status) && descendant?.outcome_error ? { outcome_error: descendant.outcome_error } : {})
240
241
  });
242
+ if (spawned && typeof value.model === "string" && value.model.trim()) {
243
+ return { sessionId: id, parentSessionId: parent, attemptId: attempt ?? null, parentPromptId: value.parent_prompt_id ?? null, model: value.model.trim() };
244
+ }
241
245
  }
242
246
  observeSubagentEvent(message) {
243
247
  const root = message?.params?.sessionId, update = message?.params?.update;
244
248
  const kind = String(update?.sessionUpdate ?? update?.type ?? message?.method ?? "");
245
- if (!root) return;
249
+ if (!root) return [];
250
+ const modelFacts = [];
246
251
  if (/subagent/i.test(kind)) {
247
- for (const value of [update?.subagent, update?.subagentInfo, update?.child, update?.result, update]) this.recordDescendant(value, root, "x.ai/subagent/event", /fail|error/i.test(kind) ? "failed" : /cancel/i.test(kind) ? "cancelled" : /complete|finish|done/i.test(kind) ? "completed" : "running");
252
+ for (const value of [update?.subagent, update?.subagentInfo, update?.child, update?.result, update]) {
253
+ const fact = this.recordDescendant(value, root, "x.ai/subagent/event", /fail|error/i.test(kind) ? "failed" : /cancel/i.test(kind) ? "cancelled" : /complete|finish|done/i.test(kind) ? "completed" : "running");
254
+ if (fact) modelFacts.push(fact);
255
+ }
248
256
  }
249
257
  // list_running omits completed children. Retain native tool receipts, never
250
258
  // extract identities from the model's prose or from arbitrary shell output.
@@ -254,7 +262,7 @@ export class Runtime {
254
262
  if (["spawn_subagent", "get_command_or_subagent_output"].includes(tool) && !this.subagentCalls.has(key)) this.subagentCalls.set(key, { name: tool, input: update.rawInput, attempts: new Map([...this.descendants].map(([id, child]) => [id, child.attempt_id])) });
255
263
  const call = this.subagentCalls.get(key), name = call?.name, output = update?.rawOutput;
256
264
  if (call && update.rawInput) call.input = update.rawInput;
257
- if (kind !== "tool_call_update" || update.status !== "completed") return;
265
+ if (kind !== "tool_call_update" || update.status !== "completed") return modelFacts;
258
266
  if (name === "spawn_subagent" && output?.type === "Text") {
259
267
  const id = /^subagent_id:\s*(\S+)\s*$/m.exec(output.text ?? "")?.[1];
260
268
  if (id) this.recordDescendant({ session_id: id, stale_activity: !call.attempts.has(id) || call.attempts.get(id) !== this.descendants.get(id)?.attempt_id }, root, "x.ai/tool/spawn_subagent", "running");
@@ -280,6 +288,7 @@ export class Runtime {
280
288
  }
281
289
  if (firstError) throw firstError;
282
290
  }
291
+ return modelFacts;
283
292
  }
284
293
  topology(root) { if (this.state.observation_error) { const error = this.state.observation_error; throw new DaemonError(error.code, error.message, false, error.details); } return [...this.descendants.values()].filter((child) => child.parent_provider_session_id === root); }
285
294
  observeToolCall(message) { this.toolObservations ??= new ToolObservations(); this.toolObservations.add(acpToolObservation(message, { harness: "grok-acp" })); const sessionId = message?.params?.sessionId; if (sessionId) this.toolUsage.set(sessionId, this.toolObservations.summary(sessionId)); }
@@ -377,7 +386,14 @@ export async function serveDaemon(stateDir) {
377
386
  const sessionId = message?.params?.sessionId;
378
387
  if (!runtime || !sessionId) return;
379
388
  try {
380
- runtime.observeToolCall(message); runtime.observeSubagentEvent(message);
389
+ runtime.observeToolCall(message);
390
+ for (const fact of runtime.observeSubagentEvent(message)) {
391
+ const source = { channel: "x.ai/subagent_spawned", scope: "configured", attempt_id: fact.attemptId, parent_prompt_id: fact.parentPromptId,
392
+ event_sha256: createHash("sha256").update(JSON.stringify(fact)).digest("hex") };
393
+ await observeModel({ journal: state.config.journal, harness: "grok-acp", sessionId: fact.sessionId, parentSessionId: fact.parentSessionId,
394
+ requested: state.config, observed: { model: fact.model }, source, evidence: "configured" });
395
+ bridge.modelProfiles?.set(fact.sessionId, { model: fact.model });
396
+ }
381
397
  } catch (error) {
382
398
  // Native notifications are fire-and-forget. Persist topology conflicts
383
399
  // for the next inspect/status call instead of creating an unhandled
@@ -388,6 +404,12 @@ export async function serveDaemon(stateDir) {
388
404
  const rootProviderSessionId = session?.root_provider_session_id ?? sessionId;
389
405
  await observeAcpToolCall(state.config, { daemonId: state.daemon_id, rootProviderSessionId, parentProviderSessionId: session?.parent_provider_session_id ?? null, observedProfile: bridge.modelProfiles?.get(sessionId) ?? {} }, message);
390
406
  } }); const initialized = await startManagedBridge({ config: state.config, state, bridge, kind: "grok-provider", publish: snapshot => writeState(paths.state, snapshot) }); runtime = new Runtime(paths, state, bridge, initialized);
407
+ for (const fact of await readModelObservations(state.config.journal)) {
408
+ if (fact.harness === "grok-acp" && fact.source?.channel === "x.ai/subagent_spawned" && runtime.sessions.has(fact.session_id)
409
+ && (runtime.descendants.get(fact.session_id)?.attempt_id ?? null) === (fact.source.attempt_id ?? null) && fact.observed?.model) {
410
+ bridge.modelProfiles?.set(fact.session_id, { model: fact.observed.model });
411
+ }
412
+ }
391
413
  scheduleDaemonProcessHeartbeat(state.config, state.resource_process);
392
414
  await removeSocket(paths.socket, paths.dir); const server = net.createServer((connection) => { connection.setEncoding("utf8"); let buffer = ""; connection.on("data", (chunk) => { buffer += chunk; const newline = buffer.indexOf("\n"); if (newline < 0) return; const line = buffer.slice(0, newline); buffer = buffer.slice(newline + 1); void (async () => { let request; try { request = JSON.parse(line); if (request.schema_id !== REQUEST_SCHEMA || !request.id || !request.operation) throw new DaemonError("daemon_protocol_mismatch", "invalid daemon request"); const result = await durableDaemonDispatch(paths.dir, request, () => runtime.dispatch(request.operation, request.params ?? {}), () => runtime.dispatch("session.inspect", request.params ?? {})); connection.end(`${JSON.stringify({ schema_id: RESPONSE_SCHEMA, id: request.id, ok: true, result: { ...result, _shutdown: undefined } })}\n`); if (result?._shutdown) setImmediate(() => void runtime.shutdown().then(() => process.exit(0)).catch(error => runtime.persist({ shutdown_state: "cleanup_failed", cleanup_error: errorPayload(error) }))); } catch (error) { connection.end(`${JSON.stringify({ schema_id: RESPONSE_SCHEMA, id: request?.id ?? null, ok: false, error: errorPayload(error) })}\n`); } })(); }); }); runtime.server = server; await new Promise((resolve, reject) => { server.once("error", reject); server.listen(paths.socket, resolve); }); await chmod(paths.socket, 0o600); await runtime.persist({ shutdown_state: "running", active_tree: false }); runtime.ready = true; return await new Promise(() => {});
393
415
  }
@@ -77,7 +77,9 @@ async function inspect(bridge, sessionId, options) {
77
77
  const [info, topology, usage] = await Promise.all([sessionInfo(bridge, sessionId), subagents(bridge, sessionId), sessionUsage(bridge, sessionId)]);
78
78
  const native = info?.data ?? info ?? {}, model = typeof native.model === "string" ? native.model : native.model?.modelId ?? null;
79
79
  const current = model ? { model, provider: native.provider ?? null, reasoning: native.reasoningEffort ?? null, mode: native.mode ?? null } : bridge.modelProfiles?.get(sessionId) ?? { model: null, provider: null, reasoning: null, mode: null };
80
- await observeModel({ journal: options.journal, harness: "grok-acp", sessionId, requested: profile(options), observed: current, source: model ? "x.ai/session/info" : "native.model_source_incomplete", evidence: model ? "configured" : "unavailable", reason: model ? null : "session_info_omits_model" });
80
+ await observeModel({ journal: options.journal, harness: "grok-acp", sessionId, requested: profile(options), observed: model ? current : {},
81
+ source: { channel: "x.ai/session/info", scope: "configured", ...(!model ? { non_asserting: true } : {}) },
82
+ evidence: model ? "configured" : "unavailable", reason: model ? null : "session_info_omits_model" });
81
83
  return { info, subagents: topology, usage, observed_profile: model ? current : { model: null, provider: null, reasoning: null, mode: null } };
82
84
  }
83
85
 
@@ -139,7 +141,6 @@ export async function createSessionWithBridge(bridge, options, initialized) {
139
141
  const created = await bridge.request("session/new", { cwd, mcpServers: [], ...(options.authMethodId ? { authMethodId: options.authMethodId } : {}) });
140
142
  const providerSessionId = text(created.sessionId, "Grok Build sessionId");
141
143
  bridge.registerModelSession?.(providerSessionId, providerSessionId, requested);
142
- await observeModel({ journal: options.journal, harness: "grok-acp", sessionId: providerSessionId, requested, observed: observedProfile(initialized, requested), source: "acp.initialize.modelState" });
143
144
  const profileReceipt = assertProfile(requested, observedProfile(initialized, requested));
144
145
  await options.onSessionCreated?.({ provider_session_id: providerSessionId, adapter_session_id: providerSessionId, cwd });
145
146
  const local = { ...options, initialized, rootProviderSessionId: providerSessionId }; const initial_usage_error = await forwardUsageBestEffort(local, providerSessionId, await sessionUsage(bridge, providerSessionId), sessionToolCalls(local, bridge, providerSessionId));
@@ -805,6 +805,7 @@ export function zcodeLifecycleQualification(sha256, bridgeCommit, harnessContrac
805
805
  ["5a8c89ab716a4b2ccc50df81cb4d204ff5c06368", "dd-zcode-harness@2"],
806
806
  ["91a1470650f5fd6105f9863fd31d1cad7b90a109", "dd-zcode-harness@2"],
807
807
  ["b9e0f5106dddb439b76186278bd9be5f45dbf3ef", "dd-zcode-harness@2"],
808
+ ["636b14181d5c9a95476055079836fd6b58169f58", "dd-zcode-harness@2"],
808
809
  ]);
809
810
  const qualified = qualifiedBridgeCommits.get(bridgeCommit) === harnessContract;
810
811
  return { status: qualified ? "qualified" : "unqualified", mechanism: "cli-invocation@1", sha256,
@@ -51,7 +51,7 @@ export function delegationContract(harness) {
51
51
 
52
52
  export const lifecycleCommandCompletionInstruction = "Run runtime-issued lifecycle CLI calls in the foreground, never with a native background or detached-shell option. A shell tool may yield before the lifecycle process exits. Retain and poll its exact process/session handle to completion, including inside a tool wrapper; empty interim output is not a CLI result. Read final stdout or response_file before deciding continuation, and never reissue the command while it is pending.";
53
53
 
54
- export const lifecycleRetryInstruction = `${lifecycleCommandCompletionInstruction} A failure is not permission to replay a lifecycle command. For proven effect=no_effect with recoverable=true and retry_command, correct the input and execute that exact issued successor. A check failure can retain receipts while leaving Work running: correct the evidenced cause first, then use the CLI's exact retry_command for an intentional fresh check attempt; never replay a check merely because Work is unsettled. If continuation.kind=repair_required, or a successful result has outcome=repair_required and a repair Work ID, end this Turn for the registered repair child; use any retry_command only after that repair settles and verification is refreshed. Unknown effects require reconciliation, never replay. A committed effect without an explicit continuation is not retryable. A native shell launch refusal before the CLI starts permits correcting only that shell-form error and retrying the same exact command. Otherwise report the primary error and stop; never invent IDs, call hooks directly, or run status as a substitute.`;
54
+ export const lifecycleRetryInstruction = `${lifecycleCommandCompletionInstruction} A failure is not permission to replay a lifecycle command. For proven effect=no_effect with recoverable=true and retry_command, correct the input and execute that exact issued successor. A check failure can retain receipts while leaving Work running: correct the evidenced cause first, then use the CLI's exact retry_command for an intentional fresh check attempt; never replay a check merely because Work is unsettled. If a successful CODE finish returns outcome=verification_required, inspect the retained receipts, update its verification file, and use only next.finish_command: the Stage is still running and the previous invocation ID is consumed. If continuation.kind=repair_required, or a successful result has outcome=repair_required and a repair Work ID, end this Turn for the registered repair child; use any retry_command only after that repair settles and verification is refreshed. Unknown effects require reconciliation, never replay. A committed effect without an explicit continuation is not retryable. A native shell launch refusal before the CLI starts permits correcting only that shell-form error and retrying the same exact command. Otherwise report the primary error and stop; never invent IDs, call hooks directly, or run status as a substitute.`;
55
55
 
56
56
  export const stageLifecycleInstruction = [
57
57
  "The project root identifies this RUN; use the Stage packet's declared execution workspace and absolute artifact paths. Do not change a native Session's working directory to a workspace provisioned for a later Stage.",
@@ -35,10 +35,11 @@ export async function observeModel({ journal, harness, sessionId, parentSessionI
35
35
  await mkdir(path.dirname(file), { recursive: true, mode: 0o700 });
36
36
  return withRunnerLock(file, async () => {
37
37
  const events = await readModelObservations(journal);
38
- const replay = source?.event_sha256 ? events.find(item => item.session_id === sessionId && item.source?.event_sha256 === source.event_sha256) : null;
38
+ const replay = source?.event_sha256 ? events.find(item => item.harness === harness && item.session_id === sessionId && item.source?.event_sha256 === source.event_sha256) : null;
39
39
  if (replay) return replay;
40
40
  const channel = source?.channel ?? evidence;
41
- const previous = events.findLast(item => item.harness === harness && item.session_id === sessionId && (item.source?.channel ?? item.evidence) === channel);
41
+ const previous = events.findLast(item => item.harness === harness && item.session_id === sessionId && (item.source?.channel ?? item.evidence) === channel
42
+ && (item.source?.attempt_id ?? null) === (source?.attempt_id ?? null));
42
43
  const normalizedReason = text(reason)?.slice(0, 200) ?? null;
43
44
  if (previous && JSON.stringify(previous.observed) === JSON.stringify(profile) && previous.reason === normalizedReason && previous.evidence === evidence) return previous;
44
45
  const previousKnown = events.findLast(item => item.harness === harness && item.session_id === sessionId && item.evidence === evidence && item.observed?.model);
@@ -62,16 +63,32 @@ export async function observeModel({ journal, harness, sessionId, parentSessionI
62
63
  });
63
64
  }
64
65
 
65
- export function modelAttribution(events) {
66
+ export function modelAttribution(events, expectedSessions = []) {
66
67
  const sessions = new Map();
67
68
  for (const event of events) {
68
69
  const key = `${event.harness}:${event.session_id}`;
69
70
  const session = sessions.get(key) ?? { harness: event.harness, session_id: event.session_id, parent_session_id: event.parent_session_id, requested: event.requested, observations: [] };
71
+ session.parent_session_id ??= event.parent_session_id;
72
+ if (!session.requested || !Object.values(session.requested).some(Boolean)) session.requested = event.requested;
70
73
  session.observations.push(event); sessions.set(key, session);
71
74
  }
75
+ const observedIds = new Set([...sessions.values()].map(session => session.session_id));
76
+ const missingSessions = expectedSessions.filter(session => typeof session === "string" ? !observedIds.has(session)
77
+ : !sessions.has(`${session.harness}:${session.session_id}`));
72
78
  const models = [...new Set(events.map(event => event.observed?.model).filter(Boolean))];
73
- const incomplete = !events.length || [...sessions.values()].some(session => session.observations.some(event => event.evidence === "unavailable" || !event.observed?.model));
79
+ const incomplete = !events.length || missingSessions.length > 0 || [...sessions.values()].some(session => {
80
+ const known = session.observations.filter(event => event.observed?.model);
81
+ if (!known.length) return true;
82
+ const knownScopes = new Set(known.map(event => `${event.evidence}:${event.source?.scope ?? "session"}:${event.source?.attempt_id ?? ""}:${event.source?.turn_id ?? ""}`));
83
+ return session.observations.some(event => {
84
+ if (event.evidence !== "unavailable") return false;
85
+ if (event.source?.non_asserting) return false;
86
+ const scope = event.source?.scope;
87
+ return !scope || !knownScopes.has(`configured:${scope}:${event.source?.attempt_id ?? ""}:${event.source?.turn_id ?? ""}`)
88
+ && !knownScopes.has(`response:${scope}:${event.source?.attempt_id ?? ""}:${event.source?.turn_id ?? ""}`);
89
+ });
90
+ });
74
91
  return { models, mixed: models.length > 1 || events.some(event => event.observed?.model && event.requested?.model && event.observed.model !== event.requested.model),
75
92
  observation_completeness: incomplete ? "incomplete" : "available_native_sources", transitions: events.filter(event => event.kind === "model_changed"),
76
- sessions: [...sessions.values()], usage_attribution: "session_totals_unattributed_to_model", note: "Native settings/events do not prove undisclosed server routing or per-model token costs." };
93
+ sessions: [...sessions.values()], missing_sessions: missingSessions, usage_attribution: "session_totals_unattributed_to_model", note: "Native settings/events do not prove undisclosed server routing or per-model token costs." };
77
94
  }
@@ -8,10 +8,12 @@ export interface ToolEvent {
8
8
  }
9
9
  export class ToolObservations {
10
10
  calls: Map<string, ToolEvent>;
11
+ turns: Map<string, { session: { harness_id: string; session_id: string }; turn_id: string; started: boolean; completed: boolean }>;
11
12
  gaps: Set<string>;
12
13
  add(event: ToolEvent | null): void;
13
- summary(sessionId?: string, cursor?: Set<string>): { total: number; failures: number; pending: number; by_tool: Record<string, number>; completeness: string; sessions: Array<{ session: { harness_id: string; session_id: string }; total: number; failures: number; pending: number; by_tool: Record<string, number> }>; reasons: string[] };
14
+ observeTurn(event: { session: { harness_id: string; session_id: string }; turn_id: string; started?: boolean; completed?: boolean }): void;
15
+ summary(sessionId?: string, cursor?: Set<string>): { total: number; failures: number; pending: number; by_tool: Record<string, number>; completeness: string; outcome_completeness: string; sessions: Array<{ session: { harness_id: string; session_id: string }; total: number; failures: number; pending: number; by_tool: Record<string, number> }>; reasons: string[] };
14
16
  cursor(): Set<string>;
15
17
  }
16
18
  export function codexToolObservation(record: Record<string, unknown>, sessionId: string): ToolEvent | null;
17
- export function replayToolJournal(file: string, options: { harness: string; rootSessionId: string; adapterSessionId?: string }, reducer?: ToolObservations): ToolObservations;
19
+ export function replayToolJournal(file: string, options: { harness: string; rootSessionId: string; adapterSessionId?: string; text?: string; expectedSessionIds?: string[] }, reducer?: ToolObservations): ToolObservations;
@@ -3,7 +3,14 @@ import fs from "node:fs";
3
3
  /** Canonical physical-session tool accounting. Native decoding lives below;
4
4
  * consumers never have to inspect provider envelopes or aggregate tokens. */
5
5
  export class ToolObservations {
6
- constructor() { this.calls = new Map(); this.gaps = new Set(); }
6
+ constructor() { this.calls = new Map(); this.turns = new Map(); this.gaps = new Set(); }
7
+ observeTurn(event) {
8
+ if (!event?.session?.harness_id || !event.session.session_id || !event.turn_id) { this.gaps.add("turn_identity_missing"); return; }
9
+ const key = JSON.stringify([event.session, event.turn_id]);
10
+ const old = this.turns.get(key);
11
+ this.turns.set(key, { ...old, ...event, started: old?.started || event.started || false,
12
+ completed: old?.completed || event.completed || false });
13
+ }
7
14
  add(event) {
8
15
  if (!event) return;
9
16
  if (event.reason) { this.gaps.add(event.reason); return; }
@@ -13,10 +20,12 @@ export class ToolObservations {
13
20
  const old = this.calls.get(key);
14
21
  // Replay and late starts cannot undo a terminal outcome. Conflicting
15
22
  // terminal facts remain visible rather than silently choosing success.
16
- const terminal = value => ["completed", "failed"].includes(value);
17
- if (old && terminal(old.status) && terminal(event.status) && old.status !== event.status) this.gaps.add("conflicting_tool_outcomes");
23
+ const proven = value => ["completed", "failed"].includes(value);
24
+ if (old && proven(old.status) && proven(event.status) && old.status !== event.status) this.gaps.add("conflicting_tool_outcomes");
18
25
  this.calls.set(key, { ...old, ...event, name: event.name ?? old?.name ?? "unknown",
19
- status: old?.status === "failed" || event.status === "failed" ? "failed" : terminal(old?.status) ? old.status : event.status ?? old?.status ?? "unknown" });
26
+ status: old?.status === "failed" || event.status === "failed" ? "failed"
27
+ : old?.status === "completed" || event.status === "completed" ? "completed"
28
+ : old?.status === "unknown" || event.status === "unknown" ? "unknown" : "running" });
20
29
  }
21
30
  summary(sessionId, cursor) {
22
31
  const by_tool = {}, sessions = new Map(), reasons = new Set(this.gaps); let total = 0, failures = 0, pending = 0;
@@ -30,6 +39,12 @@ export class ToolObservations {
30
39
  if (!["completed", "failed"].includes(value.status)) { pending++; session.pending++; }
31
40
  sessions.set(id, session);
32
41
  }
42
+ if (!cursor) for (const turn of this.turns.values()) {
43
+ if (sessionId !== undefined && turn.session.session_id !== sessionId) continue;
44
+ const id = JSON.stringify(turn.session);
45
+ sessions.set(id, sessions.get(id) ?? { session: turn.session, parent_session: null, total: 0, failures: 0, pending: 0, by_tool: {} });
46
+ if (!turn.started || !turn.completed) reasons.add("turn_lifecycle_incomplete");
47
+ }
33
48
  return { schema_id: "dd-flow/tool-summary@1", scope: "physical_sessions", measurement: "observed_invocations", origin: "native_tool_events", aggregation: cursor ? "delta" : "cumulative", total, failures, pending, by_tool,
34
49
  sessions: [...sessions.values()], completeness: reasons.size ? "partial" : "complete", outcome_completeness: pending ? "partial" : "complete", reasons: [...reasons] };
35
50
  }
@@ -66,7 +81,7 @@ export function acpToolObservation(message, { harness = "zcode-acp", rootSession
66
81
  const transport = message.params?.sessionId;
67
82
  if (adapterSessionId && transport !== adapterSessionId) return null;
68
83
  const root = rootSessionId ?? transport;
69
- const session = harness === "zcode-acp" ? zcodePhysicalSession(update, root) : root;
84
+ const session = harness === "zcode-acp" ? zcodePhysicalSession(update, root) : harness === "grok-acp" ? transport : root;
70
85
  if (!session || !update.toolCallId) return { reason: "tool_identity_missing" };
71
86
  return { session: identity(harness, session), parent_session: session !== root ? identity(harness, root) : null,
72
87
  transport_session_id: transport, call_id: update.toolCallId, name: update._meta?.["x.ai/tool"]?.name ?? update._meta?.claudeCode?.toolName ?? update.title?.split(":", 1)[0], status: status(update.status) };
@@ -85,22 +100,31 @@ export function codexToolObservation(record, sessionId) {
85
100
  const p = record.payload ?? {};
86
101
  if (["function_call", "custom_tool_call"].includes(p.type)) return { session: identity("codex-desktop", sessionId), call_id: p.call_id, name: p.name, status: "running" };
87
102
  if (["function_call_output", "custom_tool_call_output"].includes(p.type)) {
88
- return { session: identity("codex-desktop", sessionId), call_id: p.call_id, status: containsFailure(p) ? "failed" : "completed" };
103
+ return { session: identity("codex-desktop", sessionId), call_id: p.call_id, status: structuredFailure(p.output) ? "failed" : "unknown" };
89
104
  }
90
105
  return null;
91
106
  }
92
107
 
93
- function containsFailure(value) {
94
- if (Array.isArray(value)) return value.some(containsFailure);
95
- if (typeof value === "string") {
96
- if (/(?:^|\n)exit=(?:[1-9]\d*)\b|Process exited with code [1-9]\d*/.test(value)) return true;
97
- if (/^[{[]/.test(value.trim())) { try { return containsFailure(JSON.parse(value)); } catch {} }
98
- return false;
99
- }
108
+ function structuredFailure(value) {
100
109
  if (!value || typeof value !== "object") return false;
101
- if (value.ok === false || value.is_error === true || value.isError === true || value.error === true || typeof value.error === "string") return true;
102
- if (typeof value.exit_code === "number" && value.exit_code !== 0) return true;
103
- return Object.values(value).some(containsFailure);
110
+ return value.ok === false || value.is_error === true || value.isError === true || value.error === true
111
+ || typeof value.exit_code === "number" && value.exit_code !== 0;
112
+ }
113
+
114
+ const codexTools = new Set(["commandExecution", "fileChange", "mcpToolCall", "dynamicToolCall", "collabAgentToolCall", "webSearch", "imageView", "imageGeneration", "sleep"]);
115
+ const codexNonTools = new Set(["userMessage", "hookPrompt", "agentMessage", "functionCallOutput", "plan", "reasoning", "subAgentActivity", "enteredReviewMode", "exitedReviewMode", "contextCompaction"]);
116
+ export function codexAppServerObservation(message) {
117
+ if (!["item/started", "item/completed"].includes(message?.method)) return null;
118
+ const { threadId, turnId, item } = message.params ?? {};
119
+ if (!item || codexNonTools.has(item.type)) return null;
120
+ if (!codexTools.has(item.type)) return { reason: "codex_item_type_unsupported" };
121
+ if (!threadId || !item.id) return { reason: "tool_identity_missing" };
122
+ const native = item.status;
123
+ const failed = ["failed", "declined", "interrupted"].includes(native) || item.type === "commandExecution" && typeof item.exitCode === "number" && item.exitCode !== 0;
124
+ const completed = message.method === "item/completed" && (native === undefined || native === "completed");
125
+ const known = native === undefined || ["inProgress", "completed", "failed", "declined", "interrupted"].includes(native);
126
+ return { session: identity("codex-desktop", threadId), epoch: turnId ?? "", call_id: item.id, name: item.type,
127
+ status: !known ? "unknown" : failed ? "failed" : completed ? "completed" : "running", ...(!known ? { reason: "codex_item_status_unsupported" } : {}) };
104
128
  }
105
129
 
106
130
  export function openCodeToolObservation(part, sessionId, messageId) {
@@ -121,7 +145,7 @@ export function agyToolObservation(step, rootSessionId) {
121
145
  * carries physical identities; an ACP connection identity never crosses out. */
122
146
  export function replayToolJournal(file, options, reducer = new ToolObservations()) {
123
147
  let bytes;
124
- try { bytes = fs.readFileSync(file, "utf8"); } catch (error) { reducer.gaps.add(error.code === "ENOENT" ? "journal_missing" : "journal_unreadable"); return reducer; }
148
+ try { bytes = options.text ?? fs.readFileSync(file, "utf8"); } catch (error) { reducer.gaps.add(error.code === "ENOENT" ? "journal_missing" : "journal_unreadable"); return reducer; }
125
149
  const lines = bytes.split("\n"); let session = options.rootSessionId, transport = options.adapterSessionId;
126
150
  for (const [index, line] of lines.entries()) {
127
151
  if (!line.trim()) continue;
@@ -135,7 +159,14 @@ export function replayToolJournal(file, options, reducer = new ToolObservations(
135
159
  if (e.kind === "inbound" || m.method) add(acpToolObservation(m, { ...options, rootSessionId: session, adapterSessionId: transport }));
136
160
  } else if (options.harness === "codex-desktop") {
137
161
  if (e.type === "session_meta" && e.payload?.id !== session) { reducer.gaps.add("session_identity_mismatch"); break; }
138
- add(codexToolObservation(e, session));
162
+ if (m.params?.threadId && options.expectedSessionIds && !options.expectedSessionIds.includes(m.params.threadId)) continue;
163
+ if (["turn/started", "turn/completed"].includes(m.method)) {
164
+ const params = m.params ?? {};
165
+ if (params.threadId && params.turn?.id) reducer.observeTurn({ session: identity("codex-desktop", params.threadId), turn_id: params.turn.id,
166
+ started: m.method === "turn/started", completed: m.method === "turn/completed" && params.turn.status === "completed" });
167
+ else reducer.gaps.add("turn_identity_missing");
168
+ }
169
+ add(codexAppServerObservation(m) ?? codexToolObservation(e, session));
139
170
  } else if (options.harness === "droid-cli") {
140
171
  if (e.type === "session_start" && e.id !== session) { reducer.gaps.add("session_identity_mismatch"); break; }
141
172
  for (const event of droidToolObservations(e, session)) add(event);
@@ -1,14 +1,16 @@
1
1
  {
2
2
  "$schema": "https://json-schema.org/draft/2020-12/schema",
3
- "$id": "dd-flow/code-verification@2",
3
+ "$id": "dd-flow/code-verification@3",
4
4
  "type": "object",
5
5
  "additionalProperties": false,
6
6
  "required": ["schema_id", "verdict", "summary", "unresolved", "deviations"],
7
7
  "properties": {
8
- "schema_id": {"const": "dd-flow/code-verification@2"},
9
- "verdict": {"enum": ["passed", "needs_repair", "blocked"]},
8
+ "schema_id": {"enum": ["dd-flow/code-verification@2", "dd-flow/code-verification@3"]},
9
+ "verdict": {"enum": ["passed", "needs_repair", "needs_checks", "blocked"]},
10
10
  "summary": {"type": "string", "minLength": 1},
11
11
  "unresolved": {"type": "array", "items": {"type": "string", "minLength": 1}},
12
- "deviations": {"type": "array", "items": {"type": "string", "minLength": 1}}
13
- }
12
+ "deviations": {"type": "array", "items": {"type": "string", "minLength": 1}},
13
+ "document_updates": {"type": "array", "uniqueItems": true, "items": {"type": "string", "minLength": 1}}
14
+ },
15
+ "allOf": [{"if": {"properties": {"schema_id": {"const": "dd-flow/code-verification@2"}}}, "then": {"properties": {"verdict": {"enum": ["passed", "needs_repair", "blocked"]}}, "not": {"required": ["document_updates"]}}}]
14
16
  }
@@ -79,32 +79,67 @@ export function runObservations(context, projectId, runId) {
79
79
  reasons.push("lineage_evidence_invalid");
80
80
  }
81
81
  }
82
- // Native transcript adapters need no managed controller journal.
83
- for (const session of sessions)
84
- if (session.transcript_path && ![...sources.values()].some(s => s.session.harness_id === session.harness && s.session.session_id === (session.provider_session_id ?? session.session_id))) {
85
- add(`transcript:${session.harness}:${session.provider_session_id ?? session.session_id}`, session.provider_session_id ?? session.session_id, session.transcript_path, home, "current", session.harness);
86
- }
87
82
  const reducer = new ToolObservations();
88
83
  for (const reason of reasons)
89
84
  reducer.gaps.add(reason);
85
+ const expectedCodexIds = sessions.filter(session => session.harness === "codex-desktop").map(session => session.provider_session_id ?? session.session_id);
86
+ const replay = (source, sessionId) => {
87
+ const file = path.join(home, source.path);
88
+ let fd = null;
89
+ try {
90
+ const actual = fs.realpathSync(file);
91
+ if (!inside(home, actual))
92
+ throw new Error("journal escaped owned home");
93
+ fd = fs.openSync(actual, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW);
94
+ if (!fs.fstatSync(fd).isFile())
95
+ throw new Error("journal is not a regular file");
96
+ replayToolJournal(file, { harness: source.session.harness_id, rootSessionId: sessionId, text: fs.readFileSync(fd, "utf8"), ...(source.session.harness_id === "codex-desktop" ? { expectedSessionIds: expectedCodexIds } : {}) }, reducer);
97
+ }
98
+ catch {
99
+ source.status = "unavailable";
100
+ source.reason = "journal_unavailable_during_read";
101
+ }
102
+ finally {
103
+ if (fd !== null)
104
+ fs.closeSync(fd);
105
+ }
106
+ };
90
107
  for (const source of sources.values()) {
91
- if (source.status !== "available") {
92
- reducer.gaps.add(source.reason ?? "journal_unavailable");
108
+ if (source.status !== "available")
93
109
  continue;
94
- }
95
- replayToolJournal(path.join(home, source.path), { harness: source.session.harness_id, rootSessionId: source.session.session_id }, reducer);
110
+ replay(source, source.session.session_id);
111
+ }
112
+ const covered = () => new Set(reducer.summary().sessions.map(item => JSON.stringify(item.session)));
113
+ // A managed root journal can contain all of its children's native events.
114
+ // Select transcript fallback only after replay reveals an actual gap.
115
+ for (const session of sessions) {
116
+ const nativeId = session.provider_session_id ?? session.session_id;
117
+ if (!session.transcript_path || covered().has(JSON.stringify({ harness_id: session.harness, session_id: nativeId })))
118
+ continue;
119
+ const id = `transcript:${session.harness}:${nativeId}`;
120
+ add(id, nativeId, session.transcript_path, home, "current", session.harness);
121
+ const source = sources.get(id);
122
+ if (source?.status === "available")
123
+ replay(source, nativeId);
96
124
  }
125
+ for (const source of sources.values())
126
+ if (source.status !== "available"
127
+ && !covered().has(JSON.stringify(source.session)))
128
+ reducer.gaps.add(source.reason ?? "journal_unavailable");
97
129
  const expected = [...new Map(sessions.map(s => { const session = { harness_id: s.harness, session_id: s.provider_session_id ?? s.session_id }; return [JSON.stringify(session), session]; })).values()];
98
130
  const expectedIds = new Set(expected.map(s => JSON.stringify(s)));
99
131
  for (const [key, event] of reducer.calls)
100
132
  if (!expectedIds.has(JSON.stringify(event.session))) {
101
133
  reducer.calls.delete(key);
102
- reducer.gaps.add("tool_session_not_bound_to_run");
134
+ }
135
+ for (const [key, event] of reducer.turns)
136
+ if (!expectedIds.has(JSON.stringify(event.session))) {
137
+ reducer.turns.delete(key);
103
138
  }
104
139
  const summary = reducer.summary();
105
140
  const observed = new Set(summary.sessions.map(s => JSON.stringify(s.session)));
106
141
  const missing = expected.filter(s => !observed.has(JSON.stringify(s)));
107
- const completeness = missing.length || summary.completeness !== "complete" ? "partial" : "complete";
142
+ const completeness = missing.length || summary.completeness !== "complete" || summary.outcome_completeness !== "complete" ? "partial" : "complete";
108
143
  return { schema_id: "dd-flow/run-observations@1", home, sources: [...sources.values()], tools: { ...summary,
109
144
  completeness, status: observed.size === 0 ? "unavailable" : completeness === "complete" ? "measured" : "partial",
110
145
  observed_sessions: expected.length - missing.length, expected_sessions: expected.length, unavailable_sessions: missing.map(s => s.session_id), missing_sessions: missing,
@@ -2,6 +2,7 @@ import { managedLifecycleCommand } from "./lifecycle-invocations.js";
2
2
  import { stageLifecycleInstruction } from "../harness-runtime/lib/delegation-instructions.mjs";
3
3
  import { ensureAggregateRepair, ensureRepairIntent } from "./repair-intents.js";
4
4
  import { runWorkspaceBootstrap } from "./workspace-bootstrap.js";
5
+ import crypto from "node:crypto";
5
6
  import fs from "node:fs";
6
7
  import path from "node:path";
7
8
  import { AppError } from "../shared/errors.js";
@@ -220,6 +221,12 @@ export async function finishVnextCode(context, input) {
220
221
  });
221
222
  }
222
223
  const finalReceipts = receipts;
224
+ if (verification.verdict === "needs_checks") {
225
+ return { ok: true, run_id: run.id, stage, outcome: "verification_required", verification,
226
+ receipts: finalReceipts.map(receipt => ({ id: receipt.id, receipt_path: receipt.receipt_path, input_hash: receipt.input_hash, status: receipt.status })),
227
+ next: { kind: "verification_required", finish_command: finishCommand(context, run.id, projectRoot, input.verificationFile) },
228
+ instruction: "Aggregate receipts are available. Keep CODE running. Inspect them, update code-verification.json to passed or needs_repair, then invoke only the issued successor finish_command. A material document change belongs to the registered fresh repair Work, not this coordinator Session." };
229
+ }
223
230
  const projectedVerification = verificationProjection(works, finalReceipts, { historicalReceipts: finalReceipts, runId: run.id, runHome: home });
224
231
  const unresolvedAcceptance = (projectedVerification.acceptance ?? []).filter((item) => item.status === "unresolved");
225
232
  if (unresolvedAcceptance.length)
@@ -331,6 +338,15 @@ export function prepareVnextCodeRepair(context, input) {
331
338
  if (previous)
332
339
  return { kind: "replay", project, run, previous };
333
340
  }
341
+ if (input.semanticUnresolved?.length && input.verificationPath) {
342
+ const previous = codeWorks(context, project.id, run.id).find(work => {
343
+ const repair = packet(work)?.repair;
344
+ return Boolean(repair && repair.verification_path === input.verificationPath
345
+ && JSON.stringify(repair.semantic_unresolved) === JSON.stringify(unique(input.semanticUnresolved ?? [])));
346
+ });
347
+ if (previous)
348
+ return { kind: "replay", project, run, previous };
349
+ }
334
350
  if (typeof input.objective !== "string" || !input.objective.trim())
335
351
  throw new AppError("validation", "Repair objective must not be empty", 2);
336
352
  if (input.semanticUnresolved !== undefined && (!Array.isArray(input.semanticUnresolved) || input.semanticUnresolved.some(value => typeof value !== "string" || !value.trim())))
@@ -346,10 +362,12 @@ export function prepareVnextCodeRepair(context, input) {
346
362
  }
347
363
  if ([Boolean(input.checkReceiptId), Boolean(input.reviewFindings?.length), Boolean(input.semanticUnresolved?.length)].filter(Boolean).length !== 1)
348
364
  throw new AppError("validation", "Repair requires exactly one evidence source", 2);
365
+ let semanticVerification = input.verification;
349
366
  if (input.semanticUnresolved?.length) {
350
367
  if (!input.verificationPath)
351
368
  throw new AppError("usage", "Semantic repair requires a verification file", 2);
352
369
  const verification = input.verification ?? readVerification(context, { file: input.verificationPath, projectRoot: run.workspace_root, runId: run.id, runRoot: home });
370
+ semanticVerification = verification;
353
371
  if (input.verification !== undefined)
354
372
  validateSchema({ schemaName: "code-verification", file: input.verificationPath, data: verification, projectRoot: run.workspace_root, ddFlowHome: context.ddFlowHome, runId: run.id, runRoot: home });
355
373
  const unresolved = verification.unresolved.length ? verification.unresolved : [verification.summary];
@@ -417,7 +435,17 @@ export function prepareVnextCodeRepair(context, input) {
417
435
  acceptance: uniqueBy(invariantPackets.flatMap((value) => value.acceptance), (value) => JSON.stringify(value)),
418
436
  // A CODE-REVIEW repair changes delivered code/evidence, never the
419
437
  // already accepted PLAN or its ownership declaration.
420
- document_updates: [],
438
+ document_updates: semanticRepair ? uniqueBy((semanticVerification?.document_updates ?? []).map(relative => {
439
+ const declared = invariantPackets.flatMap(value => value.document_updates).find(value => value.path === relative);
440
+ if (!declared)
441
+ throw new AppError("document_update_unknown", "CODE verification selected a document absent from accepted Work ownership", 2, { path: relative });
442
+ const file = path.resolve(run.workspace_root, relative);
443
+ if (!file.startsWith(`${path.resolve(run.workspace_root)}${path.sep}`))
444
+ throw new AppError("document_update_invalid", "CODE verification document path escapes the workspace", 2, { path: relative });
445
+ const exists = fs.existsSync(file);
446
+ return { ...declared, action: exists ? "update" : "create",
447
+ baseline_sha256: exists ? crypto.createHash("sha256").update(fs.readFileSync(file)).digest("hex") : null };
448
+ }), value => value.path) : [],
421
449
  required_read: unique([...(receipt ? [receipt.receipt_path] : input.reviewFindings?.flatMap((finding) => finding.evidence_refs) ?? []), ...(semanticRepair && input.verificationPath ? [input.verificationPath] : []), ...packets.flatMap((value) => value.required_read)]),
422
450
  discovery_boundary: unique(packets.flatMap((value) => value.discovery_boundary)),
423
451
  // These are collision-avoidance hints. Receipt paths enrich the coordinator
@@ -517,15 +545,15 @@ function coordinatorPrompt(context, input) {
517
545
  "Do not choose a provider delegation tool from this stage prompt. At each work_fanout boundary, stop at the Work-graph boundary; the shared controller supplies the selected adapter's native tool, exact parameters, child task, and wait contract. It launches only entries listed in graph.ready and uses at most the qualified capacity. Every registered CODE Work runs in a fresh child Session, including a serial dependency chain. The coordinator launches those children only through the controller-supplied native delegation instruction; it must never invoke a Work start_command itself. Each child receives its complete packet from dd-flow and runs its own exact start_command. After a Work finishes, the controller uses the graph returned by work finish to launch newly ready Work. To refresh the parent graph yourself use the exact command: " + runtimeCommand(context, ["work", "ls", "--run", input.run.id, "--ready", "--project-root", input.projectRoot, "--json"]),
518
546
  "A quiet child is still running until the harness reports its turn completed, failed, cancelled or explicitly needs attention. An elapsed nominal wait, silence, or no new artifact is not an unresponsive-worker failure. Never interrupt, replace, relaunch, or stage-block a still-running child for that reason, even if an external controller asks. Long work finish and stage finish commands emit check progress on stderr. After you issue the exact CODE stage finish command, wait for that same command to return once: a completed command with a non-zero exit and structured `code_gate_failed` output is its terminal result, not a reason to keep waiting. A repair_required continuation already registered the repair: end this Turn. Do not invoke that Work's start_command from the coordinator Session, inspect its PID, start a second finish command, or infer failure from quiet output. Close a disposable child only after its Work is accepted or explicitly failed/cancelled and the harness reports the turn settled.",
519
547
  `A repairable engine, harness, or environment failure is not a user question. Write the evidence summary to ${path.join(requireHome(input.run), "works", input.rootWork.work_id, "block-summary.md")}, then record it with a separate command: ${runtimeCommand(context, ["stage", "block", input.run.id, "--stage", "code", "--work", input.rootWork.work_id, "--kind", "<engine|harness|environment>", "--code", "<stable-code>", "--summary-file", `@run/works/${input.rootWork.work_id}/block-summary.md`, "--retryable", "--project-root", input.projectRoot, "--json"])}. Choose kind and stable-code from evidence; this example is not an executable continuation. Repair it externally, then run the exact unblock_command returned by dd-flow and continue this same stage.`,
520
- `When every CODE and repair Work is completed, write ${path.join(input.root, "code-verification.json")} using the exact contract below. Mark passed only when all accepted requirements and current-gate acceptance criteria are implemented or explicitly evidenced; list every remaining issue in unresolved. Every evidence_refs item must already exist as a relative workspace path or run://${input.run.id}/ path. Do not claim a browser or other check receipt that was not retained. Then finish: ${finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "code-verification.json"))}`,
548
+ `When every CODE and repair Work is completed, write ${path.join(input.root, "code-verification.json")} using the exact contract below. If closing an accepted document depends on aggregate checks that have not run yet, use verdict needs_checks; the first finish returns receipts and a successor finish_command while CODE remains open. After checking those receipts, use needs_repair with document_updates naming only assigned documents that must change, or passed if the current documents already prove acceptance. Mark passed only when all accepted requirements and current-gate acceptance criteria are implemented or explicitly evidenced; list every remaining issue in unresolved. Every evidence_refs item must already exist as a relative workspace path or run://${input.run.id}/ path. Do not claim a browser or other check receipt that was not retained. Then finish: ${finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "code-verification.json"))}`,
521
549
  "A code_gate_failed response with continuation.kind=repair_required has already registered one repair Work with all failed receipts. End this Turn so the controller can dispatch it in a fresh child. Do not recreate the repair, copy origin IDs, write a repair task file or execute the child start command yourself. After repair, update semantic verification and finish again.",
522
550
  "</execution_commands>",
523
551
  "",
524
552
  "<verification_contract>",
525
553
  "```json",
526
- JSON.stringify({ schema_id: "dd-flow/code-verification@2", verdict: "passed", summary: "Evidence-backed conclusion.", unresolved: [], deviations: [] }, null, 2),
554
+ JSON.stringify({ schema_id: "dd-flow/code-verification@3", verdict: "passed", summary: "Evidence-backed conclusion.", unresolved: [], deviations: [], document_updates: [] }, null, 2),
527
555
  "```",
528
- "Allowed verdict values are passed, needs_repair, and blocked. The example is valid JSON using passed; choose exactly one value.",
556
+ "Allowed verdict values are passed, needs_checks, needs_repair, and blocked. The example is valid JSON using passed; choose exactly one value. A verification_required result is successful receipt gathering, not Stage completion; use its issued successor command after updating this file.",
529
557
  "The CLI verifies graph coverage and executes all deterministic checks. Do not duplicate its report; record only your semantic conclusion and evidence.",
530
558
  "</verification_contract>",
531
559
  "",
@@ -331,9 +331,11 @@ function preparedExisting(context, input) {
331
331
  return { ok: true, resumed: true, run_id: input.run.id, stage, outcome: "review_required", work_id: input.workId, id: binding.work_session_id, prompt_path: promptPath, worker_prompt_markdown: fs.readFileSync(promptPath, "utf8"), orchestration, plan_review: { requested_mode: input.requested, effective_mode: input.effective, policy_source: policySource(input.run), groups: input.groups }, next: { dispatch_command: dispatchCommand(context, input.run.id, input.projectRoot), finish_command: finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "decision.json")) } };
332
332
  }
333
333
  function orchestratorPrompt(context, input) {
334
- const { template, revision, capacity } = input;
334
+ const { revision, capacity } = input;
335
335
  const decision = path.join(input.root, "decision.json");
336
336
  const workspaceContract = ["<workspace_contract>", `- route: ${input.workspaceRoute.route}`, `- feature branch: ${input.workspaceRoute.feature_branch ?? "not applicable"}`, `- base commit: ${input.workspaceRoute.base_ref ?? "not applicable"}`, `- read/write workspace: ${input.run.workspace_root}`, "The CLI verified this frozen route. All plan and correction writes belong in the named workspace; project root remains only the stable lifecycle identity. Do not create, switch, merge or delete branches/worktrees.", "</workspace_contract>"].join("\n");
337
+ const acceptanceOwnership = "For every deferred acceptance or document update, identify the named CODE Work owner, its causal checks, and when it can truthfully close the obligation. Aggregate checks run only after implementation Work completes: assign a CODE owner to pending documents and use CODE needs_checks → verification_required → repair after those receipts. A promise that MERGE or an unnamed readiness owner will finish acceptance is a material PLAN finding; bootstrap workspace readiness is not proof of the new feature.";
338
+ const template = [acceptanceOwnership, input.template].join("\n\n");
337
339
  const reviewerLaunch = capacity.source === "external_policy"
338
340
  ? `After dispatch, return at the Work-graph boundary. The shared runtime launches the queued reviewer Works as separate external Sessions, at most ${capacity.available_slots} in parallel under the RUN's frozen profiles. Do not launch native children, create provider roots, run a capacity probe, or record this external limit as native capacity. Reviewers are read-only leaf workers. Continue the semantic decision only after their Work receipts settle; do not substitute a missing result or relaunch a settled reviewer.`
339
341
  : "After dispatch, stop at the Work-graph boundary. The shared controller supplies the selected adapter's native tool, exact parameters, child task, and wait contract; do not choose a provider tool from this stage prompt or substitute a shell command. It launches at most the qualified capacity at once and starts unchanged queued Works only after the current wave settles. A launch rejected before it starts is not review evidence: do not create a replacement. Each reviewer must be a genuinely fresh harness child Session; the lifecycle adapter binds that observed Session, so do not bind or supply a Session ID manually. Reviewers are read-only and must not create children. As soon as a reviewer result is accepted, release that reviewer Session when the harness permits.";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@deksden-com/dd-flow-cli",
3
- "version": "0.9.0-beta.116",
3
+ "version": "0.9.0-beta.118",
4
4
  "description": "Mechanical runtime CLI for dd-flow workflows.",
5
5
  "type": "module",
6
6
  "bin": {