@deksden-com/dd-flow-cli 0.9.0-beta.116 → 0.9.0-beta.118
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/build-info.json +3 -3
- package/dist/harness-runtime/lib/dd-agy-daemon.mjs +2 -2
- package/dist/harness-runtime/lib/dd-agy.mjs +2 -1
- package/dist/harness-runtime/lib/dd-grok-daemon.mjs +36 -14
- package/dist/harness-runtime/lib/dd-grok.mjs +3 -2
- package/dist/harness-runtime/lib/dd-zcode.mjs +1 -0
- package/dist/harness-runtime/lib/delegation-instructions.mjs +1 -1
- package/dist/harness-runtime/lib/model-observations.mjs +22 -5
- package/dist/harness-runtime/lib/tool-observations.d.mts +4 -2
- package/dist/harness-runtime/lib/tool-observations.mjs +49 -18
- package/dist/schemas/code-verification.schema.json +7 -5
- package/dist/services/run-observations.js +46 -11
- package/dist/services/vnext-code.js +32 -4
- package/dist/services/vnext-plan-review.js +3 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
# @deksden-com/dd-flow-cli
|
|
2
2
|
|
|
3
|
+
## 0.9.0-beta.118
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- Admit the tested ZCode bridge fix for managed turns that remain active after a frozen event watermark, and allow 30 seconds for Antigravity's version check during doctor and daemon startup.
|
|
8
|
+
|
|
9
|
+
## 0.9.0-beta.117
|
|
10
|
+
|
|
11
|
+
### Patch Changes
|
|
12
|
+
|
|
13
|
+
- Preserve Grok child model facts, replay owned Codex tool events, and let CODE collect aggregate-check evidence before semantic acceptance.
|
|
14
|
+
|
|
3
15
|
## 0.9.0-beta.116
|
|
4
16
|
|
|
5
17
|
### Patch Changes
|
package/dist/build-info.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"cli_package": "@deksden-com/dd-flow-cli",
|
|
3
|
-
"cli_version": "0.9.0-beta.
|
|
4
|
-
"cli_commit": "
|
|
5
|
-
"built_at": "2026-09-
|
|
3
|
+
"cli_version": "0.9.0-beta.118",
|
|
4
|
+
"cli_commit": "19b1d443644710c4d17e02a7175a5e556e7d8f8e",
|
|
5
|
+
"built_at": "2026-09-28T12:22:30.459Z",
|
|
6
6
|
"built_with_canon": {
|
|
7
7
|
"version": "4.1.1",
|
|
8
8
|
"commit": "678daa038287c948ada5b2d785a6dcc925c7b891",
|
|
@@ -11,7 +11,7 @@ import { agyWorkspaceHooks } from "./agy-hooks.mjs";
|
|
|
11
11
|
import { durableDaemonDispatch } from "./daemon-operations.mjs";
|
|
12
12
|
import path from "node:path";
|
|
13
13
|
import { observeModel } from "./model-observations.mjs";
|
|
14
|
-
import { AgyError, assertProfile, executable, observedProfile, runAgy, usageSnapshot } from "./dd-agy.mjs";
|
|
14
|
+
import { AGY_VERSION_TIMEOUT_MS, AgyError, assertProfile, executable, observedProfile, runAgy, usageSnapshot } from "./dd-agy.mjs";
|
|
15
15
|
import { prepareRuntimeOwner, cleanupFailedStart, cleanupFailedDaemonStart, confirmDaemonProcess, confirmDaemonStopped, finishDaemonProcess, scheduleDaemonProcessHeartbeat, registerDaemonProcess, stopProcessGroup } from "./managed-daemon.mjs";
|
|
16
16
|
|
|
17
17
|
const REQUEST_SCHEMA = "dd-agy/daemon-request@1", RESPONSE_SCHEMA = "dd-agy/daemon-response@1", STATE_SCHEMA = "dd-agy/daemon-state@1";
|
|
@@ -189,7 +189,7 @@ export async function startDaemon(options) {
|
|
|
189
189
|
if (previous?.shutdown_state === "clean") await authorizeRetainedDaemonResume(paths.dir, previous, options.sessionId, config);
|
|
190
190
|
if (previous?.shutdown_state === "running" && previous.active_tree) throw new DaemonError("invalid_harness_crash", "previous Antigravity daemon died with an active or unproven tree");
|
|
191
191
|
await removeSocket(paths); await prepare(paths, config);
|
|
192
|
-
const version = (await runAgy(config.bin, ["--version"], { cwd: config.cwd, env: sanitizedEnv(config), timeoutMs:
|
|
192
|
+
const version = (await runAgy(config.bin, ["--version"], { cwd: config.cwd, env: sanitizedEnv(config), timeoutMs: AGY_VERSION_TIMEOUT_MS })).stdout.trim();
|
|
193
193
|
const resourceProcess = await registerDaemonProcess(config, { kind: "agy-daemon", owner: `agy:${path.basename(paths.dir)}`, operation: `agy-daemon:${path.basename(paths.dir)}`, stdout: paths.log, stderr: paths.log });
|
|
194
194
|
const retained = retainedAgyState(previous, config.resumeSessionId);
|
|
195
195
|
await writeJson(paths.state, { schema_id: STATE_SCHEMA, daemon_id: config.daemonId, shutdown_state: "starting", active_tree: false, unclaimed_activity: retained.unclaimed_activity ?? false, retained_tree_settlement: retained.retained_tree_settlement ?? null, versions: { agy: version }, config, sessions: [], session_observations: retained.session_observations ?? [], descendants: retained.descendants ?? [], tool_calls: retained.tool_calls ?? [], completed_tool_steps: retained.completed_tool_steps ?? [], tool_identity_complete: retained.tool_identity_complete !== false, turn_generation: retained.turn_generation ?? 0, last_result: retained.last_result ?? null, resource_process: resourceProcess });
|
|
@@ -6,6 +6,7 @@ import path from "node:path";
|
|
|
6
6
|
import { checkObservedProfile } from "./model-observations.mjs";
|
|
7
7
|
|
|
8
8
|
const DEFAULT_MODEL = "gemini-3.1-pro-high";
|
|
9
|
+
export const AGY_VERSION_TIMEOUT_MS = 30_000;
|
|
9
10
|
|
|
10
11
|
export class AgyError extends Error {
|
|
11
12
|
constructor(code, message, retryable = false, details) { super(message); this.code = code; this.retryable = retryable; this.details = details; }
|
|
@@ -50,7 +51,7 @@ export function assertProfile(requested, observed) { return checkObservedProfile
|
|
|
50
51
|
export async function doctor(options = {}) {
|
|
51
52
|
const bin = options.bin ?? process.env.DD_AGY_BIN ?? "agy", temporary = await mkdtemp(path.join(os.tmpdir(), "dd-agy-doctor-"));
|
|
52
53
|
try {
|
|
53
|
-
const version = (await runAgy(bin, ["--version"], { timeoutMs:
|
|
54
|
+
const version = (await runAgy(bin, ["--version"], { timeoutMs: AGY_VERSION_TIMEOUT_MS })).stdout.trim();
|
|
54
55
|
const models = await runAgy(bin, [`--gemini_dir=${temporary}`, "--app_data_dir=runtime", "models"], { timeoutMs: options.timeoutMs ?? 30_000 });
|
|
55
56
|
const modelIds = models.stdout.split(/\r?\n/).map(line => line.split(/\s+/)[0]).filter(Boolean);
|
|
56
57
|
if (!modelIds.includes(options.model ?? DEFAULT_MODEL)) throw new AgyError("agy_model_unavailable", `required Antigravity model is unavailable: ${options.model ?? DEFAULT_MODEL}`);
|
|
@@ -11,6 +11,7 @@ import os from "node:os";
|
|
|
11
11
|
import path from "node:path";
|
|
12
12
|
|
|
13
13
|
import { AcpBridge } from "./dd-zcode.mjs";
|
|
14
|
+
import { observeModel, readModelObservations } from "./model-observations.mjs";
|
|
14
15
|
import { cancelSessionWithBridge, createSessionWithBridge, doctor, forkSessionWithBridge, inspectSessionWithBridge, promptSessionWithBridge } from "./dd-grok.mjs";
|
|
15
16
|
import { prepareRuntimeOwner, cleanupFailedDaemonStart, closeManagedBridge, confirmDaemonProcess, confirmDaemonStopped, finishDaemonProcess, scheduleDaemonProcessHeartbeat, registerDaemonProcess, startManagedBridge, stopProcessGroup } from "./managed-daemon.mjs";
|
|
16
17
|
|
|
@@ -201,16 +202,6 @@ export class Runtime {
|
|
|
201
202
|
// A descendant is a managed native Session, not merely a liveness item.
|
|
202
203
|
// Register it in the same registry used by inspect/hook.resolve so a
|
|
203
204
|
// parentless inspection cannot downgrade it into a second root.
|
|
204
|
-
this.sessions.set(id, {
|
|
205
|
-
...prior,
|
|
206
|
-
provider_session_id: id,
|
|
207
|
-
adapter_session_id: prior?.adapter_session_id ?? id,
|
|
208
|
-
parent_provider_session_id: parent,
|
|
209
|
-
root_provider_session_id: rootProviderSessionId,
|
|
210
|
-
cwd,
|
|
211
|
-
native_root_receipt: prior?.native_root_receipt ?? null
|
|
212
|
-
});
|
|
213
|
-
this.bridge.registerModelSession?.(id, id, this.state.config, parent);
|
|
214
205
|
const descendant = this.descendants.get(id);
|
|
215
206
|
const attempt = value.attempt_id;
|
|
216
207
|
const retired = descendant?.retired_attempt_ids ?? [];
|
|
@@ -229,6 +220,16 @@ export class Runtime {
|
|
|
229
220
|
return;
|
|
230
221
|
}
|
|
231
222
|
if (!newAttempt && terminal(descendant?.status) && terminal(status) && descendant.status !== status) throw new DaemonError("native_child_outcome_conflict", "Grok child has contradictory terminal outcomes", false, { session_id: id, previous: descendant.status, observed: status });
|
|
223
|
+
this.sessions.set(id, {
|
|
224
|
+
...prior,
|
|
225
|
+
provider_session_id: id,
|
|
226
|
+
adapter_session_id: prior?.adapter_session_id ?? id,
|
|
227
|
+
parent_provider_session_id: parent,
|
|
228
|
+
root_provider_session_id: rootProviderSessionId,
|
|
229
|
+
cwd,
|
|
230
|
+
native_root_receipt: prior?.native_root_receipt ?? null
|
|
231
|
+
});
|
|
232
|
+
this.bridge.registerModelSession?.(id, id, this.state.config, parent);
|
|
232
233
|
this.descendants.set(id, {
|
|
233
234
|
provider_session_id: id,
|
|
234
235
|
parent_provider_session_id: parent,
|
|
@@ -238,13 +239,20 @@ export class Runtime {
|
|
|
238
239
|
...(newAttempt && descendant?.attempt_id ? { retired_attempt_ids: [...retired, descendant.attempt_id] } : retired.length ? { retired_attempt_ids: retired } : {}),
|
|
239
240
|
...(!newAttempt && !terminal(status) && descendant?.outcome_error ? { outcome_error: descendant.outcome_error } : {})
|
|
240
241
|
});
|
|
242
|
+
if (spawned && typeof value.model === "string" && value.model.trim()) {
|
|
243
|
+
return { sessionId: id, parentSessionId: parent, attemptId: attempt ?? null, parentPromptId: value.parent_prompt_id ?? null, model: value.model.trim() };
|
|
244
|
+
}
|
|
241
245
|
}
|
|
242
246
|
observeSubagentEvent(message) {
|
|
243
247
|
const root = message?.params?.sessionId, update = message?.params?.update;
|
|
244
248
|
const kind = String(update?.sessionUpdate ?? update?.type ?? message?.method ?? "");
|
|
245
|
-
if (!root) return;
|
|
249
|
+
if (!root) return [];
|
|
250
|
+
const modelFacts = [];
|
|
246
251
|
if (/subagent/i.test(kind)) {
|
|
247
|
-
for (const value of [update?.subagent, update?.subagentInfo, update?.child, update?.result, update])
|
|
252
|
+
for (const value of [update?.subagent, update?.subagentInfo, update?.child, update?.result, update]) {
|
|
253
|
+
const fact = this.recordDescendant(value, root, "x.ai/subagent/event", /fail|error/i.test(kind) ? "failed" : /cancel/i.test(kind) ? "cancelled" : /complete|finish|done/i.test(kind) ? "completed" : "running");
|
|
254
|
+
if (fact) modelFacts.push(fact);
|
|
255
|
+
}
|
|
248
256
|
}
|
|
249
257
|
// list_running omits completed children. Retain native tool receipts, never
|
|
250
258
|
// extract identities from the model's prose or from arbitrary shell output.
|
|
@@ -254,7 +262,7 @@ export class Runtime {
|
|
|
254
262
|
if (["spawn_subagent", "get_command_or_subagent_output"].includes(tool) && !this.subagentCalls.has(key)) this.subagentCalls.set(key, { name: tool, input: update.rawInput, attempts: new Map([...this.descendants].map(([id, child]) => [id, child.attempt_id])) });
|
|
255
263
|
const call = this.subagentCalls.get(key), name = call?.name, output = update?.rawOutput;
|
|
256
264
|
if (call && update.rawInput) call.input = update.rawInput;
|
|
257
|
-
if (kind !== "tool_call_update" || update.status !== "completed") return;
|
|
265
|
+
if (kind !== "tool_call_update" || update.status !== "completed") return modelFacts;
|
|
258
266
|
if (name === "spawn_subagent" && output?.type === "Text") {
|
|
259
267
|
const id = /^subagent_id:\s*(\S+)\s*$/m.exec(output.text ?? "")?.[1];
|
|
260
268
|
if (id) this.recordDescendant({ session_id: id, stale_activity: !call.attempts.has(id) || call.attempts.get(id) !== this.descendants.get(id)?.attempt_id }, root, "x.ai/tool/spawn_subagent", "running");
|
|
@@ -280,6 +288,7 @@ export class Runtime {
|
|
|
280
288
|
}
|
|
281
289
|
if (firstError) throw firstError;
|
|
282
290
|
}
|
|
291
|
+
return modelFacts;
|
|
283
292
|
}
|
|
284
293
|
topology(root) { if (this.state.observation_error) { const error = this.state.observation_error; throw new DaemonError(error.code, error.message, false, error.details); } return [...this.descendants.values()].filter((child) => child.parent_provider_session_id === root); }
|
|
285
294
|
observeToolCall(message) { this.toolObservations ??= new ToolObservations(); this.toolObservations.add(acpToolObservation(message, { harness: "grok-acp" })); const sessionId = message?.params?.sessionId; if (sessionId) this.toolUsage.set(sessionId, this.toolObservations.summary(sessionId)); }
|
|
@@ -377,7 +386,14 @@ export async function serveDaemon(stateDir) {
|
|
|
377
386
|
const sessionId = message?.params?.sessionId;
|
|
378
387
|
if (!runtime || !sessionId) return;
|
|
379
388
|
try {
|
|
380
|
-
runtime.observeToolCall(message);
|
|
389
|
+
runtime.observeToolCall(message);
|
|
390
|
+
for (const fact of runtime.observeSubagentEvent(message)) {
|
|
391
|
+
const source = { channel: "x.ai/subagent_spawned", scope: "configured", attempt_id: fact.attemptId, parent_prompt_id: fact.parentPromptId,
|
|
392
|
+
event_sha256: createHash("sha256").update(JSON.stringify(fact)).digest("hex") };
|
|
393
|
+
await observeModel({ journal: state.config.journal, harness: "grok-acp", sessionId: fact.sessionId, parentSessionId: fact.parentSessionId,
|
|
394
|
+
requested: state.config, observed: { model: fact.model }, source, evidence: "configured" });
|
|
395
|
+
bridge.modelProfiles?.set(fact.sessionId, { model: fact.model });
|
|
396
|
+
}
|
|
381
397
|
} catch (error) {
|
|
382
398
|
// Native notifications are fire-and-forget. Persist topology conflicts
|
|
383
399
|
// for the next inspect/status call instead of creating an unhandled
|
|
@@ -388,6 +404,12 @@ export async function serveDaemon(stateDir) {
|
|
|
388
404
|
const rootProviderSessionId = session?.root_provider_session_id ?? sessionId;
|
|
389
405
|
await observeAcpToolCall(state.config, { daemonId: state.daemon_id, rootProviderSessionId, parentProviderSessionId: session?.parent_provider_session_id ?? null, observedProfile: bridge.modelProfiles?.get(sessionId) ?? {} }, message);
|
|
390
406
|
} }); const initialized = await startManagedBridge({ config: state.config, state, bridge, kind: "grok-provider", publish: snapshot => writeState(paths.state, snapshot) }); runtime = new Runtime(paths, state, bridge, initialized);
|
|
407
|
+
for (const fact of await readModelObservations(state.config.journal)) {
|
|
408
|
+
if (fact.harness === "grok-acp" && fact.source?.channel === "x.ai/subagent_spawned" && runtime.sessions.has(fact.session_id)
|
|
409
|
+
&& (runtime.descendants.get(fact.session_id)?.attempt_id ?? null) === (fact.source.attempt_id ?? null) && fact.observed?.model) {
|
|
410
|
+
bridge.modelProfiles?.set(fact.session_id, { model: fact.observed.model });
|
|
411
|
+
}
|
|
412
|
+
}
|
|
391
413
|
scheduleDaemonProcessHeartbeat(state.config, state.resource_process);
|
|
392
414
|
await removeSocket(paths.socket, paths.dir); const server = net.createServer((connection) => { connection.setEncoding("utf8"); let buffer = ""; connection.on("data", (chunk) => { buffer += chunk; const newline = buffer.indexOf("\n"); if (newline < 0) return; const line = buffer.slice(0, newline); buffer = buffer.slice(newline + 1); void (async () => { let request; try { request = JSON.parse(line); if (request.schema_id !== REQUEST_SCHEMA || !request.id || !request.operation) throw new DaemonError("daemon_protocol_mismatch", "invalid daemon request"); const result = await durableDaemonDispatch(paths.dir, request, () => runtime.dispatch(request.operation, request.params ?? {}), () => runtime.dispatch("session.inspect", request.params ?? {})); connection.end(`${JSON.stringify({ schema_id: RESPONSE_SCHEMA, id: request.id, ok: true, result: { ...result, _shutdown: undefined } })}\n`); if (result?._shutdown) setImmediate(() => void runtime.shutdown().then(() => process.exit(0)).catch(error => runtime.persist({ shutdown_state: "cleanup_failed", cleanup_error: errorPayload(error) }))); } catch (error) { connection.end(`${JSON.stringify({ schema_id: RESPONSE_SCHEMA, id: request?.id ?? null, ok: false, error: errorPayload(error) })}\n`); } })(); }); }); runtime.server = server; await new Promise((resolve, reject) => { server.once("error", reject); server.listen(paths.socket, resolve); }); await chmod(paths.socket, 0o600); await runtime.persist({ shutdown_state: "running", active_tree: false }); runtime.ready = true; return await new Promise(() => {});
|
|
393
415
|
}
|
|
@@ -77,7 +77,9 @@ async function inspect(bridge, sessionId, options) {
|
|
|
77
77
|
const [info, topology, usage] = await Promise.all([sessionInfo(bridge, sessionId), subagents(bridge, sessionId), sessionUsage(bridge, sessionId)]);
|
|
78
78
|
const native = info?.data ?? info ?? {}, model = typeof native.model === "string" ? native.model : native.model?.modelId ?? null;
|
|
79
79
|
const current = model ? { model, provider: native.provider ?? null, reasoning: native.reasoningEffort ?? null, mode: native.mode ?? null } : bridge.modelProfiles?.get(sessionId) ?? { model: null, provider: null, reasoning: null, mode: null };
|
|
80
|
-
await observeModel({ journal: options.journal, harness: "grok-acp", sessionId, requested: profile(options), observed:
|
|
80
|
+
await observeModel({ journal: options.journal, harness: "grok-acp", sessionId, requested: profile(options), observed: model ? current : {},
|
|
81
|
+
source: { channel: "x.ai/session/info", scope: "configured", ...(!model ? { non_asserting: true } : {}) },
|
|
82
|
+
evidence: model ? "configured" : "unavailable", reason: model ? null : "session_info_omits_model" });
|
|
81
83
|
return { info, subagents: topology, usage, observed_profile: model ? current : { model: null, provider: null, reasoning: null, mode: null } };
|
|
82
84
|
}
|
|
83
85
|
|
|
@@ -139,7 +141,6 @@ export async function createSessionWithBridge(bridge, options, initialized) {
|
|
|
139
141
|
const created = await bridge.request("session/new", { cwd, mcpServers: [], ...(options.authMethodId ? { authMethodId: options.authMethodId } : {}) });
|
|
140
142
|
const providerSessionId = text(created.sessionId, "Grok Build sessionId");
|
|
141
143
|
bridge.registerModelSession?.(providerSessionId, providerSessionId, requested);
|
|
142
|
-
await observeModel({ journal: options.journal, harness: "grok-acp", sessionId: providerSessionId, requested, observed: observedProfile(initialized, requested), source: "acp.initialize.modelState" });
|
|
143
144
|
const profileReceipt = assertProfile(requested, observedProfile(initialized, requested));
|
|
144
145
|
await options.onSessionCreated?.({ provider_session_id: providerSessionId, adapter_session_id: providerSessionId, cwd });
|
|
145
146
|
const local = { ...options, initialized, rootProviderSessionId: providerSessionId }; const initial_usage_error = await forwardUsageBestEffort(local, providerSessionId, await sessionUsage(bridge, providerSessionId), sessionToolCalls(local, bridge, providerSessionId));
|
|
@@ -805,6 +805,7 @@ export function zcodeLifecycleQualification(sha256, bridgeCommit, harnessContrac
|
|
|
805
805
|
["5a8c89ab716a4b2ccc50df81cb4d204ff5c06368", "dd-zcode-harness@2"],
|
|
806
806
|
["91a1470650f5fd6105f9863fd31d1cad7b90a109", "dd-zcode-harness@2"],
|
|
807
807
|
["b9e0f5106dddb439b76186278bd9be5f45dbf3ef", "dd-zcode-harness@2"],
|
|
808
|
+
["636b14181d5c9a95476055079836fd6b58169f58", "dd-zcode-harness@2"],
|
|
808
809
|
]);
|
|
809
810
|
const qualified = qualifiedBridgeCommits.get(bridgeCommit) === harnessContract;
|
|
810
811
|
return { status: qualified ? "qualified" : "unqualified", mechanism: "cli-invocation@1", sha256,
|
|
@@ -51,7 +51,7 @@ export function delegationContract(harness) {
|
|
|
51
51
|
|
|
52
52
|
export const lifecycleCommandCompletionInstruction = "Run runtime-issued lifecycle CLI calls in the foreground, never with a native background or detached-shell option. A shell tool may yield before the lifecycle process exits. Retain and poll its exact process/session handle to completion, including inside a tool wrapper; empty interim output is not a CLI result. Read final stdout or response_file before deciding continuation, and never reissue the command while it is pending.";
|
|
53
53
|
|
|
54
|
-
export const lifecycleRetryInstruction = `${lifecycleCommandCompletionInstruction} A failure is not permission to replay a lifecycle command. For proven effect=no_effect with recoverable=true and retry_command, correct the input and execute that exact issued successor. A check failure can retain receipts while leaving Work running: correct the evidenced cause first, then use the CLI's exact retry_command for an intentional fresh check attempt; never replay a check merely because Work is unsettled. If continuation.kind=repair_required, or a successful result has outcome=repair_required and a repair Work ID, end this Turn for the registered repair child; use any retry_command only after that repair settles and verification is refreshed. Unknown effects require reconciliation, never replay. A committed effect without an explicit continuation is not retryable. A native shell launch refusal before the CLI starts permits correcting only that shell-form error and retrying the same exact command. Otherwise report the primary error and stop; never invent IDs, call hooks directly, or run status as a substitute.`;
|
|
54
|
+
export const lifecycleRetryInstruction = `${lifecycleCommandCompletionInstruction} A failure is not permission to replay a lifecycle command. For proven effect=no_effect with recoverable=true and retry_command, correct the input and execute that exact issued successor. A check failure can retain receipts while leaving Work running: correct the evidenced cause first, then use the CLI's exact retry_command for an intentional fresh check attempt; never replay a check merely because Work is unsettled. If a successful CODE finish returns outcome=verification_required, inspect the retained receipts, update its verification file, and use only next.finish_command: the Stage is still running and the previous invocation ID is consumed. If continuation.kind=repair_required, or a successful result has outcome=repair_required and a repair Work ID, end this Turn for the registered repair child; use any retry_command only after that repair settles and verification is refreshed. Unknown effects require reconciliation, never replay. A committed effect without an explicit continuation is not retryable. A native shell launch refusal before the CLI starts permits correcting only that shell-form error and retrying the same exact command. Otherwise report the primary error and stop; never invent IDs, call hooks directly, or run status as a substitute.`;
|
|
55
55
|
|
|
56
56
|
export const stageLifecycleInstruction = [
|
|
57
57
|
"The project root identifies this RUN; use the Stage packet's declared execution workspace and absolute artifact paths. Do not change a native Session's working directory to a workspace provisioned for a later Stage.",
|
|
@@ -35,10 +35,11 @@ export async function observeModel({ journal, harness, sessionId, parentSessionI
|
|
|
35
35
|
await mkdir(path.dirname(file), { recursive: true, mode: 0o700 });
|
|
36
36
|
return withRunnerLock(file, async () => {
|
|
37
37
|
const events = await readModelObservations(journal);
|
|
38
|
-
const replay = source?.event_sha256 ? events.find(item => item.session_id === sessionId && item.source?.event_sha256 === source.event_sha256) : null;
|
|
38
|
+
const replay = source?.event_sha256 ? events.find(item => item.harness === harness && item.session_id === sessionId && item.source?.event_sha256 === source.event_sha256) : null;
|
|
39
39
|
if (replay) return replay;
|
|
40
40
|
const channel = source?.channel ?? evidence;
|
|
41
|
-
const previous = events.findLast(item => item.harness === harness && item.session_id === sessionId && (item.source?.channel ?? item.evidence) === channel
|
|
41
|
+
const previous = events.findLast(item => item.harness === harness && item.session_id === sessionId && (item.source?.channel ?? item.evidence) === channel
|
|
42
|
+
&& (item.source?.attempt_id ?? null) === (source?.attempt_id ?? null));
|
|
42
43
|
const normalizedReason = text(reason)?.slice(0, 200) ?? null;
|
|
43
44
|
if (previous && JSON.stringify(previous.observed) === JSON.stringify(profile) && previous.reason === normalizedReason && previous.evidence === evidence) return previous;
|
|
44
45
|
const previousKnown = events.findLast(item => item.harness === harness && item.session_id === sessionId && item.evidence === evidence && item.observed?.model);
|
|
@@ -62,16 +63,32 @@ export async function observeModel({ journal, harness, sessionId, parentSessionI
|
|
|
62
63
|
});
|
|
63
64
|
}
|
|
64
65
|
|
|
65
|
-
export function modelAttribution(events) {
|
|
66
|
+
export function modelAttribution(events, expectedSessions = []) {
|
|
66
67
|
const sessions = new Map();
|
|
67
68
|
for (const event of events) {
|
|
68
69
|
const key = `${event.harness}:${event.session_id}`;
|
|
69
70
|
const session = sessions.get(key) ?? { harness: event.harness, session_id: event.session_id, parent_session_id: event.parent_session_id, requested: event.requested, observations: [] };
|
|
71
|
+
session.parent_session_id ??= event.parent_session_id;
|
|
72
|
+
if (!session.requested || !Object.values(session.requested).some(Boolean)) session.requested = event.requested;
|
|
70
73
|
session.observations.push(event); sessions.set(key, session);
|
|
71
74
|
}
|
|
75
|
+
const observedIds = new Set([...sessions.values()].map(session => session.session_id));
|
|
76
|
+
const missingSessions = expectedSessions.filter(session => typeof session === "string" ? !observedIds.has(session)
|
|
77
|
+
: !sessions.has(`${session.harness}:${session.session_id}`));
|
|
72
78
|
const models = [...new Set(events.map(event => event.observed?.model).filter(Boolean))];
|
|
73
|
-
const incomplete = !events.length || [...sessions.values()].some(session =>
|
|
79
|
+
const incomplete = !events.length || missingSessions.length > 0 || [...sessions.values()].some(session => {
|
|
80
|
+
const known = session.observations.filter(event => event.observed?.model);
|
|
81
|
+
if (!known.length) return true;
|
|
82
|
+
const knownScopes = new Set(known.map(event => `${event.evidence}:${event.source?.scope ?? "session"}:${event.source?.attempt_id ?? ""}:${event.source?.turn_id ?? ""}`));
|
|
83
|
+
return session.observations.some(event => {
|
|
84
|
+
if (event.evidence !== "unavailable") return false;
|
|
85
|
+
if (event.source?.non_asserting) return false;
|
|
86
|
+
const scope = event.source?.scope;
|
|
87
|
+
return !scope || !knownScopes.has(`configured:${scope}:${event.source?.attempt_id ?? ""}:${event.source?.turn_id ?? ""}`)
|
|
88
|
+
&& !knownScopes.has(`response:${scope}:${event.source?.attempt_id ?? ""}:${event.source?.turn_id ?? ""}`);
|
|
89
|
+
});
|
|
90
|
+
});
|
|
74
91
|
return { models, mixed: models.length > 1 || events.some(event => event.observed?.model && event.requested?.model && event.observed.model !== event.requested.model),
|
|
75
92
|
observation_completeness: incomplete ? "incomplete" : "available_native_sources", transitions: events.filter(event => event.kind === "model_changed"),
|
|
76
|
-
sessions: [...sessions.values()], usage_attribution: "session_totals_unattributed_to_model", note: "Native settings/events do not prove undisclosed server routing or per-model token costs." };
|
|
93
|
+
sessions: [...sessions.values()], missing_sessions: missingSessions, usage_attribution: "session_totals_unattributed_to_model", note: "Native settings/events do not prove undisclosed server routing or per-model token costs." };
|
|
77
94
|
}
|
|
@@ -8,10 +8,12 @@ export interface ToolEvent {
|
|
|
8
8
|
}
|
|
9
9
|
export class ToolObservations {
|
|
10
10
|
calls: Map<string, ToolEvent>;
|
|
11
|
+
turns: Map<string, { session: { harness_id: string; session_id: string }; turn_id: string; started: boolean; completed: boolean }>;
|
|
11
12
|
gaps: Set<string>;
|
|
12
13
|
add(event: ToolEvent | null): void;
|
|
13
|
-
|
|
14
|
+
observeTurn(event: { session: { harness_id: string; session_id: string }; turn_id: string; started?: boolean; completed?: boolean }): void;
|
|
15
|
+
summary(sessionId?: string, cursor?: Set<string>): { total: number; failures: number; pending: number; by_tool: Record<string, number>; completeness: string; outcome_completeness: string; sessions: Array<{ session: { harness_id: string; session_id: string }; total: number; failures: number; pending: number; by_tool: Record<string, number> }>; reasons: string[] };
|
|
14
16
|
cursor(): Set<string>;
|
|
15
17
|
}
|
|
16
18
|
export function codexToolObservation(record: Record<string, unknown>, sessionId: string): ToolEvent | null;
|
|
17
|
-
export function replayToolJournal(file: string, options: { harness: string; rootSessionId: string; adapterSessionId?: string }, reducer?: ToolObservations): ToolObservations;
|
|
19
|
+
export function replayToolJournal(file: string, options: { harness: string; rootSessionId: string; adapterSessionId?: string; text?: string; expectedSessionIds?: string[] }, reducer?: ToolObservations): ToolObservations;
|
|
@@ -3,7 +3,14 @@ import fs from "node:fs";
|
|
|
3
3
|
/** Canonical physical-session tool accounting. Native decoding lives below;
|
|
4
4
|
* consumers never have to inspect provider envelopes or aggregate tokens. */
|
|
5
5
|
export class ToolObservations {
|
|
6
|
-
constructor() { this.calls = new Map(); this.gaps = new Set(); }
|
|
6
|
+
constructor() { this.calls = new Map(); this.turns = new Map(); this.gaps = new Set(); }
|
|
7
|
+
observeTurn(event) {
|
|
8
|
+
if (!event?.session?.harness_id || !event.session.session_id || !event.turn_id) { this.gaps.add("turn_identity_missing"); return; }
|
|
9
|
+
const key = JSON.stringify([event.session, event.turn_id]);
|
|
10
|
+
const old = this.turns.get(key);
|
|
11
|
+
this.turns.set(key, { ...old, ...event, started: old?.started || event.started || false,
|
|
12
|
+
completed: old?.completed || event.completed || false });
|
|
13
|
+
}
|
|
7
14
|
add(event) {
|
|
8
15
|
if (!event) return;
|
|
9
16
|
if (event.reason) { this.gaps.add(event.reason); return; }
|
|
@@ -13,10 +20,12 @@ export class ToolObservations {
|
|
|
13
20
|
const old = this.calls.get(key);
|
|
14
21
|
// Replay and late starts cannot undo a terminal outcome. Conflicting
|
|
15
22
|
// terminal facts remain visible rather than silently choosing success.
|
|
16
|
-
const
|
|
17
|
-
if (old &&
|
|
23
|
+
const proven = value => ["completed", "failed"].includes(value);
|
|
24
|
+
if (old && proven(old.status) && proven(event.status) && old.status !== event.status) this.gaps.add("conflicting_tool_outcomes");
|
|
18
25
|
this.calls.set(key, { ...old, ...event, name: event.name ?? old?.name ?? "unknown",
|
|
19
|
-
status: old?.status === "failed" || event.status === "failed" ? "failed"
|
|
26
|
+
status: old?.status === "failed" || event.status === "failed" ? "failed"
|
|
27
|
+
: old?.status === "completed" || event.status === "completed" ? "completed"
|
|
28
|
+
: old?.status === "unknown" || event.status === "unknown" ? "unknown" : "running" });
|
|
20
29
|
}
|
|
21
30
|
summary(sessionId, cursor) {
|
|
22
31
|
const by_tool = {}, sessions = new Map(), reasons = new Set(this.gaps); let total = 0, failures = 0, pending = 0;
|
|
@@ -30,6 +39,12 @@ export class ToolObservations {
|
|
|
30
39
|
if (!["completed", "failed"].includes(value.status)) { pending++; session.pending++; }
|
|
31
40
|
sessions.set(id, session);
|
|
32
41
|
}
|
|
42
|
+
if (!cursor) for (const turn of this.turns.values()) {
|
|
43
|
+
if (sessionId !== undefined && turn.session.session_id !== sessionId) continue;
|
|
44
|
+
const id = JSON.stringify(turn.session);
|
|
45
|
+
sessions.set(id, sessions.get(id) ?? { session: turn.session, parent_session: null, total: 0, failures: 0, pending: 0, by_tool: {} });
|
|
46
|
+
if (!turn.started || !turn.completed) reasons.add("turn_lifecycle_incomplete");
|
|
47
|
+
}
|
|
33
48
|
return { schema_id: "dd-flow/tool-summary@1", scope: "physical_sessions", measurement: "observed_invocations", origin: "native_tool_events", aggregation: cursor ? "delta" : "cumulative", total, failures, pending, by_tool,
|
|
34
49
|
sessions: [...sessions.values()], completeness: reasons.size ? "partial" : "complete", outcome_completeness: pending ? "partial" : "complete", reasons: [...reasons] };
|
|
35
50
|
}
|
|
@@ -66,7 +81,7 @@ export function acpToolObservation(message, { harness = "zcode-acp", rootSession
|
|
|
66
81
|
const transport = message.params?.sessionId;
|
|
67
82
|
if (adapterSessionId && transport !== adapterSessionId) return null;
|
|
68
83
|
const root = rootSessionId ?? transport;
|
|
69
|
-
const session = harness === "zcode-acp" ? zcodePhysicalSession(update, root) : root;
|
|
84
|
+
const session = harness === "zcode-acp" ? zcodePhysicalSession(update, root) : harness === "grok-acp" ? transport : root;
|
|
70
85
|
if (!session || !update.toolCallId) return { reason: "tool_identity_missing" };
|
|
71
86
|
return { session: identity(harness, session), parent_session: session !== root ? identity(harness, root) : null,
|
|
72
87
|
transport_session_id: transport, call_id: update.toolCallId, name: update._meta?.["x.ai/tool"]?.name ?? update._meta?.claudeCode?.toolName ?? update.title?.split(":", 1)[0], status: status(update.status) };
|
|
@@ -85,22 +100,31 @@ export function codexToolObservation(record, sessionId) {
|
|
|
85
100
|
const p = record.payload ?? {};
|
|
86
101
|
if (["function_call", "custom_tool_call"].includes(p.type)) return { session: identity("codex-desktop", sessionId), call_id: p.call_id, name: p.name, status: "running" };
|
|
87
102
|
if (["function_call_output", "custom_tool_call_output"].includes(p.type)) {
|
|
88
|
-
return { session: identity("codex-desktop", sessionId), call_id: p.call_id, status:
|
|
103
|
+
return { session: identity("codex-desktop", sessionId), call_id: p.call_id, status: structuredFailure(p.output) ? "failed" : "unknown" };
|
|
89
104
|
}
|
|
90
105
|
return null;
|
|
91
106
|
}
|
|
92
107
|
|
|
93
|
-
function
|
|
94
|
-
if (Array.isArray(value)) return value.some(containsFailure);
|
|
95
|
-
if (typeof value === "string") {
|
|
96
|
-
if (/(?:^|\n)exit=(?:[1-9]\d*)\b|Process exited with code [1-9]\d*/.test(value)) return true;
|
|
97
|
-
if (/^[{[]/.test(value.trim())) { try { return containsFailure(JSON.parse(value)); } catch {} }
|
|
98
|
-
return false;
|
|
99
|
-
}
|
|
108
|
+
function structuredFailure(value) {
|
|
100
109
|
if (!value || typeof value !== "object") return false;
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
110
|
+
return value.ok === false || value.is_error === true || value.isError === true || value.error === true
|
|
111
|
+
|| typeof value.exit_code === "number" && value.exit_code !== 0;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const codexTools = new Set(["commandExecution", "fileChange", "mcpToolCall", "dynamicToolCall", "collabAgentToolCall", "webSearch", "imageView", "imageGeneration", "sleep"]);
|
|
115
|
+
const codexNonTools = new Set(["userMessage", "hookPrompt", "agentMessage", "functionCallOutput", "plan", "reasoning", "subAgentActivity", "enteredReviewMode", "exitedReviewMode", "contextCompaction"]);
|
|
116
|
+
export function codexAppServerObservation(message) {
|
|
117
|
+
if (!["item/started", "item/completed"].includes(message?.method)) return null;
|
|
118
|
+
const { threadId, turnId, item } = message.params ?? {};
|
|
119
|
+
if (!item || codexNonTools.has(item.type)) return null;
|
|
120
|
+
if (!codexTools.has(item.type)) return { reason: "codex_item_type_unsupported" };
|
|
121
|
+
if (!threadId || !item.id) return { reason: "tool_identity_missing" };
|
|
122
|
+
const native = item.status;
|
|
123
|
+
const failed = ["failed", "declined", "interrupted"].includes(native) || item.type === "commandExecution" && typeof item.exitCode === "number" && item.exitCode !== 0;
|
|
124
|
+
const completed = message.method === "item/completed" && (native === undefined || native === "completed");
|
|
125
|
+
const known = native === undefined || ["inProgress", "completed", "failed", "declined", "interrupted"].includes(native);
|
|
126
|
+
return { session: identity("codex-desktop", threadId), epoch: turnId ?? "", call_id: item.id, name: item.type,
|
|
127
|
+
status: !known ? "unknown" : failed ? "failed" : completed ? "completed" : "running", ...(!known ? { reason: "codex_item_status_unsupported" } : {}) };
|
|
104
128
|
}
|
|
105
129
|
|
|
106
130
|
export function openCodeToolObservation(part, sessionId, messageId) {
|
|
@@ -121,7 +145,7 @@ export function agyToolObservation(step, rootSessionId) {
|
|
|
121
145
|
* carries physical identities; an ACP connection identity never crosses out. */
|
|
122
146
|
export function replayToolJournal(file, options, reducer = new ToolObservations()) {
|
|
123
147
|
let bytes;
|
|
124
|
-
try { bytes = fs.readFileSync(file, "utf8"); } catch (error) { reducer.gaps.add(error.code === "ENOENT" ? "journal_missing" : "journal_unreadable"); return reducer; }
|
|
148
|
+
try { bytes = options.text ?? fs.readFileSync(file, "utf8"); } catch (error) { reducer.gaps.add(error.code === "ENOENT" ? "journal_missing" : "journal_unreadable"); return reducer; }
|
|
125
149
|
const lines = bytes.split("\n"); let session = options.rootSessionId, transport = options.adapterSessionId;
|
|
126
150
|
for (const [index, line] of lines.entries()) {
|
|
127
151
|
if (!line.trim()) continue;
|
|
@@ -135,7 +159,14 @@ export function replayToolJournal(file, options, reducer = new ToolObservations(
|
|
|
135
159
|
if (e.kind === "inbound" || m.method) add(acpToolObservation(m, { ...options, rootSessionId: session, adapterSessionId: transport }));
|
|
136
160
|
} else if (options.harness === "codex-desktop") {
|
|
137
161
|
if (e.type === "session_meta" && e.payload?.id !== session) { reducer.gaps.add("session_identity_mismatch"); break; }
|
|
138
|
-
|
|
162
|
+
if (m.params?.threadId && options.expectedSessionIds && !options.expectedSessionIds.includes(m.params.threadId)) continue;
|
|
163
|
+
if (["turn/started", "turn/completed"].includes(m.method)) {
|
|
164
|
+
const params = m.params ?? {};
|
|
165
|
+
if (params.threadId && params.turn?.id) reducer.observeTurn({ session: identity("codex-desktop", params.threadId), turn_id: params.turn.id,
|
|
166
|
+
started: m.method === "turn/started", completed: m.method === "turn/completed" && params.turn.status === "completed" });
|
|
167
|
+
else reducer.gaps.add("turn_identity_missing");
|
|
168
|
+
}
|
|
169
|
+
add(codexAppServerObservation(m) ?? codexToolObservation(e, session));
|
|
139
170
|
} else if (options.harness === "droid-cli") {
|
|
140
171
|
if (e.type === "session_start" && e.id !== session) { reducer.gaps.add("session_identity_mismatch"); break; }
|
|
141
172
|
for (const event of droidToolObservations(e, session)) add(event);
|
|
@@ -1,14 +1,16 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
-
"$id": "dd-flow/code-verification@
|
|
3
|
+
"$id": "dd-flow/code-verification@3",
|
|
4
4
|
"type": "object",
|
|
5
5
|
"additionalProperties": false,
|
|
6
6
|
"required": ["schema_id", "verdict", "summary", "unresolved", "deviations"],
|
|
7
7
|
"properties": {
|
|
8
|
-
"schema_id": {"
|
|
9
|
-
"verdict": {"enum": ["passed", "needs_repair", "blocked"]},
|
|
8
|
+
"schema_id": {"enum": ["dd-flow/code-verification@2", "dd-flow/code-verification@3"]},
|
|
9
|
+
"verdict": {"enum": ["passed", "needs_repair", "needs_checks", "blocked"]},
|
|
10
10
|
"summary": {"type": "string", "minLength": 1},
|
|
11
11
|
"unresolved": {"type": "array", "items": {"type": "string", "minLength": 1}},
|
|
12
|
-
"deviations": {"type": "array", "items": {"type": "string", "minLength": 1}}
|
|
13
|
-
|
|
12
|
+
"deviations": {"type": "array", "items": {"type": "string", "minLength": 1}},
|
|
13
|
+
"document_updates": {"type": "array", "uniqueItems": true, "items": {"type": "string", "minLength": 1}}
|
|
14
|
+
},
|
|
15
|
+
"allOf": [{"if": {"properties": {"schema_id": {"const": "dd-flow/code-verification@2"}}}, "then": {"properties": {"verdict": {"enum": ["passed", "needs_repair", "blocked"]}}, "not": {"required": ["document_updates"]}}}]
|
|
14
16
|
}
|
|
@@ -79,32 +79,67 @@ export function runObservations(context, projectId, runId) {
|
|
|
79
79
|
reasons.push("lineage_evidence_invalid");
|
|
80
80
|
}
|
|
81
81
|
}
|
|
82
|
-
// Native transcript adapters need no managed controller journal.
|
|
83
|
-
for (const session of sessions)
|
|
84
|
-
if (session.transcript_path && ![...sources.values()].some(s => s.session.harness_id === session.harness && s.session.session_id === (session.provider_session_id ?? session.session_id))) {
|
|
85
|
-
add(`transcript:${session.harness}:${session.provider_session_id ?? session.session_id}`, session.provider_session_id ?? session.session_id, session.transcript_path, home, "current", session.harness);
|
|
86
|
-
}
|
|
87
82
|
const reducer = new ToolObservations();
|
|
88
83
|
for (const reason of reasons)
|
|
89
84
|
reducer.gaps.add(reason);
|
|
85
|
+
const expectedCodexIds = sessions.filter(session => session.harness === "codex-desktop").map(session => session.provider_session_id ?? session.session_id);
|
|
86
|
+
const replay = (source, sessionId) => {
|
|
87
|
+
const file = path.join(home, source.path);
|
|
88
|
+
let fd = null;
|
|
89
|
+
try {
|
|
90
|
+
const actual = fs.realpathSync(file);
|
|
91
|
+
if (!inside(home, actual))
|
|
92
|
+
throw new Error("journal escaped owned home");
|
|
93
|
+
fd = fs.openSync(actual, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW);
|
|
94
|
+
if (!fs.fstatSync(fd).isFile())
|
|
95
|
+
throw new Error("journal is not a regular file");
|
|
96
|
+
replayToolJournal(file, { harness: source.session.harness_id, rootSessionId: sessionId, text: fs.readFileSync(fd, "utf8"), ...(source.session.harness_id === "codex-desktop" ? { expectedSessionIds: expectedCodexIds } : {}) }, reducer);
|
|
97
|
+
}
|
|
98
|
+
catch {
|
|
99
|
+
source.status = "unavailable";
|
|
100
|
+
source.reason = "journal_unavailable_during_read";
|
|
101
|
+
}
|
|
102
|
+
finally {
|
|
103
|
+
if (fd !== null)
|
|
104
|
+
fs.closeSync(fd);
|
|
105
|
+
}
|
|
106
|
+
};
|
|
90
107
|
for (const source of sources.values()) {
|
|
91
|
-
if (source.status !== "available")
|
|
92
|
-
reducer.gaps.add(source.reason ?? "journal_unavailable");
|
|
108
|
+
if (source.status !== "available")
|
|
93
109
|
continue;
|
|
94
|
-
|
|
95
|
-
|
|
110
|
+
replay(source, source.session.session_id);
|
|
111
|
+
}
|
|
112
|
+
const covered = () => new Set(reducer.summary().sessions.map(item => JSON.stringify(item.session)));
|
|
113
|
+
// A managed root journal can contain all of its children's native events.
|
|
114
|
+
// Select transcript fallback only after replay reveals an actual gap.
|
|
115
|
+
for (const session of sessions) {
|
|
116
|
+
const nativeId = session.provider_session_id ?? session.session_id;
|
|
117
|
+
if (!session.transcript_path || covered().has(JSON.stringify({ harness_id: session.harness, session_id: nativeId })))
|
|
118
|
+
continue;
|
|
119
|
+
const id = `transcript:${session.harness}:${nativeId}`;
|
|
120
|
+
add(id, nativeId, session.transcript_path, home, "current", session.harness);
|
|
121
|
+
const source = sources.get(id);
|
|
122
|
+
if (source?.status === "available")
|
|
123
|
+
replay(source, nativeId);
|
|
96
124
|
}
|
|
125
|
+
for (const source of sources.values())
|
|
126
|
+
if (source.status !== "available"
|
|
127
|
+
&& !covered().has(JSON.stringify(source.session)))
|
|
128
|
+
reducer.gaps.add(source.reason ?? "journal_unavailable");
|
|
97
129
|
const expected = [...new Map(sessions.map(s => { const session = { harness_id: s.harness, session_id: s.provider_session_id ?? s.session_id }; return [JSON.stringify(session), session]; })).values()];
|
|
98
130
|
const expectedIds = new Set(expected.map(s => JSON.stringify(s)));
|
|
99
131
|
for (const [key, event] of reducer.calls)
|
|
100
132
|
if (!expectedIds.has(JSON.stringify(event.session))) {
|
|
101
133
|
reducer.calls.delete(key);
|
|
102
|
-
|
|
134
|
+
}
|
|
135
|
+
for (const [key, event] of reducer.turns)
|
|
136
|
+
if (!expectedIds.has(JSON.stringify(event.session))) {
|
|
137
|
+
reducer.turns.delete(key);
|
|
103
138
|
}
|
|
104
139
|
const summary = reducer.summary();
|
|
105
140
|
const observed = new Set(summary.sessions.map(s => JSON.stringify(s.session)));
|
|
106
141
|
const missing = expected.filter(s => !observed.has(JSON.stringify(s)));
|
|
107
|
-
const completeness = missing.length || summary.completeness !== "complete" ? "partial" : "complete";
|
|
142
|
+
const completeness = missing.length || summary.completeness !== "complete" || summary.outcome_completeness !== "complete" ? "partial" : "complete";
|
|
108
143
|
return { schema_id: "dd-flow/run-observations@1", home, sources: [...sources.values()], tools: { ...summary,
|
|
109
144
|
completeness, status: observed.size === 0 ? "unavailable" : completeness === "complete" ? "measured" : "partial",
|
|
110
145
|
observed_sessions: expected.length - missing.length, expected_sessions: expected.length, unavailable_sessions: missing.map(s => s.session_id), missing_sessions: missing,
|
|
@@ -2,6 +2,7 @@ import { managedLifecycleCommand } from "./lifecycle-invocations.js";
|
|
|
2
2
|
import { stageLifecycleInstruction } from "../harness-runtime/lib/delegation-instructions.mjs";
|
|
3
3
|
import { ensureAggregateRepair, ensureRepairIntent } from "./repair-intents.js";
|
|
4
4
|
import { runWorkspaceBootstrap } from "./workspace-bootstrap.js";
|
|
5
|
+
import crypto from "node:crypto";
|
|
5
6
|
import fs from "node:fs";
|
|
6
7
|
import path from "node:path";
|
|
7
8
|
import { AppError } from "../shared/errors.js";
|
|
@@ -220,6 +221,12 @@ export async function finishVnextCode(context, input) {
|
|
|
220
221
|
});
|
|
221
222
|
}
|
|
222
223
|
const finalReceipts = receipts;
|
|
224
|
+
if (verification.verdict === "needs_checks") {
|
|
225
|
+
return { ok: true, run_id: run.id, stage, outcome: "verification_required", verification,
|
|
226
|
+
receipts: finalReceipts.map(receipt => ({ id: receipt.id, receipt_path: receipt.receipt_path, input_hash: receipt.input_hash, status: receipt.status })),
|
|
227
|
+
next: { kind: "verification_required", finish_command: finishCommand(context, run.id, projectRoot, input.verificationFile) },
|
|
228
|
+
instruction: "Aggregate receipts are available. Keep CODE running. Inspect them, update code-verification.json to passed or needs_repair, then invoke only the issued successor finish_command. A material document change belongs to the registered fresh repair Work, not this coordinator Session." };
|
|
229
|
+
}
|
|
223
230
|
const projectedVerification = verificationProjection(works, finalReceipts, { historicalReceipts: finalReceipts, runId: run.id, runHome: home });
|
|
224
231
|
const unresolvedAcceptance = (projectedVerification.acceptance ?? []).filter((item) => item.status === "unresolved");
|
|
225
232
|
if (unresolvedAcceptance.length)
|
|
@@ -331,6 +338,15 @@ export function prepareVnextCodeRepair(context, input) {
|
|
|
331
338
|
if (previous)
|
|
332
339
|
return { kind: "replay", project, run, previous };
|
|
333
340
|
}
|
|
341
|
+
if (input.semanticUnresolved?.length && input.verificationPath) {
|
|
342
|
+
const previous = codeWorks(context, project.id, run.id).find(work => {
|
|
343
|
+
const repair = packet(work)?.repair;
|
|
344
|
+
return Boolean(repair && repair.verification_path === input.verificationPath
|
|
345
|
+
&& JSON.stringify(repair.semantic_unresolved) === JSON.stringify(unique(input.semanticUnresolved ?? [])));
|
|
346
|
+
});
|
|
347
|
+
if (previous)
|
|
348
|
+
return { kind: "replay", project, run, previous };
|
|
349
|
+
}
|
|
334
350
|
if (typeof input.objective !== "string" || !input.objective.trim())
|
|
335
351
|
throw new AppError("validation", "Repair objective must not be empty", 2);
|
|
336
352
|
if (input.semanticUnresolved !== undefined && (!Array.isArray(input.semanticUnresolved) || input.semanticUnresolved.some(value => typeof value !== "string" || !value.trim())))
|
|
@@ -346,10 +362,12 @@ export function prepareVnextCodeRepair(context, input) {
|
|
|
346
362
|
}
|
|
347
363
|
if ([Boolean(input.checkReceiptId), Boolean(input.reviewFindings?.length), Boolean(input.semanticUnresolved?.length)].filter(Boolean).length !== 1)
|
|
348
364
|
throw new AppError("validation", "Repair requires exactly one evidence source", 2);
|
|
365
|
+
let semanticVerification = input.verification;
|
|
349
366
|
if (input.semanticUnresolved?.length) {
|
|
350
367
|
if (!input.verificationPath)
|
|
351
368
|
throw new AppError("usage", "Semantic repair requires a verification file", 2);
|
|
352
369
|
const verification = input.verification ?? readVerification(context, { file: input.verificationPath, projectRoot: run.workspace_root, runId: run.id, runRoot: home });
|
|
370
|
+
semanticVerification = verification;
|
|
353
371
|
if (input.verification !== undefined)
|
|
354
372
|
validateSchema({ schemaName: "code-verification", file: input.verificationPath, data: verification, projectRoot: run.workspace_root, ddFlowHome: context.ddFlowHome, runId: run.id, runRoot: home });
|
|
355
373
|
const unresolved = verification.unresolved.length ? verification.unresolved : [verification.summary];
|
|
@@ -417,7 +435,17 @@ export function prepareVnextCodeRepair(context, input) {
|
|
|
417
435
|
acceptance: uniqueBy(invariantPackets.flatMap((value) => value.acceptance), (value) => JSON.stringify(value)),
|
|
418
436
|
// A CODE-REVIEW repair changes delivered code/evidence, never the
|
|
419
437
|
// already accepted PLAN or its ownership declaration.
|
|
420
|
-
document_updates: []
|
|
438
|
+
document_updates: semanticRepair ? uniqueBy((semanticVerification?.document_updates ?? []).map(relative => {
|
|
439
|
+
const declared = invariantPackets.flatMap(value => value.document_updates).find(value => value.path === relative);
|
|
440
|
+
if (!declared)
|
|
441
|
+
throw new AppError("document_update_unknown", "CODE verification selected a document absent from accepted Work ownership", 2, { path: relative });
|
|
442
|
+
const file = path.resolve(run.workspace_root, relative);
|
|
443
|
+
if (!file.startsWith(`${path.resolve(run.workspace_root)}${path.sep}`))
|
|
444
|
+
throw new AppError("document_update_invalid", "CODE verification document path escapes the workspace", 2, { path: relative });
|
|
445
|
+
const exists = fs.existsSync(file);
|
|
446
|
+
return { ...declared, action: exists ? "update" : "create",
|
|
447
|
+
baseline_sha256: exists ? crypto.createHash("sha256").update(fs.readFileSync(file)).digest("hex") : null };
|
|
448
|
+
}), value => value.path) : [],
|
|
421
449
|
required_read: unique([...(receipt ? [receipt.receipt_path] : input.reviewFindings?.flatMap((finding) => finding.evidence_refs) ?? []), ...(semanticRepair && input.verificationPath ? [input.verificationPath] : []), ...packets.flatMap((value) => value.required_read)]),
|
|
422
450
|
discovery_boundary: unique(packets.flatMap((value) => value.discovery_boundary)),
|
|
423
451
|
// These are collision-avoidance hints. Receipt paths enrich the coordinator
|
|
@@ -517,15 +545,15 @@ function coordinatorPrompt(context, input) {
|
|
|
517
545
|
"Do not choose a provider delegation tool from this stage prompt. At each work_fanout boundary, stop at the Work-graph boundary; the shared controller supplies the selected adapter's native tool, exact parameters, child task, and wait contract. It launches only entries listed in graph.ready and uses at most the qualified capacity. Every registered CODE Work runs in a fresh child Session, including a serial dependency chain. The coordinator launches those children only through the controller-supplied native delegation instruction; it must never invoke a Work start_command itself. Each child receives its complete packet from dd-flow and runs its own exact start_command. After a Work finishes, the controller uses the graph returned by work finish to launch newly ready Work. To refresh the parent graph yourself use the exact command: " + runtimeCommand(context, ["work", "ls", "--run", input.run.id, "--ready", "--project-root", input.projectRoot, "--json"]),
|
|
518
546
|
"A quiet child is still running until the harness reports its turn completed, failed, cancelled or explicitly needs attention. An elapsed nominal wait, silence, or no new artifact is not an unresponsive-worker failure. Never interrupt, replace, relaunch, or stage-block a still-running child for that reason, even if an external controller asks. Long work finish and stage finish commands emit check progress on stderr. After you issue the exact CODE stage finish command, wait for that same command to return once: a completed command with a non-zero exit and structured `code_gate_failed` output is its terminal result, not a reason to keep waiting. A repair_required continuation already registered the repair: end this Turn. Do not invoke that Work's start_command from the coordinator Session, inspect its PID, start a second finish command, or infer failure from quiet output. Close a disposable child only after its Work is accepted or explicitly failed/cancelled and the harness reports the turn settled.",
|
|
519
547
|
`A repairable engine, harness, or environment failure is not a user question. Write the evidence summary to ${path.join(requireHome(input.run), "works", input.rootWork.work_id, "block-summary.md")}, then record it with a separate command: ${runtimeCommand(context, ["stage", "block", input.run.id, "--stage", "code", "--work", input.rootWork.work_id, "--kind", "<engine|harness|environment>", "--code", "<stable-code>", "--summary-file", `@run/works/${input.rootWork.work_id}/block-summary.md`, "--retryable", "--project-root", input.projectRoot, "--json"])}. Choose kind and stable-code from evidence; this example is not an executable continuation. Repair it externally, then run the exact unblock_command returned by dd-flow and continue this same stage.`,
|
|
520
|
-
`When every CODE and repair Work is completed, write ${path.join(input.root, "code-verification.json")} using the exact contract below. Mark passed only when all accepted requirements and current-gate acceptance criteria are implemented or explicitly evidenced; list every remaining issue in unresolved. Every evidence_refs item must already exist as a relative workspace path or run://${input.run.id}/ path. Do not claim a browser or other check receipt that was not retained. Then finish: ${finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "code-verification.json"))}`,
|
|
548
|
+
`When every CODE and repair Work is completed, write ${path.join(input.root, "code-verification.json")} using the exact contract below. If closing an accepted document depends on aggregate checks that have not run yet, use verdict needs_checks; the first finish returns receipts and a successor finish_command while CODE remains open. After checking those receipts, use needs_repair with document_updates naming only assigned documents that must change, or passed if the current documents already prove acceptance. Mark passed only when all accepted requirements and current-gate acceptance criteria are implemented or explicitly evidenced; list every remaining issue in unresolved. Every evidence_refs item must already exist as a relative workspace path or run://${input.run.id}/ path. Do not claim a browser or other check receipt that was not retained. Then finish: ${finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "code-verification.json"))}`,
|
|
521
549
|
"A code_gate_failed response with continuation.kind=repair_required has already registered one repair Work with all failed receipts. End this Turn so the controller can dispatch it in a fresh child. Do not recreate the repair, copy origin IDs, write a repair task file or execute the child start command yourself. After repair, update semantic verification and finish again.",
|
|
522
550
|
"</execution_commands>",
|
|
523
551
|
"",
|
|
524
552
|
"<verification_contract>",
|
|
525
553
|
"```json",
|
|
526
|
-
JSON.stringify({ schema_id: "dd-flow/code-verification@
|
|
554
|
+
JSON.stringify({ schema_id: "dd-flow/code-verification@3", verdict: "passed", summary: "Evidence-backed conclusion.", unresolved: [], deviations: [], document_updates: [] }, null, 2),
|
|
527
555
|
"```",
|
|
528
|
-
"Allowed verdict values are passed, needs_repair, and blocked. The example is valid JSON using passed; choose exactly one value.",
|
|
556
|
+
"Allowed verdict values are passed, needs_checks, needs_repair, and blocked. The example is valid JSON using passed; choose exactly one value. A verification_required result is successful receipt gathering, not Stage completion; use its issued successor command after updating this file.",
|
|
529
557
|
"The CLI verifies graph coverage and executes all deterministic checks. Do not duplicate its report; record only your semantic conclusion and evidence.",
|
|
530
558
|
"</verification_contract>",
|
|
531
559
|
"",
|
|
@@ -331,9 +331,11 @@ function preparedExisting(context, input) {
|
|
|
331
331
|
return { ok: true, resumed: true, run_id: input.run.id, stage, outcome: "review_required", work_id: input.workId, id: binding.work_session_id, prompt_path: promptPath, worker_prompt_markdown: fs.readFileSync(promptPath, "utf8"), orchestration, plan_review: { requested_mode: input.requested, effective_mode: input.effective, policy_source: policySource(input.run), groups: input.groups }, next: { dispatch_command: dispatchCommand(context, input.run.id, input.projectRoot), finish_command: finishCommand(context, input.run.id, input.projectRoot, path.join(input.root, "decision.json")) } };
|
|
332
332
|
}
|
|
333
333
|
function orchestratorPrompt(context, input) {
|
|
334
|
-
const {
|
|
334
|
+
const { revision, capacity } = input;
|
|
335
335
|
const decision = path.join(input.root, "decision.json");
|
|
336
336
|
const workspaceContract = ["<workspace_contract>", `- route: ${input.workspaceRoute.route}`, `- feature branch: ${input.workspaceRoute.feature_branch ?? "not applicable"}`, `- base commit: ${input.workspaceRoute.base_ref ?? "not applicable"}`, `- read/write workspace: ${input.run.workspace_root}`, "The CLI verified this frozen route. All plan and correction writes belong in the named workspace; project root remains only the stable lifecycle identity. Do not create, switch, merge or delete branches/worktrees.", "</workspace_contract>"].join("\n");
|
|
337
|
+
const acceptanceOwnership = "For every deferred acceptance or document update, identify the named CODE Work owner, its causal checks, and when it can truthfully close the obligation. Aggregate checks run only after implementation Work completes: assign a CODE owner to pending documents and use CODE needs_checks → verification_required → repair after those receipts. A promise that MERGE or an unnamed readiness owner will finish acceptance is a material PLAN finding; bootstrap workspace readiness is not proof of the new feature.";
|
|
338
|
+
const template = [acceptanceOwnership, input.template].join("\n\n");
|
|
337
339
|
const reviewerLaunch = capacity.source === "external_policy"
|
|
338
340
|
? `After dispatch, return at the Work-graph boundary. The shared runtime launches the queued reviewer Works as separate external Sessions, at most ${capacity.available_slots} in parallel under the RUN's frozen profiles. Do not launch native children, create provider roots, run a capacity probe, or record this external limit as native capacity. Reviewers are read-only leaf workers. Continue the semantic decision only after their Work receipts settle; do not substitute a missing result or relaunch a settled reviewer.`
|
|
339
341
|
: "After dispatch, stop at the Work-graph boundary. The shared controller supplies the selected adapter's native tool, exact parameters, child task, and wait contract; do not choose a provider tool from this stage prompt or substitute a shell command. It launches at most the qualified capacity at once and starts unchanged queued Works only after the current wave settles. A launch rejected before it starts is not review evidence: do not create a replacement. Each reviewer must be a genuinely fresh harness child Session; the lifecycle adapter binds that observed Session, so do not bind or supply a Session ID manually. Reviewers are read-only and must not create children. As soon as a reviewer result is accepted, release that reviewer Session when the harness permits.";
|