@adhdev/daemon-core 0.9.82-rc.406 → 0.9.82-rc.408

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -97,33 +97,86 @@ function resolveReconcileIntervalMs(): number {
97
97
  return DEFAULT_RECONCILE_INTERVAL_MS;
98
98
  }
99
99
 
100
- // RECONCILE-SYNTH-PREEMPTS-COMPLETION in-flight debounce. PHASE 4 only synthesizes a missing
101
- // completion when the worker session reads `idle`. But a worker that is GENUINELY generating
102
- // (it emitted agent:generating_started — the dispatch row is 'acked' — and has not yet
103
- // completed) can momentarily read `idle` mid-turn: a CLI PTY parser sees an inter-turn blip
104
- // between tool calls, or the read_chat probe races a brief settle. A single such idle read
105
- // must NOT fabricate a completion for an in-flight task doing so writes a synthesized
106
- // terminal that then masks the REAL completion when it lands seconds later
100
+ // R4f (GENERATING-BOUNDARY, acked-hold redesign). PHASE 4 only synthesizes a missing completion
101
+ // when the worker session reads `idle`. But a worker that is GENUINELY generating (it emitted
102
+ // agent:generating_started — the dispatch row is 'acked' — and has not yet completed) can
103
+ // momentarily read `idle` mid-turn (a CLI PTY inter-tool-call settle, or the final assistant text
104
+ // already rendered while the turn's generating_completed lifecycle close still lags). A premature
105
+ // synth writes a terminal that then masks the worker's REAL completion when it lands seconds later
107
106
  // (drop:duplicate_completion_terminal_ledger; the observed 71s task a250fb44 lost its [System]
108
- // notification this way). We require the session to read `idle` on CONSECUTIVE reconcile ticks
109
- // before synthesizing for an `acked` (started-generating) dispatch — a transient mid-turn idle
110
- // flicker clears on the next ~4s tick, whereas a genuinely-settled (completed-but-lost) or dead
111
- // session reads idle every tick. A dispatch that was never acked (worker never started) is NOT
112
- // debounced here: there is no in-flight generation to protect, and the downstream grace gate +
113
- // stale-summary guard remain the backstops for that lost-dispatch case.
107
+ // notification this way; the R4e 53s task synth fired 16s BEFORE the worker's real emit).
114
108
  //
115
- // Keyed by `${meshId}::${taskId}`. The map is pruned each PHASE-4 pass to the set of currently
116
- // active dispatches, so a completed/pruned task's counter is dropped (no unbounded growth).
117
- const REQUIRED_CONSECUTIVE_IDLE_TICKS_FOR_INFLIGHT_SYNTH = 2;
118
- const inFlightIdleObservationCounts = new Map<string, number>();
109
+ // R4 R4e used FINITE timers (consecutive ticks / MIN_IDLE_SETTLE / ACKED_TURN_SETTLE) to delay the
110
+ // synth. That class of fix is fundamentally a RACE: the worker's real emit latency is variable and
111
+ // unbounded (win32 idle reads can flip before the emit arrives), so ANY finite timer eventually
112
+ // loses to a slow-enough turn — and the synth pre-empts the real completion. R4e live-FAILED for
113
+ // exactly this reason.
114
+ //
115
+ // R4f redesign (direction B). An `acked` task means the worker ECHOED generating_started (the
116
+ // taskId flip) — it is alive and mid-turn, so it WILL eventually emit a real terminal. We therefore
117
+ // HOLD the synth INDEFINITELY for an acked task. This is safe against the emit actually arriving:
118
+ // when the worker's real generating_completed lands, it writes a terminal ledger, and
119
+ // reconcileDirectDispatchCompletionFromTranscript's hasTerminalLedgerAfterDispatch check makes any
120
+ // later synth an idempotent no-op (alreadyTerminal). So the hold never costs a missed notification —
121
+ // the real emit always wins, no matter how late.
122
+ //
123
+ // The indefinite hold is released ONLY by a genuine-DEATH / emit-loss BACKSTOP — never a finite
124
+ // timer that races normal lag:
125
+ // (a) liveness failure — read_chat reports the session is gone, OR N consecutive read failures
126
+ // accumulate (a transport/session-gone signal, counted as death rather than swallowed via
127
+ // `continue`). A worker that died mid-turn will never emit, so the synth must eventually fire.
128
+ // (b) an absolute LONG death-deadline — time since the generating_started ack exceeds
129
+ // ACKED_DEATH_DEADLINE_MS, a backstop set FAR above any observed emit latency (default 8 min)
130
+ // so it does not race a normal slow turn; it only catches a worker that is genuinely wedged or
131
+ // whose emit was permanently lost. This is a notification-loss net, not a completion timer.
132
+ //
133
+ // A dispatch that was never acked (worker never started) is NOT held here: there is no in-flight
134
+ // generation to protect, so it keeps the existing first-idle-tick synth behavior (its lost-dispatch
135
+ // case is covered by the downstream grace + stale-summary guards). The map is pruned each PHASE-4
136
+ // pass to the set of currently active dispatches, so a completed/pruned task's state is dropped (no
137
+ // unbounded growth). Keyed by `${meshId}::${taskId}`.
138
+
139
+ // R4f backstop (a): how many CONSECUTIVE read_chat failures (transport error / success:false /
140
+ // no payload) for an acked task are treated as a death signal that releases the indefinite hold.
141
+ // A single failed read is a transient probe blip; a session that genuinely died reads-fail every
142
+ // tick, so a small streak distinguishes the two without racing a live-but-slow worker.
143
+ const ACKED_DEATH_CONSECUTIVE_READ_FAILURES = 3;
144
+
145
+ // R4f backstop (b): the absolute death-deadline. An acked task is held indefinitely until this much
146
+ // time has elapsed since its generating_started ack (dispatch.updatedAt); past it, a persistently
147
+ // idle session is synthesized as a notification-loss net. This is set FAR above any observed emit
148
+ // latency (R4e's worst case was ~16s) so it does NOT race a normal slow turn — it only catches a
149
+ // genuinely wedged worker or a permanently-lost emit. Read at call time so tests can tune it.
150
+ function resolveTunedReconcileMs(envName: string, def: number, min: number, max: number): number {
151
+ const raw = readNonEmptyString(process.env[envName]);
152
+ if (raw) {
153
+ const parsed = Number.parseInt(raw, 10);
154
+ if (Number.isFinite(parsed) && parsed >= min && parsed <= max) return parsed;
155
+ }
156
+ return def;
157
+ }
158
+ function resolveAckedDeathDeadlineMs(): number {
159
+ // Default 8 min — FAR above the variable emit latency the finite R4..R4e timers raced (R4e's
160
+ // worst case was ~16s); by the time this fires a live worker would long since have emitted its
161
+ // real terminal. The env-override floor is 0 so tests can force the deadline (production never
162
+ // sets it that low); the ceiling is 60min so a mis-set env cannot disable the loss-net forever.
163
+ return resolveTunedReconcileMs('MESH_INFLIGHT_ACKED_DEATH_DEADLINE_MS', 8 * 60_000, 0, 60 * 60_000);
164
+ }
165
+
166
+ // Per-task in-flight hold state for an acked dispatch:
167
+ // - liveConfirmedSinceAck: we have seen at least one conclusive read (idle OR generating) since
168
+ // the ack — proves the session is reachable, so a later read FAILURE is a genuine liveness loss
169
+ // rather than a node that was never reachable.
170
+ // - consecutiveReadFailures: streak of inconclusive read_chat results (death backstop (a)).
171
+ const inFlightAckedHoldState = new Map<string, { liveConfirmedSinceAck: boolean; consecutiveReadFailures: number }>();
119
172
 
120
173
  function inFlightSynthKey(meshId: string, taskId: string): string {
121
174
  return `${meshId}::${taskId}`;
122
175
  }
123
176
 
124
- // Test hook: clear the in-flight idle debounce state between cases.
177
+ // Test hook: clear the in-flight acked-hold state between cases.
125
178
  export function __resetReconcileInFlightSynthDebounceForTests(): void {
126
- inFlightIdleObservationCounts.clear();
179
+ inFlightAckedHoldState.clear();
127
180
  }
128
181
 
129
182
  interface LiveCoordinator {
@@ -1231,6 +1284,50 @@ function readChatPayloadStatus(payload: Record<string, unknown> | null): string
1231
1284
  return readNonEmptyString(payload?.status).toLowerCase();
1232
1285
  }
1233
1286
 
1287
+ // R4e fix (3): peek the pending-events queue for a REAL (worker-emitted) terminal completion
1288
+ // already queued for a task — used to yield the in-flight synth to the worker's own emit. Broad
1289
+ // peek (no daemon-id scoping) matched precisely by taskId, so a worker stamp in any daemon-id form
1290
+ // is still recognized. Best-effort: a peek failure returns false (proceed to synth — never block
1291
+ // delivery). A prior SYNTH's still-queued pending event also names this taskId, but a synth always
1292
+ // writes its terminal ledger atomically, so hasTerminalLedgerAfterDispatch downstream already
1293
+ // no-ops that case — this guard is specifically for an as-yet-unledgered worker emit in flight.
1294
+ function realTerminalEmitPendingForTask(meshId: string, taskId: string): boolean {
1295
+ let pending: readonly PendingMeshCoordinatorEvent[];
1296
+ try {
1297
+ pending = getPendingMeshCoordinatorEvents(meshId);
1298
+ } catch {
1299
+ return false;
1300
+ }
1301
+ return pending.some(e =>
1302
+ readNonEmptyString(e.metadataEvent?.taskId) === taskId
1303
+ && (e.event === 'agent:generating_completed' || e.event === 'agent:stopped'));
1304
+ }
1305
+
1306
+ // R4e fix (2): one fresh read_chat status read for the worker session, via the same local/remote
1307
+ // transport PHASE 4 uses. Returns the lowercased status, or null when the read is inconclusive
1308
+ // (transport error, success:false, no payload) — callers treat null as "no new evidence, proceed".
1309
+ async function reprobeWorkerStatus(
1310
+ components: DaemonComponents,
1311
+ args: { isLocalNode: boolean; nodeDaemonId: string; readArgs: Record<string, unknown> },
1312
+ ): Promise<string | null> {
1313
+ try {
1314
+ if (args.isLocalNode) {
1315
+ const r = await components.commandHandler.handle('read_chat', args.readArgs);
1316
+ if (r && (r as { success?: boolean }).success === false) return null;
1317
+ return readChatPayloadStatus(unwrapReadChatPayload(r));
1318
+ }
1319
+ if (components.dispatchMeshCommand) {
1320
+ const r = await components.dispatchMeshCommand(args.nodeDaemonId, 'read_chat', args.readArgs);
1321
+ const p = unwrapReadChatPayload(r);
1322
+ if (p && (p as { success?: boolean }).success === false) return null;
1323
+ return readChatPayloadStatus(p);
1324
+ }
1325
+ } catch {
1326
+ return null;
1327
+ }
1328
+ return null;
1329
+ }
1330
+
1234
1331
  // PHASE 4 helper. For every active (non-terminal) direct dispatch this daemon
1235
1332
  // hosts, confirm the worker session is idle via a read_chat and — if a final
1236
1333
  // assistant summary is present but no terminal ledger exists for that dispatch —
@@ -1251,17 +1348,17 @@ async function reconcileUnterminatedDirectDispatches(
1251
1348
  const dispatches = getActiveDirectDispatches(mesh.id);
1252
1349
  if (dispatches.length === 0) return; // cheap exit — nothing dispatched, nothing to reconcile
1253
1350
 
1254
- // Prune the in-flight idle debounce map to the tasks still active in THIS mesh, so a
1255
- // completed/pruned task's counter is dropped (the map never grows without bound).
1351
+ // Prune the in-flight acked-hold map to the tasks still active in THIS mesh, so a
1352
+ // completed/pruned task's state is dropped (the map never grows without bound).
1256
1353
  const activeTaskKeys = new Set(
1257
1354
  dispatches
1258
1355
  .map(d => readNonEmptyString(d.taskId))
1259
1356
  .filter(Boolean)
1260
1357
  .map(taskId => inFlightSynthKey(mesh.id, taskId)),
1261
1358
  );
1262
- for (const key of inFlightIdleObservationCounts.keys()) {
1359
+ for (const key of inFlightAckedHoldState.keys()) {
1263
1360
  if (key.startsWith(`${mesh.id}::`) && !activeTaskKeys.has(key)) {
1264
- inFlightIdleObservationCounts.delete(key);
1361
+ inFlightAckedHoldState.delete(key);
1265
1362
  }
1266
1363
  }
1267
1364
 
@@ -1292,51 +1389,111 @@ async function reconcileUnterminatedDirectDispatches(
1292
1389
  ...(providerType ? { agentType: providerType, providerType } : {}),
1293
1390
  };
1294
1391
 
1392
+ const synthKey = inFlightSynthKey(mesh.id, taskId);
1393
+ const isAcked = dispatch.status === 'acked';
1394
+
1395
+ // R4f: read the worker session. A FAILED read (transport error / success:false / no payload)
1396
+ // is no longer silently swallowed for an acked task — it is the liveness side of the
1397
+ // death backstop (a). We classify the read result and route an acked failure into the
1398
+ // failure counter; a never-acked (or non-acked) failure keeps the old best-effort `continue`.
1295
1399
  let payload: Record<string, unknown> | null = null;
1400
+ let readFailed = false;
1296
1401
  try {
1297
1402
  if (isLocalNode) {
1298
1403
  const result = await components.commandHandler.handle('read_chat', readArgs);
1299
- if (result && (result as { success?: boolean }).success === false) continue;
1300
- payload = unwrapReadChatPayload(result);
1404
+ if (result && (result as { success?: boolean }).success === false) {
1405
+ readFailed = true;
1406
+ } else {
1407
+ payload = unwrapReadChatPayload(result);
1408
+ }
1301
1409
  } else if (dispatchMeshCommand) {
1302
1410
  const result = await dispatchMeshCommand(nodeDaemonId, 'read_chat', readArgs);
1303
1411
  payload = unwrapReadChatPayload(result);
1304
- if (payload && (payload as { success?: boolean }).success === false) continue;
1412
+ if (payload && (payload as { success?: boolean }).success === false) { payload = null; readFailed = true; }
1305
1413
  } else {
1306
- continue; // remote node but no P2P transport — can't read; retry next tick
1414
+ continue; // remote node but no P2P transport — can't read; retry next tick (not a death signal)
1307
1415
  }
1308
1416
  } catch {
1309
- continue; // best-effort; session may be gone or node offline — retry next tick
1417
+ readFailed = true; // session may be gone or node offline
1310
1418
  }
1311
- if (!payload) continue;
1419
+ if (!payload && !readFailed) continue; // null payload that wasn't a hard failure — retry next tick
1420
+
1421
+ if (readFailed || !payload) {
1422
+ // R4f backstop (a) — liveness failure. For a never-acked dispatch there is no in-flight
1423
+ // turn to protect, so a read failure is a transient probe blip → retry next tick (old
1424
+ // behavior). For an ACKED dispatch that we had previously confirmed live, a streak of
1425
+ // consecutive read failures means the worker session genuinely went away mid-turn and
1426
+ // will never emit its real completion — count it. The actual terminal cleanup of a
1427
+ // gone session is owned by PHASE 2.5 (stranded reclaim) / PHASE 5 (orphan prune); here
1428
+ // we only record the death observation and STOP holding so those nets can take over,
1429
+ // rather than pinning the row on an indefinite hold for a session that is already gone.
1430
+ if (isAcked) {
1431
+ const prior = inFlightAckedHoldState.get(synthKey);
1432
+ const failures = (prior?.consecutiveReadFailures ?? 0) + 1;
1433
+ const liveConfirmedSinceAck = prior?.liveConfirmedSinceAck ?? false;
1434
+ inFlightAckedHoldState.set(synthKey, { liveConfirmedSinceAck, consecutiveReadFailures: failures });
1435
+ if (liveConfirmedSinceAck && failures >= ACKED_DEATH_CONSECUTIVE_READ_FAILURES) {
1436
+ LOG.warn('MeshReconcile', `Acked-hold death signal: task ${taskId} on node ${nodeId} (mesh ${mesh.id}) read_chat failed ${failures}x consecutively after a live-confirmed ack — worker session presumed gone mid-turn; releasing the indefinite synth hold to the stranded-reclaim / orphan-prune nets`);
1437
+ }
1438
+ }
1439
+ continue; // no readable transcript this tick → cannot synth here; retry / let backstops act
1440
+ }
1441
+
1442
+ // Read succeeded (a conclusive idle/generating status) → the session is reachable: reset the
1443
+ // failure streak and mark it live-confirmed-since-ack, so a LATER read failure is recognized
1444
+ // as a genuine liveness loss (backstop a) rather than a node that was never reachable.
1445
+ inFlightAckedHoldState.set(synthKey, { liveConfirmedSinceAck: true, consecutiveReadFailures: 0 });
1312
1446
 
1313
1447
  // Only act on a session that has actually settled to idle. A generating /
1314
1448
  // waiting_approval session is mid-turn — synthesizing a completion now would
1315
1449
  // be wrong. (idle is the only status the MCP poll path reconciles too.)
1316
- const synthKey = inFlightSynthKey(mesh.id, taskId);
1450
+ const nowMs = Date.now();
1317
1451
  if (readChatPayloadStatus(payload) !== 'idle') {
1318
- // Not idle → the worker is mid-turn. Reset any partial idle streak so a single
1319
- // idle blip during a long generation never accumulates toward the synth threshold.
1320
- inFlightIdleObservationCounts.delete(synthKey);
1452
+ // Not idle → the worker is genuinely mid-turn (a clear live signal). Keep the
1453
+ // live-confirmed flag set (above) but otherwise just wait for the real emit.
1321
1454
  continue;
1322
1455
  }
1323
1456
 
1324
- // RECONCILE-SYNTH-PREEMPTS-COMPLETION: a dispatch whose worker was OBSERVED to start
1325
- // generating (the agent:generating_started ack flipped the row to 'acked') and has no
1326
- // terminal yet is potentially still in-flight its `idle` read here may be a transient
1327
- // mid-turn flicker, not a settled completion. Require CONSECUTIVE idle observations
1328
- // before synthesizing for such a task: a flicker clears next tick (counter reset above),
1329
- // while a genuinely-settled (completed-but-lost) or dead session reads idle every tick
1330
- // and crosses the threshold within ~one extra interval. A never-acked dispatch (worker
1331
- // never started) is exempt there is no in-flight generation to pre-empt, and its
1332
- // lost-dispatch case is still covered by the downstream grace + stale-summary guards.
1333
- if (dispatch.status === 'acked') {
1334
- const idleStreak = (inFlightIdleObservationCounts.get(synthKey) ?? 0) + 1;
1335
- inFlightIdleObservationCounts.set(synthKey, idleStreak);
1336
- if (idleStreak < REQUIRED_CONSECUTIVE_IDLE_TICKS_FOR_INFLIGHT_SYNTH) {
1337
- LOG.info('MeshReconcile', `In-flight synth hold: task ${taskId} on node ${nodeId} (mesh ${mesh.id}) read idle ${idleStreak}/${REQUIRED_CONSECUTIVE_IDLE_TICKS_FOR_INFLIGHT_SYNTH} consecutive tick(s) after generating_started — deferring completion synth until the idle settle is confirmed (guards against a mid-turn idle flicker pre-empting the real completion)`);
1457
+ // R4f GENERATING-BOUNDARY (acked-hold): a dispatch whose worker was OBSERVED to start
1458
+ // generating (the agent:generating_started ack flipped the row to 'acked') is ALIVE and
1459
+ // mid-turn it WILL eventually emit a real terminal. An `idle` read here is therefore
1460
+ // presumed a TRANSIENT mid-turn window (a PTY inter-tool-call settle, or final text already
1461
+ // rendered while the lifecycle close lags), NOT a settled completion. We HOLD the synth
1462
+ // INDEFINITELY rather than racing the worker's (variable, unbounded) emit latency with a
1463
+ // finite timer the failure mode of R4..R4e. This is safe: when the worker's real emit
1464
+ // lands it writes a terminal ledger, and reconcileDirectDispatchCompletionFromTranscript's
1465
+ // hasTerminalLedgerAfterDispatch makes any later synth an idempotent no-op, so the real emit
1466
+ // always wins no matter how late. The hold is released ONLY by the death backstops:
1467
+ // (a) consecutive read failures after a live-confirmed ack (handled above), or
1468
+ // (b) the absolute ACKED_DEATH_DEADLINE_MS since the ack — a notification-loss net set FAR
1469
+ // above any observed emit latency, so it catches a genuinely-wedged worker / lost emit
1470
+ // without racing a normal slow turn.
1471
+ // A never-acked dispatch (worker never started) is exempt — no in-flight generation to
1472
+ // pre-empt; it keeps the first-idle-tick synth, with the downstream grace + stale-summary
1473
+ // guards as its backstops.
1474
+ if (isAcked) {
1475
+ const ackedAtMs = Date.parse(readNonEmptyString(dispatch.updatedAt));
1476
+ const sinceAckMs = Number.isFinite(ackedAtMs) ? nowMs - ackedAtMs : Number.POSITIVE_INFINITY;
1477
+ const deathDeadlineMs = resolveAckedDeathDeadlineMs();
1478
+ if (sinceAckMs < deathDeadlineMs) {
1479
+ LOG.info('MeshReconcile', `Acked-hold: task ${taskId} on node ${nodeId} (mesh ${mesh.id}) read idle ${Number.isFinite(sinceAckMs) ? Math.round(sinceAckMs / 1000) + 's' : '∞'} since the generating_started ack — HOLDING synth indefinitely (worker is alive and will emit; a later real emit is idempotent). Death backstop fires at ${Math.round(deathDeadlineMs / 1000)}s or on consecutive read failures.`);
1338
1480
  continue;
1339
1481
  }
1482
+ LOG.warn('MeshReconcile', `Acked-hold death deadline reached: task ${taskId} on node ${nodeId} (mesh ${mesh.id}) still idle ${Math.round(sinceAckMs / 1000)}s after the ack (deadline ${Math.round(deathDeadlineMs / 1000)}s) — synthesizing the missing completion as a notification-loss net (a real emit, if it ever lands, no-ops idempotently).`);
1483
+ }
1484
+
1485
+ // R4f (auxiliary, was R4e fix 3) — worker-emit priority. Secondary check: if the worker's
1486
+ // REAL terminal emit for this task has already arrived in the pending-events queue (queued
1487
+ // for delivery to the coordinator) but not yet written a terminal ledger, YIELD — let the
1488
+ // genuine emit surface rather than racing it with a synth that would win the taskId-anchored
1489
+ // fingerprint dedup and mask it. Under the R4f acked-hold this is now an auxiliary belt-and-
1490
+ // suspenders check (the indefinite hold already defers an acked synth); it still guards the
1491
+ // never-acked path and the post-death-deadline acked synth from racing an emit caught in
1492
+ // flight at synth-commit time.
1493
+ if (realTerminalEmitPendingForTask(mesh.id, taskId)) {
1494
+ inFlightAckedHoldState.delete(synthKey);
1495
+ LOG.info('MeshReconcile', `Worker-emit priority: task ${taskId} on node ${nodeId} (mesh ${mesh.id}) has a real terminal completion already queued — yielding synth to the worker's own emit`);
1496
+ continue;
1340
1497
  }
1341
1498
 
1342
1499
  const messages = Array.isArray(payload.messages) ? payload.messages as ChatMessage[] : [];
@@ -1369,6 +1526,21 @@ async function reconcileUnterminatedDirectDispatches(
1369
1526
  continue;
1370
1527
  }
1371
1528
 
1529
+ // R4f (auxiliary, was R4e fix 2) — live re-probe immediately before committing the synth. A
1530
+ // fresh read right now catches a worker that resumed generating since this tick's first read
1531
+ // so it is never falsely completed off a stale snapshot. Best-effort: an inconclusive
1532
+ // re-probe (transport error/null) falls through to the synth — we already hold a valid idle
1533
+ // read from the top of THIS tick, so a re-probe failure must not re-introduce a
1534
+ // notification-miss. Under the R4f acked-hold this matters mainly for the never-acked path
1535
+ // and the post-death-deadline acked synth (the indefinite hold already deferred a live acked
1536
+ // turn); it stays as a final live-state guard at synth-commit time.
1537
+ const reprobeStatus = await reprobeWorkerStatus(components, { isLocalNode, nodeDaemonId, readArgs });
1538
+ if (reprobeStatus && reprobeStatus !== 'idle') {
1539
+ inFlightAckedHoldState.delete(synthKey);
1540
+ LOG.info('MeshReconcile', `Live re-probe defer: task ${taskId} on node ${nodeId} (mesh ${mesh.id}) read '${reprobeStatus}' at synth-commit time — worker resumed generating; deferring synth to a later tick`);
1541
+ continue;
1542
+ }
1543
+
1372
1544
  const providerSessionId = readNonEmptyString(payload.providerSessionId);
1373
1545
  const coordinatorDaemonId = selfIds.find(id => !!id);
1374
1546
  try {
@@ -1160,6 +1160,9 @@ export async function alignRefinerySubmodulesAfterMerge(
1160
1160
  includeSubmodules: true,
1161
1161
  submoduleIgnorePaths: options.submoduleIgnorePaths,
1162
1162
  timeoutMs: 15_000,
1163
+ // Decision path — the out-of-sync submodule set drives a mutating `submodule
1164
+ // update`. Must not act on a TTL-cached status; bypass the C1 cache.
1165
+ forceFresh: true,
1163
1166
  });
1164
1167
  const outOfSyncPaths = (preStatus.submodules || [])
1165
1168
  .filter(submodule => submodule.dirty || submodule.outOfSync || !!submodule.error)
@@ -1193,6 +1196,9 @@ export async function alignRefinerySubmodulesAfterMerge(
1193
1196
  includeSubmodules: true,
1194
1197
  submoduleIgnorePaths: options.submoduleIgnorePaths,
1195
1198
  timeoutMs: 15_000,
1199
+ // Re-read AFTER `submodule update` mutated the tree — MUST be fresh, never the
1200
+ // cached preStatus from moments ago (which would falsely report still-dirty).
1201
+ forceFresh: true,
1196
1202
  });
1197
1203
  const remaining = (postStatus.submodules || [])
1198
1204
  .filter(submodule => updatePaths.includes(submodule.path) && (submodule.dirty || submodule.outOfSync || !!submodule.error));
@@ -3,7 +3,7 @@ import { dirname, join } from 'path';
3
3
  import { LOG } from '../logging/logger.js';
4
4
  import { loadBetterSqlite3 } from '../system/load-better-sqlite3.js';
5
5
  import { getLedgerDir } from './mesh-ledger.js';
6
- import { nodeSatisfiesRequiredTags } from './mesh-work-queue.js';
6
+ import { nodeSatisfiesRequiredTags, isTaskReadonly } from './mesh-work-queue.js';
7
7
  import { meshNodeIdMatches, daemonIdsEquivalent, expandDaemonIdForms } from '@adhdev/mesh-shared';
8
8
  import type { MeshTaskStatus, MeshWorkQueueEntry } from './mesh-work-queue.js';
9
9
  import type BetterSqlite3 from 'better-sqlite3';
@@ -752,11 +752,12 @@ export class MeshRuntimeStore {
752
752
  return deps.every(depId => depStatus.get(depId) === 'completed');
753
753
  };
754
754
 
755
- // Per-candidate node-conflict gate: write tasks (anything other than
756
- // live_debug_readonly) require an idle node; read-only tasks bypass the
757
- // node-busy check so N read-only diagnoses can run on one node at once.
755
+ // Per-candidate node-conflict gate: write tasks require an idle node; read-only
756
+ // tasks bypass the node-busy check so N read-only diagnoses can run on one node
757
+ // at once. Read-only classification is decided solely by isTaskReadonly (the
758
+ // single predicate shared with the cap counters / auto-launch / guardrail).
758
759
  const nodeConflictAllows = (candidate: MeshWorkQueueEntry): boolean => {
759
- if (candidate.taskMode === 'live_debug_readonly') return true;
760
+ if (isTaskReadonly(candidate)) return true;
760
761
  return !nodeBusy;
761
762
  };
762
763
 
@@ -22,6 +22,7 @@ import {
22
22
  } from '../repo-mesh-types.js';
23
23
  import { normalizeMeshNodeId } from '@adhdev/mesh-shared';
24
24
  import type { MeshWorkQueueEntry } from './mesh-work-queue.js';
25
+ import { isTaskReadonly } from './mesh-work-queue.js';
25
26
 
26
27
  /** Per-(node, provider) cap and its current consumption. */
27
28
  export interface MeshNodeProviderSchedulingRuntime {
@@ -80,10 +81,6 @@ interface MeshLike {
80
81
  nodes?: Array<{ id?: string; nodeId?: string; node_id?: string; policy?: RepoMeshNodePolicy | null; isLocalWorktree?: boolean }> | null;
81
82
  }
82
83
 
83
- function isReadonly(task: MeshWorkQueueEntry): boolean {
84
- return task.taskMode === 'live_debug_readonly';
85
- }
86
-
87
84
  function isAssigned(task: MeshWorkQueueEntry): boolean {
88
85
  return task.status === 'assigned';
89
86
  }
@@ -112,8 +109,8 @@ export function buildMeshSchedulingRuntime(
112
109
  const maxReadonlyParallelTasks = resolveMaxReadonlyParallelTasks(maxParallelTasks);
113
110
 
114
111
  const assignedTasks = (Array.isArray(queue) ? queue : []).filter(isAssigned);
115
- const activeWriteAssigned = assignedTasks.filter(t => !isReadonly(t)).length;
116
- const activeReadonlyAssigned = assignedTasks.filter(isReadonly).length;
112
+ const activeWriteAssigned = assignedTasks.filter(t => !isTaskReadonly(t)).length;
113
+ const activeReadonlyAssigned = assignedTasks.filter(isTaskReadonly).length;
117
114
  const globalWriteCapReached = activeWriteAssigned >= maxParallelTasks;
118
115
  const globalReadonlyCapReached = activeReadonlyAssigned >= maxReadonlyParallelTasks;
119
116
 
@@ -126,7 +123,7 @@ export function buildMeshSchedulingRuntime(
126
123
  const nodeId = typeof task.assignedNodeId === 'string' ? task.assignedNodeId.trim() : '';
127
124
  if (!nodeId) continue;
128
125
  assignedByNode.set(nodeId, (assignedByNode.get(nodeId) ?? 0) + 1);
129
- if (!isReadonly(task)) {
126
+ if (!isTaskReadonly(task)) {
130
127
  writeAssignedByNode.set(nodeId, (writeAssignedByNode.get(nodeId) ?? 0) + 1);
131
128
  }
132
129
  const provider = typeof task.assignedProviderType === 'string' ? task.assignedProviderType : '';
@@ -18,6 +18,29 @@ export const ACTIVE_MESH_QUEUE_STATUSES: MeshActiveTaskStatus[] = ['pending', 'a
18
18
  export const HISTORICAL_MESH_QUEUE_STATUSES: MeshHistoricalTaskStatus[] = ['completed', 'failed', 'cancelled'];
19
19
  export const MESH_TASK_MODES: MeshTaskMode[] = ['code_change', 'validation', 'live_debug_readonly', 'launch_app', 'convergence'];
20
20
 
21
+ /**
22
+ * QUEUE-NODE-SERIALIZATION: single source of truth for "is this task read-only?".
23
+ *
24
+ * Read-only classification used to be inlined as `task.taskMode === 'live_debug_readonly'`
25
+ * at every enforcement site (node-conflict claim gate, auto-launch isolation, the
26
+ * write/readonly cap counters, the write guardrail). That spread-out comparison is the
27
+ * exact recurring-defect class — one site drifting from the others silently makes the same
28
+ * task read-only at some gates and write at others, i.e. partial serialization. All sites
29
+ * MUST call this predicate so the classification is decided in exactly one place.
30
+ *
31
+ * Two orthogonal inputs feed the same boolean axis (kept backward-compatible):
32
+ * • `readonly === true` — the explicit boolean axis (new API surface).
33
+ * • `taskMode === 'live_debug_readonly'` — the original enum value, preserved as an
34
+ * OR-fallback so existing live_debug_readonly tasks keep behaving identically.
35
+ *
36
+ * Accepts any task-like shape (full {@link MeshWorkQueueEntry} or a bare
37
+ * `{ readonly?, taskMode? }`) so the daemon-core and mcp-server boundaries can share it.
38
+ */
39
+ export function isTaskReadonly(task: { readonly?: boolean; taskMode?: MeshTaskMode | string } | null | undefined): boolean {
40
+ if (!task) return false;
41
+ return task.readonly === true || task.taskMode === 'live_debug_readonly';
42
+ }
43
+
21
44
  export interface MeshTaskModeValidationResult {
22
45
  valid: boolean;
23
46
  taskMode?: MeshTaskMode;
@@ -410,13 +433,15 @@ export function normalizeMeshTaskMode(value: unknown): MeshTaskMode | undefined
410
433
  return (MESH_TASK_MODES as string[]).includes(normalized) ? normalized : undefined;
411
434
  }
412
435
 
413
- export function validateMeshTaskModeRequest(mode: unknown, message: string): MeshTaskModeValidationResult {
436
+ export function validateMeshTaskModeRequest(mode: unknown, message: string, readonly?: boolean): MeshTaskModeValidationResult {
414
437
  const taskMode = normalizeMeshTaskMode(mode);
415
- if (!taskMode) {
416
- return { valid: true, violations: [] };
417
- }
418
- if (taskMode !== 'live_debug_readonly') {
419
- return { valid: true, taskMode, violations: [] };
438
+ // QUEUE-NODE-SERIALIZATION: the write guardrail (reject deploy/push/edit commands on a
439
+ // read-only task) is driven by the unified read-only axis, not by the enum alone — so a
440
+ // task flagged read-only via the explicit `readonly:true` boolean is guarded identically
441
+ // to a legacy live_debug_readonly task. isTaskReadonly is the single classifier.
442
+ const isReadonly = isTaskReadonly({ readonly, taskMode });
443
+ if (!isReadonly) {
444
+ return taskMode ? { valid: true, taskMode, violations: [] } : { valid: true, violations: [] };
420
445
  }
421
446
  const text = message || '';
422
447
  // Only flag keywords that look like real commands (code/command context) and
@@ -447,6 +472,14 @@ export interface MeshWorkQueueEntry {
447
472
  message: string;
448
473
  status: MeshTaskStatus;
449
474
  taskMode?: MeshTaskMode;
475
+ /**
476
+ * QUEUE-NODE-SERIALIZATION: explicit read-only axis, orthogonal to taskMode. When
477
+ * true the task is treated as read-only by every scheduling gate (no node-busy
478
+ * isolation, counted under the read-only cap, write commands rejected) regardless of
479
+ * its taskMode. Decided exclusively through {@link isTaskReadonly}; `taskMode ===
480
+ * 'live_debug_readonly'` remains an OR-fallback so legacy rows behave unchanged.
481
+ */
482
+ readonly?: boolean;
450
483
  /** If specified, only this node can claim the task (used by legacy mesh_send_task) */
451
484
  targetNodeId?: string;
452
485
  /** If specified, only this runtime session can claim the task */
@@ -722,6 +755,8 @@ export function enqueueTask(
722
755
  targetNodeId?: string;
723
756
  targetSessionId?: string;
724
757
  taskMode?: MeshTaskMode | string;
758
+ /** QUEUE-NODE-SERIALIZATION: explicit read-only axis (orthogonal to taskMode). */
759
+ readonly?: boolean;
725
760
  requiredTags?: string[];
726
761
  /** M1: tasks that must complete before this one is claimable. */
727
762
  dependsOn?: string[];
@@ -734,7 +769,8 @@ export function enqueueTask(
734
769
  } & MeshQueueMutationOptions,
735
770
  ): MeshWorkQueueEntry {
736
771
  requireMeshHostQueueOwner(opts);
737
- const modeValidation = validateMeshTaskModeRequest(opts?.taskMode, message);
772
+ const readonly = opts?.readonly === true;
773
+ const modeValidation = validateMeshTaskModeRequest(opts?.taskMode, message, readonly);
738
774
  if (!modeValidation.valid) {
739
775
  throw new Error(`live_debug_readonly_guardrail_violation: forbidden operations (${modeValidation.violations.join(', ')})`);
740
776
  }
@@ -763,6 +799,7 @@ export function enqueueTask(
763
799
  message,
764
800
  status: 'pending',
765
801
  taskMode: modeValidation.taskMode,
802
+ ...(readonly ? { readonly: true } : {}),
766
803
  targetNodeId: opts?.targetNodeId,
767
804
  targetSessionId: opts?.targetSessionId,
768
805
  requiredTags: resolvedRequiredTags,
@@ -806,6 +843,8 @@ export function recordDirectDispatchTask(
806
843
  assignedNodeId?: string;
807
844
  assignedSessionId?: string;
808
845
  taskMode?: MeshTaskMode | string;
846
+ /** QUEUE-NODE-SERIALIZATION: explicit read-only axis (orthogonal to taskMode). */
847
+ readonly?: boolean;
809
848
  dispatchedAt?: string;
810
849
  },
811
850
  ): MeshWorkQueueEntry | null {
@@ -813,7 +852,8 @@ export function recordDirectDispatchTask(
813
852
  if (!missionId) return null;
814
853
  const taskId = typeof opts.id === 'string' ? opts.id.trim() : '';
815
854
  if (!taskId) return null;
816
- const modeValidation = validateMeshTaskModeRequest(opts.taskMode, message);
855
+ const readonly = opts.readonly === true;
856
+ const modeValidation = validateMeshTaskModeRequest(opts.taskMode, message, readonly);
817
857
  if (!modeValidation.valid) {
818
858
  throw new Error(`live_debug_readonly_guardrail_violation: forbidden operations (${modeValidation.violations.join(', ')})`);
819
859
  }
@@ -829,6 +869,7 @@ export function recordDirectDispatchTask(
829
869
  message,
830
870
  status: 'assigned',
831
871
  ...(modeValidation.taskMode ? { taskMode: modeValidation.taskMode } : {}),
872
+ ...(readonly ? { readonly: true } : {}),
832
873
  missionId,
833
874
  ...(opts.assignedNodeId ? { targetNodeId: opts.assignedNodeId, assignedNodeId: opts.assignedNodeId } : {}),
834
875
  ...(opts.assignedSessionId ? { targetSessionId: opts.assignedSessionId, assignedSessionId: opts.assignedSessionId } : {}),