@memberjunction/task-graph 6.1.0-edge.2 → 6.1.0-edge.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/LICENSE +180 -4
  2. package/README.md +30 -1
  3. package/dist/TaskClaimStore.d.ts +376 -4
  4. package/dist/TaskClaimStore.d.ts.map +1 -1
  5. package/dist/TaskClaimStore.js +600 -20
  6. package/dist/TaskClaimStore.js.map +1 -1
  7. package/dist/TaskGraphDispatcher.d.ts +326 -25
  8. package/dist/TaskGraphDispatcher.d.ts.map +1 -1
  9. package/dist/TaskGraphDispatcher.js +1604 -286
  10. package/dist/TaskGraphDispatcher.js.map +1 -1
  11. package/dist/TaskGraphService.d.ts +255 -5
  12. package/dist/TaskGraphService.d.ts.map +1 -1
  13. package/dist/TaskGraphService.js +583 -26
  14. package/dist/TaskGraphService.js.map +1 -1
  15. package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -1
  16. package/dist/TaskGraphSubmitterImpl.js +5 -0
  17. package/dist/TaskGraphSubmitterImpl.js.map +1 -1
  18. package/dist/condition-gate.d.ts +128 -0
  19. package/dist/condition-gate.d.ts.map +1 -0
  20. package/dist/condition-gate.js +257 -0
  21. package/dist/condition-gate.js.map +1 -0
  22. package/dist/debug-state.d.ts +102 -0
  23. package/dist/debug-state.d.ts.map +1 -0
  24. package/dist/debug-state.js +135 -0
  25. package/dist/debug-state.js.map +1 -0
  26. package/dist/index.d.ts +6 -0
  27. package/dist/index.d.ts.map +1 -1
  28. package/dist/index.js +6 -0
  29. package/dist/index.js.map +1 -1
  30. package/dist/operations/TaskGraphDebugOperations.d.ts +99 -0
  31. package/dist/operations/TaskGraphDebugOperations.d.ts.map +1 -0
  32. package/dist/operations/TaskGraphDebugOperations.js +310 -0
  33. package/dist/operations/TaskGraphDebugOperations.js.map +1 -0
  34. package/dist/operations/TaskGraphOperations.d.ts +20 -2
  35. package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
  36. package/dist/operations/TaskGraphOperations.js +47 -8
  37. package/dist/operations/TaskGraphOperations.js.map +1 -1
  38. package/dist/settlement-rescue.d.ts +85 -0
  39. package/dist/settlement-rescue.d.ts.map +1 -0
  40. package/dist/settlement-rescue.js +119 -0
  41. package/dist/settlement-rescue.js.map +1 -0
  42. package/dist/task-graph-kick.d.ts +3 -0
  43. package/dist/task-graph-kick.d.ts.map +1 -0
  44. package/dist/task-graph-kick.js +17 -0
  45. package/dist/task-graph-kick.js.map +1 -0
  46. package/dist/task-predicates.d.ts +77 -0
  47. package/dist/task-predicates.d.ts.map +1 -0
  48. package/dist/task-predicates.js +75 -0
  49. package/dist/task-predicates.js.map +1 -0
  50. package/dist/types.d.ts +110 -1
  51. package/dist/types.d.ts.map +1 -1
  52. package/dist/types.js.map +1 -1
  53. package/package.json +12 -11
@@ -18,15 +18,44 @@
18
18
  *
19
19
  * @module @memberjunction/task-graph
20
20
  */
21
- import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
21
+ import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, ConfirmSkipSeeds, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
22
22
  import { LogError, LogStatus, RunView } from '@memberjunction/core';
23
23
  import { ShutdownRegistry, UUIDsEqual } from '@memberjunction/global';
24
- import { TaskClaimStore } from './TaskClaimStore.js';
24
+ import { TaskClaimStore, TERMINAL_PARENT_STATUSES, TERMINAL_PARENT_STATUS_SQL } from './TaskClaimStore.js';
25
+ import { BuildConditionContext, DecideGate, IsBrokenGuard, ParseConditionOutput, } from './condition-gate.js';
26
+ import { HumanTaskSQL, IsHumanTask } from './task-predicates.js';
27
+ import { IsSettlementExpired, IsSubmittingRunReady, SelectUnsettledGraphIDs, SweepCutoff, UNSETTLED_SWEEP_WINDOW_HOURS, UNSETTLED_STARTUP_WINDOW_HOURS, } from './settlement-rescue.js';
25
28
  import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
29
+ import { DecideClaimGate, OverrideVerdictFor, ParseTaskGraphDebugState, } from './debug-state.js';
26
30
  import { RunForEachLoop, RunWhileLoop } from './TaskLoopExecutor.js';
31
+ import { RegisterTaskGraphKick } from './task-graph-kick.js';
27
32
  import { NotificationEngine } from '@memberjunction/notifications';
28
33
  /** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
29
34
  const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
35
+ /**
36
+ * Statuses a graph parent has stopped moving from.
37
+ *
38
+ * Shared by the guarded terminal write and the unsettled-graph sweep, so "terminal" means exactly
39
+ * one thing in both — the two disagreeing is how a graph becomes invisible to the machinery that is
40
+ * supposed to rescue it.
41
+ */
42
+ const TERMINAL_TASK_STATUSES = new Set(TERMINAL_PARENT_STATUSES);
43
+ /**
44
+ * How long `Stop()` waits for in-flight tasks and timer passes before giving up and saying so.
45
+ *
46
+ * Generous, because the alternative to waiting is a dispatcher that writes after its host believes
47
+ * it has shut down — settling graphs onto a connection somebody else now owns.
48
+ */
49
+ const STOP_DRAIN_TIMEOUT_MS = 30_000;
50
+ /**
51
+ * How many consecutive failing passes a graph gets before this instance stops re-queueing it.
52
+ *
53
+ * Not a giving-up threshold so much as a stop-shouting one: past this the graph has failed to settle
54
+ * on every attempt for minutes, so something is wrong that another identical attempt will not fix,
55
+ * and continuing costs a full graph load per poll forever. It is reported and left to the startup
56
+ * sweep, which is the wider net.
57
+ */
58
+ const MAX_SETTLEMENT_RETRY_PASSES = 20;
30
59
  /**
31
60
  * Written to a human task's `ClaimedBy` once its assignee has been told it is ready.
32
61
  *
@@ -47,7 +76,7 @@ function asRunQueryProvider(provider) {
47
76
  const candidate = provider;
48
77
  return typeof candidate.RunQuery === 'function' ? candidate : undefined;
49
78
  }
50
- import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata } from './TaskGraphService.js';
79
+ import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata, TASK_TYPE_NAME } from './TaskGraphService.js';
51
80
  import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
52
81
  /**
53
82
  * Renders a loop's bindings as template values.
@@ -132,16 +161,9 @@ function deepMergePayload(base, incoming) {
132
161
  }
133
162
  return out;
134
163
  }
135
- /**
136
- * Statuses at which an origin's outgoing conditions may be decided.
137
- *
138
- * `Skipped` is included: a branch that was not taken IS settled, and a condition on an edge leaving
139
- * it should resolve rather than hang the graph forever.
140
- */
141
- const TERMINAL_FOR_CONDITIONS = new Set([
142
- 'Complete', 'Failed', 'Cancelled', 'Skipped',
143
- ]);
144
164
  export class TaskGraphDispatcher {
165
+ /** Minimum interval between `NodeProgress` frames for one task. */
166
+ static { this.NODE_PROGRESS_MIN_INTERVAL_MS = 1_000; }
145
167
  constructor(providerFactory, agentRunner, contextUser, config,
146
168
  /**
147
169
  * Optional. Absent means a host that cannot post messages or start agent turns — a worker,
@@ -171,24 +193,90 @@ export class TaskGraphDispatcher {
171
193
  this.observer = observer;
172
194
  this.actionRunner = actionRunner;
173
195
  this.promptRunner = promptRunner;
196
+ /**
197
+ * Edges already reported as unevaluable, so the report is once per transition and not once per
198
+ * poll. Per-instance and in-memory by design: a restart re-reports, which is the right amount of
199
+ * noise for a condition that is still broken after a restart.
200
+ */
201
+ this.reportedUnevaluableConditions = new Set();
202
+ /** Resolved once it EXISTS; null while it does not, so a fresh install is not cached blind. */
203
+ this.cachedWorkflowTaskTypeID = null;
204
+ /**
205
+ * Graphs this instance is still trying to settle, with how many passes it has spent trying.
206
+ *
207
+ * **The sweep's window is on `__mj_UpdatedAt`, and a failing pass writes nothing** — the terminal
208
+ * write returns rowcount 0 because the row is already terminal, the layout pass touches only
209
+ * children, a refused CAS writes nothing at all. So a graph that fails to settle stops advancing
210
+ * its own timestamp and, after 24h of futile retries, ages out of the steady-state window while
211
+ * the process is up. The doc comment claimed the bound was "on abandonment, not age"; for this
212
+ * case it was on age, and R2-2's deferral made the case ordinary rather than exotic.
213
+ *
214
+ * In memory rather than a touch column because the alternative is a write on every failed
215
+ * attempt — more load exactly when something is already wrong — and because a restart is covered
216
+ * by the wide startup sweep, which is the durable backstop this leans on.
217
+ */
218
+ this.retryingSettlement = new Map();
219
+ /**
220
+ * Graphs whose settled-branch ANNOUNCEMENTS have already been made by this process.
221
+ *
222
+ * Re-entry is the point of the rescue, but only the parts that failed should repeat. Layout and
223
+ * the `GraphSettled` frame are idempotent facts about a finished graph, so a graph stuck in
224
+ * retry was re-persisting geometry and re-emitting the same frame every poll — for the whole
225
+ * 24h window, for as long as it kept failing.
226
+ */
227
+ this.announcedSettlements = new Set();
228
+ /** Graphs already reported as settled-but-undeliverable by this instance. */
229
+ this.reportedUndeliverable = new Set();
230
+ /** Live claim heartbeats by task ID, so the drain can silence the ones it gives up waiting for. */
231
+ this.heartbeats = new Map();
232
+ /** Latched once the drain has given up waiting, so a late arrival does not re-register. */
233
+ this.heartbeatsPurged = false;
174
234
  this.running = false;
175
235
  this.pollTimer = null;
176
236
  this.reconcileTimer = null;
237
+ this.unregisterKick = null;
177
238
  /** Tasks this instance is currently executing — bounds concurrency and drives heartbeats. */
178
239
  this.inFlight = new Set();
179
240
  /** Guards against a slow poll overlapping the next tick. */
180
241
  this.polling = false;
181
242
  /**
182
- * The poll pass currently running, so `Stop` can wait for it.
243
+ * Timer-driven passes currently running — poll and reconcile alike.
183
244
  *
184
- * `clearInterval` cannot cancel a tick that has already fired, and a pass is a long sequence of
185
- * awaits (provider, rollup, claim query) — so without this, `Stop` returns while a pass is still
186
- * mid-flight and about to claim. Its tasks then land in `inFlight` AFTER the drain loop already
187
- * saw an empty set, which is precisely the state the drain exists to prevent.
245
+ * A counter rather than a boolean because the two timers overlap by design, and `Stop()` has to
246
+ * wait for BOTH. Neither pass is held by anything else: they are launched `void`-ed from
247
+ * `setInterval`, so without this they are unobservable from the outside and a stopped dispatcher
248
+ * keeps writing.
188
249
  */
189
- this.pollPass = null;
250
+ this.activePasses = 0;
190
251
  /** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
191
252
  this.ownerByParentID = new Map();
253
+ /** Monotonic pass counter for `PassCompleted` frames, so a viewer can order and gap-detect ticks. */
254
+ this.passCounter = 0;
255
+ /**
256
+ * Debug state per graph, cached for ONE pass. `pollOnce` clears it at entry, so within a pass
257
+ * the claim filter and the propagation loop read the same state (loading it twice could see a
258
+ * pause land between them and gate half a pass), and across passes a control verb written by any
259
+ * instance is picked up within one poll interval.
260
+ */
261
+ this.debugStateByGraph = new Map();
262
+ /**
263
+ * The pause state last announced per graph, so `GraphPaused`/`GraphResumed` are emitted on the
264
+ * TRANSITION rather than every pass — the verbs write durable state, not events, and it is this
265
+ * instance's job to notice the change and say so exactly once.
266
+ */
267
+ this.announcedPaused = new Map();
268
+ /**
269
+ * The verdict last emitted per gating edge. `GateDecision` frames announce CHANGES: edges are
270
+ * re-resolved every pass, and an unconditional emission would repeat every few seconds for as
271
+ * long as the graph lives — the frame-topic version of the log flood
272
+ * `logUnevaluableConditionOnce` exists to prevent.
273
+ *
274
+ * Nested by graph so a settled run's entries go in one delete. Flat per-edge maps in a process
275
+ * that runs for weeks are a slow leak with no upper bound but the table's size.
276
+ */
277
+ this.emittedGateVerdicts = new Map();
278
+ /** Last `NodeProgress` emission per task, for rate limiting chatty runners. Nested by graph. */
279
+ this.nodeProgressLastEmit = new Map();
192
280
  /** Name shown in the shutdown drain log. */
193
281
  this.ShutdownName = 'TaskGraphDispatcher';
194
282
  this.config = { ...DEFAULT_DISPATCHER_CONFIG, ...config };
@@ -212,6 +300,37 @@ export class TaskGraphDispatcher {
212
300
  LogError(`[TaskGraphDispatcher] Observer threw on ${frame.Kind} (ignored): ${e instanceof Error ? e.message : String(e)}`);
213
301
  }
214
302
  }
303
+ /**
304
+ * A progress sink for one task's runner, rate-limited into `NodeProgress` frames.
305
+ *
306
+ * Rate-limited HERE rather than asking every runner to be polite, for the same reason `emit`
307
+ * swallows observer throws in one place: a chatty runner (an agent streaming token-level
308
+ * updates) must not be able to flood the topic, and the limit belongs to the announcement, not
309
+ * the work. A 100% report always passes — the terminal update is the one a viewer must not lose.
310
+ */
311
+ nodeProgressEmitter(graphID, ownerUserID, taskID, taskName) {
312
+ return (message, percent) => {
313
+ const now = Date.now();
314
+ let perTask = this.nodeProgressLastEmit.get(graphID);
315
+ if (!perTask) {
316
+ perTask = new Map();
317
+ this.nodeProgressLastEmit.set(graphID, perTask);
318
+ }
319
+ const last = perTask.get(taskID) ?? 0;
320
+ if (percent !== 100 && now - last < TaskGraphDispatcher.NODE_PROGRESS_MIN_INTERVAL_MS)
321
+ return;
322
+ perTask.set(taskID, now);
323
+ this.emit({
324
+ Kind: 'NodeProgress',
325
+ ParentTaskID: graphID,
326
+ OwnerUserID: ownerUserID,
327
+ TaskID: taskID,
328
+ TaskName: taskName,
329
+ ProgressMessage: message,
330
+ ProgressPercent: percent,
331
+ });
332
+ };
333
+ }
215
334
  /**
216
335
  * Who a graph belongs to, memoized for the process's lifetime.
217
336
  *
@@ -230,18 +349,106 @@ export class TaskGraphDispatcher {
230
349
  const cached = this.ownerByParentID.get(parentTaskID);
231
350
  if (cached !== undefined)
232
351
  return cached;
233
- let owner = null;
234
352
  try {
235
353
  const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
236
- if (await parent.Load(parentTaskID)) {
237
- owner = this.readParentMetadata(parent).submittedByUserID ?? null;
354
+ if (!(await parent.Load(parentTaskID))) {
355
+ // NOT CACHED (C1). A failed load is not an answer, and caching it as one is
356
+ // permanent for the life of the process: the delivery filter fails closed on a null
357
+ // owner, so every frame for this graph reaches nobody until a restart. One
358
+ // transient blip, and a viewer watches a workflow that never appears to move.
359
+ LogError(`[TaskGraphDispatcher] Could not load graph ${parentTaskID} to resolve its owner; frames for it are unaddressed this pass.`);
360
+ return null;
238
361
  }
362
+ const owner = this.readParentMetadata(parent).submittedByUserID ?? null;
363
+ // A successfully-read graph with no owner IS an answer — a scheduled or remote-triggered
364
+ // graph legitimately has none — so that one caches.
365
+ this.ownerByParentID.set(parentTaskID, owner);
366
+ return owner;
239
367
  }
240
368
  catch (e) {
241
369
  LogError(`[TaskGraphDispatcher] Could not resolve owner for graph ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
370
+ return null;
242
371
  }
243
- this.ownerByParentID.set(parentTaskID, owner);
244
- return owner;
372
+ }
373
+ /**
374
+ * A graph's debug state, cached for the current pass.
375
+ *
376
+ * Read fresh (BypassCache) because the state is written by direct `JSON_MODIFY` statements that
377
+ * fire no cache invalidation — the same reason every row read in this class bypasses the cache.
378
+ */
379
+ async readDebugState(provider, parentTaskID) {
380
+ const cached = this.debugStateByGraph.get(parentTaskID);
381
+ if (cached !== undefined)
382
+ return cached;
383
+ await this.primeDebugStates(provider, [parentTaskID]);
384
+ return this.debugStateByGraph.get(parentTaskID) ?? {};
385
+ }
386
+ /**
387
+ * Reads the debug bag for a set of graphs in ONE query, priming the per-pass cache.
388
+ *
389
+ * Batched because both loops in a pass — propagation and claiming — walk every active graph, so
390
+ * a per-graph read made observability cost scale with the number of live workflows on a path
391
+ * that is meant to be flat. One `ID IN (…)` per pass costs the same whether a server is running
392
+ * one workflow or fifty.
393
+ *
394
+ * Already-cached graphs are skipped, so calling this from both loops is free the second time.
395
+ */
396
+ async primeDebugStates(provider, parentTaskIDs) {
397
+ const missing = parentTaskIDs.filter((id) => !this.debugStateByGraph.has(id));
398
+ if (missing.length === 0)
399
+ return;
400
+ try {
401
+ const idList = missing.map((id) => `'${id}'`).join(',');
402
+ const rows = await RunView.FromMetadataProvider(provider).RunView({
403
+ EntityName: 'MJ: Tasks',
404
+ ExtraFilter: `ID IN (${idList})`,
405
+ Fields: ['ID', 'InputPayload'],
406
+ ResultType: 'simple',
407
+ // The bag is written by direct JSON_MODIFY statements, which fire no cache
408
+ // invalidation — a cached read here would gate on state a verb already changed.
409
+ BypassCache: true,
410
+ }, this.contextUser);
411
+ const byID = new Map((rows.Success ? rows.Results ?? [] : []).map((r) => [r.ID, r.InputPayload]));
412
+ for (const id of missing) {
413
+ this.debugStateByGraph.set(id, ParseTaskGraphDebugState(byID.get(id) ?? null));
414
+ }
415
+ }
416
+ catch (e) {
417
+ // "Not being debugged" is the safe reading of "could not read": gating real work on a
418
+ // transient read failure would turn a database hiccup into a paused workflow. Cached as
419
+ // empty for this pass only, so the next pass tries again.
420
+ LogError(`[TaskGraphDispatcher] Could not read debug state for ${missing.length} graph(s): ${e instanceof Error ? e.message : String(e)}`);
421
+ for (const id of missing)
422
+ this.debugStateByGraph.set(id, {});
423
+ }
424
+ }
425
+ /**
426
+ * Announces a graph's pause-state TRANSITION, once, whichever instance notices first in its own
427
+ * frame stream.
428
+ *
429
+ * Per-instance dedup rather than a CAS: frames are advisory commentary, and a viewer receiving
430
+ * the transition from two instances is a duplicate line, not a duplicate execution — the price
431
+ * of a cross-instance guard here would be a write on every pass for a purely cosmetic guarantee.
432
+ */
433
+ async announcePauseTransition(provider, parentTaskID, debug) {
434
+ const paused = debug.paused === true;
435
+ const previous = this.announcedPaused.get(parentTaskID);
436
+ if (previous === paused)
437
+ return;
438
+ this.announcedPaused.set(parentTaskID, paused);
439
+ // First sighting of an unpaused graph needs no announcement — "running" is the default a
440
+ // viewer already assumes; only a transition is information.
441
+ if (previous === undefined && !paused)
442
+ return;
443
+ this.emit({
444
+ Kind: paused ? 'GraphPaused' : 'GraphResumed',
445
+ ParentTaskID: parentTaskID,
446
+ OwnerUserID: await this.resolveOwner(provider, parentTaskID),
447
+ TaskID: debug.pausedAtTaskID ?? undefined,
448
+ Reason: paused
449
+ ? (debug.pausedReason === 'breakpoint' ? 'breakpoint' : 'paused by user')
450
+ : 'resumed',
451
+ });
245
452
  }
246
453
  /**
247
454
  * Begins dispatching.
@@ -260,11 +467,72 @@ export class TaskGraphDispatcher {
260
467
  ShutdownRegistry.Instance.Register(this);
261
468
  LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
262
469
  await this.Reconcile();
263
- this.pollTimer = setInterval(() => { this.pollPass = this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
264
- this.reconcileTimer = setInterval(() => { void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
470
+ // One wide pass over graphs that reached terminal without settling, mirroring what claim
471
+ // reconciliation above already does for tasks. The realistic producer of a >24h-stale
472
+ // unsettled graph is this process having been DOWN — an outage, a long deploy — which the
473
+ // steady-state window cannot see and which would otherwise leave those runs parked forever.
474
+ // Counted as a pass (R2-13). It settles graphs, delivers continuations and can start fresh
475
+ // reinvoke turns, and it runs AFTER this instance registers for shutdown — so a `Stop()`
476
+ // landing during it used to return immediately while the sweep carried on doing all of that
477
+ // against a host that believed the dispatcher had stopped.
478
+ this.activePasses++;
479
+ try {
480
+ await this.sweepUnsettledGraphs(UNSETTLED_STARTUP_WINDOW_HOURS);
481
+ }
482
+ finally {
483
+ this.activePasses--;
484
+ }
485
+ // A `Stop()` LANDING DURING THE BOOT AWAITS MUST NOT BE UNDONE HERE (R3-4).
486
+ //
487
+ // Everything above this line is awaited — reconciliation and the counted startup sweep,
488
+ // which R2-13's own fix makes `Stop()` wait out. So a host that shuts down during boot
489
+ // drains correctly, logs "Stopped.", and returns with both timer fields null — and then
490
+ // this continuation ran anyway and installed both timers on the stopped instance.
491
+ //
492
+ // `pollOnce` was inert (its own `running` guard), but `Reconcile` had no such guard: it
493
+ // minted a provider and ran `ReleaseExpiredClaims` — a real UPDATE returning tasks to
494
+ // Pending — every two minutes forever, against a pool the host may have torn down. Nothing
495
+ // would ever call `Stop()` again, since `ShutdownRegistry.ShutdownAll` clears its items
496
+ // after one pass, and the intervals pinned the event loop so the process could not exit.
497
+ if (!this.running) {
498
+ LogStatus(`[TaskGraphDispatcher] Stopped during startup; not installing timers.`);
499
+ return;
500
+ }
501
+ this.pollTimer = setInterval(() => { void this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
502
+ this.unregisterKick = RegisterTaskGraphKick(() => { this.Kick(); });
503
+ // Do not wait a full interval for work that already exists (or is about to be submitted).
504
+ this.Kick();
505
+ this.reconcileTimer = setInterval(
506
+ // Guarded HERE rather than inside `Reconcile` (R3-4). The defect is a stopped
507
+ // instance's TIMER executing `ReleaseExpiredClaims` — a real UPDATE — forever; the
508
+ // public method itself stays callable, because reconciling on demand before starting is
509
+ // a legitimate use (IT74's crash-recovery check does exactly that) and a guard there
510
+ // would silently no-op it, which is the class of failure this whole effort is about.
511
+ () => { if (this.running)
512
+ void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
513
+ // Belt and suspenders: `Stop()` remains the real teardown, but an un-`unref`'d interval
514
+ // keeps the event loop alive on its own, so a leaked instance can prevent process exit.
515
+ this.pollTimer.unref?.();
516
+ this.reconcileTimer.unref?.();
265
517
  }
266
518
  /**
267
- * Stops accepting new work and waits for in-flight tasks to finish.
519
+ * Stops accepting new work and waits for everything already started to finish.
520
+ *
521
+ * **"Everything" includes the timer passes, and that is the fix.** This waited only on
522
+ * `inFlight` — the task executions — while a poll pass is a `void`-ed promise nothing held. So
523
+ * `Stop()` returned while a pass was mid-flight, and that pass went on to settle graphs, emit
524
+ * lifecycle frames and CLAIM NEW TASKS afterwards. Three consequences, all of them quiet:
525
+ *
526
+ * - a `GraphSettled` frame arrived after every subscriber had gone, so the settlement was
527
+ * invisible to exactly the viewer watching for it;
528
+ * - a process shutting down claimed work it was about to abandon, leaving claims to expire —
529
+ * the orphaned-claim state reconciliation exists to clean up, manufactured by the shutdown;
530
+ * - the host reused the connection the moment `Stop()` resolved, and the still-running pass's
531
+ * statements collided with it (`Requests can only be made in the LoggedIn state`).
532
+ *
533
+ * A pass is bookkeeping for work that already happened, so it is DRAINED rather than cancelled:
534
+ * abandoning one halfway is the crash window the unsettled sweep exists to rescue, and choosing
535
+ * to open it on every clean shutdown would be perverse.
268
536
  *
269
537
  * Deliberately does NOT release claims on the way out: an abandoned claim expires on its own,
270
538
  * and releasing eagerly would hand a still-running task to another instance mid-execution.
@@ -272,6 +540,8 @@ export class TaskGraphDispatcher {
272
540
  */
273
541
  async Stop() {
274
542
  this.running = false;
543
+ this.unregisterKick?.();
544
+ this.unregisterKick = null;
275
545
  if (this.pollTimer) {
276
546
  clearInterval(this.pollTimer);
277
547
  this.pollTimer = null;
@@ -280,20 +550,35 @@ export class TaskGraphDispatcher {
280
550
  clearInterval(this.reconcileTimer);
281
551
  this.reconcileTimer = null;
282
552
  }
283
- // Drain the poll pass BEFORE the task drain below, not after: a pass still running has not
284
- // necessarily claimed anything yet, so `inFlight` can be empty while work is moments from
285
- // starting. Clearing `running` above stops that pass claiming anything further; this waits
286
- // for it to notice. Its own failures are already logged inside `pollOnce`.
287
- if (this.pollPass) {
288
- await this.pollPass.catch(() => undefined);
289
- this.pollPass = null;
290
- }
291
- const deadline = Date.now() + 30_000;
292
- while (this.inFlight.size > 0 && Date.now() < deadline) {
293
- await new Promise((r) => setTimeout(r, 250));
553
+ // Short poll interval: a pass is usually milliseconds from done, and the old 250ms granularity
554
+ // was most of the cost of stopping a dispatcher that had nothing left to do.
555
+ const deadline = Date.now() + STOP_DRAIN_TIMEOUT_MS;
556
+ while ((this.activePasses > 0 || this.inFlight.size > 0) && Date.now() < deadline) {
557
+ await new Promise((r) => setTimeout(r, 25));
294
558
  }
295
559
  if (this.inFlight.size > 0) {
296
- LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will expire.`);
560
+ // The promise in this message was FALSE while the process lived (R2-13): each in-flight
561
+ // task heartbeats its own claim on its own timer, so an over-drain task renewed its lease
562
+ // indefinitely and the claim never expired — reconciliation could not reclaim the work,
563
+ // and the host's shutdown was waiting on something that had stopped being reclaimable.
564
+ // Stopping the heartbeats makes the sentence true. The task itself keeps running; its
565
+ // completion write is guarded on still owning the claim, so if another instance reclaims
566
+ // the task in the meantime, the abandoned executor's result is refused rather than raced.
567
+ // Latched, because the purge RACES the registration it is purging (C5). A task stalled
568
+ // in its two pre-heartbeat awaits — creating a provider, loading the row — registers
569
+ // AFTER this line and would renew its claim for the rest of the process's life, which is
570
+ // exactly the state the drain timeout means to end, under exactly the database duress
571
+ // that causes drain timeouts in the first place.
572
+ this.heartbeatsPurged = true;
573
+ for (const stop of this.heartbeats.values())
574
+ clearInterval(stop);
575
+ this.heartbeats.clear();
576
+ LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will now expire.`);
577
+ }
578
+ if (this.activePasses > 0) {
579
+ // Loud, because from here on this instance writes to a database the host believes it has
580
+ // finished with — the precise shape that produced connection-state errors downstream.
581
+ LogError(`[TaskGraphDispatcher] Stopped with ${this.activePasses} pass(es) still running; their writes may land after shutdown.`);
297
582
  }
298
583
  LogStatus(`[TaskGraphDispatcher] Stopped.`);
299
584
  }
@@ -309,6 +594,51 @@ export class TaskGraphDispatcher {
309
594
  * that shape indicates tampering or a bug and Record Changes already carries the audit trail.
310
595
  */
311
596
  async Reconcile() {
597
+ this.activePasses++;
598
+ try {
599
+ await this.reconcileOnce();
600
+ }
601
+ finally {
602
+ this.activePasses--;
603
+ }
604
+ }
605
+ /**
606
+ * Announces expired claims the sweep just released, so a viewer watching the graph sees "the
607
+ * step's worker vanished and the engine requeued it" as it happens.
608
+ *
609
+ * Best-effort by contract: the release already succeeded and is the durable truth; a frame that
610
+ * cannot be addressed (row unloadable, no parent) is dropped, never retried.
611
+ */
612
+ async announceReclaims(provider, released) {
613
+ if (!this.observer || released.length === 0)
614
+ return;
615
+ try {
616
+ const idList = released.map((r) => `'${r.TaskID}'`).join(',');
617
+ const rows = await RunView.FromMetadataProvider(provider).RunView({
618
+ EntityName: 'MJ: Tasks',
619
+ ExtraFilter: `ID IN (${idList})`,
620
+ Fields: ['ID', 'ParentID', 'Name'],
621
+ ResultType: 'simple',
622
+ BypassCache: true,
623
+ }, this.contextUser);
624
+ for (const row of (rows.Success ? rows.Results : []) ?? []) {
625
+ const graphID = row.ParentID ?? row.ID;
626
+ this.emit({
627
+ Kind: 'ClaimChanged',
628
+ ParentTaskID: graphID,
629
+ OwnerUserID: await this.resolveOwner(provider, graphID),
630
+ TaskID: row.ID,
631
+ TaskName: row.Name,
632
+ ClaimEvent: 'reclaimed',
633
+ });
634
+ }
635
+ }
636
+ catch (e) {
637
+ LogError(`[TaskGraphDispatcher] Could not announce reclaimed task(s) (ignored): ${e instanceof Error ? e.message : String(e)}`);
638
+ }
639
+ }
640
+ /** The reconciliation body. Wrapped by {@link Reconcile} so `Stop()` can drain it. */
641
+ async reconcileOnce() {
312
642
  let provider = null;
313
643
  try {
314
644
  provider = await this.providerFactory.CreateProvider();
@@ -318,11 +648,21 @@ export class TaskGraphDispatcher {
318
648
  LogStatus(`[TaskGraphDispatcher] Reconciliation: ${released.length} expired claim(s) released, ` +
319
649
  `${orphaned.length} orphaned task(s) reported.`);
320
650
  }
651
+ await this.announceReclaims(provider, released);
321
652
  }
322
653
  catch (e) {
323
654
  LogError(`[TaskGraphDispatcher] Reconciliation failed: ${e instanceof Error ? e.message : String(e)}`);
324
655
  }
325
656
  }
657
+ /**
658
+ * Run a pass now instead of waiting for the next poll tick.
659
+ *
660
+ * Submit calls this (via {@link KickTaskGraphDispatchers}) so a just-written graph is claimed
661
+ * in milliseconds rather than up to {@link TaskGraphDispatcherConfig.PollIntervalSeconds}.
662
+ */
663
+ Kick() {
664
+ void this.pollOnce();
665
+ }
326
666
  /**
327
667
  * One dispatch pass: find claimable work, claim what fits under the concurrency cap, execute.
328
668
  *
@@ -332,10 +672,11 @@ export class TaskGraphDispatcher {
332
672
  async pollOnce() {
333
673
  if (!this.running || this.polling)
334
674
  return;
335
- const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
336
- if (capacity <= 0)
337
- return;
338
675
  this.polling = true;
676
+ this.activePasses++;
677
+ // One pass, one read of each graph's debug state — see `readDebugState`.
678
+ this.debugStateByGraph.clear();
679
+ const passNumber = ++this.passCounter;
339
680
  try {
340
681
  const provider = await this.providerFactory.CreateProvider();
341
682
  // `running` is re-read after every await from here on. The entry check above only proves
@@ -346,15 +687,35 @@ export class TaskGraphDispatcher {
346
687
  // then sit claimed until their lease expires.
347
688
  if (!this.running)
348
689
  return;
349
- // Settle graphs before picking new work, so a failure earlier in this pass stops its
350
- // branch immediately rather than after another wave has already launched.
690
+ // SETTLEMENT IS NOT GATED ON CAPACITY (R2-11).
691
+ //
692
+ // This used to return at `capacity <= 0` before reaching the rollup, so a handful of
693
+ // wedged long-running tasks froze EVERYTHING for the whole instance: no settlement, no
694
+ // skip or block propagation, no human-task settlement, no continuation delivery — for
695
+ // graphs that had nothing to do with the tasks holding the slots. A per-task hang is an
696
+ // accepted limitation; "one hung task stops every workflow on this host" is not, and the
697
+ // two were the same line of code.
698
+ //
699
+ // Only CLAIMING consumes capacity, because only claiming starts work.
351
700
  await this.propagateAndRollup(provider);
701
+ const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
702
+ if (capacity <= 0)
703
+ return;
704
+ // The rollup above can take seconds, and `Stop()` may have been called during it. Claiming
705
+ // now would start work the process has already decided to abandon — the claim then sits
706
+ // until its TTL expires and another instance reclaims it. Settling first and checking
707
+ // here is the right order: bookkeeping for finished work always completes, new work never
708
+ // starts after the decision to stop.
352
709
  if (!this.running)
353
710
  return;
354
- const candidates = await this.findClaimableTasks(provider, capacity);
711
+ const { tasks: candidates, stats } = await this.findClaimableTasks(provider, capacity);
712
+ const claimedByGraph = new Map();
355
713
  for (const task of candidates) {
356
- // Re-checked per iteration, not just before the loop: claiming is itself awaited, so
357
- // a multi-task wave can straddle a Stop.
714
+ // Re-checked EVERY iteration, not once before the loop (R2-13). Claiming is itself
715
+ // awaited, so a multi-task wave can straddle a `Stop`; and `findClaimableTasks` loads
716
+ // and resolves every active graph, so the scan before this loop can run for seconds.
717
+ // Unchecked, a shutting-down process takes ownership of work it is about to abandon,
718
+ // manufacturing the orphaned claims reconciliation exists to clean up.
358
719
  if (!this.running)
359
720
  break;
360
721
  if (this.inFlight.size >= this.config.MaxConcurrentTasks)
@@ -363,16 +724,34 @@ export class TaskGraphDispatcher {
363
724
  // Another instance won the race, or the task is no longer Pending. Normal.
364
725
  continue;
365
726
  }
727
+ const graphID = task.ParentID ?? task.ID;
728
+ claimedByGraph.set(graphID, (claimedByGraph.get(graphID) ?? 0) + 1);
366
729
  this.inFlight.add(task.ID);
367
730
  // Intentionally not awaited — the poll loop must keep dispatching while this runs.
368
731
  void this.executeClaimed(task.ID).finally(() => this.inFlight.delete(task.ID));
369
732
  }
733
+ // The engine's heartbeat, per watched graph: what was ready, what was held, what this
734
+ // instance took. A stuck run is a strip of these ticking with nothing moving, which is
735
+ // the honest visual of a stall — and the reason this frame exists.
736
+ for (const [graphID, s] of stats) {
737
+ this.emit({
738
+ Kind: 'PassCompleted',
739
+ ParentTaskID: graphID,
740
+ OwnerUserID: await this.resolveOwner(provider, graphID),
741
+ PassNumber: passNumber,
742
+ EligibleCount: s.eligible,
743
+ HeldCount: s.held,
744
+ ClaimedCount: claimedByGraph.get(graphID) ?? 0,
745
+ InstanceInFlightCount: this.inFlight.size,
746
+ });
747
+ }
370
748
  }
371
749
  catch (e) {
372
750
  LogError(`[TaskGraphDispatcher] Poll failed: ${e instanceof Error ? e.message : String(e)}`);
373
751
  }
374
752
  finally {
375
753
  this.polling = false;
754
+ this.activePasses--;
376
755
  }
377
756
  }
378
757
  /**
@@ -390,19 +769,44 @@ export class TaskGraphDispatcher {
390
769
  LogError(`[TaskGraphDispatcher] Claimed task ${taskID} could not be loaded.`);
391
770
  return;
392
771
  }
772
+ // Emitted after the claim is held, not before: a frame saying "started" for work another
773
+ // instance actually took would be a lie a viewer cannot detect.
774
+ const graphID = task.ParentID ?? taskID;
775
+ const ownerUserID = await this.resolveOwner(provider, graphID);
393
776
  heartbeat = setInterval(() => {
394
777
  void this.claims.Heartbeat(provider, taskID, this.contextUser).then((ok) => {
395
778
  if (!ok) {
396
779
  // Lost ownership — reconciliation reclaimed it, or a human intervened.
397
780
  LogError(`[TaskGraphDispatcher] Lost claim on task ${taskID} while executing; another instance may take it over.`);
781
+ // Announced so a viewer sees "this step's worker lost its lease" the moment
782
+ // it happens instead of discovering it in a forensic query later — the R2-1
783
+ // wedge class, made visible.
784
+ this.emit({
785
+ Kind: 'ClaimChanged', ParentTaskID: graphID, OwnerUserID: ownerUserID,
786
+ TaskID: taskID, TaskName: task.Name,
787
+ ClaimEvent: 'heartbeat-lost', ClaimedBy: this.config.InstanceID,
788
+ });
398
789
  }
399
790
  });
400
791
  }, this.config.HeartbeatIntervalSeconds * 1000);
401
- // Emitted after the claim is held, not before: a frame saying "started" for work another
402
- // instance actually took would be a lie a viewer cannot detect.
403
- const graphID = task.ParentID ?? taskID;
404
- const ownerUserID = await this.resolveOwner(provider, graphID);
792
+ // Registered so `Stop()` can reach it — unless the drain has already given up, in which
793
+ // case this task arrived too late to be waited for and must not renew its lease (C5).
794
+ // Its completion write stays guarded, so if another instance reclaims the task in the
795
+ // meantime this executor's result is refused rather than raced.
796
+ if (this.heartbeatsPurged || !this.running) {
797
+ clearInterval(heartbeat);
798
+ heartbeat = null;
799
+ }
800
+ else {
801
+ this.heartbeats.set(taskID, heartbeat);
802
+ }
405
803
  this.emit({ Kind: 'TaskStarted', ParentTaskID: graphID, OwnerUserID: ownerUserID, TaskID: taskID, TaskName: task.Name, Status: 'In Progress' });
804
+ this.emit({
805
+ Kind: 'ClaimChanged', ParentTaskID: graphID, OwnerUserID: ownerUserID,
806
+ TaskID: taskID, TaskName: task.Name,
807
+ ClaimEvent: 'claimed', ClaimedBy: this.config.InstanceID,
808
+ ClaimExpiresAt: new Date(Date.now() + this.config.ClaimTTLSeconds * 1000).toISOString(),
809
+ });
406
810
  const dependencyOutputs = await this.loadDependencyOutputs(provider, taskID);
407
811
  let inputPayload = null;
408
812
  if (task.InputPayload) {
@@ -413,13 +817,18 @@ export class TaskGraphDispatcher {
413
817
  LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
414
818
  }
415
819
  }
416
- const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs);
417
- // A prompt can end the workflow early and say why. Honour it before recording the
418
- // outcome, so the remaining tasks are already Skipped by the time the rollup runs and
419
- // the graph settles Complete rather than looking abandoned with work left Pending.
420
- if (result.ChatMessage) {
421
- await this.endGraphEarly(provider, task, result.ChatMessage);
422
- }
820
+ const onProgress = this.nodeProgressEmitter(graphID, ownerUserID, taskID, task.Name);
821
+ const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs, onProgress);
822
+ // ONLY THE CONFIRMED OWNER MUTATES THE GRAPH (R2-10).
823
+ //
824
+ // The early-finish skips used to run BEFORE this, so a lapsed claim produced the worst
825
+ // possible pair: the siblings were terminally Skipped and satisfying dependents, while
826
+ // the completion was refused and the task re-ran on another instance — where it might
827
+ // not end early at all. The graph would then be missing steps nobody decided to skip.
828
+ //
829
+ // Recording first costs a poll: the skips now land after the completion, so a rollup
830
+ // that lands in between sees work still Pending and settles one pass later. That is a
831
+ // delay; the other order was a wrong graph.
423
832
  const recorded = await this.claims.CompleteClaimed(provider, taskID, {
424
833
  Status: result.Success ? 'Complete' : 'Failed',
425
834
  OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
@@ -445,6 +854,13 @@ export class TaskGraphDispatcher {
445
854
  Status: result.Success ? 'Complete' : 'Failed',
446
855
  ErrorMessage: result.Success ? undefined : (result.ErrorMessage ?? undefined),
447
856
  });
857
+ // A prompt can end the workflow early and say why — honoured only now that this
858
+ // instance is the confirmed owner of the outcome. The remaining tasks are Skipped
859
+ // here so the graph settles Complete rather than looking abandoned with work left
860
+ // Pending; a rollup that lands between the two simply settles one pass later.
861
+ if (result.ChatMessage) {
862
+ await this.endGraphEarly(provider, task, result.ChatMessage);
863
+ }
448
864
  }
449
865
  }
450
866
  catch (e) {
@@ -458,6 +874,7 @@ export class TaskGraphDispatcher {
458
874
  finally {
459
875
  if (heartbeat)
460
876
  clearInterval(heartbeat);
877
+ this.heartbeats.delete(taskID);
461
878
  }
462
879
  }
463
880
  /**
@@ -466,8 +883,11 @@ export class TaskGraphDispatcher {
466
883
  * All four decisions — what is eligible, what must block, what the parent status is, whether the
467
884
  * graph is wedged — are delegated to the pure algorithms, unchanged from Phase 1.
468
885
  */
469
- async propagateAndRollup(provider) {
470
- for (const parentID of await this.findActiveGraphIDs(provider)) {
886
+ async propagateAndRollup(provider, graphIDs) {
887
+ const graphs = graphIDs ?? await this.findActiveGraphIDs(provider);
888
+ // One read for every graph this pass touches, rather than one per graph — see primeDebugStates.
889
+ await this.primeDebugStates(provider, graphs);
890
+ for (const parentID of graphs) {
471
891
  // Human steps settle BEFORE the graph state is read, so an answer given since the last
472
892
  // poll is already reflected when eligibility and rollup are computed. Doing it after
473
893
  // would delay every dependent branch by a full poll interval for no reason — and on a
@@ -475,8 +895,11 @@ export class TaskGraphDispatcher {
475
895
  // between "answered and moving" and "answered and apparently still stuck".
476
896
  await this.expireOverdueRequests(provider, parentID);
477
897
  await this.settleAnsweredHumanTasks(provider, parentID);
478
- await this.reopenCancelledHumanTasks(provider, parentID);
479
- const graph = await this.loadGraphState(provider, parentID);
898
+ await this.reconcileWaitingHumanTasks(provider, parentID);
899
+ // Edge overrides apply to propagation exactly as they apply to claiming — a branch the
900
+ // operator answered 'false' must cascade its skips here, not merely stop being claimed.
901
+ const debug = await this.readDebugState(provider, parentID);
902
+ const graph = await this.loadGraphState(provider, parentID, debug);
480
903
  if (graph.nodes.length === 0)
481
904
  continue;
482
905
  // SKIPS FIRST — before blocking, before eligibility. A task whose gating predecessors
@@ -490,17 +913,22 @@ export class TaskGraphDispatcher {
490
913
  // route" and "something upstream broke". A reader cannot tell those apart, so every
491
914
  // conditional workflow looked half-failed and people went hunting for bugs that did not
492
915
  // exist. `Blocked` is now reserved for FAILURE-driven unsatisfiability.
493
- const skipSeeds = new Set([...graph.skipSeedTaskIDs, ...graph.unreachableTaskIDs]);
494
- const toSkip = new Set([
495
- ...skipSeeds,
496
- ...ComputeSkipCascade(graph.nodes, graph.edges, [...skipSeeds]),
497
- ]);
916
+ // Computed once in `loadGraphState` so the claim filter sees the same set this pass is
917
+ // about to write — see R2-14 there.
918
+ const toSkip = graph.cascadeSkipTaskIDs;
919
+ const skippedByRoute = [];
498
920
  for (const taskID of toSkip) {
499
921
  const entity = graph.entityById.get(taskID);
500
922
  if (!entity || entity.Status !== 'Pending')
501
923
  continue;
502
- entity.Status = 'Skipped';
503
- if (await entity.Save()) {
924
+ // The in-memory check above is a cheap pre-filter; the guard that matters is IN the
925
+ // statement (R3-1's audit item). This snapshot was loaded at the top of the pass and
926
+ // a task can be claimed and started before its skip write lands — R2-14 closed that
927
+ // window for the claim filter, and this closes it for the write itself.
928
+ const skipTypeID = await this.workflowTaskTypeID(provider);
929
+ if (skipTypeID && await this.claims.TrySkipPending(provider, taskID, skipTypeID, this.contextUser)) {
930
+ entity.Status = 'Skipped';
931
+ skippedByRoute.push(taskID);
504
932
  LogStatus(`[TaskGraphDispatcher] Skipped '${entity.Name}' (${taskID}) — another branch was taken.`);
505
933
  // Announced separately from TaskBlocked because it means something different to
506
934
  // a viewer: nothing went wrong, this route simply was not the one chosen.
@@ -519,6 +947,11 @@ export class TaskGraphDispatcher {
519
947
  node.status = 'Skipped';
520
948
  }
521
949
  }
950
+ // A human step reached by a route the workflow did not take has the same zombie request
951
+ // as one skipped by an early finish (R2-10): notified, `Requested` forever, and invisible
952
+ // to the settle and expiry sweeps because they filter on Pending tasks and this one is
953
+ // not Pending any more. Same treatment, different reason.
954
+ await this.withdrawOpenRequests(provider, skippedByRoute, 'The workflow took a different route, so this step is no longer needed.');
522
955
  // Only failure-driven unsatisfiability reaches here now; not-taken branches were skipped
523
956
  // above. A task already Skipped is left alone rather than overwritten — the two passes
524
957
  // must not fight over the same row.
@@ -543,10 +976,14 @@ export class TaskGraphDispatcher {
543
976
  });
544
977
  }
545
978
  }
546
- if (IsGraphStalled(graph.nodes, graph.edges)) {
979
+ // Holds are passed in, or the detector reports a held graph as healthy: a held target's
980
+ // gating edge is still live and its origin Complete, so ComputeEligibleTasks counts it
981
+ // as eligible and "something is eligible" reads as "not stalled". A graph waiting
982
+ // forever on a broken condition then produced no diagnostics at all.
983
+ if (IsGraphStalled(graph.nodes, graph.edges, graph.holdTaskIDs)) {
547
984
  LogError(`[TaskGraphDispatcher] Graph ${parentID} is stalled: pending work with no satisfiable path.`);
548
985
  }
549
- const fresh = await this.loadGraphState(provider, parentID);
986
+ const fresh = await this.loadGraphState(provider, parentID, debug);
550
987
  // ComputeParentRollup treats an empty child set as Complete-and-terminal, which is right
551
988
  // for a graph that genuinely has no children and catastrophic for one whose reload came
552
989
  // back empty transiently — it would mark live work finished and fire its continuation.
@@ -568,40 +1005,127 @@ export class TaskGraphDispatcher {
568
1005
  // Taken from the earliest child rather than from the clock, because that is when work
569
1006
  // genuinely began — a graph can sit Pending for a long time between submission (already
570
1007
  // recorded as CreatedAt) and a dispatcher picking up its first task.
1008
+ // Column-scoped for the same reason the terminal write is: a full-row save here would
1009
+ // carry this instance's `InputPayload` snapshot and could erase a continuation marker
1010
+ // another instance had just claimed. Guarded on `StartedAt IS NULL`, so calling it on
1011
+ // every pass is free.
571
1012
  const earliestChildStart = this.earliestStart(fresh.entityById);
572
- const startedAtChanged = parent.StartedAt == null && earliestChildStart != null;
573
- if (startedAtChanged)
1013
+ if (parent.StartedAt == null && earliestChildStart != null) {
1014
+ await this.claims.TryStampParentStart(provider, parentID, earliestChildStart, this.contextUser);
574
1015
  parent.StartedAt = earliestChildStart;
575
- if (startedAtChanged || parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
1016
+ }
1017
+ // THE TERMINAL WRITE IS GUARDED AND COLUMN-SCOPED, not a full-row save.
1018
+ //
1019
+ // `GenerateSaveSQL` sends every updateable column on every save, so a full-row save
1020
+ // carries the whole in-memory snapshot — including `InputPayload`, where the continuation
1021
+ // marker lives. Two instances both compute the terminal rollup; if one claims the marker
1022
+ // and the other then saves its pre-marker snapshot, the marker is ERASED and the
1023
+ // settlement delivers twice. For `reinvoke` that is a second billed agent turn.
1024
+ //
1025
+ // Guarding on "not already terminal" also makes the write idempotent across the
1026
+ // re-entrant settle path below, and replaces an unchecked `Save()` whose failure left the
1027
+ // graph active — re-emitting frames and recomputing cost every poll, forever.
1028
+ if (rollup.outcome === 'settled') {
1029
+ const settled = await this.claims.TrySettleParent(provider, parentID, rollup.status, rollup.percentComplete, this.contextUser);
1030
+ if (!settled && !TERMINAL_TASK_STATUSES.has(parent.Status)) {
1031
+ // Neither "already terminal" nor a successful write: the statement failed. Leave
1032
+ // the graph active so the next pass retries rather than settling on a status the
1033
+ // database never accepted. Re-queued explicitly, because a failed write is
1034
+ // exactly the case where the row's own timestamp does not advance.
1035
+ LogError(`[TaskGraphDispatcher] Could not write terminal status for graph ${parentID}; leaving it active to retry.`);
1036
+ this.keepRetryingSettlement(parentID);
1037
+ continue;
1038
+ }
576
1039
  parent.Status = rollup.status;
577
- parent.PercentComplete = rollup.percentComplete;
578
- if (rollup.isTerminal)
579
- parent.CompletedAt = new Date();
580
- await parent.Save();
581
- }
582
- if (rollup.isTerminal) {
583
- // Geometry is settled once, here, so every viewer of this run agrees on it.
584
- await this.persistComputedLayout(fresh);
585
- // Emitted before the continuation is delivered, and outside its once-only guard: a
586
- // viewer watching the run should learn it finished whether or not this instance is
587
- // the one that wins the delivery CAS.
588
- this.emit({
589
- Kind: 'GraphSettled',
590
- ParentTaskID: parentID,
591
- OwnerUserID: await this.resolveOwner(provider, parentID),
592
- Status: rollup.status,
593
- CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
594
- TotalCount: fresh.nodes.length,
595
- });
596
- await this.rollUpCostToSubmittingRun(provider, parent);
1040
+ }
1041
+ else if (parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
1042
+ // Guarded and column-scoped for the same reason the terminal write is — and the race
1043
+ // here needs no exotic timing. This instance may have computed a non-terminal rollup
1044
+ // from a snapshot taken before another instance settled the graph; a full-row save
1045
+ // would then REVERT the status and erase the continuation marker with it, and the
1046
+ // next pass would settle and deliver a second time. See TryUpdateParentProgress.
1047
+ await this.claims.TryUpdateParentProgress(provider, parentID, rollup.status, rollup.percentComplete, this.contextUser);
1048
+ }
1049
+ if (rollup.outcome === 'settled') {
1050
+ // ANNOUNCE ONCE PER PROCESS, RETRY THE REST (R2-12). Layout and the frame are
1051
+ // idempotent facts about a finished graph; the cost, lifecycle and delivery writes
1052
+ // below are the ones re-entry exists to retry. Without this split, a graph that keeps
1053
+ // failing to settle re-persisted geometry and re-emitted the same frame every poll
1054
+ // for the whole rescue window.
1055
+ if (!this.announcedSettlements.has(parentID)) {
1056
+ // Geometry is settled once, here, so every viewer of this run agrees on it.
1057
+ await this.persistComputedLayout(fresh);
1058
+ // Emitted before the continuation is delivered, and outside its once-only guard: a
1059
+ // viewer watching the run should learn it finished whether or not this instance is
1060
+ // the one that wins the delivery CAS.
1061
+ this.emit({
1062
+ Kind: 'GraphSettled',
1063
+ ParentTaskID: parentID,
1064
+ OwnerUserID: await this.resolveOwner(provider, parentID),
1065
+ Status: rollup.status,
1066
+ CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
1067
+ TotalCount: fresh.nodes.length,
1068
+ });
1069
+ this.announcedSettlements.add(parentID);
1070
+ }
1071
+ // READ-ONLY GATE, before any write to the submitting run's half (R2-2).
1072
+ //
1073
+ // A graph can settle before the run that submitted it has parked at all. `BaseAgent`
1074
+ // sets `Paused` in `finalizeAgentRun`, AFTER the graph is durable and dispatchable —
1075
+ // so a fast graph finishes first, and both writes below then land wrong: the
1076
+ // lifecycle write silently returns (its guard is `Status === 'Paused'`), and the cost
1077
+ // write is overwritten moments later by finalize's own full-row save, which carries
1078
+ // the in-memory nulls it had before the dispatcher wrote anything.
1079
+ //
1080
+ // Deferring the whole half — rather than doing the parts that happen to work — is
1081
+ // what keeps the marker honest: nothing below claims it, so the graph stays
1082
+ // terminal-and-undelivered and the rescue sweep brings it back next pass, by which
1083
+ // time finalize has parked the run and both writes land.
1084
+ const readiness = await this.submittingRunReadiness(provider, parent);
1085
+ if (readiness.Verdict === 'defer') {
1086
+ this.keepRetryingSettlement(parentID);
1087
+ continue;
1088
+ }
1089
+ // A settled graph will not produce another verdict, claim or progress report, so the
1090
+ // per-graph observability caches for it are dead weight from here. Unbounded, they
1091
+ // are a slow leak in a process that runs for weeks — one that grows with every
1092
+ // workflow the server has ever seen rather than with anything live. (A deferred
1093
+ // settlement below repopulates them naturally on the retry pass.)
1094
+ this.forgetGraphObservability(parentID);
1095
+ // R3-8: the rollup gets a verdict, and a TRANSIENT failure defers exactly as a
1096
+ // failed lifecycle write does. Continuing past one would settle the run and claim
1097
+ // the marker, which permanently excludes the graph from the rescue sweep — making
1098
+ // the rollup's own "retrying on a later settlement" log a promise it could not keep.
1099
+ // Permanent refusals (truncated tree, graph not in the tree) proceed as before:
1100
+ // those do not clear on their own, and deferring on them would stall forever.
1101
+ if (await this.rollUpCostToSubmittingRun(provider, parent) === 'failed-transient') {
1102
+ this.keepRetryingSettlement(parentID);
1103
+ continue;
1104
+ }
597
1105
  // Deliberately AFTER the rollup and OUTSIDE its refusal paths. The rollup declines
598
1106
  // to write a number it cannot stand behind — a truncated tree, an unreachable graph
599
1107
  // — and every one of those returns early. If the run's lifecycle were settled in
600
1108
  // there, a refused rollup would strand the run parked forever, which is a far worse
601
1109
  // failure than a missing cost figure. Cost and lifecycle are separate concerns with
602
1110
  // separate failure modes, so they get separate writes.
603
- await this.settleSubmittingRun(provider, parent, rollup.status);
604
- await this.deliverContinuation(provider, parent, fresh);
1111
+ if (await this.settleSubmittingRun(provider, parent, rollup.status) === 'defer') {
1112
+ // The lifecycle write did not land. Delivering now would claim the marker and
1113
+ // make this the LAST pass to look at the graph — leaving the run Paused forever,
1114
+ // which is the exact permanence R2-2 removes. Leave the marker unset and retry.
1115
+ this.keepRetryingSettlement(parentID);
1116
+ continue;
1117
+ }
1118
+ if (await this.deliverContinuation(provider, parent, fresh, readiness.SubmitterCancelled)) {
1119
+ // Delivered, expired, or lost the CAS to a peer — every one of those means this
1120
+ // graph is somebody's finished business and needs nothing further from here.
1121
+ this.retryingSettlement.delete(parentID);
1122
+ this.announcedSettlements.delete(parentID);
1123
+ }
1124
+ else {
1125
+ // This instance cannot deliver. Stay quiet about it — the frame is already out —
1126
+ // and leave the graph for a capable peer via the sweep.
1127
+ this.keepRetryingSettlement(parentID);
1128
+ }
605
1129
  }
606
1130
  }
607
1131
  }
@@ -647,13 +1171,15 @@ export class TaskGraphDispatcher {
647
1171
  async rollUpCostToSubmittingRun(provider, parent) {
648
1172
  const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
649
1173
  if (!meta.submittedByAgentRunID)
650
- return;
1174
+ return 'landed';
651
1175
  const runID = meta.submittedByAgentRunID;
652
1176
  try {
653
1177
  const runQuery = asRunQueryProvider(provider);
654
1178
  if (!runQuery) {
1179
+ // Permanent for this host: a provider that cannot run queries will not grow the
1180
+ // ability on the next pass, so deferring would stall the graph forever.
655
1181
  LogError(`[TaskGraphDispatcher] Cannot roll up cost for run ${runID}: provider cannot run queries.`);
656
- return;
1182
+ return 'refused-permanent';
657
1183
  }
658
1184
  const tree = await LoadAgentRunTree(runID, runQuery, this.contextUser);
659
1185
  // Each of these means the sum would be a LOWER BOUND, and the column's whole contract is
@@ -667,41 +1193,49 @@ export class TaskGraphDispatcher {
667
1193
  // absent, and `?? TotalCost` cannot save a reader from a non-null wrong number. So a
668
1194
  // refusal CLEARS it, restoring the fallback's honest meaning: not settled.
669
1195
  if (tree.ErrorMessage || !tree.Root) {
670
- await this.clearStaleRollup(provider, runID, tree.ErrorMessage ?? 'the run tree came back empty');
671
- return;
1196
+ // TRANSIENT — nothing cleared (R2-15), and now nothing delivered either (R3-8).
1197
+ //
1198
+ // R2-15 stopped this path erasing a correct total. What it did not stop was the pass
1199
+ // CONTINUING: settlement flipped the run terminal and delivery claimed the marker,
1200
+ // which permanently excludes the graph from the rescue sweep — so this function's
1201
+ // own promise of "retrying on a later settlement" was structurally impossible to
1202
+ // keep. One transient DB error, no interleaving, and a multi-graph run kept a wrong
1203
+ // non-null authoritative total forever while a first-graph run stayed null.
1204
+ LogError(`[TaskGraphDispatcher] Could not load the run tree for ${runID} to roll up graph ` +
1205
+ `${parent.ID}: ${tree.ErrorMessage ?? 'the run tree came back empty'}. Leaving any ` +
1206
+ `existing rollup alone and deferring settlement so a later pass can retry.`);
1207
+ return 'failed-transient';
672
1208
  }
673
1209
  if (tree.Truncated) {
1210
+ // PERMANENT: the tree is genuinely too deep, and it will be just as deep next pass.
674
1211
  await this.clearStaleRollup(provider, runID, `the run tree hit the depth cap, so any total would silently under-report ` +
675
1212
  `(graph ${parent.ID} still carries its own costs)`);
676
- return;
1213
+ return 'refused-permanent';
677
1214
  }
678
1215
  // The graph that just settled must appear in the tree. If it does not, the tree stopped
679
1216
  // at the run — the submitting step never recorded its parentTaskID — and the sum is
680
1217
  // merely the run's own spend wearing the name of a rollup. That is precisely the silent
681
1218
  // under-count this rewrite exists to remove, so it is reported rather than written.
682
1219
  if (!this.treeContainsGraph(tree.Root, parent.ID)) {
1220
+ // PERMANENT: a missing parentTaskID link is a fact about how the graph was
1221
+ // submitted, not a condition that clears on its own.
683
1222
  await this.clearStaleRollup(provider, runID, `graph ${parent.ID} is not reachable from it, so the tree cannot see the work. ` +
684
1223
  `Did the submitting step record parentTaskID?`);
685
- return;
1224
+ return 'refused-permanent';
686
1225
  }
687
1226
  const totals = SumAgentRunTreeCost(tree.Root);
688
- const submitting = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
689
- if (!(await submitting.Load(runID))) {
690
- LogError(`[TaskGraphDispatcher] Could not load run ${runID} to record graph cost against it.`);
691
- return;
692
- }
693
1227
  // Assignment, never accumulation. The tree already contains the run's own spend as its
694
1228
  // ROOT node, and it reads own-cost everywhere, so recomputing from scratch on every
695
1229
  // settlement lands on the same answer — which is what makes this safe to call again when
696
1230
  // a second graph settles, or when the terminal check is re-evaluated after a HITL wait.
697
- submitting.TotalCostRollup = totals.Cost;
698
- submitting.TotalTokensUsedRollup = totals.Tokens;
699
- submitting.TotalPromptTokensUsedRollup = totals.PromptTokens;
700
- submitting.TotalCompletionTokensUsedRollup = totals.CompletionTokens;
701
- if (!(await submitting.Save())) {
702
- LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}: ` +
703
- `${submitting.LatestResult?.CompleteMessage ?? 'unknown error'}`);
704
- return;
1231
+ //
1232
+ // COLUMN-SCOPED (C4). A full-row save here carried a whole snapshot of the run, and two
1233
+ // instances entering the settled branch for one graph is by design — so a peer's rollup,
1234
+ // loaded before this instance settled the run, would write `Paused` back over
1235
+ // `Completed` along with everything else it had read.
1236
+ if (!(await this.claims.TrySetRunCostRollup(provider, runID, totals, this.contextUser))) {
1237
+ LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}.`);
1238
+ return 'failed-transient';
705
1239
  }
706
1240
  LogStatus(`[TaskGraphDispatcher] Credited graph ${parent.ID} to run ${runID}: ` +
707
1241
  `${tree.Rows.length} node(s), ${totals.Tokens} token(s), cost ${totals.Cost}.`);
@@ -710,8 +1244,13 @@ export class TaskGraphDispatcher {
710
1244
  // A failed rollup must never fail the graph. The work finished; only the accounting for
711
1245
  // it is missing, and a graph marked Failed because its cost could not be summed would be
712
1246
  // a far worse lie than a cost of null.
1247
+ // A throw is transient by default: nothing here proves the condition will persist, and
1248
+ // the cost of being wrong in this direction is one deferred pass rather than a
1249
+ // permanently wrong authoritative total.
713
1250
  LogError(`[TaskGraphDispatcher] Cost rollup failed for graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
1251
+ return 'failed-transient';
714
1252
  }
1253
+ return 'landed';
715
1254
  }
716
1255
  /**
717
1256
  * Clears a rollup that can no longer be trusted, and says why.
@@ -783,25 +1322,70 @@ export class TaskGraphDispatcher {
783
1322
  return;
784
1323
  try {
785
1324
  LogStatus(`[TaskGraphDispatcher] '${task.Name}' ended the workflow early: ${message}`);
1325
+ // DECLARE BEFORE MUTATING (R3-1).
1326
+ //
1327
+ // The early finish is decided by one task's result and known to nothing else: skip seeds
1328
+ // come from durable condition and exclusive-group state, so no claim filter anywhere —
1329
+ // including this instance's own, since `executeClaimed` is not awaited and the poll loop
1330
+ // runs concurrently — can tell the remaining steps are about to be skipped. Stamping it
1331
+ // first is what lets `loadGraphState` fold them into the filter, which closes the window
1332
+ // for every instance instead of narrowing it for one.
1333
+ const typeID = await this.workflowTaskTypeID(provider);
1334
+ if (typeID)
1335
+ await this.claims.TryDeclareEarlyFinish(provider, task.ParentID, typeID, this.contextUser);
1336
+ const skipped = [];
1337
+ const claimedMeanwhile = [];
786
1338
  for (const sibling of await this.loadChildTasks(provider, task.ParentID)) {
787
- if (sibling.ID === task.ID || sibling.Status !== 'Pending')
1339
+ if (UUIDsEqual(sibling.ID, task.ID) || sibling.Status !== 'Pending')
1340
+ continue;
1341
+ // GUARDED, not a full-row save against the snapshot above. A sibling claimed between
1342
+ // that load and this write is mid-execution; overwriting it to `Skipped` discards a
1343
+ // running step's outcome while its side effects have already fired. Rowcount is the
1344
+ // verdict — see TaskClaimStore.TrySkipPending.
1345
+ if (!typeID || !(await this.claims.TrySkipPending(provider, sibling.ID, typeID, this.contextUser))) {
1346
+ claimedMeanwhile.push(sibling.Name);
788
1347
  continue;
789
- sibling.Status = 'Skipped';
790
- if (await sibling.Save()) {
791
- this.emit({
792
- Kind: 'TaskSkipped',
793
- ParentTaskID: task.ParentID,
794
- OwnerUserID: await this.resolveOwner(provider, task.ParentID),
795
- TaskID: sibling.ID,
796
- TaskName: sibling.Name,
797
- Status: 'Skipped',
798
- });
799
1348
  }
1349
+ skipped.push(sibling.ID);
1350
+ this.emit({
1351
+ Kind: 'TaskSkipped',
1352
+ ParentTaskID: task.ParentID,
1353
+ OwnerUserID: await this.resolveOwner(provider, task.ParentID),
1354
+ TaskID: sibling.ID,
1355
+ TaskName: sibling.Name,
1356
+ Status: 'Skipped',
1357
+ });
800
1358
  }
801
- const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
802
- if (await parent.Load(task.ParentID)) {
803
- parent.OutputPayload = JSON.stringify({ message });
804
- await parent.Save();
1359
+ if (claimedMeanwhile.length > 0) {
1360
+ // Reported rather than forced. Those steps were already running when the workflow
1361
+ // decided to stop; their results are real and are allowed to land. The graph settles
1362
+ // once they finish, which is a pass later than it would have and correct.
1363
+ LogStatus(`[TaskGraphDispatcher] Early finish left ${claimedMeanwhile.length} step(s) running ` +
1364
+ `(${claimedMeanwhile.join(', ')}) — they were claimed before the skip and their outcomes stand.`);
1365
+ }
1366
+ // WITHDRAW WHAT WE JUST SKIPPED (R2-10).
1367
+ //
1368
+ // A skipped human step leaves its `MJ: AI Agent Requests` row `Requested` forever:
1369
+ // un-answerable, because answering settles nothing once the task is terminal, and
1370
+ // immortal, because the human settle and expiry sweeps both filter on `Status='Pending'`
1371
+ // tasks and this one no longer is. The person keeps seeing "a workflow is waiting on
1372
+ // you" for a workflow that finished without them. `Cancel` has always done this; the
1373
+ // early-finish path skipped exactly the same rows and did not.
1374
+ await this.withdrawOpenRequests(provider, skipped, 'The workflow finished before this step was needed.');
1375
+ // Column-scoped, because skipping the siblings above just made this graph fully
1376
+ // terminal — so another instance can settle it and claim the marker before this line
1377
+ // runs. A full-row save from the snapshot we loaded first would undo both. See
1378
+ // TaskClaimStore.TrySetParentOutput.
1379
+ if (typeID) {
1380
+ await this.claims.TrySetParentOutput(provider, task.ParentID, JSON.stringify({ message }), typeID, this.contextUser);
1381
+ }
1382
+ else {
1383
+ // Surfaced rather than dropped (R2-10). The graph still ends early — the siblings
1384
+ // are already Skipped — but the reason it ended goes nowhere, and a workflow that
1385
+ // stopped for a stated reason with no stated reason recorded is exactly the kind of
1386
+ // silence this round exists to remove.
1387
+ LogError(`[TaskGraphDispatcher] Could not resolve the workflow task type, so the early-finish ` +
1388
+ `reason for graph ${task.ParentID} was not recorded: ${message}`);
805
1389
  }
806
1390
  }
807
1391
  catch (e) {
@@ -849,17 +1433,9 @@ export class TaskGraphDispatcher {
849
1433
  * as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
850
1434
  * graph sail past a failure it never anticipated.
851
1435
  */
852
- async computeHandledFailures(provider, parentTaskID, nodes, edges) {
1436
+ computeHandledFailures(failureSemantics, nodes, edges) {
853
1437
  const handled = new Set();
854
- // Cheap exit before touching the database: with no failures there is nothing to handle, and
855
- // this runs on every poll for every active graph.
856
- if (!nodes.some((n) => n.status === 'Failed'))
857
- return handled;
858
- const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
859
- if (!(await parent.Load(parentTaskID)))
860
- return handled;
861
- const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
862
- if (meta.failureSemantics !== 'edges')
1438
+ if (failureSemantics !== 'edges')
863
1439
  return handled;
864
1440
  for (const node of nodes) {
865
1441
  if (node.status !== 'Failed')
@@ -893,28 +1469,77 @@ export class TaskGraphDispatcher {
893
1469
  * user-visible noise and, for `reinvoke`, an unbounded agent-run loop. Given one of the two has
894
1470
  * to be chosen, the quiet failure is the safe one.
895
1471
  *
896
- * The marker is written with a compare-and-swap read-back, so two instances reconciling the same
897
- * completed graph produce one winner rather than two.
1472
+ * The marker is claimed with a real compare-and-swap (one guarded UPDATE, rowcount as verdict),
1473
+ * so two instances reconciling the same completed graph produce one winner rather than two.
898
1474
  */
899
- async deliverContinuation(provider, parent, graph) {
1475
+ async deliverContinuation(provider, parent, graph, submitterCancelled) {
900
1476
  const meta = this.readParentMetadata(parent);
901
1477
  if (meta.continuationDeliveredAt)
902
- return;
903
- // At the cap, downgrade rather than refuse: the results still reach the user, the chain just
1478
+ return true;
1479
+ // Nobody is waiting: the run that submitted this graph was cancelled. Claim the marker so
1480
+ // nothing re-offers the graph, and record WHY nothing was announced — "we chose not to" and
1481
+ // "we found it too late" are different facts about a settlement, and a reader afterwards
1482
+ // should be able to tell them apart.
1483
+ if (submitterCancelled) {
1484
+ if (await this.claimContinuation(provider, parent.ID, 'cancelled')) {
1485
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} settled, but the run that submitted it was ` +
1486
+ `cancelled — no message posted and no reinvoke started.`);
1487
+ }
1488
+ return true;
1489
+ }
1490
+ // At the cap, DOWNGRADE rather than refuse: the results still reach the user, the chain just
904
1491
  // stops growing. Refusing outright would lose the outcome of work that actually completed.
905
- const mode = IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
1492
+ //
1493
+ // But a downgrade only applies to something that was going to be delivered (C2). Mapping the
1494
+ // cap straight onto `'message'` also promoted `continuation: 'none'` — a graph that asked
1495
+ // for silence — into a message nobody requested. Latent today because `Submit` refuses to
1496
+ // create a graph past the cap, and exactly the kind of latent that stops being latent the
1497
+ // moment a producer bypasses that check.
1498
+ const mode = meta.continuation !== 'none' && IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
906
1499
  if (mode !== 'none' && IsReinvokeCapReached(meta) && meta.continuation === 'reinvoke') {
907
1500
  LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} hit the reinvoke cap (${MAX_REINVOKE_DEPTH}); ` +
908
1501
  `delivering results as a message instead of starting another turn.`);
909
1502
  }
910
- if (!(await this.claimContinuation(provider, parent.ID, meta)))
911
- return;
1503
+ // A SETTLEMENT NOBODY IS WAITING FOR STILL SETTLES — it just does not get announced.
1504
+ //
1505
+ // Run settlement and cost rollup are status corrections and are always safe to apply late; a
1506
+ // run left `Paused` forever is strictly worse than a stale notification skipped. A stale
1507
+ // NOTIFICATION is not: posting a day-old "your workflow finished" into a live conversation,
1508
+ // or worse starting a fresh billed agent turn for it, is the outcome the age-out exists to
1509
+ // avoid. So an aged-out settlement claims the marker as `expired` and logs, which both
1510
+ // records what happened and stops any later pass delivering it. Second rung on the ladder
1511
+ // the reinvoke cap already established.
1512
+ const expired = IsSettlementExpired(parent.CompletedAt, new Date());
1513
+ // ONLY AN INSTANCE THAT CAN DELIVER MAY CLAIM THE RIGHT TO (R2-6).
1514
+ //
1515
+ // The claim ran before the deliverer check, so an instance constructed WITHOUT one — a
1516
+ // worker tier, an integration bundle, a second dev session — could observe the settlement
1517
+ // first, win the CAS, mark the graph `delivered`, and discard the message or reinvoke a
1518
+ // capable peer would have made moments later. Permanently, decided by poll timing.
1519
+ //
1520
+ // Declining leaves the marker unset, so the rescue sweep keeps offering the graph until an
1521
+ // instance that can deliver takes it. Run settlement and cost rollup have already happened
1522
+ // above and are not held up by this — what is deferred is the announcement, which is the only
1523
+ // part this instance genuinely cannot do.
1524
+ //
1525
+ // `expired` is exempt: recording "too old to deliver" requires no deliverer, and a graph past
1526
+ // its window has nothing left for a capable peer to do.
1527
+ if (!expired && mode !== 'none' && !this.continuationDeliverer) {
1528
+ this.reportUndeliverableOnce(parent.ID);
1529
+ return false;
1530
+ }
1531
+ if (!(await this.claimContinuation(provider, parent.ID, expired ? 'expired' : 'delivered')))
1532
+ return true;
1533
+ if (expired) {
1534
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} settled after its delivery window ` +
1535
+ `(${UNSETTLED_SWEEP_WINDOW_HOURS}h); the run and its cost were corrected, but the ` +
1536
+ `continuation was NOT delivered. Marked expired.`);
1537
+ return true;
1538
+ }
912
1539
  if (mode === 'none')
913
- return;
1540
+ return true;
914
1541
  const summary = this.buildContinuationSummary(parent, graph);
915
1542
  LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} finished — ${summary}`);
916
- if (!this.continuationDeliverer)
917
- return;
918
1543
  const params = {
919
1544
  ParentTaskID: parent.ID,
920
1545
  WorkflowName: parent.Name,
@@ -951,6 +1576,12 @@ export class TaskGraphDispatcher {
951
1576
  // on the marker: a missed notification visible in the record beats one repeated forever.
952
1577
  LogError(`[TaskGraphDispatcher] Continuation delivery failed for ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
953
1578
  }
1579
+ // C1: the function is declared `Promise<boolean>` and fell off the end here, so EVERY
1580
+ // successfully delivered graph resolved `undefined` — read as "not resolved" by the caller,
1581
+ // which then re-queued it and paid another full settle pass (run-tree cost query included)
1582
+ // and polluted R2-6's retry accounting. No double delivery, because the CAS holds; just a
1583
+ // wasted pass per settlement and a retry counter measuring the wrong thing.
1584
+ return true;
954
1585
  }
955
1586
  /** Reads the parent's durable continuation metadata through the shared parser. */
956
1587
  readParentMetadata(parent) {
@@ -962,19 +1593,26 @@ export class TaskGraphDispatcher {
962
1593
  * `MJ: Tasks` stays user-writable (D20), so a plain "read, decide, write" is not enough — the
963
1594
  * read-back is what makes a lost race observable instead of producing a duplicate delivery.
964
1595
  */
965
- async claimContinuation(provider, parentID, meta) {
966
- const row = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
967
- if (!(await row.Load(parentID)))
968
- return false;
969
- const current = this.readParentMetadata(row);
970
- if (current.continuationDeliveredAt)
971
- return false; // a peer got there first
972
- row.InputPayload = JSON.stringify({ ...meta, continuationDeliveredAt: new Date().toISOString() });
973
- if (!(await row.Save())) {
974
- LogError(`[TaskGraphDispatcher] Could not mark continuation delivered for ${parentID}; skipping to avoid a duplicate.`);
1596
+ /**
1597
+ * True when this graph finished so long ago that announcing it would surprise rather than inform.
1598
+ *
1599
+ * Measured from the parent's completion, not from when we noticed: the point is how stale the
1600
+ * NEWS is to whoever would receive it.
1601
+ */
1602
+ async claimContinuation(provider, parentID, deliveredAs = 'delivered') {
1603
+ // ONE GUARDED STATEMENT — see TaskClaimStore.TryClaimContinuation.
1604
+ //
1605
+ // This was Load → check the marker → `Save()`: an unconditional last-write-wins UPDATE that
1606
+ // two dispatchers could both pass. The comments here and at the call site called it a
1607
+ // compare-and-swap read-back; it was read-check-write, and for `continuation: 'reinvoke'`
1608
+ // losing that race means two fresh agent turns billed for one settlement, each able to
1609
+ // submit further graphs.
1610
+ // The type discriminator is part of the guard, not a caller-side filter — see
1611
+ // TryClaimContinuation. Nothing to claim if the type does not exist: no graph was submitted.
1612
+ const typeID = await this.workflowTaskTypeID(provider);
1613
+ if (!typeID)
975
1614
  return false;
976
- }
977
- return true;
1615
+ return this.claims.TryClaimContinuation(provider, parentID, deliveredAs, typeID, this.contextUser);
978
1616
  }
979
1617
  /** One line describing how the graph ended, for the completion log and message delivery. */
980
1618
  buildContinuationSummary(parent, graph) {
@@ -1016,11 +1654,11 @@ export class TaskGraphDispatcher {
1016
1654
  // Nothing will change on a retry. Mark it so the loop stops, and leave the task Pending
1017
1655
  // and visible — a person can still see it in the Tasks UI, which is the fallback the
1018
1656
  // notification was only ever an accelerant for.
1019
- await this.markHumanTaskNotified(task);
1657
+ await this.markHumanTaskNotified(task, provider);
1020
1658
  return;
1021
1659
  }
1022
1660
  if (!task.UserID) {
1023
- await this.markHumanTaskNotified(task);
1661
+ await this.markHumanTaskNotified(task, provider);
1024
1662
  return;
1025
1663
  }
1026
1664
  try {
@@ -1036,7 +1674,7 @@ export class TaskGraphDispatcher {
1036
1674
  catch (e) {
1037
1675
  LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
1038
1676
  }
1039
- await this.markHumanTaskNotified(task);
1677
+ await this.markHumanTaskNotified(task, provider);
1040
1678
  // Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
1041
1679
  // than appearing to stall for no reason.
1042
1680
  this.emit({
@@ -1060,8 +1698,88 @@ export class TaskGraphDispatcher {
1060
1698
  * invalidate. Left cached, a completed task keeps reading as `In Progress` and the graph never
1061
1699
  * rolls up: submitted work simply never settles.
1062
1700
  */
1063
- async findActiveGraphIDs(provider) {
1701
+ /**
1702
+ * Settles graphs that reached terminal without completing their post-settlement sequence.
1703
+ *
1704
+ * Runs the ordinary propagation path, which is safe to re-enter by construction: the terminal
1705
+ * write is guarded on not-already-terminal, the cost rollup assigns rather than accumulates, run
1706
+ * settlement is guarded on `Paused`, and delivery is guarded by the continuation CAS. A revisit
1707
+ * therefore corrects whatever is missing and does nothing where nothing is.
1708
+ *
1709
+ * @param windowHours how far back to look — wide once at startup, narrow in steady state
1710
+ */
1711
+ async sweepUnsettledGraphs(windowHours) {
1712
+ try {
1713
+ const provider = await this.providerFactory.CreateProvider();
1714
+ const ids = await this.findActiveGraphIDs(provider, windowHours);
1715
+ if (ids.length === 0)
1716
+ return;
1717
+ LogStatus(`[TaskGraphDispatcher] Startup sweep: reviewing ${ids.length} graph(s), including any that reached terminal without settling.`);
1718
+ await this.propagateAndRollup(provider, ids);
1719
+ }
1720
+ catch (e) {
1721
+ LogError(`[TaskGraphDispatcher] Unsettled-graph sweep failed: ${e instanceof Error ? e.message : String(e)}`);
1722
+ }
1723
+ }
1724
+ /**
1725
+ * The `AI Workflow` task type, resolved once per process.
1726
+ *
1727
+ * `MJ: Tasks` is a GENERAL-PURPOSE entity — conversations and user to-dos live there too — so an
1728
+ * unscoped sweep treats every root task hierarchy as a workflow: rolling up and overwriting the
1729
+ * status of somebody's to-do list, raising agent requests against plain tasks, and (once the
1730
+ * continuation CAS exists) injecting marker keys into a user's own `InputPayload`.
1731
+ *
1732
+ * `Submit` has always stamped this type on the parent and every child (`ensureTaskType`, which
1733
+ * runs before the persist transaction), so the discriminator D3 called for already exists on
1734
+ * every dispatcher-owned row. Verified against the live database: every parent graph carries it.
1735
+ *
1736
+ * Null when the type row does not exist yet — no graph has ever been submitted — in which case
1737
+ * there is nothing for the dispatcher to find and the sweep returns empty rather than unscoped.
1738
+ *
1739
+ * **A miss is never cached**, and that is not a micro-optimisation. `TaskGraphService.Submit`
1740
+ * creates the row on first use, so on a fresh install the ordinary sequence is: dispatcher
1741
+ * starts, looks, finds nothing — then somebody submits the first workflow. Caching that first
1742
+ * `null` would blind this process to every graph until it was restarted, with each poll reporting
1743
+ * a clean, empty sweep. The row is created once and never removed, so the retry costs one
1744
+ * `MaxRows: 1` lookup per poll for exactly as long as there is genuinely nothing to dispatch.
1745
+ */
1746
+ async workflowTaskTypeID(provider) {
1747
+ if (this.cachedWorkflowTaskTypeID)
1748
+ return this.cachedWorkflowTaskTypeID;
1749
+ const result = await RunView.FromMetadataProvider(provider).RunView({
1750
+ EntityName: 'MJ: Task Types',
1751
+ ExtraFilter: `Name='${TASK_TYPE_NAME}'`,
1752
+ Fields: ['ID'],
1753
+ // Ordered, and reading two (R2-7). An unordered `MaxRows: 1` against two rows sharing
1754
+ // the name lets this instance bind a different ID than `Submit` did — after which
1755
+ // every graph the other stamped is invisible to all three sweep arms here.
1756
+ OrderBy: '__mj_CreatedAt ASC, ID ASC',
1757
+ ResultType: 'simple',
1758
+ MaxRows: 2,
1759
+ }, this.contextUser);
1760
+ if (!result.Success) {
1761
+ // A failed lookup is not "no such type" — saying so would silently skip a poll cycle's
1762
+ // worth of real work. Report it, and let the next cycle ask again.
1763
+ LogError(`[TaskGraphDispatcher] Could not resolve the '${TASK_TYPE_NAME}' task type: ${result.ErrorMessage}`);
1764
+ return null;
1765
+ }
1766
+ const rows = result.Results ?? [];
1767
+ if (rows.length > 1) {
1768
+ LogError(`[TaskGraphDispatcher] More than one '${TASK_TYPE_NAME}' task type exists. Binding the ` +
1769
+ `oldest (${rows[0].ID}); any graph stamped with the other is invisible to this sweep and ` +
1770
+ `will never settle. Merge them.`);
1771
+ }
1772
+ this.cachedWorkflowTaskTypeID = rows[0]?.ID ?? null;
1773
+ return this.cachedWorkflowTaskTypeID;
1774
+ }
1775
+ async findActiveGraphIDs(provider, windowHours = UNSETTLED_SWEEP_WINDOW_HOURS) {
1064
1776
  const rv = RunView.FromMetadataProvider(provider);
1777
+ // EVERY arm is scoped to workflow graphs. Unscoped, the dispatcher rewrites tasks that are
1778
+ // none of its business — see workflowTaskTypeID.
1779
+ const typeID = await this.workflowTaskTypeID(provider);
1780
+ if (!typeID)
1781
+ return [];
1782
+ const ofWorkflowType = `TypeID='${typeID}'`;
1065
1783
  // TWO queries, because "has work left to do" and "needs attention" are not the same set.
1066
1784
  //
1067
1785
  // Selecting only graphs with non-terminal CHILDREN looks right and is subtly fatal: the
@@ -1073,23 +1791,50 @@ export class TaskGraphDispatcher {
1073
1791
  //
1074
1792
  // The second query closes it: a parent that is itself non-terminal still needs looking at,
1075
1793
  // whatever its children are doing.
1076
- const [withPendingWork, unsettledParents] = await rv.RunViews([
1794
+ // THREE queries. The third rescues a graph that reached terminal without settling.
1795
+ //
1796
+ // The post-settlement sequence — cost rollup, run settlement, continuation delivery — runs
1797
+ // AFTER the parent's terminal write, and a terminal parent with all-terminal children
1798
+ // matches neither query above. So a process that died in that window left the submitting
1799
+ // agent run `Paused` FOREVER: no rollup, no notification, and nothing that would ever look
1800
+ // again. The metadata's own doc comment promised "the next sweep retries"; that sweep did
1801
+ // not exist.
1802
+ //
1803
+ // Bounded rather than unbounded, because the marker lives in `InputPayload` JSON and cannot
1804
+ // be filtered in SQL: the window is what keeps this a targeted rescue instead of a re-parse
1805
+ // of every graph ever run. `__mj_UpdatedAt` advances on each settle attempt, so a graph
1806
+ // being actively retried stays in the window — the bound is on ABANDONMENT, not on age.
1807
+ const cutoff = SweepCutoff(new Date(), windowHours);
1808
+ const [withPendingWork, unsettledParents, terminalRecent] = await rv.RunViews([
1077
1809
  {
1078
1810
  EntityName: 'MJ: Tasks',
1079
- ExtraFilter: `ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
1811
+ ExtraFilter: `${ofWorkflowType} AND ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
1080
1812
  Fields: ['ParentID'],
1081
1813
  ResultType: 'simple',
1082
1814
  BypassCache: true,
1083
1815
  },
1084
1816
  {
1085
1817
  EntityName: 'MJ: Tasks',
1086
- ExtraFilter: `ParentID IS NULL AND Status IN ('Pending','In Progress')`,
1818
+ ExtraFilter: `${ofWorkflowType} AND ParentID IS NULL AND Status IN ('Pending','In Progress')`,
1087
1819
  Fields: ['ID'],
1088
1820
  ResultType: 'simple',
1089
1821
  BypassCache: true,
1090
1822
  },
1823
+ {
1824
+ EntityName: 'MJ: Tasks',
1825
+ ExtraFilter: `${ofWorkflowType} AND ParentID IS NULL AND Status IN (${TERMINAL_PARENT_STATUS_SQL}) ` +
1826
+ `AND __mj_UpdatedAt >= '${cutoff}'`,
1827
+ Fields: ['ID', 'InputPayload'],
1828
+ ResultType: 'simple',
1829
+ BypassCache: true,
1830
+ },
1091
1831
  ], this.contextUser);
1092
1832
  const ids = new Set();
1833
+ // Graphs this instance is mid-retry on, whatever the window says (R2-12). A failing pass
1834
+ // writes nothing, so their `__mj_UpdatedAt` has stopped advancing and the third arm below
1835
+ // will eventually stop finding them — which would turn a retry into a silent abandonment.
1836
+ for (const id of this.retryingSettlement.keys())
1837
+ ids.add(id);
1093
1838
  for (const r of (withPendingWork?.Results ?? [])) {
1094
1839
  if (r.ParentID)
1095
1840
  ids.add(r.ParentID);
@@ -1100,6 +1845,11 @@ export class TaskGraphDispatcher {
1100
1845
  if (r.ID)
1101
1846
  ids.add(r.ID);
1102
1847
  }
1848
+ // The marker is JSON, so the filter is in TypeScript rather than in SQL — see
1849
+ // SelectUnsettledGraphIDs, which owns that decision and is tested directly.
1850
+ for (const id of SelectUnsettledGraphIDs((terminalRecent?.Results ?? []))) {
1851
+ ids.add(id);
1852
+ }
1103
1853
  return [...ids];
1104
1854
  }
1105
1855
  /**
@@ -1111,10 +1861,16 @@ export class TaskGraphDispatcher {
1111
1861
  */
1112
1862
  async findClaimableTasks(provider, limit) {
1113
1863
  const claimable = [];
1114
- for (const parentID of await this.findActiveGraphIDs(provider)) {
1864
+ const stats = new Map();
1865
+ const activeGraphs = await this.findActiveGraphIDs(provider);
1866
+ // Usually a no-op — propagateAndRollup primed these earlier in the same pass.
1867
+ await this.primeDebugStates(provider, activeGraphs);
1868
+ for (const parentID of activeGraphs) {
1115
1869
  if (claimable.length >= limit)
1116
1870
  break;
1117
- const graph = await this.loadGraphState(provider, parentID);
1871
+ const debug = await this.readDebugState(provider, parentID);
1872
+ await this.announcePauseTransition(provider, parentID, debug);
1873
+ const graph = await this.loadGraphState(provider, parentID, debug);
1118
1874
  // HOLD is what makes "a broken condition stalls visibly" true rather than merely stated.
1119
1875
  // An undecided exclusive group keeps all its edges, and a kept edge on a Complete origin
1120
1876
  // is a SATISFIED prerequisite — so without this filter every branch of the fork would be
@@ -1134,9 +1890,71 @@ export class TaskGraphDispatcher {
1134
1890
  // window before it was marked.
1135
1891
  const eligible = ComputeEligibleTasks(graph.nodes, graph.edges, graph.handledFailureIDs)
1136
1892
  .filter((n) => !graph.holdTaskIDs.has(n.id) &&
1893
+ // CONFIRMED seeds, not raw ones (P1). Holding a decided loser out of claiming
1894
+ // closes a real race — the loser could be claimed between eligibility and the
1895
+ // skip write — and that role is unchanged. What changed is which targets count
1896
+ // as decided: a task another live route still reaches was never a loser, so it
1897
+ // must stay claimable and run when its own prerequisites are met.
1137
1898
  !graph.skipSeedTaskIDs.has(n.id) &&
1138
- !graph.unreachableTaskIDs.has(n.id));
1899
+ !graph.unreachableTaskIDs.has(n.id) &&
1900
+ // ...and everything the cascade is about to reach (R2-14). A descendant of a
1901
+ // seed is eligible for the moments between its ancestor's skip landing and its
1902
+ // own, because Skipped satisfies prerequisites — a window another instance can
1903
+ // and does claim inside.
1904
+ !graph.cascadeSkipTaskIDs.has(n.id));
1905
+ stats.set(parentID, { eligible: eligible.length, held: graph.holdTaskIDs.size });
1906
+ // THE DEBUG GATE — pause, single-step, breakpoints — decided by the pure function, with
1907
+ // the CAS writes staying here. Every control is a gate on CLAIMING: a claimed task can
1908
+ // never be interrupted mid-flight anyway, so "paused" means nothing new starts while
1909
+ // in-flight work finishes and its completions land. That is also why the gate sits
1910
+ // BEFORE the runner checks and the human notification below: pausing a graph must not
1911
+ // keep notifying assignees — a notification is starting something.
1912
+ const gate = DecideClaimGate(debug, eligible.map((n) => n.id));
1913
+ if (gate.mode === 'closed')
1914
+ continue;
1915
+ let allowedTaskIDs = null;
1916
+ if (gate.mode === 'breakpoint') {
1917
+ const typeID = await this.workflowTaskTypeID(provider);
1918
+ // The pause is a CAS so two instances arriving at the same breakpoint in the same
1919
+ // interval produce one announcement — the loser simply sees a paused graph next pass.
1920
+ if (typeID && await this.claims.TryPauseAtBreakpoint(provider, parentID, gate.taskID, typeID, this.contextUser)) {
1921
+ const owner = await this.resolveOwner(provider, parentID);
1922
+ const name = graph.entityById.get(gate.taskID)?.Name;
1923
+ LogStatus(`[TaskGraphDispatcher] Graph ${parentID} paused at breakpoint on '${name}' (${gate.taskID}).`);
1924
+ this.emit({ Kind: 'BreakpointHit', ParentTaskID: parentID, OwnerUserID: owner, TaskID: gate.taskID, TaskName: name });
1925
+ this.emit({ Kind: 'GraphPaused', ParentTaskID: parentID, OwnerUserID: owner, TaskID: gate.taskID, Reason: 'breakpoint' });
1926
+ this.announcedPaused.set(parentID, true);
1927
+ }
1928
+ continue;
1929
+ }
1930
+ if (gate.mode === 'step') {
1931
+ allowedTaskIDs = new Set(gate.taskIDs);
1932
+ // THE ALLOWANCE IS CONSUMED ONLY IF SOMETHING WILL ACTUALLY MOVE.
1933
+ //
1934
+ // "Eligible" is a graph-shape answer; whether this host can run the step is a
1935
+ // separate one, decided below by the runner checks. Consuming first meant a step
1936
+ // onto a node this instance has no runner for — or one already in flight — spent
1937
+ // the allowance and released nothing, leaving the operator pressing a button that
1938
+ // did nothing and no reason anywhere. Deciding first costs one pre-pass over a set
1939
+ // that is at most the frontier.
1940
+ const releasable = eligible.filter((n) => {
1941
+ const entity = graph.entityById.get(n.id);
1942
+ return entity ? allowedTaskIDs.has(n.id) && this.canActOn(entity) : false;
1943
+ });
1944
+ if (releasable.length === 0) {
1945
+ await this.reportStepReleasedNothing(provider, parentID, gate.taskIDs, graph);
1946
+ continue;
1947
+ }
1948
+ const typeID = await this.workflowTaskTypeID(provider);
1949
+ // Consuming the allowance is the race: exactly one instance clears the marker and
1950
+ // releases work; the loser waits for the next allowance. A lost consume is normal.
1951
+ if (!typeID || !(await this.claims.TryConsumeStepMarker(provider, parentID, typeID, this.contextUser))) {
1952
+ continue;
1953
+ }
1954
+ }
1139
1955
  for (const node of eligible) {
1956
+ if (allowedTaskIDs && !allowedTaskIDs.has(node.id))
1957
+ continue;
1140
1958
  const entity = graph.entityById.get(node.id);
1141
1959
  if (!entity)
1142
1960
  continue;
@@ -1165,6 +1983,17 @@ export class TaskGraphDispatcher {
1165
1983
  continue;
1166
1984
  }
1167
1985
  else if (!entity.AgentID) {
1986
+ // No runner column at all — a person completes this one. Asked through the same
1987
+ // predicate the human settle/expiry sweeps use, so a task that gets NOTIFIED here
1988
+ // is a task those sweeps can later see; the two disagreeing is how a human task
1989
+ // ends up asked and then never settled.
1990
+ if (!IsHumanTask(entity)) {
1991
+ // Neither a runner nor a person: nothing can ever move this. Loud, because
1992
+ // the alternative is a graph that waits forever on nobody.
1993
+ LogError(`[TaskGraphDispatcher] Task '${entity.Name}' (${entity.ID}) has no runner ` +
1994
+ `assignment and is not a human step — nothing can execute it. The graph will stall.`);
1995
+ continue;
1996
+ }
1168
1997
  await this.notifyHumanTaskReady(entity, provider);
1169
1998
  continue;
1170
1999
  }
@@ -1174,8 +2003,82 @@ export class TaskGraphDispatcher {
1174
2003
  if (claimable.length >= limit)
1175
2004
  break;
1176
2005
  }
2006
+ if (debug.skipBreakpointTaskID) {
2007
+ const skip = debug.skipBreakpointTaskID;
2008
+ const claimedThisPass = claimable.some((t) => UUIDsEqual(t.ID, skip));
2009
+ const stillEligible = eligible.some((n) => UUIDsEqual(n.id, skip));
2010
+ if (claimedThisPass || !stillEligible) {
2011
+ await this.clearSkipBreakpoint(provider, parentID);
2012
+ }
2013
+ }
1177
2014
  }
1178
- return claimable;
2015
+ return { tasks: claimable, stats };
2016
+ }
2017
+ async clearSkipBreakpoint(provider, parentTaskID) {
2018
+ const typeID = await this.workflowTaskTypeID(provider);
2019
+ if (!typeID)
2020
+ return;
2021
+ await this.claims.TryWriteDebugFields(provider, parentTaskID, [TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' })], typeID, this.contextUser);
2022
+ }
2023
+ /**
2024
+ * Drops the per-graph frame-dedup state for a graph that has settled.
2025
+ *
2026
+ * `ownerByParentID` is deliberately NOT purged here: it is the delivery key for the
2027
+ * `GraphSettled` frame emitted moments earlier and for any rescue-sweep pass that revisits the
2028
+ * graph, it is one small string per graph, and ownership never changes — the cost of keeping it
2029
+ * is bounded and the cost of losing it is a re-query on a path that is meant to be cheap.
2030
+ */
2031
+ forgetGraphObservability(parentTaskID) {
2032
+ this.emittedGateVerdicts.delete(parentTaskID);
2033
+ this.nodeProgressLastEmit.delete(parentTaskID);
2034
+ this.announcedPaused.delete(parentTaskID);
2035
+ this.debugStateByGraph.delete(parentTaskID);
2036
+ }
2037
+ /**
2038
+ * Whether THIS host can act on a task right now — the runner-availability question, asked
2039
+ * without acting on it.
2040
+ *
2041
+ * Mirrors the checks in the claim loop so a step allowance is spent only when something will
2042
+ * actually move. A human step counts as actionable: stepping onto one legitimately produces a
2043
+ * notification rather than a claim.
2044
+ */
2045
+ canActOn(entity) {
2046
+ if (this.inFlight.has(entity.ID))
2047
+ return false;
2048
+ if (entity.ActionID)
2049
+ return !!this.actionRunner;
2050
+ if (entity.PromptID)
2051
+ return !!this.promptRunner;
2052
+ if (entity.AgentID)
2053
+ return true;
2054
+ return IsHumanTask(entity);
2055
+ }
2056
+ /**
2057
+ * Says why a step press released nothing, instead of leaving the allowance spent and the
2058
+ * console silent.
2059
+ *
2060
+ * The allowance is deliberately NOT consumed on this path — the operator's intent stands, and
2061
+ * the step will release as soon as the named work becomes actionable (a runner arrives, an
2062
+ * in-flight task finishes). Announced once per pass rather than logged only, because the person
2063
+ * waiting is looking at the console, not the server log.
2064
+ */
2065
+ async reportStepReleasedNothing(provider, parentTaskID, requestedTaskIDs, graph) {
2066
+ const names = requestedTaskIDs
2067
+ .map((id) => graph.entityById.get(id)?.Name)
2068
+ .filter((n) => !!n);
2069
+ const subject = names.length > 0 ? `"${names.join('", "')}"` : 'the next step';
2070
+ const reason = `Step is still waiting: ${subject} cannot start on this server yet — it is already ` +
2071
+ `running, or no runner for that step type is loaded here. The step will release as ` +
2072
+ `soon as it can; nothing was lost.`;
2073
+ LogStatus(`[TaskGraphDispatcher] Step on graph ${parentTaskID} released nothing: ${reason}`);
2074
+ this.emit({
2075
+ Kind: 'StepRefused',
2076
+ ParentTaskID: parentTaskID,
2077
+ OwnerUserID: await this.resolveOwner(provider, parentTaskID),
2078
+ TaskID: requestedTaskIDs[0],
2079
+ TaskName: names[0],
2080
+ Reason: reason,
2081
+ });
1179
2082
  }
1180
2083
  /**
1181
2084
  * Marks a human task as notified, so the request is raised exactly once.
@@ -1184,11 +2087,16 @@ export class TaskGraphDispatcher {
1184
2087
  * notification: the task stays visible in the inbox either way, whereas a notification storm is
1185
2088
  * not self-correcting.
1186
2089
  */
1187
- async markHumanTaskNotified(task) {
1188
- task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
1189
- if (!(await task.Save())) {
1190
- LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
2090
+ async markHumanTaskNotified(task, provider) {
2091
+ // Guarded, not a full-row save against a snapshot (R3-5). This row was loaded at the top of
2092
+ // the pass; a full-row write could revert a status it has reached since, and two instances
2093
+ // could both stamp it after both having seen it absent. The predicate makes it once-only.
2094
+ if (await this.claims.TryMarkHumanNotified(provider, task.ID, HUMAN_TASK_NOTIFIED_MARKER, this.contextUser)) {
2095
+ task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
2096
+ return;
1191
2097
  }
2098
+ // Rowcount 0 is ordinary: another instance marked it, or the task is no longer Pending.
2099
+ // Either way this instance has nothing left to do about the notification.
1192
2100
  }
1193
2101
  /**
1194
2102
  * Raises the `MJ: AI Agent Requests` row a person answers to release this step.
@@ -1205,9 +2113,13 @@ export class TaskGraphDispatcher {
1205
2113
  */
1206
2114
  async raiseHumanRequest(task, provider) {
1207
2115
  try {
1208
- const existing = await this.findOpenRequest(provider, task.ID);
1209
- if (existing)
1210
- return 'raised'; // already waiting on someone
2116
+ const existing = await this.findOpenRequests(provider, task.ID);
2117
+ if (existing.length > 0) {
2118
+ // Somebody IS waiting on this task — but "somebody" may be two rows, so collapse
2119
+ // before returning. Free: the rows are already in hand.
2120
+ await this.withdrawDuplicateRequests(provider, task.ID, existing, existing[0].ID);
2121
+ return 'raised';
2122
+ }
1211
2123
  const request = await provider.GetEntityObject('MJ: AI Agent Requests', this.contextUser);
1212
2124
  request.NewRecord();
1213
2125
  request.OriginatingTaskID = task.ID;
@@ -1240,7 +2152,24 @@ export class TaskGraphDispatcher {
1240
2152
  if (expiresInHours && expiresInHours > 0) {
1241
2153
  request.ExpiresAt = new Date(Date.now() + expiresInHours * 60 * 60 * 1000);
1242
2154
  }
1243
- if (!(await request.Save())) {
2155
+ if (await request.Save()) {
2156
+ // INSERT-THEN-RESELECT (R3-5). The check above is read-then-write in a system whose
2157
+ // every other cross-instance write is a CAS, and there is no unique index behind it —
2158
+ // so two overlapping instances both read "none open", both insert, and both ping the
2159
+ // assignee. When one is answered, `settleAnsweredHumanTasks` settles from the single
2160
+ // latest terminal request and NOTHING ever touches the other: the withdrawal paths
2161
+ // fire only on skips and cancels, and both human sweeps scope to `Pending` tasks,
2162
+ // which the settled task no longer is. The duplicate becomes a durable, unanswerable,
2163
+ // immortal inbox item — the zombie class R2-10 removed from the skip paths.
2164
+ //
2165
+ // Checking harder is what created the window, so the resolution is to let both
2166
+ // inserts happen and then agree on a winner: the oldest open row. A loser withdraws
2167
+ // its own row and returns `raised` — somebody IS waiting on this task, which is what
2168
+ // the caller needs to know.
2169
+ await this.withdrawDuplicateRequests(provider, task.ID, await this.findOpenRequests(provider, task.ID), request.ID);
2170
+ return 'raised';
2171
+ }
2172
+ {
1244
2173
  LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ` +
1245
2174
  `${request.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1246
2175
  // A failed SAVE may be transient (deadlock, contention), so this one earns a retry.
@@ -1281,15 +2210,25 @@ export class TaskGraphDispatcher {
1281
2210
  return null;
1282
2211
  }
1283
2212
  }
1284
- /** The still-open request for a task, if one exists. */
1285
- async findOpenRequest(provider, taskID) {
2213
+ /**
2214
+ * Every still-open request for a task, oldest first.
2215
+ *
2216
+ * Plural, and ordered, for one reason each. Ordered, because the oldest row is the one every
2217
+ * instance must agree is "the" request — it is the one the assignee most likely already saw,
2218
+ * and the one `withdrawDuplicateRequests` keeps; unordered, two instances could each decide a
2219
+ * different duplicate was the keeper and withdraw each other's. Plural, because a caller that
2220
+ * only ever sees the first cannot notice there are two, which is how the duplicate below
2221
+ * survived: every reader of this took `[0]` and moved on.
2222
+ */
2223
+ async findOpenRequests(provider, taskID) {
1286
2224
  const result = await RunView.FromMetadataProvider(provider).RunView({
1287
2225
  EntityName: 'MJ: AI Agent Requests',
1288
2226
  ExtraFilter: `OriginatingTaskID='${taskID}' AND Status='Requested'`,
2227
+ OrderBy: '__mj_CreatedAt ASC, ID ASC',
1289
2228
  ResultType: 'entity_object',
1290
2229
  BypassCache: true,
1291
2230
  }, this.contextUser);
1292
- return (result.Success ? result.Results?.[0] : null) ?? null;
2231
+ return (result.Success ? result.Results : null) ?? [];
1293
2232
  }
1294
2233
  /**
1295
2234
  * Settles a human task from the request a person answered.
@@ -1306,7 +2245,7 @@ export class TaskGraphDispatcher {
1306
2245
  async settleAnsweredHumanTasks(provider, graphID) {
1307
2246
  const waiting = await RunView.FromMetadataProvider(provider).RunView({
1308
2247
  EntityName: 'MJ: Tasks',
1309
- ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
2248
+ ExtraFilter: `ParentID='${graphID}' AND ${HumanTaskSQL()} AND Status='Pending'`,
1310
2249
  ResultType: 'entity_object',
1311
2250
  BypassCache: true,
1312
2251
  }, this.contextUser);
@@ -1336,11 +2275,27 @@ export class TaskGraphDispatcher {
1336
2275
  if (!(await task.Save())) {
1337
2276
  LogError(`[TaskGraphDispatcher] Could not settle human task ${task.ID}: ` +
1338
2277
  `${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
2278
+ continue;
1339
2279
  }
2280
+ // WITHDRAW EVERY OTHER OPEN ASK FOR THIS STEP (R3-5).
2281
+ //
2282
+ // The step is terminal now, so both human sweeps — which scope to `Pending` tasks — will
2283
+ // never look at it again, and any request still `Requested` is un-answerable and
2284
+ // immortal: the assignee keeps seeing "a workflow is waiting on you" for a step that is
2285
+ // finished. Duplicates only arose from the raise race fixed above, but this also
2286
+ // retroactively cleans the ones already minted, which the raise-side fix cannot reach.
2287
+ await this.withdrawOpenRequests(provider, [task.ID], 'This step has been settled; the request is no longer open.');
1340
2288
  }
1341
2289
  }
1342
2290
  /**
1343
- * Re-opens a human step whose request was CANCELLED.
2291
+ * Reconciles the requests behind human steps that are waiting on somebody.
2292
+ *
2293
+ * Two things can be wrong with a waiting step, and both are silent. It can have NO open request
2294
+ * — the cancel case below — or it can have MORE than one, which the raise cannot fix because it
2295
+ * never runs again for a notified task. Both are corrected here, on the only sweep that visits
2296
+ * these tasks every pass.
2297
+ *
2298
+ * **Re-opening a human step whose request was CANCELLED.**
1344
2299
  *
1345
2300
  * `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
1346
2301
  * rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
@@ -1353,7 +2308,7 @@ export class TaskGraphDispatcher {
1353
2308
  * and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
1354
2309
  * human action: it takes another person cancelling again to come back here.
1355
2310
  */
1356
- async reopenCancelledHumanTasks(provider, graphID) {
2311
+ async reconcileWaitingHumanTasks(provider, graphID) {
1357
2312
  const waiting = await RunView.FromMetadataProvider(provider).RunView({
1358
2313
  EntityName: 'MJ: Tasks',
1359
2314
  // `StepType` is NULLABLE, and rows predating the column exist (4 in the reference
@@ -1364,7 +2319,7 @@ export class TaskGraphDispatcher {
1364
2319
  // this to tasks the dispatcher raised a request for, so the widening cannot pull in
1365
2320
  // unrelated work.
1366
2321
  ExtraFilter: `ParentID='${graphID}' AND Status='Pending' ` +
1367
- `AND (StepType='Human' OR (StepType IS NULL AND UserID IS NOT NULL)) ` +
2322
+ `AND ${HumanTaskSQL()} ` +
1368
2323
  `AND ClaimedBy='${HUMAN_TASK_NOTIFIED_MARKER}'`,
1369
2324
  ResultType: 'entity_object',
1370
2325
  BypassCache: true,
@@ -1375,8 +2330,13 @@ export class TaskGraphDispatcher {
1375
2330
  // Only when there is nothing live AND nothing terminal. A task with an open request is
1376
2331
  // simply waiting; one with a terminal request is settled on the next pass by
1377
2332
  // settleAnsweredHumanTasks, and re-raising either would ask the same question twice.
1378
- if (await this.findOpenRequest(provider, task.ID))
2333
+ const open = await this.findOpenRequests(provider, task.ID);
2334
+ if (open.length > 0) {
2335
+ // Waiting, correctly — but on however many asks happen to exist. Collapse them here
2336
+ // or nothing ever will: the raise is behind the notified marker for good.
2337
+ await this.withdrawDuplicateRequests(provider, task.ID, open, open[0].ID);
1379
2338
  continue;
2339
+ }
1380
2340
  if (await this.answeredRequestFor(provider, task.ID))
1381
2341
  continue;
1382
2342
  LogStatus(`[TaskGraphDispatcher] The request for '${task.Name}' was cancelled and nothing ` +
@@ -1414,7 +2374,7 @@ export class TaskGraphDispatcher {
1414
2374
  const humanTasks = await RunView.FromMetadataProvider(provider).RunView({
1415
2375
  EntityName: 'MJ: Tasks',
1416
2376
  Fields: ['ID'],
1417
- ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
2377
+ ExtraFilter: `ParentID='${graphID}' AND ${HumanTaskSQL()} AND Status='Pending'`,
1418
2378
  ResultType: 'simple',
1419
2379
  }, this.contextUser);
1420
2380
  const ids = (humanTasks.Results ?? []).map((r) => `'${r.ID}'`);
@@ -1450,7 +2410,7 @@ export class TaskGraphDispatcher {
1450
2410
  }
1451
2411
  }
1452
2412
  /** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
1453
- async loadGraphState(provider, parentTaskID) {
2413
+ async loadGraphState(provider, parentTaskID, debug) {
1454
2414
  const rv = RunView.FromMetadataProvider(provider);
1455
2415
  // BypassCache throughout: task status is written by the claim protocol's direct SQL, which
1456
2416
  // fires no cache invalidation. See findActiveGraphIDs.
@@ -1459,7 +2419,8 @@ export class TaskGraphDispatcher {
1459
2419
  if (children.length === 0) {
1460
2420
  return {
1461
2421
  nodes: [], edges: [], entityById: new Map(),
1462
- unreachableTaskIDs: new Set(), skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
2422
+ unreachableTaskIDs: new Set(), cascadeSkipTaskIDs: new Set(),
2423
+ skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
1463
2424
  handledFailureIDs: new Set(),
1464
2425
  };
1465
2426
  }
@@ -1467,6 +2428,19 @@ export class TaskGraphDispatcher {
1467
2428
  const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
1468
2429
  const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
1469
2430
  const entityById = new Map(children.map((c) => [c.ID, c]));
2431
+ // Read ONCE, and only when it can change an answer (R2-4). Both consumers below — which
2432
+ // origin statuses may decide an exclusive group, and which failures count as handled — are
2433
+ // no-ops unless something has actually failed, and this runs on every poll for every active
2434
+ // graph, so the parent load stays behind the same cheap exit `computeHandledFailures` used.
2435
+ // One parent read serves both questions this pass asks of the metadata bag.
2436
+ const parentMeta = await this.readParentMetadataFor(provider, parentTaskID);
2437
+ const failureSemantics = parentMeta.failureSemantics;
2438
+ // The invocation's own parameters, carried on the parent so a condition evaluated by any
2439
+ // instance sees what the walker saw (R3-3).
2440
+ const invocation = {
2441
+ Data: parentMeta.invocation?.data,
2442
+ Context: parentMeta.invocation?.context,
2443
+ };
1470
2444
  // Conditional edges are resolved HERE, before eligibility runs, by dropping edges whose
1471
2445
  // condition does not hold. Expressing it as edge removal rather than as a second rule inside
1472
2446
  // the eligibility algorithm is what keeps one definition of "ready": a task with no live
@@ -1491,6 +2465,12 @@ export class TaskGraphDispatcher {
1491
2465
  // only ResolveExclusiveGroups can decide.
1492
2466
  const exclusive = deps.filter((d) => !!d.ExclusiveGroup);
1493
2467
  const ordinary = deps.filter((d) => !d.ExclusiveGroup);
2468
+ // Named once and consumed twice — by the resolution below and by the `GateDecision` frame
2469
+ // emission further down. Two copies of this rule is how the console starts narrating
2470
+ // decisions the engine no longer makes.
2471
+ const decidingStatuses = failureSemantics === 'edges'
2472
+ ? new Set(['Complete', 'Failed'])
2473
+ : new Set(['Complete']);
1494
2474
  const resolution = ResolveExclusiveGroups(exclusive.map((d) => ({
1495
2475
  id: d.ID,
1496
2476
  taskId: d.TaskID,
@@ -1499,19 +2479,46 @@ export class TaskGraphDispatcher {
1499
2479
  originStatus: (entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending'),
1500
2480
  priority: d.Priority ?? 0,
1501
2481
  sequence: d.Sequence ?? 0,
1502
- conditionOutcome: this.evaluateExclusiveCondition(d, entityById),
2482
+ conditionOutcome: this.evaluateExclusiveCondition(d, entityById, invocation, debug),
1503
2483
  })),
1504
- // A flow's failure handling is its outgoing edges, so a Failed origin still decides its
1505
- // group. For a loop-agent graph the set is Complete-only and nothing changes.
1506
- new Set(['Complete', 'Failed']));
2484
+ // WHICH STATUSES MAY DECIDE — the graph's own failure dialect, not a constant.
2485
+ //
2486
+ // Under `'edges'`, a flow's failure handling IS its outgoing edges, so a Failed origin
2487
+ // decides its group and the drawn recovery path runs. Under `'block'` — the spec's
2488
+ // DEFAULT — a failure is terminal for everything downstream, and letting it decide was
2489
+ // silently catastrophic: the losers were removed and seeded, `ComputeSkipCascade`
2490
+ // confirmed them `Skipped`, `Skipped` satisfies dependents, and because the removed
2491
+ // loser edges also sever `ComputeTasksToBlock`'s forward walk, a join fed by an
2492
+ // independent healthy route EXECUTED downstream of an unhandled failure. The parent
2493
+ // still rolled up Failed, so the verdict looked right while the side effects had fired.
2494
+ //
2495
+ // The old comment claimed a loop-agent graph saw Complete-only. It did not; the same
2496
+ // hardcoded set was passed for every graph.
2497
+ decidingStatuses);
1507
2498
  const loserEdgeIDs = new Set(resolution.loserEdgeIDs);
2499
+ // Targets of an edge whose condition could not be evaluated (P2). Neither eligible nor
2500
+ // skipped: the edge stays live so the target is not mistaken for unreachable, and the target
2501
+ // joins the hold set so nothing claims it.
2502
+ const heldByCondition = new Set();
2503
+ // Decisions collected for `GateDecision` frames — announced after the state is assembled,
2504
+ // change-only, so a viewer learns WHY a branch ran (or is held) the moment it is decided.
2505
+ const gateDecisions = [];
1508
2506
  for (const d of ordinary) {
1509
2507
  if (d.Condition?.trim()) {
1510
- const outcome = this.evaluateEdgeCondition(d, entityById);
1511
- if (outcome === 'drop') {
2508
+ const decision = this.evaluateEdgeCondition(d, entityById, failureSemantics, invocation, debug);
2509
+ if (decision.decided) {
2510
+ gateDecisions.push({
2511
+ edge: d,
2512
+ verdict: decision.outcome === 'keep' ? 'satisfied' : decision.outcome === 'drop' ? 'notTaken' : 'held',
2513
+ reason: decision.reason,
2514
+ });
2515
+ }
2516
+ if (decision.outcome === 'drop') {
1512
2517
  droppedInto.add(d.TaskID);
1513
2518
  continue;
1514
2519
  }
2520
+ if (decision.outcome === 'hold')
2521
+ heldByCondition.add(d.TaskID);
1515
2522
  }
1516
2523
  stillReachable.add(d.TaskID);
1517
2524
  liveEdges.push({
@@ -1535,17 +2542,105 @@ export class TaskGraphDispatcher {
1535
2542
  // Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
1536
2543
  // waiting on it, and a node reached by an alternate branch is genuinely reachable.
1537
2544
  const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
2545
+ // EXCLUSIVE LOSERS GET THE SAME TEST — they did not, and that is P1.
2546
+ //
2547
+ // A loser's target was seeded and written `Skipped` unconditionally, with no "does another
2548
+ // live route reach it?" check. The shape that breaks: `A →(cond)→ Review → Publish` and
2549
+ // `A →(else)→ Publish`. With the condition true, the losing edge `A→Publish` skipped
2550
+ // **Publish** while Review was still running; Review completed, Publish was already
2551
+ // terminal, and `Skipped` satisfies dependents — so the graph settled Complete with the
2552
+ // publish step never executed. No error and no stall.
2553
+ //
2554
+ // Confirmed against `liveEdges`, which by this point has both losers and definitely-false
2555
+ // edges removed, so "a live gating edge still points here" is exactly the surviving-route
2556
+ // question. A genuine loser has none and is still skipped.
2557
+ const confirmedSkipSeeds = new Set(ConfirmSkipSeeds([...resolution.skipSeedTaskIDs], liveEdges));
2558
+ // Exclusive edges get verdicts too, once their origin can decide them: a loser is a branch
2559
+ // not taken, a member of an undecided group is held, a surviving edge is satisfied. Same
2560
+ // vocabulary as ordinary edges so a viewer never needs to know which dialect an edge was.
2561
+ //
2562
+ // GATED ON THE SAME `terminalDecides` SET THE RESOLUTION USED — not on
2563
+ // `TERMINAL_FOR_CONDITIONS`. Since R2-4 a `Failed` origin decides its group only under
2564
+ // `failureSemantics: 'edges'`; announcing from the wider set would tell a viewer the fork
2565
+ // resolved while under `'block'` the engine deliberately left it undecided and let the
2566
+ // ordinary block cascade own everything downstream. A console that narrates decisions the
2567
+ // engine did not make is worse than one that stays quiet.
2568
+ const exclusiveHolds = new Set(resolution.holdTaskIDs);
2569
+ for (const d of exclusive) {
2570
+ const originStatus = entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending';
2571
+ if (!decidingStatuses.has(originStatus) && !OverrideVerdictFor(debug ?? {}, d.ID))
2572
+ continue;
2573
+ gateDecisions.push({
2574
+ edge: d,
2575
+ verdict: loserEdgeIDs.has(d.ID)
2576
+ ? 'notTaken'
2577
+ : exclusiveHolds.has(d.TaskID) ? 'held' : 'satisfied',
2578
+ reason: exclusiveHolds.has(d.TaskID)
2579
+ ? 'this fork is undecided — a path in its group cannot be answered yet'
2580
+ : undefined,
2581
+ });
2582
+ }
2583
+ this.emitGateDecisions(provider, parentTaskID, gateDecisions, entityById);
1538
2584
  const nodes = children.map((c) => ({ id: c.ID, status: c.Status }));
2585
+ // THE CASCADE IS COMPUTED HERE, NOT ONLY AT SKIP TIME (R2-14).
2586
+ //
2587
+ // The claim filter covered seeds, holds and unreachable targets but not the cascade's
2588
+ // DESCENDANTS, and the skip writes are sequential per-entity saves. Between a seed's
2589
+ // `Skipped` landing and its descendants', another instance's fresh load sees
2590
+ // Skipped-satisfies-prerequisites and finds those descendants eligible — so it claims and
2591
+ // executes a branch that was never taken, irreversibly if the step has side effects.
2592
+ //
2593
+ // The set is already needed by the propagation pass, so computing it once here costs
2594
+ // nothing and closes the window by construction: nothing that is about to be skipped is
2595
+ // claimable, whichever instance is looking.
2596
+ const allSkipSeeds = [...confirmedSkipSeeds, ...unreachableTaskIDs];
2597
+ const cascadeSkipTaskIDs = new Set([
2598
+ ...allSkipSeeds,
2599
+ ...ComputeSkipCascade(nodes, liveEdges, allSkipSeeds),
2600
+ ]);
2601
+ // A DECLARED EARLY FINISH MAKES EVERY REMAINING STEP UNCLAIMABLE (R3-1).
2602
+ //
2603
+ // The declaration is durable, so this holds for every instance rather than only the one that
2604
+ // decided it — which is the whole point. Folded into the same set the claim filter already
2605
+ // consults, so nothing about to be skipped can be claimed and started in the window between
2606
+ // the decision and the skip writes.
2607
+ if (parentMeta.earlyFinishedAt) {
2608
+ for (const node of nodes) {
2609
+ if (node.status === 'Pending')
2610
+ cascadeSkipTaskIDs.add(node.id);
2611
+ }
2612
+ }
1539
2613
  return {
1540
2614
  nodes,
1541
2615
  edges: liveEdges,
1542
2616
  entityById,
1543
2617
  unreachableTaskIDs,
1544
- skipSeedTaskIDs: new Set(resolution.skipSeedTaskIDs),
1545
- holdTaskIDs: new Set(resolution.holdTaskIDs),
1546
- handledFailureIDs: await this.computeHandledFailures(provider, parentTaskID, nodes, liveEdges),
2618
+ cascadeSkipTaskIDs,
2619
+ skipSeedTaskIDs: confirmedSkipSeeds,
2620
+ // Exclusive holds and ordinary-condition holds are the same state and share one set:
2621
+ // "we cannot tell yet, so nothing may claim this."
2622
+ holdTaskIDs: new Set([...resolution.holdTaskIDs, ...heldByCondition]),
2623
+ handledFailureIDs: this.computeHandledFailures(failureSemantics, nodes, liveEdges),
1547
2624
  };
1548
2625
  }
2626
+ /**
2627
+ * Reports an unevaluable condition ONCE per edge, not once per poll.
2628
+ *
2629
+ * Eligibility is recomputed every cycle, so an unqualified LogError here would repeat every few
2630
+ * seconds for as long as the graph is held — which buries the one line that matters under
2631
+ * thousands of copies of itself. Keyed by edge id plus the failure text, so a condition that
2632
+ * starts failing differently is reported again.
2633
+ */
2634
+ logUnevaluableConditionOnce(dep, errorMessage) {
2635
+ const key = `${dep.ID}:${errorMessage ?? ''}`;
2636
+ if (this.reportedUnevaluableConditions.has(key))
2637
+ return;
2638
+ this.reportedUnevaluableConditions.add(key);
2639
+ LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
2640
+ `(${errorMessage}); condition text: ${JSON.stringify(dep.Condition)}. ` +
2641
+ `Task ${dep.TaskID} is HELD — it will not run and will not be skipped until the ` +
2642
+ `condition can be evaluated. The graph reports as stalled while this holds.`);
2643
+ }
1549
2644
  /**
1550
2645
  * Decides whether a conditional dependency edge is live.
1551
2646
  *
@@ -1553,38 +2648,55 @@ export class TaskGraphDispatcher {
1553
2648
  * only information a runtime graph has to branch on. Returns `'drop'` only on a definite false;
1554
2649
  * an unevaluable condition keeps the edge for the reason stated at the call site.
1555
2650
  */
1556
- evaluateEdgeCondition(dep, entityById) {
2651
+ evaluateEdgeCondition(dep, entityById, failureSemantics, invocation, debug) {
2652
+ // An operator's override answers the edge BEFORE the condition is consulted — an override
2653
+ // exists precisely because the condition cannot be answered (or answered wrongly), so
2654
+ // evaluating first would re-produce the hold the override exists to end.
2655
+ const override = OverrideVerdictFor(debug ?? {}, dep.ID);
2656
+ if (override) {
2657
+ return {
2658
+ outcome: override === 'true' ? 'keep' : 'drop',
2659
+ reason: `answered '${override}' by an operator override`,
2660
+ decided: true,
2661
+ };
2662
+ }
1557
2663
  const upstream = entityById.get(dep.DependsOnTaskID);
1558
2664
  if (!upstream)
1559
- return 'keep';
1560
- // TERMINALITY GUARD — fixes a latent bug, not a hypothetical one.
1561
- //
1562
- // Without it, every conditional edge is evaluated on every poll cycle, including while its
1563
- // origin is still Pending. A condition like `succeeded` is then a DEFINITE FALSE, the edge
1564
- // is dropped, and the target is Blocked at wave one — permanently, before the origin ever
1565
- // ran. That kills any conditioned linear chain, which is the most common flow shape there
1566
- // is.
2665
+ return { outcome: 'keep', decided: false };
2666
+ // The DECISION lives in `condition-gate`; what stays here is the loading and the logging.
1567
2667
  //
1568
- // A non-terminal origin is UNDECIDED, and 'keep' is the safe reading of undecided: the
1569
- // prerequisite gate already prevents the target starting early, so keeping the edge costs
1570
- // nothing and dropping it is irreversible.
1571
- if (!TERMINAL_FOR_CONDITIONS.has(upstream.Status))
1572
- return 'keep';
1573
- let output = null;
1574
- if (upstream.OutputPayload) {
1575
- try {
1576
- output = JSON.parse(upstream.OutputPayload);
1577
- }
1578
- catch { /* a malformed payload is not grounds to drop a prerequisite */ }
1579
- }
1580
- const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
1581
- if (!result.Success) {
1582
- LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
1583
- `(${result.ErrorMessage}); keeping the edge so the graph stalls visibly rather than ` +
1584
- `running ${dep.TaskID} out of order.`);
1585
- return 'keep';
1586
- }
1587
- return result.Value ? 'keep' : 'drop';
2668
+ // `DecideGate` takes the evaluation as a thunk rather than a value, and that is the fix, not
2669
+ // a style: the terminality guard has to stop the evaluation happening at all. Evaluating
2670
+ // `succeeded` against a still-Pending origin does not fail — it returns a confident, wrong
2671
+ // `false`, the edge is dropped, and the target is Blocked at wave one before the origin ever
2672
+ // ran. That killed every conditioned linear chain, with no error anywhere.
2673
+ let unevaluableError;
2674
+ let evaluated = false;
2675
+ const outcome = DecideGate(upstream.Status, failureSemantics, () => {
2676
+ evaluated = true;
2677
+ const result = this.conditionEvaluator.Evaluate(dep.Condition, BuildConditionContext(upstream, ParseConditionOutput(upstream.OutputPayload), invocation));
2678
+ if (!result.Success)
2679
+ unevaluableError = result.ErrorMessage;
2680
+ return result;
2681
+ });
2682
+ // Reported here rather than inside the decision, so the pure part stays pure and a held edge
2683
+ // is still loud once — see logUnevaluableConditionOnce.
2684
+ if (outcome === 'hold')
2685
+ this.logUnevaluableConditionOnce(dep, unevaluableError);
2686
+ return {
2687
+ outcome,
2688
+ reason: outcome === 'hold'
2689
+ ? (unevaluableError ?? 'the condition cannot be answered yet')
2690
+ : undefined,
2691
+ // A verdict was rendered only when the gate was genuinely ASKED. Terminality alone is
2692
+ // not enough since R3-2: a Failed origin under 'block' (and a Cancelled origin under
2693
+ // either dialect) returns 'keep' WITHOUT evaluating, so the block cascade owns the
2694
+ // target — announcing that as "satisfied" would tell a viewer a gate opened that
2695
+ // DecideGate deliberately never asked. `evaluated` is set by the thunk itself, so this
2696
+ // cannot drift from the gate's own rules; a Skipped origin's unevaluated drop is still
2697
+ // a verdict a viewer needs (the branch was not taken).
2698
+ decided: evaluated || (upstream.Status === 'Skipped' && outcome === 'drop'),
2699
+ };
1588
2700
  }
1589
2701
  /**
1590
2702
  * An exclusive edge's condition as a three-way outcome.
@@ -1593,55 +2705,70 @@ export class TaskGraphDispatcher {
1593
2705
  * the branch, the second holds the whole group. The generic keep/drop path cannot express that
1594
2706
  * difference, which is why exclusive edges take this route instead.
1595
2707
  */
1596
- evaluateExclusiveCondition(dep, entityById) {
2708
+ evaluateExclusiveCondition(dep, entityById, invocation, debug) {
2709
+ // Same override-first rule as ordinary edges — see evaluateEdgeCondition.
2710
+ const override = OverrideVerdictFor(debug ?? {}, dep.ID);
2711
+ if (override)
2712
+ return override === 'true' ? 'satisfied' : 'unsatisfied';
1597
2713
  if (!dep.Condition?.trim())
1598
2714
  return 'satisfied';
1599
2715
  const upstream = entityById.get(dep.DependsOnTaskID);
1600
2716
  if (!upstream)
1601
2717
  return 'unevaluable';
1602
- let output = null;
1603
- if (upstream.OutputPayload) {
1604
- try {
1605
- output = JSON.parse(upstream.OutputPayload);
1606
- }
1607
- catch { /* malformed payload */ }
1608
- }
1609
- const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
2718
+ const result = this.conditionEvaluator.Evaluate(dep.Condition,
2719
+ // The invocation envelope rides the EXCLUSIVE dialect too. R3-3 threaded it into the
2720
+ // ordinary path; a flow's XOR branch reading `data.userApproval` is the same documented
2721
+ // condition on a different edge kind, and `BuildConditionContext`'s defaulted parameter
2722
+ // made omitting it here silently evaluate those roots against nothing.
2723
+ BuildConditionContext(upstream, ParseConditionOutput(upstream.OutputPayload), invocation));
2724
+ // SAME CLASSIFICATION AS THE ORDINARY DIALECT (R2-3). The null-safe envelope already makes
2725
+ // one level of absence read as false here, but a deeper absent chain still throws — and
2726
+ // calling that 'unevaluable' would hold the whole group forever on a terminal origin, while
2727
+ // `DecideGate` would have dropped the identical condition. Two dialects, one question.
1610
2728
  if (!result.Success)
1611
- return 'unevaluable';
2729
+ return IsBrokenGuard(result.ErrorMessage) ? 'unevaluable' : 'unsatisfied';
1612
2730
  return result.Value ? 'satisfied' : 'unsatisfied';
1613
2731
  }
1614
2732
  /**
1615
- * Everything an edge condition can see — the SUPERSET of both dialects.
2733
+ * Announces gate verdicts that CHANGED since this instance last looked.
1616
2734
  *
1617
- * A flow condition is written against `payload` / `stepResult` / `flowContext` / `data` /
1618
- * `context`; the dispatcher's own conditions are written against `status` / `succeeded` /
1619
- * `failed` / `output` / `errorMessage`. Compiling flows onto this engine without the flow
1620
- * dialect would make every `payload.x` condition evaluate against nothing — silently, since an
1621
- * undefined property is simply falsy. Both dialects are readable here so a condition means the
1622
- * same thing on either engine.
1623
- *
1624
- * `payload` is the ORIGIN task's post-step snapshot. There is deliberately no "graph-wide
1625
- * payload": each task's output is its own, and inventing a merged one would give conditions a
1626
- * value the flow engine never had.
2735
+ * Fire-and-forget by design: `loadGraphState` is synchronous graph assembly, and the owner
2736
+ * lookup the frame needs is async — so the emission floats behind rather than making state
2737
+ * loading wait on observability. Frames are commentary, never a step of the work.
1627
2738
  */
1628
- buildConditionContext(upstream, output) {
1629
- const envelope = (output && typeof output === 'object' ? output : {});
1630
- const succeeded = upstream.Status === 'Complete';
1631
- return {
1632
- // dispatcher dialect — unchanged
1633
- status: upstream.Status,
1634
- succeeded,
1635
- failed: upstream.Status === 'Failed',
1636
- output,
1637
- errorMessage: upstream.ErrorMessage ?? null,
1638
- // flow dialect
1639
- payload: envelope.payload ?? output,
1640
- stepResult: { Success: succeeded, step: upstream.Name, result: envelope.result ?? output },
1641
- flowContext: { currentStepId: upstream.ID, completedSteps: [], executionPath: [], stepCount: 0 },
1642
- data: envelope.data ?? {},
1643
- context: envelope.context ?? {},
1644
- };
2739
+ emitGateDecisions(provider, parentTaskID, decisions, entityById) {
2740
+ if (!this.observer || decisions.length === 0)
2741
+ return;
2742
+ let perEdge = this.emittedGateVerdicts.get(parentTaskID);
2743
+ if (!perEdge) {
2744
+ perEdge = new Map();
2745
+ this.emittedGateVerdicts.set(parentTaskID, perEdge);
2746
+ }
2747
+ const changed = decisions.filter((d) => {
2748
+ const key = `${d.verdict}|${d.reason ?? ''}`;
2749
+ if (perEdge.get(d.edge.ID) === key)
2750
+ return false;
2751
+ perEdge.set(d.edge.ID, key);
2752
+ return true;
2753
+ });
2754
+ if (changed.length === 0)
2755
+ return;
2756
+ void this.resolveOwner(provider, parentTaskID).then((owner) => {
2757
+ for (const d of changed) {
2758
+ this.emit({
2759
+ Kind: 'GateDecision',
2760
+ ParentTaskID: parentTaskID,
2761
+ OwnerUserID: owner,
2762
+ TaskID: d.edge.TaskID,
2763
+ TaskName: entityById.get(d.edge.TaskID)?.Name,
2764
+ EdgeID: d.edge.ID,
2765
+ DependsOnTaskID: d.edge.DependsOnTaskID,
2766
+ Verdict: d.verdict,
2767
+ ConditionText: d.edge.Condition ?? undefined,
2768
+ Reason: d.reason,
2769
+ });
2770
+ }
2771
+ }).catch(() => { });
1645
2772
  }
1646
2773
  /** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
1647
2774
  async loadDependencyOutputs(provider, taskID) {
@@ -1676,7 +2803,7 @@ export class TaskGraphDispatcher {
1676
2803
  * Every branch is normalized to one shape so the recording path above stays single: an action has
1677
2804
  * no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
1678
2805
  */
1679
- async runTaskBody(task, provider, inputPayload, dependencyOutputs) {
2806
+ async runTaskBody(task, provider, inputPayload, dependencyOutputs, onProgress) {
1680
2807
  const payload = this.mergedPayload(inputPayload, dependencyOutputs);
1681
2808
  const config = task.ConfigurationObject;
1682
2809
  // A loop's own step type decides how many times its body runs; the body itself is dispatched
@@ -1714,6 +2841,7 @@ export class TaskGraphDispatcher {
1714
2841
  TemplateParameters: config?.prompt?.templateParameters,
1715
2842
  Provider: provider,
1716
2843
  ContextUser: this.contextUser,
2844
+ OnProgress: onProgress,
1717
2845
  });
1718
2846
  // A prompt's response is DEEP-MERGED into the payload rather than replacing it. A prompt
1719
2847
  // answers one question; replacing the payload with its answer would discard everything
@@ -1742,8 +2870,9 @@ export class TaskGraphDispatcher {
1742
2870
  DependencyOutputs: dependencyOutputs,
1743
2871
  Provider: provider,
1744
2872
  ContextUser: this.contextUser,
2873
+ OnProgress: onProgress,
1745
2874
  }), AgentRunID: null }
1746
- : await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs);
2875
+ : await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs, onProgress);
1747
2876
  return {
1748
2877
  ...raw,
1749
2878
  Output: this.applyStepOutputMapping(task, payload, raw.Output, config?.outputMapping),
@@ -2015,7 +3144,7 @@ export class TaskGraphDispatcher {
2015
3144
  * Depth and provenance are read together because they come from the same row: the graph's parent
2016
3145
  * task knows both how many continuation hops led here and which run submitted it.
2017
3146
  */
2018
- async runAgentNode(task, provider, effectiveInput, dependencyOutputs) {
3147
+ async runAgentNode(task, provider, effectiveInput, dependencyOutputs, onProgress) {
2019
3148
  const context = await this.graphContext(provider, task);
2020
3149
  return this.agentRunner.RunAgentForTask({
2021
3150
  TaskID: task.ID,
@@ -2026,8 +3155,193 @@ export class TaskGraphDispatcher {
2026
3155
  SubmittingAgentRunID: context.SubmittingAgentRunID,
2027
3156
  Provider: provider,
2028
3157
  ContextUser: this.contextUser,
3158
+ OnProgress: onProgress,
2029
3159
  });
2030
3160
  }
3161
+ /**
3162
+ * Leaves exactly one open request standing for a task, withdrawing any others.
3163
+ *
3164
+ * The oldest wins — it is the one whose notification the assignee most likely already saw.
3165
+ *
3166
+ * **Called from every path that reads a task's open requests**, not only from the raise. That is
3167
+ * deliberate and it is the half R3-5 first got wrong: a task is notified exactly once, and
3168
+ * `notifyHumanTaskReady` returns at the marker forever after, so a duplicate minted after that
3169
+ * pass — by an instance that crashed between its insert and its de-dup, or by any build older
3170
+ * than this one — was never looked at again by the only code that could have collapsed it. The
3171
+ * waiting-task sweep is what actually reaches those.
3172
+ *
3173
+ * @param open the task's open requests, oldest first
3174
+ * @param keepIfSole the row this caller is responsible for, named only for the log
3175
+ */
3176
+ async withdrawDuplicateRequests(provider, taskID, open, keepIfSole) {
3177
+ if (open.length <= 1)
3178
+ return;
3179
+ try {
3180
+ LogStatus(`[TaskGraphDispatcher] Task ${taskID} had ${open.length} open requests — keeping the ` +
3181
+ `oldest (${open[0].ID}${UUIDsEqual(open[0].ID, keepIfSole) ? ', this instance\'s' : ''}) and withdrawing the rest.`);
3182
+ for (const duplicate of open.slice(1)) {
3183
+ duplicate.Status = 'Canceled';
3184
+ duplicate.Comments = 'A duplicate request for the same step; the earlier one stands.';
3185
+ if (!(await duplicate.Save())) {
3186
+ LogError(`[TaskGraphDispatcher] Could not withdraw duplicate request ${duplicate.ID}: ` +
3187
+ `${duplicate.LatestResult?.CompleteMessage ?? 'unknown error'}`);
3188
+ }
3189
+ }
3190
+ }
3191
+ catch (e) {
3192
+ LogError(`[TaskGraphDispatcher] Could not de-duplicate requests for task ${taskID}: ${e instanceof Error ? e.message : String(e)}`);
3193
+ }
3194
+ }
3195
+ /**
3196
+ * Closes the still-open asks raised for tasks that will never be answered.
3197
+ *
3198
+ * `Canceled` rather than `Expired`: nobody ran out of time, the ask was withdrawn — and the two
3199
+ * mean different things downstream, since an expired human step is treated as a FAILURE that a
3200
+ * give-up edge can route around, which would be a lie about a step the workflow decided it no
3201
+ * longer needed.
3202
+ *
3203
+ * Failures are logged and never propagated. The graph's outcome is already decided; refusing to
3204
+ * finish over an inbox row would trade a stale notification for a stalled workflow.
3205
+ */
3206
+ async withdrawOpenRequests(provider, taskIDs, reason) {
3207
+ if (taskIDs.length === 0)
3208
+ return;
3209
+ try {
3210
+ const idList = taskIDs.map((id) => `'${id}'`).join(',');
3211
+ const open = await RunView.FromMetadataProvider(provider).RunView({
3212
+ EntityName: 'MJ: AI Agent Requests',
3213
+ ExtraFilter: `Status='Requested' AND OriginatingTaskID IN (${idList})`,
3214
+ ResultType: 'entity_object',
3215
+ BypassCache: true,
3216
+ }, this.contextUser);
3217
+ if (!open.Success) {
3218
+ LogError(`[TaskGraphDispatcher] Could not read open requests to withdraw: ${open.ErrorMessage}`);
3219
+ return;
3220
+ }
3221
+ for (const request of open.Results ?? []) {
3222
+ request.Status = 'Canceled';
3223
+ request.Comments = reason;
3224
+ if (!(await request.Save())) {
3225
+ LogError(`[TaskGraphDispatcher] Could not withdraw request ${request.ID}: ` +
3226
+ `${request.LatestResult?.CompleteMessage ?? 'unknown error'}. It will keep showing ` +
3227
+ `in someone's inbox for a step that will never run.`);
3228
+ }
3229
+ }
3230
+ }
3231
+ catch (e) {
3232
+ LogError(`[TaskGraphDispatcher] Could not withdraw open requests: ${e instanceof Error ? e.message : String(e)}`);
3233
+ }
3234
+ }
3235
+ /**
3236
+ * Says once, per graph, that this instance settled work it cannot announce.
3237
+ *
3238
+ * Once because the sweep re-offers the graph every poll for the rest of its window, and a line
3239
+ * per poll would bury the thing it is trying to report — which is a DEPLOYMENT fact, not a graph
3240
+ * fact: if no instance anywhere carries a deliverer, these settlements never reach anyone.
3241
+ */
3242
+ reportUndeliverableOnce(parentID) {
3243
+ if (this.reportedUndeliverable.has(parentID))
3244
+ return;
3245
+ this.reportedUndeliverable.add(parentID);
3246
+ LogStatus(`[TaskGraphDispatcher] Graph ${parentID} has settled but this instance has no continuation ` +
3247
+ `deliverer, so it is leaving the announcement to a peer that has one. If no instance in ` +
3248
+ `this deployment can deliver, the settlement will never be announced.`);
3249
+ }
3250
+ /**
3251
+ * Keeps a graph in this instance's sweep regardless of what its row timestamp says.
3252
+ *
3253
+ * Bounded, and the bound is about noise rather than surrender: past the cap the graph has failed
3254
+ * on every attempt for minutes, so another identical attempt will not fix it, and continuing
3255
+ * costs a full graph load per poll forever. It is reported once and left to the startup sweep.
3256
+ */
3257
+ keepRetryingSettlement(parentID) {
3258
+ const passes = (this.retryingSettlement.get(parentID) ?? 0) + 1;
3259
+ if (passes > MAX_SETTLEMENT_RETRY_PASSES) {
3260
+ this.retryingSettlement.delete(parentID);
3261
+ LogError(`[TaskGraphDispatcher] Graph ${parentID} has failed to settle on ${MAX_SETTLEMENT_RETRY_PASSES} ` +
3262
+ `consecutive passes; this instance will stop re-queueing it. Its submitting run may be ` +
3263
+ `left Paused. A restart's startup sweep will try again.`);
3264
+ return;
3265
+ }
3266
+ this.retryingSettlement.set(parentID, passes);
3267
+ }
3268
+ /**
3269
+ * A graph's durable metadata bag, for the questions a pass asks of it.
3270
+ *
3271
+ * Defaults on any failure to read it, and the defaults are the safe directions: `'block'` means
3272
+ * a failed step decides nothing, so a graph whose metadata we cannot read stalls visibly instead
3273
+ * of resolving forks on the say-so of a failure; and no early-finish declaration means nothing
3274
+ * is removed from the claim filter on the strength of a read that did not work.
3275
+ */
3276
+ async readParentMetadataFor(provider, parentTaskID) {
3277
+ try {
3278
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
3279
+ if (await parent.Load(parentTaskID))
3280
+ return ParseTaskGraphParentMetadata(parent.InputPayload);
3281
+ }
3282
+ catch { /* fall through to the safe defaults */ }
3283
+ return ParseTaskGraphParentMetadata(null);
3284
+ }
3285
+ /**
3286
+ * Whether the submitting run is in a state where this pass's writes to it will mean anything.
3287
+ *
3288
+ * **Read-only on purpose.** The settled branch's write order — layout, frame, cost, lifecycle,
3289
+ * delivery — is load-bearing and documented at each step; this asks the question those writes
3290
+ * depend on without joining them. What it prevents is a pass that goes through the motions and
3291
+ * then claims the delivery marker, making itself the last pass ever to look at the graph.
3292
+ *
3293
+ * Three answers, and the middle one is the bug:
3294
+ *
3295
+ * - **no run** — a scheduled or remote-triggered graph has nobody waiting. Proceed.
3296
+ * - **still `Running`** — `finalizeAgentRun` has not parked it yet. The graph beat its own
3297
+ * submitter to the finish line, which is ordinary for a fast graph and lasts milliseconds.
3298
+ * Defer: one poll later the run is parked and everything lands.
3299
+ * - **anything else** — `Paused` (settle it), or already `Completed`/`Failed`/`Cancelled` for
3300
+ * its own reasons (leave it; the lifecycle write's own guard declines). Proceed.
3301
+ *
3302
+ * **The deferral is bounded**, because "not parked yet" and "the submitting process died before
3303
+ * it could park" look identical from here. Waiting forever on the second would lose the outcome
3304
+ * of work that actually completed — strictly worse than announcing it late — so past the grace
3305
+ * period this proceeds and says why. The run itself stays `Running`, which is visibly wrong and
3306
+ * belongs to whatever reconciles abandoned runs, not to the graph that finished correctly.
3307
+ */
3308
+ async submittingRunReadiness(provider, parent) {
3309
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
3310
+ if (!meta.submittedByAgentRunID)
3311
+ return { Verdict: 'ready', SubmitterCancelled: false };
3312
+ try {
3313
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
3314
+ if (!(await run.Load(meta.submittedByAgentRunID))) {
3315
+ // Transient, most likely. Deferring costs a poll; proceeding costs the marker.
3316
+ LogError(`[TaskGraphDispatcher] Could not read run ${meta.submittedByAgentRunID} to check whether graph ${parent.ID} may settle it; retrying next pass.`);
3317
+ return { Verdict: 'defer', SubmitterCancelled: false };
3318
+ }
3319
+ // A CANCELLED SUBMITTER HAS NOBODY WAITING (R2-9). Settlement still runs — the graph's
3320
+ // own bookkeeping is owed either way — but announcing it would message a conversation
3321
+ // about a workflow the user stopped, and for `reinvoke` would start a fresh billed turn
3322
+ // for the agent they cancelled.
3323
+ const cancelled = run.Status === 'Cancelled';
3324
+ const settledFor = parent.CompletedAt ? Date.now() - parent.CompletedAt.getTime() : 0;
3325
+ if (!IsSubmittingRunReady(run.Status, settledFor)) {
3326
+ // Still `Running` and inside the grace: `finalizeAgentRun` has not parked it yet.
3327
+ // Defer the whole run-half so nothing claims the marker — see the call site.
3328
+ return { Verdict: 'defer', SubmitterCancelled: cancelled };
3329
+ }
3330
+ if (run.Status === 'Running') {
3331
+ // Ready DESPITE being unparked means the grace has expired: the submitting process
3332
+ // most likely died before it could park. Proceeding loses nothing that is still
3333
+ // recoverable and stops a dead submitter holding a finished workflow's outcome.
3334
+ LogError(`[TaskGraphDispatcher] Run ${run.ID} has been Running for ${Math.round(settledFor / 1000)}s ` +
3335
+ `since graph ${parent.ID} settled — it never parked, so its submitting process most ` +
3336
+ `likely died. Settling and delivering the graph anyway; the run needs separate attention.`);
3337
+ }
3338
+ return { Verdict: 'ready', SubmitterCancelled: cancelled };
3339
+ }
3340
+ catch (e) {
3341
+ LogError(`[TaskGraphDispatcher] Could not check the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
3342
+ return { Verdict: 'defer', SubmitterCancelled: false };
3343
+ }
3344
+ }
2031
3345
  /**
2032
3346
  * Completes the agent run that parked on this graph.
2033
3347
  *
@@ -2053,38 +3367,42 @@ export class TaskGraphDispatcher {
2053
3367
  async settleSubmittingRun(provider, parent, graphStatus) {
2054
3368
  const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
2055
3369
  if (!meta.submittedByAgentRunID)
2056
- return; // a scheduled or remote-triggered graph has nobody waiting
3370
+ return 'done'; // a scheduled or remote-triggered graph has nobody waiting
2057
3371
  try {
2058
3372
  const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
2059
3373
  if (!(await run.Load(meta.submittedByAgentRunID))) {
2060
3374
  LogError(`[TaskGraphDispatcher] Could not load run ${meta.submittedByAgentRunID} to settle it against graph ${parent.ID}.`);
2061
- return;
3375
+ return 'defer';
2062
3376
  }
2063
3377
  if (run.Status !== 'Paused')
2064
- return;
3378
+ return 'done';
2065
3379
  // The workflow's outcome becomes the run's outcome. A graph that ended any way other than
2066
3380
  // Complete did not do what the run started it to do, and a run reporting success over it
2067
3381
  // would be the same untruth in a different place.
2068
3382
  const succeeded = graphStatus === 'Complete';
2069
- run.Status = succeeded ? 'Completed' : 'Failed';
2070
- run.Success = succeeded;
2071
- run.CompletedAt = new Date();
2072
- if (!succeeded) {
2073
- const reason = `The workflow "${parent.Name}" ended ${graphStatus}.`;
2074
- run.ErrorMessage = run.ErrorMessage ? `${run.ErrorMessage}\n\n${reason}` : reason;
2075
- }
2076
- if (!(await run.Save())) {
3383
+ // COLUMN-SCOPED AND GUARDED ON `Paused` (C4), for the same reason every parent write has
3384
+ // been since Round 1: a full-row save carries a stale snapshot of a row another instance
3385
+ // may have moved, and the predicate makes the transition once-only rather than
3386
+ // last-write-wins.
3387
+ const settled = await this.claims.TrySettleRun(provider, run.ID, succeeded, succeeded ? null : `The workflow "${parent.Name}" ended ${graphStatus}.`, this.contextUser);
3388
+ if (!settled) {
2077
3389
  // Left parked rather than forced. A run stuck at Paused is visibly unfinished, which
2078
3390
  // is a state someone can investigate; a run flipped to Completed by a write that did
2079
3391
  // not land would be the same lie this whole change removes.
2080
- LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}: ` +
2081
- `${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It remains Paused.`);
2082
- return;
3392
+ // Rowcount 0 is either "a peer settled it first" — fine, and the status read above
3393
+ // would have caught the common case — or a write that did not land. Deferring covers
3394
+ // both: a peer's settle makes the next pass's `Paused` check return `done`.
3395
+ LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}; ` +
3396
+ `it is no longer Paused or the write did not land. Retrying next pass.`);
3397
+ return 'defer';
2083
3398
  }
2084
- LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${run.Status} — workflow "${parent.Name}" ended ${graphStatus}.`);
3399
+ LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${succeeded ? 'Completed' : 'Failed'} — ` +
3400
+ `workflow "${parent.Name}" ended ${graphStatus}.`);
3401
+ return 'done';
2085
3402
  }
2086
3403
  catch (e) {
2087
3404
  LogError(`[TaskGraphDispatcher] Could not settle the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
3405
+ return 'defer';
2088
3406
  }
2089
3407
  }
2090
3408
  /**