@memberjunction/task-graph 6.1.0-edge.1 → 6.1.0-edge.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/LICENSE +180 -4
  2. package/README.md +214 -0
  3. package/dist/TaskClaimStore.d.ts +387 -4
  4. package/dist/TaskClaimStore.d.ts.map +1 -1
  5. package/dist/TaskClaimStore.js +605 -20
  6. package/dist/TaskClaimStore.js.map +1 -1
  7. package/dist/TaskGraphDispatcher.d.ts +668 -5
  8. package/dist/TaskGraphDispatcher.d.ts.map +1 -1
  9. package/dist/TaskGraphDispatcher.js +2942 -127
  10. package/dist/TaskGraphDispatcher.js.map +1 -1
  11. package/dist/TaskGraphService.d.ts +364 -5
  12. package/dist/TaskGraphService.d.ts.map +1 -1
  13. package/dist/TaskGraphService.js +1039 -43
  14. package/dist/TaskGraphService.js.map +1 -1
  15. package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -1
  16. package/dist/TaskGraphSubmitterImpl.js +5 -0
  17. package/dist/TaskGraphSubmitterImpl.js.map +1 -1
  18. package/dist/TaskLoopExecutor.d.ts +62 -0
  19. package/dist/TaskLoopExecutor.d.ts.map +1 -0
  20. package/dist/TaskLoopExecutor.js +248 -0
  21. package/dist/TaskLoopExecutor.js.map +1 -0
  22. package/dist/WorkflowSpecSync.d.ts +28 -2
  23. package/dist/WorkflowSpecSync.d.ts.map +1 -1
  24. package/dist/WorkflowSpecSync.js +83 -2
  25. package/dist/WorkflowSpecSync.js.map +1 -1
  26. package/dist/condition-gate.d.ts +128 -0
  27. package/dist/condition-gate.d.ts.map +1 -0
  28. package/dist/condition-gate.js +257 -0
  29. package/dist/condition-gate.js.map +1 -0
  30. package/dist/debug-state.d.ts +102 -0
  31. package/dist/debug-state.d.ts.map +1 -0
  32. package/dist/debug-state.js +135 -0
  33. package/dist/debug-state.js.map +1 -0
  34. package/dist/index.d.ts +7 -0
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +7 -0
  37. package/dist/index.js.map +1 -1
  38. package/dist/operations/TaskGraphDebugOperations.d.ts +99 -0
  39. package/dist/operations/TaskGraphDebugOperations.d.ts.map +1 -0
  40. package/dist/operations/TaskGraphDebugOperations.js +310 -0
  41. package/dist/operations/TaskGraphDebugOperations.js.map +1 -0
  42. package/dist/operations/TaskGraphOperations.d.ts +20 -2
  43. package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
  44. package/dist/operations/TaskGraphOperations.js +51 -8
  45. package/dist/operations/TaskGraphOperations.js.map +1 -1
  46. package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
  47. package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
  48. package/dist/operations/WorkflowDraftOperation.js +141 -0
  49. package/dist/operations/WorkflowDraftOperation.js.map +1 -0
  50. package/dist/settlement-rescue.d.ts +85 -0
  51. package/dist/settlement-rescue.d.ts.map +1 -0
  52. package/dist/settlement-rescue.js +119 -0
  53. package/dist/settlement-rescue.js.map +1 -0
  54. package/dist/task-graph-kick.d.ts +3 -0
  55. package/dist/task-graph-kick.d.ts.map +1 -0
  56. package/dist/task-graph-kick.js +17 -0
  57. package/dist/task-graph-kick.js.map +1 -0
  58. package/dist/task-predicates.d.ts +77 -0
  59. package/dist/task-predicates.d.ts.map +1 -0
  60. package/dist/task-predicates.js +75 -0
  61. package/dist/task-predicates.js.map +1 -0
  62. package/dist/types.d.ts +224 -1
  63. package/dist/types.d.ts.map +1 -1
  64. package/dist/types.js.map +1 -1
  65. package/package.json +12 -8
@@ -18,14 +18,44 @@
18
18
  *
19
19
  * @module @memberjunction/task-graph
20
20
  */
21
- import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, } from '@memberjunction/ai-core-plus';
21
+ import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, ConfirmSkipSeeds, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
22
22
  import { LogError, LogStatus, RunView } from '@memberjunction/core';
23
- import { ShutdownRegistry } from '@memberjunction/global';
24
- import { TaskClaimStore } from './TaskClaimStore.js';
23
+ import { ShutdownRegistry, UUIDsEqual } from '@memberjunction/global';
24
+ import { TaskClaimStore, TERMINAL_PARENT_STATUSES, TERMINAL_PARENT_STATUS_SQL } from './TaskClaimStore.js';
25
+ import { BuildConditionContext, DecideGate, IsBrokenGuard, ParseConditionOutput, } from './condition-gate.js';
26
+ import { HumanTaskSQL, IsHumanTask } from './task-predicates.js';
27
+ import { IsSettlementExpired, IsSubmittingRunReady, SelectUnsettledGraphIDs, SweepCutoff, UNSETTLED_SWEEP_WINDOW_HOURS, UNSETTLED_STARTUP_WINDOW_HOURS, } from './settlement-rescue.js';
25
28
  import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
29
+ import { DecideClaimGate, OverrideVerdictFor, ParseTaskGraphDebugState, } from './debug-state.js';
30
+ import { RunForEachLoop, RunWhileLoop } from './TaskLoopExecutor.js';
31
+ import { RegisterTaskGraphKick } from './task-graph-kick.js';
26
32
  import { NotificationEngine } from '@memberjunction/notifications';
27
33
  /** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
28
34
  const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
35
+ /**
36
+ * Statuses a graph parent has stopped moving from.
37
+ *
38
+ * Shared by the guarded terminal write and the unsettled-graph sweep, so "terminal" means exactly
39
+ * one thing in both — the two disagreeing is how a graph becomes invisible to the machinery that is
40
+ * supposed to rescue it.
41
+ */
42
+ const TERMINAL_TASK_STATUSES = new Set(TERMINAL_PARENT_STATUSES);
43
+ /**
44
+ * How long `Stop()` waits for in-flight tasks and timer passes before giving up and saying so.
45
+ *
46
+ * Generous, because the alternative to waiting is a dispatcher that writes after its host believes
47
+ * it has shut down — settling graphs onto a connection somebody else now owns.
48
+ */
49
+ const STOP_DRAIN_TIMEOUT_MS = 30_000;
50
+ /**
51
+ * How many consecutive failing passes a graph gets before this instance stops re-queueing it.
52
+ *
53
+ * Not a giving-up threshold so much as a stop-shouting one: past this the graph has failed to settle
54
+ * on every attempt for minutes, so something is wrong that another identical attempt will not fix,
55
+ * and continuing costs a full graph load per poll forever. It is reported and left to the startup
56
+ * sweep, which is the wider net.
57
+ */
58
+ const MAX_SETTLEMENT_RETRY_PASSES = 20;
29
59
  /**
30
60
  * Written to a human task's `ClaimedBy` once its assignee has been told it is ready.
31
61
  *
@@ -34,9 +64,106 @@ const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
34
64
  * from reclamation, so this value is never mistaken for a live claim.
35
65
  */
36
66
  const HUMAN_TASK_NOTIFIED_MARKER = '__human-notified__';
37
- import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata } from './TaskGraphService.js';
67
+ /**
68
+ * The run-query capability of a provider, when it has one.
69
+ *
70
+ * `IMetadataProvider` does not extend `IRunQueryProvider`, but every provider that ships implements
71
+ * both. Narrowing by CAPABILITY rather than casting states that honestly: a provider that genuinely
72
+ * cannot run queries returns undefined and the caller reports it, instead of the call failing later
73
+ * behind a type assertion that claimed it could.
74
+ */
75
+ function asRunQueryProvider(provider) {
76
+ const candidate = provider;
77
+ return typeof candidate.RunQuery === 'function' ? candidate : undefined;
78
+ }
79
+ import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata, TASK_TYPE_NAME } from './TaskGraphService.js';
38
80
  import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
81
+ /**
82
+ * Renders a loop's bindings as template values.
83
+ *
84
+ * Template parameters are strings; an item is usually an object. Objects are JSON-encoded rather
85
+ * than dropped, because `{{ field }}` printing `[object Object]` — or nothing at all — is exactly
86
+ * the silent failure this exists to prevent.
87
+ */
88
+ function stringifyBindings(bindings) {
89
+ const out = {};
90
+ for (const [key, value] of Object.entries(bindings)) {
91
+ out[key] = typeof value === 'string' ? value : JSON.stringify(value, null, 2);
92
+ }
93
+ return out;
94
+ }
95
+ /**
96
+ * How much of a loop's per-pass payloads may be kept, and what happens when that runs out.
97
+ *
98
+ * **Why a budget exists at all.** A loop's trace lives inside one `Configuration` column, and its
99
+ * size is the product of two things nobody bounds: how many passes the loop runs, and how large the
100
+ * body's input and output are. A hundred-pass loop over documents would put megabytes in a column
101
+ * that the run tree, the timeline, the canvas and the Workflows list all read — punishing every
102
+ * reader of the row for a detail only someone inspecting one pass will ever open.
103
+ *
104
+ * **What it protects.** Only the payloads. `promptRunID` / `agentRunID` / `actionLogID` / `success`
105
+ * are always recorded: those point at the durable rows where the real forensics live, and they are
106
+ * what cost roll-up and the timeline traverse. Losing a payload costs a reader some detail; losing a
107
+ * pointer would lose the pass.
108
+ *
109
+ * **Omission is stated, never silent.** Once the budget is spent, further passes record a marker
110
+ * saying so and how large the value was, because a pass showing nothing is indistinguishable from a
111
+ * pass that produced nothing — and that ambiguity is exactly the failure this whole area keeps
112
+ * hitting.
113
+ */
114
+ const ITERATION_PAYLOAD_BUDGET_BYTES = 128 * 1024;
115
+ /** Per-value cap, so one enormous pass cannot consume the whole budget by itself. */
116
+ const ITERATION_PAYLOAD_VALUE_BYTES = 16 * 1024;
117
+ class IterationPayloadBudget {
118
+ constructor() {
119
+ this.spent = 0;
120
+ }
121
+ /**
122
+ * The value if it fits, or a marker describing what was left out.
123
+ *
124
+ * @returns the value, a marker object, or undefined when there was nothing to record
125
+ */
126
+ Take(value) {
127
+ if (value == null)
128
+ return undefined;
129
+ const asRecord = value && typeof value === 'object' && !Array.isArray(value)
130
+ ? value
131
+ : { value };
132
+ let size;
133
+ try {
134
+ size = JSON.stringify(asRecord)?.length ?? 0;
135
+ }
136
+ catch {
137
+ // Circular or otherwise unserializable. It could not be persisted anyway, and saying so
138
+ // is better than a pass that silently shows nothing.
139
+ return { __omitted: 'unserializable' };
140
+ }
141
+ if (size > ITERATION_PAYLOAD_VALUE_BYTES) {
142
+ return { __omitted: 'too-large', __bytes: size, __limit: ITERATION_PAYLOAD_VALUE_BYTES };
143
+ }
144
+ if (this.spent + size > ITERATION_PAYLOAD_BUDGET_BYTES) {
145
+ return { __omitted: 'budget-exhausted', __bytes: size, __limit: ITERATION_PAYLOAD_BUDGET_BYTES };
146
+ }
147
+ this.spent += size;
148
+ return asRecord;
149
+ }
150
+ }
151
+ /** Deep-merges a prompt's JSON response into the payload, preserving what earlier steps established. */
152
+ function deepMergePayload(base, incoming) {
153
+ const out = { ...base };
154
+ for (const [key, value] of Object.entries(incoming)) {
155
+ const existing = out[key];
156
+ const bothPlainObjects = existing && typeof existing === 'object' && !Array.isArray(existing) &&
157
+ value && typeof value === 'object' && !Array.isArray(value);
158
+ out[key] = bothPlainObjects
159
+ ? deepMergePayload(existing, value)
160
+ : value;
161
+ }
162
+ return out;
163
+ }
39
164
  export class TaskGraphDispatcher {
165
+ /** Minimum interval between `NodeProgress` frames for one task. */
166
+ static { this.NODE_PROGRESS_MIN_INTERVAL_MS = 1_000; }
40
167
  constructor(providerFactory, agentRunner, contextUser, config,
41
168
  /**
42
169
  * Optional. Absent means a host that cannot post messages or start agent turns — a worker,
@@ -48,21 +175,108 @@ export class TaskGraphDispatcher {
48
175
  * Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
49
176
  * announces nothing.
50
177
  */
51
- observer) {
178
+ observer,
179
+ /**
180
+ * Optional. Absent means this host cannot run action nodes; they stay Pending and visible
181
+ * rather than being failed, because "nobody here can run this" is not "this ran and broke".
182
+ */
183
+ actionRunner,
184
+ /**
185
+ * Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
186
+ * rather than being failed, for the same reason action nodes do.
187
+ */
188
+ promptRunner) {
52
189
  this.providerFactory = providerFactory;
53
190
  this.agentRunner = agentRunner;
54
191
  this.contextUser = contextUser;
55
192
  this.continuationDeliverer = continuationDeliverer;
56
193
  this.observer = observer;
194
+ this.actionRunner = actionRunner;
195
+ this.promptRunner = promptRunner;
196
+ /**
197
+ * Edges already reported as unevaluable, so the report is once per transition and not once per
198
+ * poll. Per-instance and in-memory by design: a restart re-reports, which is the right amount of
199
+ * noise for a condition that is still broken after a restart.
200
+ */
201
+ this.reportedUnevaluableConditions = new Set();
202
+ /** Resolved once it EXISTS; null while it does not, so a fresh install is not cached blind. */
203
+ this.cachedWorkflowTaskTypeID = null;
204
+ /**
205
+ * Graphs this instance is still trying to settle, with how many passes it has spent trying.
206
+ *
207
+ * **The sweep's window is on `__mj_UpdatedAt`, and a failing pass writes nothing** — the terminal
208
+ * write returns rowcount 0 because the row is already terminal, the layout pass touches only
209
+ * children, a refused CAS writes nothing at all. So a graph that fails to settle stops advancing
210
+ * its own timestamp and, after 24h of futile retries, ages out of the steady-state window while
211
+ * the process is up. The doc comment claimed the bound was "on abandonment, not age"; for this
212
+ * case it was on age, and R2-2's deferral made the case ordinary rather than exotic.
213
+ *
214
+ * In memory rather than a touch column because the alternative is a write on every failed
215
+ * attempt — more load exactly when something is already wrong — and because a restart is covered
216
+ * by the wide startup sweep, which is the durable backstop this leans on.
217
+ */
218
+ this.retryingSettlement = new Map();
219
+ /**
220
+ * Graphs whose settled-branch ANNOUNCEMENTS have already been made by this process.
221
+ *
222
+ * Re-entry is the point of the rescue, but only the parts that failed should repeat. Layout and
223
+ * the `GraphSettled` frame are idempotent facts about a finished graph, so a graph stuck in
224
+ * retry was re-persisting geometry and re-emitting the same frame every poll — for the whole
225
+ * 24h window, for as long as it kept failing.
226
+ */
227
+ this.announcedSettlements = new Set();
228
+ /** Graphs already reported as settled-but-undeliverable by this instance. */
229
+ this.reportedUndeliverable = new Set();
230
+ /** Live claim heartbeats by task ID, so the drain can silence the ones it gives up waiting for. */
231
+ this.heartbeats = new Map();
232
+ /** Latched once the drain has given up waiting, so a late arrival does not re-register. */
233
+ this.heartbeatsPurged = false;
57
234
  this.running = false;
58
235
  this.pollTimer = null;
59
236
  this.reconcileTimer = null;
237
+ this.unregisterKick = null;
60
238
  /** Tasks this instance is currently executing — bounds concurrency and drives heartbeats. */
61
239
  this.inFlight = new Set();
62
240
  /** Guards against a slow poll overlapping the next tick. */
63
241
  this.polling = false;
242
+ /**
243
+ * Timer-driven passes currently running — poll and reconcile alike.
244
+ *
245
+ * A counter rather than a boolean because the two timers overlap by design, and `Stop()` has to
246
+ * wait for BOTH. Neither pass is held by anything else: they are launched `void`-ed from
247
+ * `setInterval`, so without this they are unobservable from the outside and a stopped dispatcher
248
+ * keeps writing.
249
+ */
250
+ this.activePasses = 0;
64
251
  /** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
65
252
  this.ownerByParentID = new Map();
253
+ /** Monotonic pass counter for `PassCompleted` frames, so a viewer can order and gap-detect ticks. */
254
+ this.passCounter = 0;
255
+ /**
256
+ * Debug state per graph, cached for ONE pass. `pollOnce` clears it at entry, so within a pass
257
+ * the claim filter and the propagation loop read the same state (loading it twice could see a
258
+ * pause land between them and gate half a pass), and across passes a control verb written by any
259
+ * instance is picked up within one poll interval.
260
+ */
261
+ this.debugStateByGraph = new Map();
262
+ /**
263
+ * The pause state last announced per graph, so `GraphPaused`/`GraphResumed` are emitted on the
264
+ * TRANSITION rather than every pass — the verbs write durable state, not events, and it is this
265
+ * instance's job to notice the change and say so exactly once.
266
+ */
267
+ this.announcedPaused = new Map();
268
+ /**
269
+ * The verdict last emitted per gating edge. `GateDecision` frames announce CHANGES: edges are
270
+ * re-resolved every pass, and an unconditional emission would repeat every few seconds for as
271
+ * long as the graph lives — the frame-topic version of the log flood
272
+ * `logUnevaluableConditionOnce` exists to prevent.
273
+ *
274
+ * Nested by graph so a settled run's entries go in one delete. Flat per-edge maps in a process
275
+ * that runs for weeks are a slow leak with no upper bound but the table's size.
276
+ */
277
+ this.emittedGateVerdicts = new Map();
278
+ /** Last `NodeProgress` emission per task, for rate limiting chatty runners. Nested by graph. */
279
+ this.nodeProgressLastEmit = new Map();
66
280
  /** Name shown in the shutdown drain log. */
67
281
  this.ShutdownName = 'TaskGraphDispatcher';
68
282
  this.config = { ...DEFAULT_DISPATCHER_CONFIG, ...config };
@@ -86,6 +300,37 @@ export class TaskGraphDispatcher {
86
300
  LogError(`[TaskGraphDispatcher] Observer threw on ${frame.Kind} (ignored): ${e instanceof Error ? e.message : String(e)}`);
87
301
  }
88
302
  }
303
+ /**
304
+ * A progress sink for one task's runner, rate-limited into `NodeProgress` frames.
305
+ *
306
+ * Rate-limited HERE rather than asking every runner to be polite, for the same reason `emit`
307
+ * swallows observer throws in one place: a chatty runner (an agent streaming token-level
308
+ * updates) must not be able to flood the topic, and the limit belongs to the announcement, not
309
+ * the work. A 100% report always passes — the terminal update is the one a viewer must not lose.
310
+ */
311
+ nodeProgressEmitter(graphID, ownerUserID, taskID, taskName) {
312
+ return (message, percent) => {
313
+ const now = Date.now();
314
+ let perTask = this.nodeProgressLastEmit.get(graphID);
315
+ if (!perTask) {
316
+ perTask = new Map();
317
+ this.nodeProgressLastEmit.set(graphID, perTask);
318
+ }
319
+ const last = perTask.get(taskID) ?? 0;
320
+ if (percent !== 100 && now - last < TaskGraphDispatcher.NODE_PROGRESS_MIN_INTERVAL_MS)
321
+ return;
322
+ perTask.set(taskID, now);
323
+ this.emit({
324
+ Kind: 'NodeProgress',
325
+ ParentTaskID: graphID,
326
+ OwnerUserID: ownerUserID,
327
+ TaskID: taskID,
328
+ TaskName: taskName,
329
+ ProgressMessage: message,
330
+ ProgressPercent: percent,
331
+ });
332
+ };
333
+ }
89
334
  /**
90
335
  * Who a graph belongs to, memoized for the process's lifetime.
91
336
  *
@@ -104,18 +349,106 @@ export class TaskGraphDispatcher {
104
349
  const cached = this.ownerByParentID.get(parentTaskID);
105
350
  if (cached !== undefined)
106
351
  return cached;
107
- let owner = null;
108
352
  try {
109
353
  const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
110
- if (await parent.Load(parentTaskID)) {
111
- owner = this.readParentMetadata(parent).submittedByUserID ?? null;
354
+ if (!(await parent.Load(parentTaskID))) {
355
+ // NOT CACHED (C1). A failed load is not an answer, and caching it as one is
356
+ // permanent for the life of the process: the delivery filter fails closed on a null
357
+ // owner, so every frame for this graph reaches nobody until a restart. One
358
+ // transient blip, and a viewer watches a workflow that never appears to move.
359
+ LogError(`[TaskGraphDispatcher] Could not load graph ${parentTaskID} to resolve its owner; frames for it are unaddressed this pass.`);
360
+ return null;
112
361
  }
362
+ const owner = this.readParentMetadata(parent).submittedByUserID ?? null;
363
+ // A successfully-read graph with no owner IS an answer — a scheduled or remote-triggered
364
+ // graph legitimately has none — so that one caches.
365
+ this.ownerByParentID.set(parentTaskID, owner);
366
+ return owner;
113
367
  }
114
368
  catch (e) {
115
369
  LogError(`[TaskGraphDispatcher] Could not resolve owner for graph ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
370
+ return null;
371
+ }
372
+ }
373
+ /**
374
+ * A graph's debug state, cached for the current pass.
375
+ *
376
+ * Read fresh (BypassCache) because the state is written by direct `JSON_MODIFY` statements that
377
+ * fire no cache invalidation — the same reason every row read in this class bypasses the cache.
378
+ */
379
+ async readDebugState(provider, parentTaskID) {
380
+ const cached = this.debugStateByGraph.get(parentTaskID);
381
+ if (cached !== undefined)
382
+ return cached;
383
+ await this.primeDebugStates(provider, [parentTaskID]);
384
+ return this.debugStateByGraph.get(parentTaskID) ?? {};
385
+ }
386
+ /**
387
+ * Reads the debug bag for a set of graphs in ONE query, priming the per-pass cache.
388
+ *
389
+ * Batched because both loops in a pass — propagation and claiming — walk every active graph, so
390
+ * a per-graph read made observability cost scale with the number of live workflows on a path
391
+ * that is meant to be flat. One `ID IN (…)` per pass costs the same whether a server is running
392
+ * one workflow or fifty.
393
+ *
394
+ * Already-cached graphs are skipped, so calling this from both loops is free the second time.
395
+ */
396
+ async primeDebugStates(provider, parentTaskIDs) {
397
+ const missing = parentTaskIDs.filter((id) => !this.debugStateByGraph.has(id));
398
+ if (missing.length === 0)
399
+ return;
400
+ try {
401
+ const idList = missing.map((id) => `'${id}'`).join(',');
402
+ const rows = await RunView.FromMetadataProvider(provider).RunView({
403
+ EntityName: 'MJ: Tasks',
404
+ ExtraFilter: `ID IN (${idList})`,
405
+ Fields: ['ID', 'InputPayload'],
406
+ ResultType: 'simple',
407
+ // The bag is written by direct JSON_MODIFY statements, which fire no cache
408
+ // invalidation — a cached read here would gate on state a verb already changed.
409
+ BypassCache: true,
410
+ }, this.contextUser);
411
+ const byID = new Map((rows.Success ? rows.Results ?? [] : []).map((r) => [r.ID, r.InputPayload]));
412
+ for (const id of missing) {
413
+ this.debugStateByGraph.set(id, ParseTaskGraphDebugState(byID.get(id) ?? null));
414
+ }
415
+ }
416
+ catch (e) {
417
+ // "Not being debugged" is the safe reading of "could not read": gating real work on a
418
+ // transient read failure would turn a database hiccup into a paused workflow. Cached as
419
+ // empty for this pass only, so the next pass tries again.
420
+ LogError(`[TaskGraphDispatcher] Could not read debug state for ${missing.length} graph(s): ${e instanceof Error ? e.message : String(e)}`);
421
+ for (const id of missing)
422
+ this.debugStateByGraph.set(id, {});
116
423
  }
117
- this.ownerByParentID.set(parentTaskID, owner);
118
- return owner;
424
+ }
425
+ /**
426
+ * Announces a graph's pause-state TRANSITION, once, whichever instance notices first in its own
427
+ * frame stream.
428
+ *
429
+ * Per-instance dedup rather than a CAS: frames are advisory commentary, and a viewer receiving
430
+ * the transition from two instances is a duplicate line, not a duplicate execution — the price
431
+ * of a cross-instance guard here would be a write on every pass for a purely cosmetic guarantee.
432
+ */
433
+ async announcePauseTransition(provider, parentTaskID, debug) {
434
+ const paused = debug.paused === true;
435
+ const previous = this.announcedPaused.get(parentTaskID);
436
+ if (previous === paused)
437
+ return;
438
+ this.announcedPaused.set(parentTaskID, paused);
439
+ // First sighting of an unpaused graph needs no announcement — "running" is the default a
440
+ // viewer already assumes; only a transition is information.
441
+ if (previous === undefined && !paused)
442
+ return;
443
+ this.emit({
444
+ Kind: paused ? 'GraphPaused' : 'GraphResumed',
445
+ ParentTaskID: parentTaskID,
446
+ OwnerUserID: await this.resolveOwner(provider, parentTaskID),
447
+ TaskID: debug.pausedAtTaskID ?? undefined,
448
+ Reason: paused
449
+ ? (debug.pausedReason === 'breakpoint' ? 'breakpoint' : 'paused by user')
450
+ : 'resumed',
451
+ });
119
452
  }
120
453
  /**
121
454
  * Begins dispatching.
@@ -134,11 +467,72 @@ export class TaskGraphDispatcher {
134
467
  ShutdownRegistry.Instance.Register(this);
135
468
  LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
136
469
  await this.Reconcile();
470
+ // One wide pass over graphs that reached terminal without settling, mirroring what claim
471
+ // reconciliation above already does for tasks. The realistic producer of a >24h-stale
472
+ // unsettled graph is this process having been DOWN — an outage, a long deploy — which the
473
+ // steady-state window cannot see and which would otherwise leave those runs parked forever.
474
+ // Counted as a pass (R2-13). It settles graphs, delivers continuations and can start fresh
475
+ // reinvoke turns, and it runs AFTER this instance registers for shutdown — so a `Stop()`
476
+ // landing during it used to return immediately while the sweep carried on doing all of that
477
+ // against a host that believed the dispatcher had stopped.
478
+ this.activePasses++;
479
+ try {
480
+ await this.sweepUnsettledGraphs(UNSETTLED_STARTUP_WINDOW_HOURS);
481
+ }
482
+ finally {
483
+ this.activePasses--;
484
+ }
485
+ // A `Stop()` LANDING DURING THE BOOT AWAITS MUST NOT BE UNDONE HERE (R3-4).
486
+ //
487
+ // Everything above this line is awaited — reconciliation and the counted startup sweep,
488
+ // which R2-13's own fix makes `Stop()` wait out. So a host that shuts down during boot
489
+ // drains correctly, logs "Stopped.", and returns with both timer fields null — and then
490
+ // this continuation ran anyway and installed both timers on the stopped instance.
491
+ //
492
+ // `pollOnce` was inert (its own `running` guard), but `Reconcile` had no such guard: it
493
+ // minted a provider and ran `ReleaseExpiredClaims` — a real UPDATE returning tasks to
494
+ // Pending — every two minutes forever, against a pool the host may have torn down. Nothing
495
+ // would ever call `Stop()` again, since `ShutdownRegistry.ShutdownAll` clears its items
496
+ // after one pass, and the intervals pinned the event loop so the process could not exit.
497
+ if (!this.running) {
498
+ LogStatus(`[TaskGraphDispatcher] Stopped during startup; not installing timers.`);
499
+ return;
500
+ }
137
501
  this.pollTimer = setInterval(() => { void this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
138
- this.reconcileTimer = setInterval(() => { void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
502
+ this.unregisterKick = RegisterTaskGraphKick(() => { this.Kick(); });
503
+ // Do not wait a full interval for work that already exists (or is about to be submitted).
504
+ this.Kick();
505
+ this.reconcileTimer = setInterval(
506
+ // Guarded HERE rather than inside `Reconcile` (R3-4). The defect is a stopped
507
+ // instance's TIMER executing `ReleaseExpiredClaims` — a real UPDATE — forever; the
508
+ // public method itself stays callable, because reconciling on demand before starting is
509
+ // a legitimate use (IT74's crash-recovery check does exactly that) and a guard there
510
+ // would silently no-op it, which is the class of failure this whole effort is about.
511
+ () => { if (this.running)
512
+ void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
513
+ // Belt and suspenders: `Stop()` remains the real teardown, but an un-`unref`'d interval
514
+ // keeps the event loop alive on its own, so a leaked instance can prevent process exit.
515
+ this.pollTimer.unref?.();
516
+ this.reconcileTimer.unref?.();
139
517
  }
140
518
  /**
141
- * Stops accepting new work and waits for in-flight tasks to finish.
519
+ * Stops accepting new work and waits for everything already started to finish.
520
+ *
521
+ * **"Everything" includes the timer passes, and that is the fix.** This waited only on
522
+ * `inFlight` — the task executions — while a poll pass is a `void`-ed promise nothing held. So
523
+ * `Stop()` returned while a pass was mid-flight, and that pass went on to settle graphs, emit
524
+ * lifecycle frames and CLAIM NEW TASKS afterwards. Three consequences, all of them quiet:
525
+ *
526
+ * - a `GraphSettled` frame arrived after every subscriber had gone, so the settlement was
527
+ * invisible to exactly the viewer watching for it;
528
+ * - a process shutting down claimed work it was about to abandon, leaving claims to expire —
529
+ * the orphaned-claim state reconciliation exists to clean up, manufactured by the shutdown;
530
+ * - the host reused the connection the moment `Stop()` resolved, and the still-running pass's
531
+ * statements collided with it (`Requests can only be made in the LoggedIn state`).
532
+ *
533
+ * A pass is bookkeeping for work that already happened, so it is DRAINED rather than cancelled:
534
+ * abandoning one halfway is the crash window the unsettled sweep exists to rescue, and choosing
535
+ * to open it on every clean shutdown would be perverse.
142
536
  *
143
537
  * Deliberately does NOT release claims on the way out: an abandoned claim expires on its own,
144
538
  * and releasing eagerly would hand a still-running task to another instance mid-execution.
@@ -146,6 +540,8 @@ export class TaskGraphDispatcher {
146
540
  */
147
541
  async Stop() {
148
542
  this.running = false;
543
+ this.unregisterKick?.();
544
+ this.unregisterKick = null;
149
545
  if (this.pollTimer) {
150
546
  clearInterval(this.pollTimer);
151
547
  this.pollTimer = null;
@@ -154,12 +550,35 @@ export class TaskGraphDispatcher {
154
550
  clearInterval(this.reconcileTimer);
155
551
  this.reconcileTimer = null;
156
552
  }
157
- const deadline = Date.now() + 30_000;
158
- while (this.inFlight.size > 0 && Date.now() < deadline) {
159
- await new Promise((r) => setTimeout(r, 250));
553
+ // Short poll interval: a pass is usually milliseconds from done, and the old 250ms granularity
554
+ // was most of the cost of stopping a dispatcher that had nothing left to do.
555
+ const deadline = Date.now() + STOP_DRAIN_TIMEOUT_MS;
556
+ while ((this.activePasses > 0 || this.inFlight.size > 0) && Date.now() < deadline) {
557
+ await new Promise((r) => setTimeout(r, 25));
160
558
  }
161
559
  if (this.inFlight.size > 0) {
162
- LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will expire.`);
560
+ // The promise in this message was FALSE while the process lived (R2-13): each in-flight
561
+ // task heartbeats its own claim on its own timer, so an over-drain task renewed its lease
562
+ // indefinitely and the claim never expired — reconciliation could not reclaim the work,
563
+ // and the host's shutdown was waiting on something that had stopped being reclaimable.
564
+ // Stopping the heartbeats makes the sentence true. The task itself keeps running; its
565
+ // completion write is guarded on still owning the claim, so if another instance reclaims
566
+ // the task in the meantime, the abandoned executor's result is refused rather than raced.
567
+ // Latched, because the purge RACES the registration it is purging (C5). A task stalled
568
+ // in its two pre-heartbeat awaits — creating a provider, loading the row — registers
569
+ // AFTER this line and would renew its claim for the rest of the process's life, which is
570
+ // exactly the state the drain timeout means to end, under exactly the database duress
571
+ // that causes drain timeouts in the first place.
572
+ this.heartbeatsPurged = true;
573
+ for (const stop of this.heartbeats.values())
574
+ clearInterval(stop);
575
+ this.heartbeats.clear();
576
+ LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will now expire.`);
577
+ }
578
+ if (this.activePasses > 0) {
579
+ // Loud, because from here on this instance writes to a database the host believes it has
580
+ // finished with — the precise shape that produced connection-state errors downstream.
581
+ LogError(`[TaskGraphDispatcher] Stopped with ${this.activePasses} pass(es) still running; their writes may land after shutdown.`);
163
582
  }
164
583
  LogStatus(`[TaskGraphDispatcher] Stopped.`);
165
584
  }
@@ -175,6 +594,51 @@ export class TaskGraphDispatcher {
175
594
  * that shape indicates tampering or a bug and Record Changes already carries the audit trail.
176
595
  */
177
596
  async Reconcile() {
597
+ this.activePasses++;
598
+ try {
599
+ await this.reconcileOnce();
600
+ }
601
+ finally {
602
+ this.activePasses--;
603
+ }
604
+ }
605
+ /**
606
+ * Announces expired claims the sweep just released, so a viewer watching the graph sees "the
607
+ * step's worker vanished and the engine requeued it" as it happens.
608
+ *
609
+ * Best-effort by contract: the release already succeeded and is the durable truth; a frame that
610
+ * cannot be addressed (row unloadable, no parent) is dropped, never retried.
611
+ */
612
+ async announceReclaims(provider, released) {
613
+ if (!this.observer || released.length === 0)
614
+ return;
615
+ try {
616
+ const idList = released.map((r) => `'${r.TaskID}'`).join(',');
617
+ const rows = await RunView.FromMetadataProvider(provider).RunView({
618
+ EntityName: 'MJ: Tasks',
619
+ ExtraFilter: `ID IN (${idList})`,
620
+ Fields: ['ID', 'ParentID', 'Name'],
621
+ ResultType: 'simple',
622
+ BypassCache: true,
623
+ }, this.contextUser);
624
+ for (const row of (rows.Success ? rows.Results : []) ?? []) {
625
+ const graphID = row.ParentID ?? row.ID;
626
+ this.emit({
627
+ Kind: 'ClaimChanged',
628
+ ParentTaskID: graphID,
629
+ OwnerUserID: await this.resolveOwner(provider, graphID),
630
+ TaskID: row.ID,
631
+ TaskName: row.Name,
632
+ ClaimEvent: 'reclaimed',
633
+ });
634
+ }
635
+ }
636
+ catch (e) {
637
+ LogError(`[TaskGraphDispatcher] Could not announce reclaimed task(s) (ignored): ${e instanceof Error ? e.message : String(e)}`);
638
+ }
639
+ }
640
+ /** The reconciliation body. Wrapped by {@link Reconcile} so `Stop()` can drain it. */
641
+ async reconcileOnce() {
178
642
  let provider = null;
179
643
  try {
180
644
  provider = await this.providerFactory.CreateProvider();
@@ -184,11 +648,21 @@ export class TaskGraphDispatcher {
184
648
  LogStatus(`[TaskGraphDispatcher] Reconciliation: ${released.length} expired claim(s) released, ` +
185
649
  `${orphaned.length} orphaned task(s) reported.`);
186
650
  }
651
+ await this.announceReclaims(provider, released);
187
652
  }
188
653
  catch (e) {
189
654
  LogError(`[TaskGraphDispatcher] Reconciliation failed: ${e instanceof Error ? e.message : String(e)}`);
190
655
  }
191
656
  }
657
+ /**
658
+ * Run a pass now instead of waiting for the next poll tick.
659
+ *
660
+ * Submit calls this (via {@link KickTaskGraphDispatchers}) so a just-written graph is claimed
661
+ * in milliseconds rather than up to {@link TaskGraphDispatcherConfig.PollIntervalSeconds}.
662
+ */
663
+ Kick() {
664
+ void this.pollOnce();
665
+ }
192
666
  /**
193
667
  * One dispatch pass: find claimable work, claim what fits under the concurrency cap, execute.
194
668
  *
@@ -198,33 +672,86 @@ export class TaskGraphDispatcher {
198
672
  async pollOnce() {
199
673
  if (!this.running || this.polling)
200
674
  return;
201
- const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
202
- if (capacity <= 0)
203
- return;
204
675
  this.polling = true;
676
+ this.activePasses++;
677
+ // One pass, one read of each graph's debug state — see `readDebugState`.
678
+ this.debugStateByGraph.clear();
679
+ const passNumber = ++this.passCounter;
205
680
  try {
206
681
  const provider = await this.providerFactory.CreateProvider();
207
- // Settle graphs before picking new work, so a failure earlier in this pass stops its
208
- // branch immediately rather than after another wave has already launched.
682
+ // `running` is re-read after every await from here on. The entry check above only proves
683
+ // the dispatcher was live when the tick fired; each await is a point where `Stop` can
684
+ // land, and a stopped instance must neither mutate graph state nor take new work. Left
685
+ // unchecked, a stopped dispatcher goes on to roll up graphs (emitting GraphSettled to an
686
+ // observer nobody is listening to any more) and to claim tasks it will never run — which
687
+ // then sit claimed until their lease expires.
688
+ if (!this.running)
689
+ return;
690
+ // SETTLEMENT IS NOT GATED ON CAPACITY (R2-11).
691
+ //
692
+ // This used to return at `capacity <= 0` before reaching the rollup, so a handful of
693
+ // wedged long-running tasks froze EVERYTHING for the whole instance: no settlement, no
694
+ // skip or block propagation, no human-task settlement, no continuation delivery — for
695
+ // graphs that had nothing to do with the tasks holding the slots. A per-task hang is an
696
+ // accepted limitation; "one hung task stops every workflow on this host" is not, and the
697
+ // two were the same line of code.
698
+ //
699
+ // Only CLAIMING consumes capacity, because only claiming starts work.
209
700
  await this.propagateAndRollup(provider);
210
- const candidates = await this.findClaimableTasks(provider, capacity);
701
+ const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
702
+ if (capacity <= 0)
703
+ return;
704
+ // The rollup above can take seconds, and `Stop()` may have been called during it. Claiming
705
+ // now would start work the process has already decided to abandon — the claim then sits
706
+ // until its TTL expires and another instance reclaims it. Settling first and checking
707
+ // here is the right order: bookkeeping for finished work always completes, new work never
708
+ // starts after the decision to stop.
709
+ if (!this.running)
710
+ return;
711
+ const { tasks: candidates, stats } = await this.findClaimableTasks(provider, capacity);
712
+ const claimedByGraph = new Map();
211
713
  for (const task of candidates) {
714
+ // Re-checked EVERY iteration, not once before the loop (R2-13). Claiming is itself
715
+ // awaited, so a multi-task wave can straddle a `Stop`; and `findClaimableTasks` loads
716
+ // and resolves every active graph, so the scan before this loop can run for seconds.
717
+ // Unchecked, a shutting-down process takes ownership of work it is about to abandon,
718
+ // manufacturing the orphaned claims reconciliation exists to clean up.
719
+ if (!this.running)
720
+ break;
212
721
  if (this.inFlight.size >= this.config.MaxConcurrentTasks)
213
722
  break;
214
723
  if (!(await this.claims.TryClaim(provider, task.ID, this.contextUser))) {
215
724
  // Another instance won the race, or the task is no longer Pending. Normal.
216
725
  continue;
217
726
  }
727
+ const graphID = task.ParentID ?? task.ID;
728
+ claimedByGraph.set(graphID, (claimedByGraph.get(graphID) ?? 0) + 1);
218
729
  this.inFlight.add(task.ID);
219
730
  // Intentionally not awaited — the poll loop must keep dispatching while this runs.
220
731
  void this.executeClaimed(task.ID).finally(() => this.inFlight.delete(task.ID));
221
732
  }
733
+ // The engine's heartbeat, per watched graph: what was ready, what was held, what this
734
+ // instance took. A stuck run is a strip of these ticking with nothing moving, which is
735
+ // the honest visual of a stall — and the reason this frame exists.
736
+ for (const [graphID, s] of stats) {
737
+ this.emit({
738
+ Kind: 'PassCompleted',
739
+ ParentTaskID: graphID,
740
+ OwnerUserID: await this.resolveOwner(provider, graphID),
741
+ PassNumber: passNumber,
742
+ EligibleCount: s.eligible,
743
+ HeldCount: s.held,
744
+ ClaimedCount: claimedByGraph.get(graphID) ?? 0,
745
+ InstanceInFlightCount: this.inFlight.size,
746
+ });
747
+ }
222
748
  }
223
749
  catch (e) {
224
750
  LogError(`[TaskGraphDispatcher] Poll failed: ${e instanceof Error ? e.message : String(e)}`);
225
751
  }
226
752
  finally {
227
753
  this.polling = false;
754
+ this.activePasses--;
228
755
  }
229
756
  }
230
757
  /**
@@ -242,19 +769,44 @@ export class TaskGraphDispatcher {
242
769
  LogError(`[TaskGraphDispatcher] Claimed task ${taskID} could not be loaded.`);
243
770
  return;
244
771
  }
772
+ // Emitted after the claim is held, not before: a frame saying "started" for work another
773
+ // instance actually took would be a lie a viewer cannot detect.
774
+ const graphID = task.ParentID ?? taskID;
775
+ const ownerUserID = await this.resolveOwner(provider, graphID);
245
776
  heartbeat = setInterval(() => {
246
777
  void this.claims.Heartbeat(provider, taskID, this.contextUser).then((ok) => {
247
778
  if (!ok) {
248
779
  // Lost ownership — reconciliation reclaimed it, or a human intervened.
249
780
  LogError(`[TaskGraphDispatcher] Lost claim on task ${taskID} while executing; another instance may take it over.`);
781
+ // Announced so a viewer sees "this step's worker lost its lease" the moment
782
+ // it happens instead of discovering it in a forensic query later — the R2-1
783
+ // wedge class, made visible.
784
+ this.emit({
785
+ Kind: 'ClaimChanged', ParentTaskID: graphID, OwnerUserID: ownerUserID,
786
+ TaskID: taskID, TaskName: task.Name,
787
+ ClaimEvent: 'heartbeat-lost', ClaimedBy: this.config.InstanceID,
788
+ });
250
789
  }
251
790
  });
252
791
  }, this.config.HeartbeatIntervalSeconds * 1000);
253
- // Emitted after the claim is held, not before: a frame saying "started" for work another
254
- // instance actually took would be a lie a viewer cannot detect.
255
- const graphID = task.ParentID ?? taskID;
256
- const ownerUserID = await this.resolveOwner(provider, graphID);
792
+ // Registered so `Stop()` can reach it — unless the drain has already given up, in which
793
+ // case this task arrived too late to be waited for and must not renew its lease (C5).
794
+ // Its completion write stays guarded, so if another instance reclaims the task in the
795
+ // meantime this executor's result is refused rather than raced.
796
+ if (this.heartbeatsPurged || !this.running) {
797
+ clearInterval(heartbeat);
798
+ heartbeat = null;
799
+ }
800
+ else {
801
+ this.heartbeats.set(taskID, heartbeat);
802
+ }
257
803
  this.emit({ Kind: 'TaskStarted', ParentTaskID: graphID, OwnerUserID: ownerUserID, TaskID: taskID, TaskName: task.Name, Status: 'In Progress' });
804
+ this.emit({
805
+ Kind: 'ClaimChanged', ParentTaskID: graphID, OwnerUserID: ownerUserID,
806
+ TaskID: taskID, TaskName: task.Name,
807
+ ClaimEvent: 'claimed', ClaimedBy: this.config.InstanceID,
808
+ ClaimExpiresAt: new Date(Date.now() + this.config.ClaimTTLSeconds * 1000).toISOString(),
809
+ });
258
810
  const dependencyOutputs = await this.loadDependencyOutputs(provider, taskID);
259
811
  let inputPayload = null;
260
812
  if (task.InputPayload) {
@@ -265,19 +817,24 @@ export class TaskGraphDispatcher {
265
817
  LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
266
818
  }
267
819
  }
268
- const result = await this.agentRunner.RunAgentForTask({
269
- TaskID: taskID,
270
- AgentID: task.AgentID,
271
- InputPayload: inputPayload,
272
- DependencyOutputs: dependencyOutputs,
273
- Provider: provider,
274
- ContextUser: this.contextUser,
275
- });
820
+ const onProgress = this.nodeProgressEmitter(graphID, ownerUserID, taskID, task.Name);
821
+ const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs, onProgress);
822
+ // ONLY THE CONFIRMED OWNER MUTATES THE GRAPH (R2-10).
823
+ //
824
+ // The early-finish skips used to run BEFORE this, so a lapsed claim produced the worst
825
+ // possible pair: the siblings were terminally Skipped and satisfying dependents, while
826
+ // the completion was refused and the task re-ran on another instance — where it might
827
+ // not end early at all. The graph would then be missing steps nobody decided to skip.
828
+ //
829
+ // Recording first costs a poll: the skips now land after the completion, so a rollup
830
+ // that lands in between sees work still Pending and settles one pass later. That is a
831
+ // delay; the other order was a wrong graph.
276
832
  const recorded = await this.claims.CompleteClaimed(provider, taskID, {
277
833
  Status: result.Success ? 'Complete' : 'Failed',
278
834
  OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
279
835
  ErrorMessage: result.ErrorMessage ?? null,
280
836
  AgentRunID: result.AgentRunID ?? null,
837
+ Configuration: this.configurationWithRuntime(task, result.PromptRunID, result.ActionLogID, result.Iterations, result.PayloadAtStart),
281
838
  }, this.contextUser);
282
839
  if (!recorded) {
283
840
  // The guarded write refused: the row changed underneath us (cancelled, reassigned,
@@ -297,6 +854,13 @@ export class TaskGraphDispatcher {
297
854
  Status: result.Success ? 'Complete' : 'Failed',
298
855
  ErrorMessage: result.Success ? undefined : (result.ErrorMessage ?? undefined),
299
856
  });
857
+ // A prompt can end the workflow early and say why — honoured only now that this
858
+ // instance is the confirmed owner of the outcome. The remaining tasks are Skipped
859
+ // here so the graph settles Complete rather than looking abandoned with work left
860
+ // Pending; a rollup that lands between the two simply settles one pass later.
861
+ if (result.ChatMessage) {
862
+ await this.endGraphEarly(provider, task, result.ChatMessage);
863
+ }
300
864
  }
301
865
  }
302
866
  catch (e) {
@@ -310,6 +874,7 @@ export class TaskGraphDispatcher {
310
874
  finally {
311
875
  if (heartbeat)
312
876
  clearInterval(heartbeat);
877
+ this.heartbeats.delete(taskID);
313
878
  }
314
879
  }
315
880
  /**
@@ -318,12 +883,80 @@ export class TaskGraphDispatcher {
318
883
  * All four decisions — what is eligible, what must block, what the parent status is, whether the
319
884
  * graph is wedged — are delegated to the pure algorithms, unchanged from Phase 1.
320
885
  */
321
- async propagateAndRollup(provider) {
322
- for (const parentID of await this.findActiveGraphIDs(provider)) {
323
- const graph = await this.loadGraphState(provider, parentID);
886
+ async propagateAndRollup(provider, graphIDs) {
887
+ const graphs = graphIDs ?? await this.findActiveGraphIDs(provider);
888
+ // One read for every graph this pass touches, rather than one per graph — see primeDebugStates.
889
+ await this.primeDebugStates(provider, graphs);
890
+ for (const parentID of graphs) {
891
+ // Human steps settle BEFORE the graph state is read, so an answer given since the last
892
+ // poll is already reflected when eligibility and rollup are computed. Doing it after
893
+ // would delay every dependent branch by a full poll interval for no reason — and on a
894
+ // graph whose only remaining work is downstream of a person, that is the difference
895
+ // between "answered and moving" and "answered and apparently still stuck".
896
+ await this.expireOverdueRequests(provider, parentID);
897
+ await this.settleAnsweredHumanTasks(provider, parentID);
898
+ await this.reconcileWaitingHumanTasks(provider, parentID);
899
+ // Edge overrides apply to propagation exactly as they apply to claiming — a branch the
900
+ // operator answered 'false' must cascade its skips here, not merely stop being claimed.
901
+ const debug = await this.readDebugState(provider, parentID);
902
+ const graph = await this.loadGraphState(provider, parentID, debug);
324
903
  if (graph.nodes.length === 0)
325
904
  continue;
326
- const toBlock = new Set([...ComputeTasksToBlock(graph.nodes, graph.edges), ...graph.unreachableTaskIDs]);
905
+ // SKIPS FIRST — before blocking, before eligibility. A task whose gating predecessors
906
+ // are all Skipped is simultaneously "eligible" (Skipped satisfies a prerequisite) and
907
+ // "to be skipped"; deciding eligibility first would dispatch the branch nobody took.
908
+ //
909
+ // `unreachableTaskIDs` seeds this too, and that is a correction (R6). A target whose only
910
+ // route in was an ordinary conditional edge that evaluated DEFINITELY FALSE is a branch
911
+ // that was not taken — semantically identical to an XOR loser — yet it used to settle
912
+ // `Blocked`. That made `Blocked` mean two unrelated things: "the workflow chose another
913
+ // route" and "something upstream broke". A reader cannot tell those apart, so every
914
+ // conditional workflow looked half-failed and people went hunting for bugs that did not
915
+ // exist. `Blocked` is now reserved for FAILURE-driven unsatisfiability.
916
+ // Computed once in `loadGraphState` so the claim filter sees the same set this pass is
917
+ // about to write — see R2-14 there.
918
+ const toSkip = graph.cascadeSkipTaskIDs;
919
+ const skippedByRoute = [];
920
+ for (const taskID of toSkip) {
921
+ const entity = graph.entityById.get(taskID);
922
+ if (!entity || entity.Status !== 'Pending')
923
+ continue;
924
+ // The in-memory check above is a cheap pre-filter; the guard that matters is IN the
925
+ // statement (R3-1's audit item). This snapshot was loaded at the top of the pass and
926
+ // a task can be claimed and started before its skip write lands — R2-14 closed that
927
+ // window for the claim filter, and this closes it for the write itself.
928
+ const skipTypeID = await this.workflowTaskTypeID(provider);
929
+ if (skipTypeID && await this.claims.TrySkipPending(provider, taskID, skipTypeID, this.contextUser)) {
930
+ entity.Status = 'Skipped';
931
+ skippedByRoute.push(taskID);
932
+ LogStatus(`[TaskGraphDispatcher] Skipped '${entity.Name}' (${taskID}) — another branch was taken.`);
933
+ // Announced separately from TaskBlocked because it means something different to
934
+ // a viewer: nothing went wrong, this route simply was not the one chosen.
935
+ this.emit({
936
+ Kind: 'TaskSkipped',
937
+ ParentTaskID: parentID,
938
+ OwnerUserID: await this.resolveOwner(provider, parentID),
939
+ TaskID: taskID,
940
+ TaskName: entity.Name,
941
+ Status: 'Skipped',
942
+ });
943
+ // Keep the in-memory graph consistent so the blocking pass below and the rollup
944
+ // both see the skip rather than a stale Pending.
945
+ const node = graph.nodes.find((n) => n.id === taskID);
946
+ if (node)
947
+ node.status = 'Skipped';
948
+ }
949
+ }
950
+ // A human step reached by a route the workflow did not take has the same zombie request
951
+ // as one skipped by an early finish (R2-10): notified, `Requested` forever, and invisible
952
+ // to the settle and expiry sweeps because they filter on Pending tasks and this one is
953
+ // not Pending any more. Same treatment, different reason.
954
+ await this.withdrawOpenRequests(provider, skippedByRoute, 'The workflow took a different route, so this step is no longer needed.');
955
+ // Only failure-driven unsatisfiability reaches here now; not-taken branches were skipped
956
+ // above. A task already Skipped is left alone rather than overwritten — the two passes
957
+ // must not fight over the same row.
958
+ const toBlock = [...ComputeTasksToBlock(graph.nodes, graph.edges, graph.handledFailureIDs)]
959
+ .filter((id) => !toSkip.has(id));
327
960
  for (const taskID of toBlock) {
328
961
  const entity = graph.entityById.get(taskID);
329
962
  if (!entity)
@@ -343,42 +976,487 @@ export class TaskGraphDispatcher {
343
976
  });
344
977
  }
345
978
  }
346
- if (IsGraphStalled(graph.nodes, graph.edges)) {
979
+ // Holds are passed in, or the detector reports a held graph as healthy: a held target's
980
+ // gating edge is still live and its origin Complete, so ComputeEligibleTasks counts it
981
+ // as eligible and "something is eligible" reads as "not stalled". A graph waiting
982
+ // forever on a broken condition then produced no diagnostics at all.
983
+ if (IsGraphStalled(graph.nodes, graph.edges, graph.holdTaskIDs)) {
347
984
  LogError(`[TaskGraphDispatcher] Graph ${parentID} is stalled: pending work with no satisfiable path.`);
348
985
  }
349
- const fresh = await this.loadGraphState(provider, parentID);
986
+ const fresh = await this.loadGraphState(provider, parentID, debug);
350
987
  // ComputeParentRollup treats an empty child set as Complete-and-terminal, which is right
351
988
  // for a graph that genuinely has no children and catastrophic for one whose reload came
352
989
  // back empty transiently — it would mark live work finished and fire its continuation.
353
990
  // The outer guard covered the first load only.
354
991
  if (fresh.nodes.length === 0)
355
992
  continue;
356
- const rollup = ComputeParentRollup(fresh.nodes);
993
+ const rollup = ComputeParentRollup(fresh.nodes, fresh.handledFailureIDs);
357
994
  const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
358
995
  if (!(await parent.Load(parentID)))
359
996
  continue;
360
- if (parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
997
+ // A graph starts when its first step does.
998
+ //
999
+ // `StartedAt` is stamped by the CLAIM, and a parent is never claimed — it is a container,
1000
+ // not a unit of work — so the graph row carried no start time even after it completed.
1001
+ // A settled workflow therefore reported a CompletedAt with no beginning: it sorted as
1002
+ // "not started" in the run tree, showed no timestamp, and no duration could be computed
1003
+ // for the thing whose duration people actually ask about.
1004
+ //
1005
+ // Taken from the earliest child rather than from the clock, because that is when work
1006
+ // genuinely began — a graph can sit Pending for a long time between submission (already
1007
+ // recorded as CreatedAt) and a dispatcher picking up its first task.
1008
+ // Column-scoped for the same reason the terminal write is: a full-row save here would
1009
+ // carry this instance's `InputPayload` snapshot and could erase a continuation marker
1010
+ // another instance had just claimed. Guarded on `StartedAt IS NULL`, so calling it on
1011
+ // every pass is free.
1012
+ const earliestChildStart = this.earliestStart(fresh.entityById);
1013
+ if (parent.StartedAt == null && earliestChildStart != null) {
1014
+ await this.claims.TryStampParentStart(provider, parentID, earliestChildStart, this.contextUser);
1015
+ parent.StartedAt = earliestChildStart;
1016
+ }
1017
+ // THE TERMINAL WRITE IS GUARDED AND COLUMN-SCOPED, not a full-row save.
1018
+ //
1019
+ // `GenerateSaveSQL` sends every updateable column on every save, so a full-row save
1020
+ // carries the whole in-memory snapshot — including `InputPayload`, where the continuation
1021
+ // marker lives. Two instances both compute the terminal rollup; if one claims the marker
1022
+ // and the other then saves its pre-marker snapshot, the marker is ERASED and the
1023
+ // settlement delivers twice. For `reinvoke` that is a second billed agent turn.
1024
+ //
1025
+ // Guarding on "not already terminal" also makes the write idempotent across the
1026
+ // re-entrant settle path below, and replaces an unchecked `Save()` whose failure left the
1027
+ // graph active — re-emitting frames and recomputing cost every poll, forever.
1028
+ if (rollup.outcome === 'settled') {
1029
+ const settled = await this.claims.TrySettleParent(provider, parentID, rollup.status, rollup.percentComplete, this.contextUser);
1030
+ if (!settled && !TERMINAL_TASK_STATUSES.has(parent.Status)) {
1031
+ // Neither "already terminal" nor a successful write: the statement failed. Leave
1032
+ // the graph active so the next pass retries rather than settling on a status the
1033
+ // database never accepted. Re-queued explicitly, because a failed write is
1034
+ // exactly the case where the row's own timestamp does not advance.
1035
+ LogError(`[TaskGraphDispatcher] Could not write terminal status for graph ${parentID}; leaving it active to retry.`);
1036
+ this.keepRetryingSettlement(parentID);
1037
+ continue;
1038
+ }
361
1039
  parent.Status = rollup.status;
362
- parent.PercentComplete = rollup.percentComplete;
363
- if (rollup.isTerminal)
364
- parent.CompletedAt = new Date();
365
- await parent.Save();
366
- }
367
- if (rollup.isTerminal) {
368
- // Emitted before the continuation is delivered, and outside its once-only guard: a
369
- // viewer watching the run should learn it finished whether or not this instance is
370
- // the one that wins the delivery CAS.
1040
+ }
1041
+ else if (parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
1042
+ // Guarded and column-scoped for the same reason the terminal write is — and the race
1043
+ // here needs no exotic timing. This instance may have computed a non-terminal rollup
1044
+ // from a snapshot taken before another instance settled the graph; a full-row save
1045
+ // would then REVERT the status and erase the continuation marker with it, and the
1046
+ // next pass would settle and deliver a second time. See TryUpdateParentProgress.
1047
+ await this.claims.TryUpdateParentProgress(provider, parentID, rollup.status, rollup.percentComplete, this.contextUser);
1048
+ }
1049
+ if (rollup.outcome === 'settled') {
1050
+ // ANNOUNCE ONCE PER PROCESS, RETRY THE REST (R2-12). Layout and the frame are
1051
+ // idempotent facts about a finished graph; the cost, lifecycle and delivery writes
1052
+ // below are the ones re-entry exists to retry. Without this split, a graph that keeps
1053
+ // failing to settle re-persisted geometry and re-emitted the same frame every poll
1054
+ // for the whole rescue window.
1055
+ if (!this.announcedSettlements.has(parentID)) {
1056
+ // Geometry is settled once, here, so every viewer of this run agrees on it.
1057
+ await this.persistComputedLayout(fresh);
1058
+ // Emitted before the continuation is delivered, and outside its once-only guard: a
1059
+ // viewer watching the run should learn it finished whether or not this instance is
1060
+ // the one that wins the delivery CAS.
1061
+ this.emit({
1062
+ Kind: 'GraphSettled',
1063
+ ParentTaskID: parentID,
1064
+ OwnerUserID: await this.resolveOwner(provider, parentID),
1065
+ Status: rollup.status,
1066
+ CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
1067
+ TotalCount: fresh.nodes.length,
1068
+ });
1069
+ this.announcedSettlements.add(parentID);
1070
+ }
1071
+ // READ-ONLY GATE, before any write to the submitting run's half (R2-2).
1072
+ //
1073
+ // A graph can settle before the run that submitted it has parked at all. `BaseAgent`
1074
+ // sets `Paused` in `finalizeAgentRun`, AFTER the graph is durable and dispatchable —
1075
+ // so a fast graph finishes first, and both writes below then land wrong: the
1076
+ // lifecycle write silently returns (its guard is `Status === 'Paused'`), and the cost
1077
+ // write is overwritten moments later by finalize's own full-row save, which carries
1078
+ // the in-memory nulls it had before the dispatcher wrote anything.
1079
+ //
1080
+ // Deferring the whole half — rather than doing the parts that happen to work — is
1081
+ // what keeps the marker honest: nothing below claims it, so the graph stays
1082
+ // terminal-and-undelivered and the rescue sweep brings it back next pass, by which
1083
+ // time finalize has parked the run and both writes land.
1084
+ const readiness = await this.submittingRunReadiness(provider, parent);
1085
+ if (readiness.Verdict === 'defer') {
1086
+ this.keepRetryingSettlement(parentID);
1087
+ continue;
1088
+ }
1089
+ // A settled graph will not produce another verdict, claim or progress report, so the
1090
+ // per-graph observability caches for it are dead weight from here. Unbounded, they
1091
+ // are a slow leak in a process that runs for weeks — one that grows with every
1092
+ // workflow the server has ever seen rather than with anything live. (A deferred
1093
+ // settlement below repopulates them naturally on the retry pass.)
1094
+ this.forgetGraphObservability(parentID);
1095
+ // R3-8: the rollup gets a verdict, and a TRANSIENT failure defers exactly as a
1096
+ // failed lifecycle write does. Continuing past one would settle the run and claim
1097
+ // the marker, which permanently excludes the graph from the rescue sweep — making
1098
+ // the rollup's own "retrying on a later settlement" log a promise it could not keep.
1099
+ // Permanent refusals (truncated tree, graph not in the tree) proceed as before:
1100
+ // those do not clear on their own, and deferring on them would stall forever.
1101
+ if (await this.rollUpCostToSubmittingRun(provider, parent) === 'failed-transient') {
1102
+ this.keepRetryingSettlement(parentID);
1103
+ continue;
1104
+ }
1105
+ // Deliberately AFTER the rollup and OUTSIDE its refusal paths. The rollup declines
1106
+ // to write a number it cannot stand behind — a truncated tree, an unreachable graph
1107
+ // — and every one of those returns early. If the run's lifecycle were settled in
1108
+ // there, a refused rollup would strand the run parked forever, which is a far worse
1109
+ // failure than a missing cost figure. Cost and lifecycle are separate concerns with
1110
+ // separate failure modes, so they get separate writes.
1111
+ if (await this.settleSubmittingRun(provider, parent, rollup.status) === 'defer') {
1112
+ // The lifecycle write did not land. Delivering now would claim the marker and
1113
+ // make this the LAST pass to look at the graph — leaving the run Paused forever,
1114
+ // which is the exact permanence R2-2 removes. Leave the marker unset and retry.
1115
+ this.keepRetryingSettlement(parentID);
1116
+ continue;
1117
+ }
1118
+ if (await this.deliverContinuation(provider, parent, fresh, readiness.SubmitterCancelled)) {
1119
+ // Delivered, expired, or lost the CAS to a peer — every one of those means this
1120
+ // graph is somebody's finished business and needs nothing further from here.
1121
+ this.retryingSettlement.delete(parentID);
1122
+ this.announcedSettlements.delete(parentID);
1123
+ }
1124
+ else {
1125
+ // This instance cannot deliver. Stay quiet about it — the frame is already out —
1126
+ // and leave the graph for a capable peer via the sweep.
1127
+ this.keepRetryingSettlement(parentID);
1128
+ }
1129
+ }
1130
+ }
1131
+ }
1132
+ /**
1133
+ * Credits a finished graph's spending back to the agent run that submitted it.
1134
+ *
1135
+ * **Why this cannot happen during the run.** `BaseAgent` totals a run by walking its steps in
1136
+ * memory at finalization — but a submitting run *ends at submission*. Submit-and-detach is the
1137
+ * point: the run returns as soon as the graph is durable, and the graph executes afterwards,
1138
+ * possibly minutes later on a different instance. At the moment the run computes its totals the
1139
+ * spending has not happened yet, so there is nothing to count. The only place the number can be
1140
+ * known is here, when the graph settles.
1141
+ *
1142
+ * **Why the `…Rollup` columns and not the plain ones.** `AIAgentRun` has carried six `…Rollup`
1143
+ * columns since v3 that nothing has ever written — they exist for exactly this distinction:
1144
+ *
1145
+ * - `TotalCost` — what the run itself spent. For a Flow agent that is genuinely near zero: it
1146
+ * compiled a graph and handed it off. This value is already final and is never rewritten here,
1147
+ * so nothing that reads it today changes meaning, and no guardrail that already evaluated
1148
+ * against it is retroactively falsified.
1149
+ * - `TotalCostRollup` — the run plus everything it caused. Provisional until the graph settles,
1150
+ * which is now.
1151
+ *
1152
+ * **The tree is the authority; these columns are its settlement-time cache.** The total is a SUM
1153
+ * over `GetAgentRunTree`, not arithmetic of its own. The previous version walked the graph's
1154
+ * child tasks and added each one's agent run, which was wrong in two ways that no test could
1155
+ * see: a `Prompt` task has no agent run at all, so every prompt step's spend was simply missing;
1156
+ * and it read each nested run's `…Rollup ?? …Total`, mixing a descendant-inclusive number with an
1157
+ * own-spend one and depending on whether that nested graph happened to have settled yet. The
1158
+ * tree already models every one of those cases — it reaches prompt runs through
1159
+ * `Configuration.runtime.promptRunID`, and it descends into nested runs and their graphs
1160
+ * structurally — so summing it cannot disagree with what the run viewer shows, because it IS
1161
+ * what the run viewer shows.
1162
+ *
1163
+ * **This refuses rather than guesses.** A tree that failed to load, hit the depth cap, or does
1164
+ * not contain the settling graph would still produce a number — a lower bound. Writing one would
1165
+ * put an authoritative-looking total in a column every cost surface reads. Each of those cases
1166
+ * logs and leaves the column alone, so `?? TotalCost` keeps its honest meaning: not settled.
1167
+ *
1168
+ * A graph with no submitting run (a scheduled job, a remote-operation caller) simply has nobody
1169
+ * to credit — its own Task rows still carry the truth, and this returns quietly.
1170
+ */
1171
+ async rollUpCostToSubmittingRun(provider, parent) {
1172
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
1173
+ if (!meta.submittedByAgentRunID)
1174
+ return 'landed';
1175
+ const runID = meta.submittedByAgentRunID;
1176
+ try {
1177
+ const runQuery = asRunQueryProvider(provider);
1178
+ if (!runQuery) {
1179
+ // Permanent for this host: a provider that cannot run queries will not grow the
1180
+ // ability on the next pass, so deferring would stall the graph forever.
1181
+ LogError(`[TaskGraphDispatcher] Cannot roll up cost for run ${runID}: provider cannot run queries.`);
1182
+ return 'refused-permanent';
1183
+ }
1184
+ const tree = await LoadAgentRunTree(runID, runQuery, this.contextUser);
1185
+ // Each of these means the sum would be a LOWER BOUND, and the column's whole contract is
1186
+ // that it equals the tree. A known-low number presented as a total is worse than no
1187
+ // number: the readers all fall back to TotalCost when this is null, which at least
1188
+ // *says* it is the run's own spend rather than claiming to be the whole story.
1189
+ //
1190
+ // Refusing is NOT the same as leaving the column alone. A run that submitted two graphs
1191
+ // has a rollup from the first; if the second cannot be summed, the first graph's total
1192
+ // sits in the authoritative column excluding work that has since happened — stale, not
1193
+ // absent, and `?? TotalCost` cannot save a reader from a non-null wrong number. So a
1194
+ // refusal CLEARS it, restoring the fallback's honest meaning: not settled.
1195
+ if (tree.ErrorMessage || !tree.Root) {
1196
+ // TRANSIENT — nothing cleared (R2-15), and now nothing delivered either (R3-8).
1197
+ //
1198
+ // R2-15 stopped this path erasing a correct total. What it did not stop was the pass
1199
+ // CONTINUING: settlement flipped the run terminal and delivery claimed the marker,
1200
+ // which permanently excludes the graph from the rescue sweep — so this function's
1201
+ // own promise of "retrying on a later settlement" was structurally impossible to
1202
+ // keep. One transient DB error, no interleaving, and a multi-graph run kept a wrong
1203
+ // non-null authoritative total forever while a first-graph run stayed null.
1204
+ LogError(`[TaskGraphDispatcher] Could not load the run tree for ${runID} to roll up graph ` +
1205
+ `${parent.ID}: ${tree.ErrorMessage ?? 'the run tree came back empty'}. Leaving any ` +
1206
+ `existing rollup alone and deferring settlement so a later pass can retry.`);
1207
+ return 'failed-transient';
1208
+ }
1209
+ if (tree.Truncated) {
1210
+ // PERMANENT: the tree is genuinely too deep, and it will be just as deep next pass.
1211
+ await this.clearStaleRollup(provider, runID, `the run tree hit the depth cap, so any total would silently under-report ` +
1212
+ `(graph ${parent.ID} still carries its own costs)`);
1213
+ return 'refused-permanent';
1214
+ }
1215
+ // The graph that just settled must appear in the tree. If it does not, the tree stopped
1216
+ // at the run — the submitting step never recorded its parentTaskID — and the sum is
1217
+ // merely the run's own spend wearing the name of a rollup. That is precisely the silent
1218
+ // under-count this rewrite exists to remove, so it is reported rather than written.
1219
+ if (!this.treeContainsGraph(tree.Root, parent.ID)) {
1220
+ // PERMANENT: a missing parentTaskID link is a fact about how the graph was
1221
+ // submitted, not a condition that clears on its own.
1222
+ await this.clearStaleRollup(provider, runID, `graph ${parent.ID} is not reachable from it, so the tree cannot see the work. ` +
1223
+ `Did the submitting step record parentTaskID?`);
1224
+ return 'refused-permanent';
1225
+ }
1226
+ const totals = SumAgentRunTreeCost(tree.Root);
1227
+ // Assignment, never accumulation. The tree already contains the run's own spend as its
1228
+ // ROOT node, and it reads own-cost everywhere, so recomputing from scratch on every
1229
+ // settlement lands on the same answer — which is what makes this safe to call again when
1230
+ // a second graph settles, or when the terminal check is re-evaluated after a HITL wait.
1231
+ //
1232
+ // COLUMN-SCOPED (C4). A full-row save here carried a whole snapshot of the run, and two
1233
+ // instances entering the settled branch for one graph is by design — so a peer's rollup,
1234
+ // loaded before this instance settled the run, would write `Paused` back over
1235
+ // `Completed` along with everything else it had read.
1236
+ if (!(await this.claims.TrySetRunCostRollup(provider, runID, totals, this.contextUser))) {
1237
+ LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}.`);
1238
+ return 'failed-transient';
1239
+ }
1240
+ LogStatus(`[TaskGraphDispatcher] Credited graph ${parent.ID} to run ${runID}: ` +
1241
+ `${tree.Rows.length} node(s), ${totals.Tokens} token(s), cost ${totals.Cost}.`);
1242
+ }
1243
+ catch (e) {
1244
+ // A failed rollup must never fail the graph. The work finished; only the accounting for
1245
+ // it is missing, and a graph marked Failed because its cost could not be summed would be
1246
+ // a far worse lie than a cost of null.
1247
+ // A throw is transient by default: nothing here proves the condition will persist, and
1248
+ // the cost of being wrong in this direction is one deferred pass rather than a
1249
+ // permanently wrong authoritative total.
1250
+ LogError(`[TaskGraphDispatcher] Cost rollup failed for graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
1251
+ return 'failed-transient';
1252
+ }
1253
+ return 'landed';
1254
+ }
1255
+ /**
1256
+ * Clears a rollup that can no longer be trusted, and says why.
1257
+ *
1258
+ * **Why clear rather than leave.** The four `…Rollup` columns are a cache of the run tree, and
1259
+ * every reader treats a value there as the total. When the tree cannot be summed, any value
1260
+ * already in the column was computed from an EARLIER settlement — it excludes the graph that
1261
+ * just finished, so it is not merely incomplete, it is a wrong total presented as a right one.
1262
+ * `?? TotalCost` protects a reader from null, not from stale.
1263
+ *
1264
+ * Nulling restores the invariant this whole design rests on: **when the column is present, it
1265
+ * equals the tree.** Absent means not settled, which is exactly what a reader should conclude.
1266
+ * A run with no rollup yet is untouched — there is nothing stale to clear, and writing nulls
1267
+ * over nulls would churn Record Changes for nothing.
1268
+ */
1269
+ async clearStaleRollup(provider, runID, reason) {
1270
+ LogError(`[TaskGraphDispatcher] Not recording cost for run ${runID}: ${reason}.`);
1271
+ try {
1272
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
1273
+ if (!(await run.Load(runID)))
1274
+ return;
1275
+ if (run.TotalCostRollup == null && run.TotalTokensUsedRollup == null)
1276
+ return; // nothing stale
1277
+ run.TotalCostRollup = null;
1278
+ run.TotalTokensUsedRollup = null;
1279
+ run.TotalPromptTokensUsedRollup = null;
1280
+ run.TotalCompletionTokensUsedRollup = null;
1281
+ if (!(await run.Save())) {
1282
+ LogError(`[TaskGraphDispatcher] Could not clear the now-stale rollup on run ${runID}: ` +
1283
+ `${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It still shows a total that ` +
1284
+ `excludes the graph that just settled.`);
1285
+ return;
1286
+ }
1287
+ LogStatus(`[TaskGraphDispatcher] Cleared the rollup on run ${runID}: it was computed before this ` +
1288
+ `graph settled and can no longer be recomputed, so it would have under-reported.`);
1289
+ }
1290
+ catch (e) {
1291
+ LogError(`[TaskGraphDispatcher] Could not clear the rollup on run ${runID}: ${e instanceof Error ? e.message : String(e)}`);
1292
+ }
1293
+ }
1294
+ /**
1295
+ * Whether the settling graph is actually reachable from the submitting run's tree.
1296
+ *
1297
+ * Matched on the graph's parent Task id, which is the node the `TaskGraph` member of the query
1298
+ * emits. A run that submitted a graph but recorded no `parentTaskID` produces a tree that stops
1299
+ * at the run — structurally indistinguishable, at the SUM, from a run that never dispatched
1300
+ * anything. This is the check that tells those two apart.
1301
+ */
1302
+ treeContainsGraph(root, parentTaskID) {
1303
+ for (const node of WalkAgentRunTree(root)) {
1304
+ if (node.NodeType === 'TaskGraph' && UUIDsEqual(node.NodeID, parentTaskID))
1305
+ return true;
1306
+ }
1307
+ return false;
1308
+ }
1309
+ /**
1310
+ * Ends a graph early because a prompt said the work is finished.
1311
+ *
1312
+ * **Why `Skipped` and not `Cancelled`.** Nothing went wrong and nobody intervened — the workflow
1313
+ * reached its own conclusion before running every drawn step, which is exactly what a reasoning
1314
+ * step is for. `Cancelled` would tell a reader someone stopped it; `Skipped` says these routes
1315
+ * were not taken, which is true and already the vocabulary the fork machinery uses.
1316
+ *
1317
+ * The message is written to the parent so the graph carries its own answer, rather than the
1318
+ * answer living only on the step that produced it.
1319
+ */
1320
+ async endGraphEarly(provider, task, message) {
1321
+ if (!task.ParentID)
1322
+ return;
1323
+ try {
1324
+ LogStatus(`[TaskGraphDispatcher] '${task.Name}' ended the workflow early: ${message}`);
1325
+ // DECLARE BEFORE MUTATING (R3-1).
1326
+ //
1327
+ // The early finish is decided by one task's result and known to nothing else: skip seeds
1328
+ // come from durable condition and exclusive-group state, so no claim filter anywhere —
1329
+ // including this instance's own, since `executeClaimed` is not awaited and the poll loop
1330
+ // runs concurrently — can tell the remaining steps are about to be skipped. Stamping it
1331
+ // first is what lets `loadGraphState` fold them into the filter, which closes the window
1332
+ // for every instance instead of narrowing it for one.
1333
+ const typeID = await this.workflowTaskTypeID(provider);
1334
+ if (typeID)
1335
+ await this.claims.TryDeclareEarlyFinish(provider, task.ParentID, typeID, this.contextUser);
1336
+ const skipped = [];
1337
+ const claimedMeanwhile = [];
1338
+ for (const sibling of await this.loadChildTasks(provider, task.ParentID)) {
1339
+ if (UUIDsEqual(sibling.ID, task.ID) || sibling.Status !== 'Pending')
1340
+ continue;
1341
+ // GUARDED, not a full-row save against the snapshot above. A sibling claimed between
1342
+ // that load and this write is mid-execution; overwriting it to `Skipped` discards a
1343
+ // running step's outcome while its side effects have already fired. Rowcount is the
1344
+ // verdict — see TaskClaimStore.TrySkipPending.
1345
+ if (!typeID || !(await this.claims.TrySkipPending(provider, sibling.ID, typeID, this.contextUser))) {
1346
+ claimedMeanwhile.push(sibling.Name);
1347
+ continue;
1348
+ }
1349
+ skipped.push(sibling.ID);
371
1350
  this.emit({
372
- Kind: 'GraphSettled',
373
- ParentTaskID: parentID,
374
- OwnerUserID: await this.resolveOwner(provider, parentID),
375
- Status: rollup.status,
376
- CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
377
- TotalCount: fresh.nodes.length,
1351
+ Kind: 'TaskSkipped',
1352
+ ParentTaskID: task.ParentID,
1353
+ OwnerUserID: await this.resolveOwner(provider, task.ParentID),
1354
+ TaskID: sibling.ID,
1355
+ TaskName: sibling.Name,
1356
+ Status: 'Skipped',
378
1357
  });
379
- await this.deliverContinuation(provider, parent, fresh);
380
1358
  }
1359
+ if (claimedMeanwhile.length > 0) {
1360
+ // Reported rather than forced. Those steps were already running when the workflow
1361
+ // decided to stop; their results are real and are allowed to land. The graph settles
1362
+ // once they finish, which is a pass later than it would have and correct.
1363
+ LogStatus(`[TaskGraphDispatcher] Early finish left ${claimedMeanwhile.length} step(s) running ` +
1364
+ `(${claimedMeanwhile.join(', ')}) — they were claimed before the skip and their outcomes stand.`);
1365
+ }
1366
+ // WITHDRAW WHAT WE JUST SKIPPED (R2-10).
1367
+ //
1368
+ // A skipped human step leaves its `MJ: AI Agent Requests` row `Requested` forever:
1369
+ // un-answerable, because answering settles nothing once the task is terminal, and
1370
+ // immortal, because the human settle and expiry sweeps both filter on `Status='Pending'`
1371
+ // tasks and this one no longer is. The person keeps seeing "a workflow is waiting on
1372
+ // you" for a workflow that finished without them. `Cancel` has always done this; the
1373
+ // early-finish path skipped exactly the same rows and did not.
1374
+ await this.withdrawOpenRequests(provider, skipped, 'The workflow finished before this step was needed.');
1375
+ // Column-scoped, because skipping the siblings above just made this graph fully
1376
+ // terminal — so another instance can settle it and claim the marker before this line
1377
+ // runs. A full-row save from the snapshot we loaded first would undo both. See
1378
+ // TaskClaimStore.TrySetParentOutput.
1379
+ if (typeID) {
1380
+ await this.claims.TrySetParentOutput(provider, task.ParentID, JSON.stringify({ message }), typeID, this.contextUser);
1381
+ }
1382
+ else {
1383
+ // Surfaced rather than dropped (R2-10). The graph still ends early — the siblings
1384
+ // are already Skipped — but the reason it ended goes nowhere, and a workflow that
1385
+ // stopped for a stated reason with no stated reason recorded is exactly the kind of
1386
+ // silence this round exists to remove.
1387
+ LogError(`[TaskGraphDispatcher] Could not resolve the workflow task type, so the early-finish ` +
1388
+ `reason for graph ${task.ParentID} was not recorded: ${message}`);
1389
+ }
1390
+ }
1391
+ catch (e) {
1392
+ // The work itself succeeded; only the early-finish bookkeeping failed. Failing the task
1393
+ // over that would discard a completed step's result.
1394
+ LogError(`[TaskGraphDispatcher] Could not end graph early for ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
1395
+ }
1396
+ }
1397
+ /**
1398
+ * How deep the continuation chain already is, read from the graph's parent metadata.
1399
+ *
1400
+ * A run started by a graph inherits that graph's depth **plus one**. Without this every spawned
1401
+ * run begins at zero, so a self-referencing flow — one that dispatches a graph containing itself
1402
+ * — recurses without bound while the cap it should be hitting compares against a permanent zero.
1403
+ */
1404
+ async graphContext(provider, task) {
1405
+ if (!task.ParentID)
1406
+ return { Depth: 0, SubmittingAgentRunID: null };
1407
+ try {
1408
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
1409
+ if (!(await parent.Load(task.ParentID)))
1410
+ return { Depth: 0, SubmittingAgentRunID: null };
1411
+ return {
1412
+ Depth: ParseTaskGraphParentMetadata(parent.InputPayload).reinvokeDepth + 1,
1413
+ // The graph's own row carries the run that submitted it. One load answers both
1414
+ // questions, which is why they are resolved together rather than in two passes.
1415
+ SubmittingAgentRunID: parent.AgentRunID,
1416
+ };
1417
+ }
1418
+ catch {
1419
+ // An unreadable parent must not stop the work; depth zero is the safe reading, and the
1420
+ // submit-time cap still guards the next hop.
1421
+ return { Depth: 0, SubmittingAgentRunID: null };
1422
+ }
1423
+ }
1424
+ /**
1425
+ * Which failures the workflow drew a way out of.
1426
+ *
1427
+ * A Failed task with a **satisfied outgoing edge** is a handled failure: its author drew a
1428
+ * recovery route and that route is now live. Downstream work should be released along it, and the
1429
+ * parent should not roll up Failed because of a step the workflow explicitly planned around.
1430
+ *
1431
+ * Scoped to `failureSemantics: 'edges'` on purpose. Under `'block'` — every agent-emitted graph —
1432
+ * a failure is terminal for its dependents whatever edges exist, because nobody drew those edges
1433
+ * as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
1434
+ * graph sail past a failure it never anticipated.
1435
+ */
1436
+ computeHandledFailures(failureSemantics, nodes, edges) {
1437
+ const handled = new Set();
1438
+ if (failureSemantics !== 'edges')
1439
+ return handled;
1440
+ for (const node of nodes) {
1441
+ if (node.status !== 'Failed')
1442
+ continue;
1443
+ // "Has somewhere to go" is the test. An edge out of a failed step that survived condition
1444
+ // evaluation IS the drawn recovery route; a failed step with no outgoing edges has none,
1445
+ // and stays terminal.
1446
+ if (edges.some((e) => e.dependsOnTaskId === node.id))
1447
+ handled.add(node.id);
381
1448
  }
1449
+ return handled;
1450
+ }
1451
+ /** The graph's child tasks, with the fields the rollup needs. */
1452
+ async loadChildTasks(provider, parentID) {
1453
+ const result = await RunView.FromMetadataProvider(provider).RunView({
1454
+ EntityName: 'MJ: Tasks',
1455
+ ExtraFilter: `ParentID='${parentID}'`,
1456
+ ResultType: 'entity_object',
1457
+ BypassCache: true,
1458
+ }, this.contextUser);
1459
+ return (result.Success ? result.Results : []) ?? [];
382
1460
  }
383
1461
  /**
384
1462
  * Runs the graph's continuation exactly once, now that it has settled.
@@ -391,28 +1469,77 @@ export class TaskGraphDispatcher {
391
1469
  * user-visible noise and, for `reinvoke`, an unbounded agent-run loop. Given one of the two has
392
1470
  * to be chosen, the quiet failure is the safe one.
393
1471
  *
394
- * The marker is written with a compare-and-swap read-back, so two instances reconciling the same
395
- * completed graph produce one winner rather than two.
1472
+ * The marker is claimed with a real compare-and-swap (one guarded UPDATE, rowcount as verdict),
1473
+ * so two instances reconciling the same completed graph produce one winner rather than two.
396
1474
  */
397
- async deliverContinuation(provider, parent, graph) {
1475
+ async deliverContinuation(provider, parent, graph, submitterCancelled) {
398
1476
  const meta = this.readParentMetadata(parent);
399
1477
  if (meta.continuationDeliveredAt)
400
- return;
401
- // At the cap, downgrade rather than refuse: the results still reach the user, the chain just
1478
+ return true;
1479
+ // Nobody is waiting: the run that submitted this graph was cancelled. Claim the marker so
1480
+ // nothing re-offers the graph, and record WHY nothing was announced — "we chose not to" and
1481
+ // "we found it too late" are different facts about a settlement, and a reader afterwards
1482
+ // should be able to tell them apart.
1483
+ if (submitterCancelled) {
1484
+ if (await this.claimContinuation(provider, parent.ID, 'cancelled')) {
1485
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} settled, but the run that submitted it was ` +
1486
+ `cancelled — no message posted and no reinvoke started.`);
1487
+ }
1488
+ return true;
1489
+ }
1490
+ // At the cap, DOWNGRADE rather than refuse: the results still reach the user, the chain just
402
1491
  // stops growing. Refusing outright would lose the outcome of work that actually completed.
403
- const mode = IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
1492
+ //
1493
+ // But a downgrade only applies to something that was going to be delivered (C2). Mapping the
1494
+ // cap straight onto `'message'` also promoted `continuation: 'none'` — a graph that asked
1495
+ // for silence — into a message nobody requested. Latent today because `Submit` refuses to
1496
+ // create a graph past the cap, and exactly the kind of latent that stops being latent the
1497
+ // moment a producer bypasses that check.
1498
+ const mode = meta.continuation !== 'none' && IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
404
1499
  if (mode !== 'none' && IsReinvokeCapReached(meta) && meta.continuation === 'reinvoke') {
405
1500
  LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} hit the reinvoke cap (${MAX_REINVOKE_DEPTH}); ` +
406
1501
  `delivering results as a message instead of starting another turn.`);
407
1502
  }
408
- if (!(await this.claimContinuation(provider, parent.ID, meta)))
409
- return;
1503
+ // A SETTLEMENT NOBODY IS WAITING FOR STILL SETTLES — it just does not get announced.
1504
+ //
1505
+ // Run settlement and cost rollup are status corrections and are always safe to apply late; a
1506
+ // run left `Paused` forever is strictly worse than a stale notification skipped. A stale
1507
+ // NOTIFICATION is not: posting a day-old "your workflow finished" into a live conversation,
1508
+ // or worse starting a fresh billed agent turn for it, is the outcome the age-out exists to
1509
+ // avoid. So an aged-out settlement claims the marker as `expired` and logs, which both
1510
+ // records what happened and stops any later pass delivering it. Second rung on the ladder
1511
+ // the reinvoke cap already established.
1512
+ const expired = IsSettlementExpired(parent.CompletedAt, new Date());
1513
+ // ONLY AN INSTANCE THAT CAN DELIVER MAY CLAIM THE RIGHT TO (R2-6).
1514
+ //
1515
+ // The claim ran before the deliverer check, so an instance constructed WITHOUT one — a
1516
+ // worker tier, an integration bundle, a second dev session — could observe the settlement
1517
+ // first, win the CAS, mark the graph `delivered`, and discard the message or reinvoke a
1518
+ // capable peer would have made moments later. Permanently, decided by poll timing.
1519
+ //
1520
+ // Declining leaves the marker unset, so the rescue sweep keeps offering the graph until an
1521
+ // instance that can deliver takes it. Run settlement and cost rollup have already happened
1522
+ // above and are not held up by this — what is deferred is the announcement, which is the only
1523
+ // part this instance genuinely cannot do.
1524
+ //
1525
+ // `expired` is exempt: recording "too old to deliver" requires no deliverer, and a graph past
1526
+ // its window has nothing left for a capable peer to do.
1527
+ if (!expired && mode !== 'none' && !this.continuationDeliverer) {
1528
+ this.reportUndeliverableOnce(parent.ID);
1529
+ return false;
1530
+ }
1531
+ if (!(await this.claimContinuation(provider, parent.ID, expired ? 'expired' : 'delivered')))
1532
+ return true;
1533
+ if (expired) {
1534
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} settled after its delivery window ` +
1535
+ `(${UNSETTLED_SWEEP_WINDOW_HOURS}h); the run and its cost were corrected, but the ` +
1536
+ `continuation was NOT delivered. Marked expired.`);
1537
+ return true;
1538
+ }
410
1539
  if (mode === 'none')
411
- return;
1540
+ return true;
412
1541
  const summary = this.buildContinuationSummary(parent, graph);
413
1542
  LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} finished — ${summary}`);
414
- if (!this.continuationDeliverer)
415
- return;
416
1543
  const params = {
417
1544
  ParentTaskID: parent.ID,
418
1545
  WorkflowName: parent.Name,
@@ -449,6 +1576,12 @@ export class TaskGraphDispatcher {
449
1576
  // on the marker: a missed notification visible in the record beats one repeated forever.
450
1577
  LogError(`[TaskGraphDispatcher] Continuation delivery failed for ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
451
1578
  }
1579
+ // C1: the function is declared `Promise<boolean>` and fell off the end here, so EVERY
1580
+ // successfully delivered graph resolved `undefined` — read as "not resolved" by the caller,
1581
+ // which then re-queued it and paid another full settle pass (run-tree cost query included)
1582
+ // and polluted R2-6's retry accounting. No double delivery, because the CAS holds; just a
1583
+ // wasted pass per settlement and a retry counter measuring the wrong thing.
1584
+ return true;
452
1585
  }
453
1586
  /** Reads the parent's durable continuation metadata through the shared parser. */
454
1587
  readParentMetadata(parent) {
@@ -460,19 +1593,26 @@ export class TaskGraphDispatcher {
460
1593
  * `MJ: Tasks` stays user-writable (D20), so a plain "read, decide, write" is not enough — the
461
1594
  * read-back is what makes a lost race observable instead of producing a duplicate delivery.
462
1595
  */
463
- async claimContinuation(provider, parentID, meta) {
464
- const row = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
465
- if (!(await row.Load(parentID)))
466
- return false;
467
- const current = this.readParentMetadata(row);
468
- if (current.continuationDeliveredAt)
469
- return false; // a peer got there first
470
- row.InputPayload = JSON.stringify({ ...meta, continuationDeliveredAt: new Date().toISOString() });
471
- if (!(await row.Save())) {
472
- LogError(`[TaskGraphDispatcher] Could not mark continuation delivered for ${parentID}; skipping to avoid a duplicate.`);
1596
+ /**
1597
+ * True when this graph finished so long ago that announcing it would surprise rather than inform.
1598
+ *
1599
+ * Measured from the parent's completion, not from when we noticed: the point is how stale the
1600
+ * NEWS is to whoever would receive it.
1601
+ */
1602
+ async claimContinuation(provider, parentID, deliveredAs = 'delivered') {
1603
+ // ONE GUARDED STATEMENT — see TaskClaimStore.TryClaimContinuation.
1604
+ //
1605
+ // This was Load → check the marker → `Save()`: an unconditional last-write-wins UPDATE that
1606
+ // two dispatchers could both pass. The comments here and at the call site called it a
1607
+ // compare-and-swap read-back; it was read-check-write, and for `continuation: 'reinvoke'`
1608
+ // losing that race means two fresh agent turns billed for one settlement, each able to
1609
+ // submit further graphs.
1610
+ // The type discriminator is part of the guard, not a caller-side filter — see
1611
+ // TryClaimContinuation. Nothing to claim if the type does not exist: no graph was submitted.
1612
+ const typeID = await this.workflowTaskTypeID(provider);
1613
+ if (!typeID)
473
1614
  return false;
474
- }
475
- return true;
1615
+ return this.claims.TryClaimContinuation(provider, parentID, deliveredAs, typeID, this.contextUser);
476
1616
  }
477
1617
  /** One line describing how the graph ended, for the completion log and message delivery. */
478
1618
  buildContinuationSummary(parent, graph) {
@@ -498,8 +1638,29 @@ export class TaskGraphDispatcher {
498
1638
  async notifyHumanTaskReady(task, provider) {
499
1639
  if (task.ClaimedBy === HUMAN_TASK_NOTIFIED_MARKER)
500
1640
  return;
501
- if (!task.UserID)
502
- return; // unassigned human task — nobody to tell
1641
+ // The REQUEST is raised whether or not the task names an assignee. An unassigned human step
1642
+ // is a legitimate "somebody needs to look at this", and a request nobody was notified about
1643
+ // is still findable in the inbox — whereas returning early here is how such a step used to
1644
+ // become invisible work that stalled a workflow with nothing anywhere saying why.
1645
+ // TRANSIENT failures retry; PERMANENT ones stop. That distinction is the whole point, and
1646
+ // getting it wrong took a server down: retrying unconditionally meant a task whose workflow
1647
+ // has no owning agent — which can never succeed — was re-attempted on every poll forever,
1648
+ // each pass re-reading the graph, until the process was OOM-killed. The marker exists to
1649
+ // prevent exactly that storm; a permanent failure has to set it.
1650
+ const raised = await this.raiseHumanRequest(task, provider);
1651
+ if (raised === 'transient-failure')
1652
+ return; // try again next poll
1653
+ if (raised === 'permanent-failure') {
1654
+ // Nothing will change on a retry. Mark it so the loop stops, and leave the task Pending
1655
+ // and visible — a person can still see it in the Tasks UI, which is the fallback the
1656
+ // notification was only ever an accelerant for.
1657
+ await this.markHumanTaskNotified(task, provider);
1658
+ return;
1659
+ }
1660
+ if (!task.UserID) {
1661
+ await this.markHumanTaskNotified(task, provider);
1662
+ return;
1663
+ }
503
1664
  try {
504
1665
  await NotificationEngine.Instance.Config(false, this.contextUser);
505
1666
  await NotificationEngine.Instance.SendNotification({
@@ -513,13 +1674,7 @@ export class TaskGraphDispatcher {
513
1674
  catch (e) {
514
1675
  LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
515
1676
  }
516
- // Marked even when delivery threw. Retrying a notification on every five-second poll is a
517
- // worse failure than one that was missed: the task remains visible in the Tasks UI either
518
- // way, whereas a notification storm is not self-correcting.
519
- task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
520
- if (!(await task.Save())) {
521
- LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
522
- }
1677
+ await this.markHumanTaskNotified(task, provider);
523
1678
  // Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
524
1679
  // than appearing to stall for no reason.
525
1680
  this.emit({
@@ -543,8 +1698,88 @@ export class TaskGraphDispatcher {
543
1698
  * invalidate. Left cached, a completed task keeps reading as `In Progress` and the graph never
544
1699
  * rolls up: submitted work simply never settles.
545
1700
  */
546
- async findActiveGraphIDs(provider) {
1701
+ /**
1702
+ * Settles graphs that reached terminal without completing their post-settlement sequence.
1703
+ *
1704
+ * Runs the ordinary propagation path, which is safe to re-enter by construction: the terminal
1705
+ * write is guarded on not-already-terminal, the cost rollup assigns rather than accumulates, run
1706
+ * settlement is guarded on `Paused`, and delivery is guarded by the continuation CAS. A revisit
1707
+ * therefore corrects whatever is missing and does nothing where nothing is.
1708
+ *
1709
+ * @param windowHours how far back to look — wide once at startup, narrow in steady state
1710
+ */
1711
+ async sweepUnsettledGraphs(windowHours) {
1712
+ try {
1713
+ const provider = await this.providerFactory.CreateProvider();
1714
+ const ids = await this.findActiveGraphIDs(provider, windowHours);
1715
+ if (ids.length === 0)
1716
+ return;
1717
+ LogStatus(`[TaskGraphDispatcher] Startup sweep: reviewing ${ids.length} graph(s), including any that reached terminal without settling.`);
1718
+ await this.propagateAndRollup(provider, ids);
1719
+ }
1720
+ catch (e) {
1721
+ LogError(`[TaskGraphDispatcher] Unsettled-graph sweep failed: ${e instanceof Error ? e.message : String(e)}`);
1722
+ }
1723
+ }
1724
+ /**
1725
+ * The `AI Workflow` task type, resolved once per process.
1726
+ *
1727
+ * `MJ: Tasks` is a GENERAL-PURPOSE entity — conversations and user to-dos live there too — so an
1728
+ * unscoped sweep treats every root task hierarchy as a workflow: rolling up and overwriting the
1729
+ * status of somebody's to-do list, raising agent requests against plain tasks, and (once the
1730
+ * continuation CAS exists) injecting marker keys into a user's own `InputPayload`.
1731
+ *
1732
+ * `Submit` has always stamped this type on the parent and every child (`ensureTaskType`, which
1733
+ * runs before the persist transaction), so the discriminator D3 called for already exists on
1734
+ * every dispatcher-owned row. Verified against the live database: every parent graph carries it.
1735
+ *
1736
+ * Null when the type row does not exist yet — no graph has ever been submitted — in which case
1737
+ * there is nothing for the dispatcher to find and the sweep returns empty rather than unscoped.
1738
+ *
1739
+ * **A miss is never cached**, and that is not a micro-optimisation. `TaskGraphService.Submit`
1740
+ * creates the row on first use, so on a fresh install the ordinary sequence is: dispatcher
1741
+ * starts, looks, finds nothing — then somebody submits the first workflow. Caching that first
1742
+ * `null` would blind this process to every graph until it was restarted, with each poll reporting
1743
+ * a clean, empty sweep. The row is created once and never removed, so the retry costs one
1744
+ * `MaxRows: 1` lookup per poll for exactly as long as there is genuinely nothing to dispatch.
1745
+ */
1746
+ async workflowTaskTypeID(provider) {
1747
+ if (this.cachedWorkflowTaskTypeID)
1748
+ return this.cachedWorkflowTaskTypeID;
1749
+ const result = await RunView.FromMetadataProvider(provider).RunView({
1750
+ EntityName: 'MJ: Task Types',
1751
+ ExtraFilter: `Name='${TASK_TYPE_NAME}'`,
1752
+ Fields: ['ID'],
1753
+ // Ordered, and reading two (R2-7). An unordered `MaxRows: 1` against two rows sharing
1754
+ // the name lets this instance bind a different ID than `Submit` did — after which
1755
+ // every graph the other stamped is invisible to all three sweep arms here.
1756
+ OrderBy: '__mj_CreatedAt ASC, ID ASC',
1757
+ ResultType: 'simple',
1758
+ MaxRows: 2,
1759
+ }, this.contextUser);
1760
+ if (!result.Success) {
1761
+ // A failed lookup is not "no such type" — saying so would silently skip a poll cycle's
1762
+ // worth of real work. Report it, and let the next cycle ask again.
1763
+ LogError(`[TaskGraphDispatcher] Could not resolve the '${TASK_TYPE_NAME}' task type: ${result.ErrorMessage}`);
1764
+ return null;
1765
+ }
1766
+ const rows = result.Results ?? [];
1767
+ if (rows.length > 1) {
1768
+ LogError(`[TaskGraphDispatcher] More than one '${TASK_TYPE_NAME}' task type exists. Binding the ` +
1769
+ `oldest (${rows[0].ID}); any graph stamped with the other is invisible to this sweep and ` +
1770
+ `will never settle. Merge them.`);
1771
+ }
1772
+ this.cachedWorkflowTaskTypeID = rows[0]?.ID ?? null;
1773
+ return this.cachedWorkflowTaskTypeID;
1774
+ }
1775
+ async findActiveGraphIDs(provider, windowHours = UNSETTLED_SWEEP_WINDOW_HOURS) {
547
1776
  const rv = RunView.FromMetadataProvider(provider);
1777
+ // EVERY arm is scoped to workflow graphs. Unscoped, the dispatcher rewrites tasks that are
1778
+ // none of its business — see workflowTaskTypeID.
1779
+ const typeID = await this.workflowTaskTypeID(provider);
1780
+ if (!typeID)
1781
+ return [];
1782
+ const ofWorkflowType = `TypeID='${typeID}'`;
548
1783
  // TWO queries, because "has work left to do" and "needs attention" are not the same set.
549
1784
  //
550
1785
  // Selecting only graphs with non-terminal CHILDREN looks right and is subtly fatal: the
@@ -556,23 +1791,50 @@ export class TaskGraphDispatcher {
556
1791
  //
557
1792
  // The second query closes it: a parent that is itself non-terminal still needs looking at,
558
1793
  // whatever its children are doing.
559
- const [withPendingWork, unsettledParents] = await rv.RunViews([
1794
+ // THREE queries. The third rescues a graph that reached terminal without settling.
1795
+ //
1796
+ // The post-settlement sequence — cost rollup, run settlement, continuation delivery — runs
1797
+ // AFTER the parent's terminal write, and a terminal parent with all-terminal children
1798
+ // matches neither query above. So a process that died in that window left the submitting
1799
+ // agent run `Paused` FOREVER: no rollup, no notification, and nothing that would ever look
1800
+ // again. The metadata's own doc comment promised "the next sweep retries"; that sweep did
1801
+ // not exist.
1802
+ //
1803
+ // Bounded rather than unbounded, because the marker lives in `InputPayload` JSON and cannot
1804
+ // be filtered in SQL: the window is what keeps this a targeted rescue instead of a re-parse
1805
+ // of every graph ever run. `__mj_UpdatedAt` advances on each settle attempt, so a graph
1806
+ // being actively retried stays in the window — the bound is on ABANDONMENT, not on age.
1807
+ const cutoff = SweepCutoff(new Date(), windowHours);
1808
+ const [withPendingWork, unsettledParents, terminalRecent] = await rv.RunViews([
560
1809
  {
561
1810
  EntityName: 'MJ: Tasks',
562
- ExtraFilter: `ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
1811
+ ExtraFilter: `${ofWorkflowType} AND ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
563
1812
  Fields: ['ParentID'],
564
1813
  ResultType: 'simple',
565
1814
  BypassCache: true,
566
1815
  },
567
1816
  {
568
1817
  EntityName: 'MJ: Tasks',
569
- ExtraFilter: `ParentID IS NULL AND Status IN ('Pending','In Progress')`,
1818
+ ExtraFilter: `${ofWorkflowType} AND ParentID IS NULL AND Status IN ('Pending','In Progress')`,
570
1819
  Fields: ['ID'],
571
1820
  ResultType: 'simple',
572
1821
  BypassCache: true,
573
1822
  },
1823
+ {
1824
+ EntityName: 'MJ: Tasks',
1825
+ ExtraFilter: `${ofWorkflowType} AND ParentID IS NULL AND Status IN (${TERMINAL_PARENT_STATUS_SQL}) ` +
1826
+ `AND __mj_UpdatedAt >= '${cutoff}'`,
1827
+ Fields: ['ID', 'InputPayload'],
1828
+ ResultType: 'simple',
1829
+ BypassCache: true,
1830
+ },
574
1831
  ], this.contextUser);
575
1832
  const ids = new Set();
1833
+ // Graphs this instance is mid-retry on, whatever the window says (R2-12). A failing pass
1834
+ // writes nothing, so their `__mj_UpdatedAt` has stopped advancing and the third arm below
1835
+ // will eventually stop finding them — which would turn a retry into a silent abandonment.
1836
+ for (const id of this.retryingSettlement.keys())
1837
+ ids.add(id);
576
1838
  for (const r of (withPendingWork?.Results ?? [])) {
577
1839
  if (r.ParentID)
578
1840
  ids.add(r.ParentID);
@@ -583,6 +1845,11 @@ export class TaskGraphDispatcher {
583
1845
  if (r.ID)
584
1846
  ids.add(r.ID);
585
1847
  }
1848
+ // The marker is JSON, so the filter is in TypeScript rather than in SQL — see
1849
+ // SelectUnsettledGraphIDs, which owns that decision and is tested directly.
1850
+ for (const id of SelectUnsettledGraphIDs((terminalRecent?.Results ?? []))) {
1851
+ ids.add(id);
1852
+ }
586
1853
  return [...ids];
587
1854
  }
588
1855
  /**
@@ -594,11 +1861,100 @@ export class TaskGraphDispatcher {
594
1861
  */
595
1862
  async findClaimableTasks(provider, limit) {
596
1863
  const claimable = [];
597
- for (const parentID of await this.findActiveGraphIDs(provider)) {
1864
+ const stats = new Map();
1865
+ const activeGraphs = await this.findActiveGraphIDs(provider);
1866
+ // Usually a no-op — propagateAndRollup primed these earlier in the same pass.
1867
+ await this.primeDebugStates(provider, activeGraphs);
1868
+ for (const parentID of activeGraphs) {
598
1869
  if (claimable.length >= limit)
599
1870
  break;
600
- const graph = await this.loadGraphState(provider, parentID);
601
- for (const node of ComputeEligibleTasks(graph.nodes, graph.edges)) {
1871
+ const debug = await this.readDebugState(provider, parentID);
1872
+ await this.announcePauseTransition(provider, parentID, debug);
1873
+ const graph = await this.loadGraphState(provider, parentID, debug);
1874
+ // HOLD is what makes "a broken condition stalls visibly" true rather than merely stated.
1875
+ // An undecided exclusive group keeps all its edges, and a kept edge on a Complete origin
1876
+ // is a SATISFIED prerequisite — so without this filter every branch of the fork would be
1877
+ // eligible at once and all of them would run. A typo must not multiply a fork.
1878
+ //
1879
+ // The losers of a DECIDED group must be filtered for the same reason, and this is a race
1880
+ // rather than a rule: they are marked Skipped by the propagation pass, but between the
1881
+ // moment the group resolves and the moment that write lands, their incoming edge is still
1882
+ // a satisfied prerequisite on a Complete origin. A poll landing in that window would
1883
+ // claim and execute the branch the workflow chose NOT to take — irreversibly, since the
1884
+ // action has already run by the time Skipped is written over it.
1885
+ // `unreachableTaskIDs` joins the filter for exactly the reason above. R6 made a
1886
+ // definite-false ordinary edge seed the skip cascade rather than Block its target — but
1887
+ // until that Skipped write lands, the target has no unsatisfied prerequisite and is
1888
+ // vacuously eligible. That is the same race the XOR fix closed, reopened on the new
1889
+ // path: a branch the workflow decided against, claimed and executed irreversibly in the
1890
+ // window before it was marked.
1891
+ const eligible = ComputeEligibleTasks(graph.nodes, graph.edges, graph.handledFailureIDs)
1892
+ .filter((n) => !graph.holdTaskIDs.has(n.id) &&
1893
+ // CONFIRMED seeds, not raw ones (P1). Holding a decided loser out of claiming
1894
+ // closes a real race — the loser could be claimed between eligibility and the
1895
+ // skip write — and that role is unchanged. What changed is which targets count
1896
+ // as decided: a task another live route still reaches was never a loser, so it
1897
+ // must stay claimable and run when its own prerequisites are met.
1898
+ !graph.skipSeedTaskIDs.has(n.id) &&
1899
+ !graph.unreachableTaskIDs.has(n.id) &&
1900
+ // ...and everything the cascade is about to reach (R2-14). A descendant of a
1901
+ // seed is eligible for the moments between its ancestor's skip landing and its
1902
+ // own, because Skipped satisfies prerequisites — a window another instance can
1903
+ // and does claim inside.
1904
+ !graph.cascadeSkipTaskIDs.has(n.id));
1905
+ stats.set(parentID, { eligible: eligible.length, held: graph.holdTaskIDs.size });
1906
+ // THE DEBUG GATE — pause, single-step, breakpoints — decided by the pure function, with
1907
+ // the CAS writes staying here. Every control is a gate on CLAIMING: a claimed task can
1908
+ // never be interrupted mid-flight anyway, so "paused" means nothing new starts while
1909
+ // in-flight work finishes and its completions land. That is also why the gate sits
1910
+ // BEFORE the runner checks and the human notification below: pausing a graph must not
1911
+ // keep notifying assignees — a notification is starting something.
1912
+ const gate = DecideClaimGate(debug, eligible.map((n) => n.id));
1913
+ if (gate.mode === 'closed')
1914
+ continue;
1915
+ let allowedTaskIDs = null;
1916
+ if (gate.mode === 'breakpoint') {
1917
+ const typeID = await this.workflowTaskTypeID(provider);
1918
+ // The pause is a CAS so two instances arriving at the same breakpoint in the same
1919
+ // interval produce one announcement — the loser simply sees a paused graph next pass.
1920
+ if (typeID && await this.claims.TryPauseAtBreakpoint(provider, parentID, gate.taskID, typeID, this.contextUser)) {
1921
+ const owner = await this.resolveOwner(provider, parentID);
1922
+ const name = graph.entityById.get(gate.taskID)?.Name;
1923
+ LogStatus(`[TaskGraphDispatcher] Graph ${parentID} paused at breakpoint on '${name}' (${gate.taskID}).`);
1924
+ this.emit({ Kind: 'BreakpointHit', ParentTaskID: parentID, OwnerUserID: owner, TaskID: gate.taskID, TaskName: name });
1925
+ this.emit({ Kind: 'GraphPaused', ParentTaskID: parentID, OwnerUserID: owner, TaskID: gate.taskID, Reason: 'breakpoint' });
1926
+ this.announcedPaused.set(parentID, true);
1927
+ }
1928
+ continue;
1929
+ }
1930
+ if (gate.mode === 'step') {
1931
+ allowedTaskIDs = new Set(gate.taskIDs);
1932
+ // THE ALLOWANCE IS CONSUMED ONLY IF SOMETHING WILL ACTUALLY MOVE.
1933
+ //
1934
+ // "Eligible" is a graph-shape answer; whether this host can run the step is a
1935
+ // separate one, decided below by the runner checks. Consuming first meant a step
1936
+ // onto a node this instance has no runner for — or one already in flight — spent
1937
+ // the allowance and released nothing, leaving the operator pressing a button that
1938
+ // did nothing and no reason anywhere. Deciding first costs one pre-pass over a set
1939
+ // that is at most the frontier.
1940
+ const releasable = eligible.filter((n) => {
1941
+ const entity = graph.entityById.get(n.id);
1942
+ return entity ? allowedTaskIDs.has(n.id) && this.canActOn(entity) : false;
1943
+ });
1944
+ if (releasable.length === 0) {
1945
+ await this.reportStepReleasedNothing(provider, parentID, gate.taskIDs, graph);
1946
+ continue;
1947
+ }
1948
+ const typeID = await this.workflowTaskTypeID(provider);
1949
+ // Consuming the allowance is the race: exactly one instance clears the marker and
1950
+ // releases work; the loser waits for the next allowance. A lost consume is normal.
1951
+ if (!typeID || !(await this.claims.TryConsumeStepMarker(provider, parentID, typeID, this.contextUser))) {
1952
+ continue;
1953
+ }
1954
+ }
1955
+ for (const node of eligible) {
1956
+ if (allowedTaskIDs && !allowedTaskIDs.has(node.id))
1957
+ continue;
602
1958
  const entity = graph.entityById.get(node.id);
603
1959
  if (!entity)
604
1960
  continue;
@@ -608,7 +1964,36 @@ export class TaskGraphDispatcher {
608
1964
  // they cleared. Without a notification here a workflow simply stops, waiting on
609
1965
  // someone who was never told. That silent stall is the failure mode this exists to
610
1966
  // prevent, so it happens on the eligibility check rather than at submission.
611
- if (!entity.AgentID) {
1967
+ if (entity.ActionID) {
1968
+ // An action node this host has no runner for is left Pending rather than
1969
+ // claimed. Claiming it would take ownership of work this process cannot do, and
1970
+ // the claim would then have to expire before any host that CAN do it gets a
1971
+ // turn — a self-inflicted stall on a mixed deployment.
1972
+ if (!this.actionRunner)
1973
+ continue;
1974
+ }
1975
+ else if (entity.PromptID) {
1976
+ // A prompt node — including a loop that repeats a prompt — is assigned through
1977
+ // PromptID and carries NEITHER ActionID nor AgentID. Without this branch it fell
1978
+ // through to the test below and was treated as a task waiting on a PERSON: the
1979
+ // workflow notified a human who had nothing to do and then stopped forever.
1980
+ // That is precisely the misclassification the step-kind rules warn about, and it
1981
+ // is silent — the graph sits In Progress looking like it is still working.
1982
+ if (!this.promptRunner)
1983
+ continue;
1984
+ }
1985
+ else if (!entity.AgentID) {
1986
+ // No runner column at all — a person completes this one. Asked through the same
1987
+ // predicate the human settle/expiry sweeps use, so a task that gets NOTIFIED here
1988
+ // is a task those sweeps can later see; the two disagreeing is how a human task
1989
+ // ends up asked and then never settled.
1990
+ if (!IsHumanTask(entity)) {
1991
+ // Neither a runner nor a person: nothing can ever move this. Loud, because
1992
+ // the alternative is a graph that waits forever on nobody.
1993
+ LogError(`[TaskGraphDispatcher] Task '${entity.Name}' (${entity.ID}) has no runner ` +
1994
+ `assignment and is not a human step — nothing can execute it. The graph will stall.`);
1995
+ continue;
1996
+ }
612
1997
  await this.notifyHumanTaskReady(entity, provider);
613
1998
  continue;
614
1999
  }
@@ -618,22 +2003,444 @@ export class TaskGraphDispatcher {
618
2003
  if (claimable.length >= limit)
619
2004
  break;
620
2005
  }
2006
+ if (debug.skipBreakpointTaskID) {
2007
+ const skip = debug.skipBreakpointTaskID;
2008
+ const claimedThisPass = claimable.some((t) => UUIDsEqual(t.ID, skip));
2009
+ const stillEligible = eligible.some((n) => UUIDsEqual(n.id, skip));
2010
+ if (claimedThisPass || !stillEligible) {
2011
+ await this.clearSkipBreakpoint(provider, parentID);
2012
+ }
2013
+ }
2014
+ }
2015
+ return { tasks: claimable, stats };
2016
+ }
2017
+ async clearSkipBreakpoint(provider, parentTaskID) {
2018
+ const typeID = await this.workflowTaskTypeID(provider);
2019
+ if (!typeID)
2020
+ return;
2021
+ await this.claims.TryWriteDebugFields(provider, parentTaskID, [TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' })], typeID, this.contextUser);
2022
+ }
2023
+ /**
2024
+ * Drops the per-graph frame-dedup state for a graph that has settled.
2025
+ *
2026
+ * `ownerByParentID` is deliberately NOT purged here: it is the delivery key for the
2027
+ * `GraphSettled` frame emitted moments earlier and for any rescue-sweep pass that revisits the
2028
+ * graph, it is one small string per graph, and ownership never changes — the cost of keeping it
2029
+ * is bounded and the cost of losing it is a re-query on a path that is meant to be cheap.
2030
+ */
2031
+ forgetGraphObservability(parentTaskID) {
2032
+ this.emittedGateVerdicts.delete(parentTaskID);
2033
+ this.nodeProgressLastEmit.delete(parentTaskID);
2034
+ this.announcedPaused.delete(parentTaskID);
2035
+ this.debugStateByGraph.delete(parentTaskID);
2036
+ }
2037
+ /**
2038
+ * Whether THIS host can act on a task right now — the runner-availability question, asked
2039
+ * without acting on it.
2040
+ *
2041
+ * Mirrors the checks in the claim loop so a step allowance is spent only when something will
2042
+ * actually move. A human step counts as actionable: stepping onto one legitimately produces a
2043
+ * notification rather than a claim.
2044
+ */
2045
+ canActOn(entity) {
2046
+ if (this.inFlight.has(entity.ID))
2047
+ return false;
2048
+ if (entity.ActionID)
2049
+ return !!this.actionRunner;
2050
+ if (entity.PromptID)
2051
+ return !!this.promptRunner;
2052
+ if (entity.AgentID)
2053
+ return true;
2054
+ return IsHumanTask(entity);
2055
+ }
2056
+ /**
2057
+ * Says why a step press released nothing, instead of leaving the allowance spent and the
2058
+ * console silent.
2059
+ *
2060
+ * The allowance is deliberately NOT consumed on this path — the operator's intent stands, and
2061
+ * the step will release as soon as the named work becomes actionable (a runner arrives, an
2062
+ * in-flight task finishes). Announced once per pass rather than logged only, because the person
2063
+ * waiting is looking at the console, not the server log.
2064
+ */
2065
+ async reportStepReleasedNothing(provider, parentTaskID, requestedTaskIDs, graph) {
2066
+ const names = requestedTaskIDs
2067
+ .map((id) => graph.entityById.get(id)?.Name)
2068
+ .filter((n) => !!n);
2069
+ const subject = names.length > 0 ? `"${names.join('", "')}"` : 'the next step';
2070
+ const reason = `Step is still waiting: ${subject} cannot start on this server yet — it is already ` +
2071
+ `running, or no runner for that step type is loaded here. The step will release as ` +
2072
+ `soon as it can; nothing was lost.`;
2073
+ LogStatus(`[TaskGraphDispatcher] Step on graph ${parentTaskID} released nothing: ${reason}`);
2074
+ this.emit({
2075
+ Kind: 'StepRefused',
2076
+ ParentTaskID: parentTaskID,
2077
+ OwnerUserID: await this.resolveOwner(provider, parentTaskID),
2078
+ TaskID: requestedTaskIDs[0],
2079
+ TaskName: names[0],
2080
+ Reason: reason,
2081
+ });
2082
+ }
2083
+ /**
2084
+ * Marks a human task as notified, so the request is raised exactly once.
2085
+ *
2086
+ * Written even when delivery threw. Retrying on every poll is a worse failure than one missed
2087
+ * notification: the task stays visible in the inbox either way, whereas a notification storm is
2088
+ * not self-correcting.
2089
+ */
2090
+ async markHumanTaskNotified(task, provider) {
2091
+ // Guarded, not a full-row save against a snapshot (R3-5). This row was loaded at the top of
2092
+ // the pass; a full-row write could revert a status it has reached since, and two instances
2093
+ // could both stamp it after both having seen it absent. The predicate makes it once-only.
2094
+ if (await this.claims.TryMarkHumanNotified(provider, task.ID, HUMAN_TASK_NOTIFIED_MARKER, this.contextUser)) {
2095
+ task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
2096
+ return;
2097
+ }
2098
+ // Rowcount 0 is ordinary: another instance marked it, or the task is no longer Pending.
2099
+ // Either way this instance has nothing left to do about the notification.
2100
+ }
2101
+ /**
2102
+ * Raises the `MJ: AI Agent Requests` row a person answers to release this step.
2103
+ *
2104
+ * **Why that entity rather than something new.** It already models everything a workflow's human
2105
+ * step needs — who is being asked, what for, a typed response schema, priority, expiry, and an
2106
+ * inbox surface people already use. A second HITL substrate beside it would split the inbox in
2107
+ * two and leave one of them without expiry or permissions.
2108
+ *
2109
+ * **What it deliberately does NOT set is `ResumingAgentRunID`.** A request normally suspends an
2110
+ * agent run and resumes it. A workflow needs none of that: the graph OUTLIVES the run that
2111
+ * submitted it, so nothing is suspended — the task sits Pending, every other branch keeps
2112
+ * running, and answering settles the task. That column staying null is meaningful, not missing.
2113
+ */
2114
+ async raiseHumanRequest(task, provider) {
2115
+ try {
2116
+ const existing = await this.findOpenRequests(provider, task.ID);
2117
+ if (existing.length > 0) {
2118
+ // Somebody IS waiting on this task — but "somebody" may be two rows, so collapse
2119
+ // before returning. Free: the rows are already in hand.
2120
+ await this.withdrawDuplicateRequests(provider, task.ID, existing, existing[0].ID);
2121
+ return 'raised';
2122
+ }
2123
+ const request = await provider.GetEntityObject('MJ: AI Agent Requests', this.contextUser);
2124
+ request.NewRecord();
2125
+ request.OriginatingTaskID = task.ID;
2126
+ // A human task has NO AgentID of its own — that column names what EXECUTES a step, and
2127
+ // a person is not an agent. The request still needs one, so it carries the agent that
2128
+ // owns the workflow: the graph's own agent, which is who is asking.
2129
+ const owningAgentID = await this.owningAgentOf(provider, task);
2130
+ if (!owningAgentID) {
2131
+ // PERMANENT: a graph with no owning agent will not acquire one by being asked
2132
+ // again. Graphs submitted before the provenance stamp landed are all in this state.
2133
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} needs a person, but its workflow has no ` +
2134
+ `agent to ask on behalf of, so no request can be raised. The task stays Pending ` +
2135
+ `and visible in the Tasks UI; it will not be retried.`);
2136
+ return 'permanent-failure';
2137
+ }
2138
+ request.AgentID = owningAgentID;
2139
+ request.RequestForUserID = task.UserID;
2140
+ request.RequestedAt = new Date();
2141
+ request.Status = 'Requested';
2142
+ request.Request = task.Description || `A workflow is waiting on you to complete "${task.Name}".`;
2143
+ // The graph's own run is the provenance a reader follows back to see what led here.
2144
+ request.OriginatingAgentRunID = await this.submittingRunOf(provider, task);
2145
+ // The deadline, when the author set one. `expireOverdueRequests` has always been able to
2146
+ // enforce this — it expires the request and fails the step so a give-up edge can route
2147
+ // around it — but nothing ever WROTE the column, so that whole path had never run outside
2148
+ // a test and a workflow waiting on someone who left the company waited forever.
2149
+ // Absent means no deadline, deliberately: expiring on a timeout nobody chose would be
2150
+ // worse than waiting.
2151
+ const expiresInHours = this.parseConfiguration(task)?.human?.expiresInHours;
2152
+ if (expiresInHours && expiresInHours > 0) {
2153
+ request.ExpiresAt = new Date(Date.now() + expiresInHours * 60 * 60 * 1000);
2154
+ }
2155
+ if (await request.Save()) {
2156
+ // INSERT-THEN-RESELECT (R3-5). The check above is read-then-write in a system whose
2157
+ // every other cross-instance write is a CAS, and there is no unique index behind it —
2158
+ // so two overlapping instances both read "none open", both insert, and both ping the
2159
+ // assignee. When one is answered, `settleAnsweredHumanTasks` settles from the single
2160
+ // latest terminal request and NOTHING ever touches the other: the withdrawal paths
2161
+ // fire only on skips and cancels, and both human sweeps scope to `Pending` tasks,
2162
+ // which the settled task no longer is. The duplicate becomes a durable, unanswerable,
2163
+ // immortal inbox item — the zombie class R2-10 removed from the skip paths.
2164
+ //
2165
+ // Checking harder is what created the window, so the resolution is to let both
2166
+ // inserts happen and then agree on a winner: the oldest open row. A loser withdraws
2167
+ // its own row and returns `raised` — somebody IS waiting on this task, which is what
2168
+ // the caller needs to know.
2169
+ await this.withdrawDuplicateRequests(provider, task.ID, await this.findOpenRequests(provider, task.ID), request.ID);
2170
+ return 'raised';
2171
+ }
2172
+ {
2173
+ LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ` +
2174
+ `${request.LatestResult?.CompleteMessage ?? 'unknown error'}`);
2175
+ // A failed SAVE may be transient (deadlock, contention), so this one earns a retry.
2176
+ return 'transient-failure';
2177
+ }
2178
+ return 'raised';
2179
+ }
2180
+ catch (e) {
2181
+ // Never fatal. The task remains Pending and visible; a missing request is recoverable,
2182
+ // whereas throwing here would abort the whole dispatch pass for every other branch.
2183
+ LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
2184
+ return 'transient-failure';
2185
+ }
2186
+ }
2187
+ /**
2188
+ * The agent that owns this task's workflow — who the request is asked on behalf of.
2189
+ *
2190
+ * Reads the graph's parent row, falling back to the run that submitted it. A human step has no
2191
+ * agent of its own by design: `AgentID` names what EXECUTES a step, and a person is not an agent.
2192
+ */
2193
+ async owningAgentOf(provider, task) {
2194
+ if (task.AgentID)
2195
+ return task.AgentID;
2196
+ if (!task.ParentID)
2197
+ return null;
2198
+ try {
2199
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
2200
+ if (!(await parent.Load(task.ParentID)))
2201
+ return null;
2202
+ if (parent.AgentID)
2203
+ return parent.AgentID;
2204
+ if (!parent.AgentRunID)
2205
+ return null;
2206
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
2207
+ return (await run.Load(parent.AgentRunID)) ? run.AgentID : null;
2208
+ }
2209
+ catch {
2210
+ return null;
2211
+ }
2212
+ }
2213
+ /**
2214
+ * Every still-open request for a task, oldest first.
2215
+ *
2216
+ * Plural, and ordered, for one reason each. Ordered, because the oldest row is the one every
2217
+ * instance must agree is "the" request — it is the one the assignee most likely already saw,
2218
+ * and the one `withdrawDuplicateRequests` keeps; unordered, two instances could each decide a
2219
+ * different duplicate was the keeper and withdraw each other's. Plural, because a caller that
2220
+ * only ever sees the first cannot notice there are two, which is how the duplicate below
2221
+ * survived: every reader of this took `[0]` and moved on.
2222
+ */
2223
+ async findOpenRequests(provider, taskID) {
2224
+ const result = await RunView.FromMetadataProvider(provider).RunView({
2225
+ EntityName: 'MJ: AI Agent Requests',
2226
+ ExtraFilter: `OriginatingTaskID='${taskID}' AND Status='Requested'`,
2227
+ OrderBy: '__mj_CreatedAt ASC, ID ASC',
2228
+ ResultType: 'entity_object',
2229
+ BypassCache: true,
2230
+ }, this.contextUser);
2231
+ return (result.Success ? result.Results : null) ?? [];
2232
+ }
2233
+ /**
2234
+ * Settles a human task from the request a person answered.
2235
+ *
2236
+ * Runs on the poll rather than on a save hook, because the answer can arrive through any surface
2237
+ * — the inbox, the API, a conversation — and only the dispatcher knows how to release the rest
2238
+ * of the graph afterwards.
2239
+ *
2240
+ * **`ResponseData` becomes the task's output.** That is what makes a human step useful rather
2241
+ * than a gate: a downstream edge can branch on what the person actually said, typed by the
2242
+ * request's own ResponseSchema. A step that only recorded "approved" would force every decision
2243
+ * back into a separate action.
2244
+ */
2245
+ async settleAnsweredHumanTasks(provider, graphID) {
2246
+ const waiting = await RunView.FromMetadataProvider(provider).RunView({
2247
+ EntityName: 'MJ: Tasks',
2248
+ ExtraFilter: `ParentID='${graphID}' AND ${HumanTaskSQL()} AND Status='Pending'`,
2249
+ ResultType: 'entity_object',
2250
+ BypassCache: true,
2251
+ }, this.contextUser);
2252
+ if (!waiting.Success)
2253
+ return;
2254
+ for (const task of waiting.Results ?? []) {
2255
+ const request = await this.answeredRequestFor(provider, task.ID);
2256
+ if (!request)
2257
+ continue;
2258
+ const rejected = request.Status === 'Rejected';
2259
+ const expired = request.Status === 'Expired';
2260
+ task.Status = rejected || expired ? 'Failed' : 'Complete';
2261
+ task.CompletedAt = new Date();
2262
+ task.PercentComplete = rejected || expired ? 0 : 100;
2263
+ task.ClaimedBy = null;
2264
+ task.ClaimExpiresAt = null;
2265
+ task.OutputPayload = request.ResponseData ?? null;
2266
+ if (rejected) {
2267
+ task.ErrorMessage = request.Comments || 'A person rejected this step.';
2268
+ }
2269
+ else if (expired) {
2270
+ // Stated as a failure rather than left Pending. A workflow blocked forever on
2271
+ // someone who never answered — who may have left the company — is the silent stall
2272
+ // this whole path exists to avoid, and a give-up edge can now route around it.
2273
+ task.ErrorMessage = 'Nobody answered this step before its request expired.';
2274
+ }
2275
+ if (!(await task.Save())) {
2276
+ LogError(`[TaskGraphDispatcher] Could not settle human task ${task.ID}: ` +
2277
+ `${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
2278
+ continue;
2279
+ }
2280
+ // WITHDRAW EVERY OTHER OPEN ASK FOR THIS STEP (R3-5).
2281
+ //
2282
+ // The step is terminal now, so both human sweeps — which scope to `Pending` tasks — will
2283
+ // never look at it again, and any request still `Requested` is un-answerable and
2284
+ // immortal: the assignee keeps seeing "a workflow is waiting on you" for a step that is
2285
+ // finished. Duplicates only arose from the raise race fixed above, but this also
2286
+ // retroactively cleans the ones already minted, which the raise-side fix cannot reach.
2287
+ await this.withdrawOpenRequests(provider, [task.ID], 'This step has been settled; the request is no longer open.');
2288
+ }
2289
+ }
2290
+ /**
2291
+ * Reconciles the requests behind human steps that are waiting on somebody.
2292
+ *
2293
+ * Two things can be wrong with a waiting step, and both are silent. It can have NO open request
2294
+ * — the cancel case below — or it can have MORE than one, which the raise cannot fix because it
2295
+ * never runs again for a notified task. Both are corrected here, on the only sweep that visits
2296
+ * these tasks every pass.
2297
+ *
2298
+ * **Re-opening a human step whose request was CANCELLED.**
2299
+ *
2300
+ * `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
2301
+ * rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
2302
+ * it". Nothing replaced it. `raiseHumanRequest` refuses to raise twice (the notified marker on
2303
+ * `ClaimedBy` is what stops the notification storm), so a cancelled request left the task Pending
2304
+ * with no open request and no path to acquiring one: a workflow waiting forever on a question
2305
+ * nobody is being asked.
2306
+ *
2307
+ * Clearing the marker is the whole fix — the next poll sees an un-notified Pending human task
2308
+ * and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
2309
+ * human action: it takes another person cancelling again to come back here.
2310
+ */
2311
+ async reconcileWaitingHumanTasks(provider, graphID) {
2312
+ const waiting = await RunView.FromMetadataProvider(provider).RunView({
2313
+ EntityName: 'MJ: Tasks',
2314
+ // `StepType` is NULLABLE, and rows predating the column exist (4 in the reference
2315
+ // database at the time of writing). None currently carry a UserID, but a human task
2316
+ // written by any path that set the assignee without the discriminator would be
2317
+ // invisible to a `StepType='Human'` filter and stay dead forever after a cancel —
2318
+ // the exact stall this method exists to end. The notified marker already narrows
2319
+ // this to tasks the dispatcher raised a request for, so the widening cannot pull in
2320
+ // unrelated work.
2321
+ ExtraFilter: `ParentID='${graphID}' AND Status='Pending' ` +
2322
+ `AND ${HumanTaskSQL()} ` +
2323
+ `AND ClaimedBy='${HUMAN_TASK_NOTIFIED_MARKER}'`,
2324
+ ResultType: 'entity_object',
2325
+ BypassCache: true,
2326
+ }, this.contextUser);
2327
+ if (!waiting.Success)
2328
+ return;
2329
+ for (const task of waiting.Results ?? []) {
2330
+ // Only when there is nothing live AND nothing terminal. A task with an open request is
2331
+ // simply waiting; one with a terminal request is settled on the next pass by
2332
+ // settleAnsweredHumanTasks, and re-raising either would ask the same question twice.
2333
+ const open = await this.findOpenRequests(provider, task.ID);
2334
+ if (open.length > 0) {
2335
+ // Waiting, correctly — but on however many asks happen to exist. Collapse them here
2336
+ // or nothing ever will: the raise is behind the notified marker for good.
2337
+ await this.withdrawDuplicateRequests(provider, task.ID, open, open[0].ID);
2338
+ continue;
2339
+ }
2340
+ if (await this.answeredRequestFor(provider, task.ID))
2341
+ continue;
2342
+ LogStatus(`[TaskGraphDispatcher] The request for '${task.Name}' was cancelled and nothing ` +
2343
+ `replaced it; asking again.`);
2344
+ task.ClaimedBy = null;
2345
+ if (!(await task.Save())) {
2346
+ LogError(`[TaskGraphDispatcher] Could not re-open cancelled human task ${task.ID}: ` +
2347
+ `${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
2348
+ }
2349
+ }
2350
+ }
2351
+ /** The answered (or expired) request for a task, if any. */
2352
+ async answeredRequestFor(provider, taskID) {
2353
+ const result = await RunView.FromMetadataProvider(provider).RunView({
2354
+ EntityName: 'MJ: AI Agent Requests',
2355
+ // Everything terminal. 'Canceled' is deliberately absent: a cancelled request means
2356
+ // the ASK was withdrawn, not that the step was decided, so the task keeps waiting
2357
+ // for whatever replaces it.
2358
+ ExtraFilter: `OriginatingTaskID='${taskID}' AND Status IN ('Approved','Rejected','Responded','Expired')`,
2359
+ OrderBy: 'RespondedAt DESC',
2360
+ ResultType: 'entity_object',
2361
+ BypassCache: true,
2362
+ }, this.contextUser);
2363
+ return (result.Success ? result.Results?.[0] : null) ?? null;
2364
+ }
2365
+ /**
2366
+ * Expires requests whose deadline has passed.
2367
+ *
2368
+ * A deadline that nothing enforces is a comment. Without this an `ExpiresAt` in the past leaves
2369
+ * the request `Requested` forever and the workflow waiting on it just as long.
2370
+ */
2371
+ async expireOverdueRequests(provider, graphID) {
2372
+ // Scoped by an explicit id list rather than a subquery against a view name, so this reads
2373
+ // the same on any provider rather than assuming a SQL dialect and a physical view.
2374
+ const humanTasks = await RunView.FromMetadataProvider(provider).RunView({
2375
+ EntityName: 'MJ: Tasks',
2376
+ Fields: ['ID'],
2377
+ ExtraFilter: `ParentID='${graphID}' AND ${HumanTaskSQL()} AND Status='Pending'`,
2378
+ ResultType: 'simple',
2379
+ }, this.contextUser);
2380
+ const ids = (humanTasks.Results ?? []).map((r) => `'${r.ID}'`);
2381
+ if (ids.length === 0)
2382
+ return;
2383
+ const nowISO = new Date().toISOString();
2384
+ const overdue = await RunView.FromMetadataProvider(provider).RunView({
2385
+ EntityName: 'MJ: AI Agent Requests',
2386
+ ExtraFilter: `Status='Requested' AND ExpiresAt IS NOT NULL AND ExpiresAt < '${nowISO}' ` +
2387
+ `AND OriginatingTaskID IN (${ids.join(',')})`,
2388
+ ResultType: 'entity_object',
2389
+ BypassCache: true,
2390
+ }, this.contextUser);
2391
+ if (!overdue.Success)
2392
+ return;
2393
+ for (const request of overdue.Results ?? []) {
2394
+ request.Status = 'Expired';
2395
+ if (!(await request.Save())) {
2396
+ LogError(`[TaskGraphDispatcher] Could not expire request ${request.ID}.`);
2397
+ }
2398
+ }
2399
+ }
2400
+ /** The agent run that submitted this task's graph, for provenance on the request. */
2401
+ async submittingRunOf(provider, task) {
2402
+ if (!task.ParentID)
2403
+ return null;
2404
+ try {
2405
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
2406
+ return (await parent.Load(task.ParentID)) ? parent.AgentRunID : null;
2407
+ }
2408
+ catch {
2409
+ return null;
621
2410
  }
622
- return claimable;
623
2411
  }
624
2412
  /** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
625
- async loadGraphState(provider, parentTaskID) {
2413
+ async loadGraphState(provider, parentTaskID, debug) {
626
2414
  const rv = RunView.FromMetadataProvider(provider);
627
2415
  // BypassCache throughout: task status is written by the claim protocol's direct SQL, which
628
2416
  // fires no cache invalidation. See findActiveGraphIDs.
629
2417
  const childrenResult = await rv.RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
630
2418
  const children = (childrenResult.Success ? childrenResult.Results : []) ?? [];
631
- if (children.length === 0)
632
- return { nodes: [], edges: [], entityById: new Map(), unreachableTaskIDs: new Set() };
2419
+ if (children.length === 0) {
2420
+ return {
2421
+ nodes: [], edges: [], entityById: new Map(),
2422
+ unreachableTaskIDs: new Set(), cascadeSkipTaskIDs: new Set(),
2423
+ skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
2424
+ handledFailureIDs: new Set(),
2425
+ };
2426
+ }
633
2427
  const idList = children.map((c) => `'${c.ID}'`).join(',');
634
2428
  const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
635
2429
  const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
636
2430
  const entityById = new Map(children.map((c) => [c.ID, c]));
2431
+ // Read ONCE, and only when it can change an answer (R2-4). Both consumers below — which
2432
+ // origin statuses may decide an exclusive group, and which failures count as handled — are
2433
+ // no-ops unless something has actually failed, and this runs on every poll for every active
2434
+ // graph, so the parent load stays behind the same cheap exit `computeHandledFailures` used.
2435
+ // One parent read serves both questions this pass asks of the metadata bag.
2436
+ const parentMeta = await this.readParentMetadataFor(provider, parentTaskID);
2437
+ const failureSemantics = parentMeta.failureSemantics;
2438
+ // The invocation's own parameters, carried on the parent so a condition evaluated by any
2439
+ // instance sees what the walker saw (R3-3).
2440
+ const invocation = {
2441
+ Data: parentMeta.invocation?.data,
2442
+ Context: parentMeta.invocation?.context,
2443
+ };
637
2444
  // Conditional edges are resolved HERE, before eligibility runs, by dropping edges whose
638
2445
  // condition does not hold. Expressing it as edge removal rather than as a second rule inside
639
2446
  // the eligibility algorithm is what keeps one definition of "ready": a task with no live
@@ -651,13 +2458,67 @@ export class TaskGraphDispatcher {
651
2458
  // unreachable instead, and blocked before anything can claim it.
652
2459
  const droppedInto = new Set();
653
2460
  const stillReachable = new Set();
654
- for (const d of deps) {
2461
+ // EXCLUSIVE edges are exempt from the generic machinery below, and that exemption is
2462
+ // load-bearing. An XOR loser is by definition condition-false, so the ordinary path would
2463
+ // record it as unreachable and Block it — and a Blocked child poisons the parent rollup, so
2464
+ // every fork would settle the graph as Blocked. Losers must become Skipped instead, which
2465
+ // only ResolveExclusiveGroups can decide.
2466
+ const exclusive = deps.filter((d) => !!d.ExclusiveGroup);
2467
+ const ordinary = deps.filter((d) => !d.ExclusiveGroup);
2468
+ // Named once and consumed twice — by the resolution below and by the `GateDecision` frame
2469
+ // emission further down. Two copies of this rule is how the console starts narrating
2470
+ // decisions the engine no longer makes.
2471
+ const decidingStatuses = failureSemantics === 'edges'
2472
+ ? new Set(['Complete', 'Failed'])
2473
+ : new Set(['Complete']);
2474
+ const resolution = ResolveExclusiveGroups(exclusive.map((d) => ({
2475
+ id: d.ID,
2476
+ taskId: d.TaskID,
2477
+ dependsOnTaskId: d.DependsOnTaskID,
2478
+ exclusiveGroup: d.ExclusiveGroup,
2479
+ originStatus: (entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending'),
2480
+ priority: d.Priority ?? 0,
2481
+ sequence: d.Sequence ?? 0,
2482
+ conditionOutcome: this.evaluateExclusiveCondition(d, entityById, invocation, debug),
2483
+ })),
2484
+ // WHICH STATUSES MAY DECIDE — the graph's own failure dialect, not a constant.
2485
+ //
2486
+ // Under `'edges'`, a flow's failure handling IS its outgoing edges, so a Failed origin
2487
+ // decides its group and the drawn recovery path runs. Under `'block'` — the spec's
2488
+ // DEFAULT — a failure is terminal for everything downstream, and letting it decide was
2489
+ // silently catastrophic: the losers were removed and seeded, `ComputeSkipCascade`
2490
+ // confirmed them `Skipped`, `Skipped` satisfies dependents, and because the removed
2491
+ // loser edges also sever `ComputeTasksToBlock`'s forward walk, a join fed by an
2492
+ // independent healthy route EXECUTED downstream of an unhandled failure. The parent
2493
+ // still rolled up Failed, so the verdict looked right while the side effects had fired.
2494
+ //
2495
+ // The old comment claimed a loop-agent graph saw Complete-only. It did not; the same
2496
+ // hardcoded set was passed for every graph.
2497
+ decidingStatuses);
2498
+ const loserEdgeIDs = new Set(resolution.loserEdgeIDs);
2499
+ // Targets of an edge whose condition could not be evaluated (P2). Neither eligible nor
2500
+ // skipped: the edge stays live so the target is not mistaken for unreachable, and the target
2501
+ // joins the hold set so nothing claims it.
2502
+ const heldByCondition = new Set();
2503
+ // Decisions collected for `GateDecision` frames — announced after the state is assembled,
2504
+ // change-only, so a viewer learns WHY a branch ran (or is held) the moment it is decided.
2505
+ const gateDecisions = [];
2506
+ for (const d of ordinary) {
655
2507
  if (d.Condition?.trim()) {
656
- const outcome = this.evaluateEdgeCondition(d, entityById);
657
- if (outcome === 'drop') {
2508
+ const decision = this.evaluateEdgeCondition(d, entityById, failureSemantics, invocation, debug);
2509
+ if (decision.decided) {
2510
+ gateDecisions.push({
2511
+ edge: d,
2512
+ verdict: decision.outcome === 'keep' ? 'satisfied' : decision.outcome === 'drop' ? 'notTaken' : 'held',
2513
+ reason: decision.reason,
2514
+ });
2515
+ }
2516
+ if (decision.outcome === 'drop') {
658
2517
  droppedInto.add(d.TaskID);
659
2518
  continue;
660
2519
  }
2520
+ if (decision.outcome === 'hold')
2521
+ heldByCondition.add(d.TaskID);
661
2522
  }
662
2523
  stillReachable.add(d.TaskID);
663
2524
  liveEdges.push({
@@ -666,16 +2527,120 @@ export class TaskGraphDispatcher {
666
2527
  dependencyType: d.DependencyType,
667
2528
  });
668
2529
  }
2530
+ for (const d of exclusive) {
2531
+ // A losing edge is removed rather than left to gate: its target is being skipped, and a
2532
+ // live edge into a skipped task would keep the graph waiting on a branch nobody took.
2533
+ if (loserEdgeIDs.has(d.ID))
2534
+ continue;
2535
+ stillReachable.add(d.TaskID);
2536
+ liveEdges.push({
2537
+ taskId: d.TaskID,
2538
+ dependsOnTaskId: d.DependsOnTaskID,
2539
+ dependencyType: d.DependencyType,
2540
+ });
2541
+ }
669
2542
  // Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
670
2543
  // waiting on it, and a node reached by an alternate branch is genuinely reachable.
671
2544
  const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
2545
+ // EXCLUSIVE LOSERS GET THE SAME TEST — they did not, and that is P1.
2546
+ //
2547
+ // A loser's target was seeded and written `Skipped` unconditionally, with no "does another
2548
+ // live route reach it?" check. The shape that breaks: `A →(cond)→ Review → Publish` and
2549
+ // `A →(else)→ Publish`. With the condition true, the losing edge `A→Publish` skipped
2550
+ // **Publish** while Review was still running; Review completed, Publish was already
2551
+ // terminal, and `Skipped` satisfies dependents — so the graph settled Complete with the
2552
+ // publish step never executed. No error and no stall.
2553
+ //
2554
+ // Confirmed against `liveEdges`, which by this point has both losers and definitely-false
2555
+ // edges removed, so "a live gating edge still points here" is exactly the surviving-route
2556
+ // question. A genuine loser has none and is still skipped.
2557
+ const confirmedSkipSeeds = new Set(ConfirmSkipSeeds([...resolution.skipSeedTaskIDs], liveEdges));
2558
+ // Exclusive edges get verdicts too, once their origin can decide them: a loser is a branch
2559
+ // not taken, a member of an undecided group is held, a surviving edge is satisfied. Same
2560
+ // vocabulary as ordinary edges so a viewer never needs to know which dialect an edge was.
2561
+ //
2562
+ // GATED ON THE SAME `terminalDecides` SET THE RESOLUTION USED — not on
2563
+ // `TERMINAL_FOR_CONDITIONS`. Since R2-4 a `Failed` origin decides its group only under
2564
+ // `failureSemantics: 'edges'`; announcing from the wider set would tell a viewer the fork
2565
+ // resolved while under `'block'` the engine deliberately left it undecided and let the
2566
+ // ordinary block cascade own everything downstream. A console that narrates decisions the
2567
+ // engine did not make is worse than one that stays quiet.
2568
+ const exclusiveHolds = new Set(resolution.holdTaskIDs);
2569
+ for (const d of exclusive) {
2570
+ const originStatus = entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending';
2571
+ if (!decidingStatuses.has(originStatus) && !OverrideVerdictFor(debug ?? {}, d.ID))
2572
+ continue;
2573
+ gateDecisions.push({
2574
+ edge: d,
2575
+ verdict: loserEdgeIDs.has(d.ID)
2576
+ ? 'notTaken'
2577
+ : exclusiveHolds.has(d.TaskID) ? 'held' : 'satisfied',
2578
+ reason: exclusiveHolds.has(d.TaskID)
2579
+ ? 'this fork is undecided — a path in its group cannot be answered yet'
2580
+ : undefined,
2581
+ });
2582
+ }
2583
+ this.emitGateDecisions(provider, parentTaskID, gateDecisions, entityById);
2584
+ const nodes = children.map((c) => ({ id: c.ID, status: c.Status }));
2585
+ // THE CASCADE IS COMPUTED HERE, NOT ONLY AT SKIP TIME (R2-14).
2586
+ //
2587
+ // The claim filter covered seeds, holds and unreachable targets but not the cascade's
2588
+ // DESCENDANTS, and the skip writes are sequential per-entity saves. Between a seed's
2589
+ // `Skipped` landing and its descendants', another instance's fresh load sees
2590
+ // Skipped-satisfies-prerequisites and finds those descendants eligible — so it claims and
2591
+ // executes a branch that was never taken, irreversibly if the step has side effects.
2592
+ //
2593
+ // The set is already needed by the propagation pass, so computing it once here costs
2594
+ // nothing and closes the window by construction: nothing that is about to be skipped is
2595
+ // claimable, whichever instance is looking.
2596
+ const allSkipSeeds = [...confirmedSkipSeeds, ...unreachableTaskIDs];
2597
+ const cascadeSkipTaskIDs = new Set([
2598
+ ...allSkipSeeds,
2599
+ ...ComputeSkipCascade(nodes, liveEdges, allSkipSeeds),
2600
+ ]);
2601
+ // A DECLARED EARLY FINISH MAKES EVERY REMAINING STEP UNCLAIMABLE (R3-1).
2602
+ //
2603
+ // The declaration is durable, so this holds for every instance rather than only the one that
2604
+ // decided it — which is the whole point. Folded into the same set the claim filter already
2605
+ // consults, so nothing about to be skipped can be claimed and started in the window between
2606
+ // the decision and the skip writes.
2607
+ if (parentMeta.earlyFinishedAt) {
2608
+ for (const node of nodes) {
2609
+ if (node.status === 'Pending')
2610
+ cascadeSkipTaskIDs.add(node.id);
2611
+ }
2612
+ }
672
2613
  return {
673
- nodes: children.map((c) => ({ id: c.ID, status: c.Status })),
2614
+ nodes,
674
2615
  edges: liveEdges,
675
2616
  entityById,
676
2617
  unreachableTaskIDs,
2618
+ cascadeSkipTaskIDs,
2619
+ skipSeedTaskIDs: confirmedSkipSeeds,
2620
+ // Exclusive holds and ordinary-condition holds are the same state and share one set:
2621
+ // "we cannot tell yet, so nothing may claim this."
2622
+ holdTaskIDs: new Set([...resolution.holdTaskIDs, ...heldByCondition]),
2623
+ handledFailureIDs: this.computeHandledFailures(failureSemantics, nodes, liveEdges),
677
2624
  };
678
2625
  }
2626
+ /**
2627
+ * Reports an unevaluable condition ONCE per edge, not once per poll.
2628
+ *
2629
+ * Eligibility is recomputed every cycle, so an unqualified LogError here would repeat every few
2630
+ * seconds for as long as the graph is held — which buries the one line that matters under
2631
+ * thousands of copies of itself. Keyed by edge id plus the failure text, so a condition that
2632
+ * starts failing differently is reported again.
2633
+ */
2634
+ logUnevaluableConditionOnce(dep, errorMessage) {
2635
+ const key = `${dep.ID}:${errorMessage ?? ''}`;
2636
+ if (this.reportedUnevaluableConditions.has(key))
2637
+ return;
2638
+ this.reportedUnevaluableConditions.add(key);
2639
+ LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
2640
+ `(${errorMessage}); condition text: ${JSON.stringify(dep.Condition)}. ` +
2641
+ `Task ${dep.TaskID} is HELD — it will not run and will not be skipped until the ` +
2642
+ `condition can be evaluated. The graph reports as stalled while this holds.`);
2643
+ }
679
2644
  /**
680
2645
  * Decides whether a conditional dependency edge is live.
681
2646
  *
@@ -683,31 +2648,127 @@ export class TaskGraphDispatcher {
683
2648
  * only information a runtime graph has to branch on. Returns `'drop'` only on a definite false;
684
2649
  * an unevaluable condition keeps the edge for the reason stated at the call site.
685
2650
  */
686
- evaluateEdgeCondition(dep, entityById) {
2651
+ evaluateEdgeCondition(dep, entityById, failureSemantics, invocation, debug) {
2652
+ // An operator's override answers the edge BEFORE the condition is consulted — an override
2653
+ // exists precisely because the condition cannot be answered (or answered wrongly), so
2654
+ // evaluating first would re-produce the hold the override exists to end.
2655
+ const override = OverrideVerdictFor(debug ?? {}, dep.ID);
2656
+ if (override) {
2657
+ return {
2658
+ outcome: override === 'true' ? 'keep' : 'drop',
2659
+ reason: `answered '${override}' by an operator override`,
2660
+ decided: true,
2661
+ };
2662
+ }
687
2663
  const upstream = entityById.get(dep.DependsOnTaskID);
688
2664
  if (!upstream)
689
- return 'keep';
690
- let output = null;
691
- if (upstream.OutputPayload) {
692
- try {
693
- output = JSON.parse(upstream.OutputPayload);
694
- }
695
- catch { /* a malformed payload is not grounds to drop a prerequisite */ }
696
- }
697
- const result = this.conditionEvaluator.Evaluate(dep.Condition, {
698
- status: upstream.Status,
699
- succeeded: upstream.Status === 'Complete',
700
- failed: upstream.Status === 'Failed',
701
- output,
702
- errorMessage: upstream.ErrorMessage ?? null,
2665
+ return { outcome: 'keep', decided: false };
2666
+ // The DECISION lives in `condition-gate`; what stays here is the loading and the logging.
2667
+ //
2668
+ // `DecideGate` takes the evaluation as a thunk rather than a value, and that is the fix, not
2669
+ // a style: the terminality guard has to stop the evaluation happening at all. Evaluating
2670
+ // `succeeded` against a still-Pending origin does not fail — it returns a confident, wrong
2671
+ // `false`, the edge is dropped, and the target is Blocked at wave one before the origin ever
2672
+ // ran. That killed every conditioned linear chain, with no error anywhere.
2673
+ let unevaluableError;
2674
+ let evaluated = false;
2675
+ const outcome = DecideGate(upstream.Status, failureSemantics, () => {
2676
+ evaluated = true;
2677
+ const result = this.conditionEvaluator.Evaluate(dep.Condition, BuildConditionContext(upstream, ParseConditionOutput(upstream.OutputPayload), invocation));
2678
+ if (!result.Success)
2679
+ unevaluableError = result.ErrorMessage;
2680
+ return result;
703
2681
  });
704
- if (!result.Success) {
705
- LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
706
- `(${result.ErrorMessage}); keeping the edge so the graph stalls visibly rather than ` +
707
- `running ${dep.TaskID} out of order.`);
708
- return 'keep';
2682
+ // Reported here rather than inside the decision, so the pure part stays pure and a held edge
2683
+ // is still loud once — see logUnevaluableConditionOnce.
2684
+ if (outcome === 'hold')
2685
+ this.logUnevaluableConditionOnce(dep, unevaluableError);
2686
+ return {
2687
+ outcome,
2688
+ reason: outcome === 'hold'
2689
+ ? (unevaluableError ?? 'the condition cannot be answered yet')
2690
+ : undefined,
2691
+ // A verdict was rendered only when the gate was genuinely ASKED. Terminality alone is
2692
+ // not enough since R3-2: a Failed origin under 'block' (and a Cancelled origin under
2693
+ // either dialect) returns 'keep' WITHOUT evaluating, so the block cascade owns the
2694
+ // target — announcing that as "satisfied" would tell a viewer a gate opened that
2695
+ // DecideGate deliberately never asked. `evaluated` is set by the thunk itself, so this
2696
+ // cannot drift from the gate's own rules; a Skipped origin's unevaluated drop is still
2697
+ // a verdict a viewer needs (the branch was not taken).
2698
+ decided: evaluated || (upstream.Status === 'Skipped' && outcome === 'drop'),
2699
+ };
2700
+ }
2701
+ /**
2702
+ * An exclusive edge's condition as a three-way outcome.
2703
+ *
2704
+ * `ResolveExclusiveGroups` needs to tell "false" from "could not be evaluated": the first loses
2705
+ * the branch, the second holds the whole group. The generic keep/drop path cannot express that
2706
+ * difference, which is why exclusive edges take this route instead.
2707
+ */
2708
+ evaluateExclusiveCondition(dep, entityById, invocation, debug) {
2709
+ // Same override-first rule as ordinary edges — see evaluateEdgeCondition.
2710
+ const override = OverrideVerdictFor(debug ?? {}, dep.ID);
2711
+ if (override)
2712
+ return override === 'true' ? 'satisfied' : 'unsatisfied';
2713
+ if (!dep.Condition?.trim())
2714
+ return 'satisfied';
2715
+ const upstream = entityById.get(dep.DependsOnTaskID);
2716
+ if (!upstream)
2717
+ return 'unevaluable';
2718
+ const result = this.conditionEvaluator.Evaluate(dep.Condition,
2719
+ // The invocation envelope rides the EXCLUSIVE dialect too. R3-3 threaded it into the
2720
+ // ordinary path; a flow's XOR branch reading `data.userApproval` is the same documented
2721
+ // condition on a different edge kind, and `BuildConditionContext`'s defaulted parameter
2722
+ // made omitting it here silently evaluate those roots against nothing.
2723
+ BuildConditionContext(upstream, ParseConditionOutput(upstream.OutputPayload), invocation));
2724
+ // SAME CLASSIFICATION AS THE ORDINARY DIALECT (R2-3). The null-safe envelope already makes
2725
+ // one level of absence read as false here, but a deeper absent chain still throws — and
2726
+ // calling that 'unevaluable' would hold the whole group forever on a terminal origin, while
2727
+ // `DecideGate` would have dropped the identical condition. Two dialects, one question.
2728
+ if (!result.Success)
2729
+ return IsBrokenGuard(result.ErrorMessage) ? 'unevaluable' : 'unsatisfied';
2730
+ return result.Value ? 'satisfied' : 'unsatisfied';
2731
+ }
2732
+ /**
2733
+ * Announces gate verdicts that CHANGED since this instance last looked.
2734
+ *
2735
+ * Fire-and-forget by design: `loadGraphState` is synchronous graph assembly, and the owner
2736
+ * lookup the frame needs is async — so the emission floats behind rather than making state
2737
+ * loading wait on observability. Frames are commentary, never a step of the work.
2738
+ */
2739
+ emitGateDecisions(provider, parentTaskID, decisions, entityById) {
2740
+ if (!this.observer || decisions.length === 0)
2741
+ return;
2742
+ let perEdge = this.emittedGateVerdicts.get(parentTaskID);
2743
+ if (!perEdge) {
2744
+ perEdge = new Map();
2745
+ this.emittedGateVerdicts.set(parentTaskID, perEdge);
709
2746
  }
710
- return result.Value ? 'keep' : 'drop';
2747
+ const changed = decisions.filter((d) => {
2748
+ const key = `${d.verdict}|${d.reason ?? ''}`;
2749
+ if (perEdge.get(d.edge.ID) === key)
2750
+ return false;
2751
+ perEdge.set(d.edge.ID, key);
2752
+ return true;
2753
+ });
2754
+ if (changed.length === 0)
2755
+ return;
2756
+ void this.resolveOwner(provider, parentTaskID).then((owner) => {
2757
+ for (const d of changed) {
2758
+ this.emit({
2759
+ Kind: 'GateDecision',
2760
+ ParentTaskID: parentTaskID,
2761
+ OwnerUserID: owner,
2762
+ TaskID: d.edge.TaskID,
2763
+ TaskName: entityById.get(d.edge.TaskID)?.Name,
2764
+ EdgeID: d.edge.ID,
2765
+ DependsOnTaskID: d.edge.DependsOnTaskID,
2766
+ Verdict: d.verdict,
2767
+ ConditionText: d.edge.Condition ?? undefined,
2768
+ Reason: d.reason,
2769
+ });
2770
+ }
2771
+ }).catch(() => { });
711
2772
  }
712
2773
  /** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
713
2774
  async loadDependencyOutputs(provider, taskID) {
@@ -731,5 +2792,759 @@ export class TaskGraphDispatcher {
731
2792
  }
732
2793
  return outputs;
733
2794
  }
2795
+ /**
2796
+ * Runs one task's body, whatever kind of step it is.
2797
+ *
2798
+ * **Routing is on `StepType`, not on which key happens to be set.** A loop step carries the same
2799
+ * `ActionID` or `AgentID` as an ordinary step — that key is what the loop *repeats* — so the old
2800
+ * `task.ActionID ? action : agent` test would have run a loop exactly once and called it done.
2801
+ * `StepType` is the only field that distinguishes them.
2802
+ *
2803
+ * Every branch is normalized to one shape so the recording path above stays single: an action has
2804
+ * no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
2805
+ */
2806
+ async runTaskBody(task, provider, inputPayload, dependencyOutputs, onProgress) {
2807
+ const payload = this.mergedPayload(inputPayload, dependencyOutputs);
2808
+ const config = task.ConfigurationObject;
2809
+ // A loop's own step type decides how many times its body runs; the body itself is dispatched
2810
+ // through the very same runners as a one-shot step.
2811
+ if (task.StepType === 'ForEach' || task.StepType === 'While') {
2812
+ return { ...await this.runLoopTask(task, provider, payload, dependencyOutputs), PayloadAtStart: payload };
2813
+ }
2814
+ const { params, errors } = BuildMappedInput(config?.inputMapping, { payload });
2815
+ for (const e of errors)
2816
+ LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
2817
+ // `payload`, NOT `inputPayload` — the MERGED value computed above, which includes what every
2818
+ // dependency produced.
2819
+ //
2820
+ // A step with an input mapping got exactly the parameters it declared; a step WITHOUT one
2821
+ // fell back to the raw input and therefore saw nothing any earlier step had produced. For a
2822
+ // Prompt step — which declares no mapping by design, because it reads the whole payload
2823
+ // through `{{ _CURRENT_PAYLOAD }}` — that meant the placeholder rendered `{}` and the model
2824
+ // was asked to write from an empty brief.
2825
+ //
2826
+ // It answered anyway. The Content Pipeline's draft step said "the research data was empty",
2827
+ // which was TRUE of what it had been handed while twenty research results sat in the
2828
+ // dependency outputs beside it, and the reviewer then rejected the draft for saying so.
2829
+ // Every layer looked like it was working.
2830
+ const effectiveInput = Object.keys(params).length > 0 ? params : payload;
2831
+ if (task.StepType === 'Prompt') {
2832
+ if (!this.promptRunner) {
2833
+ // Not a failure: "nobody here can run this" is not "this ran and did not work".
2834
+ return { Success: false, AgentRunID: null, ErrorMessage: 'No prompt runner is loaded on this host.' };
2835
+ }
2836
+ const promptResult = await this.promptRunner.RunPromptForTask({
2837
+ TaskID: task.ID,
2838
+ PromptID: task.PromptID,
2839
+ InputPayload: effectiveInput,
2840
+ DependencyOutputs: dependencyOutputs,
2841
+ TemplateParameters: config?.prompt?.templateParameters,
2842
+ Provider: provider,
2843
+ ContextUser: this.contextUser,
2844
+ OnProgress: onProgress,
2845
+ });
2846
+ // A prompt's response is DEEP-MERGED into the payload rather than replacing it. A prompt
2847
+ // answers one question; replacing the payload with its answer would discard everything
2848
+ // the steps before it established, which is how a late step loses the data it depends on.
2849
+ const merged = promptResult.Success && promptResult.Output && typeof promptResult.Output === 'object'
2850
+ ? deepMergePayload(payload, promptResult.Output)
2851
+ : payload;
2852
+ return {
2853
+ Success: promptResult.Success,
2854
+ AgentRunID: null,
2855
+ ErrorMessage: promptResult.ErrorMessage,
2856
+ Output: this.applyStepOutputMapping(task, merged, merged, config?.outputMapping),
2857
+ PayloadAtStart: payload,
2858
+ ChatMessage: promptResult.ChatMessage,
2859
+ // Returned even when the prompt FAILED. A failed prompt still cost tokens, and a
2860
+ // cost rollup that silently omits failures under-reports exactly the runs someone
2861
+ // is most likely to be investigating.
2862
+ PromptRunID: promptResult.PromptRunID,
2863
+ };
2864
+ }
2865
+ const raw = task.ActionID
2866
+ ? { ...await this.actionRunner.RunActionForTask({
2867
+ TaskID: task.ID,
2868
+ ActionID: task.ActionID,
2869
+ InputPayload: effectiveInput,
2870
+ DependencyOutputs: dependencyOutputs,
2871
+ Provider: provider,
2872
+ ContextUser: this.contextUser,
2873
+ OnProgress: onProgress,
2874
+ }), AgentRunID: null }
2875
+ : await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs, onProgress);
2876
+ return {
2877
+ ...raw,
2878
+ Output: this.applyStepOutputMapping(task, payload, raw.Output, config?.outputMapping),
2879
+ PayloadAtStart: payload,
2880
+ };
2881
+ }
2882
+ /**
2883
+ * Runs a loop step: its body once per iteration, with the item and index in scope.
2884
+ *
2885
+ * The loop's own `Configuration` supplies the definition; the row's `ActionID` / `AgentID`
2886
+ * supplies what to repeat. Per-iteration inputs are resolved fresh each pass — the bindings are
2887
+ * merged into the payload before the mapping is applied, which is how a body can reference the
2888
+ * current item at all.
2889
+ */
2890
+ async runLoopTask(task, provider, payload, dependencyOutputs) {
2891
+ const config = task.ConfigurationObject;
2892
+ const op = task.StepType === 'ForEach' ? config?.forEach : config?.while;
2893
+ if (!op) {
2894
+ return {
2895
+ Success: false,
2896
+ AgentRunID: null,
2897
+ ErrorMessage: `"${task.Name}" is a ${task.StepType} step with no loop settings, so there is nothing to repeat.`,
2898
+ };
2899
+ }
2900
+ // A prompt body has no params of its own — it receives the payload (with the loop bindings
2901
+ // merged in) through the placeholder, so an empty mapping is correct rather than missing.
2902
+ const bodyMapping = (op.action?.params ?? {});
2903
+ // The BODY's output mapping, applied once per pass — see `foldIterationOutput`.
2904
+ //
2905
+ // It used to be applied a single time after the loop finished, against the accumulated
2906
+ // payload. That is the wrong moment in two ways at once: the mapping names an output
2907
+ // PARAMETER of the body, which no longer exists by then, and a mapping like
2908
+ // `"Items": "results[]"` can only append per pass. So every pass merged its raw result into
2909
+ // the shared payload instead, each overwriting the last, and the mapping matched nothing and
2910
+ // wrote nothing. A ForEach over five items reported five successes and kept item five.
2911
+ const bodyOutputMapping = op.action?.outputMapping ?? op.prompt?.outputMapping;
2912
+ // Where this step sits in its graph, resolved ONCE rather than per iteration. A loop body is
2913
+ // dispatched exactly like a one-shot step and needs the same two things: the run that
2914
+ // submitted the graph (so a spawned run gets a ParentRunID and is visible to the tree and to
2915
+ // cost), and the continuation depth (so the recursion cap still applies). Omitting them made
2916
+ // loop bodies second-class in every dimension — and reopened the unbounded-recursion hole
2917
+ // THROUGH loops, since each spawned run restarted the chain at zero.
2918
+ const graphContext = await this.graphContext(provider, task);
2919
+ // THE LOOP'S PAYLOAD ACCUMULATES. Each iteration's output merges in, and the next iteration
2920
+ // — and the While condition — sees it. Without this the condition closure re-read the
2921
+ // payload as it was when the loop STARTED, so a `while payload.brandOK !== true` could never
2922
+ // become false: the loop burned every iteration re-examining the original input and always
2923
+ // took the give-up branch, making the other branch unreachable. The loop ran, reported
2924
+ // success, and its result was predetermined.
2925
+ let livePayload = { ...payload };
2926
+ // One entry per pass, so the loop's work exists somewhere the platform can see it. Without
2927
+ // this a loop is a single childless node: the run tree reaches nested work through six links
2928
+ // and an iteration is none of them, so the passes were invisible to the timeline AND their
2929
+ // spend was missing from the settlement rollup. See ITaskStepRuntime.iterations.
2930
+ const iterationTrace = [];
2931
+ // Bounds what the trace's payloads may cost. The pointers are never budgeted — those are the
2932
+ // durable record of the work and must survive whatever the payloads do.
2933
+ const budget = new IterationPayloadBudget();
2934
+ const invokeBody = async ({ Index, Bindings }) => {
2935
+ // Bindings go INTO the payload rather than beside it, so an authored mapping reaches the
2936
+ // current item the same way it reaches anything else: `payload.<itemVariable>`.
2937
+ const iterationPayload = { ...livePayload, ...Bindings };
2938
+ const resolved = ResolveMappedInput(bodyMapping, { payload: iterationPayload });
2939
+ /**
2940
+ * Folds an iteration's output into the running payload the next pass will see, and
2941
+ * records what the pass produced.
2942
+ *
2943
+ * The trace is written HERE rather than after the loop because a loop that fails partway
2944
+ * still ran the passes before it, and their runs are real spend that must not vanish
2945
+ * because the loop as a whole did not finish.
2946
+ */
2947
+ const absorb = (outcome, bodyInput) => {
2948
+ livePayload = this.foldIterationOutput(task, livePayload, outcome.Output, bodyOutputMapping);
2949
+ iterationTrace.push({
2950
+ index: Index,
2951
+ // What THIS pass was handed and what it gave back — not the loop's running
2952
+ // payload before and after it.
2953
+ //
2954
+ // A pass has no row of its own, so without these there is nowhere its work can be
2955
+ // recorded: every iteration presented null on both sides and the run view could
2956
+ // say nothing about any single pass, which for a loop is the only interesting
2957
+ // question. But recording the RUNNING payload on both sides — the obvious reading
2958
+ // of "before and after" — is quadratic: each pass would hold a full copy of
2959
+ // everything every earlier pass accumulated. A five-iteration demo produced a
2960
+ // 121KB Configuration that way; the same loop over fifty items would produce
2961
+ // megabytes, in a column every reader of the row pays to load.
2962
+ //
2963
+ // The pass's own input and output are what a reader actually wants ("what did
2964
+ // pass three do?"), and they are constant-sized per pass.
2965
+ payloadAtStart: budget.Take(bodyInput),
2966
+ payloadAtEnd: budget.Take(outcome.Output),
2967
+ promptRunID: outcome.PromptRunID,
2968
+ agentRunID: outcome.AgentRunID,
2969
+ // An ACTION body records its log here. Omitting it left an action-bodied pass
2970
+ // with no pointer at all — no cost, no timing, nothing to open — and the tree,
2971
+ // seeing neither a prompt run nor an agent run, fell through to its last branch
2972
+ // and called the pass a Sub-Agent. A loop over a web search then showed five
2973
+ // sub-agent runs that never existed.
2974
+ actionLogID: outcome.ActionLogID,
2975
+ success: outcome.Success,
2976
+ errorMessage: outcome.ErrorMessage,
2977
+ });
2978
+ return outcome;
2979
+ };
2980
+ // A prompt body is checked FIRST because it is the only one whose id lives in its own
2981
+ // column: a loop repeating a prompt has PromptID set and both ActionID and AgentID null,
2982
+ // so falling through to the agent branch would dereference a null agent id.
2983
+ if (task.StepType && task.PromptID && !task.ActionID) {
2984
+ if (!this.promptRunner) {
2985
+ return { Success: false, ErrorMessage: 'No prompt runner is loaded on this host.' };
2986
+ }
2987
+ return absorb(await this.promptRunner.RunPromptForTask({
2988
+ TaskID: task.ID,
2989
+ PromptID: task.PromptID,
2990
+ // The ITERATION payload, not the mapped params. An action body declares its
2991
+ // inputs and gets exactly those; a prompt body declares none — it receives the
2992
+ // whole payload through the placeholder, and the loop's item and index are
2993
+ // merged INTO that payload. Passing the mapped result here handed the prompt an
2994
+ // empty object, so every iteration asked the model to describe nothing and got
2995
+ // five confident answers about nothing back.
2996
+ InputPayload: iterationPayload,
2997
+ DependencyOutputs: dependencyOutputs,
2998
+ // The loop's bindings become TEMPLATE VARIABLES, so an author writes
2999
+ // `{{ field }}` for the item the loop is on — which is what `itemVariable` is
3000
+ // for, and what anyone reading the step's configuration expects. Reaching it
3001
+ // through the payload placeholder instead works but is not discoverable, and
3002
+ // getting it wrong is silent: the variable renders empty and the model answers
3003
+ // confidently about nothing.
3004
+ TemplateParameters: { ...stringifyBindings(Bindings), ...op.prompt?.templateParameters },
3005
+ Provider: provider,
3006
+ ContextUser: this.contextUser,
3007
+ }), iterationPayload);
3008
+ }
3009
+ if (task.ActionID) {
3010
+ return absorb(await this.actionRunner.RunActionForTask({
3011
+ TaskID: task.ID,
3012
+ ActionID: task.ActionID,
3013
+ InputPayload: resolved,
3014
+ DependencyOutputs: dependencyOutputs,
3015
+ Provider: provider,
3016
+ ContextUser: this.contextUser,
3017
+ }), resolved);
3018
+ }
3019
+ const agentInput = Object.keys(resolved).length > 0 ? resolved : iterationPayload;
3020
+ return absorb(await this.agentRunner.RunAgentForTask({
3021
+ TaskID: task.ID,
3022
+ AgentID: task.AgentID,
3023
+ // The ITERATION payload when the body declares no inputs of its own. A sub-agent
3024
+ // body has no `params`, so the mapped result is `{}` — every iteration was handing
3025
+ // the agent nothing and asking it to work from that.
3026
+ InputPayload: agentInput,
3027
+ DependencyOutputs: dependencyOutputs,
3028
+ ContinuationDepth: graphContext.Depth,
3029
+ SubmittingAgentRunID: graphContext.SubmittingAgentRunID,
3030
+ Provider: provider,
3031
+ ContextUser: this.contextUser,
3032
+ }), agentInput);
3033
+ };
3034
+ const outcome = task.StepType === 'ForEach'
3035
+ ? await RunForEachLoop(op, { payload }, invokeBody)
3036
+ : await RunWhileLoop(op, (iteration) => this.conditionEvaluator.Evaluate(op.condition,
3037
+ // BOTH forms, because a workflow should not have two condition dialects. An
3038
+ // EDGE condition is written `payload.brandOK !== true`; a loop condition used
3039
+ // to see the payload's keys spread at the top level and nothing named `payload`,
3040
+ // so the same expression that routes an edge failed here with
3041
+ // "payload is not defined". The spread stays for conditions already written
3042
+ // against it.
3043
+ { ...livePayload, payload: livePayload, iteration }), invokeBody);
3044
+ return {
3045
+ Success: outcome.Success,
3046
+ AgentRunID: null,
3047
+ ErrorMessage: outcome.ErrorMessage,
3048
+ // Every pass that ran, including those before a failure — see `iterationTrace`.
3049
+ Iterations: iterationTrace.length > 0 ? iterationTrace : undefined,
3050
+ // The ACCUMULATED payload — everything the iterations established — not the one the
3051
+ // loop started with, which would discard the loop's whole effect on the workflow.
3052
+ //
3053
+ // Only the STEP's own mapping is applied here. The body's mapping already ran once per
3054
+ // pass inside `foldIterationOutput`; applying it again against the accumulated payload
3055
+ // is what used to make it match nothing.
3056
+ Output: this.applyStepOutputMapping(task, livePayload, outcome.Output, config?.outputMapping),
3057
+ };
3058
+ }
3059
+ /**
3060
+ * Folds one pass's result into the loop's running payload.
3061
+ *
3062
+ * **With a body mapping**, the pass's declared outputs are filed where the author said to put
3063
+ * them — including `name[]`, which appends, so a ForEach can collect one entry per item. That is
3064
+ * the whole point of a loop over a collection, and it is only expressible per pass.
3065
+ *
3066
+ * **Without one**, the raw result is deep-merged, which is the pre-existing behaviour and the
3067
+ * right default for a `While` that converges on a value: each pass refines what the condition
3068
+ * reads. It is the wrong default for a ForEach that collects — hence the mapping.
3069
+ *
3070
+ * An unmapped output is reported per pass rather than swallowed, for the same reason
3071
+ * {@link applyStepOutputMapping} reports it: a mapping that names something the body never
3072
+ * returned means the pass did work that went nowhere, while everything reports success.
3073
+ */
3074
+ foldIterationOutput(task, livePayload, output, bodyOutputMapping) {
3075
+ if (!output || typeof output !== 'object' || Array.isArray(output))
3076
+ return livePayload;
3077
+ const source = output;
3078
+ if (!bodyOutputMapping)
3079
+ return deepMergePayload(livePayload, source);
3080
+ // Applied ONTO a deep copy of the running payload, not into a fresh object: `name[]` appends,
3081
+ // and appending is meaningless without the list already there. The copy is deep because the
3082
+ // trace has already recorded earlier passes' payloads — mutating a shared nested array would
3083
+ // retroactively rewrite what those passes are recorded as having seen.
3084
+ const { updates, errors, unmapped } = ApplyOutputMapping(source, bodyOutputMapping, structuredClone(livePayload));
3085
+ for (const e of errors)
3086
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} loop body: ${e}`);
3087
+ if (unmapped?.length) {
3088
+ LogError(`[TaskGraphDispatcher] '${task.Name}' loop body mapped output(s) it did not return: ` +
3089
+ `${unmapped.join(', ')}. The pass returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
3090
+ `Those payload values were NOT written, so anything downstream reading them sees nothing.`);
3091
+ }
3092
+ // `updates` IS the copy that was applied onto, so it is already the complete next payload.
3093
+ return updates;
3094
+ }
3095
+ /**
3096
+ * Files a step's result into the payload it hands downstream.
3097
+ *
3098
+ * **This is what makes a branch condition possible.** A workflow that branches on
3099
+ * `payload.stockPrice` has that value only because this step mapped `CurrentPrice -> stockPrice`.
3100
+ * Without it the condition reads `undefined` — merely falsy — so the workflow takes the other
3101
+ * branch, finishes, and reports success with nothing to indicate anything went wrong.
3102
+ *
3103
+ * The incoming payload is carried through as well as the update, so a value written three steps
3104
+ * back is still readable here. Returning only this step's own output is what used to limit a
3105
+ * condition's view to its immediate predecessor.
3106
+ */
3107
+ applyStepOutputMapping(task, payload, output, outputMapping) {
3108
+ // No mapping: MERGE the step's output over the payload rather than replacing it.
3109
+ //
3110
+ // Replacing is what made the Content Pipeline's exclusive pair unreachable. A While loop's
3111
+ // own output is a SUMMARY — `{iterations, succeeded, failed, results}` — so returning it
3112
+ // discarded the payload the iterations had built, including the `brandOK` the reviewer had
3113
+ // just set to true. The edges read `payload.brandOK === true` and `!== true`; against a
3114
+ // summary the first is false and the second is true, so the give-up branch won on EVERY run
3115
+ // no matter what the reviewer decided. The approved branch was unreachable in practice while
3116
+ // being perfectly reachable on the canvas.
3117
+ //
3118
+ // This is the same rule the mapped path already follows two lines down, and the same rule
3119
+ // the doc comment above states. The no-mapping branch was simply not following it.
3120
+ if (!outputMapping) {
3121
+ return output && typeof output === 'object' && !Array.isArray(output)
3122
+ ? { ...payload, ...output }
3123
+ : output ?? payload;
3124
+ }
3125
+ const source = output && typeof output === 'object' ? output : { value: output };
3126
+ const { updates, errors, unmapped } = ApplyOutputMapping(source, outputMapping);
3127
+ for (const e of errors)
3128
+ LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
3129
+ // A mapping that names an output the step never produced discards that step's work while
3130
+ // the step reports Complete. It is not fatal — an action may emit a parameter only on some
3131
+ // paths — but it must not be silent, and naming what WAS returned turns a multi-table
3132
+ // forensic exercise into one line. The Content Pipeline demo lost an entire research pass
3133
+ // this way, every run, because its mapping named another action's parameter.
3134
+ if (unmapped?.length) {
3135
+ LogError(`[TaskGraphDispatcher] '${task.Name}' mapped output(s) the step did not return: ` +
3136
+ `${unmapped.join(', ')}. The step returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
3137
+ `Those payload values were NOT written, so anything downstream reading them sees nothing.`);
3138
+ }
3139
+ return { ...payload, ...updates };
3140
+ }
3141
+ /**
3142
+ * Runs an Agent step, telling the runner where in the graph it sits.
3143
+ *
3144
+ * Depth and provenance are read together because they come from the same row: the graph's parent
3145
+ * task knows both how many continuation hops led here and which run submitted it.
3146
+ */
3147
+ async runAgentNode(task, provider, effectiveInput, dependencyOutputs, onProgress) {
3148
+ const context = await this.graphContext(provider, task);
3149
+ return this.agentRunner.RunAgentForTask({
3150
+ TaskID: task.ID,
3151
+ AgentID: task.AgentID,
3152
+ InputPayload: effectiveInput,
3153
+ DependencyOutputs: dependencyOutputs,
3154
+ ContinuationDepth: context.Depth,
3155
+ SubmittingAgentRunID: context.SubmittingAgentRunID,
3156
+ Provider: provider,
3157
+ ContextUser: this.contextUser,
3158
+ OnProgress: onProgress,
3159
+ });
3160
+ }
3161
+ /**
3162
+ * Leaves exactly one open request standing for a task, withdrawing any others.
3163
+ *
3164
+ * The oldest wins — it is the one whose notification the assignee most likely already saw.
3165
+ *
3166
+ * **Called from every path that reads a task's open requests**, not only from the raise. That is
3167
+ * deliberate and it is the half R3-5 first got wrong: a task is notified exactly once, and
3168
+ * `notifyHumanTaskReady` returns at the marker forever after, so a duplicate minted after that
3169
+ * pass — by an instance that crashed between its insert and its de-dup, or by any build older
3170
+ * than this one — was never looked at again by the only code that could have collapsed it. The
3171
+ * waiting-task sweep is what actually reaches those.
3172
+ *
3173
+ * @param open the task's open requests, oldest first
3174
+ * @param keepIfSole the row this caller is responsible for, named only for the log
3175
+ */
3176
+ async withdrawDuplicateRequests(provider, taskID, open, keepIfSole) {
3177
+ if (open.length <= 1)
3178
+ return;
3179
+ try {
3180
+ LogStatus(`[TaskGraphDispatcher] Task ${taskID} had ${open.length} open requests — keeping the ` +
3181
+ `oldest (${open[0].ID}${UUIDsEqual(open[0].ID, keepIfSole) ? ', this instance\'s' : ''}) and withdrawing the rest.`);
3182
+ for (const duplicate of open.slice(1)) {
3183
+ duplicate.Status = 'Canceled';
3184
+ duplicate.Comments = 'A duplicate request for the same step; the earlier one stands.';
3185
+ if (!(await duplicate.Save())) {
3186
+ LogError(`[TaskGraphDispatcher] Could not withdraw duplicate request ${duplicate.ID}: ` +
3187
+ `${duplicate.LatestResult?.CompleteMessage ?? 'unknown error'}`);
3188
+ }
3189
+ }
3190
+ }
3191
+ catch (e) {
3192
+ LogError(`[TaskGraphDispatcher] Could not de-duplicate requests for task ${taskID}: ${e instanceof Error ? e.message : String(e)}`);
3193
+ }
3194
+ }
3195
+ /**
3196
+ * Closes the still-open asks raised for tasks that will never be answered.
3197
+ *
3198
+ * `Canceled` rather than `Expired`: nobody ran out of time, the ask was withdrawn — and the two
3199
+ * mean different things downstream, since an expired human step is treated as a FAILURE that a
3200
+ * give-up edge can route around, which would be a lie about a step the workflow decided it no
3201
+ * longer needed.
3202
+ *
3203
+ * Failures are logged and never propagated. The graph's outcome is already decided; refusing to
3204
+ * finish over an inbox row would trade a stale notification for a stalled workflow.
3205
+ */
3206
+ async withdrawOpenRequests(provider, taskIDs, reason) {
3207
+ if (taskIDs.length === 0)
3208
+ return;
3209
+ try {
3210
+ const idList = taskIDs.map((id) => `'${id}'`).join(',');
3211
+ const open = await RunView.FromMetadataProvider(provider).RunView({
3212
+ EntityName: 'MJ: AI Agent Requests',
3213
+ ExtraFilter: `Status='Requested' AND OriginatingTaskID IN (${idList})`,
3214
+ ResultType: 'entity_object',
3215
+ BypassCache: true,
3216
+ }, this.contextUser);
3217
+ if (!open.Success) {
3218
+ LogError(`[TaskGraphDispatcher] Could not read open requests to withdraw: ${open.ErrorMessage}`);
3219
+ return;
3220
+ }
3221
+ for (const request of open.Results ?? []) {
3222
+ request.Status = 'Canceled';
3223
+ request.Comments = reason;
3224
+ if (!(await request.Save())) {
3225
+ LogError(`[TaskGraphDispatcher] Could not withdraw request ${request.ID}: ` +
3226
+ `${request.LatestResult?.CompleteMessage ?? 'unknown error'}. It will keep showing ` +
3227
+ `in someone's inbox for a step that will never run.`);
3228
+ }
3229
+ }
3230
+ }
3231
+ catch (e) {
3232
+ LogError(`[TaskGraphDispatcher] Could not withdraw open requests: ${e instanceof Error ? e.message : String(e)}`);
3233
+ }
3234
+ }
3235
+ /**
3236
+ * Says once, per graph, that this instance settled work it cannot announce.
3237
+ *
3238
+ * Once because the sweep re-offers the graph every poll for the rest of its window, and a line
3239
+ * per poll would bury the thing it is trying to report — which is a DEPLOYMENT fact, not a graph
3240
+ * fact: if no instance anywhere carries a deliverer, these settlements never reach anyone.
3241
+ */
3242
+ reportUndeliverableOnce(parentID) {
3243
+ if (this.reportedUndeliverable.has(parentID))
3244
+ return;
3245
+ this.reportedUndeliverable.add(parentID);
3246
+ LogStatus(`[TaskGraphDispatcher] Graph ${parentID} has settled but this instance has no continuation ` +
3247
+ `deliverer, so it is leaving the announcement to a peer that has one. If no instance in ` +
3248
+ `this deployment can deliver, the settlement will never be announced.`);
3249
+ }
3250
+ /**
3251
+ * Keeps a graph in this instance's sweep regardless of what its row timestamp says.
3252
+ *
3253
+ * Bounded, and the bound is about noise rather than surrender: past the cap the graph has failed
3254
+ * on every attempt for minutes, so another identical attempt will not fix it, and continuing
3255
+ * costs a full graph load per poll forever. It is reported once and left to the startup sweep.
3256
+ */
3257
+ keepRetryingSettlement(parentID) {
3258
+ const passes = (this.retryingSettlement.get(parentID) ?? 0) + 1;
3259
+ if (passes > MAX_SETTLEMENT_RETRY_PASSES) {
3260
+ this.retryingSettlement.delete(parentID);
3261
+ LogError(`[TaskGraphDispatcher] Graph ${parentID} has failed to settle on ${MAX_SETTLEMENT_RETRY_PASSES} ` +
3262
+ `consecutive passes; this instance will stop re-queueing it. Its submitting run may be ` +
3263
+ `left Paused. A restart's startup sweep will try again.`);
3264
+ return;
3265
+ }
3266
+ this.retryingSettlement.set(parentID, passes);
3267
+ }
3268
+ /**
3269
+ * A graph's durable metadata bag, for the questions a pass asks of it.
3270
+ *
3271
+ * Defaults on any failure to read it, and the defaults are the safe directions: `'block'` means
3272
+ * a failed step decides nothing, so a graph whose metadata we cannot read stalls visibly instead
3273
+ * of resolving forks on the say-so of a failure; and no early-finish declaration means nothing
3274
+ * is removed from the claim filter on the strength of a read that did not work.
3275
+ */
3276
+ async readParentMetadataFor(provider, parentTaskID) {
3277
+ try {
3278
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
3279
+ if (await parent.Load(parentTaskID))
3280
+ return ParseTaskGraphParentMetadata(parent.InputPayload);
3281
+ }
3282
+ catch { /* fall through to the safe defaults */ }
3283
+ return ParseTaskGraphParentMetadata(null);
3284
+ }
3285
+ /**
3286
+ * Whether the submitting run is in a state where this pass's writes to it will mean anything.
3287
+ *
3288
+ * **Read-only on purpose.** The settled branch's write order — layout, frame, cost, lifecycle,
3289
+ * delivery — is load-bearing and documented at each step; this asks the question those writes
3290
+ * depend on without joining them. What it prevents is a pass that goes through the motions and
3291
+ * then claims the delivery marker, making itself the last pass ever to look at the graph.
3292
+ *
3293
+ * Three answers, and the middle one is the bug:
3294
+ *
3295
+ * - **no run** — a scheduled or remote-triggered graph has nobody waiting. Proceed.
3296
+ * - **still `Running`** — `finalizeAgentRun` has not parked it yet. The graph beat its own
3297
+ * submitter to the finish line, which is ordinary for a fast graph and lasts milliseconds.
3298
+ * Defer: one poll later the run is parked and everything lands.
3299
+ * - **anything else** — `Paused` (settle it), or already `Completed`/`Failed`/`Cancelled` for
3300
+ * its own reasons (leave it; the lifecycle write's own guard declines). Proceed.
3301
+ *
3302
+ * **The deferral is bounded**, because "not parked yet" and "the submitting process died before
3303
+ * it could park" look identical from here. Waiting forever on the second would lose the outcome
3304
+ * of work that actually completed — strictly worse than announcing it late — so past the grace
3305
+ * period this proceeds and says why. The run itself stays `Running`, which is visibly wrong and
3306
+ * belongs to whatever reconciles abandoned runs, not to the graph that finished correctly.
3307
+ */
3308
+ async submittingRunReadiness(provider, parent) {
3309
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
3310
+ if (!meta.submittedByAgentRunID)
3311
+ return { Verdict: 'ready', SubmitterCancelled: false };
3312
+ try {
3313
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
3314
+ if (!(await run.Load(meta.submittedByAgentRunID))) {
3315
+ // Transient, most likely. Deferring costs a poll; proceeding costs the marker.
3316
+ LogError(`[TaskGraphDispatcher] Could not read run ${meta.submittedByAgentRunID} to check whether graph ${parent.ID} may settle it; retrying next pass.`);
3317
+ return { Verdict: 'defer', SubmitterCancelled: false };
3318
+ }
3319
+ // A CANCELLED SUBMITTER HAS NOBODY WAITING (R2-9). Settlement still runs — the graph's
3320
+ // own bookkeeping is owed either way — but announcing it would message a conversation
3321
+ // about a workflow the user stopped, and for `reinvoke` would start a fresh billed turn
3322
+ // for the agent they cancelled.
3323
+ const cancelled = run.Status === 'Cancelled';
3324
+ const settledFor = parent.CompletedAt ? Date.now() - parent.CompletedAt.getTime() : 0;
3325
+ if (!IsSubmittingRunReady(run.Status, settledFor)) {
3326
+ // Still `Running` and inside the grace: `finalizeAgentRun` has not parked it yet.
3327
+ // Defer the whole run-half so nothing claims the marker — see the call site.
3328
+ return { Verdict: 'defer', SubmitterCancelled: cancelled };
3329
+ }
3330
+ if (run.Status === 'Running') {
3331
+ // Ready DESPITE being unparked means the grace has expired: the submitting process
3332
+ // most likely died before it could park. Proceeding loses nothing that is still
3333
+ // recoverable and stops a dead submitter holding a finished workflow's outcome.
3334
+ LogError(`[TaskGraphDispatcher] Run ${run.ID} has been Running for ${Math.round(settledFor / 1000)}s ` +
3335
+ `since graph ${parent.ID} settled — it never parked, so its submitting process most ` +
3336
+ `likely died. Settling and delivering the graph anyway; the run needs separate attention.`);
3337
+ }
3338
+ return { Verdict: 'ready', SubmitterCancelled: cancelled };
3339
+ }
3340
+ catch (e) {
3341
+ LogError(`[TaskGraphDispatcher] Could not check the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
3342
+ return { Verdict: 'defer', SubmitterCancelled: false };
3343
+ }
3344
+ }
3345
+ /**
3346
+ * Completes the agent run that parked on this graph.
3347
+ *
3348
+ * **This is the other half of submit-and-detach.** A run that dispatches a graph does not
3349
+ * complete at submission — it ends `Paused`, because reporting `Completed` above a workflow
3350
+ * where nothing has happened yet is a claim the row cannot support. The run's lifecycle is
3351
+ * finished HERE, when the graph it was waiting on actually settles, which is the first moment
3352
+ * the answer exists.
3353
+ *
3354
+ * Doing it from the dispatcher rather than by awaiting in the agent is what keeps the properties
3355
+ * that made detach right in the first place: a graph containing a human approval can park for
3356
+ * days without holding a conversation turn open, and a graph reclaimed by another instance after
3357
+ * a crash still settles its submitting run, because the settling happens wherever the graph
3358
+ * finishes rather than wherever it started.
3359
+ *
3360
+ * **Only a parked run is touched.** A run that is already `Completed`, `Failed` or `Cancelled`
3361
+ * reached that state for its own reasons — a second graph settling later, a run the user
3362
+ * cancelled, a run that failed after submitting — and overwriting it would rewrite history from
3363
+ * the outside. The `Paused` predicate is the whole guard.
3364
+ *
3365
+ * @param graphStatus the parent rollup's status: what the workflow as a whole did
3366
+ */
3367
+ async settleSubmittingRun(provider, parent, graphStatus) {
3368
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
3369
+ if (!meta.submittedByAgentRunID)
3370
+ return 'done'; // a scheduled or remote-triggered graph has nobody waiting
3371
+ try {
3372
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
3373
+ if (!(await run.Load(meta.submittedByAgentRunID))) {
3374
+ LogError(`[TaskGraphDispatcher] Could not load run ${meta.submittedByAgentRunID} to settle it against graph ${parent.ID}.`);
3375
+ return 'defer';
3376
+ }
3377
+ if (run.Status !== 'Paused')
3378
+ return 'done';
3379
+ // The workflow's outcome becomes the run's outcome. A graph that ended any way other than
3380
+ // Complete did not do what the run started it to do, and a run reporting success over it
3381
+ // would be the same untruth in a different place.
3382
+ const succeeded = graphStatus === 'Complete';
3383
+ // COLUMN-SCOPED AND GUARDED ON `Paused` (C4), for the same reason every parent write has
3384
+ // been since Round 1: a full-row save carries a stale snapshot of a row another instance
3385
+ // may have moved, and the predicate makes the transition once-only rather than
3386
+ // last-write-wins.
3387
+ const settled = await this.claims.TrySettleRun(provider, run.ID, succeeded, succeeded ? null : `The workflow "${parent.Name}" ended ${graphStatus}.`, this.contextUser);
3388
+ if (!settled) {
3389
+ // Left parked rather than forced. A run stuck at Paused is visibly unfinished, which
3390
+ // is a state someone can investigate; a run flipped to Completed by a write that did
3391
+ // not land would be the same lie this whole change removes.
3392
+ // Rowcount 0 is either "a peer settled it first" — fine, and the status read above
3393
+ // would have caught the common case — or a write that did not land. Deferring covers
3394
+ // both: a peer's settle makes the next pass's `Paused` check return `done`.
3395
+ LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}; ` +
3396
+ `it is no longer Paused or the write did not land. Retrying next pass.`);
3397
+ return 'defer';
3398
+ }
3399
+ LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${succeeded ? 'Completed' : 'Failed'} — ` +
3400
+ `workflow "${parent.Name}" ended ${graphStatus}.`);
3401
+ return 'done';
3402
+ }
3403
+ catch (e) {
3404
+ LogError(`[TaskGraphDispatcher] Could not settle the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
3405
+ return 'defer';
3406
+ }
3407
+ }
3408
+ /**
3409
+ * Gives every step that lacks one a position, once the graph has finished.
3410
+ *
3411
+ * **Why the run stores geometry at all.** A `TaskGraphSpec` is a logical structure with no
3412
+ * layout field, so a graph an agent emitted has no opinion about where its boxes go. Every
3413
+ * viewer was therefore laying it out for itself at render time — and a viewer that failed to
3414
+ * (because the canvas measures nodes it has not drawn yet) fell back to every node at the
3415
+ * origin, piled on one another, with the zoom-to-fit that follows fitting a one-node bounding
3416
+ * box. Settling it once, server-side, means the agent-run canvas, the Workflows runs tab and
3417
+ * anything built later all draw the same picture, and none of them has to compute it.
3418
+ *
3419
+ * **An authored position is never overwritten.** A workflow compiled from a Flow agent carries
3420
+ * the arrangement someone dragged into place; replacing it with an algorithm's guess would
3421
+ * discard a deliberate act. Only steps with no geometry get one, so a partially-arranged graph
3422
+ * keeps what it has.
3423
+ *
3424
+ * Failure here is logged and swallowed: this is presentation. A graph whose work completed must
3425
+ * not be reported as failed because its picture could not be saved.
3426
+ */
3427
+ async persistComputedLayout(graph) {
3428
+ try {
3429
+ const needsLayout = [...graph.entityById.values()].filter((t) => !this.parseConfiguration(t)?.layout);
3430
+ if (needsLayout.length === 0)
3431
+ return;
3432
+ // Laid out over the WHOLE graph, not just the nodes missing geometry: position depends on
3433
+ // where a node sits in the topology, and a layout computed over a subset would place its
3434
+ // nodes as though the rest of the workflow did not exist.
3435
+ const edges = graph.edges.map((e) => ({ From: e.dependsOnTaskId, To: e.taskId }));
3436
+ const positions = LayoutGraphNodes([...graph.entityById.keys()], edges, { Direction: 'LR' });
3437
+ for (const task of needsLayout) {
3438
+ const position = positions.get(task.ID);
3439
+ if (!position)
3440
+ continue;
3441
+ const existing = this.parseConfiguration(task);
3442
+ const merged = {
3443
+ ...existing,
3444
+ layout: { x: position.X, y: position.Y },
3445
+ };
3446
+ task.Configuration = JSON.stringify(merged);
3447
+ if (!(await task.Save())) {
3448
+ LogError(`[TaskGraphDispatcher] Could not save computed layout for ${task.ID}: ${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
3449
+ }
3450
+ }
3451
+ }
3452
+ catch (e) {
3453
+ LogError(`[TaskGraphDispatcher] Could not compute a layout for the settled graph: ${e instanceof Error ? e.message : String(e)}`);
3454
+ }
3455
+ }
3456
+ /**
3457
+ * The earliest moment any step in the graph began, or null when none has.
3458
+ *
3459
+ * Null is a real answer — a graph whose tasks are all still Pending has not started — and is
3460
+ * deliberately not collapsed to "now", which would date the graph from whenever this pass
3461
+ * happened to run.
3462
+ */
3463
+ earliestStart(entityById) {
3464
+ let earliest = null;
3465
+ for (const entity of entityById.values()) {
3466
+ if (!entity.StartedAt)
3467
+ continue;
3468
+ if (earliest === null || entity.StartedAt < earliest)
3469
+ earliest = entity.StartedAt;
3470
+ }
3471
+ return earliest;
3472
+ }
3473
+ /**
3474
+ * The step's Configuration with this run's artefacts folded in, or `undefined` to leave it be.
3475
+ *
3476
+ * **Merged into the authored bag, never written over it.** The Configuration column holds the
3477
+ * step's definition — its loop body, its mappings, its policy, the position someone dragged it
3478
+ * to. Writing a fresh object containing only `runtime` would erase all of that the first time a
3479
+ * prompt step completed, which is the kind of loss that surfaces much later as a workflow that
3480
+ * mysteriously stopped mapping its output.
3481
+ *
3482
+ * Returns `undefined` when there is nothing to record, so the guarded write omits the column
3483
+ * rather than rewriting it with what it already held.
3484
+ */
3485
+ configurationWithRuntime(task, promptRunID, actionLogID, iterations, payloadAtStart) {
3486
+ if (!promptRunID && !actionLogID && !iterations?.length && !payloadAtStart)
3487
+ return undefined;
3488
+ const existing = this.parseConfiguration(task);
3489
+ const merged = {
3490
+ ...existing,
3491
+ runtime: {
3492
+ ...existing?.runtime,
3493
+ ...(promptRunID ? { promptRunID } : {}),
3494
+ ...(actionLogID ? { actionLogID } : {}),
3495
+ // Replaced wholesale rather than appended: this is the trace of the loop's LAST
3496
+ // execution, and a retried step that concatenated would report a loop that ran twice
3497
+ // as many passes as it did.
3498
+ ...(iterations?.length ? { iterations } : {}),
3499
+ // The resolved before-state, so the run view has something to diff the output
3500
+ // against. NOT written to Task.InputPayload, which holds the AUTHORED input and
3501
+ // round-trips back out as part of the spec.
3502
+ ...(payloadAtStart ? { payloadAtStart } : {}),
3503
+ },
3504
+ };
3505
+ return JSON.stringify(merged);
3506
+ }
3507
+ /**
3508
+ * Reads a step's Configuration bag, tolerating a row whose JSON cannot be parsed.
3509
+ *
3510
+ * Unparseable configuration is logged rather than thrown: the step has already RUN by the time
3511
+ * this is called, and refusing to record its outcome because its definition is malformed would
3512
+ * discard the result of real work and leave the task claimed until the claim lapsed.
3513
+ */
3514
+ parseConfiguration(task) {
3515
+ if (!task.Configuration)
3516
+ return undefined;
3517
+ try {
3518
+ return JSON.parse(task.Configuration);
3519
+ }
3520
+ catch (e) {
3521
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} has unparseable Configuration; ` +
3522
+ `recording runtime artefacts against an empty bag. ${e instanceof Error ? e.message : String(e)}`);
3523
+ return undefined;
3524
+ }
3525
+ }
3526
+ /**
3527
+ * The payload a step sees: everything its prerequisites produced, plus its own declared input.
3528
+ *
3529
+ * **Why the outputs are merged rather than kept per-task.** A flow carried ONE payload that
3530
+ * accumulated as it went, so a condition on the edge into step C could read a value step A wrote.
3531
+ * Handing each task only its immediate predecessor's output would silently narrow that: the
3532
+ * condition reads `undefined`, which is falsy, and the workflow quietly takes a different route
3533
+ * than the flow it was compiled from. Merging in dependency order restores the accumulation.
3534
+ *
3535
+ * Later prerequisites win on a key collision, matching a flow's own last-write-wins behaviour.
3536
+ */
3537
+ mergedPayload(inputPayload, dependencyOutputs) {
3538
+ const merged = {};
3539
+ for (const output of dependencyOutputs.values()) {
3540
+ if (output && typeof output === 'object' && !Array.isArray(output)) {
3541
+ Object.assign(merged, output);
3542
+ }
3543
+ }
3544
+ if (inputPayload && typeof inputPayload === 'object' && !Array.isArray(inputPayload)) {
3545
+ Object.assign(merged, inputPayload);
3546
+ }
3547
+ return merged;
3548
+ }
734
3549
  }
735
3550
  //# sourceMappingURL=TaskGraphDispatcher.js.map