@memberjunction/task-graph 0.0.0 → 6.1.0-edge.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/LICENSE +7 -0
  2. package/README.md +167 -27
  3. package/dist/DispatcherConditionEvaluator.d.ts +10 -0
  4. package/dist/DispatcherConditionEvaluator.d.ts.map +1 -0
  5. package/dist/DispatcherConditionEvaluator.js +27 -0
  6. package/dist/DispatcherConditionEvaluator.js.map +1 -0
  7. package/dist/TaskClaimStore.d.ts +125 -0
  8. package/dist/TaskClaimStore.d.ts.map +1 -0
  9. package/dist/TaskClaimStore.js +223 -0
  10. package/dist/TaskClaimStore.js.map +1 -0
  11. package/dist/TaskGraphDispatcher.d.ts +548 -0
  12. package/dist/TaskGraphDispatcher.d.ts.map +1 -0
  13. package/dist/TaskGraphDispatcher.js +2232 -0
  14. package/dist/TaskGraphDispatcher.js.map +1 -0
  15. package/dist/TaskGraphService.d.ts +247 -0
  16. package/dist/TaskGraphService.d.ts.map +1 -0
  17. package/dist/TaskGraphService.js +753 -0
  18. package/dist/TaskGraphService.js.map +1 -0
  19. package/dist/TaskGraphSubmitterImpl.d.ts +7 -0
  20. package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -0
  21. package/dist/TaskGraphSubmitterImpl.js +52 -0
  22. package/dist/TaskGraphSubmitterImpl.js.map +1 -0
  23. package/dist/TaskLoopExecutor.d.ts +62 -0
  24. package/dist/TaskLoopExecutor.d.ts.map +1 -0
  25. package/dist/TaskLoopExecutor.js +248 -0
  26. package/dist/TaskLoopExecutor.js.map +1 -0
  27. package/dist/WorkflowSpecSync.d.ts +197 -0
  28. package/dist/WorkflowSpecSync.d.ts.map +1 -0
  29. package/dist/WorkflowSpecSync.js +474 -0
  30. package/dist/WorkflowSpecSync.js.map +1 -0
  31. package/dist/index.d.ts +19 -0
  32. package/dist/index.d.ts.map +1 -0
  33. package/dist/index.js +19 -0
  34. package/dist/index.js.map +1 -0
  35. package/dist/operations/TaskGraphOperations.d.ts +39 -0
  36. package/dist/operations/TaskGraphOperations.d.ts.map +1 -0
  37. package/dist/operations/TaskGraphOperations.js +168 -0
  38. package/dist/operations/TaskGraphOperations.js.map +1 -0
  39. package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
  40. package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
  41. package/dist/operations/WorkflowDraftOperation.js +141 -0
  42. package/dist/operations/WorkflowDraftOperation.js.map +1 -0
  43. package/dist/operations/WorkflowOperations.d.ts +22 -0
  44. package/dist/operations/WorkflowOperations.d.ts.map +1 -0
  45. package/dist/operations/WorkflowOperations.js +99 -0
  46. package/dist/operations/WorkflowOperations.js.map +1 -0
  47. package/dist/types.d.ts +328 -0
  48. package/dist/types.d.ts.map +1 -0
  49. package/dist/types.js +9 -0
  50. package/dist/types.js.map +1 -0
  51. package/package.json +36 -8
@@ -0,0 +1,2232 @@
1
+ /**
2
+ * @fileoverview Durable, host-agnostic execution of submitted task graphs.
3
+ *
4
+ * The dispatcher is what makes task graphs survive things the old client-driven path could not: a
5
+ * page reload, the submitting agent run ending, a server restart, or a second server instance
6
+ * running the same table. It polls for claimable work, claims atomically, executes with a fresh
7
+ * provider per task, and reconciles orphaned state on a timer.
8
+ *
9
+ * **What it deliberately does not do.** It does not decide graph semantics. Eligibility, failure
10
+ * propagation, parent rollup and stall detection all come from the pure algorithms in
11
+ * `@memberjunction/ai-core-plus` — the same functions the in-run executor consumes. That is
12
+ * the whole reason those were factored out dependency-free: the in-run executor and the durable
13
+ * executor cannot drift apart if neither owns the rules.
14
+ *
15
+ * **Host-agnostic by construction.** Provider minting and agent execution arrive as injected
16
+ * dependencies (`ProviderFactory`, `TaskAgentRunner`), so this package never imports MJServer. The
17
+ * dependency runs MJServer -> task-graph, never the reverse.
18
+ *
19
+ * @module @memberjunction/task-graph
20
+ */
21
+ import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
22
+ import { LogError, LogStatus, RunView } from '@memberjunction/core';
23
+ import { ShutdownRegistry, UUIDsEqual } from '@memberjunction/global';
24
+ import { TaskClaimStore } from './TaskClaimStore.js';
25
+ import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
26
+ import { RunForEachLoop, RunWhileLoop } from './TaskLoopExecutor.js';
27
+ import { NotificationEngine } from '@memberjunction/notifications';
28
+ /** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
29
+ const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
30
+ /**
31
+ * Written to a human task's `ClaimedBy` once its assignee has been told it is ready.
32
+ *
33
+ * A human task has no executor, so the claim column is otherwise unused — which makes it the natural
34
+ * place to record a fact that must survive a restart. Reconciliation already exempts human tasks
35
+ * from reclamation, so this value is never mistaken for a live claim.
36
+ */
37
+ const HUMAN_TASK_NOTIFIED_MARKER = '__human-notified__';
38
+ /**
39
+ * The run-query capability of a provider, when it has one.
40
+ *
41
+ * `IMetadataProvider` does not extend `IRunQueryProvider`, but every provider that ships implements
42
+ * both. Narrowing by CAPABILITY rather than casting states that honestly: a provider that genuinely
43
+ * cannot run queries returns undefined and the caller reports it, instead of the call failing later
44
+ * behind a type assertion that claimed it could.
45
+ */
46
+ function asRunQueryProvider(provider) {
47
+ const candidate = provider;
48
+ return typeof candidate.RunQuery === 'function' ? candidate : undefined;
49
+ }
50
+ import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata } from './TaskGraphService.js';
51
+ import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
52
+ /**
53
+ * Renders a loop's bindings as template values.
54
+ *
55
+ * Template parameters are strings; an item is usually an object. Objects are JSON-encoded rather
56
+ * than dropped, because `{{ field }}` printing `[object Object]` — or nothing at all — is exactly
57
+ * the silent failure this exists to prevent.
58
+ */
59
+ function stringifyBindings(bindings) {
60
+ const out = {};
61
+ for (const [key, value] of Object.entries(bindings)) {
62
+ out[key] = typeof value === 'string' ? value : JSON.stringify(value, null, 2);
63
+ }
64
+ return out;
65
+ }
66
+ /**
67
+ * How much of a loop's per-pass payloads may be kept, and what happens when that runs out.
68
+ *
69
+ * **Why a budget exists at all.** A loop's trace lives inside one `Configuration` column, and its
70
+ * size is the product of two things nobody bounds: how many passes the loop runs, and how large the
71
+ * body's input and output are. A hundred-pass loop over documents would put megabytes in a column
72
+ * that the run tree, the timeline, the canvas and the Workflows list all read — punishing every
73
+ * reader of the row for a detail only someone inspecting one pass will ever open.
74
+ *
75
+ * **What it protects.** Only the payloads. `promptRunID` / `agentRunID` / `actionLogID` / `success`
76
+ * are always recorded: those point at the durable rows where the real forensics live, and they are
77
+ * what cost roll-up and the timeline traverse. Losing a payload costs a reader some detail; losing a
78
+ * pointer would lose the pass.
79
+ *
80
+ * **Omission is stated, never silent.** Once the budget is spent, further passes record a marker
81
+ * saying so and how large the value was, because a pass showing nothing is indistinguishable from a
82
+ * pass that produced nothing — and that ambiguity is exactly the failure this whole area keeps
83
+ * hitting.
84
+ */
85
+ const ITERATION_PAYLOAD_BUDGET_BYTES = 128 * 1024;
86
+ /** Per-value cap, so one enormous pass cannot consume the whole budget by itself. */
87
+ const ITERATION_PAYLOAD_VALUE_BYTES = 16 * 1024;
88
+ class IterationPayloadBudget {
89
+ constructor() {
90
+ this.spent = 0;
91
+ }
92
+ /**
93
+ * The value if it fits, or a marker describing what was left out.
94
+ *
95
+ * @returns the value, a marker object, or undefined when there was nothing to record
96
+ */
97
+ Take(value) {
98
+ if (value == null)
99
+ return undefined;
100
+ const asRecord = value && typeof value === 'object' && !Array.isArray(value)
101
+ ? value
102
+ : { value };
103
+ let size;
104
+ try {
105
+ size = JSON.stringify(asRecord)?.length ?? 0;
106
+ }
107
+ catch {
108
+ // Circular or otherwise unserializable. It could not be persisted anyway, and saying so
109
+ // is better than a pass that silently shows nothing.
110
+ return { __omitted: 'unserializable' };
111
+ }
112
+ if (size > ITERATION_PAYLOAD_VALUE_BYTES) {
113
+ return { __omitted: 'too-large', __bytes: size, __limit: ITERATION_PAYLOAD_VALUE_BYTES };
114
+ }
115
+ if (this.spent + size > ITERATION_PAYLOAD_BUDGET_BYTES) {
116
+ return { __omitted: 'budget-exhausted', __bytes: size, __limit: ITERATION_PAYLOAD_BUDGET_BYTES };
117
+ }
118
+ this.spent += size;
119
+ return asRecord;
120
+ }
121
+ }
122
+ /** Deep-merges a prompt's JSON response into the payload, preserving what earlier steps established. */
123
+ function deepMergePayload(base, incoming) {
124
+ const out = { ...base };
125
+ for (const [key, value] of Object.entries(incoming)) {
126
+ const existing = out[key];
127
+ const bothPlainObjects = existing && typeof existing === 'object' && !Array.isArray(existing) &&
128
+ value && typeof value === 'object' && !Array.isArray(value);
129
+ out[key] = bothPlainObjects
130
+ ? deepMergePayload(existing, value)
131
+ : value;
132
+ }
133
+ return out;
134
+ }
135
+ /**
136
+ * Statuses at which an origin's outgoing conditions may be decided.
137
+ *
138
+ * `Skipped` is included: a branch that was not taken IS settled, and a condition on an edge leaving
139
+ * it should resolve rather than hang the graph forever.
140
+ */
141
+ const TERMINAL_FOR_CONDITIONS = new Set([
142
+ 'Complete', 'Failed', 'Cancelled', 'Skipped',
143
+ ]);
144
+ export class TaskGraphDispatcher {
145
+ constructor(providerFactory, agentRunner, contextUser, config,
146
+ /**
147
+ * Optional. Absent means a host that cannot post messages or start agent turns — a worker,
148
+ * a test. The dispatcher still records and logs every completion, so a graph's outcome is
149
+ * never lost; it simply is not announced.
150
+ */
151
+ continuationDeliverer,
152
+ /**
153
+ * Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
154
+ * announces nothing.
155
+ */
156
+ observer,
157
+ /**
158
+ * Optional. Absent means this host cannot run action nodes; they stay Pending and visible
159
+ * rather than being failed, because "nobody here can run this" is not "this ran and broke".
160
+ */
161
+ actionRunner,
162
+ /**
163
+ * Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
164
+ * rather than being failed, for the same reason action nodes do.
165
+ */
166
+ promptRunner) {
167
+ this.providerFactory = providerFactory;
168
+ this.agentRunner = agentRunner;
169
+ this.contextUser = contextUser;
170
+ this.continuationDeliverer = continuationDeliverer;
171
+ this.observer = observer;
172
+ this.actionRunner = actionRunner;
173
+ this.promptRunner = promptRunner;
174
+ this.running = false;
175
+ this.pollTimer = null;
176
+ this.reconcileTimer = null;
177
+ /** Tasks this instance is currently executing — bounds concurrency and drives heartbeats. */
178
+ this.inFlight = new Set();
179
+ /** Guards against a slow poll overlapping the next tick. */
180
+ this.polling = false;
181
+ /**
182
+ * The poll pass currently running, so `Stop` can wait for it.
183
+ *
184
+ * `clearInterval` cannot cancel a tick that has already fired, and a pass is a long sequence of
185
+ * awaits (provider, rollup, claim query) — so without this, `Stop` returns while a pass is still
186
+ * mid-flight and about to claim. Its tasks then land in `inFlight` AFTER the drain loop already
187
+ * saw an empty set, which is precisely the state the drain exists to prevent.
188
+ */
189
+ this.pollPass = null;
190
+ /** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
191
+ this.ownerByParentID = new Map();
192
+ /** Name shown in the shutdown drain log. */
193
+ this.ShutdownName = 'TaskGraphDispatcher';
194
+ this.config = { ...DEFAULT_DISPATCHER_CONFIG, ...config };
195
+ this.claims = new TaskClaimStore(this.config.InstanceID, this.config.ClaimTTLSeconds);
196
+ this.conditionEvaluator = new DispatcherConditionEvaluator();
197
+ }
198
+ /**
199
+ * Announce something that happened, and never let the announcement matter.
200
+ *
201
+ * A frame is commentary on work, never a step of it, so an observer that throws must not be able
202
+ * to fail a task or stall a graph. Swallowing here rather than asking every implementation to be
203
+ * careful means one place enforces it.
204
+ */
205
+ emit(frame) {
206
+ if (!this.observer)
207
+ return;
208
+ try {
209
+ this.observer.OnFrame(frame);
210
+ }
211
+ catch (e) {
212
+ LogError(`[TaskGraphDispatcher] Observer threw on ${frame.Kind} (ignored): ${e instanceof Error ? e.message : String(e)}`);
213
+ }
214
+ }
215
+ /**
216
+ * Who a graph belongs to, memoized for the process's lifetime.
217
+ *
218
+ * Read from the parent's durable metadata rather than a column, because `Task.UserID` means
219
+ * "the person this task is waiting on" — setting it on a parent would make every graph look
220
+ * like a human task. Memoized because frames are emitted per step: without the cache, watching
221
+ * a run would cost one query per event, and observability that scales with work is the thing a
222
+ * push mechanism exists to avoid. Ownership never changes for a given graph, so the cache can
223
+ * never go stale.
224
+ *
225
+ * Skipped entirely when nobody is observing — the lookup exists only to address frames.
226
+ */
227
+ async resolveOwner(provider, parentTaskID) {
228
+ if (!this.observer)
229
+ return null;
230
+ const cached = this.ownerByParentID.get(parentTaskID);
231
+ if (cached !== undefined)
232
+ return cached;
233
+ let owner = null;
234
+ try {
235
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
236
+ if (await parent.Load(parentTaskID)) {
237
+ owner = this.readParentMetadata(parent).submittedByUserID ?? null;
238
+ }
239
+ }
240
+ catch (e) {
241
+ LogError(`[TaskGraphDispatcher] Could not resolve owner for graph ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
242
+ }
243
+ this.ownerByParentID.set(parentTaskID, owner);
244
+ return owner;
245
+ }
246
+ /**
247
+ * Begins dispatching.
248
+ *
249
+ * Runs reconciliation FIRST, before accepting any new work. On a restart this instance may be
250
+ * looking at tasks its own previous incarnation claimed and never released — reclaiming those
251
+ * up front is what turns a crash from "work stranded forever" into "work resumes".
252
+ */
253
+ async Start() {
254
+ if (this.running)
255
+ return;
256
+ this.running = true;
257
+ // Self-register rather than make each host remember to stop us. A dispatcher that keeps
258
+ // polling through a graceful shutdown would claim work the process is about to abandon,
259
+ // which is exactly the orphaned-claim state reconciliation exists to clean up.
260
+ ShutdownRegistry.Instance.Register(this);
261
+ LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
262
+ await this.Reconcile();
263
+ this.pollTimer = setInterval(() => { this.pollPass = this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
264
+ this.reconcileTimer = setInterval(() => { void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
265
+ }
266
+ /**
267
+ * Stops accepting new work and waits for in-flight tasks to finish.
268
+ *
269
+ * Deliberately does NOT release claims on the way out: an abandoned claim expires on its own,
270
+ * and releasing eagerly would hand a still-running task to another instance mid-execution.
271
+ * Letting the TTL do it is the safer failure mode.
272
+ */
273
+ async Stop() {
274
+ this.running = false;
275
+ if (this.pollTimer) {
276
+ clearInterval(this.pollTimer);
277
+ this.pollTimer = null;
278
+ }
279
+ if (this.reconcileTimer) {
280
+ clearInterval(this.reconcileTimer);
281
+ this.reconcileTimer = null;
282
+ }
283
+ // Drain the poll pass BEFORE the task drain below, not after: a pass still running has not
284
+ // necessarily claimed anything yet, so `inFlight` can be empty while work is moments from
285
+ // starting. Clearing `running` above stops that pass claiming anything further; this waits
286
+ // for it to notice. Its own failures are already logged inside `pollOnce`.
287
+ if (this.pollPass) {
288
+ await this.pollPass.catch(() => undefined);
289
+ this.pollPass = null;
290
+ }
291
+ const deadline = Date.now() + 30_000;
292
+ while (this.inFlight.size > 0 && Date.now() < deadline) {
293
+ await new Promise((r) => setTimeout(r, 250));
294
+ }
295
+ if (this.inFlight.size > 0) {
296
+ LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will expire.`);
297
+ }
298
+ LogStatus(`[TaskGraphDispatcher] Stopped.`);
299
+ }
300
+ /** {@link IShutdownable} — idempotent by way of `Stop`'s `running` guard. */
301
+ async Shutdown() {
302
+ await this.Stop();
303
+ }
304
+ /**
305
+ * Reclaims expired claims and reports anomalies.
306
+ *
307
+ * Also enforces the two schema promises that previously had no enforcer anywhere: agent tasks
308
+ * left `In Progress` with no claim are surfaced loudly rather than silently corrected, since
309
+ * that shape indicates tampering or a bug and Record Changes already carries the audit trail.
310
+ */
311
+ async Reconcile() {
312
+ let provider = null;
313
+ try {
314
+ provider = await this.providerFactory.CreateProvider();
315
+ const released = await this.claims.ReleaseExpiredClaims(provider, this.contextUser);
316
+ const orphaned = await this.claims.FindOrphanedInProgress(provider, this.contextUser);
317
+ if (released.length > 0 || orphaned.length > 0) {
318
+ LogStatus(`[TaskGraphDispatcher] Reconciliation: ${released.length} expired claim(s) released, ` +
319
+ `${orphaned.length} orphaned task(s) reported.`);
320
+ }
321
+ }
322
+ catch (e) {
323
+ LogError(`[TaskGraphDispatcher] Reconciliation failed: ${e instanceof Error ? e.message : String(e)}`);
324
+ }
325
+ }
326
+ /**
327
+ * One dispatch pass: find claimable work, claim what fits under the concurrency cap, execute.
328
+ *
329
+ * Overlap-guarded — a pass that runs long simply skips the next tick rather than stacking, which
330
+ * would otherwise let a slow database multiply in-flight work past the cap.
331
+ */
332
+ async pollOnce() {
333
+ if (!this.running || this.polling)
334
+ return;
335
+ const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
336
+ if (capacity <= 0)
337
+ return;
338
+ this.polling = true;
339
+ try {
340
+ const provider = await this.providerFactory.CreateProvider();
341
+ // `running` is re-read after every await from here on. The entry check above only proves
342
+ // the dispatcher was live when the tick fired; each await is a point where `Stop` can
343
+ // land, and a stopped instance must neither mutate graph state nor take new work. Left
344
+ // unchecked, a stopped dispatcher goes on to roll up graphs (emitting GraphSettled to an
345
+ // observer nobody is listening to any more) and to claim tasks it will never run — which
346
+ // then sit claimed until their lease expires.
347
+ if (!this.running)
348
+ return;
349
+ // Settle graphs before picking new work, so a failure earlier in this pass stops its
350
+ // branch immediately rather than after another wave has already launched.
351
+ await this.propagateAndRollup(provider);
352
+ if (!this.running)
353
+ return;
354
+ const candidates = await this.findClaimableTasks(provider, capacity);
355
+ for (const task of candidates) {
356
+ // Re-checked per iteration, not just before the loop: claiming is itself awaited, so
357
+ // a multi-task wave can straddle a Stop.
358
+ if (!this.running)
359
+ break;
360
+ if (this.inFlight.size >= this.config.MaxConcurrentTasks)
361
+ break;
362
+ if (!(await this.claims.TryClaim(provider, task.ID, this.contextUser))) {
363
+ // Another instance won the race, or the task is no longer Pending. Normal.
364
+ continue;
365
+ }
366
+ this.inFlight.add(task.ID);
367
+ // Intentionally not awaited — the poll loop must keep dispatching while this runs.
368
+ void this.executeClaimed(task.ID).finally(() => this.inFlight.delete(task.ID));
369
+ }
370
+ }
371
+ catch (e) {
372
+ LogError(`[TaskGraphDispatcher] Poll failed: ${e instanceof Error ? e.message : String(e)}`);
373
+ }
374
+ finally {
375
+ this.polling = false;
376
+ }
377
+ }
378
+ /**
379
+ * Executes one claimed task on its own provider, heartbeating until it settles.
380
+ *
381
+ * A fresh provider per task is the point of `ProviderFactory`: parallel tasks must not share a
382
+ * transaction scope or entity instances, or one task's work becomes visible inside another's.
383
+ */
384
+ async executeClaimed(taskID) {
385
+ let heartbeat = null;
386
+ try {
387
+ const provider = await this.providerFactory.CreateProvider();
388
+ const task = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
389
+ if (!(await task.Load(taskID))) {
390
+ LogError(`[TaskGraphDispatcher] Claimed task ${taskID} could not be loaded.`);
391
+ return;
392
+ }
393
+ heartbeat = setInterval(() => {
394
+ void this.claims.Heartbeat(provider, taskID, this.contextUser).then((ok) => {
395
+ if (!ok) {
396
+ // Lost ownership — reconciliation reclaimed it, or a human intervened.
397
+ LogError(`[TaskGraphDispatcher] Lost claim on task ${taskID} while executing; another instance may take it over.`);
398
+ }
399
+ });
400
+ }, this.config.HeartbeatIntervalSeconds * 1000);
401
+ // Emitted after the claim is held, not before: a frame saying "started" for work another
402
+ // instance actually took would be a lie a viewer cannot detect.
403
+ const graphID = task.ParentID ?? taskID;
404
+ const ownerUserID = await this.resolveOwner(provider, graphID);
405
+ this.emit({ Kind: 'TaskStarted', ParentTaskID: graphID, OwnerUserID: ownerUserID, TaskID: taskID, TaskName: task.Name, Status: 'In Progress' });
406
+ const dependencyOutputs = await this.loadDependencyOutputs(provider, taskID);
407
+ let inputPayload = null;
408
+ if (task.InputPayload) {
409
+ try {
410
+ inputPayload = JSON.parse(task.InputPayload);
411
+ }
412
+ catch (e) {
413
+ LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
414
+ }
415
+ }
416
+ const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs);
417
+ // A prompt can end the workflow early and say why. Honour it before recording the
418
+ // outcome, so the remaining tasks are already Skipped by the time the rollup runs and
419
+ // the graph settles Complete rather than looking abandoned with work left Pending.
420
+ if (result.ChatMessage) {
421
+ await this.endGraphEarly(provider, task, result.ChatMessage);
422
+ }
423
+ const recorded = await this.claims.CompleteClaimed(provider, taskID, {
424
+ Status: result.Success ? 'Complete' : 'Failed',
425
+ OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
426
+ ErrorMessage: result.ErrorMessage ?? null,
427
+ AgentRunID: result.AgentRunID ?? null,
428
+ Configuration: this.configurationWithRuntime(task, result.PromptRunID, result.ActionLogID, result.Iterations, result.PayloadAtStart),
429
+ }, this.contextUser);
430
+ if (!recorded) {
431
+ // The guarded write refused: the row changed underneath us (cancelled, reassigned,
432
+ // or reclaimed). Deferring to whoever owns it now is correct — overwriting would
433
+ // undo a newer, deliberate decision.
434
+ LogError(`[TaskGraphDispatcher] Could not record outcome for ${taskID}; the task is no longer owned by this instance.`);
435
+ }
436
+ else {
437
+ // Only announced when the guarded write actually landed. Announcing an outcome we
438
+ // failed to persist would show a viewer a completion the database never recorded.
439
+ this.emit({
440
+ Kind: result.Success ? 'TaskCompleted' : 'TaskFailed',
441
+ ParentTaskID: graphID,
442
+ OwnerUserID: ownerUserID,
443
+ TaskID: taskID,
444
+ TaskName: task.Name,
445
+ Status: result.Success ? 'Complete' : 'Failed',
446
+ ErrorMessage: result.Success ? undefined : (result.ErrorMessage ?? undefined),
447
+ });
448
+ }
449
+ }
450
+ catch (e) {
451
+ LogError(`[TaskGraphDispatcher] Execution failed for ${taskID}: ${e instanceof Error ? e.message : String(e)}`);
452
+ try {
453
+ const provider = await this.providerFactory.CreateProvider();
454
+ await this.claims.CompleteClaimed(provider, taskID, { Status: 'Failed', ErrorMessage: e instanceof Error ? e.message : String(e) }, this.contextUser);
455
+ }
456
+ catch { /* already logged; nothing further to do */ }
457
+ }
458
+ finally {
459
+ if (heartbeat)
460
+ clearInterval(heartbeat);
461
+ }
462
+ }
463
+ /**
464
+ * Applies failure propagation and parent rollup across every graph with active work.
465
+ *
466
+ * All four decisions — what is eligible, what must block, what the parent status is, whether the
467
+ * graph is wedged — are delegated to the pure algorithms, unchanged from Phase 1.
468
+ */
469
+ async propagateAndRollup(provider) {
470
+ for (const parentID of await this.findActiveGraphIDs(provider)) {
471
+ // Human steps settle BEFORE the graph state is read, so an answer given since the last
472
+ // poll is already reflected when eligibility and rollup are computed. Doing it after
473
+ // would delay every dependent branch by a full poll interval for no reason — and on a
474
+ // graph whose only remaining work is downstream of a person, that is the difference
475
+ // between "answered and moving" and "answered and apparently still stuck".
476
+ await this.expireOverdueRequests(provider, parentID);
477
+ await this.settleAnsweredHumanTasks(provider, parentID);
478
+ await this.reopenCancelledHumanTasks(provider, parentID);
479
+ const graph = await this.loadGraphState(provider, parentID);
480
+ if (graph.nodes.length === 0)
481
+ continue;
482
+ // SKIPS FIRST — before blocking, before eligibility. A task whose gating predecessors
483
+ // are all Skipped is simultaneously "eligible" (Skipped satisfies a prerequisite) and
484
+ // "to be skipped"; deciding eligibility first would dispatch the branch nobody took.
485
+ //
486
+ // `unreachableTaskIDs` seeds this too, and that is a correction (R6). A target whose only
487
+ // route in was an ordinary conditional edge that evaluated DEFINITELY FALSE is a branch
488
+ // that was not taken — semantically identical to an XOR loser — yet it used to settle
489
+ // `Blocked`. That made `Blocked` mean two unrelated things: "the workflow chose another
490
+ // route" and "something upstream broke". A reader cannot tell those apart, so every
491
+ // conditional workflow looked half-failed and people went hunting for bugs that did not
492
+ // exist. `Blocked` is now reserved for FAILURE-driven unsatisfiability.
493
+ const skipSeeds = new Set([...graph.skipSeedTaskIDs, ...graph.unreachableTaskIDs]);
494
+ const toSkip = new Set([
495
+ ...skipSeeds,
496
+ ...ComputeSkipCascade(graph.nodes, graph.edges, [...skipSeeds]),
497
+ ]);
498
+ for (const taskID of toSkip) {
499
+ const entity = graph.entityById.get(taskID);
500
+ if (!entity || entity.Status !== 'Pending')
501
+ continue;
502
+ entity.Status = 'Skipped';
503
+ if (await entity.Save()) {
504
+ LogStatus(`[TaskGraphDispatcher] Skipped '${entity.Name}' (${taskID}) — another branch was taken.`);
505
+ // Announced separately from TaskBlocked because it means something different to
506
+ // a viewer: nothing went wrong, this route simply was not the one chosen.
507
+ this.emit({
508
+ Kind: 'TaskSkipped',
509
+ ParentTaskID: parentID,
510
+ OwnerUserID: await this.resolveOwner(provider, parentID),
511
+ TaskID: taskID,
512
+ TaskName: entity.Name,
513
+ Status: 'Skipped',
514
+ });
515
+ // Keep the in-memory graph consistent so the blocking pass below and the rollup
516
+ // both see the skip rather than a stale Pending.
517
+ const node = graph.nodes.find((n) => n.id === taskID);
518
+ if (node)
519
+ node.status = 'Skipped';
520
+ }
521
+ }
522
+ // Only failure-driven unsatisfiability reaches here now; not-taken branches were skipped
523
+ // above. A task already Skipped is left alone rather than overwritten — the two passes
524
+ // must not fight over the same row.
525
+ const toBlock = [...ComputeTasksToBlock(graph.nodes, graph.edges, graph.handledFailureIDs)]
526
+ .filter((id) => !toSkip.has(id));
527
+ for (const taskID of toBlock) {
528
+ const entity = graph.entityById.get(taskID);
529
+ if (!entity)
530
+ continue;
531
+ entity.Status = 'Blocked';
532
+ if (await entity.Save()) {
533
+ LogStatus(`[TaskGraphDispatcher] Blocked '${entity.Name}' (${taskID}) — a dependency can never be satisfied.`);
534
+ // Worth announcing on its own: a blocked step is the one outcome a viewer would
535
+ // otherwise see as a task that simply never starts.
536
+ this.emit({
537
+ Kind: 'TaskBlocked',
538
+ ParentTaskID: parentID,
539
+ OwnerUserID: await this.resolveOwner(provider, parentID),
540
+ TaskID: taskID,
541
+ TaskName: entity.Name,
542
+ Status: 'Blocked',
543
+ });
544
+ }
545
+ }
546
+ if (IsGraphStalled(graph.nodes, graph.edges)) {
547
+ LogError(`[TaskGraphDispatcher] Graph ${parentID} is stalled: pending work with no satisfiable path.`);
548
+ }
549
+ const fresh = await this.loadGraphState(provider, parentID);
550
+ // ComputeParentRollup treats an empty child set as Complete-and-terminal, which is right
551
+ // for a graph that genuinely has no children and catastrophic for one whose reload came
552
+ // back empty transiently — it would mark live work finished and fire its continuation.
553
+ // The outer guard covered the first load only.
554
+ if (fresh.nodes.length === 0)
555
+ continue;
556
+ const rollup = ComputeParentRollup(fresh.nodes, fresh.handledFailureIDs);
557
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
558
+ if (!(await parent.Load(parentID)))
559
+ continue;
560
+ // A graph starts when its first step does.
561
+ //
562
+ // `StartedAt` is stamped by the CLAIM, and a parent is never claimed — it is a container,
563
+ // not a unit of work — so the graph row carried no start time even after it completed.
564
+ // A settled workflow therefore reported a CompletedAt with no beginning: it sorted as
565
+ // "not started" in the run tree, showed no timestamp, and no duration could be computed
566
+ // for the thing whose duration people actually ask about.
567
+ //
568
+ // Taken from the earliest child rather than from the clock, because that is when work
569
+ // genuinely began — a graph can sit Pending for a long time between submission (already
570
+ // recorded as CreatedAt) and a dispatcher picking up its first task.
571
+ const earliestChildStart = this.earliestStart(fresh.entityById);
572
+ const startedAtChanged = parent.StartedAt == null && earliestChildStart != null;
573
+ if (startedAtChanged)
574
+ parent.StartedAt = earliestChildStart;
575
+ if (startedAtChanged || parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
576
+ parent.Status = rollup.status;
577
+ parent.PercentComplete = rollup.percentComplete;
578
+ if (rollup.isTerminal)
579
+ parent.CompletedAt = new Date();
580
+ await parent.Save();
581
+ }
582
+ if (rollup.isTerminal) {
583
+ // Geometry is settled once, here, so every viewer of this run agrees on it.
584
+ await this.persistComputedLayout(fresh);
585
+ // Emitted before the continuation is delivered, and outside its once-only guard: a
586
+ // viewer watching the run should learn it finished whether or not this instance is
587
+ // the one that wins the delivery CAS.
588
+ this.emit({
589
+ Kind: 'GraphSettled',
590
+ ParentTaskID: parentID,
591
+ OwnerUserID: await this.resolveOwner(provider, parentID),
592
+ Status: rollup.status,
593
+ CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
594
+ TotalCount: fresh.nodes.length,
595
+ });
596
+ await this.rollUpCostToSubmittingRun(provider, parent);
597
+ // Deliberately AFTER the rollup and OUTSIDE its refusal paths. The rollup declines
598
+ // to write a number it cannot stand behind — a truncated tree, an unreachable graph
599
+ // — and every one of those returns early. If the run's lifecycle were settled in
600
+ // there, a refused rollup would strand the run parked forever, which is a far worse
601
+ // failure than a missing cost figure. Cost and lifecycle are separate concerns with
602
+ // separate failure modes, so they get separate writes.
603
+ await this.settleSubmittingRun(provider, parent, rollup.status);
604
+ await this.deliverContinuation(provider, parent, fresh);
605
+ }
606
+ }
607
+ }
608
+ /**
609
+ * Credits a finished graph's spending back to the agent run that submitted it.
610
+ *
611
+ * **Why this cannot happen during the run.** `BaseAgent` totals a run by walking its steps in
612
+ * memory at finalization — but a submitting run *ends at submission*. Submit-and-detach is the
613
+ * point: the run returns as soon as the graph is durable, and the graph executes afterwards,
614
+ * possibly minutes later on a different instance. At the moment the run computes its totals the
615
+ * spending has not happened yet, so there is nothing to count. The only place the number can be
616
+ * known is here, when the graph settles.
617
+ *
618
+ * **Why the `…Rollup` columns and not the plain ones.** `AIAgentRun` has carried six `…Rollup`
619
+ * columns since v3 that nothing has ever written — they exist for exactly this distinction:
620
+ *
621
+ * - `TotalCost` — what the run itself spent. For a Flow agent that is genuinely near zero: it
622
+ * compiled a graph and handed it off. This value is already final and is never rewritten here,
623
+ * so nothing that reads it today changes meaning, and no guardrail that already evaluated
624
+ * against it is retroactively falsified.
625
+ * - `TotalCostRollup` — the run plus everything it caused. Provisional until the graph settles,
626
+ * which is now.
627
+ *
628
+ * **The tree is the authority; these columns are its settlement-time cache.** The total is a SUM
629
+ * over `GetAgentRunTree`, not arithmetic of its own. The previous version walked the graph's
630
+ * child tasks and added each one's agent run, which was wrong in two ways that no test could
631
+ * see: a `Prompt` task has no agent run at all, so every prompt step's spend was simply missing;
632
+ * and it read each nested run's `…Rollup ?? …Total`, mixing a descendant-inclusive number with an
633
+ * own-spend one and depending on whether that nested graph happened to have settled yet. The
634
+ * tree already models every one of those cases — it reaches prompt runs through
635
+ * `Configuration.runtime.promptRunID`, and it descends into nested runs and their graphs
636
+ * structurally — so summing it cannot disagree with what the run viewer shows, because it IS
637
+ * what the run viewer shows.
638
+ *
639
+ * **This refuses rather than guesses.** A tree that failed to load, hit the depth cap, or does
640
+ * not contain the settling graph would still produce a number — a lower bound. Writing one would
641
+ * put an authoritative-looking total in a column every cost surface reads. Each of those cases
642
+ * logs and leaves the column alone, so `?? TotalCost` keeps its honest meaning: not settled.
643
+ *
644
+ * A graph with no submitting run (a scheduled job, a remote-operation caller) simply has nobody
645
+ * to credit — its own Task rows still carry the truth, and this returns quietly.
646
+ */
647
+ async rollUpCostToSubmittingRun(provider, parent) {
648
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
649
+ if (!meta.submittedByAgentRunID)
650
+ return;
651
+ const runID = meta.submittedByAgentRunID;
652
+ try {
653
+ const runQuery = asRunQueryProvider(provider);
654
+ if (!runQuery) {
655
+ LogError(`[TaskGraphDispatcher] Cannot roll up cost for run ${runID}: provider cannot run queries.`);
656
+ return;
657
+ }
658
+ const tree = await LoadAgentRunTree(runID, runQuery, this.contextUser);
659
+ // Each of these means the sum would be a LOWER BOUND, and the column's whole contract is
660
+ // that it equals the tree. A known-low number presented as a total is worse than no
661
+ // number: the readers all fall back to TotalCost when this is null, which at least
662
+ // *says* it is the run's own spend rather than claiming to be the whole story.
663
+ //
664
+ // Refusing is NOT the same as leaving the column alone. A run that submitted two graphs
665
+ // has a rollup from the first; if the second cannot be summed, the first graph's total
666
+ // sits in the authoritative column excluding work that has since happened — stale, not
667
+ // absent, and `?? TotalCost` cannot save a reader from a non-null wrong number. So a
668
+ // refusal CLEARS it, restoring the fallback's honest meaning: not settled.
669
+ if (tree.ErrorMessage || !tree.Root) {
670
+ await this.clearStaleRollup(provider, runID, tree.ErrorMessage ?? 'the run tree came back empty');
671
+ return;
672
+ }
673
+ if (tree.Truncated) {
674
+ await this.clearStaleRollup(provider, runID, `the run tree hit the depth cap, so any total would silently under-report ` +
675
+ `(graph ${parent.ID} still carries its own costs)`);
676
+ return;
677
+ }
678
+ // The graph that just settled must appear in the tree. If it does not, the tree stopped
679
+ // at the run — the submitting step never recorded its parentTaskID — and the sum is
680
+ // merely the run's own spend wearing the name of a rollup. That is precisely the silent
681
+ // under-count this rewrite exists to remove, so it is reported rather than written.
682
+ if (!this.treeContainsGraph(tree.Root, parent.ID)) {
683
+ await this.clearStaleRollup(provider, runID, `graph ${parent.ID} is not reachable from it, so the tree cannot see the work. ` +
684
+ `Did the submitting step record parentTaskID?`);
685
+ return;
686
+ }
687
+ const totals = SumAgentRunTreeCost(tree.Root);
688
+ const submitting = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
689
+ if (!(await submitting.Load(runID))) {
690
+ LogError(`[TaskGraphDispatcher] Could not load run ${runID} to record graph cost against it.`);
691
+ return;
692
+ }
693
+ // Assignment, never accumulation. The tree already contains the run's own spend as its
694
+ // ROOT node, and it reads own-cost everywhere, so recomputing from scratch on every
695
+ // settlement lands on the same answer — which is what makes this safe to call again when
696
+ // a second graph settles, or when the terminal check is re-evaluated after a HITL wait.
697
+ submitting.TotalCostRollup = totals.Cost;
698
+ submitting.TotalTokensUsedRollup = totals.Tokens;
699
+ submitting.TotalPromptTokensUsedRollup = totals.PromptTokens;
700
+ submitting.TotalCompletionTokensUsedRollup = totals.CompletionTokens;
701
+ if (!(await submitting.Save())) {
702
+ LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}: ` +
703
+ `${submitting.LatestResult?.CompleteMessage ?? 'unknown error'}`);
704
+ return;
705
+ }
706
+ LogStatus(`[TaskGraphDispatcher] Credited graph ${parent.ID} to run ${runID}: ` +
707
+ `${tree.Rows.length} node(s), ${totals.Tokens} token(s), cost ${totals.Cost}.`);
708
+ }
709
+ catch (e) {
710
+ // A failed rollup must never fail the graph. The work finished; only the accounting for
711
+ // it is missing, and a graph marked Failed because its cost could not be summed would be
712
+ // a far worse lie than a cost of null.
713
+ LogError(`[TaskGraphDispatcher] Cost rollup failed for graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
714
+ }
715
+ }
716
+ /**
717
+ * Clears a rollup that can no longer be trusted, and says why.
718
+ *
719
+ * **Why clear rather than leave.** The four `…Rollup` columns are a cache of the run tree, and
720
+ * every reader treats a value there as the total. When the tree cannot be summed, any value
721
+ * already in the column was computed from an EARLIER settlement — it excludes the graph that
722
+ * just finished, so it is not merely incomplete, it is a wrong total presented as a right one.
723
+ * `?? TotalCost` protects a reader from null, not from stale.
724
+ *
725
+ * Nulling restores the invariant this whole design rests on: **when the column is present, it
726
+ * equals the tree.** Absent means not settled, which is exactly what a reader should conclude.
727
+ * A run with no rollup yet is untouched — there is nothing stale to clear, and writing nulls
728
+ * over nulls would churn Record Changes for nothing.
729
+ */
730
+ async clearStaleRollup(provider, runID, reason) {
731
+ LogError(`[TaskGraphDispatcher] Not recording cost for run ${runID}: ${reason}.`);
732
+ try {
733
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
734
+ if (!(await run.Load(runID)))
735
+ return;
736
+ if (run.TotalCostRollup == null && run.TotalTokensUsedRollup == null)
737
+ return; // nothing stale
738
+ run.TotalCostRollup = null;
739
+ run.TotalTokensUsedRollup = null;
740
+ run.TotalPromptTokensUsedRollup = null;
741
+ run.TotalCompletionTokensUsedRollup = null;
742
+ if (!(await run.Save())) {
743
+ LogError(`[TaskGraphDispatcher] Could not clear the now-stale rollup on run ${runID}: ` +
744
+ `${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It still shows a total that ` +
745
+ `excludes the graph that just settled.`);
746
+ return;
747
+ }
748
+ LogStatus(`[TaskGraphDispatcher] Cleared the rollup on run ${runID}: it was computed before this ` +
749
+ `graph settled and can no longer be recomputed, so it would have under-reported.`);
750
+ }
751
+ catch (e) {
752
+ LogError(`[TaskGraphDispatcher] Could not clear the rollup on run ${runID}: ${e instanceof Error ? e.message : String(e)}`);
753
+ }
754
+ }
755
+ /**
756
+ * Whether the settling graph is actually reachable from the submitting run's tree.
757
+ *
758
+ * Matched on the graph's parent Task id, which is the node the `TaskGraph` member of the query
759
+ * emits. A run that submitted a graph but recorded no `parentTaskID` produces a tree that stops
760
+ * at the run — structurally indistinguishable, at the SUM, from a run that never dispatched
761
+ * anything. This is the check that tells those two apart.
762
+ */
763
+ treeContainsGraph(root, parentTaskID) {
764
+ for (const node of WalkAgentRunTree(root)) {
765
+ if (node.NodeType === 'TaskGraph' && UUIDsEqual(node.NodeID, parentTaskID))
766
+ return true;
767
+ }
768
+ return false;
769
+ }
770
+ /**
771
+ * Ends a graph early because a prompt said the work is finished.
772
+ *
773
+ * **Why `Skipped` and not `Cancelled`.** Nothing went wrong and nobody intervened — the workflow
774
+ * reached its own conclusion before running every drawn step, which is exactly what a reasoning
775
+ * step is for. `Cancelled` would tell a reader someone stopped it; `Skipped` says these routes
776
+ * were not taken, which is true and already the vocabulary the fork machinery uses.
777
+ *
778
+ * The message is written to the parent so the graph carries its own answer, rather than the
779
+ * answer living only on the step that produced it.
780
+ */
781
+ async endGraphEarly(provider, task, message) {
782
+ if (!task.ParentID)
783
+ return;
784
+ try {
785
+ LogStatus(`[TaskGraphDispatcher] '${task.Name}' ended the workflow early: ${message}`);
786
+ for (const sibling of await this.loadChildTasks(provider, task.ParentID)) {
787
+ if (sibling.ID === task.ID || sibling.Status !== 'Pending')
788
+ continue;
789
+ sibling.Status = 'Skipped';
790
+ if (await sibling.Save()) {
791
+ this.emit({
792
+ Kind: 'TaskSkipped',
793
+ ParentTaskID: task.ParentID,
794
+ OwnerUserID: await this.resolveOwner(provider, task.ParentID),
795
+ TaskID: sibling.ID,
796
+ TaskName: sibling.Name,
797
+ Status: 'Skipped',
798
+ });
799
+ }
800
+ }
801
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
802
+ if (await parent.Load(task.ParentID)) {
803
+ parent.OutputPayload = JSON.stringify({ message });
804
+ await parent.Save();
805
+ }
806
+ }
807
+ catch (e) {
808
+ // The work itself succeeded; only the early-finish bookkeeping failed. Failing the task
809
+ // over that would discard a completed step's result.
810
+ LogError(`[TaskGraphDispatcher] Could not end graph early for ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
811
+ }
812
+ }
813
+ /**
814
+ * How deep the continuation chain already is, read from the graph's parent metadata.
815
+ *
816
+ * A run started by a graph inherits that graph's depth **plus one**. Without this every spawned
817
+ * run begins at zero, so a self-referencing flow — one that dispatches a graph containing itself
818
+ * — recurses without bound while the cap it should be hitting compares against a permanent zero.
819
+ */
820
+ async graphContext(provider, task) {
821
+ if (!task.ParentID)
822
+ return { Depth: 0, SubmittingAgentRunID: null };
823
+ try {
824
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
825
+ if (!(await parent.Load(task.ParentID)))
826
+ return { Depth: 0, SubmittingAgentRunID: null };
827
+ return {
828
+ Depth: ParseTaskGraphParentMetadata(parent.InputPayload).reinvokeDepth + 1,
829
+ // The graph's own row carries the run that submitted it. One load answers both
830
+ // questions, which is why they are resolved together rather than in two passes.
831
+ SubmittingAgentRunID: parent.AgentRunID,
832
+ };
833
+ }
834
+ catch {
835
+ // An unreadable parent must not stop the work; depth zero is the safe reading, and the
836
+ // submit-time cap still guards the next hop.
837
+ return { Depth: 0, SubmittingAgentRunID: null };
838
+ }
839
+ }
840
+ /**
841
+ * Which failures the workflow drew a way out of.
842
+ *
843
+ * A Failed task with a **satisfied outgoing edge** is a handled failure: its author drew a
844
+ * recovery route and that route is now live. Downstream work should be released along it, and the
845
+ * parent should not roll up Failed because of a step the workflow explicitly planned around.
846
+ *
847
+ * Scoped to `failureSemantics: 'edges'` on purpose. Under `'block'` — every agent-emitted graph —
848
+ * a failure is terminal for its dependents whatever edges exist, because nobody drew those edges
849
+ * as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
850
+ * graph sail past a failure it never anticipated.
851
+ */
852
+ async computeHandledFailures(provider, parentTaskID, nodes, edges) {
853
+ const handled = new Set();
854
+ // Cheap exit before touching the database: with no failures there is nothing to handle, and
855
+ // this runs on every poll for every active graph.
856
+ if (!nodes.some((n) => n.status === 'Failed'))
857
+ return handled;
858
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
859
+ if (!(await parent.Load(parentTaskID)))
860
+ return handled;
861
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
862
+ if (meta.failureSemantics !== 'edges')
863
+ return handled;
864
+ for (const node of nodes) {
865
+ if (node.status !== 'Failed')
866
+ continue;
867
+ // "Has somewhere to go" is the test. An edge out of a failed step that survived condition
868
+ // evaluation IS the drawn recovery route; a failed step with no outgoing edges has none,
869
+ // and stays terminal.
870
+ if (edges.some((e) => e.dependsOnTaskId === node.id))
871
+ handled.add(node.id);
872
+ }
873
+ return handled;
874
+ }
875
+ /** The graph's child tasks, with the fields the rollup needs. */
876
+ async loadChildTasks(provider, parentID) {
877
+ const result = await RunView.FromMetadataProvider(provider).RunView({
878
+ EntityName: 'MJ: Tasks',
879
+ ExtraFilter: `ParentID='${parentID}'`,
880
+ ResultType: 'entity_object',
881
+ BypassCache: true,
882
+ }, this.contextUser);
883
+ return (result.Success ? result.Results : []) ?? [];
884
+ }
885
+ /**
886
+ * Runs the graph's continuation exactly once, now that it has settled.
887
+ *
888
+ * **Why the delivery marker is written before the side effect.** Delivery is at-least-once by
889
+ * nature: the process can die between "the graph is done" and "the user has been told". Marking
890
+ * first and acting second means the worst case is a *missed* notification that shows up in the
891
+ * task record as delivered — recoverable, visible, and inspectable. Marking after would make the
892
+ * worst case a *repeated* notification on every reconciliation sweep, forever, which is both
893
+ * user-visible noise and, for `reinvoke`, an unbounded agent-run loop. Given one of the two has
894
+ * to be chosen, the quiet failure is the safe one.
895
+ *
896
+ * The marker is written with a compare-and-swap read-back, so two instances reconciling the same
897
+ * completed graph produce one winner rather than two.
898
+ */
899
+ async deliverContinuation(provider, parent, graph) {
900
+ const meta = this.readParentMetadata(parent);
901
+ if (meta.continuationDeliveredAt)
902
+ return;
903
+ // At the cap, downgrade rather than refuse: the results still reach the user, the chain just
904
+ // stops growing. Refusing outright would lose the outcome of work that actually completed.
905
+ const mode = IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
906
+ if (mode !== 'none' && IsReinvokeCapReached(meta) && meta.continuation === 'reinvoke') {
907
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} hit the reinvoke cap (${MAX_REINVOKE_DEPTH}); ` +
908
+ `delivering results as a message instead of starting another turn.`);
909
+ }
910
+ if (!(await this.claimContinuation(provider, parent.ID, meta)))
911
+ return;
912
+ if (mode === 'none')
913
+ return;
914
+ const summary = this.buildContinuationSummary(parent, graph);
915
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} finished — ${summary}`);
916
+ if (!this.continuationDeliverer)
917
+ return;
918
+ const params = {
919
+ ParentTaskID: parent.ID,
920
+ WorkflowName: parent.Name,
921
+ ConversationDetailID: parent.ConversationDetailID ?? null,
922
+ SubmittedByAgentRunID: meta.submittedByAgentRunID,
923
+ ReinvokeDepth: meta.reinvokeDepth,
924
+ Tasks: [...graph.entityById.values()].map((t) => ({
925
+ TaskID: t.ID,
926
+ Name: t.Name,
927
+ Status: t.Status,
928
+ // A reference, not the payload. Inlining every task's output would swamp the
929
+ // continuation turn's context; the agent pulls what it needs by task ID.
930
+ Summary: t.OutputPayload ? `output available (${t.OutputPayload.length} chars)` : undefined,
931
+ ErrorMessage: t.ErrorMessage ?? undefined,
932
+ })),
933
+ Summary: summary,
934
+ };
935
+ try {
936
+ // Reinvoke degrades to a message when the host cannot start agent turns. Degrading is
937
+ // right rather than throwing: the work genuinely ran, and the user losing the results
938
+ // because nobody could start a follow-up turn would be the worse outcome.
939
+ if (mode === 'reinvoke' && this.continuationDeliverer.Reinvoke) {
940
+ await this.continuationDeliverer.Reinvoke(params);
941
+ }
942
+ else {
943
+ if (mode === 'reinvoke') {
944
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID}: host cannot reinvoke; delivering as a message.`);
945
+ }
946
+ await this.continuationDeliverer.PostMessage(params);
947
+ }
948
+ }
949
+ catch (e) {
950
+ // Already marked delivered, so this will not retry. That is the deliberate trade stated
951
+ // on the marker: a missed notification visible in the record beats one repeated forever.
952
+ LogError(`[TaskGraphDispatcher] Continuation delivery failed for ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
953
+ }
954
+ }
955
+ /** Reads the parent's durable continuation metadata through the shared parser. */
956
+ readParentMetadata(parent) {
957
+ return ParseTaskGraphParentMetadata(parent.InputPayload);
958
+ }
959
+ /**
960
+ * Stamps the delivery marker and confirms this instance won the race.
961
+ *
962
+ * `MJ: Tasks` stays user-writable (D20), so a plain "read, decide, write" is not enough — the
963
+ * read-back is what makes a lost race observable instead of producing a duplicate delivery.
964
+ */
965
+ async claimContinuation(provider, parentID, meta) {
966
+ const row = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
967
+ if (!(await row.Load(parentID)))
968
+ return false;
969
+ const current = this.readParentMetadata(row);
970
+ if (current.continuationDeliveredAt)
971
+ return false; // a peer got there first
972
+ row.InputPayload = JSON.stringify({ ...meta, continuationDeliveredAt: new Date().toISOString() });
973
+ if (!(await row.Save())) {
974
+ LogError(`[TaskGraphDispatcher] Could not mark continuation delivered for ${parentID}; skipping to avoid a duplicate.`);
975
+ return false;
976
+ }
977
+ return true;
978
+ }
979
+ /** One line describing how the graph ended, for the completion log and message delivery. */
980
+ buildContinuationSummary(parent, graph) {
981
+ const counts = new Map();
982
+ for (const node of graph.nodes)
983
+ counts.set(node.status, (counts.get(node.status) ?? 0) + 1);
984
+ const breakdown = [...counts.entries()].map(([status, n]) => `${n} ${status}`).join(', ');
985
+ return `"${parent.Name}": ${graph.nodes.length} task(s) — ${breakdown}.`;
986
+ }
987
+ /**
988
+ * Tells the assignee that a human task is ready, exactly once.
989
+ *
990
+ * **Once** matters more than it looks: eligibility is recomputed on every poll, so a task parked
991
+ * on a person for three days would otherwise re-notify every five seconds until they acted. The
992
+ * marker is the task's own `ClaimedBy` — a human task has no executor to claim it, so the column
993
+ * is free, and reusing it means the "already notified" fact is as durable and as crash-safe as
994
+ * every other piece of graph state. A restart cannot resend.
995
+ *
996
+ * Best-effort by design. A notification that fails to send must not stop the graph or the poll
997
+ * loop; the task is still visible in the Tasks UI, so the work is discoverable even when the
998
+ * nudge does not arrive.
999
+ */
1000
+ async notifyHumanTaskReady(task, provider) {
1001
+ if (task.ClaimedBy === HUMAN_TASK_NOTIFIED_MARKER)
1002
+ return;
1003
+ // The REQUEST is raised whether or not the task names an assignee. An unassigned human step
1004
+ // is a legitimate "somebody needs to look at this", and a request nobody was notified about
1005
+ // is still findable in the inbox — whereas returning early here is how such a step used to
1006
+ // become invisible work that stalled a workflow with nothing anywhere saying why.
1007
+ // TRANSIENT failures retry; PERMANENT ones stop. That distinction is the whole point, and
1008
+ // getting it wrong took a server down: retrying unconditionally meant a task whose workflow
1009
+ // has no owning agent — which can never succeed — was re-attempted on every poll forever,
1010
+ // each pass re-reading the graph, until the process was OOM-killed. The marker exists to
1011
+ // prevent exactly that storm; a permanent failure has to set it.
1012
+ const raised = await this.raiseHumanRequest(task, provider);
1013
+ if (raised === 'transient-failure')
1014
+ return; // try again next poll
1015
+ if (raised === 'permanent-failure') {
1016
+ // Nothing will change on a retry. Mark it so the loop stops, and leave the task Pending
1017
+ // and visible — a person can still see it in the Tasks UI, which is the fallback the
1018
+ // notification was only ever an accelerant for.
1019
+ await this.markHumanTaskNotified(task);
1020
+ return;
1021
+ }
1022
+ if (!task.UserID) {
1023
+ await this.markHumanTaskNotified(task);
1024
+ return;
1025
+ }
1026
+ try {
1027
+ await NotificationEngine.Instance.Config(false, this.contextUser);
1028
+ await NotificationEngine.Instance.SendNotification({
1029
+ userId: task.UserID,
1030
+ typeNameOrId: HUMAN_TASK_NOTIFICATION_TYPE,
1031
+ title: `Action needed: ${task.Name}`,
1032
+ message: task.Description || 'A workflow is waiting on you to complete this task.',
1033
+ resourceConfiguration: { type: 'Task', taskId: task.ID, parentTaskId: task.ParentID ?? '' },
1034
+ }, this.contextUser);
1035
+ }
1036
+ catch (e) {
1037
+ LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
1038
+ }
1039
+ await this.markHumanTaskNotified(task);
1040
+ // Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
1041
+ // than appearing to stall for no reason.
1042
+ this.emit({
1043
+ Kind: 'TaskAwaitingHuman',
1044
+ ParentTaskID: task.ParentID ?? task.ID,
1045
+ OwnerUserID: await this.resolveOwner(provider, task.ParentID ?? task.ID),
1046
+ TaskID: task.ID,
1047
+ TaskName: task.Name,
1048
+ Status: task.Status,
1049
+ AssignedUserID: task.UserID,
1050
+ });
1051
+ }
1052
+ /**
1053
+ * Parent tasks that still have work to do.
1054
+ *
1055
+ * `BypassCache` for the reason the caching guide names explicitly: **the claim protocol mutates
1056
+ * these rows through direct SQL**, because the CAS guarantee IS the database's atomicity and a
1057
+ * `BaseEntity.Save()` cannot express a guarded UPDATE. Direct DML fires no invalidation event,
1058
+ * so a cached read of this query is stale the instant any task is claimed or completed — and
1059
+ * the dispatcher would then be reading its own work queue through a cache its own writes never
1060
+ * invalidate. Left cached, a completed task keeps reading as `In Progress` and the graph never
1061
+ * rolls up: submitted work simply never settles.
1062
+ */
1063
+ async findActiveGraphIDs(provider) {
1064
+ const rv = RunView.FromMetadataProvider(provider);
1065
+ // TWO queries, because "has work left to do" and "needs attention" are not the same set.
1066
+ //
1067
+ // Selecting only graphs with non-terminal CHILDREN looks right and is subtly fatal: the
1068
+ // moment the last child completes, the graph leaves that set — so the pass that would have
1069
+ // rolled the parent up never sees it. A graph whose tasks all succeed therefore stays
1070
+ // In Progress forever and its continuation never fires. (A graph that FAILS happened to
1071
+ // survive this, because blocking its dependents left them non-terminal for one more pass —
1072
+ // which is why the bug hid behind a passing failure-path test.)
1073
+ //
1074
+ // The second query closes it: a parent that is itself non-terminal still needs looking at,
1075
+ // whatever its children are doing.
1076
+ const [withPendingWork, unsettledParents] = await rv.RunViews([
1077
+ {
1078
+ EntityName: 'MJ: Tasks',
1079
+ ExtraFilter: `ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
1080
+ Fields: ['ParentID'],
1081
+ ResultType: 'simple',
1082
+ BypassCache: true,
1083
+ },
1084
+ {
1085
+ EntityName: 'MJ: Tasks',
1086
+ ExtraFilter: `ParentID IS NULL AND Status IN ('Pending','In Progress')`,
1087
+ Fields: ['ID'],
1088
+ ResultType: 'simple',
1089
+ BypassCache: true,
1090
+ },
1091
+ ], this.contextUser);
1092
+ const ids = new Set();
1093
+ for (const r of (withPendingWork?.Results ?? [])) {
1094
+ if (r.ParentID)
1095
+ ids.add(r.ParentID);
1096
+ }
1097
+ // Childless tasks match the second query too; propagateAndRollup skips anything with no
1098
+ // nodes, so they cost one empty load and nothing else.
1099
+ for (const r of (unsettledParents?.Results ?? [])) {
1100
+ if (r.ID)
1101
+ ids.add(r.ID);
1102
+ }
1103
+ return [...ids];
1104
+ }
1105
+ /**
1106
+ * Tasks eligible to claim right now, across all active graphs.
1107
+ *
1108
+ * Eligibility is decided by the pure algorithm rather than by SQL: expressing "all prerequisites
1109
+ * complete" as a query is possible but would be a second, independently-maintained definition of
1110
+ * the same rule, free to drift from the one the in-run executor uses.
1111
+ */
1112
+ async findClaimableTasks(provider, limit) {
1113
+ const claimable = [];
1114
+ for (const parentID of await this.findActiveGraphIDs(provider)) {
1115
+ if (claimable.length >= limit)
1116
+ break;
1117
+ const graph = await this.loadGraphState(provider, parentID);
1118
+ // HOLD is what makes "a broken condition stalls visibly" true rather than merely stated.
1119
+ // An undecided exclusive group keeps all its edges, and a kept edge on a Complete origin
1120
+ // is a SATISFIED prerequisite — so without this filter every branch of the fork would be
1121
+ // eligible at once and all of them would run. A typo must not multiply a fork.
1122
+ //
1123
+ // The losers of a DECIDED group must be filtered for the same reason, and this is a race
1124
+ // rather than a rule: they are marked Skipped by the propagation pass, but between the
1125
+ // moment the group resolves and the moment that write lands, their incoming edge is still
1126
+ // a satisfied prerequisite on a Complete origin. A poll landing in that window would
1127
+ // claim and execute the branch the workflow chose NOT to take — irreversibly, since the
1128
+ // action has already run by the time Skipped is written over it.
1129
+ // `unreachableTaskIDs` joins the filter for exactly the reason above. R6 made a
1130
+ // definite-false ordinary edge seed the skip cascade rather than Block its target — but
1131
+ // until that Skipped write lands, the target has no unsatisfied prerequisite and is
1132
+ // vacuously eligible. That is the same race the XOR fix closed, reopened on the new
1133
+ // path: a branch the workflow decided against, claimed and executed irreversibly in the
1134
+ // window before it was marked.
1135
+ const eligible = ComputeEligibleTasks(graph.nodes, graph.edges, graph.handledFailureIDs)
1136
+ .filter((n) => !graph.holdTaskIDs.has(n.id) &&
1137
+ !graph.skipSeedTaskIDs.has(n.id) &&
1138
+ !graph.unreachableTaskIDs.has(n.id));
1139
+ for (const node of eligible) {
1140
+ const entity = graph.entityById.get(node.id);
1141
+ if (!entity)
1142
+ continue;
1143
+ // Human tasks are never dispatched — a person completes them. But "eligible" is the
1144
+ // moment that person can finally act, and nothing else in the system knows it has
1145
+ // arrived: the task sat Pending behind prerequisites, and no save touched it when
1146
+ // they cleared. Without a notification here a workflow simply stops, waiting on
1147
+ // someone who was never told. That silent stall is the failure mode this exists to
1148
+ // prevent, so it happens on the eligibility check rather than at submission.
1149
+ if (entity.ActionID) {
1150
+ // An action node this host has no runner for is left Pending rather than
1151
+ // claimed. Claiming it would take ownership of work this process cannot do, and
1152
+ // the claim would then have to expire before any host that CAN do it gets a
1153
+ // turn — a self-inflicted stall on a mixed deployment.
1154
+ if (!this.actionRunner)
1155
+ continue;
1156
+ }
1157
+ else if (entity.PromptID) {
1158
+ // A prompt node — including a loop that repeats a prompt — is assigned through
1159
+ // PromptID and carries NEITHER ActionID nor AgentID. Without this branch it fell
1160
+ // through to the test below and was treated as a task waiting on a PERSON: the
1161
+ // workflow notified a human who had nothing to do and then stopped forever.
1162
+ // That is precisely the misclassification the step-kind rules warn about, and it
1163
+ // is silent — the graph sits In Progress looking like it is still working.
1164
+ if (!this.promptRunner)
1165
+ continue;
1166
+ }
1167
+ else if (!entity.AgentID) {
1168
+ await this.notifyHumanTaskReady(entity, provider);
1169
+ continue;
1170
+ }
1171
+ if (this.inFlight.has(entity.ID))
1172
+ continue;
1173
+ claimable.push(entity);
1174
+ if (claimable.length >= limit)
1175
+ break;
1176
+ }
1177
+ }
1178
+ return claimable;
1179
+ }
1180
+ /**
1181
+ * Marks a human task as notified, so the request is raised exactly once.
1182
+ *
1183
+ * Written even when delivery threw. Retrying on every poll is a worse failure than one missed
1184
+ * notification: the task stays visible in the inbox either way, whereas a notification storm is
1185
+ * not self-correcting.
1186
+ */
1187
+ async markHumanTaskNotified(task) {
1188
+ task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
1189
+ if (!(await task.Save())) {
1190
+ LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
1191
+ }
1192
+ }
1193
+ /**
1194
+ * Raises the `MJ: AI Agent Requests` row a person answers to release this step.
1195
+ *
1196
+ * **Why that entity rather than something new.** It already models everything a workflow's human
1197
+ * step needs — who is being asked, what for, a typed response schema, priority, expiry, and an
1198
+ * inbox surface people already use. A second HITL substrate beside it would split the inbox in
1199
+ * two and leave one of them without expiry or permissions.
1200
+ *
1201
+ * **What it deliberately does NOT set is `ResumingAgentRunID`.** A request normally suspends an
1202
+ * agent run and resumes it. A workflow needs none of that: the graph OUTLIVES the run that
1203
+ * submitted it, so nothing is suspended — the task sits Pending, every other branch keeps
1204
+ * running, and answering settles the task. That column staying null is meaningful, not missing.
1205
+ */
1206
+ async raiseHumanRequest(task, provider) {
1207
+ try {
1208
+ const existing = await this.findOpenRequest(provider, task.ID);
1209
+ if (existing)
1210
+ return 'raised'; // already waiting on someone
1211
+ const request = await provider.GetEntityObject('MJ: AI Agent Requests', this.contextUser);
1212
+ request.NewRecord();
1213
+ request.OriginatingTaskID = task.ID;
1214
+ // A human task has NO AgentID of its own — that column names what EXECUTES a step, and
1215
+ // a person is not an agent. The request still needs one, so it carries the agent that
1216
+ // owns the workflow: the graph's own agent, which is who is asking.
1217
+ const owningAgentID = await this.owningAgentOf(provider, task);
1218
+ if (!owningAgentID) {
1219
+ // PERMANENT: a graph with no owning agent will not acquire one by being asked
1220
+ // again. Graphs submitted before the provenance stamp landed are all in this state.
1221
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} needs a person, but its workflow has no ` +
1222
+ `agent to ask on behalf of, so no request can be raised. The task stays Pending ` +
1223
+ `and visible in the Tasks UI; it will not be retried.`);
1224
+ return 'permanent-failure';
1225
+ }
1226
+ request.AgentID = owningAgentID;
1227
+ request.RequestForUserID = task.UserID;
1228
+ request.RequestedAt = new Date();
1229
+ request.Status = 'Requested';
1230
+ request.Request = task.Description || `A workflow is waiting on you to complete "${task.Name}".`;
1231
+ // The graph's own run is the provenance a reader follows back to see what led here.
1232
+ request.OriginatingAgentRunID = await this.submittingRunOf(provider, task);
1233
+ // The deadline, when the author set one. `expireOverdueRequests` has always been able to
1234
+ // enforce this — it expires the request and fails the step so a give-up edge can route
1235
+ // around it — but nothing ever WROTE the column, so that whole path had never run outside
1236
+ // a test and a workflow waiting on someone who left the company waited forever.
1237
+ // Absent means no deadline, deliberately: expiring on a timeout nobody chose would be
1238
+ // worse than waiting.
1239
+ const expiresInHours = this.parseConfiguration(task)?.human?.expiresInHours;
1240
+ if (expiresInHours && expiresInHours > 0) {
1241
+ request.ExpiresAt = new Date(Date.now() + expiresInHours * 60 * 60 * 1000);
1242
+ }
1243
+ if (!(await request.Save())) {
1244
+ LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ` +
1245
+ `${request.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1246
+ // A failed SAVE may be transient (deadlock, contention), so this one earns a retry.
1247
+ return 'transient-failure';
1248
+ }
1249
+ return 'raised';
1250
+ }
1251
+ catch (e) {
1252
+ // Never fatal. The task remains Pending and visible; a missing request is recoverable,
1253
+ // whereas throwing here would abort the whole dispatch pass for every other branch.
1254
+ LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
1255
+ return 'transient-failure';
1256
+ }
1257
+ }
1258
+ /**
1259
+ * The agent that owns this task's workflow — who the request is asked on behalf of.
1260
+ *
1261
+ * Reads the graph's parent row, falling back to the run that submitted it. A human step has no
1262
+ * agent of its own by design: `AgentID` names what EXECUTES a step, and a person is not an agent.
1263
+ */
1264
+ async owningAgentOf(provider, task) {
1265
+ if (task.AgentID)
1266
+ return task.AgentID;
1267
+ if (!task.ParentID)
1268
+ return null;
1269
+ try {
1270
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
1271
+ if (!(await parent.Load(task.ParentID)))
1272
+ return null;
1273
+ if (parent.AgentID)
1274
+ return parent.AgentID;
1275
+ if (!parent.AgentRunID)
1276
+ return null;
1277
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
1278
+ return (await run.Load(parent.AgentRunID)) ? run.AgentID : null;
1279
+ }
1280
+ catch {
1281
+ return null;
1282
+ }
1283
+ }
1284
+ /** The still-open request for a task, if one exists. */
1285
+ async findOpenRequest(provider, taskID) {
1286
+ const result = await RunView.FromMetadataProvider(provider).RunView({
1287
+ EntityName: 'MJ: AI Agent Requests',
1288
+ ExtraFilter: `OriginatingTaskID='${taskID}' AND Status='Requested'`,
1289
+ ResultType: 'entity_object',
1290
+ BypassCache: true,
1291
+ }, this.contextUser);
1292
+ return (result.Success ? result.Results?.[0] : null) ?? null;
1293
+ }
1294
+ /**
1295
+ * Settles a human task from the request a person answered.
1296
+ *
1297
+ * Runs on the poll rather than on a save hook, because the answer can arrive through any surface
1298
+ * — the inbox, the API, a conversation — and only the dispatcher knows how to release the rest
1299
+ * of the graph afterwards.
1300
+ *
1301
+ * **`ResponseData` becomes the task's output.** That is what makes a human step useful rather
1302
+ * than a gate: a downstream edge can branch on what the person actually said, typed by the
1303
+ * request's own ResponseSchema. A step that only recorded "approved" would force every decision
1304
+ * back into a separate action.
1305
+ */
1306
+ async settleAnsweredHumanTasks(provider, graphID) {
1307
+ const waiting = await RunView.FromMetadataProvider(provider).RunView({
1308
+ EntityName: 'MJ: Tasks',
1309
+ ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
1310
+ ResultType: 'entity_object',
1311
+ BypassCache: true,
1312
+ }, this.contextUser);
1313
+ if (!waiting.Success)
1314
+ return;
1315
+ for (const task of waiting.Results ?? []) {
1316
+ const request = await this.answeredRequestFor(provider, task.ID);
1317
+ if (!request)
1318
+ continue;
1319
+ const rejected = request.Status === 'Rejected';
1320
+ const expired = request.Status === 'Expired';
1321
+ task.Status = rejected || expired ? 'Failed' : 'Complete';
1322
+ task.CompletedAt = new Date();
1323
+ task.PercentComplete = rejected || expired ? 0 : 100;
1324
+ task.ClaimedBy = null;
1325
+ task.ClaimExpiresAt = null;
1326
+ task.OutputPayload = request.ResponseData ?? null;
1327
+ if (rejected) {
1328
+ task.ErrorMessage = request.Comments || 'A person rejected this step.';
1329
+ }
1330
+ else if (expired) {
1331
+ // Stated as a failure rather than left Pending. A workflow blocked forever on
1332
+ // someone who never answered — who may have left the company — is the silent stall
1333
+ // this whole path exists to avoid, and a give-up edge can now route around it.
1334
+ task.ErrorMessage = 'Nobody answered this step before its request expired.';
1335
+ }
1336
+ if (!(await task.Save())) {
1337
+ LogError(`[TaskGraphDispatcher] Could not settle human task ${task.ID}: ` +
1338
+ `${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1339
+ }
1340
+ }
1341
+ }
1342
+ /**
1343
+ * Re-opens a human step whose request was CANCELLED.
1344
+ *
1345
+ * `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
1346
+ * rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
1347
+ * it". Nothing replaced it. `raiseHumanRequest` refuses to raise twice (the notified marker on
1348
+ * `ClaimedBy` is what stops the notification storm), so a cancelled request left the task Pending
1349
+ * with no open request and no path to acquiring one: a workflow waiting forever on a question
1350
+ * nobody is being asked.
1351
+ *
1352
+ * Clearing the marker is the whole fix — the next poll sees an un-notified Pending human task
1353
+ * and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
1354
+ * human action: it takes another person cancelling again to come back here.
1355
+ */
1356
+ async reopenCancelledHumanTasks(provider, graphID) {
1357
+ const waiting = await RunView.FromMetadataProvider(provider).RunView({
1358
+ EntityName: 'MJ: Tasks',
1359
+ // `StepType` is NULLABLE, and rows predating the column exist (4 in the reference
1360
+ // database at the time of writing). None currently carry a UserID, but a human task
1361
+ // written by any path that set the assignee without the discriminator would be
1362
+ // invisible to a `StepType='Human'` filter and stay dead forever after a cancel —
1363
+ // the exact stall this method exists to end. The notified marker already narrows
1364
+ // this to tasks the dispatcher raised a request for, so the widening cannot pull in
1365
+ // unrelated work.
1366
+ ExtraFilter: `ParentID='${graphID}' AND Status='Pending' ` +
1367
+ `AND (StepType='Human' OR (StepType IS NULL AND UserID IS NOT NULL)) ` +
1368
+ `AND ClaimedBy='${HUMAN_TASK_NOTIFIED_MARKER}'`,
1369
+ ResultType: 'entity_object',
1370
+ BypassCache: true,
1371
+ }, this.contextUser);
1372
+ if (!waiting.Success)
1373
+ return;
1374
+ for (const task of waiting.Results ?? []) {
1375
+ // Only when there is nothing live AND nothing terminal. A task with an open request is
1376
+ // simply waiting; one with a terminal request is settled on the next pass by
1377
+ // settleAnsweredHumanTasks, and re-raising either would ask the same question twice.
1378
+ if (await this.findOpenRequest(provider, task.ID))
1379
+ continue;
1380
+ if (await this.answeredRequestFor(provider, task.ID))
1381
+ continue;
1382
+ LogStatus(`[TaskGraphDispatcher] The request for '${task.Name}' was cancelled and nothing ` +
1383
+ `replaced it; asking again.`);
1384
+ task.ClaimedBy = null;
1385
+ if (!(await task.Save())) {
1386
+ LogError(`[TaskGraphDispatcher] Could not re-open cancelled human task ${task.ID}: ` +
1387
+ `${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1388
+ }
1389
+ }
1390
+ }
1391
+ /** The answered (or expired) request for a task, if any. */
1392
+ async answeredRequestFor(provider, taskID) {
1393
+ const result = await RunView.FromMetadataProvider(provider).RunView({
1394
+ EntityName: 'MJ: AI Agent Requests',
1395
+ // Everything terminal. 'Canceled' is deliberately absent: a cancelled request means
1396
+ // the ASK was withdrawn, not that the step was decided, so the task keeps waiting
1397
+ // for whatever replaces it.
1398
+ ExtraFilter: `OriginatingTaskID='${taskID}' AND Status IN ('Approved','Rejected','Responded','Expired')`,
1399
+ OrderBy: 'RespondedAt DESC',
1400
+ ResultType: 'entity_object',
1401
+ BypassCache: true,
1402
+ }, this.contextUser);
1403
+ return (result.Success ? result.Results?.[0] : null) ?? null;
1404
+ }
1405
+ /**
1406
+ * Expires requests whose deadline has passed.
1407
+ *
1408
+ * A deadline that nothing enforces is a comment. Without this an `ExpiresAt` in the past leaves
1409
+ * the request `Requested` forever and the workflow waiting on it just as long.
1410
+ */
1411
+ async expireOverdueRequests(provider, graphID) {
1412
+ // Scoped by an explicit id list rather than a subquery against a view name, so this reads
1413
+ // the same on any provider rather than assuming a SQL dialect and a physical view.
1414
+ const humanTasks = await RunView.FromMetadataProvider(provider).RunView({
1415
+ EntityName: 'MJ: Tasks',
1416
+ Fields: ['ID'],
1417
+ ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
1418
+ ResultType: 'simple',
1419
+ }, this.contextUser);
1420
+ const ids = (humanTasks.Results ?? []).map((r) => `'${r.ID}'`);
1421
+ if (ids.length === 0)
1422
+ return;
1423
+ const nowISO = new Date().toISOString();
1424
+ const overdue = await RunView.FromMetadataProvider(provider).RunView({
1425
+ EntityName: 'MJ: AI Agent Requests',
1426
+ ExtraFilter: `Status='Requested' AND ExpiresAt IS NOT NULL AND ExpiresAt < '${nowISO}' ` +
1427
+ `AND OriginatingTaskID IN (${ids.join(',')})`,
1428
+ ResultType: 'entity_object',
1429
+ BypassCache: true,
1430
+ }, this.contextUser);
1431
+ if (!overdue.Success)
1432
+ return;
1433
+ for (const request of overdue.Results ?? []) {
1434
+ request.Status = 'Expired';
1435
+ if (!(await request.Save())) {
1436
+ LogError(`[TaskGraphDispatcher] Could not expire request ${request.ID}.`);
1437
+ }
1438
+ }
1439
+ }
1440
+ /** The agent run that submitted this task's graph, for provenance on the request. */
1441
+ async submittingRunOf(provider, task) {
1442
+ if (!task.ParentID)
1443
+ return null;
1444
+ try {
1445
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
1446
+ return (await parent.Load(task.ParentID)) ? parent.AgentRunID : null;
1447
+ }
1448
+ catch {
1449
+ return null;
1450
+ }
1451
+ }
1452
+ /** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
1453
+ async loadGraphState(provider, parentTaskID) {
1454
+ const rv = RunView.FromMetadataProvider(provider);
1455
+ // BypassCache throughout: task status is written by the claim protocol's direct SQL, which
1456
+ // fires no cache invalidation. See findActiveGraphIDs.
1457
+ const childrenResult = await rv.RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
1458
+ const children = (childrenResult.Success ? childrenResult.Results : []) ?? [];
1459
+ if (children.length === 0) {
1460
+ return {
1461
+ nodes: [], edges: [], entityById: new Map(),
1462
+ unreachableTaskIDs: new Set(), skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
1463
+ handledFailureIDs: new Set(),
1464
+ };
1465
+ }
1466
+ const idList = children.map((c) => `'${c.ID}'`).join(',');
1467
+ const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
1468
+ const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
1469
+ const entityById = new Map(children.map((c) => [c.ID, c]));
1470
+ // Conditional edges are resolved HERE, before eligibility runs, by dropping edges whose
1471
+ // condition does not hold. Expressing it as edge removal rather than as a second rule inside
1472
+ // the eligibility algorithm is what keeps one definition of "ready": a task with no live
1473
+ // incoming edges is ready for exactly the same reason a task with no edges at all is.
1474
+ //
1475
+ // An edge whose condition cannot be evaluated is KEPT, which is the opposite of the flow
1476
+ // executor's choice and deliberately so. There, a broken condition means an edge is not
1477
+ // followed and the graph moves on. Here it would mean a prerequisite silently disappears and
1478
+ // the dependent task runs early — turning a typo into out-of-order execution. Keeping the
1479
+ // edge instead stalls the graph, which the stall detector already reports loudly.
1480
+ const liveEdges = [];
1481
+ // A definitely-false edge must not merely disappear. Removing a task's only prerequisite
1482
+ // makes it eligible in the very next wave — so "this branch was not taken" would execute the
1483
+ // branch, potentially before the node that gated it. The dependent is recorded as
1484
+ // unreachable instead, and blocked before anything can claim it.
1485
+ const droppedInto = new Set();
1486
+ const stillReachable = new Set();
1487
+ // EXCLUSIVE edges are exempt from the generic machinery below, and that exemption is
1488
+ // load-bearing. An XOR loser is by definition condition-false, so the ordinary path would
1489
+ // record it as unreachable and Block it — and a Blocked child poisons the parent rollup, so
1490
+ // every fork would settle the graph as Blocked. Losers must become Skipped instead, which
1491
+ // only ResolveExclusiveGroups can decide.
1492
+ const exclusive = deps.filter((d) => !!d.ExclusiveGroup);
1493
+ const ordinary = deps.filter((d) => !d.ExclusiveGroup);
1494
+ const resolution = ResolveExclusiveGroups(exclusive.map((d) => ({
1495
+ id: d.ID,
1496
+ taskId: d.TaskID,
1497
+ dependsOnTaskId: d.DependsOnTaskID,
1498
+ exclusiveGroup: d.ExclusiveGroup,
1499
+ originStatus: (entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending'),
1500
+ priority: d.Priority ?? 0,
1501
+ sequence: d.Sequence ?? 0,
1502
+ conditionOutcome: this.evaluateExclusiveCondition(d, entityById),
1503
+ })),
1504
+ // A flow's failure handling is its outgoing edges, so a Failed origin still decides its
1505
+ // group. For a loop-agent graph the set is Complete-only and nothing changes.
1506
+ new Set(['Complete', 'Failed']));
1507
+ const loserEdgeIDs = new Set(resolution.loserEdgeIDs);
1508
+ for (const d of ordinary) {
1509
+ if (d.Condition?.trim()) {
1510
+ const outcome = this.evaluateEdgeCondition(d, entityById);
1511
+ if (outcome === 'drop') {
1512
+ droppedInto.add(d.TaskID);
1513
+ continue;
1514
+ }
1515
+ }
1516
+ stillReachable.add(d.TaskID);
1517
+ liveEdges.push({
1518
+ taskId: d.TaskID,
1519
+ dependsOnTaskId: d.DependsOnTaskID,
1520
+ dependencyType: d.DependencyType,
1521
+ });
1522
+ }
1523
+ for (const d of exclusive) {
1524
+ // A losing edge is removed rather than left to gate: its target is being skipped, and a
1525
+ // live edge into a skipped task would keep the graph waiting on a branch nobody took.
1526
+ if (loserEdgeIDs.has(d.ID))
1527
+ continue;
1528
+ stillReachable.add(d.TaskID);
1529
+ liveEdges.push({
1530
+ taskId: d.TaskID,
1531
+ dependsOnTaskId: d.DependsOnTaskID,
1532
+ dependencyType: d.DependencyType,
1533
+ });
1534
+ }
1535
+ // Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
1536
+ // waiting on it, and a node reached by an alternate branch is genuinely reachable.
1537
+ const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
1538
+ const nodes = children.map((c) => ({ id: c.ID, status: c.Status }));
1539
+ return {
1540
+ nodes,
1541
+ edges: liveEdges,
1542
+ entityById,
1543
+ unreachableTaskIDs,
1544
+ skipSeedTaskIDs: new Set(resolution.skipSeedTaskIDs),
1545
+ holdTaskIDs: new Set(resolution.holdTaskIDs),
1546
+ handledFailureIDs: await this.computeHandledFailures(provider, parentTaskID, nodes, liveEdges),
1547
+ };
1548
+ }
1549
+ /**
1550
+ * Decides whether a conditional dependency edge is live.
1551
+ *
1552
+ * The condition sees the upstream task's outcome — its status and parsed output — which is the
1553
+ * only information a runtime graph has to branch on. Returns `'drop'` only on a definite false;
1554
+ * an unevaluable condition keeps the edge for the reason stated at the call site.
1555
+ */
1556
+ evaluateEdgeCondition(dep, entityById) {
1557
+ const upstream = entityById.get(dep.DependsOnTaskID);
1558
+ if (!upstream)
1559
+ return 'keep';
1560
+ // TERMINALITY GUARD — fixes a latent bug, not a hypothetical one.
1561
+ //
1562
+ // Without it, every conditional edge is evaluated on every poll cycle, including while its
1563
+ // origin is still Pending. A condition like `succeeded` is then a DEFINITE FALSE, the edge
1564
+ // is dropped, and the target is Blocked at wave one — permanently, before the origin ever
1565
+ // ran. That kills any conditioned linear chain, which is the most common flow shape there
1566
+ // is.
1567
+ //
1568
+ // A non-terminal origin is UNDECIDED, and 'keep' is the safe reading of undecided: the
1569
+ // prerequisite gate already prevents the target starting early, so keeping the edge costs
1570
+ // nothing and dropping it is irreversible.
1571
+ if (!TERMINAL_FOR_CONDITIONS.has(upstream.Status))
1572
+ return 'keep';
1573
+ let output = null;
1574
+ if (upstream.OutputPayload) {
1575
+ try {
1576
+ output = JSON.parse(upstream.OutputPayload);
1577
+ }
1578
+ catch { /* a malformed payload is not grounds to drop a prerequisite */ }
1579
+ }
1580
+ const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
1581
+ if (!result.Success) {
1582
+ LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
1583
+ `(${result.ErrorMessage}); keeping the edge so the graph stalls visibly rather than ` +
1584
+ `running ${dep.TaskID} out of order.`);
1585
+ return 'keep';
1586
+ }
1587
+ return result.Value ? 'keep' : 'drop';
1588
+ }
1589
+ /**
1590
+ * An exclusive edge's condition as a three-way outcome.
1591
+ *
1592
+ * `ResolveExclusiveGroups` needs to tell "false" from "could not be evaluated": the first loses
1593
+ * the branch, the second holds the whole group. The generic keep/drop path cannot express that
1594
+ * difference, which is why exclusive edges take this route instead.
1595
+ */
1596
+ evaluateExclusiveCondition(dep, entityById) {
1597
+ if (!dep.Condition?.trim())
1598
+ return 'satisfied';
1599
+ const upstream = entityById.get(dep.DependsOnTaskID);
1600
+ if (!upstream)
1601
+ return 'unevaluable';
1602
+ let output = null;
1603
+ if (upstream.OutputPayload) {
1604
+ try {
1605
+ output = JSON.parse(upstream.OutputPayload);
1606
+ }
1607
+ catch { /* malformed payload */ }
1608
+ }
1609
+ const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
1610
+ if (!result.Success)
1611
+ return 'unevaluable';
1612
+ return result.Value ? 'satisfied' : 'unsatisfied';
1613
+ }
1614
+ /**
1615
+ * Everything an edge condition can see — the SUPERSET of both dialects.
1616
+ *
1617
+ * A flow condition is written against `payload` / `stepResult` / `flowContext` / `data` /
1618
+ * `context`; the dispatcher's own conditions are written against `status` / `succeeded` /
1619
+ * `failed` / `output` / `errorMessage`. Compiling flows onto this engine without the flow
1620
+ * dialect would make every `payload.x` condition evaluate against nothing — silently, since an
1621
+ * undefined property is simply falsy. Both dialects are readable here so a condition means the
1622
+ * same thing on either engine.
1623
+ *
1624
+ * `payload` is the ORIGIN task's post-step snapshot. There is deliberately no "graph-wide
1625
+ * payload": each task's output is its own, and inventing a merged one would give conditions a
1626
+ * value the flow engine never had.
1627
+ */
1628
+ buildConditionContext(upstream, output) {
1629
+ const envelope = (output && typeof output === 'object' ? output : {});
1630
+ const succeeded = upstream.Status === 'Complete';
1631
+ return {
1632
+ // dispatcher dialect — unchanged
1633
+ status: upstream.Status,
1634
+ succeeded,
1635
+ failed: upstream.Status === 'Failed',
1636
+ output,
1637
+ errorMessage: upstream.ErrorMessage ?? null,
1638
+ // flow dialect
1639
+ payload: envelope.payload ?? output,
1640
+ stepResult: { Success: succeeded, step: upstream.Name, result: envelope.result ?? output },
1641
+ flowContext: { currentStepId: upstream.ID, completedSteps: [], executionPath: [], stepCount: 0 },
1642
+ data: envelope.data ?? {},
1643
+ context: envelope.context ?? {},
1644
+ };
1645
+ }
1646
+ /** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
1647
+ async loadDependencyOutputs(provider, taskID) {
1648
+ const outputs = new Map();
1649
+ const rv = RunView.FromMetadataProvider(provider);
1650
+ // BypassCache: an upstream task's OutputPayload is written on the completion path, so a
1651
+ // cached read here can hand a dependent task the previous run's output — or none at all.
1652
+ const deps = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID='${taskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
1653
+ for (const dep of (deps.Success ? deps.Results : []) ?? []) {
1654
+ const upstream = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
1655
+ if (!(await upstream.Load(dep.DependsOnTaskID)))
1656
+ continue;
1657
+ if (!upstream.OutputPayload)
1658
+ continue;
1659
+ try {
1660
+ outputs.set(dep.DependsOnTaskID, JSON.parse(upstream.OutputPayload));
1661
+ }
1662
+ catch (e) {
1663
+ LogError(`[TaskGraphDispatcher] Task ${dep.DependsOnTaskID} has malformed OutputPayload: ${e}`);
1664
+ }
1665
+ }
1666
+ return outputs;
1667
+ }
1668
+ /**
1669
+ * Runs one task's body, whatever kind of step it is.
1670
+ *
1671
+ * **Routing is on `StepType`, not on which key happens to be set.** A loop step carries the same
1672
+ * `ActionID` or `AgentID` as an ordinary step — that key is what the loop *repeats* — so the old
1673
+ * `task.ActionID ? action : agent` test would have run a loop exactly once and called it done.
1674
+ * `StepType` is the only field that distinguishes them.
1675
+ *
1676
+ * Every branch is normalized to one shape so the recording path above stays single: an action has
1677
+ * no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
1678
+ */
1679
+ async runTaskBody(task, provider, inputPayload, dependencyOutputs) {
1680
+ const payload = this.mergedPayload(inputPayload, dependencyOutputs);
1681
+ const config = task.ConfigurationObject;
1682
+ // A loop's own step type decides how many times its body runs; the body itself is dispatched
1683
+ // through the very same runners as a one-shot step.
1684
+ if (task.StepType === 'ForEach' || task.StepType === 'While') {
1685
+ return { ...await this.runLoopTask(task, provider, payload, dependencyOutputs), PayloadAtStart: payload };
1686
+ }
1687
+ const { params, errors } = BuildMappedInput(config?.inputMapping, { payload });
1688
+ for (const e of errors)
1689
+ LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
1690
+ // `payload`, NOT `inputPayload` — the MERGED value computed above, which includes what every
1691
+ // dependency produced.
1692
+ //
1693
+ // A step with an input mapping got exactly the parameters it declared; a step WITHOUT one
1694
+ // fell back to the raw input and therefore saw nothing any earlier step had produced. For a
1695
+ // Prompt step — which declares no mapping by design, because it reads the whole payload
1696
+ // through `{{ _CURRENT_PAYLOAD }}` — that meant the placeholder rendered `{}` and the model
1697
+ // was asked to write from an empty brief.
1698
+ //
1699
+ // It answered anyway. The Content Pipeline's draft step said "the research data was empty",
1700
+ // which was TRUE of what it had been handed while twenty research results sat in the
1701
+ // dependency outputs beside it, and the reviewer then rejected the draft for saying so.
1702
+ // Every layer looked like it was working.
1703
+ const effectiveInput = Object.keys(params).length > 0 ? params : payload;
1704
+ if (task.StepType === 'Prompt') {
1705
+ if (!this.promptRunner) {
1706
+ // Not a failure: "nobody here can run this" is not "this ran and did not work".
1707
+ return { Success: false, AgentRunID: null, ErrorMessage: 'No prompt runner is loaded on this host.' };
1708
+ }
1709
+ const promptResult = await this.promptRunner.RunPromptForTask({
1710
+ TaskID: task.ID,
1711
+ PromptID: task.PromptID,
1712
+ InputPayload: effectiveInput,
1713
+ DependencyOutputs: dependencyOutputs,
1714
+ TemplateParameters: config?.prompt?.templateParameters,
1715
+ Provider: provider,
1716
+ ContextUser: this.contextUser,
1717
+ });
1718
+ // A prompt's response is DEEP-MERGED into the payload rather than replacing it. A prompt
1719
+ // answers one question; replacing the payload with its answer would discard everything
1720
+ // the steps before it established, which is how a late step loses the data it depends on.
1721
+ const merged = promptResult.Success && promptResult.Output && typeof promptResult.Output === 'object'
1722
+ ? deepMergePayload(payload, promptResult.Output)
1723
+ : payload;
1724
+ return {
1725
+ Success: promptResult.Success,
1726
+ AgentRunID: null,
1727
+ ErrorMessage: promptResult.ErrorMessage,
1728
+ Output: this.applyStepOutputMapping(task, merged, merged, config?.outputMapping),
1729
+ PayloadAtStart: payload,
1730
+ ChatMessage: promptResult.ChatMessage,
1731
+ // Returned even when the prompt FAILED. A failed prompt still cost tokens, and a
1732
+ // cost rollup that silently omits failures under-reports exactly the runs someone
1733
+ // is most likely to be investigating.
1734
+ PromptRunID: promptResult.PromptRunID,
1735
+ };
1736
+ }
1737
+ const raw = task.ActionID
1738
+ ? { ...await this.actionRunner.RunActionForTask({
1739
+ TaskID: task.ID,
1740
+ ActionID: task.ActionID,
1741
+ InputPayload: effectiveInput,
1742
+ DependencyOutputs: dependencyOutputs,
1743
+ Provider: provider,
1744
+ ContextUser: this.contextUser,
1745
+ }), AgentRunID: null }
1746
+ : await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs);
1747
+ return {
1748
+ ...raw,
1749
+ Output: this.applyStepOutputMapping(task, payload, raw.Output, config?.outputMapping),
1750
+ PayloadAtStart: payload,
1751
+ };
1752
+ }
1753
+ /**
1754
+ * Runs a loop step: its body once per iteration, with the item and index in scope.
1755
+ *
1756
+ * The loop's own `Configuration` supplies the definition; the row's `ActionID` / `AgentID`
1757
+ * supplies what to repeat. Per-iteration inputs are resolved fresh each pass — the bindings are
1758
+ * merged into the payload before the mapping is applied, which is how a body can reference the
1759
+ * current item at all.
1760
+ */
1761
+ async runLoopTask(task, provider, payload, dependencyOutputs) {
1762
+ const config = task.ConfigurationObject;
1763
+ const op = task.StepType === 'ForEach' ? config?.forEach : config?.while;
1764
+ if (!op) {
1765
+ return {
1766
+ Success: false,
1767
+ AgentRunID: null,
1768
+ ErrorMessage: `"${task.Name}" is a ${task.StepType} step with no loop settings, so there is nothing to repeat.`,
1769
+ };
1770
+ }
1771
+ // A prompt body has no params of its own — it receives the payload (with the loop bindings
1772
+ // merged in) through the placeholder, so an empty mapping is correct rather than missing.
1773
+ const bodyMapping = (op.action?.params ?? {});
1774
+ // The BODY's output mapping, applied once per pass — see `foldIterationOutput`.
1775
+ //
1776
+ // It used to be applied a single time after the loop finished, against the accumulated
1777
+ // payload. That is the wrong moment in two ways at once: the mapping names an output
1778
+ // PARAMETER of the body, which no longer exists by then, and a mapping like
1779
+ // `"Items": "results[]"` can only append per pass. So every pass merged its raw result into
1780
+ // the shared payload instead, each overwriting the last, and the mapping matched nothing and
1781
+ // wrote nothing. A ForEach over five items reported five successes and kept item five.
1782
+ const bodyOutputMapping = op.action?.outputMapping ?? op.prompt?.outputMapping;
1783
+ // Where this step sits in its graph, resolved ONCE rather than per iteration. A loop body is
1784
+ // dispatched exactly like a one-shot step and needs the same two things: the run that
1785
+ // submitted the graph (so a spawned run gets a ParentRunID and is visible to the tree and to
1786
+ // cost), and the continuation depth (so the recursion cap still applies). Omitting them made
1787
+ // loop bodies second-class in every dimension — and reopened the unbounded-recursion hole
1788
+ // THROUGH loops, since each spawned run restarted the chain at zero.
1789
+ const graphContext = await this.graphContext(provider, task);
1790
+ // THE LOOP'S PAYLOAD ACCUMULATES. Each iteration's output merges in, and the next iteration
1791
+ // — and the While condition — sees it. Without this the condition closure re-read the
1792
+ // payload as it was when the loop STARTED, so a `while payload.brandOK !== true` could never
1793
+ // become false: the loop burned every iteration re-examining the original input and always
1794
+ // took the give-up branch, making the other branch unreachable. The loop ran, reported
1795
+ // success, and its result was predetermined.
1796
+ let livePayload = { ...payload };
1797
+ // One entry per pass, so the loop's work exists somewhere the platform can see it. Without
1798
+ // this a loop is a single childless node: the run tree reaches nested work through six links
1799
+ // and an iteration is none of them, so the passes were invisible to the timeline AND their
1800
+ // spend was missing from the settlement rollup. See ITaskStepRuntime.iterations.
1801
+ const iterationTrace = [];
1802
+ // Bounds what the trace's payloads may cost. The pointers are never budgeted — those are the
1803
+ // durable record of the work and must survive whatever the payloads do.
1804
+ const budget = new IterationPayloadBudget();
1805
+ const invokeBody = async ({ Index, Bindings }) => {
1806
+ // Bindings go INTO the payload rather than beside it, so an authored mapping reaches the
1807
+ // current item the same way it reaches anything else: `payload.<itemVariable>`.
1808
+ const iterationPayload = { ...livePayload, ...Bindings };
1809
+ const resolved = ResolveMappedInput(bodyMapping, { payload: iterationPayload });
1810
+ /**
1811
+ * Folds an iteration's output into the running payload the next pass will see, and
1812
+ * records what the pass produced.
1813
+ *
1814
+ * The trace is written HERE rather than after the loop because a loop that fails partway
1815
+ * still ran the passes before it, and their runs are real spend that must not vanish
1816
+ * because the loop as a whole did not finish.
1817
+ */
1818
+ const absorb = (outcome, bodyInput) => {
1819
+ livePayload = this.foldIterationOutput(task, livePayload, outcome.Output, bodyOutputMapping);
1820
+ iterationTrace.push({
1821
+ index: Index,
1822
+ // What THIS pass was handed and what it gave back — not the loop's running
1823
+ // payload before and after it.
1824
+ //
1825
+ // A pass has no row of its own, so without these there is nowhere its work can be
1826
+ // recorded: every iteration presented null on both sides and the run view could
1827
+ // say nothing about any single pass, which for a loop is the only interesting
1828
+ // question. But recording the RUNNING payload on both sides — the obvious reading
1829
+ // of "before and after" — is quadratic: each pass would hold a full copy of
1830
+ // everything every earlier pass accumulated. A five-iteration demo produced a
1831
+ // 121KB Configuration that way; the same loop over fifty items would produce
1832
+ // megabytes, in a column every reader of the row pays to load.
1833
+ //
1834
+ // The pass's own input and output are what a reader actually wants ("what did
1835
+ // pass three do?"), and they are constant-sized per pass.
1836
+ payloadAtStart: budget.Take(bodyInput),
1837
+ payloadAtEnd: budget.Take(outcome.Output),
1838
+ promptRunID: outcome.PromptRunID,
1839
+ agentRunID: outcome.AgentRunID,
1840
+ // An ACTION body records its log here. Omitting it left an action-bodied pass
1841
+ // with no pointer at all — no cost, no timing, nothing to open — and the tree,
1842
+ // seeing neither a prompt run nor an agent run, fell through to its last branch
1843
+ // and called the pass a Sub-Agent. A loop over a web search then showed five
1844
+ // sub-agent runs that never existed.
1845
+ actionLogID: outcome.ActionLogID,
1846
+ success: outcome.Success,
1847
+ errorMessage: outcome.ErrorMessage,
1848
+ });
1849
+ return outcome;
1850
+ };
1851
+ // A prompt body is checked FIRST because it is the only one whose id lives in its own
1852
+ // column: a loop repeating a prompt has PromptID set and both ActionID and AgentID null,
1853
+ // so falling through to the agent branch would dereference a null agent id.
1854
+ if (task.StepType && task.PromptID && !task.ActionID) {
1855
+ if (!this.promptRunner) {
1856
+ return { Success: false, ErrorMessage: 'No prompt runner is loaded on this host.' };
1857
+ }
1858
+ return absorb(await this.promptRunner.RunPromptForTask({
1859
+ TaskID: task.ID,
1860
+ PromptID: task.PromptID,
1861
+ // The ITERATION payload, not the mapped params. An action body declares its
1862
+ // inputs and gets exactly those; a prompt body declares none — it receives the
1863
+ // whole payload through the placeholder, and the loop's item and index are
1864
+ // merged INTO that payload. Passing the mapped result here handed the prompt an
1865
+ // empty object, so every iteration asked the model to describe nothing and got
1866
+ // five confident answers about nothing back.
1867
+ InputPayload: iterationPayload,
1868
+ DependencyOutputs: dependencyOutputs,
1869
+ // The loop's bindings become TEMPLATE VARIABLES, so an author writes
1870
+ // `{{ field }}` for the item the loop is on — which is what `itemVariable` is
1871
+ // for, and what anyone reading the step's configuration expects. Reaching it
1872
+ // through the payload placeholder instead works but is not discoverable, and
1873
+ // getting it wrong is silent: the variable renders empty and the model answers
1874
+ // confidently about nothing.
1875
+ TemplateParameters: { ...stringifyBindings(Bindings), ...op.prompt?.templateParameters },
1876
+ Provider: provider,
1877
+ ContextUser: this.contextUser,
1878
+ }), iterationPayload);
1879
+ }
1880
+ if (task.ActionID) {
1881
+ return absorb(await this.actionRunner.RunActionForTask({
1882
+ TaskID: task.ID,
1883
+ ActionID: task.ActionID,
1884
+ InputPayload: resolved,
1885
+ DependencyOutputs: dependencyOutputs,
1886
+ Provider: provider,
1887
+ ContextUser: this.contextUser,
1888
+ }), resolved);
1889
+ }
1890
+ const agentInput = Object.keys(resolved).length > 0 ? resolved : iterationPayload;
1891
+ return absorb(await this.agentRunner.RunAgentForTask({
1892
+ TaskID: task.ID,
1893
+ AgentID: task.AgentID,
1894
+ // The ITERATION payload when the body declares no inputs of its own. A sub-agent
1895
+ // body has no `params`, so the mapped result is `{}` — every iteration was handing
1896
+ // the agent nothing and asking it to work from that.
1897
+ InputPayload: agentInput,
1898
+ DependencyOutputs: dependencyOutputs,
1899
+ ContinuationDepth: graphContext.Depth,
1900
+ SubmittingAgentRunID: graphContext.SubmittingAgentRunID,
1901
+ Provider: provider,
1902
+ ContextUser: this.contextUser,
1903
+ }), agentInput);
1904
+ };
1905
+ const outcome = task.StepType === 'ForEach'
1906
+ ? await RunForEachLoop(op, { payload }, invokeBody)
1907
+ : await RunWhileLoop(op, (iteration) => this.conditionEvaluator.Evaluate(op.condition,
1908
+ // BOTH forms, because a workflow should not have two condition dialects. An
1909
+ // EDGE condition is written `payload.brandOK !== true`; a loop condition used
1910
+ // to see the payload's keys spread at the top level and nothing named `payload`,
1911
+ // so the same expression that routes an edge failed here with
1912
+ // "payload is not defined". The spread stays for conditions already written
1913
+ // against it.
1914
+ { ...livePayload, payload: livePayload, iteration }), invokeBody);
1915
+ return {
1916
+ Success: outcome.Success,
1917
+ AgentRunID: null,
1918
+ ErrorMessage: outcome.ErrorMessage,
1919
+ // Every pass that ran, including those before a failure — see `iterationTrace`.
1920
+ Iterations: iterationTrace.length > 0 ? iterationTrace : undefined,
1921
+ // The ACCUMULATED payload — everything the iterations established — not the one the
1922
+ // loop started with, which would discard the loop's whole effect on the workflow.
1923
+ //
1924
+ // Only the STEP's own mapping is applied here. The body's mapping already ran once per
1925
+ // pass inside `foldIterationOutput`; applying it again against the accumulated payload
1926
+ // is what used to make it match nothing.
1927
+ Output: this.applyStepOutputMapping(task, livePayload, outcome.Output, config?.outputMapping),
1928
+ };
1929
+ }
1930
+ /**
1931
+ * Folds one pass's result into the loop's running payload.
1932
+ *
1933
+ * **With a body mapping**, the pass's declared outputs are filed where the author said to put
1934
+ * them — including `name[]`, which appends, so a ForEach can collect one entry per item. That is
1935
+ * the whole point of a loop over a collection, and it is only expressible per pass.
1936
+ *
1937
+ * **Without one**, the raw result is deep-merged, which is the pre-existing behaviour and the
1938
+ * right default for a `While` that converges on a value: each pass refines what the condition
1939
+ * reads. It is the wrong default for a ForEach that collects — hence the mapping.
1940
+ *
1941
+ * An unmapped output is reported per pass rather than swallowed, for the same reason
1942
+ * {@link applyStepOutputMapping} reports it: a mapping that names something the body never
1943
+ * returned means the pass did work that went nowhere, while everything reports success.
1944
+ */
1945
+ foldIterationOutput(task, livePayload, output, bodyOutputMapping) {
1946
+ if (!output || typeof output !== 'object' || Array.isArray(output))
1947
+ return livePayload;
1948
+ const source = output;
1949
+ if (!bodyOutputMapping)
1950
+ return deepMergePayload(livePayload, source);
1951
+ // Applied ONTO a deep copy of the running payload, not into a fresh object: `name[]` appends,
1952
+ // and appending is meaningless without the list already there. The copy is deep because the
1953
+ // trace has already recorded earlier passes' payloads — mutating a shared nested array would
1954
+ // retroactively rewrite what those passes are recorded as having seen.
1955
+ const { updates, errors, unmapped } = ApplyOutputMapping(source, bodyOutputMapping, structuredClone(livePayload));
1956
+ for (const e of errors)
1957
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} loop body: ${e}`);
1958
+ if (unmapped?.length) {
1959
+ LogError(`[TaskGraphDispatcher] '${task.Name}' loop body mapped output(s) it did not return: ` +
1960
+ `${unmapped.join(', ')}. The pass returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
1961
+ `Those payload values were NOT written, so anything downstream reading them sees nothing.`);
1962
+ }
1963
+ // `updates` IS the copy that was applied onto, so it is already the complete next payload.
1964
+ return updates;
1965
+ }
1966
+ /**
1967
+ * Files a step's result into the payload it hands downstream.
1968
+ *
1969
+ * **This is what makes a branch condition possible.** A workflow that branches on
1970
+ * `payload.stockPrice` has that value only because this step mapped `CurrentPrice -> stockPrice`.
1971
+ * Without it the condition reads `undefined` — merely falsy — so the workflow takes the other
1972
+ * branch, finishes, and reports success with nothing to indicate anything went wrong.
1973
+ *
1974
+ * The incoming payload is carried through as well as the update, so a value written three steps
1975
+ * back is still readable here. Returning only this step's own output is what used to limit a
1976
+ * condition's view to its immediate predecessor.
1977
+ */
1978
+ applyStepOutputMapping(task, payload, output, outputMapping) {
1979
+ // No mapping: MERGE the step's output over the payload rather than replacing it.
1980
+ //
1981
+ // Replacing is what made the Content Pipeline's exclusive pair unreachable. A While loop's
1982
+ // own output is a SUMMARY — `{iterations, succeeded, failed, results}` — so returning it
1983
+ // discarded the payload the iterations had built, including the `brandOK` the reviewer had
1984
+ // just set to true. The edges read `payload.brandOK === true` and `!== true`; against a
1985
+ // summary the first is false and the second is true, so the give-up branch won on EVERY run
1986
+ // no matter what the reviewer decided. The approved branch was unreachable in practice while
1987
+ // being perfectly reachable on the canvas.
1988
+ //
1989
+ // This is the same rule the mapped path already follows two lines down, and the same rule
1990
+ // the doc comment above states. The no-mapping branch was simply not following it.
1991
+ if (!outputMapping) {
1992
+ return output && typeof output === 'object' && !Array.isArray(output)
1993
+ ? { ...payload, ...output }
1994
+ : output ?? payload;
1995
+ }
1996
+ const source = output && typeof output === 'object' ? output : { value: output };
1997
+ const { updates, errors, unmapped } = ApplyOutputMapping(source, outputMapping);
1998
+ for (const e of errors)
1999
+ LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
2000
+ // A mapping that names an output the step never produced discards that step's work while
2001
+ // the step reports Complete. It is not fatal — an action may emit a parameter only on some
2002
+ // paths — but it must not be silent, and naming what WAS returned turns a multi-table
2003
+ // forensic exercise into one line. The Content Pipeline demo lost an entire research pass
2004
+ // this way, every run, because its mapping named another action's parameter.
2005
+ if (unmapped?.length) {
2006
+ LogError(`[TaskGraphDispatcher] '${task.Name}' mapped output(s) the step did not return: ` +
2007
+ `${unmapped.join(', ')}. The step returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
2008
+ `Those payload values were NOT written, so anything downstream reading them sees nothing.`);
2009
+ }
2010
+ return { ...payload, ...updates };
2011
+ }
2012
+ /**
2013
+ * Runs an Agent step, telling the runner where in the graph it sits.
2014
+ *
2015
+ * Depth and provenance are read together because they come from the same row: the graph's parent
2016
+ * task knows both how many continuation hops led here and which run submitted it.
2017
+ */
2018
+ async runAgentNode(task, provider, effectiveInput, dependencyOutputs) {
2019
+ const context = await this.graphContext(provider, task);
2020
+ return this.agentRunner.RunAgentForTask({
2021
+ TaskID: task.ID,
2022
+ AgentID: task.AgentID,
2023
+ InputPayload: effectiveInput,
2024
+ DependencyOutputs: dependencyOutputs,
2025
+ ContinuationDepth: context.Depth,
2026
+ SubmittingAgentRunID: context.SubmittingAgentRunID,
2027
+ Provider: provider,
2028
+ ContextUser: this.contextUser,
2029
+ });
2030
+ }
2031
+ /**
2032
+ * Completes the agent run that parked on this graph.
2033
+ *
2034
+ * **This is the other half of submit-and-detach.** A run that dispatches a graph does not
2035
+ * complete at submission — it ends `Paused`, because reporting `Completed` above a workflow
2036
+ * where nothing has happened yet is a claim the row cannot support. The run's lifecycle is
2037
+ * finished HERE, when the graph it was waiting on actually settles, which is the first moment
2038
+ * the answer exists.
2039
+ *
2040
+ * Doing it from the dispatcher rather than by awaiting in the agent is what keeps the properties
2041
+ * that made detach right in the first place: a graph containing a human approval can park for
2042
+ * days without holding a conversation turn open, and a graph reclaimed by another instance after
2043
+ * a crash still settles its submitting run, because the settling happens wherever the graph
2044
+ * finishes rather than wherever it started.
2045
+ *
2046
+ * **Only a parked run is touched.** A run that is already `Completed`, `Failed` or `Cancelled`
2047
+ * reached that state for its own reasons — a second graph settling later, a run the user
2048
+ * cancelled, a run that failed after submitting — and overwriting it would rewrite history from
2049
+ * the outside. The `Paused` predicate is the whole guard.
2050
+ *
2051
+ * @param graphStatus the parent rollup's status: what the workflow as a whole did
2052
+ */
2053
+ async settleSubmittingRun(provider, parent, graphStatus) {
2054
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
2055
+ if (!meta.submittedByAgentRunID)
2056
+ return; // a scheduled or remote-triggered graph has nobody waiting
2057
+ try {
2058
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
2059
+ if (!(await run.Load(meta.submittedByAgentRunID))) {
2060
+ LogError(`[TaskGraphDispatcher] Could not load run ${meta.submittedByAgentRunID} to settle it against graph ${parent.ID}.`);
2061
+ return;
2062
+ }
2063
+ if (run.Status !== 'Paused')
2064
+ return;
2065
+ // The workflow's outcome becomes the run's outcome. A graph that ended any way other than
2066
+ // Complete did not do what the run started it to do, and a run reporting success over it
2067
+ // would be the same untruth in a different place.
2068
+ const succeeded = graphStatus === 'Complete';
2069
+ run.Status = succeeded ? 'Completed' : 'Failed';
2070
+ run.Success = succeeded;
2071
+ run.CompletedAt = new Date();
2072
+ if (!succeeded) {
2073
+ const reason = `The workflow "${parent.Name}" ended ${graphStatus}.`;
2074
+ run.ErrorMessage = run.ErrorMessage ? `${run.ErrorMessage}\n\n${reason}` : reason;
2075
+ }
2076
+ if (!(await run.Save())) {
2077
+ // Left parked rather than forced. A run stuck at Paused is visibly unfinished, which
2078
+ // is a state someone can investigate; a run flipped to Completed by a write that did
2079
+ // not land would be the same lie this whole change removes.
2080
+ LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}: ` +
2081
+ `${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It remains Paused.`);
2082
+ return;
2083
+ }
2084
+ LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${run.Status} — workflow "${parent.Name}" ended ${graphStatus}.`);
2085
+ }
2086
+ catch (e) {
2087
+ LogError(`[TaskGraphDispatcher] Could not settle the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
2088
+ }
2089
+ }
2090
+ /**
2091
+ * Gives every step that lacks one a position, once the graph has finished.
2092
+ *
2093
+ * **Why the run stores geometry at all.** A `TaskGraphSpec` is a logical structure with no
2094
+ * layout field, so a graph an agent emitted has no opinion about where its boxes go. Every
2095
+ * viewer was therefore laying it out for itself at render time — and a viewer that failed to
2096
+ * (because the canvas measures nodes it has not drawn yet) fell back to every node at the
2097
+ * origin, piled on one another, with the zoom-to-fit that follows fitting a one-node bounding
2098
+ * box. Settling it once, server-side, means the agent-run canvas, the Workflows runs tab and
2099
+ * anything built later all draw the same picture, and none of them has to compute it.
2100
+ *
2101
+ * **An authored position is never overwritten.** A workflow compiled from a Flow agent carries
2102
+ * the arrangement someone dragged into place; replacing it with an algorithm's guess would
2103
+ * discard a deliberate act. Only steps with no geometry get one, so a partially-arranged graph
2104
+ * keeps what it has.
2105
+ *
2106
+ * Failure here is logged and swallowed: this is presentation. A graph whose work completed must
2107
+ * not be reported as failed because its picture could not be saved.
2108
+ */
2109
+ async persistComputedLayout(graph) {
2110
+ try {
2111
+ const needsLayout = [...graph.entityById.values()].filter((t) => !this.parseConfiguration(t)?.layout);
2112
+ if (needsLayout.length === 0)
2113
+ return;
2114
+ // Laid out over the WHOLE graph, not just the nodes missing geometry: position depends on
2115
+ // where a node sits in the topology, and a layout computed over a subset would place its
2116
+ // nodes as though the rest of the workflow did not exist.
2117
+ const edges = graph.edges.map((e) => ({ From: e.dependsOnTaskId, To: e.taskId }));
2118
+ const positions = LayoutGraphNodes([...graph.entityById.keys()], edges, { Direction: 'LR' });
2119
+ for (const task of needsLayout) {
2120
+ const position = positions.get(task.ID);
2121
+ if (!position)
2122
+ continue;
2123
+ const existing = this.parseConfiguration(task);
2124
+ const merged = {
2125
+ ...existing,
2126
+ layout: { x: position.X, y: position.Y },
2127
+ };
2128
+ task.Configuration = JSON.stringify(merged);
2129
+ if (!(await task.Save())) {
2130
+ LogError(`[TaskGraphDispatcher] Could not save computed layout for ${task.ID}: ${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
2131
+ }
2132
+ }
2133
+ }
2134
+ catch (e) {
2135
+ LogError(`[TaskGraphDispatcher] Could not compute a layout for the settled graph: ${e instanceof Error ? e.message : String(e)}`);
2136
+ }
2137
+ }
2138
+ /**
2139
+ * The earliest moment any step in the graph began, or null when none has.
2140
+ *
2141
+ * Null is a real answer — a graph whose tasks are all still Pending has not started — and is
2142
+ * deliberately not collapsed to "now", which would date the graph from whenever this pass
2143
+ * happened to run.
2144
+ */
2145
+ earliestStart(entityById) {
2146
+ let earliest = null;
2147
+ for (const entity of entityById.values()) {
2148
+ if (!entity.StartedAt)
2149
+ continue;
2150
+ if (earliest === null || entity.StartedAt < earliest)
2151
+ earliest = entity.StartedAt;
2152
+ }
2153
+ return earliest;
2154
+ }
2155
+ /**
2156
+ * The step's Configuration with this run's artefacts folded in, or `undefined` to leave it be.
2157
+ *
2158
+ * **Merged into the authored bag, never written over it.** The Configuration column holds the
2159
+ * step's definition — its loop body, its mappings, its policy, the position someone dragged it
2160
+ * to. Writing a fresh object containing only `runtime` would erase all of that the first time a
2161
+ * prompt step completed, which is the kind of loss that surfaces much later as a workflow that
2162
+ * mysteriously stopped mapping its output.
2163
+ *
2164
+ * Returns `undefined` when there is nothing to record, so the guarded write omits the column
2165
+ * rather than rewriting it with what it already held.
2166
+ */
2167
+ configurationWithRuntime(task, promptRunID, actionLogID, iterations, payloadAtStart) {
2168
+ if (!promptRunID && !actionLogID && !iterations?.length && !payloadAtStart)
2169
+ return undefined;
2170
+ const existing = this.parseConfiguration(task);
2171
+ const merged = {
2172
+ ...existing,
2173
+ runtime: {
2174
+ ...existing?.runtime,
2175
+ ...(promptRunID ? { promptRunID } : {}),
2176
+ ...(actionLogID ? { actionLogID } : {}),
2177
+ // Replaced wholesale rather than appended: this is the trace of the loop's LAST
2178
+ // execution, and a retried step that concatenated would report a loop that ran twice
2179
+ // as many passes as it did.
2180
+ ...(iterations?.length ? { iterations } : {}),
2181
+ // The resolved before-state, so the run view has something to diff the output
2182
+ // against. NOT written to Task.InputPayload, which holds the AUTHORED input and
2183
+ // round-trips back out as part of the spec.
2184
+ ...(payloadAtStart ? { payloadAtStart } : {}),
2185
+ },
2186
+ };
2187
+ return JSON.stringify(merged);
2188
+ }
2189
+ /**
2190
+ * Reads a step's Configuration bag, tolerating a row whose JSON cannot be parsed.
2191
+ *
2192
+ * Unparseable configuration is logged rather than thrown: the step has already RUN by the time
2193
+ * this is called, and refusing to record its outcome because its definition is malformed would
2194
+ * discard the result of real work and leave the task claimed until the claim lapsed.
2195
+ */
2196
+ parseConfiguration(task) {
2197
+ if (!task.Configuration)
2198
+ return undefined;
2199
+ try {
2200
+ return JSON.parse(task.Configuration);
2201
+ }
2202
+ catch (e) {
2203
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} has unparseable Configuration; ` +
2204
+ `recording runtime artefacts against an empty bag. ${e instanceof Error ? e.message : String(e)}`);
2205
+ return undefined;
2206
+ }
2207
+ }
2208
+ /**
2209
+ * The payload a step sees: everything its prerequisites produced, plus its own declared input.
2210
+ *
2211
+ * **Why the outputs are merged rather than kept per-task.** A flow carried ONE payload that
2212
+ * accumulated as it went, so a condition on the edge into step C could read a value step A wrote.
2213
+ * Handing each task only its immediate predecessor's output would silently narrow that: the
2214
+ * condition reads `undefined`, which is falsy, and the workflow quietly takes a different route
2215
+ * than the flow it was compiled from. Merging in dependency order restores the accumulation.
2216
+ *
2217
+ * Later prerequisites win on a key collision, matching a flow's own last-write-wins behaviour.
2218
+ */
2219
+ mergedPayload(inputPayload, dependencyOutputs) {
2220
+ const merged = {};
2221
+ for (const output of dependencyOutputs.values()) {
2222
+ if (output && typeof output === 'object' && !Array.isArray(output)) {
2223
+ Object.assign(merged, output);
2224
+ }
2225
+ }
2226
+ if (inputPayload && typeof inputPayload === 'object' && !Array.isArray(inputPayload)) {
2227
+ Object.assign(merged, inputPayload);
2228
+ }
2229
+ return merged;
2230
+ }
2231
+ }
2232
+ //# sourceMappingURL=TaskGraphDispatcher.js.map