@memberjunction/task-graph 6.1.0-edge.1 → 6.1.0-edge.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/README.md +185 -0
  2. package/dist/TaskClaimStore.d.ts +11 -0
  3. package/dist/TaskClaimStore.d.ts.map +1 -1
  4. package/dist/TaskClaimStore.js +8 -3
  5. package/dist/TaskClaimStore.js.map +1 -1
  6. package/dist/TaskGraphDispatcher.d.ts +364 -2
  7. package/dist/TaskGraphDispatcher.d.ts.map +1 -1
  8. package/dist/TaskGraphDispatcher.js +1534 -37
  9. package/dist/TaskGraphDispatcher.js.map +1 -1
  10. package/dist/TaskGraphService.d.ts +110 -1
  11. package/dist/TaskGraphService.d.ts.map +1 -1
  12. package/dist/TaskGraphService.js +458 -19
  13. package/dist/TaskGraphService.js.map +1 -1
  14. package/dist/TaskLoopExecutor.d.ts +62 -0
  15. package/dist/TaskLoopExecutor.d.ts.map +1 -0
  16. package/dist/TaskLoopExecutor.js +248 -0
  17. package/dist/TaskLoopExecutor.js.map +1 -0
  18. package/dist/WorkflowSpecSync.d.ts +28 -2
  19. package/dist/WorkflowSpecSync.d.ts.map +1 -1
  20. package/dist/WorkflowSpecSync.js +83 -2
  21. package/dist/WorkflowSpecSync.js.map +1 -1
  22. package/dist/index.d.ts +1 -0
  23. package/dist/index.d.ts.map +1 -1
  24. package/dist/index.js +1 -0
  25. package/dist/index.js.map +1 -1
  26. package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
  27. package/dist/operations/TaskGraphOperations.js +4 -0
  28. package/dist/operations/TaskGraphOperations.js.map +1 -1
  29. package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
  30. package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
  31. package/dist/operations/WorkflowDraftOperation.js +141 -0
  32. package/dist/operations/WorkflowDraftOperation.js.map +1 -0
  33. package/dist/types.d.ts +114 -0
  34. package/dist/types.d.ts.map +1 -1
  35. package/dist/types.js.map +1 -1
  36. package/package.json +10 -7
@@ -18,11 +18,12 @@
18
18
  *
19
19
  * @module @memberjunction/task-graph
20
20
  */
21
- import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, } from '@memberjunction/ai-core-plus';
21
+ import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
22
22
  import { LogError, LogStatus, RunView } from '@memberjunction/core';
23
- import { ShutdownRegistry } from '@memberjunction/global';
23
+ import { ShutdownRegistry, UUIDsEqual } from '@memberjunction/global';
24
24
  import { TaskClaimStore } from './TaskClaimStore.js';
25
25
  import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
26
+ import { RunForEachLoop, RunWhileLoop } from './TaskLoopExecutor.js';
26
27
  import { NotificationEngine } from '@memberjunction/notifications';
27
28
  /** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
28
29
  const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
@@ -34,8 +35,112 @@ const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
34
35
  * from reclamation, so this value is never mistaken for a live claim.
35
36
  */
36
37
  const HUMAN_TASK_NOTIFIED_MARKER = '__human-notified__';
38
+ /**
39
+ * The run-query capability of a provider, when it has one.
40
+ *
41
+ * `IMetadataProvider` does not extend `IRunQueryProvider`, but every provider that ships implements
42
+ * both. Narrowing by CAPABILITY rather than casting states that honestly: a provider that genuinely
43
+ * cannot run queries returns undefined and the caller reports it, instead of the call failing later
44
+ * behind a type assertion that claimed it could.
45
+ */
46
+ function asRunQueryProvider(provider) {
47
+ const candidate = provider;
48
+ return typeof candidate.RunQuery === 'function' ? candidate : undefined;
49
+ }
37
50
  import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata } from './TaskGraphService.js';
38
51
  import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
52
+ /**
53
+ * Renders a loop's bindings as template values.
54
+ *
55
+ * Template parameters are strings; an item is usually an object. Objects are JSON-encoded rather
56
+ * than dropped, because `{{ field }}` printing `[object Object]` — or nothing at all — is exactly
57
+ * the silent failure this exists to prevent.
58
+ */
59
+ function stringifyBindings(bindings) {
60
+ const out = {};
61
+ for (const [key, value] of Object.entries(bindings)) {
62
+ out[key] = typeof value === 'string' ? value : JSON.stringify(value, null, 2);
63
+ }
64
+ return out;
65
+ }
66
+ /**
67
+ * How much of a loop's per-pass payloads may be kept, and what happens when that runs out.
68
+ *
69
+ * **Why a budget exists at all.** A loop's trace lives inside one `Configuration` column, and its
70
+ * size is the product of two things nobody bounds: how many passes the loop runs, and how large the
71
+ * body's input and output are. A hundred-pass loop over documents would put megabytes in a column
72
+ * that the run tree, the timeline, the canvas and the Workflows list all read — punishing every
73
+ * reader of the row for a detail only someone inspecting one pass will ever open.
74
+ *
75
+ * **What it protects.** Only the payloads. `promptRunID` / `agentRunID` / `actionLogID` / `success`
76
+ * are always recorded: those point at the durable rows where the real forensics live, and they are
77
+ * what cost roll-up and the timeline traverse. Losing a payload costs a reader some detail; losing a
78
+ * pointer would lose the pass.
79
+ *
80
+ * **Omission is stated, never silent.** Once the budget is spent, further passes record a marker
81
+ * saying so and how large the value was, because a pass showing nothing is indistinguishable from a
82
+ * pass that produced nothing — and that ambiguity is exactly the failure this whole area keeps
83
+ * hitting.
84
+ */
85
+ const ITERATION_PAYLOAD_BUDGET_BYTES = 128 * 1024;
86
+ /** Per-value cap, so one enormous pass cannot consume the whole budget by itself. */
87
+ const ITERATION_PAYLOAD_VALUE_BYTES = 16 * 1024;
88
+ class IterationPayloadBudget {
89
+ constructor() {
90
+ this.spent = 0;
91
+ }
92
+ /**
93
+ * The value if it fits, or a marker describing what was left out.
94
+ *
95
+ * @returns the value, a marker object, or undefined when there was nothing to record
96
+ */
97
+ Take(value) {
98
+ if (value == null)
99
+ return undefined;
100
+ const asRecord = value && typeof value === 'object' && !Array.isArray(value)
101
+ ? value
102
+ : { value };
103
+ let size;
104
+ try {
105
+ size = JSON.stringify(asRecord)?.length ?? 0;
106
+ }
107
+ catch {
108
+ // Circular or otherwise unserializable. It could not be persisted anyway, and saying so
109
+ // is better than a pass that silently shows nothing.
110
+ return { __omitted: 'unserializable' };
111
+ }
112
+ if (size > ITERATION_PAYLOAD_VALUE_BYTES) {
113
+ return { __omitted: 'too-large', __bytes: size, __limit: ITERATION_PAYLOAD_VALUE_BYTES };
114
+ }
115
+ if (this.spent + size > ITERATION_PAYLOAD_BUDGET_BYTES) {
116
+ return { __omitted: 'budget-exhausted', __bytes: size, __limit: ITERATION_PAYLOAD_BUDGET_BYTES };
117
+ }
118
+ this.spent += size;
119
+ return asRecord;
120
+ }
121
+ }
122
+ /** Deep-merges a prompt's JSON response into the payload, preserving what earlier steps established. */
123
+ function deepMergePayload(base, incoming) {
124
+ const out = { ...base };
125
+ for (const [key, value] of Object.entries(incoming)) {
126
+ const existing = out[key];
127
+ const bothPlainObjects = existing && typeof existing === 'object' && !Array.isArray(existing) &&
128
+ value && typeof value === 'object' && !Array.isArray(value);
129
+ out[key] = bothPlainObjects
130
+ ? deepMergePayload(existing, value)
131
+ : value;
132
+ }
133
+ return out;
134
+ }
135
+ /**
136
+ * Statuses at which an origin's outgoing conditions may be decided.
137
+ *
138
+ * `Skipped` is included: a branch that was not taken IS settled, and a condition on an edge leaving
139
+ * it should resolve rather than hang the graph forever.
140
+ */
141
+ const TERMINAL_FOR_CONDITIONS = new Set([
142
+ 'Complete', 'Failed', 'Cancelled', 'Skipped',
143
+ ]);
39
144
  export class TaskGraphDispatcher {
40
145
  constructor(providerFactory, agentRunner, contextUser, config,
41
146
  /**
@@ -48,12 +153,24 @@ export class TaskGraphDispatcher {
48
153
  * Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
49
154
  * announces nothing.
50
155
  */
51
- observer) {
156
+ observer,
157
+ /**
158
+ * Optional. Absent means this host cannot run action nodes; they stay Pending and visible
159
+ * rather than being failed, because "nobody here can run this" is not "this ran and broke".
160
+ */
161
+ actionRunner,
162
+ /**
163
+ * Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
164
+ * rather than being failed, for the same reason action nodes do.
165
+ */
166
+ promptRunner) {
52
167
  this.providerFactory = providerFactory;
53
168
  this.agentRunner = agentRunner;
54
169
  this.contextUser = contextUser;
55
170
  this.continuationDeliverer = continuationDeliverer;
56
171
  this.observer = observer;
172
+ this.actionRunner = actionRunner;
173
+ this.promptRunner = promptRunner;
57
174
  this.running = false;
58
175
  this.pollTimer = null;
59
176
  this.reconcileTimer = null;
@@ -61,6 +178,15 @@ export class TaskGraphDispatcher {
61
178
  this.inFlight = new Set();
62
179
  /** Guards against a slow poll overlapping the next tick. */
63
180
  this.polling = false;
181
+ /**
182
+ * The poll pass currently running, so `Stop` can wait for it.
183
+ *
184
+ * `clearInterval` cannot cancel a tick that has already fired, and a pass is a long sequence of
185
+ * awaits (provider, rollup, claim query) — so without this, `Stop` returns while a pass is still
186
+ * mid-flight and about to claim. Its tasks then land in `inFlight` AFTER the drain loop already
187
+ * saw an empty set, which is precisely the state the drain exists to prevent.
188
+ */
189
+ this.pollPass = null;
64
190
  /** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
65
191
  this.ownerByParentID = new Map();
66
192
  /** Name shown in the shutdown drain log. */
@@ -134,7 +260,7 @@ export class TaskGraphDispatcher {
134
260
  ShutdownRegistry.Instance.Register(this);
135
261
  LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
136
262
  await this.Reconcile();
137
- this.pollTimer = setInterval(() => { void this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
263
+ this.pollTimer = setInterval(() => { this.pollPass = this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
138
264
  this.reconcileTimer = setInterval(() => { void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
139
265
  }
140
266
  /**
@@ -154,6 +280,14 @@ export class TaskGraphDispatcher {
154
280
  clearInterval(this.reconcileTimer);
155
281
  this.reconcileTimer = null;
156
282
  }
283
+ // Drain the poll pass BEFORE the task drain below, not after: a pass still running has not
284
+ // necessarily claimed anything yet, so `inFlight` can be empty while work is moments from
285
+ // starting. Clearing `running` above stops that pass claiming anything further; this waits
286
+ // for it to notice. Its own failures are already logged inside `pollOnce`.
287
+ if (this.pollPass) {
288
+ await this.pollPass.catch(() => undefined);
289
+ this.pollPass = null;
290
+ }
157
291
  const deadline = Date.now() + 30_000;
158
292
  while (this.inFlight.size > 0 && Date.now() < deadline) {
159
293
  await new Promise((r) => setTimeout(r, 250));
@@ -204,11 +338,25 @@ export class TaskGraphDispatcher {
204
338
  this.polling = true;
205
339
  try {
206
340
  const provider = await this.providerFactory.CreateProvider();
341
+ // `running` is re-read after every await from here on. The entry check above only proves
342
+ // the dispatcher was live when the tick fired; each await is a point where `Stop` can
343
+ // land, and a stopped instance must neither mutate graph state nor take new work. Left
344
+ // unchecked, a stopped dispatcher goes on to roll up graphs (emitting GraphSettled to an
345
+ // observer nobody is listening to any more) and to claim tasks it will never run — which
346
+ // then sit claimed until their lease expires.
347
+ if (!this.running)
348
+ return;
207
349
  // Settle graphs before picking new work, so a failure earlier in this pass stops its
208
350
  // branch immediately rather than after another wave has already launched.
209
351
  await this.propagateAndRollup(provider);
352
+ if (!this.running)
353
+ return;
210
354
  const candidates = await this.findClaimableTasks(provider, capacity);
211
355
  for (const task of candidates) {
356
+ // Re-checked per iteration, not just before the loop: claiming is itself awaited, so
357
+ // a multi-task wave can straddle a Stop.
358
+ if (!this.running)
359
+ break;
212
360
  if (this.inFlight.size >= this.config.MaxConcurrentTasks)
213
361
  break;
214
362
  if (!(await this.claims.TryClaim(provider, task.ID, this.contextUser))) {
@@ -265,19 +413,19 @@ export class TaskGraphDispatcher {
265
413
  LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
266
414
  }
267
415
  }
268
- const result = await this.agentRunner.RunAgentForTask({
269
- TaskID: taskID,
270
- AgentID: task.AgentID,
271
- InputPayload: inputPayload,
272
- DependencyOutputs: dependencyOutputs,
273
- Provider: provider,
274
- ContextUser: this.contextUser,
275
- });
416
+ const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs);
417
+ // A prompt can end the workflow early and say why. Honour it before recording the
418
+ // outcome, so the remaining tasks are already Skipped by the time the rollup runs and
419
+ // the graph settles Complete rather than looking abandoned with work left Pending.
420
+ if (result.ChatMessage) {
421
+ await this.endGraphEarly(provider, task, result.ChatMessage);
422
+ }
276
423
  const recorded = await this.claims.CompleteClaimed(provider, taskID, {
277
424
  Status: result.Success ? 'Complete' : 'Failed',
278
425
  OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
279
426
  ErrorMessage: result.ErrorMessage ?? null,
280
427
  AgentRunID: result.AgentRunID ?? null,
428
+ Configuration: this.configurationWithRuntime(task, result.PromptRunID, result.ActionLogID, result.Iterations, result.PayloadAtStart),
281
429
  }, this.contextUser);
282
430
  if (!recorded) {
283
431
  // The guarded write refused: the row changed underneath us (cancelled, reassigned,
@@ -320,10 +468,62 @@ export class TaskGraphDispatcher {
320
468
  */
321
469
  async propagateAndRollup(provider) {
322
470
  for (const parentID of await this.findActiveGraphIDs(provider)) {
471
+ // Human steps settle BEFORE the graph state is read, so an answer given since the last
472
+ // poll is already reflected when eligibility and rollup are computed. Doing it after
473
+ // would delay every dependent branch by a full poll interval for no reason — and on a
474
+ // graph whose only remaining work is downstream of a person, that is the difference
475
+ // between "answered and moving" and "answered and apparently still stuck".
476
+ await this.expireOverdueRequests(provider, parentID);
477
+ await this.settleAnsweredHumanTasks(provider, parentID);
478
+ await this.reopenCancelledHumanTasks(provider, parentID);
323
479
  const graph = await this.loadGraphState(provider, parentID);
324
480
  if (graph.nodes.length === 0)
325
481
  continue;
326
- const toBlock = new Set([...ComputeTasksToBlock(graph.nodes, graph.edges), ...graph.unreachableTaskIDs]);
482
+ // SKIPS FIRST — before blocking, before eligibility. A task whose gating predecessors
483
+ // are all Skipped is simultaneously "eligible" (Skipped satisfies a prerequisite) and
484
+ // "to be skipped"; deciding eligibility first would dispatch the branch nobody took.
485
+ //
486
+ // `unreachableTaskIDs` seeds this too, and that is a correction (R6). A target whose only
487
+ // route in was an ordinary conditional edge that evaluated DEFINITELY FALSE is a branch
488
+ // that was not taken — semantically identical to an XOR loser — yet it used to settle
489
+ // `Blocked`. That made `Blocked` mean two unrelated things: "the workflow chose another
490
+ // route" and "something upstream broke". A reader cannot tell those apart, so every
491
+ // conditional workflow looked half-failed and people went hunting for bugs that did not
492
+ // exist. `Blocked` is now reserved for FAILURE-driven unsatisfiability.
493
+ const skipSeeds = new Set([...graph.skipSeedTaskIDs, ...graph.unreachableTaskIDs]);
494
+ const toSkip = new Set([
495
+ ...skipSeeds,
496
+ ...ComputeSkipCascade(graph.nodes, graph.edges, [...skipSeeds]),
497
+ ]);
498
+ for (const taskID of toSkip) {
499
+ const entity = graph.entityById.get(taskID);
500
+ if (!entity || entity.Status !== 'Pending')
501
+ continue;
502
+ entity.Status = 'Skipped';
503
+ if (await entity.Save()) {
504
+ LogStatus(`[TaskGraphDispatcher] Skipped '${entity.Name}' (${taskID}) — another branch was taken.`);
505
+ // Announced separately from TaskBlocked because it means something different to
506
+ // a viewer: nothing went wrong, this route simply was not the one chosen.
507
+ this.emit({
508
+ Kind: 'TaskSkipped',
509
+ ParentTaskID: parentID,
510
+ OwnerUserID: await this.resolveOwner(provider, parentID),
511
+ TaskID: taskID,
512
+ TaskName: entity.Name,
513
+ Status: 'Skipped',
514
+ });
515
+ // Keep the in-memory graph consistent so the blocking pass below and the rollup
516
+ // both see the skip rather than a stale Pending.
517
+ const node = graph.nodes.find((n) => n.id === taskID);
518
+ if (node)
519
+ node.status = 'Skipped';
520
+ }
521
+ }
522
+ // Only failure-driven unsatisfiability reaches here now; not-taken branches were skipped
523
+ // above. A task already Skipped is left alone rather than overwritten — the two passes
524
+ // must not fight over the same row.
525
+ const toBlock = [...ComputeTasksToBlock(graph.nodes, graph.edges, graph.handledFailureIDs)]
526
+ .filter((id) => !toSkip.has(id));
327
527
  for (const taskID of toBlock) {
328
528
  const entity = graph.entityById.get(taskID);
329
529
  if (!entity)
@@ -353,11 +553,26 @@ export class TaskGraphDispatcher {
353
553
  // The outer guard covered the first load only.
354
554
  if (fresh.nodes.length === 0)
355
555
  continue;
356
- const rollup = ComputeParentRollup(fresh.nodes);
556
+ const rollup = ComputeParentRollup(fresh.nodes, fresh.handledFailureIDs);
357
557
  const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
358
558
  if (!(await parent.Load(parentID)))
359
559
  continue;
360
- if (parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
560
+ // A graph starts when its first step does.
561
+ //
562
+ // `StartedAt` is stamped by the CLAIM, and a parent is never claimed — it is a container,
563
+ // not a unit of work — so the graph row carried no start time even after it completed.
564
+ // A settled workflow therefore reported a CompletedAt with no beginning: it sorted as
565
+ // "not started" in the run tree, showed no timestamp, and no duration could be computed
566
+ // for the thing whose duration people actually ask about.
567
+ //
568
+ // Taken from the earliest child rather than from the clock, because that is when work
569
+ // genuinely began — a graph can sit Pending for a long time between submission (already
570
+ // recorded as CreatedAt) and a dispatcher picking up its first task.
571
+ const earliestChildStart = this.earliestStart(fresh.entityById);
572
+ const startedAtChanged = parent.StartedAt == null && earliestChildStart != null;
573
+ if (startedAtChanged)
574
+ parent.StartedAt = earliestChildStart;
575
+ if (startedAtChanged || parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
361
576
  parent.Status = rollup.status;
362
577
  parent.PercentComplete = rollup.percentComplete;
363
578
  if (rollup.isTerminal)
@@ -365,6 +580,8 @@ export class TaskGraphDispatcher {
365
580
  await parent.Save();
366
581
  }
367
582
  if (rollup.isTerminal) {
583
+ // Geometry is settled once, here, so every viewer of this run agrees on it.
584
+ await this.persistComputedLayout(fresh);
368
585
  // Emitted before the continuation is delivered, and outside its once-only guard: a
369
586
  // viewer watching the run should learn it finished whether or not this instance is
370
587
  // the one that wins the delivery CAS.
@@ -376,10 +593,295 @@ export class TaskGraphDispatcher {
376
593
  CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
377
594
  TotalCount: fresh.nodes.length,
378
595
  });
596
+ await this.rollUpCostToSubmittingRun(provider, parent);
597
+ // Deliberately AFTER the rollup and OUTSIDE its refusal paths. The rollup declines
598
+ // to write a number it cannot stand behind — a truncated tree, an unreachable graph
599
+ // — and every one of those returns early. If the run's lifecycle were settled in
600
+ // there, a refused rollup would strand the run parked forever, which is a far worse
601
+ // failure than a missing cost figure. Cost and lifecycle are separate concerns with
602
+ // separate failure modes, so they get separate writes.
603
+ await this.settleSubmittingRun(provider, parent, rollup.status);
379
604
  await this.deliverContinuation(provider, parent, fresh);
380
605
  }
381
606
  }
382
607
  }
608
+ /**
609
+ * Credits a finished graph's spending back to the agent run that submitted it.
610
+ *
611
+ * **Why this cannot happen during the run.** `BaseAgent` totals a run by walking its steps in
612
+ * memory at finalization — but a submitting run *ends at submission*. Submit-and-detach is the
613
+ * point: the run returns as soon as the graph is durable, and the graph executes afterwards,
614
+ * possibly minutes later on a different instance. At the moment the run computes its totals the
615
+ * spending has not happened yet, so there is nothing to count. The only place the number can be
616
+ * known is here, when the graph settles.
617
+ *
618
+ * **Why the `…Rollup` columns and not the plain ones.** `AIAgentRun` has carried six `…Rollup`
619
+ * columns since v3 that nothing has ever written — they exist for exactly this distinction:
620
+ *
621
+ * - `TotalCost` — what the run itself spent. For a Flow agent that is genuinely near zero: it
622
+ * compiled a graph and handed it off. This value is already final and is never rewritten here,
623
+ * so nothing that reads it today changes meaning, and no guardrail that already evaluated
624
+ * against it is retroactively falsified.
625
+ * - `TotalCostRollup` — the run plus everything it caused. Provisional until the graph settles,
626
+ * which is now.
627
+ *
628
+ * **The tree is the authority; these columns are its settlement-time cache.** The total is a SUM
629
+ * over `GetAgentRunTree`, not arithmetic of its own. The previous version walked the graph's
630
+ * child tasks and added each one's agent run, which was wrong in two ways that no test could
631
+ * see: a `Prompt` task has no agent run at all, so every prompt step's spend was simply missing;
632
+ * and it read each nested run's `…Rollup ?? …Total`, mixing a descendant-inclusive number with an
633
+ * own-spend one and depending on whether that nested graph happened to have settled yet. The
634
+ * tree already models every one of those cases — it reaches prompt runs through
635
+ * `Configuration.runtime.promptRunID`, and it descends into nested runs and their graphs
636
+ * structurally — so summing it cannot disagree with what the run viewer shows, because it IS
637
+ * what the run viewer shows.
638
+ *
639
+ * **This refuses rather than guesses.** A tree that failed to load, hit the depth cap, or does
640
+ * not contain the settling graph would still produce a number — a lower bound. Writing one would
641
+ * put an authoritative-looking total in a column every cost surface reads. Each of those cases
642
+ * logs and leaves the column alone, so `?? TotalCost` keeps its honest meaning: not settled.
643
+ *
644
+ * A graph with no submitting run (a scheduled job, a remote-operation caller) simply has nobody
645
+ * to credit — its own Task rows still carry the truth, and this returns quietly.
646
+ */
647
+ async rollUpCostToSubmittingRun(provider, parent) {
648
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
649
+ if (!meta.submittedByAgentRunID)
650
+ return;
651
+ const runID = meta.submittedByAgentRunID;
652
+ try {
653
+ const runQuery = asRunQueryProvider(provider);
654
+ if (!runQuery) {
655
+ LogError(`[TaskGraphDispatcher] Cannot roll up cost for run ${runID}: provider cannot run queries.`);
656
+ return;
657
+ }
658
+ const tree = await LoadAgentRunTree(runID, runQuery, this.contextUser);
659
+ // Each of these means the sum would be a LOWER BOUND, and the column's whole contract is
660
+ // that it equals the tree. A known-low number presented as a total is worse than no
661
+ // number: the readers all fall back to TotalCost when this is null, which at least
662
+ // *says* it is the run's own spend rather than claiming to be the whole story.
663
+ //
664
+ // Refusing is NOT the same as leaving the column alone. A run that submitted two graphs
665
+ // has a rollup from the first; if the second cannot be summed, the first graph's total
666
+ // sits in the authoritative column excluding work that has since happened — stale, not
667
+ // absent, and `?? TotalCost` cannot save a reader from a non-null wrong number. So a
668
+ // refusal CLEARS it, restoring the fallback's honest meaning: not settled.
669
+ if (tree.ErrorMessage || !tree.Root) {
670
+ await this.clearStaleRollup(provider, runID, tree.ErrorMessage ?? 'the run tree came back empty');
671
+ return;
672
+ }
673
+ if (tree.Truncated) {
674
+ await this.clearStaleRollup(provider, runID, `the run tree hit the depth cap, so any total would silently under-report ` +
675
+ `(graph ${parent.ID} still carries its own costs)`);
676
+ return;
677
+ }
678
+ // The graph that just settled must appear in the tree. If it does not, the tree stopped
679
+ // at the run — the submitting step never recorded its parentTaskID — and the sum is
680
+ // merely the run's own spend wearing the name of a rollup. That is precisely the silent
681
+ // under-count this rewrite exists to remove, so it is reported rather than written.
682
+ if (!this.treeContainsGraph(tree.Root, parent.ID)) {
683
+ await this.clearStaleRollup(provider, runID, `graph ${parent.ID} is not reachable from it, so the tree cannot see the work. ` +
684
+ `Did the submitting step record parentTaskID?`);
685
+ return;
686
+ }
687
+ const totals = SumAgentRunTreeCost(tree.Root);
688
+ const submitting = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
689
+ if (!(await submitting.Load(runID))) {
690
+ LogError(`[TaskGraphDispatcher] Could not load run ${runID} to record graph cost against it.`);
691
+ return;
692
+ }
693
+ // Assignment, never accumulation. The tree already contains the run's own spend as its
694
+ // ROOT node, and it reads own-cost everywhere, so recomputing from scratch on every
695
+ // settlement lands on the same answer — which is what makes this safe to call again when
696
+ // a second graph settles, or when the terminal check is re-evaluated after a HITL wait.
697
+ submitting.TotalCostRollup = totals.Cost;
698
+ submitting.TotalTokensUsedRollup = totals.Tokens;
699
+ submitting.TotalPromptTokensUsedRollup = totals.PromptTokens;
700
+ submitting.TotalCompletionTokensUsedRollup = totals.CompletionTokens;
701
+ if (!(await submitting.Save())) {
702
+ LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}: ` +
703
+ `${submitting.LatestResult?.CompleteMessage ?? 'unknown error'}`);
704
+ return;
705
+ }
706
+ LogStatus(`[TaskGraphDispatcher] Credited graph ${parent.ID} to run ${runID}: ` +
707
+ `${tree.Rows.length} node(s), ${totals.Tokens} token(s), cost ${totals.Cost}.`);
708
+ }
709
+ catch (e) {
710
+ // A failed rollup must never fail the graph. The work finished; only the accounting for
711
+ // it is missing, and a graph marked Failed because its cost could not be summed would be
712
+ // a far worse lie than a cost of null.
713
+ LogError(`[TaskGraphDispatcher] Cost rollup failed for graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
714
+ }
715
+ }
716
+ /**
717
+ * Clears a rollup that can no longer be trusted, and says why.
718
+ *
719
+ * **Why clear rather than leave.** The four `…Rollup` columns are a cache of the run tree, and
720
+ * every reader treats a value there as the total. When the tree cannot be summed, any value
721
+ * already in the column was computed from an EARLIER settlement — it excludes the graph that
722
+ * just finished, so it is not merely incomplete, it is a wrong total presented as a right one.
723
+ * `?? TotalCost` protects a reader from null, not from stale.
724
+ *
725
+ * Nulling restores the invariant this whole design rests on: **when the column is present, it
726
+ * equals the tree.** Absent means not settled, which is exactly what a reader should conclude.
727
+ * A run with no rollup yet is untouched — there is nothing stale to clear, and writing nulls
728
+ * over nulls would churn Record Changes for nothing.
729
+ */
730
+ async clearStaleRollup(provider, runID, reason) {
731
+ LogError(`[TaskGraphDispatcher] Not recording cost for run ${runID}: ${reason}.`);
732
+ try {
733
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
734
+ if (!(await run.Load(runID)))
735
+ return;
736
+ if (run.TotalCostRollup == null && run.TotalTokensUsedRollup == null)
737
+ return; // nothing stale
738
+ run.TotalCostRollup = null;
739
+ run.TotalTokensUsedRollup = null;
740
+ run.TotalPromptTokensUsedRollup = null;
741
+ run.TotalCompletionTokensUsedRollup = null;
742
+ if (!(await run.Save())) {
743
+ LogError(`[TaskGraphDispatcher] Could not clear the now-stale rollup on run ${runID}: ` +
744
+ `${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It still shows a total that ` +
745
+ `excludes the graph that just settled.`);
746
+ return;
747
+ }
748
+ LogStatus(`[TaskGraphDispatcher] Cleared the rollup on run ${runID}: it was computed before this ` +
749
+ `graph settled and can no longer be recomputed, so it would have under-reported.`);
750
+ }
751
+ catch (e) {
752
+ LogError(`[TaskGraphDispatcher] Could not clear the rollup on run ${runID}: ${e instanceof Error ? e.message : String(e)}`);
753
+ }
754
+ }
755
+ /**
756
+ * Whether the settling graph is actually reachable from the submitting run's tree.
757
+ *
758
+ * Matched on the graph's parent Task id, which is the node the `TaskGraph` member of the query
759
+ * emits. A run that submitted a graph but recorded no `parentTaskID` produces a tree that stops
760
+ * at the run — structurally indistinguishable, at the SUM, from a run that never dispatched
761
+ * anything. This is the check that tells those two apart.
762
+ */
763
+ treeContainsGraph(root, parentTaskID) {
764
+ for (const node of WalkAgentRunTree(root)) {
765
+ if (node.NodeType === 'TaskGraph' && UUIDsEqual(node.NodeID, parentTaskID))
766
+ return true;
767
+ }
768
+ return false;
769
+ }
770
+ /**
771
+ * Ends a graph early because a prompt said the work is finished.
772
+ *
773
+ * **Why `Skipped` and not `Cancelled`.** Nothing went wrong and nobody intervened — the workflow
774
+ * reached its own conclusion before running every drawn step, which is exactly what a reasoning
775
+ * step is for. `Cancelled` would tell a reader someone stopped it; `Skipped` says these routes
776
+ * were not taken, which is true and already the vocabulary the fork machinery uses.
777
+ *
778
+ * The message is written to the parent so the graph carries its own answer, rather than the
779
+ * answer living only on the step that produced it.
780
+ */
781
+ async endGraphEarly(provider, task, message) {
782
+ if (!task.ParentID)
783
+ return;
784
+ try {
785
+ LogStatus(`[TaskGraphDispatcher] '${task.Name}' ended the workflow early: ${message}`);
786
+ for (const sibling of await this.loadChildTasks(provider, task.ParentID)) {
787
+ if (sibling.ID === task.ID || sibling.Status !== 'Pending')
788
+ continue;
789
+ sibling.Status = 'Skipped';
790
+ if (await sibling.Save()) {
791
+ this.emit({
792
+ Kind: 'TaskSkipped',
793
+ ParentTaskID: task.ParentID,
794
+ OwnerUserID: await this.resolveOwner(provider, task.ParentID),
795
+ TaskID: sibling.ID,
796
+ TaskName: sibling.Name,
797
+ Status: 'Skipped',
798
+ });
799
+ }
800
+ }
801
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
802
+ if (await parent.Load(task.ParentID)) {
803
+ parent.OutputPayload = JSON.stringify({ message });
804
+ await parent.Save();
805
+ }
806
+ }
807
+ catch (e) {
808
+ // The work itself succeeded; only the early-finish bookkeeping failed. Failing the task
809
+ // over that would discard a completed step's result.
810
+ LogError(`[TaskGraphDispatcher] Could not end graph early for ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
811
+ }
812
+ }
813
+ /**
814
+ * How deep the continuation chain already is, read from the graph's parent metadata.
815
+ *
816
+ * A run started by a graph inherits that graph's depth **plus one**. Without this every spawned
817
+ * run begins at zero, so a self-referencing flow — one that dispatches a graph containing itself
818
+ * — recurses without bound while the cap it should be hitting compares against a permanent zero.
819
+ */
820
+ async graphContext(provider, task) {
821
+ if (!task.ParentID)
822
+ return { Depth: 0, SubmittingAgentRunID: null };
823
+ try {
824
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
825
+ if (!(await parent.Load(task.ParentID)))
826
+ return { Depth: 0, SubmittingAgentRunID: null };
827
+ return {
828
+ Depth: ParseTaskGraphParentMetadata(parent.InputPayload).reinvokeDepth + 1,
829
+ // The graph's own row carries the run that submitted it. One load answers both
830
+ // questions, which is why they are resolved together rather than in two passes.
831
+ SubmittingAgentRunID: parent.AgentRunID,
832
+ };
833
+ }
834
+ catch {
835
+ // An unreadable parent must not stop the work; depth zero is the safe reading, and the
836
+ // submit-time cap still guards the next hop.
837
+ return { Depth: 0, SubmittingAgentRunID: null };
838
+ }
839
+ }
840
+ /**
841
+ * Which failures the workflow drew a way out of.
842
+ *
843
+ * A Failed task with a **satisfied outgoing edge** is a handled failure: its author drew a
844
+ * recovery route and that route is now live. Downstream work should be released along it, and the
845
+ * parent should not roll up Failed because of a step the workflow explicitly planned around.
846
+ *
847
+ * Scoped to `failureSemantics: 'edges'` on purpose. Under `'block'` — every agent-emitted graph —
848
+ * a failure is terminal for its dependents whatever edges exist, because nobody drew those edges
849
+ * as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
850
+ * graph sail past a failure it never anticipated.
851
+ */
852
+ async computeHandledFailures(provider, parentTaskID, nodes, edges) {
853
+ const handled = new Set();
854
+ // Cheap exit before touching the database: with no failures there is nothing to handle, and
855
+ // this runs on every poll for every active graph.
856
+ if (!nodes.some((n) => n.status === 'Failed'))
857
+ return handled;
858
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
859
+ if (!(await parent.Load(parentTaskID)))
860
+ return handled;
861
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
862
+ if (meta.failureSemantics !== 'edges')
863
+ return handled;
864
+ for (const node of nodes) {
865
+ if (node.status !== 'Failed')
866
+ continue;
867
+ // "Has somewhere to go" is the test. An edge out of a failed step that survived condition
868
+ // evaluation IS the drawn recovery route; a failed step with no outgoing edges has none,
869
+ // and stays terminal.
870
+ if (edges.some((e) => e.dependsOnTaskId === node.id))
871
+ handled.add(node.id);
872
+ }
873
+ return handled;
874
+ }
875
+ /** The graph's child tasks, with the fields the rollup needs. */
876
+ async loadChildTasks(provider, parentID) {
877
+ const result = await RunView.FromMetadataProvider(provider).RunView({
878
+ EntityName: 'MJ: Tasks',
879
+ ExtraFilter: `ParentID='${parentID}'`,
880
+ ResultType: 'entity_object',
881
+ BypassCache: true,
882
+ }, this.contextUser);
883
+ return (result.Success ? result.Results : []) ?? [];
884
+ }
383
885
  /**
384
886
  * Runs the graph's continuation exactly once, now that it has settled.
385
887
  *
@@ -498,8 +1000,29 @@ export class TaskGraphDispatcher {
498
1000
  async notifyHumanTaskReady(task, provider) {
499
1001
  if (task.ClaimedBy === HUMAN_TASK_NOTIFIED_MARKER)
500
1002
  return;
501
- if (!task.UserID)
502
- return; // unassigned human task — nobody to tell
1003
+ // The REQUEST is raised whether or not the task names an assignee. An unassigned human step
1004
+ // is a legitimate "somebody needs to look at this", and a request nobody was notified about
1005
+ // is still findable in the inbox — whereas returning early here is how such a step used to
1006
+ // become invisible work that stalled a workflow with nothing anywhere saying why.
1007
+ // TRANSIENT failures retry; PERMANENT ones stop. That distinction is the whole point, and
1008
+ // getting it wrong took a server down: retrying unconditionally meant a task whose workflow
1009
+ // has no owning agent — which can never succeed — was re-attempted on every poll forever,
1010
+ // each pass re-reading the graph, until the process was OOM-killed. The marker exists to
1011
+ // prevent exactly that storm; a permanent failure has to set it.
1012
+ const raised = await this.raiseHumanRequest(task, provider);
1013
+ if (raised === 'transient-failure')
1014
+ return; // try again next poll
1015
+ if (raised === 'permanent-failure') {
1016
+ // Nothing will change on a retry. Mark it so the loop stops, and leave the task Pending
1017
+ // and visible — a person can still see it in the Tasks UI, which is the fallback the
1018
+ // notification was only ever an accelerant for.
1019
+ await this.markHumanTaskNotified(task);
1020
+ return;
1021
+ }
1022
+ if (!task.UserID) {
1023
+ await this.markHumanTaskNotified(task);
1024
+ return;
1025
+ }
503
1026
  try {
504
1027
  await NotificationEngine.Instance.Config(false, this.contextUser);
505
1028
  await NotificationEngine.Instance.SendNotification({
@@ -513,13 +1036,7 @@ export class TaskGraphDispatcher {
513
1036
  catch (e) {
514
1037
  LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
515
1038
  }
516
- // Marked even when delivery threw. Retrying a notification on every five-second poll is a
517
- // worse failure than one that was missed: the task remains visible in the Tasks UI either
518
- // way, whereas a notification storm is not self-correcting.
519
- task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
520
- if (!(await task.Save())) {
521
- LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
522
- }
1039
+ await this.markHumanTaskNotified(task);
523
1040
  // Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
524
1041
  // than appearing to stall for no reason.
525
1042
  this.emit({
@@ -598,7 +1115,28 @@ export class TaskGraphDispatcher {
598
1115
  if (claimable.length >= limit)
599
1116
  break;
600
1117
  const graph = await this.loadGraphState(provider, parentID);
601
- for (const node of ComputeEligibleTasks(graph.nodes, graph.edges)) {
1118
+ // HOLD is what makes "a broken condition stalls visibly" true rather than merely stated.
1119
+ // An undecided exclusive group keeps all its edges, and a kept edge on a Complete origin
1120
+ // is a SATISFIED prerequisite — so without this filter every branch of the fork would be
1121
+ // eligible at once and all of them would run. A typo must not multiply a fork.
1122
+ //
1123
+ // The losers of a DECIDED group must be filtered for the same reason, and this is a race
1124
+ // rather than a rule: they are marked Skipped by the propagation pass, but between the
1125
+ // moment the group resolves and the moment that write lands, their incoming edge is still
1126
+ // a satisfied prerequisite on a Complete origin. A poll landing in that window would
1127
+ // claim and execute the branch the workflow chose NOT to take — irreversibly, since the
1128
+ // action has already run by the time Skipped is written over it.
1129
+ // `unreachableTaskIDs` joins the filter for exactly the reason above. R6 made a
1130
+ // definite-false ordinary edge seed the skip cascade rather than Block its target — but
1131
+ // until that Skipped write lands, the target has no unsatisfied prerequisite and is
1132
+ // vacuously eligible. That is the same race the XOR fix closed, reopened on the new
1133
+ // path: a branch the workflow decided against, claimed and executed irreversibly in the
1134
+ // window before it was marked.
1135
+ const eligible = ComputeEligibleTasks(graph.nodes, graph.edges, graph.handledFailureIDs)
1136
+ .filter((n) => !graph.holdTaskIDs.has(n.id) &&
1137
+ !graph.skipSeedTaskIDs.has(n.id) &&
1138
+ !graph.unreachableTaskIDs.has(n.id));
1139
+ for (const node of eligible) {
602
1140
  const entity = graph.entityById.get(node.id);
603
1141
  if (!entity)
604
1142
  continue;
@@ -608,7 +1146,25 @@ export class TaskGraphDispatcher {
608
1146
  // they cleared. Without a notification here a workflow simply stops, waiting on
609
1147
  // someone who was never told. That silent stall is the failure mode this exists to
610
1148
  // prevent, so it happens on the eligibility check rather than at submission.
611
- if (!entity.AgentID) {
1149
+ if (entity.ActionID) {
1150
+ // An action node this host has no runner for is left Pending rather than
1151
+ // claimed. Claiming it would take ownership of work this process cannot do, and
1152
+ // the claim would then have to expire before any host that CAN do it gets a
1153
+ // turn — a self-inflicted stall on a mixed deployment.
1154
+ if (!this.actionRunner)
1155
+ continue;
1156
+ }
1157
+ else if (entity.PromptID) {
1158
+ // A prompt node — including a loop that repeats a prompt — is assigned through
1159
+ // PromptID and carries NEITHER ActionID nor AgentID. Without this branch it fell
1160
+ // through to the test below and was treated as a task waiting on a PERSON: the
1161
+ // workflow notified a human who had nothing to do and then stopped forever.
1162
+ // That is precisely the misclassification the step-kind rules warn about, and it
1163
+ // is silent — the graph sits In Progress looking like it is still working.
1164
+ if (!this.promptRunner)
1165
+ continue;
1166
+ }
1167
+ else if (!entity.AgentID) {
612
1168
  await this.notifyHumanTaskReady(entity, provider);
613
1169
  continue;
614
1170
  }
@@ -621,6 +1177,278 @@ export class TaskGraphDispatcher {
621
1177
  }
622
1178
  return claimable;
623
1179
  }
1180
+ /**
1181
+ * Marks a human task as notified, so the request is raised exactly once.
1182
+ *
1183
+ * Written even when delivery threw. Retrying on every poll is a worse failure than one missed
1184
+ * notification: the task stays visible in the inbox either way, whereas a notification storm is
1185
+ * not self-correcting.
1186
+ */
1187
+ async markHumanTaskNotified(task) {
1188
+ task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
1189
+ if (!(await task.Save())) {
1190
+ LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
1191
+ }
1192
+ }
1193
+ /**
1194
+ * Raises the `MJ: AI Agent Requests` row a person answers to release this step.
1195
+ *
1196
+ * **Why that entity rather than something new.** It already models everything a workflow's human
1197
+ * step needs — who is being asked, what for, a typed response schema, priority, expiry, and an
1198
+ * inbox surface people already use. A second HITL substrate beside it would split the inbox in
1199
+ * two and leave one of them without expiry or permissions.
1200
+ *
1201
+ * **What it deliberately does NOT set is `ResumingAgentRunID`.** A request normally suspends an
1202
+ * agent run and resumes it. A workflow needs none of that: the graph OUTLIVES the run that
1203
+ * submitted it, so nothing is suspended — the task sits Pending, every other branch keeps
1204
+ * running, and answering settles the task. That column staying null is meaningful, not missing.
1205
+ */
1206
+ async raiseHumanRequest(task, provider) {
1207
+ try {
1208
+ const existing = await this.findOpenRequest(provider, task.ID);
1209
+ if (existing)
1210
+ return 'raised'; // already waiting on someone
1211
+ const request = await provider.GetEntityObject('MJ: AI Agent Requests', this.contextUser);
1212
+ request.NewRecord();
1213
+ request.OriginatingTaskID = task.ID;
1214
+ // A human task has NO AgentID of its own — that column names what EXECUTES a step, and
1215
+ // a person is not an agent. The request still needs one, so it carries the agent that
1216
+ // owns the workflow: the graph's own agent, which is who is asking.
1217
+ const owningAgentID = await this.owningAgentOf(provider, task);
1218
+ if (!owningAgentID) {
1219
+ // PERMANENT: a graph with no owning agent will not acquire one by being asked
1220
+ // again. Graphs submitted before the provenance stamp landed are all in this state.
1221
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} needs a person, but its workflow has no ` +
1222
+ `agent to ask on behalf of, so no request can be raised. The task stays Pending ` +
1223
+ `and visible in the Tasks UI; it will not be retried.`);
1224
+ return 'permanent-failure';
1225
+ }
1226
+ request.AgentID = owningAgentID;
1227
+ request.RequestForUserID = task.UserID;
1228
+ request.RequestedAt = new Date();
1229
+ request.Status = 'Requested';
1230
+ request.Request = task.Description || `A workflow is waiting on you to complete "${task.Name}".`;
1231
+ // The graph's own run is the provenance a reader follows back to see what led here.
1232
+ request.OriginatingAgentRunID = await this.submittingRunOf(provider, task);
1233
+ // The deadline, when the author set one. `expireOverdueRequests` has always been able to
1234
+ // enforce this — it expires the request and fails the step so a give-up edge can route
1235
+ // around it — but nothing ever WROTE the column, so that whole path had never run outside
1236
+ // a test and a workflow waiting on someone who left the company waited forever.
1237
+ // Absent means no deadline, deliberately: expiring on a timeout nobody chose would be
1238
+ // worse than waiting.
1239
+ const expiresInHours = this.parseConfiguration(task)?.human?.expiresInHours;
1240
+ if (expiresInHours && expiresInHours > 0) {
1241
+ request.ExpiresAt = new Date(Date.now() + expiresInHours * 60 * 60 * 1000);
1242
+ }
1243
+ if (!(await request.Save())) {
1244
+ LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ` +
1245
+ `${request.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1246
+ // A failed SAVE may be transient (deadlock, contention), so this one earns a retry.
1247
+ return 'transient-failure';
1248
+ }
1249
+ return 'raised';
1250
+ }
1251
+ catch (e) {
1252
+ // Never fatal. The task remains Pending and visible; a missing request is recoverable,
1253
+ // whereas throwing here would abort the whole dispatch pass for every other branch.
1254
+ LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
1255
+ return 'transient-failure';
1256
+ }
1257
+ }
1258
+ /**
1259
+ * The agent that owns this task's workflow — who the request is asked on behalf of.
1260
+ *
1261
+ * Reads the graph's parent row, falling back to the run that submitted it. A human step has no
1262
+ * agent of its own by design: `AgentID` names what EXECUTES a step, and a person is not an agent.
1263
+ */
1264
+ async owningAgentOf(provider, task) {
1265
+ if (task.AgentID)
1266
+ return task.AgentID;
1267
+ if (!task.ParentID)
1268
+ return null;
1269
+ try {
1270
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
1271
+ if (!(await parent.Load(task.ParentID)))
1272
+ return null;
1273
+ if (parent.AgentID)
1274
+ return parent.AgentID;
1275
+ if (!parent.AgentRunID)
1276
+ return null;
1277
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
1278
+ return (await run.Load(parent.AgentRunID)) ? run.AgentID : null;
1279
+ }
1280
+ catch {
1281
+ return null;
1282
+ }
1283
+ }
1284
+ /** The still-open request for a task, if one exists. */
1285
+ async findOpenRequest(provider, taskID) {
1286
+ const result = await RunView.FromMetadataProvider(provider).RunView({
1287
+ EntityName: 'MJ: AI Agent Requests',
1288
+ ExtraFilter: `OriginatingTaskID='${taskID}' AND Status='Requested'`,
1289
+ ResultType: 'entity_object',
1290
+ BypassCache: true,
1291
+ }, this.contextUser);
1292
+ return (result.Success ? result.Results?.[0] : null) ?? null;
1293
+ }
1294
+ /**
1295
+ * Settles a human task from the request a person answered.
1296
+ *
1297
+ * Runs on the poll rather than on a save hook, because the answer can arrive through any surface
1298
+ * — the inbox, the API, a conversation — and only the dispatcher knows how to release the rest
1299
+ * of the graph afterwards.
1300
+ *
1301
+ * **`ResponseData` becomes the task's output.** That is what makes a human step useful rather
1302
+ * than a gate: a downstream edge can branch on what the person actually said, typed by the
1303
+ * request's own ResponseSchema. A step that only recorded "approved" would force every decision
1304
+ * back into a separate action.
1305
+ */
1306
+ async settleAnsweredHumanTasks(provider, graphID) {
1307
+ const waiting = await RunView.FromMetadataProvider(provider).RunView({
1308
+ EntityName: 'MJ: Tasks',
1309
+ ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
1310
+ ResultType: 'entity_object',
1311
+ BypassCache: true,
1312
+ }, this.contextUser);
1313
+ if (!waiting.Success)
1314
+ return;
1315
+ for (const task of waiting.Results ?? []) {
1316
+ const request = await this.answeredRequestFor(provider, task.ID);
1317
+ if (!request)
1318
+ continue;
1319
+ const rejected = request.Status === 'Rejected';
1320
+ const expired = request.Status === 'Expired';
1321
+ task.Status = rejected || expired ? 'Failed' : 'Complete';
1322
+ task.CompletedAt = new Date();
1323
+ task.PercentComplete = rejected || expired ? 0 : 100;
1324
+ task.ClaimedBy = null;
1325
+ task.ClaimExpiresAt = null;
1326
+ task.OutputPayload = request.ResponseData ?? null;
1327
+ if (rejected) {
1328
+ task.ErrorMessage = request.Comments || 'A person rejected this step.';
1329
+ }
1330
+ else if (expired) {
1331
+ // Stated as a failure rather than left Pending. A workflow blocked forever on
1332
+ // someone who never answered — who may have left the company — is the silent stall
1333
+ // this whole path exists to avoid, and a give-up edge can now route around it.
1334
+ task.ErrorMessage = 'Nobody answered this step before its request expired.';
1335
+ }
1336
+ if (!(await task.Save())) {
1337
+ LogError(`[TaskGraphDispatcher] Could not settle human task ${task.ID}: ` +
1338
+ `${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1339
+ }
1340
+ }
1341
+ }
1342
+ /**
1343
+ * Re-opens a human step whose request was CANCELLED.
1344
+ *
1345
+ * `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
1346
+ * rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
1347
+ * it". Nothing replaced it. `raiseHumanRequest` refuses to raise twice (the notified marker on
1348
+ * `ClaimedBy` is what stops the notification storm), so a cancelled request left the task Pending
1349
+ * with no open request and no path to acquiring one: a workflow waiting forever on a question
1350
+ * nobody is being asked.
1351
+ *
1352
+ * Clearing the marker is the whole fix — the next poll sees an un-notified Pending human task
1353
+ * and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
1354
+ * human action: it takes another person cancelling again to come back here.
1355
+ */
1356
+ async reopenCancelledHumanTasks(provider, graphID) {
1357
+ const waiting = await RunView.FromMetadataProvider(provider).RunView({
1358
+ EntityName: 'MJ: Tasks',
1359
+ // `StepType` is NULLABLE, and rows predating the column exist (4 in the reference
1360
+ // database at the time of writing). None currently carry a UserID, but a human task
1361
+ // written by any path that set the assignee without the discriminator would be
1362
+ // invisible to a `StepType='Human'` filter and stay dead forever after a cancel —
1363
+ // the exact stall this method exists to end. The notified marker already narrows
1364
+ // this to tasks the dispatcher raised a request for, so the widening cannot pull in
1365
+ // unrelated work.
1366
+ ExtraFilter: `ParentID='${graphID}' AND Status='Pending' ` +
1367
+ `AND (StepType='Human' OR (StepType IS NULL AND UserID IS NOT NULL)) ` +
1368
+ `AND ClaimedBy='${HUMAN_TASK_NOTIFIED_MARKER}'`,
1369
+ ResultType: 'entity_object',
1370
+ BypassCache: true,
1371
+ }, this.contextUser);
1372
+ if (!waiting.Success)
1373
+ return;
1374
+ for (const task of waiting.Results ?? []) {
1375
+ // Only when there is nothing live AND nothing terminal. A task with an open request is
1376
+ // simply waiting; one with a terminal request is settled on the next pass by
1377
+ // settleAnsweredHumanTasks, and re-raising either would ask the same question twice.
1378
+ if (await this.findOpenRequest(provider, task.ID))
1379
+ continue;
1380
+ if (await this.answeredRequestFor(provider, task.ID))
1381
+ continue;
1382
+ LogStatus(`[TaskGraphDispatcher] The request for '${task.Name}' was cancelled and nothing ` +
1383
+ `replaced it; asking again.`);
1384
+ task.ClaimedBy = null;
1385
+ if (!(await task.Save())) {
1386
+ LogError(`[TaskGraphDispatcher] Could not re-open cancelled human task ${task.ID}: ` +
1387
+ `${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1388
+ }
1389
+ }
1390
+ }
1391
+ /** The answered (or expired) request for a task, if any. */
1392
+ async answeredRequestFor(provider, taskID) {
1393
+ const result = await RunView.FromMetadataProvider(provider).RunView({
1394
+ EntityName: 'MJ: AI Agent Requests',
1395
+ // Everything terminal. 'Canceled' is deliberately absent: a cancelled request means
1396
+ // the ASK was withdrawn, not that the step was decided, so the task keeps waiting
1397
+ // for whatever replaces it.
1398
+ ExtraFilter: `OriginatingTaskID='${taskID}' AND Status IN ('Approved','Rejected','Responded','Expired')`,
1399
+ OrderBy: 'RespondedAt DESC',
1400
+ ResultType: 'entity_object',
1401
+ BypassCache: true,
1402
+ }, this.contextUser);
1403
+ return (result.Success ? result.Results?.[0] : null) ?? null;
1404
+ }
1405
+ /**
1406
+ * Expires requests whose deadline has passed.
1407
+ *
1408
+ * A deadline that nothing enforces is a comment. Without this an `ExpiresAt` in the past leaves
1409
+ * the request `Requested` forever and the workflow waiting on it just as long.
1410
+ */
1411
+ async expireOverdueRequests(provider, graphID) {
1412
+ // Scoped by an explicit id list rather than a subquery against a view name, so this reads
1413
+ // the same on any provider rather than assuming a SQL dialect and a physical view.
1414
+ const humanTasks = await RunView.FromMetadataProvider(provider).RunView({
1415
+ EntityName: 'MJ: Tasks',
1416
+ Fields: ['ID'],
1417
+ ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
1418
+ ResultType: 'simple',
1419
+ }, this.contextUser);
1420
+ const ids = (humanTasks.Results ?? []).map((r) => `'${r.ID}'`);
1421
+ if (ids.length === 0)
1422
+ return;
1423
+ const nowISO = new Date().toISOString();
1424
+ const overdue = await RunView.FromMetadataProvider(provider).RunView({
1425
+ EntityName: 'MJ: AI Agent Requests',
1426
+ ExtraFilter: `Status='Requested' AND ExpiresAt IS NOT NULL AND ExpiresAt < '${nowISO}' ` +
1427
+ `AND OriginatingTaskID IN (${ids.join(',')})`,
1428
+ ResultType: 'entity_object',
1429
+ BypassCache: true,
1430
+ }, this.contextUser);
1431
+ if (!overdue.Success)
1432
+ return;
1433
+ for (const request of overdue.Results ?? []) {
1434
+ request.Status = 'Expired';
1435
+ if (!(await request.Save())) {
1436
+ LogError(`[TaskGraphDispatcher] Could not expire request ${request.ID}.`);
1437
+ }
1438
+ }
1439
+ }
1440
+ /** The agent run that submitted this task's graph, for provenance on the request. */
1441
+ async submittingRunOf(provider, task) {
1442
+ if (!task.ParentID)
1443
+ return null;
1444
+ try {
1445
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
1446
+ return (await parent.Load(task.ParentID)) ? parent.AgentRunID : null;
1447
+ }
1448
+ catch {
1449
+ return null;
1450
+ }
1451
+ }
624
1452
  /** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
625
1453
  async loadGraphState(provider, parentTaskID) {
626
1454
  const rv = RunView.FromMetadataProvider(provider);
@@ -628,8 +1456,13 @@ export class TaskGraphDispatcher {
628
1456
  // fires no cache invalidation. See findActiveGraphIDs.
629
1457
  const childrenResult = await rv.RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
630
1458
  const children = (childrenResult.Success ? childrenResult.Results : []) ?? [];
631
- if (children.length === 0)
632
- return { nodes: [], edges: [], entityById: new Map(), unreachableTaskIDs: new Set() };
1459
+ if (children.length === 0) {
1460
+ return {
1461
+ nodes: [], edges: [], entityById: new Map(),
1462
+ unreachableTaskIDs: new Set(), skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
1463
+ handledFailureIDs: new Set(),
1464
+ };
1465
+ }
633
1466
  const idList = children.map((c) => `'${c.ID}'`).join(',');
634
1467
  const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
635
1468
  const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
@@ -651,7 +1484,28 @@ export class TaskGraphDispatcher {
651
1484
  // unreachable instead, and blocked before anything can claim it.
652
1485
  const droppedInto = new Set();
653
1486
  const stillReachable = new Set();
654
- for (const d of deps) {
1487
+ // EXCLUSIVE edges are exempt from the generic machinery below, and that exemption is
1488
+ // load-bearing. An XOR loser is by definition condition-false, so the ordinary path would
1489
+ // record it as unreachable and Block it — and a Blocked child poisons the parent rollup, so
1490
+ // every fork would settle the graph as Blocked. Losers must become Skipped instead, which
1491
+ // only ResolveExclusiveGroups can decide.
1492
+ const exclusive = deps.filter((d) => !!d.ExclusiveGroup);
1493
+ const ordinary = deps.filter((d) => !d.ExclusiveGroup);
1494
+ const resolution = ResolveExclusiveGroups(exclusive.map((d) => ({
1495
+ id: d.ID,
1496
+ taskId: d.TaskID,
1497
+ dependsOnTaskId: d.DependsOnTaskID,
1498
+ exclusiveGroup: d.ExclusiveGroup,
1499
+ originStatus: (entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending'),
1500
+ priority: d.Priority ?? 0,
1501
+ sequence: d.Sequence ?? 0,
1502
+ conditionOutcome: this.evaluateExclusiveCondition(d, entityById),
1503
+ })),
1504
+ // A flow's failure handling is its outgoing edges, so a Failed origin still decides its
1505
+ // group. For a loop-agent graph the set is Complete-only and nothing changes.
1506
+ new Set(['Complete', 'Failed']));
1507
+ const loserEdgeIDs = new Set(resolution.loserEdgeIDs);
1508
+ for (const d of ordinary) {
655
1509
  if (d.Condition?.trim()) {
656
1510
  const outcome = this.evaluateEdgeCondition(d, entityById);
657
1511
  if (outcome === 'drop') {
@@ -666,14 +1520,30 @@ export class TaskGraphDispatcher {
666
1520
  dependencyType: d.DependencyType,
667
1521
  });
668
1522
  }
1523
+ for (const d of exclusive) {
1524
+ // A losing edge is removed rather than left to gate: its target is being skipped, and a
1525
+ // live edge into a skipped task would keep the graph waiting on a branch nobody took.
1526
+ if (loserEdgeIDs.has(d.ID))
1527
+ continue;
1528
+ stillReachable.add(d.TaskID);
1529
+ liveEdges.push({
1530
+ taskId: d.TaskID,
1531
+ dependsOnTaskId: d.DependsOnTaskID,
1532
+ dependencyType: d.DependencyType,
1533
+ });
1534
+ }
669
1535
  // Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
670
1536
  // waiting on it, and a node reached by an alternate branch is genuinely reachable.
671
1537
  const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
1538
+ const nodes = children.map((c) => ({ id: c.ID, status: c.Status }));
672
1539
  return {
673
- nodes: children.map((c) => ({ id: c.ID, status: c.Status })),
1540
+ nodes,
674
1541
  edges: liveEdges,
675
1542
  entityById,
676
1543
  unreachableTaskIDs,
1544
+ skipSeedTaskIDs: new Set(resolution.skipSeedTaskIDs),
1545
+ holdTaskIDs: new Set(resolution.holdTaskIDs),
1546
+ handledFailureIDs: await this.computeHandledFailures(provider, parentTaskID, nodes, liveEdges),
677
1547
  };
678
1548
  }
679
1549
  /**
@@ -687,6 +1557,19 @@ export class TaskGraphDispatcher {
687
1557
  const upstream = entityById.get(dep.DependsOnTaskID);
688
1558
  if (!upstream)
689
1559
  return 'keep';
1560
+ // TERMINALITY GUARD — fixes a latent bug, not a hypothetical one.
1561
+ //
1562
+ // Without it, every conditional edge is evaluated on every poll cycle, including while its
1563
+ // origin is still Pending. A condition like `succeeded` is then a DEFINITE FALSE, the edge
1564
+ // is dropped, and the target is Blocked at wave one — permanently, before the origin ever
1565
+ // ran. That kills any conditioned linear chain, which is the most common flow shape there
1566
+ // is.
1567
+ //
1568
+ // A non-terminal origin is UNDECIDED, and 'keep' is the safe reading of undecided: the
1569
+ // prerequisite gate already prevents the target starting early, so keeping the edge costs
1570
+ // nothing and dropping it is irreversible.
1571
+ if (!TERMINAL_FOR_CONDITIONS.has(upstream.Status))
1572
+ return 'keep';
690
1573
  let output = null;
691
1574
  if (upstream.OutputPayload) {
692
1575
  try {
@@ -694,13 +1577,7 @@ export class TaskGraphDispatcher {
694
1577
  }
695
1578
  catch { /* a malformed payload is not grounds to drop a prerequisite */ }
696
1579
  }
697
- const result = this.conditionEvaluator.Evaluate(dep.Condition, {
698
- status: upstream.Status,
699
- succeeded: upstream.Status === 'Complete',
700
- failed: upstream.Status === 'Failed',
701
- output,
702
- errorMessage: upstream.ErrorMessage ?? null,
703
- });
1580
+ const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
704
1581
  if (!result.Success) {
705
1582
  LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
706
1583
  `(${result.ErrorMessage}); keeping the edge so the graph stalls visibly rather than ` +
@@ -709,6 +1586,63 @@ export class TaskGraphDispatcher {
709
1586
  }
710
1587
  return result.Value ? 'keep' : 'drop';
711
1588
  }
1589
+ /**
1590
+ * An exclusive edge's condition as a three-way outcome.
1591
+ *
1592
+ * `ResolveExclusiveGroups` needs to tell "false" from "could not be evaluated": the first loses
1593
+ * the branch, the second holds the whole group. The generic keep/drop path cannot express that
1594
+ * difference, which is why exclusive edges take this route instead.
1595
+ */
1596
+ evaluateExclusiveCondition(dep, entityById) {
1597
+ if (!dep.Condition?.trim())
1598
+ return 'satisfied';
1599
+ const upstream = entityById.get(dep.DependsOnTaskID);
1600
+ if (!upstream)
1601
+ return 'unevaluable';
1602
+ let output = null;
1603
+ if (upstream.OutputPayload) {
1604
+ try {
1605
+ output = JSON.parse(upstream.OutputPayload);
1606
+ }
1607
+ catch { /* malformed payload */ }
1608
+ }
1609
+ const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
1610
+ if (!result.Success)
1611
+ return 'unevaluable';
1612
+ return result.Value ? 'satisfied' : 'unsatisfied';
1613
+ }
1614
+ /**
1615
+ * Everything an edge condition can see — the SUPERSET of both dialects.
1616
+ *
1617
+ * A flow condition is written against `payload` / `stepResult` / `flowContext` / `data` /
1618
+ * `context`; the dispatcher's own conditions are written against `status` / `succeeded` /
1619
+ * `failed` / `output` / `errorMessage`. Compiling flows onto this engine without the flow
1620
+ * dialect would make every `payload.x` condition evaluate against nothing — silently, since an
1621
+ * undefined property is simply falsy. Both dialects are readable here so a condition means the
1622
+ * same thing on either engine.
1623
+ *
1624
+ * `payload` is the ORIGIN task's post-step snapshot. There is deliberately no "graph-wide
1625
+ * payload": each task's output is its own, and inventing a merged one would give conditions a
1626
+ * value the flow engine never had.
1627
+ */
1628
+ buildConditionContext(upstream, output) {
1629
+ const envelope = (output && typeof output === 'object' ? output : {});
1630
+ const succeeded = upstream.Status === 'Complete';
1631
+ return {
1632
+ // dispatcher dialect — unchanged
1633
+ status: upstream.Status,
1634
+ succeeded,
1635
+ failed: upstream.Status === 'Failed',
1636
+ output,
1637
+ errorMessage: upstream.ErrorMessage ?? null,
1638
+ // flow dialect
1639
+ payload: envelope.payload ?? output,
1640
+ stepResult: { Success: succeeded, step: upstream.Name, result: envelope.result ?? output },
1641
+ flowContext: { currentStepId: upstream.ID, completedSteps: [], executionPath: [], stepCount: 0 },
1642
+ data: envelope.data ?? {},
1643
+ context: envelope.context ?? {},
1644
+ };
1645
+ }
712
1646
  /** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
713
1647
  async loadDependencyOutputs(provider, taskID) {
714
1648
  const outputs = new Map();
@@ -731,5 +1665,568 @@ export class TaskGraphDispatcher {
731
1665
  }
732
1666
  return outputs;
733
1667
  }
1668
+ /**
1669
+ * Runs one task's body, whatever kind of step it is.
1670
+ *
1671
+ * **Routing is on `StepType`, not on which key happens to be set.** A loop step carries the same
1672
+ * `ActionID` or `AgentID` as an ordinary step — that key is what the loop *repeats* — so the old
1673
+ * `task.ActionID ? action : agent` test would have run a loop exactly once and called it done.
1674
+ * `StepType` is the only field that distinguishes them.
1675
+ *
1676
+ * Every branch is normalized to one shape so the recording path above stays single: an action has
1677
+ * no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
1678
+ */
1679
+ async runTaskBody(task, provider, inputPayload, dependencyOutputs) {
1680
+ const payload = this.mergedPayload(inputPayload, dependencyOutputs);
1681
+ const config = task.ConfigurationObject;
1682
+ // A loop's own step type decides how many times its body runs; the body itself is dispatched
1683
+ // through the very same runners as a one-shot step.
1684
+ if (task.StepType === 'ForEach' || task.StepType === 'While') {
1685
+ return { ...await this.runLoopTask(task, provider, payload, dependencyOutputs), PayloadAtStart: payload };
1686
+ }
1687
+ const { params, errors } = BuildMappedInput(config?.inputMapping, { payload });
1688
+ for (const e of errors)
1689
+ LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
1690
+ // `payload`, NOT `inputPayload` — the MERGED value computed above, which includes what every
1691
+ // dependency produced.
1692
+ //
1693
+ // A step with an input mapping got exactly the parameters it declared; a step WITHOUT one
1694
+ // fell back to the raw input and therefore saw nothing any earlier step had produced. For a
1695
+ // Prompt step — which declares no mapping by design, because it reads the whole payload
1696
+ // through `{{ _CURRENT_PAYLOAD }}` — that meant the placeholder rendered `{}` and the model
1697
+ // was asked to write from an empty brief.
1698
+ //
1699
+ // It answered anyway. The Content Pipeline's draft step said "the research data was empty",
1700
+ // which was TRUE of what it had been handed while twenty research results sat in the
1701
+ // dependency outputs beside it, and the reviewer then rejected the draft for saying so.
1702
+ // Every layer looked like it was working.
1703
+ const effectiveInput = Object.keys(params).length > 0 ? params : payload;
1704
+ if (task.StepType === 'Prompt') {
1705
+ if (!this.promptRunner) {
1706
+ // Not a failure: "nobody here can run this" is not "this ran and did not work".
1707
+ return { Success: false, AgentRunID: null, ErrorMessage: 'No prompt runner is loaded on this host.' };
1708
+ }
1709
+ const promptResult = await this.promptRunner.RunPromptForTask({
1710
+ TaskID: task.ID,
1711
+ PromptID: task.PromptID,
1712
+ InputPayload: effectiveInput,
1713
+ DependencyOutputs: dependencyOutputs,
1714
+ TemplateParameters: config?.prompt?.templateParameters,
1715
+ Provider: provider,
1716
+ ContextUser: this.contextUser,
1717
+ });
1718
+ // A prompt's response is DEEP-MERGED into the payload rather than replacing it. A prompt
1719
+ // answers one question; replacing the payload with its answer would discard everything
1720
+ // the steps before it established, which is how a late step loses the data it depends on.
1721
+ const merged = promptResult.Success && promptResult.Output && typeof promptResult.Output === 'object'
1722
+ ? deepMergePayload(payload, promptResult.Output)
1723
+ : payload;
1724
+ return {
1725
+ Success: promptResult.Success,
1726
+ AgentRunID: null,
1727
+ ErrorMessage: promptResult.ErrorMessage,
1728
+ Output: this.applyStepOutputMapping(task, merged, merged, config?.outputMapping),
1729
+ PayloadAtStart: payload,
1730
+ ChatMessage: promptResult.ChatMessage,
1731
+ // Returned even when the prompt FAILED. A failed prompt still cost tokens, and a
1732
+ // cost rollup that silently omits failures under-reports exactly the runs someone
1733
+ // is most likely to be investigating.
1734
+ PromptRunID: promptResult.PromptRunID,
1735
+ };
1736
+ }
1737
+ const raw = task.ActionID
1738
+ ? { ...await this.actionRunner.RunActionForTask({
1739
+ TaskID: task.ID,
1740
+ ActionID: task.ActionID,
1741
+ InputPayload: effectiveInput,
1742
+ DependencyOutputs: dependencyOutputs,
1743
+ Provider: provider,
1744
+ ContextUser: this.contextUser,
1745
+ }), AgentRunID: null }
1746
+ : await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs);
1747
+ return {
1748
+ ...raw,
1749
+ Output: this.applyStepOutputMapping(task, payload, raw.Output, config?.outputMapping),
1750
+ PayloadAtStart: payload,
1751
+ };
1752
+ }
1753
+ /**
1754
+ * Runs a loop step: its body once per iteration, with the item and index in scope.
1755
+ *
1756
+ * The loop's own `Configuration` supplies the definition; the row's `ActionID` / `AgentID`
1757
+ * supplies what to repeat. Per-iteration inputs are resolved fresh each pass — the bindings are
1758
+ * merged into the payload before the mapping is applied, which is how a body can reference the
1759
+ * current item at all.
1760
+ */
1761
+ async runLoopTask(task, provider, payload, dependencyOutputs) {
1762
+ const config = task.ConfigurationObject;
1763
+ const op = task.StepType === 'ForEach' ? config?.forEach : config?.while;
1764
+ if (!op) {
1765
+ return {
1766
+ Success: false,
1767
+ AgentRunID: null,
1768
+ ErrorMessage: `"${task.Name}" is a ${task.StepType} step with no loop settings, so there is nothing to repeat.`,
1769
+ };
1770
+ }
1771
+ // A prompt body has no params of its own — it receives the payload (with the loop bindings
1772
+ // merged in) through the placeholder, so an empty mapping is correct rather than missing.
1773
+ const bodyMapping = (op.action?.params ?? {});
1774
+ // The BODY's output mapping, applied once per pass — see `foldIterationOutput`.
1775
+ //
1776
+ // It used to be applied a single time after the loop finished, against the accumulated
1777
+ // payload. That is the wrong moment in two ways at once: the mapping names an output
1778
+ // PARAMETER of the body, which no longer exists by then, and a mapping like
1779
+ // `"Items": "results[]"` can only append per pass. So every pass merged its raw result into
1780
+ // the shared payload instead, each overwriting the last, and the mapping matched nothing and
1781
+ // wrote nothing. A ForEach over five items reported five successes and kept item five.
1782
+ const bodyOutputMapping = op.action?.outputMapping ?? op.prompt?.outputMapping;
1783
+ // Where this step sits in its graph, resolved ONCE rather than per iteration. A loop body is
1784
+ // dispatched exactly like a one-shot step and needs the same two things: the run that
1785
+ // submitted the graph (so a spawned run gets a ParentRunID and is visible to the tree and to
1786
+ // cost), and the continuation depth (so the recursion cap still applies). Omitting them made
1787
+ // loop bodies second-class in every dimension — and reopened the unbounded-recursion hole
1788
+ // THROUGH loops, since each spawned run restarted the chain at zero.
1789
+ const graphContext = await this.graphContext(provider, task);
1790
+ // THE LOOP'S PAYLOAD ACCUMULATES. Each iteration's output merges in, and the next iteration
1791
+ // — and the While condition — sees it. Without this the condition closure re-read the
1792
+ // payload as it was when the loop STARTED, so a `while payload.brandOK !== true` could never
1793
+ // become false: the loop burned every iteration re-examining the original input and always
1794
+ // took the give-up branch, making the other branch unreachable. The loop ran, reported
1795
+ // success, and its result was predetermined.
1796
+ let livePayload = { ...payload };
1797
+ // One entry per pass, so the loop's work exists somewhere the platform can see it. Without
1798
+ // this a loop is a single childless node: the run tree reaches nested work through six links
1799
+ // and an iteration is none of them, so the passes were invisible to the timeline AND their
1800
+ // spend was missing from the settlement rollup. See ITaskStepRuntime.iterations.
1801
+ const iterationTrace = [];
1802
+ // Bounds what the trace's payloads may cost. The pointers are never budgeted — those are the
1803
+ // durable record of the work and must survive whatever the payloads do.
1804
+ const budget = new IterationPayloadBudget();
1805
+ const invokeBody = async ({ Index, Bindings }) => {
1806
+ // Bindings go INTO the payload rather than beside it, so an authored mapping reaches the
1807
+ // current item the same way it reaches anything else: `payload.<itemVariable>`.
1808
+ const iterationPayload = { ...livePayload, ...Bindings };
1809
+ const resolved = ResolveMappedInput(bodyMapping, { payload: iterationPayload });
1810
+ /**
1811
+ * Folds an iteration's output into the running payload the next pass will see, and
1812
+ * records what the pass produced.
1813
+ *
1814
+ * The trace is written HERE rather than after the loop because a loop that fails partway
1815
+ * still ran the passes before it, and their runs are real spend that must not vanish
1816
+ * because the loop as a whole did not finish.
1817
+ */
1818
+ const absorb = (outcome, bodyInput) => {
1819
+ livePayload = this.foldIterationOutput(task, livePayload, outcome.Output, bodyOutputMapping);
1820
+ iterationTrace.push({
1821
+ index: Index,
1822
+ // What THIS pass was handed and what it gave back — not the loop's running
1823
+ // payload before and after it.
1824
+ //
1825
+ // A pass has no row of its own, so without these there is nowhere its work can be
1826
+ // recorded: every iteration presented null on both sides and the run view could
1827
+ // say nothing about any single pass, which for a loop is the only interesting
1828
+ // question. But recording the RUNNING payload on both sides — the obvious reading
1829
+ // of "before and after" — is quadratic: each pass would hold a full copy of
1830
+ // everything every earlier pass accumulated. A five-iteration demo produced a
1831
+ // 121KB Configuration that way; the same loop over fifty items would produce
1832
+ // megabytes, in a column every reader of the row pays to load.
1833
+ //
1834
+ // The pass's own input and output are what a reader actually wants ("what did
1835
+ // pass three do?"), and they are constant-sized per pass.
1836
+ payloadAtStart: budget.Take(bodyInput),
1837
+ payloadAtEnd: budget.Take(outcome.Output),
1838
+ promptRunID: outcome.PromptRunID,
1839
+ agentRunID: outcome.AgentRunID,
1840
+ // An ACTION body records its log here. Omitting it left an action-bodied pass
1841
+ // with no pointer at all — no cost, no timing, nothing to open — and the tree,
1842
+ // seeing neither a prompt run nor an agent run, fell through to its last branch
1843
+ // and called the pass a Sub-Agent. A loop over a web search then showed five
1844
+ // sub-agent runs that never existed.
1845
+ actionLogID: outcome.ActionLogID,
1846
+ success: outcome.Success,
1847
+ errorMessage: outcome.ErrorMessage,
1848
+ });
1849
+ return outcome;
1850
+ };
1851
+ // A prompt body is checked FIRST because it is the only one whose id lives in its own
1852
+ // column: a loop repeating a prompt has PromptID set and both ActionID and AgentID null,
1853
+ // so falling through to the agent branch would dereference a null agent id.
1854
+ if (task.StepType && task.PromptID && !task.ActionID) {
1855
+ if (!this.promptRunner) {
1856
+ return { Success: false, ErrorMessage: 'No prompt runner is loaded on this host.' };
1857
+ }
1858
+ return absorb(await this.promptRunner.RunPromptForTask({
1859
+ TaskID: task.ID,
1860
+ PromptID: task.PromptID,
1861
+ // The ITERATION payload, not the mapped params. An action body declares its
1862
+ // inputs and gets exactly those; a prompt body declares none — it receives the
1863
+ // whole payload through the placeholder, and the loop's item and index are
1864
+ // merged INTO that payload. Passing the mapped result here handed the prompt an
1865
+ // empty object, so every iteration asked the model to describe nothing and got
1866
+ // five confident answers about nothing back.
1867
+ InputPayload: iterationPayload,
1868
+ DependencyOutputs: dependencyOutputs,
1869
+ // The loop's bindings become TEMPLATE VARIABLES, so an author writes
1870
+ // `{{ field }}` for the item the loop is on — which is what `itemVariable` is
1871
+ // for, and what anyone reading the step's configuration expects. Reaching it
1872
+ // through the payload placeholder instead works but is not discoverable, and
1873
+ // getting it wrong is silent: the variable renders empty and the model answers
1874
+ // confidently about nothing.
1875
+ TemplateParameters: { ...stringifyBindings(Bindings), ...op.prompt?.templateParameters },
1876
+ Provider: provider,
1877
+ ContextUser: this.contextUser,
1878
+ }), iterationPayload);
1879
+ }
1880
+ if (task.ActionID) {
1881
+ return absorb(await this.actionRunner.RunActionForTask({
1882
+ TaskID: task.ID,
1883
+ ActionID: task.ActionID,
1884
+ InputPayload: resolved,
1885
+ DependencyOutputs: dependencyOutputs,
1886
+ Provider: provider,
1887
+ ContextUser: this.contextUser,
1888
+ }), resolved);
1889
+ }
1890
+ const agentInput = Object.keys(resolved).length > 0 ? resolved : iterationPayload;
1891
+ return absorb(await this.agentRunner.RunAgentForTask({
1892
+ TaskID: task.ID,
1893
+ AgentID: task.AgentID,
1894
+ // The ITERATION payload when the body declares no inputs of its own. A sub-agent
1895
+ // body has no `params`, so the mapped result is `{}` — every iteration was handing
1896
+ // the agent nothing and asking it to work from that.
1897
+ InputPayload: agentInput,
1898
+ DependencyOutputs: dependencyOutputs,
1899
+ ContinuationDepth: graphContext.Depth,
1900
+ SubmittingAgentRunID: graphContext.SubmittingAgentRunID,
1901
+ Provider: provider,
1902
+ ContextUser: this.contextUser,
1903
+ }), agentInput);
1904
+ };
1905
+ const outcome = task.StepType === 'ForEach'
1906
+ ? await RunForEachLoop(op, { payload }, invokeBody)
1907
+ : await RunWhileLoop(op, (iteration) => this.conditionEvaluator.Evaluate(op.condition,
1908
+ // BOTH forms, because a workflow should not have two condition dialects. An
1909
+ // EDGE condition is written `payload.brandOK !== true`; a loop condition used
1910
+ // to see the payload's keys spread at the top level and nothing named `payload`,
1911
+ // so the same expression that routes an edge failed here with
1912
+ // "payload is not defined". The spread stays for conditions already written
1913
+ // against it.
1914
+ { ...livePayload, payload: livePayload, iteration }), invokeBody);
1915
+ return {
1916
+ Success: outcome.Success,
1917
+ AgentRunID: null,
1918
+ ErrorMessage: outcome.ErrorMessage,
1919
+ // Every pass that ran, including those before a failure — see `iterationTrace`.
1920
+ Iterations: iterationTrace.length > 0 ? iterationTrace : undefined,
1921
+ // The ACCUMULATED payload — everything the iterations established — not the one the
1922
+ // loop started with, which would discard the loop's whole effect on the workflow.
1923
+ //
1924
+ // Only the STEP's own mapping is applied here. The body's mapping already ran once per
1925
+ // pass inside `foldIterationOutput`; applying it again against the accumulated payload
1926
+ // is what used to make it match nothing.
1927
+ Output: this.applyStepOutputMapping(task, livePayload, outcome.Output, config?.outputMapping),
1928
+ };
1929
+ }
1930
+ /**
1931
+ * Folds one pass's result into the loop's running payload.
1932
+ *
1933
+ * **With a body mapping**, the pass's declared outputs are filed where the author said to put
1934
+ * them — including `name[]`, which appends, so a ForEach can collect one entry per item. That is
1935
+ * the whole point of a loop over a collection, and it is only expressible per pass.
1936
+ *
1937
+ * **Without one**, the raw result is deep-merged, which is the pre-existing behaviour and the
1938
+ * right default for a `While` that converges on a value: each pass refines what the condition
1939
+ * reads. It is the wrong default for a ForEach that collects — hence the mapping.
1940
+ *
1941
+ * An unmapped output is reported per pass rather than swallowed, for the same reason
1942
+ * {@link applyStepOutputMapping} reports it: a mapping that names something the body never
1943
+ * returned means the pass did work that went nowhere, while everything reports success.
1944
+ */
1945
+ foldIterationOutput(task, livePayload, output, bodyOutputMapping) {
1946
+ if (!output || typeof output !== 'object' || Array.isArray(output))
1947
+ return livePayload;
1948
+ const source = output;
1949
+ if (!bodyOutputMapping)
1950
+ return deepMergePayload(livePayload, source);
1951
+ // Applied ONTO a deep copy of the running payload, not into a fresh object: `name[]` appends,
1952
+ // and appending is meaningless without the list already there. The copy is deep because the
1953
+ // trace has already recorded earlier passes' payloads — mutating a shared nested array would
1954
+ // retroactively rewrite what those passes are recorded as having seen.
1955
+ const { updates, errors, unmapped } = ApplyOutputMapping(source, bodyOutputMapping, structuredClone(livePayload));
1956
+ for (const e of errors)
1957
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} loop body: ${e}`);
1958
+ if (unmapped?.length) {
1959
+ LogError(`[TaskGraphDispatcher] '${task.Name}' loop body mapped output(s) it did not return: ` +
1960
+ `${unmapped.join(', ')}. The pass returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
1961
+ `Those payload values were NOT written, so anything downstream reading them sees nothing.`);
1962
+ }
1963
+ // `updates` IS the copy that was applied onto, so it is already the complete next payload.
1964
+ return updates;
1965
+ }
1966
+ /**
1967
+ * Files a step's result into the payload it hands downstream.
1968
+ *
1969
+ * **This is what makes a branch condition possible.** A workflow that branches on
1970
+ * `payload.stockPrice` has that value only because this step mapped `CurrentPrice -> stockPrice`.
1971
+ * Without it the condition reads `undefined` — merely falsy — so the workflow takes the other
1972
+ * branch, finishes, and reports success with nothing to indicate anything went wrong.
1973
+ *
1974
+ * The incoming payload is carried through as well as the update, so a value written three steps
1975
+ * back is still readable here. Returning only this step's own output is what used to limit a
1976
+ * condition's view to its immediate predecessor.
1977
+ */
1978
+ applyStepOutputMapping(task, payload, output, outputMapping) {
1979
+ // No mapping: MERGE the step's output over the payload rather than replacing it.
1980
+ //
1981
+ // Replacing is what made the Content Pipeline's exclusive pair unreachable. A While loop's
1982
+ // own output is a SUMMARY — `{iterations, succeeded, failed, results}` — so returning it
1983
+ // discarded the payload the iterations had built, including the `brandOK` the reviewer had
1984
+ // just set to true. The edges read `payload.brandOK === true` and `!== true`; against a
1985
+ // summary the first is false and the second is true, so the give-up branch won on EVERY run
1986
+ // no matter what the reviewer decided. The approved branch was unreachable in practice while
1987
+ // being perfectly reachable on the canvas.
1988
+ //
1989
+ // This is the same rule the mapped path already follows two lines down, and the same rule
1990
+ // the doc comment above states. The no-mapping branch was simply not following it.
1991
+ if (!outputMapping) {
1992
+ return output && typeof output === 'object' && !Array.isArray(output)
1993
+ ? { ...payload, ...output }
1994
+ : output ?? payload;
1995
+ }
1996
+ const source = output && typeof output === 'object' ? output : { value: output };
1997
+ const { updates, errors, unmapped } = ApplyOutputMapping(source, outputMapping);
1998
+ for (const e of errors)
1999
+ LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
2000
+ // A mapping that names an output the step never produced discards that step's work while
2001
+ // the step reports Complete. It is not fatal — an action may emit a parameter only on some
2002
+ // paths — but it must not be silent, and naming what WAS returned turns a multi-table
2003
+ // forensic exercise into one line. The Content Pipeline demo lost an entire research pass
2004
+ // this way, every run, because its mapping named another action's parameter.
2005
+ if (unmapped?.length) {
2006
+ LogError(`[TaskGraphDispatcher] '${task.Name}' mapped output(s) the step did not return: ` +
2007
+ `${unmapped.join(', ')}. The step returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
2008
+ `Those payload values were NOT written, so anything downstream reading them sees nothing.`);
2009
+ }
2010
+ return { ...payload, ...updates };
2011
+ }
2012
+ /**
2013
+ * Runs an Agent step, telling the runner where in the graph it sits.
2014
+ *
2015
+ * Depth and provenance are read together because they come from the same row: the graph's parent
2016
+ * task knows both how many continuation hops led here and which run submitted it.
2017
+ */
2018
+ async runAgentNode(task, provider, effectiveInput, dependencyOutputs) {
2019
+ const context = await this.graphContext(provider, task);
2020
+ return this.agentRunner.RunAgentForTask({
2021
+ TaskID: task.ID,
2022
+ AgentID: task.AgentID,
2023
+ InputPayload: effectiveInput,
2024
+ DependencyOutputs: dependencyOutputs,
2025
+ ContinuationDepth: context.Depth,
2026
+ SubmittingAgentRunID: context.SubmittingAgentRunID,
2027
+ Provider: provider,
2028
+ ContextUser: this.contextUser,
2029
+ });
2030
+ }
2031
+ /**
2032
+ * Completes the agent run that parked on this graph.
2033
+ *
2034
+ * **This is the other half of submit-and-detach.** A run that dispatches a graph does not
2035
+ * complete at submission — it ends `Paused`, because reporting `Completed` above a workflow
2036
+ * where nothing has happened yet is a claim the row cannot support. The run's lifecycle is
2037
+ * finished HERE, when the graph it was waiting on actually settles, which is the first moment
2038
+ * the answer exists.
2039
+ *
2040
+ * Doing it from the dispatcher rather than by awaiting in the agent is what keeps the properties
2041
+ * that made detach right in the first place: a graph containing a human approval can park for
2042
+ * days without holding a conversation turn open, and a graph reclaimed by another instance after
2043
+ * a crash still settles its submitting run, because the settling happens wherever the graph
2044
+ * finishes rather than wherever it started.
2045
+ *
2046
+ * **Only a parked run is touched.** A run that is already `Completed`, `Failed` or `Cancelled`
2047
+ * reached that state for its own reasons — a second graph settling later, a run the user
2048
+ * cancelled, a run that failed after submitting — and overwriting it would rewrite history from
2049
+ * the outside. The `Paused` predicate is the whole guard.
2050
+ *
2051
+ * @param graphStatus the parent rollup's status: what the workflow as a whole did
2052
+ */
2053
+ async settleSubmittingRun(provider, parent, graphStatus) {
2054
+ const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
2055
+ if (!meta.submittedByAgentRunID)
2056
+ return; // a scheduled or remote-triggered graph has nobody waiting
2057
+ try {
2058
+ const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
2059
+ if (!(await run.Load(meta.submittedByAgentRunID))) {
2060
+ LogError(`[TaskGraphDispatcher] Could not load run ${meta.submittedByAgentRunID} to settle it against graph ${parent.ID}.`);
2061
+ return;
2062
+ }
2063
+ if (run.Status !== 'Paused')
2064
+ return;
2065
+ // The workflow's outcome becomes the run's outcome. A graph that ended any way other than
2066
+ // Complete did not do what the run started it to do, and a run reporting success over it
2067
+ // would be the same untruth in a different place.
2068
+ const succeeded = graphStatus === 'Complete';
2069
+ run.Status = succeeded ? 'Completed' : 'Failed';
2070
+ run.Success = succeeded;
2071
+ run.CompletedAt = new Date();
2072
+ if (!succeeded) {
2073
+ const reason = `The workflow "${parent.Name}" ended ${graphStatus}.`;
2074
+ run.ErrorMessage = run.ErrorMessage ? `${run.ErrorMessage}\n\n${reason}` : reason;
2075
+ }
2076
+ if (!(await run.Save())) {
2077
+ // Left parked rather than forced. A run stuck at Paused is visibly unfinished, which
2078
+ // is a state someone can investigate; a run flipped to Completed by a write that did
2079
+ // not land would be the same lie this whole change removes.
2080
+ LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}: ` +
2081
+ `${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It remains Paused.`);
2082
+ return;
2083
+ }
2084
+ LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${run.Status} — workflow "${parent.Name}" ended ${graphStatus}.`);
2085
+ }
2086
+ catch (e) {
2087
+ LogError(`[TaskGraphDispatcher] Could not settle the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
2088
+ }
2089
+ }
2090
+ /**
2091
+ * Gives every step that lacks one a position, once the graph has finished.
2092
+ *
2093
+ * **Why the run stores geometry at all.** A `TaskGraphSpec` is a logical structure with no
2094
+ * layout field, so a graph an agent emitted has no opinion about where its boxes go. Every
2095
+ * viewer was therefore laying it out for itself at render time — and a viewer that failed to
2096
+ * (because the canvas measures nodes it has not drawn yet) fell back to every node at the
2097
+ * origin, piled on one another, with the zoom-to-fit that follows fitting a one-node bounding
2098
+ * box. Settling it once, server-side, means the agent-run canvas, the Workflows runs tab and
2099
+ * anything built later all draw the same picture, and none of them has to compute it.
2100
+ *
2101
+ * **An authored position is never overwritten.** A workflow compiled from a Flow agent carries
2102
+ * the arrangement someone dragged into place; replacing it with an algorithm's guess would
2103
+ * discard a deliberate act. Only steps with no geometry get one, so a partially-arranged graph
2104
+ * keeps what it has.
2105
+ *
2106
+ * Failure here is logged and swallowed: this is presentation. A graph whose work completed must
2107
+ * not be reported as failed because its picture could not be saved.
2108
+ */
2109
+ async persistComputedLayout(graph) {
2110
+ try {
2111
+ const needsLayout = [...graph.entityById.values()].filter((t) => !this.parseConfiguration(t)?.layout);
2112
+ if (needsLayout.length === 0)
2113
+ return;
2114
+ // Laid out over the WHOLE graph, not just the nodes missing geometry: position depends on
2115
+ // where a node sits in the topology, and a layout computed over a subset would place its
2116
+ // nodes as though the rest of the workflow did not exist.
2117
+ const edges = graph.edges.map((e) => ({ From: e.dependsOnTaskId, To: e.taskId }));
2118
+ const positions = LayoutGraphNodes([...graph.entityById.keys()], edges, { Direction: 'LR' });
2119
+ for (const task of needsLayout) {
2120
+ const position = positions.get(task.ID);
2121
+ if (!position)
2122
+ continue;
2123
+ const existing = this.parseConfiguration(task);
2124
+ const merged = {
2125
+ ...existing,
2126
+ layout: { x: position.X, y: position.Y },
2127
+ };
2128
+ task.Configuration = JSON.stringify(merged);
2129
+ if (!(await task.Save())) {
2130
+ LogError(`[TaskGraphDispatcher] Could not save computed layout for ${task.ID}: ${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
2131
+ }
2132
+ }
2133
+ }
2134
+ catch (e) {
2135
+ LogError(`[TaskGraphDispatcher] Could not compute a layout for the settled graph: ${e instanceof Error ? e.message : String(e)}`);
2136
+ }
2137
+ }
2138
+ /**
2139
+ * The earliest moment any step in the graph began, or null when none has.
2140
+ *
2141
+ * Null is a real answer — a graph whose tasks are all still Pending has not started — and is
2142
+ * deliberately not collapsed to "now", which would date the graph from whenever this pass
2143
+ * happened to run.
2144
+ */
2145
+ earliestStart(entityById) {
2146
+ let earliest = null;
2147
+ for (const entity of entityById.values()) {
2148
+ if (!entity.StartedAt)
2149
+ continue;
2150
+ if (earliest === null || entity.StartedAt < earliest)
2151
+ earliest = entity.StartedAt;
2152
+ }
2153
+ return earliest;
2154
+ }
2155
+ /**
2156
+ * The step's Configuration with this run's artefacts folded in, or `undefined` to leave it be.
2157
+ *
2158
+ * **Merged into the authored bag, never written over it.** The Configuration column holds the
2159
+ * step's definition — its loop body, its mappings, its policy, the position someone dragged it
2160
+ * to. Writing a fresh object containing only `runtime` would erase all of that the first time a
2161
+ * prompt step completed, which is the kind of loss that surfaces much later as a workflow that
2162
+ * mysteriously stopped mapping its output.
2163
+ *
2164
+ * Returns `undefined` when there is nothing to record, so the guarded write omits the column
2165
+ * rather than rewriting it with what it already held.
2166
+ */
2167
+ configurationWithRuntime(task, promptRunID, actionLogID, iterations, payloadAtStart) {
2168
+ if (!promptRunID && !actionLogID && !iterations?.length && !payloadAtStart)
2169
+ return undefined;
2170
+ const existing = this.parseConfiguration(task);
2171
+ const merged = {
2172
+ ...existing,
2173
+ runtime: {
2174
+ ...existing?.runtime,
2175
+ ...(promptRunID ? { promptRunID } : {}),
2176
+ ...(actionLogID ? { actionLogID } : {}),
2177
+ // Replaced wholesale rather than appended: this is the trace of the loop's LAST
2178
+ // execution, and a retried step that concatenated would report a loop that ran twice
2179
+ // as many passes as it did.
2180
+ ...(iterations?.length ? { iterations } : {}),
2181
+ // The resolved before-state, so the run view has something to diff the output
2182
+ // against. NOT written to Task.InputPayload, which holds the AUTHORED input and
2183
+ // round-trips back out as part of the spec.
2184
+ ...(payloadAtStart ? { payloadAtStart } : {}),
2185
+ },
2186
+ };
2187
+ return JSON.stringify(merged);
2188
+ }
2189
+ /**
2190
+ * Reads a step's Configuration bag, tolerating a row whose JSON cannot be parsed.
2191
+ *
2192
+ * Unparseable configuration is logged rather than thrown: the step has already RUN by the time
2193
+ * this is called, and refusing to record its outcome because its definition is malformed would
2194
+ * discard the result of real work and leave the task claimed until the claim lapsed.
2195
+ */
2196
+ parseConfiguration(task) {
2197
+ if (!task.Configuration)
2198
+ return undefined;
2199
+ try {
2200
+ return JSON.parse(task.Configuration);
2201
+ }
2202
+ catch (e) {
2203
+ LogError(`[TaskGraphDispatcher] Task ${task.ID} has unparseable Configuration; ` +
2204
+ `recording runtime artefacts against an empty bag. ${e instanceof Error ? e.message : String(e)}`);
2205
+ return undefined;
2206
+ }
2207
+ }
2208
+ /**
2209
+ * The payload a step sees: everything its prerequisites produced, plus its own declared input.
2210
+ *
2211
+ * **Why the outputs are merged rather than kept per-task.** A flow carried ONE payload that
2212
+ * accumulated as it went, so a condition on the edge into step C could read a value step A wrote.
2213
+ * Handing each task only its immediate predecessor's output would silently narrow that: the
2214
+ * condition reads `undefined`, which is falsy, and the workflow quietly takes a different route
2215
+ * than the flow it was compiled from. Merging in dependency order restores the accumulation.
2216
+ *
2217
+ * Later prerequisites win on a key collision, matching a flow's own last-write-wins behaviour.
2218
+ */
2219
+ mergedPayload(inputPayload, dependencyOutputs) {
2220
+ const merged = {};
2221
+ for (const output of dependencyOutputs.values()) {
2222
+ if (output && typeof output === 'object' && !Array.isArray(output)) {
2223
+ Object.assign(merged, output);
2224
+ }
2225
+ }
2226
+ if (inputPayload && typeof inputPayload === 'object' && !Array.isArray(inputPayload)) {
2227
+ Object.assign(merged, inputPayload);
2228
+ }
2229
+ return merged;
2230
+ }
734
2231
  }
735
2232
  //# sourceMappingURL=TaskGraphDispatcher.js.map