@memberjunction/task-graph 6.1.0-edge.1 → 6.1.0-edge.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +185 -0
- package/dist/TaskClaimStore.d.ts +11 -0
- package/dist/TaskClaimStore.d.ts.map +1 -1
- package/dist/TaskClaimStore.js +8 -3
- package/dist/TaskClaimStore.js.map +1 -1
- package/dist/TaskGraphDispatcher.d.ts +364 -2
- package/dist/TaskGraphDispatcher.d.ts.map +1 -1
- package/dist/TaskGraphDispatcher.js +1534 -37
- package/dist/TaskGraphDispatcher.js.map +1 -1
- package/dist/TaskGraphService.d.ts +110 -1
- package/dist/TaskGraphService.d.ts.map +1 -1
- package/dist/TaskGraphService.js +458 -19
- package/dist/TaskGraphService.js.map +1 -1
- package/dist/TaskLoopExecutor.d.ts +62 -0
- package/dist/TaskLoopExecutor.d.ts.map +1 -0
- package/dist/TaskLoopExecutor.js +248 -0
- package/dist/TaskLoopExecutor.js.map +1 -0
- package/dist/WorkflowSpecSync.d.ts +28 -2
- package/dist/WorkflowSpecSync.d.ts.map +1 -1
- package/dist/WorkflowSpecSync.js +83 -2
- package/dist/WorkflowSpecSync.js.map +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
- package/dist/operations/TaskGraphOperations.js +4 -0
- package/dist/operations/TaskGraphOperations.js.map +1 -1
- package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
- package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
- package/dist/operations/WorkflowDraftOperation.js +141 -0
- package/dist/operations/WorkflowDraftOperation.js.map +1 -0
- package/dist/types.d.ts +114 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js.map +1 -1
- package/package.json +10 -7
|
@@ -18,11 +18,12 @@
|
|
|
18
18
|
*
|
|
19
19
|
* @module @memberjunction/task-graph
|
|
20
20
|
*/
|
|
21
|
-
import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, } from '@memberjunction/ai-core-plus';
|
|
21
|
+
import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
|
|
22
22
|
import { LogError, LogStatus, RunView } from '@memberjunction/core';
|
|
23
|
-
import { ShutdownRegistry } from '@memberjunction/global';
|
|
23
|
+
import { ShutdownRegistry, UUIDsEqual } from '@memberjunction/global';
|
|
24
24
|
import { TaskClaimStore } from './TaskClaimStore.js';
|
|
25
25
|
import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
|
|
26
|
+
import { RunForEachLoop, RunWhileLoop } from './TaskLoopExecutor.js';
|
|
26
27
|
import { NotificationEngine } from '@memberjunction/notifications';
|
|
27
28
|
/** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
|
|
28
29
|
const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
|
|
@@ -34,8 +35,112 @@ const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
|
|
|
34
35
|
* from reclamation, so this value is never mistaken for a live claim.
|
|
35
36
|
*/
|
|
36
37
|
const HUMAN_TASK_NOTIFIED_MARKER = '__human-notified__';
|
|
38
|
+
/**
|
|
39
|
+
* The run-query capability of a provider, when it has one.
|
|
40
|
+
*
|
|
41
|
+
* `IMetadataProvider` does not extend `IRunQueryProvider`, but every provider that ships implements
|
|
42
|
+
* both. Narrowing by CAPABILITY rather than casting states that honestly: a provider that genuinely
|
|
43
|
+
* cannot run queries returns undefined and the caller reports it, instead of the call failing later
|
|
44
|
+
* behind a type assertion that claimed it could.
|
|
45
|
+
*/
|
|
46
|
+
function asRunQueryProvider(provider) {
|
|
47
|
+
const candidate = provider;
|
|
48
|
+
return typeof candidate.RunQuery === 'function' ? candidate : undefined;
|
|
49
|
+
}
|
|
37
50
|
import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata } from './TaskGraphService.js';
|
|
38
51
|
import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
|
|
52
|
+
/**
|
|
53
|
+
* Renders a loop's bindings as template values.
|
|
54
|
+
*
|
|
55
|
+
* Template parameters are strings; an item is usually an object. Objects are JSON-encoded rather
|
|
56
|
+
* than dropped, because `{{ field }}` printing `[object Object]` — or nothing at all — is exactly
|
|
57
|
+
* the silent failure this exists to prevent.
|
|
58
|
+
*/
|
|
59
|
+
function stringifyBindings(bindings) {
|
|
60
|
+
const out = {};
|
|
61
|
+
for (const [key, value] of Object.entries(bindings)) {
|
|
62
|
+
out[key] = typeof value === 'string' ? value : JSON.stringify(value, null, 2);
|
|
63
|
+
}
|
|
64
|
+
return out;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* How much of a loop's per-pass payloads may be kept, and what happens when that runs out.
|
|
68
|
+
*
|
|
69
|
+
* **Why a budget exists at all.** A loop's trace lives inside one `Configuration` column, and its
|
|
70
|
+
* size is the product of two things nobody bounds: how many passes the loop runs, and how large the
|
|
71
|
+
* body's input and output are. A hundred-pass loop over documents would put megabytes in a column
|
|
72
|
+
* that the run tree, the timeline, the canvas and the Workflows list all read — punishing every
|
|
73
|
+
* reader of the row for a detail only someone inspecting one pass will ever open.
|
|
74
|
+
*
|
|
75
|
+
* **What it protects.** Only the payloads. `promptRunID` / `agentRunID` / `actionLogID` / `success`
|
|
76
|
+
* are always recorded: those point at the durable rows where the real forensics live, and they are
|
|
77
|
+
* what cost roll-up and the timeline traverse. Losing a payload costs a reader some detail; losing a
|
|
78
|
+
* pointer would lose the pass.
|
|
79
|
+
*
|
|
80
|
+
* **Omission is stated, never silent.** Once the budget is spent, further passes record a marker
|
|
81
|
+
* saying so and how large the value was, because a pass showing nothing is indistinguishable from a
|
|
82
|
+
* pass that produced nothing — and that ambiguity is exactly the failure this whole area keeps
|
|
83
|
+
* hitting.
|
|
84
|
+
*/
|
|
85
|
+
const ITERATION_PAYLOAD_BUDGET_BYTES = 128 * 1024;
|
|
86
|
+
/** Per-value cap, so one enormous pass cannot consume the whole budget by itself. */
|
|
87
|
+
const ITERATION_PAYLOAD_VALUE_BYTES = 16 * 1024;
|
|
88
|
+
class IterationPayloadBudget {
|
|
89
|
+
constructor() {
|
|
90
|
+
this.spent = 0;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* The value if it fits, or a marker describing what was left out.
|
|
94
|
+
*
|
|
95
|
+
* @returns the value, a marker object, or undefined when there was nothing to record
|
|
96
|
+
*/
|
|
97
|
+
Take(value) {
|
|
98
|
+
if (value == null)
|
|
99
|
+
return undefined;
|
|
100
|
+
const asRecord = value && typeof value === 'object' && !Array.isArray(value)
|
|
101
|
+
? value
|
|
102
|
+
: { value };
|
|
103
|
+
let size;
|
|
104
|
+
try {
|
|
105
|
+
size = JSON.stringify(asRecord)?.length ?? 0;
|
|
106
|
+
}
|
|
107
|
+
catch {
|
|
108
|
+
// Circular or otherwise unserializable. It could not be persisted anyway, and saying so
|
|
109
|
+
// is better than a pass that silently shows nothing.
|
|
110
|
+
return { __omitted: 'unserializable' };
|
|
111
|
+
}
|
|
112
|
+
if (size > ITERATION_PAYLOAD_VALUE_BYTES) {
|
|
113
|
+
return { __omitted: 'too-large', __bytes: size, __limit: ITERATION_PAYLOAD_VALUE_BYTES };
|
|
114
|
+
}
|
|
115
|
+
if (this.spent + size > ITERATION_PAYLOAD_BUDGET_BYTES) {
|
|
116
|
+
return { __omitted: 'budget-exhausted', __bytes: size, __limit: ITERATION_PAYLOAD_BUDGET_BYTES };
|
|
117
|
+
}
|
|
118
|
+
this.spent += size;
|
|
119
|
+
return asRecord;
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
/** Deep-merges a prompt's JSON response into the payload, preserving what earlier steps established. */
|
|
123
|
+
function deepMergePayload(base, incoming) {
|
|
124
|
+
const out = { ...base };
|
|
125
|
+
for (const [key, value] of Object.entries(incoming)) {
|
|
126
|
+
const existing = out[key];
|
|
127
|
+
const bothPlainObjects = existing && typeof existing === 'object' && !Array.isArray(existing) &&
|
|
128
|
+
value && typeof value === 'object' && !Array.isArray(value);
|
|
129
|
+
out[key] = bothPlainObjects
|
|
130
|
+
? deepMergePayload(existing, value)
|
|
131
|
+
: value;
|
|
132
|
+
}
|
|
133
|
+
return out;
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Statuses at which an origin's outgoing conditions may be decided.
|
|
137
|
+
*
|
|
138
|
+
* `Skipped` is included: a branch that was not taken IS settled, and a condition on an edge leaving
|
|
139
|
+
* it should resolve rather than hang the graph forever.
|
|
140
|
+
*/
|
|
141
|
+
const TERMINAL_FOR_CONDITIONS = new Set([
|
|
142
|
+
'Complete', 'Failed', 'Cancelled', 'Skipped',
|
|
143
|
+
]);
|
|
39
144
|
export class TaskGraphDispatcher {
|
|
40
145
|
constructor(providerFactory, agentRunner, contextUser, config,
|
|
41
146
|
/**
|
|
@@ -48,12 +153,24 @@ export class TaskGraphDispatcher {
|
|
|
48
153
|
* Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
|
|
49
154
|
* announces nothing.
|
|
50
155
|
*/
|
|
51
|
-
observer
|
|
156
|
+
observer,
|
|
157
|
+
/**
|
|
158
|
+
* Optional. Absent means this host cannot run action nodes; they stay Pending and visible
|
|
159
|
+
* rather than being failed, because "nobody here can run this" is not "this ran and broke".
|
|
160
|
+
*/
|
|
161
|
+
actionRunner,
|
|
162
|
+
/**
|
|
163
|
+
* Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
|
|
164
|
+
* rather than being failed, for the same reason action nodes do.
|
|
165
|
+
*/
|
|
166
|
+
promptRunner) {
|
|
52
167
|
this.providerFactory = providerFactory;
|
|
53
168
|
this.agentRunner = agentRunner;
|
|
54
169
|
this.contextUser = contextUser;
|
|
55
170
|
this.continuationDeliverer = continuationDeliverer;
|
|
56
171
|
this.observer = observer;
|
|
172
|
+
this.actionRunner = actionRunner;
|
|
173
|
+
this.promptRunner = promptRunner;
|
|
57
174
|
this.running = false;
|
|
58
175
|
this.pollTimer = null;
|
|
59
176
|
this.reconcileTimer = null;
|
|
@@ -61,6 +178,15 @@ export class TaskGraphDispatcher {
|
|
|
61
178
|
this.inFlight = new Set();
|
|
62
179
|
/** Guards against a slow poll overlapping the next tick. */
|
|
63
180
|
this.polling = false;
|
|
181
|
+
/**
|
|
182
|
+
* The poll pass currently running, so `Stop` can wait for it.
|
|
183
|
+
*
|
|
184
|
+
* `clearInterval` cannot cancel a tick that has already fired, and a pass is a long sequence of
|
|
185
|
+
* awaits (provider, rollup, claim query) — so without this, `Stop` returns while a pass is still
|
|
186
|
+
* mid-flight and about to claim. Its tasks then land in `inFlight` AFTER the drain loop already
|
|
187
|
+
* saw an empty set, which is precisely the state the drain exists to prevent.
|
|
188
|
+
*/
|
|
189
|
+
this.pollPass = null;
|
|
64
190
|
/** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
|
|
65
191
|
this.ownerByParentID = new Map();
|
|
66
192
|
/** Name shown in the shutdown drain log. */
|
|
@@ -134,7 +260,7 @@ export class TaskGraphDispatcher {
|
|
|
134
260
|
ShutdownRegistry.Instance.Register(this);
|
|
135
261
|
LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
|
|
136
262
|
await this.Reconcile();
|
|
137
|
-
this.pollTimer = setInterval(() => {
|
|
263
|
+
this.pollTimer = setInterval(() => { this.pollPass = this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
|
|
138
264
|
this.reconcileTimer = setInterval(() => { void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
|
|
139
265
|
}
|
|
140
266
|
/**
|
|
@@ -154,6 +280,14 @@ export class TaskGraphDispatcher {
|
|
|
154
280
|
clearInterval(this.reconcileTimer);
|
|
155
281
|
this.reconcileTimer = null;
|
|
156
282
|
}
|
|
283
|
+
// Drain the poll pass BEFORE the task drain below, not after: a pass still running has not
|
|
284
|
+
// necessarily claimed anything yet, so `inFlight` can be empty while work is moments from
|
|
285
|
+
// starting. Clearing `running` above stops that pass claiming anything further; this waits
|
|
286
|
+
// for it to notice. Its own failures are already logged inside `pollOnce`.
|
|
287
|
+
if (this.pollPass) {
|
|
288
|
+
await this.pollPass.catch(() => undefined);
|
|
289
|
+
this.pollPass = null;
|
|
290
|
+
}
|
|
157
291
|
const deadline = Date.now() + 30_000;
|
|
158
292
|
while (this.inFlight.size > 0 && Date.now() < deadline) {
|
|
159
293
|
await new Promise((r) => setTimeout(r, 250));
|
|
@@ -204,11 +338,25 @@ export class TaskGraphDispatcher {
|
|
|
204
338
|
this.polling = true;
|
|
205
339
|
try {
|
|
206
340
|
const provider = await this.providerFactory.CreateProvider();
|
|
341
|
+
// `running` is re-read after every await from here on. The entry check above only proves
|
|
342
|
+
// the dispatcher was live when the tick fired; each await is a point where `Stop` can
|
|
343
|
+
// land, and a stopped instance must neither mutate graph state nor take new work. Left
|
|
344
|
+
// unchecked, a stopped dispatcher goes on to roll up graphs (emitting GraphSettled to an
|
|
345
|
+
// observer nobody is listening to any more) and to claim tasks it will never run — which
|
|
346
|
+
// then sit claimed until their lease expires.
|
|
347
|
+
if (!this.running)
|
|
348
|
+
return;
|
|
207
349
|
// Settle graphs before picking new work, so a failure earlier in this pass stops its
|
|
208
350
|
// branch immediately rather than after another wave has already launched.
|
|
209
351
|
await this.propagateAndRollup(provider);
|
|
352
|
+
if (!this.running)
|
|
353
|
+
return;
|
|
210
354
|
const candidates = await this.findClaimableTasks(provider, capacity);
|
|
211
355
|
for (const task of candidates) {
|
|
356
|
+
// Re-checked per iteration, not just before the loop: claiming is itself awaited, so
|
|
357
|
+
// a multi-task wave can straddle a Stop.
|
|
358
|
+
if (!this.running)
|
|
359
|
+
break;
|
|
212
360
|
if (this.inFlight.size >= this.config.MaxConcurrentTasks)
|
|
213
361
|
break;
|
|
214
362
|
if (!(await this.claims.TryClaim(provider, task.ID, this.contextUser))) {
|
|
@@ -265,19 +413,19 @@ export class TaskGraphDispatcher {
|
|
|
265
413
|
LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
|
|
266
414
|
}
|
|
267
415
|
}
|
|
268
|
-
const result = await this.
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
});
|
|
416
|
+
const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs);
|
|
417
|
+
// A prompt can end the workflow early and say why. Honour it before recording the
|
|
418
|
+
// outcome, so the remaining tasks are already Skipped by the time the rollup runs and
|
|
419
|
+
// the graph settles Complete rather than looking abandoned with work left Pending.
|
|
420
|
+
if (result.ChatMessage) {
|
|
421
|
+
await this.endGraphEarly(provider, task, result.ChatMessage);
|
|
422
|
+
}
|
|
276
423
|
const recorded = await this.claims.CompleteClaimed(provider, taskID, {
|
|
277
424
|
Status: result.Success ? 'Complete' : 'Failed',
|
|
278
425
|
OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
|
|
279
426
|
ErrorMessage: result.ErrorMessage ?? null,
|
|
280
427
|
AgentRunID: result.AgentRunID ?? null,
|
|
428
|
+
Configuration: this.configurationWithRuntime(task, result.PromptRunID, result.ActionLogID, result.Iterations, result.PayloadAtStart),
|
|
281
429
|
}, this.contextUser);
|
|
282
430
|
if (!recorded) {
|
|
283
431
|
// The guarded write refused: the row changed underneath us (cancelled, reassigned,
|
|
@@ -320,10 +468,62 @@ export class TaskGraphDispatcher {
|
|
|
320
468
|
*/
|
|
321
469
|
async propagateAndRollup(provider) {
|
|
322
470
|
for (const parentID of await this.findActiveGraphIDs(provider)) {
|
|
471
|
+
// Human steps settle BEFORE the graph state is read, so an answer given since the last
|
|
472
|
+
// poll is already reflected when eligibility and rollup are computed. Doing it after
|
|
473
|
+
// would delay every dependent branch by a full poll interval for no reason — and on a
|
|
474
|
+
// graph whose only remaining work is downstream of a person, that is the difference
|
|
475
|
+
// between "answered and moving" and "answered and apparently still stuck".
|
|
476
|
+
await this.expireOverdueRequests(provider, parentID);
|
|
477
|
+
await this.settleAnsweredHumanTasks(provider, parentID);
|
|
478
|
+
await this.reopenCancelledHumanTasks(provider, parentID);
|
|
323
479
|
const graph = await this.loadGraphState(provider, parentID);
|
|
324
480
|
if (graph.nodes.length === 0)
|
|
325
481
|
continue;
|
|
326
|
-
|
|
482
|
+
// SKIPS FIRST — before blocking, before eligibility. A task whose gating predecessors
|
|
483
|
+
// are all Skipped is simultaneously "eligible" (Skipped satisfies a prerequisite) and
|
|
484
|
+
// "to be skipped"; deciding eligibility first would dispatch the branch nobody took.
|
|
485
|
+
//
|
|
486
|
+
// `unreachableTaskIDs` seeds this too, and that is a correction (R6). A target whose only
|
|
487
|
+
// route in was an ordinary conditional edge that evaluated DEFINITELY FALSE is a branch
|
|
488
|
+
// that was not taken — semantically identical to an XOR loser — yet it used to settle
|
|
489
|
+
// `Blocked`. That made `Blocked` mean two unrelated things: "the workflow chose another
|
|
490
|
+
// route" and "something upstream broke". A reader cannot tell those apart, so every
|
|
491
|
+
// conditional workflow looked half-failed and people went hunting for bugs that did not
|
|
492
|
+
// exist. `Blocked` is now reserved for FAILURE-driven unsatisfiability.
|
|
493
|
+
const skipSeeds = new Set([...graph.skipSeedTaskIDs, ...graph.unreachableTaskIDs]);
|
|
494
|
+
const toSkip = new Set([
|
|
495
|
+
...skipSeeds,
|
|
496
|
+
...ComputeSkipCascade(graph.nodes, graph.edges, [...skipSeeds]),
|
|
497
|
+
]);
|
|
498
|
+
for (const taskID of toSkip) {
|
|
499
|
+
const entity = graph.entityById.get(taskID);
|
|
500
|
+
if (!entity || entity.Status !== 'Pending')
|
|
501
|
+
continue;
|
|
502
|
+
entity.Status = 'Skipped';
|
|
503
|
+
if (await entity.Save()) {
|
|
504
|
+
LogStatus(`[TaskGraphDispatcher] Skipped '${entity.Name}' (${taskID}) — another branch was taken.`);
|
|
505
|
+
// Announced separately from TaskBlocked because it means something different to
|
|
506
|
+
// a viewer: nothing went wrong, this route simply was not the one chosen.
|
|
507
|
+
this.emit({
|
|
508
|
+
Kind: 'TaskSkipped',
|
|
509
|
+
ParentTaskID: parentID,
|
|
510
|
+
OwnerUserID: await this.resolveOwner(provider, parentID),
|
|
511
|
+
TaskID: taskID,
|
|
512
|
+
TaskName: entity.Name,
|
|
513
|
+
Status: 'Skipped',
|
|
514
|
+
});
|
|
515
|
+
// Keep the in-memory graph consistent so the blocking pass below and the rollup
|
|
516
|
+
// both see the skip rather than a stale Pending.
|
|
517
|
+
const node = graph.nodes.find((n) => n.id === taskID);
|
|
518
|
+
if (node)
|
|
519
|
+
node.status = 'Skipped';
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
// Only failure-driven unsatisfiability reaches here now; not-taken branches were skipped
|
|
523
|
+
// above. A task already Skipped is left alone rather than overwritten — the two passes
|
|
524
|
+
// must not fight over the same row.
|
|
525
|
+
const toBlock = [...ComputeTasksToBlock(graph.nodes, graph.edges, graph.handledFailureIDs)]
|
|
526
|
+
.filter((id) => !toSkip.has(id));
|
|
327
527
|
for (const taskID of toBlock) {
|
|
328
528
|
const entity = graph.entityById.get(taskID);
|
|
329
529
|
if (!entity)
|
|
@@ -353,11 +553,26 @@ export class TaskGraphDispatcher {
|
|
|
353
553
|
// The outer guard covered the first load only.
|
|
354
554
|
if (fresh.nodes.length === 0)
|
|
355
555
|
continue;
|
|
356
|
-
const rollup = ComputeParentRollup(fresh.nodes);
|
|
556
|
+
const rollup = ComputeParentRollup(fresh.nodes, fresh.handledFailureIDs);
|
|
357
557
|
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
358
558
|
if (!(await parent.Load(parentID)))
|
|
359
559
|
continue;
|
|
360
|
-
|
|
560
|
+
// A graph starts when its first step does.
|
|
561
|
+
//
|
|
562
|
+
// `StartedAt` is stamped by the CLAIM, and a parent is never claimed — it is a container,
|
|
563
|
+
// not a unit of work — so the graph row carried no start time even after it completed.
|
|
564
|
+
// A settled workflow therefore reported a CompletedAt with no beginning: it sorted as
|
|
565
|
+
// "not started" in the run tree, showed no timestamp, and no duration could be computed
|
|
566
|
+
// for the thing whose duration people actually ask about.
|
|
567
|
+
//
|
|
568
|
+
// Taken from the earliest child rather than from the clock, because that is when work
|
|
569
|
+
// genuinely began — a graph can sit Pending for a long time between submission (already
|
|
570
|
+
// recorded as CreatedAt) and a dispatcher picking up its first task.
|
|
571
|
+
const earliestChildStart = this.earliestStart(fresh.entityById);
|
|
572
|
+
const startedAtChanged = parent.StartedAt == null && earliestChildStart != null;
|
|
573
|
+
if (startedAtChanged)
|
|
574
|
+
parent.StartedAt = earliestChildStart;
|
|
575
|
+
if (startedAtChanged || parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
|
|
361
576
|
parent.Status = rollup.status;
|
|
362
577
|
parent.PercentComplete = rollup.percentComplete;
|
|
363
578
|
if (rollup.isTerminal)
|
|
@@ -365,6 +580,8 @@ export class TaskGraphDispatcher {
|
|
|
365
580
|
await parent.Save();
|
|
366
581
|
}
|
|
367
582
|
if (rollup.isTerminal) {
|
|
583
|
+
// Geometry is settled once, here, so every viewer of this run agrees on it.
|
|
584
|
+
await this.persistComputedLayout(fresh);
|
|
368
585
|
// Emitted before the continuation is delivered, and outside its once-only guard: a
|
|
369
586
|
// viewer watching the run should learn it finished whether or not this instance is
|
|
370
587
|
// the one that wins the delivery CAS.
|
|
@@ -376,10 +593,295 @@ export class TaskGraphDispatcher {
|
|
|
376
593
|
CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
|
|
377
594
|
TotalCount: fresh.nodes.length,
|
|
378
595
|
});
|
|
596
|
+
await this.rollUpCostToSubmittingRun(provider, parent);
|
|
597
|
+
// Deliberately AFTER the rollup and OUTSIDE its refusal paths. The rollup declines
|
|
598
|
+
// to write a number it cannot stand behind — a truncated tree, an unreachable graph
|
|
599
|
+
// — and every one of those returns early. If the run's lifecycle were settled in
|
|
600
|
+
// there, a refused rollup would strand the run parked forever, which is a far worse
|
|
601
|
+
// failure than a missing cost figure. Cost and lifecycle are separate concerns with
|
|
602
|
+
// separate failure modes, so they get separate writes.
|
|
603
|
+
await this.settleSubmittingRun(provider, parent, rollup.status);
|
|
379
604
|
await this.deliverContinuation(provider, parent, fresh);
|
|
380
605
|
}
|
|
381
606
|
}
|
|
382
607
|
}
|
|
608
|
+
/**
|
|
609
|
+
* Credits a finished graph's spending back to the agent run that submitted it.
|
|
610
|
+
*
|
|
611
|
+
* **Why this cannot happen during the run.** `BaseAgent` totals a run by walking its steps in
|
|
612
|
+
* memory at finalization — but a submitting run *ends at submission*. Submit-and-detach is the
|
|
613
|
+
* point: the run returns as soon as the graph is durable, and the graph executes afterwards,
|
|
614
|
+
* possibly minutes later on a different instance. At the moment the run computes its totals the
|
|
615
|
+
* spending has not happened yet, so there is nothing to count. The only place the number can be
|
|
616
|
+
* known is here, when the graph settles.
|
|
617
|
+
*
|
|
618
|
+
* **Why the `…Rollup` columns and not the plain ones.** `AIAgentRun` has carried six `…Rollup`
|
|
619
|
+
* columns since v3 that nothing has ever written — they exist for exactly this distinction:
|
|
620
|
+
*
|
|
621
|
+
* - `TotalCost` — what the run itself spent. For a Flow agent that is genuinely near zero: it
|
|
622
|
+
* compiled a graph and handed it off. This value is already final and is never rewritten here,
|
|
623
|
+
* so nothing that reads it today changes meaning, and no guardrail that already evaluated
|
|
624
|
+
* against it is retroactively falsified.
|
|
625
|
+
* - `TotalCostRollup` — the run plus everything it caused. Provisional until the graph settles,
|
|
626
|
+
* which is now.
|
|
627
|
+
*
|
|
628
|
+
* **The tree is the authority; these columns are its settlement-time cache.** The total is a SUM
|
|
629
|
+
* over `GetAgentRunTree`, not arithmetic of its own. The previous version walked the graph's
|
|
630
|
+
* child tasks and added each one's agent run, which was wrong in two ways that no test could
|
|
631
|
+
* see: a `Prompt` task has no agent run at all, so every prompt step's spend was simply missing;
|
|
632
|
+
* and it read each nested run's `…Rollup ?? …Total`, mixing a descendant-inclusive number with an
|
|
633
|
+
* own-spend one and depending on whether that nested graph happened to have settled yet. The
|
|
634
|
+
* tree already models every one of those cases — it reaches prompt runs through
|
|
635
|
+
* `Configuration.runtime.promptRunID`, and it descends into nested runs and their graphs
|
|
636
|
+
* structurally — so summing it cannot disagree with what the run viewer shows, because it IS
|
|
637
|
+
* what the run viewer shows.
|
|
638
|
+
*
|
|
639
|
+
* **This refuses rather than guesses.** A tree that failed to load, hit the depth cap, or does
|
|
640
|
+
* not contain the settling graph would still produce a number — a lower bound. Writing one would
|
|
641
|
+
* put an authoritative-looking total in a column every cost surface reads. Each of those cases
|
|
642
|
+
* logs and leaves the column alone, so `?? TotalCost` keeps its honest meaning: not settled.
|
|
643
|
+
*
|
|
644
|
+
* A graph with no submitting run (a scheduled job, a remote-operation caller) simply has nobody
|
|
645
|
+
* to credit — its own Task rows still carry the truth, and this returns quietly.
|
|
646
|
+
*/
|
|
647
|
+
async rollUpCostToSubmittingRun(provider, parent) {
|
|
648
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
649
|
+
if (!meta.submittedByAgentRunID)
|
|
650
|
+
return;
|
|
651
|
+
const runID = meta.submittedByAgentRunID;
|
|
652
|
+
try {
|
|
653
|
+
const runQuery = asRunQueryProvider(provider);
|
|
654
|
+
if (!runQuery) {
|
|
655
|
+
LogError(`[TaskGraphDispatcher] Cannot roll up cost for run ${runID}: provider cannot run queries.`);
|
|
656
|
+
return;
|
|
657
|
+
}
|
|
658
|
+
const tree = await LoadAgentRunTree(runID, runQuery, this.contextUser);
|
|
659
|
+
// Each of these means the sum would be a LOWER BOUND, and the column's whole contract is
|
|
660
|
+
// that it equals the tree. A known-low number presented as a total is worse than no
|
|
661
|
+
// number: the readers all fall back to TotalCost when this is null, which at least
|
|
662
|
+
// *says* it is the run's own spend rather than claiming to be the whole story.
|
|
663
|
+
//
|
|
664
|
+
// Refusing is NOT the same as leaving the column alone. A run that submitted two graphs
|
|
665
|
+
// has a rollup from the first; if the second cannot be summed, the first graph's total
|
|
666
|
+
// sits in the authoritative column excluding work that has since happened — stale, not
|
|
667
|
+
// absent, and `?? TotalCost` cannot save a reader from a non-null wrong number. So a
|
|
668
|
+
// refusal CLEARS it, restoring the fallback's honest meaning: not settled.
|
|
669
|
+
if (tree.ErrorMessage || !tree.Root) {
|
|
670
|
+
await this.clearStaleRollup(provider, runID, tree.ErrorMessage ?? 'the run tree came back empty');
|
|
671
|
+
return;
|
|
672
|
+
}
|
|
673
|
+
if (tree.Truncated) {
|
|
674
|
+
await this.clearStaleRollup(provider, runID, `the run tree hit the depth cap, so any total would silently under-report ` +
|
|
675
|
+
`(graph ${parent.ID} still carries its own costs)`);
|
|
676
|
+
return;
|
|
677
|
+
}
|
|
678
|
+
// The graph that just settled must appear in the tree. If it does not, the tree stopped
|
|
679
|
+
// at the run — the submitting step never recorded its parentTaskID — and the sum is
|
|
680
|
+
// merely the run's own spend wearing the name of a rollup. That is precisely the silent
|
|
681
|
+
// under-count this rewrite exists to remove, so it is reported rather than written.
|
|
682
|
+
if (!this.treeContainsGraph(tree.Root, parent.ID)) {
|
|
683
|
+
await this.clearStaleRollup(provider, runID, `graph ${parent.ID} is not reachable from it, so the tree cannot see the work. ` +
|
|
684
|
+
`Did the submitting step record parentTaskID?`);
|
|
685
|
+
return;
|
|
686
|
+
}
|
|
687
|
+
const totals = SumAgentRunTreeCost(tree.Root);
|
|
688
|
+
const submitting = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
689
|
+
if (!(await submitting.Load(runID))) {
|
|
690
|
+
LogError(`[TaskGraphDispatcher] Could not load run ${runID} to record graph cost against it.`);
|
|
691
|
+
return;
|
|
692
|
+
}
|
|
693
|
+
// Assignment, never accumulation. The tree already contains the run's own spend as its
|
|
694
|
+
// ROOT node, and it reads own-cost everywhere, so recomputing from scratch on every
|
|
695
|
+
// settlement lands on the same answer — which is what makes this safe to call again when
|
|
696
|
+
// a second graph settles, or when the terminal check is re-evaluated after a HITL wait.
|
|
697
|
+
submitting.TotalCostRollup = totals.Cost;
|
|
698
|
+
submitting.TotalTokensUsedRollup = totals.Tokens;
|
|
699
|
+
submitting.TotalPromptTokensUsedRollup = totals.PromptTokens;
|
|
700
|
+
submitting.TotalCompletionTokensUsedRollup = totals.CompletionTokens;
|
|
701
|
+
if (!(await submitting.Save())) {
|
|
702
|
+
LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}: ` +
|
|
703
|
+
`${submitting.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
704
|
+
return;
|
|
705
|
+
}
|
|
706
|
+
LogStatus(`[TaskGraphDispatcher] Credited graph ${parent.ID} to run ${runID}: ` +
|
|
707
|
+
`${tree.Rows.length} node(s), ${totals.Tokens} token(s), cost ${totals.Cost}.`);
|
|
708
|
+
}
|
|
709
|
+
catch (e) {
|
|
710
|
+
// A failed rollup must never fail the graph. The work finished; only the accounting for
|
|
711
|
+
// it is missing, and a graph marked Failed because its cost could not be summed would be
|
|
712
|
+
// a far worse lie than a cost of null.
|
|
713
|
+
LogError(`[TaskGraphDispatcher] Cost rollup failed for graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
714
|
+
}
|
|
715
|
+
}
|
|
716
|
+
/**
|
|
717
|
+
* Clears a rollup that can no longer be trusted, and says why.
|
|
718
|
+
*
|
|
719
|
+
* **Why clear rather than leave.** The four `…Rollup` columns are a cache of the run tree, and
|
|
720
|
+
* every reader treats a value there as the total. When the tree cannot be summed, any value
|
|
721
|
+
* already in the column was computed from an EARLIER settlement — it excludes the graph that
|
|
722
|
+
* just finished, so it is not merely incomplete, it is a wrong total presented as a right one.
|
|
723
|
+
* `?? TotalCost` protects a reader from null, not from stale.
|
|
724
|
+
*
|
|
725
|
+
* Nulling restores the invariant this whole design rests on: **when the column is present, it
|
|
726
|
+
* equals the tree.** Absent means not settled, which is exactly what a reader should conclude.
|
|
727
|
+
* A run with no rollup yet is untouched — there is nothing stale to clear, and writing nulls
|
|
728
|
+
* over nulls would churn Record Changes for nothing.
|
|
729
|
+
*/
|
|
730
|
+
async clearStaleRollup(provider, runID, reason) {
|
|
731
|
+
LogError(`[TaskGraphDispatcher] Not recording cost for run ${runID}: ${reason}.`);
|
|
732
|
+
try {
|
|
733
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
734
|
+
if (!(await run.Load(runID)))
|
|
735
|
+
return;
|
|
736
|
+
if (run.TotalCostRollup == null && run.TotalTokensUsedRollup == null)
|
|
737
|
+
return; // nothing stale
|
|
738
|
+
run.TotalCostRollup = null;
|
|
739
|
+
run.TotalTokensUsedRollup = null;
|
|
740
|
+
run.TotalPromptTokensUsedRollup = null;
|
|
741
|
+
run.TotalCompletionTokensUsedRollup = null;
|
|
742
|
+
if (!(await run.Save())) {
|
|
743
|
+
LogError(`[TaskGraphDispatcher] Could not clear the now-stale rollup on run ${runID}: ` +
|
|
744
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It still shows a total that ` +
|
|
745
|
+
`excludes the graph that just settled.`);
|
|
746
|
+
return;
|
|
747
|
+
}
|
|
748
|
+
LogStatus(`[TaskGraphDispatcher] Cleared the rollup on run ${runID}: it was computed before this ` +
|
|
749
|
+
`graph settled and can no longer be recomputed, so it would have under-reported.`);
|
|
750
|
+
}
|
|
751
|
+
catch (e) {
|
|
752
|
+
LogError(`[TaskGraphDispatcher] Could not clear the rollup on run ${runID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
/**
|
|
756
|
+
* Whether the settling graph is actually reachable from the submitting run's tree.
|
|
757
|
+
*
|
|
758
|
+
* Matched on the graph's parent Task id, which is the node the `TaskGraph` member of the query
|
|
759
|
+
* emits. A run that submitted a graph but recorded no `parentTaskID` produces a tree that stops
|
|
760
|
+
* at the run — structurally indistinguishable, at the SUM, from a run that never dispatched
|
|
761
|
+
* anything. This is the check that tells those two apart.
|
|
762
|
+
*/
|
|
763
|
+
treeContainsGraph(root, parentTaskID) {
|
|
764
|
+
for (const node of WalkAgentRunTree(root)) {
|
|
765
|
+
if (node.NodeType === 'TaskGraph' && UUIDsEqual(node.NodeID, parentTaskID))
|
|
766
|
+
return true;
|
|
767
|
+
}
|
|
768
|
+
return false;
|
|
769
|
+
}
|
|
770
|
+
/**
|
|
771
|
+
* Ends a graph early because a prompt said the work is finished.
|
|
772
|
+
*
|
|
773
|
+
* **Why `Skipped` and not `Cancelled`.** Nothing went wrong and nobody intervened — the workflow
|
|
774
|
+
* reached its own conclusion before running every drawn step, which is exactly what a reasoning
|
|
775
|
+
* step is for. `Cancelled` would tell a reader someone stopped it; `Skipped` says these routes
|
|
776
|
+
* were not taken, which is true and already the vocabulary the fork machinery uses.
|
|
777
|
+
*
|
|
778
|
+
* The message is written to the parent so the graph carries its own answer, rather than the
|
|
779
|
+
* answer living only on the step that produced it.
|
|
780
|
+
*/
|
|
781
|
+
async endGraphEarly(provider, task, message) {
|
|
782
|
+
if (!task.ParentID)
|
|
783
|
+
return;
|
|
784
|
+
try {
|
|
785
|
+
LogStatus(`[TaskGraphDispatcher] '${task.Name}' ended the workflow early: ${message}`);
|
|
786
|
+
for (const sibling of await this.loadChildTasks(provider, task.ParentID)) {
|
|
787
|
+
if (sibling.ID === task.ID || sibling.Status !== 'Pending')
|
|
788
|
+
continue;
|
|
789
|
+
sibling.Status = 'Skipped';
|
|
790
|
+
if (await sibling.Save()) {
|
|
791
|
+
this.emit({
|
|
792
|
+
Kind: 'TaskSkipped',
|
|
793
|
+
ParentTaskID: task.ParentID,
|
|
794
|
+
OwnerUserID: await this.resolveOwner(provider, task.ParentID),
|
|
795
|
+
TaskID: sibling.ID,
|
|
796
|
+
TaskName: sibling.Name,
|
|
797
|
+
Status: 'Skipped',
|
|
798
|
+
});
|
|
799
|
+
}
|
|
800
|
+
}
|
|
801
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
802
|
+
if (await parent.Load(task.ParentID)) {
|
|
803
|
+
parent.OutputPayload = JSON.stringify({ message });
|
|
804
|
+
await parent.Save();
|
|
805
|
+
}
|
|
806
|
+
}
|
|
807
|
+
catch (e) {
|
|
808
|
+
// The work itself succeeded; only the early-finish bookkeeping failed. Failing the task
|
|
809
|
+
// over that would discard a completed step's result.
|
|
810
|
+
LogError(`[TaskGraphDispatcher] Could not end graph early for ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
/**
|
|
814
|
+
* How deep the continuation chain already is, read from the graph's parent metadata.
|
|
815
|
+
*
|
|
816
|
+
* A run started by a graph inherits that graph's depth **plus one**. Without this every spawned
|
|
817
|
+
* run begins at zero, so a self-referencing flow — one that dispatches a graph containing itself
|
|
818
|
+
* — recurses without bound while the cap it should be hitting compares against a permanent zero.
|
|
819
|
+
*/
|
|
820
|
+
async graphContext(provider, task) {
|
|
821
|
+
if (!task.ParentID)
|
|
822
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
823
|
+
try {
|
|
824
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
825
|
+
if (!(await parent.Load(task.ParentID)))
|
|
826
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
827
|
+
return {
|
|
828
|
+
Depth: ParseTaskGraphParentMetadata(parent.InputPayload).reinvokeDepth + 1,
|
|
829
|
+
// The graph's own row carries the run that submitted it. One load answers both
|
|
830
|
+
// questions, which is why they are resolved together rather than in two passes.
|
|
831
|
+
SubmittingAgentRunID: parent.AgentRunID,
|
|
832
|
+
};
|
|
833
|
+
}
|
|
834
|
+
catch {
|
|
835
|
+
// An unreadable parent must not stop the work; depth zero is the safe reading, and the
|
|
836
|
+
// submit-time cap still guards the next hop.
|
|
837
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
838
|
+
}
|
|
839
|
+
}
|
|
840
|
+
/**
|
|
841
|
+
* Which failures the workflow drew a way out of.
|
|
842
|
+
*
|
|
843
|
+
* A Failed task with a **satisfied outgoing edge** is a handled failure: its author drew a
|
|
844
|
+
* recovery route and that route is now live. Downstream work should be released along it, and the
|
|
845
|
+
* parent should not roll up Failed because of a step the workflow explicitly planned around.
|
|
846
|
+
*
|
|
847
|
+
* Scoped to `failureSemantics: 'edges'` on purpose. Under `'block'` — every agent-emitted graph —
|
|
848
|
+
* a failure is terminal for its dependents whatever edges exist, because nobody drew those edges
|
|
849
|
+
* as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
|
|
850
|
+
* graph sail past a failure it never anticipated.
|
|
851
|
+
*/
|
|
852
|
+
async computeHandledFailures(provider, parentTaskID, nodes, edges) {
|
|
853
|
+
const handled = new Set();
|
|
854
|
+
// Cheap exit before touching the database: with no failures there is nothing to handle, and
|
|
855
|
+
// this runs on every poll for every active graph.
|
|
856
|
+
if (!nodes.some((n) => n.status === 'Failed'))
|
|
857
|
+
return handled;
|
|
858
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
859
|
+
if (!(await parent.Load(parentTaskID)))
|
|
860
|
+
return handled;
|
|
861
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
862
|
+
if (meta.failureSemantics !== 'edges')
|
|
863
|
+
return handled;
|
|
864
|
+
for (const node of nodes) {
|
|
865
|
+
if (node.status !== 'Failed')
|
|
866
|
+
continue;
|
|
867
|
+
// "Has somewhere to go" is the test. An edge out of a failed step that survived condition
|
|
868
|
+
// evaluation IS the drawn recovery route; a failed step with no outgoing edges has none,
|
|
869
|
+
// and stays terminal.
|
|
870
|
+
if (edges.some((e) => e.dependsOnTaskId === node.id))
|
|
871
|
+
handled.add(node.id);
|
|
872
|
+
}
|
|
873
|
+
return handled;
|
|
874
|
+
}
|
|
875
|
+
/** The graph's child tasks, with the fields the rollup needs. */
|
|
876
|
+
async loadChildTasks(provider, parentID) {
|
|
877
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
878
|
+
EntityName: 'MJ: Tasks',
|
|
879
|
+
ExtraFilter: `ParentID='${parentID}'`,
|
|
880
|
+
ResultType: 'entity_object',
|
|
881
|
+
BypassCache: true,
|
|
882
|
+
}, this.contextUser);
|
|
883
|
+
return (result.Success ? result.Results : []) ?? [];
|
|
884
|
+
}
|
|
383
885
|
/**
|
|
384
886
|
* Runs the graph's continuation exactly once, now that it has settled.
|
|
385
887
|
*
|
|
@@ -498,8 +1000,29 @@ export class TaskGraphDispatcher {
|
|
|
498
1000
|
async notifyHumanTaskReady(task, provider) {
|
|
499
1001
|
if (task.ClaimedBy === HUMAN_TASK_NOTIFIED_MARKER)
|
|
500
1002
|
return;
|
|
501
|
-
|
|
502
|
-
|
|
1003
|
+
// The REQUEST is raised whether or not the task names an assignee. An unassigned human step
|
|
1004
|
+
// is a legitimate "somebody needs to look at this", and a request nobody was notified about
|
|
1005
|
+
// is still findable in the inbox — whereas returning early here is how such a step used to
|
|
1006
|
+
// become invisible work that stalled a workflow with nothing anywhere saying why.
|
|
1007
|
+
// TRANSIENT failures retry; PERMANENT ones stop. That distinction is the whole point, and
|
|
1008
|
+
// getting it wrong took a server down: retrying unconditionally meant a task whose workflow
|
|
1009
|
+
// has no owning agent — which can never succeed — was re-attempted on every poll forever,
|
|
1010
|
+
// each pass re-reading the graph, until the process was OOM-killed. The marker exists to
|
|
1011
|
+
// prevent exactly that storm; a permanent failure has to set it.
|
|
1012
|
+
const raised = await this.raiseHumanRequest(task, provider);
|
|
1013
|
+
if (raised === 'transient-failure')
|
|
1014
|
+
return; // try again next poll
|
|
1015
|
+
if (raised === 'permanent-failure') {
|
|
1016
|
+
// Nothing will change on a retry. Mark it so the loop stops, and leave the task Pending
|
|
1017
|
+
// and visible — a person can still see it in the Tasks UI, which is the fallback the
|
|
1018
|
+
// notification was only ever an accelerant for.
|
|
1019
|
+
await this.markHumanTaskNotified(task);
|
|
1020
|
+
return;
|
|
1021
|
+
}
|
|
1022
|
+
if (!task.UserID) {
|
|
1023
|
+
await this.markHumanTaskNotified(task);
|
|
1024
|
+
return;
|
|
1025
|
+
}
|
|
503
1026
|
try {
|
|
504
1027
|
await NotificationEngine.Instance.Config(false, this.contextUser);
|
|
505
1028
|
await NotificationEngine.Instance.SendNotification({
|
|
@@ -513,13 +1036,7 @@ export class TaskGraphDispatcher {
|
|
|
513
1036
|
catch (e) {
|
|
514
1037
|
LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
515
1038
|
}
|
|
516
|
-
|
|
517
|
-
// worse failure than one that was missed: the task remains visible in the Tasks UI either
|
|
518
|
-
// way, whereas a notification storm is not self-correcting.
|
|
519
|
-
task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
|
|
520
|
-
if (!(await task.Save())) {
|
|
521
|
-
LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
|
|
522
|
-
}
|
|
1039
|
+
await this.markHumanTaskNotified(task);
|
|
523
1040
|
// Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
|
|
524
1041
|
// than appearing to stall for no reason.
|
|
525
1042
|
this.emit({
|
|
@@ -598,7 +1115,28 @@ export class TaskGraphDispatcher {
|
|
|
598
1115
|
if (claimable.length >= limit)
|
|
599
1116
|
break;
|
|
600
1117
|
const graph = await this.loadGraphState(provider, parentID);
|
|
601
|
-
|
|
1118
|
+
// HOLD is what makes "a broken condition stalls visibly" true rather than merely stated.
|
|
1119
|
+
// An undecided exclusive group keeps all its edges, and a kept edge on a Complete origin
|
|
1120
|
+
// is a SATISFIED prerequisite — so without this filter every branch of the fork would be
|
|
1121
|
+
// eligible at once and all of them would run. A typo must not multiply a fork.
|
|
1122
|
+
//
|
|
1123
|
+
// The losers of a DECIDED group must be filtered for the same reason, and this is a race
|
|
1124
|
+
// rather than a rule: they are marked Skipped by the propagation pass, but between the
|
|
1125
|
+
// moment the group resolves and the moment that write lands, their incoming edge is still
|
|
1126
|
+
// a satisfied prerequisite on a Complete origin. A poll landing in that window would
|
|
1127
|
+
// claim and execute the branch the workflow chose NOT to take — irreversibly, since the
|
|
1128
|
+
// action has already run by the time Skipped is written over it.
|
|
1129
|
+
// `unreachableTaskIDs` joins the filter for exactly the reason above. R6 made a
|
|
1130
|
+
// definite-false ordinary edge seed the skip cascade rather than Block its target — but
|
|
1131
|
+
// until that Skipped write lands, the target has no unsatisfied prerequisite and is
|
|
1132
|
+
// vacuously eligible. That is the same race the XOR fix closed, reopened on the new
|
|
1133
|
+
// path: a branch the workflow decided against, claimed and executed irreversibly in the
|
|
1134
|
+
// window before it was marked.
|
|
1135
|
+
const eligible = ComputeEligibleTasks(graph.nodes, graph.edges, graph.handledFailureIDs)
|
|
1136
|
+
.filter((n) => !graph.holdTaskIDs.has(n.id) &&
|
|
1137
|
+
!graph.skipSeedTaskIDs.has(n.id) &&
|
|
1138
|
+
!graph.unreachableTaskIDs.has(n.id));
|
|
1139
|
+
for (const node of eligible) {
|
|
602
1140
|
const entity = graph.entityById.get(node.id);
|
|
603
1141
|
if (!entity)
|
|
604
1142
|
continue;
|
|
@@ -608,7 +1146,25 @@ export class TaskGraphDispatcher {
|
|
|
608
1146
|
// they cleared. Without a notification here a workflow simply stops, waiting on
|
|
609
1147
|
// someone who was never told. That silent stall is the failure mode this exists to
|
|
610
1148
|
// prevent, so it happens on the eligibility check rather than at submission.
|
|
611
|
-
if (
|
|
1149
|
+
if (entity.ActionID) {
|
|
1150
|
+
// An action node this host has no runner for is left Pending rather than
|
|
1151
|
+
// claimed. Claiming it would take ownership of work this process cannot do, and
|
|
1152
|
+
// the claim would then have to expire before any host that CAN do it gets a
|
|
1153
|
+
// turn — a self-inflicted stall on a mixed deployment.
|
|
1154
|
+
if (!this.actionRunner)
|
|
1155
|
+
continue;
|
|
1156
|
+
}
|
|
1157
|
+
else if (entity.PromptID) {
|
|
1158
|
+
// A prompt node — including a loop that repeats a prompt — is assigned through
|
|
1159
|
+
// PromptID and carries NEITHER ActionID nor AgentID. Without this branch it fell
|
|
1160
|
+
// through to the test below and was treated as a task waiting on a PERSON: the
|
|
1161
|
+
// workflow notified a human who had nothing to do and then stopped forever.
|
|
1162
|
+
// That is precisely the misclassification the step-kind rules warn about, and it
|
|
1163
|
+
// is silent — the graph sits In Progress looking like it is still working.
|
|
1164
|
+
if (!this.promptRunner)
|
|
1165
|
+
continue;
|
|
1166
|
+
}
|
|
1167
|
+
else if (!entity.AgentID) {
|
|
612
1168
|
await this.notifyHumanTaskReady(entity, provider);
|
|
613
1169
|
continue;
|
|
614
1170
|
}
|
|
@@ -621,6 +1177,278 @@ export class TaskGraphDispatcher {
|
|
|
621
1177
|
}
|
|
622
1178
|
return claimable;
|
|
623
1179
|
}
|
|
1180
|
+
/**
|
|
1181
|
+
* Marks a human task as notified, so the request is raised exactly once.
|
|
1182
|
+
*
|
|
1183
|
+
* Written even when delivery threw. Retrying on every poll is a worse failure than one missed
|
|
1184
|
+
* notification: the task stays visible in the inbox either way, whereas a notification storm is
|
|
1185
|
+
* not self-correcting.
|
|
1186
|
+
*/
|
|
1187
|
+
async markHumanTaskNotified(task) {
|
|
1188
|
+
task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
|
|
1189
|
+
if (!(await task.Save())) {
|
|
1190
|
+
LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
|
|
1191
|
+
}
|
|
1192
|
+
}
|
|
1193
|
+
/**
|
|
1194
|
+
* Raises the `MJ: AI Agent Requests` row a person answers to release this step.
|
|
1195
|
+
*
|
|
1196
|
+
* **Why that entity rather than something new.** It already models everything a workflow's human
|
|
1197
|
+
* step needs — who is being asked, what for, a typed response schema, priority, expiry, and an
|
|
1198
|
+
* inbox surface people already use. A second HITL substrate beside it would split the inbox in
|
|
1199
|
+
* two and leave one of them without expiry or permissions.
|
|
1200
|
+
*
|
|
1201
|
+
* **What it deliberately does NOT set is `ResumingAgentRunID`.** A request normally suspends an
|
|
1202
|
+
* agent run and resumes it. A workflow needs none of that: the graph OUTLIVES the run that
|
|
1203
|
+
* submitted it, so nothing is suspended — the task sits Pending, every other branch keeps
|
|
1204
|
+
* running, and answering settles the task. That column staying null is meaningful, not missing.
|
|
1205
|
+
*/
|
|
1206
|
+
async raiseHumanRequest(task, provider) {
|
|
1207
|
+
try {
|
|
1208
|
+
const existing = await this.findOpenRequest(provider, task.ID);
|
|
1209
|
+
if (existing)
|
|
1210
|
+
return 'raised'; // already waiting on someone
|
|
1211
|
+
const request = await provider.GetEntityObject('MJ: AI Agent Requests', this.contextUser);
|
|
1212
|
+
request.NewRecord();
|
|
1213
|
+
request.OriginatingTaskID = task.ID;
|
|
1214
|
+
// A human task has NO AgentID of its own — that column names what EXECUTES a step, and
|
|
1215
|
+
// a person is not an agent. The request still needs one, so it carries the agent that
|
|
1216
|
+
// owns the workflow: the graph's own agent, which is who is asking.
|
|
1217
|
+
const owningAgentID = await this.owningAgentOf(provider, task);
|
|
1218
|
+
if (!owningAgentID) {
|
|
1219
|
+
// PERMANENT: a graph with no owning agent will not acquire one by being asked
|
|
1220
|
+
// again. Graphs submitted before the provenance stamp landed are all in this state.
|
|
1221
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} needs a person, but its workflow has no ` +
|
|
1222
|
+
`agent to ask on behalf of, so no request can be raised. The task stays Pending ` +
|
|
1223
|
+
`and visible in the Tasks UI; it will not be retried.`);
|
|
1224
|
+
return 'permanent-failure';
|
|
1225
|
+
}
|
|
1226
|
+
request.AgentID = owningAgentID;
|
|
1227
|
+
request.RequestForUserID = task.UserID;
|
|
1228
|
+
request.RequestedAt = new Date();
|
|
1229
|
+
request.Status = 'Requested';
|
|
1230
|
+
request.Request = task.Description || `A workflow is waiting on you to complete "${task.Name}".`;
|
|
1231
|
+
// The graph's own run is the provenance a reader follows back to see what led here.
|
|
1232
|
+
request.OriginatingAgentRunID = await this.submittingRunOf(provider, task);
|
|
1233
|
+
// The deadline, when the author set one. `expireOverdueRequests` has always been able to
|
|
1234
|
+
// enforce this — it expires the request and fails the step so a give-up edge can route
|
|
1235
|
+
// around it — but nothing ever WROTE the column, so that whole path had never run outside
|
|
1236
|
+
// a test and a workflow waiting on someone who left the company waited forever.
|
|
1237
|
+
// Absent means no deadline, deliberately: expiring on a timeout nobody chose would be
|
|
1238
|
+
// worse than waiting.
|
|
1239
|
+
const expiresInHours = this.parseConfiguration(task)?.human?.expiresInHours;
|
|
1240
|
+
if (expiresInHours && expiresInHours > 0) {
|
|
1241
|
+
request.ExpiresAt = new Date(Date.now() + expiresInHours * 60 * 60 * 1000);
|
|
1242
|
+
}
|
|
1243
|
+
if (!(await request.Save())) {
|
|
1244
|
+
LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ` +
|
|
1245
|
+
`${request.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
1246
|
+
// A failed SAVE may be transient (deadlock, contention), so this one earns a retry.
|
|
1247
|
+
return 'transient-failure';
|
|
1248
|
+
}
|
|
1249
|
+
return 'raised';
|
|
1250
|
+
}
|
|
1251
|
+
catch (e) {
|
|
1252
|
+
// Never fatal. The task remains Pending and visible; a missing request is recoverable,
|
|
1253
|
+
// whereas throwing here would abort the whole dispatch pass for every other branch.
|
|
1254
|
+
LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
1255
|
+
return 'transient-failure';
|
|
1256
|
+
}
|
|
1257
|
+
}
|
|
1258
|
+
/**
|
|
1259
|
+
* The agent that owns this task's workflow — who the request is asked on behalf of.
|
|
1260
|
+
*
|
|
1261
|
+
* Reads the graph's parent row, falling back to the run that submitted it. A human step has no
|
|
1262
|
+
* agent of its own by design: `AgentID` names what EXECUTES a step, and a person is not an agent.
|
|
1263
|
+
*/
|
|
1264
|
+
async owningAgentOf(provider, task) {
|
|
1265
|
+
if (task.AgentID)
|
|
1266
|
+
return task.AgentID;
|
|
1267
|
+
if (!task.ParentID)
|
|
1268
|
+
return null;
|
|
1269
|
+
try {
|
|
1270
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
1271
|
+
if (!(await parent.Load(task.ParentID)))
|
|
1272
|
+
return null;
|
|
1273
|
+
if (parent.AgentID)
|
|
1274
|
+
return parent.AgentID;
|
|
1275
|
+
if (!parent.AgentRunID)
|
|
1276
|
+
return null;
|
|
1277
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
1278
|
+
return (await run.Load(parent.AgentRunID)) ? run.AgentID : null;
|
|
1279
|
+
}
|
|
1280
|
+
catch {
|
|
1281
|
+
return null;
|
|
1282
|
+
}
|
|
1283
|
+
}
|
|
1284
|
+
/** The still-open request for a task, if one exists. */
|
|
1285
|
+
async findOpenRequest(provider, taskID) {
|
|
1286
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
1287
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
1288
|
+
ExtraFilter: `OriginatingTaskID='${taskID}' AND Status='Requested'`,
|
|
1289
|
+
ResultType: 'entity_object',
|
|
1290
|
+
BypassCache: true,
|
|
1291
|
+
}, this.contextUser);
|
|
1292
|
+
return (result.Success ? result.Results?.[0] : null) ?? null;
|
|
1293
|
+
}
|
|
1294
|
+
/**
|
|
1295
|
+
* Settles a human task from the request a person answered.
|
|
1296
|
+
*
|
|
1297
|
+
* Runs on the poll rather than on a save hook, because the answer can arrive through any surface
|
|
1298
|
+
* — the inbox, the API, a conversation — and only the dispatcher knows how to release the rest
|
|
1299
|
+
* of the graph afterwards.
|
|
1300
|
+
*
|
|
1301
|
+
* **`ResponseData` becomes the task's output.** That is what makes a human step useful rather
|
|
1302
|
+
* than a gate: a downstream edge can branch on what the person actually said, typed by the
|
|
1303
|
+
* request's own ResponseSchema. A step that only recorded "approved" would force every decision
|
|
1304
|
+
* back into a separate action.
|
|
1305
|
+
*/
|
|
1306
|
+
async settleAnsweredHumanTasks(provider, graphID) {
|
|
1307
|
+
const waiting = await RunView.FromMetadataProvider(provider).RunView({
|
|
1308
|
+
EntityName: 'MJ: Tasks',
|
|
1309
|
+
ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
|
|
1310
|
+
ResultType: 'entity_object',
|
|
1311
|
+
BypassCache: true,
|
|
1312
|
+
}, this.contextUser);
|
|
1313
|
+
if (!waiting.Success)
|
|
1314
|
+
return;
|
|
1315
|
+
for (const task of waiting.Results ?? []) {
|
|
1316
|
+
const request = await this.answeredRequestFor(provider, task.ID);
|
|
1317
|
+
if (!request)
|
|
1318
|
+
continue;
|
|
1319
|
+
const rejected = request.Status === 'Rejected';
|
|
1320
|
+
const expired = request.Status === 'Expired';
|
|
1321
|
+
task.Status = rejected || expired ? 'Failed' : 'Complete';
|
|
1322
|
+
task.CompletedAt = new Date();
|
|
1323
|
+
task.PercentComplete = rejected || expired ? 0 : 100;
|
|
1324
|
+
task.ClaimedBy = null;
|
|
1325
|
+
task.ClaimExpiresAt = null;
|
|
1326
|
+
task.OutputPayload = request.ResponseData ?? null;
|
|
1327
|
+
if (rejected) {
|
|
1328
|
+
task.ErrorMessage = request.Comments || 'A person rejected this step.';
|
|
1329
|
+
}
|
|
1330
|
+
else if (expired) {
|
|
1331
|
+
// Stated as a failure rather than left Pending. A workflow blocked forever on
|
|
1332
|
+
// someone who never answered — who may have left the company — is the silent stall
|
|
1333
|
+
// this whole path exists to avoid, and a give-up edge can now route around it.
|
|
1334
|
+
task.ErrorMessage = 'Nobody answered this step before its request expired.';
|
|
1335
|
+
}
|
|
1336
|
+
if (!(await task.Save())) {
|
|
1337
|
+
LogError(`[TaskGraphDispatcher] Could not settle human task ${task.ID}: ` +
|
|
1338
|
+
`${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
1339
|
+
}
|
|
1340
|
+
}
|
|
1341
|
+
}
|
|
1342
|
+
/**
|
|
1343
|
+
* Re-opens a human step whose request was CANCELLED.
|
|
1344
|
+
*
|
|
1345
|
+
* `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
|
|
1346
|
+
* rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
|
|
1347
|
+
* it". Nothing replaced it. `raiseHumanRequest` refuses to raise twice (the notified marker on
|
|
1348
|
+
* `ClaimedBy` is what stops the notification storm), so a cancelled request left the task Pending
|
|
1349
|
+
* with no open request and no path to acquiring one: a workflow waiting forever on a question
|
|
1350
|
+
* nobody is being asked.
|
|
1351
|
+
*
|
|
1352
|
+
* Clearing the marker is the whole fix — the next poll sees an un-notified Pending human task
|
|
1353
|
+
* and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
|
|
1354
|
+
* human action: it takes another person cancelling again to come back here.
|
|
1355
|
+
*/
|
|
1356
|
+
async reopenCancelledHumanTasks(provider, graphID) {
|
|
1357
|
+
const waiting = await RunView.FromMetadataProvider(provider).RunView({
|
|
1358
|
+
EntityName: 'MJ: Tasks',
|
|
1359
|
+
// `StepType` is NULLABLE, and rows predating the column exist (4 in the reference
|
|
1360
|
+
// database at the time of writing). None currently carry a UserID, but a human task
|
|
1361
|
+
// written by any path that set the assignee without the discriminator would be
|
|
1362
|
+
// invisible to a `StepType='Human'` filter and stay dead forever after a cancel —
|
|
1363
|
+
// the exact stall this method exists to end. The notified marker already narrows
|
|
1364
|
+
// this to tasks the dispatcher raised a request for, so the widening cannot pull in
|
|
1365
|
+
// unrelated work.
|
|
1366
|
+
ExtraFilter: `ParentID='${graphID}' AND Status='Pending' ` +
|
|
1367
|
+
`AND (StepType='Human' OR (StepType IS NULL AND UserID IS NOT NULL)) ` +
|
|
1368
|
+
`AND ClaimedBy='${HUMAN_TASK_NOTIFIED_MARKER}'`,
|
|
1369
|
+
ResultType: 'entity_object',
|
|
1370
|
+
BypassCache: true,
|
|
1371
|
+
}, this.contextUser);
|
|
1372
|
+
if (!waiting.Success)
|
|
1373
|
+
return;
|
|
1374
|
+
for (const task of waiting.Results ?? []) {
|
|
1375
|
+
// Only when there is nothing live AND nothing terminal. A task with an open request is
|
|
1376
|
+
// simply waiting; one with a terminal request is settled on the next pass by
|
|
1377
|
+
// settleAnsweredHumanTasks, and re-raising either would ask the same question twice.
|
|
1378
|
+
if (await this.findOpenRequest(provider, task.ID))
|
|
1379
|
+
continue;
|
|
1380
|
+
if (await this.answeredRequestFor(provider, task.ID))
|
|
1381
|
+
continue;
|
|
1382
|
+
LogStatus(`[TaskGraphDispatcher] The request for '${task.Name}' was cancelled and nothing ` +
|
|
1383
|
+
`replaced it; asking again.`);
|
|
1384
|
+
task.ClaimedBy = null;
|
|
1385
|
+
if (!(await task.Save())) {
|
|
1386
|
+
LogError(`[TaskGraphDispatcher] Could not re-open cancelled human task ${task.ID}: ` +
|
|
1387
|
+
`${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
1388
|
+
}
|
|
1389
|
+
}
|
|
1390
|
+
}
|
|
1391
|
+
/** The answered (or expired) request for a task, if any. */
|
|
1392
|
+
async answeredRequestFor(provider, taskID) {
|
|
1393
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
1394
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
1395
|
+
// Everything terminal. 'Canceled' is deliberately absent: a cancelled request means
|
|
1396
|
+
// the ASK was withdrawn, not that the step was decided, so the task keeps waiting
|
|
1397
|
+
// for whatever replaces it.
|
|
1398
|
+
ExtraFilter: `OriginatingTaskID='${taskID}' AND Status IN ('Approved','Rejected','Responded','Expired')`,
|
|
1399
|
+
OrderBy: 'RespondedAt DESC',
|
|
1400
|
+
ResultType: 'entity_object',
|
|
1401
|
+
BypassCache: true,
|
|
1402
|
+
}, this.contextUser);
|
|
1403
|
+
return (result.Success ? result.Results?.[0] : null) ?? null;
|
|
1404
|
+
}
|
|
1405
|
+
/**
|
|
1406
|
+
* Expires requests whose deadline has passed.
|
|
1407
|
+
*
|
|
1408
|
+
* A deadline that nothing enforces is a comment. Without this an `ExpiresAt` in the past leaves
|
|
1409
|
+
* the request `Requested` forever and the workflow waiting on it just as long.
|
|
1410
|
+
*/
|
|
1411
|
+
async expireOverdueRequests(provider, graphID) {
|
|
1412
|
+
// Scoped by an explicit id list rather than a subquery against a view name, so this reads
|
|
1413
|
+
// the same on any provider rather than assuming a SQL dialect and a physical view.
|
|
1414
|
+
const humanTasks = await RunView.FromMetadataProvider(provider).RunView({
|
|
1415
|
+
EntityName: 'MJ: Tasks',
|
|
1416
|
+
Fields: ['ID'],
|
|
1417
|
+
ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
|
|
1418
|
+
ResultType: 'simple',
|
|
1419
|
+
}, this.contextUser);
|
|
1420
|
+
const ids = (humanTasks.Results ?? []).map((r) => `'${r.ID}'`);
|
|
1421
|
+
if (ids.length === 0)
|
|
1422
|
+
return;
|
|
1423
|
+
const nowISO = new Date().toISOString();
|
|
1424
|
+
const overdue = await RunView.FromMetadataProvider(provider).RunView({
|
|
1425
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
1426
|
+
ExtraFilter: `Status='Requested' AND ExpiresAt IS NOT NULL AND ExpiresAt < '${nowISO}' ` +
|
|
1427
|
+
`AND OriginatingTaskID IN (${ids.join(',')})`,
|
|
1428
|
+
ResultType: 'entity_object',
|
|
1429
|
+
BypassCache: true,
|
|
1430
|
+
}, this.contextUser);
|
|
1431
|
+
if (!overdue.Success)
|
|
1432
|
+
return;
|
|
1433
|
+
for (const request of overdue.Results ?? []) {
|
|
1434
|
+
request.Status = 'Expired';
|
|
1435
|
+
if (!(await request.Save())) {
|
|
1436
|
+
LogError(`[TaskGraphDispatcher] Could not expire request ${request.ID}.`);
|
|
1437
|
+
}
|
|
1438
|
+
}
|
|
1439
|
+
}
|
|
1440
|
+
/** The agent run that submitted this task's graph, for provenance on the request. */
|
|
1441
|
+
async submittingRunOf(provider, task) {
|
|
1442
|
+
if (!task.ParentID)
|
|
1443
|
+
return null;
|
|
1444
|
+
try {
|
|
1445
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
1446
|
+
return (await parent.Load(task.ParentID)) ? parent.AgentRunID : null;
|
|
1447
|
+
}
|
|
1448
|
+
catch {
|
|
1449
|
+
return null;
|
|
1450
|
+
}
|
|
1451
|
+
}
|
|
624
1452
|
/** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
|
|
625
1453
|
async loadGraphState(provider, parentTaskID) {
|
|
626
1454
|
const rv = RunView.FromMetadataProvider(provider);
|
|
@@ -628,8 +1456,13 @@ export class TaskGraphDispatcher {
|
|
|
628
1456
|
// fires no cache invalidation. See findActiveGraphIDs.
|
|
629
1457
|
const childrenResult = await rv.RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
|
|
630
1458
|
const children = (childrenResult.Success ? childrenResult.Results : []) ?? [];
|
|
631
|
-
if (children.length === 0)
|
|
632
|
-
return {
|
|
1459
|
+
if (children.length === 0) {
|
|
1460
|
+
return {
|
|
1461
|
+
nodes: [], edges: [], entityById: new Map(),
|
|
1462
|
+
unreachableTaskIDs: new Set(), skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
|
|
1463
|
+
handledFailureIDs: new Set(),
|
|
1464
|
+
};
|
|
1465
|
+
}
|
|
633
1466
|
const idList = children.map((c) => `'${c.ID}'`).join(',');
|
|
634
1467
|
const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
|
|
635
1468
|
const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
|
|
@@ -651,7 +1484,28 @@ export class TaskGraphDispatcher {
|
|
|
651
1484
|
// unreachable instead, and blocked before anything can claim it.
|
|
652
1485
|
const droppedInto = new Set();
|
|
653
1486
|
const stillReachable = new Set();
|
|
654
|
-
|
|
1487
|
+
// EXCLUSIVE edges are exempt from the generic machinery below, and that exemption is
|
|
1488
|
+
// load-bearing. An XOR loser is by definition condition-false, so the ordinary path would
|
|
1489
|
+
// record it as unreachable and Block it — and a Blocked child poisons the parent rollup, so
|
|
1490
|
+
// every fork would settle the graph as Blocked. Losers must become Skipped instead, which
|
|
1491
|
+
// only ResolveExclusiveGroups can decide.
|
|
1492
|
+
const exclusive = deps.filter((d) => !!d.ExclusiveGroup);
|
|
1493
|
+
const ordinary = deps.filter((d) => !d.ExclusiveGroup);
|
|
1494
|
+
const resolution = ResolveExclusiveGroups(exclusive.map((d) => ({
|
|
1495
|
+
id: d.ID,
|
|
1496
|
+
taskId: d.TaskID,
|
|
1497
|
+
dependsOnTaskId: d.DependsOnTaskID,
|
|
1498
|
+
exclusiveGroup: d.ExclusiveGroup,
|
|
1499
|
+
originStatus: (entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending'),
|
|
1500
|
+
priority: d.Priority ?? 0,
|
|
1501
|
+
sequence: d.Sequence ?? 0,
|
|
1502
|
+
conditionOutcome: this.evaluateExclusiveCondition(d, entityById),
|
|
1503
|
+
})),
|
|
1504
|
+
// A flow's failure handling is its outgoing edges, so a Failed origin still decides its
|
|
1505
|
+
// group. For a loop-agent graph the set is Complete-only and nothing changes.
|
|
1506
|
+
new Set(['Complete', 'Failed']));
|
|
1507
|
+
const loserEdgeIDs = new Set(resolution.loserEdgeIDs);
|
|
1508
|
+
for (const d of ordinary) {
|
|
655
1509
|
if (d.Condition?.trim()) {
|
|
656
1510
|
const outcome = this.evaluateEdgeCondition(d, entityById);
|
|
657
1511
|
if (outcome === 'drop') {
|
|
@@ -666,14 +1520,30 @@ export class TaskGraphDispatcher {
|
|
|
666
1520
|
dependencyType: d.DependencyType,
|
|
667
1521
|
});
|
|
668
1522
|
}
|
|
1523
|
+
for (const d of exclusive) {
|
|
1524
|
+
// A losing edge is removed rather than left to gate: its target is being skipped, and a
|
|
1525
|
+
// live edge into a skipped task would keep the graph waiting on a branch nobody took.
|
|
1526
|
+
if (loserEdgeIDs.has(d.ID))
|
|
1527
|
+
continue;
|
|
1528
|
+
stillReachable.add(d.TaskID);
|
|
1529
|
+
liveEdges.push({
|
|
1530
|
+
taskId: d.TaskID,
|
|
1531
|
+
dependsOnTaskId: d.DependsOnTaskID,
|
|
1532
|
+
dependencyType: d.DependencyType,
|
|
1533
|
+
});
|
|
1534
|
+
}
|
|
669
1535
|
// Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
|
|
670
1536
|
// waiting on it, and a node reached by an alternate branch is genuinely reachable.
|
|
671
1537
|
const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
|
|
1538
|
+
const nodes = children.map((c) => ({ id: c.ID, status: c.Status }));
|
|
672
1539
|
return {
|
|
673
|
-
nodes
|
|
1540
|
+
nodes,
|
|
674
1541
|
edges: liveEdges,
|
|
675
1542
|
entityById,
|
|
676
1543
|
unreachableTaskIDs,
|
|
1544
|
+
skipSeedTaskIDs: new Set(resolution.skipSeedTaskIDs),
|
|
1545
|
+
holdTaskIDs: new Set(resolution.holdTaskIDs),
|
|
1546
|
+
handledFailureIDs: await this.computeHandledFailures(provider, parentTaskID, nodes, liveEdges),
|
|
677
1547
|
};
|
|
678
1548
|
}
|
|
679
1549
|
/**
|
|
@@ -687,6 +1557,19 @@ export class TaskGraphDispatcher {
|
|
|
687
1557
|
const upstream = entityById.get(dep.DependsOnTaskID);
|
|
688
1558
|
if (!upstream)
|
|
689
1559
|
return 'keep';
|
|
1560
|
+
// TERMINALITY GUARD — fixes a latent bug, not a hypothetical one.
|
|
1561
|
+
//
|
|
1562
|
+
// Without it, every conditional edge is evaluated on every poll cycle, including while its
|
|
1563
|
+
// origin is still Pending. A condition like `succeeded` is then a DEFINITE FALSE, the edge
|
|
1564
|
+
// is dropped, and the target is Blocked at wave one — permanently, before the origin ever
|
|
1565
|
+
// ran. That kills any conditioned linear chain, which is the most common flow shape there
|
|
1566
|
+
// is.
|
|
1567
|
+
//
|
|
1568
|
+
// A non-terminal origin is UNDECIDED, and 'keep' is the safe reading of undecided: the
|
|
1569
|
+
// prerequisite gate already prevents the target starting early, so keeping the edge costs
|
|
1570
|
+
// nothing and dropping it is irreversible.
|
|
1571
|
+
if (!TERMINAL_FOR_CONDITIONS.has(upstream.Status))
|
|
1572
|
+
return 'keep';
|
|
690
1573
|
let output = null;
|
|
691
1574
|
if (upstream.OutputPayload) {
|
|
692
1575
|
try {
|
|
@@ -694,13 +1577,7 @@ export class TaskGraphDispatcher {
|
|
|
694
1577
|
}
|
|
695
1578
|
catch { /* a malformed payload is not grounds to drop a prerequisite */ }
|
|
696
1579
|
}
|
|
697
|
-
const result = this.conditionEvaluator.Evaluate(dep.Condition,
|
|
698
|
-
status: upstream.Status,
|
|
699
|
-
succeeded: upstream.Status === 'Complete',
|
|
700
|
-
failed: upstream.Status === 'Failed',
|
|
701
|
-
output,
|
|
702
|
-
errorMessage: upstream.ErrorMessage ?? null,
|
|
703
|
-
});
|
|
1580
|
+
const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
|
|
704
1581
|
if (!result.Success) {
|
|
705
1582
|
LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
|
|
706
1583
|
`(${result.ErrorMessage}); keeping the edge so the graph stalls visibly rather than ` +
|
|
@@ -709,6 +1586,63 @@ export class TaskGraphDispatcher {
|
|
|
709
1586
|
}
|
|
710
1587
|
return result.Value ? 'keep' : 'drop';
|
|
711
1588
|
}
|
|
1589
|
+
/**
|
|
1590
|
+
* An exclusive edge's condition as a three-way outcome.
|
|
1591
|
+
*
|
|
1592
|
+
* `ResolveExclusiveGroups` needs to tell "false" from "could not be evaluated": the first loses
|
|
1593
|
+
* the branch, the second holds the whole group. The generic keep/drop path cannot express that
|
|
1594
|
+
* difference, which is why exclusive edges take this route instead.
|
|
1595
|
+
*/
|
|
1596
|
+
evaluateExclusiveCondition(dep, entityById) {
|
|
1597
|
+
if (!dep.Condition?.trim())
|
|
1598
|
+
return 'satisfied';
|
|
1599
|
+
const upstream = entityById.get(dep.DependsOnTaskID);
|
|
1600
|
+
if (!upstream)
|
|
1601
|
+
return 'unevaluable';
|
|
1602
|
+
let output = null;
|
|
1603
|
+
if (upstream.OutputPayload) {
|
|
1604
|
+
try {
|
|
1605
|
+
output = JSON.parse(upstream.OutputPayload);
|
|
1606
|
+
}
|
|
1607
|
+
catch { /* malformed payload */ }
|
|
1608
|
+
}
|
|
1609
|
+
const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
|
|
1610
|
+
if (!result.Success)
|
|
1611
|
+
return 'unevaluable';
|
|
1612
|
+
return result.Value ? 'satisfied' : 'unsatisfied';
|
|
1613
|
+
}
|
|
1614
|
+
/**
|
|
1615
|
+
* Everything an edge condition can see — the SUPERSET of both dialects.
|
|
1616
|
+
*
|
|
1617
|
+
* A flow condition is written against `payload` / `stepResult` / `flowContext` / `data` /
|
|
1618
|
+
* `context`; the dispatcher's own conditions are written against `status` / `succeeded` /
|
|
1619
|
+
* `failed` / `output` / `errorMessage`. Compiling flows onto this engine without the flow
|
|
1620
|
+
* dialect would make every `payload.x` condition evaluate against nothing — silently, since an
|
|
1621
|
+
* undefined property is simply falsy. Both dialects are readable here so a condition means the
|
|
1622
|
+
* same thing on either engine.
|
|
1623
|
+
*
|
|
1624
|
+
* `payload` is the ORIGIN task's post-step snapshot. There is deliberately no "graph-wide
|
|
1625
|
+
* payload": each task's output is its own, and inventing a merged one would give conditions a
|
|
1626
|
+
* value the flow engine never had.
|
|
1627
|
+
*/
|
|
1628
|
+
buildConditionContext(upstream, output) {
|
|
1629
|
+
const envelope = (output && typeof output === 'object' ? output : {});
|
|
1630
|
+
const succeeded = upstream.Status === 'Complete';
|
|
1631
|
+
return {
|
|
1632
|
+
// dispatcher dialect — unchanged
|
|
1633
|
+
status: upstream.Status,
|
|
1634
|
+
succeeded,
|
|
1635
|
+
failed: upstream.Status === 'Failed',
|
|
1636
|
+
output,
|
|
1637
|
+
errorMessage: upstream.ErrorMessage ?? null,
|
|
1638
|
+
// flow dialect
|
|
1639
|
+
payload: envelope.payload ?? output,
|
|
1640
|
+
stepResult: { Success: succeeded, step: upstream.Name, result: envelope.result ?? output },
|
|
1641
|
+
flowContext: { currentStepId: upstream.ID, completedSteps: [], executionPath: [], stepCount: 0 },
|
|
1642
|
+
data: envelope.data ?? {},
|
|
1643
|
+
context: envelope.context ?? {},
|
|
1644
|
+
};
|
|
1645
|
+
}
|
|
712
1646
|
/** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
|
|
713
1647
|
async loadDependencyOutputs(provider, taskID) {
|
|
714
1648
|
const outputs = new Map();
|
|
@@ -731,5 +1665,568 @@ export class TaskGraphDispatcher {
|
|
|
731
1665
|
}
|
|
732
1666
|
return outputs;
|
|
733
1667
|
}
|
|
1668
|
+
/**
|
|
1669
|
+
* Runs one task's body, whatever kind of step it is.
|
|
1670
|
+
*
|
|
1671
|
+
* **Routing is on `StepType`, not on which key happens to be set.** A loop step carries the same
|
|
1672
|
+
* `ActionID` or `AgentID` as an ordinary step — that key is what the loop *repeats* — so the old
|
|
1673
|
+
* `task.ActionID ? action : agent` test would have run a loop exactly once and called it done.
|
|
1674
|
+
* `StepType` is the only field that distinguishes them.
|
|
1675
|
+
*
|
|
1676
|
+
* Every branch is normalized to one shape so the recording path above stays single: an action has
|
|
1677
|
+
* no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
|
|
1678
|
+
*/
|
|
1679
|
+
async runTaskBody(task, provider, inputPayload, dependencyOutputs) {
|
|
1680
|
+
const payload = this.mergedPayload(inputPayload, dependencyOutputs);
|
|
1681
|
+
const config = task.ConfigurationObject;
|
|
1682
|
+
// A loop's own step type decides how many times its body runs; the body itself is dispatched
|
|
1683
|
+
// through the very same runners as a one-shot step.
|
|
1684
|
+
if (task.StepType === 'ForEach' || task.StepType === 'While') {
|
|
1685
|
+
return { ...await this.runLoopTask(task, provider, payload, dependencyOutputs), PayloadAtStart: payload };
|
|
1686
|
+
}
|
|
1687
|
+
const { params, errors } = BuildMappedInput(config?.inputMapping, { payload });
|
|
1688
|
+
for (const e of errors)
|
|
1689
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
|
|
1690
|
+
// `payload`, NOT `inputPayload` — the MERGED value computed above, which includes what every
|
|
1691
|
+
// dependency produced.
|
|
1692
|
+
//
|
|
1693
|
+
// A step with an input mapping got exactly the parameters it declared; a step WITHOUT one
|
|
1694
|
+
// fell back to the raw input and therefore saw nothing any earlier step had produced. For a
|
|
1695
|
+
// Prompt step — which declares no mapping by design, because it reads the whole payload
|
|
1696
|
+
// through `{{ _CURRENT_PAYLOAD }}` — that meant the placeholder rendered `{}` and the model
|
|
1697
|
+
// was asked to write from an empty brief.
|
|
1698
|
+
//
|
|
1699
|
+
// It answered anyway. The Content Pipeline's draft step said "the research data was empty",
|
|
1700
|
+
// which was TRUE of what it had been handed while twenty research results sat in the
|
|
1701
|
+
// dependency outputs beside it, and the reviewer then rejected the draft for saying so.
|
|
1702
|
+
// Every layer looked like it was working.
|
|
1703
|
+
const effectiveInput = Object.keys(params).length > 0 ? params : payload;
|
|
1704
|
+
if (task.StepType === 'Prompt') {
|
|
1705
|
+
if (!this.promptRunner) {
|
|
1706
|
+
// Not a failure: "nobody here can run this" is not "this ran and did not work".
|
|
1707
|
+
return { Success: false, AgentRunID: null, ErrorMessage: 'No prompt runner is loaded on this host.' };
|
|
1708
|
+
}
|
|
1709
|
+
const promptResult = await this.promptRunner.RunPromptForTask({
|
|
1710
|
+
TaskID: task.ID,
|
|
1711
|
+
PromptID: task.PromptID,
|
|
1712
|
+
InputPayload: effectiveInput,
|
|
1713
|
+
DependencyOutputs: dependencyOutputs,
|
|
1714
|
+
TemplateParameters: config?.prompt?.templateParameters,
|
|
1715
|
+
Provider: provider,
|
|
1716
|
+
ContextUser: this.contextUser,
|
|
1717
|
+
});
|
|
1718
|
+
// A prompt's response is DEEP-MERGED into the payload rather than replacing it. A prompt
|
|
1719
|
+
// answers one question; replacing the payload with its answer would discard everything
|
|
1720
|
+
// the steps before it established, which is how a late step loses the data it depends on.
|
|
1721
|
+
const merged = promptResult.Success && promptResult.Output && typeof promptResult.Output === 'object'
|
|
1722
|
+
? deepMergePayload(payload, promptResult.Output)
|
|
1723
|
+
: payload;
|
|
1724
|
+
return {
|
|
1725
|
+
Success: promptResult.Success,
|
|
1726
|
+
AgentRunID: null,
|
|
1727
|
+
ErrorMessage: promptResult.ErrorMessage,
|
|
1728
|
+
Output: this.applyStepOutputMapping(task, merged, merged, config?.outputMapping),
|
|
1729
|
+
PayloadAtStart: payload,
|
|
1730
|
+
ChatMessage: promptResult.ChatMessage,
|
|
1731
|
+
// Returned even when the prompt FAILED. A failed prompt still cost tokens, and a
|
|
1732
|
+
// cost rollup that silently omits failures under-reports exactly the runs someone
|
|
1733
|
+
// is most likely to be investigating.
|
|
1734
|
+
PromptRunID: promptResult.PromptRunID,
|
|
1735
|
+
};
|
|
1736
|
+
}
|
|
1737
|
+
const raw = task.ActionID
|
|
1738
|
+
? { ...await this.actionRunner.RunActionForTask({
|
|
1739
|
+
TaskID: task.ID,
|
|
1740
|
+
ActionID: task.ActionID,
|
|
1741
|
+
InputPayload: effectiveInput,
|
|
1742
|
+
DependencyOutputs: dependencyOutputs,
|
|
1743
|
+
Provider: provider,
|
|
1744
|
+
ContextUser: this.contextUser,
|
|
1745
|
+
}), AgentRunID: null }
|
|
1746
|
+
: await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs);
|
|
1747
|
+
return {
|
|
1748
|
+
...raw,
|
|
1749
|
+
Output: this.applyStepOutputMapping(task, payload, raw.Output, config?.outputMapping),
|
|
1750
|
+
PayloadAtStart: payload,
|
|
1751
|
+
};
|
|
1752
|
+
}
|
|
1753
|
+
/**
|
|
1754
|
+
* Runs a loop step: its body once per iteration, with the item and index in scope.
|
|
1755
|
+
*
|
|
1756
|
+
* The loop's own `Configuration` supplies the definition; the row's `ActionID` / `AgentID`
|
|
1757
|
+
* supplies what to repeat. Per-iteration inputs are resolved fresh each pass — the bindings are
|
|
1758
|
+
* merged into the payload before the mapping is applied, which is how a body can reference the
|
|
1759
|
+
* current item at all.
|
|
1760
|
+
*/
|
|
1761
|
+
async runLoopTask(task, provider, payload, dependencyOutputs) {
|
|
1762
|
+
const config = task.ConfigurationObject;
|
|
1763
|
+
const op = task.StepType === 'ForEach' ? config?.forEach : config?.while;
|
|
1764
|
+
if (!op) {
|
|
1765
|
+
return {
|
|
1766
|
+
Success: false,
|
|
1767
|
+
AgentRunID: null,
|
|
1768
|
+
ErrorMessage: `"${task.Name}" is a ${task.StepType} step with no loop settings, so there is nothing to repeat.`,
|
|
1769
|
+
};
|
|
1770
|
+
}
|
|
1771
|
+
// A prompt body has no params of its own — it receives the payload (with the loop bindings
|
|
1772
|
+
// merged in) through the placeholder, so an empty mapping is correct rather than missing.
|
|
1773
|
+
const bodyMapping = (op.action?.params ?? {});
|
|
1774
|
+
// The BODY's output mapping, applied once per pass — see `foldIterationOutput`.
|
|
1775
|
+
//
|
|
1776
|
+
// It used to be applied a single time after the loop finished, against the accumulated
|
|
1777
|
+
// payload. That is the wrong moment in two ways at once: the mapping names an output
|
|
1778
|
+
// PARAMETER of the body, which no longer exists by then, and a mapping like
|
|
1779
|
+
// `"Items": "results[]"` can only append per pass. So every pass merged its raw result into
|
|
1780
|
+
// the shared payload instead, each overwriting the last, and the mapping matched nothing and
|
|
1781
|
+
// wrote nothing. A ForEach over five items reported five successes and kept item five.
|
|
1782
|
+
const bodyOutputMapping = op.action?.outputMapping ?? op.prompt?.outputMapping;
|
|
1783
|
+
// Where this step sits in its graph, resolved ONCE rather than per iteration. A loop body is
|
|
1784
|
+
// dispatched exactly like a one-shot step and needs the same two things: the run that
|
|
1785
|
+
// submitted the graph (so a spawned run gets a ParentRunID and is visible to the tree and to
|
|
1786
|
+
// cost), and the continuation depth (so the recursion cap still applies). Omitting them made
|
|
1787
|
+
// loop bodies second-class in every dimension — and reopened the unbounded-recursion hole
|
|
1788
|
+
// THROUGH loops, since each spawned run restarted the chain at zero.
|
|
1789
|
+
const graphContext = await this.graphContext(provider, task);
|
|
1790
|
+
// THE LOOP'S PAYLOAD ACCUMULATES. Each iteration's output merges in, and the next iteration
|
|
1791
|
+
// — and the While condition — sees it. Without this the condition closure re-read the
|
|
1792
|
+
// payload as it was when the loop STARTED, so a `while payload.brandOK !== true` could never
|
|
1793
|
+
// become false: the loop burned every iteration re-examining the original input and always
|
|
1794
|
+
// took the give-up branch, making the other branch unreachable. The loop ran, reported
|
|
1795
|
+
// success, and its result was predetermined.
|
|
1796
|
+
let livePayload = { ...payload };
|
|
1797
|
+
// One entry per pass, so the loop's work exists somewhere the platform can see it. Without
|
|
1798
|
+
// this a loop is a single childless node: the run tree reaches nested work through six links
|
|
1799
|
+
// and an iteration is none of them, so the passes were invisible to the timeline AND their
|
|
1800
|
+
// spend was missing from the settlement rollup. See ITaskStepRuntime.iterations.
|
|
1801
|
+
const iterationTrace = [];
|
|
1802
|
+
// Bounds what the trace's payloads may cost. The pointers are never budgeted — those are the
|
|
1803
|
+
// durable record of the work and must survive whatever the payloads do.
|
|
1804
|
+
const budget = new IterationPayloadBudget();
|
|
1805
|
+
const invokeBody = async ({ Index, Bindings }) => {
|
|
1806
|
+
// Bindings go INTO the payload rather than beside it, so an authored mapping reaches the
|
|
1807
|
+
// current item the same way it reaches anything else: `payload.<itemVariable>`.
|
|
1808
|
+
const iterationPayload = { ...livePayload, ...Bindings };
|
|
1809
|
+
const resolved = ResolveMappedInput(bodyMapping, { payload: iterationPayload });
|
|
1810
|
+
/**
|
|
1811
|
+
* Folds an iteration's output into the running payload the next pass will see, and
|
|
1812
|
+
* records what the pass produced.
|
|
1813
|
+
*
|
|
1814
|
+
* The trace is written HERE rather than after the loop because a loop that fails partway
|
|
1815
|
+
* still ran the passes before it, and their runs are real spend that must not vanish
|
|
1816
|
+
* because the loop as a whole did not finish.
|
|
1817
|
+
*/
|
|
1818
|
+
const absorb = (outcome, bodyInput) => {
|
|
1819
|
+
livePayload = this.foldIterationOutput(task, livePayload, outcome.Output, bodyOutputMapping);
|
|
1820
|
+
iterationTrace.push({
|
|
1821
|
+
index: Index,
|
|
1822
|
+
// What THIS pass was handed and what it gave back — not the loop's running
|
|
1823
|
+
// payload before and after it.
|
|
1824
|
+
//
|
|
1825
|
+
// A pass has no row of its own, so without these there is nowhere its work can be
|
|
1826
|
+
// recorded: every iteration presented null on both sides and the run view could
|
|
1827
|
+
// say nothing about any single pass, which for a loop is the only interesting
|
|
1828
|
+
// question. But recording the RUNNING payload on both sides — the obvious reading
|
|
1829
|
+
// of "before and after" — is quadratic: each pass would hold a full copy of
|
|
1830
|
+
// everything every earlier pass accumulated. A five-iteration demo produced a
|
|
1831
|
+
// 121KB Configuration that way; the same loop over fifty items would produce
|
|
1832
|
+
// megabytes, in a column every reader of the row pays to load.
|
|
1833
|
+
//
|
|
1834
|
+
// The pass's own input and output are what a reader actually wants ("what did
|
|
1835
|
+
// pass three do?"), and they are constant-sized per pass.
|
|
1836
|
+
payloadAtStart: budget.Take(bodyInput),
|
|
1837
|
+
payloadAtEnd: budget.Take(outcome.Output),
|
|
1838
|
+
promptRunID: outcome.PromptRunID,
|
|
1839
|
+
agentRunID: outcome.AgentRunID,
|
|
1840
|
+
// An ACTION body records its log here. Omitting it left an action-bodied pass
|
|
1841
|
+
// with no pointer at all — no cost, no timing, nothing to open — and the tree,
|
|
1842
|
+
// seeing neither a prompt run nor an agent run, fell through to its last branch
|
|
1843
|
+
// and called the pass a Sub-Agent. A loop over a web search then showed five
|
|
1844
|
+
// sub-agent runs that never existed.
|
|
1845
|
+
actionLogID: outcome.ActionLogID,
|
|
1846
|
+
success: outcome.Success,
|
|
1847
|
+
errorMessage: outcome.ErrorMessage,
|
|
1848
|
+
});
|
|
1849
|
+
return outcome;
|
|
1850
|
+
};
|
|
1851
|
+
// A prompt body is checked FIRST because it is the only one whose id lives in its own
|
|
1852
|
+
// column: a loop repeating a prompt has PromptID set and both ActionID and AgentID null,
|
|
1853
|
+
// so falling through to the agent branch would dereference a null agent id.
|
|
1854
|
+
if (task.StepType && task.PromptID && !task.ActionID) {
|
|
1855
|
+
if (!this.promptRunner) {
|
|
1856
|
+
return { Success: false, ErrorMessage: 'No prompt runner is loaded on this host.' };
|
|
1857
|
+
}
|
|
1858
|
+
return absorb(await this.promptRunner.RunPromptForTask({
|
|
1859
|
+
TaskID: task.ID,
|
|
1860
|
+
PromptID: task.PromptID,
|
|
1861
|
+
// The ITERATION payload, not the mapped params. An action body declares its
|
|
1862
|
+
// inputs and gets exactly those; a prompt body declares none — it receives the
|
|
1863
|
+
// whole payload through the placeholder, and the loop's item and index are
|
|
1864
|
+
// merged INTO that payload. Passing the mapped result here handed the prompt an
|
|
1865
|
+
// empty object, so every iteration asked the model to describe nothing and got
|
|
1866
|
+
// five confident answers about nothing back.
|
|
1867
|
+
InputPayload: iterationPayload,
|
|
1868
|
+
DependencyOutputs: dependencyOutputs,
|
|
1869
|
+
// The loop's bindings become TEMPLATE VARIABLES, so an author writes
|
|
1870
|
+
// `{{ field }}` for the item the loop is on — which is what `itemVariable` is
|
|
1871
|
+
// for, and what anyone reading the step's configuration expects. Reaching it
|
|
1872
|
+
// through the payload placeholder instead works but is not discoverable, and
|
|
1873
|
+
// getting it wrong is silent: the variable renders empty and the model answers
|
|
1874
|
+
// confidently about nothing.
|
|
1875
|
+
TemplateParameters: { ...stringifyBindings(Bindings), ...op.prompt?.templateParameters },
|
|
1876
|
+
Provider: provider,
|
|
1877
|
+
ContextUser: this.contextUser,
|
|
1878
|
+
}), iterationPayload);
|
|
1879
|
+
}
|
|
1880
|
+
if (task.ActionID) {
|
|
1881
|
+
return absorb(await this.actionRunner.RunActionForTask({
|
|
1882
|
+
TaskID: task.ID,
|
|
1883
|
+
ActionID: task.ActionID,
|
|
1884
|
+
InputPayload: resolved,
|
|
1885
|
+
DependencyOutputs: dependencyOutputs,
|
|
1886
|
+
Provider: provider,
|
|
1887
|
+
ContextUser: this.contextUser,
|
|
1888
|
+
}), resolved);
|
|
1889
|
+
}
|
|
1890
|
+
const agentInput = Object.keys(resolved).length > 0 ? resolved : iterationPayload;
|
|
1891
|
+
return absorb(await this.agentRunner.RunAgentForTask({
|
|
1892
|
+
TaskID: task.ID,
|
|
1893
|
+
AgentID: task.AgentID,
|
|
1894
|
+
// The ITERATION payload when the body declares no inputs of its own. A sub-agent
|
|
1895
|
+
// body has no `params`, so the mapped result is `{}` — every iteration was handing
|
|
1896
|
+
// the agent nothing and asking it to work from that.
|
|
1897
|
+
InputPayload: agentInput,
|
|
1898
|
+
DependencyOutputs: dependencyOutputs,
|
|
1899
|
+
ContinuationDepth: graphContext.Depth,
|
|
1900
|
+
SubmittingAgentRunID: graphContext.SubmittingAgentRunID,
|
|
1901
|
+
Provider: provider,
|
|
1902
|
+
ContextUser: this.contextUser,
|
|
1903
|
+
}), agentInput);
|
|
1904
|
+
};
|
|
1905
|
+
const outcome = task.StepType === 'ForEach'
|
|
1906
|
+
? await RunForEachLoop(op, { payload }, invokeBody)
|
|
1907
|
+
: await RunWhileLoop(op, (iteration) => this.conditionEvaluator.Evaluate(op.condition,
|
|
1908
|
+
// BOTH forms, because a workflow should not have two condition dialects. An
|
|
1909
|
+
// EDGE condition is written `payload.brandOK !== true`; a loop condition used
|
|
1910
|
+
// to see the payload's keys spread at the top level and nothing named `payload`,
|
|
1911
|
+
// so the same expression that routes an edge failed here with
|
|
1912
|
+
// "payload is not defined". The spread stays for conditions already written
|
|
1913
|
+
// against it.
|
|
1914
|
+
{ ...livePayload, payload: livePayload, iteration }), invokeBody);
|
|
1915
|
+
return {
|
|
1916
|
+
Success: outcome.Success,
|
|
1917
|
+
AgentRunID: null,
|
|
1918
|
+
ErrorMessage: outcome.ErrorMessage,
|
|
1919
|
+
// Every pass that ran, including those before a failure — see `iterationTrace`.
|
|
1920
|
+
Iterations: iterationTrace.length > 0 ? iterationTrace : undefined,
|
|
1921
|
+
// The ACCUMULATED payload — everything the iterations established — not the one the
|
|
1922
|
+
// loop started with, which would discard the loop's whole effect on the workflow.
|
|
1923
|
+
//
|
|
1924
|
+
// Only the STEP's own mapping is applied here. The body's mapping already ran once per
|
|
1925
|
+
// pass inside `foldIterationOutput`; applying it again against the accumulated payload
|
|
1926
|
+
// is what used to make it match nothing.
|
|
1927
|
+
Output: this.applyStepOutputMapping(task, livePayload, outcome.Output, config?.outputMapping),
|
|
1928
|
+
};
|
|
1929
|
+
}
|
|
1930
|
+
/**
|
|
1931
|
+
* Folds one pass's result into the loop's running payload.
|
|
1932
|
+
*
|
|
1933
|
+
* **With a body mapping**, the pass's declared outputs are filed where the author said to put
|
|
1934
|
+
* them — including `name[]`, which appends, so a ForEach can collect one entry per item. That is
|
|
1935
|
+
* the whole point of a loop over a collection, and it is only expressible per pass.
|
|
1936
|
+
*
|
|
1937
|
+
* **Without one**, the raw result is deep-merged, which is the pre-existing behaviour and the
|
|
1938
|
+
* right default for a `While` that converges on a value: each pass refines what the condition
|
|
1939
|
+
* reads. It is the wrong default for a ForEach that collects — hence the mapping.
|
|
1940
|
+
*
|
|
1941
|
+
* An unmapped output is reported per pass rather than swallowed, for the same reason
|
|
1942
|
+
* {@link applyStepOutputMapping} reports it: a mapping that names something the body never
|
|
1943
|
+
* returned means the pass did work that went nowhere, while everything reports success.
|
|
1944
|
+
*/
|
|
1945
|
+
foldIterationOutput(task, livePayload, output, bodyOutputMapping) {
|
|
1946
|
+
if (!output || typeof output !== 'object' || Array.isArray(output))
|
|
1947
|
+
return livePayload;
|
|
1948
|
+
const source = output;
|
|
1949
|
+
if (!bodyOutputMapping)
|
|
1950
|
+
return deepMergePayload(livePayload, source);
|
|
1951
|
+
// Applied ONTO a deep copy of the running payload, not into a fresh object: `name[]` appends,
|
|
1952
|
+
// and appending is meaningless without the list already there. The copy is deep because the
|
|
1953
|
+
// trace has already recorded earlier passes' payloads — mutating a shared nested array would
|
|
1954
|
+
// retroactively rewrite what those passes are recorded as having seen.
|
|
1955
|
+
const { updates, errors, unmapped } = ApplyOutputMapping(source, bodyOutputMapping, structuredClone(livePayload));
|
|
1956
|
+
for (const e of errors)
|
|
1957
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} loop body: ${e}`);
|
|
1958
|
+
if (unmapped?.length) {
|
|
1959
|
+
LogError(`[TaskGraphDispatcher] '${task.Name}' loop body mapped output(s) it did not return: ` +
|
|
1960
|
+
`${unmapped.join(', ')}. The pass returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
|
|
1961
|
+
`Those payload values were NOT written, so anything downstream reading them sees nothing.`);
|
|
1962
|
+
}
|
|
1963
|
+
// `updates` IS the copy that was applied onto, so it is already the complete next payload.
|
|
1964
|
+
return updates;
|
|
1965
|
+
}
|
|
1966
|
+
/**
|
|
1967
|
+
* Files a step's result into the payload it hands downstream.
|
|
1968
|
+
*
|
|
1969
|
+
* **This is what makes a branch condition possible.** A workflow that branches on
|
|
1970
|
+
* `payload.stockPrice` has that value only because this step mapped `CurrentPrice -> stockPrice`.
|
|
1971
|
+
* Without it the condition reads `undefined` — merely falsy — so the workflow takes the other
|
|
1972
|
+
* branch, finishes, and reports success with nothing to indicate anything went wrong.
|
|
1973
|
+
*
|
|
1974
|
+
* The incoming payload is carried through as well as the update, so a value written three steps
|
|
1975
|
+
* back is still readable here. Returning only this step's own output is what used to limit a
|
|
1976
|
+
* condition's view to its immediate predecessor.
|
|
1977
|
+
*/
|
|
1978
|
+
applyStepOutputMapping(task, payload, output, outputMapping) {
|
|
1979
|
+
// No mapping: MERGE the step's output over the payload rather than replacing it.
|
|
1980
|
+
//
|
|
1981
|
+
// Replacing is what made the Content Pipeline's exclusive pair unreachable. A While loop's
|
|
1982
|
+
// own output is a SUMMARY — `{iterations, succeeded, failed, results}` — so returning it
|
|
1983
|
+
// discarded the payload the iterations had built, including the `brandOK` the reviewer had
|
|
1984
|
+
// just set to true. The edges read `payload.brandOK === true` and `!== true`; against a
|
|
1985
|
+
// summary the first is false and the second is true, so the give-up branch won on EVERY run
|
|
1986
|
+
// no matter what the reviewer decided. The approved branch was unreachable in practice while
|
|
1987
|
+
// being perfectly reachable on the canvas.
|
|
1988
|
+
//
|
|
1989
|
+
// This is the same rule the mapped path already follows two lines down, and the same rule
|
|
1990
|
+
// the doc comment above states. The no-mapping branch was simply not following it.
|
|
1991
|
+
if (!outputMapping) {
|
|
1992
|
+
return output && typeof output === 'object' && !Array.isArray(output)
|
|
1993
|
+
? { ...payload, ...output }
|
|
1994
|
+
: output ?? payload;
|
|
1995
|
+
}
|
|
1996
|
+
const source = output && typeof output === 'object' ? output : { value: output };
|
|
1997
|
+
const { updates, errors, unmapped } = ApplyOutputMapping(source, outputMapping);
|
|
1998
|
+
for (const e of errors)
|
|
1999
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
|
|
2000
|
+
// A mapping that names an output the step never produced discards that step's work while
|
|
2001
|
+
// the step reports Complete. It is not fatal — an action may emit a parameter only on some
|
|
2002
|
+
// paths — but it must not be silent, and naming what WAS returned turns a multi-table
|
|
2003
|
+
// forensic exercise into one line. The Content Pipeline demo lost an entire research pass
|
|
2004
|
+
// this way, every run, because its mapping named another action's parameter.
|
|
2005
|
+
if (unmapped?.length) {
|
|
2006
|
+
LogError(`[TaskGraphDispatcher] '${task.Name}' mapped output(s) the step did not return: ` +
|
|
2007
|
+
`${unmapped.join(', ')}. The step returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
|
|
2008
|
+
`Those payload values were NOT written, so anything downstream reading them sees nothing.`);
|
|
2009
|
+
}
|
|
2010
|
+
return { ...payload, ...updates };
|
|
2011
|
+
}
|
|
2012
|
+
/**
|
|
2013
|
+
* Runs an Agent step, telling the runner where in the graph it sits.
|
|
2014
|
+
*
|
|
2015
|
+
* Depth and provenance are read together because they come from the same row: the graph's parent
|
|
2016
|
+
* task knows both how many continuation hops led here and which run submitted it.
|
|
2017
|
+
*/
|
|
2018
|
+
async runAgentNode(task, provider, effectiveInput, dependencyOutputs) {
|
|
2019
|
+
const context = await this.graphContext(provider, task);
|
|
2020
|
+
return this.agentRunner.RunAgentForTask({
|
|
2021
|
+
TaskID: task.ID,
|
|
2022
|
+
AgentID: task.AgentID,
|
|
2023
|
+
InputPayload: effectiveInput,
|
|
2024
|
+
DependencyOutputs: dependencyOutputs,
|
|
2025
|
+
ContinuationDepth: context.Depth,
|
|
2026
|
+
SubmittingAgentRunID: context.SubmittingAgentRunID,
|
|
2027
|
+
Provider: provider,
|
|
2028
|
+
ContextUser: this.contextUser,
|
|
2029
|
+
});
|
|
2030
|
+
}
|
|
2031
|
+
/**
|
|
2032
|
+
* Completes the agent run that parked on this graph.
|
|
2033
|
+
*
|
|
2034
|
+
* **This is the other half of submit-and-detach.** A run that dispatches a graph does not
|
|
2035
|
+
* complete at submission — it ends `Paused`, because reporting `Completed` above a workflow
|
|
2036
|
+
* where nothing has happened yet is a claim the row cannot support. The run's lifecycle is
|
|
2037
|
+
* finished HERE, when the graph it was waiting on actually settles, which is the first moment
|
|
2038
|
+
* the answer exists.
|
|
2039
|
+
*
|
|
2040
|
+
* Doing it from the dispatcher rather than by awaiting in the agent is what keeps the properties
|
|
2041
|
+
* that made detach right in the first place: a graph containing a human approval can park for
|
|
2042
|
+
* days without holding a conversation turn open, and a graph reclaimed by another instance after
|
|
2043
|
+
* a crash still settles its submitting run, because the settling happens wherever the graph
|
|
2044
|
+
* finishes rather than wherever it started.
|
|
2045
|
+
*
|
|
2046
|
+
* **Only a parked run is touched.** A run that is already `Completed`, `Failed` or `Cancelled`
|
|
2047
|
+
* reached that state for its own reasons — a second graph settling later, a run the user
|
|
2048
|
+
* cancelled, a run that failed after submitting — and overwriting it would rewrite history from
|
|
2049
|
+
* the outside. The `Paused` predicate is the whole guard.
|
|
2050
|
+
*
|
|
2051
|
+
* @param graphStatus the parent rollup's status: what the workflow as a whole did
|
|
2052
|
+
*/
|
|
2053
|
+
async settleSubmittingRun(provider, parent, graphStatus) {
|
|
2054
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
2055
|
+
if (!meta.submittedByAgentRunID)
|
|
2056
|
+
return; // a scheduled or remote-triggered graph has nobody waiting
|
|
2057
|
+
try {
|
|
2058
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
2059
|
+
if (!(await run.Load(meta.submittedByAgentRunID))) {
|
|
2060
|
+
LogError(`[TaskGraphDispatcher] Could not load run ${meta.submittedByAgentRunID} to settle it against graph ${parent.ID}.`);
|
|
2061
|
+
return;
|
|
2062
|
+
}
|
|
2063
|
+
if (run.Status !== 'Paused')
|
|
2064
|
+
return;
|
|
2065
|
+
// The workflow's outcome becomes the run's outcome. A graph that ended any way other than
|
|
2066
|
+
// Complete did not do what the run started it to do, and a run reporting success over it
|
|
2067
|
+
// would be the same untruth in a different place.
|
|
2068
|
+
const succeeded = graphStatus === 'Complete';
|
|
2069
|
+
run.Status = succeeded ? 'Completed' : 'Failed';
|
|
2070
|
+
run.Success = succeeded;
|
|
2071
|
+
run.CompletedAt = new Date();
|
|
2072
|
+
if (!succeeded) {
|
|
2073
|
+
const reason = `The workflow "${parent.Name}" ended ${graphStatus}.`;
|
|
2074
|
+
run.ErrorMessage = run.ErrorMessage ? `${run.ErrorMessage}\n\n${reason}` : reason;
|
|
2075
|
+
}
|
|
2076
|
+
if (!(await run.Save())) {
|
|
2077
|
+
// Left parked rather than forced. A run stuck at Paused is visibly unfinished, which
|
|
2078
|
+
// is a state someone can investigate; a run flipped to Completed by a write that did
|
|
2079
|
+
// not land would be the same lie this whole change removes.
|
|
2080
|
+
LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}: ` +
|
|
2081
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It remains Paused.`);
|
|
2082
|
+
return;
|
|
2083
|
+
}
|
|
2084
|
+
LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${run.Status} — workflow "${parent.Name}" ended ${graphStatus}.`);
|
|
2085
|
+
}
|
|
2086
|
+
catch (e) {
|
|
2087
|
+
LogError(`[TaskGraphDispatcher] Could not settle the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
2088
|
+
}
|
|
2089
|
+
}
|
|
2090
|
+
/**
|
|
2091
|
+
* Gives every step that lacks one a position, once the graph has finished.
|
|
2092
|
+
*
|
|
2093
|
+
* **Why the run stores geometry at all.** A `TaskGraphSpec` is a logical structure with no
|
|
2094
|
+
* layout field, so a graph an agent emitted has no opinion about where its boxes go. Every
|
|
2095
|
+
* viewer was therefore laying it out for itself at render time — and a viewer that failed to
|
|
2096
|
+
* (because the canvas measures nodes it has not drawn yet) fell back to every node at the
|
|
2097
|
+
* origin, piled on one another, with the zoom-to-fit that follows fitting a one-node bounding
|
|
2098
|
+
* box. Settling it once, server-side, means the agent-run canvas, the Workflows runs tab and
|
|
2099
|
+
* anything built later all draw the same picture, and none of them has to compute it.
|
|
2100
|
+
*
|
|
2101
|
+
* **An authored position is never overwritten.** A workflow compiled from a Flow agent carries
|
|
2102
|
+
* the arrangement someone dragged into place; replacing it with an algorithm's guess would
|
|
2103
|
+
* discard a deliberate act. Only steps with no geometry get one, so a partially-arranged graph
|
|
2104
|
+
* keeps what it has.
|
|
2105
|
+
*
|
|
2106
|
+
* Failure here is logged and swallowed: this is presentation. A graph whose work completed must
|
|
2107
|
+
* not be reported as failed because its picture could not be saved.
|
|
2108
|
+
*/
|
|
2109
|
+
async persistComputedLayout(graph) {
|
|
2110
|
+
try {
|
|
2111
|
+
const needsLayout = [...graph.entityById.values()].filter((t) => !this.parseConfiguration(t)?.layout);
|
|
2112
|
+
if (needsLayout.length === 0)
|
|
2113
|
+
return;
|
|
2114
|
+
// Laid out over the WHOLE graph, not just the nodes missing geometry: position depends on
|
|
2115
|
+
// where a node sits in the topology, and a layout computed over a subset would place its
|
|
2116
|
+
// nodes as though the rest of the workflow did not exist.
|
|
2117
|
+
const edges = graph.edges.map((e) => ({ From: e.dependsOnTaskId, To: e.taskId }));
|
|
2118
|
+
const positions = LayoutGraphNodes([...graph.entityById.keys()], edges, { Direction: 'LR' });
|
|
2119
|
+
for (const task of needsLayout) {
|
|
2120
|
+
const position = positions.get(task.ID);
|
|
2121
|
+
if (!position)
|
|
2122
|
+
continue;
|
|
2123
|
+
const existing = this.parseConfiguration(task);
|
|
2124
|
+
const merged = {
|
|
2125
|
+
...existing,
|
|
2126
|
+
layout: { x: position.X, y: position.Y },
|
|
2127
|
+
};
|
|
2128
|
+
task.Configuration = JSON.stringify(merged);
|
|
2129
|
+
if (!(await task.Save())) {
|
|
2130
|
+
LogError(`[TaskGraphDispatcher] Could not save computed layout for ${task.ID}: ${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
2131
|
+
}
|
|
2132
|
+
}
|
|
2133
|
+
}
|
|
2134
|
+
catch (e) {
|
|
2135
|
+
LogError(`[TaskGraphDispatcher] Could not compute a layout for the settled graph: ${e instanceof Error ? e.message : String(e)}`);
|
|
2136
|
+
}
|
|
2137
|
+
}
|
|
2138
|
+
/**
|
|
2139
|
+
* The earliest moment any step in the graph began, or null when none has.
|
|
2140
|
+
*
|
|
2141
|
+
* Null is a real answer — a graph whose tasks are all still Pending has not started — and is
|
|
2142
|
+
* deliberately not collapsed to "now", which would date the graph from whenever this pass
|
|
2143
|
+
* happened to run.
|
|
2144
|
+
*/
|
|
2145
|
+
earliestStart(entityById) {
|
|
2146
|
+
let earliest = null;
|
|
2147
|
+
for (const entity of entityById.values()) {
|
|
2148
|
+
if (!entity.StartedAt)
|
|
2149
|
+
continue;
|
|
2150
|
+
if (earliest === null || entity.StartedAt < earliest)
|
|
2151
|
+
earliest = entity.StartedAt;
|
|
2152
|
+
}
|
|
2153
|
+
return earliest;
|
|
2154
|
+
}
|
|
2155
|
+
/**
|
|
2156
|
+
* The step's Configuration with this run's artefacts folded in, or `undefined` to leave it be.
|
|
2157
|
+
*
|
|
2158
|
+
* **Merged into the authored bag, never written over it.** The Configuration column holds the
|
|
2159
|
+
* step's definition — its loop body, its mappings, its policy, the position someone dragged it
|
|
2160
|
+
* to. Writing a fresh object containing only `runtime` would erase all of that the first time a
|
|
2161
|
+
* prompt step completed, which is the kind of loss that surfaces much later as a workflow that
|
|
2162
|
+
* mysteriously stopped mapping its output.
|
|
2163
|
+
*
|
|
2164
|
+
* Returns `undefined` when there is nothing to record, so the guarded write omits the column
|
|
2165
|
+
* rather than rewriting it with what it already held.
|
|
2166
|
+
*/
|
|
2167
|
+
configurationWithRuntime(task, promptRunID, actionLogID, iterations, payloadAtStart) {
|
|
2168
|
+
if (!promptRunID && !actionLogID && !iterations?.length && !payloadAtStart)
|
|
2169
|
+
return undefined;
|
|
2170
|
+
const existing = this.parseConfiguration(task);
|
|
2171
|
+
const merged = {
|
|
2172
|
+
...existing,
|
|
2173
|
+
runtime: {
|
|
2174
|
+
...existing?.runtime,
|
|
2175
|
+
...(promptRunID ? { promptRunID } : {}),
|
|
2176
|
+
...(actionLogID ? { actionLogID } : {}),
|
|
2177
|
+
// Replaced wholesale rather than appended: this is the trace of the loop's LAST
|
|
2178
|
+
// execution, and a retried step that concatenated would report a loop that ran twice
|
|
2179
|
+
// as many passes as it did.
|
|
2180
|
+
...(iterations?.length ? { iterations } : {}),
|
|
2181
|
+
// The resolved before-state, so the run view has something to diff the output
|
|
2182
|
+
// against. NOT written to Task.InputPayload, which holds the AUTHORED input and
|
|
2183
|
+
// round-trips back out as part of the spec.
|
|
2184
|
+
...(payloadAtStart ? { payloadAtStart } : {}),
|
|
2185
|
+
},
|
|
2186
|
+
};
|
|
2187
|
+
return JSON.stringify(merged);
|
|
2188
|
+
}
|
|
2189
|
+
/**
|
|
2190
|
+
* Reads a step's Configuration bag, tolerating a row whose JSON cannot be parsed.
|
|
2191
|
+
*
|
|
2192
|
+
* Unparseable configuration is logged rather than thrown: the step has already RUN by the time
|
|
2193
|
+
* this is called, and refusing to record its outcome because its definition is malformed would
|
|
2194
|
+
* discard the result of real work and leave the task claimed until the claim lapsed.
|
|
2195
|
+
*/
|
|
2196
|
+
parseConfiguration(task) {
|
|
2197
|
+
if (!task.Configuration)
|
|
2198
|
+
return undefined;
|
|
2199
|
+
try {
|
|
2200
|
+
return JSON.parse(task.Configuration);
|
|
2201
|
+
}
|
|
2202
|
+
catch (e) {
|
|
2203
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} has unparseable Configuration; ` +
|
|
2204
|
+
`recording runtime artefacts against an empty bag. ${e instanceof Error ? e.message : String(e)}`);
|
|
2205
|
+
return undefined;
|
|
2206
|
+
}
|
|
2207
|
+
}
|
|
2208
|
+
/**
|
|
2209
|
+
* The payload a step sees: everything its prerequisites produced, plus its own declared input.
|
|
2210
|
+
*
|
|
2211
|
+
* **Why the outputs are merged rather than kept per-task.** A flow carried ONE payload that
|
|
2212
|
+
* accumulated as it went, so a condition on the edge into step C could read a value step A wrote.
|
|
2213
|
+
* Handing each task only its immediate predecessor's output would silently narrow that: the
|
|
2214
|
+
* condition reads `undefined`, which is falsy, and the workflow quietly takes a different route
|
|
2215
|
+
* than the flow it was compiled from. Merging in dependency order restores the accumulation.
|
|
2216
|
+
*
|
|
2217
|
+
* Later prerequisites win on a key collision, matching a flow's own last-write-wins behaviour.
|
|
2218
|
+
*/
|
|
2219
|
+
mergedPayload(inputPayload, dependencyOutputs) {
|
|
2220
|
+
const merged = {};
|
|
2221
|
+
for (const output of dependencyOutputs.values()) {
|
|
2222
|
+
if (output && typeof output === 'object' && !Array.isArray(output)) {
|
|
2223
|
+
Object.assign(merged, output);
|
|
2224
|
+
}
|
|
2225
|
+
}
|
|
2226
|
+
if (inputPayload && typeof inputPayload === 'object' && !Array.isArray(inputPayload)) {
|
|
2227
|
+
Object.assign(merged, inputPayload);
|
|
2228
|
+
}
|
|
2229
|
+
return merged;
|
|
2230
|
+
}
|
|
734
2231
|
}
|
|
735
2232
|
//# sourceMappingURL=TaskGraphDispatcher.js.map
|