@memberjunction/task-graph 0.0.0 → 6.1.0-edge.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +7 -0
- package/README.md +167 -27
- package/dist/DispatcherConditionEvaluator.d.ts +10 -0
- package/dist/DispatcherConditionEvaluator.d.ts.map +1 -0
- package/dist/DispatcherConditionEvaluator.js +27 -0
- package/dist/DispatcherConditionEvaluator.js.map +1 -0
- package/dist/TaskClaimStore.d.ts +125 -0
- package/dist/TaskClaimStore.d.ts.map +1 -0
- package/dist/TaskClaimStore.js +223 -0
- package/dist/TaskClaimStore.js.map +1 -0
- package/dist/TaskGraphDispatcher.d.ts +548 -0
- package/dist/TaskGraphDispatcher.d.ts.map +1 -0
- package/dist/TaskGraphDispatcher.js +2232 -0
- package/dist/TaskGraphDispatcher.js.map +1 -0
- package/dist/TaskGraphService.d.ts +247 -0
- package/dist/TaskGraphService.d.ts.map +1 -0
- package/dist/TaskGraphService.js +753 -0
- package/dist/TaskGraphService.js.map +1 -0
- package/dist/TaskGraphSubmitterImpl.d.ts +7 -0
- package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -0
- package/dist/TaskGraphSubmitterImpl.js +52 -0
- package/dist/TaskGraphSubmitterImpl.js.map +1 -0
- package/dist/TaskLoopExecutor.d.ts +62 -0
- package/dist/TaskLoopExecutor.d.ts.map +1 -0
- package/dist/TaskLoopExecutor.js +248 -0
- package/dist/TaskLoopExecutor.js.map +1 -0
- package/dist/WorkflowSpecSync.d.ts +197 -0
- package/dist/WorkflowSpecSync.d.ts.map +1 -0
- package/dist/WorkflowSpecSync.js +474 -0
- package/dist/WorkflowSpecSync.js.map +1 -0
- package/dist/index.d.ts +19 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +19 -0
- package/dist/index.js.map +1 -0
- package/dist/operations/TaskGraphOperations.d.ts +39 -0
- package/dist/operations/TaskGraphOperations.d.ts.map +1 -0
- package/dist/operations/TaskGraphOperations.js +168 -0
- package/dist/operations/TaskGraphOperations.js.map +1 -0
- package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
- package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
- package/dist/operations/WorkflowDraftOperation.js +141 -0
- package/dist/operations/WorkflowDraftOperation.js.map +1 -0
- package/dist/operations/WorkflowOperations.d.ts +22 -0
- package/dist/operations/WorkflowOperations.d.ts.map +1 -0
- package/dist/operations/WorkflowOperations.js +99 -0
- package/dist/operations/WorkflowOperations.js.map +1 -0
- package/dist/types.d.ts +328 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +9 -0
- package/dist/types.js.map +1 -0
- package/package.json +36 -8
|
@@ -0,0 +1,2232 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Durable, host-agnostic execution of submitted task graphs.
|
|
3
|
+
*
|
|
4
|
+
* The dispatcher is what makes task graphs survive things the old client-driven path could not: a
|
|
5
|
+
* page reload, the submitting agent run ending, a server restart, or a second server instance
|
|
6
|
+
* running the same table. It polls for claimable work, claims atomically, executes with a fresh
|
|
7
|
+
* provider per task, and reconciles orphaned state on a timer.
|
|
8
|
+
*
|
|
9
|
+
* **What it deliberately does not do.** It does not decide graph semantics. Eligibility, failure
|
|
10
|
+
* propagation, parent rollup and stall detection all come from the pure algorithms in
|
|
11
|
+
* `@memberjunction/ai-core-plus` — the same functions the in-run executor consumes. That is
|
|
12
|
+
* the whole reason those were factored out dependency-free: the in-run executor and the durable
|
|
13
|
+
* executor cannot drift apart if neither owns the rules.
|
|
14
|
+
*
|
|
15
|
+
* **Host-agnostic by construction.** Provider minting and agent execution arrive as injected
|
|
16
|
+
* dependencies (`ProviderFactory`, `TaskAgentRunner`), so this package never imports MJServer. The
|
|
17
|
+
* dependency runs MJServer -> task-graph, never the reverse.
|
|
18
|
+
*
|
|
19
|
+
* @module @memberjunction/task-graph
|
|
20
|
+
*/
|
|
21
|
+
import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
|
|
22
|
+
import { LogError, LogStatus, RunView } from '@memberjunction/core';
|
|
23
|
+
import { ShutdownRegistry, UUIDsEqual } from '@memberjunction/global';
|
|
24
|
+
import { TaskClaimStore } from './TaskClaimStore.js';
|
|
25
|
+
import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
|
|
26
|
+
import { RunForEachLoop, RunWhileLoop } from './TaskLoopExecutor.js';
|
|
27
|
+
import { NotificationEngine } from '@memberjunction/notifications';
|
|
28
|
+
/** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
|
|
29
|
+
const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
|
|
30
|
+
/**
|
|
31
|
+
* Written to a human task's `ClaimedBy` once its assignee has been told it is ready.
|
|
32
|
+
*
|
|
33
|
+
* A human task has no executor, so the claim column is otherwise unused — which makes it the natural
|
|
34
|
+
* place to record a fact that must survive a restart. Reconciliation already exempts human tasks
|
|
35
|
+
* from reclamation, so this value is never mistaken for a live claim.
|
|
36
|
+
*/
|
|
37
|
+
const HUMAN_TASK_NOTIFIED_MARKER = '__human-notified__';
|
|
38
|
+
/**
|
|
39
|
+
* The run-query capability of a provider, when it has one.
|
|
40
|
+
*
|
|
41
|
+
* `IMetadataProvider` does not extend `IRunQueryProvider`, but every provider that ships implements
|
|
42
|
+
* both. Narrowing by CAPABILITY rather than casting states that honestly: a provider that genuinely
|
|
43
|
+
* cannot run queries returns undefined and the caller reports it, instead of the call failing later
|
|
44
|
+
* behind a type assertion that claimed it could.
|
|
45
|
+
*/
|
|
46
|
+
function asRunQueryProvider(provider) {
|
|
47
|
+
const candidate = provider;
|
|
48
|
+
return typeof candidate.RunQuery === 'function' ? candidate : undefined;
|
|
49
|
+
}
|
|
50
|
+
import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata } from './TaskGraphService.js';
|
|
51
|
+
import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
|
|
52
|
+
/**
|
|
53
|
+
* Renders a loop's bindings as template values.
|
|
54
|
+
*
|
|
55
|
+
* Template parameters are strings; an item is usually an object. Objects are JSON-encoded rather
|
|
56
|
+
* than dropped, because `{{ field }}` printing `[object Object]` — or nothing at all — is exactly
|
|
57
|
+
* the silent failure this exists to prevent.
|
|
58
|
+
*/
|
|
59
|
+
function stringifyBindings(bindings) {
|
|
60
|
+
const out = {};
|
|
61
|
+
for (const [key, value] of Object.entries(bindings)) {
|
|
62
|
+
out[key] = typeof value === 'string' ? value : JSON.stringify(value, null, 2);
|
|
63
|
+
}
|
|
64
|
+
return out;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* How much of a loop's per-pass payloads may be kept, and what happens when that runs out.
|
|
68
|
+
*
|
|
69
|
+
* **Why a budget exists at all.** A loop's trace lives inside one `Configuration` column, and its
|
|
70
|
+
* size is the product of two things nobody bounds: how many passes the loop runs, and how large the
|
|
71
|
+
* body's input and output are. A hundred-pass loop over documents would put megabytes in a column
|
|
72
|
+
* that the run tree, the timeline, the canvas and the Workflows list all read — punishing every
|
|
73
|
+
* reader of the row for a detail only someone inspecting one pass will ever open.
|
|
74
|
+
*
|
|
75
|
+
* **What it protects.** Only the payloads. `promptRunID` / `agentRunID` / `actionLogID` / `success`
|
|
76
|
+
* are always recorded: those point at the durable rows where the real forensics live, and they are
|
|
77
|
+
* what cost roll-up and the timeline traverse. Losing a payload costs a reader some detail; losing a
|
|
78
|
+
* pointer would lose the pass.
|
|
79
|
+
*
|
|
80
|
+
* **Omission is stated, never silent.** Once the budget is spent, further passes record a marker
|
|
81
|
+
* saying so and how large the value was, because a pass showing nothing is indistinguishable from a
|
|
82
|
+
* pass that produced nothing — and that ambiguity is exactly the failure this whole area keeps
|
|
83
|
+
* hitting.
|
|
84
|
+
*/
|
|
85
|
+
const ITERATION_PAYLOAD_BUDGET_BYTES = 128 * 1024;
|
|
86
|
+
/** Per-value cap, so one enormous pass cannot consume the whole budget by itself. */
|
|
87
|
+
const ITERATION_PAYLOAD_VALUE_BYTES = 16 * 1024;
|
|
88
|
+
class IterationPayloadBudget {
|
|
89
|
+
constructor() {
|
|
90
|
+
this.spent = 0;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* The value if it fits, or a marker describing what was left out.
|
|
94
|
+
*
|
|
95
|
+
* @returns the value, a marker object, or undefined when there was nothing to record
|
|
96
|
+
*/
|
|
97
|
+
Take(value) {
|
|
98
|
+
if (value == null)
|
|
99
|
+
return undefined;
|
|
100
|
+
const asRecord = value && typeof value === 'object' && !Array.isArray(value)
|
|
101
|
+
? value
|
|
102
|
+
: { value };
|
|
103
|
+
let size;
|
|
104
|
+
try {
|
|
105
|
+
size = JSON.stringify(asRecord)?.length ?? 0;
|
|
106
|
+
}
|
|
107
|
+
catch {
|
|
108
|
+
// Circular or otherwise unserializable. It could not be persisted anyway, and saying so
|
|
109
|
+
// is better than a pass that silently shows nothing.
|
|
110
|
+
return { __omitted: 'unserializable' };
|
|
111
|
+
}
|
|
112
|
+
if (size > ITERATION_PAYLOAD_VALUE_BYTES) {
|
|
113
|
+
return { __omitted: 'too-large', __bytes: size, __limit: ITERATION_PAYLOAD_VALUE_BYTES };
|
|
114
|
+
}
|
|
115
|
+
if (this.spent + size > ITERATION_PAYLOAD_BUDGET_BYTES) {
|
|
116
|
+
return { __omitted: 'budget-exhausted', __bytes: size, __limit: ITERATION_PAYLOAD_BUDGET_BYTES };
|
|
117
|
+
}
|
|
118
|
+
this.spent += size;
|
|
119
|
+
return asRecord;
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
/** Deep-merges a prompt's JSON response into the payload, preserving what earlier steps established. */
|
|
123
|
+
function deepMergePayload(base, incoming) {
|
|
124
|
+
const out = { ...base };
|
|
125
|
+
for (const [key, value] of Object.entries(incoming)) {
|
|
126
|
+
const existing = out[key];
|
|
127
|
+
const bothPlainObjects = existing && typeof existing === 'object' && !Array.isArray(existing) &&
|
|
128
|
+
value && typeof value === 'object' && !Array.isArray(value);
|
|
129
|
+
out[key] = bothPlainObjects
|
|
130
|
+
? deepMergePayload(existing, value)
|
|
131
|
+
: value;
|
|
132
|
+
}
|
|
133
|
+
return out;
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Statuses at which an origin's outgoing conditions may be decided.
|
|
137
|
+
*
|
|
138
|
+
* `Skipped` is included: a branch that was not taken IS settled, and a condition on an edge leaving
|
|
139
|
+
* it should resolve rather than hang the graph forever.
|
|
140
|
+
*/
|
|
141
|
+
const TERMINAL_FOR_CONDITIONS = new Set([
|
|
142
|
+
'Complete', 'Failed', 'Cancelled', 'Skipped',
|
|
143
|
+
]);
|
|
144
|
+
export class TaskGraphDispatcher {
|
|
145
|
+
constructor(providerFactory, agentRunner, contextUser, config,
|
|
146
|
+
/**
|
|
147
|
+
* Optional. Absent means a host that cannot post messages or start agent turns — a worker,
|
|
148
|
+
* a test. The dispatcher still records and logs every completion, so a graph's outcome is
|
|
149
|
+
* never lost; it simply is not announced.
|
|
150
|
+
*/
|
|
151
|
+
continuationDeliverer,
|
|
152
|
+
/**
|
|
153
|
+
* Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
|
|
154
|
+
* announces nothing.
|
|
155
|
+
*/
|
|
156
|
+
observer,
|
|
157
|
+
/**
|
|
158
|
+
* Optional. Absent means this host cannot run action nodes; they stay Pending and visible
|
|
159
|
+
* rather than being failed, because "nobody here can run this" is not "this ran and broke".
|
|
160
|
+
*/
|
|
161
|
+
actionRunner,
|
|
162
|
+
/**
|
|
163
|
+
* Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
|
|
164
|
+
* rather than being failed, for the same reason action nodes do.
|
|
165
|
+
*/
|
|
166
|
+
promptRunner) {
|
|
167
|
+
this.providerFactory = providerFactory;
|
|
168
|
+
this.agentRunner = agentRunner;
|
|
169
|
+
this.contextUser = contextUser;
|
|
170
|
+
this.continuationDeliverer = continuationDeliverer;
|
|
171
|
+
this.observer = observer;
|
|
172
|
+
this.actionRunner = actionRunner;
|
|
173
|
+
this.promptRunner = promptRunner;
|
|
174
|
+
this.running = false;
|
|
175
|
+
this.pollTimer = null;
|
|
176
|
+
this.reconcileTimer = null;
|
|
177
|
+
/** Tasks this instance is currently executing — bounds concurrency and drives heartbeats. */
|
|
178
|
+
this.inFlight = new Set();
|
|
179
|
+
/** Guards against a slow poll overlapping the next tick. */
|
|
180
|
+
this.polling = false;
|
|
181
|
+
/**
|
|
182
|
+
* The poll pass currently running, so `Stop` can wait for it.
|
|
183
|
+
*
|
|
184
|
+
* `clearInterval` cannot cancel a tick that has already fired, and a pass is a long sequence of
|
|
185
|
+
* awaits (provider, rollup, claim query) — so without this, `Stop` returns while a pass is still
|
|
186
|
+
* mid-flight and about to claim. Its tasks then land in `inFlight` AFTER the drain loop already
|
|
187
|
+
* saw an empty set, which is precisely the state the drain exists to prevent.
|
|
188
|
+
*/
|
|
189
|
+
this.pollPass = null;
|
|
190
|
+
/** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
|
|
191
|
+
this.ownerByParentID = new Map();
|
|
192
|
+
/** Name shown in the shutdown drain log. */
|
|
193
|
+
this.ShutdownName = 'TaskGraphDispatcher';
|
|
194
|
+
this.config = { ...DEFAULT_DISPATCHER_CONFIG, ...config };
|
|
195
|
+
this.claims = new TaskClaimStore(this.config.InstanceID, this.config.ClaimTTLSeconds);
|
|
196
|
+
this.conditionEvaluator = new DispatcherConditionEvaluator();
|
|
197
|
+
}
|
|
198
|
+
/**
|
|
199
|
+
* Announce something that happened, and never let the announcement matter.
|
|
200
|
+
*
|
|
201
|
+
* A frame is commentary on work, never a step of it, so an observer that throws must not be able
|
|
202
|
+
* to fail a task or stall a graph. Swallowing here rather than asking every implementation to be
|
|
203
|
+
* careful means one place enforces it.
|
|
204
|
+
*/
|
|
205
|
+
emit(frame) {
|
|
206
|
+
if (!this.observer)
|
|
207
|
+
return;
|
|
208
|
+
try {
|
|
209
|
+
this.observer.OnFrame(frame);
|
|
210
|
+
}
|
|
211
|
+
catch (e) {
|
|
212
|
+
LogError(`[TaskGraphDispatcher] Observer threw on ${frame.Kind} (ignored): ${e instanceof Error ? e.message : String(e)}`);
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
/**
|
|
216
|
+
* Who a graph belongs to, memoized for the process's lifetime.
|
|
217
|
+
*
|
|
218
|
+
* Read from the parent's durable metadata rather than a column, because `Task.UserID` means
|
|
219
|
+
* "the person this task is waiting on" — setting it on a parent would make every graph look
|
|
220
|
+
* like a human task. Memoized because frames are emitted per step: without the cache, watching
|
|
221
|
+
* a run would cost one query per event, and observability that scales with work is the thing a
|
|
222
|
+
* push mechanism exists to avoid. Ownership never changes for a given graph, so the cache can
|
|
223
|
+
* never go stale.
|
|
224
|
+
*
|
|
225
|
+
* Skipped entirely when nobody is observing — the lookup exists only to address frames.
|
|
226
|
+
*/
|
|
227
|
+
async resolveOwner(provider, parentTaskID) {
|
|
228
|
+
if (!this.observer)
|
|
229
|
+
return null;
|
|
230
|
+
const cached = this.ownerByParentID.get(parentTaskID);
|
|
231
|
+
if (cached !== undefined)
|
|
232
|
+
return cached;
|
|
233
|
+
let owner = null;
|
|
234
|
+
try {
|
|
235
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
236
|
+
if (await parent.Load(parentTaskID)) {
|
|
237
|
+
owner = this.readParentMetadata(parent).submittedByUserID ?? null;
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
catch (e) {
|
|
241
|
+
LogError(`[TaskGraphDispatcher] Could not resolve owner for graph ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
242
|
+
}
|
|
243
|
+
this.ownerByParentID.set(parentTaskID, owner);
|
|
244
|
+
return owner;
|
|
245
|
+
}
|
|
246
|
+
/**
|
|
247
|
+
* Begins dispatching.
|
|
248
|
+
*
|
|
249
|
+
* Runs reconciliation FIRST, before accepting any new work. On a restart this instance may be
|
|
250
|
+
* looking at tasks its own previous incarnation claimed and never released — reclaiming those
|
|
251
|
+
* up front is what turns a crash from "work stranded forever" into "work resumes".
|
|
252
|
+
*/
|
|
253
|
+
async Start() {
|
|
254
|
+
if (this.running)
|
|
255
|
+
return;
|
|
256
|
+
this.running = true;
|
|
257
|
+
// Self-register rather than make each host remember to stop us. A dispatcher that keeps
|
|
258
|
+
// polling through a graceful shutdown would claim work the process is about to abandon,
|
|
259
|
+
// which is exactly the orphaned-claim state reconciliation exists to clean up.
|
|
260
|
+
ShutdownRegistry.Instance.Register(this);
|
|
261
|
+
LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
|
|
262
|
+
await this.Reconcile();
|
|
263
|
+
this.pollTimer = setInterval(() => { this.pollPass = this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
|
|
264
|
+
this.reconcileTimer = setInterval(() => { void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
|
|
265
|
+
}
|
|
266
|
+
/**
|
|
267
|
+
* Stops accepting new work and waits for in-flight tasks to finish.
|
|
268
|
+
*
|
|
269
|
+
* Deliberately does NOT release claims on the way out: an abandoned claim expires on its own,
|
|
270
|
+
* and releasing eagerly would hand a still-running task to another instance mid-execution.
|
|
271
|
+
* Letting the TTL do it is the safer failure mode.
|
|
272
|
+
*/
|
|
273
|
+
async Stop() {
|
|
274
|
+
this.running = false;
|
|
275
|
+
if (this.pollTimer) {
|
|
276
|
+
clearInterval(this.pollTimer);
|
|
277
|
+
this.pollTimer = null;
|
|
278
|
+
}
|
|
279
|
+
if (this.reconcileTimer) {
|
|
280
|
+
clearInterval(this.reconcileTimer);
|
|
281
|
+
this.reconcileTimer = null;
|
|
282
|
+
}
|
|
283
|
+
// Drain the poll pass BEFORE the task drain below, not after: a pass still running has not
|
|
284
|
+
// necessarily claimed anything yet, so `inFlight` can be empty while work is moments from
|
|
285
|
+
// starting. Clearing `running` above stops that pass claiming anything further; this waits
|
|
286
|
+
// for it to notice. Its own failures are already logged inside `pollOnce`.
|
|
287
|
+
if (this.pollPass) {
|
|
288
|
+
await this.pollPass.catch(() => undefined);
|
|
289
|
+
this.pollPass = null;
|
|
290
|
+
}
|
|
291
|
+
const deadline = Date.now() + 30_000;
|
|
292
|
+
while (this.inFlight.size > 0 && Date.now() < deadline) {
|
|
293
|
+
await new Promise((r) => setTimeout(r, 250));
|
|
294
|
+
}
|
|
295
|
+
if (this.inFlight.size > 0) {
|
|
296
|
+
LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will expire.`);
|
|
297
|
+
}
|
|
298
|
+
LogStatus(`[TaskGraphDispatcher] Stopped.`);
|
|
299
|
+
}
|
|
300
|
+
/** {@link IShutdownable} — idempotent by way of `Stop`'s `running` guard. */
|
|
301
|
+
async Shutdown() {
|
|
302
|
+
await this.Stop();
|
|
303
|
+
}
|
|
304
|
+
/**
|
|
305
|
+
* Reclaims expired claims and reports anomalies.
|
|
306
|
+
*
|
|
307
|
+
* Also enforces the two schema promises that previously had no enforcer anywhere: agent tasks
|
|
308
|
+
* left `In Progress` with no claim are surfaced loudly rather than silently corrected, since
|
|
309
|
+
* that shape indicates tampering or a bug and Record Changes already carries the audit trail.
|
|
310
|
+
*/
|
|
311
|
+
async Reconcile() {
|
|
312
|
+
let provider = null;
|
|
313
|
+
try {
|
|
314
|
+
provider = await this.providerFactory.CreateProvider();
|
|
315
|
+
const released = await this.claims.ReleaseExpiredClaims(provider, this.contextUser);
|
|
316
|
+
const orphaned = await this.claims.FindOrphanedInProgress(provider, this.contextUser);
|
|
317
|
+
if (released.length > 0 || orphaned.length > 0) {
|
|
318
|
+
LogStatus(`[TaskGraphDispatcher] Reconciliation: ${released.length} expired claim(s) released, ` +
|
|
319
|
+
`${orphaned.length} orphaned task(s) reported.`);
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
catch (e) {
|
|
323
|
+
LogError(`[TaskGraphDispatcher] Reconciliation failed: ${e instanceof Error ? e.message : String(e)}`);
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
/**
|
|
327
|
+
* One dispatch pass: find claimable work, claim what fits under the concurrency cap, execute.
|
|
328
|
+
*
|
|
329
|
+
* Overlap-guarded — a pass that runs long simply skips the next tick rather than stacking, which
|
|
330
|
+
* would otherwise let a slow database multiply in-flight work past the cap.
|
|
331
|
+
*/
|
|
332
|
+
async pollOnce() {
|
|
333
|
+
if (!this.running || this.polling)
|
|
334
|
+
return;
|
|
335
|
+
const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
|
|
336
|
+
if (capacity <= 0)
|
|
337
|
+
return;
|
|
338
|
+
this.polling = true;
|
|
339
|
+
try {
|
|
340
|
+
const provider = await this.providerFactory.CreateProvider();
|
|
341
|
+
// `running` is re-read after every await from here on. The entry check above only proves
|
|
342
|
+
// the dispatcher was live when the tick fired; each await is a point where `Stop` can
|
|
343
|
+
// land, and a stopped instance must neither mutate graph state nor take new work. Left
|
|
344
|
+
// unchecked, a stopped dispatcher goes on to roll up graphs (emitting GraphSettled to an
|
|
345
|
+
// observer nobody is listening to any more) and to claim tasks it will never run — which
|
|
346
|
+
// then sit claimed until their lease expires.
|
|
347
|
+
if (!this.running)
|
|
348
|
+
return;
|
|
349
|
+
// Settle graphs before picking new work, so a failure earlier in this pass stops its
|
|
350
|
+
// branch immediately rather than after another wave has already launched.
|
|
351
|
+
await this.propagateAndRollup(provider);
|
|
352
|
+
if (!this.running)
|
|
353
|
+
return;
|
|
354
|
+
const candidates = await this.findClaimableTasks(provider, capacity);
|
|
355
|
+
for (const task of candidates) {
|
|
356
|
+
// Re-checked per iteration, not just before the loop: claiming is itself awaited, so
|
|
357
|
+
// a multi-task wave can straddle a Stop.
|
|
358
|
+
if (!this.running)
|
|
359
|
+
break;
|
|
360
|
+
if (this.inFlight.size >= this.config.MaxConcurrentTasks)
|
|
361
|
+
break;
|
|
362
|
+
if (!(await this.claims.TryClaim(provider, task.ID, this.contextUser))) {
|
|
363
|
+
// Another instance won the race, or the task is no longer Pending. Normal.
|
|
364
|
+
continue;
|
|
365
|
+
}
|
|
366
|
+
this.inFlight.add(task.ID);
|
|
367
|
+
// Intentionally not awaited — the poll loop must keep dispatching while this runs.
|
|
368
|
+
void this.executeClaimed(task.ID).finally(() => this.inFlight.delete(task.ID));
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
catch (e) {
|
|
372
|
+
LogError(`[TaskGraphDispatcher] Poll failed: ${e instanceof Error ? e.message : String(e)}`);
|
|
373
|
+
}
|
|
374
|
+
finally {
|
|
375
|
+
this.polling = false;
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
/**
|
|
379
|
+
* Executes one claimed task on its own provider, heartbeating until it settles.
|
|
380
|
+
*
|
|
381
|
+
* A fresh provider per task is the point of `ProviderFactory`: parallel tasks must not share a
|
|
382
|
+
* transaction scope or entity instances, or one task's work becomes visible inside another's.
|
|
383
|
+
*/
|
|
384
|
+
async executeClaimed(taskID) {
|
|
385
|
+
let heartbeat = null;
|
|
386
|
+
try {
|
|
387
|
+
const provider = await this.providerFactory.CreateProvider();
|
|
388
|
+
const task = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
389
|
+
if (!(await task.Load(taskID))) {
|
|
390
|
+
LogError(`[TaskGraphDispatcher] Claimed task ${taskID} could not be loaded.`);
|
|
391
|
+
return;
|
|
392
|
+
}
|
|
393
|
+
heartbeat = setInterval(() => {
|
|
394
|
+
void this.claims.Heartbeat(provider, taskID, this.contextUser).then((ok) => {
|
|
395
|
+
if (!ok) {
|
|
396
|
+
// Lost ownership — reconciliation reclaimed it, or a human intervened.
|
|
397
|
+
LogError(`[TaskGraphDispatcher] Lost claim on task ${taskID} while executing; another instance may take it over.`);
|
|
398
|
+
}
|
|
399
|
+
});
|
|
400
|
+
}, this.config.HeartbeatIntervalSeconds * 1000);
|
|
401
|
+
// Emitted after the claim is held, not before: a frame saying "started" for work another
|
|
402
|
+
// instance actually took would be a lie a viewer cannot detect.
|
|
403
|
+
const graphID = task.ParentID ?? taskID;
|
|
404
|
+
const ownerUserID = await this.resolveOwner(provider, graphID);
|
|
405
|
+
this.emit({ Kind: 'TaskStarted', ParentTaskID: graphID, OwnerUserID: ownerUserID, TaskID: taskID, TaskName: task.Name, Status: 'In Progress' });
|
|
406
|
+
const dependencyOutputs = await this.loadDependencyOutputs(provider, taskID);
|
|
407
|
+
let inputPayload = null;
|
|
408
|
+
if (task.InputPayload) {
|
|
409
|
+
try {
|
|
410
|
+
inputPayload = JSON.parse(task.InputPayload);
|
|
411
|
+
}
|
|
412
|
+
catch (e) {
|
|
413
|
+
LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs);
|
|
417
|
+
// A prompt can end the workflow early and say why. Honour it before recording the
|
|
418
|
+
// outcome, so the remaining tasks are already Skipped by the time the rollup runs and
|
|
419
|
+
// the graph settles Complete rather than looking abandoned with work left Pending.
|
|
420
|
+
if (result.ChatMessage) {
|
|
421
|
+
await this.endGraphEarly(provider, task, result.ChatMessage);
|
|
422
|
+
}
|
|
423
|
+
const recorded = await this.claims.CompleteClaimed(provider, taskID, {
|
|
424
|
+
Status: result.Success ? 'Complete' : 'Failed',
|
|
425
|
+
OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
|
|
426
|
+
ErrorMessage: result.ErrorMessage ?? null,
|
|
427
|
+
AgentRunID: result.AgentRunID ?? null,
|
|
428
|
+
Configuration: this.configurationWithRuntime(task, result.PromptRunID, result.ActionLogID, result.Iterations, result.PayloadAtStart),
|
|
429
|
+
}, this.contextUser);
|
|
430
|
+
if (!recorded) {
|
|
431
|
+
// The guarded write refused: the row changed underneath us (cancelled, reassigned,
|
|
432
|
+
// or reclaimed). Deferring to whoever owns it now is correct — overwriting would
|
|
433
|
+
// undo a newer, deliberate decision.
|
|
434
|
+
LogError(`[TaskGraphDispatcher] Could not record outcome for ${taskID}; the task is no longer owned by this instance.`);
|
|
435
|
+
}
|
|
436
|
+
else {
|
|
437
|
+
// Only announced when the guarded write actually landed. Announcing an outcome we
|
|
438
|
+
// failed to persist would show a viewer a completion the database never recorded.
|
|
439
|
+
this.emit({
|
|
440
|
+
Kind: result.Success ? 'TaskCompleted' : 'TaskFailed',
|
|
441
|
+
ParentTaskID: graphID,
|
|
442
|
+
OwnerUserID: ownerUserID,
|
|
443
|
+
TaskID: taskID,
|
|
444
|
+
TaskName: task.Name,
|
|
445
|
+
Status: result.Success ? 'Complete' : 'Failed',
|
|
446
|
+
ErrorMessage: result.Success ? undefined : (result.ErrorMessage ?? undefined),
|
|
447
|
+
});
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
catch (e) {
|
|
451
|
+
LogError(`[TaskGraphDispatcher] Execution failed for ${taskID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
452
|
+
try {
|
|
453
|
+
const provider = await this.providerFactory.CreateProvider();
|
|
454
|
+
await this.claims.CompleteClaimed(provider, taskID, { Status: 'Failed', ErrorMessage: e instanceof Error ? e.message : String(e) }, this.contextUser);
|
|
455
|
+
}
|
|
456
|
+
catch { /* already logged; nothing further to do */ }
|
|
457
|
+
}
|
|
458
|
+
finally {
|
|
459
|
+
if (heartbeat)
|
|
460
|
+
clearInterval(heartbeat);
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
/**
|
|
464
|
+
* Applies failure propagation and parent rollup across every graph with active work.
|
|
465
|
+
*
|
|
466
|
+
* All four decisions — what is eligible, what must block, what the parent status is, whether the
|
|
467
|
+
* graph is wedged — are delegated to the pure algorithms, unchanged from Phase 1.
|
|
468
|
+
*/
|
|
469
|
+
async propagateAndRollup(provider) {
|
|
470
|
+
for (const parentID of await this.findActiveGraphIDs(provider)) {
|
|
471
|
+
// Human steps settle BEFORE the graph state is read, so an answer given since the last
|
|
472
|
+
// poll is already reflected when eligibility and rollup are computed. Doing it after
|
|
473
|
+
// would delay every dependent branch by a full poll interval for no reason — and on a
|
|
474
|
+
// graph whose only remaining work is downstream of a person, that is the difference
|
|
475
|
+
// between "answered and moving" and "answered and apparently still stuck".
|
|
476
|
+
await this.expireOverdueRequests(provider, parentID);
|
|
477
|
+
await this.settleAnsweredHumanTasks(provider, parentID);
|
|
478
|
+
await this.reopenCancelledHumanTasks(provider, parentID);
|
|
479
|
+
const graph = await this.loadGraphState(provider, parentID);
|
|
480
|
+
if (graph.nodes.length === 0)
|
|
481
|
+
continue;
|
|
482
|
+
// SKIPS FIRST — before blocking, before eligibility. A task whose gating predecessors
|
|
483
|
+
// are all Skipped is simultaneously "eligible" (Skipped satisfies a prerequisite) and
|
|
484
|
+
// "to be skipped"; deciding eligibility first would dispatch the branch nobody took.
|
|
485
|
+
//
|
|
486
|
+
// `unreachableTaskIDs` seeds this too, and that is a correction (R6). A target whose only
|
|
487
|
+
// route in was an ordinary conditional edge that evaluated DEFINITELY FALSE is a branch
|
|
488
|
+
// that was not taken — semantically identical to an XOR loser — yet it used to settle
|
|
489
|
+
// `Blocked`. That made `Blocked` mean two unrelated things: "the workflow chose another
|
|
490
|
+
// route" and "something upstream broke". A reader cannot tell those apart, so every
|
|
491
|
+
// conditional workflow looked half-failed and people went hunting for bugs that did not
|
|
492
|
+
// exist. `Blocked` is now reserved for FAILURE-driven unsatisfiability.
|
|
493
|
+
const skipSeeds = new Set([...graph.skipSeedTaskIDs, ...graph.unreachableTaskIDs]);
|
|
494
|
+
const toSkip = new Set([
|
|
495
|
+
...skipSeeds,
|
|
496
|
+
...ComputeSkipCascade(graph.nodes, graph.edges, [...skipSeeds]),
|
|
497
|
+
]);
|
|
498
|
+
for (const taskID of toSkip) {
|
|
499
|
+
const entity = graph.entityById.get(taskID);
|
|
500
|
+
if (!entity || entity.Status !== 'Pending')
|
|
501
|
+
continue;
|
|
502
|
+
entity.Status = 'Skipped';
|
|
503
|
+
if (await entity.Save()) {
|
|
504
|
+
LogStatus(`[TaskGraphDispatcher] Skipped '${entity.Name}' (${taskID}) — another branch was taken.`);
|
|
505
|
+
// Announced separately from TaskBlocked because it means something different to
|
|
506
|
+
// a viewer: nothing went wrong, this route simply was not the one chosen.
|
|
507
|
+
this.emit({
|
|
508
|
+
Kind: 'TaskSkipped',
|
|
509
|
+
ParentTaskID: parentID,
|
|
510
|
+
OwnerUserID: await this.resolveOwner(provider, parentID),
|
|
511
|
+
TaskID: taskID,
|
|
512
|
+
TaskName: entity.Name,
|
|
513
|
+
Status: 'Skipped',
|
|
514
|
+
});
|
|
515
|
+
// Keep the in-memory graph consistent so the blocking pass below and the rollup
|
|
516
|
+
// both see the skip rather than a stale Pending.
|
|
517
|
+
const node = graph.nodes.find((n) => n.id === taskID);
|
|
518
|
+
if (node)
|
|
519
|
+
node.status = 'Skipped';
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
// Only failure-driven unsatisfiability reaches here now; not-taken branches were skipped
|
|
523
|
+
// above. A task already Skipped is left alone rather than overwritten — the two passes
|
|
524
|
+
// must not fight over the same row.
|
|
525
|
+
const toBlock = [...ComputeTasksToBlock(graph.nodes, graph.edges, graph.handledFailureIDs)]
|
|
526
|
+
.filter((id) => !toSkip.has(id));
|
|
527
|
+
for (const taskID of toBlock) {
|
|
528
|
+
const entity = graph.entityById.get(taskID);
|
|
529
|
+
if (!entity)
|
|
530
|
+
continue;
|
|
531
|
+
entity.Status = 'Blocked';
|
|
532
|
+
if (await entity.Save()) {
|
|
533
|
+
LogStatus(`[TaskGraphDispatcher] Blocked '${entity.Name}' (${taskID}) — a dependency can never be satisfied.`);
|
|
534
|
+
// Worth announcing on its own: a blocked step is the one outcome a viewer would
|
|
535
|
+
// otherwise see as a task that simply never starts.
|
|
536
|
+
this.emit({
|
|
537
|
+
Kind: 'TaskBlocked',
|
|
538
|
+
ParentTaskID: parentID,
|
|
539
|
+
OwnerUserID: await this.resolveOwner(provider, parentID),
|
|
540
|
+
TaskID: taskID,
|
|
541
|
+
TaskName: entity.Name,
|
|
542
|
+
Status: 'Blocked',
|
|
543
|
+
});
|
|
544
|
+
}
|
|
545
|
+
}
|
|
546
|
+
if (IsGraphStalled(graph.nodes, graph.edges)) {
|
|
547
|
+
LogError(`[TaskGraphDispatcher] Graph ${parentID} is stalled: pending work with no satisfiable path.`);
|
|
548
|
+
}
|
|
549
|
+
const fresh = await this.loadGraphState(provider, parentID);
|
|
550
|
+
// ComputeParentRollup treats an empty child set as Complete-and-terminal, which is right
|
|
551
|
+
// for a graph that genuinely has no children and catastrophic for one whose reload came
|
|
552
|
+
// back empty transiently — it would mark live work finished and fire its continuation.
|
|
553
|
+
// The outer guard covered the first load only.
|
|
554
|
+
if (fresh.nodes.length === 0)
|
|
555
|
+
continue;
|
|
556
|
+
const rollup = ComputeParentRollup(fresh.nodes, fresh.handledFailureIDs);
|
|
557
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
558
|
+
if (!(await parent.Load(parentID)))
|
|
559
|
+
continue;
|
|
560
|
+
// A graph starts when its first step does.
|
|
561
|
+
//
|
|
562
|
+
// `StartedAt` is stamped by the CLAIM, and a parent is never claimed — it is a container,
|
|
563
|
+
// not a unit of work — so the graph row carried no start time even after it completed.
|
|
564
|
+
// A settled workflow therefore reported a CompletedAt with no beginning: it sorted as
|
|
565
|
+
// "not started" in the run tree, showed no timestamp, and no duration could be computed
|
|
566
|
+
// for the thing whose duration people actually ask about.
|
|
567
|
+
//
|
|
568
|
+
// Taken from the earliest child rather than from the clock, because that is when work
|
|
569
|
+
// genuinely began — a graph can sit Pending for a long time between submission (already
|
|
570
|
+
// recorded as CreatedAt) and a dispatcher picking up its first task.
|
|
571
|
+
const earliestChildStart = this.earliestStart(fresh.entityById);
|
|
572
|
+
const startedAtChanged = parent.StartedAt == null && earliestChildStart != null;
|
|
573
|
+
if (startedAtChanged)
|
|
574
|
+
parent.StartedAt = earliestChildStart;
|
|
575
|
+
if (startedAtChanged || parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
|
|
576
|
+
parent.Status = rollup.status;
|
|
577
|
+
parent.PercentComplete = rollup.percentComplete;
|
|
578
|
+
if (rollup.isTerminal)
|
|
579
|
+
parent.CompletedAt = new Date();
|
|
580
|
+
await parent.Save();
|
|
581
|
+
}
|
|
582
|
+
if (rollup.isTerminal) {
|
|
583
|
+
// Geometry is settled once, here, so every viewer of this run agrees on it.
|
|
584
|
+
await this.persistComputedLayout(fresh);
|
|
585
|
+
// Emitted before the continuation is delivered, and outside its once-only guard: a
|
|
586
|
+
// viewer watching the run should learn it finished whether or not this instance is
|
|
587
|
+
// the one that wins the delivery CAS.
|
|
588
|
+
this.emit({
|
|
589
|
+
Kind: 'GraphSettled',
|
|
590
|
+
ParentTaskID: parentID,
|
|
591
|
+
OwnerUserID: await this.resolveOwner(provider, parentID),
|
|
592
|
+
Status: rollup.status,
|
|
593
|
+
CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
|
|
594
|
+
TotalCount: fresh.nodes.length,
|
|
595
|
+
});
|
|
596
|
+
await this.rollUpCostToSubmittingRun(provider, parent);
|
|
597
|
+
// Deliberately AFTER the rollup and OUTSIDE its refusal paths. The rollup declines
|
|
598
|
+
// to write a number it cannot stand behind — a truncated tree, an unreachable graph
|
|
599
|
+
// — and every one of those returns early. If the run's lifecycle were settled in
|
|
600
|
+
// there, a refused rollup would strand the run parked forever, which is a far worse
|
|
601
|
+
// failure than a missing cost figure. Cost and lifecycle are separate concerns with
|
|
602
|
+
// separate failure modes, so they get separate writes.
|
|
603
|
+
await this.settleSubmittingRun(provider, parent, rollup.status);
|
|
604
|
+
await this.deliverContinuation(provider, parent, fresh);
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
/**
|
|
609
|
+
* Credits a finished graph's spending back to the agent run that submitted it.
|
|
610
|
+
*
|
|
611
|
+
* **Why this cannot happen during the run.** `BaseAgent` totals a run by walking its steps in
|
|
612
|
+
* memory at finalization — but a submitting run *ends at submission*. Submit-and-detach is the
|
|
613
|
+
* point: the run returns as soon as the graph is durable, and the graph executes afterwards,
|
|
614
|
+
* possibly minutes later on a different instance. At the moment the run computes its totals the
|
|
615
|
+
* spending has not happened yet, so there is nothing to count. The only place the number can be
|
|
616
|
+
* known is here, when the graph settles.
|
|
617
|
+
*
|
|
618
|
+
* **Why the `…Rollup` columns and not the plain ones.** `AIAgentRun` has carried six `…Rollup`
|
|
619
|
+
* columns since v3 that nothing has ever written — they exist for exactly this distinction:
|
|
620
|
+
*
|
|
621
|
+
* - `TotalCost` — what the run itself spent. For a Flow agent that is genuinely near zero: it
|
|
622
|
+
* compiled a graph and handed it off. This value is already final and is never rewritten here,
|
|
623
|
+
* so nothing that reads it today changes meaning, and no guardrail that already evaluated
|
|
624
|
+
* against it is retroactively falsified.
|
|
625
|
+
* - `TotalCostRollup` — the run plus everything it caused. Provisional until the graph settles,
|
|
626
|
+
* which is now.
|
|
627
|
+
*
|
|
628
|
+
* **The tree is the authority; these columns are its settlement-time cache.** The total is a SUM
|
|
629
|
+
* over `GetAgentRunTree`, not arithmetic of its own. The previous version walked the graph's
|
|
630
|
+
* child tasks and added each one's agent run, which was wrong in two ways that no test could
|
|
631
|
+
* see: a `Prompt` task has no agent run at all, so every prompt step's spend was simply missing;
|
|
632
|
+
* and it read each nested run's `…Rollup ?? …Total`, mixing a descendant-inclusive number with an
|
|
633
|
+
* own-spend one and depending on whether that nested graph happened to have settled yet. The
|
|
634
|
+
* tree already models every one of those cases — it reaches prompt runs through
|
|
635
|
+
* `Configuration.runtime.promptRunID`, and it descends into nested runs and their graphs
|
|
636
|
+
* structurally — so summing it cannot disagree with what the run viewer shows, because it IS
|
|
637
|
+
* what the run viewer shows.
|
|
638
|
+
*
|
|
639
|
+
* **This refuses rather than guesses.** A tree that failed to load, hit the depth cap, or does
|
|
640
|
+
* not contain the settling graph would still produce a number — a lower bound. Writing one would
|
|
641
|
+
* put an authoritative-looking total in a column every cost surface reads. Each of those cases
|
|
642
|
+
* logs and leaves the column alone, so `?? TotalCost` keeps its honest meaning: not settled.
|
|
643
|
+
*
|
|
644
|
+
* A graph with no submitting run (a scheduled job, a remote-operation caller) simply has nobody
|
|
645
|
+
* to credit — its own Task rows still carry the truth, and this returns quietly.
|
|
646
|
+
*/
|
|
647
|
+
async rollUpCostToSubmittingRun(provider, parent) {
|
|
648
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
649
|
+
if (!meta.submittedByAgentRunID)
|
|
650
|
+
return;
|
|
651
|
+
const runID = meta.submittedByAgentRunID;
|
|
652
|
+
try {
|
|
653
|
+
const runQuery = asRunQueryProvider(provider);
|
|
654
|
+
if (!runQuery) {
|
|
655
|
+
LogError(`[TaskGraphDispatcher] Cannot roll up cost for run ${runID}: provider cannot run queries.`);
|
|
656
|
+
return;
|
|
657
|
+
}
|
|
658
|
+
const tree = await LoadAgentRunTree(runID, runQuery, this.contextUser);
|
|
659
|
+
// Each of these means the sum would be a LOWER BOUND, and the column's whole contract is
|
|
660
|
+
// that it equals the tree. A known-low number presented as a total is worse than no
|
|
661
|
+
// number: the readers all fall back to TotalCost when this is null, which at least
|
|
662
|
+
// *says* it is the run's own spend rather than claiming to be the whole story.
|
|
663
|
+
//
|
|
664
|
+
// Refusing is NOT the same as leaving the column alone. A run that submitted two graphs
|
|
665
|
+
// has a rollup from the first; if the second cannot be summed, the first graph's total
|
|
666
|
+
// sits in the authoritative column excluding work that has since happened — stale, not
|
|
667
|
+
// absent, and `?? TotalCost` cannot save a reader from a non-null wrong number. So a
|
|
668
|
+
// refusal CLEARS it, restoring the fallback's honest meaning: not settled.
|
|
669
|
+
if (tree.ErrorMessage || !tree.Root) {
|
|
670
|
+
await this.clearStaleRollup(provider, runID, tree.ErrorMessage ?? 'the run tree came back empty');
|
|
671
|
+
return;
|
|
672
|
+
}
|
|
673
|
+
if (tree.Truncated) {
|
|
674
|
+
await this.clearStaleRollup(provider, runID, `the run tree hit the depth cap, so any total would silently under-report ` +
|
|
675
|
+
`(graph ${parent.ID} still carries its own costs)`);
|
|
676
|
+
return;
|
|
677
|
+
}
|
|
678
|
+
// The graph that just settled must appear in the tree. If it does not, the tree stopped
|
|
679
|
+
// at the run — the submitting step never recorded its parentTaskID — and the sum is
|
|
680
|
+
// merely the run's own spend wearing the name of a rollup. That is precisely the silent
|
|
681
|
+
// under-count this rewrite exists to remove, so it is reported rather than written.
|
|
682
|
+
if (!this.treeContainsGraph(tree.Root, parent.ID)) {
|
|
683
|
+
await this.clearStaleRollup(provider, runID, `graph ${parent.ID} is not reachable from it, so the tree cannot see the work. ` +
|
|
684
|
+
`Did the submitting step record parentTaskID?`);
|
|
685
|
+
return;
|
|
686
|
+
}
|
|
687
|
+
const totals = SumAgentRunTreeCost(tree.Root);
|
|
688
|
+
const submitting = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
689
|
+
if (!(await submitting.Load(runID))) {
|
|
690
|
+
LogError(`[TaskGraphDispatcher] Could not load run ${runID} to record graph cost against it.`);
|
|
691
|
+
return;
|
|
692
|
+
}
|
|
693
|
+
// Assignment, never accumulation. The tree already contains the run's own spend as its
|
|
694
|
+
// ROOT node, and it reads own-cost everywhere, so recomputing from scratch on every
|
|
695
|
+
// settlement lands on the same answer — which is what makes this safe to call again when
|
|
696
|
+
// a second graph settles, or when the terminal check is re-evaluated after a HITL wait.
|
|
697
|
+
submitting.TotalCostRollup = totals.Cost;
|
|
698
|
+
submitting.TotalTokensUsedRollup = totals.Tokens;
|
|
699
|
+
submitting.TotalPromptTokensUsedRollup = totals.PromptTokens;
|
|
700
|
+
submitting.TotalCompletionTokensUsedRollup = totals.CompletionTokens;
|
|
701
|
+
if (!(await submitting.Save())) {
|
|
702
|
+
LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}: ` +
|
|
703
|
+
`${submitting.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
704
|
+
return;
|
|
705
|
+
}
|
|
706
|
+
LogStatus(`[TaskGraphDispatcher] Credited graph ${parent.ID} to run ${runID}: ` +
|
|
707
|
+
`${tree.Rows.length} node(s), ${totals.Tokens} token(s), cost ${totals.Cost}.`);
|
|
708
|
+
}
|
|
709
|
+
catch (e) {
|
|
710
|
+
// A failed rollup must never fail the graph. The work finished; only the accounting for
|
|
711
|
+
// it is missing, and a graph marked Failed because its cost could not be summed would be
|
|
712
|
+
// a far worse lie than a cost of null.
|
|
713
|
+
LogError(`[TaskGraphDispatcher] Cost rollup failed for graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
714
|
+
}
|
|
715
|
+
}
|
|
716
|
+
/**
|
|
717
|
+
* Clears a rollup that can no longer be trusted, and says why.
|
|
718
|
+
*
|
|
719
|
+
* **Why clear rather than leave.** The four `…Rollup` columns are a cache of the run tree, and
|
|
720
|
+
* every reader treats a value there as the total. When the tree cannot be summed, any value
|
|
721
|
+
* already in the column was computed from an EARLIER settlement — it excludes the graph that
|
|
722
|
+
* just finished, so it is not merely incomplete, it is a wrong total presented as a right one.
|
|
723
|
+
* `?? TotalCost` protects a reader from null, not from stale.
|
|
724
|
+
*
|
|
725
|
+
* Nulling restores the invariant this whole design rests on: **when the column is present, it
|
|
726
|
+
* equals the tree.** Absent means not settled, which is exactly what a reader should conclude.
|
|
727
|
+
* A run with no rollup yet is untouched — there is nothing stale to clear, and writing nulls
|
|
728
|
+
* over nulls would churn Record Changes for nothing.
|
|
729
|
+
*/
|
|
730
|
+
async clearStaleRollup(provider, runID, reason) {
|
|
731
|
+
LogError(`[TaskGraphDispatcher] Not recording cost for run ${runID}: ${reason}.`);
|
|
732
|
+
try {
|
|
733
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
734
|
+
if (!(await run.Load(runID)))
|
|
735
|
+
return;
|
|
736
|
+
if (run.TotalCostRollup == null && run.TotalTokensUsedRollup == null)
|
|
737
|
+
return; // nothing stale
|
|
738
|
+
run.TotalCostRollup = null;
|
|
739
|
+
run.TotalTokensUsedRollup = null;
|
|
740
|
+
run.TotalPromptTokensUsedRollup = null;
|
|
741
|
+
run.TotalCompletionTokensUsedRollup = null;
|
|
742
|
+
if (!(await run.Save())) {
|
|
743
|
+
LogError(`[TaskGraphDispatcher] Could not clear the now-stale rollup on run ${runID}: ` +
|
|
744
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It still shows a total that ` +
|
|
745
|
+
`excludes the graph that just settled.`);
|
|
746
|
+
return;
|
|
747
|
+
}
|
|
748
|
+
LogStatus(`[TaskGraphDispatcher] Cleared the rollup on run ${runID}: it was computed before this ` +
|
|
749
|
+
`graph settled and can no longer be recomputed, so it would have under-reported.`);
|
|
750
|
+
}
|
|
751
|
+
catch (e) {
|
|
752
|
+
LogError(`[TaskGraphDispatcher] Could not clear the rollup on run ${runID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
/**
|
|
756
|
+
* Whether the settling graph is actually reachable from the submitting run's tree.
|
|
757
|
+
*
|
|
758
|
+
* Matched on the graph's parent Task id, which is the node the `TaskGraph` member of the query
|
|
759
|
+
* emits. A run that submitted a graph but recorded no `parentTaskID` produces a tree that stops
|
|
760
|
+
* at the run — structurally indistinguishable, at the SUM, from a run that never dispatched
|
|
761
|
+
* anything. This is the check that tells those two apart.
|
|
762
|
+
*/
|
|
763
|
+
treeContainsGraph(root, parentTaskID) {
|
|
764
|
+
for (const node of WalkAgentRunTree(root)) {
|
|
765
|
+
if (node.NodeType === 'TaskGraph' && UUIDsEqual(node.NodeID, parentTaskID))
|
|
766
|
+
return true;
|
|
767
|
+
}
|
|
768
|
+
return false;
|
|
769
|
+
}
|
|
770
|
+
/**
|
|
771
|
+
* Ends a graph early because a prompt said the work is finished.
|
|
772
|
+
*
|
|
773
|
+
* **Why `Skipped` and not `Cancelled`.** Nothing went wrong and nobody intervened — the workflow
|
|
774
|
+
* reached its own conclusion before running every drawn step, which is exactly what a reasoning
|
|
775
|
+
* step is for. `Cancelled` would tell a reader someone stopped it; `Skipped` says these routes
|
|
776
|
+
* were not taken, which is true and already the vocabulary the fork machinery uses.
|
|
777
|
+
*
|
|
778
|
+
* The message is written to the parent so the graph carries its own answer, rather than the
|
|
779
|
+
* answer living only on the step that produced it.
|
|
780
|
+
*/
|
|
781
|
+
async endGraphEarly(provider, task, message) {
|
|
782
|
+
if (!task.ParentID)
|
|
783
|
+
return;
|
|
784
|
+
try {
|
|
785
|
+
LogStatus(`[TaskGraphDispatcher] '${task.Name}' ended the workflow early: ${message}`);
|
|
786
|
+
for (const sibling of await this.loadChildTasks(provider, task.ParentID)) {
|
|
787
|
+
if (sibling.ID === task.ID || sibling.Status !== 'Pending')
|
|
788
|
+
continue;
|
|
789
|
+
sibling.Status = 'Skipped';
|
|
790
|
+
if (await sibling.Save()) {
|
|
791
|
+
this.emit({
|
|
792
|
+
Kind: 'TaskSkipped',
|
|
793
|
+
ParentTaskID: task.ParentID,
|
|
794
|
+
OwnerUserID: await this.resolveOwner(provider, task.ParentID),
|
|
795
|
+
TaskID: sibling.ID,
|
|
796
|
+
TaskName: sibling.Name,
|
|
797
|
+
Status: 'Skipped',
|
|
798
|
+
});
|
|
799
|
+
}
|
|
800
|
+
}
|
|
801
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
802
|
+
if (await parent.Load(task.ParentID)) {
|
|
803
|
+
parent.OutputPayload = JSON.stringify({ message });
|
|
804
|
+
await parent.Save();
|
|
805
|
+
}
|
|
806
|
+
}
|
|
807
|
+
catch (e) {
|
|
808
|
+
// The work itself succeeded; only the early-finish bookkeeping failed. Failing the task
|
|
809
|
+
// over that would discard a completed step's result.
|
|
810
|
+
LogError(`[TaskGraphDispatcher] Could not end graph early for ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
811
|
+
}
|
|
812
|
+
}
|
|
813
|
+
/**
|
|
814
|
+
* How deep the continuation chain already is, read from the graph's parent metadata.
|
|
815
|
+
*
|
|
816
|
+
* A run started by a graph inherits that graph's depth **plus one**. Without this every spawned
|
|
817
|
+
* run begins at zero, so a self-referencing flow — one that dispatches a graph containing itself
|
|
818
|
+
* — recurses without bound while the cap it should be hitting compares against a permanent zero.
|
|
819
|
+
*/
|
|
820
|
+
async graphContext(provider, task) {
|
|
821
|
+
if (!task.ParentID)
|
|
822
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
823
|
+
try {
|
|
824
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
825
|
+
if (!(await parent.Load(task.ParentID)))
|
|
826
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
827
|
+
return {
|
|
828
|
+
Depth: ParseTaskGraphParentMetadata(parent.InputPayload).reinvokeDepth + 1,
|
|
829
|
+
// The graph's own row carries the run that submitted it. One load answers both
|
|
830
|
+
// questions, which is why they are resolved together rather than in two passes.
|
|
831
|
+
SubmittingAgentRunID: parent.AgentRunID,
|
|
832
|
+
};
|
|
833
|
+
}
|
|
834
|
+
catch {
|
|
835
|
+
// An unreadable parent must not stop the work; depth zero is the safe reading, and the
|
|
836
|
+
// submit-time cap still guards the next hop.
|
|
837
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
838
|
+
}
|
|
839
|
+
}
|
|
840
|
+
/**
|
|
841
|
+
* Which failures the workflow drew a way out of.
|
|
842
|
+
*
|
|
843
|
+
* A Failed task with a **satisfied outgoing edge** is a handled failure: its author drew a
|
|
844
|
+
* recovery route and that route is now live. Downstream work should be released along it, and the
|
|
845
|
+
* parent should not roll up Failed because of a step the workflow explicitly planned around.
|
|
846
|
+
*
|
|
847
|
+
* Scoped to `failureSemantics: 'edges'` on purpose. Under `'block'` — every agent-emitted graph —
|
|
848
|
+
* a failure is terminal for its dependents whatever edges exist, because nobody drew those edges
|
|
849
|
+
* as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
|
|
850
|
+
* graph sail past a failure it never anticipated.
|
|
851
|
+
*/
|
|
852
|
+
async computeHandledFailures(provider, parentTaskID, nodes, edges) {
|
|
853
|
+
const handled = new Set();
|
|
854
|
+
// Cheap exit before touching the database: with no failures there is nothing to handle, and
|
|
855
|
+
// this runs on every poll for every active graph.
|
|
856
|
+
if (!nodes.some((n) => n.status === 'Failed'))
|
|
857
|
+
return handled;
|
|
858
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
859
|
+
if (!(await parent.Load(parentTaskID)))
|
|
860
|
+
return handled;
|
|
861
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
862
|
+
if (meta.failureSemantics !== 'edges')
|
|
863
|
+
return handled;
|
|
864
|
+
for (const node of nodes) {
|
|
865
|
+
if (node.status !== 'Failed')
|
|
866
|
+
continue;
|
|
867
|
+
// "Has somewhere to go" is the test. An edge out of a failed step that survived condition
|
|
868
|
+
// evaluation IS the drawn recovery route; a failed step with no outgoing edges has none,
|
|
869
|
+
// and stays terminal.
|
|
870
|
+
if (edges.some((e) => e.dependsOnTaskId === node.id))
|
|
871
|
+
handled.add(node.id);
|
|
872
|
+
}
|
|
873
|
+
return handled;
|
|
874
|
+
}
|
|
875
|
+
/** The graph's child tasks, with the fields the rollup needs. */
|
|
876
|
+
async loadChildTasks(provider, parentID) {
|
|
877
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
878
|
+
EntityName: 'MJ: Tasks',
|
|
879
|
+
ExtraFilter: `ParentID='${parentID}'`,
|
|
880
|
+
ResultType: 'entity_object',
|
|
881
|
+
BypassCache: true,
|
|
882
|
+
}, this.contextUser);
|
|
883
|
+
return (result.Success ? result.Results : []) ?? [];
|
|
884
|
+
}
|
|
885
|
+
/**
|
|
886
|
+
* Runs the graph's continuation exactly once, now that it has settled.
|
|
887
|
+
*
|
|
888
|
+
* **Why the delivery marker is written before the side effect.** Delivery is at-least-once by
|
|
889
|
+
* nature: the process can die between "the graph is done" and "the user has been told". Marking
|
|
890
|
+
* first and acting second means the worst case is a *missed* notification that shows up in the
|
|
891
|
+
* task record as delivered — recoverable, visible, and inspectable. Marking after would make the
|
|
892
|
+
* worst case a *repeated* notification on every reconciliation sweep, forever, which is both
|
|
893
|
+
* user-visible noise and, for `reinvoke`, an unbounded agent-run loop. Given one of the two has
|
|
894
|
+
* to be chosen, the quiet failure is the safe one.
|
|
895
|
+
*
|
|
896
|
+
* The marker is written with a compare-and-swap read-back, so two instances reconciling the same
|
|
897
|
+
* completed graph produce one winner rather than two.
|
|
898
|
+
*/
|
|
899
|
+
async deliverContinuation(provider, parent, graph) {
|
|
900
|
+
const meta = this.readParentMetadata(parent);
|
|
901
|
+
if (meta.continuationDeliveredAt)
|
|
902
|
+
return;
|
|
903
|
+
// At the cap, downgrade rather than refuse: the results still reach the user, the chain just
|
|
904
|
+
// stops growing. Refusing outright would lose the outcome of work that actually completed.
|
|
905
|
+
const mode = IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
|
|
906
|
+
if (mode !== 'none' && IsReinvokeCapReached(meta) && meta.continuation === 'reinvoke') {
|
|
907
|
+
LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} hit the reinvoke cap (${MAX_REINVOKE_DEPTH}); ` +
|
|
908
|
+
`delivering results as a message instead of starting another turn.`);
|
|
909
|
+
}
|
|
910
|
+
if (!(await this.claimContinuation(provider, parent.ID, meta)))
|
|
911
|
+
return;
|
|
912
|
+
if (mode === 'none')
|
|
913
|
+
return;
|
|
914
|
+
const summary = this.buildContinuationSummary(parent, graph);
|
|
915
|
+
LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} finished — ${summary}`);
|
|
916
|
+
if (!this.continuationDeliverer)
|
|
917
|
+
return;
|
|
918
|
+
const params = {
|
|
919
|
+
ParentTaskID: parent.ID,
|
|
920
|
+
WorkflowName: parent.Name,
|
|
921
|
+
ConversationDetailID: parent.ConversationDetailID ?? null,
|
|
922
|
+
SubmittedByAgentRunID: meta.submittedByAgentRunID,
|
|
923
|
+
ReinvokeDepth: meta.reinvokeDepth,
|
|
924
|
+
Tasks: [...graph.entityById.values()].map((t) => ({
|
|
925
|
+
TaskID: t.ID,
|
|
926
|
+
Name: t.Name,
|
|
927
|
+
Status: t.Status,
|
|
928
|
+
// A reference, not the payload. Inlining every task's output would swamp the
|
|
929
|
+
// continuation turn's context; the agent pulls what it needs by task ID.
|
|
930
|
+
Summary: t.OutputPayload ? `output available (${t.OutputPayload.length} chars)` : undefined,
|
|
931
|
+
ErrorMessage: t.ErrorMessage ?? undefined,
|
|
932
|
+
})),
|
|
933
|
+
Summary: summary,
|
|
934
|
+
};
|
|
935
|
+
try {
|
|
936
|
+
// Reinvoke degrades to a message when the host cannot start agent turns. Degrading is
|
|
937
|
+
// right rather than throwing: the work genuinely ran, and the user losing the results
|
|
938
|
+
// because nobody could start a follow-up turn would be the worse outcome.
|
|
939
|
+
if (mode === 'reinvoke' && this.continuationDeliverer.Reinvoke) {
|
|
940
|
+
await this.continuationDeliverer.Reinvoke(params);
|
|
941
|
+
}
|
|
942
|
+
else {
|
|
943
|
+
if (mode === 'reinvoke') {
|
|
944
|
+
LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID}: host cannot reinvoke; delivering as a message.`);
|
|
945
|
+
}
|
|
946
|
+
await this.continuationDeliverer.PostMessage(params);
|
|
947
|
+
}
|
|
948
|
+
}
|
|
949
|
+
catch (e) {
|
|
950
|
+
// Already marked delivered, so this will not retry. That is the deliberate trade stated
|
|
951
|
+
// on the marker: a missed notification visible in the record beats one repeated forever.
|
|
952
|
+
LogError(`[TaskGraphDispatcher] Continuation delivery failed for ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
953
|
+
}
|
|
954
|
+
}
|
|
955
|
+
/** Reads the parent's durable continuation metadata through the shared parser. */
|
|
956
|
+
readParentMetadata(parent) {
|
|
957
|
+
return ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
958
|
+
}
|
|
959
|
+
/**
|
|
960
|
+
* Stamps the delivery marker and confirms this instance won the race.
|
|
961
|
+
*
|
|
962
|
+
* `MJ: Tasks` stays user-writable (D20), so a plain "read, decide, write" is not enough — the
|
|
963
|
+
* read-back is what makes a lost race observable instead of producing a duplicate delivery.
|
|
964
|
+
*/
|
|
965
|
+
async claimContinuation(provider, parentID, meta) {
|
|
966
|
+
const row = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
967
|
+
if (!(await row.Load(parentID)))
|
|
968
|
+
return false;
|
|
969
|
+
const current = this.readParentMetadata(row);
|
|
970
|
+
if (current.continuationDeliveredAt)
|
|
971
|
+
return false; // a peer got there first
|
|
972
|
+
row.InputPayload = JSON.stringify({ ...meta, continuationDeliveredAt: new Date().toISOString() });
|
|
973
|
+
if (!(await row.Save())) {
|
|
974
|
+
LogError(`[TaskGraphDispatcher] Could not mark continuation delivered for ${parentID}; skipping to avoid a duplicate.`);
|
|
975
|
+
return false;
|
|
976
|
+
}
|
|
977
|
+
return true;
|
|
978
|
+
}
|
|
979
|
+
/** One line describing how the graph ended, for the completion log and message delivery. */
|
|
980
|
+
buildContinuationSummary(parent, graph) {
|
|
981
|
+
const counts = new Map();
|
|
982
|
+
for (const node of graph.nodes)
|
|
983
|
+
counts.set(node.status, (counts.get(node.status) ?? 0) + 1);
|
|
984
|
+
const breakdown = [...counts.entries()].map(([status, n]) => `${n} ${status}`).join(', ');
|
|
985
|
+
return `"${parent.Name}": ${graph.nodes.length} task(s) — ${breakdown}.`;
|
|
986
|
+
}
|
|
987
|
+
/**
|
|
988
|
+
* Tells the assignee that a human task is ready, exactly once.
|
|
989
|
+
*
|
|
990
|
+
* **Once** matters more than it looks: eligibility is recomputed on every poll, so a task parked
|
|
991
|
+
* on a person for three days would otherwise re-notify every five seconds until they acted. The
|
|
992
|
+
* marker is the task's own `ClaimedBy` — a human task has no executor to claim it, so the column
|
|
993
|
+
* is free, and reusing it means the "already notified" fact is as durable and as crash-safe as
|
|
994
|
+
* every other piece of graph state. A restart cannot resend.
|
|
995
|
+
*
|
|
996
|
+
* Best-effort by design. A notification that fails to send must not stop the graph or the poll
|
|
997
|
+
* loop; the task is still visible in the Tasks UI, so the work is discoverable even when the
|
|
998
|
+
* nudge does not arrive.
|
|
999
|
+
*/
|
|
1000
|
+
async notifyHumanTaskReady(task, provider) {
|
|
1001
|
+
if (task.ClaimedBy === HUMAN_TASK_NOTIFIED_MARKER)
|
|
1002
|
+
return;
|
|
1003
|
+
// The REQUEST is raised whether or not the task names an assignee. An unassigned human step
|
|
1004
|
+
// is a legitimate "somebody needs to look at this", and a request nobody was notified about
|
|
1005
|
+
// is still findable in the inbox — whereas returning early here is how such a step used to
|
|
1006
|
+
// become invisible work that stalled a workflow with nothing anywhere saying why.
|
|
1007
|
+
// TRANSIENT failures retry; PERMANENT ones stop. That distinction is the whole point, and
|
|
1008
|
+
// getting it wrong took a server down: retrying unconditionally meant a task whose workflow
|
|
1009
|
+
// has no owning agent — which can never succeed — was re-attempted on every poll forever,
|
|
1010
|
+
// each pass re-reading the graph, until the process was OOM-killed. The marker exists to
|
|
1011
|
+
// prevent exactly that storm; a permanent failure has to set it.
|
|
1012
|
+
const raised = await this.raiseHumanRequest(task, provider);
|
|
1013
|
+
if (raised === 'transient-failure')
|
|
1014
|
+
return; // try again next poll
|
|
1015
|
+
if (raised === 'permanent-failure') {
|
|
1016
|
+
// Nothing will change on a retry. Mark it so the loop stops, and leave the task Pending
|
|
1017
|
+
// and visible — a person can still see it in the Tasks UI, which is the fallback the
|
|
1018
|
+
// notification was only ever an accelerant for.
|
|
1019
|
+
await this.markHumanTaskNotified(task);
|
|
1020
|
+
return;
|
|
1021
|
+
}
|
|
1022
|
+
if (!task.UserID) {
|
|
1023
|
+
await this.markHumanTaskNotified(task);
|
|
1024
|
+
return;
|
|
1025
|
+
}
|
|
1026
|
+
try {
|
|
1027
|
+
await NotificationEngine.Instance.Config(false, this.contextUser);
|
|
1028
|
+
await NotificationEngine.Instance.SendNotification({
|
|
1029
|
+
userId: task.UserID,
|
|
1030
|
+
typeNameOrId: HUMAN_TASK_NOTIFICATION_TYPE,
|
|
1031
|
+
title: `Action needed: ${task.Name}`,
|
|
1032
|
+
message: task.Description || 'A workflow is waiting on you to complete this task.',
|
|
1033
|
+
resourceConfiguration: { type: 'Task', taskId: task.ID, parentTaskId: task.ParentID ?? '' },
|
|
1034
|
+
}, this.contextUser);
|
|
1035
|
+
}
|
|
1036
|
+
catch (e) {
|
|
1037
|
+
LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
1038
|
+
}
|
|
1039
|
+
await this.markHumanTaskNotified(task);
|
|
1040
|
+
// Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
|
|
1041
|
+
// than appearing to stall for no reason.
|
|
1042
|
+
this.emit({
|
|
1043
|
+
Kind: 'TaskAwaitingHuman',
|
|
1044
|
+
ParentTaskID: task.ParentID ?? task.ID,
|
|
1045
|
+
OwnerUserID: await this.resolveOwner(provider, task.ParentID ?? task.ID),
|
|
1046
|
+
TaskID: task.ID,
|
|
1047
|
+
TaskName: task.Name,
|
|
1048
|
+
Status: task.Status,
|
|
1049
|
+
AssignedUserID: task.UserID,
|
|
1050
|
+
});
|
|
1051
|
+
}
|
|
1052
|
+
/**
|
|
1053
|
+
* Parent tasks that still have work to do.
|
|
1054
|
+
*
|
|
1055
|
+
* `BypassCache` for the reason the caching guide names explicitly: **the claim protocol mutates
|
|
1056
|
+
* these rows through direct SQL**, because the CAS guarantee IS the database's atomicity and a
|
|
1057
|
+
* `BaseEntity.Save()` cannot express a guarded UPDATE. Direct DML fires no invalidation event,
|
|
1058
|
+
* so a cached read of this query is stale the instant any task is claimed or completed — and
|
|
1059
|
+
* the dispatcher would then be reading its own work queue through a cache its own writes never
|
|
1060
|
+
* invalidate. Left cached, a completed task keeps reading as `In Progress` and the graph never
|
|
1061
|
+
* rolls up: submitted work simply never settles.
|
|
1062
|
+
*/
|
|
1063
|
+
async findActiveGraphIDs(provider) {
|
|
1064
|
+
const rv = RunView.FromMetadataProvider(provider);
|
|
1065
|
+
// TWO queries, because "has work left to do" and "needs attention" are not the same set.
|
|
1066
|
+
//
|
|
1067
|
+
// Selecting only graphs with non-terminal CHILDREN looks right and is subtly fatal: the
|
|
1068
|
+
// moment the last child completes, the graph leaves that set — so the pass that would have
|
|
1069
|
+
// rolled the parent up never sees it. A graph whose tasks all succeed therefore stays
|
|
1070
|
+
// In Progress forever and its continuation never fires. (A graph that FAILS happened to
|
|
1071
|
+
// survive this, because blocking its dependents left them non-terminal for one more pass —
|
|
1072
|
+
// which is why the bug hid behind a passing failure-path test.)
|
|
1073
|
+
//
|
|
1074
|
+
// The second query closes it: a parent that is itself non-terminal still needs looking at,
|
|
1075
|
+
// whatever its children are doing.
|
|
1076
|
+
const [withPendingWork, unsettledParents] = await rv.RunViews([
|
|
1077
|
+
{
|
|
1078
|
+
EntityName: 'MJ: Tasks',
|
|
1079
|
+
ExtraFilter: `ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
|
|
1080
|
+
Fields: ['ParentID'],
|
|
1081
|
+
ResultType: 'simple',
|
|
1082
|
+
BypassCache: true,
|
|
1083
|
+
},
|
|
1084
|
+
{
|
|
1085
|
+
EntityName: 'MJ: Tasks',
|
|
1086
|
+
ExtraFilter: `ParentID IS NULL AND Status IN ('Pending','In Progress')`,
|
|
1087
|
+
Fields: ['ID'],
|
|
1088
|
+
ResultType: 'simple',
|
|
1089
|
+
BypassCache: true,
|
|
1090
|
+
},
|
|
1091
|
+
], this.contextUser);
|
|
1092
|
+
const ids = new Set();
|
|
1093
|
+
for (const r of (withPendingWork?.Results ?? [])) {
|
|
1094
|
+
if (r.ParentID)
|
|
1095
|
+
ids.add(r.ParentID);
|
|
1096
|
+
}
|
|
1097
|
+
// Childless tasks match the second query too; propagateAndRollup skips anything with no
|
|
1098
|
+
// nodes, so they cost one empty load and nothing else.
|
|
1099
|
+
for (const r of (unsettledParents?.Results ?? [])) {
|
|
1100
|
+
if (r.ID)
|
|
1101
|
+
ids.add(r.ID);
|
|
1102
|
+
}
|
|
1103
|
+
return [...ids];
|
|
1104
|
+
}
|
|
1105
|
+
/**
|
|
1106
|
+
* Tasks eligible to claim right now, across all active graphs.
|
|
1107
|
+
*
|
|
1108
|
+
* Eligibility is decided by the pure algorithm rather than by SQL: expressing "all prerequisites
|
|
1109
|
+
* complete" as a query is possible but would be a second, independently-maintained definition of
|
|
1110
|
+
* the same rule, free to drift from the one the in-run executor uses.
|
|
1111
|
+
*/
|
|
1112
|
+
async findClaimableTasks(provider, limit) {
|
|
1113
|
+
const claimable = [];
|
|
1114
|
+
for (const parentID of await this.findActiveGraphIDs(provider)) {
|
|
1115
|
+
if (claimable.length >= limit)
|
|
1116
|
+
break;
|
|
1117
|
+
const graph = await this.loadGraphState(provider, parentID);
|
|
1118
|
+
// HOLD is what makes "a broken condition stalls visibly" true rather than merely stated.
|
|
1119
|
+
// An undecided exclusive group keeps all its edges, and a kept edge on a Complete origin
|
|
1120
|
+
// is a SATISFIED prerequisite — so without this filter every branch of the fork would be
|
|
1121
|
+
// eligible at once and all of them would run. A typo must not multiply a fork.
|
|
1122
|
+
//
|
|
1123
|
+
// The losers of a DECIDED group must be filtered for the same reason, and this is a race
|
|
1124
|
+
// rather than a rule: they are marked Skipped by the propagation pass, but between the
|
|
1125
|
+
// moment the group resolves and the moment that write lands, their incoming edge is still
|
|
1126
|
+
// a satisfied prerequisite on a Complete origin. A poll landing in that window would
|
|
1127
|
+
// claim and execute the branch the workflow chose NOT to take — irreversibly, since the
|
|
1128
|
+
// action has already run by the time Skipped is written over it.
|
|
1129
|
+
// `unreachableTaskIDs` joins the filter for exactly the reason above. R6 made a
|
|
1130
|
+
// definite-false ordinary edge seed the skip cascade rather than Block its target — but
|
|
1131
|
+
// until that Skipped write lands, the target has no unsatisfied prerequisite and is
|
|
1132
|
+
// vacuously eligible. That is the same race the XOR fix closed, reopened on the new
|
|
1133
|
+
// path: a branch the workflow decided against, claimed and executed irreversibly in the
|
|
1134
|
+
// window before it was marked.
|
|
1135
|
+
const eligible = ComputeEligibleTasks(graph.nodes, graph.edges, graph.handledFailureIDs)
|
|
1136
|
+
.filter((n) => !graph.holdTaskIDs.has(n.id) &&
|
|
1137
|
+
!graph.skipSeedTaskIDs.has(n.id) &&
|
|
1138
|
+
!graph.unreachableTaskIDs.has(n.id));
|
|
1139
|
+
for (const node of eligible) {
|
|
1140
|
+
const entity = graph.entityById.get(node.id);
|
|
1141
|
+
if (!entity)
|
|
1142
|
+
continue;
|
|
1143
|
+
// Human tasks are never dispatched — a person completes them. But "eligible" is the
|
|
1144
|
+
// moment that person can finally act, and nothing else in the system knows it has
|
|
1145
|
+
// arrived: the task sat Pending behind prerequisites, and no save touched it when
|
|
1146
|
+
// they cleared. Without a notification here a workflow simply stops, waiting on
|
|
1147
|
+
// someone who was never told. That silent stall is the failure mode this exists to
|
|
1148
|
+
// prevent, so it happens on the eligibility check rather than at submission.
|
|
1149
|
+
if (entity.ActionID) {
|
|
1150
|
+
// An action node this host has no runner for is left Pending rather than
|
|
1151
|
+
// claimed. Claiming it would take ownership of work this process cannot do, and
|
|
1152
|
+
// the claim would then have to expire before any host that CAN do it gets a
|
|
1153
|
+
// turn — a self-inflicted stall on a mixed deployment.
|
|
1154
|
+
if (!this.actionRunner)
|
|
1155
|
+
continue;
|
|
1156
|
+
}
|
|
1157
|
+
else if (entity.PromptID) {
|
|
1158
|
+
// A prompt node — including a loop that repeats a prompt — is assigned through
|
|
1159
|
+
// PromptID and carries NEITHER ActionID nor AgentID. Without this branch it fell
|
|
1160
|
+
// through to the test below and was treated as a task waiting on a PERSON: the
|
|
1161
|
+
// workflow notified a human who had nothing to do and then stopped forever.
|
|
1162
|
+
// That is precisely the misclassification the step-kind rules warn about, and it
|
|
1163
|
+
// is silent — the graph sits In Progress looking like it is still working.
|
|
1164
|
+
if (!this.promptRunner)
|
|
1165
|
+
continue;
|
|
1166
|
+
}
|
|
1167
|
+
else if (!entity.AgentID) {
|
|
1168
|
+
await this.notifyHumanTaskReady(entity, provider);
|
|
1169
|
+
continue;
|
|
1170
|
+
}
|
|
1171
|
+
if (this.inFlight.has(entity.ID))
|
|
1172
|
+
continue;
|
|
1173
|
+
claimable.push(entity);
|
|
1174
|
+
if (claimable.length >= limit)
|
|
1175
|
+
break;
|
|
1176
|
+
}
|
|
1177
|
+
}
|
|
1178
|
+
return claimable;
|
|
1179
|
+
}
|
|
1180
|
+
/**
|
|
1181
|
+
* Marks a human task as notified, so the request is raised exactly once.
|
|
1182
|
+
*
|
|
1183
|
+
* Written even when delivery threw. Retrying on every poll is a worse failure than one missed
|
|
1184
|
+
* notification: the task stays visible in the inbox either way, whereas a notification storm is
|
|
1185
|
+
* not self-correcting.
|
|
1186
|
+
*/
|
|
1187
|
+
async markHumanTaskNotified(task) {
|
|
1188
|
+
task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
|
|
1189
|
+
if (!(await task.Save())) {
|
|
1190
|
+
LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
|
|
1191
|
+
}
|
|
1192
|
+
}
|
|
1193
|
+
/**
|
|
1194
|
+
* Raises the `MJ: AI Agent Requests` row a person answers to release this step.
|
|
1195
|
+
*
|
|
1196
|
+
* **Why that entity rather than something new.** It already models everything a workflow's human
|
|
1197
|
+
* step needs — who is being asked, what for, a typed response schema, priority, expiry, and an
|
|
1198
|
+
* inbox surface people already use. A second HITL substrate beside it would split the inbox in
|
|
1199
|
+
* two and leave one of them without expiry or permissions.
|
|
1200
|
+
*
|
|
1201
|
+
* **What it deliberately does NOT set is `ResumingAgentRunID`.** A request normally suspends an
|
|
1202
|
+
* agent run and resumes it. A workflow needs none of that: the graph OUTLIVES the run that
|
|
1203
|
+
* submitted it, so nothing is suspended — the task sits Pending, every other branch keeps
|
|
1204
|
+
* running, and answering settles the task. That column staying null is meaningful, not missing.
|
|
1205
|
+
*/
|
|
1206
|
+
async raiseHumanRequest(task, provider) {
|
|
1207
|
+
try {
|
|
1208
|
+
const existing = await this.findOpenRequest(provider, task.ID);
|
|
1209
|
+
if (existing)
|
|
1210
|
+
return 'raised'; // already waiting on someone
|
|
1211
|
+
const request = await provider.GetEntityObject('MJ: AI Agent Requests', this.contextUser);
|
|
1212
|
+
request.NewRecord();
|
|
1213
|
+
request.OriginatingTaskID = task.ID;
|
|
1214
|
+
// A human task has NO AgentID of its own — that column names what EXECUTES a step, and
|
|
1215
|
+
// a person is not an agent. The request still needs one, so it carries the agent that
|
|
1216
|
+
// owns the workflow: the graph's own agent, which is who is asking.
|
|
1217
|
+
const owningAgentID = await this.owningAgentOf(provider, task);
|
|
1218
|
+
if (!owningAgentID) {
|
|
1219
|
+
// PERMANENT: a graph with no owning agent will not acquire one by being asked
|
|
1220
|
+
// again. Graphs submitted before the provenance stamp landed are all in this state.
|
|
1221
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} needs a person, but its workflow has no ` +
|
|
1222
|
+
`agent to ask on behalf of, so no request can be raised. The task stays Pending ` +
|
|
1223
|
+
`and visible in the Tasks UI; it will not be retried.`);
|
|
1224
|
+
return 'permanent-failure';
|
|
1225
|
+
}
|
|
1226
|
+
request.AgentID = owningAgentID;
|
|
1227
|
+
request.RequestForUserID = task.UserID;
|
|
1228
|
+
request.RequestedAt = new Date();
|
|
1229
|
+
request.Status = 'Requested';
|
|
1230
|
+
request.Request = task.Description || `A workflow is waiting on you to complete "${task.Name}".`;
|
|
1231
|
+
// The graph's own run is the provenance a reader follows back to see what led here.
|
|
1232
|
+
request.OriginatingAgentRunID = await this.submittingRunOf(provider, task);
|
|
1233
|
+
// The deadline, when the author set one. `expireOverdueRequests` has always been able to
|
|
1234
|
+
// enforce this — it expires the request and fails the step so a give-up edge can route
|
|
1235
|
+
// around it — but nothing ever WROTE the column, so that whole path had never run outside
|
|
1236
|
+
// a test and a workflow waiting on someone who left the company waited forever.
|
|
1237
|
+
// Absent means no deadline, deliberately: expiring on a timeout nobody chose would be
|
|
1238
|
+
// worse than waiting.
|
|
1239
|
+
const expiresInHours = this.parseConfiguration(task)?.human?.expiresInHours;
|
|
1240
|
+
if (expiresInHours && expiresInHours > 0) {
|
|
1241
|
+
request.ExpiresAt = new Date(Date.now() + expiresInHours * 60 * 60 * 1000);
|
|
1242
|
+
}
|
|
1243
|
+
if (!(await request.Save())) {
|
|
1244
|
+
LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ` +
|
|
1245
|
+
`${request.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
1246
|
+
// A failed SAVE may be transient (deadlock, contention), so this one earns a retry.
|
|
1247
|
+
return 'transient-failure';
|
|
1248
|
+
}
|
|
1249
|
+
return 'raised';
|
|
1250
|
+
}
|
|
1251
|
+
catch (e) {
|
|
1252
|
+
// Never fatal. The task remains Pending and visible; a missing request is recoverable,
|
|
1253
|
+
// whereas throwing here would abort the whole dispatch pass for every other branch.
|
|
1254
|
+
LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
1255
|
+
return 'transient-failure';
|
|
1256
|
+
}
|
|
1257
|
+
}
|
|
1258
|
+
/**
|
|
1259
|
+
* The agent that owns this task's workflow — who the request is asked on behalf of.
|
|
1260
|
+
*
|
|
1261
|
+
* Reads the graph's parent row, falling back to the run that submitted it. A human step has no
|
|
1262
|
+
* agent of its own by design: `AgentID` names what EXECUTES a step, and a person is not an agent.
|
|
1263
|
+
*/
|
|
1264
|
+
async owningAgentOf(provider, task) {
|
|
1265
|
+
if (task.AgentID)
|
|
1266
|
+
return task.AgentID;
|
|
1267
|
+
if (!task.ParentID)
|
|
1268
|
+
return null;
|
|
1269
|
+
try {
|
|
1270
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
1271
|
+
if (!(await parent.Load(task.ParentID)))
|
|
1272
|
+
return null;
|
|
1273
|
+
if (parent.AgentID)
|
|
1274
|
+
return parent.AgentID;
|
|
1275
|
+
if (!parent.AgentRunID)
|
|
1276
|
+
return null;
|
|
1277
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
1278
|
+
return (await run.Load(parent.AgentRunID)) ? run.AgentID : null;
|
|
1279
|
+
}
|
|
1280
|
+
catch {
|
|
1281
|
+
return null;
|
|
1282
|
+
}
|
|
1283
|
+
}
|
|
1284
|
+
/** The still-open request for a task, if one exists. */
|
|
1285
|
+
async findOpenRequest(provider, taskID) {
|
|
1286
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
1287
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
1288
|
+
ExtraFilter: `OriginatingTaskID='${taskID}' AND Status='Requested'`,
|
|
1289
|
+
ResultType: 'entity_object',
|
|
1290
|
+
BypassCache: true,
|
|
1291
|
+
}, this.contextUser);
|
|
1292
|
+
return (result.Success ? result.Results?.[0] : null) ?? null;
|
|
1293
|
+
}
|
|
1294
|
+
/**
|
|
1295
|
+
* Settles a human task from the request a person answered.
|
|
1296
|
+
*
|
|
1297
|
+
* Runs on the poll rather than on a save hook, because the answer can arrive through any surface
|
|
1298
|
+
* — the inbox, the API, a conversation — and only the dispatcher knows how to release the rest
|
|
1299
|
+
* of the graph afterwards.
|
|
1300
|
+
*
|
|
1301
|
+
* **`ResponseData` becomes the task's output.** That is what makes a human step useful rather
|
|
1302
|
+
* than a gate: a downstream edge can branch on what the person actually said, typed by the
|
|
1303
|
+
* request's own ResponseSchema. A step that only recorded "approved" would force every decision
|
|
1304
|
+
* back into a separate action.
|
|
1305
|
+
*/
|
|
1306
|
+
async settleAnsweredHumanTasks(provider, graphID) {
|
|
1307
|
+
const waiting = await RunView.FromMetadataProvider(provider).RunView({
|
|
1308
|
+
EntityName: 'MJ: Tasks',
|
|
1309
|
+
ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
|
|
1310
|
+
ResultType: 'entity_object',
|
|
1311
|
+
BypassCache: true,
|
|
1312
|
+
}, this.contextUser);
|
|
1313
|
+
if (!waiting.Success)
|
|
1314
|
+
return;
|
|
1315
|
+
for (const task of waiting.Results ?? []) {
|
|
1316
|
+
const request = await this.answeredRequestFor(provider, task.ID);
|
|
1317
|
+
if (!request)
|
|
1318
|
+
continue;
|
|
1319
|
+
const rejected = request.Status === 'Rejected';
|
|
1320
|
+
const expired = request.Status === 'Expired';
|
|
1321
|
+
task.Status = rejected || expired ? 'Failed' : 'Complete';
|
|
1322
|
+
task.CompletedAt = new Date();
|
|
1323
|
+
task.PercentComplete = rejected || expired ? 0 : 100;
|
|
1324
|
+
task.ClaimedBy = null;
|
|
1325
|
+
task.ClaimExpiresAt = null;
|
|
1326
|
+
task.OutputPayload = request.ResponseData ?? null;
|
|
1327
|
+
if (rejected) {
|
|
1328
|
+
task.ErrorMessage = request.Comments || 'A person rejected this step.';
|
|
1329
|
+
}
|
|
1330
|
+
else if (expired) {
|
|
1331
|
+
// Stated as a failure rather than left Pending. A workflow blocked forever on
|
|
1332
|
+
// someone who never answered — who may have left the company — is the silent stall
|
|
1333
|
+
// this whole path exists to avoid, and a give-up edge can now route around it.
|
|
1334
|
+
task.ErrorMessage = 'Nobody answered this step before its request expired.';
|
|
1335
|
+
}
|
|
1336
|
+
if (!(await task.Save())) {
|
|
1337
|
+
LogError(`[TaskGraphDispatcher] Could not settle human task ${task.ID}: ` +
|
|
1338
|
+
`${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
1339
|
+
}
|
|
1340
|
+
}
|
|
1341
|
+
}
|
|
1342
|
+
/**
|
|
1343
|
+
* Re-opens a human step whose request was CANCELLED.
|
|
1344
|
+
*
|
|
1345
|
+
* `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
|
|
1346
|
+
* rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
|
|
1347
|
+
* it". Nothing replaced it. `raiseHumanRequest` refuses to raise twice (the notified marker on
|
|
1348
|
+
* `ClaimedBy` is what stops the notification storm), so a cancelled request left the task Pending
|
|
1349
|
+
* with no open request and no path to acquiring one: a workflow waiting forever on a question
|
|
1350
|
+
* nobody is being asked.
|
|
1351
|
+
*
|
|
1352
|
+
* Clearing the marker is the whole fix — the next poll sees an un-notified Pending human task
|
|
1353
|
+
* and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
|
|
1354
|
+
* human action: it takes another person cancelling again to come back here.
|
|
1355
|
+
*/
|
|
1356
|
+
async reopenCancelledHumanTasks(provider, graphID) {
|
|
1357
|
+
const waiting = await RunView.FromMetadataProvider(provider).RunView({
|
|
1358
|
+
EntityName: 'MJ: Tasks',
|
|
1359
|
+
// `StepType` is NULLABLE, and rows predating the column exist (4 in the reference
|
|
1360
|
+
// database at the time of writing). None currently carry a UserID, but a human task
|
|
1361
|
+
// written by any path that set the assignee without the discriminator would be
|
|
1362
|
+
// invisible to a `StepType='Human'` filter and stay dead forever after a cancel —
|
|
1363
|
+
// the exact stall this method exists to end. The notified marker already narrows
|
|
1364
|
+
// this to tasks the dispatcher raised a request for, so the widening cannot pull in
|
|
1365
|
+
// unrelated work.
|
|
1366
|
+
ExtraFilter: `ParentID='${graphID}' AND Status='Pending' ` +
|
|
1367
|
+
`AND (StepType='Human' OR (StepType IS NULL AND UserID IS NOT NULL)) ` +
|
|
1368
|
+
`AND ClaimedBy='${HUMAN_TASK_NOTIFIED_MARKER}'`,
|
|
1369
|
+
ResultType: 'entity_object',
|
|
1370
|
+
BypassCache: true,
|
|
1371
|
+
}, this.contextUser);
|
|
1372
|
+
if (!waiting.Success)
|
|
1373
|
+
return;
|
|
1374
|
+
for (const task of waiting.Results ?? []) {
|
|
1375
|
+
// Only when there is nothing live AND nothing terminal. A task with an open request is
|
|
1376
|
+
// simply waiting; one with a terminal request is settled on the next pass by
|
|
1377
|
+
// settleAnsweredHumanTasks, and re-raising either would ask the same question twice.
|
|
1378
|
+
if (await this.findOpenRequest(provider, task.ID))
|
|
1379
|
+
continue;
|
|
1380
|
+
if (await this.answeredRequestFor(provider, task.ID))
|
|
1381
|
+
continue;
|
|
1382
|
+
LogStatus(`[TaskGraphDispatcher] The request for '${task.Name}' was cancelled and nothing ` +
|
|
1383
|
+
`replaced it; asking again.`);
|
|
1384
|
+
task.ClaimedBy = null;
|
|
1385
|
+
if (!(await task.Save())) {
|
|
1386
|
+
LogError(`[TaskGraphDispatcher] Could not re-open cancelled human task ${task.ID}: ` +
|
|
1387
|
+
`${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
1388
|
+
}
|
|
1389
|
+
}
|
|
1390
|
+
}
|
|
1391
|
+
/** The answered (or expired) request for a task, if any. */
|
|
1392
|
+
async answeredRequestFor(provider, taskID) {
|
|
1393
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
1394
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
1395
|
+
// Everything terminal. 'Canceled' is deliberately absent: a cancelled request means
|
|
1396
|
+
// the ASK was withdrawn, not that the step was decided, so the task keeps waiting
|
|
1397
|
+
// for whatever replaces it.
|
|
1398
|
+
ExtraFilter: `OriginatingTaskID='${taskID}' AND Status IN ('Approved','Rejected','Responded','Expired')`,
|
|
1399
|
+
OrderBy: 'RespondedAt DESC',
|
|
1400
|
+
ResultType: 'entity_object',
|
|
1401
|
+
BypassCache: true,
|
|
1402
|
+
}, this.contextUser);
|
|
1403
|
+
return (result.Success ? result.Results?.[0] : null) ?? null;
|
|
1404
|
+
}
|
|
1405
|
+
/**
|
|
1406
|
+
* Expires requests whose deadline has passed.
|
|
1407
|
+
*
|
|
1408
|
+
* A deadline that nothing enforces is a comment. Without this an `ExpiresAt` in the past leaves
|
|
1409
|
+
* the request `Requested` forever and the workflow waiting on it just as long.
|
|
1410
|
+
*/
|
|
1411
|
+
async expireOverdueRequests(provider, graphID) {
|
|
1412
|
+
// Scoped by an explicit id list rather than a subquery against a view name, so this reads
|
|
1413
|
+
// the same on any provider rather than assuming a SQL dialect and a physical view.
|
|
1414
|
+
const humanTasks = await RunView.FromMetadataProvider(provider).RunView({
|
|
1415
|
+
EntityName: 'MJ: Tasks',
|
|
1416
|
+
Fields: ['ID'],
|
|
1417
|
+
ExtraFilter: `ParentID='${graphID}' AND StepType='Human' AND Status='Pending'`,
|
|
1418
|
+
ResultType: 'simple',
|
|
1419
|
+
}, this.contextUser);
|
|
1420
|
+
const ids = (humanTasks.Results ?? []).map((r) => `'${r.ID}'`);
|
|
1421
|
+
if (ids.length === 0)
|
|
1422
|
+
return;
|
|
1423
|
+
const nowISO = new Date().toISOString();
|
|
1424
|
+
const overdue = await RunView.FromMetadataProvider(provider).RunView({
|
|
1425
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
1426
|
+
ExtraFilter: `Status='Requested' AND ExpiresAt IS NOT NULL AND ExpiresAt < '${nowISO}' ` +
|
|
1427
|
+
`AND OriginatingTaskID IN (${ids.join(',')})`,
|
|
1428
|
+
ResultType: 'entity_object',
|
|
1429
|
+
BypassCache: true,
|
|
1430
|
+
}, this.contextUser);
|
|
1431
|
+
if (!overdue.Success)
|
|
1432
|
+
return;
|
|
1433
|
+
for (const request of overdue.Results ?? []) {
|
|
1434
|
+
request.Status = 'Expired';
|
|
1435
|
+
if (!(await request.Save())) {
|
|
1436
|
+
LogError(`[TaskGraphDispatcher] Could not expire request ${request.ID}.`);
|
|
1437
|
+
}
|
|
1438
|
+
}
|
|
1439
|
+
}
|
|
1440
|
+
/** The agent run that submitted this task's graph, for provenance on the request. */
|
|
1441
|
+
async submittingRunOf(provider, task) {
|
|
1442
|
+
if (!task.ParentID)
|
|
1443
|
+
return null;
|
|
1444
|
+
try {
|
|
1445
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
1446
|
+
return (await parent.Load(task.ParentID)) ? parent.AgentRunID : null;
|
|
1447
|
+
}
|
|
1448
|
+
catch {
|
|
1449
|
+
return null;
|
|
1450
|
+
}
|
|
1451
|
+
}
|
|
1452
|
+
/** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
|
|
1453
|
+
async loadGraphState(provider, parentTaskID) {
|
|
1454
|
+
const rv = RunView.FromMetadataProvider(provider);
|
|
1455
|
+
// BypassCache throughout: task status is written by the claim protocol's direct SQL, which
|
|
1456
|
+
// fires no cache invalidation. See findActiveGraphIDs.
|
|
1457
|
+
const childrenResult = await rv.RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
|
|
1458
|
+
const children = (childrenResult.Success ? childrenResult.Results : []) ?? [];
|
|
1459
|
+
if (children.length === 0) {
|
|
1460
|
+
return {
|
|
1461
|
+
nodes: [], edges: [], entityById: new Map(),
|
|
1462
|
+
unreachableTaskIDs: new Set(), skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
|
|
1463
|
+
handledFailureIDs: new Set(),
|
|
1464
|
+
};
|
|
1465
|
+
}
|
|
1466
|
+
const idList = children.map((c) => `'${c.ID}'`).join(',');
|
|
1467
|
+
const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
|
|
1468
|
+
const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
|
|
1469
|
+
const entityById = new Map(children.map((c) => [c.ID, c]));
|
|
1470
|
+
// Conditional edges are resolved HERE, before eligibility runs, by dropping edges whose
|
|
1471
|
+
// condition does not hold. Expressing it as edge removal rather than as a second rule inside
|
|
1472
|
+
// the eligibility algorithm is what keeps one definition of "ready": a task with no live
|
|
1473
|
+
// incoming edges is ready for exactly the same reason a task with no edges at all is.
|
|
1474
|
+
//
|
|
1475
|
+
// An edge whose condition cannot be evaluated is KEPT, which is the opposite of the flow
|
|
1476
|
+
// executor's choice and deliberately so. There, a broken condition means an edge is not
|
|
1477
|
+
// followed and the graph moves on. Here it would mean a prerequisite silently disappears and
|
|
1478
|
+
// the dependent task runs early — turning a typo into out-of-order execution. Keeping the
|
|
1479
|
+
// edge instead stalls the graph, which the stall detector already reports loudly.
|
|
1480
|
+
const liveEdges = [];
|
|
1481
|
+
// A definitely-false edge must not merely disappear. Removing a task's only prerequisite
|
|
1482
|
+
// makes it eligible in the very next wave — so "this branch was not taken" would execute the
|
|
1483
|
+
// branch, potentially before the node that gated it. The dependent is recorded as
|
|
1484
|
+
// unreachable instead, and blocked before anything can claim it.
|
|
1485
|
+
const droppedInto = new Set();
|
|
1486
|
+
const stillReachable = new Set();
|
|
1487
|
+
// EXCLUSIVE edges are exempt from the generic machinery below, and that exemption is
|
|
1488
|
+
// load-bearing. An XOR loser is by definition condition-false, so the ordinary path would
|
|
1489
|
+
// record it as unreachable and Block it — and a Blocked child poisons the parent rollup, so
|
|
1490
|
+
// every fork would settle the graph as Blocked. Losers must become Skipped instead, which
|
|
1491
|
+
// only ResolveExclusiveGroups can decide.
|
|
1492
|
+
const exclusive = deps.filter((d) => !!d.ExclusiveGroup);
|
|
1493
|
+
const ordinary = deps.filter((d) => !d.ExclusiveGroup);
|
|
1494
|
+
const resolution = ResolveExclusiveGroups(exclusive.map((d) => ({
|
|
1495
|
+
id: d.ID,
|
|
1496
|
+
taskId: d.TaskID,
|
|
1497
|
+
dependsOnTaskId: d.DependsOnTaskID,
|
|
1498
|
+
exclusiveGroup: d.ExclusiveGroup,
|
|
1499
|
+
originStatus: (entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending'),
|
|
1500
|
+
priority: d.Priority ?? 0,
|
|
1501
|
+
sequence: d.Sequence ?? 0,
|
|
1502
|
+
conditionOutcome: this.evaluateExclusiveCondition(d, entityById),
|
|
1503
|
+
})),
|
|
1504
|
+
// A flow's failure handling is its outgoing edges, so a Failed origin still decides its
|
|
1505
|
+
// group. For a loop-agent graph the set is Complete-only and nothing changes.
|
|
1506
|
+
new Set(['Complete', 'Failed']));
|
|
1507
|
+
const loserEdgeIDs = new Set(resolution.loserEdgeIDs);
|
|
1508
|
+
for (const d of ordinary) {
|
|
1509
|
+
if (d.Condition?.trim()) {
|
|
1510
|
+
const outcome = this.evaluateEdgeCondition(d, entityById);
|
|
1511
|
+
if (outcome === 'drop') {
|
|
1512
|
+
droppedInto.add(d.TaskID);
|
|
1513
|
+
continue;
|
|
1514
|
+
}
|
|
1515
|
+
}
|
|
1516
|
+
stillReachable.add(d.TaskID);
|
|
1517
|
+
liveEdges.push({
|
|
1518
|
+
taskId: d.TaskID,
|
|
1519
|
+
dependsOnTaskId: d.DependsOnTaskID,
|
|
1520
|
+
dependencyType: d.DependencyType,
|
|
1521
|
+
});
|
|
1522
|
+
}
|
|
1523
|
+
for (const d of exclusive) {
|
|
1524
|
+
// A losing edge is removed rather than left to gate: its target is being skipped, and a
|
|
1525
|
+
// live edge into a skipped task would keep the graph waiting on a branch nobody took.
|
|
1526
|
+
if (loserEdgeIDs.has(d.ID))
|
|
1527
|
+
continue;
|
|
1528
|
+
stillReachable.add(d.TaskID);
|
|
1529
|
+
liveEdges.push({
|
|
1530
|
+
taskId: d.TaskID,
|
|
1531
|
+
dependsOnTaskId: d.DependsOnTaskID,
|
|
1532
|
+
dependencyType: d.DependencyType,
|
|
1533
|
+
});
|
|
1534
|
+
}
|
|
1535
|
+
// Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
|
|
1536
|
+
// waiting on it, and a node reached by an alternate branch is genuinely reachable.
|
|
1537
|
+
const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
|
|
1538
|
+
const nodes = children.map((c) => ({ id: c.ID, status: c.Status }));
|
|
1539
|
+
return {
|
|
1540
|
+
nodes,
|
|
1541
|
+
edges: liveEdges,
|
|
1542
|
+
entityById,
|
|
1543
|
+
unreachableTaskIDs,
|
|
1544
|
+
skipSeedTaskIDs: new Set(resolution.skipSeedTaskIDs),
|
|
1545
|
+
holdTaskIDs: new Set(resolution.holdTaskIDs),
|
|
1546
|
+
handledFailureIDs: await this.computeHandledFailures(provider, parentTaskID, nodes, liveEdges),
|
|
1547
|
+
};
|
|
1548
|
+
}
|
|
1549
|
+
/**
|
|
1550
|
+
* Decides whether a conditional dependency edge is live.
|
|
1551
|
+
*
|
|
1552
|
+
* The condition sees the upstream task's outcome — its status and parsed output — which is the
|
|
1553
|
+
* only information a runtime graph has to branch on. Returns `'drop'` only on a definite false;
|
|
1554
|
+
* an unevaluable condition keeps the edge for the reason stated at the call site.
|
|
1555
|
+
*/
|
|
1556
|
+
evaluateEdgeCondition(dep, entityById) {
|
|
1557
|
+
const upstream = entityById.get(dep.DependsOnTaskID);
|
|
1558
|
+
if (!upstream)
|
|
1559
|
+
return 'keep';
|
|
1560
|
+
// TERMINALITY GUARD — fixes a latent bug, not a hypothetical one.
|
|
1561
|
+
//
|
|
1562
|
+
// Without it, every conditional edge is evaluated on every poll cycle, including while its
|
|
1563
|
+
// origin is still Pending. A condition like `succeeded` is then a DEFINITE FALSE, the edge
|
|
1564
|
+
// is dropped, and the target is Blocked at wave one — permanently, before the origin ever
|
|
1565
|
+
// ran. That kills any conditioned linear chain, which is the most common flow shape there
|
|
1566
|
+
// is.
|
|
1567
|
+
//
|
|
1568
|
+
// A non-terminal origin is UNDECIDED, and 'keep' is the safe reading of undecided: the
|
|
1569
|
+
// prerequisite gate already prevents the target starting early, so keeping the edge costs
|
|
1570
|
+
// nothing and dropping it is irreversible.
|
|
1571
|
+
if (!TERMINAL_FOR_CONDITIONS.has(upstream.Status))
|
|
1572
|
+
return 'keep';
|
|
1573
|
+
let output = null;
|
|
1574
|
+
if (upstream.OutputPayload) {
|
|
1575
|
+
try {
|
|
1576
|
+
output = JSON.parse(upstream.OutputPayload);
|
|
1577
|
+
}
|
|
1578
|
+
catch { /* a malformed payload is not grounds to drop a prerequisite */ }
|
|
1579
|
+
}
|
|
1580
|
+
const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
|
|
1581
|
+
if (!result.Success) {
|
|
1582
|
+
LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
|
|
1583
|
+
`(${result.ErrorMessage}); keeping the edge so the graph stalls visibly rather than ` +
|
|
1584
|
+
`running ${dep.TaskID} out of order.`);
|
|
1585
|
+
return 'keep';
|
|
1586
|
+
}
|
|
1587
|
+
return result.Value ? 'keep' : 'drop';
|
|
1588
|
+
}
|
|
1589
|
+
/**
|
|
1590
|
+
* An exclusive edge's condition as a three-way outcome.
|
|
1591
|
+
*
|
|
1592
|
+
* `ResolveExclusiveGroups` needs to tell "false" from "could not be evaluated": the first loses
|
|
1593
|
+
* the branch, the second holds the whole group. The generic keep/drop path cannot express that
|
|
1594
|
+
* difference, which is why exclusive edges take this route instead.
|
|
1595
|
+
*/
|
|
1596
|
+
evaluateExclusiveCondition(dep, entityById) {
|
|
1597
|
+
if (!dep.Condition?.trim())
|
|
1598
|
+
return 'satisfied';
|
|
1599
|
+
const upstream = entityById.get(dep.DependsOnTaskID);
|
|
1600
|
+
if (!upstream)
|
|
1601
|
+
return 'unevaluable';
|
|
1602
|
+
let output = null;
|
|
1603
|
+
if (upstream.OutputPayload) {
|
|
1604
|
+
try {
|
|
1605
|
+
output = JSON.parse(upstream.OutputPayload);
|
|
1606
|
+
}
|
|
1607
|
+
catch { /* malformed payload */ }
|
|
1608
|
+
}
|
|
1609
|
+
const result = this.conditionEvaluator.Evaluate(dep.Condition, this.buildConditionContext(upstream, output));
|
|
1610
|
+
if (!result.Success)
|
|
1611
|
+
return 'unevaluable';
|
|
1612
|
+
return result.Value ? 'satisfied' : 'unsatisfied';
|
|
1613
|
+
}
|
|
1614
|
+
/**
|
|
1615
|
+
* Everything an edge condition can see — the SUPERSET of both dialects.
|
|
1616
|
+
*
|
|
1617
|
+
* A flow condition is written against `payload` / `stepResult` / `flowContext` / `data` /
|
|
1618
|
+
* `context`; the dispatcher's own conditions are written against `status` / `succeeded` /
|
|
1619
|
+
* `failed` / `output` / `errorMessage`. Compiling flows onto this engine without the flow
|
|
1620
|
+
* dialect would make every `payload.x` condition evaluate against nothing — silently, since an
|
|
1621
|
+
* undefined property is simply falsy. Both dialects are readable here so a condition means the
|
|
1622
|
+
* same thing on either engine.
|
|
1623
|
+
*
|
|
1624
|
+
* `payload` is the ORIGIN task's post-step snapshot. There is deliberately no "graph-wide
|
|
1625
|
+
* payload": each task's output is its own, and inventing a merged one would give conditions a
|
|
1626
|
+
* value the flow engine never had.
|
|
1627
|
+
*/
|
|
1628
|
+
buildConditionContext(upstream, output) {
|
|
1629
|
+
const envelope = (output && typeof output === 'object' ? output : {});
|
|
1630
|
+
const succeeded = upstream.Status === 'Complete';
|
|
1631
|
+
return {
|
|
1632
|
+
// dispatcher dialect — unchanged
|
|
1633
|
+
status: upstream.Status,
|
|
1634
|
+
succeeded,
|
|
1635
|
+
failed: upstream.Status === 'Failed',
|
|
1636
|
+
output,
|
|
1637
|
+
errorMessage: upstream.ErrorMessage ?? null,
|
|
1638
|
+
// flow dialect
|
|
1639
|
+
payload: envelope.payload ?? output,
|
|
1640
|
+
stepResult: { Success: succeeded, step: upstream.Name, result: envelope.result ?? output },
|
|
1641
|
+
flowContext: { currentStepId: upstream.ID, completedSteps: [], executionPath: [], stepCount: 0 },
|
|
1642
|
+
data: envelope.data ?? {},
|
|
1643
|
+
context: envelope.context ?? {},
|
|
1644
|
+
};
|
|
1645
|
+
}
|
|
1646
|
+
/** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
|
|
1647
|
+
async loadDependencyOutputs(provider, taskID) {
|
|
1648
|
+
const outputs = new Map();
|
|
1649
|
+
const rv = RunView.FromMetadataProvider(provider);
|
|
1650
|
+
// BypassCache: an upstream task's OutputPayload is written on the completion path, so a
|
|
1651
|
+
// cached read here can hand a dependent task the previous run's output — or none at all.
|
|
1652
|
+
const deps = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID='${taskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
|
|
1653
|
+
for (const dep of (deps.Success ? deps.Results : []) ?? []) {
|
|
1654
|
+
const upstream = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
1655
|
+
if (!(await upstream.Load(dep.DependsOnTaskID)))
|
|
1656
|
+
continue;
|
|
1657
|
+
if (!upstream.OutputPayload)
|
|
1658
|
+
continue;
|
|
1659
|
+
try {
|
|
1660
|
+
outputs.set(dep.DependsOnTaskID, JSON.parse(upstream.OutputPayload));
|
|
1661
|
+
}
|
|
1662
|
+
catch (e) {
|
|
1663
|
+
LogError(`[TaskGraphDispatcher] Task ${dep.DependsOnTaskID} has malformed OutputPayload: ${e}`);
|
|
1664
|
+
}
|
|
1665
|
+
}
|
|
1666
|
+
return outputs;
|
|
1667
|
+
}
|
|
1668
|
+
/**
|
|
1669
|
+
* Runs one task's body, whatever kind of step it is.
|
|
1670
|
+
*
|
|
1671
|
+
* **Routing is on `StepType`, not on which key happens to be set.** A loop step carries the same
|
|
1672
|
+
* `ActionID` or `AgentID` as an ordinary step — that key is what the loop *repeats* — so the old
|
|
1673
|
+
* `task.ActionID ? action : agent` test would have run a loop exactly once and called it done.
|
|
1674
|
+
* `StepType` is the only field that distinguishes them.
|
|
1675
|
+
*
|
|
1676
|
+
* Every branch is normalized to one shape so the recording path above stays single: an action has
|
|
1677
|
+
* no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
|
|
1678
|
+
*/
|
|
1679
|
+
async runTaskBody(task, provider, inputPayload, dependencyOutputs) {
|
|
1680
|
+
const payload = this.mergedPayload(inputPayload, dependencyOutputs);
|
|
1681
|
+
const config = task.ConfigurationObject;
|
|
1682
|
+
// A loop's own step type decides how many times its body runs; the body itself is dispatched
|
|
1683
|
+
// through the very same runners as a one-shot step.
|
|
1684
|
+
if (task.StepType === 'ForEach' || task.StepType === 'While') {
|
|
1685
|
+
return { ...await this.runLoopTask(task, provider, payload, dependencyOutputs), PayloadAtStart: payload };
|
|
1686
|
+
}
|
|
1687
|
+
const { params, errors } = BuildMappedInput(config?.inputMapping, { payload });
|
|
1688
|
+
for (const e of errors)
|
|
1689
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
|
|
1690
|
+
// `payload`, NOT `inputPayload` — the MERGED value computed above, which includes what every
|
|
1691
|
+
// dependency produced.
|
|
1692
|
+
//
|
|
1693
|
+
// A step with an input mapping got exactly the parameters it declared; a step WITHOUT one
|
|
1694
|
+
// fell back to the raw input and therefore saw nothing any earlier step had produced. For a
|
|
1695
|
+
// Prompt step — which declares no mapping by design, because it reads the whole payload
|
|
1696
|
+
// through `{{ _CURRENT_PAYLOAD }}` — that meant the placeholder rendered `{}` and the model
|
|
1697
|
+
// was asked to write from an empty brief.
|
|
1698
|
+
//
|
|
1699
|
+
// It answered anyway. The Content Pipeline's draft step said "the research data was empty",
|
|
1700
|
+
// which was TRUE of what it had been handed while twenty research results sat in the
|
|
1701
|
+
// dependency outputs beside it, and the reviewer then rejected the draft for saying so.
|
|
1702
|
+
// Every layer looked like it was working.
|
|
1703
|
+
const effectiveInput = Object.keys(params).length > 0 ? params : payload;
|
|
1704
|
+
if (task.StepType === 'Prompt') {
|
|
1705
|
+
if (!this.promptRunner) {
|
|
1706
|
+
// Not a failure: "nobody here can run this" is not "this ran and did not work".
|
|
1707
|
+
return { Success: false, AgentRunID: null, ErrorMessage: 'No prompt runner is loaded on this host.' };
|
|
1708
|
+
}
|
|
1709
|
+
const promptResult = await this.promptRunner.RunPromptForTask({
|
|
1710
|
+
TaskID: task.ID,
|
|
1711
|
+
PromptID: task.PromptID,
|
|
1712
|
+
InputPayload: effectiveInput,
|
|
1713
|
+
DependencyOutputs: dependencyOutputs,
|
|
1714
|
+
TemplateParameters: config?.prompt?.templateParameters,
|
|
1715
|
+
Provider: provider,
|
|
1716
|
+
ContextUser: this.contextUser,
|
|
1717
|
+
});
|
|
1718
|
+
// A prompt's response is DEEP-MERGED into the payload rather than replacing it. A prompt
|
|
1719
|
+
// answers one question; replacing the payload with its answer would discard everything
|
|
1720
|
+
// the steps before it established, which is how a late step loses the data it depends on.
|
|
1721
|
+
const merged = promptResult.Success && promptResult.Output && typeof promptResult.Output === 'object'
|
|
1722
|
+
? deepMergePayload(payload, promptResult.Output)
|
|
1723
|
+
: payload;
|
|
1724
|
+
return {
|
|
1725
|
+
Success: promptResult.Success,
|
|
1726
|
+
AgentRunID: null,
|
|
1727
|
+
ErrorMessage: promptResult.ErrorMessage,
|
|
1728
|
+
Output: this.applyStepOutputMapping(task, merged, merged, config?.outputMapping),
|
|
1729
|
+
PayloadAtStart: payload,
|
|
1730
|
+
ChatMessage: promptResult.ChatMessage,
|
|
1731
|
+
// Returned even when the prompt FAILED. A failed prompt still cost tokens, and a
|
|
1732
|
+
// cost rollup that silently omits failures under-reports exactly the runs someone
|
|
1733
|
+
// is most likely to be investigating.
|
|
1734
|
+
PromptRunID: promptResult.PromptRunID,
|
|
1735
|
+
};
|
|
1736
|
+
}
|
|
1737
|
+
const raw = task.ActionID
|
|
1738
|
+
? { ...await this.actionRunner.RunActionForTask({
|
|
1739
|
+
TaskID: task.ID,
|
|
1740
|
+
ActionID: task.ActionID,
|
|
1741
|
+
InputPayload: effectiveInput,
|
|
1742
|
+
DependencyOutputs: dependencyOutputs,
|
|
1743
|
+
Provider: provider,
|
|
1744
|
+
ContextUser: this.contextUser,
|
|
1745
|
+
}), AgentRunID: null }
|
|
1746
|
+
: await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs);
|
|
1747
|
+
return {
|
|
1748
|
+
...raw,
|
|
1749
|
+
Output: this.applyStepOutputMapping(task, payload, raw.Output, config?.outputMapping),
|
|
1750
|
+
PayloadAtStart: payload,
|
|
1751
|
+
};
|
|
1752
|
+
}
|
|
1753
|
+
/**
|
|
1754
|
+
* Runs a loop step: its body once per iteration, with the item and index in scope.
|
|
1755
|
+
*
|
|
1756
|
+
* The loop's own `Configuration` supplies the definition; the row's `ActionID` / `AgentID`
|
|
1757
|
+
* supplies what to repeat. Per-iteration inputs are resolved fresh each pass — the bindings are
|
|
1758
|
+
* merged into the payload before the mapping is applied, which is how a body can reference the
|
|
1759
|
+
* current item at all.
|
|
1760
|
+
*/
|
|
1761
|
+
async runLoopTask(task, provider, payload, dependencyOutputs) {
|
|
1762
|
+
const config = task.ConfigurationObject;
|
|
1763
|
+
const op = task.StepType === 'ForEach' ? config?.forEach : config?.while;
|
|
1764
|
+
if (!op) {
|
|
1765
|
+
return {
|
|
1766
|
+
Success: false,
|
|
1767
|
+
AgentRunID: null,
|
|
1768
|
+
ErrorMessage: `"${task.Name}" is a ${task.StepType} step with no loop settings, so there is nothing to repeat.`,
|
|
1769
|
+
};
|
|
1770
|
+
}
|
|
1771
|
+
// A prompt body has no params of its own — it receives the payload (with the loop bindings
|
|
1772
|
+
// merged in) through the placeholder, so an empty mapping is correct rather than missing.
|
|
1773
|
+
const bodyMapping = (op.action?.params ?? {});
|
|
1774
|
+
// The BODY's output mapping, applied once per pass — see `foldIterationOutput`.
|
|
1775
|
+
//
|
|
1776
|
+
// It used to be applied a single time after the loop finished, against the accumulated
|
|
1777
|
+
// payload. That is the wrong moment in two ways at once: the mapping names an output
|
|
1778
|
+
// PARAMETER of the body, which no longer exists by then, and a mapping like
|
|
1779
|
+
// `"Items": "results[]"` can only append per pass. So every pass merged its raw result into
|
|
1780
|
+
// the shared payload instead, each overwriting the last, and the mapping matched nothing and
|
|
1781
|
+
// wrote nothing. A ForEach over five items reported five successes and kept item five.
|
|
1782
|
+
const bodyOutputMapping = op.action?.outputMapping ?? op.prompt?.outputMapping;
|
|
1783
|
+
// Where this step sits in its graph, resolved ONCE rather than per iteration. A loop body is
|
|
1784
|
+
// dispatched exactly like a one-shot step and needs the same two things: the run that
|
|
1785
|
+
// submitted the graph (so a spawned run gets a ParentRunID and is visible to the tree and to
|
|
1786
|
+
// cost), and the continuation depth (so the recursion cap still applies). Omitting them made
|
|
1787
|
+
// loop bodies second-class in every dimension — and reopened the unbounded-recursion hole
|
|
1788
|
+
// THROUGH loops, since each spawned run restarted the chain at zero.
|
|
1789
|
+
const graphContext = await this.graphContext(provider, task);
|
|
1790
|
+
// THE LOOP'S PAYLOAD ACCUMULATES. Each iteration's output merges in, and the next iteration
|
|
1791
|
+
// — and the While condition — sees it. Without this the condition closure re-read the
|
|
1792
|
+
// payload as it was when the loop STARTED, so a `while payload.brandOK !== true` could never
|
|
1793
|
+
// become false: the loop burned every iteration re-examining the original input and always
|
|
1794
|
+
// took the give-up branch, making the other branch unreachable. The loop ran, reported
|
|
1795
|
+
// success, and its result was predetermined.
|
|
1796
|
+
let livePayload = { ...payload };
|
|
1797
|
+
// One entry per pass, so the loop's work exists somewhere the platform can see it. Without
|
|
1798
|
+
// this a loop is a single childless node: the run tree reaches nested work through six links
|
|
1799
|
+
// and an iteration is none of them, so the passes were invisible to the timeline AND their
|
|
1800
|
+
// spend was missing from the settlement rollup. See ITaskStepRuntime.iterations.
|
|
1801
|
+
const iterationTrace = [];
|
|
1802
|
+
// Bounds what the trace's payloads may cost. The pointers are never budgeted — those are the
|
|
1803
|
+
// durable record of the work and must survive whatever the payloads do.
|
|
1804
|
+
const budget = new IterationPayloadBudget();
|
|
1805
|
+
const invokeBody = async ({ Index, Bindings }) => {
|
|
1806
|
+
// Bindings go INTO the payload rather than beside it, so an authored mapping reaches the
|
|
1807
|
+
// current item the same way it reaches anything else: `payload.<itemVariable>`.
|
|
1808
|
+
const iterationPayload = { ...livePayload, ...Bindings };
|
|
1809
|
+
const resolved = ResolveMappedInput(bodyMapping, { payload: iterationPayload });
|
|
1810
|
+
/**
|
|
1811
|
+
* Folds an iteration's output into the running payload the next pass will see, and
|
|
1812
|
+
* records what the pass produced.
|
|
1813
|
+
*
|
|
1814
|
+
* The trace is written HERE rather than after the loop because a loop that fails partway
|
|
1815
|
+
* still ran the passes before it, and their runs are real spend that must not vanish
|
|
1816
|
+
* because the loop as a whole did not finish.
|
|
1817
|
+
*/
|
|
1818
|
+
const absorb = (outcome, bodyInput) => {
|
|
1819
|
+
livePayload = this.foldIterationOutput(task, livePayload, outcome.Output, bodyOutputMapping);
|
|
1820
|
+
iterationTrace.push({
|
|
1821
|
+
index: Index,
|
|
1822
|
+
// What THIS pass was handed and what it gave back — not the loop's running
|
|
1823
|
+
// payload before and after it.
|
|
1824
|
+
//
|
|
1825
|
+
// A pass has no row of its own, so without these there is nowhere its work can be
|
|
1826
|
+
// recorded: every iteration presented null on both sides and the run view could
|
|
1827
|
+
// say nothing about any single pass, which for a loop is the only interesting
|
|
1828
|
+
// question. But recording the RUNNING payload on both sides — the obvious reading
|
|
1829
|
+
// of "before and after" — is quadratic: each pass would hold a full copy of
|
|
1830
|
+
// everything every earlier pass accumulated. A five-iteration demo produced a
|
|
1831
|
+
// 121KB Configuration that way; the same loop over fifty items would produce
|
|
1832
|
+
// megabytes, in a column every reader of the row pays to load.
|
|
1833
|
+
//
|
|
1834
|
+
// The pass's own input and output are what a reader actually wants ("what did
|
|
1835
|
+
// pass three do?"), and they are constant-sized per pass.
|
|
1836
|
+
payloadAtStart: budget.Take(bodyInput),
|
|
1837
|
+
payloadAtEnd: budget.Take(outcome.Output),
|
|
1838
|
+
promptRunID: outcome.PromptRunID,
|
|
1839
|
+
agentRunID: outcome.AgentRunID,
|
|
1840
|
+
// An ACTION body records its log here. Omitting it left an action-bodied pass
|
|
1841
|
+
// with no pointer at all — no cost, no timing, nothing to open — and the tree,
|
|
1842
|
+
// seeing neither a prompt run nor an agent run, fell through to its last branch
|
|
1843
|
+
// and called the pass a Sub-Agent. A loop over a web search then showed five
|
|
1844
|
+
// sub-agent runs that never existed.
|
|
1845
|
+
actionLogID: outcome.ActionLogID,
|
|
1846
|
+
success: outcome.Success,
|
|
1847
|
+
errorMessage: outcome.ErrorMessage,
|
|
1848
|
+
});
|
|
1849
|
+
return outcome;
|
|
1850
|
+
};
|
|
1851
|
+
// A prompt body is checked FIRST because it is the only one whose id lives in its own
|
|
1852
|
+
// column: a loop repeating a prompt has PromptID set and both ActionID and AgentID null,
|
|
1853
|
+
// so falling through to the agent branch would dereference a null agent id.
|
|
1854
|
+
if (task.StepType && task.PromptID && !task.ActionID) {
|
|
1855
|
+
if (!this.promptRunner) {
|
|
1856
|
+
return { Success: false, ErrorMessage: 'No prompt runner is loaded on this host.' };
|
|
1857
|
+
}
|
|
1858
|
+
return absorb(await this.promptRunner.RunPromptForTask({
|
|
1859
|
+
TaskID: task.ID,
|
|
1860
|
+
PromptID: task.PromptID,
|
|
1861
|
+
// The ITERATION payload, not the mapped params. An action body declares its
|
|
1862
|
+
// inputs and gets exactly those; a prompt body declares none — it receives the
|
|
1863
|
+
// whole payload through the placeholder, and the loop's item and index are
|
|
1864
|
+
// merged INTO that payload. Passing the mapped result here handed the prompt an
|
|
1865
|
+
// empty object, so every iteration asked the model to describe nothing and got
|
|
1866
|
+
// five confident answers about nothing back.
|
|
1867
|
+
InputPayload: iterationPayload,
|
|
1868
|
+
DependencyOutputs: dependencyOutputs,
|
|
1869
|
+
// The loop's bindings become TEMPLATE VARIABLES, so an author writes
|
|
1870
|
+
// `{{ field }}` for the item the loop is on — which is what `itemVariable` is
|
|
1871
|
+
// for, and what anyone reading the step's configuration expects. Reaching it
|
|
1872
|
+
// through the payload placeholder instead works but is not discoverable, and
|
|
1873
|
+
// getting it wrong is silent: the variable renders empty and the model answers
|
|
1874
|
+
// confidently about nothing.
|
|
1875
|
+
TemplateParameters: { ...stringifyBindings(Bindings), ...op.prompt?.templateParameters },
|
|
1876
|
+
Provider: provider,
|
|
1877
|
+
ContextUser: this.contextUser,
|
|
1878
|
+
}), iterationPayload);
|
|
1879
|
+
}
|
|
1880
|
+
if (task.ActionID) {
|
|
1881
|
+
return absorb(await this.actionRunner.RunActionForTask({
|
|
1882
|
+
TaskID: task.ID,
|
|
1883
|
+
ActionID: task.ActionID,
|
|
1884
|
+
InputPayload: resolved,
|
|
1885
|
+
DependencyOutputs: dependencyOutputs,
|
|
1886
|
+
Provider: provider,
|
|
1887
|
+
ContextUser: this.contextUser,
|
|
1888
|
+
}), resolved);
|
|
1889
|
+
}
|
|
1890
|
+
const agentInput = Object.keys(resolved).length > 0 ? resolved : iterationPayload;
|
|
1891
|
+
return absorb(await this.agentRunner.RunAgentForTask({
|
|
1892
|
+
TaskID: task.ID,
|
|
1893
|
+
AgentID: task.AgentID,
|
|
1894
|
+
// The ITERATION payload when the body declares no inputs of its own. A sub-agent
|
|
1895
|
+
// body has no `params`, so the mapped result is `{}` — every iteration was handing
|
|
1896
|
+
// the agent nothing and asking it to work from that.
|
|
1897
|
+
InputPayload: agentInput,
|
|
1898
|
+
DependencyOutputs: dependencyOutputs,
|
|
1899
|
+
ContinuationDepth: graphContext.Depth,
|
|
1900
|
+
SubmittingAgentRunID: graphContext.SubmittingAgentRunID,
|
|
1901
|
+
Provider: provider,
|
|
1902
|
+
ContextUser: this.contextUser,
|
|
1903
|
+
}), agentInput);
|
|
1904
|
+
};
|
|
1905
|
+
const outcome = task.StepType === 'ForEach'
|
|
1906
|
+
? await RunForEachLoop(op, { payload }, invokeBody)
|
|
1907
|
+
: await RunWhileLoop(op, (iteration) => this.conditionEvaluator.Evaluate(op.condition,
|
|
1908
|
+
// BOTH forms, because a workflow should not have two condition dialects. An
|
|
1909
|
+
// EDGE condition is written `payload.brandOK !== true`; a loop condition used
|
|
1910
|
+
// to see the payload's keys spread at the top level and nothing named `payload`,
|
|
1911
|
+
// so the same expression that routes an edge failed here with
|
|
1912
|
+
// "payload is not defined". The spread stays for conditions already written
|
|
1913
|
+
// against it.
|
|
1914
|
+
{ ...livePayload, payload: livePayload, iteration }), invokeBody);
|
|
1915
|
+
return {
|
|
1916
|
+
Success: outcome.Success,
|
|
1917
|
+
AgentRunID: null,
|
|
1918
|
+
ErrorMessage: outcome.ErrorMessage,
|
|
1919
|
+
// Every pass that ran, including those before a failure — see `iterationTrace`.
|
|
1920
|
+
Iterations: iterationTrace.length > 0 ? iterationTrace : undefined,
|
|
1921
|
+
// The ACCUMULATED payload — everything the iterations established — not the one the
|
|
1922
|
+
// loop started with, which would discard the loop's whole effect on the workflow.
|
|
1923
|
+
//
|
|
1924
|
+
// Only the STEP's own mapping is applied here. The body's mapping already ran once per
|
|
1925
|
+
// pass inside `foldIterationOutput`; applying it again against the accumulated payload
|
|
1926
|
+
// is what used to make it match nothing.
|
|
1927
|
+
Output: this.applyStepOutputMapping(task, livePayload, outcome.Output, config?.outputMapping),
|
|
1928
|
+
};
|
|
1929
|
+
}
|
|
1930
|
+
/**
|
|
1931
|
+
* Folds one pass's result into the loop's running payload.
|
|
1932
|
+
*
|
|
1933
|
+
* **With a body mapping**, the pass's declared outputs are filed where the author said to put
|
|
1934
|
+
* them — including `name[]`, which appends, so a ForEach can collect one entry per item. That is
|
|
1935
|
+
* the whole point of a loop over a collection, and it is only expressible per pass.
|
|
1936
|
+
*
|
|
1937
|
+
* **Without one**, the raw result is deep-merged, which is the pre-existing behaviour and the
|
|
1938
|
+
* right default for a `While` that converges on a value: each pass refines what the condition
|
|
1939
|
+
* reads. It is the wrong default for a ForEach that collects — hence the mapping.
|
|
1940
|
+
*
|
|
1941
|
+
* An unmapped output is reported per pass rather than swallowed, for the same reason
|
|
1942
|
+
* {@link applyStepOutputMapping} reports it: a mapping that names something the body never
|
|
1943
|
+
* returned means the pass did work that went nowhere, while everything reports success.
|
|
1944
|
+
*/
|
|
1945
|
+
foldIterationOutput(task, livePayload, output, bodyOutputMapping) {
|
|
1946
|
+
if (!output || typeof output !== 'object' || Array.isArray(output))
|
|
1947
|
+
return livePayload;
|
|
1948
|
+
const source = output;
|
|
1949
|
+
if (!bodyOutputMapping)
|
|
1950
|
+
return deepMergePayload(livePayload, source);
|
|
1951
|
+
// Applied ONTO a deep copy of the running payload, not into a fresh object: `name[]` appends,
|
|
1952
|
+
// and appending is meaningless without the list already there. The copy is deep because the
|
|
1953
|
+
// trace has already recorded earlier passes' payloads — mutating a shared nested array would
|
|
1954
|
+
// retroactively rewrite what those passes are recorded as having seen.
|
|
1955
|
+
const { updates, errors, unmapped } = ApplyOutputMapping(source, bodyOutputMapping, structuredClone(livePayload));
|
|
1956
|
+
for (const e of errors)
|
|
1957
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} loop body: ${e}`);
|
|
1958
|
+
if (unmapped?.length) {
|
|
1959
|
+
LogError(`[TaskGraphDispatcher] '${task.Name}' loop body mapped output(s) it did not return: ` +
|
|
1960
|
+
`${unmapped.join(', ')}. The pass returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
|
|
1961
|
+
`Those payload values were NOT written, so anything downstream reading them sees nothing.`);
|
|
1962
|
+
}
|
|
1963
|
+
// `updates` IS the copy that was applied onto, so it is already the complete next payload.
|
|
1964
|
+
return updates;
|
|
1965
|
+
}
|
|
1966
|
+
/**
|
|
1967
|
+
* Files a step's result into the payload it hands downstream.
|
|
1968
|
+
*
|
|
1969
|
+
* **This is what makes a branch condition possible.** A workflow that branches on
|
|
1970
|
+
* `payload.stockPrice` has that value only because this step mapped `CurrentPrice -> stockPrice`.
|
|
1971
|
+
* Without it the condition reads `undefined` — merely falsy — so the workflow takes the other
|
|
1972
|
+
* branch, finishes, and reports success with nothing to indicate anything went wrong.
|
|
1973
|
+
*
|
|
1974
|
+
* The incoming payload is carried through as well as the update, so a value written three steps
|
|
1975
|
+
* back is still readable here. Returning only this step's own output is what used to limit a
|
|
1976
|
+
* condition's view to its immediate predecessor.
|
|
1977
|
+
*/
|
|
1978
|
+
applyStepOutputMapping(task, payload, output, outputMapping) {
|
|
1979
|
+
// No mapping: MERGE the step's output over the payload rather than replacing it.
|
|
1980
|
+
//
|
|
1981
|
+
// Replacing is what made the Content Pipeline's exclusive pair unreachable. A While loop's
|
|
1982
|
+
// own output is a SUMMARY — `{iterations, succeeded, failed, results}` — so returning it
|
|
1983
|
+
// discarded the payload the iterations had built, including the `brandOK` the reviewer had
|
|
1984
|
+
// just set to true. The edges read `payload.brandOK === true` and `!== true`; against a
|
|
1985
|
+
// summary the first is false and the second is true, so the give-up branch won on EVERY run
|
|
1986
|
+
// no matter what the reviewer decided. The approved branch was unreachable in practice while
|
|
1987
|
+
// being perfectly reachable on the canvas.
|
|
1988
|
+
//
|
|
1989
|
+
// This is the same rule the mapped path already follows two lines down, and the same rule
|
|
1990
|
+
// the doc comment above states. The no-mapping branch was simply not following it.
|
|
1991
|
+
if (!outputMapping) {
|
|
1992
|
+
return output && typeof output === 'object' && !Array.isArray(output)
|
|
1993
|
+
? { ...payload, ...output }
|
|
1994
|
+
: output ?? payload;
|
|
1995
|
+
}
|
|
1996
|
+
const source = output && typeof output === 'object' ? output : { value: output };
|
|
1997
|
+
const { updates, errors, unmapped } = ApplyOutputMapping(source, outputMapping);
|
|
1998
|
+
for (const e of errors)
|
|
1999
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
|
|
2000
|
+
// A mapping that names an output the step never produced discards that step's work while
|
|
2001
|
+
// the step reports Complete. It is not fatal — an action may emit a parameter only on some
|
|
2002
|
+
// paths — but it must not be silent, and naming what WAS returned turns a multi-table
|
|
2003
|
+
// forensic exercise into one line. The Content Pipeline demo lost an entire research pass
|
|
2004
|
+
// this way, every run, because its mapping named another action's parameter.
|
|
2005
|
+
if (unmapped?.length) {
|
|
2006
|
+
LogError(`[TaskGraphDispatcher] '${task.Name}' mapped output(s) the step did not return: ` +
|
|
2007
|
+
`${unmapped.join(', ')}. The step returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
|
|
2008
|
+
`Those payload values were NOT written, so anything downstream reading them sees nothing.`);
|
|
2009
|
+
}
|
|
2010
|
+
return { ...payload, ...updates };
|
|
2011
|
+
}
|
|
2012
|
+
/**
|
|
2013
|
+
* Runs an Agent step, telling the runner where in the graph it sits.
|
|
2014
|
+
*
|
|
2015
|
+
* Depth and provenance are read together because they come from the same row: the graph's parent
|
|
2016
|
+
* task knows both how many continuation hops led here and which run submitted it.
|
|
2017
|
+
*/
|
|
2018
|
+
async runAgentNode(task, provider, effectiveInput, dependencyOutputs) {
|
|
2019
|
+
const context = await this.graphContext(provider, task);
|
|
2020
|
+
return this.agentRunner.RunAgentForTask({
|
|
2021
|
+
TaskID: task.ID,
|
|
2022
|
+
AgentID: task.AgentID,
|
|
2023
|
+
InputPayload: effectiveInput,
|
|
2024
|
+
DependencyOutputs: dependencyOutputs,
|
|
2025
|
+
ContinuationDepth: context.Depth,
|
|
2026
|
+
SubmittingAgentRunID: context.SubmittingAgentRunID,
|
|
2027
|
+
Provider: provider,
|
|
2028
|
+
ContextUser: this.contextUser,
|
|
2029
|
+
});
|
|
2030
|
+
}
|
|
2031
|
+
/**
|
|
2032
|
+
* Completes the agent run that parked on this graph.
|
|
2033
|
+
*
|
|
2034
|
+
* **This is the other half of submit-and-detach.** A run that dispatches a graph does not
|
|
2035
|
+
* complete at submission — it ends `Paused`, because reporting `Completed` above a workflow
|
|
2036
|
+
* where nothing has happened yet is a claim the row cannot support. The run's lifecycle is
|
|
2037
|
+
* finished HERE, when the graph it was waiting on actually settles, which is the first moment
|
|
2038
|
+
* the answer exists.
|
|
2039
|
+
*
|
|
2040
|
+
* Doing it from the dispatcher rather than by awaiting in the agent is what keeps the properties
|
|
2041
|
+
* that made detach right in the first place: a graph containing a human approval can park for
|
|
2042
|
+
* days without holding a conversation turn open, and a graph reclaimed by another instance after
|
|
2043
|
+
* a crash still settles its submitting run, because the settling happens wherever the graph
|
|
2044
|
+
* finishes rather than wherever it started.
|
|
2045
|
+
*
|
|
2046
|
+
* **Only a parked run is touched.** A run that is already `Completed`, `Failed` or `Cancelled`
|
|
2047
|
+
* reached that state for its own reasons — a second graph settling later, a run the user
|
|
2048
|
+
* cancelled, a run that failed after submitting — and overwriting it would rewrite history from
|
|
2049
|
+
* the outside. The `Paused` predicate is the whole guard.
|
|
2050
|
+
*
|
|
2051
|
+
* @param graphStatus the parent rollup's status: what the workflow as a whole did
|
|
2052
|
+
*/
|
|
2053
|
+
async settleSubmittingRun(provider, parent, graphStatus) {
|
|
2054
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
2055
|
+
if (!meta.submittedByAgentRunID)
|
|
2056
|
+
return; // a scheduled or remote-triggered graph has nobody waiting
|
|
2057
|
+
try {
|
|
2058
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
2059
|
+
if (!(await run.Load(meta.submittedByAgentRunID))) {
|
|
2060
|
+
LogError(`[TaskGraphDispatcher] Could not load run ${meta.submittedByAgentRunID} to settle it against graph ${parent.ID}.`);
|
|
2061
|
+
return;
|
|
2062
|
+
}
|
|
2063
|
+
if (run.Status !== 'Paused')
|
|
2064
|
+
return;
|
|
2065
|
+
// The workflow's outcome becomes the run's outcome. A graph that ended any way other than
|
|
2066
|
+
// Complete did not do what the run started it to do, and a run reporting success over it
|
|
2067
|
+
// would be the same untruth in a different place.
|
|
2068
|
+
const succeeded = graphStatus === 'Complete';
|
|
2069
|
+
run.Status = succeeded ? 'Completed' : 'Failed';
|
|
2070
|
+
run.Success = succeeded;
|
|
2071
|
+
run.CompletedAt = new Date();
|
|
2072
|
+
if (!succeeded) {
|
|
2073
|
+
const reason = `The workflow "${parent.Name}" ended ${graphStatus}.`;
|
|
2074
|
+
run.ErrorMessage = run.ErrorMessage ? `${run.ErrorMessage}\n\n${reason}` : reason;
|
|
2075
|
+
}
|
|
2076
|
+
if (!(await run.Save())) {
|
|
2077
|
+
// Left parked rather than forced. A run stuck at Paused is visibly unfinished, which
|
|
2078
|
+
// is a state someone can investigate; a run flipped to Completed by a write that did
|
|
2079
|
+
// not land would be the same lie this whole change removes.
|
|
2080
|
+
LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}: ` +
|
|
2081
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It remains Paused.`);
|
|
2082
|
+
return;
|
|
2083
|
+
}
|
|
2084
|
+
LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${run.Status} — workflow "${parent.Name}" ended ${graphStatus}.`);
|
|
2085
|
+
}
|
|
2086
|
+
catch (e) {
|
|
2087
|
+
LogError(`[TaskGraphDispatcher] Could not settle the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
2088
|
+
}
|
|
2089
|
+
}
|
|
2090
|
+
/**
|
|
2091
|
+
* Gives every step that lacks one a position, once the graph has finished.
|
|
2092
|
+
*
|
|
2093
|
+
* **Why the run stores geometry at all.** A `TaskGraphSpec` is a logical structure with no
|
|
2094
|
+
* layout field, so a graph an agent emitted has no opinion about where its boxes go. Every
|
|
2095
|
+
* viewer was therefore laying it out for itself at render time — and a viewer that failed to
|
|
2096
|
+
* (because the canvas measures nodes it has not drawn yet) fell back to every node at the
|
|
2097
|
+
* origin, piled on one another, with the zoom-to-fit that follows fitting a one-node bounding
|
|
2098
|
+
* box. Settling it once, server-side, means the agent-run canvas, the Workflows runs tab and
|
|
2099
|
+
* anything built later all draw the same picture, and none of them has to compute it.
|
|
2100
|
+
*
|
|
2101
|
+
* **An authored position is never overwritten.** A workflow compiled from a Flow agent carries
|
|
2102
|
+
* the arrangement someone dragged into place; replacing it with an algorithm's guess would
|
|
2103
|
+
* discard a deliberate act. Only steps with no geometry get one, so a partially-arranged graph
|
|
2104
|
+
* keeps what it has.
|
|
2105
|
+
*
|
|
2106
|
+
* Failure here is logged and swallowed: this is presentation. A graph whose work completed must
|
|
2107
|
+
* not be reported as failed because its picture could not be saved.
|
|
2108
|
+
*/
|
|
2109
|
+
async persistComputedLayout(graph) {
|
|
2110
|
+
try {
|
|
2111
|
+
const needsLayout = [...graph.entityById.values()].filter((t) => !this.parseConfiguration(t)?.layout);
|
|
2112
|
+
if (needsLayout.length === 0)
|
|
2113
|
+
return;
|
|
2114
|
+
// Laid out over the WHOLE graph, not just the nodes missing geometry: position depends on
|
|
2115
|
+
// where a node sits in the topology, and a layout computed over a subset would place its
|
|
2116
|
+
// nodes as though the rest of the workflow did not exist.
|
|
2117
|
+
const edges = graph.edges.map((e) => ({ From: e.dependsOnTaskId, To: e.taskId }));
|
|
2118
|
+
const positions = LayoutGraphNodes([...graph.entityById.keys()], edges, { Direction: 'LR' });
|
|
2119
|
+
for (const task of needsLayout) {
|
|
2120
|
+
const position = positions.get(task.ID);
|
|
2121
|
+
if (!position)
|
|
2122
|
+
continue;
|
|
2123
|
+
const existing = this.parseConfiguration(task);
|
|
2124
|
+
const merged = {
|
|
2125
|
+
...existing,
|
|
2126
|
+
layout: { x: position.X, y: position.Y },
|
|
2127
|
+
};
|
|
2128
|
+
task.Configuration = JSON.stringify(merged);
|
|
2129
|
+
if (!(await task.Save())) {
|
|
2130
|
+
LogError(`[TaskGraphDispatcher] Could not save computed layout for ${task.ID}: ${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
2131
|
+
}
|
|
2132
|
+
}
|
|
2133
|
+
}
|
|
2134
|
+
catch (e) {
|
|
2135
|
+
LogError(`[TaskGraphDispatcher] Could not compute a layout for the settled graph: ${e instanceof Error ? e.message : String(e)}`);
|
|
2136
|
+
}
|
|
2137
|
+
}
|
|
2138
|
+
/**
|
|
2139
|
+
* The earliest moment any step in the graph began, or null when none has.
|
|
2140
|
+
*
|
|
2141
|
+
* Null is a real answer — a graph whose tasks are all still Pending has not started — and is
|
|
2142
|
+
* deliberately not collapsed to "now", which would date the graph from whenever this pass
|
|
2143
|
+
* happened to run.
|
|
2144
|
+
*/
|
|
2145
|
+
earliestStart(entityById) {
|
|
2146
|
+
let earliest = null;
|
|
2147
|
+
for (const entity of entityById.values()) {
|
|
2148
|
+
if (!entity.StartedAt)
|
|
2149
|
+
continue;
|
|
2150
|
+
if (earliest === null || entity.StartedAt < earliest)
|
|
2151
|
+
earliest = entity.StartedAt;
|
|
2152
|
+
}
|
|
2153
|
+
return earliest;
|
|
2154
|
+
}
|
|
2155
|
+
/**
|
|
2156
|
+
* The step's Configuration with this run's artefacts folded in, or `undefined` to leave it be.
|
|
2157
|
+
*
|
|
2158
|
+
* **Merged into the authored bag, never written over it.** The Configuration column holds the
|
|
2159
|
+
* step's definition — its loop body, its mappings, its policy, the position someone dragged it
|
|
2160
|
+
* to. Writing a fresh object containing only `runtime` would erase all of that the first time a
|
|
2161
|
+
* prompt step completed, which is the kind of loss that surfaces much later as a workflow that
|
|
2162
|
+
* mysteriously stopped mapping its output.
|
|
2163
|
+
*
|
|
2164
|
+
* Returns `undefined` when there is nothing to record, so the guarded write omits the column
|
|
2165
|
+
* rather than rewriting it with what it already held.
|
|
2166
|
+
*/
|
|
2167
|
+
configurationWithRuntime(task, promptRunID, actionLogID, iterations, payloadAtStart) {
|
|
2168
|
+
if (!promptRunID && !actionLogID && !iterations?.length && !payloadAtStart)
|
|
2169
|
+
return undefined;
|
|
2170
|
+
const existing = this.parseConfiguration(task);
|
|
2171
|
+
const merged = {
|
|
2172
|
+
...existing,
|
|
2173
|
+
runtime: {
|
|
2174
|
+
...existing?.runtime,
|
|
2175
|
+
...(promptRunID ? { promptRunID } : {}),
|
|
2176
|
+
...(actionLogID ? { actionLogID } : {}),
|
|
2177
|
+
// Replaced wholesale rather than appended: this is the trace of the loop's LAST
|
|
2178
|
+
// execution, and a retried step that concatenated would report a loop that ran twice
|
|
2179
|
+
// as many passes as it did.
|
|
2180
|
+
...(iterations?.length ? { iterations } : {}),
|
|
2181
|
+
// The resolved before-state, so the run view has something to diff the output
|
|
2182
|
+
// against. NOT written to Task.InputPayload, which holds the AUTHORED input and
|
|
2183
|
+
// round-trips back out as part of the spec.
|
|
2184
|
+
...(payloadAtStart ? { payloadAtStart } : {}),
|
|
2185
|
+
},
|
|
2186
|
+
};
|
|
2187
|
+
return JSON.stringify(merged);
|
|
2188
|
+
}
|
|
2189
|
+
/**
|
|
2190
|
+
* Reads a step's Configuration bag, tolerating a row whose JSON cannot be parsed.
|
|
2191
|
+
*
|
|
2192
|
+
* Unparseable configuration is logged rather than thrown: the step has already RUN by the time
|
|
2193
|
+
* this is called, and refusing to record its outcome because its definition is malformed would
|
|
2194
|
+
* discard the result of real work and leave the task claimed until the claim lapsed.
|
|
2195
|
+
*/
|
|
2196
|
+
parseConfiguration(task) {
|
|
2197
|
+
if (!task.Configuration)
|
|
2198
|
+
return undefined;
|
|
2199
|
+
try {
|
|
2200
|
+
return JSON.parse(task.Configuration);
|
|
2201
|
+
}
|
|
2202
|
+
catch (e) {
|
|
2203
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} has unparseable Configuration; ` +
|
|
2204
|
+
`recording runtime artefacts against an empty bag. ${e instanceof Error ? e.message : String(e)}`);
|
|
2205
|
+
return undefined;
|
|
2206
|
+
}
|
|
2207
|
+
}
|
|
2208
|
+
/**
|
|
2209
|
+
* The payload a step sees: everything its prerequisites produced, plus its own declared input.
|
|
2210
|
+
*
|
|
2211
|
+
* **Why the outputs are merged rather than kept per-task.** A flow carried ONE payload that
|
|
2212
|
+
* accumulated as it went, so a condition on the edge into step C could read a value step A wrote.
|
|
2213
|
+
* Handing each task only its immediate predecessor's output would silently narrow that: the
|
|
2214
|
+
* condition reads `undefined`, which is falsy, and the workflow quietly takes a different route
|
|
2215
|
+
* than the flow it was compiled from. Merging in dependency order restores the accumulation.
|
|
2216
|
+
*
|
|
2217
|
+
* Later prerequisites win on a key collision, matching a flow's own last-write-wins behaviour.
|
|
2218
|
+
*/
|
|
2219
|
+
mergedPayload(inputPayload, dependencyOutputs) {
|
|
2220
|
+
const merged = {};
|
|
2221
|
+
for (const output of dependencyOutputs.values()) {
|
|
2222
|
+
if (output && typeof output === 'object' && !Array.isArray(output)) {
|
|
2223
|
+
Object.assign(merged, output);
|
|
2224
|
+
}
|
|
2225
|
+
}
|
|
2226
|
+
if (inputPayload && typeof inputPayload === 'object' && !Array.isArray(inputPayload)) {
|
|
2227
|
+
Object.assign(merged, inputPayload);
|
|
2228
|
+
}
|
|
2229
|
+
return merged;
|
|
2230
|
+
}
|
|
2231
|
+
}
|
|
2232
|
+
//# sourceMappingURL=TaskGraphDispatcher.js.map
|