@memberjunction/task-graph 6.1.0-edge.1 → 6.1.0-edge.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +180 -4
- package/README.md +214 -0
- package/dist/TaskClaimStore.d.ts +387 -4
- package/dist/TaskClaimStore.d.ts.map +1 -1
- package/dist/TaskClaimStore.js +605 -20
- package/dist/TaskClaimStore.js.map +1 -1
- package/dist/TaskGraphDispatcher.d.ts +668 -5
- package/dist/TaskGraphDispatcher.d.ts.map +1 -1
- package/dist/TaskGraphDispatcher.js +2942 -127
- package/dist/TaskGraphDispatcher.js.map +1 -1
- package/dist/TaskGraphService.d.ts +364 -5
- package/dist/TaskGraphService.d.ts.map +1 -1
- package/dist/TaskGraphService.js +1039 -43
- package/dist/TaskGraphService.js.map +1 -1
- package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -1
- package/dist/TaskGraphSubmitterImpl.js +5 -0
- package/dist/TaskGraphSubmitterImpl.js.map +1 -1
- package/dist/TaskLoopExecutor.d.ts +62 -0
- package/dist/TaskLoopExecutor.d.ts.map +1 -0
- package/dist/TaskLoopExecutor.js +248 -0
- package/dist/TaskLoopExecutor.js.map +1 -0
- package/dist/WorkflowSpecSync.d.ts +28 -2
- package/dist/WorkflowSpecSync.d.ts.map +1 -1
- package/dist/WorkflowSpecSync.js +83 -2
- package/dist/WorkflowSpecSync.js.map +1 -1
- package/dist/condition-gate.d.ts +128 -0
- package/dist/condition-gate.d.ts.map +1 -0
- package/dist/condition-gate.js +257 -0
- package/dist/condition-gate.js.map +1 -0
- package/dist/debug-state.d.ts +102 -0
- package/dist/debug-state.d.ts.map +1 -0
- package/dist/debug-state.js +135 -0
- package/dist/debug-state.js.map +1 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -0
- package/dist/index.js.map +1 -1
- package/dist/operations/TaskGraphDebugOperations.d.ts +99 -0
- package/dist/operations/TaskGraphDebugOperations.d.ts.map +1 -0
- package/dist/operations/TaskGraphDebugOperations.js +310 -0
- package/dist/operations/TaskGraphDebugOperations.js.map +1 -0
- package/dist/operations/TaskGraphOperations.d.ts +20 -2
- package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
- package/dist/operations/TaskGraphOperations.js +51 -8
- package/dist/operations/TaskGraphOperations.js.map +1 -1
- package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
- package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
- package/dist/operations/WorkflowDraftOperation.js +141 -0
- package/dist/operations/WorkflowDraftOperation.js.map +1 -0
- package/dist/settlement-rescue.d.ts +85 -0
- package/dist/settlement-rescue.d.ts.map +1 -0
- package/dist/settlement-rescue.js +119 -0
- package/dist/settlement-rescue.js.map +1 -0
- package/dist/task-graph-kick.d.ts +3 -0
- package/dist/task-graph-kick.d.ts.map +1 -0
- package/dist/task-graph-kick.js +17 -0
- package/dist/task-graph-kick.js.map +1 -0
- package/dist/task-predicates.d.ts +77 -0
- package/dist/task-predicates.d.ts.map +1 -0
- package/dist/task-predicates.js +75 -0
- package/dist/task-predicates.js.map +1 -0
- package/dist/types.d.ts +224 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js.map +1 -1
- package/package.json +12 -8
|
@@ -18,14 +18,44 @@
|
|
|
18
18
|
*
|
|
19
19
|
* @module @memberjunction/task-graph
|
|
20
20
|
*/
|
|
21
|
-
import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, } from '@memberjunction/ai-core-plus';
|
|
21
|
+
import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, ResolveExclusiveGroups, ComputeSkipCascade, ConfirmSkipSeeds, LayoutGraphNodes, ApplyOutputMapping, BuildMappedInput, ResolveMappedInput, LoadAgentRunTree, SumAgentRunTreeCost, WalkAgentRunTree, } from '@memberjunction/ai-core-plus';
|
|
22
22
|
import { LogError, LogStatus, RunView } from '@memberjunction/core';
|
|
23
|
-
import { ShutdownRegistry } from '@memberjunction/global';
|
|
24
|
-
import { TaskClaimStore } from './TaskClaimStore.js';
|
|
23
|
+
import { ShutdownRegistry, UUIDsEqual } from '@memberjunction/global';
|
|
24
|
+
import { TaskClaimStore, TERMINAL_PARENT_STATUSES, TERMINAL_PARENT_STATUS_SQL } from './TaskClaimStore.js';
|
|
25
|
+
import { BuildConditionContext, DecideGate, IsBrokenGuard, ParseConditionOutput, } from './condition-gate.js';
|
|
26
|
+
import { HumanTaskSQL, IsHumanTask } from './task-predicates.js';
|
|
27
|
+
import { IsSettlementExpired, IsSubmittingRunReady, SelectUnsettledGraphIDs, SweepCutoff, UNSETTLED_SWEEP_WINDOW_HOURS, UNSETTLED_STARTUP_WINDOW_HOURS, } from './settlement-rescue.js';
|
|
25
28
|
import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
|
|
29
|
+
import { DecideClaimGate, OverrideVerdictFor, ParseTaskGraphDebugState, } from './debug-state.js';
|
|
30
|
+
import { RunForEachLoop, RunWhileLoop } from './TaskLoopExecutor.js';
|
|
31
|
+
import { RegisterTaskGraphKick } from './task-graph-kick.js';
|
|
26
32
|
import { NotificationEngine } from '@memberjunction/notifications';
|
|
27
33
|
/** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
|
|
28
34
|
const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
|
|
35
|
+
/**
|
|
36
|
+
* Statuses a graph parent has stopped moving from.
|
|
37
|
+
*
|
|
38
|
+
* Shared by the guarded terminal write and the unsettled-graph sweep, so "terminal" means exactly
|
|
39
|
+
* one thing in both — the two disagreeing is how a graph becomes invisible to the machinery that is
|
|
40
|
+
* supposed to rescue it.
|
|
41
|
+
*/
|
|
42
|
+
const TERMINAL_TASK_STATUSES = new Set(TERMINAL_PARENT_STATUSES);
|
|
43
|
+
/**
|
|
44
|
+
* How long `Stop()` waits for in-flight tasks and timer passes before giving up and saying so.
|
|
45
|
+
*
|
|
46
|
+
* Generous, because the alternative to waiting is a dispatcher that writes after its host believes
|
|
47
|
+
* it has shut down — settling graphs onto a connection somebody else now owns.
|
|
48
|
+
*/
|
|
49
|
+
const STOP_DRAIN_TIMEOUT_MS = 30_000;
|
|
50
|
+
/**
|
|
51
|
+
* How many consecutive failing passes a graph gets before this instance stops re-queueing it.
|
|
52
|
+
*
|
|
53
|
+
* Not a giving-up threshold so much as a stop-shouting one: past this the graph has failed to settle
|
|
54
|
+
* on every attempt for minutes, so something is wrong that another identical attempt will not fix,
|
|
55
|
+
* and continuing costs a full graph load per poll forever. It is reported and left to the startup
|
|
56
|
+
* sweep, which is the wider net.
|
|
57
|
+
*/
|
|
58
|
+
const MAX_SETTLEMENT_RETRY_PASSES = 20;
|
|
29
59
|
/**
|
|
30
60
|
* Written to a human task's `ClaimedBy` once its assignee has been told it is ready.
|
|
31
61
|
*
|
|
@@ -34,9 +64,106 @@ const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
|
|
|
34
64
|
* from reclamation, so this value is never mistaken for a live claim.
|
|
35
65
|
*/
|
|
36
66
|
const HUMAN_TASK_NOTIFIED_MARKER = '__human-notified__';
|
|
37
|
-
|
|
67
|
+
/**
|
|
68
|
+
* The run-query capability of a provider, when it has one.
|
|
69
|
+
*
|
|
70
|
+
* `IMetadataProvider` does not extend `IRunQueryProvider`, but every provider that ships implements
|
|
71
|
+
* both. Narrowing by CAPABILITY rather than casting states that honestly: a provider that genuinely
|
|
72
|
+
* cannot run queries returns undefined and the caller reports it, instead of the call failing later
|
|
73
|
+
* behind a type assertion that claimed it could.
|
|
74
|
+
*/
|
|
75
|
+
function asRunQueryProvider(provider) {
|
|
76
|
+
const candidate = provider;
|
|
77
|
+
return typeof candidate.RunQuery === 'function' ? candidate : undefined;
|
|
78
|
+
}
|
|
79
|
+
import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata, TASK_TYPE_NAME } from './TaskGraphService.js';
|
|
38
80
|
import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
|
|
81
|
+
/**
|
|
82
|
+
* Renders a loop's bindings as template values.
|
|
83
|
+
*
|
|
84
|
+
* Template parameters are strings; an item is usually an object. Objects are JSON-encoded rather
|
|
85
|
+
* than dropped, because `{{ field }}` printing `[object Object]` — or nothing at all — is exactly
|
|
86
|
+
* the silent failure this exists to prevent.
|
|
87
|
+
*/
|
|
88
|
+
function stringifyBindings(bindings) {
|
|
89
|
+
const out = {};
|
|
90
|
+
for (const [key, value] of Object.entries(bindings)) {
|
|
91
|
+
out[key] = typeof value === 'string' ? value : JSON.stringify(value, null, 2);
|
|
92
|
+
}
|
|
93
|
+
return out;
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* How much of a loop's per-pass payloads may be kept, and what happens when that runs out.
|
|
97
|
+
*
|
|
98
|
+
* **Why a budget exists at all.** A loop's trace lives inside one `Configuration` column, and its
|
|
99
|
+
* size is the product of two things nobody bounds: how many passes the loop runs, and how large the
|
|
100
|
+
* body's input and output are. A hundred-pass loop over documents would put megabytes in a column
|
|
101
|
+
* that the run tree, the timeline, the canvas and the Workflows list all read — punishing every
|
|
102
|
+
* reader of the row for a detail only someone inspecting one pass will ever open.
|
|
103
|
+
*
|
|
104
|
+
* **What it protects.** Only the payloads. `promptRunID` / `agentRunID` / `actionLogID` / `success`
|
|
105
|
+
* are always recorded: those point at the durable rows where the real forensics live, and they are
|
|
106
|
+
* what cost roll-up and the timeline traverse. Losing a payload costs a reader some detail; losing a
|
|
107
|
+
* pointer would lose the pass.
|
|
108
|
+
*
|
|
109
|
+
* **Omission is stated, never silent.** Once the budget is spent, further passes record a marker
|
|
110
|
+
* saying so and how large the value was, because a pass showing nothing is indistinguishable from a
|
|
111
|
+
* pass that produced nothing — and that ambiguity is exactly the failure this whole area keeps
|
|
112
|
+
* hitting.
|
|
113
|
+
*/
|
|
114
|
+
const ITERATION_PAYLOAD_BUDGET_BYTES = 128 * 1024;
|
|
115
|
+
/** Per-value cap, so one enormous pass cannot consume the whole budget by itself. */
|
|
116
|
+
const ITERATION_PAYLOAD_VALUE_BYTES = 16 * 1024;
|
|
117
|
+
class IterationPayloadBudget {
|
|
118
|
+
constructor() {
|
|
119
|
+
this.spent = 0;
|
|
120
|
+
}
|
|
121
|
+
/**
|
|
122
|
+
* The value if it fits, or a marker describing what was left out.
|
|
123
|
+
*
|
|
124
|
+
* @returns the value, a marker object, or undefined when there was nothing to record
|
|
125
|
+
*/
|
|
126
|
+
Take(value) {
|
|
127
|
+
if (value == null)
|
|
128
|
+
return undefined;
|
|
129
|
+
const asRecord = value && typeof value === 'object' && !Array.isArray(value)
|
|
130
|
+
? value
|
|
131
|
+
: { value };
|
|
132
|
+
let size;
|
|
133
|
+
try {
|
|
134
|
+
size = JSON.stringify(asRecord)?.length ?? 0;
|
|
135
|
+
}
|
|
136
|
+
catch {
|
|
137
|
+
// Circular or otherwise unserializable. It could not be persisted anyway, and saying so
|
|
138
|
+
// is better than a pass that silently shows nothing.
|
|
139
|
+
return { __omitted: 'unserializable' };
|
|
140
|
+
}
|
|
141
|
+
if (size > ITERATION_PAYLOAD_VALUE_BYTES) {
|
|
142
|
+
return { __omitted: 'too-large', __bytes: size, __limit: ITERATION_PAYLOAD_VALUE_BYTES };
|
|
143
|
+
}
|
|
144
|
+
if (this.spent + size > ITERATION_PAYLOAD_BUDGET_BYTES) {
|
|
145
|
+
return { __omitted: 'budget-exhausted', __bytes: size, __limit: ITERATION_PAYLOAD_BUDGET_BYTES };
|
|
146
|
+
}
|
|
147
|
+
this.spent += size;
|
|
148
|
+
return asRecord;
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
/** Deep-merges a prompt's JSON response into the payload, preserving what earlier steps established. */
|
|
152
|
+
function deepMergePayload(base, incoming) {
|
|
153
|
+
const out = { ...base };
|
|
154
|
+
for (const [key, value] of Object.entries(incoming)) {
|
|
155
|
+
const existing = out[key];
|
|
156
|
+
const bothPlainObjects = existing && typeof existing === 'object' && !Array.isArray(existing) &&
|
|
157
|
+
value && typeof value === 'object' && !Array.isArray(value);
|
|
158
|
+
out[key] = bothPlainObjects
|
|
159
|
+
? deepMergePayload(existing, value)
|
|
160
|
+
: value;
|
|
161
|
+
}
|
|
162
|
+
return out;
|
|
163
|
+
}
|
|
39
164
|
export class TaskGraphDispatcher {
|
|
165
|
+
/** Minimum interval between `NodeProgress` frames for one task. */
|
|
166
|
+
static { this.NODE_PROGRESS_MIN_INTERVAL_MS = 1_000; }
|
|
40
167
|
constructor(providerFactory, agentRunner, contextUser, config,
|
|
41
168
|
/**
|
|
42
169
|
* Optional. Absent means a host that cannot post messages or start agent turns — a worker,
|
|
@@ -48,21 +175,108 @@ export class TaskGraphDispatcher {
|
|
|
48
175
|
* Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
|
|
49
176
|
* announces nothing.
|
|
50
177
|
*/
|
|
51
|
-
observer
|
|
178
|
+
observer,
|
|
179
|
+
/**
|
|
180
|
+
* Optional. Absent means this host cannot run action nodes; they stay Pending and visible
|
|
181
|
+
* rather than being failed, because "nobody here can run this" is not "this ran and broke".
|
|
182
|
+
*/
|
|
183
|
+
actionRunner,
|
|
184
|
+
/**
|
|
185
|
+
* Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
|
|
186
|
+
* rather than being failed, for the same reason action nodes do.
|
|
187
|
+
*/
|
|
188
|
+
promptRunner) {
|
|
52
189
|
this.providerFactory = providerFactory;
|
|
53
190
|
this.agentRunner = agentRunner;
|
|
54
191
|
this.contextUser = contextUser;
|
|
55
192
|
this.continuationDeliverer = continuationDeliverer;
|
|
56
193
|
this.observer = observer;
|
|
194
|
+
this.actionRunner = actionRunner;
|
|
195
|
+
this.promptRunner = promptRunner;
|
|
196
|
+
/**
|
|
197
|
+
* Edges already reported as unevaluable, so the report is once per transition and not once per
|
|
198
|
+
* poll. Per-instance and in-memory by design: a restart re-reports, which is the right amount of
|
|
199
|
+
* noise for a condition that is still broken after a restart.
|
|
200
|
+
*/
|
|
201
|
+
this.reportedUnevaluableConditions = new Set();
|
|
202
|
+
/** Resolved once it EXISTS; null while it does not, so a fresh install is not cached blind. */
|
|
203
|
+
this.cachedWorkflowTaskTypeID = null;
|
|
204
|
+
/**
|
|
205
|
+
* Graphs this instance is still trying to settle, with how many passes it has spent trying.
|
|
206
|
+
*
|
|
207
|
+
* **The sweep's window is on `__mj_UpdatedAt`, and a failing pass writes nothing** — the terminal
|
|
208
|
+
* write returns rowcount 0 because the row is already terminal, the layout pass touches only
|
|
209
|
+
* children, a refused CAS writes nothing at all. So a graph that fails to settle stops advancing
|
|
210
|
+
* its own timestamp and, after 24h of futile retries, ages out of the steady-state window while
|
|
211
|
+
* the process is up. The doc comment claimed the bound was "on abandonment, not age"; for this
|
|
212
|
+
* case it was on age, and R2-2's deferral made the case ordinary rather than exotic.
|
|
213
|
+
*
|
|
214
|
+
* In memory rather than a touch column because the alternative is a write on every failed
|
|
215
|
+
* attempt — more load exactly when something is already wrong — and because a restart is covered
|
|
216
|
+
* by the wide startup sweep, which is the durable backstop this leans on.
|
|
217
|
+
*/
|
|
218
|
+
this.retryingSettlement = new Map();
|
|
219
|
+
/**
|
|
220
|
+
* Graphs whose settled-branch ANNOUNCEMENTS have already been made by this process.
|
|
221
|
+
*
|
|
222
|
+
* Re-entry is the point of the rescue, but only the parts that failed should repeat. Layout and
|
|
223
|
+
* the `GraphSettled` frame are idempotent facts about a finished graph, so a graph stuck in
|
|
224
|
+
* retry was re-persisting geometry and re-emitting the same frame every poll — for the whole
|
|
225
|
+
* 24h window, for as long as it kept failing.
|
|
226
|
+
*/
|
|
227
|
+
this.announcedSettlements = new Set();
|
|
228
|
+
/** Graphs already reported as settled-but-undeliverable by this instance. */
|
|
229
|
+
this.reportedUndeliverable = new Set();
|
|
230
|
+
/** Live claim heartbeats by task ID, so the drain can silence the ones it gives up waiting for. */
|
|
231
|
+
this.heartbeats = new Map();
|
|
232
|
+
/** Latched once the drain has given up waiting, so a late arrival does not re-register. */
|
|
233
|
+
this.heartbeatsPurged = false;
|
|
57
234
|
this.running = false;
|
|
58
235
|
this.pollTimer = null;
|
|
59
236
|
this.reconcileTimer = null;
|
|
237
|
+
this.unregisterKick = null;
|
|
60
238
|
/** Tasks this instance is currently executing — bounds concurrency and drives heartbeats. */
|
|
61
239
|
this.inFlight = new Set();
|
|
62
240
|
/** Guards against a slow poll overlapping the next tick. */
|
|
63
241
|
this.polling = false;
|
|
242
|
+
/**
|
|
243
|
+
* Timer-driven passes currently running — poll and reconcile alike.
|
|
244
|
+
*
|
|
245
|
+
* A counter rather than a boolean because the two timers overlap by design, and `Stop()` has to
|
|
246
|
+
* wait for BOTH. Neither pass is held by anything else: they are launched `void`-ed from
|
|
247
|
+
* `setInterval`, so without this they are unobservable from the outside and a stopped dispatcher
|
|
248
|
+
* keeps writing.
|
|
249
|
+
*/
|
|
250
|
+
this.activePasses = 0;
|
|
64
251
|
/** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
|
|
65
252
|
this.ownerByParentID = new Map();
|
|
253
|
+
/** Monotonic pass counter for `PassCompleted` frames, so a viewer can order and gap-detect ticks. */
|
|
254
|
+
this.passCounter = 0;
|
|
255
|
+
/**
|
|
256
|
+
* Debug state per graph, cached for ONE pass. `pollOnce` clears it at entry, so within a pass
|
|
257
|
+
* the claim filter and the propagation loop read the same state (loading it twice could see a
|
|
258
|
+
* pause land between them and gate half a pass), and across passes a control verb written by any
|
|
259
|
+
* instance is picked up within one poll interval.
|
|
260
|
+
*/
|
|
261
|
+
this.debugStateByGraph = new Map();
|
|
262
|
+
/**
|
|
263
|
+
* The pause state last announced per graph, so `GraphPaused`/`GraphResumed` are emitted on the
|
|
264
|
+
* TRANSITION rather than every pass — the verbs write durable state, not events, and it is this
|
|
265
|
+
* instance's job to notice the change and say so exactly once.
|
|
266
|
+
*/
|
|
267
|
+
this.announcedPaused = new Map();
|
|
268
|
+
/**
|
|
269
|
+
* The verdict last emitted per gating edge. `GateDecision` frames announce CHANGES: edges are
|
|
270
|
+
* re-resolved every pass, and an unconditional emission would repeat every few seconds for as
|
|
271
|
+
* long as the graph lives — the frame-topic version of the log flood
|
|
272
|
+
* `logUnevaluableConditionOnce` exists to prevent.
|
|
273
|
+
*
|
|
274
|
+
* Nested by graph so a settled run's entries go in one delete. Flat per-edge maps in a process
|
|
275
|
+
* that runs for weeks are a slow leak with no upper bound but the table's size.
|
|
276
|
+
*/
|
|
277
|
+
this.emittedGateVerdicts = new Map();
|
|
278
|
+
/** Last `NodeProgress` emission per task, for rate limiting chatty runners. Nested by graph. */
|
|
279
|
+
this.nodeProgressLastEmit = new Map();
|
|
66
280
|
/** Name shown in the shutdown drain log. */
|
|
67
281
|
this.ShutdownName = 'TaskGraphDispatcher';
|
|
68
282
|
this.config = { ...DEFAULT_DISPATCHER_CONFIG, ...config };
|
|
@@ -86,6 +300,37 @@ export class TaskGraphDispatcher {
|
|
|
86
300
|
LogError(`[TaskGraphDispatcher] Observer threw on ${frame.Kind} (ignored): ${e instanceof Error ? e.message : String(e)}`);
|
|
87
301
|
}
|
|
88
302
|
}
|
|
303
|
+
/**
|
|
304
|
+
* A progress sink for one task's runner, rate-limited into `NodeProgress` frames.
|
|
305
|
+
*
|
|
306
|
+
* Rate-limited HERE rather than asking every runner to be polite, for the same reason `emit`
|
|
307
|
+
* swallows observer throws in one place: a chatty runner (an agent streaming token-level
|
|
308
|
+
* updates) must not be able to flood the topic, and the limit belongs to the announcement, not
|
|
309
|
+
* the work. A 100% report always passes — the terminal update is the one a viewer must not lose.
|
|
310
|
+
*/
|
|
311
|
+
nodeProgressEmitter(graphID, ownerUserID, taskID, taskName) {
|
|
312
|
+
return (message, percent) => {
|
|
313
|
+
const now = Date.now();
|
|
314
|
+
let perTask = this.nodeProgressLastEmit.get(graphID);
|
|
315
|
+
if (!perTask) {
|
|
316
|
+
perTask = new Map();
|
|
317
|
+
this.nodeProgressLastEmit.set(graphID, perTask);
|
|
318
|
+
}
|
|
319
|
+
const last = perTask.get(taskID) ?? 0;
|
|
320
|
+
if (percent !== 100 && now - last < TaskGraphDispatcher.NODE_PROGRESS_MIN_INTERVAL_MS)
|
|
321
|
+
return;
|
|
322
|
+
perTask.set(taskID, now);
|
|
323
|
+
this.emit({
|
|
324
|
+
Kind: 'NodeProgress',
|
|
325
|
+
ParentTaskID: graphID,
|
|
326
|
+
OwnerUserID: ownerUserID,
|
|
327
|
+
TaskID: taskID,
|
|
328
|
+
TaskName: taskName,
|
|
329
|
+
ProgressMessage: message,
|
|
330
|
+
ProgressPercent: percent,
|
|
331
|
+
});
|
|
332
|
+
};
|
|
333
|
+
}
|
|
89
334
|
/**
|
|
90
335
|
* Who a graph belongs to, memoized for the process's lifetime.
|
|
91
336
|
*
|
|
@@ -104,18 +349,106 @@ export class TaskGraphDispatcher {
|
|
|
104
349
|
const cached = this.ownerByParentID.get(parentTaskID);
|
|
105
350
|
if (cached !== undefined)
|
|
106
351
|
return cached;
|
|
107
|
-
let owner = null;
|
|
108
352
|
try {
|
|
109
353
|
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
110
|
-
if (await parent.Load(parentTaskID)) {
|
|
111
|
-
|
|
354
|
+
if (!(await parent.Load(parentTaskID))) {
|
|
355
|
+
// NOT CACHED (C1). A failed load is not an answer, and caching it as one is
|
|
356
|
+
// permanent for the life of the process: the delivery filter fails closed on a null
|
|
357
|
+
// owner, so every frame for this graph reaches nobody until a restart. One
|
|
358
|
+
// transient blip, and a viewer watches a workflow that never appears to move.
|
|
359
|
+
LogError(`[TaskGraphDispatcher] Could not load graph ${parentTaskID} to resolve its owner; frames for it are unaddressed this pass.`);
|
|
360
|
+
return null;
|
|
112
361
|
}
|
|
362
|
+
const owner = this.readParentMetadata(parent).submittedByUserID ?? null;
|
|
363
|
+
// A successfully-read graph with no owner IS an answer — a scheduled or remote-triggered
|
|
364
|
+
// graph legitimately has none — so that one caches.
|
|
365
|
+
this.ownerByParentID.set(parentTaskID, owner);
|
|
366
|
+
return owner;
|
|
113
367
|
}
|
|
114
368
|
catch (e) {
|
|
115
369
|
LogError(`[TaskGraphDispatcher] Could not resolve owner for graph ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
370
|
+
return null;
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
/**
|
|
374
|
+
* A graph's debug state, cached for the current pass.
|
|
375
|
+
*
|
|
376
|
+
* Read fresh (BypassCache) because the state is written by direct `JSON_MODIFY` statements that
|
|
377
|
+
* fire no cache invalidation — the same reason every row read in this class bypasses the cache.
|
|
378
|
+
*/
|
|
379
|
+
async readDebugState(provider, parentTaskID) {
|
|
380
|
+
const cached = this.debugStateByGraph.get(parentTaskID);
|
|
381
|
+
if (cached !== undefined)
|
|
382
|
+
return cached;
|
|
383
|
+
await this.primeDebugStates(provider, [parentTaskID]);
|
|
384
|
+
return this.debugStateByGraph.get(parentTaskID) ?? {};
|
|
385
|
+
}
|
|
386
|
+
/**
|
|
387
|
+
* Reads the debug bag for a set of graphs in ONE query, priming the per-pass cache.
|
|
388
|
+
*
|
|
389
|
+
* Batched because both loops in a pass — propagation and claiming — walk every active graph, so
|
|
390
|
+
* a per-graph read made observability cost scale with the number of live workflows on a path
|
|
391
|
+
* that is meant to be flat. One `ID IN (…)` per pass costs the same whether a server is running
|
|
392
|
+
* one workflow or fifty.
|
|
393
|
+
*
|
|
394
|
+
* Already-cached graphs are skipped, so calling this from both loops is free the second time.
|
|
395
|
+
*/
|
|
396
|
+
async primeDebugStates(provider, parentTaskIDs) {
|
|
397
|
+
const missing = parentTaskIDs.filter((id) => !this.debugStateByGraph.has(id));
|
|
398
|
+
if (missing.length === 0)
|
|
399
|
+
return;
|
|
400
|
+
try {
|
|
401
|
+
const idList = missing.map((id) => `'${id}'`).join(',');
|
|
402
|
+
const rows = await RunView.FromMetadataProvider(provider).RunView({
|
|
403
|
+
EntityName: 'MJ: Tasks',
|
|
404
|
+
ExtraFilter: `ID IN (${idList})`,
|
|
405
|
+
Fields: ['ID', 'InputPayload'],
|
|
406
|
+
ResultType: 'simple',
|
|
407
|
+
// The bag is written by direct JSON_MODIFY statements, which fire no cache
|
|
408
|
+
// invalidation — a cached read here would gate on state a verb already changed.
|
|
409
|
+
BypassCache: true,
|
|
410
|
+
}, this.contextUser);
|
|
411
|
+
const byID = new Map((rows.Success ? rows.Results ?? [] : []).map((r) => [r.ID, r.InputPayload]));
|
|
412
|
+
for (const id of missing) {
|
|
413
|
+
this.debugStateByGraph.set(id, ParseTaskGraphDebugState(byID.get(id) ?? null));
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
catch (e) {
|
|
417
|
+
// "Not being debugged" is the safe reading of "could not read": gating real work on a
|
|
418
|
+
// transient read failure would turn a database hiccup into a paused workflow. Cached as
|
|
419
|
+
// empty for this pass only, so the next pass tries again.
|
|
420
|
+
LogError(`[TaskGraphDispatcher] Could not read debug state for ${missing.length} graph(s): ${e instanceof Error ? e.message : String(e)}`);
|
|
421
|
+
for (const id of missing)
|
|
422
|
+
this.debugStateByGraph.set(id, {});
|
|
116
423
|
}
|
|
117
|
-
|
|
118
|
-
|
|
424
|
+
}
|
|
425
|
+
/**
|
|
426
|
+
* Announces a graph's pause-state TRANSITION, once, whichever instance notices first in its own
|
|
427
|
+
* frame stream.
|
|
428
|
+
*
|
|
429
|
+
* Per-instance dedup rather than a CAS: frames are advisory commentary, and a viewer receiving
|
|
430
|
+
* the transition from two instances is a duplicate line, not a duplicate execution — the price
|
|
431
|
+
* of a cross-instance guard here would be a write on every pass for a purely cosmetic guarantee.
|
|
432
|
+
*/
|
|
433
|
+
async announcePauseTransition(provider, parentTaskID, debug) {
|
|
434
|
+
const paused = debug.paused === true;
|
|
435
|
+
const previous = this.announcedPaused.get(parentTaskID);
|
|
436
|
+
if (previous === paused)
|
|
437
|
+
return;
|
|
438
|
+
this.announcedPaused.set(parentTaskID, paused);
|
|
439
|
+
// First sighting of an unpaused graph needs no announcement — "running" is the default a
|
|
440
|
+
// viewer already assumes; only a transition is information.
|
|
441
|
+
if (previous === undefined && !paused)
|
|
442
|
+
return;
|
|
443
|
+
this.emit({
|
|
444
|
+
Kind: paused ? 'GraphPaused' : 'GraphResumed',
|
|
445
|
+
ParentTaskID: parentTaskID,
|
|
446
|
+
OwnerUserID: await this.resolveOwner(provider, parentTaskID),
|
|
447
|
+
TaskID: debug.pausedAtTaskID ?? undefined,
|
|
448
|
+
Reason: paused
|
|
449
|
+
? (debug.pausedReason === 'breakpoint' ? 'breakpoint' : 'paused by user')
|
|
450
|
+
: 'resumed',
|
|
451
|
+
});
|
|
119
452
|
}
|
|
120
453
|
/**
|
|
121
454
|
* Begins dispatching.
|
|
@@ -134,11 +467,72 @@ export class TaskGraphDispatcher {
|
|
|
134
467
|
ShutdownRegistry.Instance.Register(this);
|
|
135
468
|
LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
|
|
136
469
|
await this.Reconcile();
|
|
470
|
+
// One wide pass over graphs that reached terminal without settling, mirroring what claim
|
|
471
|
+
// reconciliation above already does for tasks. The realistic producer of a >24h-stale
|
|
472
|
+
// unsettled graph is this process having been DOWN — an outage, a long deploy — which the
|
|
473
|
+
// steady-state window cannot see and which would otherwise leave those runs parked forever.
|
|
474
|
+
// Counted as a pass (R2-13). It settles graphs, delivers continuations and can start fresh
|
|
475
|
+
// reinvoke turns, and it runs AFTER this instance registers for shutdown — so a `Stop()`
|
|
476
|
+
// landing during it used to return immediately while the sweep carried on doing all of that
|
|
477
|
+
// against a host that believed the dispatcher had stopped.
|
|
478
|
+
this.activePasses++;
|
|
479
|
+
try {
|
|
480
|
+
await this.sweepUnsettledGraphs(UNSETTLED_STARTUP_WINDOW_HOURS);
|
|
481
|
+
}
|
|
482
|
+
finally {
|
|
483
|
+
this.activePasses--;
|
|
484
|
+
}
|
|
485
|
+
// A `Stop()` LANDING DURING THE BOOT AWAITS MUST NOT BE UNDONE HERE (R3-4).
|
|
486
|
+
//
|
|
487
|
+
// Everything above this line is awaited — reconciliation and the counted startup sweep,
|
|
488
|
+
// which R2-13's own fix makes `Stop()` wait out. So a host that shuts down during boot
|
|
489
|
+
// drains correctly, logs "Stopped.", and returns with both timer fields null — and then
|
|
490
|
+
// this continuation ran anyway and installed both timers on the stopped instance.
|
|
491
|
+
//
|
|
492
|
+
// `pollOnce` was inert (its own `running` guard), but `Reconcile` had no such guard: it
|
|
493
|
+
// minted a provider and ran `ReleaseExpiredClaims` — a real UPDATE returning tasks to
|
|
494
|
+
// Pending — every two minutes forever, against a pool the host may have torn down. Nothing
|
|
495
|
+
// would ever call `Stop()` again, since `ShutdownRegistry.ShutdownAll` clears its items
|
|
496
|
+
// after one pass, and the intervals pinned the event loop so the process could not exit.
|
|
497
|
+
if (!this.running) {
|
|
498
|
+
LogStatus(`[TaskGraphDispatcher] Stopped during startup; not installing timers.`);
|
|
499
|
+
return;
|
|
500
|
+
}
|
|
137
501
|
this.pollTimer = setInterval(() => { void this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
|
|
138
|
-
this.
|
|
502
|
+
this.unregisterKick = RegisterTaskGraphKick(() => { this.Kick(); });
|
|
503
|
+
// Do not wait a full interval for work that already exists (or is about to be submitted).
|
|
504
|
+
this.Kick();
|
|
505
|
+
this.reconcileTimer = setInterval(
|
|
506
|
+
// Guarded HERE rather than inside `Reconcile` (R3-4). The defect is a stopped
|
|
507
|
+
// instance's TIMER executing `ReleaseExpiredClaims` — a real UPDATE — forever; the
|
|
508
|
+
// public method itself stays callable, because reconciling on demand before starting is
|
|
509
|
+
// a legitimate use (IT74's crash-recovery check does exactly that) and a guard there
|
|
510
|
+
// would silently no-op it, which is the class of failure this whole effort is about.
|
|
511
|
+
() => { if (this.running)
|
|
512
|
+
void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
|
|
513
|
+
// Belt and suspenders: `Stop()` remains the real teardown, but an un-`unref`'d interval
|
|
514
|
+
// keeps the event loop alive on its own, so a leaked instance can prevent process exit.
|
|
515
|
+
this.pollTimer.unref?.();
|
|
516
|
+
this.reconcileTimer.unref?.();
|
|
139
517
|
}
|
|
140
518
|
/**
|
|
141
|
-
* Stops accepting new work and waits for
|
|
519
|
+
* Stops accepting new work and waits for everything already started to finish.
|
|
520
|
+
*
|
|
521
|
+
* **"Everything" includes the timer passes, and that is the fix.** This waited only on
|
|
522
|
+
* `inFlight` — the task executions — while a poll pass is a `void`-ed promise nothing held. So
|
|
523
|
+
* `Stop()` returned while a pass was mid-flight, and that pass went on to settle graphs, emit
|
|
524
|
+
* lifecycle frames and CLAIM NEW TASKS afterwards. Three consequences, all of them quiet:
|
|
525
|
+
*
|
|
526
|
+
* - a `GraphSettled` frame arrived after every subscriber had gone, so the settlement was
|
|
527
|
+
* invisible to exactly the viewer watching for it;
|
|
528
|
+
* - a process shutting down claimed work it was about to abandon, leaving claims to expire —
|
|
529
|
+
* the orphaned-claim state reconciliation exists to clean up, manufactured by the shutdown;
|
|
530
|
+
* - the host reused the connection the moment `Stop()` resolved, and the still-running pass's
|
|
531
|
+
* statements collided with it (`Requests can only be made in the LoggedIn state`).
|
|
532
|
+
*
|
|
533
|
+
* A pass is bookkeeping for work that already happened, so it is DRAINED rather than cancelled:
|
|
534
|
+
* abandoning one halfway is the crash window the unsettled sweep exists to rescue, and choosing
|
|
535
|
+
* to open it on every clean shutdown would be perverse.
|
|
142
536
|
*
|
|
143
537
|
* Deliberately does NOT release claims on the way out: an abandoned claim expires on its own,
|
|
144
538
|
* and releasing eagerly would hand a still-running task to another instance mid-execution.
|
|
@@ -146,6 +540,8 @@ export class TaskGraphDispatcher {
|
|
|
146
540
|
*/
|
|
147
541
|
async Stop() {
|
|
148
542
|
this.running = false;
|
|
543
|
+
this.unregisterKick?.();
|
|
544
|
+
this.unregisterKick = null;
|
|
149
545
|
if (this.pollTimer) {
|
|
150
546
|
clearInterval(this.pollTimer);
|
|
151
547
|
this.pollTimer = null;
|
|
@@ -154,12 +550,35 @@ export class TaskGraphDispatcher {
|
|
|
154
550
|
clearInterval(this.reconcileTimer);
|
|
155
551
|
this.reconcileTimer = null;
|
|
156
552
|
}
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
553
|
+
// Short poll interval: a pass is usually milliseconds from done, and the old 250ms granularity
|
|
554
|
+
// was most of the cost of stopping a dispatcher that had nothing left to do.
|
|
555
|
+
const deadline = Date.now() + STOP_DRAIN_TIMEOUT_MS;
|
|
556
|
+
while ((this.activePasses > 0 || this.inFlight.size > 0) && Date.now() < deadline) {
|
|
557
|
+
await new Promise((r) => setTimeout(r, 25));
|
|
160
558
|
}
|
|
161
559
|
if (this.inFlight.size > 0) {
|
|
162
|
-
|
|
560
|
+
// The promise in this message was FALSE while the process lived (R2-13): each in-flight
|
|
561
|
+
// task heartbeats its own claim on its own timer, so an over-drain task renewed its lease
|
|
562
|
+
// indefinitely and the claim never expired — reconciliation could not reclaim the work,
|
|
563
|
+
// and the host's shutdown was waiting on something that had stopped being reclaimable.
|
|
564
|
+
// Stopping the heartbeats makes the sentence true. The task itself keeps running; its
|
|
565
|
+
// completion write is guarded on still owning the claim, so if another instance reclaims
|
|
566
|
+
// the task in the meantime, the abandoned executor's result is refused rather than raced.
|
|
567
|
+
// Latched, because the purge RACES the registration it is purging (C5). A task stalled
|
|
568
|
+
// in its two pre-heartbeat awaits — creating a provider, loading the row — registers
|
|
569
|
+
// AFTER this line and would renew its claim for the rest of the process's life, which is
|
|
570
|
+
// exactly the state the drain timeout means to end, under exactly the database duress
|
|
571
|
+
// that causes drain timeouts in the first place.
|
|
572
|
+
this.heartbeatsPurged = true;
|
|
573
|
+
for (const stop of this.heartbeats.values())
|
|
574
|
+
clearInterval(stop);
|
|
575
|
+
this.heartbeats.clear();
|
|
576
|
+
LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will now expire.`);
|
|
577
|
+
}
|
|
578
|
+
if (this.activePasses > 0) {
|
|
579
|
+
// Loud, because from here on this instance writes to a database the host believes it has
|
|
580
|
+
// finished with — the precise shape that produced connection-state errors downstream.
|
|
581
|
+
LogError(`[TaskGraphDispatcher] Stopped with ${this.activePasses} pass(es) still running; their writes may land after shutdown.`);
|
|
163
582
|
}
|
|
164
583
|
LogStatus(`[TaskGraphDispatcher] Stopped.`);
|
|
165
584
|
}
|
|
@@ -175,6 +594,51 @@ export class TaskGraphDispatcher {
|
|
|
175
594
|
* that shape indicates tampering or a bug and Record Changes already carries the audit trail.
|
|
176
595
|
*/
|
|
177
596
|
async Reconcile() {
|
|
597
|
+
this.activePasses++;
|
|
598
|
+
try {
|
|
599
|
+
await this.reconcileOnce();
|
|
600
|
+
}
|
|
601
|
+
finally {
|
|
602
|
+
this.activePasses--;
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
/**
|
|
606
|
+
* Announces expired claims the sweep just released, so a viewer watching the graph sees "the
|
|
607
|
+
* step's worker vanished and the engine requeued it" as it happens.
|
|
608
|
+
*
|
|
609
|
+
* Best-effort by contract: the release already succeeded and is the durable truth; a frame that
|
|
610
|
+
* cannot be addressed (row unloadable, no parent) is dropped, never retried.
|
|
611
|
+
*/
|
|
612
|
+
async announceReclaims(provider, released) {
|
|
613
|
+
if (!this.observer || released.length === 0)
|
|
614
|
+
return;
|
|
615
|
+
try {
|
|
616
|
+
const idList = released.map((r) => `'${r.TaskID}'`).join(',');
|
|
617
|
+
const rows = await RunView.FromMetadataProvider(provider).RunView({
|
|
618
|
+
EntityName: 'MJ: Tasks',
|
|
619
|
+
ExtraFilter: `ID IN (${idList})`,
|
|
620
|
+
Fields: ['ID', 'ParentID', 'Name'],
|
|
621
|
+
ResultType: 'simple',
|
|
622
|
+
BypassCache: true,
|
|
623
|
+
}, this.contextUser);
|
|
624
|
+
for (const row of (rows.Success ? rows.Results : []) ?? []) {
|
|
625
|
+
const graphID = row.ParentID ?? row.ID;
|
|
626
|
+
this.emit({
|
|
627
|
+
Kind: 'ClaimChanged',
|
|
628
|
+
ParentTaskID: graphID,
|
|
629
|
+
OwnerUserID: await this.resolveOwner(provider, graphID),
|
|
630
|
+
TaskID: row.ID,
|
|
631
|
+
TaskName: row.Name,
|
|
632
|
+
ClaimEvent: 'reclaimed',
|
|
633
|
+
});
|
|
634
|
+
}
|
|
635
|
+
}
|
|
636
|
+
catch (e) {
|
|
637
|
+
LogError(`[TaskGraphDispatcher] Could not announce reclaimed task(s) (ignored): ${e instanceof Error ? e.message : String(e)}`);
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
/** The reconciliation body. Wrapped by {@link Reconcile} so `Stop()` can drain it. */
|
|
641
|
+
async reconcileOnce() {
|
|
178
642
|
let provider = null;
|
|
179
643
|
try {
|
|
180
644
|
provider = await this.providerFactory.CreateProvider();
|
|
@@ -184,11 +648,21 @@ export class TaskGraphDispatcher {
|
|
|
184
648
|
LogStatus(`[TaskGraphDispatcher] Reconciliation: ${released.length} expired claim(s) released, ` +
|
|
185
649
|
`${orphaned.length} orphaned task(s) reported.`);
|
|
186
650
|
}
|
|
651
|
+
await this.announceReclaims(provider, released);
|
|
187
652
|
}
|
|
188
653
|
catch (e) {
|
|
189
654
|
LogError(`[TaskGraphDispatcher] Reconciliation failed: ${e instanceof Error ? e.message : String(e)}`);
|
|
190
655
|
}
|
|
191
656
|
}
|
|
657
|
+
/**
|
|
658
|
+
* Run a pass now instead of waiting for the next poll tick.
|
|
659
|
+
*
|
|
660
|
+
* Submit calls this (via {@link KickTaskGraphDispatchers}) so a just-written graph is claimed
|
|
661
|
+
* in milliseconds rather than up to {@link TaskGraphDispatcherConfig.PollIntervalSeconds}.
|
|
662
|
+
*/
|
|
663
|
+
Kick() {
|
|
664
|
+
void this.pollOnce();
|
|
665
|
+
}
|
|
192
666
|
/**
|
|
193
667
|
* One dispatch pass: find claimable work, claim what fits under the concurrency cap, execute.
|
|
194
668
|
*
|
|
@@ -198,33 +672,86 @@ export class TaskGraphDispatcher {
|
|
|
198
672
|
async pollOnce() {
|
|
199
673
|
if (!this.running || this.polling)
|
|
200
674
|
return;
|
|
201
|
-
const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
|
|
202
|
-
if (capacity <= 0)
|
|
203
|
-
return;
|
|
204
675
|
this.polling = true;
|
|
676
|
+
this.activePasses++;
|
|
677
|
+
// One pass, one read of each graph's debug state — see `readDebugState`.
|
|
678
|
+
this.debugStateByGraph.clear();
|
|
679
|
+
const passNumber = ++this.passCounter;
|
|
205
680
|
try {
|
|
206
681
|
const provider = await this.providerFactory.CreateProvider();
|
|
207
|
-
//
|
|
208
|
-
//
|
|
682
|
+
// `running` is re-read after every await from here on. The entry check above only proves
|
|
683
|
+
// the dispatcher was live when the tick fired; each await is a point where `Stop` can
|
|
684
|
+
// land, and a stopped instance must neither mutate graph state nor take new work. Left
|
|
685
|
+
// unchecked, a stopped dispatcher goes on to roll up graphs (emitting GraphSettled to an
|
|
686
|
+
// observer nobody is listening to any more) and to claim tasks it will never run — which
|
|
687
|
+
// then sit claimed until their lease expires.
|
|
688
|
+
if (!this.running)
|
|
689
|
+
return;
|
|
690
|
+
// SETTLEMENT IS NOT GATED ON CAPACITY (R2-11).
|
|
691
|
+
//
|
|
692
|
+
// This used to return at `capacity <= 0` before reaching the rollup, so a handful of
|
|
693
|
+
// wedged long-running tasks froze EVERYTHING for the whole instance: no settlement, no
|
|
694
|
+
// skip or block propagation, no human-task settlement, no continuation delivery — for
|
|
695
|
+
// graphs that had nothing to do with the tasks holding the slots. A per-task hang is an
|
|
696
|
+
// accepted limitation; "one hung task stops every workflow on this host" is not, and the
|
|
697
|
+
// two were the same line of code.
|
|
698
|
+
//
|
|
699
|
+
// Only CLAIMING consumes capacity, because only claiming starts work.
|
|
209
700
|
await this.propagateAndRollup(provider);
|
|
210
|
-
const
|
|
701
|
+
const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
|
|
702
|
+
if (capacity <= 0)
|
|
703
|
+
return;
|
|
704
|
+
// The rollup above can take seconds, and `Stop()` may have been called during it. Claiming
|
|
705
|
+
// now would start work the process has already decided to abandon — the claim then sits
|
|
706
|
+
// until its TTL expires and another instance reclaims it. Settling first and checking
|
|
707
|
+
// here is the right order: bookkeeping for finished work always completes, new work never
|
|
708
|
+
// starts after the decision to stop.
|
|
709
|
+
if (!this.running)
|
|
710
|
+
return;
|
|
711
|
+
const { tasks: candidates, stats } = await this.findClaimableTasks(provider, capacity);
|
|
712
|
+
const claimedByGraph = new Map();
|
|
211
713
|
for (const task of candidates) {
|
|
714
|
+
// Re-checked EVERY iteration, not once before the loop (R2-13). Claiming is itself
|
|
715
|
+
// awaited, so a multi-task wave can straddle a `Stop`; and `findClaimableTasks` loads
|
|
716
|
+
// and resolves every active graph, so the scan before this loop can run for seconds.
|
|
717
|
+
// Unchecked, a shutting-down process takes ownership of work it is about to abandon,
|
|
718
|
+
// manufacturing the orphaned claims reconciliation exists to clean up.
|
|
719
|
+
if (!this.running)
|
|
720
|
+
break;
|
|
212
721
|
if (this.inFlight.size >= this.config.MaxConcurrentTasks)
|
|
213
722
|
break;
|
|
214
723
|
if (!(await this.claims.TryClaim(provider, task.ID, this.contextUser))) {
|
|
215
724
|
// Another instance won the race, or the task is no longer Pending. Normal.
|
|
216
725
|
continue;
|
|
217
726
|
}
|
|
727
|
+
const graphID = task.ParentID ?? task.ID;
|
|
728
|
+
claimedByGraph.set(graphID, (claimedByGraph.get(graphID) ?? 0) + 1);
|
|
218
729
|
this.inFlight.add(task.ID);
|
|
219
730
|
// Intentionally not awaited — the poll loop must keep dispatching while this runs.
|
|
220
731
|
void this.executeClaimed(task.ID).finally(() => this.inFlight.delete(task.ID));
|
|
221
732
|
}
|
|
733
|
+
// The engine's heartbeat, per watched graph: what was ready, what was held, what this
|
|
734
|
+
// instance took. A stuck run is a strip of these ticking with nothing moving, which is
|
|
735
|
+
// the honest visual of a stall — and the reason this frame exists.
|
|
736
|
+
for (const [graphID, s] of stats) {
|
|
737
|
+
this.emit({
|
|
738
|
+
Kind: 'PassCompleted',
|
|
739
|
+
ParentTaskID: graphID,
|
|
740
|
+
OwnerUserID: await this.resolveOwner(provider, graphID),
|
|
741
|
+
PassNumber: passNumber,
|
|
742
|
+
EligibleCount: s.eligible,
|
|
743
|
+
HeldCount: s.held,
|
|
744
|
+
ClaimedCount: claimedByGraph.get(graphID) ?? 0,
|
|
745
|
+
InstanceInFlightCount: this.inFlight.size,
|
|
746
|
+
});
|
|
747
|
+
}
|
|
222
748
|
}
|
|
223
749
|
catch (e) {
|
|
224
750
|
LogError(`[TaskGraphDispatcher] Poll failed: ${e instanceof Error ? e.message : String(e)}`);
|
|
225
751
|
}
|
|
226
752
|
finally {
|
|
227
753
|
this.polling = false;
|
|
754
|
+
this.activePasses--;
|
|
228
755
|
}
|
|
229
756
|
}
|
|
230
757
|
/**
|
|
@@ -242,19 +769,44 @@ export class TaskGraphDispatcher {
|
|
|
242
769
|
LogError(`[TaskGraphDispatcher] Claimed task ${taskID} could not be loaded.`);
|
|
243
770
|
return;
|
|
244
771
|
}
|
|
772
|
+
// Emitted after the claim is held, not before: a frame saying "started" for work another
|
|
773
|
+
// instance actually took would be a lie a viewer cannot detect.
|
|
774
|
+
const graphID = task.ParentID ?? taskID;
|
|
775
|
+
const ownerUserID = await this.resolveOwner(provider, graphID);
|
|
245
776
|
heartbeat = setInterval(() => {
|
|
246
777
|
void this.claims.Heartbeat(provider, taskID, this.contextUser).then((ok) => {
|
|
247
778
|
if (!ok) {
|
|
248
779
|
// Lost ownership — reconciliation reclaimed it, or a human intervened.
|
|
249
780
|
LogError(`[TaskGraphDispatcher] Lost claim on task ${taskID} while executing; another instance may take it over.`);
|
|
781
|
+
// Announced so a viewer sees "this step's worker lost its lease" the moment
|
|
782
|
+
// it happens instead of discovering it in a forensic query later — the R2-1
|
|
783
|
+
// wedge class, made visible.
|
|
784
|
+
this.emit({
|
|
785
|
+
Kind: 'ClaimChanged', ParentTaskID: graphID, OwnerUserID: ownerUserID,
|
|
786
|
+
TaskID: taskID, TaskName: task.Name,
|
|
787
|
+
ClaimEvent: 'heartbeat-lost', ClaimedBy: this.config.InstanceID,
|
|
788
|
+
});
|
|
250
789
|
}
|
|
251
790
|
});
|
|
252
791
|
}, this.config.HeartbeatIntervalSeconds * 1000);
|
|
253
|
-
//
|
|
254
|
-
//
|
|
255
|
-
|
|
256
|
-
|
|
792
|
+
// Registered so `Stop()` can reach it — unless the drain has already given up, in which
|
|
793
|
+
// case this task arrived too late to be waited for and must not renew its lease (C5).
|
|
794
|
+
// Its completion write stays guarded, so if another instance reclaims the task in the
|
|
795
|
+
// meantime this executor's result is refused rather than raced.
|
|
796
|
+
if (this.heartbeatsPurged || !this.running) {
|
|
797
|
+
clearInterval(heartbeat);
|
|
798
|
+
heartbeat = null;
|
|
799
|
+
}
|
|
800
|
+
else {
|
|
801
|
+
this.heartbeats.set(taskID, heartbeat);
|
|
802
|
+
}
|
|
257
803
|
this.emit({ Kind: 'TaskStarted', ParentTaskID: graphID, OwnerUserID: ownerUserID, TaskID: taskID, TaskName: task.Name, Status: 'In Progress' });
|
|
804
|
+
this.emit({
|
|
805
|
+
Kind: 'ClaimChanged', ParentTaskID: graphID, OwnerUserID: ownerUserID,
|
|
806
|
+
TaskID: taskID, TaskName: task.Name,
|
|
807
|
+
ClaimEvent: 'claimed', ClaimedBy: this.config.InstanceID,
|
|
808
|
+
ClaimExpiresAt: new Date(Date.now() + this.config.ClaimTTLSeconds * 1000).toISOString(),
|
|
809
|
+
});
|
|
258
810
|
const dependencyOutputs = await this.loadDependencyOutputs(provider, taskID);
|
|
259
811
|
let inputPayload = null;
|
|
260
812
|
if (task.InputPayload) {
|
|
@@ -265,19 +817,24 @@ export class TaskGraphDispatcher {
|
|
|
265
817
|
LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
|
|
266
818
|
}
|
|
267
819
|
}
|
|
268
|
-
const
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
820
|
+
const onProgress = this.nodeProgressEmitter(graphID, ownerUserID, taskID, task.Name);
|
|
821
|
+
const result = await this.runTaskBody(task, provider, inputPayload, dependencyOutputs, onProgress);
|
|
822
|
+
// ONLY THE CONFIRMED OWNER MUTATES THE GRAPH (R2-10).
|
|
823
|
+
//
|
|
824
|
+
// The early-finish skips used to run BEFORE this, so a lapsed claim produced the worst
|
|
825
|
+
// possible pair: the siblings were terminally Skipped and satisfying dependents, while
|
|
826
|
+
// the completion was refused and the task re-ran on another instance — where it might
|
|
827
|
+
// not end early at all. The graph would then be missing steps nobody decided to skip.
|
|
828
|
+
//
|
|
829
|
+
// Recording first costs a poll: the skips now land after the completion, so a rollup
|
|
830
|
+
// that lands in between sees work still Pending and settles one pass later. That is a
|
|
831
|
+
// delay; the other order was a wrong graph.
|
|
276
832
|
const recorded = await this.claims.CompleteClaimed(provider, taskID, {
|
|
277
833
|
Status: result.Success ? 'Complete' : 'Failed',
|
|
278
834
|
OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
|
|
279
835
|
ErrorMessage: result.ErrorMessage ?? null,
|
|
280
836
|
AgentRunID: result.AgentRunID ?? null,
|
|
837
|
+
Configuration: this.configurationWithRuntime(task, result.PromptRunID, result.ActionLogID, result.Iterations, result.PayloadAtStart),
|
|
281
838
|
}, this.contextUser);
|
|
282
839
|
if (!recorded) {
|
|
283
840
|
// The guarded write refused: the row changed underneath us (cancelled, reassigned,
|
|
@@ -297,6 +854,13 @@ export class TaskGraphDispatcher {
|
|
|
297
854
|
Status: result.Success ? 'Complete' : 'Failed',
|
|
298
855
|
ErrorMessage: result.Success ? undefined : (result.ErrorMessage ?? undefined),
|
|
299
856
|
});
|
|
857
|
+
// A prompt can end the workflow early and say why — honoured only now that this
|
|
858
|
+
// instance is the confirmed owner of the outcome. The remaining tasks are Skipped
|
|
859
|
+
// here so the graph settles Complete rather than looking abandoned with work left
|
|
860
|
+
// Pending; a rollup that lands between the two simply settles one pass later.
|
|
861
|
+
if (result.ChatMessage) {
|
|
862
|
+
await this.endGraphEarly(provider, task, result.ChatMessage);
|
|
863
|
+
}
|
|
300
864
|
}
|
|
301
865
|
}
|
|
302
866
|
catch (e) {
|
|
@@ -310,6 +874,7 @@ export class TaskGraphDispatcher {
|
|
|
310
874
|
finally {
|
|
311
875
|
if (heartbeat)
|
|
312
876
|
clearInterval(heartbeat);
|
|
877
|
+
this.heartbeats.delete(taskID);
|
|
313
878
|
}
|
|
314
879
|
}
|
|
315
880
|
/**
|
|
@@ -318,12 +883,80 @@ export class TaskGraphDispatcher {
|
|
|
318
883
|
* All four decisions — what is eligible, what must block, what the parent status is, whether the
|
|
319
884
|
* graph is wedged — are delegated to the pure algorithms, unchanged from Phase 1.
|
|
320
885
|
*/
|
|
321
|
-
async propagateAndRollup(provider) {
|
|
322
|
-
|
|
323
|
-
|
|
886
|
+
async propagateAndRollup(provider, graphIDs) {
|
|
887
|
+
const graphs = graphIDs ?? await this.findActiveGraphIDs(provider);
|
|
888
|
+
// One read for every graph this pass touches, rather than one per graph — see primeDebugStates.
|
|
889
|
+
await this.primeDebugStates(provider, graphs);
|
|
890
|
+
for (const parentID of graphs) {
|
|
891
|
+
// Human steps settle BEFORE the graph state is read, so an answer given since the last
|
|
892
|
+
// poll is already reflected when eligibility and rollup are computed. Doing it after
|
|
893
|
+
// would delay every dependent branch by a full poll interval for no reason — and on a
|
|
894
|
+
// graph whose only remaining work is downstream of a person, that is the difference
|
|
895
|
+
// between "answered and moving" and "answered and apparently still stuck".
|
|
896
|
+
await this.expireOverdueRequests(provider, parentID);
|
|
897
|
+
await this.settleAnsweredHumanTasks(provider, parentID);
|
|
898
|
+
await this.reconcileWaitingHumanTasks(provider, parentID);
|
|
899
|
+
// Edge overrides apply to propagation exactly as they apply to claiming — a branch the
|
|
900
|
+
// operator answered 'false' must cascade its skips here, not merely stop being claimed.
|
|
901
|
+
const debug = await this.readDebugState(provider, parentID);
|
|
902
|
+
const graph = await this.loadGraphState(provider, parentID, debug);
|
|
324
903
|
if (graph.nodes.length === 0)
|
|
325
904
|
continue;
|
|
326
|
-
|
|
905
|
+
// SKIPS FIRST — before blocking, before eligibility. A task whose gating predecessors
|
|
906
|
+
// are all Skipped is simultaneously "eligible" (Skipped satisfies a prerequisite) and
|
|
907
|
+
// "to be skipped"; deciding eligibility first would dispatch the branch nobody took.
|
|
908
|
+
//
|
|
909
|
+
// `unreachableTaskIDs` seeds this too, and that is a correction (R6). A target whose only
|
|
910
|
+
// route in was an ordinary conditional edge that evaluated DEFINITELY FALSE is a branch
|
|
911
|
+
// that was not taken — semantically identical to an XOR loser — yet it used to settle
|
|
912
|
+
// `Blocked`. That made `Blocked` mean two unrelated things: "the workflow chose another
|
|
913
|
+
// route" and "something upstream broke". A reader cannot tell those apart, so every
|
|
914
|
+
// conditional workflow looked half-failed and people went hunting for bugs that did not
|
|
915
|
+
// exist. `Blocked` is now reserved for FAILURE-driven unsatisfiability.
|
|
916
|
+
// Computed once in `loadGraphState` so the claim filter sees the same set this pass is
|
|
917
|
+
// about to write — see R2-14 there.
|
|
918
|
+
const toSkip = graph.cascadeSkipTaskIDs;
|
|
919
|
+
const skippedByRoute = [];
|
|
920
|
+
for (const taskID of toSkip) {
|
|
921
|
+
const entity = graph.entityById.get(taskID);
|
|
922
|
+
if (!entity || entity.Status !== 'Pending')
|
|
923
|
+
continue;
|
|
924
|
+
// The in-memory check above is a cheap pre-filter; the guard that matters is IN the
|
|
925
|
+
// statement (R3-1's audit item). This snapshot was loaded at the top of the pass and
|
|
926
|
+
// a task can be claimed and started before its skip write lands — R2-14 closed that
|
|
927
|
+
// window for the claim filter, and this closes it for the write itself.
|
|
928
|
+
const skipTypeID = await this.workflowTaskTypeID(provider);
|
|
929
|
+
if (skipTypeID && await this.claims.TrySkipPending(provider, taskID, skipTypeID, this.contextUser)) {
|
|
930
|
+
entity.Status = 'Skipped';
|
|
931
|
+
skippedByRoute.push(taskID);
|
|
932
|
+
LogStatus(`[TaskGraphDispatcher] Skipped '${entity.Name}' (${taskID}) — another branch was taken.`);
|
|
933
|
+
// Announced separately from TaskBlocked because it means something different to
|
|
934
|
+
// a viewer: nothing went wrong, this route simply was not the one chosen.
|
|
935
|
+
this.emit({
|
|
936
|
+
Kind: 'TaskSkipped',
|
|
937
|
+
ParentTaskID: parentID,
|
|
938
|
+
OwnerUserID: await this.resolveOwner(provider, parentID),
|
|
939
|
+
TaskID: taskID,
|
|
940
|
+
TaskName: entity.Name,
|
|
941
|
+
Status: 'Skipped',
|
|
942
|
+
});
|
|
943
|
+
// Keep the in-memory graph consistent so the blocking pass below and the rollup
|
|
944
|
+
// both see the skip rather than a stale Pending.
|
|
945
|
+
const node = graph.nodes.find((n) => n.id === taskID);
|
|
946
|
+
if (node)
|
|
947
|
+
node.status = 'Skipped';
|
|
948
|
+
}
|
|
949
|
+
}
|
|
950
|
+
// A human step reached by a route the workflow did not take has the same zombie request
|
|
951
|
+
// as one skipped by an early finish (R2-10): notified, `Requested` forever, and invisible
|
|
952
|
+
// to the settle and expiry sweeps because they filter on Pending tasks and this one is
|
|
953
|
+
// not Pending any more. Same treatment, different reason.
|
|
954
|
+
await this.withdrawOpenRequests(provider, skippedByRoute, 'The workflow took a different route, so this step is no longer needed.');
|
|
955
|
+
// Only failure-driven unsatisfiability reaches here now; not-taken branches were skipped
|
|
956
|
+
// above. A task already Skipped is left alone rather than overwritten — the two passes
|
|
957
|
+
// must not fight over the same row.
|
|
958
|
+
const toBlock = [...ComputeTasksToBlock(graph.nodes, graph.edges, graph.handledFailureIDs)]
|
|
959
|
+
.filter((id) => !toSkip.has(id));
|
|
327
960
|
for (const taskID of toBlock) {
|
|
328
961
|
const entity = graph.entityById.get(taskID);
|
|
329
962
|
if (!entity)
|
|
@@ -343,42 +976,487 @@ export class TaskGraphDispatcher {
|
|
|
343
976
|
});
|
|
344
977
|
}
|
|
345
978
|
}
|
|
346
|
-
|
|
979
|
+
// Holds are passed in, or the detector reports a held graph as healthy: a held target's
|
|
980
|
+
// gating edge is still live and its origin Complete, so ComputeEligibleTasks counts it
|
|
981
|
+
// as eligible and "something is eligible" reads as "not stalled". A graph waiting
|
|
982
|
+
// forever on a broken condition then produced no diagnostics at all.
|
|
983
|
+
if (IsGraphStalled(graph.nodes, graph.edges, graph.holdTaskIDs)) {
|
|
347
984
|
LogError(`[TaskGraphDispatcher] Graph ${parentID} is stalled: pending work with no satisfiable path.`);
|
|
348
985
|
}
|
|
349
|
-
const fresh = await this.loadGraphState(provider, parentID);
|
|
986
|
+
const fresh = await this.loadGraphState(provider, parentID, debug);
|
|
350
987
|
// ComputeParentRollup treats an empty child set as Complete-and-terminal, which is right
|
|
351
988
|
// for a graph that genuinely has no children and catastrophic for one whose reload came
|
|
352
989
|
// back empty transiently — it would mark live work finished and fire its continuation.
|
|
353
990
|
// The outer guard covered the first load only.
|
|
354
991
|
if (fresh.nodes.length === 0)
|
|
355
992
|
continue;
|
|
356
|
-
const rollup = ComputeParentRollup(fresh.nodes);
|
|
993
|
+
const rollup = ComputeParentRollup(fresh.nodes, fresh.handledFailureIDs);
|
|
357
994
|
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
358
995
|
if (!(await parent.Load(parentID)))
|
|
359
996
|
continue;
|
|
360
|
-
|
|
997
|
+
// A graph starts when its first step does.
|
|
998
|
+
//
|
|
999
|
+
// `StartedAt` is stamped by the CLAIM, and a parent is never claimed — it is a container,
|
|
1000
|
+
// not a unit of work — so the graph row carried no start time even after it completed.
|
|
1001
|
+
// A settled workflow therefore reported a CompletedAt with no beginning: it sorted as
|
|
1002
|
+
// "not started" in the run tree, showed no timestamp, and no duration could be computed
|
|
1003
|
+
// for the thing whose duration people actually ask about.
|
|
1004
|
+
//
|
|
1005
|
+
// Taken from the earliest child rather than from the clock, because that is when work
|
|
1006
|
+
// genuinely began — a graph can sit Pending for a long time between submission (already
|
|
1007
|
+
// recorded as CreatedAt) and a dispatcher picking up its first task.
|
|
1008
|
+
// Column-scoped for the same reason the terminal write is: a full-row save here would
|
|
1009
|
+
// carry this instance's `InputPayload` snapshot and could erase a continuation marker
|
|
1010
|
+
// another instance had just claimed. Guarded on `StartedAt IS NULL`, so calling it on
|
|
1011
|
+
// every pass is free.
|
|
1012
|
+
const earliestChildStart = this.earliestStart(fresh.entityById);
|
|
1013
|
+
if (parent.StartedAt == null && earliestChildStart != null) {
|
|
1014
|
+
await this.claims.TryStampParentStart(provider, parentID, earliestChildStart, this.contextUser);
|
|
1015
|
+
parent.StartedAt = earliestChildStart;
|
|
1016
|
+
}
|
|
1017
|
+
// THE TERMINAL WRITE IS GUARDED AND COLUMN-SCOPED, not a full-row save.
|
|
1018
|
+
//
|
|
1019
|
+
// `GenerateSaveSQL` sends every updateable column on every save, so a full-row save
|
|
1020
|
+
// carries the whole in-memory snapshot — including `InputPayload`, where the continuation
|
|
1021
|
+
// marker lives. Two instances both compute the terminal rollup; if one claims the marker
|
|
1022
|
+
// and the other then saves its pre-marker snapshot, the marker is ERASED and the
|
|
1023
|
+
// settlement delivers twice. For `reinvoke` that is a second billed agent turn.
|
|
1024
|
+
//
|
|
1025
|
+
// Guarding on "not already terminal" also makes the write idempotent across the
|
|
1026
|
+
// re-entrant settle path below, and replaces an unchecked `Save()` whose failure left the
|
|
1027
|
+
// graph active — re-emitting frames and recomputing cost every poll, forever.
|
|
1028
|
+
if (rollup.outcome === 'settled') {
|
|
1029
|
+
const settled = await this.claims.TrySettleParent(provider, parentID, rollup.status, rollup.percentComplete, this.contextUser);
|
|
1030
|
+
if (!settled && !TERMINAL_TASK_STATUSES.has(parent.Status)) {
|
|
1031
|
+
// Neither "already terminal" nor a successful write: the statement failed. Leave
|
|
1032
|
+
// the graph active so the next pass retries rather than settling on a status the
|
|
1033
|
+
// database never accepted. Re-queued explicitly, because a failed write is
|
|
1034
|
+
// exactly the case where the row's own timestamp does not advance.
|
|
1035
|
+
LogError(`[TaskGraphDispatcher] Could not write terminal status for graph ${parentID}; leaving it active to retry.`);
|
|
1036
|
+
this.keepRetryingSettlement(parentID);
|
|
1037
|
+
continue;
|
|
1038
|
+
}
|
|
361
1039
|
parent.Status = rollup.status;
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
//
|
|
369
|
-
|
|
370
|
-
|
|
1040
|
+
}
|
|
1041
|
+
else if (parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
|
|
1042
|
+
// Guarded and column-scoped for the same reason the terminal write is — and the race
|
|
1043
|
+
// here needs no exotic timing. This instance may have computed a non-terminal rollup
|
|
1044
|
+
// from a snapshot taken before another instance settled the graph; a full-row save
|
|
1045
|
+
// would then REVERT the status and erase the continuation marker with it, and the
|
|
1046
|
+
// next pass would settle and deliver a second time. See TryUpdateParentProgress.
|
|
1047
|
+
await this.claims.TryUpdateParentProgress(provider, parentID, rollup.status, rollup.percentComplete, this.contextUser);
|
|
1048
|
+
}
|
|
1049
|
+
if (rollup.outcome === 'settled') {
|
|
1050
|
+
// ANNOUNCE ONCE PER PROCESS, RETRY THE REST (R2-12). Layout and the frame are
|
|
1051
|
+
// idempotent facts about a finished graph; the cost, lifecycle and delivery writes
|
|
1052
|
+
// below are the ones re-entry exists to retry. Without this split, a graph that keeps
|
|
1053
|
+
// failing to settle re-persisted geometry and re-emitted the same frame every poll
|
|
1054
|
+
// for the whole rescue window.
|
|
1055
|
+
if (!this.announcedSettlements.has(parentID)) {
|
|
1056
|
+
// Geometry is settled once, here, so every viewer of this run agrees on it.
|
|
1057
|
+
await this.persistComputedLayout(fresh);
|
|
1058
|
+
// Emitted before the continuation is delivered, and outside its once-only guard: a
|
|
1059
|
+
// viewer watching the run should learn it finished whether or not this instance is
|
|
1060
|
+
// the one that wins the delivery CAS.
|
|
1061
|
+
this.emit({
|
|
1062
|
+
Kind: 'GraphSettled',
|
|
1063
|
+
ParentTaskID: parentID,
|
|
1064
|
+
OwnerUserID: await this.resolveOwner(provider, parentID),
|
|
1065
|
+
Status: rollup.status,
|
|
1066
|
+
CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
|
|
1067
|
+
TotalCount: fresh.nodes.length,
|
|
1068
|
+
});
|
|
1069
|
+
this.announcedSettlements.add(parentID);
|
|
1070
|
+
}
|
|
1071
|
+
// READ-ONLY GATE, before any write to the submitting run's half (R2-2).
|
|
1072
|
+
//
|
|
1073
|
+
// A graph can settle before the run that submitted it has parked at all. `BaseAgent`
|
|
1074
|
+
// sets `Paused` in `finalizeAgentRun`, AFTER the graph is durable and dispatchable —
|
|
1075
|
+
// so a fast graph finishes first, and both writes below then land wrong: the
|
|
1076
|
+
// lifecycle write silently returns (its guard is `Status === 'Paused'`), and the cost
|
|
1077
|
+
// write is overwritten moments later by finalize's own full-row save, which carries
|
|
1078
|
+
// the in-memory nulls it had before the dispatcher wrote anything.
|
|
1079
|
+
//
|
|
1080
|
+
// Deferring the whole half — rather than doing the parts that happen to work — is
|
|
1081
|
+
// what keeps the marker honest: nothing below claims it, so the graph stays
|
|
1082
|
+
// terminal-and-undelivered and the rescue sweep brings it back next pass, by which
|
|
1083
|
+
// time finalize has parked the run and both writes land.
|
|
1084
|
+
const readiness = await this.submittingRunReadiness(provider, parent);
|
|
1085
|
+
if (readiness.Verdict === 'defer') {
|
|
1086
|
+
this.keepRetryingSettlement(parentID);
|
|
1087
|
+
continue;
|
|
1088
|
+
}
|
|
1089
|
+
// A settled graph will not produce another verdict, claim or progress report, so the
|
|
1090
|
+
// per-graph observability caches for it are dead weight from here. Unbounded, they
|
|
1091
|
+
// are a slow leak in a process that runs for weeks — one that grows with every
|
|
1092
|
+
// workflow the server has ever seen rather than with anything live. (A deferred
|
|
1093
|
+
// settlement below repopulates them naturally on the retry pass.)
|
|
1094
|
+
this.forgetGraphObservability(parentID);
|
|
1095
|
+
// R3-8: the rollup gets a verdict, and a TRANSIENT failure defers exactly as a
|
|
1096
|
+
// failed lifecycle write does. Continuing past one would settle the run and claim
|
|
1097
|
+
// the marker, which permanently excludes the graph from the rescue sweep — making
|
|
1098
|
+
// the rollup's own "retrying on a later settlement" log a promise it could not keep.
|
|
1099
|
+
// Permanent refusals (truncated tree, graph not in the tree) proceed as before:
|
|
1100
|
+
// those do not clear on their own, and deferring on them would stall forever.
|
|
1101
|
+
if (await this.rollUpCostToSubmittingRun(provider, parent) === 'failed-transient') {
|
|
1102
|
+
this.keepRetryingSettlement(parentID);
|
|
1103
|
+
continue;
|
|
1104
|
+
}
|
|
1105
|
+
// Deliberately AFTER the rollup and OUTSIDE its refusal paths. The rollup declines
|
|
1106
|
+
// to write a number it cannot stand behind — a truncated tree, an unreachable graph
|
|
1107
|
+
// — and every one of those returns early. If the run's lifecycle were settled in
|
|
1108
|
+
// there, a refused rollup would strand the run parked forever, which is a far worse
|
|
1109
|
+
// failure than a missing cost figure. Cost and lifecycle are separate concerns with
|
|
1110
|
+
// separate failure modes, so they get separate writes.
|
|
1111
|
+
if (await this.settleSubmittingRun(provider, parent, rollup.status) === 'defer') {
|
|
1112
|
+
// The lifecycle write did not land. Delivering now would claim the marker and
|
|
1113
|
+
// make this the LAST pass to look at the graph — leaving the run Paused forever,
|
|
1114
|
+
// which is the exact permanence R2-2 removes. Leave the marker unset and retry.
|
|
1115
|
+
this.keepRetryingSettlement(parentID);
|
|
1116
|
+
continue;
|
|
1117
|
+
}
|
|
1118
|
+
if (await this.deliverContinuation(provider, parent, fresh, readiness.SubmitterCancelled)) {
|
|
1119
|
+
// Delivered, expired, or lost the CAS to a peer — every one of those means this
|
|
1120
|
+
// graph is somebody's finished business and needs nothing further from here.
|
|
1121
|
+
this.retryingSettlement.delete(parentID);
|
|
1122
|
+
this.announcedSettlements.delete(parentID);
|
|
1123
|
+
}
|
|
1124
|
+
else {
|
|
1125
|
+
// This instance cannot deliver. Stay quiet about it — the frame is already out —
|
|
1126
|
+
// and leave the graph for a capable peer via the sweep.
|
|
1127
|
+
this.keepRetryingSettlement(parentID);
|
|
1128
|
+
}
|
|
1129
|
+
}
|
|
1130
|
+
}
|
|
1131
|
+
}
|
|
1132
|
+
/**
|
|
1133
|
+
* Credits a finished graph's spending back to the agent run that submitted it.
|
|
1134
|
+
*
|
|
1135
|
+
* **Why this cannot happen during the run.** `BaseAgent` totals a run by walking its steps in
|
|
1136
|
+
* memory at finalization — but a submitting run *ends at submission*. Submit-and-detach is the
|
|
1137
|
+
* point: the run returns as soon as the graph is durable, and the graph executes afterwards,
|
|
1138
|
+
* possibly minutes later on a different instance. At the moment the run computes its totals the
|
|
1139
|
+
* spending has not happened yet, so there is nothing to count. The only place the number can be
|
|
1140
|
+
* known is here, when the graph settles.
|
|
1141
|
+
*
|
|
1142
|
+
* **Why the `…Rollup` columns and not the plain ones.** `AIAgentRun` has carried six `…Rollup`
|
|
1143
|
+
* columns since v3 that nothing has ever written — they exist for exactly this distinction:
|
|
1144
|
+
*
|
|
1145
|
+
* - `TotalCost` — what the run itself spent. For a Flow agent that is genuinely near zero: it
|
|
1146
|
+
* compiled a graph and handed it off. This value is already final and is never rewritten here,
|
|
1147
|
+
* so nothing that reads it today changes meaning, and no guardrail that already evaluated
|
|
1148
|
+
* against it is retroactively falsified.
|
|
1149
|
+
* - `TotalCostRollup` — the run plus everything it caused. Provisional until the graph settles,
|
|
1150
|
+
* which is now.
|
|
1151
|
+
*
|
|
1152
|
+
* **The tree is the authority; these columns are its settlement-time cache.** The total is a SUM
|
|
1153
|
+
* over `GetAgentRunTree`, not arithmetic of its own. The previous version walked the graph's
|
|
1154
|
+
* child tasks and added each one's agent run, which was wrong in two ways that no test could
|
|
1155
|
+
* see: a `Prompt` task has no agent run at all, so every prompt step's spend was simply missing;
|
|
1156
|
+
* and it read each nested run's `…Rollup ?? …Total`, mixing a descendant-inclusive number with an
|
|
1157
|
+
* own-spend one and depending on whether that nested graph happened to have settled yet. The
|
|
1158
|
+
* tree already models every one of those cases — it reaches prompt runs through
|
|
1159
|
+
* `Configuration.runtime.promptRunID`, and it descends into nested runs and their graphs
|
|
1160
|
+
* structurally — so summing it cannot disagree with what the run viewer shows, because it IS
|
|
1161
|
+
* what the run viewer shows.
|
|
1162
|
+
*
|
|
1163
|
+
* **This refuses rather than guesses.** A tree that failed to load, hit the depth cap, or does
|
|
1164
|
+
* not contain the settling graph would still produce a number — a lower bound. Writing one would
|
|
1165
|
+
* put an authoritative-looking total in a column every cost surface reads. Each of those cases
|
|
1166
|
+
* logs and leaves the column alone, so `?? TotalCost` keeps its honest meaning: not settled.
|
|
1167
|
+
*
|
|
1168
|
+
* A graph with no submitting run (a scheduled job, a remote-operation caller) simply has nobody
|
|
1169
|
+
* to credit — its own Task rows still carry the truth, and this returns quietly.
|
|
1170
|
+
*/
|
|
1171
|
+
async rollUpCostToSubmittingRun(provider, parent) {
|
|
1172
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
1173
|
+
if (!meta.submittedByAgentRunID)
|
|
1174
|
+
return 'landed';
|
|
1175
|
+
const runID = meta.submittedByAgentRunID;
|
|
1176
|
+
try {
|
|
1177
|
+
const runQuery = asRunQueryProvider(provider);
|
|
1178
|
+
if (!runQuery) {
|
|
1179
|
+
// Permanent for this host: a provider that cannot run queries will not grow the
|
|
1180
|
+
// ability on the next pass, so deferring would stall the graph forever.
|
|
1181
|
+
LogError(`[TaskGraphDispatcher] Cannot roll up cost for run ${runID}: provider cannot run queries.`);
|
|
1182
|
+
return 'refused-permanent';
|
|
1183
|
+
}
|
|
1184
|
+
const tree = await LoadAgentRunTree(runID, runQuery, this.contextUser);
|
|
1185
|
+
// Each of these means the sum would be a LOWER BOUND, and the column's whole contract is
|
|
1186
|
+
// that it equals the tree. A known-low number presented as a total is worse than no
|
|
1187
|
+
// number: the readers all fall back to TotalCost when this is null, which at least
|
|
1188
|
+
// *says* it is the run's own spend rather than claiming to be the whole story.
|
|
1189
|
+
//
|
|
1190
|
+
// Refusing is NOT the same as leaving the column alone. A run that submitted two graphs
|
|
1191
|
+
// has a rollup from the first; if the second cannot be summed, the first graph's total
|
|
1192
|
+
// sits in the authoritative column excluding work that has since happened — stale, not
|
|
1193
|
+
// absent, and `?? TotalCost` cannot save a reader from a non-null wrong number. So a
|
|
1194
|
+
// refusal CLEARS it, restoring the fallback's honest meaning: not settled.
|
|
1195
|
+
if (tree.ErrorMessage || !tree.Root) {
|
|
1196
|
+
// TRANSIENT — nothing cleared (R2-15), and now nothing delivered either (R3-8).
|
|
1197
|
+
//
|
|
1198
|
+
// R2-15 stopped this path erasing a correct total. What it did not stop was the pass
|
|
1199
|
+
// CONTINUING: settlement flipped the run terminal and delivery claimed the marker,
|
|
1200
|
+
// which permanently excludes the graph from the rescue sweep — so this function's
|
|
1201
|
+
// own promise of "retrying on a later settlement" was structurally impossible to
|
|
1202
|
+
// keep. One transient DB error, no interleaving, and a multi-graph run kept a wrong
|
|
1203
|
+
// non-null authoritative total forever while a first-graph run stayed null.
|
|
1204
|
+
LogError(`[TaskGraphDispatcher] Could not load the run tree for ${runID} to roll up graph ` +
|
|
1205
|
+
`${parent.ID}: ${tree.ErrorMessage ?? 'the run tree came back empty'}. Leaving any ` +
|
|
1206
|
+
`existing rollup alone and deferring settlement so a later pass can retry.`);
|
|
1207
|
+
return 'failed-transient';
|
|
1208
|
+
}
|
|
1209
|
+
if (tree.Truncated) {
|
|
1210
|
+
// PERMANENT: the tree is genuinely too deep, and it will be just as deep next pass.
|
|
1211
|
+
await this.clearStaleRollup(provider, runID, `the run tree hit the depth cap, so any total would silently under-report ` +
|
|
1212
|
+
`(graph ${parent.ID} still carries its own costs)`);
|
|
1213
|
+
return 'refused-permanent';
|
|
1214
|
+
}
|
|
1215
|
+
// The graph that just settled must appear in the tree. If it does not, the tree stopped
|
|
1216
|
+
// at the run — the submitting step never recorded its parentTaskID — and the sum is
|
|
1217
|
+
// merely the run's own spend wearing the name of a rollup. That is precisely the silent
|
|
1218
|
+
// under-count this rewrite exists to remove, so it is reported rather than written.
|
|
1219
|
+
if (!this.treeContainsGraph(tree.Root, parent.ID)) {
|
|
1220
|
+
// PERMANENT: a missing parentTaskID link is a fact about how the graph was
|
|
1221
|
+
// submitted, not a condition that clears on its own.
|
|
1222
|
+
await this.clearStaleRollup(provider, runID, `graph ${parent.ID} is not reachable from it, so the tree cannot see the work. ` +
|
|
1223
|
+
`Did the submitting step record parentTaskID?`);
|
|
1224
|
+
return 'refused-permanent';
|
|
1225
|
+
}
|
|
1226
|
+
const totals = SumAgentRunTreeCost(tree.Root);
|
|
1227
|
+
// Assignment, never accumulation. The tree already contains the run's own spend as its
|
|
1228
|
+
// ROOT node, and it reads own-cost everywhere, so recomputing from scratch on every
|
|
1229
|
+
// settlement lands on the same answer — which is what makes this safe to call again when
|
|
1230
|
+
// a second graph settles, or when the terminal check is re-evaluated after a HITL wait.
|
|
1231
|
+
//
|
|
1232
|
+
// COLUMN-SCOPED (C4). A full-row save here carried a whole snapshot of the run, and two
|
|
1233
|
+
// instances entering the settled branch for one graph is by design — so a peer's rollup,
|
|
1234
|
+
// loaded before this instance settled the run, would write `Paused` back over
|
|
1235
|
+
// `Completed` along with everything else it had read.
|
|
1236
|
+
if (!(await this.claims.TrySetRunCostRollup(provider, runID, totals, this.contextUser))) {
|
|
1237
|
+
LogError(`[TaskGraphDispatcher] Could not record graph cost against run ${runID}.`);
|
|
1238
|
+
return 'failed-transient';
|
|
1239
|
+
}
|
|
1240
|
+
LogStatus(`[TaskGraphDispatcher] Credited graph ${parent.ID} to run ${runID}: ` +
|
|
1241
|
+
`${tree.Rows.length} node(s), ${totals.Tokens} token(s), cost ${totals.Cost}.`);
|
|
1242
|
+
}
|
|
1243
|
+
catch (e) {
|
|
1244
|
+
// A failed rollup must never fail the graph. The work finished; only the accounting for
|
|
1245
|
+
// it is missing, and a graph marked Failed because its cost could not be summed would be
|
|
1246
|
+
// a far worse lie than a cost of null.
|
|
1247
|
+
// A throw is transient by default: nothing here proves the condition will persist, and
|
|
1248
|
+
// the cost of being wrong in this direction is one deferred pass rather than a
|
|
1249
|
+
// permanently wrong authoritative total.
|
|
1250
|
+
LogError(`[TaskGraphDispatcher] Cost rollup failed for graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
1251
|
+
return 'failed-transient';
|
|
1252
|
+
}
|
|
1253
|
+
return 'landed';
|
|
1254
|
+
}
|
|
1255
|
+
/**
|
|
1256
|
+
* Clears a rollup that can no longer be trusted, and says why.
|
|
1257
|
+
*
|
|
1258
|
+
* **Why clear rather than leave.** The four `…Rollup` columns are a cache of the run tree, and
|
|
1259
|
+
* every reader treats a value there as the total. When the tree cannot be summed, any value
|
|
1260
|
+
* already in the column was computed from an EARLIER settlement — it excludes the graph that
|
|
1261
|
+
* just finished, so it is not merely incomplete, it is a wrong total presented as a right one.
|
|
1262
|
+
* `?? TotalCost` protects a reader from null, not from stale.
|
|
1263
|
+
*
|
|
1264
|
+
* Nulling restores the invariant this whole design rests on: **when the column is present, it
|
|
1265
|
+
* equals the tree.** Absent means not settled, which is exactly what a reader should conclude.
|
|
1266
|
+
* A run with no rollup yet is untouched — there is nothing stale to clear, and writing nulls
|
|
1267
|
+
* over nulls would churn Record Changes for nothing.
|
|
1268
|
+
*/
|
|
1269
|
+
async clearStaleRollup(provider, runID, reason) {
|
|
1270
|
+
LogError(`[TaskGraphDispatcher] Not recording cost for run ${runID}: ${reason}.`);
|
|
1271
|
+
try {
|
|
1272
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
1273
|
+
if (!(await run.Load(runID)))
|
|
1274
|
+
return;
|
|
1275
|
+
if (run.TotalCostRollup == null && run.TotalTokensUsedRollup == null)
|
|
1276
|
+
return; // nothing stale
|
|
1277
|
+
run.TotalCostRollup = null;
|
|
1278
|
+
run.TotalTokensUsedRollup = null;
|
|
1279
|
+
run.TotalPromptTokensUsedRollup = null;
|
|
1280
|
+
run.TotalCompletionTokensUsedRollup = null;
|
|
1281
|
+
if (!(await run.Save())) {
|
|
1282
|
+
LogError(`[TaskGraphDispatcher] Could not clear the now-stale rollup on run ${runID}: ` +
|
|
1283
|
+
`${run.LatestResult?.CompleteMessage ?? 'unknown error'}. It still shows a total that ` +
|
|
1284
|
+
`excludes the graph that just settled.`);
|
|
1285
|
+
return;
|
|
1286
|
+
}
|
|
1287
|
+
LogStatus(`[TaskGraphDispatcher] Cleared the rollup on run ${runID}: it was computed before this ` +
|
|
1288
|
+
`graph settled and can no longer be recomputed, so it would have under-reported.`);
|
|
1289
|
+
}
|
|
1290
|
+
catch (e) {
|
|
1291
|
+
LogError(`[TaskGraphDispatcher] Could not clear the rollup on run ${runID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
1292
|
+
}
|
|
1293
|
+
}
|
|
1294
|
+
/**
|
|
1295
|
+
* Whether the settling graph is actually reachable from the submitting run's tree.
|
|
1296
|
+
*
|
|
1297
|
+
* Matched on the graph's parent Task id, which is the node the `TaskGraph` member of the query
|
|
1298
|
+
* emits. A run that submitted a graph but recorded no `parentTaskID` produces a tree that stops
|
|
1299
|
+
* at the run — structurally indistinguishable, at the SUM, from a run that never dispatched
|
|
1300
|
+
* anything. This is the check that tells those two apart.
|
|
1301
|
+
*/
|
|
1302
|
+
treeContainsGraph(root, parentTaskID) {
|
|
1303
|
+
for (const node of WalkAgentRunTree(root)) {
|
|
1304
|
+
if (node.NodeType === 'TaskGraph' && UUIDsEqual(node.NodeID, parentTaskID))
|
|
1305
|
+
return true;
|
|
1306
|
+
}
|
|
1307
|
+
return false;
|
|
1308
|
+
}
|
|
1309
|
+
/**
|
|
1310
|
+
* Ends a graph early because a prompt said the work is finished.
|
|
1311
|
+
*
|
|
1312
|
+
* **Why `Skipped` and not `Cancelled`.** Nothing went wrong and nobody intervened — the workflow
|
|
1313
|
+
* reached its own conclusion before running every drawn step, which is exactly what a reasoning
|
|
1314
|
+
* step is for. `Cancelled` would tell a reader someone stopped it; `Skipped` says these routes
|
|
1315
|
+
* were not taken, which is true and already the vocabulary the fork machinery uses.
|
|
1316
|
+
*
|
|
1317
|
+
* The message is written to the parent so the graph carries its own answer, rather than the
|
|
1318
|
+
* answer living only on the step that produced it.
|
|
1319
|
+
*/
|
|
1320
|
+
async endGraphEarly(provider, task, message) {
|
|
1321
|
+
if (!task.ParentID)
|
|
1322
|
+
return;
|
|
1323
|
+
try {
|
|
1324
|
+
LogStatus(`[TaskGraphDispatcher] '${task.Name}' ended the workflow early: ${message}`);
|
|
1325
|
+
// DECLARE BEFORE MUTATING (R3-1).
|
|
1326
|
+
//
|
|
1327
|
+
// The early finish is decided by one task's result and known to nothing else: skip seeds
|
|
1328
|
+
// come from durable condition and exclusive-group state, so no claim filter anywhere —
|
|
1329
|
+
// including this instance's own, since `executeClaimed` is not awaited and the poll loop
|
|
1330
|
+
// runs concurrently — can tell the remaining steps are about to be skipped. Stamping it
|
|
1331
|
+
// first is what lets `loadGraphState` fold them into the filter, which closes the window
|
|
1332
|
+
// for every instance instead of narrowing it for one.
|
|
1333
|
+
const typeID = await this.workflowTaskTypeID(provider);
|
|
1334
|
+
if (typeID)
|
|
1335
|
+
await this.claims.TryDeclareEarlyFinish(provider, task.ParentID, typeID, this.contextUser);
|
|
1336
|
+
const skipped = [];
|
|
1337
|
+
const claimedMeanwhile = [];
|
|
1338
|
+
for (const sibling of await this.loadChildTasks(provider, task.ParentID)) {
|
|
1339
|
+
if (UUIDsEqual(sibling.ID, task.ID) || sibling.Status !== 'Pending')
|
|
1340
|
+
continue;
|
|
1341
|
+
// GUARDED, not a full-row save against the snapshot above. A sibling claimed between
|
|
1342
|
+
// that load and this write is mid-execution; overwriting it to `Skipped` discards a
|
|
1343
|
+
// running step's outcome while its side effects have already fired. Rowcount is the
|
|
1344
|
+
// verdict — see TaskClaimStore.TrySkipPending.
|
|
1345
|
+
if (!typeID || !(await this.claims.TrySkipPending(provider, sibling.ID, typeID, this.contextUser))) {
|
|
1346
|
+
claimedMeanwhile.push(sibling.Name);
|
|
1347
|
+
continue;
|
|
1348
|
+
}
|
|
1349
|
+
skipped.push(sibling.ID);
|
|
371
1350
|
this.emit({
|
|
372
|
-
Kind: '
|
|
373
|
-
ParentTaskID:
|
|
374
|
-
OwnerUserID: await this.resolveOwner(provider,
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
1351
|
+
Kind: 'TaskSkipped',
|
|
1352
|
+
ParentTaskID: task.ParentID,
|
|
1353
|
+
OwnerUserID: await this.resolveOwner(provider, task.ParentID),
|
|
1354
|
+
TaskID: sibling.ID,
|
|
1355
|
+
TaskName: sibling.Name,
|
|
1356
|
+
Status: 'Skipped',
|
|
378
1357
|
});
|
|
379
|
-
await this.deliverContinuation(provider, parent, fresh);
|
|
380
1358
|
}
|
|
1359
|
+
if (claimedMeanwhile.length > 0) {
|
|
1360
|
+
// Reported rather than forced. Those steps were already running when the workflow
|
|
1361
|
+
// decided to stop; their results are real and are allowed to land. The graph settles
|
|
1362
|
+
// once they finish, which is a pass later than it would have and correct.
|
|
1363
|
+
LogStatus(`[TaskGraphDispatcher] Early finish left ${claimedMeanwhile.length} step(s) running ` +
|
|
1364
|
+
`(${claimedMeanwhile.join(', ')}) — they were claimed before the skip and their outcomes stand.`);
|
|
1365
|
+
}
|
|
1366
|
+
// WITHDRAW WHAT WE JUST SKIPPED (R2-10).
|
|
1367
|
+
//
|
|
1368
|
+
// A skipped human step leaves its `MJ: AI Agent Requests` row `Requested` forever:
|
|
1369
|
+
// un-answerable, because answering settles nothing once the task is terminal, and
|
|
1370
|
+
// immortal, because the human settle and expiry sweeps both filter on `Status='Pending'`
|
|
1371
|
+
// tasks and this one no longer is. The person keeps seeing "a workflow is waiting on
|
|
1372
|
+
// you" for a workflow that finished without them. `Cancel` has always done this; the
|
|
1373
|
+
// early-finish path skipped exactly the same rows and did not.
|
|
1374
|
+
await this.withdrawOpenRequests(provider, skipped, 'The workflow finished before this step was needed.');
|
|
1375
|
+
// Column-scoped, because skipping the siblings above just made this graph fully
|
|
1376
|
+
// terminal — so another instance can settle it and claim the marker before this line
|
|
1377
|
+
// runs. A full-row save from the snapshot we loaded first would undo both. See
|
|
1378
|
+
// TaskClaimStore.TrySetParentOutput.
|
|
1379
|
+
if (typeID) {
|
|
1380
|
+
await this.claims.TrySetParentOutput(provider, task.ParentID, JSON.stringify({ message }), typeID, this.contextUser);
|
|
1381
|
+
}
|
|
1382
|
+
else {
|
|
1383
|
+
// Surfaced rather than dropped (R2-10). The graph still ends early — the siblings
|
|
1384
|
+
// are already Skipped — but the reason it ended goes nowhere, and a workflow that
|
|
1385
|
+
// stopped for a stated reason with no stated reason recorded is exactly the kind of
|
|
1386
|
+
// silence this round exists to remove.
|
|
1387
|
+
LogError(`[TaskGraphDispatcher] Could not resolve the workflow task type, so the early-finish ` +
|
|
1388
|
+
`reason for graph ${task.ParentID} was not recorded: ${message}`);
|
|
1389
|
+
}
|
|
1390
|
+
}
|
|
1391
|
+
catch (e) {
|
|
1392
|
+
// The work itself succeeded; only the early-finish bookkeeping failed. Failing the task
|
|
1393
|
+
// over that would discard a completed step's result.
|
|
1394
|
+
LogError(`[TaskGraphDispatcher] Could not end graph early for ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
1395
|
+
}
|
|
1396
|
+
}
|
|
1397
|
+
/**
|
|
1398
|
+
* How deep the continuation chain already is, read from the graph's parent metadata.
|
|
1399
|
+
*
|
|
1400
|
+
* A run started by a graph inherits that graph's depth **plus one**. Without this every spawned
|
|
1401
|
+
* run begins at zero, so a self-referencing flow — one that dispatches a graph containing itself
|
|
1402
|
+
* — recurses without bound while the cap it should be hitting compares against a permanent zero.
|
|
1403
|
+
*/
|
|
1404
|
+
async graphContext(provider, task) {
|
|
1405
|
+
if (!task.ParentID)
|
|
1406
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
1407
|
+
try {
|
|
1408
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
1409
|
+
if (!(await parent.Load(task.ParentID)))
|
|
1410
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
1411
|
+
return {
|
|
1412
|
+
Depth: ParseTaskGraphParentMetadata(parent.InputPayload).reinvokeDepth + 1,
|
|
1413
|
+
// The graph's own row carries the run that submitted it. One load answers both
|
|
1414
|
+
// questions, which is why they are resolved together rather than in two passes.
|
|
1415
|
+
SubmittingAgentRunID: parent.AgentRunID,
|
|
1416
|
+
};
|
|
1417
|
+
}
|
|
1418
|
+
catch {
|
|
1419
|
+
// An unreadable parent must not stop the work; depth zero is the safe reading, and the
|
|
1420
|
+
// submit-time cap still guards the next hop.
|
|
1421
|
+
return { Depth: 0, SubmittingAgentRunID: null };
|
|
1422
|
+
}
|
|
1423
|
+
}
|
|
1424
|
+
/**
|
|
1425
|
+
* Which failures the workflow drew a way out of.
|
|
1426
|
+
*
|
|
1427
|
+
* A Failed task with a **satisfied outgoing edge** is a handled failure: its author drew a
|
|
1428
|
+
* recovery route and that route is now live. Downstream work should be released along it, and the
|
|
1429
|
+
* parent should not roll up Failed because of a step the workflow explicitly planned around.
|
|
1430
|
+
*
|
|
1431
|
+
* Scoped to `failureSemantics: 'edges'` on purpose. Under `'block'` — every agent-emitted graph —
|
|
1432
|
+
* a failure is terminal for its dependents whatever edges exist, because nobody drew those edges
|
|
1433
|
+
* as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
|
|
1434
|
+
* graph sail past a failure it never anticipated.
|
|
1435
|
+
*/
|
|
1436
|
+
computeHandledFailures(failureSemantics, nodes, edges) {
|
|
1437
|
+
const handled = new Set();
|
|
1438
|
+
if (failureSemantics !== 'edges')
|
|
1439
|
+
return handled;
|
|
1440
|
+
for (const node of nodes) {
|
|
1441
|
+
if (node.status !== 'Failed')
|
|
1442
|
+
continue;
|
|
1443
|
+
// "Has somewhere to go" is the test. An edge out of a failed step that survived condition
|
|
1444
|
+
// evaluation IS the drawn recovery route; a failed step with no outgoing edges has none,
|
|
1445
|
+
// and stays terminal.
|
|
1446
|
+
if (edges.some((e) => e.dependsOnTaskId === node.id))
|
|
1447
|
+
handled.add(node.id);
|
|
381
1448
|
}
|
|
1449
|
+
return handled;
|
|
1450
|
+
}
|
|
1451
|
+
/** The graph's child tasks, with the fields the rollup needs. */
|
|
1452
|
+
async loadChildTasks(provider, parentID) {
|
|
1453
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
1454
|
+
EntityName: 'MJ: Tasks',
|
|
1455
|
+
ExtraFilter: `ParentID='${parentID}'`,
|
|
1456
|
+
ResultType: 'entity_object',
|
|
1457
|
+
BypassCache: true,
|
|
1458
|
+
}, this.contextUser);
|
|
1459
|
+
return (result.Success ? result.Results : []) ?? [];
|
|
382
1460
|
}
|
|
383
1461
|
/**
|
|
384
1462
|
* Runs the graph's continuation exactly once, now that it has settled.
|
|
@@ -391,28 +1469,77 @@ export class TaskGraphDispatcher {
|
|
|
391
1469
|
* user-visible noise and, for `reinvoke`, an unbounded agent-run loop. Given one of the two has
|
|
392
1470
|
* to be chosen, the quiet failure is the safe one.
|
|
393
1471
|
*
|
|
394
|
-
* The marker is
|
|
395
|
-
* completed graph produce one winner rather than two.
|
|
1472
|
+
* The marker is claimed with a real compare-and-swap (one guarded UPDATE, rowcount as verdict),
|
|
1473
|
+
* so two instances reconciling the same completed graph produce one winner rather than two.
|
|
396
1474
|
*/
|
|
397
|
-
async deliverContinuation(provider, parent, graph) {
|
|
1475
|
+
async deliverContinuation(provider, parent, graph, submitterCancelled) {
|
|
398
1476
|
const meta = this.readParentMetadata(parent);
|
|
399
1477
|
if (meta.continuationDeliveredAt)
|
|
400
|
-
return;
|
|
401
|
-
//
|
|
1478
|
+
return true;
|
|
1479
|
+
// Nobody is waiting: the run that submitted this graph was cancelled. Claim the marker so
|
|
1480
|
+
// nothing re-offers the graph, and record WHY nothing was announced — "we chose not to" and
|
|
1481
|
+
// "we found it too late" are different facts about a settlement, and a reader afterwards
|
|
1482
|
+
// should be able to tell them apart.
|
|
1483
|
+
if (submitterCancelled) {
|
|
1484
|
+
if (await this.claimContinuation(provider, parent.ID, 'cancelled')) {
|
|
1485
|
+
LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} settled, but the run that submitted it was ` +
|
|
1486
|
+
`cancelled — no message posted and no reinvoke started.`);
|
|
1487
|
+
}
|
|
1488
|
+
return true;
|
|
1489
|
+
}
|
|
1490
|
+
// At the cap, DOWNGRADE rather than refuse: the results still reach the user, the chain just
|
|
402
1491
|
// stops growing. Refusing outright would lose the outcome of work that actually completed.
|
|
403
|
-
|
|
1492
|
+
//
|
|
1493
|
+
// But a downgrade only applies to something that was going to be delivered (C2). Mapping the
|
|
1494
|
+
// cap straight onto `'message'` also promoted `continuation: 'none'` — a graph that asked
|
|
1495
|
+
// for silence — into a message nobody requested. Latent today because `Submit` refuses to
|
|
1496
|
+
// create a graph past the cap, and exactly the kind of latent that stops being latent the
|
|
1497
|
+
// moment a producer bypasses that check.
|
|
1498
|
+
const mode = meta.continuation !== 'none' && IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
|
|
404
1499
|
if (mode !== 'none' && IsReinvokeCapReached(meta) && meta.continuation === 'reinvoke') {
|
|
405
1500
|
LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} hit the reinvoke cap (${MAX_REINVOKE_DEPTH}); ` +
|
|
406
1501
|
`delivering results as a message instead of starting another turn.`);
|
|
407
1502
|
}
|
|
408
|
-
|
|
409
|
-
|
|
1503
|
+
// A SETTLEMENT NOBODY IS WAITING FOR STILL SETTLES — it just does not get announced.
|
|
1504
|
+
//
|
|
1505
|
+
// Run settlement and cost rollup are status corrections and are always safe to apply late; a
|
|
1506
|
+
// run left `Paused` forever is strictly worse than a stale notification skipped. A stale
|
|
1507
|
+
// NOTIFICATION is not: posting a day-old "your workflow finished" into a live conversation,
|
|
1508
|
+
// or worse starting a fresh billed agent turn for it, is the outcome the age-out exists to
|
|
1509
|
+
// avoid. So an aged-out settlement claims the marker as `expired` and logs, which both
|
|
1510
|
+
// records what happened and stops any later pass delivering it. Second rung on the ladder
|
|
1511
|
+
// the reinvoke cap already established.
|
|
1512
|
+
const expired = IsSettlementExpired(parent.CompletedAt, new Date());
|
|
1513
|
+
// ONLY AN INSTANCE THAT CAN DELIVER MAY CLAIM THE RIGHT TO (R2-6).
|
|
1514
|
+
//
|
|
1515
|
+
// The claim ran before the deliverer check, so an instance constructed WITHOUT one — a
|
|
1516
|
+
// worker tier, an integration bundle, a second dev session — could observe the settlement
|
|
1517
|
+
// first, win the CAS, mark the graph `delivered`, and discard the message or reinvoke a
|
|
1518
|
+
// capable peer would have made moments later. Permanently, decided by poll timing.
|
|
1519
|
+
//
|
|
1520
|
+
// Declining leaves the marker unset, so the rescue sweep keeps offering the graph until an
|
|
1521
|
+
// instance that can deliver takes it. Run settlement and cost rollup have already happened
|
|
1522
|
+
// above and are not held up by this — what is deferred is the announcement, which is the only
|
|
1523
|
+
// part this instance genuinely cannot do.
|
|
1524
|
+
//
|
|
1525
|
+
// `expired` is exempt: recording "too old to deliver" requires no deliverer, and a graph past
|
|
1526
|
+
// its window has nothing left for a capable peer to do.
|
|
1527
|
+
if (!expired && mode !== 'none' && !this.continuationDeliverer) {
|
|
1528
|
+
this.reportUndeliverableOnce(parent.ID);
|
|
1529
|
+
return false;
|
|
1530
|
+
}
|
|
1531
|
+
if (!(await this.claimContinuation(provider, parent.ID, expired ? 'expired' : 'delivered')))
|
|
1532
|
+
return true;
|
|
1533
|
+
if (expired) {
|
|
1534
|
+
LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} settled after its delivery window ` +
|
|
1535
|
+
`(${UNSETTLED_SWEEP_WINDOW_HOURS}h); the run and its cost were corrected, but the ` +
|
|
1536
|
+
`continuation was NOT delivered. Marked expired.`);
|
|
1537
|
+
return true;
|
|
1538
|
+
}
|
|
410
1539
|
if (mode === 'none')
|
|
411
|
-
return;
|
|
1540
|
+
return true;
|
|
412
1541
|
const summary = this.buildContinuationSummary(parent, graph);
|
|
413
1542
|
LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} finished — ${summary}`);
|
|
414
|
-
if (!this.continuationDeliverer)
|
|
415
|
-
return;
|
|
416
1543
|
const params = {
|
|
417
1544
|
ParentTaskID: parent.ID,
|
|
418
1545
|
WorkflowName: parent.Name,
|
|
@@ -449,6 +1576,12 @@ export class TaskGraphDispatcher {
|
|
|
449
1576
|
// on the marker: a missed notification visible in the record beats one repeated forever.
|
|
450
1577
|
LogError(`[TaskGraphDispatcher] Continuation delivery failed for ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
451
1578
|
}
|
|
1579
|
+
// C1: the function is declared `Promise<boolean>` and fell off the end here, so EVERY
|
|
1580
|
+
// successfully delivered graph resolved `undefined` — read as "not resolved" by the caller,
|
|
1581
|
+
// which then re-queued it and paid another full settle pass (run-tree cost query included)
|
|
1582
|
+
// and polluted R2-6's retry accounting. No double delivery, because the CAS holds; just a
|
|
1583
|
+
// wasted pass per settlement and a retry counter measuring the wrong thing.
|
|
1584
|
+
return true;
|
|
452
1585
|
}
|
|
453
1586
|
/** Reads the parent's durable continuation metadata through the shared parser. */
|
|
454
1587
|
readParentMetadata(parent) {
|
|
@@ -460,19 +1593,26 @@ export class TaskGraphDispatcher {
|
|
|
460
1593
|
* `MJ: Tasks` stays user-writable (D20), so a plain "read, decide, write" is not enough — the
|
|
461
1594
|
* read-back is what makes a lost race observable instead of producing a duplicate delivery.
|
|
462
1595
|
*/
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
1596
|
+
/**
|
|
1597
|
+
* True when this graph finished so long ago that announcing it would surprise rather than inform.
|
|
1598
|
+
*
|
|
1599
|
+
* Measured from the parent's completion, not from when we noticed: the point is how stale the
|
|
1600
|
+
* NEWS is to whoever would receive it.
|
|
1601
|
+
*/
|
|
1602
|
+
async claimContinuation(provider, parentID, deliveredAs = 'delivered') {
|
|
1603
|
+
// ONE GUARDED STATEMENT — see TaskClaimStore.TryClaimContinuation.
|
|
1604
|
+
//
|
|
1605
|
+
// This was Load → check the marker → `Save()`: an unconditional last-write-wins UPDATE that
|
|
1606
|
+
// two dispatchers could both pass. The comments here and at the call site called it a
|
|
1607
|
+
// compare-and-swap read-back; it was read-check-write, and for `continuation: 'reinvoke'`
|
|
1608
|
+
// losing that race means two fresh agent turns billed for one settlement, each able to
|
|
1609
|
+
// submit further graphs.
|
|
1610
|
+
// The type discriminator is part of the guard, not a caller-side filter — see
|
|
1611
|
+
// TryClaimContinuation. Nothing to claim if the type does not exist: no graph was submitted.
|
|
1612
|
+
const typeID = await this.workflowTaskTypeID(provider);
|
|
1613
|
+
if (!typeID)
|
|
473
1614
|
return false;
|
|
474
|
-
|
|
475
|
-
return true;
|
|
1615
|
+
return this.claims.TryClaimContinuation(provider, parentID, deliveredAs, typeID, this.contextUser);
|
|
476
1616
|
}
|
|
477
1617
|
/** One line describing how the graph ended, for the completion log and message delivery. */
|
|
478
1618
|
buildContinuationSummary(parent, graph) {
|
|
@@ -498,8 +1638,29 @@ export class TaskGraphDispatcher {
|
|
|
498
1638
|
async notifyHumanTaskReady(task, provider) {
|
|
499
1639
|
if (task.ClaimedBy === HUMAN_TASK_NOTIFIED_MARKER)
|
|
500
1640
|
return;
|
|
501
|
-
|
|
502
|
-
|
|
1641
|
+
// The REQUEST is raised whether or not the task names an assignee. An unassigned human step
|
|
1642
|
+
// is a legitimate "somebody needs to look at this", and a request nobody was notified about
|
|
1643
|
+
// is still findable in the inbox — whereas returning early here is how such a step used to
|
|
1644
|
+
// become invisible work that stalled a workflow with nothing anywhere saying why.
|
|
1645
|
+
// TRANSIENT failures retry; PERMANENT ones stop. That distinction is the whole point, and
|
|
1646
|
+
// getting it wrong took a server down: retrying unconditionally meant a task whose workflow
|
|
1647
|
+
// has no owning agent — which can never succeed — was re-attempted on every poll forever,
|
|
1648
|
+
// each pass re-reading the graph, until the process was OOM-killed. The marker exists to
|
|
1649
|
+
// prevent exactly that storm; a permanent failure has to set it.
|
|
1650
|
+
const raised = await this.raiseHumanRequest(task, provider);
|
|
1651
|
+
if (raised === 'transient-failure')
|
|
1652
|
+
return; // try again next poll
|
|
1653
|
+
if (raised === 'permanent-failure') {
|
|
1654
|
+
// Nothing will change on a retry. Mark it so the loop stops, and leave the task Pending
|
|
1655
|
+
// and visible — a person can still see it in the Tasks UI, which is the fallback the
|
|
1656
|
+
// notification was only ever an accelerant for.
|
|
1657
|
+
await this.markHumanTaskNotified(task, provider);
|
|
1658
|
+
return;
|
|
1659
|
+
}
|
|
1660
|
+
if (!task.UserID) {
|
|
1661
|
+
await this.markHumanTaskNotified(task, provider);
|
|
1662
|
+
return;
|
|
1663
|
+
}
|
|
503
1664
|
try {
|
|
504
1665
|
await NotificationEngine.Instance.Config(false, this.contextUser);
|
|
505
1666
|
await NotificationEngine.Instance.SendNotification({
|
|
@@ -513,13 +1674,7 @@ export class TaskGraphDispatcher {
|
|
|
513
1674
|
catch (e) {
|
|
514
1675
|
LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
515
1676
|
}
|
|
516
|
-
|
|
517
|
-
// worse failure than one that was missed: the task remains visible in the Tasks UI either
|
|
518
|
-
// way, whereas a notification storm is not self-correcting.
|
|
519
|
-
task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
|
|
520
|
-
if (!(await task.Save())) {
|
|
521
|
-
LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
|
|
522
|
-
}
|
|
1677
|
+
await this.markHumanTaskNotified(task, provider);
|
|
523
1678
|
// Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
|
|
524
1679
|
// than appearing to stall for no reason.
|
|
525
1680
|
this.emit({
|
|
@@ -543,8 +1698,88 @@ export class TaskGraphDispatcher {
|
|
|
543
1698
|
* invalidate. Left cached, a completed task keeps reading as `In Progress` and the graph never
|
|
544
1699
|
* rolls up: submitted work simply never settles.
|
|
545
1700
|
*/
|
|
546
|
-
|
|
1701
|
+
/**
|
|
1702
|
+
* Settles graphs that reached terminal without completing their post-settlement sequence.
|
|
1703
|
+
*
|
|
1704
|
+
* Runs the ordinary propagation path, which is safe to re-enter by construction: the terminal
|
|
1705
|
+
* write is guarded on not-already-terminal, the cost rollup assigns rather than accumulates, run
|
|
1706
|
+
* settlement is guarded on `Paused`, and delivery is guarded by the continuation CAS. A revisit
|
|
1707
|
+
* therefore corrects whatever is missing and does nothing where nothing is.
|
|
1708
|
+
*
|
|
1709
|
+
* @param windowHours how far back to look — wide once at startup, narrow in steady state
|
|
1710
|
+
*/
|
|
1711
|
+
async sweepUnsettledGraphs(windowHours) {
|
|
1712
|
+
try {
|
|
1713
|
+
const provider = await this.providerFactory.CreateProvider();
|
|
1714
|
+
const ids = await this.findActiveGraphIDs(provider, windowHours);
|
|
1715
|
+
if (ids.length === 0)
|
|
1716
|
+
return;
|
|
1717
|
+
LogStatus(`[TaskGraphDispatcher] Startup sweep: reviewing ${ids.length} graph(s), including any that reached terminal without settling.`);
|
|
1718
|
+
await this.propagateAndRollup(provider, ids);
|
|
1719
|
+
}
|
|
1720
|
+
catch (e) {
|
|
1721
|
+
LogError(`[TaskGraphDispatcher] Unsettled-graph sweep failed: ${e instanceof Error ? e.message : String(e)}`);
|
|
1722
|
+
}
|
|
1723
|
+
}
|
|
1724
|
+
/**
|
|
1725
|
+
* The `AI Workflow` task type, resolved once per process.
|
|
1726
|
+
*
|
|
1727
|
+
* `MJ: Tasks` is a GENERAL-PURPOSE entity — conversations and user to-dos live there too — so an
|
|
1728
|
+
* unscoped sweep treats every root task hierarchy as a workflow: rolling up and overwriting the
|
|
1729
|
+
* status of somebody's to-do list, raising agent requests against plain tasks, and (once the
|
|
1730
|
+
* continuation CAS exists) injecting marker keys into a user's own `InputPayload`.
|
|
1731
|
+
*
|
|
1732
|
+
* `Submit` has always stamped this type on the parent and every child (`ensureTaskType`, which
|
|
1733
|
+
* runs before the persist transaction), so the discriminator D3 called for already exists on
|
|
1734
|
+
* every dispatcher-owned row. Verified against the live database: every parent graph carries it.
|
|
1735
|
+
*
|
|
1736
|
+
* Null when the type row does not exist yet — no graph has ever been submitted — in which case
|
|
1737
|
+
* there is nothing for the dispatcher to find and the sweep returns empty rather than unscoped.
|
|
1738
|
+
*
|
|
1739
|
+
* **A miss is never cached**, and that is not a micro-optimisation. `TaskGraphService.Submit`
|
|
1740
|
+
* creates the row on first use, so on a fresh install the ordinary sequence is: dispatcher
|
|
1741
|
+
* starts, looks, finds nothing — then somebody submits the first workflow. Caching that first
|
|
1742
|
+
* `null` would blind this process to every graph until it was restarted, with each poll reporting
|
|
1743
|
+
* a clean, empty sweep. The row is created once and never removed, so the retry costs one
|
|
1744
|
+
* `MaxRows: 1` lookup per poll for exactly as long as there is genuinely nothing to dispatch.
|
|
1745
|
+
*/
|
|
1746
|
+
async workflowTaskTypeID(provider) {
|
|
1747
|
+
if (this.cachedWorkflowTaskTypeID)
|
|
1748
|
+
return this.cachedWorkflowTaskTypeID;
|
|
1749
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
1750
|
+
EntityName: 'MJ: Task Types',
|
|
1751
|
+
ExtraFilter: `Name='${TASK_TYPE_NAME}'`,
|
|
1752
|
+
Fields: ['ID'],
|
|
1753
|
+
// Ordered, and reading two (R2-7). An unordered `MaxRows: 1` against two rows sharing
|
|
1754
|
+
// the name lets this instance bind a different ID than `Submit` did — after which
|
|
1755
|
+
// every graph the other stamped is invisible to all three sweep arms here.
|
|
1756
|
+
OrderBy: '__mj_CreatedAt ASC, ID ASC',
|
|
1757
|
+
ResultType: 'simple',
|
|
1758
|
+
MaxRows: 2,
|
|
1759
|
+
}, this.contextUser);
|
|
1760
|
+
if (!result.Success) {
|
|
1761
|
+
// A failed lookup is not "no such type" — saying so would silently skip a poll cycle's
|
|
1762
|
+
// worth of real work. Report it, and let the next cycle ask again.
|
|
1763
|
+
LogError(`[TaskGraphDispatcher] Could not resolve the '${TASK_TYPE_NAME}' task type: ${result.ErrorMessage}`);
|
|
1764
|
+
return null;
|
|
1765
|
+
}
|
|
1766
|
+
const rows = result.Results ?? [];
|
|
1767
|
+
if (rows.length > 1) {
|
|
1768
|
+
LogError(`[TaskGraphDispatcher] More than one '${TASK_TYPE_NAME}' task type exists. Binding the ` +
|
|
1769
|
+
`oldest (${rows[0].ID}); any graph stamped with the other is invisible to this sweep and ` +
|
|
1770
|
+
`will never settle. Merge them.`);
|
|
1771
|
+
}
|
|
1772
|
+
this.cachedWorkflowTaskTypeID = rows[0]?.ID ?? null;
|
|
1773
|
+
return this.cachedWorkflowTaskTypeID;
|
|
1774
|
+
}
|
|
1775
|
+
async findActiveGraphIDs(provider, windowHours = UNSETTLED_SWEEP_WINDOW_HOURS) {
|
|
547
1776
|
const rv = RunView.FromMetadataProvider(provider);
|
|
1777
|
+
// EVERY arm is scoped to workflow graphs. Unscoped, the dispatcher rewrites tasks that are
|
|
1778
|
+
// none of its business — see workflowTaskTypeID.
|
|
1779
|
+
const typeID = await this.workflowTaskTypeID(provider);
|
|
1780
|
+
if (!typeID)
|
|
1781
|
+
return [];
|
|
1782
|
+
const ofWorkflowType = `TypeID='${typeID}'`;
|
|
548
1783
|
// TWO queries, because "has work left to do" and "needs attention" are not the same set.
|
|
549
1784
|
//
|
|
550
1785
|
// Selecting only graphs with non-terminal CHILDREN looks right and is subtly fatal: the
|
|
@@ -556,23 +1791,50 @@ export class TaskGraphDispatcher {
|
|
|
556
1791
|
//
|
|
557
1792
|
// The second query closes it: a parent that is itself non-terminal still needs looking at,
|
|
558
1793
|
// whatever its children are doing.
|
|
559
|
-
|
|
1794
|
+
// THREE queries. The third rescues a graph that reached terminal without settling.
|
|
1795
|
+
//
|
|
1796
|
+
// The post-settlement sequence — cost rollup, run settlement, continuation delivery — runs
|
|
1797
|
+
// AFTER the parent's terminal write, and a terminal parent with all-terminal children
|
|
1798
|
+
// matches neither query above. So a process that died in that window left the submitting
|
|
1799
|
+
// agent run `Paused` FOREVER: no rollup, no notification, and nothing that would ever look
|
|
1800
|
+
// again. The metadata's own doc comment promised "the next sweep retries"; that sweep did
|
|
1801
|
+
// not exist.
|
|
1802
|
+
//
|
|
1803
|
+
// Bounded rather than unbounded, because the marker lives in `InputPayload` JSON and cannot
|
|
1804
|
+
// be filtered in SQL: the window is what keeps this a targeted rescue instead of a re-parse
|
|
1805
|
+
// of every graph ever run. `__mj_UpdatedAt` advances on each settle attempt, so a graph
|
|
1806
|
+
// being actively retried stays in the window — the bound is on ABANDONMENT, not on age.
|
|
1807
|
+
const cutoff = SweepCutoff(new Date(), windowHours);
|
|
1808
|
+
const [withPendingWork, unsettledParents, terminalRecent] = await rv.RunViews([
|
|
560
1809
|
{
|
|
561
1810
|
EntityName: 'MJ: Tasks',
|
|
562
|
-
ExtraFilter:
|
|
1811
|
+
ExtraFilter: `${ofWorkflowType} AND ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
|
|
563
1812
|
Fields: ['ParentID'],
|
|
564
1813
|
ResultType: 'simple',
|
|
565
1814
|
BypassCache: true,
|
|
566
1815
|
},
|
|
567
1816
|
{
|
|
568
1817
|
EntityName: 'MJ: Tasks',
|
|
569
|
-
ExtraFilter:
|
|
1818
|
+
ExtraFilter: `${ofWorkflowType} AND ParentID IS NULL AND Status IN ('Pending','In Progress')`,
|
|
570
1819
|
Fields: ['ID'],
|
|
571
1820
|
ResultType: 'simple',
|
|
572
1821
|
BypassCache: true,
|
|
573
1822
|
},
|
|
1823
|
+
{
|
|
1824
|
+
EntityName: 'MJ: Tasks',
|
|
1825
|
+
ExtraFilter: `${ofWorkflowType} AND ParentID IS NULL AND Status IN (${TERMINAL_PARENT_STATUS_SQL}) ` +
|
|
1826
|
+
`AND __mj_UpdatedAt >= '${cutoff}'`,
|
|
1827
|
+
Fields: ['ID', 'InputPayload'],
|
|
1828
|
+
ResultType: 'simple',
|
|
1829
|
+
BypassCache: true,
|
|
1830
|
+
},
|
|
574
1831
|
], this.contextUser);
|
|
575
1832
|
const ids = new Set();
|
|
1833
|
+
// Graphs this instance is mid-retry on, whatever the window says (R2-12). A failing pass
|
|
1834
|
+
// writes nothing, so their `__mj_UpdatedAt` has stopped advancing and the third arm below
|
|
1835
|
+
// will eventually stop finding them — which would turn a retry into a silent abandonment.
|
|
1836
|
+
for (const id of this.retryingSettlement.keys())
|
|
1837
|
+
ids.add(id);
|
|
576
1838
|
for (const r of (withPendingWork?.Results ?? [])) {
|
|
577
1839
|
if (r.ParentID)
|
|
578
1840
|
ids.add(r.ParentID);
|
|
@@ -583,6 +1845,11 @@ export class TaskGraphDispatcher {
|
|
|
583
1845
|
if (r.ID)
|
|
584
1846
|
ids.add(r.ID);
|
|
585
1847
|
}
|
|
1848
|
+
// The marker is JSON, so the filter is in TypeScript rather than in SQL — see
|
|
1849
|
+
// SelectUnsettledGraphIDs, which owns that decision and is tested directly.
|
|
1850
|
+
for (const id of SelectUnsettledGraphIDs((terminalRecent?.Results ?? []))) {
|
|
1851
|
+
ids.add(id);
|
|
1852
|
+
}
|
|
586
1853
|
return [...ids];
|
|
587
1854
|
}
|
|
588
1855
|
/**
|
|
@@ -594,11 +1861,100 @@ export class TaskGraphDispatcher {
|
|
|
594
1861
|
*/
|
|
595
1862
|
async findClaimableTasks(provider, limit) {
|
|
596
1863
|
const claimable = [];
|
|
597
|
-
|
|
1864
|
+
const stats = new Map();
|
|
1865
|
+
const activeGraphs = await this.findActiveGraphIDs(provider);
|
|
1866
|
+
// Usually a no-op — propagateAndRollup primed these earlier in the same pass.
|
|
1867
|
+
await this.primeDebugStates(provider, activeGraphs);
|
|
1868
|
+
for (const parentID of activeGraphs) {
|
|
598
1869
|
if (claimable.length >= limit)
|
|
599
1870
|
break;
|
|
600
|
-
const
|
|
601
|
-
|
|
1871
|
+
const debug = await this.readDebugState(provider, parentID);
|
|
1872
|
+
await this.announcePauseTransition(provider, parentID, debug);
|
|
1873
|
+
const graph = await this.loadGraphState(provider, parentID, debug);
|
|
1874
|
+
// HOLD is what makes "a broken condition stalls visibly" true rather than merely stated.
|
|
1875
|
+
// An undecided exclusive group keeps all its edges, and a kept edge on a Complete origin
|
|
1876
|
+
// is a SATISFIED prerequisite — so without this filter every branch of the fork would be
|
|
1877
|
+
// eligible at once and all of them would run. A typo must not multiply a fork.
|
|
1878
|
+
//
|
|
1879
|
+
// The losers of a DECIDED group must be filtered for the same reason, and this is a race
|
|
1880
|
+
// rather than a rule: they are marked Skipped by the propagation pass, but between the
|
|
1881
|
+
// moment the group resolves and the moment that write lands, their incoming edge is still
|
|
1882
|
+
// a satisfied prerequisite on a Complete origin. A poll landing in that window would
|
|
1883
|
+
// claim and execute the branch the workflow chose NOT to take — irreversibly, since the
|
|
1884
|
+
// action has already run by the time Skipped is written over it.
|
|
1885
|
+
// `unreachableTaskIDs` joins the filter for exactly the reason above. R6 made a
|
|
1886
|
+
// definite-false ordinary edge seed the skip cascade rather than Block its target — but
|
|
1887
|
+
// until that Skipped write lands, the target has no unsatisfied prerequisite and is
|
|
1888
|
+
// vacuously eligible. That is the same race the XOR fix closed, reopened on the new
|
|
1889
|
+
// path: a branch the workflow decided against, claimed and executed irreversibly in the
|
|
1890
|
+
// window before it was marked.
|
|
1891
|
+
const eligible = ComputeEligibleTasks(graph.nodes, graph.edges, graph.handledFailureIDs)
|
|
1892
|
+
.filter((n) => !graph.holdTaskIDs.has(n.id) &&
|
|
1893
|
+
// CONFIRMED seeds, not raw ones (P1). Holding a decided loser out of claiming
|
|
1894
|
+
// closes a real race — the loser could be claimed between eligibility and the
|
|
1895
|
+
// skip write — and that role is unchanged. What changed is which targets count
|
|
1896
|
+
// as decided: a task another live route still reaches was never a loser, so it
|
|
1897
|
+
// must stay claimable and run when its own prerequisites are met.
|
|
1898
|
+
!graph.skipSeedTaskIDs.has(n.id) &&
|
|
1899
|
+
!graph.unreachableTaskIDs.has(n.id) &&
|
|
1900
|
+
// ...and everything the cascade is about to reach (R2-14). A descendant of a
|
|
1901
|
+
// seed is eligible for the moments between its ancestor's skip landing and its
|
|
1902
|
+
// own, because Skipped satisfies prerequisites — a window another instance can
|
|
1903
|
+
// and does claim inside.
|
|
1904
|
+
!graph.cascadeSkipTaskIDs.has(n.id));
|
|
1905
|
+
stats.set(parentID, { eligible: eligible.length, held: graph.holdTaskIDs.size });
|
|
1906
|
+
// THE DEBUG GATE — pause, single-step, breakpoints — decided by the pure function, with
|
|
1907
|
+
// the CAS writes staying here. Every control is a gate on CLAIMING: a claimed task can
|
|
1908
|
+
// never be interrupted mid-flight anyway, so "paused" means nothing new starts while
|
|
1909
|
+
// in-flight work finishes and its completions land. That is also why the gate sits
|
|
1910
|
+
// BEFORE the runner checks and the human notification below: pausing a graph must not
|
|
1911
|
+
// keep notifying assignees — a notification is starting something.
|
|
1912
|
+
const gate = DecideClaimGate(debug, eligible.map((n) => n.id));
|
|
1913
|
+
if (gate.mode === 'closed')
|
|
1914
|
+
continue;
|
|
1915
|
+
let allowedTaskIDs = null;
|
|
1916
|
+
if (gate.mode === 'breakpoint') {
|
|
1917
|
+
const typeID = await this.workflowTaskTypeID(provider);
|
|
1918
|
+
// The pause is a CAS so two instances arriving at the same breakpoint in the same
|
|
1919
|
+
// interval produce one announcement — the loser simply sees a paused graph next pass.
|
|
1920
|
+
if (typeID && await this.claims.TryPauseAtBreakpoint(provider, parentID, gate.taskID, typeID, this.contextUser)) {
|
|
1921
|
+
const owner = await this.resolveOwner(provider, parentID);
|
|
1922
|
+
const name = graph.entityById.get(gate.taskID)?.Name;
|
|
1923
|
+
LogStatus(`[TaskGraphDispatcher] Graph ${parentID} paused at breakpoint on '${name}' (${gate.taskID}).`);
|
|
1924
|
+
this.emit({ Kind: 'BreakpointHit', ParentTaskID: parentID, OwnerUserID: owner, TaskID: gate.taskID, TaskName: name });
|
|
1925
|
+
this.emit({ Kind: 'GraphPaused', ParentTaskID: parentID, OwnerUserID: owner, TaskID: gate.taskID, Reason: 'breakpoint' });
|
|
1926
|
+
this.announcedPaused.set(parentID, true);
|
|
1927
|
+
}
|
|
1928
|
+
continue;
|
|
1929
|
+
}
|
|
1930
|
+
if (gate.mode === 'step') {
|
|
1931
|
+
allowedTaskIDs = new Set(gate.taskIDs);
|
|
1932
|
+
// THE ALLOWANCE IS CONSUMED ONLY IF SOMETHING WILL ACTUALLY MOVE.
|
|
1933
|
+
//
|
|
1934
|
+
// "Eligible" is a graph-shape answer; whether this host can run the step is a
|
|
1935
|
+
// separate one, decided below by the runner checks. Consuming first meant a step
|
|
1936
|
+
// onto a node this instance has no runner for — or one already in flight — spent
|
|
1937
|
+
// the allowance and released nothing, leaving the operator pressing a button that
|
|
1938
|
+
// did nothing and no reason anywhere. Deciding first costs one pre-pass over a set
|
|
1939
|
+
// that is at most the frontier.
|
|
1940
|
+
const releasable = eligible.filter((n) => {
|
|
1941
|
+
const entity = graph.entityById.get(n.id);
|
|
1942
|
+
return entity ? allowedTaskIDs.has(n.id) && this.canActOn(entity) : false;
|
|
1943
|
+
});
|
|
1944
|
+
if (releasable.length === 0) {
|
|
1945
|
+
await this.reportStepReleasedNothing(provider, parentID, gate.taskIDs, graph);
|
|
1946
|
+
continue;
|
|
1947
|
+
}
|
|
1948
|
+
const typeID = await this.workflowTaskTypeID(provider);
|
|
1949
|
+
// Consuming the allowance is the race: exactly one instance clears the marker and
|
|
1950
|
+
// releases work; the loser waits for the next allowance. A lost consume is normal.
|
|
1951
|
+
if (!typeID || !(await this.claims.TryConsumeStepMarker(provider, parentID, typeID, this.contextUser))) {
|
|
1952
|
+
continue;
|
|
1953
|
+
}
|
|
1954
|
+
}
|
|
1955
|
+
for (const node of eligible) {
|
|
1956
|
+
if (allowedTaskIDs && !allowedTaskIDs.has(node.id))
|
|
1957
|
+
continue;
|
|
602
1958
|
const entity = graph.entityById.get(node.id);
|
|
603
1959
|
if (!entity)
|
|
604
1960
|
continue;
|
|
@@ -608,7 +1964,36 @@ export class TaskGraphDispatcher {
|
|
|
608
1964
|
// they cleared. Without a notification here a workflow simply stops, waiting on
|
|
609
1965
|
// someone who was never told. That silent stall is the failure mode this exists to
|
|
610
1966
|
// prevent, so it happens on the eligibility check rather than at submission.
|
|
611
|
-
if (
|
|
1967
|
+
if (entity.ActionID) {
|
|
1968
|
+
// An action node this host has no runner for is left Pending rather than
|
|
1969
|
+
// claimed. Claiming it would take ownership of work this process cannot do, and
|
|
1970
|
+
// the claim would then have to expire before any host that CAN do it gets a
|
|
1971
|
+
// turn — a self-inflicted stall on a mixed deployment.
|
|
1972
|
+
if (!this.actionRunner)
|
|
1973
|
+
continue;
|
|
1974
|
+
}
|
|
1975
|
+
else if (entity.PromptID) {
|
|
1976
|
+
// A prompt node — including a loop that repeats a prompt — is assigned through
|
|
1977
|
+
// PromptID and carries NEITHER ActionID nor AgentID. Without this branch it fell
|
|
1978
|
+
// through to the test below and was treated as a task waiting on a PERSON: the
|
|
1979
|
+
// workflow notified a human who had nothing to do and then stopped forever.
|
|
1980
|
+
// That is precisely the misclassification the step-kind rules warn about, and it
|
|
1981
|
+
// is silent — the graph sits In Progress looking like it is still working.
|
|
1982
|
+
if (!this.promptRunner)
|
|
1983
|
+
continue;
|
|
1984
|
+
}
|
|
1985
|
+
else if (!entity.AgentID) {
|
|
1986
|
+
// No runner column at all — a person completes this one. Asked through the same
|
|
1987
|
+
// predicate the human settle/expiry sweeps use, so a task that gets NOTIFIED here
|
|
1988
|
+
// is a task those sweeps can later see; the two disagreeing is how a human task
|
|
1989
|
+
// ends up asked and then never settled.
|
|
1990
|
+
if (!IsHumanTask(entity)) {
|
|
1991
|
+
// Neither a runner nor a person: nothing can ever move this. Loud, because
|
|
1992
|
+
// the alternative is a graph that waits forever on nobody.
|
|
1993
|
+
LogError(`[TaskGraphDispatcher] Task '${entity.Name}' (${entity.ID}) has no runner ` +
|
|
1994
|
+
`assignment and is not a human step — nothing can execute it. The graph will stall.`);
|
|
1995
|
+
continue;
|
|
1996
|
+
}
|
|
612
1997
|
await this.notifyHumanTaskReady(entity, provider);
|
|
613
1998
|
continue;
|
|
614
1999
|
}
|
|
@@ -618,22 +2003,444 @@ export class TaskGraphDispatcher {
|
|
|
618
2003
|
if (claimable.length >= limit)
|
|
619
2004
|
break;
|
|
620
2005
|
}
|
|
2006
|
+
if (debug.skipBreakpointTaskID) {
|
|
2007
|
+
const skip = debug.skipBreakpointTaskID;
|
|
2008
|
+
const claimedThisPass = claimable.some((t) => UUIDsEqual(t.ID, skip));
|
|
2009
|
+
const stillEligible = eligible.some((n) => UUIDsEqual(n.id, skip));
|
|
2010
|
+
if (claimedThisPass || !stillEligible) {
|
|
2011
|
+
await this.clearSkipBreakpoint(provider, parentID);
|
|
2012
|
+
}
|
|
2013
|
+
}
|
|
2014
|
+
}
|
|
2015
|
+
return { tasks: claimable, stats };
|
|
2016
|
+
}
|
|
2017
|
+
async clearSkipBreakpoint(provider, parentTaskID) {
|
|
2018
|
+
const typeID = await this.workflowTaskTypeID(provider);
|
|
2019
|
+
if (!typeID)
|
|
2020
|
+
return;
|
|
2021
|
+
await this.claims.TryWriteDebugFields(provider, parentTaskID, [TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' })], typeID, this.contextUser);
|
|
2022
|
+
}
|
|
2023
|
+
/**
|
|
2024
|
+
* Drops the per-graph frame-dedup state for a graph that has settled.
|
|
2025
|
+
*
|
|
2026
|
+
* `ownerByParentID` is deliberately NOT purged here: it is the delivery key for the
|
|
2027
|
+
* `GraphSettled` frame emitted moments earlier and for any rescue-sweep pass that revisits the
|
|
2028
|
+
* graph, it is one small string per graph, and ownership never changes — the cost of keeping it
|
|
2029
|
+
* is bounded and the cost of losing it is a re-query on a path that is meant to be cheap.
|
|
2030
|
+
*/
|
|
2031
|
+
forgetGraphObservability(parentTaskID) {
|
|
2032
|
+
this.emittedGateVerdicts.delete(parentTaskID);
|
|
2033
|
+
this.nodeProgressLastEmit.delete(parentTaskID);
|
|
2034
|
+
this.announcedPaused.delete(parentTaskID);
|
|
2035
|
+
this.debugStateByGraph.delete(parentTaskID);
|
|
2036
|
+
}
|
|
2037
|
+
/**
|
|
2038
|
+
* Whether THIS host can act on a task right now — the runner-availability question, asked
|
|
2039
|
+
* without acting on it.
|
|
2040
|
+
*
|
|
2041
|
+
* Mirrors the checks in the claim loop so a step allowance is spent only when something will
|
|
2042
|
+
* actually move. A human step counts as actionable: stepping onto one legitimately produces a
|
|
2043
|
+
* notification rather than a claim.
|
|
2044
|
+
*/
|
|
2045
|
+
canActOn(entity) {
|
|
2046
|
+
if (this.inFlight.has(entity.ID))
|
|
2047
|
+
return false;
|
|
2048
|
+
if (entity.ActionID)
|
|
2049
|
+
return !!this.actionRunner;
|
|
2050
|
+
if (entity.PromptID)
|
|
2051
|
+
return !!this.promptRunner;
|
|
2052
|
+
if (entity.AgentID)
|
|
2053
|
+
return true;
|
|
2054
|
+
return IsHumanTask(entity);
|
|
2055
|
+
}
|
|
2056
|
+
/**
|
|
2057
|
+
* Says why a step press released nothing, instead of leaving the allowance spent and the
|
|
2058
|
+
* console silent.
|
|
2059
|
+
*
|
|
2060
|
+
* The allowance is deliberately NOT consumed on this path — the operator's intent stands, and
|
|
2061
|
+
* the step will release as soon as the named work becomes actionable (a runner arrives, an
|
|
2062
|
+
* in-flight task finishes). Announced once per pass rather than logged only, because the person
|
|
2063
|
+
* waiting is looking at the console, not the server log.
|
|
2064
|
+
*/
|
|
2065
|
+
async reportStepReleasedNothing(provider, parentTaskID, requestedTaskIDs, graph) {
|
|
2066
|
+
const names = requestedTaskIDs
|
|
2067
|
+
.map((id) => graph.entityById.get(id)?.Name)
|
|
2068
|
+
.filter((n) => !!n);
|
|
2069
|
+
const subject = names.length > 0 ? `"${names.join('", "')}"` : 'the next step';
|
|
2070
|
+
const reason = `Step is still waiting: ${subject} cannot start on this server yet — it is already ` +
|
|
2071
|
+
`running, or no runner for that step type is loaded here. The step will release as ` +
|
|
2072
|
+
`soon as it can; nothing was lost.`;
|
|
2073
|
+
LogStatus(`[TaskGraphDispatcher] Step on graph ${parentTaskID} released nothing: ${reason}`);
|
|
2074
|
+
this.emit({
|
|
2075
|
+
Kind: 'StepRefused',
|
|
2076
|
+
ParentTaskID: parentTaskID,
|
|
2077
|
+
OwnerUserID: await this.resolveOwner(provider, parentTaskID),
|
|
2078
|
+
TaskID: requestedTaskIDs[0],
|
|
2079
|
+
TaskName: names[0],
|
|
2080
|
+
Reason: reason,
|
|
2081
|
+
});
|
|
2082
|
+
}
|
|
2083
|
+
/**
|
|
2084
|
+
* Marks a human task as notified, so the request is raised exactly once.
|
|
2085
|
+
*
|
|
2086
|
+
* Written even when delivery threw. Retrying on every poll is a worse failure than one missed
|
|
2087
|
+
* notification: the task stays visible in the inbox either way, whereas a notification storm is
|
|
2088
|
+
* not self-correcting.
|
|
2089
|
+
*/
|
|
2090
|
+
async markHumanTaskNotified(task, provider) {
|
|
2091
|
+
// Guarded, not a full-row save against a snapshot (R3-5). This row was loaded at the top of
|
|
2092
|
+
// the pass; a full-row write could revert a status it has reached since, and two instances
|
|
2093
|
+
// could both stamp it after both having seen it absent. The predicate makes it once-only.
|
|
2094
|
+
if (await this.claims.TryMarkHumanNotified(provider, task.ID, HUMAN_TASK_NOTIFIED_MARKER, this.contextUser)) {
|
|
2095
|
+
task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
|
|
2096
|
+
return;
|
|
2097
|
+
}
|
|
2098
|
+
// Rowcount 0 is ordinary: another instance marked it, or the task is no longer Pending.
|
|
2099
|
+
// Either way this instance has nothing left to do about the notification.
|
|
2100
|
+
}
|
|
2101
|
+
/**
|
|
2102
|
+
* Raises the `MJ: AI Agent Requests` row a person answers to release this step.
|
|
2103
|
+
*
|
|
2104
|
+
* **Why that entity rather than something new.** It already models everything a workflow's human
|
|
2105
|
+
* step needs — who is being asked, what for, a typed response schema, priority, expiry, and an
|
|
2106
|
+
* inbox surface people already use. A second HITL substrate beside it would split the inbox in
|
|
2107
|
+
* two and leave one of them without expiry or permissions.
|
|
2108
|
+
*
|
|
2109
|
+
* **What it deliberately does NOT set is `ResumingAgentRunID`.** A request normally suspends an
|
|
2110
|
+
* agent run and resumes it. A workflow needs none of that: the graph OUTLIVES the run that
|
|
2111
|
+
* submitted it, so nothing is suspended — the task sits Pending, every other branch keeps
|
|
2112
|
+
* running, and answering settles the task. That column staying null is meaningful, not missing.
|
|
2113
|
+
*/
|
|
2114
|
+
async raiseHumanRequest(task, provider) {
|
|
2115
|
+
try {
|
|
2116
|
+
const existing = await this.findOpenRequests(provider, task.ID);
|
|
2117
|
+
if (existing.length > 0) {
|
|
2118
|
+
// Somebody IS waiting on this task — but "somebody" may be two rows, so collapse
|
|
2119
|
+
// before returning. Free: the rows are already in hand.
|
|
2120
|
+
await this.withdrawDuplicateRequests(provider, task.ID, existing, existing[0].ID);
|
|
2121
|
+
return 'raised';
|
|
2122
|
+
}
|
|
2123
|
+
const request = await provider.GetEntityObject('MJ: AI Agent Requests', this.contextUser);
|
|
2124
|
+
request.NewRecord();
|
|
2125
|
+
request.OriginatingTaskID = task.ID;
|
|
2126
|
+
// A human task has NO AgentID of its own — that column names what EXECUTES a step, and
|
|
2127
|
+
// a person is not an agent. The request still needs one, so it carries the agent that
|
|
2128
|
+
// owns the workflow: the graph's own agent, which is who is asking.
|
|
2129
|
+
const owningAgentID = await this.owningAgentOf(provider, task);
|
|
2130
|
+
if (!owningAgentID) {
|
|
2131
|
+
// PERMANENT: a graph with no owning agent will not acquire one by being asked
|
|
2132
|
+
// again. Graphs submitted before the provenance stamp landed are all in this state.
|
|
2133
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} needs a person, but its workflow has no ` +
|
|
2134
|
+
`agent to ask on behalf of, so no request can be raised. The task stays Pending ` +
|
|
2135
|
+
`and visible in the Tasks UI; it will not be retried.`);
|
|
2136
|
+
return 'permanent-failure';
|
|
2137
|
+
}
|
|
2138
|
+
request.AgentID = owningAgentID;
|
|
2139
|
+
request.RequestForUserID = task.UserID;
|
|
2140
|
+
request.RequestedAt = new Date();
|
|
2141
|
+
request.Status = 'Requested';
|
|
2142
|
+
request.Request = task.Description || `A workflow is waiting on you to complete "${task.Name}".`;
|
|
2143
|
+
// The graph's own run is the provenance a reader follows back to see what led here.
|
|
2144
|
+
request.OriginatingAgentRunID = await this.submittingRunOf(provider, task);
|
|
2145
|
+
// The deadline, when the author set one. `expireOverdueRequests` has always been able to
|
|
2146
|
+
// enforce this — it expires the request and fails the step so a give-up edge can route
|
|
2147
|
+
// around it — but nothing ever WROTE the column, so that whole path had never run outside
|
|
2148
|
+
// a test and a workflow waiting on someone who left the company waited forever.
|
|
2149
|
+
// Absent means no deadline, deliberately: expiring on a timeout nobody chose would be
|
|
2150
|
+
// worse than waiting.
|
|
2151
|
+
const expiresInHours = this.parseConfiguration(task)?.human?.expiresInHours;
|
|
2152
|
+
if (expiresInHours && expiresInHours > 0) {
|
|
2153
|
+
request.ExpiresAt = new Date(Date.now() + expiresInHours * 60 * 60 * 1000);
|
|
2154
|
+
}
|
|
2155
|
+
if (await request.Save()) {
|
|
2156
|
+
// INSERT-THEN-RESELECT (R3-5). The check above is read-then-write in a system whose
|
|
2157
|
+
// every other cross-instance write is a CAS, and there is no unique index behind it —
|
|
2158
|
+
// so two overlapping instances both read "none open", both insert, and both ping the
|
|
2159
|
+
// assignee. When one is answered, `settleAnsweredHumanTasks` settles from the single
|
|
2160
|
+
// latest terminal request and NOTHING ever touches the other: the withdrawal paths
|
|
2161
|
+
// fire only on skips and cancels, and both human sweeps scope to `Pending` tasks,
|
|
2162
|
+
// which the settled task no longer is. The duplicate becomes a durable, unanswerable,
|
|
2163
|
+
// immortal inbox item — the zombie class R2-10 removed from the skip paths.
|
|
2164
|
+
//
|
|
2165
|
+
// Checking harder is what created the window, so the resolution is to let both
|
|
2166
|
+
// inserts happen and then agree on a winner: the oldest open row. A loser withdraws
|
|
2167
|
+
// its own row and returns `raised` — somebody IS waiting on this task, which is what
|
|
2168
|
+
// the caller needs to know.
|
|
2169
|
+
await this.withdrawDuplicateRequests(provider, task.ID, await this.findOpenRequests(provider, task.ID), request.ID);
|
|
2170
|
+
return 'raised';
|
|
2171
|
+
}
|
|
2172
|
+
{
|
|
2173
|
+
LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ` +
|
|
2174
|
+
`${request.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
2175
|
+
// A failed SAVE may be transient (deadlock, contention), so this one earns a retry.
|
|
2176
|
+
return 'transient-failure';
|
|
2177
|
+
}
|
|
2178
|
+
return 'raised';
|
|
2179
|
+
}
|
|
2180
|
+
catch (e) {
|
|
2181
|
+
// Never fatal. The task remains Pending and visible; a missing request is recoverable,
|
|
2182
|
+
// whereas throwing here would abort the whole dispatch pass for every other branch.
|
|
2183
|
+
LogError(`[TaskGraphDispatcher] Could not raise a request for task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
2184
|
+
return 'transient-failure';
|
|
2185
|
+
}
|
|
2186
|
+
}
|
|
2187
|
+
/**
|
|
2188
|
+
* The agent that owns this task's workflow — who the request is asked on behalf of.
|
|
2189
|
+
*
|
|
2190
|
+
* Reads the graph's parent row, falling back to the run that submitted it. A human step has no
|
|
2191
|
+
* agent of its own by design: `AgentID` names what EXECUTES a step, and a person is not an agent.
|
|
2192
|
+
*/
|
|
2193
|
+
async owningAgentOf(provider, task) {
|
|
2194
|
+
if (task.AgentID)
|
|
2195
|
+
return task.AgentID;
|
|
2196
|
+
if (!task.ParentID)
|
|
2197
|
+
return null;
|
|
2198
|
+
try {
|
|
2199
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
2200
|
+
if (!(await parent.Load(task.ParentID)))
|
|
2201
|
+
return null;
|
|
2202
|
+
if (parent.AgentID)
|
|
2203
|
+
return parent.AgentID;
|
|
2204
|
+
if (!parent.AgentRunID)
|
|
2205
|
+
return null;
|
|
2206
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
2207
|
+
return (await run.Load(parent.AgentRunID)) ? run.AgentID : null;
|
|
2208
|
+
}
|
|
2209
|
+
catch {
|
|
2210
|
+
return null;
|
|
2211
|
+
}
|
|
2212
|
+
}
|
|
2213
|
+
/**
|
|
2214
|
+
* Every still-open request for a task, oldest first.
|
|
2215
|
+
*
|
|
2216
|
+
* Plural, and ordered, for one reason each. Ordered, because the oldest row is the one every
|
|
2217
|
+
* instance must agree is "the" request — it is the one the assignee most likely already saw,
|
|
2218
|
+
* and the one `withdrawDuplicateRequests` keeps; unordered, two instances could each decide a
|
|
2219
|
+
* different duplicate was the keeper and withdraw each other's. Plural, because a caller that
|
|
2220
|
+
* only ever sees the first cannot notice there are two, which is how the duplicate below
|
|
2221
|
+
* survived: every reader of this took `[0]` and moved on.
|
|
2222
|
+
*/
|
|
2223
|
+
async findOpenRequests(provider, taskID) {
|
|
2224
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
2225
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
2226
|
+
ExtraFilter: `OriginatingTaskID='${taskID}' AND Status='Requested'`,
|
|
2227
|
+
OrderBy: '__mj_CreatedAt ASC, ID ASC',
|
|
2228
|
+
ResultType: 'entity_object',
|
|
2229
|
+
BypassCache: true,
|
|
2230
|
+
}, this.contextUser);
|
|
2231
|
+
return (result.Success ? result.Results : null) ?? [];
|
|
2232
|
+
}
|
|
2233
|
+
/**
|
|
2234
|
+
* Settles a human task from the request a person answered.
|
|
2235
|
+
*
|
|
2236
|
+
* Runs on the poll rather than on a save hook, because the answer can arrive through any surface
|
|
2237
|
+
* — the inbox, the API, a conversation — and only the dispatcher knows how to release the rest
|
|
2238
|
+
* of the graph afterwards.
|
|
2239
|
+
*
|
|
2240
|
+
* **`ResponseData` becomes the task's output.** That is what makes a human step useful rather
|
|
2241
|
+
* than a gate: a downstream edge can branch on what the person actually said, typed by the
|
|
2242
|
+
* request's own ResponseSchema. A step that only recorded "approved" would force every decision
|
|
2243
|
+
* back into a separate action.
|
|
2244
|
+
*/
|
|
2245
|
+
async settleAnsweredHumanTasks(provider, graphID) {
|
|
2246
|
+
const waiting = await RunView.FromMetadataProvider(provider).RunView({
|
|
2247
|
+
EntityName: 'MJ: Tasks',
|
|
2248
|
+
ExtraFilter: `ParentID='${graphID}' AND ${HumanTaskSQL()} AND Status='Pending'`,
|
|
2249
|
+
ResultType: 'entity_object',
|
|
2250
|
+
BypassCache: true,
|
|
2251
|
+
}, this.contextUser);
|
|
2252
|
+
if (!waiting.Success)
|
|
2253
|
+
return;
|
|
2254
|
+
for (const task of waiting.Results ?? []) {
|
|
2255
|
+
const request = await this.answeredRequestFor(provider, task.ID);
|
|
2256
|
+
if (!request)
|
|
2257
|
+
continue;
|
|
2258
|
+
const rejected = request.Status === 'Rejected';
|
|
2259
|
+
const expired = request.Status === 'Expired';
|
|
2260
|
+
task.Status = rejected || expired ? 'Failed' : 'Complete';
|
|
2261
|
+
task.CompletedAt = new Date();
|
|
2262
|
+
task.PercentComplete = rejected || expired ? 0 : 100;
|
|
2263
|
+
task.ClaimedBy = null;
|
|
2264
|
+
task.ClaimExpiresAt = null;
|
|
2265
|
+
task.OutputPayload = request.ResponseData ?? null;
|
|
2266
|
+
if (rejected) {
|
|
2267
|
+
task.ErrorMessage = request.Comments || 'A person rejected this step.';
|
|
2268
|
+
}
|
|
2269
|
+
else if (expired) {
|
|
2270
|
+
// Stated as a failure rather than left Pending. A workflow blocked forever on
|
|
2271
|
+
// someone who never answered — who may have left the company — is the silent stall
|
|
2272
|
+
// this whole path exists to avoid, and a give-up edge can now route around it.
|
|
2273
|
+
task.ErrorMessage = 'Nobody answered this step before its request expired.';
|
|
2274
|
+
}
|
|
2275
|
+
if (!(await task.Save())) {
|
|
2276
|
+
LogError(`[TaskGraphDispatcher] Could not settle human task ${task.ID}: ` +
|
|
2277
|
+
`${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
2278
|
+
continue;
|
|
2279
|
+
}
|
|
2280
|
+
// WITHDRAW EVERY OTHER OPEN ASK FOR THIS STEP (R3-5).
|
|
2281
|
+
//
|
|
2282
|
+
// The step is terminal now, so both human sweeps — which scope to `Pending` tasks — will
|
|
2283
|
+
// never look at it again, and any request still `Requested` is un-answerable and
|
|
2284
|
+
// immortal: the assignee keeps seeing "a workflow is waiting on you" for a step that is
|
|
2285
|
+
// finished. Duplicates only arose from the raise race fixed above, but this also
|
|
2286
|
+
// retroactively cleans the ones already minted, which the raise-side fix cannot reach.
|
|
2287
|
+
await this.withdrawOpenRequests(provider, [task.ID], 'This step has been settled; the request is no longer open.');
|
|
2288
|
+
}
|
|
2289
|
+
}
|
|
2290
|
+
/**
|
|
2291
|
+
* Reconciles the requests behind human steps that are waiting on somebody.
|
|
2292
|
+
*
|
|
2293
|
+
* Two things can be wrong with a waiting step, and both are silent. It can have NO open request
|
|
2294
|
+
* — the cancel case below — or it can have MORE than one, which the raise cannot fix because it
|
|
2295
|
+
* never runs again for a notified task. Both are corrected here, on the only sweep that visits
|
|
2296
|
+
* these tasks every pass.
|
|
2297
|
+
*
|
|
2298
|
+
* **Re-opening a human step whose request was CANCELLED.**
|
|
2299
|
+
*
|
|
2300
|
+
* `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
|
|
2301
|
+
* rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
|
|
2302
|
+
* it". Nothing replaced it. `raiseHumanRequest` refuses to raise twice (the notified marker on
|
|
2303
|
+
* `ClaimedBy` is what stops the notification storm), so a cancelled request left the task Pending
|
|
2304
|
+
* with no open request and no path to acquiring one: a workflow waiting forever on a question
|
|
2305
|
+
* nobody is being asked.
|
|
2306
|
+
*
|
|
2307
|
+
* Clearing the marker is the whole fix — the next poll sees an un-notified Pending human task
|
|
2308
|
+
* and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
|
|
2309
|
+
* human action: it takes another person cancelling again to come back here.
|
|
2310
|
+
*/
|
|
2311
|
+
async reconcileWaitingHumanTasks(provider, graphID) {
|
|
2312
|
+
const waiting = await RunView.FromMetadataProvider(provider).RunView({
|
|
2313
|
+
EntityName: 'MJ: Tasks',
|
|
2314
|
+
// `StepType` is NULLABLE, and rows predating the column exist (4 in the reference
|
|
2315
|
+
// database at the time of writing). None currently carry a UserID, but a human task
|
|
2316
|
+
// written by any path that set the assignee without the discriminator would be
|
|
2317
|
+
// invisible to a `StepType='Human'` filter and stay dead forever after a cancel —
|
|
2318
|
+
// the exact stall this method exists to end. The notified marker already narrows
|
|
2319
|
+
// this to tasks the dispatcher raised a request for, so the widening cannot pull in
|
|
2320
|
+
// unrelated work.
|
|
2321
|
+
ExtraFilter: `ParentID='${graphID}' AND Status='Pending' ` +
|
|
2322
|
+
`AND ${HumanTaskSQL()} ` +
|
|
2323
|
+
`AND ClaimedBy='${HUMAN_TASK_NOTIFIED_MARKER}'`,
|
|
2324
|
+
ResultType: 'entity_object',
|
|
2325
|
+
BypassCache: true,
|
|
2326
|
+
}, this.contextUser);
|
|
2327
|
+
if (!waiting.Success)
|
|
2328
|
+
return;
|
|
2329
|
+
for (const task of waiting.Results ?? []) {
|
|
2330
|
+
// Only when there is nothing live AND nothing terminal. A task with an open request is
|
|
2331
|
+
// simply waiting; one with a terminal request is settled on the next pass by
|
|
2332
|
+
// settleAnsweredHumanTasks, and re-raising either would ask the same question twice.
|
|
2333
|
+
const open = await this.findOpenRequests(provider, task.ID);
|
|
2334
|
+
if (open.length > 0) {
|
|
2335
|
+
// Waiting, correctly — but on however many asks happen to exist. Collapse them here
|
|
2336
|
+
// or nothing ever will: the raise is behind the notified marker for good.
|
|
2337
|
+
await this.withdrawDuplicateRequests(provider, task.ID, open, open[0].ID);
|
|
2338
|
+
continue;
|
|
2339
|
+
}
|
|
2340
|
+
if (await this.answeredRequestFor(provider, task.ID))
|
|
2341
|
+
continue;
|
|
2342
|
+
LogStatus(`[TaskGraphDispatcher] The request for '${task.Name}' was cancelled and nothing ` +
|
|
2343
|
+
`replaced it; asking again.`);
|
|
2344
|
+
task.ClaimedBy = null;
|
|
2345
|
+
if (!(await task.Save())) {
|
|
2346
|
+
LogError(`[TaskGraphDispatcher] Could not re-open cancelled human task ${task.ID}: ` +
|
|
2347
|
+
`${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
2348
|
+
}
|
|
2349
|
+
}
|
|
2350
|
+
}
|
|
2351
|
+
/** The answered (or expired) request for a task, if any. */
|
|
2352
|
+
async answeredRequestFor(provider, taskID) {
|
|
2353
|
+
const result = await RunView.FromMetadataProvider(provider).RunView({
|
|
2354
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
2355
|
+
// Everything terminal. 'Canceled' is deliberately absent: a cancelled request means
|
|
2356
|
+
// the ASK was withdrawn, not that the step was decided, so the task keeps waiting
|
|
2357
|
+
// for whatever replaces it.
|
|
2358
|
+
ExtraFilter: `OriginatingTaskID='${taskID}' AND Status IN ('Approved','Rejected','Responded','Expired')`,
|
|
2359
|
+
OrderBy: 'RespondedAt DESC',
|
|
2360
|
+
ResultType: 'entity_object',
|
|
2361
|
+
BypassCache: true,
|
|
2362
|
+
}, this.contextUser);
|
|
2363
|
+
return (result.Success ? result.Results?.[0] : null) ?? null;
|
|
2364
|
+
}
|
|
2365
|
+
/**
|
|
2366
|
+
* Expires requests whose deadline has passed.
|
|
2367
|
+
*
|
|
2368
|
+
* A deadline that nothing enforces is a comment. Without this an `ExpiresAt` in the past leaves
|
|
2369
|
+
* the request `Requested` forever and the workflow waiting on it just as long.
|
|
2370
|
+
*/
|
|
2371
|
+
async expireOverdueRequests(provider, graphID) {
|
|
2372
|
+
// Scoped by an explicit id list rather than a subquery against a view name, so this reads
|
|
2373
|
+
// the same on any provider rather than assuming a SQL dialect and a physical view.
|
|
2374
|
+
const humanTasks = await RunView.FromMetadataProvider(provider).RunView({
|
|
2375
|
+
EntityName: 'MJ: Tasks',
|
|
2376
|
+
Fields: ['ID'],
|
|
2377
|
+
ExtraFilter: `ParentID='${graphID}' AND ${HumanTaskSQL()} AND Status='Pending'`,
|
|
2378
|
+
ResultType: 'simple',
|
|
2379
|
+
}, this.contextUser);
|
|
2380
|
+
const ids = (humanTasks.Results ?? []).map((r) => `'${r.ID}'`);
|
|
2381
|
+
if (ids.length === 0)
|
|
2382
|
+
return;
|
|
2383
|
+
const nowISO = new Date().toISOString();
|
|
2384
|
+
const overdue = await RunView.FromMetadataProvider(provider).RunView({
|
|
2385
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
2386
|
+
ExtraFilter: `Status='Requested' AND ExpiresAt IS NOT NULL AND ExpiresAt < '${nowISO}' ` +
|
|
2387
|
+
`AND OriginatingTaskID IN (${ids.join(',')})`,
|
|
2388
|
+
ResultType: 'entity_object',
|
|
2389
|
+
BypassCache: true,
|
|
2390
|
+
}, this.contextUser);
|
|
2391
|
+
if (!overdue.Success)
|
|
2392
|
+
return;
|
|
2393
|
+
for (const request of overdue.Results ?? []) {
|
|
2394
|
+
request.Status = 'Expired';
|
|
2395
|
+
if (!(await request.Save())) {
|
|
2396
|
+
LogError(`[TaskGraphDispatcher] Could not expire request ${request.ID}.`);
|
|
2397
|
+
}
|
|
2398
|
+
}
|
|
2399
|
+
}
|
|
2400
|
+
/** The agent run that submitted this task's graph, for provenance on the request. */
|
|
2401
|
+
async submittingRunOf(provider, task) {
|
|
2402
|
+
if (!task.ParentID)
|
|
2403
|
+
return null;
|
|
2404
|
+
try {
|
|
2405
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
2406
|
+
return (await parent.Load(task.ParentID)) ? parent.AgentRunID : null;
|
|
2407
|
+
}
|
|
2408
|
+
catch {
|
|
2409
|
+
return null;
|
|
621
2410
|
}
|
|
622
|
-
return claimable;
|
|
623
2411
|
}
|
|
624
2412
|
/** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
|
|
625
|
-
async loadGraphState(provider, parentTaskID) {
|
|
2413
|
+
async loadGraphState(provider, parentTaskID, debug) {
|
|
626
2414
|
const rv = RunView.FromMetadataProvider(provider);
|
|
627
2415
|
// BypassCache throughout: task status is written by the claim protocol's direct SQL, which
|
|
628
2416
|
// fires no cache invalidation. See findActiveGraphIDs.
|
|
629
2417
|
const childrenResult = await rv.RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
|
|
630
2418
|
const children = (childrenResult.Success ? childrenResult.Results : []) ?? [];
|
|
631
|
-
if (children.length === 0)
|
|
632
|
-
return {
|
|
2419
|
+
if (children.length === 0) {
|
|
2420
|
+
return {
|
|
2421
|
+
nodes: [], edges: [], entityById: new Map(),
|
|
2422
|
+
unreachableTaskIDs: new Set(), cascadeSkipTaskIDs: new Set(),
|
|
2423
|
+
skipSeedTaskIDs: new Set(), holdTaskIDs: new Set(),
|
|
2424
|
+
handledFailureIDs: new Set(),
|
|
2425
|
+
};
|
|
2426
|
+
}
|
|
633
2427
|
const idList = children.map((c) => `'${c.ID}'`).join(',');
|
|
634
2428
|
const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
|
|
635
2429
|
const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
|
|
636
2430
|
const entityById = new Map(children.map((c) => [c.ID, c]));
|
|
2431
|
+
// Read ONCE, and only when it can change an answer (R2-4). Both consumers below — which
|
|
2432
|
+
// origin statuses may decide an exclusive group, and which failures count as handled — are
|
|
2433
|
+
// no-ops unless something has actually failed, and this runs on every poll for every active
|
|
2434
|
+
// graph, so the parent load stays behind the same cheap exit `computeHandledFailures` used.
|
|
2435
|
+
// One parent read serves both questions this pass asks of the metadata bag.
|
|
2436
|
+
const parentMeta = await this.readParentMetadataFor(provider, parentTaskID);
|
|
2437
|
+
const failureSemantics = parentMeta.failureSemantics;
|
|
2438
|
+
// The invocation's own parameters, carried on the parent so a condition evaluated by any
|
|
2439
|
+
// instance sees what the walker saw (R3-3).
|
|
2440
|
+
const invocation = {
|
|
2441
|
+
Data: parentMeta.invocation?.data,
|
|
2442
|
+
Context: parentMeta.invocation?.context,
|
|
2443
|
+
};
|
|
637
2444
|
// Conditional edges are resolved HERE, before eligibility runs, by dropping edges whose
|
|
638
2445
|
// condition does not hold. Expressing it as edge removal rather than as a second rule inside
|
|
639
2446
|
// the eligibility algorithm is what keeps one definition of "ready": a task with no live
|
|
@@ -651,13 +2458,67 @@ export class TaskGraphDispatcher {
|
|
|
651
2458
|
// unreachable instead, and blocked before anything can claim it.
|
|
652
2459
|
const droppedInto = new Set();
|
|
653
2460
|
const stillReachable = new Set();
|
|
654
|
-
|
|
2461
|
+
// EXCLUSIVE edges are exempt from the generic machinery below, and that exemption is
|
|
2462
|
+
// load-bearing. An XOR loser is by definition condition-false, so the ordinary path would
|
|
2463
|
+
// record it as unreachable and Block it — and a Blocked child poisons the parent rollup, so
|
|
2464
|
+
// every fork would settle the graph as Blocked. Losers must become Skipped instead, which
|
|
2465
|
+
// only ResolveExclusiveGroups can decide.
|
|
2466
|
+
const exclusive = deps.filter((d) => !!d.ExclusiveGroup);
|
|
2467
|
+
const ordinary = deps.filter((d) => !d.ExclusiveGroup);
|
|
2468
|
+
// Named once and consumed twice — by the resolution below and by the `GateDecision` frame
|
|
2469
|
+
// emission further down. Two copies of this rule is how the console starts narrating
|
|
2470
|
+
// decisions the engine no longer makes.
|
|
2471
|
+
const decidingStatuses = failureSemantics === 'edges'
|
|
2472
|
+
? new Set(['Complete', 'Failed'])
|
|
2473
|
+
: new Set(['Complete']);
|
|
2474
|
+
const resolution = ResolveExclusiveGroups(exclusive.map((d) => ({
|
|
2475
|
+
id: d.ID,
|
|
2476
|
+
taskId: d.TaskID,
|
|
2477
|
+
dependsOnTaskId: d.DependsOnTaskID,
|
|
2478
|
+
exclusiveGroup: d.ExclusiveGroup,
|
|
2479
|
+
originStatus: (entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending'),
|
|
2480
|
+
priority: d.Priority ?? 0,
|
|
2481
|
+
sequence: d.Sequence ?? 0,
|
|
2482
|
+
conditionOutcome: this.evaluateExclusiveCondition(d, entityById, invocation, debug),
|
|
2483
|
+
})),
|
|
2484
|
+
// WHICH STATUSES MAY DECIDE — the graph's own failure dialect, not a constant.
|
|
2485
|
+
//
|
|
2486
|
+
// Under `'edges'`, a flow's failure handling IS its outgoing edges, so a Failed origin
|
|
2487
|
+
// decides its group and the drawn recovery path runs. Under `'block'` — the spec's
|
|
2488
|
+
// DEFAULT — a failure is terminal for everything downstream, and letting it decide was
|
|
2489
|
+
// silently catastrophic: the losers were removed and seeded, `ComputeSkipCascade`
|
|
2490
|
+
// confirmed them `Skipped`, `Skipped` satisfies dependents, and because the removed
|
|
2491
|
+
// loser edges also sever `ComputeTasksToBlock`'s forward walk, a join fed by an
|
|
2492
|
+
// independent healthy route EXECUTED downstream of an unhandled failure. The parent
|
|
2493
|
+
// still rolled up Failed, so the verdict looked right while the side effects had fired.
|
|
2494
|
+
//
|
|
2495
|
+
// The old comment claimed a loop-agent graph saw Complete-only. It did not; the same
|
|
2496
|
+
// hardcoded set was passed for every graph.
|
|
2497
|
+
decidingStatuses);
|
|
2498
|
+
const loserEdgeIDs = new Set(resolution.loserEdgeIDs);
|
|
2499
|
+
// Targets of an edge whose condition could not be evaluated (P2). Neither eligible nor
|
|
2500
|
+
// skipped: the edge stays live so the target is not mistaken for unreachable, and the target
|
|
2501
|
+
// joins the hold set so nothing claims it.
|
|
2502
|
+
const heldByCondition = new Set();
|
|
2503
|
+
// Decisions collected for `GateDecision` frames — announced after the state is assembled,
|
|
2504
|
+
// change-only, so a viewer learns WHY a branch ran (or is held) the moment it is decided.
|
|
2505
|
+
const gateDecisions = [];
|
|
2506
|
+
for (const d of ordinary) {
|
|
655
2507
|
if (d.Condition?.trim()) {
|
|
656
|
-
const
|
|
657
|
-
if (
|
|
2508
|
+
const decision = this.evaluateEdgeCondition(d, entityById, failureSemantics, invocation, debug);
|
|
2509
|
+
if (decision.decided) {
|
|
2510
|
+
gateDecisions.push({
|
|
2511
|
+
edge: d,
|
|
2512
|
+
verdict: decision.outcome === 'keep' ? 'satisfied' : decision.outcome === 'drop' ? 'notTaken' : 'held',
|
|
2513
|
+
reason: decision.reason,
|
|
2514
|
+
});
|
|
2515
|
+
}
|
|
2516
|
+
if (decision.outcome === 'drop') {
|
|
658
2517
|
droppedInto.add(d.TaskID);
|
|
659
2518
|
continue;
|
|
660
2519
|
}
|
|
2520
|
+
if (decision.outcome === 'hold')
|
|
2521
|
+
heldByCondition.add(d.TaskID);
|
|
661
2522
|
}
|
|
662
2523
|
stillReachable.add(d.TaskID);
|
|
663
2524
|
liveEdges.push({
|
|
@@ -666,16 +2527,120 @@ export class TaskGraphDispatcher {
|
|
|
666
2527
|
dependencyType: d.DependencyType,
|
|
667
2528
|
});
|
|
668
2529
|
}
|
|
2530
|
+
for (const d of exclusive) {
|
|
2531
|
+
// A losing edge is removed rather than left to gate: its target is being skipped, and a
|
|
2532
|
+
// live edge into a skipped task would keep the graph waiting on a branch nobody took.
|
|
2533
|
+
if (loserEdgeIDs.has(d.ID))
|
|
2534
|
+
continue;
|
|
2535
|
+
stillReachable.add(d.TaskID);
|
|
2536
|
+
liveEdges.push({
|
|
2537
|
+
taskId: d.TaskID,
|
|
2538
|
+
dependsOnTaskId: d.DependsOnTaskID,
|
|
2539
|
+
dependencyType: d.DependencyType,
|
|
2540
|
+
});
|
|
2541
|
+
}
|
|
669
2542
|
// Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
|
|
670
2543
|
// waiting on it, and a node reached by an alternate branch is genuinely reachable.
|
|
671
2544
|
const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
|
|
2545
|
+
// EXCLUSIVE LOSERS GET THE SAME TEST — they did not, and that is P1.
|
|
2546
|
+
//
|
|
2547
|
+
// A loser's target was seeded and written `Skipped` unconditionally, with no "does another
|
|
2548
|
+
// live route reach it?" check. The shape that breaks: `A →(cond)→ Review → Publish` and
|
|
2549
|
+
// `A →(else)→ Publish`. With the condition true, the losing edge `A→Publish` skipped
|
|
2550
|
+
// **Publish** while Review was still running; Review completed, Publish was already
|
|
2551
|
+
// terminal, and `Skipped` satisfies dependents — so the graph settled Complete with the
|
|
2552
|
+
// publish step never executed. No error and no stall.
|
|
2553
|
+
//
|
|
2554
|
+
// Confirmed against `liveEdges`, which by this point has both losers and definitely-false
|
|
2555
|
+
// edges removed, so "a live gating edge still points here" is exactly the surviving-route
|
|
2556
|
+
// question. A genuine loser has none and is still skipped.
|
|
2557
|
+
const confirmedSkipSeeds = new Set(ConfirmSkipSeeds([...resolution.skipSeedTaskIDs], liveEdges));
|
|
2558
|
+
// Exclusive edges get verdicts too, once their origin can decide them: a loser is a branch
|
|
2559
|
+
// not taken, a member of an undecided group is held, a surviving edge is satisfied. Same
|
|
2560
|
+
// vocabulary as ordinary edges so a viewer never needs to know which dialect an edge was.
|
|
2561
|
+
//
|
|
2562
|
+
// GATED ON THE SAME `terminalDecides` SET THE RESOLUTION USED — not on
|
|
2563
|
+
// `TERMINAL_FOR_CONDITIONS`. Since R2-4 a `Failed` origin decides its group only under
|
|
2564
|
+
// `failureSemantics: 'edges'`; announcing from the wider set would tell a viewer the fork
|
|
2565
|
+
// resolved while under `'block'` the engine deliberately left it undecided and let the
|
|
2566
|
+
// ordinary block cascade own everything downstream. A console that narrates decisions the
|
|
2567
|
+
// engine did not make is worse than one that stays quiet.
|
|
2568
|
+
const exclusiveHolds = new Set(resolution.holdTaskIDs);
|
|
2569
|
+
for (const d of exclusive) {
|
|
2570
|
+
const originStatus = entityById.get(d.DependsOnTaskID)?.Status ?? 'Pending';
|
|
2571
|
+
if (!decidingStatuses.has(originStatus) && !OverrideVerdictFor(debug ?? {}, d.ID))
|
|
2572
|
+
continue;
|
|
2573
|
+
gateDecisions.push({
|
|
2574
|
+
edge: d,
|
|
2575
|
+
verdict: loserEdgeIDs.has(d.ID)
|
|
2576
|
+
? 'notTaken'
|
|
2577
|
+
: exclusiveHolds.has(d.TaskID) ? 'held' : 'satisfied',
|
|
2578
|
+
reason: exclusiveHolds.has(d.TaskID)
|
|
2579
|
+
? 'this fork is undecided — a path in its group cannot be answered yet'
|
|
2580
|
+
: undefined,
|
|
2581
|
+
});
|
|
2582
|
+
}
|
|
2583
|
+
this.emitGateDecisions(provider, parentTaskID, gateDecisions, entityById);
|
|
2584
|
+
const nodes = children.map((c) => ({ id: c.ID, status: c.Status }));
|
|
2585
|
+
// THE CASCADE IS COMPUTED HERE, NOT ONLY AT SKIP TIME (R2-14).
|
|
2586
|
+
//
|
|
2587
|
+
// The claim filter covered seeds, holds and unreachable targets but not the cascade's
|
|
2588
|
+
// DESCENDANTS, and the skip writes are sequential per-entity saves. Between a seed's
|
|
2589
|
+
// `Skipped` landing and its descendants', another instance's fresh load sees
|
|
2590
|
+
// Skipped-satisfies-prerequisites and finds those descendants eligible — so it claims and
|
|
2591
|
+
// executes a branch that was never taken, irreversibly if the step has side effects.
|
|
2592
|
+
//
|
|
2593
|
+
// The set is already needed by the propagation pass, so computing it once here costs
|
|
2594
|
+
// nothing and closes the window by construction: nothing that is about to be skipped is
|
|
2595
|
+
// claimable, whichever instance is looking.
|
|
2596
|
+
const allSkipSeeds = [...confirmedSkipSeeds, ...unreachableTaskIDs];
|
|
2597
|
+
const cascadeSkipTaskIDs = new Set([
|
|
2598
|
+
...allSkipSeeds,
|
|
2599
|
+
...ComputeSkipCascade(nodes, liveEdges, allSkipSeeds),
|
|
2600
|
+
]);
|
|
2601
|
+
// A DECLARED EARLY FINISH MAKES EVERY REMAINING STEP UNCLAIMABLE (R3-1).
|
|
2602
|
+
//
|
|
2603
|
+
// The declaration is durable, so this holds for every instance rather than only the one that
|
|
2604
|
+
// decided it — which is the whole point. Folded into the same set the claim filter already
|
|
2605
|
+
// consults, so nothing about to be skipped can be claimed and started in the window between
|
|
2606
|
+
// the decision and the skip writes.
|
|
2607
|
+
if (parentMeta.earlyFinishedAt) {
|
|
2608
|
+
for (const node of nodes) {
|
|
2609
|
+
if (node.status === 'Pending')
|
|
2610
|
+
cascadeSkipTaskIDs.add(node.id);
|
|
2611
|
+
}
|
|
2612
|
+
}
|
|
672
2613
|
return {
|
|
673
|
-
nodes
|
|
2614
|
+
nodes,
|
|
674
2615
|
edges: liveEdges,
|
|
675
2616
|
entityById,
|
|
676
2617
|
unreachableTaskIDs,
|
|
2618
|
+
cascadeSkipTaskIDs,
|
|
2619
|
+
skipSeedTaskIDs: confirmedSkipSeeds,
|
|
2620
|
+
// Exclusive holds and ordinary-condition holds are the same state and share one set:
|
|
2621
|
+
// "we cannot tell yet, so nothing may claim this."
|
|
2622
|
+
holdTaskIDs: new Set([...resolution.holdTaskIDs, ...heldByCondition]),
|
|
2623
|
+
handledFailureIDs: this.computeHandledFailures(failureSemantics, nodes, liveEdges),
|
|
677
2624
|
};
|
|
678
2625
|
}
|
|
2626
|
+
/**
|
|
2627
|
+
* Reports an unevaluable condition ONCE per edge, not once per poll.
|
|
2628
|
+
*
|
|
2629
|
+
* Eligibility is recomputed every cycle, so an unqualified LogError here would repeat every few
|
|
2630
|
+
* seconds for as long as the graph is held — which buries the one line that matters under
|
|
2631
|
+
* thousands of copies of itself. Keyed by edge id plus the failure text, so a condition that
|
|
2632
|
+
* starts failing differently is reported again.
|
|
2633
|
+
*/
|
|
2634
|
+
logUnevaluableConditionOnce(dep, errorMessage) {
|
|
2635
|
+
const key = `${dep.ID}:${errorMessage ?? ''}`;
|
|
2636
|
+
if (this.reportedUnevaluableConditions.has(key))
|
|
2637
|
+
return;
|
|
2638
|
+
this.reportedUnevaluableConditions.add(key);
|
|
2639
|
+
LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
|
|
2640
|
+
`(${errorMessage}); condition text: ${JSON.stringify(dep.Condition)}. ` +
|
|
2641
|
+
`Task ${dep.TaskID} is HELD — it will not run and will not be skipped until the ` +
|
|
2642
|
+
`condition can be evaluated. The graph reports as stalled while this holds.`);
|
|
2643
|
+
}
|
|
679
2644
|
/**
|
|
680
2645
|
* Decides whether a conditional dependency edge is live.
|
|
681
2646
|
*
|
|
@@ -683,31 +2648,127 @@ export class TaskGraphDispatcher {
|
|
|
683
2648
|
* only information a runtime graph has to branch on. Returns `'drop'` only on a definite false;
|
|
684
2649
|
* an unevaluable condition keeps the edge for the reason stated at the call site.
|
|
685
2650
|
*/
|
|
686
|
-
evaluateEdgeCondition(dep, entityById) {
|
|
2651
|
+
evaluateEdgeCondition(dep, entityById, failureSemantics, invocation, debug) {
|
|
2652
|
+
// An operator's override answers the edge BEFORE the condition is consulted — an override
|
|
2653
|
+
// exists precisely because the condition cannot be answered (or answered wrongly), so
|
|
2654
|
+
// evaluating first would re-produce the hold the override exists to end.
|
|
2655
|
+
const override = OverrideVerdictFor(debug ?? {}, dep.ID);
|
|
2656
|
+
if (override) {
|
|
2657
|
+
return {
|
|
2658
|
+
outcome: override === 'true' ? 'keep' : 'drop',
|
|
2659
|
+
reason: `answered '${override}' by an operator override`,
|
|
2660
|
+
decided: true,
|
|
2661
|
+
};
|
|
2662
|
+
}
|
|
687
2663
|
const upstream = entityById.get(dep.DependsOnTaskID);
|
|
688
2664
|
if (!upstream)
|
|
689
|
-
return 'keep';
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
2665
|
+
return { outcome: 'keep', decided: false };
|
|
2666
|
+
// The DECISION lives in `condition-gate`; what stays here is the loading and the logging.
|
|
2667
|
+
//
|
|
2668
|
+
// `DecideGate` takes the evaluation as a thunk rather than a value, and that is the fix, not
|
|
2669
|
+
// a style: the terminality guard has to stop the evaluation happening at all. Evaluating
|
|
2670
|
+
// `succeeded` against a still-Pending origin does not fail — it returns a confident, wrong
|
|
2671
|
+
// `false`, the edge is dropped, and the target is Blocked at wave one before the origin ever
|
|
2672
|
+
// ran. That killed every conditioned linear chain, with no error anywhere.
|
|
2673
|
+
let unevaluableError;
|
|
2674
|
+
let evaluated = false;
|
|
2675
|
+
const outcome = DecideGate(upstream.Status, failureSemantics, () => {
|
|
2676
|
+
evaluated = true;
|
|
2677
|
+
const result = this.conditionEvaluator.Evaluate(dep.Condition, BuildConditionContext(upstream, ParseConditionOutput(upstream.OutputPayload), invocation));
|
|
2678
|
+
if (!result.Success)
|
|
2679
|
+
unevaluableError = result.ErrorMessage;
|
|
2680
|
+
return result;
|
|
703
2681
|
});
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
2682
|
+
// Reported here rather than inside the decision, so the pure part stays pure and a held edge
|
|
2683
|
+
// is still loud once — see logUnevaluableConditionOnce.
|
|
2684
|
+
if (outcome === 'hold')
|
|
2685
|
+
this.logUnevaluableConditionOnce(dep, unevaluableError);
|
|
2686
|
+
return {
|
|
2687
|
+
outcome,
|
|
2688
|
+
reason: outcome === 'hold'
|
|
2689
|
+
? (unevaluableError ?? 'the condition cannot be answered yet')
|
|
2690
|
+
: undefined,
|
|
2691
|
+
// A verdict was rendered only when the gate was genuinely ASKED. Terminality alone is
|
|
2692
|
+
// not enough since R3-2: a Failed origin under 'block' (and a Cancelled origin under
|
|
2693
|
+
// either dialect) returns 'keep' WITHOUT evaluating, so the block cascade owns the
|
|
2694
|
+
// target — announcing that as "satisfied" would tell a viewer a gate opened that
|
|
2695
|
+
// DecideGate deliberately never asked. `evaluated` is set by the thunk itself, so this
|
|
2696
|
+
// cannot drift from the gate's own rules; a Skipped origin's unevaluated drop is still
|
|
2697
|
+
// a verdict a viewer needs (the branch was not taken).
|
|
2698
|
+
decided: evaluated || (upstream.Status === 'Skipped' && outcome === 'drop'),
|
|
2699
|
+
};
|
|
2700
|
+
}
|
|
2701
|
+
/**
|
|
2702
|
+
* An exclusive edge's condition as a three-way outcome.
|
|
2703
|
+
*
|
|
2704
|
+
* `ResolveExclusiveGroups` needs to tell "false" from "could not be evaluated": the first loses
|
|
2705
|
+
* the branch, the second holds the whole group. The generic keep/drop path cannot express that
|
|
2706
|
+
* difference, which is why exclusive edges take this route instead.
|
|
2707
|
+
*/
|
|
2708
|
+
evaluateExclusiveCondition(dep, entityById, invocation, debug) {
|
|
2709
|
+
// Same override-first rule as ordinary edges — see evaluateEdgeCondition.
|
|
2710
|
+
const override = OverrideVerdictFor(debug ?? {}, dep.ID);
|
|
2711
|
+
if (override)
|
|
2712
|
+
return override === 'true' ? 'satisfied' : 'unsatisfied';
|
|
2713
|
+
if (!dep.Condition?.trim())
|
|
2714
|
+
return 'satisfied';
|
|
2715
|
+
const upstream = entityById.get(dep.DependsOnTaskID);
|
|
2716
|
+
if (!upstream)
|
|
2717
|
+
return 'unevaluable';
|
|
2718
|
+
const result = this.conditionEvaluator.Evaluate(dep.Condition,
|
|
2719
|
+
// The invocation envelope rides the EXCLUSIVE dialect too. R3-3 threaded it into the
|
|
2720
|
+
// ordinary path; a flow's XOR branch reading `data.userApproval` is the same documented
|
|
2721
|
+
// condition on a different edge kind, and `BuildConditionContext`'s defaulted parameter
|
|
2722
|
+
// made omitting it here silently evaluate those roots against nothing.
|
|
2723
|
+
BuildConditionContext(upstream, ParseConditionOutput(upstream.OutputPayload), invocation));
|
|
2724
|
+
// SAME CLASSIFICATION AS THE ORDINARY DIALECT (R2-3). The null-safe envelope already makes
|
|
2725
|
+
// one level of absence read as false here, but a deeper absent chain still throws — and
|
|
2726
|
+
// calling that 'unevaluable' would hold the whole group forever on a terminal origin, while
|
|
2727
|
+
// `DecideGate` would have dropped the identical condition. Two dialects, one question.
|
|
2728
|
+
if (!result.Success)
|
|
2729
|
+
return IsBrokenGuard(result.ErrorMessage) ? 'unevaluable' : 'unsatisfied';
|
|
2730
|
+
return result.Value ? 'satisfied' : 'unsatisfied';
|
|
2731
|
+
}
|
|
2732
|
+
/**
|
|
2733
|
+
* Announces gate verdicts that CHANGED since this instance last looked.
|
|
2734
|
+
*
|
|
2735
|
+
* Fire-and-forget by design: `loadGraphState` is synchronous graph assembly, and the owner
|
|
2736
|
+
* lookup the frame needs is async — so the emission floats behind rather than making state
|
|
2737
|
+
* loading wait on observability. Frames are commentary, never a step of the work.
|
|
2738
|
+
*/
|
|
2739
|
+
emitGateDecisions(provider, parentTaskID, decisions, entityById) {
|
|
2740
|
+
if (!this.observer || decisions.length === 0)
|
|
2741
|
+
return;
|
|
2742
|
+
let perEdge = this.emittedGateVerdicts.get(parentTaskID);
|
|
2743
|
+
if (!perEdge) {
|
|
2744
|
+
perEdge = new Map();
|
|
2745
|
+
this.emittedGateVerdicts.set(parentTaskID, perEdge);
|
|
709
2746
|
}
|
|
710
|
-
|
|
2747
|
+
const changed = decisions.filter((d) => {
|
|
2748
|
+
const key = `${d.verdict}|${d.reason ?? ''}`;
|
|
2749
|
+
if (perEdge.get(d.edge.ID) === key)
|
|
2750
|
+
return false;
|
|
2751
|
+
perEdge.set(d.edge.ID, key);
|
|
2752
|
+
return true;
|
|
2753
|
+
});
|
|
2754
|
+
if (changed.length === 0)
|
|
2755
|
+
return;
|
|
2756
|
+
void this.resolveOwner(provider, parentTaskID).then((owner) => {
|
|
2757
|
+
for (const d of changed) {
|
|
2758
|
+
this.emit({
|
|
2759
|
+
Kind: 'GateDecision',
|
|
2760
|
+
ParentTaskID: parentTaskID,
|
|
2761
|
+
OwnerUserID: owner,
|
|
2762
|
+
TaskID: d.edge.TaskID,
|
|
2763
|
+
TaskName: entityById.get(d.edge.TaskID)?.Name,
|
|
2764
|
+
EdgeID: d.edge.ID,
|
|
2765
|
+
DependsOnTaskID: d.edge.DependsOnTaskID,
|
|
2766
|
+
Verdict: d.verdict,
|
|
2767
|
+
ConditionText: d.edge.Condition ?? undefined,
|
|
2768
|
+
Reason: d.reason,
|
|
2769
|
+
});
|
|
2770
|
+
}
|
|
2771
|
+
}).catch(() => { });
|
|
711
2772
|
}
|
|
712
2773
|
/** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
|
|
713
2774
|
async loadDependencyOutputs(provider, taskID) {
|
|
@@ -731,5 +2792,759 @@ export class TaskGraphDispatcher {
|
|
|
731
2792
|
}
|
|
732
2793
|
return outputs;
|
|
733
2794
|
}
|
|
2795
|
+
/**
|
|
2796
|
+
* Runs one task's body, whatever kind of step it is.
|
|
2797
|
+
*
|
|
2798
|
+
* **Routing is on `StepType`, not on which key happens to be set.** A loop step carries the same
|
|
2799
|
+
* `ActionID` or `AgentID` as an ordinary step — that key is what the loop *repeats* — so the old
|
|
2800
|
+
* `task.ActionID ? action : agent` test would have run a loop exactly once and called it done.
|
|
2801
|
+
* `StepType` is the only field that distinguishes them.
|
|
2802
|
+
*
|
|
2803
|
+
* Every branch is normalized to one shape so the recording path above stays single: an action has
|
|
2804
|
+
* no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
|
|
2805
|
+
*/
|
|
2806
|
+
async runTaskBody(task, provider, inputPayload, dependencyOutputs, onProgress) {
|
|
2807
|
+
const payload = this.mergedPayload(inputPayload, dependencyOutputs);
|
|
2808
|
+
const config = task.ConfigurationObject;
|
|
2809
|
+
// A loop's own step type decides how many times its body runs; the body itself is dispatched
|
|
2810
|
+
// through the very same runners as a one-shot step.
|
|
2811
|
+
if (task.StepType === 'ForEach' || task.StepType === 'While') {
|
|
2812
|
+
return { ...await this.runLoopTask(task, provider, payload, dependencyOutputs), PayloadAtStart: payload };
|
|
2813
|
+
}
|
|
2814
|
+
const { params, errors } = BuildMappedInput(config?.inputMapping, { payload });
|
|
2815
|
+
for (const e of errors)
|
|
2816
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
|
|
2817
|
+
// `payload`, NOT `inputPayload` — the MERGED value computed above, which includes what every
|
|
2818
|
+
// dependency produced.
|
|
2819
|
+
//
|
|
2820
|
+
// A step with an input mapping got exactly the parameters it declared; a step WITHOUT one
|
|
2821
|
+
// fell back to the raw input and therefore saw nothing any earlier step had produced. For a
|
|
2822
|
+
// Prompt step — which declares no mapping by design, because it reads the whole payload
|
|
2823
|
+
// through `{{ _CURRENT_PAYLOAD }}` — that meant the placeholder rendered `{}` and the model
|
|
2824
|
+
// was asked to write from an empty brief.
|
|
2825
|
+
//
|
|
2826
|
+
// It answered anyway. The Content Pipeline's draft step said "the research data was empty",
|
|
2827
|
+
// which was TRUE of what it had been handed while twenty research results sat in the
|
|
2828
|
+
// dependency outputs beside it, and the reviewer then rejected the draft for saying so.
|
|
2829
|
+
// Every layer looked like it was working.
|
|
2830
|
+
const effectiveInput = Object.keys(params).length > 0 ? params : payload;
|
|
2831
|
+
if (task.StepType === 'Prompt') {
|
|
2832
|
+
if (!this.promptRunner) {
|
|
2833
|
+
// Not a failure: "nobody here can run this" is not "this ran and did not work".
|
|
2834
|
+
return { Success: false, AgentRunID: null, ErrorMessage: 'No prompt runner is loaded on this host.' };
|
|
2835
|
+
}
|
|
2836
|
+
const promptResult = await this.promptRunner.RunPromptForTask({
|
|
2837
|
+
TaskID: task.ID,
|
|
2838
|
+
PromptID: task.PromptID,
|
|
2839
|
+
InputPayload: effectiveInput,
|
|
2840
|
+
DependencyOutputs: dependencyOutputs,
|
|
2841
|
+
TemplateParameters: config?.prompt?.templateParameters,
|
|
2842
|
+
Provider: provider,
|
|
2843
|
+
ContextUser: this.contextUser,
|
|
2844
|
+
OnProgress: onProgress,
|
|
2845
|
+
});
|
|
2846
|
+
// A prompt's response is DEEP-MERGED into the payload rather than replacing it. A prompt
|
|
2847
|
+
// answers one question; replacing the payload with its answer would discard everything
|
|
2848
|
+
// the steps before it established, which is how a late step loses the data it depends on.
|
|
2849
|
+
const merged = promptResult.Success && promptResult.Output && typeof promptResult.Output === 'object'
|
|
2850
|
+
? deepMergePayload(payload, promptResult.Output)
|
|
2851
|
+
: payload;
|
|
2852
|
+
return {
|
|
2853
|
+
Success: promptResult.Success,
|
|
2854
|
+
AgentRunID: null,
|
|
2855
|
+
ErrorMessage: promptResult.ErrorMessage,
|
|
2856
|
+
Output: this.applyStepOutputMapping(task, merged, merged, config?.outputMapping),
|
|
2857
|
+
PayloadAtStart: payload,
|
|
2858
|
+
ChatMessage: promptResult.ChatMessage,
|
|
2859
|
+
// Returned even when the prompt FAILED. A failed prompt still cost tokens, and a
|
|
2860
|
+
// cost rollup that silently omits failures under-reports exactly the runs someone
|
|
2861
|
+
// is most likely to be investigating.
|
|
2862
|
+
PromptRunID: promptResult.PromptRunID,
|
|
2863
|
+
};
|
|
2864
|
+
}
|
|
2865
|
+
const raw = task.ActionID
|
|
2866
|
+
? { ...await this.actionRunner.RunActionForTask({
|
|
2867
|
+
TaskID: task.ID,
|
|
2868
|
+
ActionID: task.ActionID,
|
|
2869
|
+
InputPayload: effectiveInput,
|
|
2870
|
+
DependencyOutputs: dependencyOutputs,
|
|
2871
|
+
Provider: provider,
|
|
2872
|
+
ContextUser: this.contextUser,
|
|
2873
|
+
OnProgress: onProgress,
|
|
2874
|
+
}), AgentRunID: null }
|
|
2875
|
+
: await this.runAgentNode(task, provider, effectiveInput, dependencyOutputs, onProgress);
|
|
2876
|
+
return {
|
|
2877
|
+
...raw,
|
|
2878
|
+
Output: this.applyStepOutputMapping(task, payload, raw.Output, config?.outputMapping),
|
|
2879
|
+
PayloadAtStart: payload,
|
|
2880
|
+
};
|
|
2881
|
+
}
|
|
2882
|
+
/**
|
|
2883
|
+
* Runs a loop step: its body once per iteration, with the item and index in scope.
|
|
2884
|
+
*
|
|
2885
|
+
* The loop's own `Configuration` supplies the definition; the row's `ActionID` / `AgentID`
|
|
2886
|
+
* supplies what to repeat. Per-iteration inputs are resolved fresh each pass — the bindings are
|
|
2887
|
+
* merged into the payload before the mapping is applied, which is how a body can reference the
|
|
2888
|
+
* current item at all.
|
|
2889
|
+
*/
|
|
2890
|
+
async runLoopTask(task, provider, payload, dependencyOutputs) {
|
|
2891
|
+
const config = task.ConfigurationObject;
|
|
2892
|
+
const op = task.StepType === 'ForEach' ? config?.forEach : config?.while;
|
|
2893
|
+
if (!op) {
|
|
2894
|
+
return {
|
|
2895
|
+
Success: false,
|
|
2896
|
+
AgentRunID: null,
|
|
2897
|
+
ErrorMessage: `"${task.Name}" is a ${task.StepType} step with no loop settings, so there is nothing to repeat.`,
|
|
2898
|
+
};
|
|
2899
|
+
}
|
|
2900
|
+
// A prompt body has no params of its own — it receives the payload (with the loop bindings
|
|
2901
|
+
// merged in) through the placeholder, so an empty mapping is correct rather than missing.
|
|
2902
|
+
const bodyMapping = (op.action?.params ?? {});
|
|
2903
|
+
// The BODY's output mapping, applied once per pass — see `foldIterationOutput`.
|
|
2904
|
+
//
|
|
2905
|
+
// It used to be applied a single time after the loop finished, against the accumulated
|
|
2906
|
+
// payload. That is the wrong moment in two ways at once: the mapping names an output
|
|
2907
|
+
// PARAMETER of the body, which no longer exists by then, and a mapping like
|
|
2908
|
+
// `"Items": "results[]"` can only append per pass. So every pass merged its raw result into
|
|
2909
|
+
// the shared payload instead, each overwriting the last, and the mapping matched nothing and
|
|
2910
|
+
// wrote nothing. A ForEach over five items reported five successes and kept item five.
|
|
2911
|
+
const bodyOutputMapping = op.action?.outputMapping ?? op.prompt?.outputMapping;
|
|
2912
|
+
// Where this step sits in its graph, resolved ONCE rather than per iteration. A loop body is
|
|
2913
|
+
// dispatched exactly like a one-shot step and needs the same two things: the run that
|
|
2914
|
+
// submitted the graph (so a spawned run gets a ParentRunID and is visible to the tree and to
|
|
2915
|
+
// cost), and the continuation depth (so the recursion cap still applies). Omitting them made
|
|
2916
|
+
// loop bodies second-class in every dimension — and reopened the unbounded-recursion hole
|
|
2917
|
+
// THROUGH loops, since each spawned run restarted the chain at zero.
|
|
2918
|
+
const graphContext = await this.graphContext(provider, task);
|
|
2919
|
+
// THE LOOP'S PAYLOAD ACCUMULATES. Each iteration's output merges in, and the next iteration
|
|
2920
|
+
// — and the While condition — sees it. Without this the condition closure re-read the
|
|
2921
|
+
// payload as it was when the loop STARTED, so a `while payload.brandOK !== true` could never
|
|
2922
|
+
// become false: the loop burned every iteration re-examining the original input and always
|
|
2923
|
+
// took the give-up branch, making the other branch unreachable. The loop ran, reported
|
|
2924
|
+
// success, and its result was predetermined.
|
|
2925
|
+
let livePayload = { ...payload };
|
|
2926
|
+
// One entry per pass, so the loop's work exists somewhere the platform can see it. Without
|
|
2927
|
+
// this a loop is a single childless node: the run tree reaches nested work through six links
|
|
2928
|
+
// and an iteration is none of them, so the passes were invisible to the timeline AND their
|
|
2929
|
+
// spend was missing from the settlement rollup. See ITaskStepRuntime.iterations.
|
|
2930
|
+
const iterationTrace = [];
|
|
2931
|
+
// Bounds what the trace's payloads may cost. The pointers are never budgeted — those are the
|
|
2932
|
+
// durable record of the work and must survive whatever the payloads do.
|
|
2933
|
+
const budget = new IterationPayloadBudget();
|
|
2934
|
+
const invokeBody = async ({ Index, Bindings }) => {
|
|
2935
|
+
// Bindings go INTO the payload rather than beside it, so an authored mapping reaches the
|
|
2936
|
+
// current item the same way it reaches anything else: `payload.<itemVariable>`.
|
|
2937
|
+
const iterationPayload = { ...livePayload, ...Bindings };
|
|
2938
|
+
const resolved = ResolveMappedInput(bodyMapping, { payload: iterationPayload });
|
|
2939
|
+
/**
|
|
2940
|
+
* Folds an iteration's output into the running payload the next pass will see, and
|
|
2941
|
+
* records what the pass produced.
|
|
2942
|
+
*
|
|
2943
|
+
* The trace is written HERE rather than after the loop because a loop that fails partway
|
|
2944
|
+
* still ran the passes before it, and their runs are real spend that must not vanish
|
|
2945
|
+
* because the loop as a whole did not finish.
|
|
2946
|
+
*/
|
|
2947
|
+
const absorb = (outcome, bodyInput) => {
|
|
2948
|
+
livePayload = this.foldIterationOutput(task, livePayload, outcome.Output, bodyOutputMapping);
|
|
2949
|
+
iterationTrace.push({
|
|
2950
|
+
index: Index,
|
|
2951
|
+
// What THIS pass was handed and what it gave back — not the loop's running
|
|
2952
|
+
// payload before and after it.
|
|
2953
|
+
//
|
|
2954
|
+
// A pass has no row of its own, so without these there is nowhere its work can be
|
|
2955
|
+
// recorded: every iteration presented null on both sides and the run view could
|
|
2956
|
+
// say nothing about any single pass, which for a loop is the only interesting
|
|
2957
|
+
// question. But recording the RUNNING payload on both sides — the obvious reading
|
|
2958
|
+
// of "before and after" — is quadratic: each pass would hold a full copy of
|
|
2959
|
+
// everything every earlier pass accumulated. A five-iteration demo produced a
|
|
2960
|
+
// 121KB Configuration that way; the same loop over fifty items would produce
|
|
2961
|
+
// megabytes, in a column every reader of the row pays to load.
|
|
2962
|
+
//
|
|
2963
|
+
// The pass's own input and output are what a reader actually wants ("what did
|
|
2964
|
+
// pass three do?"), and they are constant-sized per pass.
|
|
2965
|
+
payloadAtStart: budget.Take(bodyInput),
|
|
2966
|
+
payloadAtEnd: budget.Take(outcome.Output),
|
|
2967
|
+
promptRunID: outcome.PromptRunID,
|
|
2968
|
+
agentRunID: outcome.AgentRunID,
|
|
2969
|
+
// An ACTION body records its log here. Omitting it left an action-bodied pass
|
|
2970
|
+
// with no pointer at all — no cost, no timing, nothing to open — and the tree,
|
|
2971
|
+
// seeing neither a prompt run nor an agent run, fell through to its last branch
|
|
2972
|
+
// and called the pass a Sub-Agent. A loop over a web search then showed five
|
|
2973
|
+
// sub-agent runs that never existed.
|
|
2974
|
+
actionLogID: outcome.ActionLogID,
|
|
2975
|
+
success: outcome.Success,
|
|
2976
|
+
errorMessage: outcome.ErrorMessage,
|
|
2977
|
+
});
|
|
2978
|
+
return outcome;
|
|
2979
|
+
};
|
|
2980
|
+
// A prompt body is checked FIRST because it is the only one whose id lives in its own
|
|
2981
|
+
// column: a loop repeating a prompt has PromptID set and both ActionID and AgentID null,
|
|
2982
|
+
// so falling through to the agent branch would dereference a null agent id.
|
|
2983
|
+
if (task.StepType && task.PromptID && !task.ActionID) {
|
|
2984
|
+
if (!this.promptRunner) {
|
|
2985
|
+
return { Success: false, ErrorMessage: 'No prompt runner is loaded on this host.' };
|
|
2986
|
+
}
|
|
2987
|
+
return absorb(await this.promptRunner.RunPromptForTask({
|
|
2988
|
+
TaskID: task.ID,
|
|
2989
|
+
PromptID: task.PromptID,
|
|
2990
|
+
// The ITERATION payload, not the mapped params. An action body declares its
|
|
2991
|
+
// inputs and gets exactly those; a prompt body declares none — it receives the
|
|
2992
|
+
// whole payload through the placeholder, and the loop's item and index are
|
|
2993
|
+
// merged INTO that payload. Passing the mapped result here handed the prompt an
|
|
2994
|
+
// empty object, so every iteration asked the model to describe nothing and got
|
|
2995
|
+
// five confident answers about nothing back.
|
|
2996
|
+
InputPayload: iterationPayload,
|
|
2997
|
+
DependencyOutputs: dependencyOutputs,
|
|
2998
|
+
// The loop's bindings become TEMPLATE VARIABLES, so an author writes
|
|
2999
|
+
// `{{ field }}` for the item the loop is on — which is what `itemVariable` is
|
|
3000
|
+
// for, and what anyone reading the step's configuration expects. Reaching it
|
|
3001
|
+
// through the payload placeholder instead works but is not discoverable, and
|
|
3002
|
+
// getting it wrong is silent: the variable renders empty and the model answers
|
|
3003
|
+
// confidently about nothing.
|
|
3004
|
+
TemplateParameters: { ...stringifyBindings(Bindings), ...op.prompt?.templateParameters },
|
|
3005
|
+
Provider: provider,
|
|
3006
|
+
ContextUser: this.contextUser,
|
|
3007
|
+
}), iterationPayload);
|
|
3008
|
+
}
|
|
3009
|
+
if (task.ActionID) {
|
|
3010
|
+
return absorb(await this.actionRunner.RunActionForTask({
|
|
3011
|
+
TaskID: task.ID,
|
|
3012
|
+
ActionID: task.ActionID,
|
|
3013
|
+
InputPayload: resolved,
|
|
3014
|
+
DependencyOutputs: dependencyOutputs,
|
|
3015
|
+
Provider: provider,
|
|
3016
|
+
ContextUser: this.contextUser,
|
|
3017
|
+
}), resolved);
|
|
3018
|
+
}
|
|
3019
|
+
const agentInput = Object.keys(resolved).length > 0 ? resolved : iterationPayload;
|
|
3020
|
+
return absorb(await this.agentRunner.RunAgentForTask({
|
|
3021
|
+
TaskID: task.ID,
|
|
3022
|
+
AgentID: task.AgentID,
|
|
3023
|
+
// The ITERATION payload when the body declares no inputs of its own. A sub-agent
|
|
3024
|
+
// body has no `params`, so the mapped result is `{}` — every iteration was handing
|
|
3025
|
+
// the agent nothing and asking it to work from that.
|
|
3026
|
+
InputPayload: agentInput,
|
|
3027
|
+
DependencyOutputs: dependencyOutputs,
|
|
3028
|
+
ContinuationDepth: graphContext.Depth,
|
|
3029
|
+
SubmittingAgentRunID: graphContext.SubmittingAgentRunID,
|
|
3030
|
+
Provider: provider,
|
|
3031
|
+
ContextUser: this.contextUser,
|
|
3032
|
+
}), agentInput);
|
|
3033
|
+
};
|
|
3034
|
+
const outcome = task.StepType === 'ForEach'
|
|
3035
|
+
? await RunForEachLoop(op, { payload }, invokeBody)
|
|
3036
|
+
: await RunWhileLoop(op, (iteration) => this.conditionEvaluator.Evaluate(op.condition,
|
|
3037
|
+
// BOTH forms, because a workflow should not have two condition dialects. An
|
|
3038
|
+
// EDGE condition is written `payload.brandOK !== true`; a loop condition used
|
|
3039
|
+
// to see the payload's keys spread at the top level and nothing named `payload`,
|
|
3040
|
+
// so the same expression that routes an edge failed here with
|
|
3041
|
+
// "payload is not defined". The spread stays for conditions already written
|
|
3042
|
+
// against it.
|
|
3043
|
+
{ ...livePayload, payload: livePayload, iteration }), invokeBody);
|
|
3044
|
+
return {
|
|
3045
|
+
Success: outcome.Success,
|
|
3046
|
+
AgentRunID: null,
|
|
3047
|
+
ErrorMessage: outcome.ErrorMessage,
|
|
3048
|
+
// Every pass that ran, including those before a failure — see `iterationTrace`.
|
|
3049
|
+
Iterations: iterationTrace.length > 0 ? iterationTrace : undefined,
|
|
3050
|
+
// The ACCUMULATED payload — everything the iterations established — not the one the
|
|
3051
|
+
// loop started with, which would discard the loop's whole effect on the workflow.
|
|
3052
|
+
//
|
|
3053
|
+
// Only the STEP's own mapping is applied here. The body's mapping already ran once per
|
|
3054
|
+
// pass inside `foldIterationOutput`; applying it again against the accumulated payload
|
|
3055
|
+
// is what used to make it match nothing.
|
|
3056
|
+
Output: this.applyStepOutputMapping(task, livePayload, outcome.Output, config?.outputMapping),
|
|
3057
|
+
};
|
|
3058
|
+
}
|
|
3059
|
+
/**
|
|
3060
|
+
* Folds one pass's result into the loop's running payload.
|
|
3061
|
+
*
|
|
3062
|
+
* **With a body mapping**, the pass's declared outputs are filed where the author said to put
|
|
3063
|
+
* them — including `name[]`, which appends, so a ForEach can collect one entry per item. That is
|
|
3064
|
+
* the whole point of a loop over a collection, and it is only expressible per pass.
|
|
3065
|
+
*
|
|
3066
|
+
* **Without one**, the raw result is deep-merged, which is the pre-existing behaviour and the
|
|
3067
|
+
* right default for a `While` that converges on a value: each pass refines what the condition
|
|
3068
|
+
* reads. It is the wrong default for a ForEach that collects — hence the mapping.
|
|
3069
|
+
*
|
|
3070
|
+
* An unmapped output is reported per pass rather than swallowed, for the same reason
|
|
3071
|
+
* {@link applyStepOutputMapping} reports it: a mapping that names something the body never
|
|
3072
|
+
* returned means the pass did work that went nowhere, while everything reports success.
|
|
3073
|
+
*/
|
|
3074
|
+
foldIterationOutput(task, livePayload, output, bodyOutputMapping) {
|
|
3075
|
+
if (!output || typeof output !== 'object' || Array.isArray(output))
|
|
3076
|
+
return livePayload;
|
|
3077
|
+
const source = output;
|
|
3078
|
+
if (!bodyOutputMapping)
|
|
3079
|
+
return deepMergePayload(livePayload, source);
|
|
3080
|
+
// Applied ONTO a deep copy of the running payload, not into a fresh object: `name[]` appends,
|
|
3081
|
+
// and appending is meaningless without the list already there. The copy is deep because the
|
|
3082
|
+
// trace has already recorded earlier passes' payloads — mutating a shared nested array would
|
|
3083
|
+
// retroactively rewrite what those passes are recorded as having seen.
|
|
3084
|
+
const { updates, errors, unmapped } = ApplyOutputMapping(source, bodyOutputMapping, structuredClone(livePayload));
|
|
3085
|
+
for (const e of errors)
|
|
3086
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} loop body: ${e}`);
|
|
3087
|
+
if (unmapped?.length) {
|
|
3088
|
+
LogError(`[TaskGraphDispatcher] '${task.Name}' loop body mapped output(s) it did not return: ` +
|
|
3089
|
+
`${unmapped.join(', ')}. The pass returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
|
|
3090
|
+
`Those payload values were NOT written, so anything downstream reading them sees nothing.`);
|
|
3091
|
+
}
|
|
3092
|
+
// `updates` IS the copy that was applied onto, so it is already the complete next payload.
|
|
3093
|
+
return updates;
|
|
3094
|
+
}
|
|
3095
|
+
/**
|
|
3096
|
+
* Files a step's result into the payload it hands downstream.
|
|
3097
|
+
*
|
|
3098
|
+
* **This is what makes a branch condition possible.** A workflow that branches on
|
|
3099
|
+
* `payload.stockPrice` has that value only because this step mapped `CurrentPrice -> stockPrice`.
|
|
3100
|
+
* Without it the condition reads `undefined` — merely falsy — so the workflow takes the other
|
|
3101
|
+
* branch, finishes, and reports success with nothing to indicate anything went wrong.
|
|
3102
|
+
*
|
|
3103
|
+
* The incoming payload is carried through as well as the update, so a value written three steps
|
|
3104
|
+
* back is still readable here. Returning only this step's own output is what used to limit a
|
|
3105
|
+
* condition's view to its immediate predecessor.
|
|
3106
|
+
*/
|
|
3107
|
+
applyStepOutputMapping(task, payload, output, outputMapping) {
|
|
3108
|
+
// No mapping: MERGE the step's output over the payload rather than replacing it.
|
|
3109
|
+
//
|
|
3110
|
+
// Replacing is what made the Content Pipeline's exclusive pair unreachable. A While loop's
|
|
3111
|
+
// own output is a SUMMARY — `{iterations, succeeded, failed, results}` — so returning it
|
|
3112
|
+
// discarded the payload the iterations had built, including the `brandOK` the reviewer had
|
|
3113
|
+
// just set to true. The edges read `payload.brandOK === true` and `!== true`; against a
|
|
3114
|
+
// summary the first is false and the second is true, so the give-up branch won on EVERY run
|
|
3115
|
+
// no matter what the reviewer decided. The approved branch was unreachable in practice while
|
|
3116
|
+
// being perfectly reachable on the canvas.
|
|
3117
|
+
//
|
|
3118
|
+
// This is the same rule the mapped path already follows two lines down, and the same rule
|
|
3119
|
+
// the doc comment above states. The no-mapping branch was simply not following it.
|
|
3120
|
+
if (!outputMapping) {
|
|
3121
|
+
return output && typeof output === 'object' && !Array.isArray(output)
|
|
3122
|
+
? { ...payload, ...output }
|
|
3123
|
+
: output ?? payload;
|
|
3124
|
+
}
|
|
3125
|
+
const source = output && typeof output === 'object' ? output : { value: output };
|
|
3126
|
+
const { updates, errors, unmapped } = ApplyOutputMapping(source, outputMapping);
|
|
3127
|
+
for (const e of errors)
|
|
3128
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID}: ${e}`);
|
|
3129
|
+
// A mapping that names an output the step never produced discards that step's work while
|
|
3130
|
+
// the step reports Complete. It is not fatal — an action may emit a parameter only on some
|
|
3131
|
+
// paths — but it must not be silent, and naming what WAS returned turns a multi-table
|
|
3132
|
+
// forensic exercise into one line. The Content Pipeline demo lost an entire research pass
|
|
3133
|
+
// this way, every run, because its mapping named another action's parameter.
|
|
3134
|
+
if (unmapped?.length) {
|
|
3135
|
+
LogError(`[TaskGraphDispatcher] '${task.Name}' mapped output(s) the step did not return: ` +
|
|
3136
|
+
`${unmapped.join(', ')}. The step returned: ${Object.keys(source).join(', ') || '(nothing)'}. ` +
|
|
3137
|
+
`Those payload values were NOT written, so anything downstream reading them sees nothing.`);
|
|
3138
|
+
}
|
|
3139
|
+
return { ...payload, ...updates };
|
|
3140
|
+
}
|
|
3141
|
+
/**
|
|
3142
|
+
* Runs an Agent step, telling the runner where in the graph it sits.
|
|
3143
|
+
*
|
|
3144
|
+
* Depth and provenance are read together because they come from the same row: the graph's parent
|
|
3145
|
+
* task knows both how many continuation hops led here and which run submitted it.
|
|
3146
|
+
*/
|
|
3147
|
+
async runAgentNode(task, provider, effectiveInput, dependencyOutputs, onProgress) {
|
|
3148
|
+
const context = await this.graphContext(provider, task);
|
|
3149
|
+
return this.agentRunner.RunAgentForTask({
|
|
3150
|
+
TaskID: task.ID,
|
|
3151
|
+
AgentID: task.AgentID,
|
|
3152
|
+
InputPayload: effectiveInput,
|
|
3153
|
+
DependencyOutputs: dependencyOutputs,
|
|
3154
|
+
ContinuationDepth: context.Depth,
|
|
3155
|
+
SubmittingAgentRunID: context.SubmittingAgentRunID,
|
|
3156
|
+
Provider: provider,
|
|
3157
|
+
ContextUser: this.contextUser,
|
|
3158
|
+
OnProgress: onProgress,
|
|
3159
|
+
});
|
|
3160
|
+
}
|
|
3161
|
+
/**
|
|
3162
|
+
* Leaves exactly one open request standing for a task, withdrawing any others.
|
|
3163
|
+
*
|
|
3164
|
+
* The oldest wins — it is the one whose notification the assignee most likely already saw.
|
|
3165
|
+
*
|
|
3166
|
+
* **Called from every path that reads a task's open requests**, not only from the raise. That is
|
|
3167
|
+
* deliberate and it is the half R3-5 first got wrong: a task is notified exactly once, and
|
|
3168
|
+
* `notifyHumanTaskReady` returns at the marker forever after, so a duplicate minted after that
|
|
3169
|
+
* pass — by an instance that crashed between its insert and its de-dup, or by any build older
|
|
3170
|
+
* than this one — was never looked at again by the only code that could have collapsed it. The
|
|
3171
|
+
* waiting-task sweep is what actually reaches those.
|
|
3172
|
+
*
|
|
3173
|
+
* @param open the task's open requests, oldest first
|
|
3174
|
+
* @param keepIfSole the row this caller is responsible for, named only for the log
|
|
3175
|
+
*/
|
|
3176
|
+
async withdrawDuplicateRequests(provider, taskID, open, keepIfSole) {
|
|
3177
|
+
if (open.length <= 1)
|
|
3178
|
+
return;
|
|
3179
|
+
try {
|
|
3180
|
+
LogStatus(`[TaskGraphDispatcher] Task ${taskID} had ${open.length} open requests — keeping the ` +
|
|
3181
|
+
`oldest (${open[0].ID}${UUIDsEqual(open[0].ID, keepIfSole) ? ', this instance\'s' : ''}) and withdrawing the rest.`);
|
|
3182
|
+
for (const duplicate of open.slice(1)) {
|
|
3183
|
+
duplicate.Status = 'Canceled';
|
|
3184
|
+
duplicate.Comments = 'A duplicate request for the same step; the earlier one stands.';
|
|
3185
|
+
if (!(await duplicate.Save())) {
|
|
3186
|
+
LogError(`[TaskGraphDispatcher] Could not withdraw duplicate request ${duplicate.ID}: ` +
|
|
3187
|
+
`${duplicate.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
3188
|
+
}
|
|
3189
|
+
}
|
|
3190
|
+
}
|
|
3191
|
+
catch (e) {
|
|
3192
|
+
LogError(`[TaskGraphDispatcher] Could not de-duplicate requests for task ${taskID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
3193
|
+
}
|
|
3194
|
+
}
|
|
3195
|
+
/**
|
|
3196
|
+
* Closes the still-open asks raised for tasks that will never be answered.
|
|
3197
|
+
*
|
|
3198
|
+
* `Canceled` rather than `Expired`: nobody ran out of time, the ask was withdrawn — and the two
|
|
3199
|
+
* mean different things downstream, since an expired human step is treated as a FAILURE that a
|
|
3200
|
+
* give-up edge can route around, which would be a lie about a step the workflow decided it no
|
|
3201
|
+
* longer needed.
|
|
3202
|
+
*
|
|
3203
|
+
* Failures are logged and never propagated. The graph's outcome is already decided; refusing to
|
|
3204
|
+
* finish over an inbox row would trade a stale notification for a stalled workflow.
|
|
3205
|
+
*/
|
|
3206
|
+
async withdrawOpenRequests(provider, taskIDs, reason) {
|
|
3207
|
+
if (taskIDs.length === 0)
|
|
3208
|
+
return;
|
|
3209
|
+
try {
|
|
3210
|
+
const idList = taskIDs.map((id) => `'${id}'`).join(',');
|
|
3211
|
+
const open = await RunView.FromMetadataProvider(provider).RunView({
|
|
3212
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
3213
|
+
ExtraFilter: `Status='Requested' AND OriginatingTaskID IN (${idList})`,
|
|
3214
|
+
ResultType: 'entity_object',
|
|
3215
|
+
BypassCache: true,
|
|
3216
|
+
}, this.contextUser);
|
|
3217
|
+
if (!open.Success) {
|
|
3218
|
+
LogError(`[TaskGraphDispatcher] Could not read open requests to withdraw: ${open.ErrorMessage}`);
|
|
3219
|
+
return;
|
|
3220
|
+
}
|
|
3221
|
+
for (const request of open.Results ?? []) {
|
|
3222
|
+
request.Status = 'Canceled';
|
|
3223
|
+
request.Comments = reason;
|
|
3224
|
+
if (!(await request.Save())) {
|
|
3225
|
+
LogError(`[TaskGraphDispatcher] Could not withdraw request ${request.ID}: ` +
|
|
3226
|
+
`${request.LatestResult?.CompleteMessage ?? 'unknown error'}. It will keep showing ` +
|
|
3227
|
+
`in someone's inbox for a step that will never run.`);
|
|
3228
|
+
}
|
|
3229
|
+
}
|
|
3230
|
+
}
|
|
3231
|
+
catch (e) {
|
|
3232
|
+
LogError(`[TaskGraphDispatcher] Could not withdraw open requests: ${e instanceof Error ? e.message : String(e)}`);
|
|
3233
|
+
}
|
|
3234
|
+
}
|
|
3235
|
+
/**
|
|
3236
|
+
* Says once, per graph, that this instance settled work it cannot announce.
|
|
3237
|
+
*
|
|
3238
|
+
* Once because the sweep re-offers the graph every poll for the rest of its window, and a line
|
|
3239
|
+
* per poll would bury the thing it is trying to report — which is a DEPLOYMENT fact, not a graph
|
|
3240
|
+
* fact: if no instance anywhere carries a deliverer, these settlements never reach anyone.
|
|
3241
|
+
*/
|
|
3242
|
+
reportUndeliverableOnce(parentID) {
|
|
3243
|
+
if (this.reportedUndeliverable.has(parentID))
|
|
3244
|
+
return;
|
|
3245
|
+
this.reportedUndeliverable.add(parentID);
|
|
3246
|
+
LogStatus(`[TaskGraphDispatcher] Graph ${parentID} has settled but this instance has no continuation ` +
|
|
3247
|
+
`deliverer, so it is leaving the announcement to a peer that has one. If no instance in ` +
|
|
3248
|
+
`this deployment can deliver, the settlement will never be announced.`);
|
|
3249
|
+
}
|
|
3250
|
+
/**
|
|
3251
|
+
* Keeps a graph in this instance's sweep regardless of what its row timestamp says.
|
|
3252
|
+
*
|
|
3253
|
+
* Bounded, and the bound is about noise rather than surrender: past the cap the graph has failed
|
|
3254
|
+
* on every attempt for minutes, so another identical attempt will not fix it, and continuing
|
|
3255
|
+
* costs a full graph load per poll forever. It is reported once and left to the startup sweep.
|
|
3256
|
+
*/
|
|
3257
|
+
keepRetryingSettlement(parentID) {
|
|
3258
|
+
const passes = (this.retryingSettlement.get(parentID) ?? 0) + 1;
|
|
3259
|
+
if (passes > MAX_SETTLEMENT_RETRY_PASSES) {
|
|
3260
|
+
this.retryingSettlement.delete(parentID);
|
|
3261
|
+
LogError(`[TaskGraphDispatcher] Graph ${parentID} has failed to settle on ${MAX_SETTLEMENT_RETRY_PASSES} ` +
|
|
3262
|
+
`consecutive passes; this instance will stop re-queueing it. Its submitting run may be ` +
|
|
3263
|
+
`left Paused. A restart's startup sweep will try again.`);
|
|
3264
|
+
return;
|
|
3265
|
+
}
|
|
3266
|
+
this.retryingSettlement.set(parentID, passes);
|
|
3267
|
+
}
|
|
3268
|
+
/**
|
|
3269
|
+
* A graph's durable metadata bag, for the questions a pass asks of it.
|
|
3270
|
+
*
|
|
3271
|
+
* Defaults on any failure to read it, and the defaults are the safe directions: `'block'` means
|
|
3272
|
+
* a failed step decides nothing, so a graph whose metadata we cannot read stalls visibly instead
|
|
3273
|
+
* of resolving forks on the say-so of a failure; and no early-finish declaration means nothing
|
|
3274
|
+
* is removed from the claim filter on the strength of a read that did not work.
|
|
3275
|
+
*/
|
|
3276
|
+
async readParentMetadataFor(provider, parentTaskID) {
|
|
3277
|
+
try {
|
|
3278
|
+
const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
|
|
3279
|
+
if (await parent.Load(parentTaskID))
|
|
3280
|
+
return ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
3281
|
+
}
|
|
3282
|
+
catch { /* fall through to the safe defaults */ }
|
|
3283
|
+
return ParseTaskGraphParentMetadata(null);
|
|
3284
|
+
}
|
|
3285
|
+
/**
|
|
3286
|
+
* Whether the submitting run is in a state where this pass's writes to it will mean anything.
|
|
3287
|
+
*
|
|
3288
|
+
* **Read-only on purpose.** The settled branch's write order — layout, frame, cost, lifecycle,
|
|
3289
|
+
* delivery — is load-bearing and documented at each step; this asks the question those writes
|
|
3290
|
+
* depend on without joining them. What it prevents is a pass that goes through the motions and
|
|
3291
|
+
* then claims the delivery marker, making itself the last pass ever to look at the graph.
|
|
3292
|
+
*
|
|
3293
|
+
* Three answers, and the middle one is the bug:
|
|
3294
|
+
*
|
|
3295
|
+
* - **no run** — a scheduled or remote-triggered graph has nobody waiting. Proceed.
|
|
3296
|
+
* - **still `Running`** — `finalizeAgentRun` has not parked it yet. The graph beat its own
|
|
3297
|
+
* submitter to the finish line, which is ordinary for a fast graph and lasts milliseconds.
|
|
3298
|
+
* Defer: one poll later the run is parked and everything lands.
|
|
3299
|
+
* - **anything else** — `Paused` (settle it), or already `Completed`/`Failed`/`Cancelled` for
|
|
3300
|
+
* its own reasons (leave it; the lifecycle write's own guard declines). Proceed.
|
|
3301
|
+
*
|
|
3302
|
+
* **The deferral is bounded**, because "not parked yet" and "the submitting process died before
|
|
3303
|
+
* it could park" look identical from here. Waiting forever on the second would lose the outcome
|
|
3304
|
+
* of work that actually completed — strictly worse than announcing it late — so past the grace
|
|
3305
|
+
* period this proceeds and says why. The run itself stays `Running`, which is visibly wrong and
|
|
3306
|
+
* belongs to whatever reconciles abandoned runs, not to the graph that finished correctly.
|
|
3307
|
+
*/
|
|
3308
|
+
async submittingRunReadiness(provider, parent) {
|
|
3309
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
3310
|
+
if (!meta.submittedByAgentRunID)
|
|
3311
|
+
return { Verdict: 'ready', SubmitterCancelled: false };
|
|
3312
|
+
try {
|
|
3313
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
3314
|
+
if (!(await run.Load(meta.submittedByAgentRunID))) {
|
|
3315
|
+
// Transient, most likely. Deferring costs a poll; proceeding costs the marker.
|
|
3316
|
+
LogError(`[TaskGraphDispatcher] Could not read run ${meta.submittedByAgentRunID} to check whether graph ${parent.ID} may settle it; retrying next pass.`);
|
|
3317
|
+
return { Verdict: 'defer', SubmitterCancelled: false };
|
|
3318
|
+
}
|
|
3319
|
+
// A CANCELLED SUBMITTER HAS NOBODY WAITING (R2-9). Settlement still runs — the graph's
|
|
3320
|
+
// own bookkeeping is owed either way — but announcing it would message a conversation
|
|
3321
|
+
// about a workflow the user stopped, and for `reinvoke` would start a fresh billed turn
|
|
3322
|
+
// for the agent they cancelled.
|
|
3323
|
+
const cancelled = run.Status === 'Cancelled';
|
|
3324
|
+
const settledFor = parent.CompletedAt ? Date.now() - parent.CompletedAt.getTime() : 0;
|
|
3325
|
+
if (!IsSubmittingRunReady(run.Status, settledFor)) {
|
|
3326
|
+
// Still `Running` and inside the grace: `finalizeAgentRun` has not parked it yet.
|
|
3327
|
+
// Defer the whole run-half so nothing claims the marker — see the call site.
|
|
3328
|
+
return { Verdict: 'defer', SubmitterCancelled: cancelled };
|
|
3329
|
+
}
|
|
3330
|
+
if (run.Status === 'Running') {
|
|
3331
|
+
// Ready DESPITE being unparked means the grace has expired: the submitting process
|
|
3332
|
+
// most likely died before it could park. Proceeding loses nothing that is still
|
|
3333
|
+
// recoverable and stops a dead submitter holding a finished workflow's outcome.
|
|
3334
|
+
LogError(`[TaskGraphDispatcher] Run ${run.ID} has been Running for ${Math.round(settledFor / 1000)}s ` +
|
|
3335
|
+
`since graph ${parent.ID} settled — it never parked, so its submitting process most ` +
|
|
3336
|
+
`likely died. Settling and delivering the graph anyway; the run needs separate attention.`);
|
|
3337
|
+
}
|
|
3338
|
+
return { Verdict: 'ready', SubmitterCancelled: cancelled };
|
|
3339
|
+
}
|
|
3340
|
+
catch (e) {
|
|
3341
|
+
LogError(`[TaskGraphDispatcher] Could not check the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
3342
|
+
return { Verdict: 'defer', SubmitterCancelled: false };
|
|
3343
|
+
}
|
|
3344
|
+
}
|
|
3345
|
+
/**
|
|
3346
|
+
* Completes the agent run that parked on this graph.
|
|
3347
|
+
*
|
|
3348
|
+
* **This is the other half of submit-and-detach.** A run that dispatches a graph does not
|
|
3349
|
+
* complete at submission — it ends `Paused`, because reporting `Completed` above a workflow
|
|
3350
|
+
* where nothing has happened yet is a claim the row cannot support. The run's lifecycle is
|
|
3351
|
+
* finished HERE, when the graph it was waiting on actually settles, which is the first moment
|
|
3352
|
+
* the answer exists.
|
|
3353
|
+
*
|
|
3354
|
+
* Doing it from the dispatcher rather than by awaiting in the agent is what keeps the properties
|
|
3355
|
+
* that made detach right in the first place: a graph containing a human approval can park for
|
|
3356
|
+
* days without holding a conversation turn open, and a graph reclaimed by another instance after
|
|
3357
|
+
* a crash still settles its submitting run, because the settling happens wherever the graph
|
|
3358
|
+
* finishes rather than wherever it started.
|
|
3359
|
+
*
|
|
3360
|
+
* **Only a parked run is touched.** A run that is already `Completed`, `Failed` or `Cancelled`
|
|
3361
|
+
* reached that state for its own reasons — a second graph settling later, a run the user
|
|
3362
|
+
* cancelled, a run that failed after submitting — and overwriting it would rewrite history from
|
|
3363
|
+
* the outside. The `Paused` predicate is the whole guard.
|
|
3364
|
+
*
|
|
3365
|
+
* @param graphStatus the parent rollup's status: what the workflow as a whole did
|
|
3366
|
+
*/
|
|
3367
|
+
async settleSubmittingRun(provider, parent, graphStatus) {
|
|
3368
|
+
const meta = ParseTaskGraphParentMetadata(parent.InputPayload);
|
|
3369
|
+
if (!meta.submittedByAgentRunID)
|
|
3370
|
+
return 'done'; // a scheduled or remote-triggered graph has nobody waiting
|
|
3371
|
+
try {
|
|
3372
|
+
const run = await provider.GetEntityObject('MJ: AI Agent Runs', this.contextUser);
|
|
3373
|
+
if (!(await run.Load(meta.submittedByAgentRunID))) {
|
|
3374
|
+
LogError(`[TaskGraphDispatcher] Could not load run ${meta.submittedByAgentRunID} to settle it against graph ${parent.ID}.`);
|
|
3375
|
+
return 'defer';
|
|
3376
|
+
}
|
|
3377
|
+
if (run.Status !== 'Paused')
|
|
3378
|
+
return 'done';
|
|
3379
|
+
// The workflow's outcome becomes the run's outcome. A graph that ended any way other than
|
|
3380
|
+
// Complete did not do what the run started it to do, and a run reporting success over it
|
|
3381
|
+
// would be the same untruth in a different place.
|
|
3382
|
+
const succeeded = graphStatus === 'Complete';
|
|
3383
|
+
// COLUMN-SCOPED AND GUARDED ON `Paused` (C4), for the same reason every parent write has
|
|
3384
|
+
// been since Round 1: a full-row save carries a stale snapshot of a row another instance
|
|
3385
|
+
// may have moved, and the predicate makes the transition once-only rather than
|
|
3386
|
+
// last-write-wins.
|
|
3387
|
+
const settled = await this.claims.TrySettleRun(provider, run.ID, succeeded, succeeded ? null : `The workflow "${parent.Name}" ended ${graphStatus}.`, this.contextUser);
|
|
3388
|
+
if (!settled) {
|
|
3389
|
+
// Left parked rather than forced. A run stuck at Paused is visibly unfinished, which
|
|
3390
|
+
// is a state someone can investigate; a run flipped to Completed by a write that did
|
|
3391
|
+
// not land would be the same lie this whole change removes.
|
|
3392
|
+
// Rowcount 0 is either "a peer settled it first" — fine, and the status read above
|
|
3393
|
+
// would have caught the common case — or a write that did not land. Deferring covers
|
|
3394
|
+
// both: a peer's settle makes the next pass's `Paused` check return `done`.
|
|
3395
|
+
LogError(`[TaskGraphDispatcher] Could not settle run ${run.ID} against graph ${parent.ID}; ` +
|
|
3396
|
+
`it is no longer Paused or the write did not land. Retrying next pass.`);
|
|
3397
|
+
return 'defer';
|
|
3398
|
+
}
|
|
3399
|
+
LogStatus(`[TaskGraphDispatcher] Run ${run.ID} settled ${succeeded ? 'Completed' : 'Failed'} — ` +
|
|
3400
|
+
`workflow "${parent.Name}" ended ${graphStatus}.`);
|
|
3401
|
+
return 'done';
|
|
3402
|
+
}
|
|
3403
|
+
catch (e) {
|
|
3404
|
+
LogError(`[TaskGraphDispatcher] Could not settle the run waiting on graph ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
|
|
3405
|
+
return 'defer';
|
|
3406
|
+
}
|
|
3407
|
+
}
|
|
3408
|
+
/**
|
|
3409
|
+
* Gives every step that lacks one a position, once the graph has finished.
|
|
3410
|
+
*
|
|
3411
|
+
* **Why the run stores geometry at all.** A `TaskGraphSpec` is a logical structure with no
|
|
3412
|
+
* layout field, so a graph an agent emitted has no opinion about where its boxes go. Every
|
|
3413
|
+
* viewer was therefore laying it out for itself at render time — and a viewer that failed to
|
|
3414
|
+
* (because the canvas measures nodes it has not drawn yet) fell back to every node at the
|
|
3415
|
+
* origin, piled on one another, with the zoom-to-fit that follows fitting a one-node bounding
|
|
3416
|
+
* box. Settling it once, server-side, means the agent-run canvas, the Workflows runs tab and
|
|
3417
|
+
* anything built later all draw the same picture, and none of them has to compute it.
|
|
3418
|
+
*
|
|
3419
|
+
* **An authored position is never overwritten.** A workflow compiled from a Flow agent carries
|
|
3420
|
+
* the arrangement someone dragged into place; replacing it with an algorithm's guess would
|
|
3421
|
+
* discard a deliberate act. Only steps with no geometry get one, so a partially-arranged graph
|
|
3422
|
+
* keeps what it has.
|
|
3423
|
+
*
|
|
3424
|
+
* Failure here is logged and swallowed: this is presentation. A graph whose work completed must
|
|
3425
|
+
* not be reported as failed because its picture could not be saved.
|
|
3426
|
+
*/
|
|
3427
|
+
async persistComputedLayout(graph) {
|
|
3428
|
+
try {
|
|
3429
|
+
const needsLayout = [...graph.entityById.values()].filter((t) => !this.parseConfiguration(t)?.layout);
|
|
3430
|
+
if (needsLayout.length === 0)
|
|
3431
|
+
return;
|
|
3432
|
+
// Laid out over the WHOLE graph, not just the nodes missing geometry: position depends on
|
|
3433
|
+
// where a node sits in the topology, and a layout computed over a subset would place its
|
|
3434
|
+
// nodes as though the rest of the workflow did not exist.
|
|
3435
|
+
const edges = graph.edges.map((e) => ({ From: e.dependsOnTaskId, To: e.taskId }));
|
|
3436
|
+
const positions = LayoutGraphNodes([...graph.entityById.keys()], edges, { Direction: 'LR' });
|
|
3437
|
+
for (const task of needsLayout) {
|
|
3438
|
+
const position = positions.get(task.ID);
|
|
3439
|
+
if (!position)
|
|
3440
|
+
continue;
|
|
3441
|
+
const existing = this.parseConfiguration(task);
|
|
3442
|
+
const merged = {
|
|
3443
|
+
...existing,
|
|
3444
|
+
layout: { x: position.X, y: position.Y },
|
|
3445
|
+
};
|
|
3446
|
+
task.Configuration = JSON.stringify(merged);
|
|
3447
|
+
if (!(await task.Save())) {
|
|
3448
|
+
LogError(`[TaskGraphDispatcher] Could not save computed layout for ${task.ID}: ${task.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
3449
|
+
}
|
|
3450
|
+
}
|
|
3451
|
+
}
|
|
3452
|
+
catch (e) {
|
|
3453
|
+
LogError(`[TaskGraphDispatcher] Could not compute a layout for the settled graph: ${e instanceof Error ? e.message : String(e)}`);
|
|
3454
|
+
}
|
|
3455
|
+
}
|
|
3456
|
+
/**
|
|
3457
|
+
* The earliest moment any step in the graph began, or null when none has.
|
|
3458
|
+
*
|
|
3459
|
+
* Null is a real answer — a graph whose tasks are all still Pending has not started — and is
|
|
3460
|
+
* deliberately not collapsed to "now", which would date the graph from whenever this pass
|
|
3461
|
+
* happened to run.
|
|
3462
|
+
*/
|
|
3463
|
+
earliestStart(entityById) {
|
|
3464
|
+
let earliest = null;
|
|
3465
|
+
for (const entity of entityById.values()) {
|
|
3466
|
+
if (!entity.StartedAt)
|
|
3467
|
+
continue;
|
|
3468
|
+
if (earliest === null || entity.StartedAt < earliest)
|
|
3469
|
+
earliest = entity.StartedAt;
|
|
3470
|
+
}
|
|
3471
|
+
return earliest;
|
|
3472
|
+
}
|
|
3473
|
+
/**
|
|
3474
|
+
* The step's Configuration with this run's artefacts folded in, or `undefined` to leave it be.
|
|
3475
|
+
*
|
|
3476
|
+
* **Merged into the authored bag, never written over it.** The Configuration column holds the
|
|
3477
|
+
* step's definition — its loop body, its mappings, its policy, the position someone dragged it
|
|
3478
|
+
* to. Writing a fresh object containing only `runtime` would erase all of that the first time a
|
|
3479
|
+
* prompt step completed, which is the kind of loss that surfaces much later as a workflow that
|
|
3480
|
+
* mysteriously stopped mapping its output.
|
|
3481
|
+
*
|
|
3482
|
+
* Returns `undefined` when there is nothing to record, so the guarded write omits the column
|
|
3483
|
+
* rather than rewriting it with what it already held.
|
|
3484
|
+
*/
|
|
3485
|
+
configurationWithRuntime(task, promptRunID, actionLogID, iterations, payloadAtStart) {
|
|
3486
|
+
if (!promptRunID && !actionLogID && !iterations?.length && !payloadAtStart)
|
|
3487
|
+
return undefined;
|
|
3488
|
+
const existing = this.parseConfiguration(task);
|
|
3489
|
+
const merged = {
|
|
3490
|
+
...existing,
|
|
3491
|
+
runtime: {
|
|
3492
|
+
...existing?.runtime,
|
|
3493
|
+
...(promptRunID ? { promptRunID } : {}),
|
|
3494
|
+
...(actionLogID ? { actionLogID } : {}),
|
|
3495
|
+
// Replaced wholesale rather than appended: this is the trace of the loop's LAST
|
|
3496
|
+
// execution, and a retried step that concatenated would report a loop that ran twice
|
|
3497
|
+
// as many passes as it did.
|
|
3498
|
+
...(iterations?.length ? { iterations } : {}),
|
|
3499
|
+
// The resolved before-state, so the run view has something to diff the output
|
|
3500
|
+
// against. NOT written to Task.InputPayload, which holds the AUTHORED input and
|
|
3501
|
+
// round-trips back out as part of the spec.
|
|
3502
|
+
...(payloadAtStart ? { payloadAtStart } : {}),
|
|
3503
|
+
},
|
|
3504
|
+
};
|
|
3505
|
+
return JSON.stringify(merged);
|
|
3506
|
+
}
|
|
3507
|
+
/**
|
|
3508
|
+
* Reads a step's Configuration bag, tolerating a row whose JSON cannot be parsed.
|
|
3509
|
+
*
|
|
3510
|
+
* Unparseable configuration is logged rather than thrown: the step has already RUN by the time
|
|
3511
|
+
* this is called, and refusing to record its outcome because its definition is malformed would
|
|
3512
|
+
* discard the result of real work and leave the task claimed until the claim lapsed.
|
|
3513
|
+
*/
|
|
3514
|
+
parseConfiguration(task) {
|
|
3515
|
+
if (!task.Configuration)
|
|
3516
|
+
return undefined;
|
|
3517
|
+
try {
|
|
3518
|
+
return JSON.parse(task.Configuration);
|
|
3519
|
+
}
|
|
3520
|
+
catch (e) {
|
|
3521
|
+
LogError(`[TaskGraphDispatcher] Task ${task.ID} has unparseable Configuration; ` +
|
|
3522
|
+
`recording runtime artefacts against an empty bag. ${e instanceof Error ? e.message : String(e)}`);
|
|
3523
|
+
return undefined;
|
|
3524
|
+
}
|
|
3525
|
+
}
|
|
3526
|
+
/**
|
|
3527
|
+
* The payload a step sees: everything its prerequisites produced, plus its own declared input.
|
|
3528
|
+
*
|
|
3529
|
+
* **Why the outputs are merged rather than kept per-task.** A flow carried ONE payload that
|
|
3530
|
+
* accumulated as it went, so a condition on the edge into step C could read a value step A wrote.
|
|
3531
|
+
* Handing each task only its immediate predecessor's output would silently narrow that: the
|
|
3532
|
+
* condition reads `undefined`, which is falsy, and the workflow quietly takes a different route
|
|
3533
|
+
* than the flow it was compiled from. Merging in dependency order restores the accumulation.
|
|
3534
|
+
*
|
|
3535
|
+
* Later prerequisites win on a key collision, matching a flow's own last-write-wins behaviour.
|
|
3536
|
+
*/
|
|
3537
|
+
mergedPayload(inputPayload, dependencyOutputs) {
|
|
3538
|
+
const merged = {};
|
|
3539
|
+
for (const output of dependencyOutputs.values()) {
|
|
3540
|
+
if (output && typeof output === 'object' && !Array.isArray(output)) {
|
|
3541
|
+
Object.assign(merged, output);
|
|
3542
|
+
}
|
|
3543
|
+
}
|
|
3544
|
+
if (inputPayload && typeof inputPayload === 'object' && !Array.isArray(inputPayload)) {
|
|
3545
|
+
Object.assign(merged, inputPayload);
|
|
3546
|
+
}
|
|
3547
|
+
return merged;
|
|
3548
|
+
}
|
|
734
3549
|
}
|
|
735
3550
|
//# sourceMappingURL=TaskGraphDispatcher.js.map
|