@memberjunction/task-graph 6.1.0-edge.1 → 6.1.0-edge.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/LICENSE +180 -4
  2. package/README.md +214 -0
  3. package/dist/TaskClaimStore.d.ts +387 -4
  4. package/dist/TaskClaimStore.d.ts.map +1 -1
  5. package/dist/TaskClaimStore.js +605 -20
  6. package/dist/TaskClaimStore.js.map +1 -1
  7. package/dist/TaskGraphDispatcher.d.ts +668 -5
  8. package/dist/TaskGraphDispatcher.d.ts.map +1 -1
  9. package/dist/TaskGraphDispatcher.js +2942 -127
  10. package/dist/TaskGraphDispatcher.js.map +1 -1
  11. package/dist/TaskGraphService.d.ts +364 -5
  12. package/dist/TaskGraphService.d.ts.map +1 -1
  13. package/dist/TaskGraphService.js +1039 -43
  14. package/dist/TaskGraphService.js.map +1 -1
  15. package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -1
  16. package/dist/TaskGraphSubmitterImpl.js +5 -0
  17. package/dist/TaskGraphSubmitterImpl.js.map +1 -1
  18. package/dist/TaskLoopExecutor.d.ts +62 -0
  19. package/dist/TaskLoopExecutor.d.ts.map +1 -0
  20. package/dist/TaskLoopExecutor.js +248 -0
  21. package/dist/TaskLoopExecutor.js.map +1 -0
  22. package/dist/WorkflowSpecSync.d.ts +28 -2
  23. package/dist/WorkflowSpecSync.d.ts.map +1 -1
  24. package/dist/WorkflowSpecSync.js +83 -2
  25. package/dist/WorkflowSpecSync.js.map +1 -1
  26. package/dist/condition-gate.d.ts +128 -0
  27. package/dist/condition-gate.d.ts.map +1 -0
  28. package/dist/condition-gate.js +257 -0
  29. package/dist/condition-gate.js.map +1 -0
  30. package/dist/debug-state.d.ts +102 -0
  31. package/dist/debug-state.d.ts.map +1 -0
  32. package/dist/debug-state.js +135 -0
  33. package/dist/debug-state.js.map +1 -0
  34. package/dist/index.d.ts +7 -0
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +7 -0
  37. package/dist/index.js.map +1 -1
  38. package/dist/operations/TaskGraphDebugOperations.d.ts +99 -0
  39. package/dist/operations/TaskGraphDebugOperations.d.ts.map +1 -0
  40. package/dist/operations/TaskGraphDebugOperations.js +310 -0
  41. package/dist/operations/TaskGraphDebugOperations.js.map +1 -0
  42. package/dist/operations/TaskGraphOperations.d.ts +20 -2
  43. package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
  44. package/dist/operations/TaskGraphOperations.js +51 -8
  45. package/dist/operations/TaskGraphOperations.js.map +1 -1
  46. package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
  47. package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
  48. package/dist/operations/WorkflowDraftOperation.js +141 -0
  49. package/dist/operations/WorkflowDraftOperation.js.map +1 -0
  50. package/dist/settlement-rescue.d.ts +85 -0
  51. package/dist/settlement-rescue.d.ts.map +1 -0
  52. package/dist/settlement-rescue.js +119 -0
  53. package/dist/settlement-rescue.js.map +1 -0
  54. package/dist/task-graph-kick.d.ts +3 -0
  55. package/dist/task-graph-kick.d.ts.map +1 -0
  56. package/dist/task-graph-kick.js +17 -0
  57. package/dist/task-graph-kick.js.map +1 -0
  58. package/dist/task-predicates.d.ts +77 -0
  59. package/dist/task-predicates.d.ts.map +1 -0
  60. package/dist/task-predicates.js +75 -0
  61. package/dist/task-predicates.js.map +1 -0
  62. package/dist/types.d.ts +224 -1
  63. package/dist/types.d.ts.map +1 -1
  64. package/dist/types.js.map +1 -1
  65. package/package.json +12 -8
@@ -15,8 +15,27 @@
15
15
  *
16
16
  * @module @memberjunction/task-graph
17
17
  */
18
- import { LogError, LogStatus, RunView, } from '@memberjunction/core';
19
- import { FormatValidationErrors, NormalizeDependency, ValidateTaskGraphSpec, } from '@memberjunction/ai-core-plus';
18
+ import { LogError, LogStatus, RunInEntityTransaction, RunView, } from '@memberjunction/core';
19
+ import { FormatValidationErrors, NormalizeDependency, RankGraphNodes, SanitizeInvocationEnvelope, ValidateTaskGraphSpec, ConfigOf, } from '@memberjunction/ai-core-plus';
20
+ import { UUIDsEqual } from '@memberjunction/global';
21
+ import { TaskClaimStore } from './TaskClaimStore.js';
22
+ import { ParseTaskGraphDebugState } from './debug-state.js';
23
+ import { KickTaskGraphDispatchers } from './task-graph-kick.js';
24
+ /**
25
+ * Normalizes a caller-supplied reinvoke depth to a safe cap seed.
26
+ *
27
+ * The runaway-loop cap is a signed comparison over a persisted-verbatim seed, and the remote Submit
28
+ * operation is the one seam where the seed is caller-supplied rather than computed by the engine —
29
+ * a negative value would BUY hops (`-1000` turns a 5-hop cap into 1005). Clamped rather than
30
+ * refused so a stale client sending zero-adjacent noise keeps working, and no accepted value can
31
+ * ever weaken the cap. Exported pure because a boundary rule that cannot be tested directly is a
32
+ * boundary nobody will notice moving.
33
+ */
34
+ export function ClampReinvokeDepth(value) {
35
+ return typeof value === 'number' && Number.isFinite(value)
36
+ ? Math.max(0, Math.floor(value))
37
+ : undefined;
38
+ }
20
39
  /**
21
40
  * Continuation chains are bounded separately from graph nesting.
22
41
  *
@@ -31,6 +50,7 @@ export const MAX_REINVOKE_DEPTH = 5;
31
50
  const DEFAULT_PARENT_METADATA = {
32
51
  continuation: 'message',
33
52
  reinvokeDepth: 0,
53
+ failureSemantics: 'block',
34
54
  submittedByAgentRunID: null,
35
55
  submittedByUserID: null,
36
56
  };
@@ -61,20 +81,259 @@ export function ParseTaskGraphParentMetadata(raw) {
61
81
  continuation: parsed.continuation === 'reinvoke' || parsed.continuation === 'none'
62
82
  ? parsed.continuation
63
83
  : 'message',
84
+ failureSemantics: parsed.failureSemantics === 'edges' ? 'edges' : 'block',
64
85
  reinvokeDepth: Number.isFinite(parsed.reinvokeDepth) ? Number(parsed.reinvokeDepth) : 0,
86
+ // Guarded like the others: this is read to explain a settlement after the fact, and an
87
+ // arbitrary string arriving from a hand edit should read as "unknown", not be echoed.
88
+ continuationDeliveredAs: DELIVERY_OUTCOMES.has(parsed.continuationDeliveredAs)
89
+ ? parsed.continuationDeliveredAs
90
+ : undefined,
65
91
  };
66
92
  }
67
93
  catch {
68
94
  return { ...DEFAULT_PARENT_METADATA };
69
95
  }
70
96
  }
97
+ /**
98
+ * The JSON bag `persistParent` writes. Exported so start-paused is testable without a database —
99
+ * Pause-after-submit races the first dispatcher poll, so this bag is the only place `paused: true`
100
+ * is guaranteed to land before anyone claims.
101
+ */
102
+ export function BuildTaskGraphParentInputPayload(args) {
103
+ return {
104
+ continuation: args.continuation,
105
+ reinvokeDepth: args.reinvokeDepth,
106
+ failureSemantics: args.failureSemantics,
107
+ submittedByAgentRunID: args.submittedByAgentRunID,
108
+ submittedByUserID: args.submittedByUserID,
109
+ ...(args.invocation
110
+ ? { invocation: { data: args.invocation.data, context: args.invocation.context } }
111
+ : {}),
112
+ ...(args.startPaused
113
+ ? {
114
+ debug: {
115
+ paused: true,
116
+ pausedReason: 'user',
117
+ pausedBy: args.submittedByUserID,
118
+ },
119
+ }
120
+ : {}),
121
+ };
122
+ }
123
+ /**
124
+ * How a settlement's announcement ended — the values `TryClaimContinuation` may record.
125
+ *
126
+ * `expired` means found too late to announce; `cancelled` means there was deliberately nobody left
127
+ * to announce to, because the run that submitted the graph was cancelled. Both are settlements that
128
+ * completed WITHOUT an announcement, and keeping them distinct is the difference between "we missed
129
+ * it" and "we chose not to".
130
+ */
131
+ const DELIVERY_OUTCOMES = new Set(['delivered', 'expired', 'cancelled']);
71
132
  /** True when a continuation chain has gone as far as it may. */
72
133
  export function IsReinvokeCapReached(meta) {
73
134
  return meta.reinvokeDepth >= MAX_REINVOKE_DEPTH;
74
135
  }
75
136
  /** Name of the task type used for agent-orchestrated graphs. */
76
- const TASK_TYPE_NAME = 'AI Workflow';
137
+ export const TASK_TYPE_NAME = 'AI Workflow';
138
+ /**
139
+ * The node kinds a `Task` row can actually represent, and therefore the ones the dispatcher can run.
140
+ *
141
+ * `Agent` and `Action` have their own foreign keys; `Human` is a task with an assignee; `ForEach`
142
+ * and `While` carry their loop definition in `Task.Configuration` and their repeated body in the
143
+ * same `AgentID` / `ActionID` keys.
144
+ *
145
+ * `Prompt` joins them now that `TaskPromptRunner` exists — it carries `Task.PromptID`. `External`
146
+ * remains absent: it is completed by a system that has no way to report back, so persisting one
147
+ * would produce a task that waits forever.
148
+ */
149
+ const DISPATCHABLE_KINDS = [
150
+ 'Agent', 'Action', 'Human', 'ForEach', 'While', 'Prompt',
151
+ ];
152
+ /**
153
+ * Reports the node kinds this dispatcher cannot execute, or `null` when the graph is fully runnable.
154
+ *
155
+ * **Why this is a submit-time check and not a validation rule.** `ValidateTaskGraphSpec` is a pure
156
+ * function over the spec: it answers "is this a well-formed graph?", and its answer has to be the
157
+ * same in a browser, a CLI and a server. "Can it run *here*?" is a different question whose answer
158
+ * changes as runners are added, so it belongs to the runtime that owns the runners.
159
+ *
160
+ * **Why refuse rather than persist-and-stall.** Before this existed, `persistTasks` keyed off which
161
+ * configuration field happened to be populated and fell through to "Human task assigned to the
162
+ * submitter" for everything else — so a loop step silently became an approval request nobody asked
163
+ * for and nothing would ever complete. A graph that hangs forever while *looking* like it is waiting
164
+ * on a person is the most expensive failure available here. Refusing at the door instead names the
165
+ * offending step while the workflow is still the author's to edit.
166
+ */
167
+ /**
168
+ * Projects a spec node onto the `Task.Configuration` bag.
169
+ *
170
+ * Everything the row cannot hold in a column of its own: the kind-specific settings, the payload
171
+ * mappings, and the execution policy. Returning `null` for an empty result keeps `Configuration`
172
+ * NULL rather than `"{}"`, so "this step has no settings" reads the same in the database as it does
173
+ * in the spec.
174
+ *
175
+ * **The mappings are the point.** They are how a step's result reaches the payload, and every
176
+ * branch condition downstream reads the payload — so a step persisted without them produces a
177
+ * workflow whose conditions all evaluate against nothing. Undefined is falsy, so that failure looks
178
+ * exactly like a branch legitimately not being taken.
179
+ */
180
+ export function BuildStepConfiguration(node) {
181
+ const config = {};
182
+ const agent = ConfigOf(node, 'Agent');
183
+ if (agent?.message || agent?.templateParameters) {
184
+ config.agent = { message: agent.message, templateParameters: agent.templateParameters };
185
+ }
186
+ const prompt = ConfigOf(node, 'Prompt');
187
+ if (prompt?.templateParameters)
188
+ config.prompt = { templateParameters: prompt.templateParameters };
189
+ const forEach = ConfigOf(node, 'ForEach');
190
+ if (forEach)
191
+ config.forEach = forEach;
192
+ const whileOp = ConfigOf(node, 'While');
193
+ if (whileOp)
194
+ config.while = whileOp;
195
+ // `expiresInHours` as well as instructions: dropping it here is what would leave the deadline in
196
+ // the spec and out of the row the dispatcher actually reads, so the request would be raised with
197
+ // no ExpiresAt and the author's timeout would silently not exist.
198
+ const human = ConfigOf(node, 'Human');
199
+ if (human?.instructions || human?.expiresInHours) {
200
+ config.human = { instructions: human.instructions, expiresInHours: human.expiresInHours };
201
+ }
202
+ const external = ConfigOf(node, 'External');
203
+ if (external)
204
+ config.external = external;
205
+ // Mappings live on the Action arm of the spec, but they are not action-specific: a loop step
206
+ // carries them too, which is how its per-iteration inputs and results are wired.
207
+ const action = ConfigOf(node, 'Action');
208
+ const inputMapping = action?.inputMapping;
209
+ const outputMapping = action?.outputMapping;
210
+ if (inputMapping)
211
+ config.inputMapping = inputMapping;
212
+ if (outputMapping)
213
+ config.outputMapping = outputMapping;
214
+ if (node.policy) {
215
+ config.policy = {
216
+ timeoutSeconds: node.policy.timeoutSeconds,
217
+ retryCount: node.policy.retryCount,
218
+ onError: node.policy.onError,
219
+ };
220
+ }
221
+ // The author's own arrangement, carried through so a hand-drawn workflow runs — and appears in
222
+ // run history — in the shape they drew. Dropping it (as this did) meant a workflow someone had
223
+ // laid out carefully came back as a machine-arranged graph the first time they watched it run.
224
+ // Only authored geometry is stored: a graph with none has its layout derived at render time, and
225
+ // persisting a derived layout would freeze one rendering of a graph that can still change.
226
+ if (node.layout && Object.keys(node.layout).length > 0) {
227
+ config.layout = {
228
+ x: node.layout.x,
229
+ y: node.layout.y,
230
+ width: node.layout.width,
231
+ height: node.layout.height,
232
+ };
233
+ }
234
+ return Object.keys(config).length > 0 ? config : null;
235
+ }
236
+ /** Internal alias so the persistence path reads as a step, not as a projection. */
237
+ const buildStepConfiguration = BuildStepConfiguration;
238
+ /**
239
+ * The loop definition on a node, whichever loop kind it is.
240
+ *
241
+ * ForEach and While differ in how they decide to iterate, not in what they repeat, so everything
242
+ * downstream of that decision — body resolution, name collection, persistence — treats them alike.
243
+ */
244
+ export function LoopOperationOf(node) {
245
+ return ConfigOf(node, 'ForEach') ?? ConfigOf(node, 'While') ?? null;
246
+ }
247
+ /** Every agent name a node references, including the sub-agent a loop repeats. */
248
+ function agentNamesIn(node) {
249
+ const names = [ConfigOf(node, 'Agent')?.agentName, LoopOperationOf(node)?.subAgent?.name];
250
+ return names.filter((n) => !!n);
251
+ }
252
+ /** Every prompt name a node references, including the prompt a loop repeats. */
253
+ function promptNamesIn(node) {
254
+ const names = [ConfigOf(node, 'Prompt')?.promptName, LoopOperationOf(node)?.prompt?.name];
255
+ return names.filter((n) => !!n);
256
+ }
257
+ /** Every action name a node references, including the action a loop repeats. */
258
+ function actionNamesIn(node) {
259
+ const names = [ConfigOf(node, 'Action')?.actionName, LoopOperationOf(node)?.action?.name];
260
+ return names.filter((n) => !!n);
261
+ }
262
+ /**
263
+ * The transaction capability of a provider, when it has one.
264
+ *
265
+ * `IMetadataProvider` does not declare transaction support — a browser provider genuinely has none —
266
+ * so this narrows by CAPABILITY rather than asserting a type the interface does not promise. A
267
+ * provider without it returns undefined and `RunInEntityTransaction` runs the work directly, which
268
+ * is the honest degradation: server submissions get atomicity, and a client submission behaves
269
+ * exactly as it did before rather than failing at a call site that claimed something untrue.
270
+ */
271
+ function asTransactionCapable(provider) {
272
+ const candidate = provider;
273
+ return candidate.SupportsEntityTransactions === true && typeof candidate.BeginEntityTransaction === 'function'
274
+ ? candidate
275
+ : undefined;
276
+ }
277
+ export function FindUnrunnableKinds(spec) {
278
+ const offenders = spec.tasks.filter((t) => !DISPATCHABLE_KINDS.includes(t.kind));
279
+ if (offenders.length === 0)
280
+ return null;
281
+ const detail = offenders.map((t) => `"${t.name}" (${t.kind})`).join(', ');
282
+ return (`"${spec.workflowName}" cannot be run yet: ${detail}. ` +
283
+ `The dispatcher runs agent, action, person and loop steps. Prompt steps and steps completed ` +
284
+ `by an outside system are not supported yet — replace them, or split them out of this workflow.`);
285
+ }
286
+ /**
287
+ * Reports human steps assigned to someone other than the submitter, or `null` when there are none.
288
+ *
289
+ * Cross-user assignment needs an authorization model (#3524) — deciding that A may put work in B's
290
+ * inbox is a permissions question, not a graph question. Until it lands, a workflow can only ask the
291
+ * person who started it.
292
+ *
293
+ * **Why refuse rather than reassign.** Persist wrote `task.UserID = submitter` unconditionally, so
294
+ * an authored `assignToUserID` was overwritten in silence. Every layer above accepts the field —
295
+ * the flow compiler reads it into the spec, the validator passes it, the spec type declares it — so
296
+ * silence here is indistinguishable from support: the graph submits, a step appears in the WRONG
297
+ * person's inbox, the named person is never told, and the author has no reason to suspect any of it.
298
+ * Refusing while the graph is still the author's to edit is the only point at which saying so costs
299
+ * nothing.
300
+ */
301
+ export function FindCrossUserAssignments(spec, submitterUserID) {
302
+ const offenders = spec.tasks.filter((t) => {
303
+ const assignTo = ConfigOf(t, 'Human')?.assignToUserID;
304
+ return !!assignTo && !UUIDsEqual(assignTo, submitterUserID);
305
+ });
306
+ if (offenders.length === 0)
307
+ return null;
308
+ const detail = offenders.map((t) => `"${t.name}"`).join(', ');
309
+ return (`"${spec.workflowName}" was not started: ${detail} asks a person other than whoever runs the ` +
310
+ `workflow. Assigning a step to someone else is not available yet (#3524) — a workflow can ` +
311
+ `only ask the person who started it. Remove assignToUserID from those steps.`);
312
+ }
77
313
  export class TaskGraphService {
314
+ constructor() {
315
+ /**
316
+ * Guarded single-statement writes, shared with the dispatcher.
317
+ *
318
+ * The instance id is descriptive only — this service never CLAIMS anything, it only issues
319
+ * guarded transitions whose predicates are about the row's own status rather than about who
320
+ * holds it.
321
+ */
322
+ this.claims = new TaskClaimStore('task-graph-service', 0);
323
+ // ────────────────────────────────────────────────────────────────────────
324
+ // debug / runner control plane
325
+ //
326
+ // Every verb here is durable, declarative state the dispatcher's claim filter consults on its
327
+ // next pass — never a call into a running dispatcher. That is what makes the controls work
328
+ // across instances and restarts, and what bounds their latency to one poll interval. See
329
+ // `debug-state.ts` for the model.
330
+ // ────────────────────────────────────────────────────────────────────────
331
+ /**
332
+ * Store for the guarded JSON_MODIFY writes. The instance identity and TTL are claim-protocol
333
+ * concerns this class never exercises — the debug writes are instance-free.
334
+ */
335
+ this.debugWrites = new TaskClaimStore('task-graph-service', 0);
336
+ }
78
337
  /**
79
338
  * Validates and persists a task graph, returning as soon as it is durable.
80
339
  *
@@ -92,22 +351,90 @@ export class TaskGraphService {
92
351
  LogError(`[TaskGraphService] ${message}`);
93
352
  return { Success: false, ErrorMessage: message };
94
353
  }
354
+ // 1b. Representability. Validation asks "is this a well-formed graph?"; this asks "can THIS
355
+ // dispatcher run it?" — a different question, and one the pure validator has no business
356
+ // answering, since capability is a property of the runtime, not of the spec.
357
+ const unrunnable = this.findUnrunnableKinds(spec);
358
+ if (unrunnable) {
359
+ LogError(`[TaskGraphService] ${unrunnable}`);
360
+ return { Success: false, ErrorMessage: unrunnable };
361
+ }
362
+ // 1b-ii. Assignability. Same question, narrower: this dispatcher can only ask the person who
363
+ // submitted the graph. Persist used to overwrite an authored `assignToUserID` with
364
+ // the submitter and say nothing, so a step meant for someone else landed in the
365
+ // wrong inbox and the named person was never told. The compiler accepts the field and
366
+ // the validator passes it, which makes silence here indistinguishable from support.
367
+ const misassigned = FindCrossUserAssignments(spec, context.ContextUser.ID);
368
+ if (misassigned) {
369
+ LogError(`[TaskGraphService] ${misassigned}`);
370
+ return { Success: false, ErrorMessage: misassigned };
371
+ }
372
+ // 1c. Chain depth. A flow that dispatches a graph containing itself recurses without bound,
373
+ // and each hop costs real money and real rows before anyone notices. The cap is checked
374
+ // HERE rather than at execution because refusing to write the graph is the only point at
375
+ // which nothing has happened yet.
376
+ const depth = context.ReinvokeDepth ?? 0;
377
+ if (depth >= MAX_REINVOKE_DEPTH) {
378
+ const message = `"${spec.workflowName}" was not started: it is ${depth} levels deep in a chain of ` +
379
+ `workflows starting workflows, which is the limit. A workflow that reaches this is ` +
380
+ `almost always calling itself, directly or through another one.`;
381
+ LogError(`[TaskGraphService] ${message}`);
382
+ return { Success: false, ErrorMessage: message };
383
+ }
95
384
  try {
96
- // 2. Resolve every agent BEFORE writing anything. An unresolvable agent is a hard
97
- // error, not a skipped node: silently dropping a task executes the graph with holes
98
- // where the caller's work should have been.
385
+ // 2. Resolve every agent and action BEFORE writing anything. An unresolvable name is a
386
+ // hard error, not a skipped node: silently dropping a task executes the graph with
387
+ // holes where the caller's work should have been.
99
388
  const agentIDsByName = await this.resolveAgents(spec, context);
100
389
  if (!agentIDsByName.Success) {
101
390
  return { Success: false, ErrorMessage: agentIDsByName.ErrorMessage };
102
391
  }
392
+ const actionIDsByName = await this.resolveActions(spec, context);
393
+ if (!actionIDsByName.Success) {
394
+ return { Success: false, ErrorMessage: actionIDsByName.ErrorMessage };
395
+ }
396
+ const promptIDsByName = await this.resolvePrompts(spec, context);
397
+ if (!promptIDsByName.Success) {
398
+ return { Success: false, ErrorMessage: promptIDsByName.ErrorMessage };
399
+ }
103
400
  const taskTypeID = await this.ensureTaskType(context);
104
- // 3. Persist. Parent first so children have a ParentID, then children, then edges —
105
- // edges last because they reference two child IDs that must both exist.
106
- const parentTaskID = await this.persistParent(spec, taskTypeID, context);
107
- const taskIDMap = await this.persistChildren(spec, parentTaskID, taskTypeID, agentIDsByName.Map, context);
108
- await this.persistDependencies(spec, taskIDMap, context);
401
+ // 3. Persist — ALL of it, or none of it.
402
+ //
403
+ // ── The whole graph becomes visible together, or not at all ─────────────────────────
404
+ // The dispatcher discovers work by polling for child tasks in 'Pending'. Writing the
405
+ // children first and their dependencies afterwards leaves a window — milliseconds, but
406
+ // real — in which every task exists with NO prerequisites recorded yet. A poll landing
407
+ // there sees a graph of independent tasks and claims all of them at once.
408
+ //
409
+ // Observed, not theorised: in graph C08B36E3 the dependency gating `Research: focused`
410
+ // was written at 00:00:59.653 and that task STARTED at 00:00:59.647 — six milliseconds
411
+ // before the edge that was supposed to hold it back existed. `Close out: approved` ran
412
+ // in the same wave as the research steps, before the draft it was meant to judge had
413
+ // been written. The graph then reported Complete, having executed in an order its author
414
+ // never drew.
415
+ //
416
+ // A transaction is the whole fix: the dispatcher cannot observe a half-built graph
417
+ // because a half-built graph is never visible. `RunInEntityTransaction` degrades to
418
+ // running the work as-is on a provider that cannot transact, which is the correct
419
+ // fallback rather than a silent failure.
420
+ //
421
+ // The PARENT is inside it too. Written outside, a failure while persisting children or
422
+ // edges would roll those back and leave a childless parent durably in 'Pending' — which
423
+ // never settles (the rollup deliberately skips a graph with no nodes) and never dies, so
424
+ // the active-graph scan picks it up on every poll forever: permanent debris plus a
425
+ // permanent tick of wasted work for each failed submit. Ordering inside the transaction
426
+ // is unchanged, because edges only ever reference child IDs.
427
+ const { parentTaskID, taskIDMap } = await RunInEntityTransaction(asTransactionCapable(context.Provider), async () => {
428
+ // Parent first so children have a ParentID, then children, then edges — edges
429
+ // last because they reference two child IDs that must both exist.
430
+ const parentID = await this.persistParent(spec, taskTypeID, context);
431
+ const map = await this.persistChildren(spec, parentID, taskTypeID, agentIDsByName.Map, actionIDsByName.Map, promptIDsByName.Map, context);
432
+ await this.persistDependencies(spec, map, context);
433
+ return { parentTaskID: parentID, taskIDMap: map };
434
+ });
109
435
  LogStatus(`[TaskGraphService] Submitted "${spec.workflowName}": parent ${parentTaskID}, ${taskIDMap.size} task(s). ` +
110
436
  `Awaiting dispatcher pickup.`);
437
+ KickTaskGraphDispatchers();
111
438
  return { Success: true, ParentTaskID: parentTaskID, TaskIDMap: taskIDMap };
112
439
  }
113
440
  catch (e) {
@@ -117,34 +444,202 @@ export class TaskGraphService {
117
444
  }
118
445
  }
119
446
  /**
120
- * Cancels a graph and everything in it that has not already settled.
447
+ * Cancels a graph, everything in it that has not already settled, and everything it started.
121
448
  *
122
449
  * Cancels children first: a parent marked `Cancelled` while children are still `Pending` would
123
450
  * leave the dispatcher free to pick those children up, which is the opposite of what the caller
124
451
  * asked for.
452
+ *
453
+ * **The verdict is the outcome, not the attempt** (R2-9). This returned `true` unconditionally
454
+ * while logging each child that failed to cancel — so one failed save left that child `Pending`,
455
+ * told the caller cancellation had succeeded, and let the dispatcher run the child afterwards.
456
+ * The graph could then settle `Complete` and ANNOUNCE ITS COMPLETION into the conversation of a
457
+ * workflow the user had cancelled. A partial cancel now says so and names what survived; the
458
+ * graph stays active, so retrying is meaningful rather than cosmetic.
125
459
  */
126
460
  async Cancel(parentTaskID, context) {
461
+ return this.cancelWithDepth(parentTaskID, context, 0, new Set());
462
+ }
463
+ /**
464
+ * `Cancel`, carrying the recursion state the public entry point does not expose.
465
+ *
466
+ * **The depth cap was dead code** (R3-10): `Cancel` passed a literal 0, and the recursion
467
+ * re-entered through `this.Cancel`, which restarted at 0 — so the check could never fire and
468
+ * the "bounded by the reinvoke depth cap" promise was false. A hand-edited `AgentRunID` cycle
469
+ * recursed to stack overflow mid-cancel.
470
+ *
471
+ * The visited set is cheap armour on top: the cap bounds how DEEP a legitimate chain goes, and
472
+ * a cycle is not deep, it is circular. Arithmetic alone would eventually stop it; a visited set
473
+ * stops it immediately and covers linkage shapes the arithmetic does not anticipate.
474
+ */
475
+ async cancelWithDepth(parentTaskID, context, depth, visited) {
476
+ if (visited.has(parentTaskID)) {
477
+ return { Success: true, Cancelled: true, UncancelledTaskNames: [] };
478
+ }
479
+ visited.add(parentTaskID);
127
480
  try {
128
481
  const children = await this.loadChildren(parentTaskID, context);
482
+ const uncancelled = [];
483
+ const settledMeanwhile = [];
129
484
  for (const child of children) {
130
485
  // Terminal work is left alone — cancelling a completed task would rewrite history.
131
- if (['Complete', 'Failed', 'Cancelled'].includes(child.Status))
486
+ // The in-memory test is a cheap pre-filter; the one that MATTERS is in the statement
487
+ // (R3-9), because a child can settle between this snapshot and its own write, and
488
+ // the full-row save this replaces overwrote that outcome wholesale.
489
+ if (['Complete', 'Failed', 'Cancelled', 'Skipped'].includes(child.Status))
490
+ continue;
491
+ if (await this.claims.TryCancelTask(context.Provider, child.ID, context.ContextUser))
132
492
  continue;
133
- child.Status = 'Cancelled';
134
- if (!(await child.Save())) {
135
- LogError(`[TaskGraphService] Failed to cancel task ${child.ID}: ${child.LatestResult?.CompleteMessage ?? 'unknown error'}`);
493
+ // Rowcount 0 means it reached a terminal status while we were cancelling its
494
+ // siblings. Its outcome is real and stays; the verdict says so rather than pretending
495
+ // the cancel was total.
496
+ settledMeanwhile.push(child.Name);
497
+ }
498
+ if (settledMeanwhile.length > 0) {
499
+ LogStatus(`[TaskGraphService] ${settledMeanwhile.length} task(s) settled while the cancel ran ` +
500
+ `(${settledMeanwhile.join(', ')}); their outcomes are kept.`);
501
+ }
502
+ // Withdraw the questions too. A human step that was waiting has an open
503
+ // `MJ: AI Agent Requests` row, and cancelling only the Task left that row `Requested`
504
+ // FOREVER: the person keeps seeing "a workflow is waiting on you" in their inbox for a
505
+ // workflow that no longer exists, and answering it settles nothing because the task is
506
+ // already Cancelled. Nothing else ever closes these — the dispatcher only expires rows
507
+ // that carry a deadline, and most do not.
508
+ await this.cancelOpenRequests(children.map((c) => c.ID), context);
509
+ // THE PARENT IS LEFT TO THE DISPATCHER, DELIBERATELY.
510
+ //
511
+ // Writing it terminal here skipped the settle path entirely — no cost rollup, no run
512
+ // settlement, no notification — so the submitting agent run stayed `Paused` forever.
513
+ // Worse, it was NONDETERMINISTIC: if a dispatcher poll happened to land between the
514
+ // child cancels above and the parent write, the graph settled through the normal path
515
+ // and the run WAS failed and messaged. Cancel behaved differently run to run depending
516
+ // on timing.
517
+ //
518
+ // With the children cancelled, `ComputeParentRollup` reaches `Cancelled` on its own and
519
+ // the ordinary settle sequence runs — rollup, run settlement, continuation — exactly as
520
+ // it does for a graph that finished by itself. Less code, one path, and a deterministic
521
+ // outcome.
522
+ //
523
+ // The parent stays non-terminal until then, so the sweep still sees it as active work.
524
+ // WHAT THIS WORKFLOW STARTED IS ALSO CANCELLED (R2-9).
525
+ //
526
+ // A graph's step can be an agent that submits a graph of its own, and those sub-graphs
527
+ // persist as ROOTS — linked back only through the child task's `AgentRunID`. So
528
+ // cancelling a workflow left its descendants running, and on settlement one of them can
529
+ // REINVOKE the cancelled workflow's own agent for a fresh billed turn: the user stopped
530
+ // a workflow and it started itself again.
531
+ //
532
+ // Bounded by the reinvoke depth cap, which is what bounds the chain in the first place.
533
+ const nested = await this.cancelNestedGraphs(children, context, depth, visited);
534
+ uncancelled.push(...nested);
535
+ if (uncancelled.length > 0) {
536
+ return {
537
+ Success: false,
538
+ Cancelled: false,
539
+ UncancelledTaskNames: uncancelled,
540
+ ErrorMessage: `Cancelled what it could, but ${uncancelled.length} task(s) could not be cancelled ` +
541
+ `(${uncancelled.join(', ')}). The workflow is still active — retry the cancel.`,
542
+ };
543
+ }
544
+ return { Success: true, Cancelled: true, UncancelledTaskNames: [] };
545
+ }
546
+ catch (e) {
547
+ const message = e instanceof Error ? e.message : String(e);
548
+ LogError(`[TaskGraphService] Cancel failed for ${parentTaskID}: ${message}`);
549
+ return { Success: false, Cancelled: false, UncancelledTaskNames: [], ErrorMessage: message };
550
+ }
551
+ }
552
+ /**
553
+ * Cancels the graphs that this graph's own steps submitted, one level at a time.
554
+ *
555
+ * The linkage is `child task → AgentRunID → the graphs that run submitted`, which is exactly how
556
+ * the continuation chain finds its way back up; walking it downward is the same relation read the
557
+ * other way. Depth-capped by the same constant that caps reinvocation, so a self-referencing
558
+ * workflow cannot make cancellation recurse further than it could have spawned.
559
+ *
560
+ * @returns names of tasks in descendant graphs that could not be cancelled
561
+ */
562
+ async cancelNestedGraphs(children, context, depth, visited) {
563
+ if (depth >= MAX_REINVOKE_DEPTH) {
564
+ LogError(`[TaskGraphService] Nested cancel stopped at depth ${depth}; a deeper sub-graph chain ` +
565
+ `than the reinvoke cap allows may still be running.`);
566
+ return [];
567
+ }
568
+ const runIDs = [...new Set(children.map((c) => c.AgentRunID).filter((id) => !!id))];
569
+ if (runIDs.length === 0)
570
+ return [];
571
+ const rv = RunView.FromMetadataProvider(context.Provider);
572
+ const inList = runIDs.map((id) => `'${id}'`).join(',');
573
+ // TYPE-SCOPED, like every other graph-mutating walk over this table (R3-10). `MJ: Tasks` is
574
+ // general-purpose, and without the predicate any non-workflow root hierarchy that happens to
575
+ // carry a cancelled run's ID gets `Cancelled` written over its children and its requests
576
+ // withdrawn — the user-writable-table threat the claim store's guards exist to defend
577
+ // against. The comment here already claimed this scoping; the query did not have it.
578
+ const typeID = await this.findTaskTypeID(context);
579
+ if (!typeID)
580
+ return [];
581
+ const subGraphs = await rv.RunView({
582
+ EntityName: 'MJ: Tasks',
583
+ ExtraFilter: `TypeID='${typeID}' AND ParentID IS NULL AND AgentRunID IN (${inList})`,
584
+ Fields: ['ID'],
585
+ ResultType: 'simple',
586
+ BypassCache: true,
587
+ }, context.ContextUser);
588
+ if (!subGraphs.Success) {
589
+ LogError(`[TaskGraphService] Could not look for sub-graphs while cancelling: ${subGraphs.ErrorMessage}`);
590
+ return [];
591
+ }
592
+ const failures = [];
593
+ for (const row of subGraphs.Results ?? []) {
594
+ // Through the depth-carrying overload, so the cap actually engages.
595
+ const result = await this.cancelWithDepth(row.ID, context, depth + 1, visited);
596
+ if (!result.Success)
597
+ failures.push(...result.UncancelledTaskNames);
598
+ }
599
+ return failures;
600
+ }
601
+ /**
602
+ * Closes the still-open requests raised for a set of tasks.
603
+ *
604
+ * `Canceled` rather than `Expired`: nobody ran out of time, the ask was withdrawn — and the two
605
+ * mean different things downstream, since the dispatcher treats an expired human step as a
606
+ * FAILURE a give-up edge can route around, which would be a lie about a graph somebody stopped
607
+ * on purpose.
608
+ *
609
+ * Failures here are logged and never propagated: the graph is already cancelled, and refusing to
610
+ * finish that because an inbox row would not close would leave the graph in a worse state than
611
+ * the debris it is trying to avoid.
612
+ */
613
+ async cancelOpenRequests(taskIDs, context) {
614
+ if (taskIDs.length === 0)
615
+ return;
616
+ try {
617
+ const idList = taskIDs.map((id) => `'${id}'`).join(',');
618
+ const open = await RunView.FromMetadataProvider(context.Provider).RunView({
619
+ EntityName: 'MJ: AI Agent Requests',
620
+ ExtraFilter: `Status='Requested' AND OriginatingTaskID IN (${idList})`,
621
+ ResultType: 'entity_object',
622
+ BypassCache: true,
623
+ }, context.ContextUser);
624
+ if (!open.Success) {
625
+ LogError(`[TaskGraphService] Could not read open requests to cancel: ${open.ErrorMessage}`);
626
+ return;
627
+ }
628
+ for (const request of open.Results ?? []) {
629
+ request.Status = 'Canceled';
630
+ request.Comments = 'The workflow that asked this was cancelled.';
631
+ if (!(await request.Save())) {
632
+ LogError(`[TaskGraphService] Could not withdraw request ${request.ID}: ` +
633
+ `${request.LatestResult?.CompleteMessage ?? 'unknown error'}. It will keep showing ` +
634
+ `in someone's inbox for a workflow that no longer exists.`);
136
635
  }
137
636
  }
138
- const parent = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
139
- if (!(await parent.Load(parentTaskID)))
140
- return false;
141
- parent.Status = 'Cancelled';
142
- parent.CompletedAt = new Date();
143
- return await parent.Save();
637
+ if ((open.Results ?? []).length > 0) {
638
+ LogStatus(`[TaskGraphService] Withdrew ${open.Results.length} open request(s) for the cancelled graph.`);
639
+ }
144
640
  }
145
641
  catch (e) {
146
- LogError(`[TaskGraphService] Cancel failed for ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
147
- return false;
642
+ LogError(`[TaskGraphService] Could not withdraw open requests: ${e instanceof Error ? e.message : String(e)}`);
148
643
  }
149
644
  }
150
645
  /**
@@ -154,7 +649,7 @@ export class TaskGraphService {
154
649
  * leaving them blocked would make the retry pointless, as the graph still could not progress
155
650
  * past this node.
156
651
  */
157
- async Retry(taskID, context) {
652
+ async Retry(taskID, context, inputPayload) {
158
653
  try {
159
654
  const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
160
655
  if (!(await task.Load(taskID)))
@@ -163,6 +658,27 @@ export class TaskGraphService {
163
658
  LogError(`[TaskGraphService] Cannot retry task ${taskID}: status is ${task.Status}, expected Failed.`);
164
659
  return false;
165
660
  }
661
+ // An edited input rides the retry: the operator saw WHY it failed and is re-running the
662
+ // step with a corrected brief. Applies to this run only — the graph's spec is long gone.
663
+ //
664
+ // Written through the GUARDED statement rather than onto the in-memory row, so the edit
665
+ // cannot ride along on the full-row save below. The window here is narrower than
666
+ // `UpdateTaskInput`'s (the pre-state is `Failed`, so a concurrent claim is not the
667
+ // hazard — a concurrent human retry is), but the shape is the same and it costs one
668
+ // statement to not have it. The rest of this method's full-row save predates this PR
669
+ // and is Round 3's to purge; the new write does not add to it.
670
+ if (inputPayload !== undefined) {
671
+ const typeID = await this.ensureTaskType(context);
672
+ const json = typeof inputPayload === 'string' ? inputPayload : JSON.stringify(inputPayload);
673
+ const wrote = await this.debugWrites.TryUpdateInputPayload(context.Provider, taskID, json, 'Failed', typeID, context.ContextUser);
674
+ if (!wrote) {
675
+ LogError(`[TaskGraphService] Could not apply the edited input to task ${taskID}; retry refused rather than re-running the old brief.`);
676
+ return false;
677
+ }
678
+ // Keep the in-memory row in step with what was just written, so the save below does
679
+ // not put the old input back.
680
+ task.InputPayload = json;
681
+ }
166
682
  task.Status = 'Pending';
167
683
  task.ErrorMessage = null;
168
684
  task.StartedAt = null;
@@ -188,12 +704,275 @@ export class TaskGraphService {
188
704
  return false;
189
705
  }
190
706
  }
707
+ /** Result shape shared by the control verbs: what happened, and the state that now holds. */
708
+ controlResult(success, debug, errorMessage) {
709
+ return { Success: success, Debug: debug, ErrorMessage: errorMessage };
710
+ }
711
+ /**
712
+ * Loads a graph parent and proves it IS a workflow graph before any debug write.
713
+ *
714
+ * Read with `BypassCache` for the same reason the dispatcher reads rows that way: the debug bag
715
+ * is written by direct `JSON_MODIFY` statements that fire no cache invalidation, so a cached
716
+ * read here could merge new state over a stale copy and silently resurrect a cleared flag.
717
+ */
718
+ async loadWorkflowParent(parentTaskID, context) {
719
+ const typeID = await this.ensureTaskType(context);
720
+ const rows = await RunView.FromMetadataProvider(context.Provider).RunView({
721
+ EntityName: 'MJ: Tasks',
722
+ ExtraFilter: `ID='${parentTaskID.replace(/'/g, "''")}'`,
723
+ Fields: ['ID', 'TypeID', 'InputPayload', 'Status'],
724
+ ResultType: 'simple',
725
+ BypassCache: true,
726
+ }, context.ContextUser);
727
+ const row = rows.Success ? rows.Results?.[0] : undefined;
728
+ if (!row)
729
+ return null;
730
+ if (!UUIDsEqual(row.TypeID, typeID))
731
+ return null;
732
+ return { typeID, inputPayload: row.InputPayload, status: row.Status };
733
+ }
734
+ /**
735
+ * Writes the debug-bag fields a verb OWNS, and reports the state that results.
736
+ *
737
+ * Field-scoped on purpose — see {@link TaskClaimStore.TryWriteDebugFields}. A verb declares the
738
+ * paths it is responsible for; everything else in the bag is left exactly as the database has
739
+ * it, so a concurrent step-consume, breakpoint edit, or override cannot be undone by a verb that
740
+ * was not talking about them.
741
+ *
742
+ * The returned state is this instance's best view (read + the fields just written) and is
743
+ * advisory — the same posture the console takes toward frames.
744
+ */
745
+ async writeDebugFields(parentTaskID, context, build) {
746
+ const parent = await this.loadWorkflowParent(parentTaskID, context);
747
+ if (!parent)
748
+ return this.controlResult(false, undefined, 'Not a workflow graph this control plane can act on.');
749
+ const { Fields, Next } = build(ParseTaskGraphDebugState(parent.inputPayload));
750
+ const ok = await this.debugWrites.TryWriteDebugFields(context.Provider, parentTaskID, Fields, parent.typeID, context.ContextUser);
751
+ if (!ok)
752
+ return this.controlResult(false, undefined, 'The debug state could not be written; see the server log.');
753
+ return this.controlResult(true, Next);
754
+ }
755
+ /**
756
+ * Pauses a graph: nothing new is claimed until it is resumed. In-flight steps finish naturally
757
+ * and their completions land — a pause gates claiming and never touches a live claim, which is
758
+ * why there is no "what happens to the claim" question to answer.
759
+ */
760
+ async PauseGraph(parentTaskID, context, pausedByUserID) {
761
+ const pausedBy = pausedByUserID ?? context.ContextUser?.ID ?? null;
762
+ return this.writeDebugFields(parentTaskID, context, (current) => ({
763
+ // Pause owns the pause fields AND the step allowance: an allowance armed a moment ago is
764
+ // for a run the operator has now stopped, so clearing it is the verb's meaning rather
765
+ // than a side effect. Breakpoints and overrides are untouched — they outlive a pause.
766
+ Fields: [
767
+ TaskClaimStore.DebugField('$.debug.paused', { Kind: 'bool', Value: true }),
768
+ TaskClaimStore.DebugField('$.debug.pausedReason', { Kind: 'string', Value: 'user' }),
769
+ TaskClaimStore.DebugField('$.debug.pausedBy', pausedBy ? { Kind: 'string', Value: pausedBy } : { Kind: 'null' }),
770
+ TaskClaimStore.DebugField('$.debug.pausedAtTaskID', { Kind: 'null' }),
771
+ TaskClaimStore.DebugField('$.debug.step', { Kind: 'null' }),
772
+ TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' }),
773
+ ],
774
+ Next: { ...current, paused: true, pausedBy, pausedReason: 'user', pausedAtTaskID: null, step: undefined, skipBreakpointTaskID: undefined },
775
+ }));
776
+ }
777
+ /** Resumes a paused graph. Breakpoints and edge overrides survive — only the pause clears. */
778
+ async ResumeGraph(parentTaskID, context) {
779
+ return this.writeDebugFields(parentTaskID, context, (current) => {
780
+ const next = { ...current };
781
+ // Continue from a breakpoint must run the stopped task. Stamping it here so the next
782
+ // poll does not re-hit the same still-eligible breakpoint without claiming.
783
+ const skip = current.pausedReason === 'breakpoint' ? current.pausedAtTaskID : null;
784
+ delete next.paused;
785
+ delete next.pausedBy;
786
+ delete next.pausedReason;
787
+ delete next.pausedAtTaskID;
788
+ delete next.step;
789
+ if (skip)
790
+ next.skipBreakpointTaskID = skip;
791
+ else
792
+ delete next.skipBreakpointTaskID;
793
+ return {
794
+ Fields: [
795
+ TaskClaimStore.DebugField('$.debug.paused', { Kind: 'null' }),
796
+ TaskClaimStore.DebugField('$.debug.pausedReason', { Kind: 'null' }),
797
+ TaskClaimStore.DebugField('$.debug.pausedBy', { Kind: 'null' }),
798
+ TaskClaimStore.DebugField('$.debug.pausedAtTaskID', { Kind: 'null' }),
799
+ TaskClaimStore.DebugField('$.debug.step', { Kind: 'null' }),
800
+ skip
801
+ ? TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'string', Value: skip })
802
+ : TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' }),
803
+ ],
804
+ Next: next,
805
+ };
806
+ });
807
+ }
808
+ /**
809
+ * Arms a one-shot step allowance on a paused graph: `'one'` releases the next eligible task,
810
+ * `'wave'` releases the current frontier, a task ID releases exactly that task. The dispatcher
811
+ * consumes the allowance CAS-style, so two instances stepping the same graph release work once.
812
+ */
813
+ async StepGraph(parentTaskID, target, context) {
814
+ const parent = await this.loadWorkflowParent(parentTaskID, context);
815
+ if (!parent)
816
+ return this.controlResult(false, undefined, 'Not a workflow graph this control plane can act on.');
817
+ const current = ParseTaskGraphDebugState(parent.inputPayload);
818
+ if (!current.paused) {
819
+ return this.controlResult(false, current, 'Step only applies to a paused workflow — pause it first.');
820
+ }
821
+ return this.writeDebugFields(parentTaskID, context, (state) => ({
822
+ Fields: [TaskClaimStore.DebugField('$.debug.step', { Kind: 'string', Value: target })],
823
+ Next: { ...state, step: target },
824
+ }));
825
+ }
826
+ /**
827
+ * Replaces the graph's breakpoint set. Every ID must name a child of this graph — a breakpoint
828
+ * on a task in some other graph would gate nothing and silently lie to the person who set it.
829
+ */
830
+ async SetBreakpoints(parentTaskID, taskIDs, context) {
831
+ const children = await this.loadChildren(parentTaskID, context);
832
+ const childIDs = new Set(children.map((c) => c.ID.toLowerCase()));
833
+ const foreign = taskIDs.filter((id) => !childIDs.has(id.toLowerCase()));
834
+ if (foreign.length > 0) {
835
+ return this.controlResult(false, undefined, `Not steps of this workflow: ${foreign.join(', ')}`);
836
+ }
837
+ return this.writeDebugFields(parentTaskID, context, (current) => {
838
+ const next = { ...current };
839
+ if (taskIDs.length > 0)
840
+ next.breakpoints = [...taskIDs];
841
+ else
842
+ delete next.breakpoints;
843
+ return {
844
+ Fields: [
845
+ TaskClaimStore.DebugField('$.debug.breakpoints', taskIDs.length > 0 ? { Kind: 'json', Value: JSON.stringify(taskIDs) } : { Kind: 'null' }),
846
+ ],
847
+ Next: next,
848
+ };
849
+ });
850
+ }
851
+ /**
852
+ * Overrides one edge's condition verdict — the operator's answer for a path the engine cannot
853
+ * decide (a held graph) or decided wrongly (a broken guard). `'false'` reads as "branch not
854
+ * taken" and cascades skips; `'true'` opens the gate; `null` removes the override.
855
+ */
856
+ async SetEdgeOverride(parentTaskID, edgeID, verdict, context) {
857
+ // Prove the edge belongs to this graph before writing anything about it.
858
+ const edge = await context.Provider.GetEntityObject('MJ: Task Dependencies', context.ContextUser);
859
+ if (!(await edge.Load(edgeID)))
860
+ return this.controlResult(false, undefined, 'No such path.');
861
+ const target = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
862
+ if (!(await target.Load(edge.TaskID)) || !UUIDsEqual(target.ParentID ?? '', parentTaskID)) {
863
+ return this.controlResult(false, undefined, 'That path does not belong to this workflow.');
864
+ }
865
+ // AN UNCONDITIONAL PATH CANNOT BE OVERRIDDEN — refused here so the two dialects agree.
866
+ //
867
+ // The engine reads overrides at different depths: an ordinary edge only consults one when
868
+ // it HAS a condition, while the exclusive evaluator consults it before its no-condition
869
+ // early return. Left open, the same override would force an unconditional exclusive edge to
870
+ // lose while doing nothing at all to an unconditional ordinary one — the operator's answer
871
+ // meaning two different things depending on a property of the graph they cannot see.
872
+ // Refusing is the conservative reading, and it costs nothing: "don't take this branch" is
873
+ // already expressible, precisely, as SkipTask on the step itself.
874
+ if (verdict !== null && !edge.Condition?.trim()) {
875
+ return this.controlResult(false, undefined, 'That path has no condition to answer — it is always taken. To stop the branch, skip its step instead.');
876
+ }
877
+ return this.writeDebugFields(parentTaskID, context, (current) => {
878
+ const overrides = { ...(current.edgeOverrides ?? {}) };
879
+ if (verdict === null)
880
+ delete overrides[edgeID];
881
+ else
882
+ overrides[edgeID] = verdict;
883
+ const next = { ...current };
884
+ if (Object.keys(overrides).length > 0)
885
+ next.edgeOverrides = overrides;
886
+ else
887
+ delete next.edgeOverrides;
888
+ return {
889
+ // Scoped to THIS edge's key, not the whole map: two operators answering two
890
+ // different held paths at once must not overwrite each other's answer.
891
+ Fields: [
892
+ TaskClaimStore.DebugField(`$.debug.edgeOverrides."${edgeID}"`, verdict === null ? { Kind: 'null' } : { Kind: 'string', Value: verdict }),
893
+ ],
894
+ Next: next,
895
+ };
896
+ });
897
+ }
898
+ /**
899
+ * Declares a Pending step not-taken. Downstream dependents proceed — `Skipped` satisfies a
900
+ * prerequisite — and any open human request for the step is withdrawn so nobody keeps seeing an
901
+ * ask for work the operator decided against.
902
+ */
903
+ async SkipTask(taskID, context) {
904
+ const typeID = await this.ensureTaskType(context);
905
+ const ok = await this.debugWrites.TrySkipPending(context.Provider, taskID, typeID, context.ContextUser);
906
+ if (!ok) {
907
+ return { Success: false, ErrorMessage: 'Only a step that has not started can be skipped.' };
908
+ }
909
+ await this.cancelOpenRequests([taskID], context);
910
+ return { Success: true };
911
+ }
912
+ /**
913
+ * Marks a step Complete with an operator-supplied output.
914
+ *
915
+ * Human steps are refused here on purpose: they already have a first-class completion path
916
+ * (`TaskGraph.CompleteTask`) with the assignee/elevation check, and this verb must not become
917
+ * the door that bypasses it.
918
+ */
919
+ async ForceCompleteTask(taskID, outputPayload, context) {
920
+ const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
921
+ if (!(await task.Load(taskID)))
922
+ return { Success: false, ErrorMessage: 'No such step.' };
923
+ if (task.UserID) {
924
+ return { Success: false, ErrorMessage: 'A human step completes through its assignee — use CompleteTask.' };
925
+ }
926
+ const json = outputPayload == null
927
+ ? null
928
+ : typeof outputPayload === 'string' ? outputPayload : JSON.stringify(outputPayload);
929
+ const typeID = await this.ensureTaskType(context);
930
+ const ok = await this.debugWrites.TryForceComplete(context.Provider, taskID, json, typeID, context.ContextUser);
931
+ if (!ok) {
932
+ return {
933
+ Success: false,
934
+ ErrorMessage: 'The step is running with a live claim, or already finished. Cancel it or wait for the claim to lapse.',
935
+ };
936
+ }
937
+ return { Success: true };
938
+ }
939
+ /**
940
+ * Replaces a Pending step's input — the "edit the brief before stepping" move at a breakpoint.
941
+ * Applies to this run only; the step must not have started.
942
+ *
943
+ * **A guarded statement, not load-check-save.** The in-memory `Status === 'Pending'` check plus
944
+ * `task.Save()` is an unconditional full-row UPDATE: a task claimed in the window between the
945
+ * load and the save has its claim columns reverted to the pre-claim snapshot *while its body
946
+ * runs*, and a second instance then claims it again — the step executes twice. See
947
+ * {@link TaskClaimStore.TryUpdateInputPayload}, which makes the check and the write one atomic
948
+ * operation whose rowcount is the answer.
949
+ */
950
+ async UpdateTaskInput(taskID, inputPayload, context) {
951
+ try {
952
+ const typeID = await this.ensureTaskType(context);
953
+ const json = typeof inputPayload === 'string' ? inputPayload : JSON.stringify(inputPayload);
954
+ const ok = await this.debugWrites.TryUpdateInputPayload(context.Provider, taskID, json, 'Pending', typeID, context.ContextUser);
955
+ if (!ok) {
956
+ return {
957
+ Success: false,
958
+ ErrorMessage: 'Only a workflow step that has not started can have its input edited.',
959
+ };
960
+ }
961
+ return { Success: true };
962
+ }
963
+ catch (e) {
964
+ return { Success: false, ErrorMessage: e instanceof Error ? e.message : String(e) };
965
+ }
966
+ }
191
967
  // ────────────────────────────────────────────────────────────────────────
192
968
  // internals
193
969
  // ────────────────────────────────────────────────────────────────────────
194
970
  /** Maps every referenced agent name to its ID, or reports all unresolvable names at once. */
195
971
  async resolveAgents(spec, context) {
196
- const names = [...new Set(spec.tasks.filter((t) => !!t.agentName).map((t) => t.agentName))];
972
+ // A loop's repeated sub-agent counts as a referenced agent. Collecting only the Agent nodes
973
+ // left a loop body pointing at a name nothing had resolved, so the row got no AgentID and
974
+ // the loop had nothing to run.
975
+ const names = [...new Set(spec.tasks.flatMap(agentNamesIn))];
197
976
  if (names.length === 0)
198
977
  return { Success: true, Map: new Map() };
199
978
  const quoted = names.map((n) => `'${n.replace(/'/g, "''")}'`).join(',');
@@ -209,20 +988,111 @@ export class TaskGraphService {
209
988
  }
210
989
  return { Success: true, Map: found };
211
990
  }
212
- /** Finds or creates the task type used for orchestrated graphs. */
991
+ /**
992
+ * Maps every referenced action name to its ID, or reports all unresolvable names at once.
993
+ *
994
+ * Deliberately a mirror of {@link resolveAgents} rather than a generalization of it: the two
995
+ * read different entities and produce different error prose, and the shared shape is three
996
+ * lines. Collapsing them would trade a readable failure message for a parameterized lookup.
997
+ */
998
+ /**
999
+ * Maps every referenced prompt name to its ID.
1000
+ *
1001
+ * A prompt is addressed by name in the spec and stored as a foreign key on the row, exactly like
1002
+ * agents and actions — a name in JSON cannot be joined, checked, or survive a rename.
1003
+ */
1004
+ async resolvePrompts(spec, context) {
1005
+ const names = [...new Set(spec.tasks.flatMap(promptNamesIn))];
1006
+ if (names.length === 0)
1007
+ return { Success: true, Map: new Map() };
1008
+ const quoted = names.map((n) => `'${n.replace(/'/g, "''")}'`).join(',');
1009
+ const result = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: AI Prompts', ExtraFilter: `Name IN (${quoted})`, Fields: ['ID', 'Name'], ResultType: 'simple' }, context.ContextUser);
1010
+ const found = new Map((result.Results ?? []).map((r) => [r.Name, r.ID]));
1011
+ const missing = names.filter((n) => !found.has(n));
1012
+ if (missing.length > 0) {
1013
+ return {
1014
+ Success: false,
1015
+ ErrorMessage: `Task graph "${spec.workflowName}" references ${missing.length} unknown prompt(s): ${missing.join(', ')}. ` +
1016
+ `Submitting would execute the graph with holes where those steps should be.`,
1017
+ };
1018
+ }
1019
+ return { Success: true, Map: found };
1020
+ }
1021
+ async resolveActions(spec, context) {
1022
+ // Includes a loop's repeated action — see the note in resolveAgents.
1023
+ const names = [...new Set(spec.tasks.flatMap(actionNamesIn))];
1024
+ if (names.length === 0)
1025
+ return { Success: true, Map: new Map() };
1026
+ const quoted = names.map((n) => `'${n.replace(/'/g, "''")}'`).join(',');
1027
+ const result = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: Actions', ExtraFilter: `Name IN (${quoted})`, Fields: ['ID', 'Name'], ResultType: 'simple' }, context.ContextUser);
1028
+ const found = new Map((result.Results ?? []).map((r) => [r.Name, r.ID]));
1029
+ const missing = names.filter((n) => !found.has(n));
1030
+ if (missing.length > 0) {
1031
+ return {
1032
+ Success: false,
1033
+ ErrorMessage: `Task graph "${spec.workflowName}" references ${missing.length} unknown action(s): ${missing.join(', ')}. ` +
1034
+ `Submitting would execute the graph with holes where those tasks should be.`,
1035
+ };
1036
+ }
1037
+ return { Success: true, Map: found };
1038
+ }
1039
+ /**
1040
+ * Finds or creates the task type used for orchestrated graphs — exactly one of it, ever.
1041
+ *
1042
+ * **This resolves the engine's discriminator, not a label** (R2-7). Round 1 scoped every sweep
1043
+ * arm and both payload-writing guards to this type, so a second row sharing the name lets
1044
+ * different processes bind different IDs — and a graph stamped with the other one is invisible
1045
+ * to the sweep, never claimed, never settled, its submitting run `Paused` forever, with no
1046
+ * error anywhere.
1047
+ *
1048
+ * Race-safe by INSERT-then-reselect rather than by checking harder. Two concurrent first-ever
1049
+ * submissions both read "not there" and both insert; the unique index added in this round makes
1050
+ * the loser's insert fail, and the loser then re-reads and finds the winner's row. Checking
1051
+ * first is what created the window, so the fix cannot be a better check.
1052
+ */
213
1053
  async ensureTaskType(context) {
214
- const existing = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: Task Types', ExtraFilter: `Name='${TASK_TYPE_NAME}'`, Fields: ['ID'], ResultType: 'simple', MaxRows: 1 }, context.ContextUser);
215
- const found = existing.Results?.[0]?.ID;
1054
+ const found = await this.findTaskTypeID(context);
216
1055
  if (found)
217
1056
  return found;
218
1057
  const tt = await context.Provider.GetEntityObject('MJ: Task Types', context.ContextUser);
219
1058
  tt.NewRecord();
220
1059
  tt.Name = TASK_TYPE_NAME;
221
1060
  tt.Description = 'Tasks created by agent-orchestrated workflows.';
222
- if (!(await tt.Save())) {
223
- throw new Error(`Could not create task type: ${tt.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1061
+ if (await tt.Save())
1062
+ return tt.ID;
1063
+ // The insert lost. Almost certainly to the unique index and another process that got there
1064
+ // first — so re-read before treating it as a failure. Any other cause falls through to the
1065
+ // throw below with its own message intact.
1066
+ const winner = await this.findTaskTypeID(context);
1067
+ if (winner)
1068
+ return winner;
1069
+ throw new Error(`Could not create task type: ${tt.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1070
+ }
1071
+ /**
1072
+ * The `AI Workflow` task type's ID, resolved deterministically.
1073
+ *
1074
+ * `ORDER BY` is not decoration: two rows sharing the name come back in whatever order the engine
1075
+ * chooses, so an unordered `MaxRows: 1` lets two processes bind different IDs from the same
1076
+ * data. The index this round adds makes duplicates impossible going forward; the ordering makes
1077
+ * the resolution deterministic on a database that still has some, and the warning makes the
1078
+ * situation visible rather than merely survivable.
1079
+ */
1080
+ async findTaskTypeID(context) {
1081
+ const existing = await RunView.FromMetadataProvider(context.Provider).RunView({
1082
+ EntityName: 'MJ: Task Types',
1083
+ ExtraFilter: `Name='${TASK_TYPE_NAME}'`,
1084
+ Fields: ['ID'],
1085
+ OrderBy: '__mj_CreatedAt ASC, ID ASC',
1086
+ ResultType: 'simple',
1087
+ MaxRows: 2,
1088
+ }, context.ContextUser);
1089
+ const rows = existing.Results ?? [];
1090
+ if (rows.length > 1) {
1091
+ LogError(`[TaskGraphService] More than one '${TASK_TYPE_NAME}' task type exists. Binding the oldest ` +
1092
+ `(${rows[0].ID}), but graphs stamped with the other are invisible to the dispatcher's ` +
1093
+ `sweep and will never settle. Merge them.`);
224
1094
  }
225
- return tt.ID;
1095
+ return rows[0]?.ID ?? null;
226
1096
  }
227
1097
  /** Writes the parent task that represents the graph as a whole. */
228
1098
  async persistParent(spec, taskTypeID, context) {
@@ -235,24 +1105,52 @@ export class TaskGraphService {
235
1105
  parent.ConversationDetailID = context.ConversationDetailID ?? null;
236
1106
  parent.Status = 'In Progress';
237
1107
  parent.PercentComplete = 0;
1108
+ // The run that submitted this graph, in the COLUMN and not only in the metadata JSON below.
1109
+ // A json field cannot be joined, so provenance that lives only there is unavailable to any
1110
+ // query — which is how a human step ended up with no agent to ask on behalf of, and how a
1111
+ // dispatched run had no parent to roll its cost up to.
1112
+ parent.AgentRunID = context.AgentRunID ?? null;
238
1113
  // The parent row carries what happens AFTER the graph settles. It lives here rather than in
239
1114
  // dispatcher memory because the dispatcher that finishes a graph is frequently not the
240
1115
  // process that accepted it — a restart, a second instance, or simply a long-running graph
241
1116
  // all break that assumption. Anything the completion path needs has to be durable too.
242
- parent.InputPayload = JSON.stringify({
1117
+ // Done before the write, and reported: a value that silently left the envelope becomes a
1118
+ // condition reading absent-data and taking the other branch, with nothing saying why.
1119
+ const sanitized = SanitizeInvocationEnvelope(context.Invocation);
1120
+ if (sanitized.DroppedPaths.length > 0) {
1121
+ LogStatus(`[TaskGraphService] Invocation envelope for '${spec.workflowName}' dropped ` +
1122
+ `${sanitized.DroppedPaths.length} non-persistable value(s): ` +
1123
+ `${sanitized.DroppedPaths.join(', ')}. Conditions referencing them will read as ` +
1124
+ `absent data. Pass plain JSON values for anything a condition needs.`);
1125
+ }
1126
+ // Only written when the caller supplied one, so a graph with no invocation envelope
1127
+ // carries no key at all rather than a misleading empty object. SANITIZED first: the
1128
+ // agent's `context` is documented as possibly a class instance holding connections and
1129
+ // credentials, and carrying it verbatim threw `Converting circular structure to JSON`
1130
+ // on any context with a socket in it — killing the run at submit time — while a context
1131
+ // that happened to serialize would have written its credentials to this row.
1132
+ parent.InputPayload = JSON.stringify(BuildTaskGraphParentInputPayload({
243
1133
  continuation: spec.continuation ?? 'message',
244
1134
  reinvokeDepth: context.ReinvokeDepth ?? 0,
1135
+ failureSemantics: spec.failureSemantics ?? 'block',
245
1136
  submittedByAgentRunID: context.AgentRunID ?? null,
246
1137
  submittedByUserID: context.ContextUser?.ID ?? null,
247
- });
1138
+ invocation: sanitized.Envelope
1139
+ ? { data: sanitized.Envelope.Data, context: sanitized.Envelope.Context }
1140
+ : null,
1141
+ startPaused: context.Debug?.paused === true,
1142
+ }));
248
1143
  if (!(await parent.Save())) {
249
1144
  throw new Error(`Could not create parent task: ${parent.LatestResult?.CompleteMessage ?? 'unknown error'}`);
250
1145
  }
251
1146
  return parent.ID;
252
1147
  }
253
1148
  /** Writes each child task, returning the tempId -> real ID mapping edges will need. */
254
- async persistChildren(spec, parentTaskID, taskTypeID, agentIDsByName, context) {
1149
+ async persistChildren(spec, parentTaskID, taskTypeID, agentIDsByName, actionIDsByName, promptIDsByName, context) {
255
1150
  const map = new Map();
1151
+ // Resolved ONCE over the whole graph: a rank is a node's position in the topology, so it
1152
+ // cannot be computed per node without seeing all of them.
1153
+ const ranks = RankGraphNodes(spec.tasks.map((t) => t.tempId), spec.tasks.flatMap((t) => (t.dependsOn ?? []).map((d) => ({ From: NormalizeDependency(d).tempId, To: t.tempId }))));
256
1154
  for (const node of spec.tasks) {
257
1155
  const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
258
1156
  task.NewRecord();
@@ -264,14 +1162,69 @@ export class TaskGraphService {
264
1162
  task.ConversationDetailID = context.ConversationDetailID ?? null;
265
1163
  task.Status = 'Pending';
266
1164
  task.PercentComplete = 0;
267
- if (node.agentName) {
268
- task.AgentID = agentIDsByName.get(node.agentName);
269
- }
270
- else {
271
- // Human task. Assigned to the submitting user only — cross-user assignment stays
272
- // rejected until the authorization model in #3524 lands.
273
- task.UserID = context.ContextUser.ID;
1165
+ // The discriminator the dispatcher routes on. Written first because everything below —
1166
+ // and everything the dispatcher later does with this row — reads it.
1167
+ task.StepType = node.kind;
1168
+ // Switching on `kind` rather than on which config field happens to be populated. The
1169
+ // old shape — "agentName? then agent; actionName? then action; ELSE human" — turned
1170
+ // every kind it did not know about into a Human task assigned to the submitter, so a
1171
+ // ForEach step became an approval request that nobody had asked for and nothing would
1172
+ // ever complete. A graph that stalls forever while LOOKING like it is waiting on a
1173
+ // person is the worst available failure. `findUnrunnableKinds` refuses unsupported
1174
+ // kinds before anything is written; this switch is what keeps the two in step.
1175
+ switch (node.kind) {
1176
+ case 'Agent':
1177
+ task.AgentID = agentIDsByName.get(ConfigOf(node, 'Agent').agentName);
1178
+ break;
1179
+ case 'Action':
1180
+ task.ActionID = actionIDsByName.get(ConfigOf(node, 'Action').actionName);
1181
+ break;
1182
+ case 'Human':
1183
+ // Assigned to the submitting user only — cross-user assignment stays rejected
1184
+ // until the authorization model in #3524 lands, and `findCrossUserAssignments`
1185
+ // has already refused any graph that asked for someone else.
1186
+ task.UserID = context.ContextUser.ID;
1187
+ break;
1188
+ case 'Prompt':
1189
+ task.PromptID = promptIDsByName.get(ConfigOf(node, 'Prompt').promptName);
1190
+ break;
1191
+ case 'ForEach':
1192
+ case 'While': {
1193
+ // A loop's key points at what it REPEATS, not at the loop itself. That keeps the
1194
+ // reference a real foreign key — joinable, constrained, rename-proof — instead
1195
+ // of a name buried in JSON, and CK_Task_Assignment still holds because a loop
1196
+ // body is exactly one thing.
1197
+ const op = LoopOperationOf(node);
1198
+ if (op?.action)
1199
+ task.ActionID = actionIDsByName.get(op.action.name);
1200
+ else if (op?.subAgent)
1201
+ task.AgentID = agentIDsByName.get(op.subAgent.name);
1202
+ else if (op?.prompt)
1203
+ task.PromptID = promptIDsByName.get(op.prompt.name);
1204
+ else {
1205
+ throw new Error(`Task "${node.name}" is a loop with nothing to repeat. Choose an action, ` +
1206
+ `a prompt, or a sub-agent for it to run on each pass.`);
1207
+ }
1208
+ break;
1209
+ }
1210
+ default:
1211
+ // Unreachable: findUnrunnableKinds rejected these before any write. Throwing
1212
+ // rather than falling through means a kind added later fails loudly here
1213
+ // instead of quietly becoming somebody's to-do item.
1214
+ throw new Error(`Task "${node.name}" has kind "${node.kind}", which cannot be persisted for dispatch.`);
274
1215
  }
1216
+ // Everything about the step that has no column of its own. Dropping this is what used
1217
+ // to lose the input/output mappings — and with them the payload values every branch
1218
+ // condition downstream reads.
1219
+ //
1220
+ // The step's rank in the graph's own order rides along. `Task` has no sequence column,
1221
+ // so without it every consumer listing a graph's steps falls back to creation order —
1222
+ // the compiler's walk, which is neither the order they were drawn in nor the order they
1223
+ // run in. A graph that has not started yet has no timestamps to sort by, so its
1224
+ // structure is the only order available, and it is available from the moment it is
1225
+ // compiled.
1226
+ const configuration = { ...buildStepConfiguration(node), sequence: ranks.get(node.tempId) };
1227
+ task.Configuration = JSON.stringify(configuration);
275
1228
  // Input rides in its own column; Description stays human-readable.
276
1229
  task.InputPayload = node.inputPayload ? JSON.stringify(node.inputPayload) : null;
277
1230
  if (!(await task.Save())) {
@@ -281,6 +1234,25 @@ export class TaskGraphService {
281
1234
  }
282
1235
  return map;
283
1236
  }
1237
+ /**
1238
+ * Reports the node kinds this dispatcher cannot yet execute, or `null` when the graph is
1239
+ * entirely runnable.
1240
+ *
1241
+ * **Why this is a submit-time check and not a validation rule.** `ValidateTaskGraphSpec` is a
1242
+ * pure function over the spec: it answers "is this a well-formed graph?", and its answer must be
1243
+ * the same in a browser, a CLI and a server. "Can it run *here*?" is a different question whose
1244
+ * answer changes as runners are added, so it belongs to the runtime that owns the runners.
1245
+ *
1246
+ * **Why refuse rather than persist-and-stall.** `Prompt`, `ForEach`, `While` and `External`
1247
+ * are legitimate parts of the spec, but the `Task` row has nowhere to carry a node's `kind` or
1248
+ * its typed `configuration` — so a persisted one could not be dispatched even if a runner
1249
+ * existed. Refusing at the door tells the author which step is the problem while the graph is
1250
+ * still theirs to edit. The alternative is a graph that submits successfully and then never
1251
+ * finishes, which costs an operator an afternoon to diagnose.
1252
+ */
1253
+ findUnrunnableKinds(spec) {
1254
+ return FindUnrunnableKinds(spec);
1255
+ }
284
1256
  /** Writes the dependency edges, translating tempIds to persisted IDs. */
285
1257
  async persistDependencies(spec, taskIDMap, context) {
286
1258
  for (const node of spec.tasks) {
@@ -300,6 +1272,15 @@ export class TaskGraphService {
300
1272
  // NULL for an unconditional edge, matching AIAgentStepPath — so a graph authored in
301
1273
  // the flow editor and one emitted by an agent store the same thing.
302
1274
  dep.Condition = edge.condition ?? null;
1275
+ // The exclusive-choice fields. Dropping these was silent and severe: without an
1276
+ // ExclusiveGroup the dispatcher sees a plain fan-out and runs EVERY branch, so a
1277
+ // workflow that should pick one route would take all of them — doing work its author
1278
+ // never intended and, in the Demo workflow's case, calling two different APIs where
1279
+ // the flow calls one. Priority and Sequence are what decide which branch wins, so a
1280
+ // group without them resolves arbitrarily.
1281
+ dep.ExclusiveGroup = edge.exclusiveGroup ?? null;
1282
+ dep.Priority = edge.priority ?? 0;
1283
+ dep.Sequence = edge.sequence ?? 0;
303
1284
  if (!(await dep.Save())) {
304
1285
  throw new Error(`Could not create dependency ${node.tempId} -> ${edge.tempId}: ${dep.LatestResult?.CompleteMessage ?? 'unknown error'}`);
305
1286
  }
@@ -307,7 +1288,22 @@ export class TaskGraphService {
307
1288
  }
308
1289
  }
309
1290
  async loadChildren(parentTaskID, context) {
310
- const result = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object' }, context.ContextUser);
1291
+ // `BypassCache` for the reason the dispatcher documents at every one of its reads (C4): task
1292
+ // status is written by the claim protocol's direct SQL, which fires no cache invalidation.
1293
+ // A cached read here returns PRE-EXECUTION state, and both callers act on status — Cancel's
1294
+ // "leave terminal work alone" guard would pass for a task that has since completed, and
1295
+ // write `Cancelled` over a `Complete` row. That is precisely the history-rewriting the guard
1296
+ // exists to prevent, performed by the guard itself.
1297
+ // Type-scoped for the same reason the sub-graph walk is (R3-10): these rows are about to be
1298
+ // written, and `MJ: Tasks` holds conversation tasks and users' own to-dos too.
1299
+ const typeID = await this.findTaskTypeID(context);
1300
+ const ofType = typeID ? `TypeID='${typeID}' AND ` : '';
1301
+ const result = await RunView.FromMetadataProvider(context.Provider).RunView({
1302
+ EntityName: 'MJ: Tasks',
1303
+ ExtraFilter: `${ofType}ParentID='${parentTaskID}'`,
1304
+ ResultType: 'entity_object',
1305
+ BypassCache: true,
1306
+ }, context.ContextUser);
311
1307
  return (result.Success ? result.Results : []) ?? [];
312
1308
  }
313
1309
  }