@memberjunction/task-graph 6.1.0-edge.1 → 6.1.0-edge.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +180 -4
- package/README.md +214 -0
- package/dist/TaskClaimStore.d.ts +387 -4
- package/dist/TaskClaimStore.d.ts.map +1 -1
- package/dist/TaskClaimStore.js +605 -20
- package/dist/TaskClaimStore.js.map +1 -1
- package/dist/TaskGraphDispatcher.d.ts +668 -5
- package/dist/TaskGraphDispatcher.d.ts.map +1 -1
- package/dist/TaskGraphDispatcher.js +2942 -127
- package/dist/TaskGraphDispatcher.js.map +1 -1
- package/dist/TaskGraphService.d.ts +364 -5
- package/dist/TaskGraphService.d.ts.map +1 -1
- package/dist/TaskGraphService.js +1039 -43
- package/dist/TaskGraphService.js.map +1 -1
- package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -1
- package/dist/TaskGraphSubmitterImpl.js +5 -0
- package/dist/TaskGraphSubmitterImpl.js.map +1 -1
- package/dist/TaskLoopExecutor.d.ts +62 -0
- package/dist/TaskLoopExecutor.d.ts.map +1 -0
- package/dist/TaskLoopExecutor.js +248 -0
- package/dist/TaskLoopExecutor.js.map +1 -0
- package/dist/WorkflowSpecSync.d.ts +28 -2
- package/dist/WorkflowSpecSync.d.ts.map +1 -1
- package/dist/WorkflowSpecSync.js +83 -2
- package/dist/WorkflowSpecSync.js.map +1 -1
- package/dist/condition-gate.d.ts +128 -0
- package/dist/condition-gate.d.ts.map +1 -0
- package/dist/condition-gate.js +257 -0
- package/dist/condition-gate.js.map +1 -0
- package/dist/debug-state.d.ts +102 -0
- package/dist/debug-state.d.ts.map +1 -0
- package/dist/debug-state.js +135 -0
- package/dist/debug-state.js.map +1 -0
- package/dist/index.d.ts +7 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -0
- package/dist/index.js.map +1 -1
- package/dist/operations/TaskGraphDebugOperations.d.ts +99 -0
- package/dist/operations/TaskGraphDebugOperations.d.ts.map +1 -0
- package/dist/operations/TaskGraphDebugOperations.js +310 -0
- package/dist/operations/TaskGraphDebugOperations.js.map +1 -0
- package/dist/operations/TaskGraphOperations.d.ts +20 -2
- package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
- package/dist/operations/TaskGraphOperations.js +51 -8
- package/dist/operations/TaskGraphOperations.js.map +1 -1
- package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
- package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
- package/dist/operations/WorkflowDraftOperation.js +141 -0
- package/dist/operations/WorkflowDraftOperation.js.map +1 -0
- package/dist/settlement-rescue.d.ts +85 -0
- package/dist/settlement-rescue.d.ts.map +1 -0
- package/dist/settlement-rescue.js +119 -0
- package/dist/settlement-rescue.js.map +1 -0
- package/dist/task-graph-kick.d.ts +3 -0
- package/dist/task-graph-kick.d.ts.map +1 -0
- package/dist/task-graph-kick.js +17 -0
- package/dist/task-graph-kick.js.map +1 -0
- package/dist/task-predicates.d.ts +77 -0
- package/dist/task-predicates.d.ts.map +1 -0
- package/dist/task-predicates.js +75 -0
- package/dist/task-predicates.js.map +1 -0
- package/dist/types.d.ts +224 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js.map +1 -1
- package/package.json +12 -8
package/dist/TaskGraphService.js
CHANGED
|
@@ -15,8 +15,27 @@
|
|
|
15
15
|
*
|
|
16
16
|
* @module @memberjunction/task-graph
|
|
17
17
|
*/
|
|
18
|
-
import { LogError, LogStatus, RunView, } from '@memberjunction/core';
|
|
19
|
-
import { FormatValidationErrors, NormalizeDependency, ValidateTaskGraphSpec, } from '@memberjunction/ai-core-plus';
|
|
18
|
+
import { LogError, LogStatus, RunInEntityTransaction, RunView, } from '@memberjunction/core';
|
|
19
|
+
import { FormatValidationErrors, NormalizeDependency, RankGraphNodes, SanitizeInvocationEnvelope, ValidateTaskGraphSpec, ConfigOf, } from '@memberjunction/ai-core-plus';
|
|
20
|
+
import { UUIDsEqual } from '@memberjunction/global';
|
|
21
|
+
import { TaskClaimStore } from './TaskClaimStore.js';
|
|
22
|
+
import { ParseTaskGraphDebugState } from './debug-state.js';
|
|
23
|
+
import { KickTaskGraphDispatchers } from './task-graph-kick.js';
|
|
24
|
+
/**
|
|
25
|
+
* Normalizes a caller-supplied reinvoke depth to a safe cap seed.
|
|
26
|
+
*
|
|
27
|
+
* The runaway-loop cap is a signed comparison over a persisted-verbatim seed, and the remote Submit
|
|
28
|
+
* operation is the one seam where the seed is caller-supplied rather than computed by the engine —
|
|
29
|
+
* a negative value would BUY hops (`-1000` turns a 5-hop cap into 1005). Clamped rather than
|
|
30
|
+
* refused so a stale client sending zero-adjacent noise keeps working, and no accepted value can
|
|
31
|
+
* ever weaken the cap. Exported pure because a boundary rule that cannot be tested directly is a
|
|
32
|
+
* boundary nobody will notice moving.
|
|
33
|
+
*/
|
|
34
|
+
export function ClampReinvokeDepth(value) {
|
|
35
|
+
return typeof value === 'number' && Number.isFinite(value)
|
|
36
|
+
? Math.max(0, Math.floor(value))
|
|
37
|
+
: undefined;
|
|
38
|
+
}
|
|
20
39
|
/**
|
|
21
40
|
* Continuation chains are bounded separately from graph nesting.
|
|
22
41
|
*
|
|
@@ -31,6 +50,7 @@ export const MAX_REINVOKE_DEPTH = 5;
|
|
|
31
50
|
const DEFAULT_PARENT_METADATA = {
|
|
32
51
|
continuation: 'message',
|
|
33
52
|
reinvokeDepth: 0,
|
|
53
|
+
failureSemantics: 'block',
|
|
34
54
|
submittedByAgentRunID: null,
|
|
35
55
|
submittedByUserID: null,
|
|
36
56
|
};
|
|
@@ -61,20 +81,259 @@ export function ParseTaskGraphParentMetadata(raw) {
|
|
|
61
81
|
continuation: parsed.continuation === 'reinvoke' || parsed.continuation === 'none'
|
|
62
82
|
? parsed.continuation
|
|
63
83
|
: 'message',
|
|
84
|
+
failureSemantics: parsed.failureSemantics === 'edges' ? 'edges' : 'block',
|
|
64
85
|
reinvokeDepth: Number.isFinite(parsed.reinvokeDepth) ? Number(parsed.reinvokeDepth) : 0,
|
|
86
|
+
// Guarded like the others: this is read to explain a settlement after the fact, and an
|
|
87
|
+
// arbitrary string arriving from a hand edit should read as "unknown", not be echoed.
|
|
88
|
+
continuationDeliveredAs: DELIVERY_OUTCOMES.has(parsed.continuationDeliveredAs)
|
|
89
|
+
? parsed.continuationDeliveredAs
|
|
90
|
+
: undefined,
|
|
65
91
|
};
|
|
66
92
|
}
|
|
67
93
|
catch {
|
|
68
94
|
return { ...DEFAULT_PARENT_METADATA };
|
|
69
95
|
}
|
|
70
96
|
}
|
|
97
|
+
/**
|
|
98
|
+
* The JSON bag `persistParent` writes. Exported so start-paused is testable without a database —
|
|
99
|
+
* Pause-after-submit races the first dispatcher poll, so this bag is the only place `paused: true`
|
|
100
|
+
* is guaranteed to land before anyone claims.
|
|
101
|
+
*/
|
|
102
|
+
export function BuildTaskGraphParentInputPayload(args) {
|
|
103
|
+
return {
|
|
104
|
+
continuation: args.continuation,
|
|
105
|
+
reinvokeDepth: args.reinvokeDepth,
|
|
106
|
+
failureSemantics: args.failureSemantics,
|
|
107
|
+
submittedByAgentRunID: args.submittedByAgentRunID,
|
|
108
|
+
submittedByUserID: args.submittedByUserID,
|
|
109
|
+
...(args.invocation
|
|
110
|
+
? { invocation: { data: args.invocation.data, context: args.invocation.context } }
|
|
111
|
+
: {}),
|
|
112
|
+
...(args.startPaused
|
|
113
|
+
? {
|
|
114
|
+
debug: {
|
|
115
|
+
paused: true,
|
|
116
|
+
pausedReason: 'user',
|
|
117
|
+
pausedBy: args.submittedByUserID,
|
|
118
|
+
},
|
|
119
|
+
}
|
|
120
|
+
: {}),
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
/**
|
|
124
|
+
* How a settlement's announcement ended — the values `TryClaimContinuation` may record.
|
|
125
|
+
*
|
|
126
|
+
* `expired` means found too late to announce; `cancelled` means there was deliberately nobody left
|
|
127
|
+
* to announce to, because the run that submitted the graph was cancelled. Both are settlements that
|
|
128
|
+
* completed WITHOUT an announcement, and keeping them distinct is the difference between "we missed
|
|
129
|
+
* it" and "we chose not to".
|
|
130
|
+
*/
|
|
131
|
+
const DELIVERY_OUTCOMES = new Set(['delivered', 'expired', 'cancelled']);
|
|
71
132
|
/** True when a continuation chain has gone as far as it may. */
|
|
72
133
|
export function IsReinvokeCapReached(meta) {
|
|
73
134
|
return meta.reinvokeDepth >= MAX_REINVOKE_DEPTH;
|
|
74
135
|
}
|
|
75
136
|
/** Name of the task type used for agent-orchestrated graphs. */
|
|
76
|
-
const TASK_TYPE_NAME = 'AI Workflow';
|
|
137
|
+
export const TASK_TYPE_NAME = 'AI Workflow';
|
|
138
|
+
/**
|
|
139
|
+
* The node kinds a `Task` row can actually represent, and therefore the ones the dispatcher can run.
|
|
140
|
+
*
|
|
141
|
+
* `Agent` and `Action` have their own foreign keys; `Human` is a task with an assignee; `ForEach`
|
|
142
|
+
* and `While` carry their loop definition in `Task.Configuration` and their repeated body in the
|
|
143
|
+
* same `AgentID` / `ActionID` keys.
|
|
144
|
+
*
|
|
145
|
+
* `Prompt` joins them now that `TaskPromptRunner` exists — it carries `Task.PromptID`. `External`
|
|
146
|
+
* remains absent: it is completed by a system that has no way to report back, so persisting one
|
|
147
|
+
* would produce a task that waits forever.
|
|
148
|
+
*/
|
|
149
|
+
const DISPATCHABLE_KINDS = [
|
|
150
|
+
'Agent', 'Action', 'Human', 'ForEach', 'While', 'Prompt',
|
|
151
|
+
];
|
|
152
|
+
/**
|
|
153
|
+
* Reports the node kinds this dispatcher cannot execute, or `null` when the graph is fully runnable.
|
|
154
|
+
*
|
|
155
|
+
* **Why this is a submit-time check and not a validation rule.** `ValidateTaskGraphSpec` is a pure
|
|
156
|
+
* function over the spec: it answers "is this a well-formed graph?", and its answer has to be the
|
|
157
|
+
* same in a browser, a CLI and a server. "Can it run *here*?" is a different question whose answer
|
|
158
|
+
* changes as runners are added, so it belongs to the runtime that owns the runners.
|
|
159
|
+
*
|
|
160
|
+
* **Why refuse rather than persist-and-stall.** Before this existed, `persistTasks` keyed off which
|
|
161
|
+
* configuration field happened to be populated and fell through to "Human task assigned to the
|
|
162
|
+
* submitter" for everything else — so a loop step silently became an approval request nobody asked
|
|
163
|
+
* for and nothing would ever complete. A graph that hangs forever while *looking* like it is waiting
|
|
164
|
+
* on a person is the most expensive failure available here. Refusing at the door instead names the
|
|
165
|
+
* offending step while the workflow is still the author's to edit.
|
|
166
|
+
*/
|
|
167
|
+
/**
|
|
168
|
+
* Projects a spec node onto the `Task.Configuration` bag.
|
|
169
|
+
*
|
|
170
|
+
* Everything the row cannot hold in a column of its own: the kind-specific settings, the payload
|
|
171
|
+
* mappings, and the execution policy. Returning `null` for an empty result keeps `Configuration`
|
|
172
|
+
* NULL rather than `"{}"`, so "this step has no settings" reads the same in the database as it does
|
|
173
|
+
* in the spec.
|
|
174
|
+
*
|
|
175
|
+
* **The mappings are the point.** They are how a step's result reaches the payload, and every
|
|
176
|
+
* branch condition downstream reads the payload — so a step persisted without them produces a
|
|
177
|
+
* workflow whose conditions all evaluate against nothing. Undefined is falsy, so that failure looks
|
|
178
|
+
* exactly like a branch legitimately not being taken.
|
|
179
|
+
*/
|
|
180
|
+
export function BuildStepConfiguration(node) {
|
|
181
|
+
const config = {};
|
|
182
|
+
const agent = ConfigOf(node, 'Agent');
|
|
183
|
+
if (agent?.message || agent?.templateParameters) {
|
|
184
|
+
config.agent = { message: agent.message, templateParameters: agent.templateParameters };
|
|
185
|
+
}
|
|
186
|
+
const prompt = ConfigOf(node, 'Prompt');
|
|
187
|
+
if (prompt?.templateParameters)
|
|
188
|
+
config.prompt = { templateParameters: prompt.templateParameters };
|
|
189
|
+
const forEach = ConfigOf(node, 'ForEach');
|
|
190
|
+
if (forEach)
|
|
191
|
+
config.forEach = forEach;
|
|
192
|
+
const whileOp = ConfigOf(node, 'While');
|
|
193
|
+
if (whileOp)
|
|
194
|
+
config.while = whileOp;
|
|
195
|
+
// `expiresInHours` as well as instructions: dropping it here is what would leave the deadline in
|
|
196
|
+
// the spec and out of the row the dispatcher actually reads, so the request would be raised with
|
|
197
|
+
// no ExpiresAt and the author's timeout would silently not exist.
|
|
198
|
+
const human = ConfigOf(node, 'Human');
|
|
199
|
+
if (human?.instructions || human?.expiresInHours) {
|
|
200
|
+
config.human = { instructions: human.instructions, expiresInHours: human.expiresInHours };
|
|
201
|
+
}
|
|
202
|
+
const external = ConfigOf(node, 'External');
|
|
203
|
+
if (external)
|
|
204
|
+
config.external = external;
|
|
205
|
+
// Mappings live on the Action arm of the spec, but they are not action-specific: a loop step
|
|
206
|
+
// carries them too, which is how its per-iteration inputs and results are wired.
|
|
207
|
+
const action = ConfigOf(node, 'Action');
|
|
208
|
+
const inputMapping = action?.inputMapping;
|
|
209
|
+
const outputMapping = action?.outputMapping;
|
|
210
|
+
if (inputMapping)
|
|
211
|
+
config.inputMapping = inputMapping;
|
|
212
|
+
if (outputMapping)
|
|
213
|
+
config.outputMapping = outputMapping;
|
|
214
|
+
if (node.policy) {
|
|
215
|
+
config.policy = {
|
|
216
|
+
timeoutSeconds: node.policy.timeoutSeconds,
|
|
217
|
+
retryCount: node.policy.retryCount,
|
|
218
|
+
onError: node.policy.onError,
|
|
219
|
+
};
|
|
220
|
+
}
|
|
221
|
+
// The author's own arrangement, carried through so a hand-drawn workflow runs — and appears in
|
|
222
|
+
// run history — in the shape they drew. Dropping it (as this did) meant a workflow someone had
|
|
223
|
+
// laid out carefully came back as a machine-arranged graph the first time they watched it run.
|
|
224
|
+
// Only authored geometry is stored: a graph with none has its layout derived at render time, and
|
|
225
|
+
// persisting a derived layout would freeze one rendering of a graph that can still change.
|
|
226
|
+
if (node.layout && Object.keys(node.layout).length > 0) {
|
|
227
|
+
config.layout = {
|
|
228
|
+
x: node.layout.x,
|
|
229
|
+
y: node.layout.y,
|
|
230
|
+
width: node.layout.width,
|
|
231
|
+
height: node.layout.height,
|
|
232
|
+
};
|
|
233
|
+
}
|
|
234
|
+
return Object.keys(config).length > 0 ? config : null;
|
|
235
|
+
}
|
|
236
|
+
/** Internal alias so the persistence path reads as a step, not as a projection. */
|
|
237
|
+
const buildStepConfiguration = BuildStepConfiguration;
|
|
238
|
+
/**
|
|
239
|
+
* The loop definition on a node, whichever loop kind it is.
|
|
240
|
+
*
|
|
241
|
+
* ForEach and While differ in how they decide to iterate, not in what they repeat, so everything
|
|
242
|
+
* downstream of that decision — body resolution, name collection, persistence — treats them alike.
|
|
243
|
+
*/
|
|
244
|
+
export function LoopOperationOf(node) {
|
|
245
|
+
return ConfigOf(node, 'ForEach') ?? ConfigOf(node, 'While') ?? null;
|
|
246
|
+
}
|
|
247
|
+
/** Every agent name a node references, including the sub-agent a loop repeats. */
|
|
248
|
+
function agentNamesIn(node) {
|
|
249
|
+
const names = [ConfigOf(node, 'Agent')?.agentName, LoopOperationOf(node)?.subAgent?.name];
|
|
250
|
+
return names.filter((n) => !!n);
|
|
251
|
+
}
|
|
252
|
+
/** Every prompt name a node references, including the prompt a loop repeats. */
|
|
253
|
+
function promptNamesIn(node) {
|
|
254
|
+
const names = [ConfigOf(node, 'Prompt')?.promptName, LoopOperationOf(node)?.prompt?.name];
|
|
255
|
+
return names.filter((n) => !!n);
|
|
256
|
+
}
|
|
257
|
+
/** Every action name a node references, including the action a loop repeats. */
|
|
258
|
+
function actionNamesIn(node) {
|
|
259
|
+
const names = [ConfigOf(node, 'Action')?.actionName, LoopOperationOf(node)?.action?.name];
|
|
260
|
+
return names.filter((n) => !!n);
|
|
261
|
+
}
|
|
262
|
+
/**
|
|
263
|
+
* The transaction capability of a provider, when it has one.
|
|
264
|
+
*
|
|
265
|
+
* `IMetadataProvider` does not declare transaction support — a browser provider genuinely has none —
|
|
266
|
+
* so this narrows by CAPABILITY rather than asserting a type the interface does not promise. A
|
|
267
|
+
* provider without it returns undefined and `RunInEntityTransaction` runs the work directly, which
|
|
268
|
+
* is the honest degradation: server submissions get atomicity, and a client submission behaves
|
|
269
|
+
* exactly as it did before rather than failing at a call site that claimed something untrue.
|
|
270
|
+
*/
|
|
271
|
+
function asTransactionCapable(provider) {
|
|
272
|
+
const candidate = provider;
|
|
273
|
+
return candidate.SupportsEntityTransactions === true && typeof candidate.BeginEntityTransaction === 'function'
|
|
274
|
+
? candidate
|
|
275
|
+
: undefined;
|
|
276
|
+
}
|
|
277
|
+
export function FindUnrunnableKinds(spec) {
|
|
278
|
+
const offenders = spec.tasks.filter((t) => !DISPATCHABLE_KINDS.includes(t.kind));
|
|
279
|
+
if (offenders.length === 0)
|
|
280
|
+
return null;
|
|
281
|
+
const detail = offenders.map((t) => `"${t.name}" (${t.kind})`).join(', ');
|
|
282
|
+
return (`"${spec.workflowName}" cannot be run yet: ${detail}. ` +
|
|
283
|
+
`The dispatcher runs agent, action, person and loop steps. Prompt steps and steps completed ` +
|
|
284
|
+
`by an outside system are not supported yet — replace them, or split them out of this workflow.`);
|
|
285
|
+
}
|
|
286
|
+
/**
|
|
287
|
+
* Reports human steps assigned to someone other than the submitter, or `null` when there are none.
|
|
288
|
+
*
|
|
289
|
+
* Cross-user assignment needs an authorization model (#3524) — deciding that A may put work in B's
|
|
290
|
+
* inbox is a permissions question, not a graph question. Until it lands, a workflow can only ask the
|
|
291
|
+
* person who started it.
|
|
292
|
+
*
|
|
293
|
+
* **Why refuse rather than reassign.** Persist wrote `task.UserID = submitter` unconditionally, so
|
|
294
|
+
* an authored `assignToUserID` was overwritten in silence. Every layer above accepts the field —
|
|
295
|
+
* the flow compiler reads it into the spec, the validator passes it, the spec type declares it — so
|
|
296
|
+
* silence here is indistinguishable from support: the graph submits, a step appears in the WRONG
|
|
297
|
+
* person's inbox, the named person is never told, and the author has no reason to suspect any of it.
|
|
298
|
+
* Refusing while the graph is still the author's to edit is the only point at which saying so costs
|
|
299
|
+
* nothing.
|
|
300
|
+
*/
|
|
301
|
+
export function FindCrossUserAssignments(spec, submitterUserID) {
|
|
302
|
+
const offenders = spec.tasks.filter((t) => {
|
|
303
|
+
const assignTo = ConfigOf(t, 'Human')?.assignToUserID;
|
|
304
|
+
return !!assignTo && !UUIDsEqual(assignTo, submitterUserID);
|
|
305
|
+
});
|
|
306
|
+
if (offenders.length === 0)
|
|
307
|
+
return null;
|
|
308
|
+
const detail = offenders.map((t) => `"${t.name}"`).join(', ');
|
|
309
|
+
return (`"${spec.workflowName}" was not started: ${detail} asks a person other than whoever runs the ` +
|
|
310
|
+
`workflow. Assigning a step to someone else is not available yet (#3524) — a workflow can ` +
|
|
311
|
+
`only ask the person who started it. Remove assignToUserID from those steps.`);
|
|
312
|
+
}
|
|
77
313
|
export class TaskGraphService {
|
|
314
|
+
constructor() {
|
|
315
|
+
/**
|
|
316
|
+
* Guarded single-statement writes, shared with the dispatcher.
|
|
317
|
+
*
|
|
318
|
+
* The instance id is descriptive only — this service never CLAIMS anything, it only issues
|
|
319
|
+
* guarded transitions whose predicates are about the row's own status rather than about who
|
|
320
|
+
* holds it.
|
|
321
|
+
*/
|
|
322
|
+
this.claims = new TaskClaimStore('task-graph-service', 0);
|
|
323
|
+
// ────────────────────────────────────────────────────────────────────────
|
|
324
|
+
// debug / runner control plane
|
|
325
|
+
//
|
|
326
|
+
// Every verb here is durable, declarative state the dispatcher's claim filter consults on its
|
|
327
|
+
// next pass — never a call into a running dispatcher. That is what makes the controls work
|
|
328
|
+
// across instances and restarts, and what bounds their latency to one poll interval. See
|
|
329
|
+
// `debug-state.ts` for the model.
|
|
330
|
+
// ────────────────────────────────────────────────────────────────────────
|
|
331
|
+
/**
|
|
332
|
+
* Store for the guarded JSON_MODIFY writes. The instance identity and TTL are claim-protocol
|
|
333
|
+
* concerns this class never exercises — the debug writes are instance-free.
|
|
334
|
+
*/
|
|
335
|
+
this.debugWrites = new TaskClaimStore('task-graph-service', 0);
|
|
336
|
+
}
|
|
78
337
|
/**
|
|
79
338
|
* Validates and persists a task graph, returning as soon as it is durable.
|
|
80
339
|
*
|
|
@@ -92,22 +351,90 @@ export class TaskGraphService {
|
|
|
92
351
|
LogError(`[TaskGraphService] ${message}`);
|
|
93
352
|
return { Success: false, ErrorMessage: message };
|
|
94
353
|
}
|
|
354
|
+
// 1b. Representability. Validation asks "is this a well-formed graph?"; this asks "can THIS
|
|
355
|
+
// dispatcher run it?" — a different question, and one the pure validator has no business
|
|
356
|
+
// answering, since capability is a property of the runtime, not of the spec.
|
|
357
|
+
const unrunnable = this.findUnrunnableKinds(spec);
|
|
358
|
+
if (unrunnable) {
|
|
359
|
+
LogError(`[TaskGraphService] ${unrunnable}`);
|
|
360
|
+
return { Success: false, ErrorMessage: unrunnable };
|
|
361
|
+
}
|
|
362
|
+
// 1b-ii. Assignability. Same question, narrower: this dispatcher can only ask the person who
|
|
363
|
+
// submitted the graph. Persist used to overwrite an authored `assignToUserID` with
|
|
364
|
+
// the submitter and say nothing, so a step meant for someone else landed in the
|
|
365
|
+
// wrong inbox and the named person was never told. The compiler accepts the field and
|
|
366
|
+
// the validator passes it, which makes silence here indistinguishable from support.
|
|
367
|
+
const misassigned = FindCrossUserAssignments(spec, context.ContextUser.ID);
|
|
368
|
+
if (misassigned) {
|
|
369
|
+
LogError(`[TaskGraphService] ${misassigned}`);
|
|
370
|
+
return { Success: false, ErrorMessage: misassigned };
|
|
371
|
+
}
|
|
372
|
+
// 1c. Chain depth. A flow that dispatches a graph containing itself recurses without bound,
|
|
373
|
+
// and each hop costs real money and real rows before anyone notices. The cap is checked
|
|
374
|
+
// HERE rather than at execution because refusing to write the graph is the only point at
|
|
375
|
+
// which nothing has happened yet.
|
|
376
|
+
const depth = context.ReinvokeDepth ?? 0;
|
|
377
|
+
if (depth >= MAX_REINVOKE_DEPTH) {
|
|
378
|
+
const message = `"${spec.workflowName}" was not started: it is ${depth} levels deep in a chain of ` +
|
|
379
|
+
`workflows starting workflows, which is the limit. A workflow that reaches this is ` +
|
|
380
|
+
`almost always calling itself, directly or through another one.`;
|
|
381
|
+
LogError(`[TaskGraphService] ${message}`);
|
|
382
|
+
return { Success: false, ErrorMessage: message };
|
|
383
|
+
}
|
|
95
384
|
try {
|
|
96
|
-
// 2. Resolve every agent BEFORE writing anything. An unresolvable
|
|
97
|
-
// error, not a skipped node: silently dropping a task executes the graph with
|
|
98
|
-
// where the caller's work should have been.
|
|
385
|
+
// 2. Resolve every agent and action BEFORE writing anything. An unresolvable name is a
|
|
386
|
+
// hard error, not a skipped node: silently dropping a task executes the graph with
|
|
387
|
+
// holes where the caller's work should have been.
|
|
99
388
|
const agentIDsByName = await this.resolveAgents(spec, context);
|
|
100
389
|
if (!agentIDsByName.Success) {
|
|
101
390
|
return { Success: false, ErrorMessage: agentIDsByName.ErrorMessage };
|
|
102
391
|
}
|
|
392
|
+
const actionIDsByName = await this.resolveActions(spec, context);
|
|
393
|
+
if (!actionIDsByName.Success) {
|
|
394
|
+
return { Success: false, ErrorMessage: actionIDsByName.ErrorMessage };
|
|
395
|
+
}
|
|
396
|
+
const promptIDsByName = await this.resolvePrompts(spec, context);
|
|
397
|
+
if (!promptIDsByName.Success) {
|
|
398
|
+
return { Success: false, ErrorMessage: promptIDsByName.ErrorMessage };
|
|
399
|
+
}
|
|
103
400
|
const taskTypeID = await this.ensureTaskType(context);
|
|
104
|
-
// 3. Persist
|
|
105
|
-
//
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
401
|
+
// 3. Persist — ALL of it, or none of it.
|
|
402
|
+
//
|
|
403
|
+
// ── The whole graph becomes visible together, or not at all ─────────────────────────
|
|
404
|
+
// The dispatcher discovers work by polling for child tasks in 'Pending'. Writing the
|
|
405
|
+
// children first and their dependencies afterwards leaves a window — milliseconds, but
|
|
406
|
+
// real — in which every task exists with NO prerequisites recorded yet. A poll landing
|
|
407
|
+
// there sees a graph of independent tasks and claims all of them at once.
|
|
408
|
+
//
|
|
409
|
+
// Observed, not theorised: in graph C08B36E3 the dependency gating `Research: focused`
|
|
410
|
+
// was written at 00:00:59.653 and that task STARTED at 00:00:59.647 — six milliseconds
|
|
411
|
+
// before the edge that was supposed to hold it back existed. `Close out: approved` ran
|
|
412
|
+
// in the same wave as the research steps, before the draft it was meant to judge had
|
|
413
|
+
// been written. The graph then reported Complete, having executed in an order its author
|
|
414
|
+
// never drew.
|
|
415
|
+
//
|
|
416
|
+
// A transaction is the whole fix: the dispatcher cannot observe a half-built graph
|
|
417
|
+
// because a half-built graph is never visible. `RunInEntityTransaction` degrades to
|
|
418
|
+
// running the work as-is on a provider that cannot transact, which is the correct
|
|
419
|
+
// fallback rather than a silent failure.
|
|
420
|
+
//
|
|
421
|
+
// The PARENT is inside it too. Written outside, a failure while persisting children or
|
|
422
|
+
// edges would roll those back and leave a childless parent durably in 'Pending' — which
|
|
423
|
+
// never settles (the rollup deliberately skips a graph with no nodes) and never dies, so
|
|
424
|
+
// the active-graph scan picks it up on every poll forever: permanent debris plus a
|
|
425
|
+
// permanent tick of wasted work for each failed submit. Ordering inside the transaction
|
|
426
|
+
// is unchanged, because edges only ever reference child IDs.
|
|
427
|
+
const { parentTaskID, taskIDMap } = await RunInEntityTransaction(asTransactionCapable(context.Provider), async () => {
|
|
428
|
+
// Parent first so children have a ParentID, then children, then edges — edges
|
|
429
|
+
// last because they reference two child IDs that must both exist.
|
|
430
|
+
const parentID = await this.persistParent(spec, taskTypeID, context);
|
|
431
|
+
const map = await this.persistChildren(spec, parentID, taskTypeID, agentIDsByName.Map, actionIDsByName.Map, promptIDsByName.Map, context);
|
|
432
|
+
await this.persistDependencies(spec, map, context);
|
|
433
|
+
return { parentTaskID: parentID, taskIDMap: map };
|
|
434
|
+
});
|
|
109
435
|
LogStatus(`[TaskGraphService] Submitted "${spec.workflowName}": parent ${parentTaskID}, ${taskIDMap.size} task(s). ` +
|
|
110
436
|
`Awaiting dispatcher pickup.`);
|
|
437
|
+
KickTaskGraphDispatchers();
|
|
111
438
|
return { Success: true, ParentTaskID: parentTaskID, TaskIDMap: taskIDMap };
|
|
112
439
|
}
|
|
113
440
|
catch (e) {
|
|
@@ -117,34 +444,202 @@ export class TaskGraphService {
|
|
|
117
444
|
}
|
|
118
445
|
}
|
|
119
446
|
/**
|
|
120
|
-
* Cancels a graph
|
|
447
|
+
* Cancels a graph, everything in it that has not already settled, and everything it started.
|
|
121
448
|
*
|
|
122
449
|
* Cancels children first: a parent marked `Cancelled` while children are still `Pending` would
|
|
123
450
|
* leave the dispatcher free to pick those children up, which is the opposite of what the caller
|
|
124
451
|
* asked for.
|
|
452
|
+
*
|
|
453
|
+
* **The verdict is the outcome, not the attempt** (R2-9). This returned `true` unconditionally
|
|
454
|
+
* while logging each child that failed to cancel — so one failed save left that child `Pending`,
|
|
455
|
+
* told the caller cancellation had succeeded, and let the dispatcher run the child afterwards.
|
|
456
|
+
* The graph could then settle `Complete` and ANNOUNCE ITS COMPLETION into the conversation of a
|
|
457
|
+
* workflow the user had cancelled. A partial cancel now says so and names what survived; the
|
|
458
|
+
* graph stays active, so retrying is meaningful rather than cosmetic.
|
|
125
459
|
*/
|
|
126
460
|
async Cancel(parentTaskID, context) {
|
|
461
|
+
return this.cancelWithDepth(parentTaskID, context, 0, new Set());
|
|
462
|
+
}
|
|
463
|
+
/**
|
|
464
|
+
* `Cancel`, carrying the recursion state the public entry point does not expose.
|
|
465
|
+
*
|
|
466
|
+
* **The depth cap was dead code** (R3-10): `Cancel` passed a literal 0, and the recursion
|
|
467
|
+
* re-entered through `this.Cancel`, which restarted at 0 — so the check could never fire and
|
|
468
|
+
* the "bounded by the reinvoke depth cap" promise was false. A hand-edited `AgentRunID` cycle
|
|
469
|
+
* recursed to stack overflow mid-cancel.
|
|
470
|
+
*
|
|
471
|
+
* The visited set is cheap armour on top: the cap bounds how DEEP a legitimate chain goes, and
|
|
472
|
+
* a cycle is not deep, it is circular. Arithmetic alone would eventually stop it; a visited set
|
|
473
|
+
* stops it immediately and covers linkage shapes the arithmetic does not anticipate.
|
|
474
|
+
*/
|
|
475
|
+
async cancelWithDepth(parentTaskID, context, depth, visited) {
|
|
476
|
+
if (visited.has(parentTaskID)) {
|
|
477
|
+
return { Success: true, Cancelled: true, UncancelledTaskNames: [] };
|
|
478
|
+
}
|
|
479
|
+
visited.add(parentTaskID);
|
|
127
480
|
try {
|
|
128
481
|
const children = await this.loadChildren(parentTaskID, context);
|
|
482
|
+
const uncancelled = [];
|
|
483
|
+
const settledMeanwhile = [];
|
|
129
484
|
for (const child of children) {
|
|
130
485
|
// Terminal work is left alone — cancelling a completed task would rewrite history.
|
|
131
|
-
|
|
486
|
+
// The in-memory test is a cheap pre-filter; the one that MATTERS is in the statement
|
|
487
|
+
// (R3-9), because a child can settle between this snapshot and its own write, and
|
|
488
|
+
// the full-row save this replaces overwrote that outcome wholesale.
|
|
489
|
+
if (['Complete', 'Failed', 'Cancelled', 'Skipped'].includes(child.Status))
|
|
490
|
+
continue;
|
|
491
|
+
if (await this.claims.TryCancelTask(context.Provider, child.ID, context.ContextUser))
|
|
132
492
|
continue;
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
493
|
+
// Rowcount 0 means it reached a terminal status while we were cancelling its
|
|
494
|
+
// siblings. Its outcome is real and stays; the verdict says so rather than pretending
|
|
495
|
+
// the cancel was total.
|
|
496
|
+
settledMeanwhile.push(child.Name);
|
|
497
|
+
}
|
|
498
|
+
if (settledMeanwhile.length > 0) {
|
|
499
|
+
LogStatus(`[TaskGraphService] ${settledMeanwhile.length} task(s) settled while the cancel ran ` +
|
|
500
|
+
`(${settledMeanwhile.join(', ')}); their outcomes are kept.`);
|
|
501
|
+
}
|
|
502
|
+
// Withdraw the questions too. A human step that was waiting has an open
|
|
503
|
+
// `MJ: AI Agent Requests` row, and cancelling only the Task left that row `Requested`
|
|
504
|
+
// FOREVER: the person keeps seeing "a workflow is waiting on you" in their inbox for a
|
|
505
|
+
// workflow that no longer exists, and answering it settles nothing because the task is
|
|
506
|
+
// already Cancelled. Nothing else ever closes these — the dispatcher only expires rows
|
|
507
|
+
// that carry a deadline, and most do not.
|
|
508
|
+
await this.cancelOpenRequests(children.map((c) => c.ID), context);
|
|
509
|
+
// THE PARENT IS LEFT TO THE DISPATCHER, DELIBERATELY.
|
|
510
|
+
//
|
|
511
|
+
// Writing it terminal here skipped the settle path entirely — no cost rollup, no run
|
|
512
|
+
// settlement, no notification — so the submitting agent run stayed `Paused` forever.
|
|
513
|
+
// Worse, it was NONDETERMINISTIC: if a dispatcher poll happened to land between the
|
|
514
|
+
// child cancels above and the parent write, the graph settled through the normal path
|
|
515
|
+
// and the run WAS failed and messaged. Cancel behaved differently run to run depending
|
|
516
|
+
// on timing.
|
|
517
|
+
//
|
|
518
|
+
// With the children cancelled, `ComputeParentRollup` reaches `Cancelled` on its own and
|
|
519
|
+
// the ordinary settle sequence runs — rollup, run settlement, continuation — exactly as
|
|
520
|
+
// it does for a graph that finished by itself. Less code, one path, and a deterministic
|
|
521
|
+
// outcome.
|
|
522
|
+
//
|
|
523
|
+
// The parent stays non-terminal until then, so the sweep still sees it as active work.
|
|
524
|
+
// WHAT THIS WORKFLOW STARTED IS ALSO CANCELLED (R2-9).
|
|
525
|
+
//
|
|
526
|
+
// A graph's step can be an agent that submits a graph of its own, and those sub-graphs
|
|
527
|
+
// persist as ROOTS — linked back only through the child task's `AgentRunID`. So
|
|
528
|
+
// cancelling a workflow left its descendants running, and on settlement one of them can
|
|
529
|
+
// REINVOKE the cancelled workflow's own agent for a fresh billed turn: the user stopped
|
|
530
|
+
// a workflow and it started itself again.
|
|
531
|
+
//
|
|
532
|
+
// Bounded by the reinvoke depth cap, which is what bounds the chain in the first place.
|
|
533
|
+
const nested = await this.cancelNestedGraphs(children, context, depth, visited);
|
|
534
|
+
uncancelled.push(...nested);
|
|
535
|
+
if (uncancelled.length > 0) {
|
|
536
|
+
return {
|
|
537
|
+
Success: false,
|
|
538
|
+
Cancelled: false,
|
|
539
|
+
UncancelledTaskNames: uncancelled,
|
|
540
|
+
ErrorMessage: `Cancelled what it could, but ${uncancelled.length} task(s) could not be cancelled ` +
|
|
541
|
+
`(${uncancelled.join(', ')}). The workflow is still active — retry the cancel.`,
|
|
542
|
+
};
|
|
543
|
+
}
|
|
544
|
+
return { Success: true, Cancelled: true, UncancelledTaskNames: [] };
|
|
545
|
+
}
|
|
546
|
+
catch (e) {
|
|
547
|
+
const message = e instanceof Error ? e.message : String(e);
|
|
548
|
+
LogError(`[TaskGraphService] Cancel failed for ${parentTaskID}: ${message}`);
|
|
549
|
+
return { Success: false, Cancelled: false, UncancelledTaskNames: [], ErrorMessage: message };
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
/**
|
|
553
|
+
* Cancels the graphs that this graph's own steps submitted, one level at a time.
|
|
554
|
+
*
|
|
555
|
+
* The linkage is `child task → AgentRunID → the graphs that run submitted`, which is exactly how
|
|
556
|
+
* the continuation chain finds its way back up; walking it downward is the same relation read the
|
|
557
|
+
* other way. Depth-capped by the same constant that caps reinvocation, so a self-referencing
|
|
558
|
+
* workflow cannot make cancellation recurse further than it could have spawned.
|
|
559
|
+
*
|
|
560
|
+
* @returns names of tasks in descendant graphs that could not be cancelled
|
|
561
|
+
*/
|
|
562
|
+
async cancelNestedGraphs(children, context, depth, visited) {
|
|
563
|
+
if (depth >= MAX_REINVOKE_DEPTH) {
|
|
564
|
+
LogError(`[TaskGraphService] Nested cancel stopped at depth ${depth}; a deeper sub-graph chain ` +
|
|
565
|
+
`than the reinvoke cap allows may still be running.`);
|
|
566
|
+
return [];
|
|
567
|
+
}
|
|
568
|
+
const runIDs = [...new Set(children.map((c) => c.AgentRunID).filter((id) => !!id))];
|
|
569
|
+
if (runIDs.length === 0)
|
|
570
|
+
return [];
|
|
571
|
+
const rv = RunView.FromMetadataProvider(context.Provider);
|
|
572
|
+
const inList = runIDs.map((id) => `'${id}'`).join(',');
|
|
573
|
+
// TYPE-SCOPED, like every other graph-mutating walk over this table (R3-10). `MJ: Tasks` is
|
|
574
|
+
// general-purpose, and without the predicate any non-workflow root hierarchy that happens to
|
|
575
|
+
// carry a cancelled run's ID gets `Cancelled` written over its children and its requests
|
|
576
|
+
// withdrawn — the user-writable-table threat the claim store's guards exist to defend
|
|
577
|
+
// against. The comment here already claimed this scoping; the query did not have it.
|
|
578
|
+
const typeID = await this.findTaskTypeID(context);
|
|
579
|
+
if (!typeID)
|
|
580
|
+
return [];
|
|
581
|
+
const subGraphs = await rv.RunView({
|
|
582
|
+
EntityName: 'MJ: Tasks',
|
|
583
|
+
ExtraFilter: `TypeID='${typeID}' AND ParentID IS NULL AND AgentRunID IN (${inList})`,
|
|
584
|
+
Fields: ['ID'],
|
|
585
|
+
ResultType: 'simple',
|
|
586
|
+
BypassCache: true,
|
|
587
|
+
}, context.ContextUser);
|
|
588
|
+
if (!subGraphs.Success) {
|
|
589
|
+
LogError(`[TaskGraphService] Could not look for sub-graphs while cancelling: ${subGraphs.ErrorMessage}`);
|
|
590
|
+
return [];
|
|
591
|
+
}
|
|
592
|
+
const failures = [];
|
|
593
|
+
for (const row of subGraphs.Results ?? []) {
|
|
594
|
+
// Through the depth-carrying overload, so the cap actually engages.
|
|
595
|
+
const result = await this.cancelWithDepth(row.ID, context, depth + 1, visited);
|
|
596
|
+
if (!result.Success)
|
|
597
|
+
failures.push(...result.UncancelledTaskNames);
|
|
598
|
+
}
|
|
599
|
+
return failures;
|
|
600
|
+
}
|
|
601
|
+
/**
|
|
602
|
+
* Closes the still-open requests raised for a set of tasks.
|
|
603
|
+
*
|
|
604
|
+
* `Canceled` rather than `Expired`: nobody ran out of time, the ask was withdrawn — and the two
|
|
605
|
+
* mean different things downstream, since the dispatcher treats an expired human step as a
|
|
606
|
+
* FAILURE a give-up edge can route around, which would be a lie about a graph somebody stopped
|
|
607
|
+
* on purpose.
|
|
608
|
+
*
|
|
609
|
+
* Failures here are logged and never propagated: the graph is already cancelled, and refusing to
|
|
610
|
+
* finish that because an inbox row would not close would leave the graph in a worse state than
|
|
611
|
+
* the debris it is trying to avoid.
|
|
612
|
+
*/
|
|
613
|
+
async cancelOpenRequests(taskIDs, context) {
|
|
614
|
+
if (taskIDs.length === 0)
|
|
615
|
+
return;
|
|
616
|
+
try {
|
|
617
|
+
const idList = taskIDs.map((id) => `'${id}'`).join(',');
|
|
618
|
+
const open = await RunView.FromMetadataProvider(context.Provider).RunView({
|
|
619
|
+
EntityName: 'MJ: AI Agent Requests',
|
|
620
|
+
ExtraFilter: `Status='Requested' AND OriginatingTaskID IN (${idList})`,
|
|
621
|
+
ResultType: 'entity_object',
|
|
622
|
+
BypassCache: true,
|
|
623
|
+
}, context.ContextUser);
|
|
624
|
+
if (!open.Success) {
|
|
625
|
+
LogError(`[TaskGraphService] Could not read open requests to cancel: ${open.ErrorMessage}`);
|
|
626
|
+
return;
|
|
627
|
+
}
|
|
628
|
+
for (const request of open.Results ?? []) {
|
|
629
|
+
request.Status = 'Canceled';
|
|
630
|
+
request.Comments = 'The workflow that asked this was cancelled.';
|
|
631
|
+
if (!(await request.Save())) {
|
|
632
|
+
LogError(`[TaskGraphService] Could not withdraw request ${request.ID}: ` +
|
|
633
|
+
`${request.LatestResult?.CompleteMessage ?? 'unknown error'}. It will keep showing ` +
|
|
634
|
+
`in someone's inbox for a workflow that no longer exists.`);
|
|
136
635
|
}
|
|
137
636
|
}
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
parent.Status = 'Cancelled';
|
|
142
|
-
parent.CompletedAt = new Date();
|
|
143
|
-
return await parent.Save();
|
|
637
|
+
if ((open.Results ?? []).length > 0) {
|
|
638
|
+
LogStatus(`[TaskGraphService] Withdrew ${open.Results.length} open request(s) for the cancelled graph.`);
|
|
639
|
+
}
|
|
144
640
|
}
|
|
145
641
|
catch (e) {
|
|
146
|
-
LogError(`[TaskGraphService]
|
|
147
|
-
return false;
|
|
642
|
+
LogError(`[TaskGraphService] Could not withdraw open requests: ${e instanceof Error ? e.message : String(e)}`);
|
|
148
643
|
}
|
|
149
644
|
}
|
|
150
645
|
/**
|
|
@@ -154,7 +649,7 @@ export class TaskGraphService {
|
|
|
154
649
|
* leaving them blocked would make the retry pointless, as the graph still could not progress
|
|
155
650
|
* past this node.
|
|
156
651
|
*/
|
|
157
|
-
async Retry(taskID, context) {
|
|
652
|
+
async Retry(taskID, context, inputPayload) {
|
|
158
653
|
try {
|
|
159
654
|
const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
|
|
160
655
|
if (!(await task.Load(taskID)))
|
|
@@ -163,6 +658,27 @@ export class TaskGraphService {
|
|
|
163
658
|
LogError(`[TaskGraphService] Cannot retry task ${taskID}: status is ${task.Status}, expected Failed.`);
|
|
164
659
|
return false;
|
|
165
660
|
}
|
|
661
|
+
// An edited input rides the retry: the operator saw WHY it failed and is re-running the
|
|
662
|
+
// step with a corrected brief. Applies to this run only — the graph's spec is long gone.
|
|
663
|
+
//
|
|
664
|
+
// Written through the GUARDED statement rather than onto the in-memory row, so the edit
|
|
665
|
+
// cannot ride along on the full-row save below. The window here is narrower than
|
|
666
|
+
// `UpdateTaskInput`'s (the pre-state is `Failed`, so a concurrent claim is not the
|
|
667
|
+
// hazard — a concurrent human retry is), but the shape is the same and it costs one
|
|
668
|
+
// statement to not have it. The rest of this method's full-row save predates this PR
|
|
669
|
+
// and is Round 3's to purge; the new write does not add to it.
|
|
670
|
+
if (inputPayload !== undefined) {
|
|
671
|
+
const typeID = await this.ensureTaskType(context);
|
|
672
|
+
const json = typeof inputPayload === 'string' ? inputPayload : JSON.stringify(inputPayload);
|
|
673
|
+
const wrote = await this.debugWrites.TryUpdateInputPayload(context.Provider, taskID, json, 'Failed', typeID, context.ContextUser);
|
|
674
|
+
if (!wrote) {
|
|
675
|
+
LogError(`[TaskGraphService] Could not apply the edited input to task ${taskID}; retry refused rather than re-running the old brief.`);
|
|
676
|
+
return false;
|
|
677
|
+
}
|
|
678
|
+
// Keep the in-memory row in step with what was just written, so the save below does
|
|
679
|
+
// not put the old input back.
|
|
680
|
+
task.InputPayload = json;
|
|
681
|
+
}
|
|
166
682
|
task.Status = 'Pending';
|
|
167
683
|
task.ErrorMessage = null;
|
|
168
684
|
task.StartedAt = null;
|
|
@@ -188,12 +704,275 @@ export class TaskGraphService {
|
|
|
188
704
|
return false;
|
|
189
705
|
}
|
|
190
706
|
}
|
|
707
|
+
/** Result shape shared by the control verbs: what happened, and the state that now holds. */
|
|
708
|
+
controlResult(success, debug, errorMessage) {
|
|
709
|
+
return { Success: success, Debug: debug, ErrorMessage: errorMessage };
|
|
710
|
+
}
|
|
711
|
+
/**
|
|
712
|
+
* Loads a graph parent and proves it IS a workflow graph before any debug write.
|
|
713
|
+
*
|
|
714
|
+
* Read with `BypassCache` for the same reason the dispatcher reads rows that way: the debug bag
|
|
715
|
+
* is written by direct `JSON_MODIFY` statements that fire no cache invalidation, so a cached
|
|
716
|
+
* read here could merge new state over a stale copy and silently resurrect a cleared flag.
|
|
717
|
+
*/
|
|
718
|
+
async loadWorkflowParent(parentTaskID, context) {
|
|
719
|
+
const typeID = await this.ensureTaskType(context);
|
|
720
|
+
const rows = await RunView.FromMetadataProvider(context.Provider).RunView({
|
|
721
|
+
EntityName: 'MJ: Tasks',
|
|
722
|
+
ExtraFilter: `ID='${parentTaskID.replace(/'/g, "''")}'`,
|
|
723
|
+
Fields: ['ID', 'TypeID', 'InputPayload', 'Status'],
|
|
724
|
+
ResultType: 'simple',
|
|
725
|
+
BypassCache: true,
|
|
726
|
+
}, context.ContextUser);
|
|
727
|
+
const row = rows.Success ? rows.Results?.[0] : undefined;
|
|
728
|
+
if (!row)
|
|
729
|
+
return null;
|
|
730
|
+
if (!UUIDsEqual(row.TypeID, typeID))
|
|
731
|
+
return null;
|
|
732
|
+
return { typeID, inputPayload: row.InputPayload, status: row.Status };
|
|
733
|
+
}
|
|
734
|
+
/**
|
|
735
|
+
* Writes the debug-bag fields a verb OWNS, and reports the state that results.
|
|
736
|
+
*
|
|
737
|
+
* Field-scoped on purpose — see {@link TaskClaimStore.TryWriteDebugFields}. A verb declares the
|
|
738
|
+
* paths it is responsible for; everything else in the bag is left exactly as the database has
|
|
739
|
+
* it, so a concurrent step-consume, breakpoint edit, or override cannot be undone by a verb that
|
|
740
|
+
* was not talking about them.
|
|
741
|
+
*
|
|
742
|
+
* The returned state is this instance's best view (read + the fields just written) and is
|
|
743
|
+
* advisory — the same posture the console takes toward frames.
|
|
744
|
+
*/
|
|
745
|
+
async writeDebugFields(parentTaskID, context, build) {
|
|
746
|
+
const parent = await this.loadWorkflowParent(parentTaskID, context);
|
|
747
|
+
if (!parent)
|
|
748
|
+
return this.controlResult(false, undefined, 'Not a workflow graph this control plane can act on.');
|
|
749
|
+
const { Fields, Next } = build(ParseTaskGraphDebugState(parent.inputPayload));
|
|
750
|
+
const ok = await this.debugWrites.TryWriteDebugFields(context.Provider, parentTaskID, Fields, parent.typeID, context.ContextUser);
|
|
751
|
+
if (!ok)
|
|
752
|
+
return this.controlResult(false, undefined, 'The debug state could not be written; see the server log.');
|
|
753
|
+
return this.controlResult(true, Next);
|
|
754
|
+
}
|
|
755
|
+
/**
|
|
756
|
+
* Pauses a graph: nothing new is claimed until it is resumed. In-flight steps finish naturally
|
|
757
|
+
* and their completions land — a pause gates claiming and never touches a live claim, which is
|
|
758
|
+
* why there is no "what happens to the claim" question to answer.
|
|
759
|
+
*/
|
|
760
|
+
async PauseGraph(parentTaskID, context, pausedByUserID) {
|
|
761
|
+
const pausedBy = pausedByUserID ?? context.ContextUser?.ID ?? null;
|
|
762
|
+
return this.writeDebugFields(parentTaskID, context, (current) => ({
|
|
763
|
+
// Pause owns the pause fields AND the step allowance: an allowance armed a moment ago is
|
|
764
|
+
// for a run the operator has now stopped, so clearing it is the verb's meaning rather
|
|
765
|
+
// than a side effect. Breakpoints and overrides are untouched — they outlive a pause.
|
|
766
|
+
Fields: [
|
|
767
|
+
TaskClaimStore.DebugField('$.debug.paused', { Kind: 'bool', Value: true }),
|
|
768
|
+
TaskClaimStore.DebugField('$.debug.pausedReason', { Kind: 'string', Value: 'user' }),
|
|
769
|
+
TaskClaimStore.DebugField('$.debug.pausedBy', pausedBy ? { Kind: 'string', Value: pausedBy } : { Kind: 'null' }),
|
|
770
|
+
TaskClaimStore.DebugField('$.debug.pausedAtTaskID', { Kind: 'null' }),
|
|
771
|
+
TaskClaimStore.DebugField('$.debug.step', { Kind: 'null' }),
|
|
772
|
+
TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' }),
|
|
773
|
+
],
|
|
774
|
+
Next: { ...current, paused: true, pausedBy, pausedReason: 'user', pausedAtTaskID: null, step: undefined, skipBreakpointTaskID: undefined },
|
|
775
|
+
}));
|
|
776
|
+
}
|
|
777
|
+
/** Resumes a paused graph. Breakpoints and edge overrides survive — only the pause clears. */
|
|
778
|
+
async ResumeGraph(parentTaskID, context) {
|
|
779
|
+
return this.writeDebugFields(parentTaskID, context, (current) => {
|
|
780
|
+
const next = { ...current };
|
|
781
|
+
// Continue from a breakpoint must run the stopped task. Stamping it here so the next
|
|
782
|
+
// poll does not re-hit the same still-eligible breakpoint without claiming.
|
|
783
|
+
const skip = current.pausedReason === 'breakpoint' ? current.pausedAtTaskID : null;
|
|
784
|
+
delete next.paused;
|
|
785
|
+
delete next.pausedBy;
|
|
786
|
+
delete next.pausedReason;
|
|
787
|
+
delete next.pausedAtTaskID;
|
|
788
|
+
delete next.step;
|
|
789
|
+
if (skip)
|
|
790
|
+
next.skipBreakpointTaskID = skip;
|
|
791
|
+
else
|
|
792
|
+
delete next.skipBreakpointTaskID;
|
|
793
|
+
return {
|
|
794
|
+
Fields: [
|
|
795
|
+
TaskClaimStore.DebugField('$.debug.paused', { Kind: 'null' }),
|
|
796
|
+
TaskClaimStore.DebugField('$.debug.pausedReason', { Kind: 'null' }),
|
|
797
|
+
TaskClaimStore.DebugField('$.debug.pausedBy', { Kind: 'null' }),
|
|
798
|
+
TaskClaimStore.DebugField('$.debug.pausedAtTaskID', { Kind: 'null' }),
|
|
799
|
+
TaskClaimStore.DebugField('$.debug.step', { Kind: 'null' }),
|
|
800
|
+
skip
|
|
801
|
+
? TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'string', Value: skip })
|
|
802
|
+
: TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' }),
|
|
803
|
+
],
|
|
804
|
+
Next: next,
|
|
805
|
+
};
|
|
806
|
+
});
|
|
807
|
+
}
|
|
808
|
+
/**
|
|
809
|
+
* Arms a one-shot step allowance on a paused graph: `'one'` releases the next eligible task,
|
|
810
|
+
* `'wave'` releases the current frontier, a task ID releases exactly that task. The dispatcher
|
|
811
|
+
* consumes the allowance CAS-style, so two instances stepping the same graph release work once.
|
|
812
|
+
*/
|
|
813
|
+
async StepGraph(parentTaskID, target, context) {
|
|
814
|
+
const parent = await this.loadWorkflowParent(parentTaskID, context);
|
|
815
|
+
if (!parent)
|
|
816
|
+
return this.controlResult(false, undefined, 'Not a workflow graph this control plane can act on.');
|
|
817
|
+
const current = ParseTaskGraphDebugState(parent.inputPayload);
|
|
818
|
+
if (!current.paused) {
|
|
819
|
+
return this.controlResult(false, current, 'Step only applies to a paused workflow — pause it first.');
|
|
820
|
+
}
|
|
821
|
+
return this.writeDebugFields(parentTaskID, context, (state) => ({
|
|
822
|
+
Fields: [TaskClaimStore.DebugField('$.debug.step', { Kind: 'string', Value: target })],
|
|
823
|
+
Next: { ...state, step: target },
|
|
824
|
+
}));
|
|
825
|
+
}
|
|
826
|
+
/**
|
|
827
|
+
* Replaces the graph's breakpoint set. Every ID must name a child of this graph — a breakpoint
|
|
828
|
+
* on a task in some other graph would gate nothing and silently lie to the person who set it.
|
|
829
|
+
*/
|
|
830
|
+
async SetBreakpoints(parentTaskID, taskIDs, context) {
|
|
831
|
+
const children = await this.loadChildren(parentTaskID, context);
|
|
832
|
+
const childIDs = new Set(children.map((c) => c.ID.toLowerCase()));
|
|
833
|
+
const foreign = taskIDs.filter((id) => !childIDs.has(id.toLowerCase()));
|
|
834
|
+
if (foreign.length > 0) {
|
|
835
|
+
return this.controlResult(false, undefined, `Not steps of this workflow: ${foreign.join(', ')}`);
|
|
836
|
+
}
|
|
837
|
+
return this.writeDebugFields(parentTaskID, context, (current) => {
|
|
838
|
+
const next = { ...current };
|
|
839
|
+
if (taskIDs.length > 0)
|
|
840
|
+
next.breakpoints = [...taskIDs];
|
|
841
|
+
else
|
|
842
|
+
delete next.breakpoints;
|
|
843
|
+
return {
|
|
844
|
+
Fields: [
|
|
845
|
+
TaskClaimStore.DebugField('$.debug.breakpoints', taskIDs.length > 0 ? { Kind: 'json', Value: JSON.stringify(taskIDs) } : { Kind: 'null' }),
|
|
846
|
+
],
|
|
847
|
+
Next: next,
|
|
848
|
+
};
|
|
849
|
+
});
|
|
850
|
+
}
|
|
851
|
+
/**
|
|
852
|
+
* Overrides one edge's condition verdict — the operator's answer for a path the engine cannot
|
|
853
|
+
* decide (a held graph) or decided wrongly (a broken guard). `'false'` reads as "branch not
|
|
854
|
+
* taken" and cascades skips; `'true'` opens the gate; `null` removes the override.
|
|
855
|
+
*/
|
|
856
|
+
async SetEdgeOverride(parentTaskID, edgeID, verdict, context) {
|
|
857
|
+
// Prove the edge belongs to this graph before writing anything about it.
|
|
858
|
+
const edge = await context.Provider.GetEntityObject('MJ: Task Dependencies', context.ContextUser);
|
|
859
|
+
if (!(await edge.Load(edgeID)))
|
|
860
|
+
return this.controlResult(false, undefined, 'No such path.');
|
|
861
|
+
const target = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
|
|
862
|
+
if (!(await target.Load(edge.TaskID)) || !UUIDsEqual(target.ParentID ?? '', parentTaskID)) {
|
|
863
|
+
return this.controlResult(false, undefined, 'That path does not belong to this workflow.');
|
|
864
|
+
}
|
|
865
|
+
// AN UNCONDITIONAL PATH CANNOT BE OVERRIDDEN — refused here so the two dialects agree.
|
|
866
|
+
//
|
|
867
|
+
// The engine reads overrides at different depths: an ordinary edge only consults one when
|
|
868
|
+
// it HAS a condition, while the exclusive evaluator consults it before its no-condition
|
|
869
|
+
// early return. Left open, the same override would force an unconditional exclusive edge to
|
|
870
|
+
// lose while doing nothing at all to an unconditional ordinary one — the operator's answer
|
|
871
|
+
// meaning two different things depending on a property of the graph they cannot see.
|
|
872
|
+
// Refusing is the conservative reading, and it costs nothing: "don't take this branch" is
|
|
873
|
+
// already expressible, precisely, as SkipTask on the step itself.
|
|
874
|
+
if (verdict !== null && !edge.Condition?.trim()) {
|
|
875
|
+
return this.controlResult(false, undefined, 'That path has no condition to answer — it is always taken. To stop the branch, skip its step instead.');
|
|
876
|
+
}
|
|
877
|
+
return this.writeDebugFields(parentTaskID, context, (current) => {
|
|
878
|
+
const overrides = { ...(current.edgeOverrides ?? {}) };
|
|
879
|
+
if (verdict === null)
|
|
880
|
+
delete overrides[edgeID];
|
|
881
|
+
else
|
|
882
|
+
overrides[edgeID] = verdict;
|
|
883
|
+
const next = { ...current };
|
|
884
|
+
if (Object.keys(overrides).length > 0)
|
|
885
|
+
next.edgeOverrides = overrides;
|
|
886
|
+
else
|
|
887
|
+
delete next.edgeOverrides;
|
|
888
|
+
return {
|
|
889
|
+
// Scoped to THIS edge's key, not the whole map: two operators answering two
|
|
890
|
+
// different held paths at once must not overwrite each other's answer.
|
|
891
|
+
Fields: [
|
|
892
|
+
TaskClaimStore.DebugField(`$.debug.edgeOverrides."${edgeID}"`, verdict === null ? { Kind: 'null' } : { Kind: 'string', Value: verdict }),
|
|
893
|
+
],
|
|
894
|
+
Next: next,
|
|
895
|
+
};
|
|
896
|
+
});
|
|
897
|
+
}
|
|
898
|
+
/**
|
|
899
|
+
* Declares a Pending step not-taken. Downstream dependents proceed — `Skipped` satisfies a
|
|
900
|
+
* prerequisite — and any open human request for the step is withdrawn so nobody keeps seeing an
|
|
901
|
+
* ask for work the operator decided against.
|
|
902
|
+
*/
|
|
903
|
+
async SkipTask(taskID, context) {
|
|
904
|
+
const typeID = await this.ensureTaskType(context);
|
|
905
|
+
const ok = await this.debugWrites.TrySkipPending(context.Provider, taskID, typeID, context.ContextUser);
|
|
906
|
+
if (!ok) {
|
|
907
|
+
return { Success: false, ErrorMessage: 'Only a step that has not started can be skipped.' };
|
|
908
|
+
}
|
|
909
|
+
await this.cancelOpenRequests([taskID], context);
|
|
910
|
+
return { Success: true };
|
|
911
|
+
}
|
|
912
|
+
/**
|
|
913
|
+
* Marks a step Complete with an operator-supplied output.
|
|
914
|
+
*
|
|
915
|
+
* Human steps are refused here on purpose: they already have a first-class completion path
|
|
916
|
+
* (`TaskGraph.CompleteTask`) with the assignee/elevation check, and this verb must not become
|
|
917
|
+
* the door that bypasses it.
|
|
918
|
+
*/
|
|
919
|
+
async ForceCompleteTask(taskID, outputPayload, context) {
|
|
920
|
+
const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
|
|
921
|
+
if (!(await task.Load(taskID)))
|
|
922
|
+
return { Success: false, ErrorMessage: 'No such step.' };
|
|
923
|
+
if (task.UserID) {
|
|
924
|
+
return { Success: false, ErrorMessage: 'A human step completes through its assignee — use CompleteTask.' };
|
|
925
|
+
}
|
|
926
|
+
const json = outputPayload == null
|
|
927
|
+
? null
|
|
928
|
+
: typeof outputPayload === 'string' ? outputPayload : JSON.stringify(outputPayload);
|
|
929
|
+
const typeID = await this.ensureTaskType(context);
|
|
930
|
+
const ok = await this.debugWrites.TryForceComplete(context.Provider, taskID, json, typeID, context.ContextUser);
|
|
931
|
+
if (!ok) {
|
|
932
|
+
return {
|
|
933
|
+
Success: false,
|
|
934
|
+
ErrorMessage: 'The step is running with a live claim, or already finished. Cancel it or wait for the claim to lapse.',
|
|
935
|
+
};
|
|
936
|
+
}
|
|
937
|
+
return { Success: true };
|
|
938
|
+
}
|
|
939
|
+
/**
|
|
940
|
+
* Replaces a Pending step's input — the "edit the brief before stepping" move at a breakpoint.
|
|
941
|
+
* Applies to this run only; the step must not have started.
|
|
942
|
+
*
|
|
943
|
+
* **A guarded statement, not load-check-save.** The in-memory `Status === 'Pending'` check plus
|
|
944
|
+
* `task.Save()` is an unconditional full-row UPDATE: a task claimed in the window between the
|
|
945
|
+
* load and the save has its claim columns reverted to the pre-claim snapshot *while its body
|
|
946
|
+
* runs*, and a second instance then claims it again — the step executes twice. See
|
|
947
|
+
* {@link TaskClaimStore.TryUpdateInputPayload}, which makes the check and the write one atomic
|
|
948
|
+
* operation whose rowcount is the answer.
|
|
949
|
+
*/
|
|
950
|
+
async UpdateTaskInput(taskID, inputPayload, context) {
|
|
951
|
+
try {
|
|
952
|
+
const typeID = await this.ensureTaskType(context);
|
|
953
|
+
const json = typeof inputPayload === 'string' ? inputPayload : JSON.stringify(inputPayload);
|
|
954
|
+
const ok = await this.debugWrites.TryUpdateInputPayload(context.Provider, taskID, json, 'Pending', typeID, context.ContextUser);
|
|
955
|
+
if (!ok) {
|
|
956
|
+
return {
|
|
957
|
+
Success: false,
|
|
958
|
+
ErrorMessage: 'Only a workflow step that has not started can have its input edited.',
|
|
959
|
+
};
|
|
960
|
+
}
|
|
961
|
+
return { Success: true };
|
|
962
|
+
}
|
|
963
|
+
catch (e) {
|
|
964
|
+
return { Success: false, ErrorMessage: e instanceof Error ? e.message : String(e) };
|
|
965
|
+
}
|
|
966
|
+
}
|
|
191
967
|
// ────────────────────────────────────────────────────────────────────────
|
|
192
968
|
// internals
|
|
193
969
|
// ────────────────────────────────────────────────────────────────────────
|
|
194
970
|
/** Maps every referenced agent name to its ID, or reports all unresolvable names at once. */
|
|
195
971
|
async resolveAgents(spec, context) {
|
|
196
|
-
|
|
972
|
+
// A loop's repeated sub-agent counts as a referenced agent. Collecting only the Agent nodes
|
|
973
|
+
// left a loop body pointing at a name nothing had resolved, so the row got no AgentID and
|
|
974
|
+
// the loop had nothing to run.
|
|
975
|
+
const names = [...new Set(spec.tasks.flatMap(agentNamesIn))];
|
|
197
976
|
if (names.length === 0)
|
|
198
977
|
return { Success: true, Map: new Map() };
|
|
199
978
|
const quoted = names.map((n) => `'${n.replace(/'/g, "''")}'`).join(',');
|
|
@@ -209,20 +988,111 @@ export class TaskGraphService {
|
|
|
209
988
|
}
|
|
210
989
|
return { Success: true, Map: found };
|
|
211
990
|
}
|
|
212
|
-
/**
|
|
991
|
+
/**
|
|
992
|
+
* Maps every referenced action name to its ID, or reports all unresolvable names at once.
|
|
993
|
+
*
|
|
994
|
+
* Deliberately a mirror of {@link resolveAgents} rather than a generalization of it: the two
|
|
995
|
+
* read different entities and produce different error prose, and the shared shape is three
|
|
996
|
+
* lines. Collapsing them would trade a readable failure message for a parameterized lookup.
|
|
997
|
+
*/
|
|
998
|
+
/**
|
|
999
|
+
* Maps every referenced prompt name to its ID.
|
|
1000
|
+
*
|
|
1001
|
+
* A prompt is addressed by name in the spec and stored as a foreign key on the row, exactly like
|
|
1002
|
+
* agents and actions — a name in JSON cannot be joined, checked, or survive a rename.
|
|
1003
|
+
*/
|
|
1004
|
+
async resolvePrompts(spec, context) {
|
|
1005
|
+
const names = [...new Set(spec.tasks.flatMap(promptNamesIn))];
|
|
1006
|
+
if (names.length === 0)
|
|
1007
|
+
return { Success: true, Map: new Map() };
|
|
1008
|
+
const quoted = names.map((n) => `'${n.replace(/'/g, "''")}'`).join(',');
|
|
1009
|
+
const result = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: AI Prompts', ExtraFilter: `Name IN (${quoted})`, Fields: ['ID', 'Name'], ResultType: 'simple' }, context.ContextUser);
|
|
1010
|
+
const found = new Map((result.Results ?? []).map((r) => [r.Name, r.ID]));
|
|
1011
|
+
const missing = names.filter((n) => !found.has(n));
|
|
1012
|
+
if (missing.length > 0) {
|
|
1013
|
+
return {
|
|
1014
|
+
Success: false,
|
|
1015
|
+
ErrorMessage: `Task graph "${spec.workflowName}" references ${missing.length} unknown prompt(s): ${missing.join(', ')}. ` +
|
|
1016
|
+
`Submitting would execute the graph with holes where those steps should be.`,
|
|
1017
|
+
};
|
|
1018
|
+
}
|
|
1019
|
+
return { Success: true, Map: found };
|
|
1020
|
+
}
|
|
1021
|
+
async resolveActions(spec, context) {
|
|
1022
|
+
// Includes a loop's repeated action — see the note in resolveAgents.
|
|
1023
|
+
const names = [...new Set(spec.tasks.flatMap(actionNamesIn))];
|
|
1024
|
+
if (names.length === 0)
|
|
1025
|
+
return { Success: true, Map: new Map() };
|
|
1026
|
+
const quoted = names.map((n) => `'${n.replace(/'/g, "''")}'`).join(',');
|
|
1027
|
+
const result = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: Actions', ExtraFilter: `Name IN (${quoted})`, Fields: ['ID', 'Name'], ResultType: 'simple' }, context.ContextUser);
|
|
1028
|
+
const found = new Map((result.Results ?? []).map((r) => [r.Name, r.ID]));
|
|
1029
|
+
const missing = names.filter((n) => !found.has(n));
|
|
1030
|
+
if (missing.length > 0) {
|
|
1031
|
+
return {
|
|
1032
|
+
Success: false,
|
|
1033
|
+
ErrorMessage: `Task graph "${spec.workflowName}" references ${missing.length} unknown action(s): ${missing.join(', ')}. ` +
|
|
1034
|
+
`Submitting would execute the graph with holes where those tasks should be.`,
|
|
1035
|
+
};
|
|
1036
|
+
}
|
|
1037
|
+
return { Success: true, Map: found };
|
|
1038
|
+
}
|
|
1039
|
+
/**
|
|
1040
|
+
* Finds or creates the task type used for orchestrated graphs — exactly one of it, ever.
|
|
1041
|
+
*
|
|
1042
|
+
* **This resolves the engine's discriminator, not a label** (R2-7). Round 1 scoped every sweep
|
|
1043
|
+
* arm and both payload-writing guards to this type, so a second row sharing the name lets
|
|
1044
|
+
* different processes bind different IDs — and a graph stamped with the other one is invisible
|
|
1045
|
+
* to the sweep, never claimed, never settled, its submitting run `Paused` forever, with no
|
|
1046
|
+
* error anywhere.
|
|
1047
|
+
*
|
|
1048
|
+
* Race-safe by INSERT-then-reselect rather than by checking harder. Two concurrent first-ever
|
|
1049
|
+
* submissions both read "not there" and both insert; the unique index added in this round makes
|
|
1050
|
+
* the loser's insert fail, and the loser then re-reads and finds the winner's row. Checking
|
|
1051
|
+
* first is what created the window, so the fix cannot be a better check.
|
|
1052
|
+
*/
|
|
213
1053
|
async ensureTaskType(context) {
|
|
214
|
-
const
|
|
215
|
-
const found = existing.Results?.[0]?.ID;
|
|
1054
|
+
const found = await this.findTaskTypeID(context);
|
|
216
1055
|
if (found)
|
|
217
1056
|
return found;
|
|
218
1057
|
const tt = await context.Provider.GetEntityObject('MJ: Task Types', context.ContextUser);
|
|
219
1058
|
tt.NewRecord();
|
|
220
1059
|
tt.Name = TASK_TYPE_NAME;
|
|
221
1060
|
tt.Description = 'Tasks created by agent-orchestrated workflows.';
|
|
222
|
-
if (
|
|
223
|
-
|
|
1061
|
+
if (await tt.Save())
|
|
1062
|
+
return tt.ID;
|
|
1063
|
+
// The insert lost. Almost certainly to the unique index and another process that got there
|
|
1064
|
+
// first — so re-read before treating it as a failure. Any other cause falls through to the
|
|
1065
|
+
// throw below with its own message intact.
|
|
1066
|
+
const winner = await this.findTaskTypeID(context);
|
|
1067
|
+
if (winner)
|
|
1068
|
+
return winner;
|
|
1069
|
+
throw new Error(`Could not create task type: ${tt.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
1070
|
+
}
|
|
1071
|
+
/**
|
|
1072
|
+
* The `AI Workflow` task type's ID, resolved deterministically.
|
|
1073
|
+
*
|
|
1074
|
+
* `ORDER BY` is not decoration: two rows sharing the name come back in whatever order the engine
|
|
1075
|
+
* chooses, so an unordered `MaxRows: 1` lets two processes bind different IDs from the same
|
|
1076
|
+
* data. The index this round adds makes duplicates impossible going forward; the ordering makes
|
|
1077
|
+
* the resolution deterministic on a database that still has some, and the warning makes the
|
|
1078
|
+
* situation visible rather than merely survivable.
|
|
1079
|
+
*/
|
|
1080
|
+
async findTaskTypeID(context) {
|
|
1081
|
+
const existing = await RunView.FromMetadataProvider(context.Provider).RunView({
|
|
1082
|
+
EntityName: 'MJ: Task Types',
|
|
1083
|
+
ExtraFilter: `Name='${TASK_TYPE_NAME}'`,
|
|
1084
|
+
Fields: ['ID'],
|
|
1085
|
+
OrderBy: '__mj_CreatedAt ASC, ID ASC',
|
|
1086
|
+
ResultType: 'simple',
|
|
1087
|
+
MaxRows: 2,
|
|
1088
|
+
}, context.ContextUser);
|
|
1089
|
+
const rows = existing.Results ?? [];
|
|
1090
|
+
if (rows.length > 1) {
|
|
1091
|
+
LogError(`[TaskGraphService] More than one '${TASK_TYPE_NAME}' task type exists. Binding the oldest ` +
|
|
1092
|
+
`(${rows[0].ID}), but graphs stamped with the other are invisible to the dispatcher's ` +
|
|
1093
|
+
`sweep and will never settle. Merge them.`);
|
|
224
1094
|
}
|
|
225
|
-
return
|
|
1095
|
+
return rows[0]?.ID ?? null;
|
|
226
1096
|
}
|
|
227
1097
|
/** Writes the parent task that represents the graph as a whole. */
|
|
228
1098
|
async persistParent(spec, taskTypeID, context) {
|
|
@@ -235,24 +1105,52 @@ export class TaskGraphService {
|
|
|
235
1105
|
parent.ConversationDetailID = context.ConversationDetailID ?? null;
|
|
236
1106
|
parent.Status = 'In Progress';
|
|
237
1107
|
parent.PercentComplete = 0;
|
|
1108
|
+
// The run that submitted this graph, in the COLUMN and not only in the metadata JSON below.
|
|
1109
|
+
// A json field cannot be joined, so provenance that lives only there is unavailable to any
|
|
1110
|
+
// query — which is how a human step ended up with no agent to ask on behalf of, and how a
|
|
1111
|
+
// dispatched run had no parent to roll its cost up to.
|
|
1112
|
+
parent.AgentRunID = context.AgentRunID ?? null;
|
|
238
1113
|
// The parent row carries what happens AFTER the graph settles. It lives here rather than in
|
|
239
1114
|
// dispatcher memory because the dispatcher that finishes a graph is frequently not the
|
|
240
1115
|
// process that accepted it — a restart, a second instance, or simply a long-running graph
|
|
241
1116
|
// all break that assumption. Anything the completion path needs has to be durable too.
|
|
242
|
-
|
|
1117
|
+
// Done before the write, and reported: a value that silently left the envelope becomes a
|
|
1118
|
+
// condition reading absent-data and taking the other branch, with nothing saying why.
|
|
1119
|
+
const sanitized = SanitizeInvocationEnvelope(context.Invocation);
|
|
1120
|
+
if (sanitized.DroppedPaths.length > 0) {
|
|
1121
|
+
LogStatus(`[TaskGraphService] Invocation envelope for '${spec.workflowName}' dropped ` +
|
|
1122
|
+
`${sanitized.DroppedPaths.length} non-persistable value(s): ` +
|
|
1123
|
+
`${sanitized.DroppedPaths.join(', ')}. Conditions referencing them will read as ` +
|
|
1124
|
+
`absent data. Pass plain JSON values for anything a condition needs.`);
|
|
1125
|
+
}
|
|
1126
|
+
// Only written when the caller supplied one, so a graph with no invocation envelope
|
|
1127
|
+
// carries no key at all rather than a misleading empty object. SANITIZED first: the
|
|
1128
|
+
// agent's `context` is documented as possibly a class instance holding connections and
|
|
1129
|
+
// credentials, and carrying it verbatim threw `Converting circular structure to JSON`
|
|
1130
|
+
// on any context with a socket in it — killing the run at submit time — while a context
|
|
1131
|
+
// that happened to serialize would have written its credentials to this row.
|
|
1132
|
+
parent.InputPayload = JSON.stringify(BuildTaskGraphParentInputPayload({
|
|
243
1133
|
continuation: spec.continuation ?? 'message',
|
|
244
1134
|
reinvokeDepth: context.ReinvokeDepth ?? 0,
|
|
1135
|
+
failureSemantics: spec.failureSemantics ?? 'block',
|
|
245
1136
|
submittedByAgentRunID: context.AgentRunID ?? null,
|
|
246
1137
|
submittedByUserID: context.ContextUser?.ID ?? null,
|
|
247
|
-
|
|
1138
|
+
invocation: sanitized.Envelope
|
|
1139
|
+
? { data: sanitized.Envelope.Data, context: sanitized.Envelope.Context }
|
|
1140
|
+
: null,
|
|
1141
|
+
startPaused: context.Debug?.paused === true,
|
|
1142
|
+
}));
|
|
248
1143
|
if (!(await parent.Save())) {
|
|
249
1144
|
throw new Error(`Could not create parent task: ${parent.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
250
1145
|
}
|
|
251
1146
|
return parent.ID;
|
|
252
1147
|
}
|
|
253
1148
|
/** Writes each child task, returning the tempId -> real ID mapping edges will need. */
|
|
254
|
-
async persistChildren(spec, parentTaskID, taskTypeID, agentIDsByName, context) {
|
|
1149
|
+
async persistChildren(spec, parentTaskID, taskTypeID, agentIDsByName, actionIDsByName, promptIDsByName, context) {
|
|
255
1150
|
const map = new Map();
|
|
1151
|
+
// Resolved ONCE over the whole graph: a rank is a node's position in the topology, so it
|
|
1152
|
+
// cannot be computed per node without seeing all of them.
|
|
1153
|
+
const ranks = RankGraphNodes(spec.tasks.map((t) => t.tempId), spec.tasks.flatMap((t) => (t.dependsOn ?? []).map((d) => ({ From: NormalizeDependency(d).tempId, To: t.tempId }))));
|
|
256
1154
|
for (const node of spec.tasks) {
|
|
257
1155
|
const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
|
|
258
1156
|
task.NewRecord();
|
|
@@ -264,14 +1162,69 @@ export class TaskGraphService {
|
|
|
264
1162
|
task.ConversationDetailID = context.ConversationDetailID ?? null;
|
|
265
1163
|
task.Status = 'Pending';
|
|
266
1164
|
task.PercentComplete = 0;
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
1165
|
+
// The discriminator the dispatcher routes on. Written first because everything below —
|
|
1166
|
+
// and everything the dispatcher later does with this row — reads it.
|
|
1167
|
+
task.StepType = node.kind;
|
|
1168
|
+
// Switching on `kind` rather than on which config field happens to be populated. The
|
|
1169
|
+
// old shape — "agentName? then agent; actionName? then action; ELSE human" — turned
|
|
1170
|
+
// every kind it did not know about into a Human task assigned to the submitter, so a
|
|
1171
|
+
// ForEach step became an approval request that nobody had asked for and nothing would
|
|
1172
|
+
// ever complete. A graph that stalls forever while LOOKING like it is waiting on a
|
|
1173
|
+
// person is the worst available failure. `findUnrunnableKinds` refuses unsupported
|
|
1174
|
+
// kinds before anything is written; this switch is what keeps the two in step.
|
|
1175
|
+
switch (node.kind) {
|
|
1176
|
+
case 'Agent':
|
|
1177
|
+
task.AgentID = agentIDsByName.get(ConfigOf(node, 'Agent').agentName);
|
|
1178
|
+
break;
|
|
1179
|
+
case 'Action':
|
|
1180
|
+
task.ActionID = actionIDsByName.get(ConfigOf(node, 'Action').actionName);
|
|
1181
|
+
break;
|
|
1182
|
+
case 'Human':
|
|
1183
|
+
// Assigned to the submitting user only — cross-user assignment stays rejected
|
|
1184
|
+
// until the authorization model in #3524 lands, and `findCrossUserAssignments`
|
|
1185
|
+
// has already refused any graph that asked for someone else.
|
|
1186
|
+
task.UserID = context.ContextUser.ID;
|
|
1187
|
+
break;
|
|
1188
|
+
case 'Prompt':
|
|
1189
|
+
task.PromptID = promptIDsByName.get(ConfigOf(node, 'Prompt').promptName);
|
|
1190
|
+
break;
|
|
1191
|
+
case 'ForEach':
|
|
1192
|
+
case 'While': {
|
|
1193
|
+
// A loop's key points at what it REPEATS, not at the loop itself. That keeps the
|
|
1194
|
+
// reference a real foreign key — joinable, constrained, rename-proof — instead
|
|
1195
|
+
// of a name buried in JSON, and CK_Task_Assignment still holds because a loop
|
|
1196
|
+
// body is exactly one thing.
|
|
1197
|
+
const op = LoopOperationOf(node);
|
|
1198
|
+
if (op?.action)
|
|
1199
|
+
task.ActionID = actionIDsByName.get(op.action.name);
|
|
1200
|
+
else if (op?.subAgent)
|
|
1201
|
+
task.AgentID = agentIDsByName.get(op.subAgent.name);
|
|
1202
|
+
else if (op?.prompt)
|
|
1203
|
+
task.PromptID = promptIDsByName.get(op.prompt.name);
|
|
1204
|
+
else {
|
|
1205
|
+
throw new Error(`Task "${node.name}" is a loop with nothing to repeat. Choose an action, ` +
|
|
1206
|
+
`a prompt, or a sub-agent for it to run on each pass.`);
|
|
1207
|
+
}
|
|
1208
|
+
break;
|
|
1209
|
+
}
|
|
1210
|
+
default:
|
|
1211
|
+
// Unreachable: findUnrunnableKinds rejected these before any write. Throwing
|
|
1212
|
+
// rather than falling through means a kind added later fails loudly here
|
|
1213
|
+
// instead of quietly becoming somebody's to-do item.
|
|
1214
|
+
throw new Error(`Task "${node.name}" has kind "${node.kind}", which cannot be persisted for dispatch.`);
|
|
274
1215
|
}
|
|
1216
|
+
// Everything about the step that has no column of its own. Dropping this is what used
|
|
1217
|
+
// to lose the input/output mappings — and with them the payload values every branch
|
|
1218
|
+
// condition downstream reads.
|
|
1219
|
+
//
|
|
1220
|
+
// The step's rank in the graph's own order rides along. `Task` has no sequence column,
|
|
1221
|
+
// so without it every consumer listing a graph's steps falls back to creation order —
|
|
1222
|
+
// the compiler's walk, which is neither the order they were drawn in nor the order they
|
|
1223
|
+
// run in. A graph that has not started yet has no timestamps to sort by, so its
|
|
1224
|
+
// structure is the only order available, and it is available from the moment it is
|
|
1225
|
+
// compiled.
|
|
1226
|
+
const configuration = { ...buildStepConfiguration(node), sequence: ranks.get(node.tempId) };
|
|
1227
|
+
task.Configuration = JSON.stringify(configuration);
|
|
275
1228
|
// Input rides in its own column; Description stays human-readable.
|
|
276
1229
|
task.InputPayload = node.inputPayload ? JSON.stringify(node.inputPayload) : null;
|
|
277
1230
|
if (!(await task.Save())) {
|
|
@@ -281,6 +1234,25 @@ export class TaskGraphService {
|
|
|
281
1234
|
}
|
|
282
1235
|
return map;
|
|
283
1236
|
}
|
|
1237
|
+
/**
|
|
1238
|
+
* Reports the node kinds this dispatcher cannot yet execute, or `null` when the graph is
|
|
1239
|
+
* entirely runnable.
|
|
1240
|
+
*
|
|
1241
|
+
* **Why this is a submit-time check and not a validation rule.** `ValidateTaskGraphSpec` is a
|
|
1242
|
+
* pure function over the spec: it answers "is this a well-formed graph?", and its answer must be
|
|
1243
|
+
* the same in a browser, a CLI and a server. "Can it run *here*?" is a different question whose
|
|
1244
|
+
* answer changes as runners are added, so it belongs to the runtime that owns the runners.
|
|
1245
|
+
*
|
|
1246
|
+
* **Why refuse rather than persist-and-stall.** `Prompt`, `ForEach`, `While` and `External`
|
|
1247
|
+
* are legitimate parts of the spec, but the `Task` row has nowhere to carry a node's `kind` or
|
|
1248
|
+
* its typed `configuration` — so a persisted one could not be dispatched even if a runner
|
|
1249
|
+
* existed. Refusing at the door tells the author which step is the problem while the graph is
|
|
1250
|
+
* still theirs to edit. The alternative is a graph that submits successfully and then never
|
|
1251
|
+
* finishes, which costs an operator an afternoon to diagnose.
|
|
1252
|
+
*/
|
|
1253
|
+
findUnrunnableKinds(spec) {
|
|
1254
|
+
return FindUnrunnableKinds(spec);
|
|
1255
|
+
}
|
|
284
1256
|
/** Writes the dependency edges, translating tempIds to persisted IDs. */
|
|
285
1257
|
async persistDependencies(spec, taskIDMap, context) {
|
|
286
1258
|
for (const node of spec.tasks) {
|
|
@@ -300,6 +1272,15 @@ export class TaskGraphService {
|
|
|
300
1272
|
// NULL for an unconditional edge, matching AIAgentStepPath — so a graph authored in
|
|
301
1273
|
// the flow editor and one emitted by an agent store the same thing.
|
|
302
1274
|
dep.Condition = edge.condition ?? null;
|
|
1275
|
+
// The exclusive-choice fields. Dropping these was silent and severe: without an
|
|
1276
|
+
// ExclusiveGroup the dispatcher sees a plain fan-out and runs EVERY branch, so a
|
|
1277
|
+
// workflow that should pick one route would take all of them — doing work its author
|
|
1278
|
+
// never intended and, in the Demo workflow's case, calling two different APIs where
|
|
1279
|
+
// the flow calls one. Priority and Sequence are what decide which branch wins, so a
|
|
1280
|
+
// group without them resolves arbitrarily.
|
|
1281
|
+
dep.ExclusiveGroup = edge.exclusiveGroup ?? null;
|
|
1282
|
+
dep.Priority = edge.priority ?? 0;
|
|
1283
|
+
dep.Sequence = edge.sequence ?? 0;
|
|
303
1284
|
if (!(await dep.Save())) {
|
|
304
1285
|
throw new Error(`Could not create dependency ${node.tempId} -> ${edge.tempId}: ${dep.LatestResult?.CompleteMessage ?? 'unknown error'}`);
|
|
305
1286
|
}
|
|
@@ -307,7 +1288,22 @@ export class TaskGraphService {
|
|
|
307
1288
|
}
|
|
308
1289
|
}
|
|
309
1290
|
async loadChildren(parentTaskID, context) {
|
|
310
|
-
|
|
1291
|
+
// `BypassCache` for the reason the dispatcher documents at every one of its reads (C4): task
|
|
1292
|
+
// status is written by the claim protocol's direct SQL, which fires no cache invalidation.
|
|
1293
|
+
// A cached read here returns PRE-EXECUTION state, and both callers act on status — Cancel's
|
|
1294
|
+
// "leave terminal work alone" guard would pass for a task that has since completed, and
|
|
1295
|
+
// write `Cancelled` over a `Complete` row. That is precisely the history-rewriting the guard
|
|
1296
|
+
// exists to prevent, performed by the guard itself.
|
|
1297
|
+
// Type-scoped for the same reason the sub-graph walk is (R3-10): these rows are about to be
|
|
1298
|
+
// written, and `MJ: Tasks` holds conversation tasks and users' own to-dos too.
|
|
1299
|
+
const typeID = await this.findTaskTypeID(context);
|
|
1300
|
+
const ofType = typeID ? `TypeID='${typeID}' AND ` : '';
|
|
1301
|
+
const result = await RunView.FromMetadataProvider(context.Provider).RunView({
|
|
1302
|
+
EntityName: 'MJ: Tasks',
|
|
1303
|
+
ExtraFilter: `${ofType}ParentID='${parentTaskID}'`,
|
|
1304
|
+
ResultType: 'entity_object',
|
|
1305
|
+
BypassCache: true,
|
|
1306
|
+
}, context.ContextUser);
|
|
311
1307
|
return (result.Success ? result.Results : []) ?? [];
|
|
312
1308
|
}
|
|
313
1309
|
}
|