@dotdrelle/wiki-manager 0.15.49 → 0.15.52
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -961
- package/docker-compose.yml +7 -1
- package/package.json +10 -2
- package/src/agent/graph.js +4 -36
- package/src/agent/graph.test.js +12 -54
- package/src/cli/wiki-manager.js +28 -4
- package/src/commands/slash.js +1 -1
- package/src/core/agentEvents.js +12 -1
- package/src/core/agentEvents.test.js +21 -0
- package/src/core/buildInfo.json +2 -2
- package/src/core/dockerCompose.test.js +18 -0
- package/src/core/mcp.js +1 -1
- package/src/orchestrator/.fuse_hidden0000001c00000001 +316 -0
- package/src/orchestrator/agentRegistry.js +54 -0
- package/src/orchestrator/agentRegistry.test.js +75 -0
- package/src/runtime/client.js +2 -0
- package/src/runtime/delegation.js +1 -1
- package/src/runtime/runner.js +3 -3
- package/src/runtime/server.js +5 -4
- package/src/runtime/server.test.js +12 -1
- package/src/runtime/supervisor.js +18 -1
- package/src/runtime/supervisor.test.js +43 -0
- package/src/shell/LeftPane.tsx +17 -5
- package/src/shell/StartupScreen.tsx +9 -4
- package/src/shell/repl.js +1 -0
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
2
|
+
import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
|
|
3
|
+
import { normalizeRuntimeLog } from '../core/runtimeLog.js';
|
|
4
|
+
import { assertContract } from '../contracts/schemas.js';
|
|
5
|
+
|
|
6
|
+
const AVAILABLE = 'available';
|
|
7
|
+
const UNAVAILABLE = 'unavailable';
|
|
8
|
+
/**
|
|
9
|
+
* Santé d'un agent restauré depuis le journal, tant qu'aucun `agent_describe`
|
|
10
|
+
* n'a réussi dans le processus courant.
|
|
11
|
+
*
|
|
12
|
+
* Ni `available` ni `unavailable` : on ne SAIT pas. Le distinguer de
|
|
13
|
+
* `unavailable` a une conséquence pratique — un agent inconnu redevient
|
|
14
|
+
* disponible en silence dès le premier scan réussi, là où un agent déclaré
|
|
15
|
+
* indisponible mériterait d'être signalé comme tel à l'utilisateur.
|
|
16
|
+
*/
|
|
17
|
+
const UNKNOWN = 'unknown';
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Un agent persisté n'est pas un agent joignable.
|
|
21
|
+
*
|
|
22
|
+
* Au redémarrage, `hydrateSession` rejoue le journal d'événements et
|
|
23
|
+
* reconstruit les agents avec la santé qu'ils avaient AU MOMENT où l'événement
|
|
24
|
+
* a été écrit. `cme-main` réapparaissait donc `available` alors que son
|
|
25
|
+
* endpoint n'existe plus, était retenu comme fournisseur, et la tâche partait
|
|
26
|
+
* vers un agent absent — la panne observée le 2026-08-04.
|
|
27
|
+
*
|
|
28
|
+
* La persistance dit ce qui a existé, pas ce qui répond maintenant. Seul un
|
|
29
|
+
* `agent_describe` réussi dans ce processus autorise à parler de disponibilité.
|
|
30
|
+
*/
|
|
31
|
+
export function markPersistedAgentsStale(session) {
|
|
32
|
+
if (!session || typeof session !== 'object') return [];
|
|
33
|
+
const stale = (agent) => ({
|
|
34
|
+
...agent,
|
|
35
|
+
health: UNKNOWN,
|
|
36
|
+
stale: true,
|
|
37
|
+
// La santé d'origine est conservée : elle raconte ce qu'on savait avant
|
|
38
|
+
// l'arrêt, ce qui aide à lire un journal, sans jamais servir au routage.
|
|
39
|
+
healthBeforeRestart: agent?.health ?? null,
|
|
40
|
+
});
|
|
41
|
+
/*
|
|
42
|
+
TOUTES les représentations restaurées, `agentRegistrySnapshot` compris.
|
|
43
|
+
|
|
44
|
+
J'avais d'abord épargné le snapshot, au motif qu'il pouvait porter un scan
|
|
45
|
+
vivant. C'était une inversion : à l'hydratation, aucun scan n'a encore eu
|
|
46
|
+
lieu — l'ordre est hydrate → invalidate → discover, et rien ne s'exécute
|
|
47
|
+
entre les deux premiers. Le snapshot vient donc de la même projection
|
|
48
|
+
persistée que `session.agents`. L'épargner laissait `cme-main` routable avec
|
|
49
|
+
son endpoint éteint, ce que la validation à chaud a montré.
|
|
50
|
+
|
|
51
|
+
Le seul scan qui compte est celui qui suivra : `discover()` réécrit le
|
|
52
|
+
snapshot en entier à partir des `agent_describe` réussis.
|
|
53
|
+
*/
|
|
54
|
+
session.agents = (session.agents ?? []).map(stale);
|
|
55
|
+
session.agentRegistrySnapshot = (session.agentRegistrySnapshot ?? []).map(stale);
|
|
56
|
+
return session.agentRegistrySnapshot;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function createAgentRegistry({
|
|
60
|
+
callTool = callMcpTool,
|
|
61
|
+
now = () => new Date(),
|
|
62
|
+
} = {}) {
|
|
63
|
+
const agentsByInstance = new Map();
|
|
64
|
+
const instanceByServer = new Map();
|
|
65
|
+
|
|
66
|
+
return {
|
|
67
|
+
async discover(session, { signal = null } = {}) {
|
|
68
|
+
const discovered = [];
|
|
69
|
+
const endpoints = Object.entries(session?.mcp ?? {});
|
|
70
|
+
const activeServers = new Set(endpoints.map(([serverName]) => serverName));
|
|
71
|
+
for (const [serverName, endpoint] of endpoints) {
|
|
72
|
+
const agent = await discoverServerAgent(session, serverName, endpoint, { callTool, signal, now });
|
|
73
|
+
discovered.push(registerAgent(session, agent, { agentsByInstance, instanceByServer }));
|
|
74
|
+
}
|
|
75
|
+
for (const [serverName, instanceId] of instanceByServer) {
|
|
76
|
+
if (activeServers.has(serverName)) continue;
|
|
77
|
+
const previous = agentsByInstance.get(instanceId);
|
|
78
|
+
instanceByServer.delete(serverName);
|
|
79
|
+
agentsByInstance.delete(instanceId);
|
|
80
|
+
if (previous) dispatchRegistryEvent(session, 'agent.unregistered', {
|
|
81
|
+
agentInstanceId: instanceId,
|
|
82
|
+
serverName,
|
|
83
|
+
});
|
|
84
|
+
}
|
|
85
|
+
session.agentRegistry = this;
|
|
86
|
+
session.agentRegistrySnapshot = this.snapshot();
|
|
87
|
+
return discovered;
|
|
88
|
+
},
|
|
89
|
+
snapshot() {
|
|
90
|
+
return [...agentsByInstance.values()]
|
|
91
|
+
.map((agent) => cloneAgent(agent))
|
|
92
|
+
.sort((a, b) => a.agentInstanceId.localeCompare(b.agentInstanceId));
|
|
93
|
+
},
|
|
94
|
+
get(agentInstanceId) {
|
|
95
|
+
const agent = agentsByInstance.get(String(agentInstanceId));
|
|
96
|
+
return agent ? cloneAgent(agent) : null;
|
|
97
|
+
},
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
async function discoverServerAgent(session, serverName, endpoint = {}, { callTool, signal, now }) {
|
|
102
|
+
const lastSeenAt = now().toISOString();
|
|
103
|
+
if (endpoint.status !== 'connected') {
|
|
104
|
+
return legacyAgent(serverName, endpoint, { health: UNAVAILABLE, lastSeenAt });
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const tool = findAgentDescribeTool(serverName, endpoint.tools ?? []);
|
|
108
|
+
if (!tool) {
|
|
109
|
+
return legacyAgent(serverName, endpoint, { health: AVAILABLE, lastSeenAt });
|
|
110
|
+
}
|
|
111
|
+
const toolName = tool.name;
|
|
112
|
+
|
|
113
|
+
try {
|
|
114
|
+
const result = await callTool(
|
|
115
|
+
session.mcp,
|
|
116
|
+
serverName,
|
|
117
|
+
toolName,
|
|
118
|
+
describeArguments(tool, session?.workspace),
|
|
119
|
+
signal,
|
|
120
|
+
);
|
|
121
|
+
const description = assertContract('agentDescription', parseToolJsonResult(result));
|
|
122
|
+
return {
|
|
123
|
+
serverName,
|
|
124
|
+
toolName,
|
|
125
|
+
agentInstanceId: description.agentInstanceId,
|
|
126
|
+
description,
|
|
127
|
+
health: description.health?.status ?? AVAILABLE,
|
|
128
|
+
firstSeenAt: lastSeenAt,
|
|
129
|
+
lastSeenAt,
|
|
130
|
+
legacy: false,
|
|
131
|
+
orchestrable: true,
|
|
132
|
+
};
|
|
133
|
+
} catch (error) {
|
|
134
|
+
return legacyAgent(serverName, endpoint, {
|
|
135
|
+
health: UNAVAILABLE,
|
|
136
|
+
lastSeenAt,
|
|
137
|
+
toolName,
|
|
138
|
+
error: error instanceof Error ? error.message : String(error),
|
|
139
|
+
});
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
function registerAgent(session, agent, { agentsByInstance, instanceByServer }) {
|
|
144
|
+
const previousInstanceId = instanceByServer.get(agent.serverName);
|
|
145
|
+
const previous = previousInstanceId ? agentsByInstance.get(previousInstanceId) : null;
|
|
146
|
+
|
|
147
|
+
/*
|
|
148
|
+
A failed discovery must not erase a known orchestrator agent.
|
|
149
|
+
|
|
150
|
+
When the endpoint is transiently unreachable — the runtime boots before its
|
|
151
|
+
containers (the normal boot order), or a single probe times out —
|
|
152
|
+
`discoverServerAgent` falls back to a legacy agent with no capabilities. The
|
|
153
|
+
old code replaced the orchestrator agent with that fallback, so every
|
|
154
|
+
capability silently vanished from the registry and did not come back until a
|
|
155
|
+
LATER successful discovery. Keep the orchestrator agent and only refresh its
|
|
156
|
+
probe timestamp: its capabilities are still real, only the endpoint is down.
|
|
157
|
+
*/
|
|
158
|
+
if (agent.legacy && previous && !previous.legacy) {
|
|
159
|
+
agentsByInstance.set(previous.agentInstanceId, { ...previous, lastSeenAt: agent.lastSeenAt });
|
|
160
|
+
/*
|
|
161
|
+
Preserving is right; preserving in silence is what caused the hunt.
|
|
162
|
+
|
|
163
|
+
Every defect this registry produced was invisible: capabilities vanished
|
|
164
|
+
without an event, and the resolver could only report the consequence ("no
|
|
165
|
+
agent provides X") long afterwards. Keeping the agent fixes the loss, not
|
|
166
|
+
the blindness — a probe that failed is a fact worth stating, once, where
|
|
167
|
+
the panels and the shell already read.
|
|
168
|
+
|
|
169
|
+
Deliberately NOT a health change: the endpoint is down but the agent stays
|
|
170
|
+
usable by design here, and moving `health` would make `capabilityResolver`
|
|
171
|
+
refuse it — trading a silent loss for a silent refusal.
|
|
172
|
+
*/
|
|
173
|
+
dispatchRuntimeLog(session, `agent-registry: ${agent.serverName} did not answer agent_describe`
|
|
174
|
+
+ `${agent.error ? ` (${agent.error})` : ''}; keeping its known capabilities`
|
|
175
|
+
+ ` (${(previous.description?.capabilities ?? []).map((capability) => capability.id).join(', ') || 'none'}).`);
|
|
176
|
+
return cloneAgent(previous);
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
const firstSeenAt = previous?.firstSeenAt ?? agent.firstSeenAt ?? agent.lastSeenAt;
|
|
180
|
+
const next = {
|
|
181
|
+
...agent,
|
|
182
|
+
firstSeenAt,
|
|
183
|
+
};
|
|
184
|
+
|
|
185
|
+
if (previous && previous.agentInstanceId !== next.agentInstanceId) {
|
|
186
|
+
agentsByInstance.delete(previous.agentInstanceId);
|
|
187
|
+
}
|
|
188
|
+
agentsByInstance.set(next.agentInstanceId, next);
|
|
189
|
+
instanceByServer.set(next.serverName, next.agentInstanceId);
|
|
190
|
+
|
|
191
|
+
if (!previous || previous.agentInstanceId !== next.agentInstanceId) {
|
|
192
|
+
dispatchRegistryEvent(session, 'agent.registered', { agent: next });
|
|
193
|
+
} else if (previous.health !== next.health) {
|
|
194
|
+
dispatchRegistryEvent(session, 'agent.health_changed', {
|
|
195
|
+
agent: next,
|
|
196
|
+
agentInstanceId: next.agentInstanceId,
|
|
197
|
+
previousHealth: previous.health,
|
|
198
|
+
health: next.health,
|
|
199
|
+
});
|
|
200
|
+
}
|
|
201
|
+
return cloneAgent(next);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Runtime log line, emitted without importing the supervisor.
|
|
206
|
+
*
|
|
207
|
+
* `emitRuntimeLog` lives in `runtime/supervisor.js`, which already imports THIS
|
|
208
|
+
* module: importing it back would close a cycle for one log line. The event
|
|
209
|
+
* shape is the contract, not the helper, so we build it from the same
|
|
210
|
+
* normalizer the supervisor uses.
|
|
211
|
+
*/
|
|
212
|
+
function dispatchRuntimeLog(session, message) {
|
|
213
|
+
if (!session) return;
|
|
214
|
+
const payload = normalizeRuntimeLog(message, { session });
|
|
215
|
+
dispatchAgentEvent(session, createAgentEvent('runtime_log', {
|
|
216
|
+
origin: 'runtime',
|
|
217
|
+
runId: payload.runId ?? null,
|
|
218
|
+
taskId: payload.taskId ?? null,
|
|
219
|
+
workspace: payload.workspaceId ?? null,
|
|
220
|
+
payload,
|
|
221
|
+
}));
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
function dispatchRegistryEvent(session, type, payload) {
|
|
225
|
+
if (!session) return;
|
|
226
|
+
dispatchAgentEvent(session, createAgentEvent(type, {
|
|
227
|
+
origin: 'agent_registry',
|
|
228
|
+
workspace: session.workspace ?? null,
|
|
229
|
+
payload,
|
|
230
|
+
}));
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
function findAgentDescribeTool(serverName, tools) {
|
|
234
|
+
const named = tools.filter((tool) => String(tool?.name ?? ''));
|
|
235
|
+
const byName = (predicate) => named.find((tool) => predicate(String(tool.name)));
|
|
236
|
+
return byName((name) => name === 'agent_describe')
|
|
237
|
+
?? byName((name) => name === `${serverName}__agent_describe`)
|
|
238
|
+
?? byName((name) => name.endsWith('__agent_describe'))
|
|
239
|
+
?? null;
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
// Parts of a contract are workspace-scoped — typically the closed vocabulary of
|
|
243
|
+
// an argument (the sources declared in THIS workspace). Published as a bare
|
|
244
|
+
// string, such a field is unverifiable and a planner fills it with any noun
|
|
245
|
+
// from the objective, so it is worth telling the agent which workspace we are
|
|
246
|
+
// asking about.
|
|
247
|
+
//
|
|
248
|
+
// But the orchestrator must not assume an agent accepts an argument it never
|
|
249
|
+
// declared: an agent whose agent_describe schema is `additionalProperties:
|
|
250
|
+
// false` REJECTS the call, drops out of the registry, and its capabilities
|
|
251
|
+
// silently vanish — the objective then resolves to whatever agent is left.
|
|
252
|
+
// Send the workspace only to agents whose own schema says they can take it.
|
|
253
|
+
function describeArguments(tool, workspace) {
|
|
254
|
+
if (!workspace) return {};
|
|
255
|
+
const schema = tool?.inputSchema;
|
|
256
|
+
if (!schema || typeof schema !== 'object') return {};
|
|
257
|
+
const declaresWorkspace = Object.hasOwn(schema.properties ?? {}, 'workspace');
|
|
258
|
+
const acceptsExtra = schema.additionalProperties !== false;
|
|
259
|
+
return declaresWorkspace || acceptsExtra ? { workspace: String(workspace) } : {};
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
function parseToolJsonResult(result) {
|
|
263
|
+
if (result && typeof result === 'object' && !Array.isArray(result) && !Array.isArray(result.content)) {
|
|
264
|
+
return result;
|
|
265
|
+
}
|
|
266
|
+
const text = formatMcpToolResult(result);
|
|
267
|
+
return JSON.parse(text);
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
function legacyAgent(serverName, endpoint = {}, { health, lastSeenAt, toolName = null, error = null }) {
|
|
271
|
+
const displayName = endpoint.displayName ?? serverName;
|
|
272
|
+
const description = {
|
|
273
|
+
contractVersion: 'legacy',
|
|
274
|
+
agentType: serverName,
|
|
275
|
+
agentInstanceId: `${serverName}-legacy`,
|
|
276
|
+
displayName,
|
|
277
|
+
capabilities: [],
|
|
278
|
+
orchestration: {
|
|
279
|
+
canPlan: false,
|
|
280
|
+
canExpandPlan: false,
|
|
281
|
+
canExecute: false,
|
|
282
|
+
canCancel: false,
|
|
283
|
+
canResume: false,
|
|
284
|
+
supportsIdempotency: false,
|
|
285
|
+
supportsParallelWorkers: false,
|
|
286
|
+
},
|
|
287
|
+
limits: {
|
|
288
|
+
recommendedConcurrency: 0,
|
|
289
|
+
maxConcurrency: 0,
|
|
290
|
+
},
|
|
291
|
+
health: { status: health },
|
|
292
|
+
};
|
|
293
|
+
return {
|
|
294
|
+
serverName,
|
|
295
|
+
toolName,
|
|
296
|
+
agentInstanceId: description.agentInstanceId,
|
|
297
|
+
description,
|
|
298
|
+
health,
|
|
299
|
+
firstSeenAt: lastSeenAt,
|
|
300
|
+
lastSeenAt,
|
|
301
|
+
legacy: true,
|
|
302
|
+
orchestrable: false,
|
|
303
|
+
error,
|
|
304
|
+
};
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
function cloneAgent(agent) {
|
|
308
|
+
return {
|
|
309
|
+
...agent,
|
|
310
|
+
description: cloneJson(agent.description),
|
|
311
|
+
};
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
function cloneJson(value) {
|
|
315
|
+
return value == null ? value : JSON.parse(JSON.stringify(value));
|
|
316
|
+
}
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
2
2
|
import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
|
|
3
|
+
import { normalizeRuntimeLog } from '../core/runtimeLog.js';
|
|
3
4
|
import { assertContract } from '../contracts/schemas.js';
|
|
4
5
|
|
|
5
6
|
const AVAILABLE = 'available';
|
|
@@ -142,6 +143,39 @@ async function discoverServerAgent(session, serverName, endpoint = {}, { callToo
|
|
|
142
143
|
function registerAgent(session, agent, { agentsByInstance, instanceByServer }) {
|
|
143
144
|
const previousInstanceId = instanceByServer.get(agent.serverName);
|
|
144
145
|
const previous = previousInstanceId ? agentsByInstance.get(previousInstanceId) : null;
|
|
146
|
+
|
|
147
|
+
/*
|
|
148
|
+
A failed discovery must not erase a known orchestrator agent.
|
|
149
|
+
|
|
150
|
+
When the endpoint is transiently unreachable — the runtime boots before its
|
|
151
|
+
containers (the normal boot order), or a single probe times out —
|
|
152
|
+
`discoverServerAgent` falls back to a legacy agent with no capabilities. The
|
|
153
|
+
old code replaced the orchestrator agent with that fallback, so every
|
|
154
|
+
capability silently vanished from the registry and did not come back until a
|
|
155
|
+
LATER successful discovery. Keep the orchestrator agent and only refresh its
|
|
156
|
+
probe timestamp: its capabilities are still real, only the endpoint is down.
|
|
157
|
+
*/
|
|
158
|
+
if (agent.legacy && previous && !previous.legacy) {
|
|
159
|
+
agentsByInstance.set(previous.agentInstanceId, { ...previous, lastSeenAt: agent.lastSeenAt });
|
|
160
|
+
/*
|
|
161
|
+
Preserving is right; preserving in silence is what caused the hunt.
|
|
162
|
+
|
|
163
|
+
Every defect this registry produced was invisible: capabilities vanished
|
|
164
|
+
without an event, and the resolver could only report the consequence ("no
|
|
165
|
+
agent provides X") long afterwards. Keeping the agent fixes the loss, not
|
|
166
|
+
the blindness — a probe that failed is a fact worth stating, once, where
|
|
167
|
+
the panels and the shell already read.
|
|
168
|
+
|
|
169
|
+
Deliberately NOT a health change: the endpoint is down but the agent stays
|
|
170
|
+
usable by design here, and moving `health` would make `capabilityResolver`
|
|
171
|
+
refuse it — trading a silent loss for a silent refusal.
|
|
172
|
+
*/
|
|
173
|
+
dispatchRuntimeLog(session, `agent-registry: ${agent.serverName} did not answer agent_describe`
|
|
174
|
+
+ `${agent.error ? ` (${agent.error})` : ''}; keeping its known capabilities`
|
|
175
|
+
+ ` (${(previous.description?.capabilities ?? []).map((capability) => capability.id).join(', ') || 'none'}).`);
|
|
176
|
+
return cloneAgent(previous);
|
|
177
|
+
}
|
|
178
|
+
|
|
145
179
|
const firstSeenAt = previous?.firstSeenAt ?? agent.firstSeenAt ?? agent.lastSeenAt;
|
|
146
180
|
const next = {
|
|
147
181
|
...agent,
|
|
@@ -167,6 +201,26 @@ function registerAgent(session, agent, { agentsByInstance, instanceByServer }) {
|
|
|
167
201
|
return cloneAgent(next);
|
|
168
202
|
}
|
|
169
203
|
|
|
204
|
+
/**
|
|
205
|
+
* Runtime log line, emitted without importing the supervisor.
|
|
206
|
+
*
|
|
207
|
+
* `emitRuntimeLog` lives in `runtime/supervisor.js`, which already imports THIS
|
|
208
|
+
* module: importing it back would close a cycle for one log line. The event
|
|
209
|
+
* shape is the contract, not the helper, so we build it from the same
|
|
210
|
+
* normalizer the supervisor uses.
|
|
211
|
+
*/
|
|
212
|
+
function dispatchRuntimeLog(session, message) {
|
|
213
|
+
if (!session) return;
|
|
214
|
+
const payload = normalizeRuntimeLog(message, { session });
|
|
215
|
+
dispatchAgentEvent(session, createAgentEvent('runtime_log', {
|
|
216
|
+
origin: 'runtime',
|
|
217
|
+
runId: payload.runId ?? null,
|
|
218
|
+
taskId: payload.taskId ?? null,
|
|
219
|
+
workspace: payload.workspaceId ?? null,
|
|
220
|
+
payload,
|
|
221
|
+
}));
|
|
222
|
+
}
|
|
223
|
+
|
|
170
224
|
function dispatchRegistryEvent(session, type, payload) {
|
|
171
225
|
if (!session) return;
|
|
172
226
|
dispatchAgentEvent(session, createAgentEvent(type, {
|
|
@@ -134,6 +134,81 @@ test('agentRegistry marks unavailable boot agents and emits health changes on re
|
|
|
134
134
|
assert.equal(registry.snapshot()[0].health, 'available');
|
|
135
135
|
});
|
|
136
136
|
|
|
137
|
+
test('a failed re-discovery keeps the orchestrator agent, never erases its capabilities', async () => {
|
|
138
|
+
/*
|
|
139
|
+
Boot order: the runtime starts before its containers. The first scan sees the
|
|
140
|
+
production endpoint "unavailable" and would register a legacy placeholder —
|
|
141
|
+
and the old code replaced the orchestrator agent with it, silently dropping
|
|
142
|
+
every capability until a LATER successful discovery. The agent must survive a
|
|
143
|
+
transient probe failure, because its capabilities did not change.
|
|
144
|
+
*/
|
|
145
|
+
const events = [];
|
|
146
|
+
const session = {
|
|
147
|
+
workspace: 'acpi',
|
|
148
|
+
mcp: {
|
|
149
|
+
production: { status: 'connected', tools: [{ name: 'agent_describe' }] },
|
|
150
|
+
},
|
|
151
|
+
_onAgentEvent: (event) => events.push(event),
|
|
152
|
+
};
|
|
153
|
+
const registry = createAgentRegistry({
|
|
154
|
+
callTool: async () => ({ content: [{ type: 'text', text: JSON.stringify(description()) }] }),
|
|
155
|
+
});
|
|
156
|
+
|
|
157
|
+
await registry.discover(session);
|
|
158
|
+
assert.equal(registry.snapshot()[0].agentInstanceId, 'production-main');
|
|
159
|
+
|
|
160
|
+
// The endpoint goes down: discovery now falls back to a legacy placeholder.
|
|
161
|
+
session.mcp.production.status = 'unavailable';
|
|
162
|
+
await registry.discover(session);
|
|
163
|
+
|
|
164
|
+
const [agent] = registry.snapshot();
|
|
165
|
+
assert.equal(agent.agentInstanceId, 'production-main');
|
|
166
|
+
assert.equal(agent.legacy, false);
|
|
167
|
+
assert.equal(agent.description.capabilities.length, 1);
|
|
168
|
+
// And it is still routable: the capability was not erased.
|
|
169
|
+
const capability = createCapabilityRegistry({ agents: registry.snapshot() });
|
|
170
|
+
assert.equal(capability.providersFor('knowledge.update').length, 1);
|
|
171
|
+
|
|
172
|
+
/*
|
|
173
|
+
Preserving must not be silent.
|
|
174
|
+
|
|
175
|
+
Every defect this registry produced was invisible, and that is what turned a
|
|
176
|
+
boot-order race into a debugging session: the capability vanished with no
|
|
177
|
+
event, and the only report came much later, from the resolver, as "no agent
|
|
178
|
+
provides X". A probe that failed is a fact, and it belongs where the shell
|
|
179
|
+
and the panels already read.
|
|
180
|
+
*/
|
|
181
|
+
const kept = events.find((event) => event.type === 'runtime_log'
|
|
182
|
+
&& String(event.payload?.message ?? '').includes('agent-registry:'));
|
|
183
|
+
assert.ok(kept, 'a preserved agent must leave a runtime log');
|
|
184
|
+
assert.match(String(kept.payload.message), /did not answer agent_describe/);
|
|
185
|
+
assert.match(String(kept.payload.message), /knowledge\.update/);
|
|
186
|
+
});
|
|
187
|
+
|
|
188
|
+
test('a failed re-discovery keeps a degraded orchestrator agent too', async () => {
|
|
189
|
+
// Same protection, but when the agent was discovered healthy then the probe
|
|
190
|
+
// throws (endpoint "connected" but agent_describe fails mid-flight).
|
|
191
|
+
const session = {
|
|
192
|
+
mcp: { production: { status: 'connected', tools: [{ name: 'agent_describe' }] } },
|
|
193
|
+
};
|
|
194
|
+
let fail = false;
|
|
195
|
+
const registry = createAgentRegistry({
|
|
196
|
+
callTool: async () => {
|
|
197
|
+
if (fail) throw new Error('agent_describe timeout');
|
|
198
|
+
return { content: [{ type: 'text', text: JSON.stringify(description()) }] };
|
|
199
|
+
},
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
await registry.discover(session);
|
|
203
|
+
fail = true;
|
|
204
|
+
await registry.discover(session);
|
|
205
|
+
|
|
206
|
+
const [agent] = registry.snapshot();
|
|
207
|
+
assert.equal(agent.agentInstanceId, 'production-main');
|
|
208
|
+
assert.equal(agent.legacy, false);
|
|
209
|
+
assert.equal(agent.description.capabilities.length, 1);
|
|
210
|
+
});
|
|
211
|
+
|
|
137
212
|
test('discovery sends the workspace only to agents whose schema declares it', async () => {
|
|
138
213
|
const seen = {};
|
|
139
214
|
const registry = createAgentRegistry({
|
package/src/runtime/client.js
CHANGED
|
@@ -261,6 +261,7 @@ export async function postRuntimeApprove({
|
|
|
261
261
|
scope = null,
|
|
262
262
|
planRevision = null,
|
|
263
263
|
approvalClasses = null,
|
|
264
|
+
caller = null,
|
|
264
265
|
} = {}) {
|
|
265
266
|
const endpoint = runtimeEndpoint(url, '/approve', workspace);
|
|
266
267
|
const parsed = new URL(endpoint);
|
|
@@ -278,6 +279,7 @@ export async function postRuntimeApprove({
|
|
|
278
279
|
scope,
|
|
279
280
|
planRevision,
|
|
280
281
|
approvalClasses,
|
|
282
|
+
caller,
|
|
281
283
|
}),
|
|
282
284
|
});
|
|
283
285
|
if (!response.ok) throw new Error(`Runtime approve failed: HTTP ${response.status}`);
|
|
@@ -94,7 +94,7 @@ export function integratePreparedDelegation({
|
|
|
94
94
|
session,
|
|
95
95
|
approval.approved
|
|
96
96
|
? `approval: run ${runId} auto-approved (autoApprove opt-in)`
|
|
97
|
-
: `approval: run ${runId} awaiting explicit approval before mutations (/approve
|
|
97
|
+
: `approval: run ${runId} awaiting explicit approval before mutations (/approve)`,
|
|
98
98
|
);
|
|
99
99
|
return { integrated, approval };
|
|
100
100
|
}
|
package/src/runtime/runner.js
CHANGED
|
@@ -407,8 +407,8 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
407
407
|
sanitizeSessionPlanForExecution(session, runId);
|
|
408
408
|
ensurePlanProjection(session, runId);
|
|
409
409
|
emitRuntimeLog(session, `scheduler: parallel plan enabled (concurrency ${limit}; agent=${concurrencyDetail.agentLimit ?? 'n/a'}, ceiling=${concurrencyDetail.ceiling ?? 'none'}${concurrencyDetail.cappedByCeiling ? ' → capped by manager ceiling' : ''})`);
|
|
410
|
-
// Interactive approvals do NOT expire: the user has /approve,
|
|
411
|
-
//
|
|
410
|
+
// Interactive approvals do NOT expire: the user has /approve, /cancel and
|
|
411
|
+
// /run kill — an arbitrary timer only created mystery
|
|
412
412
|
// failures. A deadline exists only when explicitly configured (headless
|
|
413
413
|
// runs, CI) via the session or the env escape hatch.
|
|
414
414
|
const configuredApprovalWait = Number(session._approvalTimeoutMs) > 0
|
|
@@ -572,7 +572,7 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
572
572
|
`⏸ Approbation requise avant exécution : ${newlyRequested.length} tâche(s) mutante(s) en attente.`,
|
|
573
573
|
...newlyRequested.slice(0, 5).map((step) => ` - ${step.description ?? step.id}`),
|
|
574
574
|
newlyRequested.length > 5 ? ` … et ${newlyRequested.length - 5} autre(s).` : null,
|
|
575
|
-
'
|
|
575
|
+
'Tape /approve (ou clique sur « Approuver ») pour lancer, « annule » pour abandonner.',
|
|
576
576
|
].filter(Boolean).join('\n'),
|
|
577
577
|
},
|
|
578
578
|
}));
|
package/src/runtime/server.js
CHANGED
|
@@ -642,6 +642,7 @@ export function startRuntimeServer({
|
|
|
642
642
|
groupId: url.searchParams.get('groupId') ?? body.groupId ?? null,
|
|
643
643
|
planRevision: readOptionalNumber(url.searchParams.get('planRevision') ?? body.planRevision),
|
|
644
644
|
approvalClasses: readOptionalList(body.approvalClasses ?? body.approvalClass ?? url.searchParams.get('approvalClass')),
|
|
645
|
+
caller: body.caller ?? url.searchParams.get('caller') ?? null,
|
|
645
646
|
});
|
|
646
647
|
sendJson(response, result?.approved ? 202 : 404, result ?? { approved: false });
|
|
647
648
|
return;
|
|
@@ -805,7 +806,10 @@ export function startRuntimeServer({
|
|
|
805
806
|
runPromise
|
|
806
807
|
.catch((err) => {
|
|
807
808
|
rejectReady?.(err);
|
|
808
|
-
|
|
809
|
+
// The runId travels with the failure: without it `finishControlByRun`
|
|
810
|
+
// had nothing to match, so a queued restore stayed "pending" forever
|
|
811
|
+
// while the run that carried it was already dead.
|
|
812
|
+
context.session?._onRuntimeError?.(err, runId);
|
|
809
813
|
})
|
|
810
814
|
.finally(() => {
|
|
811
815
|
context.running = false;
|
|
@@ -1371,9 +1375,6 @@ function classifyControlMessage(input, status, forcedIntent = null) {
|
|
|
1371
1375
|
if (explicit) {
|
|
1372
1376
|
return { kind: explicit, confidence: 1, reason: 'explicit_intent' };
|
|
1373
1377
|
}
|
|
1374
|
-
if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
|
|
1375
|
-
return { kind: 'approve', confidence: 0.86, reason: 'approval_request' };
|
|
1376
|
-
}
|
|
1377
1378
|
if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort)\b/i.test(lower)) {
|
|
1378
1379
|
return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request' };
|
|
1379
1380
|
}
|
|
@@ -1082,6 +1082,7 @@ test('runtime server exposes approval endpoint', async (t) => {
|
|
|
1082
1082
|
groupId: null,
|
|
1083
1083
|
planRevision: null,
|
|
1084
1084
|
approvalClasses: [],
|
|
1085
|
+
caller: null,
|
|
1085
1086
|
});
|
|
1086
1087
|
assert.deepEqual(await response.json(), { approved: true, runId: 'run-1', itemId: 'item-1' });
|
|
1087
1088
|
} finally {
|
|
@@ -1307,11 +1308,21 @@ test('runtime server control message handles approve and cancel intents during a
|
|
|
1307
1308
|
const approveResponse = await fetch(`http://127.0.0.1:${handle.port}/control?workspace=acme`, {
|
|
1308
1309
|
method: 'POST',
|
|
1309
1310
|
headers: { 'Content-Type': 'application/json' },
|
|
1310
|
-
body: JSON.stringify({ action: 'message', input: 'valide tout' }),
|
|
1311
|
+
body: JSON.stringify({ action: 'message', input: 'valide tout', intent: 'approve' }),
|
|
1311
1312
|
});
|
|
1312
1313
|
assert.equal(approveResponse.status, 200);
|
|
1313
1314
|
assert.equal((await approveResponse.json()).kind, 'approve');
|
|
1314
1315
|
|
|
1316
|
+
// Free-text approval phrasing is NOT honored anymore: approval is an
|
|
1317
|
+
// explicit /approve (or the approval button), never a keyword match.
|
|
1318
|
+
const freeTextResponse = await fetch(`http://127.0.0.1:${handle.port}/control?workspace=acme`, {
|
|
1319
|
+
method: 'POST',
|
|
1320
|
+
headers: { 'Content-Type': 'application/json' },
|
|
1321
|
+
body: JSON.stringify({ action: 'message', input: 'valide tout' }),
|
|
1322
|
+
});
|
|
1323
|
+
assert.equal(freeTextResponse.status, 200);
|
|
1324
|
+
assert.notEqual((await freeTextResponse.json()).kind, 'approve');
|
|
1325
|
+
|
|
1315
1326
|
const cancelResponse = await fetch(`http://127.0.0.1:${handle.port}/control?workspace=acme`, {
|
|
1316
1327
|
method: 'POST',
|
|
1317
1328
|
headers: { 'Content-Type': 'application/json' },
|
|
@@ -13,6 +13,7 @@ export function startActivitySupervisor(session, {
|
|
|
13
13
|
agentRegistryIntervalMs = registryIntervalFromEnv(),
|
|
14
14
|
agentRegistry = null,
|
|
15
15
|
callTool = callMcpTool,
|
|
16
|
+
refreshMcp = null,
|
|
16
17
|
} = {}) {
|
|
17
18
|
const pollBusy = new Set();
|
|
18
19
|
let stopped = false;
|
|
@@ -32,8 +33,24 @@ export function startActivitySupervisor(session, {
|
|
|
32
33
|
}
|
|
33
34
|
}, queueIntervalMs);
|
|
34
35
|
|
|
36
|
+
/*
|
|
37
|
+
The re-scan must probe the endpoints again, not trust a stale status.
|
|
38
|
+
|
|
39
|
+
The first discovery ran while the containers were still coming up, so the
|
|
40
|
+
production endpoint was marked "not connected" and its agent fell back to a
|
|
41
|
+
legacy placeholder. Re-scanning against that cached status kept the fallback
|
|
42
|
+
forever. `refreshMcp` re-resolves the endpoint states (and their tools)
|
|
43
|
+
before each discovery, so a container that came up is actually discovered.
|
|
44
|
+
*/
|
|
35
45
|
const agentRegistryTimer = agentRegistryIntervalMs > 0
|
|
36
|
-
? setInterval(() => {
|
|
46
|
+
? setInterval(async () => {
|
|
47
|
+
if (refreshMcp) {
|
|
48
|
+
try {
|
|
49
|
+
await refreshMcp(session);
|
|
50
|
+
} catch {
|
|
51
|
+
/* the discovery below still runs on the cached status */
|
|
52
|
+
}
|
|
53
|
+
}
|
|
37
54
|
void discoverAgentsOnce(session, { registry, signal: runSignal });
|
|
38
55
|
}, agentRegistryIntervalMs)
|
|
39
56
|
: null;
|
|
@@ -246,6 +246,49 @@ test('startActivitySupervisor periodically re-scans the agent registry', async (
|
|
|
246
246
|
}
|
|
247
247
|
});
|
|
248
248
|
|
|
249
|
+
test('startActivitySupervisor refreshes MCP before each periodic re-scan', async () => {
|
|
250
|
+
/*
|
|
251
|
+
Without the refresh, the re-scan re-reads a cached "not connected" endpoint
|
|
252
|
+
status and keeps a legacy placeholder forever, so an agent whose container
|
|
253
|
+
came up after boot is never discovered. The refresh probes the endpoints
|
|
254
|
+
again before the scan.
|
|
255
|
+
*/
|
|
256
|
+
const session = {
|
|
257
|
+
mcp: {},
|
|
258
|
+
activities: {},
|
|
259
|
+
headlessPlan: null,
|
|
260
|
+
jobQueue: [],
|
|
261
|
+
};
|
|
262
|
+
let refreshes = 0;
|
|
263
|
+
let discoveries = 0;
|
|
264
|
+
const registry = {
|
|
265
|
+
async discover() {
|
|
266
|
+
discoveries += 1;
|
|
267
|
+
return [];
|
|
268
|
+
},
|
|
269
|
+
snapshot() {
|
|
270
|
+
return [];
|
|
271
|
+
},
|
|
272
|
+
};
|
|
273
|
+
|
|
274
|
+
const supervisor = startActivitySupervisor(session, {
|
|
275
|
+
intervalMs: 1000,
|
|
276
|
+
queueIntervalMs: 1000,
|
|
277
|
+
agentRegistryIntervalMs: 10,
|
|
278
|
+
agentRegistry: registry,
|
|
279
|
+
refreshMcp: async () => {
|
|
280
|
+
refreshes += 1;
|
|
281
|
+
},
|
|
282
|
+
});
|
|
283
|
+
|
|
284
|
+
try {
|
|
285
|
+
await waitFor(() => refreshes >= 2);
|
|
286
|
+
assert.ok(discoveries >= 2);
|
|
287
|
+
} finally {
|
|
288
|
+
supervisor.stop();
|
|
289
|
+
}
|
|
290
|
+
});
|
|
291
|
+
|
|
249
292
|
test('discoverAgentsOnce uses the session registry and returns discovered agents', async () => {
|
|
250
293
|
const expected = [{ agentInstanceId: 'a' }];
|
|
251
294
|
const registry = {
|