@dotdrelle/wiki-manager 0.15.42 → 0.15.48
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +139 -27
- package/mcp.endpoints.example.json +1 -1
- package/package.json +2 -2
- package/src/agent/graph.js +290 -29
- package/src/agent/graph.test.js +551 -1
- package/src/agent/skillRecursion.test.js +98 -0
- package/src/cli/wiki-manager.js +209 -7
- package/src/cli/wiki-manager.test.js +89 -0
- package/src/commands/slash.js +28 -10
- package/src/contracts/schemas.js +1 -1
- package/src/core/agentEvents.js +50 -0
- package/src/core/agentEvents.test.js +52 -0
- package/src/core/buildInfo.json +2 -2
- package/src/core/env.js +20 -1
- package/src/core/env.test.js +34 -0
- package/src/core/mcp.js +1 -1
- package/src/core/profile.js +19 -0
- package/src/core/runtimeLog.js +15 -0
- package/src/core/runtimeLog.test.js +15 -1
- package/src/core/skillChainView.js +84 -0
- package/src/core/skillChainView.test.js +50 -0
- package/src/core/skillCompiler.js +135 -0
- package/src/core/skillCompiler.test.js +91 -0
- package/src/core/skillInvocation.js +79 -0
- package/src/core/skillInvocation.test.js +73 -0
- package/src/core/skills.js +81 -19
- package/src/core/wikiWorkspace.test.js +34 -0
- package/src/core/workspaceProfile.test.js +55 -0
- package/src/runtime/client.js +45 -4
- package/src/runtime/controlCancellation.js +33 -0
- package/src/runtime/controlCancellation.test.js +49 -0
- package/src/runtime/controlDrain.js +50 -0
- package/src/runtime/controlDrain.test.js +38 -0
- package/src/runtime/server.js +341 -20
- package/src/runtime/server.test.js +344 -2
- package/src/runtime/skillChain.e2e.test.js +394 -0
- package/src/runtime/skillRun.js +104 -0
- package/src/runtime/skillRun.test.js +84 -0
- package/src/runtime/store.js +69 -0
- package/src/runtime/store.test.js +11 -0
- package/src/runtime/workspaceIsolation.test.js +178 -0
- package/src/shell/RightPane.tsx +3 -2
- package/src/shell/repl.js +51 -6
- package/src/shell/repl.test.js +41 -0
- package/src/shell/useSession.ts +43 -9
- package/wiki-workspace +137 -1
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { mkdtempSync, mkdirSync, writeFileSync, readFileSync, copyFileSync, existsSync } from 'node:fs';
|
|
4
|
+
import { tmpdir } from 'node:os';
|
|
5
|
+
import { join, resolve } from 'node:path';
|
|
6
|
+
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
7
|
+
import { RESERVED_SLASH_COMMANDS } from '../core/skillInvocation.js';
|
|
8
|
+
import { startRuntimeServer as startRuntimeServerImpl } from './server.js';
|
|
9
|
+
import { handleRuntimeControlTool } from '../agent/graph.js';
|
|
10
|
+
|
|
11
|
+
// Plan V4.1 §58 — E2E-001..005.
|
|
12
|
+
//
|
|
13
|
+
// These drive the real HTTP surface with a stub `run` so the assertions are on
|
|
14
|
+
// what the runtime actually did: how many runs were started, in which order,
|
|
15
|
+
// under which chain, and what survived a cancel or a kill. The stub is the only
|
|
16
|
+
// fake — routing, compilation, drain and cancellation are the production code.
|
|
17
|
+
|
|
18
|
+
const SCAFFOLD_SKILLS = resolve('../llm-wiki/scaffold/workspace/.wiki/skills');
|
|
19
|
+
|
|
20
|
+
function startRuntimeServer(options) {
|
|
21
|
+
return startRuntimeServerImpl({ token: '', ...options });
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function workspaceWithSkills(files) {
|
|
25
|
+
const root = mkdtempSync(join(tmpdir(), 'skill-chain-e2e-'));
|
|
26
|
+
const skillDir = join(root, '.wiki', 'skills');
|
|
27
|
+
mkdirSync(skillDir, { recursive: true });
|
|
28
|
+
for (const [name, body] of Object.entries(files)) {
|
|
29
|
+
if (body === null) {
|
|
30
|
+
// Use the real shipped skill: the guard is worthless against a body
|
|
31
|
+
// rewritten for the test.
|
|
32
|
+
copyFileSync(join(SCAFFOLD_SKILLS, `${name}.md`), join(skillDir, `${name}.md`));
|
|
33
|
+
} else {
|
|
34
|
+
writeFileSync(join(skillDir, `${name}.md`), body);
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
return root;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// Starts the server with a run stub that records every started run and finishes
|
|
41
|
+
// it on demand, so a chain can be observed step by step.
|
|
42
|
+
async function harness(t, { skills, autoFinish = true, onRun = null } = {}) {
|
|
43
|
+
if (!existsSync(SCAFFOLD_SKILLS)) {
|
|
44
|
+
t.skip('llm-wiki is not checked out next to llm-wiki-manager');
|
|
45
|
+
return null;
|
|
46
|
+
}
|
|
47
|
+
const root = workspaceWithSkills(skills);
|
|
48
|
+
const session = { workspace: 'acme', workspacePath: root, controlQueue: [] };
|
|
49
|
+
const context = { workspace: 'acme', session, running: false, currentAbortController: null };
|
|
50
|
+
const runs = [];
|
|
51
|
+
const pending = [];
|
|
52
|
+
let handle;
|
|
53
|
+
try {
|
|
54
|
+
handle = await startRuntimeServer({
|
|
55
|
+
host: '127.0.0.1',
|
|
56
|
+
port: 0,
|
|
57
|
+
store: {
|
|
58
|
+
dbPath: ':memory:',
|
|
59
|
+
getState: () => ({ status: 'idle', plan: [], queue: [], approvals: [] }),
|
|
60
|
+
listEvents: () => [],
|
|
61
|
+
},
|
|
62
|
+
getContext: async () => context,
|
|
63
|
+
run: async (ctx, body, { runId, signal } = {}) => {
|
|
64
|
+
runs.push({ runId, input: body.input, capabilityPlan: body.capabilityPlan, skillChain: body.skillChain });
|
|
65
|
+
// Un test peut jouer le rôle de l'agent : c'est le seul moyen de
|
|
66
|
+
// parcourir réellement enchaînement -> file -> run suivant.
|
|
67
|
+
if (onRun) await onRun(ctx, body);
|
|
68
|
+
if (autoFinish) {
|
|
69
|
+
dispatchAgentEvent(ctx.session, createAgentEvent('run_done', { origin: 'runtime', runId }));
|
|
70
|
+
return;
|
|
71
|
+
}
|
|
72
|
+
await new Promise((resolveRun) => {
|
|
73
|
+
pending.push({ runId, finish: resolveRun });
|
|
74
|
+
signal?.addEventListener('abort', () => resolveRun(), { once: true });
|
|
75
|
+
});
|
|
76
|
+
},
|
|
77
|
+
cancel: async (ctx) => {
|
|
78
|
+
dispatchAgentEvent(ctx.session, createAgentEvent('run_cancelled', {
|
|
79
|
+
origin: 'runtime',
|
|
80
|
+
runId: ctx.currentRunId,
|
|
81
|
+
}));
|
|
82
|
+
},
|
|
83
|
+
});
|
|
84
|
+
} catch (err) {
|
|
85
|
+
if (err?.code === 'EPERM') {
|
|
86
|
+
t.skip('network listen is not permitted in this sandbox');
|
|
87
|
+
return null;
|
|
88
|
+
}
|
|
89
|
+
throw err;
|
|
90
|
+
}
|
|
91
|
+
const post = async (path, body) => {
|
|
92
|
+
const response = await fetch(`http://127.0.0.1:${handle.port}${path}`, {
|
|
93
|
+
method: 'POST',
|
|
94
|
+
headers: { 'content-type': 'application/json' },
|
|
95
|
+
body: JSON.stringify(body ?? {}),
|
|
96
|
+
});
|
|
97
|
+
return { status: response.status, body: await response.json() };
|
|
98
|
+
};
|
|
99
|
+
const finishRun = async (runId) => {
|
|
100
|
+
const entry = pending.find((item) => item.runId === runId);
|
|
101
|
+
dispatchAgentEvent(session, createAgentEvent('run_done', { origin: 'runtime', runId }));
|
|
102
|
+
entry?.finish();
|
|
103
|
+
await settle();
|
|
104
|
+
};
|
|
105
|
+
const settle = () => new Promise((r) => setTimeout(r, 20));
|
|
106
|
+
t.after(async () => {
|
|
107
|
+
context.currentAbortController?.abort();
|
|
108
|
+
for (const entry of pending) entry.finish();
|
|
109
|
+
await handle.close();
|
|
110
|
+
});
|
|
111
|
+
return { context, session, runs, post, finishRun, settle, chain: () => session.controlQueue };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
test('E2E-001 pipeline: one objective, one run, internal plan left untouched', async (t) => {
|
|
115
|
+
const env = await harness(t, { skills: { pipeline: null } });
|
|
116
|
+
if (!env) return;
|
|
117
|
+
|
|
118
|
+
const { status, body } = await env.post('/run?workspace=acme', { input: '/pipeline' });
|
|
119
|
+
await env.settle();
|
|
120
|
+
|
|
121
|
+
assert.equal(status, 202);
|
|
122
|
+
assert.equal(body.kind, 'skill_chain');
|
|
123
|
+
assert.equal(body.objectives, 1, 'pipeline must never be fragmented into sub-steps');
|
|
124
|
+
assert.equal(env.runs.length, 1, 'one objective must produce exactly one run');
|
|
125
|
+
// The executor receives the private business intention while public runtime
|
|
126
|
+
// projections retain only the skill invocation. Nothing here pre-resolves a
|
|
127
|
+
// capability or hands the dispatcher a plan, so the pipeline capability
|
|
128
|
+
// keeps its own DAG and its own concurrency.
|
|
129
|
+
assert.equal(body.items[0].input, '/pipeline');
|
|
130
|
+
assert.match(env.runs[0].input, /^Execute the complete wiki production pipeline/);
|
|
131
|
+
assert.equal(env.runs[0].capabilityPlan, undefined);
|
|
132
|
+
assert.equal(env.chain().length, 1);
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
test('E2E-002 wiki-sync: two objectives, two ordered runs, one chainId', async (t) => {
|
|
136
|
+
const env = await harness(t, { skills: { 'wiki-sync': null } });
|
|
137
|
+
if (!env) return;
|
|
138
|
+
|
|
139
|
+
const { body } = await env.post('/run?workspace=acme', { input: '/wiki-sync docs' });
|
|
140
|
+
await env.settle();
|
|
141
|
+
|
|
142
|
+
assert.equal(body.objectives, 2);
|
|
143
|
+
assert.equal(env.runs.length, 2, 'the second objective must run after the first');
|
|
144
|
+
assert.match(env.runs[0].input, /^Export the requested Confluence source/);
|
|
145
|
+
assert.match(env.runs[1].input, /^Ingest the newly exported Markdown/);
|
|
146
|
+
// CME first, Production second — and the parameter reaches the step that
|
|
147
|
+
// consumes it, not only the last objective.
|
|
148
|
+
for (const run of env.runs) assert.match(run.input, /User parameters:\nsource: docs/);
|
|
149
|
+
const items = env.chain();
|
|
150
|
+
assert.equal(items.length, 2);
|
|
151
|
+
assert.equal(items[0].chainId, items[1].chainId);
|
|
152
|
+
assert.equal(items[0].chainId, body.chainId);
|
|
153
|
+
assert.deepEqual(items.map((item) => item.status), ['done', 'done']);
|
|
154
|
+
assert.deepEqual(items.map((item) => item.skillName), ['wiki-sync', 'wiki-sync']);
|
|
155
|
+
});
|
|
156
|
+
|
|
157
|
+
test('E2E-003 cancel: the running step and its chain stop, unrelated queue survives', async (t) => {
|
|
158
|
+
const env = await harness(t, {
|
|
159
|
+
autoFinish: false,
|
|
160
|
+
skills: {
|
|
161
|
+
'three-step': '---\nname: three-step\nparams: []\n---\nCollect the sources.\n\nThen ingest them.\n\nThen publish the result.',
|
|
162
|
+
},
|
|
163
|
+
});
|
|
164
|
+
if (!env) return;
|
|
165
|
+
|
|
166
|
+
const { body } = await env.post('/run?workspace=acme', { input: '/three-step' });
|
|
167
|
+
await env.settle();
|
|
168
|
+
assert.equal(body.objectives, 3);
|
|
169
|
+
|
|
170
|
+
// An unrelated request queued while the chain runs must not be collateral.
|
|
171
|
+
await env.post('/run?workspace=acme', {
|
|
172
|
+
input: 'unrelated structured work',
|
|
173
|
+
intent: 'enqueue',
|
|
174
|
+
capabilityPlan: { tasks: [{ id: 't1' }] },
|
|
175
|
+
});
|
|
176
|
+
await env.settle();
|
|
177
|
+
|
|
178
|
+
await env.finishRun(env.runs[0].runId);
|
|
179
|
+
assert.equal(env.runs.length, 2, 'step 2 must have started');
|
|
180
|
+
|
|
181
|
+
const cancelled = await env.post('/cancel?workspace=acme');
|
|
182
|
+
await env.settle();
|
|
183
|
+
assert.equal(cancelled.status, 202);
|
|
184
|
+
|
|
185
|
+
const items = env.chain();
|
|
186
|
+
const chainItems = items.filter((item) => item.chainId === body.chainId);
|
|
187
|
+
assert.equal(chainItems[0].status, 'done');
|
|
188
|
+
assert.equal(chainItems[1].status, 'cancelled', 'the running step is cancelled');
|
|
189
|
+
assert.equal(chainItems[2].status, 'skipped', 'the rest of the same chain is skipped');
|
|
190
|
+
assert.equal(chainItems[2].skipReason, 'chain_cancelled');
|
|
191
|
+
|
|
192
|
+
const unrelated = items.find((item) => !item.chainId);
|
|
193
|
+
assert.ok(unrelated, 'the unrelated enqueue must still be in the queue');
|
|
194
|
+
assert.notEqual(unrelated.status, 'skipped');
|
|
195
|
+
assert.notEqual(unrelated.status, 'cancelled');
|
|
196
|
+
});
|
|
197
|
+
|
|
198
|
+
// Plan V4.1 §57 — performance gate. The compiler test already locks the
|
|
199
|
+
// objective counts; what has to be guarded here is the step after it: one
|
|
200
|
+
// objective must become exactly one run, and therefore exactly one
|
|
201
|
+
// resolveObjective call, since prepareDelegation resolves once per run. A skill
|
|
202
|
+
// that silently fragments would show up as extra runs, not as extra objectives.
|
|
203
|
+
const PERFORMANCE_TABLE = {
|
|
204
|
+
pipeline: 1,
|
|
205
|
+
'wiki-ingest': 1,
|
|
206
|
+
'wiki-build': 1,
|
|
207
|
+
deliver: 1,
|
|
208
|
+
diagnose: 1,
|
|
209
|
+
status: 1,
|
|
210
|
+
'new-template': 1,
|
|
211
|
+
'wiki-sync': 2,
|
|
212
|
+
};
|
|
213
|
+
|
|
214
|
+
for (const [name, expected] of Object.entries(PERFORMANCE_TABLE)) {
|
|
215
|
+
test(`§57 performance gate: ${name} compiles to ${expected} objective(s) and ${expected} run(s)`, async (t) => {
|
|
216
|
+
const env = await harness(t, { skills: { [name]: null } });
|
|
217
|
+
if (!env) return;
|
|
218
|
+
// `status` collides with a built-in slash command: it is reachable only
|
|
219
|
+
// through the explicit `/skills run status` path, which carries skillName.
|
|
220
|
+
const reserved = RESERVED_SLASH_COMMANDS.has(name);
|
|
221
|
+
const { body } = await env.post('/run?workspace=acme', {
|
|
222
|
+
input: `/${name}`,
|
|
223
|
+
...(reserved ? { skillName: name } : {}),
|
|
224
|
+
});
|
|
225
|
+
await env.settle();
|
|
226
|
+
assert.equal(body.kind, 'skill_chain', reserved ? 'reserved names need skillName' : 'plain invocation');
|
|
227
|
+
assert.equal(body.objectives, expected, 'objective count');
|
|
228
|
+
assert.equal(env.runs.length, expected, 'run count');
|
|
229
|
+
assert.equal(env.chain().length, expected, 'control items');
|
|
230
|
+
// One run carries one whole intention: never a pre-resolved capability plan.
|
|
231
|
+
for (const run of env.runs) assert.equal(run.capabilityPlan, undefined);
|
|
232
|
+
});
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
test('E2E-004 kill: the whole workspace queue is purged, chain or not', async (t) => {
|
|
236
|
+
// /run cancel is chain-scoped by design; /run kill deliberately is not, and
|
|
237
|
+
// that asymmetry is the invariant worth guarding.
|
|
238
|
+
const env = await harness(t, {
|
|
239
|
+
autoFinish: false,
|
|
240
|
+
skills: {
|
|
241
|
+
'two-step': '---\nname: two-step\nparams: []\n---\nCollect the sources.\n\nThen ingest them.',
|
|
242
|
+
},
|
|
243
|
+
});
|
|
244
|
+
if (!env) return;
|
|
245
|
+
|
|
246
|
+
await env.post('/run?workspace=acme', { input: '/two-step' });
|
|
247
|
+
await env.settle();
|
|
248
|
+
await env.post('/run?workspace=acme', {
|
|
249
|
+
input: 'unrelated structured work',
|
|
250
|
+
intent: 'enqueue',
|
|
251
|
+
capabilityPlan: { tasks: [{ id: 't1' }] },
|
|
252
|
+
});
|
|
253
|
+
await env.settle();
|
|
254
|
+
assert.equal(env.chain().filter((item) => item.status === 'queued').length, 2);
|
|
255
|
+
|
|
256
|
+
const killed = await env.post('/kill?workspace=acme');
|
|
257
|
+
await env.settle();
|
|
258
|
+
|
|
259
|
+
assert.equal(killed.status, 202);
|
|
260
|
+
assert.equal(killed.body.killed, true);
|
|
261
|
+
assert.equal(
|
|
262
|
+
env.chain().some((item) => item.status === 'queued'),
|
|
263
|
+
false,
|
|
264
|
+
'kill leaves nothing queued, unlike cancel',
|
|
265
|
+
);
|
|
266
|
+
});
|
|
267
|
+
|
|
268
|
+
test('E2E-005 legacy: a structured enqueue keeps carrying its capabilityPlan', async (t) => {
|
|
269
|
+
const env = await harness(t, {
|
|
270
|
+
autoFinish: false,
|
|
271
|
+
skills: { 'one-step': '---\nname: one-step\nparams: []\n---\nDo the whole thing in one go.' },
|
|
272
|
+
});
|
|
273
|
+
if (!env) return;
|
|
274
|
+
|
|
275
|
+
await env.post('/run?workspace=acme', { input: '/one-step' });
|
|
276
|
+
await env.settle();
|
|
277
|
+
|
|
278
|
+
const plan = { tasks: [{ id: 't1', requiredCapability: 'knowledge.ingest' }] };
|
|
279
|
+
const enqueued = await env.post('/run?workspace=acme', {
|
|
280
|
+
input: 'ingest the pending files',
|
|
281
|
+
intent: 'enqueue',
|
|
282
|
+
capabilityPlan: plan,
|
|
283
|
+
});
|
|
284
|
+
await env.settle();
|
|
285
|
+
|
|
286
|
+
assert.equal(enqueued.status, 202);
|
|
287
|
+
const queued = env.chain().find((item) => !item.chainId);
|
|
288
|
+
assert.deepEqual(queued.capabilityPlan, plan, 'the structured plan must survive the queue');
|
|
289
|
+
|
|
290
|
+
// …and reach the run untouched once the chain step ahead of it completes.
|
|
291
|
+
await env.finishRun(env.runs[0].runId);
|
|
292
|
+
const structuredRun = env.runs.find((run) => run.capabilityPlan);
|
|
293
|
+
assert.deepEqual(structuredRun?.capabilityPlan, plan);
|
|
294
|
+
});
|
|
295
|
+
|
|
296
|
+
/*
|
|
297
|
+
E2E-010 — la pile des compétences franchit le hand-off.
|
|
298
|
+
|
|
299
|
+
C'est la seule chose que la garde de récursion ne pouvait pas vérifier
|
|
300
|
+
elle-même : elle lisait `session._skillStack`, qui vit le temps d'un run, alors
|
|
301
|
+
qu'une compétence imbriquée est mise en FILE et démarre après le `finally` du
|
|
302
|
+
parent. Le run recevait donc une pile vide et A→B→A passait, jusqu'à épuiser le
|
|
303
|
+
budget en headless où personne n'interrompt.
|
|
304
|
+
|
|
305
|
+
On vérifie ici la donnée qui traverse la frontière, pas le code qui la produit.
|
|
306
|
+
*/
|
|
307
|
+
test('E2E-010 skill stack: every run receives the ancestors of its chain', async (t) => {
|
|
308
|
+
const env = await harness(t, { skills: { 'wiki-sync': null } });
|
|
309
|
+
if (!env) return;
|
|
310
|
+
|
|
311
|
+
await env.post('/run?workspace=acme', { input: '/wiki-sync docs' });
|
|
312
|
+
await env.settle();
|
|
313
|
+
|
|
314
|
+
assert.ok(env.runs.length >= 1);
|
|
315
|
+
for (const run of env.runs) {
|
|
316
|
+
assert.deepEqual(
|
|
317
|
+
run.skillChain?.skillStack,
|
|
318
|
+
['wiki-sync'],
|
|
319
|
+
'the run must know which skills are already open above it',
|
|
320
|
+
);
|
|
321
|
+
}
|
|
322
|
+
// Et l'élément de file la porte, puisque c'est lui qui survit au parent.
|
|
323
|
+
for (const item of env.chain()) assert.deepEqual(item.skillStack, ['wiki-sync']);
|
|
324
|
+
});
|
|
325
|
+
|
|
326
|
+
/*
|
|
327
|
+
E2E-011 — le cycle A→B→A est refusé À TRAVERS deux hand-offs réels.
|
|
328
|
+
|
|
329
|
+
E2E-010 ne prouvait que le transport d'une pile à un seul élément, et le test
|
|
330
|
+
unitaire de la garde injectait `['a', 'b']` directement dans une session : ni
|
|
331
|
+
l'un ni l'autre ne parcourait le chemin qui était cassé — enchaînement, mise en
|
|
332
|
+
file, run suivant, relecture de la pile. C'est précisément là que la pile
|
|
333
|
+
disparaissait, puisqu'elle vivait sur la session le temps d'un seul run.
|
|
334
|
+
|
|
335
|
+
Le seul élément simulé ici est la DÉCISION du modèle : « quelle compétence
|
|
336
|
+
appeler ». Tout le reste — HTTP, file de contrôle, projection d'événements,
|
|
337
|
+
démarrage du run, garde de récursion — est le code de production.
|
|
338
|
+
*/
|
|
339
|
+
test('E2E-011 skill cycle: A → B → A is refused across two real hand-offs', async (t) => {
|
|
340
|
+
const refusals = [];
|
|
341
|
+
const invoked = [];
|
|
342
|
+
// Ce que la compétence en cours demande ensuite. C'est la seule chose que le
|
|
343
|
+
// test décide à la place du modèle.
|
|
344
|
+
const nextSkill = { 'skill-a': 'skill-b', 'skill-b': 'skill-a' };
|
|
345
|
+
|
|
346
|
+
const env = await harness(t, {
|
|
347
|
+
skills: {
|
|
348
|
+
'skill-a': '---\nname: skill-a\nparams: []\n---\nCollect the pending sources.\n',
|
|
349
|
+
'skill-b': '---\nname: skill-b\nparams: []\n---\nPublish the collected result.\n',
|
|
350
|
+
},
|
|
351
|
+
onRun: async (ctx, body) => {
|
|
352
|
+
const current = body.skillChain?.skillName;
|
|
353
|
+
const target = current ? nextSkill[current] : null;
|
|
354
|
+
if (!target || invoked.length >= 4) return;
|
|
355
|
+
invoked.push(target);
|
|
356
|
+
/*
|
|
357
|
+
Exactement ce que fait `wiki-manager.js` au démarrage d'un run : la pile
|
|
358
|
+
vient de l'élément, pas de ce que la session a gardé du run précédent.
|
|
359
|
+
*/
|
|
360
|
+
ctx.session._skillStack = Array.isArray(body.skillChain?.skillStack)
|
|
361
|
+
? [...body.skillChain.skillStack]
|
|
362
|
+
: [];
|
|
363
|
+
ctx.session.runtime = { url: 'http://runtime.invalid' };
|
|
364
|
+
const raw = await handleRuntimeControlTool(ctx.session, 'run_skill', {
|
|
365
|
+
skillName: target,
|
|
366
|
+
_userInput: target,
|
|
367
|
+
});
|
|
368
|
+
const result = JSON.parse(raw);
|
|
369
|
+
if (result.code === 'skill_recursion_blocked') refusals.push({ target, stack: result.skillStack });
|
|
370
|
+
},
|
|
371
|
+
});
|
|
372
|
+
if (!env) return;
|
|
373
|
+
|
|
374
|
+
await env.post('/run?workspace=acme', { input: '/skill-a' });
|
|
375
|
+
// Chaque cran est un run de plus : on laisse la file se vider jusqu'à ce que
|
|
376
|
+
// plus rien ne bouge, plutôt que de deviner le nombre de tours.
|
|
377
|
+
for (let tick = 0; tick < 12 && refusals.length === 0; tick += 1) await env.settle();
|
|
378
|
+
|
|
379
|
+
/*
|
|
380
|
+
A a mis B en file, B a démarré depuis la file — deux hand-offs réels — et
|
|
381
|
+
c'est le retour vers A qui tombe. Le cycle se referme donc AVANT qu'un
|
|
382
|
+
troisième run existe, ce qui est le comportement voulu : on refuse la
|
|
383
|
+
ré-entrée, on n'attend pas que le budget s'épuise.
|
|
384
|
+
*/
|
|
385
|
+
assert.deepEqual(invoked, ['skill-b', 'skill-a']);
|
|
386
|
+
assert.equal(refusals.length, 1, 'the cycle must be refused, not merely bounded by the budget');
|
|
387
|
+
assert.equal(refusals[0].target, 'skill-a');
|
|
388
|
+
/*
|
|
389
|
+
Et la pile porte les DEUX ancêtres. C'est l'assertion qui échouait avant le
|
|
390
|
+
correctif : le run de B repartait de `[]`, puisque le `finally` du run de A
|
|
391
|
+
avait déjà nettoyé la session, et le retour vers A passait sans être vu.
|
|
392
|
+
*/
|
|
393
|
+
assert.deepEqual(refusals[0].stack, ['skill-a', 'skill-b']);
|
|
394
|
+
});
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
import { randomUUID } from 'node:crypto';
|
|
2
|
+
import { applyLegacySkillPlaceholders, parseSkillArguments } from '../core/skillInvocation.js';
|
|
3
|
+
import { compileSkillObjectives, createSkillCompilerFallback } from '../core/skillCompiler.js';
|
|
4
|
+
|
|
5
|
+
const SKILL_PARAM_RE = /^[a-zA-Z][a-zA-Z0-9_-]{0,63}$/;
|
|
6
|
+
const DANGEROUS_KEYS = new Set(['__proto__', 'prototype', 'constructor']);
|
|
7
|
+
const MAX_PARAM_COUNT = 12;
|
|
8
|
+
const MAX_PARAM_VALUE_LENGTH = 2_000;
|
|
9
|
+
const MAX_ARGUMENTS_LENGTH = 8_000;
|
|
10
|
+
|
|
11
|
+
export function validateNamedSkillArguments(skill, supplied = {}) {
|
|
12
|
+
if (supplied == null) supplied = {};
|
|
13
|
+
if (typeof supplied !== 'object' || Array.isArray(supplied)) throw argumentError('Skill arguments must be an object.');
|
|
14
|
+
const declared = Array.isArray(skill?.params) ? skill.params : [];
|
|
15
|
+
if (declared.length > MAX_PARAM_COUNT || declared.some((name) => !SKILL_PARAM_RE.test(name) || DANGEROUS_KEYS.has(name))) {
|
|
16
|
+
throw argumentError('The skill declares invalid parameters.');
|
|
17
|
+
}
|
|
18
|
+
const allowed = new Set(declared);
|
|
19
|
+
const entries = Object.entries(supplied);
|
|
20
|
+
if (entries.length > MAX_PARAM_COUNT) throw argumentError(`A skill accepts at most ${MAX_PARAM_COUNT} arguments.`);
|
|
21
|
+
let total = 0;
|
|
22
|
+
const normalized = Object.create(null);
|
|
23
|
+
for (const [name, value] of entries) {
|
|
24
|
+
if (!allowed.has(name) || DANGEROUS_KEYS.has(name)) throw argumentError(`Unknown skill argument: ${name}.`);
|
|
25
|
+
if (value != null && typeof value !== 'string') throw argumentError(`Skill argument ${name} must be a string.`);
|
|
26
|
+
const text = value == null ? '' : value;
|
|
27
|
+
if (text.length > MAX_PARAM_VALUE_LENGTH) throw argumentError(`Skill argument ${name} exceeds ${MAX_PARAM_VALUE_LENGTH} characters.`);
|
|
28
|
+
total += text.length;
|
|
29
|
+
normalized[name] = text;
|
|
30
|
+
}
|
|
31
|
+
if (total > MAX_ARGUMENTS_LENGTH) throw argumentError(`Skill arguments exceed ${MAX_ARGUMENTS_LENGTH} characters in total.`);
|
|
32
|
+
for (const name of declared) if (!(name in normalized)) normalized[name] = '';
|
|
33
|
+
return normalized;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export async function runSkillChain(context, skill, {
|
|
37
|
+
args = null,
|
|
38
|
+
rawArgs = null,
|
|
39
|
+
enqueueControlRequest,
|
|
40
|
+
drainControlQueue,
|
|
41
|
+
selectionKind = null,
|
|
42
|
+
/**
|
|
43
|
+
* Compétences déjà en cours d'exécution au-dessus de celle-ci. Chaque élément
|
|
44
|
+
* mis en file la porte, augmentée de la compétence courante : c'est ce qui
|
|
45
|
+
* permet au run imbriqué — qui démarre longtemps après son parent — de
|
|
46
|
+
* reconnaître un cycle.
|
|
47
|
+
*/
|
|
48
|
+
skillStack = [],
|
|
49
|
+
} = {}) {
|
|
50
|
+
if (args !== null && rawArgs !== null) {
|
|
51
|
+
const error = new Error('Provide named skill arguments or raw arguments, not both.');
|
|
52
|
+
error.code = 'skill_arguments_invalid';
|
|
53
|
+
throw error;
|
|
54
|
+
}
|
|
55
|
+
if (typeof enqueueControlRequest !== 'function' || typeof drainControlQueue !== 'function') {
|
|
56
|
+
throw new TypeError('runSkillChain requires the runtime control queue functions.');
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
const resolvedArgs = args === null ? parseSkillArguments(skill, rawArgs ?? '') : validateNamedSkillArguments(skill, args);
|
|
60
|
+
const legacy = applyLegacySkillPlaceholders(skill.body, resolvedArgs);
|
|
61
|
+
const naturalArgs = Object.fromEntries(
|
|
62
|
+
Object.entries(resolvedArgs).filter(([name]) => !legacy.deprecatedPlaceholders.includes(name)),
|
|
63
|
+
);
|
|
64
|
+
const objectives = await compileSkillObjectives(
|
|
65
|
+
{ ...skill, body: legacy.body },
|
|
66
|
+
naturalArgs,
|
|
67
|
+
{ llmFallback: createSkillCompilerFallback(context.session?.llm, { timeoutMs: 8_000 }) },
|
|
68
|
+
);
|
|
69
|
+
const chainId = `chain-${randomUUID()}`;
|
|
70
|
+
const nestedStack = [...(Array.isArray(skillStack) ? skillStack : []), skill.name];
|
|
71
|
+
const publicInput = formatPublicSkillInvocation(skill.name, resolvedArgs);
|
|
72
|
+
const items = objectives.map((objective, chainSequence) => enqueueControlRequest(context, objective.text, {
|
|
73
|
+
publicInput,
|
|
74
|
+
chainId,
|
|
75
|
+
chainSequence,
|
|
76
|
+
skillName: skill.name,
|
|
77
|
+
skillExecution: skill.execution === 'direct' ? 'direct' : 'orchestrated',
|
|
78
|
+
skillStack: nestedStack,
|
|
79
|
+
...(selectionKind ? { selectionKind } : {}),
|
|
80
|
+
optional: objective.optional,
|
|
81
|
+
continueOnFailure: objective.continueOnFailure,
|
|
82
|
+
}));
|
|
83
|
+
void drainControlQueue(context);
|
|
84
|
+
return {
|
|
85
|
+
chainId,
|
|
86
|
+
skill: skill.name,
|
|
87
|
+
objectives: objectives.length,
|
|
88
|
+
items,
|
|
89
|
+
deprecatedPlaceholders: legacy.deprecatedPlaceholders,
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function formatPublicSkillInvocation(name, args = {}) {
|
|
94
|
+
const values = Object.entries(args ?? {})
|
|
95
|
+
.filter(([, value]) => String(value ?? '').trim())
|
|
96
|
+
.map(([key, value]) => `${key}=${JSON.stringify(String(value))}`);
|
|
97
|
+
return `/${String(name ?? '').trim()}${values.length ? ` ${values.join(' ')}` : ''}`;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
function argumentError(message) {
|
|
101
|
+
const error = new Error(message);
|
|
102
|
+
error.code = 'skill_arguments_invalid';
|
|
103
|
+
return error;
|
|
104
|
+
}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { formatPublicSkillInvocation, runSkillChain, validateNamedSkillArguments } from './skillRun.js';
|
|
4
|
+
|
|
5
|
+
const skill = { name: 'deliver', params: ['template', 'polish'], body: 'Deliver the requested output.' };
|
|
6
|
+
|
|
7
|
+
test('named skill arguments preserve spaces and fill omitted declarations with empty strings', () => {
|
|
8
|
+
const args = validateNamedSkillArguments(skill, { template: 'Quarterly report' });
|
|
9
|
+
assert.equal(Object.getPrototypeOf(args), null);
|
|
10
|
+
assert.deepEqual({ ...args }, { template: 'Quarterly report', polish: '' });
|
|
11
|
+
});
|
|
12
|
+
|
|
13
|
+
test('named skill arguments reject undeclared, non-string, and oversized values', () => {
|
|
14
|
+
assert.throws(() => validateNamedSkillArguments(skill, { target: 'x' }), { code: 'skill_arguments_invalid' });
|
|
15
|
+
assert.throws(() => validateNamedSkillArguments(skill, { template: 42 }), { code: 'skill_arguments_invalid' });
|
|
16
|
+
assert.throws(() => validateNamedSkillArguments(skill, { template: 'x'.repeat(2_001) }), { code: 'skill_arguments_invalid' });
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
test('runSkillChain enqueues named arguments without exposing a skill body field', async () => {
|
|
20
|
+
const queued = [];
|
|
21
|
+
const result = await runSkillChain({ session: {} }, skill, {
|
|
22
|
+
args: { template: 'Quarterly report' },
|
|
23
|
+
enqueueControlRequest(_context, input, metadata) {
|
|
24
|
+
const item = { id: `item-${queued.length}`, input, status: 'queued', ...metadata };
|
|
25
|
+
queued.push(item);
|
|
26
|
+
return item;
|
|
27
|
+
},
|
|
28
|
+
drainControlQueue() {},
|
|
29
|
+
});
|
|
30
|
+
assert.equal(result.objectives, 1);
|
|
31
|
+
assert.match(queued[0].input, /template: Quarterly report/);
|
|
32
|
+
assert.equal(queued[0].publicInput, '/deliver template="Quarterly report"');
|
|
33
|
+
assert.equal(queued[0].skillExecution, 'orchestrated');
|
|
34
|
+
assert.equal('body' in queued[0], false);
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
test('public skill invocation contains arguments but never compiled objective prose', () => {
|
|
38
|
+
const rendered = formatPublicSkillInvocation('deliver', { template: 'Quarterly report', polish: '' });
|
|
39
|
+
assert.equal(rendered, '/deliver template="Quarterly report"');
|
|
40
|
+
assert.doesNotMatch(rendered, /Deliver the requested output/);
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
/*
|
|
44
|
+
La pile voyage avec l'élément, sinon elle n'existe plus quand il démarre.
|
|
45
|
+
|
|
46
|
+
Elle vivait sur la session le temps d'un run, restaurée par son `finally`. Une
|
|
47
|
+
compétence imbriquée n'étant pas exécutée en ligne mais MISE EN FILE, son run
|
|
48
|
+
démarrait après ce nettoyage et lisait une pile vide : la garde n'attrapait que
|
|
49
|
+
la réinvocation d'une compétence dans son propre run, et laissait passer A→B→A.
|
|
50
|
+
*/
|
|
51
|
+
test('runSkillChain stamps every queued item with the ancestor stack', async () => {
|
|
52
|
+
const queued = [];
|
|
53
|
+
await runSkillChain({ session: {} }, skill, {
|
|
54
|
+
args: {},
|
|
55
|
+
skillStack: ['pipeline', 'wiki-sync'],
|
|
56
|
+
enqueueControlRequest(_context, input, metadata) {
|
|
57
|
+
const item = { id: `item-${queued.length}`, input, ...metadata };
|
|
58
|
+
queued.push(item);
|
|
59
|
+
return item;
|
|
60
|
+
},
|
|
61
|
+
drainControlQueue() {},
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
assert.ok(queued.length > 0);
|
|
65
|
+
for (const item of queued) {
|
|
66
|
+
// La compétence lancée est empilée ici, une seule fois, au seul endroit qui
|
|
67
|
+
// sait laquelle a réellement été résolue.
|
|
68
|
+
assert.deepEqual(item.skillStack, ['pipeline', 'wiki-sync', 'deliver']);
|
|
69
|
+
}
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
test('runSkillChain starts a fresh stack for a top-level invocation', async () => {
|
|
73
|
+
const queued = [];
|
|
74
|
+
await runSkillChain({ session: {} }, skill, {
|
|
75
|
+
args: {},
|
|
76
|
+
enqueueControlRequest(_context, input, metadata) {
|
|
77
|
+
queued.push(metadata);
|
|
78
|
+
return { id: 'item-0', input, ...metadata };
|
|
79
|
+
},
|
|
80
|
+
drainControlQueue() {},
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
assert.deepEqual(queued[0].skillStack, ['deliver']);
|
|
84
|
+
});
|