@dotdrelle/wiki-manager 0.14.13 → 0.14.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +29 -4
- package/README.md +19 -0
- package/docker-compose.yml +6 -1
- package/package.json +1 -1
- package/src/activity/activityAggregator.js +50 -16
- package/src/activity/activityAggregator.test.js +67 -4
- package/src/agent/graph.js +79 -11
- package/src/agent/graph.test.js +43 -3
- package/src/cli/wiki-manager.js +74 -11
- package/src/cli/wiki-manager.test.js +40 -1
- package/src/commands/slash.js +10 -3
- package/src/commands/slash.test.js +24 -0
- package/src/core/buildInfo.json +2 -2
- package/src/core/dockerCompose.test.js +4 -0
- package/src/core/env.test.js +3 -0
- package/src/core/mcp.js +1 -1
- package/src/core/wikiSetup.js +35 -0
- package/src/core/wikiWorkspace.test.js +20 -0
- package/src/core/workflow.js +72 -0
- package/src/core/workflow.test.js +57 -0
- package/src/core/workspaces.js +10 -2
- package/src/orchestrator/dependencyResolver.js +19 -1
- package/src/orchestrator/objectiveResolver.js +24 -0
- package/src/orchestrator/objectiveResolver.test.js +23 -1
- package/src/orchestrator/scheduler.js +27 -7
- package/src/orchestrator/scheduler.test.js +45 -1
- package/src/runtime/auth.test.js +65 -1
- package/src/runtime/client.js +4 -0
- package/src/runtime/donna-contract.test.js +2 -0
- package/src/runtime/lifecycle.js +21 -12
- package/src/runtime/runner.js +141 -18
- package/src/runtime/runner.test.js +30 -0
- package/src/runtime/server.js +13 -2
- package/src/runtime/server.test.js +30 -1
- package/src/runtime/store.js +54 -0
- package/src/runtime/store.test.js +42 -0
- package/src/shell/FileEditorDialog.tsx +2 -2
- package/src/shell/LeftPane.tsx +60 -15
- package/src/shell/RightPane.tsx +168 -56
- package/src/shell/StartupScreen.tsx +3 -7
- package/src/shell/renderer.ts +1 -0
- package/src/shell/repl.js +7 -81
- package/src/shell/repl.test.js +141 -38
- package/src/shell/tui.tsx +65 -65
- package/src/shell/useSession.ts +92 -6
- package/wiki-workspace +28 -0
package/.env.example
CHANGED
|
@@ -63,10 +63,35 @@ DOCUMENTS_MCP_AUTH_TOKEN=
|
|
|
63
63
|
# Add your own entries when you declare additional endpoints.
|
|
64
64
|
|
|
65
65
|
# ── Orchestration (optional) ───────────────────────────────────────────────────
|
|
66
|
-
#
|
|
67
|
-
#
|
|
68
|
-
#
|
|
69
|
-
#
|
|
66
|
+
# The runtime runs on the host while `llm-wiki serve` runs in Docker and
|
|
67
|
+
# reaches it through host.docker.internal. Listen on all host interfaces so
|
|
68
|
+
# the container can connect; exposed runtimes are protected by the generated
|
|
69
|
+
# WIKI_MANAGER_RUNTIME_TOKEN.
|
|
70
|
+
#
|
|
71
|
+
# WIKI_MANAGER_RUNTIME_PORT=7788
|
|
72
|
+
# Set to 0 to skip pulling and renewing already-running containers at startup.
|
|
73
|
+
# WIKI_MANAGER_AUTO_UPDATE=1
|
|
74
|
+
# WIKI_MANAGER_RUNTIME_HOST=0.0.0.0
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ── Parallelism & throughput ───────────────────────────────────────────────────
|
|
78
|
+
# Effective concurrency = MIN(agent recommendedConcurrency, agent maxConcurrency,
|
|
79
|
+
# this ceiling, per-task limits). So the PRIMARY levers live on the production
|
|
80
|
+
# agent (PRODUCTION_RECOMMENDED_CONCURRENCY / PRODUCTION_MAX_CONCURRENCY); this
|
|
81
|
+
# manager variable can only LOWER the result, never raise it. Full explanation
|
|
82
|
+
# and low/high profiles in docs/configuration.md § "Parallelism & throughput".
|
|
83
|
+
#
|
|
84
|
+
# Manager ceiling — leave unset to let the agent decide. Set to constrain:
|
|
85
|
+
# WIKI_MANAGER_CAPABILITY_CONCURRENCY=4
|
|
86
|
+
#
|
|
87
|
+
# Production agent capacity (passed through by docker-compose). Intermediate
|
|
88
|
+
# defaults are 4/8 (effective ≈ 4 parallel). Profiles:
|
|
89
|
+
# low → PRODUCTION_RECOMMENDED_CONCURRENCY=2 PRODUCTION_MAX_CONCURRENCY=4
|
|
90
|
+
# high → PRODUCTION_RECOMMENDED_CONCURRENCY=8 PRODUCTION_MAX_CONCURRENCY=16
|
|
91
|
+
# The wiki LLM backend must accept this many concurrent requests, and
|
|
92
|
+
# ingest_apply stays serialized regardless (global workspace-write lock).
|
|
93
|
+
# PRODUCTION_RECOMMENDED_CONCURRENCY=4
|
|
94
|
+
# PRODUCTION_MAX_CONCURRENCY=8
|
|
70
95
|
|
|
71
96
|
# ── MCP retry policy (optional) ────────────────────────────────────────────────
|
|
72
97
|
|
package/README.md
CHANGED
|
@@ -550,6 +550,25 @@ or the shell command `/approve item <id>`. The approval timeout defaults to 10
|
|
|
550
550
|
minutes and can be changed with `WIKI_MANAGER_APPROVAL_TIMEOUT_MS` or
|
|
551
551
|
`approvalTimeoutMs` in the `/run` body.
|
|
552
552
|
|
|
553
|
+
Directly-launched capability runs (ingest, pipeline) now **wait for approval by
|
|
554
|
+
default** before their mutating tasks: reply "valide tout", run `/approve`, or
|
|
555
|
+
click Approve in either UI (Shell right-pane banner, or the `serve` banner above
|
|
556
|
+
the composer). Auto-approval only happens when the run is started with
|
|
557
|
+
`autoApprove: true` (headless/CI).
|
|
558
|
+
|
|
559
|
+
### Parallelism & throughput
|
|
560
|
+
|
|
561
|
+
The number of tasks that run at once is `MIN(agent recommendedConcurrency, agent
|
|
562
|
+
maxConcurrency, WIKI_MANAGER_CAPABILITY_CONCURRENCY, per-task limits)` — a
|
|
563
|
+
minimum, so the manager ceiling can only lower it. The production agent ships
|
|
564
|
+
intermediate defaults (`PRODUCTION_RECOMMENDED_CONCURRENCY=4` /
|
|
565
|
+
`PRODUCTION_MAX_CONCURRENCY=8`, ≈ 4 parallel); locks then cap real parallelism
|
|
566
|
+
per phase (`ingest_apply` stays serial). The resolved value is shown in both
|
|
567
|
+
UIs' run summary and on the run node of the execution graph, with an amber
|
|
568
|
+
"(ceiling)" marker when the manager ceiling binds. Low/high profiles, the lock
|
|
569
|
+
model and the LLM-backend caveat are in
|
|
570
|
+
[docs/configuration.md § "Parallelism & throughput"](docs/configuration.md).
|
|
571
|
+
|
|
553
572
|
While a run is active, `GET`/`POST /control` still answers without waiting for
|
|
554
573
|
it to finish: `{"action":"status"}` returns the current run/plan/queue state,
|
|
555
574
|
`{"action":"explain"}` adds a one-line plain-language summary, and
|
package/docker-compose.yml
CHANGED
|
@@ -59,7 +59,7 @@ services:
|
|
|
59
59
|
- DOCUMENT_MAX_UPLOAD_BYTES=${DOCUMENT_MAX_UPLOAD_BYTES:-52428800}
|
|
60
60
|
- WIKI_MCP_PROXY_URL=http://host.docker.internal:${WIKI_MCP_PORT:-3101}/mcp
|
|
61
61
|
- PRODUCTION_MCP_PROXY_URL=http://host.docker.internal:${PRODUCTION_MCP_PORT:-3102}/mcp/
|
|
62
|
-
- WIKI_MANAGER_RUNTIME_URL=http://host.docker.internal
|
|
62
|
+
- WIKI_MANAGER_RUNTIME_URL=http://host.docker.internal:${WIKI_MANAGER_RUNTIME_PORT:-7788}
|
|
63
63
|
- WIKI_MANAGER_RUNTIME_TOKEN=${WIKI_MANAGER_RUNTIME_TOKEN:-}
|
|
64
64
|
# HTTPS — set paths inside the container (e.g. /certs/server.crt) and uncomment the volume above
|
|
65
65
|
#- WIKI_SERVE_TLS_CERT_PATH=/certs/server.crt
|
|
@@ -110,6 +110,11 @@ services:
|
|
|
110
110
|
- WIKI_CONFIG_PATH=${WIKI_CONFIG_PATH:-}
|
|
111
111
|
- PRODUCTION_ALLOWED_STEPS=${PRODUCTION_ALLOWED_STEPS:-doctor,ingest,ingest_plan,ingest_apply,build,export,polish,pipeline}
|
|
112
112
|
- PRODUCTION_REQUIRE_CONFIRMATION=${PRODUCTION_REQUIRE_CONFIRMATION:-false}
|
|
113
|
+
# Parallelism levers — effective concurrency ≈ recommendedConcurrency.
|
|
114
|
+
# Intermediate defaults (4/8). Low profile 2/4, high profile 8/16.
|
|
115
|
+
# See docs/configuration.md § "Parallelism & throughput".
|
|
116
|
+
- PRODUCTION_RECOMMENDED_CONCURRENCY=${PRODUCTION_RECOMMENDED_CONCURRENCY:-4}
|
|
117
|
+
- PRODUCTION_MAX_CONCURRENCY=${PRODUCTION_MAX_CONCURRENCY:-8}
|
|
113
118
|
- PRODUCTION_JOBS_DIR=${PRODUCTION_JOBS_DIR:-/workspace/.wiki/production-jobs}
|
|
114
119
|
- PRODUCTION_LOCKS_DIR=${PRODUCTION_LOCKS_DIR:-/workspace/.wiki/production-jobs/locks}
|
|
115
120
|
ports:
|
package/package.json
CHANGED
|
@@ -71,23 +71,43 @@ function groupLine(group, activities) {
|
|
|
71
71
|
const running = group.tasks.filter((task) => ACTIVE.has(statusOf(task)));
|
|
72
72
|
const failed = group.tasks.find((task) => statusOf(task) === 'failed');
|
|
73
73
|
const waitingApproval = group.tasks.some((task) => ['pending_approval', 'waiting_approval'].includes(statusOf(task)));
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
74
|
+
// Activity polling can advance before the persisted task projection catches
|
|
75
|
+
// up. Treat a live, linked activity as authoritative instead of rendering
|
|
76
|
+
// the whole group as "validation 0%" while a worker visibly runs at 35%.
|
|
77
|
+
const activePair = group.tasks
|
|
78
|
+
.map((task) => ({ task, activity: activityForTask(task, activities) }))
|
|
79
|
+
.find(({ activity }) => activity && !activity.terminal && !DONE.has(statusOf(activity)));
|
|
80
|
+
const activeTask = activePair?.task ?? running[0] ?? null;
|
|
81
|
+
const activeActivity = activePair?.activity
|
|
82
|
+
?? running.map((task) => activityForTask(task, activities)).find(Boolean);
|
|
83
|
+
const activeAgents = new Set([
|
|
84
|
+
...running.map((task) => task.agentInstanceId),
|
|
85
|
+
activeTask?.agentInstanceId,
|
|
86
|
+
].filter(Boolean)).size;
|
|
87
|
+
const activeProgress = Number.isFinite(Number(activeActivity?.progress?.percent))
|
|
88
|
+
? Number(activeActivity.progress.percent)
|
|
89
|
+
: running.map((task) => Number(task?.progress?.percent)).find(Number.isFinite);
|
|
78
90
|
let icon = '[ ]';
|
|
79
91
|
let status = 'en attente';
|
|
80
|
-
|
|
81
|
-
|
|
92
|
+
// A group may contain an earlier failure while another independent task is
|
|
93
|
+
// still progressing. Show the live worker as running; surface the group
|
|
94
|
+
// failure once no work remains. Otherwise its business label was rendered
|
|
95
|
+
// red even though that exact task was healthy and advancing.
|
|
96
|
+
// Glyphs mirror the Shell PlanPanel so the two never disagree: done is a
|
|
97
|
+
// check (not a cross — "[x]" read as a failure X), failure is a distinct
|
|
98
|
+
// cross, and an approval wait gets its own pause glyph instead of reusing the
|
|
99
|
+
// failure "[!]".
|
|
100
|
+
if (running.length > 0 || activeActivity) {
|
|
101
|
+
icon = '[...]';
|
|
102
|
+
status = activeProgress != null ? `${Math.round(activeProgress)} %` : `${done}/${total}`;
|
|
103
|
+
} else if (failed) {
|
|
104
|
+
icon = '[✗]';
|
|
82
105
|
status = 'error';
|
|
83
106
|
} else if (done === total) {
|
|
84
|
-
icon = '[
|
|
107
|
+
icon = '[✓]';
|
|
85
108
|
status = 'done';
|
|
86
|
-
} else if (running.length > 0) {
|
|
87
|
-
icon = '[...]';
|
|
88
|
-
status = activeProgress != null ? `${Math.round(activeProgress)} %` : `${done}/${total}`;
|
|
89
109
|
} else if (waitingApproval) {
|
|
90
|
-
icon = '[
|
|
110
|
+
icon = '[⏸]';
|
|
91
111
|
status = 'validation';
|
|
92
112
|
} else if (done > 0) {
|
|
93
113
|
icon = '[...]';
|
|
@@ -100,12 +120,28 @@ function groupLine(group, activities) {
|
|
|
100
120
|
// text. Fall back to the task-completion ratio when nothing is running.
|
|
101
121
|
const percent = running.length > 0 && activeProgress != null
|
|
102
122
|
? Math.round(activeProgress)
|
|
123
|
+
: activeActivity && activeProgress != null
|
|
124
|
+
? Math.round(activeProgress)
|
|
103
125
|
: (total > 0 ? Math.round((done / total) * 100) : null);
|
|
126
|
+
const phaseTasks = activeTask
|
|
127
|
+
? group.tasks.filter((task) => String(task.operation ?? '') === String(activeTask.operation ?? ''))
|
|
128
|
+
: [];
|
|
129
|
+
const taskIndex = activeTask ? phaseTasks.indexOf(activeTask) + 1 : null;
|
|
130
|
+
const taskTotal = phaseTasks.length || null;
|
|
104
131
|
return {
|
|
105
132
|
id: `group:${group.id}`,
|
|
106
133
|
label: `${icon} ${group.label} - ${status}${agents}`,
|
|
107
134
|
status,
|
|
108
|
-
|
|
135
|
+
// Preserve the active worker's business progress (source/template/
|
|
136
|
+
// deliverable, phase and detail). ShellUI can then show the same useful
|
|
137
|
+
// information as the direct wiki CLI instead of only "knowledge.update".
|
|
138
|
+
progress: {
|
|
139
|
+
...(activeActivity?.progress ?? {}),
|
|
140
|
+
done,
|
|
141
|
+
total,
|
|
142
|
+
percent,
|
|
143
|
+
...(taskIndex ? { taskIndex, taskTotal, taskOperation: activeTask?.operation ?? null } : {}),
|
|
144
|
+
},
|
|
109
145
|
activeAgents,
|
|
110
146
|
};
|
|
111
147
|
}
|
|
@@ -121,15 +157,13 @@ function activityLine(activity) {
|
|
|
121
157
|
};
|
|
122
158
|
}
|
|
123
159
|
|
|
124
|
-
function
|
|
160
|
+
function activityForTask(task, activities) {
|
|
125
161
|
const taskId = String(task.id ?? task.step ?? '');
|
|
126
162
|
const activityKey = task.activityKey ?? task.ownerActivityKey ?? null;
|
|
127
|
-
|
|
163
|
+
return activities.find((activity) =>
|
|
128
164
|
(activityKey && (activity.key === activityKey || activity.id === activityKey))
|
|
129
165
|
|| String(activity?.progress?.stepId ?? '') === taskId,
|
|
130
166
|
);
|
|
131
|
-
const value = Number(match?.progress?.percent ?? task.progress?.percent);
|
|
132
|
-
return Number.isFinite(value) ? value : null;
|
|
133
167
|
}
|
|
134
168
|
|
|
135
169
|
function statusOf(task) {
|
|
@@ -5,6 +5,20 @@ import { aggregateActivity } from './activityAggregator.js';
|
|
|
5
5
|
import { visibleActivityEvents } from './activityDeduplicator.js';
|
|
6
6
|
import { calculateWeightedProgress } from './progressCalculator.js';
|
|
7
7
|
|
|
8
|
+
test('a group awaiting approval renders a distinct pause glyph, not a failure', () => {
|
|
9
|
+
const aggregated = aggregateActivity({
|
|
10
|
+
plan: [
|
|
11
|
+
{ id: 'publish', label: 'Publication', groupId: 'publish', status: 'pending_approval', progressWeight: 1 },
|
|
12
|
+
],
|
|
13
|
+
activities: [],
|
|
14
|
+
});
|
|
15
|
+
const line = aggregated.lines.find((item) => /publish|Publication/.test(item.label));
|
|
16
|
+
assert.ok(line, 'the awaiting-approval group is present');
|
|
17
|
+
assert.match(line.label, /\[⏸\]/, 'uses the pause glyph');
|
|
18
|
+
assert.doesNotMatch(line.label, /\[!\]|\[✗\]/, 'never reuses the failure glyph');
|
|
19
|
+
assert.equal(line.status, 'validation');
|
|
20
|
+
});
|
|
21
|
+
|
|
8
22
|
test('activityDeduplicator keeps one visible entry for repeated 2 percent polls', () => {
|
|
9
23
|
const events = Array.from({ length: 50 }, () => ({
|
|
10
24
|
type: 'activity_upserted',
|
|
@@ -37,10 +51,10 @@ test('aggregateActivity exposes initial synthesis and grouped display lines', ()
|
|
|
37
51
|
const activity = aggregateActivity({
|
|
38
52
|
plan: [
|
|
39
53
|
{ id: 'collect', label: 'Collecte externe', groupId: 'collect', status: 'done', progressWeight: 1 },
|
|
40
|
-
{ id: 'enrich', label: 'Enrichissement commercial', groupId: 'enrich', requiredCapability: 'customer-data.enrich', status: 'running', progressWeight: 1, activityKey: 'activity-enrich' },
|
|
54
|
+
{ id: 'enrich', label: 'Enrichissement commercial', groupId: 'enrich', requiredCapability: 'customer-data.enrich', operation: 'export', status: 'running', progressWeight: 1, activityKey: 'activity-enrich' },
|
|
41
55
|
{ id: 'publish', label: 'Publication', groupId: 'publish', status: 'pending', progressWeight: 1 },
|
|
42
56
|
],
|
|
43
|
-
activities: [{ key: 'activity-enrich', label: 'Enrichissement commercial', status: 'running', progress: { percent: 63, stepId: 'enrich' } }],
|
|
57
|
+
activities: [{ key: 'activity-enrich', label: 'Enrichissement commercial', status: 'running', progress: { percent: 63, stepId: 'enrich', label: 'Export rapport.md', detail: 'Rendering PDF', currentStep: 'export' } }],
|
|
44
58
|
}, [{
|
|
45
59
|
type: 'plan.received',
|
|
46
60
|
payload: {
|
|
@@ -54,8 +68,15 @@ test('aggregateActivity exposes initial synthesis and grouped display lines', ()
|
|
|
54
68
|
|
|
55
69
|
assert.deepEqual(activity.initialSynthesis, ['120 sources detectees', '6 traitements simultanes recommandes']);
|
|
56
70
|
assert.equal(activity.progress.percent, 54);
|
|
57
|
-
assert.ok(activity.lines.some((line) => /\[
|
|
71
|
+
assert.ok(activity.lines.some((line) => /\[✓\] collect - done/.test(line.label)));
|
|
58
72
|
assert.ok(activity.lines.some((line) => /\[\.\.\.\] customer-data\.enrich - 63 %/.test(line.label)));
|
|
73
|
+
const enrichLine = activity.lines.find((line) => /customer-data\.enrich/.test(line.label));
|
|
74
|
+
assert.equal(enrichLine.progress.label, 'Export rapport.md');
|
|
75
|
+
assert.equal(enrichLine.progress.detail, 'Rendering PDF');
|
|
76
|
+
assert.equal(enrichLine.progress.currentStep, 'export');
|
|
77
|
+
assert.equal(enrichLine.progress.taskIndex, 1);
|
|
78
|
+
assert.equal(enrichLine.progress.taskTotal, 1);
|
|
79
|
+
assert.equal(enrichLine.progress.taskOperation, 'export');
|
|
59
80
|
assert.ok(activity.lines.some((line) => /\[ \] publish - en attente/.test(line.label)));
|
|
60
81
|
});
|
|
61
82
|
|
|
@@ -85,8 +106,50 @@ test('aggregateActivity keeps activities not attached to any plan task visible',
|
|
|
85
106
|
|
|
86
107
|
const aggregated = aggregateActivity(state, []);
|
|
87
108
|
const labels = aggregated.lines.map((line) => line.label).join('\n');
|
|
88
|
-
assert.match(labels, /\[
|
|
109
|
+
assert.match(labels, /\[✓\] .* done/, 'the done plan group stays visible');
|
|
89
110
|
assert.match(labels, /Ingest b87acaf6/, 'the unattached running ingest must appear');
|
|
90
111
|
const ingestLine = aggregated.lines.find((line) => /Ingest/.test(line.label));
|
|
91
112
|
assert.equal(ingestLine.status, 'running');
|
|
92
113
|
});
|
|
114
|
+
|
|
115
|
+
test('aggregateActivity trusts live worker progress while the task projection lags', () => {
|
|
116
|
+
const aggregated = aggregateActivity({
|
|
117
|
+
plan: [
|
|
118
|
+
{ id: 'plan-a', groupId: 'knowledge.update', operation: 'ingest_plan', status: 'pending_approval', activityKey: 'activity-plan-a' },
|
|
119
|
+
{ id: 'plan-b', groupId: 'knowledge.update', operation: 'ingest_plan', status: 'pending' },
|
|
120
|
+
],
|
|
121
|
+
activities: [{
|
|
122
|
+
key: 'activity-plan-a',
|
|
123
|
+
status: 'running',
|
|
124
|
+
terminal: false,
|
|
125
|
+
progress: { percent: 35, stepId: 'plan-a', label: 'Ingest source-a.md', stepIndex: 1, stepTotal: 1 },
|
|
126
|
+
}],
|
|
127
|
+
}, []);
|
|
128
|
+
|
|
129
|
+
const line = aggregated.lines.find((item) => /knowledge\.update/.test(item.label));
|
|
130
|
+
assert.match(line.label, /35 %/);
|
|
131
|
+
assert.equal(line.progress.percent, 35);
|
|
132
|
+
assert.equal(line.progress.label, 'Ingest source-a.md');
|
|
133
|
+
assert.equal(line.progress.taskIndex, 1);
|
|
134
|
+
assert.equal(line.progress.taskTotal, 2);
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
test('aggregateActivity keeps a healthy active task out of the error color when a sibling failed', () => {
|
|
138
|
+
const aggregated = aggregateActivity({
|
|
139
|
+
plan: [
|
|
140
|
+
{ id: 'failed-a', groupId: 'ingest', operation: 'ingest_plan', status: 'failed' },
|
|
141
|
+
{ id: 'running-b', groupId: 'ingest', operation: 'ingest_plan', status: 'running', activityKey: 'activity-b' },
|
|
142
|
+
],
|
|
143
|
+
activities: [{
|
|
144
|
+
key: 'activity-b',
|
|
145
|
+
status: 'running',
|
|
146
|
+
terminal: false,
|
|
147
|
+
progress: { percent: 35, stepId: 'running-b', label: 'Ingest application-orea.md', detail: 'LLM running' },
|
|
148
|
+
}],
|
|
149
|
+
}, []);
|
|
150
|
+
|
|
151
|
+
const line = aggregated.lines[0];
|
|
152
|
+
assert.equal(line.status, '35 %');
|
|
153
|
+
assert.match(line.label, /^\[\.\.\.\]/);
|
|
154
|
+
assert.equal(line.progress.label, 'Ingest application-orea.md');
|
|
155
|
+
});
|
package/src/agent/graph.js
CHANGED
|
@@ -274,6 +274,7 @@ const AgentState = Annotation.Root({
|
|
|
274
274
|
invalidResponseRetries: Annotation({ default: () => 0 }),
|
|
275
275
|
invalidToolCallRetries: Annotation({ default: () => 0 }),
|
|
276
276
|
forceDelegation: Annotation({ default: () => false }),
|
|
277
|
+
terminalToolFailure: Annotation({ default: () => false }),
|
|
277
278
|
});
|
|
278
279
|
|
|
279
280
|
function invalidToolCalls(toolCalls) {
|
|
@@ -365,21 +366,62 @@ export function invalidUserFacingToolNames(content, session) {
|
|
|
365
366
|
return [...new Set([...connected, ...syntactic])].sort();
|
|
366
367
|
}
|
|
367
368
|
|
|
369
|
+
function parseActionJson(text) {
|
|
370
|
+
const cleaned = String(text ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
|
|
371
|
+
if (!cleaned) return null;
|
|
372
|
+
return JSON.parse(cleaned)?.action === true;
|
|
373
|
+
}
|
|
374
|
+
|
|
368
375
|
async function classifyRequestedAction(llm, input, signal) {
|
|
376
|
+
const system = [
|
|
377
|
+
'Classify whether the user explicitly requests a real state-changing action now.',
|
|
378
|
+
'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
|
|
379
|
+
'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
|
|
380
|
+
'Return JSON only: {"action":true} or {"action":false}.',
|
|
381
|
+
].join('\n');
|
|
382
|
+
const messages = [{ role: 'user', content: String(input ?? '') }];
|
|
383
|
+
|
|
384
|
+
// Preferred path: a forced structured tool call, reliable on providers that
|
|
385
|
+
// honour tool_choice. But an OpenAI-compatible gateway (e.g. Albert / gpt-oss)
|
|
386
|
+
// may reject a forced tool_choice or return neither tool_calls nor parsable
|
|
387
|
+
// content. Without a fallback that made EVERY request classify as a non-action
|
|
388
|
+
// (catch → false), so Donna silently stopped delegating in agent mode. Fall
|
|
389
|
+
// back to a plain JSON-text completion, and only give up if both paths fail.
|
|
369
390
|
try {
|
|
391
|
+
const classifier = {
|
|
392
|
+
type: 'function',
|
|
393
|
+
function: {
|
|
394
|
+
name: 'classify_action_request',
|
|
395
|
+
description: 'Classify whether the user explicitly requests a real state-changing action now.',
|
|
396
|
+
parameters: {
|
|
397
|
+
type: 'object',
|
|
398
|
+
additionalProperties: false,
|
|
399
|
+
properties: { action: { type: 'boolean' } },
|
|
400
|
+
required: ['action'],
|
|
401
|
+
},
|
|
402
|
+
},
|
|
403
|
+
};
|
|
370
404
|
const result = await llm.completeWithTools({
|
|
371
|
-
system
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
'Return JSON only: {"action":true} or {"action":false}.',
|
|
376
|
-
].join('\n'),
|
|
377
|
-
tools: [],
|
|
378
|
-
messages: [{ role: 'user', content: String(input ?? '') }],
|
|
405
|
+
system,
|
|
406
|
+
tools: [classifier],
|
|
407
|
+
toolChoice: { type: 'function', function: { name: 'classify_action_request' } },
|
|
408
|
+
messages,
|
|
379
409
|
signal,
|
|
380
410
|
});
|
|
381
|
-
const
|
|
382
|
-
|
|
411
|
+
const call = (result?.tool_calls ?? []).find((item) => item?.function?.name === 'classify_action_request');
|
|
412
|
+
if (call) {
|
|
413
|
+
const parsed = JSON.parse(call.function.arguments ?? '{}')?.action;
|
|
414
|
+
if (typeof parsed === 'boolean') return parsed;
|
|
415
|
+
}
|
|
416
|
+
const fromText = parseActionJson(result?.content);
|
|
417
|
+
if (fromText !== null) return fromText;
|
|
418
|
+
} catch {
|
|
419
|
+
// Fall through to the toolless path below.
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
try {
|
|
423
|
+
const result = await llm.completeWithTools({ system, tools: [], messages, signal });
|
|
424
|
+
return parseActionJson(result?.content) === true;
|
|
383
425
|
} catch {
|
|
384
426
|
return false;
|
|
385
427
|
}
|
|
@@ -1327,6 +1369,7 @@ export function createAgentGraph(options = {}) {
|
|
|
1327
1369
|
async function toolExecutorNode(state) {
|
|
1328
1370
|
const toolCalls = state.pendingToolCalls ?? [];
|
|
1329
1371
|
const toolResultMessages = [];
|
|
1372
|
+
let terminalFailure = null;
|
|
1330
1373
|
|
|
1331
1374
|
for (const call of toolCalls) {
|
|
1332
1375
|
const resolved = resolveToolCallName(state.session.mcp, call.function.name, INTERNAL_TOOL_SERVERS);
|
|
@@ -1415,6 +1458,12 @@ export function createAgentGraph(options = {}) {
|
|
|
1415
1458
|
resultText = JSON.stringify(result, null, 2);
|
|
1416
1459
|
} else if (server === 'runtime') {
|
|
1417
1460
|
resultText = await handleRuntimeControlTool(state.session, tool, args);
|
|
1461
|
+
if (tool === 'delegate' && /^Runtime control error \(delegate\):/i.test(resultText)) {
|
|
1462
|
+
terminalFailure = resultText
|
|
1463
|
+
.replace(/^Runtime control error \(delegate\):\s*/i, '')
|
|
1464
|
+
.replace(/^Delegation failed during objective_resolution:\s*/i, '');
|
|
1465
|
+
ok = false;
|
|
1466
|
+
}
|
|
1418
1467
|
} else if (server !== 'shell') {
|
|
1419
1468
|
await awaitRunApproval(state.session, { runId, tool: toolName });
|
|
1420
1469
|
await awaitToolApproval(state.session, {
|
|
@@ -1510,17 +1559,36 @@ export function createAgentGraph(options = {}) {
|
|
|
1510
1559
|
tool_call_id: call.id,
|
|
1511
1560
|
content: boundedResult,
|
|
1512
1561
|
});
|
|
1562
|
+
if (terminalFailure) break;
|
|
1513
1563
|
}
|
|
1514
1564
|
|
|
1565
|
+
if (terminalFailure) {
|
|
1566
|
+
const response = `Action non lancée : ${terminalFailure}`;
|
|
1567
|
+
emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: response });
|
|
1568
|
+
return {
|
|
1569
|
+
response,
|
|
1570
|
+
messages: toolResultMessages,
|
|
1571
|
+
pendingToolCalls: null,
|
|
1572
|
+
forceDelegation: false,
|
|
1573
|
+
terminalToolFailure: true,
|
|
1574
|
+
invalidToolCallRetries: 0,
|
|
1575
|
+
invalidResponseRetries: 0,
|
|
1576
|
+
};
|
|
1577
|
+
}
|
|
1515
1578
|
return {
|
|
1516
1579
|
messages: toolResultMessages,
|
|
1517
1580
|
pendingToolCalls: null,
|
|
1518
1581
|
forceDelegation: false,
|
|
1519
1582
|
invalidToolCallRetries: 0,
|
|
1520
1583
|
invalidResponseRetries: 0,
|
|
1584
|
+
terminalToolFailure: false,
|
|
1521
1585
|
};
|
|
1522
1586
|
}
|
|
1523
1587
|
|
|
1588
|
+
function routeToolExecutor(state) {
|
|
1589
|
+
return state.terminalToolFailure ? END : 'orchestrator';
|
|
1590
|
+
}
|
|
1591
|
+
|
|
1524
1592
|
function routeOrchestrator(state) {
|
|
1525
1593
|
if (state.pendingToolCalls?.length > 0) return 'tool_executor';
|
|
1526
1594
|
if (state.retryWithoutTool) return 'orchestrator';
|
|
@@ -1534,7 +1602,7 @@ export function createAgentGraph(options = {}) {
|
|
|
1534
1602
|
.addNode('tool_executor', toolExecutorNode)
|
|
1535
1603
|
.addEdge(START, 'orchestrator')
|
|
1536
1604
|
.addConditionalEdges('orchestrator', routeOrchestrator)
|
|
1537
|
-
.
|
|
1605
|
+
.addConditionalEdges('tool_executor', routeToolExecutor)
|
|
1538
1606
|
.compile();
|
|
1539
1607
|
|
|
1540
1608
|
// LangGraph's default recursionLimit is 25 super-steps. Each tool round
|
package/src/agent/graph.test.js
CHANGED
|
@@ -25,7 +25,9 @@ test('Donna cannot answer an explicit action with manual instructions instead of
|
|
|
25
25
|
runtime: { url: 'http://runtime.test' },
|
|
26
26
|
llm: {
|
|
27
27
|
async completeWithTools({ tools }) {
|
|
28
|
-
if (tools.
|
|
28
|
+
if (tools.some((tool) => tool.function?.name === 'classify_action_request')) {
|
|
29
|
+
return { content: null, message: { role: 'assistant', content: null }, tool_calls: [{ id: 'classify', type: 'function', function: { name: 'classify_action_request', arguments: '{"action":true}' } }] };
|
|
30
|
+
}
|
|
29
31
|
mainCalls += 1;
|
|
30
32
|
if (mainCalls === 1) {
|
|
31
33
|
return {
|
|
@@ -910,8 +912,8 @@ test('forced delegation is cleared after one valid tool call and does not loop',
|
|
|
910
912
|
commands: ['status'],
|
|
911
913
|
llm: {
|
|
912
914
|
async completeWithTools({ toolChoice, tools }) {
|
|
913
|
-
if (tools.
|
|
914
|
-
return { content:
|
|
915
|
+
if (tools.some((tool) => tool.function?.name === 'classify_action_request')) {
|
|
916
|
+
return { content: null, message: { role: 'assistant', content: null }, tool_calls: [{ id: 'classify', type: 'function', function: { name: 'classify_action_request', arguments: '{"action":true}' } }] };
|
|
915
917
|
}
|
|
916
918
|
calls += 1;
|
|
917
919
|
choices.push(toolChoice);
|
|
@@ -946,6 +948,44 @@ test('forced delegation is cleared after one valid tool call and does not loop',
|
|
|
946
948
|
}
|
|
947
949
|
});
|
|
948
950
|
|
|
951
|
+
test('a rejected runtime delegation is terminal and never loops', async () => {
|
|
952
|
+
const originalFetch = globalThis.fetch;
|
|
953
|
+
globalThis.fetch = async () => ({
|
|
954
|
+
ok: false,
|
|
955
|
+
status: 422,
|
|
956
|
+
json: async () => ({
|
|
957
|
+
error: 'Delegation failed during objective_resolution: No orchestrable capability is currently available.',
|
|
958
|
+
}),
|
|
959
|
+
});
|
|
960
|
+
let calls = 0;
|
|
961
|
+
const session = sessionBase({
|
|
962
|
+
runtime: { url: 'http://runtime.test' },
|
|
963
|
+
llm: {
|
|
964
|
+
async completeWithTools() {
|
|
965
|
+
calls += 1;
|
|
966
|
+
return {
|
|
967
|
+
content: null,
|
|
968
|
+
message: { role: 'assistant', content: null },
|
|
969
|
+
tool_calls: [{
|
|
970
|
+
id: 'delegate-failure',
|
|
971
|
+
type: 'function',
|
|
972
|
+
function: { name: 'runtime__delegate', arguments: '{"objective":"Lance ingestion"}' },
|
|
973
|
+
}],
|
|
974
|
+
};
|
|
975
|
+
},
|
|
976
|
+
},
|
|
977
|
+
});
|
|
978
|
+
|
|
979
|
+
try {
|
|
980
|
+
const result = await createAgentGraph().invoke({ input: 'lance ingestion', session });
|
|
981
|
+
assert.equal(calls, 1);
|
|
982
|
+
assert.equal(result.response, 'Action non lancée : No orchestrable capability is currently available.');
|
|
983
|
+
assert.equal(result.terminalToolFailure, true);
|
|
984
|
+
} finally {
|
|
985
|
+
globalThis.fetch = originalFetch;
|
|
986
|
+
}
|
|
987
|
+
});
|
|
988
|
+
|
|
949
989
|
// Guard: the system prompt must never show a connected tool's bare name
|
|
950
990
|
// outside its qualified server__tool form. Bare mentions are what teach the
|
|
951
991
|
// model to emit unqualified tool calls (the cme_status incident). The bare
|