@dotdrelle/wiki-manager 0.15.94 → 0.15.97
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +39 -28
- package/package.json +2 -2
- package/src/agent/graph.js +61 -21
- package/src/agent/graph.test.js +71 -1
- package/src/cli/wiki-manager.js +35 -17
- package/src/core/agentEvents.js +21 -4
- package/src/core/agentEvents.test.js +34 -0
- package/src/core/buildInfo.json +2 -2
- package/src/core/googleGrants.js +0 -3
- package/src/core/json.js +9 -0
- package/src/core/mcp.js +1 -1
- package/src/core/plan.js +0 -4
- package/src/core/progressNotes.js +0 -4
- package/src/core/skillChainView.js +3 -1
- package/src/core/toolLoop.js +56 -2
- package/src/core/toolLoop.test.js +35 -4
- package/src/orchestrator/agentRegistry.js +1 -3
- package/src/orchestrator/dependencyResolver.js +0 -3
- package/src/orchestrator/planValidator.js +1 -3
- package/src/orchestrator/providers/runtimeProvider.js +0 -14
- package/src/orchestrator/taskStatuses.js +8 -0
- package/src/runtime/client.js +0 -16
- package/src/runtime/controlDrain.js +6 -3
- package/src/runtime/deltaCoalescer.js +53 -0
- package/src/runtime/deltaCoalescer.test.js +56 -0
- package/src/runtime/loginPage.js +0 -3
- package/src/runtime/loginSession.js +1 -4
- package/src/runtime/runner.js +17 -3
- package/src/runtime/runner.test.js +32 -1
- package/src/runtime/server.js +172 -14
- package/src/runtime/server.test.js +206 -0
- package/src/runtime/skillRun.js +2 -2
- package/src/runtime/skillRun.test.js +7 -2
- package/src/shell/repl.js +7 -3
- package/src/orchestrator/.fuse_hidden0000001c00000001 +0 -316
package/src/runtime/server.js
CHANGED
|
@@ -7,6 +7,7 @@ import { validateContractInDev } from '../contracts/schemas.js';
|
|
|
7
7
|
import { runtimeTokenFromEnv } from './auth.js';
|
|
8
8
|
import { controlMessage } from './controlMessages.js';
|
|
9
9
|
import { tasksAwaitingApproval } from '../orchestrator/dependencyResolver.js';
|
|
10
|
+
import { isActive, isCancelled, isFailed, isSuccessful } from '../orchestrator/taskStatuses.js';
|
|
10
11
|
import { approvalClassForTask } from '../orchestrator/approvalPolicy.js';
|
|
11
12
|
import { RUNTIME_SHUTDOWN_ABORT_REASON } from '../orchestrator/dispatcher.js';
|
|
12
13
|
import { matchSkillInvocation } from '../core/skillInvocation.js';
|
|
@@ -502,7 +503,7 @@ export function startRuntimeServer({
|
|
|
502
503
|
}
|
|
503
504
|
if (request.method === 'POST' && url.pathname === '/turn') {
|
|
504
505
|
const { body, context } = await resolveBodyContext(request, url);
|
|
505
|
-
|
|
506
|
+
let input = String(body.input ?? body.prompt ?? '').trim();
|
|
506
507
|
if (!input) {
|
|
507
508
|
sendJson(response, 400, { error: 'Missing input.' });
|
|
508
509
|
return;
|
|
@@ -529,6 +530,16 @@ export function startRuntimeServer({
|
|
|
529
530
|
}
|
|
530
531
|
return;
|
|
531
532
|
}
|
|
533
|
+
// A run/job status question must NEVER surface the raw system text in
|
|
534
|
+
// the thread: the runtime collects the facts and hands them to Donna,
|
|
535
|
+
// who synthesizes them in the session language. `readOnlyChat` so the
|
|
536
|
+
// turn is a conversation, not a control decision — and the facts are
|
|
537
|
+
// supplied here so the model cannot mistake the runtime run id for a
|
|
538
|
+
// production job id ("job not found").
|
|
539
|
+
if (asksForRunStatus(input)) {
|
|
540
|
+
input = runtimeStatusSynthesisPrompt(input, controlStatus(context, store));
|
|
541
|
+
readOnlyChat = true;
|
|
542
|
+
}
|
|
532
543
|
if (context.running && !readOnlyChat) {
|
|
533
544
|
// Agent-mode message while a run is active. Classify once: control
|
|
534
545
|
// verbs and new tasks go to the control lane, plain conversation is
|
|
@@ -538,7 +549,13 @@ export function startRuntimeServer({
|
|
|
538
549
|
llm: context?.session?.llm,
|
|
539
550
|
session: context?.session,
|
|
540
551
|
});
|
|
541
|
-
if (classification.kind
|
|
552
|
+
if (classification.kind === 'observe') {
|
|
553
|
+
// An observation is system facts, not a control action: Donna gets
|
|
554
|
+
// them and answers in the session language, never a raw English
|
|
555
|
+
// line pushed into the thread.
|
|
556
|
+
input = runtimeStatusSynthesisPrompt(input, controlStatus(context, store));
|
|
557
|
+
readOnlyChat = true;
|
|
558
|
+
} else if (classification.kind !== 'converse') {
|
|
542
559
|
const result = await handleControlMessage(context, store, input, {
|
|
543
560
|
intent: body.intent,
|
|
544
561
|
startNextControlRequest,
|
|
@@ -547,8 +564,9 @@ export function startRuntimeServer({
|
|
|
547
564
|
});
|
|
548
565
|
sendJson(response, result.statusCode, result.body);
|
|
549
566
|
return;
|
|
567
|
+
} else {
|
|
568
|
+
readOnlyChat = true;
|
|
550
569
|
}
|
|
551
|
-
readOnlyChat = true;
|
|
552
570
|
}
|
|
553
571
|
if (typeof turn !== 'function') {
|
|
554
572
|
sendJson(response, 501, { error: 'Runtime interactive turns are unavailable.' });
|
|
@@ -566,7 +584,10 @@ export function startRuntimeServer({
|
|
|
566
584
|
llm: context?.session?.llm,
|
|
567
585
|
session: context?.session,
|
|
568
586
|
});
|
|
569
|
-
if (classification.kind
|
|
587
|
+
if (classification.kind === 'observe') {
|
|
588
|
+
input = runtimeStatusSynthesisPrompt(input, controlStatus(context, store));
|
|
589
|
+
readOnlyChat = true;
|
|
590
|
+
} else if (classification.kind !== 'converse') {
|
|
570
591
|
const result = await handleControlMessage(context, store, input, {
|
|
571
592
|
intent: body.intent,
|
|
572
593
|
startNextControlRequest,
|
|
@@ -853,12 +874,24 @@ export function startRuntimeServer({
|
|
|
853
874
|
if (context?.session) {
|
|
854
875
|
context.session._runSkillWithinRun = async (skillName, args = {}, metadata = {}) => {
|
|
855
876
|
const skill = findSkill(context.session, skillName);
|
|
856
|
-
if (!skill)
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
877
|
+
if (!skill) {
|
|
878
|
+
const available = listSkills(context.session).map((item) => item.name);
|
|
879
|
+
/*
|
|
880
|
+
A skill the model GUESSED is recoverable, not terminal: the observed
|
|
881
|
+
failure was `/diagnose` (leading slash copied from the catalogue) →
|
|
882
|
+
skill_not_found → the terminal path stripped the tools from the
|
|
883
|
+
synthesis turn → the model wrote `runtime__delegate{...}` as plain
|
|
884
|
+
text and the turn did nothing. An explicitly user-named missing skill
|
|
885
|
+
stays terminal: there is nothing to fall back to.
|
|
886
|
+
*/
|
|
887
|
+
return {
|
|
888
|
+
ok: false,
|
|
889
|
+
terminal: metadata.selectionKind === 'explicit_name',
|
|
890
|
+
code: 'skill_not_found',
|
|
891
|
+
message: `No skill named "${skillName}". Pass the exact name without a leading slash (${available.join(', ') || 'none'}), or delegate the objective with runtime__delegate.`,
|
|
892
|
+
availableSkills: available,
|
|
893
|
+
};
|
|
894
|
+
}
|
|
862
895
|
try {
|
|
863
896
|
const idempotencyKey = metadata.idempotencyKey
|
|
864
897
|
? String(metadata.idempotencyKey)
|
|
@@ -1147,22 +1180,109 @@ export function runtimeState(context, store, { workspace = null, session = null
|
|
|
1147
1180
|
// own history) so those replies surface. The log is a superset of the
|
|
1148
1181
|
// canonical run conversation, so run rendering is unaffected.
|
|
1149
1182
|
conversation: reduceAgentEvents(store.listEvents({ workspace })).conversation,
|
|
1150
|
-
|
|
1183
|
+
// `context.running` keeps the process alive while the scheduler waits for an
|
|
1184
|
+
// approval, but the run is then NOT running — the reducer already says
|
|
1185
|
+
// `pending_approval`. Prefer it over the blanket override.
|
|
1186
|
+
status: context?.running
|
|
1187
|
+
? (state.status === 'pending_approval' ? 'pending_approval' : 'running')
|
|
1188
|
+
: state.status ?? 'idle',
|
|
1151
1189
|
running: Boolean(context?.running),
|
|
1152
1190
|
runId: context?.currentRunId ?? state.runId ?? null,
|
|
1153
1191
|
workspace: context?.currentRunWorkspace ?? context?.workspace ?? state.workspace ?? workspace ?? null,
|
|
1154
1192
|
};
|
|
1155
1193
|
}
|
|
1156
1194
|
|
|
1195
|
+
// The live figures of the activity the run is on, in one line: percent, plan
|
|
1196
|
+
// step, build batch, instruction count and the stabilize counters. A status
|
|
1197
|
+
// that only named the current step could not tell 5% from 95%, nor what the
|
|
1198
|
+
// running batch had actually done.
|
|
1199
|
+
function describeActivityProgress(activity) {
|
|
1200
|
+
const progress = activity?.progress ?? {};
|
|
1201
|
+
const bits = [];
|
|
1202
|
+
if (Number.isFinite(Number(progress.percent))) bits.push(`${Number(progress.percent)}%`);
|
|
1203
|
+
if (progress.stepIndex != null && progress.stepTotal != null) bits.push(`step ${progress.stepIndex}/${progress.stepTotal}`);
|
|
1204
|
+
if (progress.batchIndex != null && progress.batchCount != null) bits.push(`batch ${Number(progress.batchIndex) + 1}/${progress.batchCount}`);
|
|
1205
|
+
if (progress.instructionCount != null) bits.push(`${progress.instructionCount} instruction${Number(progress.instructionCount) > 1 ? 's' : ''}`);
|
|
1206
|
+
const stabilize = [progress.stabilizeKept, progress.stabilizeMerged, progress.stabilizeInserted, progress.stabilizeRemoved];
|
|
1207
|
+
if (stabilize.some((value) => value != null)) {
|
|
1208
|
+
bits.push(`kept ${progress.stabilizeKept ?? 0}, merged ${progress.stabilizeMerged ?? 0}, inserted ${progress.stabilizeInserted ?? 0}, removed ${progress.stabilizeRemoved ?? 0}`);
|
|
1209
|
+
}
|
|
1210
|
+
const detail = progress.detail && !bits.includes(String(progress.detail)) ? String(progress.detail) : null;
|
|
1211
|
+
return { bits: bits.join(' · '), detail, label: activity?.label ?? null };
|
|
1212
|
+
}
|
|
1213
|
+
|
|
1214
|
+
// The facts a status answer is built from, rendered server-side ONCE. A status
|
|
1215
|
+
// question is answered by Donna, never by pushing this text into the thread:
|
|
1216
|
+
// the runtime supplies the figures, she phrases them in the session language.
|
|
1217
|
+
function runtimeStatusFacts(status) {
|
|
1218
|
+
const plan = Array.isArray(status.plan) ? status.plan : [];
|
|
1219
|
+
const approvals = Array.isArray(status.approvals) ? status.approvals : [];
|
|
1220
|
+
const queue = Array.isArray(status.controlQueue) ? status.controlQueue : [];
|
|
1221
|
+
const activities = Array.isArray(status.activities)
|
|
1222
|
+
? status.activities
|
|
1223
|
+
: Object.values(status.activities ?? {});
|
|
1224
|
+
const lines = [
|
|
1225
|
+
`Runtime status: ${status.status ?? 'idle'}`,
|
|
1226
|
+
`Workspace: ${status.workspace ?? '-'}`,
|
|
1227
|
+
];
|
|
1228
|
+
if (status.runId) lines.push(`Run id: ${status.runId}`);
|
|
1229
|
+
for (const activity of activities.filter((entry) => !entry?.terminal).slice(0, 8)) {
|
|
1230
|
+
const info = describeActivityProgress(activity);
|
|
1231
|
+
const detail = [info.bits, info.detail].filter(Boolean).join(' · ');
|
|
1232
|
+
lines.push(
|
|
1233
|
+
`Activity: ${info.label ?? activity.label ?? activity.id ?? '-'} — ${activity.status ?? '-'}${detail ? ` (${detail})` : ''}`,
|
|
1234
|
+
);
|
|
1235
|
+
}
|
|
1236
|
+
for (const [index, step] of plan.slice(0, 60).entries()) {
|
|
1237
|
+
lines.push(`Task ${step.step ?? index + 1}: ${step.status ?? 'pending'} - ${step.description ?? step.label ?? step.id ?? 'step'}`);
|
|
1238
|
+
}
|
|
1239
|
+
for (const approval of approvals.filter((entry) => entry.status === 'pending_approval')) {
|
|
1240
|
+
lines.push(`Pending approval: ${approval.reason ?? approval.taskId ?? approval.id ?? '-'}`);
|
|
1241
|
+
}
|
|
1242
|
+
for (const item of queue.filter((entry) => entry.status === 'queued')) {
|
|
1243
|
+
lines.push(`Queued: ${item.label ?? item.input ?? item.id ?? '-'}`);
|
|
1244
|
+
}
|
|
1245
|
+
return lines.join('\n');
|
|
1246
|
+
}
|
|
1247
|
+
|
|
1248
|
+
function runtimeStatusSynthesisPrompt(asked, status) {
|
|
1249
|
+
return [
|
|
1250
|
+
'The runtime facts below are the authoritative status the system just collected (this is a runtime run, not a production job — do not look up a job id).',
|
|
1251
|
+
`User question: ${asked}`,
|
|
1252
|
+
'Answer in the session language with a concise, natural status: name the requested target first, then progress, blockers (pending approvals), queued items and the next step.',
|
|
1253
|
+
'Keep every figure (percent, step, batch, instruction and stabilize counts) and every task status accurate; never invent, drop or round away a figure. Do not paste the fact block verbatim; summarize it into prose.',
|
|
1254
|
+
'',
|
|
1255
|
+
'Runtime facts:',
|
|
1256
|
+
runtimeStatusFacts(status),
|
|
1257
|
+
].join('\n');
|
|
1258
|
+
}
|
|
1259
|
+
|
|
1157
1260
|
function explainControlState(status) {
|
|
1158
1261
|
const plan = Array.isArray(status.plan) ? status.plan : [];
|
|
1262
|
+
const activities = Array.isArray(status.activities)
|
|
1263
|
+
? status.activities
|
|
1264
|
+
: Object.values(status.activities ?? {});
|
|
1265
|
+
// A run whose only outstanding work is a human decision is not "running":
|
|
1266
|
+
// `status.running` mirrors the process, which stays alive while the scheduler
|
|
1267
|
+
// waits. Mirrors the reducer's rule — pending_approval, or a pending approval
|
|
1268
|
+
// with no step actually executing.
|
|
1269
|
+
const approvals = Array.isArray(status.approvals) ? status.approvals : [];
|
|
1270
|
+
const pendingApproval = approvals.find((approval) => approval.status === 'pending_approval');
|
|
1271
|
+
const awaitingApproval = pendingApproval
|
|
1272
|
+
&& (status.status === 'pending_approval' || !plan.some((step) => isActive(step?.status)));
|
|
1273
|
+
if (awaitingApproval) {
|
|
1274
|
+
return `Runtime is waiting for approval: ${pendingApproval.reason ?? pendingApproval.id}.`;
|
|
1275
|
+
}
|
|
1159
1276
|
if (status.running) {
|
|
1160
1277
|
const runningStep = plan.find((step) => step.status === 'running');
|
|
1278
|
+
const activity = activities.find((entry) => !entry?.terminal) ?? activities[0] ?? null;
|
|
1279
|
+
const info = activity ? describeActivityProgress(activity) : { bits: '', detail: null, label: null };
|
|
1280
|
+
const detailText = [info.bits, info.detail].filter(Boolean).join(' · ');
|
|
1281
|
+
const suffix = detailText ? ` (${detailText})` : '';
|
|
1161
1282
|
return runningStep
|
|
1162
|
-
? `Runtime run is active. Current step: ${runningStep.description ?? runningStep.label ?? runningStep.step}
|
|
1163
|
-
:
|
|
1283
|
+
? `Runtime run is active. Current step: ${runningStep.description ?? runningStep.label ?? runningStep.step}.${suffix}`
|
|
1284
|
+
: `Runtime run is active. No current plan step is available yet.${suffix}`;
|
|
1164
1285
|
}
|
|
1165
|
-
const pendingApproval = status.approvals.find((approval) => approval.status === 'pending_approval');
|
|
1166
1286
|
if (pendingApproval) {
|
|
1167
1287
|
return `Runtime is waiting for approval: ${pendingApproval.reason ?? pendingApproval.id}.`;
|
|
1168
1288
|
}
|
|
@@ -1173,6 +1293,15 @@ function explainControlState(status) {
|
|
|
1173
1293
|
if (plan.some((step) => step.status === 'pending')) {
|
|
1174
1294
|
return 'Runtime is idle with pending plan steps visible from the last run.';
|
|
1175
1295
|
}
|
|
1296
|
+
// Idle at the end of a run: say what the last run did, not just "idle" — that
|
|
1297
|
+
// is the question the operator actually asks when the thread goes quiet.
|
|
1298
|
+
const failed = plan.filter((step) => isFailed(step.status) || isCancelled(step.status)).length;
|
|
1299
|
+
const done = plan.filter((step) => isSuccessful(step.status)).length;
|
|
1300
|
+
if (plan.length > 0) {
|
|
1301
|
+
return failed > 0
|
|
1302
|
+
? `Runtime is idle. Last run: ${done}/${plan.length} task(s) succeeded, ${failed} failed or cancelled.`
|
|
1303
|
+
: `Runtime is idle. Last run: ${done}/${plan.length} task(s) succeeded.`;
|
|
1304
|
+
}
|
|
1176
1305
|
return 'Runtime is idle.';
|
|
1177
1306
|
}
|
|
1178
1307
|
|
|
@@ -1632,6 +1761,27 @@ function rejectPlanPatch(context, store, patchId, reason) {
|
|
|
1632
1761
|
};
|
|
1633
1762
|
}
|
|
1634
1763
|
|
|
1764
|
+
// A question about the run/job currently executing. The free-text form is
|
|
1765
|
+
// deliberately narrow — a status word AND a run/job noun — so it never hijacks
|
|
1766
|
+
// an ordinary "explain how X works" question. Such a question must be answered
|
|
1767
|
+
// by the runtime itself: left to the model, a runtime runId was mistaken for a
|
|
1768
|
+
// production job id and reported as "not found", and a read-only chat turn had
|
|
1769
|
+
// no runtime status tool.
|
|
1770
|
+
//
|
|
1771
|
+
// The reserved built-in `/status` is ALWAYS a runtime status, in every surface:
|
|
1772
|
+
// `RESERVED_SLASH_COMMANDS` keeps the homonymous workspace skill out of
|
|
1773
|
+
// `matchSkillInvocation`, but without this branch `/turn` still handed the
|
|
1774
|
+
// literal command to the model, which ran the skill (English "status" output) or
|
|
1775
|
+
// an unrelated review instead of reporting anything. Serve types `/status` into
|
|
1776
|
+
// this endpoint; the ShellUI answers it locally.
|
|
1777
|
+
function asksForRunStatus(input) {
|
|
1778
|
+
const text = String(input ?? '').trim();
|
|
1779
|
+
if (/^\/status(?:\s|$)/i.test(text)) return true;
|
|
1780
|
+
const statusWord = /\b(status|statut|progression|progress|avancement|o[uù] en est|o[uù] en sont)\b/i;
|
|
1781
|
+
const runNoun = /\b(job|run|t[aâ]che|task|build|ingest|pipeline|export|polish|traitement)\b/i;
|
|
1782
|
+
return statusWord.test(text) && runNoun.test(text);
|
|
1783
|
+
}
|
|
1784
|
+
|
|
1635
1785
|
// Classifier for the control lane's free-text messages. The classification is
|
|
1636
1786
|
// LLM-backed: the only deterministic matches left are the runtime's own
|
|
1637
1787
|
// control verbs (cancel, an explicit "later/queue", status and plan-change
|
|
@@ -1668,6 +1818,14 @@ async function classifyControlMessage(input, status, { forcedIntent = null, llm
|
|
|
1668
1818
|
if (/\b(o[uù] en es[t-]|status|statut|progress|progression|logs?|explique|explain|inspect|show|montre|quoi de neuf)\b/i.test(lower)) {
|
|
1669
1819
|
return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request' };
|
|
1670
1820
|
}
|
|
1821
|
+
// A bare "yes" answers the runtime's own last prompt (the launch
|
|
1822
|
+
// acknowledgement used to end on "check progress or cancel?"). While a run is
|
|
1823
|
+
// active, the only thing the runtime can act on is a status check: treating
|
|
1824
|
+
// the word as ordinary conversation made the read-only chat fallback lecture
|
|
1825
|
+
// the user about switching modes instead of answering.
|
|
1826
|
+
if (status.running && /^\s*(oui|yes|yep|ok|okay|vas[- ]?y|d'accord|daccord|entendu)\b/i.test(lower)) {
|
|
1827
|
+
return { kind: 'observe', confidence: 0.7, reason: 'confirmation_of_runtime_prompt' };
|
|
1828
|
+
}
|
|
1671
1829
|
if (status.running && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|apr[eè]s|before|after|chaque|each|plan|step|t[aâ]che)\b/i.test(lower)) {
|
|
1672
1830
|
return { kind: 'modify_run', confidence: 0.78, reason: 'active_run_change_request' };
|
|
1673
1831
|
}
|
|
@@ -1733,6 +1733,212 @@ test('POST /turn keeps informational skill and build questions conversational',
|
|
|
1733
1733
|
}
|
|
1734
1734
|
});
|
|
1735
1735
|
|
|
1736
|
+
test('POST /turn hands a run status question to Donna with the runtime facts', async (t) => {
|
|
1737
|
+
const session = { workspace: 'acme', controlQueue: [] };
|
|
1738
|
+
const context = { workspace: 'acme', session, running: true, currentAbortController: null };
|
|
1739
|
+
const status = {
|
|
1740
|
+
status: 'running',
|
|
1741
|
+
running: true,
|
|
1742
|
+
plan: [{ step: 1, description: 'Build TechSections', status: 'running' }],
|
|
1743
|
+
queue: [],
|
|
1744
|
+
controlQueue: [],
|
|
1745
|
+
approvals: [],
|
|
1746
|
+
conversation: [],
|
|
1747
|
+
};
|
|
1748
|
+
let turns = 0;
|
|
1749
|
+
let turnInput = '';
|
|
1750
|
+
let turnMode = null;
|
|
1751
|
+
let handle;
|
|
1752
|
+
try {
|
|
1753
|
+
handle = await startRuntimeServer({
|
|
1754
|
+
host: '127.0.0.1', port: 0,
|
|
1755
|
+
store: { dbPath: ':memory:', getState: () => status, listEvents: () => [] },
|
|
1756
|
+
getContext: async () => context,
|
|
1757
|
+
run: async () => new Promise(() => {}),
|
|
1758
|
+
turn: async (_context, options) => { turns += 1; turnInput = options.input; turnMode = options.mode; return { ok: true }; },
|
|
1759
|
+
});
|
|
1760
|
+
} catch (err) {
|
|
1761
|
+
if (err?.code === 'EPERM') { t.skip('network listen is not permitted in this sandbox'); return; }
|
|
1762
|
+
throw err;
|
|
1763
|
+
}
|
|
1764
|
+
try {
|
|
1765
|
+
// System facts never reach the thread as raw text: the runtime supplies
|
|
1766
|
+
// them to Donna, who synthesizes the answer. The facts also prevent the
|
|
1767
|
+
// model mistaking the runtime runId for a production job id.
|
|
1768
|
+
const response = await fetch(`http://127.0.0.1:${handle.port}/turn?workspace=acme`, {
|
|
1769
|
+
method: 'POST', headers: { 'content-type': 'application/json' },
|
|
1770
|
+
body: JSON.stringify({ input: 'donne le status du job en cours', mode: 'agent' }),
|
|
1771
|
+
});
|
|
1772
|
+
const body = await response.json();
|
|
1773
|
+
assert.equal(response.status, 202);
|
|
1774
|
+
assert.equal(body.kind, 'turn');
|
|
1775
|
+
// The turn is dispatched asynchronously after the 202.
|
|
1776
|
+
await new Promise((resolve) => setTimeout(resolve, 25));
|
|
1777
|
+
assert.equal(turns, 1);
|
|
1778
|
+
assert.equal(turnMode, 'chat');
|
|
1779
|
+
assert.match(turnInput, /Build TechSections/);
|
|
1780
|
+
assert.match(turnInput, /runtime run, not a production job/i);
|
|
1781
|
+
} finally {
|
|
1782
|
+
context.currentAbortController?.abort();
|
|
1783
|
+
await handle.close();
|
|
1784
|
+
}
|
|
1785
|
+
});
|
|
1786
|
+
|
|
1787
|
+
test('a run blocked on approval is described as waiting, not as running', async (t) => {
|
|
1788
|
+
const session = { workspace: 'acme', controlQueue: [] };
|
|
1789
|
+
const context = { workspace: 'acme', session, running: true, currentAbortController: null };
|
|
1790
|
+
const status = {
|
|
1791
|
+
status: 'pending_approval',
|
|
1792
|
+
running: true,
|
|
1793
|
+
plan: [{ step: 1, description: 'Rebuild the concepts', status: 'pending_approval' }],
|
|
1794
|
+
queue: [],
|
|
1795
|
+
controlQueue: [],
|
|
1796
|
+
approvals: [{ id: 'a1', status: 'pending_approval', reason: 'a mutating task needs approval' }],
|
|
1797
|
+
conversation: [],
|
|
1798
|
+
};
|
|
1799
|
+
let turns = 0;
|
|
1800
|
+
let turnInput = '';
|
|
1801
|
+
let handle;
|
|
1802
|
+
try {
|
|
1803
|
+
handle = await startRuntimeServer({
|
|
1804
|
+
host: '127.0.0.1', port: 0,
|
|
1805
|
+
store: { dbPath: ':memory:', getState: () => status, listEvents: () => [] },
|
|
1806
|
+
getContext: async () => context,
|
|
1807
|
+
run: async () => new Promise(() => {}),
|
|
1808
|
+
turn: async (_context, options) => { turns += 1; turnInput = options.input; return { ok: true }; },
|
|
1809
|
+
});
|
|
1810
|
+
} catch (err) {
|
|
1811
|
+
if (err?.code === 'EPERM') { t.skip('network listen is not permitted in this sandbox'); return; }
|
|
1812
|
+
throw err;
|
|
1813
|
+
}
|
|
1814
|
+
try {
|
|
1815
|
+
// The controller (`explainControlState`) must not call a pending approval
|
|
1816
|
+
// "running": the scheduler keeps `context.running` true while it waits.
|
|
1817
|
+
const control = await fetch(`http://127.0.0.1:${handle.port}/control?workspace=acme`, {
|
|
1818
|
+
method: 'POST', headers: { 'content-type': 'application/json' },
|
|
1819
|
+
body: JSON.stringify({ action: 'explain' }),
|
|
1820
|
+
});
|
|
1821
|
+
const controlBody = await control.json();
|
|
1822
|
+
assert.match(controlBody.explanation, /waiting for approval/i);
|
|
1823
|
+
assert.doesNotMatch(controlBody.explanation, /is active/i);
|
|
1824
|
+
|
|
1825
|
+
// And the turn hands the same facts to Donna rather than dumping them.
|
|
1826
|
+
const response = await fetch(`http://127.0.0.1:${handle.port}/turn?workspace=acme`, {
|
|
1827
|
+
method: 'POST', headers: { 'content-type': 'application/json' },
|
|
1828
|
+
body: JSON.stringify({ input: 'donne le status du run en cours', mode: 'agent' }),
|
|
1829
|
+
});
|
|
1830
|
+
const body = await response.json();
|
|
1831
|
+
assert.equal(response.status, 202);
|
|
1832
|
+
assert.equal(body.kind, 'turn');
|
|
1833
|
+
await new Promise((resolve) => setTimeout(resolve, 25));
|
|
1834
|
+
assert.equal(turns, 1);
|
|
1835
|
+
assert.match(turnInput, /Pending approval: a mutating task needs approval/);
|
|
1836
|
+
} finally {
|
|
1837
|
+
context.currentAbortController?.abort();
|
|
1838
|
+
await handle.close();
|
|
1839
|
+
}
|
|
1840
|
+
});
|
|
1841
|
+
|
|
1842
|
+
test('POST /turn treats a bare confirmation during a run as a status check', async (t) => {
|
|
1843
|
+
const session = { workspace: 'acme', controlQueue: [] };
|
|
1844
|
+
const context = { workspace: 'acme', session, running: true, currentAbortController: null };
|
|
1845
|
+
const status = {
|
|
1846
|
+
status: 'running',
|
|
1847
|
+
running: true,
|
|
1848
|
+
plan: [{ step: 1, description: 'Rebuild the wiki', status: 'running' }],
|
|
1849
|
+
queue: [],
|
|
1850
|
+
controlQueue: [],
|
|
1851
|
+
approvals: [],
|
|
1852
|
+
conversation: [],
|
|
1853
|
+
};
|
|
1854
|
+
let turns = 0;
|
|
1855
|
+
let turnInput = '';
|
|
1856
|
+
let handle;
|
|
1857
|
+
try {
|
|
1858
|
+
handle = await startRuntimeServer({
|
|
1859
|
+
host: '127.0.0.1', port: 0,
|
|
1860
|
+
store: { dbPath: ':memory:', getState: () => status, listEvents: () => [] },
|
|
1861
|
+
getContext: async () => context,
|
|
1862
|
+
run: async () => new Promise(() => {}),
|
|
1863
|
+
turn: async (_context, options) => { turns += 1; turnInput = options.input; return { ok: true }; },
|
|
1864
|
+
});
|
|
1865
|
+
} catch (err) {
|
|
1866
|
+
if (err?.code === 'EPERM') { t.skip('network listen is not permitted in this sandbox'); return; }
|
|
1867
|
+
throw err;
|
|
1868
|
+
}
|
|
1869
|
+
try {
|
|
1870
|
+
// "oui" answers the launch acknowledgement. It is an observation, so it
|
|
1871
|
+
// reaches Donna with the runtime facts — not a deterministic English line
|
|
1872
|
+
// and not a read-only chat turn that lectures about switching modes.
|
|
1873
|
+
const response = await fetch(`http://127.0.0.1:${handle.port}/turn?workspace=acme`, {
|
|
1874
|
+
method: 'POST', headers: { 'content-type': 'application/json' },
|
|
1875
|
+
body: JSON.stringify({ input: 'oui', mode: 'agent' }),
|
|
1876
|
+
});
|
|
1877
|
+
const body = await response.json();
|
|
1878
|
+
assert.equal(response.status, 202);
|
|
1879
|
+
assert.equal(body.kind, 'turn');
|
|
1880
|
+
await new Promise((resolve) => setTimeout(resolve, 25));
|
|
1881
|
+
assert.equal(turns, 1);
|
|
1882
|
+
assert.match(turnInput, /Runtime facts:/);
|
|
1883
|
+
assert.match(turnInput, /Rebuild the wiki/);
|
|
1884
|
+
} finally {
|
|
1885
|
+
context.currentAbortController?.abort();
|
|
1886
|
+
await handle.close();
|
|
1887
|
+
}
|
|
1888
|
+
});
|
|
1889
|
+
|
|
1890
|
+
test('POST /turn answers the reserved /status command itself, never the homonymous skill', async (t) => {
|
|
1891
|
+
// A workspace skill named `status` exists precisely to prove the built-in
|
|
1892
|
+
// wins: `/status` was handed to the model, which ran that skill (English
|
|
1893
|
+
// output) or an unrelated review instead of reporting anything. Serve types
|
|
1894
|
+
// `/status` into /turn; only `/skills run status` may reach the skill.
|
|
1895
|
+
const root = mkdtempSync(join(tmpdir(), 'runtime-status-builtin-'));
|
|
1896
|
+
mkdirSync(join(root, '.wiki', 'skills'), { recursive: true });
|
|
1897
|
+
writeFileSync(join(root, '.wiki', 'skills', 'status.md'), '---\nname: status\n---\nInspect services.');
|
|
1898
|
+
const session = { workspace: 'acme', workspacePath: root, controlQueue: [] };
|
|
1899
|
+
const context = { workspace: 'acme', session, running: false, currentAbortController: null };
|
|
1900
|
+
const status = {
|
|
1901
|
+
status: 'idle',
|
|
1902
|
+
running: false,
|
|
1903
|
+
plan: [{ step: 1, description: 'Rebuild the wiki', status: 'done' }],
|
|
1904
|
+
queue: [],
|
|
1905
|
+
controlQueue: [],
|
|
1906
|
+
approvals: [],
|
|
1907
|
+
conversation: [],
|
|
1908
|
+
};
|
|
1909
|
+
let turns = 0;
|
|
1910
|
+
let turnInput = '';
|
|
1911
|
+
let handle;
|
|
1912
|
+
try {
|
|
1913
|
+
handle = await startRuntimeServer({
|
|
1914
|
+
host: '127.0.0.1', port: 0,
|
|
1915
|
+
store: { dbPath: ':memory:', getState: () => status, listEvents: () => [] },
|
|
1916
|
+
getContext: async () => context,
|
|
1917
|
+
run: async () => new Promise(() => {}),
|
|
1918
|
+
turn: async (_context, options) => { turns += 1; turnInput = options.input; return { ok: true }; },
|
|
1919
|
+
});
|
|
1920
|
+
} catch (err) {
|
|
1921
|
+
if (err?.code === 'EPERM') { t.skip('network listen is not permitted in this sandbox'); return; }
|
|
1922
|
+
throw err;
|
|
1923
|
+
}
|
|
1924
|
+
try {
|
|
1925
|
+
const response = await fetch(`http://127.0.0.1:${handle.port}/turn?workspace=acme`, {
|
|
1926
|
+
method: 'POST', headers: { 'content-type': 'application/json' },
|
|
1927
|
+
body: JSON.stringify({ input: '/status', mode: 'agent' }),
|
|
1928
|
+
});
|
|
1929
|
+
const body = await response.json();
|
|
1930
|
+
assert.equal(response.status, 202);
|
|
1931
|
+
assert.equal(body.kind, 'turn');
|
|
1932
|
+
await new Promise((resolve) => setTimeout(resolve, 25));
|
|
1933
|
+
assert.equal(turns, 1, 'the built-in status reaches Donna, never the homonymous skill');
|
|
1934
|
+
assert.match(turnInput, /Runtime facts:/);
|
|
1935
|
+
assert.doesNotMatch(turnInput, /Inspect services/);
|
|
1936
|
+
} finally {
|
|
1937
|
+
context.currentAbortController?.abort();
|
|
1938
|
+
await handle.close();
|
|
1939
|
+
}
|
|
1940
|
+
});
|
|
1941
|
+
|
|
1736
1942
|
test('POST /run accepts named skill arguments and deduplicates an explicit retry key', async (t) => {
|
|
1737
1943
|
const root = mkdtempSync(join(tmpdir(), 'runtime-named-skill-'));
|
|
1738
1944
|
mkdirSync(join(root, '.wiki', 'skills'), { recursive: true });
|
package/src/runtime/skillRun.js
CHANGED
|
@@ -165,7 +165,7 @@ export async function generateSkillAcknowledgment(session, { publicInput, object
|
|
|
165
165
|
// slow provider must not block the skill-launch HTTP response forever.
|
|
166
166
|
const reply = await llm.complete({
|
|
167
167
|
system: 'You are Donna, the workspace assistant. You acknowledge a launched workflow in the user\'s language. Be concise: exactly one short sentence.',
|
|
168
|
-
input: `The user just launched the workspace skill ${publicInput}. It was compiled into ${count} step(s) and
|
|
168
|
+
input: `The user just launched the workspace skill ${publicInput}. It was compiled into ${count} step(s) and has been queued.\n\nWrite ONE short sentence in ${language} that confirms the launch, echoes the skill and its arguments, and says progress will be reported. Do not claim the work is running, executing or done: a mutating step waits for the user's approval before it runs. Do not ask a question, do not propose options, and do not offer to check, monitor or cancel anything: the runtime reports progress on its own and this acknowledgement is not a decision point. Return only that sentence, nothing else.`,
|
|
169
169
|
signal: AbortSignal.timeout(8_000),
|
|
170
170
|
});
|
|
171
171
|
const text = String(reply ?? '').trim();
|
|
@@ -179,7 +179,7 @@ export async function generateSkillAcknowledgment(session, { publicInput, object
|
|
|
179
179
|
emitRuntimeLog(session, `skill-acknowledgment: LLM call failed, using the neutral fallback — ${err instanceof Error ? err.message : String(err)}`);
|
|
180
180
|
}
|
|
181
181
|
}
|
|
182
|
-
return `Started ${publicInput} — ${count} step(s)
|
|
182
|
+
return `Started ${publicInput} — ${count} step(s) queued.`;
|
|
183
183
|
}
|
|
184
184
|
|
|
185
185
|
function argumentError(message) {
|
|
@@ -94,17 +94,22 @@ test('generateSkillAcknowledgment asks Donna in the session language and echoes
|
|
|
94
94
|
assert.equal(calls.length, 1);
|
|
95
95
|
assert.match(calls[0].input, /es/);
|
|
96
96
|
assert.match(calls[0].input, /\/deliver deliverable="Informe"/);
|
|
97
|
+
// The acknowledgement is not a decision point: it must not invite the user
|
|
98
|
+
// into a dialog the runtime cannot act on, and it must not claim the work is
|
|
99
|
+
// already executing — a mutating step waits for approval.
|
|
100
|
+
assert.match(calls[0].input, /Do not ask a question/);
|
|
101
|
+
assert.match(calls[0].input, /Do not claim the work is running/);
|
|
97
102
|
});
|
|
98
103
|
|
|
99
104
|
test('generateSkillAcknowledgment degrades to a neutral message without an LLM client', async () => {
|
|
100
105
|
const reply = await generateSkillAcknowledgment({ language: 'fr' }, { publicInput: '/wiki-ingest docs', objectives: 2 });
|
|
101
|
-
assert.equal(reply, 'Started /wiki-ingest docs — 2 step(s)
|
|
106
|
+
assert.equal(reply, 'Started /wiki-ingest docs — 2 step(s) queued.');
|
|
102
107
|
});
|
|
103
108
|
|
|
104
109
|
test('generateSkillAcknowledgment falls back when the LLM call fails', async () => {
|
|
105
110
|
const session = { language: 'en', llm: { complete: async () => { throw new Error('down'); } } };
|
|
106
111
|
const reply = await generateSkillAcknowledgment(session, { publicInput: '/deliver', objectives: 1 });
|
|
107
|
-
assert.equal(reply, 'Started /deliver — 1 step(s)
|
|
112
|
+
assert.equal(reply, 'Started /deliver — 1 step(s) queued.');
|
|
108
113
|
});
|
|
109
114
|
|
|
110
115
|
test('generateSkillAcknowledgment announces an LLM failure instead of degrading silently', async () => {
|
package/src/shell/repl.js
CHANGED
|
@@ -1583,13 +1583,17 @@ async function runChatToolLoop({ input, session, history, donnaMessage, onUpdate
|
|
|
1583
1583
|
executeCall,
|
|
1584
1584
|
maxIterations: Math.min(8, Number(session?.chatAccess?.maxToolIterations) || 4),
|
|
1585
1585
|
signal: session._abortSignal,
|
|
1586
|
-
onStep: (
|
|
1586
|
+
onStep: () => onStep?.('Chat: consulting…'),
|
|
1587
1587
|
onTextDelta,
|
|
1588
1588
|
onTextReset,
|
|
1589
1589
|
});
|
|
1590
|
-
|
|
1590
|
+
// A capped turn now asks the model for a final answer without tools, so an
|
|
1591
|
+
// answer may exist even when the loop hit its limit: show it. Only fall back
|
|
1592
|
+
// to the honest limit notice when there is genuinely nothing to show.
|
|
1593
|
+
const answer = stripDsmlArtifacts(content).trim();
|
|
1594
|
+
donnaMessage.content = answer || (capped
|
|
1591
1595
|
? 'Could not finish within the chat mode iteration limit. Switch to /agent if needed.'
|
|
1592
|
-
:
|
|
1596
|
+
: formatLlmUnavailableMessage('empty response'));
|
|
1593
1597
|
onUpdate?.();
|
|
1594
1598
|
}
|
|
1595
1599
|
|