agent-nuvira 3.1.1 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/agents/reasoner.d.ts +8 -0
- package/dist/agents/agents/reasoner.d.ts.map +1 -1
- package/dist/agents/agents/reasoner.js +53 -2
- package/dist/agents/agents/reasoner.js.map +1 -1
- package/dist/agents/agents/writer.d.ts +24 -0
- package/dist/agents/agents/writer.d.ts.map +1 -1
- package/dist/agents/agents/writer.js +194 -0
- package/dist/agents/agents/writer.js.map +1 -1
- package/dist/agents/composite-plan.d.ts +143 -0
- package/dist/agents/composite-plan.d.ts.map +1 -0
- package/dist/agents/composite-plan.js +399 -0
- package/dist/agents/composite-plan.js.map +1 -0
- package/dist/agents/long-form-plan.d.ts +156 -0
- package/dist/agents/long-form-plan.d.ts.map +1 -0
- package/dist/agents/long-form-plan.js +274 -0
- package/dist/agents/long-form-plan.js.map +1 -0
- package/dist/agents/orchestrator.d.ts +107 -0
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +501 -35
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/agents/prompt-assembly.d.ts +8 -0
- package/dist/agents/prompt-assembly.d.ts.map +1 -1
- package/dist/agents/prompt-assembly.js +17 -0
- package/dist/agents/prompt-assembly.js.map +1 -1
- package/dist/cli/chat.d.ts +27 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +133 -8
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/execute.d.ts +12 -0
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +105 -2
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/models.d.ts.map +1 -1
- package/dist/cli/models.js +10 -0
- package/dist/cli/models.js.map +1 -1
- package/dist/cli/nlu.d.ts +8 -0
- package/dist/cli/nlu.d.ts.map +1 -1
- package/dist/cli/nlu.js +57 -0
- package/dist/cli/nlu.js.map +1 -1
- package/dist/cli/trace.d.ts.map +1 -1
- package/dist/cli/trace.js +27 -0
- package/dist/cli/trace.js.map +1 -1
- package/dist/gateway/gateway-log.d.ts +1 -1
- package/dist/gateway/gateway-log.d.ts.map +1 -1
- package/dist/gateway/gateway-log.js.map +1 -1
- package/dist/gateway/registry.d.ts +116 -1
- package/dist/gateway/registry.d.ts.map +1 -1
- package/dist/gateway/registry.js +575 -52
- package/dist/gateway/registry.js.map +1 -1
- package/dist/inference/model-entitlement.d.ts +46 -0
- package/dist/inference/model-entitlement.d.ts.map +1 -0
- package/dist/inference/model-entitlement.js +98 -0
- package/dist/inference/model-entitlement.js.map +1 -0
- package/dist/inference/tool-call-utils.d.ts.map +1 -1
- package/dist/inference/tool-call-utils.js +70 -1
- package/dist/inference/tool-call-utils.js.map +1 -1
- package/dist/learning/autonomy-policy.d.ts +334 -0
- package/dist/learning/autonomy-policy.d.ts.map +1 -0
- package/dist/learning/autonomy-policy.js +500 -0
- package/dist/learning/autonomy-policy.js.map +1 -0
- package/dist/learning/credential-fingerprint.d.ts +58 -0
- package/dist/learning/credential-fingerprint.d.ts.map +1 -0
- package/dist/learning/credential-fingerprint.js +126 -0
- package/dist/learning/credential-fingerprint.js.map +1 -0
- package/dist/learning/deferred-task.d.ts +159 -0
- package/dist/learning/deferred-task.d.ts.map +1 -0
- package/dist/learning/deferred-task.js +451 -0
- package/dist/learning/deferred-task.js.map +1 -0
- package/dist/learning/deliverable-class.d.ts +130 -0
- package/dist/learning/deliverable-class.d.ts.map +1 -0
- package/dist/learning/deliverable-class.js +432 -0
- package/dist/learning/deliverable-class.js.map +1 -0
- package/dist/learning/long-form.d.ts +252 -0
- package/dist/learning/long-form.d.ts.map +1 -0
- package/dist/learning/long-form.js +510 -0
- package/dist/learning/long-form.js.map +1 -0
- package/dist/learning/model-first-router.d.ts +17 -0
- package/dist/learning/model-first-router.d.ts.map +1 -1
- package/dist/learning/model-first-router.js +35 -0
- package/dist/learning/model-first-router.js.map +1 -1
- package/dist/learning/model-registry.d.ts +96 -4
- package/dist/learning/model-registry.d.ts.map +1 -1
- package/dist/learning/model-registry.js +128 -6
- package/dist/learning/model-registry.js.map +1 -1
- package/dist/learning/model-warmup.d.ts +97 -2
- package/dist/learning/model-warmup.d.ts.map +1 -1
- package/dist/learning/model-warmup.js +165 -60
- package/dist/learning/model-warmup.js.map +1 -1
- package/dist/learning/prompt-layers.d.ts +61 -0
- package/dist/learning/prompt-layers.d.ts.map +1 -0
- package/dist/learning/prompt-layers.js +140 -0
- package/dist/learning/prompt-layers.js.map +1 -0
- package/dist/learning/provider-limits.d.ts +66 -0
- package/dist/learning/provider-limits.d.ts.map +1 -0
- package/dist/learning/provider-limits.js +184 -0
- package/dist/learning/provider-limits.js.map +1 -0
- package/dist/learning/reasoning-trace.d.ts +36 -1
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +36 -2
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resilient-call.d.ts +36 -1
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +118 -8
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/learning/unattended-job.d.ts +293 -0
- package/dist/learning/unattended-job.d.ts.map +1 -0
- package/dist/learning/unattended-job.js +544 -0
- package/dist/learning/unattended-job.js.map +1 -0
- package/dist/learning/unattended-progress.d.ts +95 -0
- package/dist/learning/unattended-progress.d.ts.map +1 -0
- package/dist/learning/unattended-progress.js +147 -0
- package/dist/learning/unattended-progress.js.map +1 -0
- package/dist/learning/working-state.d.ts +109 -0
- package/dist/learning/working-state.d.ts.map +1 -0
- package/dist/learning/working-state.js +244 -0
- package/dist/learning/working-state.js.map +1 -0
- package/dist/nlu/conversation-gate.d.ts +46 -0
- package/dist/nlu/conversation-gate.d.ts.map +1 -1
- package/dist/nlu/conversation-gate.js +73 -6
- package/dist/nlu/conversation-gate.js.map +1 -1
- package/dist/nlu/intent-confirm.d.ts +82 -0
- package/dist/nlu/intent-confirm.d.ts.map +1 -0
- package/dist/nlu/intent-confirm.js +146 -0
- package/dist/nlu/intent-confirm.js.map +1 -0
- package/dist/nlu/learnings.d.ts +101 -0
- package/dist/nlu/learnings.d.ts.map +1 -0
- package/dist/nlu/learnings.js +283 -0
- package/dist/nlu/learnings.js.map +1 -0
- package/dist/tools/coding-tools.d.ts.map +1 -1
- package/dist/tools/coding-tools.js +95 -19
- package/dist/tools/coding-tools.js.map +1 -1
- package/dist/tools/edit-verification.d.ts +125 -0
- package/dist/tools/edit-verification.d.ts.map +1 -0
- package/dist/tools/edit-verification.js +237 -0
- package/dist/tools/edit-verification.js.map +1 -0
- package/dist/tools/git-tool.d.ts.map +1 -1
- package/dist/tools/git-tool.js +37 -4
- package/dist/tools/git-tool.js.map +1 -1
- package/dist/tools/registry.d.ts +30 -2
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +40 -11
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/run-cli.d.ts.map +1 -1
- package/dist/tools/run-cli.js +40 -13
- package/dist/tools/run-cli.js.map +1 -1
- package/dist/tools/run-terminal.d.ts +6 -1
- package/dist/tools/run-terminal.d.ts.map +1 -1
- package/dist/tools/run-terminal.js +158 -18
- package/dist/tools/run-terminal.js.map +1 -1
- package/dist/tools/tool-loop.d.ts +35 -0
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +157 -3
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/web-dashboard/chat-console.d.ts +35 -0
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +97 -18
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/chat-retry.d.ts +143 -0
- package/dist/web-dashboard/chat-retry.d.ts.map +1 -0
- package/dist/web-dashboard/chat-retry.js +227 -0
- package/dist/web-dashboard/chat-retry.js.map +1 -0
- package/dist/web-dashboard/server.d.ts +17 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +200 -9
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +81 -4
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/package.json +1 -1
- package/src/web-dashboard/public/assets/{index-CQW0qv1r.js → index-kCUkORm7.js} +27 -27
- package/src/web-dashboard/public/assets/index-kCUkORm7.js.map +1 -0
- package/src/web-dashboard/public/index.html +1 -1
- package/src/web-dashboard/public/assets/index-CQW0qv1r.js.map +0 -1
package/dist/gateway/registry.js
CHANGED
|
@@ -32,13 +32,28 @@ import { logGatewayEvent, previewText } from './gateway-log.js';
|
|
|
32
32
|
import { hasCodingAction, looksLikeAgentCliAsk, resolveAskKind } from '../nlu/conversation-gate.js';
|
|
33
33
|
import { GatewayChatStore } from './chat-store.js';
|
|
34
34
|
import { looksLikeConfusedScaffoldingReply, looksLikeReasoningLeakReply, stripLeadingReasoningTrace, stripToolCallArtifacts, stripReasoningLeak, toUserFacingGenerationError, } from '../inference/tool-call-utils.js';
|
|
35
|
-
import { markFailoverAttempts, modelBreadthReport, renderModelBreadthReport, } from '../learning/resilient-call.js';
|
|
35
|
+
import { createResilientCallLLM, markFailoverAttempts, modelBreadthReport, renderModelBreadthReport, reportWarrantsRetry, } from '../learning/resilient-call.js';
|
|
36
|
+
// The intent audit: on a REPEATED failure, ask the model what the ask really
|
|
37
|
+
// needs and feed a confirmed correction back into the NLU (see learnings.ts).
|
|
38
|
+
import { confirmRoutedIntent, intentConfirmedNote, intentCorrectedNote, } from '../nlu/intent-confirm.js';
|
|
39
|
+
// The retry queue behind the "Reply *yes* and I will keep trying" offer. LEAF
|
|
40
|
+
// import: a plain store + matcher, so the gateway does not pull routing in for it.
|
|
41
|
+
import { abandonedLine, acceptedLine, cancelTasksFor, confirmTask, deferTask, dueTasks, expiredTasks, getPendingTask, isRetryAcceptance, isRetryDecline, removeDeferredTask, retryingLine, updateDeferredTask, } from '../learning/deferred-task.js';
|
|
42
|
+
import { UnattendedRunner, } from '../learning/unattended-job.js';
|
|
43
|
+
import { measureUnattendedProgress, scheduleFromPendingWork, } from '../learning/unattended-progress.js';
|
|
36
44
|
import { logger } from '../utils/logger.js';
|
|
37
45
|
import { existsSync, mkdirSync, writeFileSync, unlinkSync } from 'node:fs';
|
|
38
46
|
import { join } from 'node:path';
|
|
39
47
|
import { resolveBuffConfigDir } from '../config/paths.js';
|
|
40
48
|
/** How often the running gateway drains due delivery entries (ms). */
|
|
41
49
|
const DELIVERY_DRAIN_INTERVAL_MS = 30_000;
|
|
50
|
+
/**
|
|
51
|
+
* How often the running gateway checks for deferred asks whose wait is over.
|
|
52
|
+
* 15s: a retry is already gated by the model's own free-up ETA, so this only
|
|
53
|
+
* bounds how late the reply can be — tight enough to feel immediate, loose
|
|
54
|
+
* enough to cost nothing on a quiet gateway.
|
|
55
|
+
*/
|
|
56
|
+
const RETRY_DRAIN_INTERVAL_MS = 15_000;
|
|
42
57
|
/** Cap the per-target send-failure map (diagnostics only — never grows). */
|
|
43
58
|
const MAX_TRACKED_SEND_ERRORS = 200;
|
|
44
59
|
// ─── Response cleanup ──────────────────────────────────────────────────────
|
|
@@ -485,6 +500,24 @@ export class GatewayRegistry {
|
|
|
485
500
|
* double-count would prematurely fail entries). */
|
|
486
501
|
drainChain = Promise.resolve();
|
|
487
502
|
started = false;
|
|
503
|
+
/**
|
|
504
|
+
* The deferred-retry drain. A failed turn is queued (see `deferFailedTask`)
|
|
505
|
+
* and re-run here when a model is expected back — the mechanism behind the
|
|
506
|
+
* "Reply *yes*" offer, which used to be a promise with nothing behind it.
|
|
507
|
+
*/
|
|
508
|
+
retryTimer = null;
|
|
509
|
+
/** Serializes retry runs so a long one is never started twice. */
|
|
510
|
+
retryChain = Promise.resolve();
|
|
511
|
+
/**
|
|
512
|
+
* Enterprise G11 — unattended continuation of unfinished work.
|
|
513
|
+
*
|
|
514
|
+
* The story audit's ask arrived HERE (WhatsApp) and died here: 39 units of a
|
|
515
|
+
* book cannot finish in one turn, so every batch used to end by asking the
|
|
516
|
+
* sender to reply "continue". The runner keeps going on its own until the
|
|
517
|
+
* deliverable is complete, a decision is genuinely needed, or the budget runs
|
|
518
|
+
* out — and reports to the same chat.
|
|
519
|
+
*/
|
|
520
|
+
unattendedTimer = null;
|
|
488
521
|
/** Epoch ms of start() — the uptime reported in the heartbeat. */
|
|
489
522
|
startedAt = 0;
|
|
490
523
|
/** Beats written this run (monotonic; a stalled count means a stalled loop). */
|
|
@@ -743,7 +776,7 @@ export class GatewayRegistry {
|
|
|
743
776
|
* channel (a second message while a run is active queues behind it).
|
|
744
777
|
* Every message is recorded in the inbox (P2) with its disposition.
|
|
745
778
|
*/
|
|
746
|
-
async handleInbound(msg) {
|
|
779
|
+
async handleInbound(msg, opts = {}) {
|
|
747
780
|
// Auto-learn Telegram chat IDs: when a message arrives from a Telegram
|
|
748
781
|
// user, update any contact/alias that used a phone number format.
|
|
749
782
|
// Also auto-registers new users with status: 'pending' for admin approval.
|
|
@@ -876,7 +909,11 @@ export class GatewayRegistry {
|
|
|
876
909
|
// must never break handling).
|
|
877
910
|
const historyKey = `${msg.platform}:${msg.channelId}`;
|
|
878
911
|
try {
|
|
879
|
-
|
|
912
|
+
// A deferred RETRY or an audit RE-ROUTE must not re-record the ask: the
|
|
913
|
+
// original arrival already wrote that user turn, and a second copy would
|
|
914
|
+
// make the model see the same request twice in the thread.
|
|
915
|
+
if (!opts.retryOf && !opts.reentry)
|
|
916
|
+
this.chatStore.recordInbound(historyKey, msg.text);
|
|
880
917
|
}
|
|
881
918
|
catch {
|
|
882
919
|
/* best-effort */
|
|
@@ -895,6 +932,19 @@ export class GatewayRegistry {
|
|
|
895
932
|
record('clarified', pendingAnswerLine);
|
|
896
933
|
return pendingAnswerLine;
|
|
897
934
|
}
|
|
935
|
+
// DEFERRED RETRY ANSWER — an earlier failure told the sender "Reply *yes*
|
|
936
|
+
// and I will keep trying", so a pending task may be waiting for exactly
|
|
937
|
+
// this message. Placed AFTER the pending-question gate (a turn ACTIVELY
|
|
938
|
+
// holding for an answer wins — it is mid-run, the queue is not) and BEFORE
|
|
939
|
+
// every routing decision, because the reply belongs to the OFFER rather
|
|
940
|
+
// than to the NLU. A message that is neither yes nor no falls through
|
|
941
|
+
// unchanged, so a real new request is never swallowed.
|
|
942
|
+
const retryLine = this.consumeRetryAnswer(msg);
|
|
943
|
+
if (retryLine) {
|
|
944
|
+
await replyTo(retryLine);
|
|
945
|
+
record('clarified', retryLine);
|
|
946
|
+
return retryLine;
|
|
947
|
+
}
|
|
898
948
|
// ── Local-CLI asks ── "run nuvira gateway status" names a command for the
|
|
899
949
|
// OPERATOR's terminal. Observed live: it was dispatched to the multi-agent
|
|
900
950
|
// pipeline as a create intent, burned 112s, failed, and wrote an approval
|
|
@@ -937,8 +987,8 @@ export class GatewayRegistry {
|
|
|
937
987
|
// `parsed.action.run`, so the two surfaces disagreed on the same ask —
|
|
938
988
|
// "how do I add JWT auth to the app?" got prose here while chat/execute
|
|
939
989
|
// did the work, and a question phrased like a task still burned a run.
|
|
940
|
-
const askKind = resolveAskKind(msg.text, parsed);
|
|
941
|
-
logger.debug(`gateway: route ${askKind} (intent ${parsed.intent} @ ${parsed.confidence.toFixed(2)}, coding=${hasCodingAction(msg.text)})`);
|
|
990
|
+
const askKind = opts.forceKind ?? resolveAskKind(msg.text, parsed);
|
|
991
|
+
logger.debug(`gateway: route ${askKind}${opts.forceKind ? ' (audit override)' : ''} (intent ${parsed.intent} @ ${parsed.confidence.toFixed(2)}, coding=${hasCodingAction(msg.text)})`);
|
|
942
992
|
// Chat ask: a REAL chat answer through the same engine as the dashboard
|
|
943
993
|
// console (ChatCommand.answerOnce) — so "write a poem and send it to Alex"
|
|
944
994
|
// on WhatsApp actually writes the poem, and the model's toolset includes
|
|
@@ -951,6 +1001,10 @@ export class GatewayRegistry {
|
|
|
951
1001
|
if (answer && answer.content.trim() && !answer.generationFailed) {
|
|
952
1002
|
await replyTo(answer.content);
|
|
953
1003
|
record('chat', answer.content);
|
|
1004
|
+
// A retry that produced a real answer is DONE — the queue entry must go,
|
|
1005
|
+
// or the drain would keep re-running a fulfilled ask.
|
|
1006
|
+
if (opts.retryOf)
|
|
1007
|
+
removeDeferredTask(opts.retryOf);
|
|
954
1008
|
return answer.content;
|
|
955
1009
|
}
|
|
956
1010
|
// Generation failed. The sender gets NO parsed intent/confidence (that
|
|
@@ -968,11 +1022,37 @@ export class GatewayRegistry {
|
|
|
968
1022
|
// A failure the sender cannot act on is a failure twice over. When a
|
|
969
1023
|
// model IS configured, say which models were tried, which are parked and
|
|
970
1024
|
// why, and offer to keep checking — instead of "couldn't get an answer".
|
|
1025
|
+
// BEFORE reporting the failure: ask the model whether a WRITTEN ANSWER is
|
|
1026
|
+
// even the right route (see `auditAndReroute`). A chat ask that keeps
|
|
1027
|
+
// failing is often a coding task the NLU misread — in which case the
|
|
1028
|
+
// honest answer is to do the work, not to report a model problem.
|
|
1029
|
+
const audit = await this.auditAndReroute(msg, 'chat', opts);
|
|
1030
|
+
if (audit?.rerouted) {
|
|
1031
|
+
// The corrected route already replied through its own branch.
|
|
1032
|
+
return audit.reply;
|
|
1033
|
+
}
|
|
971
1034
|
const line = this.generationFailureLine();
|
|
972
|
-
const
|
|
973
|
-
?
|
|
1035
|
+
const breadth = this.hasConfiguredModel()
|
|
1036
|
+
? modelBreadthReport(attemptMark, this.configManager)
|
|
974
1037
|
: undefined;
|
|
975
|
-
|
|
1038
|
+
// CHAT branch: no answer was generated at all, which is a MODEL-layer
|
|
1039
|
+
// failure by construction — so a retry is the right remedy even when the
|
|
1040
|
+
// walk recorded no attempt (it can throw before recording anything).
|
|
1041
|
+
const detail = breadth
|
|
1042
|
+
? renderModelBreadthReport(breadth, { task: msg.text, modelLayerFailure: true })
|
|
1043
|
+
: undefined;
|
|
1044
|
+
// The offer above is only honest if something ENFORCES it: queue the ask
|
|
1045
|
+
// and retry it when a model is back (see drainDeferredTasks).
|
|
1046
|
+
//
|
|
1047
|
+
// G9: and only when a retry could actually help. When the pool was fine
|
|
1048
|
+
// and the failure was the ask itself (a wrong plan, an artifact the
|
|
1049
|
+
// writer cannot emit), the report no longer offers a retry — so queueing
|
|
1050
|
+
// one would retry SILENTLY for 6 hours and fail identically every time.
|
|
1051
|
+
// The renderer and the queue must agree, so both ask this one predicate.
|
|
1052
|
+
if (detail && reportWarrantsRetry(breadth, { modelLayerFailure: true })) {
|
|
1053
|
+
this.deferFailedAsk(msg, 'chat', breadth?.nextFreeInMs, line);
|
|
1054
|
+
}
|
|
1055
|
+
const full = [line, detail, audit?.note].filter(Boolean).join('\n\n');
|
|
976
1056
|
await replyTo(full);
|
|
977
1057
|
record('chat', full);
|
|
978
1058
|
return full;
|
|
@@ -985,19 +1065,25 @@ export class GatewayRegistry {
|
|
|
985
1065
|
// task with no delivery ask stays on the fast direct-orchestrator path.
|
|
986
1066
|
if (hasDeliveryAsk(msg.text)) {
|
|
987
1067
|
const answer = await this.runInboundChat(msg);
|
|
988
|
-
const
|
|
989
|
-
|
|
990
|
-
: this.generationFailureLine();
|
|
1068
|
+
const answered = !!answer && answer.content.trim().length > 0 && !answer.generationFailed;
|
|
1069
|
+
const line = answered ? answer.content : this.generationFailureLine();
|
|
991
1070
|
await replyTo(line);
|
|
992
1071
|
record('pipeline', line);
|
|
1072
|
+
if (answered && opts.retryOf)
|
|
1073
|
+
removeDeferredTask(opts.retryOf);
|
|
993
1074
|
// Status recipients: a pipeline task (even one routed through the loop
|
|
994
1075
|
// for delivery) forwards its outcome to the configured contacts.
|
|
995
1076
|
await this.notifyStatusRecipients(line);
|
|
996
1077
|
return line;
|
|
997
1078
|
}
|
|
998
|
-
// Pure pipeline intent —
|
|
1079
|
+
// Pure pipeline intent — the RUN is serialized. Deliberately does NOT touch
|
|
999
1080
|
// activeChannel before the gate: a chat message arriving while a pipeline
|
|
1000
1081
|
// run is in flight must not redirect the run's board events.
|
|
1082
|
+
//
|
|
1083
|
+
// Only the WORK sits inside the chain. Composing and sending the reply used
|
|
1084
|
+
// to live there too, which is why a failed run could not be re-examined:
|
|
1085
|
+
// the intent audit re-enters `handleInbound`, and doing that from inside the
|
|
1086
|
+
// chain would wait on the chain it is itself blocking (a deadlock).
|
|
1001
1087
|
await replyTo(`✅ Got it — running the ${parsed.action.name} pipeline…`);
|
|
1002
1088
|
const pipelineAttemptMark = markFailoverAttempts();
|
|
1003
1089
|
const run = this.runChain.then(async () => {
|
|
@@ -1006,47 +1092,196 @@ export class GatewayRegistry {
|
|
|
1006
1092
|
// P2 — origin context: the pipeline model knows who it's talking to and
|
|
1007
1093
|
// where (so it can reply/forward to the right place via gateway_send).
|
|
1008
1094
|
const origin = `${PLATFORM_LABELS[msg.platform]} ${msg.isGroup ? 'group' : 'chat'} ${msg.from ?? msg.senderId ?? msg.channelId}`;
|
|
1009
|
-
|
|
1095
|
+
return runPipelineTool(msg.text, this.configManager, {
|
|
1010
1096
|
board: false,
|
|
1011
1097
|
mode: parsed.mode,
|
|
1012
1098
|
taskIntentHint: parsed.action.taskIntent,
|
|
1013
1099
|
origin,
|
|
1014
1100
|
});
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
|
|
1023
|
-
|
|
1024
|
-
|
|
1101
|
+
});
|
|
1102
|
+
this.runChain = run.catch(() => undefined);
|
|
1103
|
+
const result = await run;
|
|
1104
|
+
// G11 — a run that ended with unfinished long work is CONTINUED by the
|
|
1105
|
+
// unattended drain, not by asking the sender to reply "continue". Scheduled
|
|
1106
|
+
// before the reply so the sender is told the work is ongoing: unexplained
|
|
1107
|
+
// silence followed by chapters arriving hours later reads as a bug.
|
|
1108
|
+
const continued = scheduleFromPendingWork(result.result?.pendingWork, ref);
|
|
1109
|
+
const reply = composePipelineReply({
|
|
1110
|
+
success: result.success,
|
|
1111
|
+
summary: result.summary,
|
|
1112
|
+
structured: result.result
|
|
1113
|
+
? {
|
|
1114
|
+
tasksCompleted: result.result.tasksCompleted,
|
|
1115
|
+
tasksTotal: result.result.tasksTotal,
|
|
1116
|
+
agentResults: result.result.agentResults,
|
|
1117
|
+
}
|
|
1118
|
+
: null,
|
|
1119
|
+
});
|
|
1120
|
+
// A FAILED pipeline is where the routing itself deserves one audit: six
|
|
1121
|
+
// agent steps were just spent on something the rules may have misread, and
|
|
1122
|
+
// the sender's alternative is to rephrase and hope. When the confirmation
|
|
1123
|
+
// says the ask was NOT a coding task, the right move is to answer it rather
|
|
1124
|
+
// than to report the failure — so the failure report is deliberately not
|
|
1125
|
+
// sent at all in that path (one reply, not two).
|
|
1126
|
+
let intentNote;
|
|
1127
|
+
if (!result.success) {
|
|
1128
|
+
const audit = await this.auditAndReroute(msg, 'pipeline', opts);
|
|
1129
|
+
if (audit?.rerouted) {
|
|
1130
|
+
logGatewayEvent('pipeline.completed', {
|
|
1131
|
+
platform: msg.platform,
|
|
1132
|
+
channelId: msg.channelId,
|
|
1133
|
+
success: false,
|
|
1134
|
+
summary: result.summary,
|
|
1135
|
+
tools: parsed.action.name,
|
|
1136
|
+
reroutedTo: 'chat',
|
|
1137
|
+
}, 'warn');
|
|
1138
|
+
// Status recipients still hear the outcome — the ask was answered, just
|
|
1139
|
+
// by the other route.
|
|
1140
|
+
await this.notifyStatusRecipients(audit.reply);
|
|
1141
|
+
return audit.reply;
|
|
1142
|
+
}
|
|
1143
|
+
intentNote = audit?.note;
|
|
1144
|
+
}
|
|
1145
|
+
// On failure, name the models that were tried and the ones parked, so the
|
|
1146
|
+
// sender learns WHY rather than just that it failed.
|
|
1147
|
+
const report = result.success
|
|
1148
|
+
? undefined
|
|
1149
|
+
: modelBreadthReport(pipelineAttemptMark, this.configManager);
|
|
1150
|
+
// PIPELINE branch: the tasks may have run and failed. With a healthy pool
|
|
1151
|
+
// that is a task-shape failure (live: 6 identical runs in 32 minutes), so
|
|
1152
|
+
// no `modelLayerFailure` — the report decides honestly instead.
|
|
1153
|
+
const breadth = report ? renderModelBreadthReport(report, { task: msg.text }) : undefined;
|
|
1154
|
+
// Queue the ask so the retry offer in that report is actually enforced —
|
|
1155
|
+
// but only when the report actually OFFERED a retry (G9). A pipeline
|
|
1156
|
+
// failure with a healthy pool is a task-shape failure: 40 queued attempts
|
|
1157
|
+
// would each reproduce it, so nothing is queued and the reply says so.
|
|
1158
|
+
// Never BOTH: when the work is already scheduled to continue, a queued retry
|
|
1159
|
+
// would re-run the same ask on a different timer and race it on the same
|
|
1160
|
+
// files. Continuation wins — it is the mechanism that can actually finish.
|
|
1161
|
+
if (breadth && reportWarrantsRetry(report) && !continued) {
|
|
1162
|
+
this.deferFailedAsk(msg, 'pipeline', report?.nextFreeInMs, reply);
|
|
1163
|
+
}
|
|
1164
|
+
const continuationLine = continued
|
|
1165
|
+
? '🤖 This is bigger than one pass — I\'m continuing automatically and will report back here when it\'s done. No reply needed.'
|
|
1166
|
+
: undefined;
|
|
1167
|
+
const finalReply = [reply, continuationLine, breadth, intentNote].filter(Boolean).join('\n\n');
|
|
1168
|
+
await replyTo(finalReply);
|
|
1169
|
+
// Only a SUCCESSFUL run settles the retry: on failure the branch above has
|
|
1170
|
+
// already re-queued this same task with a fresh wait.
|
|
1171
|
+
if (result.success && opts.retryOf)
|
|
1172
|
+
removeDeferredTask(opts.retryOf);
|
|
1173
|
+
logGatewayEvent('pipeline.completed', {
|
|
1174
|
+
platform: msg.platform,
|
|
1175
|
+
channelId: msg.channelId,
|
|
1176
|
+
success: result.success,
|
|
1177
|
+
summary: result.summary,
|
|
1178
|
+
tools: parsed.action.name,
|
|
1179
|
+
}, result.success ? 'info' : 'warn');
|
|
1180
|
+
record('pipeline', finalReply);
|
|
1181
|
+
// Status recipients: ALWAYS forward the completion summary to the
|
|
1182
|
+
// configured contacts/groups, whoever triggered it.
|
|
1183
|
+
await this.notifyStatusRecipients(finalReply);
|
|
1184
|
+
return finalReply;
|
|
1185
|
+
}
|
|
1186
|
+
/**
|
|
1187
|
+
* On a REPEATED failure, ask the model what this ask really needs, and act on
|
|
1188
|
+
* the answer.
|
|
1189
|
+
*
|
|
1190
|
+
* THE PROBLEM IT SOLVES. A failure says nothing about whether the request was
|
|
1191
|
+
* read correctly — but a failure that happens TWICE (a queued retry, an audit
|
|
1192
|
+
* re-route, or a spent six-step pipeline) is expensive enough that the routing
|
|
1193
|
+
* itself is worth one question. Without it, a misreading is permanent: the same
|
|
1194
|
+
* ask runs down the same wrong path forever, and the user's only recourse is to
|
|
1195
|
+
* rephrase. This is the self-correction half of routing; `learnings.ts` is the
|
|
1196
|
+
* memory half, so the NEXT identical ask routes right the first time.
|
|
1197
|
+
*
|
|
1198
|
+
* Runs at most ONCE per inbound message (the `audited` flag), which is what
|
|
1199
|
+
* stops a misreading ping-ponging between the two routes.
|
|
1200
|
+
*
|
|
1201
|
+
* @returns null when no audit ran; `{rerouted: true}` when the corrected route
|
|
1202
|
+
* was run (its reply already delivered); `{rerouted: false, note}` when the
|
|
1203
|
+
* model CONFIRMED the reading, so the caller keeps its failure report and
|
|
1204
|
+
* adds the confirmation line.
|
|
1205
|
+
*/
|
|
1206
|
+
async auditAndReroute(msg, routed, opts) {
|
|
1207
|
+
if (opts.audited || opts.reentry || opts.forceKind)
|
|
1208
|
+
return null;
|
|
1209
|
+
// REPEATED failure only. A single failure mostly describes the world, and
|
|
1210
|
+
// auditing every one would spend a quota-limited model to second-guess a
|
|
1211
|
+
// decision that was usually right.
|
|
1212
|
+
const queued = getPendingTask(msg.platform, msg.channelId);
|
|
1213
|
+
const repeated = !!opts.retryOf || (queued?.attempts ?? 0) >= 1;
|
|
1214
|
+
if (!repeated)
|
|
1215
|
+
return null;
|
|
1216
|
+
if (!this.hasConfiguredModel())
|
|
1217
|
+
return null;
|
|
1218
|
+
let verdict;
|
|
1219
|
+
try {
|
|
1220
|
+
const callLLM = createResilientCallLLM(this.configManager, {
|
|
1221
|
+
// A throwaway task identity on purpose: this probe must not inherit the
|
|
1222
|
+
// failover state of the turn that just died, and it must not write
|
|
1223
|
+
// cross-pipeline exclusions from a question about intent.
|
|
1224
|
+
task: { agentType: 'chat', description: msg.text },
|
|
1225
|
+
crossPipelineMemory: false,
|
|
1226
|
+
});
|
|
1227
|
+
verdict = await confirmRoutedIntent({
|
|
1228
|
+
ask: msg.text,
|
|
1229
|
+
routed,
|
|
1230
|
+
callLLM: (prompt) => callLLM(prompt, { maxTokens: 160, temperature: 0 }),
|
|
1025
1231
|
});
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
});
|
|
1033
|
-
const finalReply = breadth ? `${reply}\n\n${breadth}` : reply;
|
|
1034
|
-
await replyTo(finalReply);
|
|
1035
|
-
logGatewayEvent('pipeline.completed', {
|
|
1232
|
+
}
|
|
1233
|
+
catch {
|
|
1234
|
+
return null;
|
|
1235
|
+
}
|
|
1236
|
+
if (verdict.agreed) {
|
|
1237
|
+
logGatewayEvent('intent.confirmed', {
|
|
1036
1238
|
platform: msg.platform,
|
|
1037
1239
|
channelId: msg.channelId,
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
},
|
|
1042
|
-
|
|
1043
|
-
//
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1240
|
+
routed,
|
|
1241
|
+
probed: !verdict.failed,
|
|
1242
|
+
reason: verdict.reason,
|
|
1243
|
+
}, 'info');
|
|
1244
|
+
// The audit ran and confirmed the reading: say so, because "it failed" and
|
|
1245
|
+
// "it failed and I checked that I understood you" are different promises.
|
|
1246
|
+
return { rerouted: false, ...(verdict.failed ? {} : { note: intentConfirmedNote(routed) }) };
|
|
1247
|
+
}
|
|
1248
|
+
logger.warn(`gateway: intent audit corrected ${routed} → ${verdict.kind} for "${msg.text.slice(0, 70)}"${verdict.reason ? ` (${verdict.reason})` : ''}`);
|
|
1249
|
+
logGatewayEvent('intent.corrected', {
|
|
1250
|
+
platform: msg.platform,
|
|
1251
|
+
channelId: msg.channelId,
|
|
1252
|
+
from: routed,
|
|
1253
|
+
to: verdict.kind,
|
|
1254
|
+
reason: verdict.reason,
|
|
1255
|
+
learningId: verdict.learning?.id,
|
|
1256
|
+
}, 'warn');
|
|
1257
|
+
const corrected = verdict.kind;
|
|
1258
|
+
const line = intentCorrectedNote(corrected, verdict.reason);
|
|
1259
|
+
try {
|
|
1260
|
+
await this.sendToRef({ platform: msg.platform, channelId: msg.channelId }, line);
|
|
1261
|
+
}
|
|
1262
|
+
catch {
|
|
1263
|
+
/* best-effort — the corrected route still runs */
|
|
1264
|
+
}
|
|
1265
|
+
try {
|
|
1266
|
+
// Re-enter through the SAME entry point: the corrected route gets the
|
|
1267
|
+
// same gates, the same reply path and its own ledger rows, so it cannot
|
|
1268
|
+
// drift from a first-time ask. Fresh message id, `reentry` (the ask is
|
|
1269
|
+
// already in history), `audited` (never ping-pong the audit).
|
|
1270
|
+
const reply = await this.handleInbound({ ...msg, messageId: `${msg.messageId ?? 'ask'}-audit-${corrected}` }, {
|
|
1271
|
+
reentry: true,
|
|
1272
|
+
audited: true,
|
|
1273
|
+
forceKind: corrected,
|
|
1274
|
+
// Carry the retry identity through: the corrected route IS this
|
|
1275
|
+
// task's attempt, so its success must settle the queue entry rather
|
|
1276
|
+
// than leave a fulfilled ask waiting to be retried again.
|
|
1277
|
+
...(opts.retryOf ? { retryOf: opts.retryOf } : {}),
|
|
1278
|
+
});
|
|
1279
|
+
return { rerouted: true, reply, to: corrected };
|
|
1280
|
+
}
|
|
1281
|
+
catch (err) {
|
|
1282
|
+
logger.warn(`gateway: audit re-route to ${corrected} failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1283
|
+
return { rerouted: false, note: line };
|
|
1284
|
+
}
|
|
1050
1285
|
}
|
|
1051
1286
|
/**
|
|
1052
1287
|
* Chat-intent answer (write/explain/ask): run ONE tool-loop turn through the
|
|
@@ -1156,17 +1391,26 @@ export class GatewayRegistry {
|
|
|
1156
1391
|
const continuation = isSuggestedFollowup(msg.text, lastFollowups);
|
|
1157
1392
|
// P2 — origin context: the chat model knows who it's talking to, so its
|
|
1158
1393
|
// gateway_send calls target the right contact/channel.
|
|
1159
|
-
//
|
|
1160
|
-
//
|
|
1161
|
-
|
|
1162
|
-
|
|
1163
|
-
|
|
1394
|
+
//
|
|
1395
|
+
// Session 3 — CHANNEL POLICY MOVED TO THE STABLE LAYER. The response-
|
|
1396
|
+
// format rules are byte-identical on EVERY inbound message, so prepending
|
|
1397
|
+
// them to the user turn re-injected ~330 chars as volatile content every
|
|
1398
|
+
// message (an audit of the last 30 traces found they were ~80% of the
|
|
1399
|
+
// visible user turn on WhatsApp turns) and buried the ask. They now ride
|
|
1400
|
+
// in the system prompt via `systemPolicy` — prompt-cacheable, and visible
|
|
1401
|
+
// in the layered trace's `systemDigest`.
|
|
1402
|
+
const CHANNEL_POLICY = [
|
|
1164
1403
|
'RESPONSE FORMAT (non-negotiable for messaging app replies):',
|
|
1165
|
-
'- If you need to think or plan, put your reasoning inside
|
|
1404
|
+
'- If you need to think or plan, put your reasoning inside thinking and </think> tags. Only the text OUTSIDE these tags is shown to the user.',
|
|
1166
1405
|
'- Deliver your answer DIRECTLY. No preamble, no "I would be happy to...", no restating the request.',
|
|
1167
|
-
'- Do NOT include planning, constraints, self-evaluation, or tool deliberation in your visible response. Put ALL reasoning in
|
|
1406
|
+
'- Do NOT include planning, constraints, self-evaluation, or tool deliberation in your visible response. Put ALL reasoning in thinking tags.',
|
|
1168
1407
|
'- For creative tasks (poems, stories, messages): just write the content. No meta-commentary about how you wrote it.',
|
|
1169
1408
|
'- End with suggest_followups (3 suggestions) — but NEVER include the suggest_followups JSON in your response text; use the tool call.',
|
|
1409
|
+
].join('\n');
|
|
1410
|
+
// The open user turn now carries ONLY what genuinely varies per message:
|
|
1411
|
+
// who it came from, and the ask itself.
|
|
1412
|
+
const prompt = [
|
|
1413
|
+
`[Origin: ${origin} — this message was sent from a messaging app (WhatsApp/Telegram/etc). Your text response is automatically delivered back to the sender — do NOT call gateway_send for this conversation unless you need to send to a DIFFERENT target.]`,
|
|
1170
1414
|
'',
|
|
1171
1415
|
msg.text,
|
|
1172
1416
|
].join('\n');
|
|
@@ -1196,6 +1440,8 @@ export class GatewayRegistry {
|
|
|
1196
1440
|
provider: useAuto ? 'auto' : providerType,
|
|
1197
1441
|
model: useAuto ? 'auto' : providerConfig.model,
|
|
1198
1442
|
history,
|
|
1443
|
+
// Session 3 — the channel rules go in the STABLE layer, not the ask.
|
|
1444
|
+
systemPolicy: CHANNEL_POLICY,
|
|
1199
1445
|
// P5 — a replied followup carries the continuation marker into the
|
|
1200
1446
|
// model thread (the previous answer is already in `history`).
|
|
1201
1447
|
...(continuation ? { continuation: true } : {}),
|
|
@@ -1401,6 +1647,19 @@ export class GatewayRegistry {
|
|
|
1401
1647
|
this.deliveryTimer = setInterval(() => {
|
|
1402
1648
|
void this.drainDelivery().catch(() => undefined);
|
|
1403
1649
|
}, DELIVERY_DRAIN_INTERVAL_MS);
|
|
1650
|
+
// Deferred retries: the queue is PERSISTED, so this also resumes asks that
|
|
1651
|
+
// were still waiting when the gateway restarted — the normal case for a
|
|
1652
|
+
// quota wait, which routinely outlives the process that offered it.
|
|
1653
|
+
this.retryTimer = setInterval(() => {
|
|
1654
|
+
void this.drainDeferredTasks().catch(() => undefined);
|
|
1655
|
+
}, RETRY_DRAIN_INTERVAL_MS);
|
|
1656
|
+
void this.drainDeferredTasks().catch(() => undefined);
|
|
1657
|
+
// G11: keep unfinished LONG work going. Persisted too, so a book that was
|
|
1658
|
+
// half-written when the gateway restarted continues instead of being lost.
|
|
1659
|
+
this.unattendedTimer = setInterval(() => {
|
|
1660
|
+
void this.drainUnattendedJobs().catch(() => undefined);
|
|
1661
|
+
}, RETRY_DRAIN_INTERVAL_MS);
|
|
1662
|
+
void this.drainUnattendedJobs().catch(() => undefined);
|
|
1404
1663
|
}
|
|
1405
1664
|
/**
|
|
1406
1665
|
* Auto-learn Telegram chat IDs: when a message arrives from a Telegram user,
|
|
@@ -1666,6 +1925,256 @@ export class GatewayRegistry {
|
|
|
1666
1925
|
entry.resolve(match);
|
|
1667
1926
|
return `👍 Got it — using "${match.answer}".`;
|
|
1668
1927
|
}
|
|
1928
|
+
// ─── Deferred retries (the "Reply *yes*" offer, enforced) ────────────────
|
|
1929
|
+
/**
|
|
1930
|
+
* Resolve a reply against a pending retry offer. Returns a user-facing line
|
|
1931
|
+
* when the message WAS an answer to that offer, or `null` to let it fall
|
|
1932
|
+
* through to normal handling.
|
|
1933
|
+
*
|
|
1934
|
+
* Failing OPEN is the invariant (same as the pending-question machinery): a
|
|
1935
|
+
* message that is not clearly yes/no is never swallowed, and the queued task
|
|
1936
|
+
* simply keeps its place — so the sender can still accept or cancel later.
|
|
1937
|
+
*/
|
|
1938
|
+
consumeRetryAnswer(msg) {
|
|
1939
|
+
const task = getPendingTask(msg.platform, msg.channelId);
|
|
1940
|
+
if (!task)
|
|
1941
|
+
return null;
|
|
1942
|
+
if (isRetryDecline(msg.text)) {
|
|
1943
|
+
const cancelled = cancelTasksFor(msg.platform, msg.channelId);
|
|
1944
|
+
logger.info(`gateway: sender declined the retry offer — ${cancelled} queued task(s) cancelled`);
|
|
1945
|
+
logGatewayEvent('retry.cancelled', {
|
|
1946
|
+
platform: msg.platform,
|
|
1947
|
+
channelId: msg.channelId,
|
|
1948
|
+
tasks: cancelled,
|
|
1949
|
+
}, 'info');
|
|
1950
|
+
return "👍 Okay — I've stopped retrying that. Ask me anything else whenever you're ready.";
|
|
1951
|
+
}
|
|
1952
|
+
if (!isRetryAcceptance(msg.text))
|
|
1953
|
+
return null;
|
|
1954
|
+
// The task was queued the moment the turn failed, so "yes" CONFIRMS it
|
|
1955
|
+
// rather than creating a second one — and confirmation is what buys the
|
|
1956
|
+
// long horizon (see `confirmTask`).
|
|
1957
|
+
const confirmed = confirmTask(task.id) ?? task;
|
|
1958
|
+
logger.info(`gateway: retry offer accepted (attempt ${confirmed.attempts}, next in ${Math.max(0, confirmed.notBefore - Date.now())}ms)`);
|
|
1959
|
+
logGatewayEvent('retry.accepted', {
|
|
1960
|
+
platform: msg.platform,
|
|
1961
|
+
channelId: msg.channelId,
|
|
1962
|
+
taskId: confirmed.id,
|
|
1963
|
+
attempts: confirmed.attempts,
|
|
1964
|
+
}, 'info');
|
|
1965
|
+
return acceptedLine(confirmed);
|
|
1966
|
+
}
|
|
1967
|
+
/**
|
|
1968
|
+
* Queue a failed ask for retry. Called from BOTH failure branches (chat and
|
|
1969
|
+
* pipeline) at the moment the breadth report — and therefore the "Reply
|
|
1970
|
+
* *yes*" offer — is being sent to the sender.
|
|
1971
|
+
*
|
|
1972
|
+
* Never throws and never awaits: a queue write must not delay or break the
|
|
1973
|
+
* failure reply the sender is owed.
|
|
1974
|
+
*/
|
|
1975
|
+
deferFailedAsk(msg, kind, nextFreeInMs, lastError) {
|
|
1976
|
+
try {
|
|
1977
|
+
const { task, created } = deferTask({
|
|
1978
|
+
platform: msg.platform,
|
|
1979
|
+
channelId: msg.channelId,
|
|
1980
|
+
text: msg.text,
|
|
1981
|
+
kind,
|
|
1982
|
+
from: msg.from,
|
|
1983
|
+
senderId: msg.senderId,
|
|
1984
|
+
isGroup: msg.isGroup,
|
|
1985
|
+
nextFreeInMs,
|
|
1986
|
+
lastError,
|
|
1987
|
+
});
|
|
1988
|
+
logger.info(`gateway: ${created ? 'queued' : 're-queued'} deferred ask (${kind}) for ${msg.platform}:${msg.channelId} — next attempt in ${Math.max(0, task.notBefore - Date.now())}ms`);
|
|
1989
|
+
}
|
|
1990
|
+
catch (err) {
|
|
1991
|
+
logger.warn(`gateway: could not queue the failed ask for retry: ${err instanceof Error ? err.message : String(err)}`);
|
|
1992
|
+
}
|
|
1993
|
+
}
|
|
1994
|
+
/**
|
|
1995
|
+
* G11 — advance every unfinished LONG job whose turn is due.
|
|
1996
|
+
*
|
|
1997
|
+
* The other half of the story fix. `drainDeferredTasks` retries a turn that
|
|
1998
|
+
* FAILED; this keeps going a turn that SUCCEEDED but is 4 units into a
|
|
1999
|
+
* 39-unit deliverable. Without it the sender was asked to type "continue" ten
|
|
2000
|
+
* times for a book they had already asked for once.
|
|
2001
|
+
*
|
|
2002
|
+
* Ownership is filtered by platform like the retry drain (the dashboard owns
|
|
2003
|
+
* its own sessions), and a job belongs to the chat it was asked in, so the
|
|
2004
|
+
* progress lands back where the ask came from.
|
|
2005
|
+
*/
|
|
2006
|
+
async drainUnattendedJobs() {
|
|
2007
|
+
const runner = new UnattendedRunner({
|
|
2008
|
+
owns: (job) => job.surface.platform !== 'dashboard',
|
|
2009
|
+
runBatch: (job) => this.runUnattendedBatch(job),
|
|
2010
|
+
notify: async (job, line) => {
|
|
2011
|
+
const ref = {
|
|
2012
|
+
platform: job.surface.platform,
|
|
2013
|
+
channelId: job.surface.channelId,
|
|
2014
|
+
};
|
|
2015
|
+
await this.sendToRef(ref, line);
|
|
2016
|
+
},
|
|
2017
|
+
});
|
|
2018
|
+
await runner.drain();
|
|
2019
|
+
}
|
|
2020
|
+
/**
|
|
2021
|
+
* Run ONE continuation batch for a job and report what it MEASURED.
|
|
2022
|
+
*
|
|
2023
|
+
* Serialized through `runChain` exactly like a first-time pipeline run: two
|
|
2024
|
+
* batches must never build the same chapters concurrently, and the board
|
|
2025
|
+
* events have to stream to the right chat.
|
|
2026
|
+
*/
|
|
2027
|
+
async runUnattendedBatch(job) {
|
|
2028
|
+
const ref = {
|
|
2029
|
+
platform: job.surface.platform,
|
|
2030
|
+
channelId: job.surface.channelId,
|
|
2031
|
+
};
|
|
2032
|
+
const origin = `${PLATFORM_LABELS[ref.platform] ?? ref.platform} continuation of a larger task`;
|
|
2033
|
+
const run = this.runChain.then(async () => {
|
|
2034
|
+
this.activeChannel = ref;
|
|
2035
|
+
return runPipelineTool(job.continuationPrompt, this.configManager, { board: false, origin });
|
|
2036
|
+
});
|
|
2037
|
+
this.runChain = run.catch(() => undefined);
|
|
2038
|
+
let outcome;
|
|
2039
|
+
try {
|
|
2040
|
+
outcome = await run;
|
|
2041
|
+
}
|
|
2042
|
+
catch (err) {
|
|
2043
|
+
return { error: err instanceof Error ? err.message : String(err) };
|
|
2044
|
+
}
|
|
2045
|
+
const measured = measureUnattendedProgress(job);
|
|
2046
|
+
// Refresh the schedule from the newest snapshot: a composite job only knows
|
|
2047
|
+
// its full artifact list once its phases have been planned.
|
|
2048
|
+
scheduleFromPendingWork(outcome.result?.pendingWork, job.surface);
|
|
2049
|
+
// Measured, not claimed (see unattended-job.ts): a failed batch that still
|
|
2050
|
+
// wrote chapters is progress; a failed batch that moved nothing is a real
|
|
2051
|
+
// failure and counts toward the failure cap.
|
|
2052
|
+
const moved = (measured.progress ?? 0) > job.progress;
|
|
2053
|
+
if (!outcome.success && !measured.finished && !moved) {
|
|
2054
|
+
return { ...measured, error: outcome.error || outcome.summary || 'continuation batch failed' };
|
|
2055
|
+
}
|
|
2056
|
+
return measured;
|
|
2057
|
+
}
|
|
2058
|
+
/**
|
|
2059
|
+
* Run every deferred ask whose wait is over (and report the ones that ran
|
|
2060
|
+
* out of road).
|
|
2061
|
+
*
|
|
2062
|
+
* Serialized through `retryChain`: two drains must never run the same ask
|
|
2063
|
+
* twice, and a retry is a full chat/pipeline turn that can take minutes.
|
|
2064
|
+
*/
|
|
2065
|
+
async drainDeferredTasks() {
|
|
2066
|
+
// The queue is SHARED with the dashboard server, which serves its own
|
|
2067
|
+
// `dashboard` tasks. This drain owns every OTHER platform (the messaging
|
|
2068
|
+
// channels this process actually has adapters for); claiming a dashboard
|
|
2069
|
+
// session id here would try to deliver it to a messaging platform.
|
|
2070
|
+
const owns = (task) => task.platform !== 'dashboard';
|
|
2071
|
+
// Expiry is reported FIRST: a sender who was told "I'll keep trying" must
|
|
2072
|
+
// hear the outcome even when the queue is busy with something else.
|
|
2073
|
+
for (const task of expiredTasks(Date.now(), owns)) {
|
|
2074
|
+
removeDeferredTask(task.id);
|
|
2075
|
+
const ref = { platform: task.platform, channelId: task.channelId };
|
|
2076
|
+
try {
|
|
2077
|
+
await this.sendToRef(ref, abandonedLine(task));
|
|
2078
|
+
}
|
|
2079
|
+
catch {
|
|
2080
|
+
/* best-effort — the report must not stop the drain */
|
|
2081
|
+
}
|
|
2082
|
+
logGatewayEvent('retry.abandoned', {
|
|
2083
|
+
platform: task.platform,
|
|
2084
|
+
channelId: task.channelId,
|
|
2085
|
+
attempts: task.attempts,
|
|
2086
|
+
confirmed: !!task.confirmed,
|
|
2087
|
+
}, 'warn');
|
|
2088
|
+
}
|
|
2089
|
+
const due = dueTasks(Date.now(), owns);
|
|
2090
|
+
if (due.length === 0)
|
|
2091
|
+
return;
|
|
2092
|
+
this.retryChain = this.retryChain
|
|
2093
|
+
.then(async () => {
|
|
2094
|
+
for (const task of due)
|
|
2095
|
+
await this.runDeferredTask(task);
|
|
2096
|
+
})
|
|
2097
|
+
.catch(() => undefined);
|
|
2098
|
+
await this.retryChain;
|
|
2099
|
+
}
|
|
2100
|
+
/**
|
|
2101
|
+
* Replay ONE deferred ask through the ordinary inbound path.
|
|
2102
|
+
*
|
|
2103
|
+
* Replaying the original message (rather than calling an engine directly) is
|
|
2104
|
+
* deliberate: the retry gets the SAME authorization, the SAME routing
|
|
2105
|
+
* verdict, the SAME pipeline serialization and the SAME reply path as a
|
|
2106
|
+
* first-time ask, so a retry can never behave differently from the request
|
|
2107
|
+
* it is retrying.
|
|
2108
|
+
*/
|
|
2109
|
+
async runDeferredTask(task) {
|
|
2110
|
+
const runStart = Date.now();
|
|
2111
|
+
const ref = { platform: task.platform, channelId: task.channelId };
|
|
2112
|
+
const attempt = task.attempts + 1;
|
|
2113
|
+
updateDeferredTask(task.id, {
|
|
2114
|
+
status: 'running',
|
|
2115
|
+
attempts: attempt,
|
|
2116
|
+
lastAttemptAt: runStart,
|
|
2117
|
+
});
|
|
2118
|
+
logger.info(`gateway: retrying deferred ask (attempt ${attempt}) for ${task.platform}:${task.channelId} — ${task.text.slice(0, 60)}`);
|
|
2119
|
+
logGatewayEvent('retry.started', {
|
|
2120
|
+
platform: task.platform,
|
|
2121
|
+
channelId: task.channelId,
|
|
2122
|
+
attempt,
|
|
2123
|
+
}, 'info');
|
|
2124
|
+
// Say so BEFORE the result: a failure report arriving out of nowhere, or an
|
|
2125
|
+
// answer to a request the sender had given up on, both read as a bug. The
|
|
2126
|
+
// line also explains the task in hand, since the original ask may be hours old.
|
|
2127
|
+
try {
|
|
2128
|
+
await this.sendToRef(ref, retryingLine({ ...task, attempts: attempt }));
|
|
2129
|
+
}
|
|
2130
|
+
catch {
|
|
2131
|
+
/* best-effort */
|
|
2132
|
+
}
|
|
2133
|
+
let outcome = '';
|
|
2134
|
+
try {
|
|
2135
|
+
outcome = await this.handleInbound({
|
|
2136
|
+
platform: task.platform,
|
|
2137
|
+
channelId: task.channelId,
|
|
2138
|
+
text: task.text,
|
|
2139
|
+
...(task.from ? { from: task.from } : {}),
|
|
2140
|
+
...(task.senderId ? { senderId: task.senderId } : {}),
|
|
2141
|
+
...(task.isGroup !== undefined ? { isGroup: task.isGroup } : {}),
|
|
2142
|
+
// A FRESH transport id: the dedup ledger consumed the original one, so
|
|
2143
|
+
// reusing it would classify the retry as a duplicate and silently
|
|
2144
|
+
// drop it — the exact "nothing happened" failure this queue exists to fix.
|
|
2145
|
+
messageId: `${task.id}-${attempt}`,
|
|
2146
|
+
}, { retryOf: task.id });
|
|
2147
|
+
}
|
|
2148
|
+
catch (err) {
|
|
2149
|
+
logger.warn(`gateway: deferred retry threw: ${err instanceof Error ? err.message : String(err)}`);
|
|
2150
|
+
}
|
|
2151
|
+
// A policy change can deauthorize the sender between the offer and the
|
|
2152
|
+
// retry. Retrying a refused sender forever would be harassment, so the task
|
|
2153
|
+
// is dropped (silently, matching the policy gate's own contract).
|
|
2154
|
+
if (outcome === 'refused') {
|
|
2155
|
+
removeDeferredTask(task.id);
|
|
2156
|
+
logger.info('gateway: deferred ask dropped — sender no longer authorized.');
|
|
2157
|
+
return;
|
|
2158
|
+
}
|
|
2159
|
+
// Did the replay FAIL again? Then the failure branch re-queued this same
|
|
2160
|
+
// task with a fresh wait, which is the ONLY thing that moves `notBefore`
|
|
2161
|
+
// past the moment we started. Anything else means the ask was answered.
|
|
2162
|
+
const after = getPendingTask(task.platform, task.channelId);
|
|
2163
|
+
if (after && after.id === task.id && after.notBefore > runStart)
|
|
2164
|
+
return;
|
|
2165
|
+
if (after && after.id === task.id) {
|
|
2166
|
+
// No decision was reached (the turn threw before its failure branch):
|
|
2167
|
+
// keep the task with a short backoff rather than dropping a real request.
|
|
2168
|
+
updateDeferredTask(task.id, { status: 'pending', notBefore: Date.now() + 60_000 });
|
|
2169
|
+
return;
|
|
2170
|
+
}
|
|
2171
|
+
removeDeferredTask(task.id);
|
|
2172
|
+
logGatewayEvent('retry.succeeded', {
|
|
2173
|
+
platform: task.platform,
|
|
2174
|
+
channelId: task.channelId,
|
|
2175
|
+
attempts: attempt,
|
|
2176
|
+
}, 'info');
|
|
2177
|
+
}
|
|
1669
2178
|
/** Stop adapters + unsubscribe + stop the delivery drain. Idempotent. */
|
|
1670
2179
|
async stop() {
|
|
1671
2180
|
// Release every awaiting question FIRST, and before the `started` guard: a
|
|
@@ -1690,6 +2199,20 @@ export class GatewayRegistry {
|
|
|
1690
2199
|
clearInterval(this.deliveryTimer);
|
|
1691
2200
|
this.deliveryTimer = null;
|
|
1692
2201
|
}
|
|
2202
|
+
// Deferred tasks are NOT cleared here: they are persisted precisely so a
|
|
2203
|
+
// restart resumes them (and `stop()` is also called on registries that were
|
|
2204
|
+
// never started).
|
|
2205
|
+
if (this.retryTimer) {
|
|
2206
|
+
clearInterval(this.retryTimer);
|
|
2207
|
+
this.retryTimer = null;
|
|
2208
|
+
}
|
|
2209
|
+
// Unattended jobs are NOT cleared either: they live in their own persisted
|
|
2210
|
+
// store so a restart resumes them (and `stop()` runs on registries that
|
|
2211
|
+
// were never started).
|
|
2212
|
+
if (this.unattendedTimer) {
|
|
2213
|
+
clearInterval(this.unattendedTimer);
|
|
2214
|
+
this.unattendedTimer = null;
|
|
2215
|
+
}
|
|
1693
2216
|
for (const adapter of this.adapters.values()) {
|
|
1694
2217
|
try {
|
|
1695
2218
|
await adapter.stop();
|