agent-nuvira 3.1.1 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (172) hide show
  1. package/dist/agents/agents/reasoner.d.ts +8 -0
  2. package/dist/agents/agents/reasoner.d.ts.map +1 -1
  3. package/dist/agents/agents/reasoner.js +53 -2
  4. package/dist/agents/agents/reasoner.js.map +1 -1
  5. package/dist/agents/agents/writer.d.ts +24 -0
  6. package/dist/agents/agents/writer.d.ts.map +1 -1
  7. package/dist/agents/agents/writer.js +194 -0
  8. package/dist/agents/agents/writer.js.map +1 -1
  9. package/dist/agents/composite-plan.d.ts +143 -0
  10. package/dist/agents/composite-plan.d.ts.map +1 -0
  11. package/dist/agents/composite-plan.js +399 -0
  12. package/dist/agents/composite-plan.js.map +1 -0
  13. package/dist/agents/long-form-plan.d.ts +156 -0
  14. package/dist/agents/long-form-plan.d.ts.map +1 -0
  15. package/dist/agents/long-form-plan.js +274 -0
  16. package/dist/agents/long-form-plan.js.map +1 -0
  17. package/dist/agents/orchestrator.d.ts +107 -0
  18. package/dist/agents/orchestrator.d.ts.map +1 -1
  19. package/dist/agents/orchestrator.js +501 -35
  20. package/dist/agents/orchestrator.js.map +1 -1
  21. package/dist/agents/prompt-assembly.d.ts +8 -0
  22. package/dist/agents/prompt-assembly.d.ts.map +1 -1
  23. package/dist/agents/prompt-assembly.js +17 -0
  24. package/dist/agents/prompt-assembly.js.map +1 -1
  25. package/dist/cli/chat.d.ts +27 -0
  26. package/dist/cli/chat.d.ts.map +1 -1
  27. package/dist/cli/chat.js +133 -8
  28. package/dist/cli/chat.js.map +1 -1
  29. package/dist/cli/execute.d.ts +12 -0
  30. package/dist/cli/execute.d.ts.map +1 -1
  31. package/dist/cli/execute.js +105 -2
  32. package/dist/cli/execute.js.map +1 -1
  33. package/dist/cli/models.d.ts.map +1 -1
  34. package/dist/cli/models.js +10 -0
  35. package/dist/cli/models.js.map +1 -1
  36. package/dist/cli/nlu.d.ts +8 -0
  37. package/dist/cli/nlu.d.ts.map +1 -1
  38. package/dist/cli/nlu.js +57 -0
  39. package/dist/cli/nlu.js.map +1 -1
  40. package/dist/cli/trace.d.ts.map +1 -1
  41. package/dist/cli/trace.js +27 -0
  42. package/dist/cli/trace.js.map +1 -1
  43. package/dist/gateway/gateway-log.d.ts +1 -1
  44. package/dist/gateway/gateway-log.d.ts.map +1 -1
  45. package/dist/gateway/gateway-log.js.map +1 -1
  46. package/dist/gateway/registry.d.ts +116 -1
  47. package/dist/gateway/registry.d.ts.map +1 -1
  48. package/dist/gateway/registry.js +575 -52
  49. package/dist/gateway/registry.js.map +1 -1
  50. package/dist/inference/model-entitlement.d.ts +46 -0
  51. package/dist/inference/model-entitlement.d.ts.map +1 -0
  52. package/dist/inference/model-entitlement.js +98 -0
  53. package/dist/inference/model-entitlement.js.map +1 -0
  54. package/dist/inference/tool-call-utils.d.ts.map +1 -1
  55. package/dist/inference/tool-call-utils.js +70 -1
  56. package/dist/inference/tool-call-utils.js.map +1 -1
  57. package/dist/learning/autonomy-policy.d.ts +334 -0
  58. package/dist/learning/autonomy-policy.d.ts.map +1 -0
  59. package/dist/learning/autonomy-policy.js +500 -0
  60. package/dist/learning/autonomy-policy.js.map +1 -0
  61. package/dist/learning/credential-fingerprint.d.ts +58 -0
  62. package/dist/learning/credential-fingerprint.d.ts.map +1 -0
  63. package/dist/learning/credential-fingerprint.js +126 -0
  64. package/dist/learning/credential-fingerprint.js.map +1 -0
  65. package/dist/learning/deferred-task.d.ts +159 -0
  66. package/dist/learning/deferred-task.d.ts.map +1 -0
  67. package/dist/learning/deferred-task.js +451 -0
  68. package/dist/learning/deferred-task.js.map +1 -0
  69. package/dist/learning/deliverable-class.d.ts +130 -0
  70. package/dist/learning/deliverable-class.d.ts.map +1 -0
  71. package/dist/learning/deliverable-class.js +432 -0
  72. package/dist/learning/deliverable-class.js.map +1 -0
  73. package/dist/learning/long-form.d.ts +252 -0
  74. package/dist/learning/long-form.d.ts.map +1 -0
  75. package/dist/learning/long-form.js +510 -0
  76. package/dist/learning/long-form.js.map +1 -0
  77. package/dist/learning/model-first-router.d.ts +17 -0
  78. package/dist/learning/model-first-router.d.ts.map +1 -1
  79. package/dist/learning/model-first-router.js +35 -0
  80. package/dist/learning/model-first-router.js.map +1 -1
  81. package/dist/learning/model-registry.d.ts +96 -4
  82. package/dist/learning/model-registry.d.ts.map +1 -1
  83. package/dist/learning/model-registry.js +128 -6
  84. package/dist/learning/model-registry.js.map +1 -1
  85. package/dist/learning/model-warmup.d.ts +97 -2
  86. package/dist/learning/model-warmup.d.ts.map +1 -1
  87. package/dist/learning/model-warmup.js +165 -60
  88. package/dist/learning/model-warmup.js.map +1 -1
  89. package/dist/learning/prompt-layers.d.ts +61 -0
  90. package/dist/learning/prompt-layers.d.ts.map +1 -0
  91. package/dist/learning/prompt-layers.js +140 -0
  92. package/dist/learning/prompt-layers.js.map +1 -0
  93. package/dist/learning/provider-limits.d.ts +66 -0
  94. package/dist/learning/provider-limits.d.ts.map +1 -0
  95. package/dist/learning/provider-limits.js +184 -0
  96. package/dist/learning/provider-limits.js.map +1 -0
  97. package/dist/learning/reasoning-trace.d.ts +36 -1
  98. package/dist/learning/reasoning-trace.d.ts.map +1 -1
  99. package/dist/learning/reasoning-trace.js +36 -2
  100. package/dist/learning/reasoning-trace.js.map +1 -1
  101. package/dist/learning/resilient-call.d.ts +36 -1
  102. package/dist/learning/resilient-call.d.ts.map +1 -1
  103. package/dist/learning/resilient-call.js +118 -8
  104. package/dist/learning/resilient-call.js.map +1 -1
  105. package/dist/learning/unattended-job.d.ts +293 -0
  106. package/dist/learning/unattended-job.d.ts.map +1 -0
  107. package/dist/learning/unattended-job.js +544 -0
  108. package/dist/learning/unattended-job.js.map +1 -0
  109. package/dist/learning/unattended-progress.d.ts +95 -0
  110. package/dist/learning/unattended-progress.d.ts.map +1 -0
  111. package/dist/learning/unattended-progress.js +147 -0
  112. package/dist/learning/unattended-progress.js.map +1 -0
  113. package/dist/learning/working-state.d.ts +109 -0
  114. package/dist/learning/working-state.d.ts.map +1 -0
  115. package/dist/learning/working-state.js +244 -0
  116. package/dist/learning/working-state.js.map +1 -0
  117. package/dist/nlu/conversation-gate.d.ts +46 -0
  118. package/dist/nlu/conversation-gate.d.ts.map +1 -1
  119. package/dist/nlu/conversation-gate.js +73 -6
  120. package/dist/nlu/conversation-gate.js.map +1 -1
  121. package/dist/nlu/intent-confirm.d.ts +82 -0
  122. package/dist/nlu/intent-confirm.d.ts.map +1 -0
  123. package/dist/nlu/intent-confirm.js +146 -0
  124. package/dist/nlu/intent-confirm.js.map +1 -0
  125. package/dist/nlu/learnings.d.ts +101 -0
  126. package/dist/nlu/learnings.d.ts.map +1 -0
  127. package/dist/nlu/learnings.js +283 -0
  128. package/dist/nlu/learnings.js.map +1 -0
  129. package/dist/tools/coding-tools.d.ts.map +1 -1
  130. package/dist/tools/coding-tools.js +95 -19
  131. package/dist/tools/coding-tools.js.map +1 -1
  132. package/dist/tools/edit-verification.d.ts +125 -0
  133. package/dist/tools/edit-verification.d.ts.map +1 -0
  134. package/dist/tools/edit-verification.js +237 -0
  135. package/dist/tools/edit-verification.js.map +1 -0
  136. package/dist/tools/git-tool.d.ts.map +1 -1
  137. package/dist/tools/git-tool.js +37 -4
  138. package/dist/tools/git-tool.js.map +1 -1
  139. package/dist/tools/registry.d.ts +30 -2
  140. package/dist/tools/registry.d.ts.map +1 -1
  141. package/dist/tools/registry.js +40 -11
  142. package/dist/tools/registry.js.map +1 -1
  143. package/dist/tools/run-cli.d.ts.map +1 -1
  144. package/dist/tools/run-cli.js +40 -13
  145. package/dist/tools/run-cli.js.map +1 -1
  146. package/dist/tools/run-terminal.d.ts +6 -1
  147. package/dist/tools/run-terminal.d.ts.map +1 -1
  148. package/dist/tools/run-terminal.js +158 -18
  149. package/dist/tools/run-terminal.js.map +1 -1
  150. package/dist/tools/tool-loop.d.ts +35 -0
  151. package/dist/tools/tool-loop.d.ts.map +1 -1
  152. package/dist/tools/tool-loop.js +157 -3
  153. package/dist/tools/tool-loop.js.map +1 -1
  154. package/dist/web-dashboard/chat-console.d.ts +35 -0
  155. package/dist/web-dashboard/chat-console.d.ts.map +1 -1
  156. package/dist/web-dashboard/chat-console.js +97 -18
  157. package/dist/web-dashboard/chat-console.js.map +1 -1
  158. package/dist/web-dashboard/chat-retry.d.ts +143 -0
  159. package/dist/web-dashboard/chat-retry.d.ts.map +1 -0
  160. package/dist/web-dashboard/chat-retry.js +227 -0
  161. package/dist/web-dashboard/chat-retry.js.map +1 -0
  162. package/dist/web-dashboard/server.d.ts +17 -0
  163. package/dist/web-dashboard/server.d.ts.map +1 -1
  164. package/dist/web-dashboard/server.js +200 -9
  165. package/dist/web-dashboard/server.js.map +1 -1
  166. package/dist/web-dashboard/src/types.d.ts +81 -4
  167. package/dist/web-dashboard/src/types.d.ts.map +1 -1
  168. package/package.json +1 -1
  169. package/src/web-dashboard/public/assets/{index-CQW0qv1r.js → index-kCUkORm7.js} +27 -27
  170. package/src/web-dashboard/public/assets/index-kCUkORm7.js.map +1 -0
  171. package/src/web-dashboard/public/index.html +1 -1
  172. package/src/web-dashboard/public/assets/index-CQW0qv1r.js.map +0 -1
@@ -32,13 +32,28 @@ import { logGatewayEvent, previewText } from './gateway-log.js';
32
32
  import { hasCodingAction, looksLikeAgentCliAsk, resolveAskKind } from '../nlu/conversation-gate.js';
33
33
  import { GatewayChatStore } from './chat-store.js';
34
34
  import { looksLikeConfusedScaffoldingReply, looksLikeReasoningLeakReply, stripLeadingReasoningTrace, stripToolCallArtifacts, stripReasoningLeak, toUserFacingGenerationError, } from '../inference/tool-call-utils.js';
35
- import { markFailoverAttempts, modelBreadthReport, renderModelBreadthReport, } from '../learning/resilient-call.js';
35
+ import { createResilientCallLLM, markFailoverAttempts, modelBreadthReport, renderModelBreadthReport, reportWarrantsRetry, } from '../learning/resilient-call.js';
36
+ // The intent audit: on a REPEATED failure, ask the model what the ask really
37
+ // needs and feed a confirmed correction back into the NLU (see learnings.ts).
38
+ import { confirmRoutedIntent, intentConfirmedNote, intentCorrectedNote, } from '../nlu/intent-confirm.js';
39
+ // The retry queue behind the "Reply *yes* and I will keep trying" offer. LEAF
40
+ // import: a plain store + matcher, so the gateway does not pull routing in for it.
41
+ import { abandonedLine, acceptedLine, cancelTasksFor, confirmTask, deferTask, dueTasks, expiredTasks, getPendingTask, isRetryAcceptance, isRetryDecline, removeDeferredTask, retryingLine, updateDeferredTask, } from '../learning/deferred-task.js';
42
+ import { UnattendedRunner, } from '../learning/unattended-job.js';
43
+ import { measureUnattendedProgress, scheduleFromPendingWork, } from '../learning/unattended-progress.js';
36
44
  import { logger } from '../utils/logger.js';
37
45
  import { existsSync, mkdirSync, writeFileSync, unlinkSync } from 'node:fs';
38
46
  import { join } from 'node:path';
39
47
  import { resolveBuffConfigDir } from '../config/paths.js';
40
48
  /** How often the running gateway drains due delivery entries (ms). */
41
49
  const DELIVERY_DRAIN_INTERVAL_MS = 30_000;
50
+ /**
51
+ * How often the running gateway checks for deferred asks whose wait is over.
52
+ * 15s: a retry is already gated by the model's own free-up ETA, so this only
53
+ * bounds how late the reply can be — tight enough to feel immediate, loose
54
+ * enough to cost nothing on a quiet gateway.
55
+ */
56
+ const RETRY_DRAIN_INTERVAL_MS = 15_000;
42
57
  /** Cap the per-target send-failure map (diagnostics only — never grows). */
43
58
  const MAX_TRACKED_SEND_ERRORS = 200;
44
59
  // ─── Response cleanup ──────────────────────────────────────────────────────
@@ -485,6 +500,24 @@ export class GatewayRegistry {
485
500
  * double-count would prematurely fail entries). */
486
501
  drainChain = Promise.resolve();
487
502
  started = false;
503
+ /**
504
+ * The deferred-retry drain. A failed turn is queued (see `deferFailedTask`)
505
+ * and re-run here when a model is expected back — the mechanism behind the
506
+ * "Reply *yes*" offer, which used to be a promise with nothing behind it.
507
+ */
508
+ retryTimer = null;
509
+ /** Serializes retry runs so a long one is never started twice. */
510
+ retryChain = Promise.resolve();
511
+ /**
512
+ * Enterprise G11 — unattended continuation of unfinished work.
513
+ *
514
+ * The story audit's ask arrived HERE (WhatsApp) and died here: 39 units of a
515
+ * book cannot finish in one turn, so every batch used to end by asking the
516
+ * sender to reply "continue". The runner keeps going on its own until the
517
+ * deliverable is complete, a decision is genuinely needed, or the budget runs
518
+ * out — and reports to the same chat.
519
+ */
520
+ unattendedTimer = null;
488
521
  /** Epoch ms of start() — the uptime reported in the heartbeat. */
489
522
  startedAt = 0;
490
523
  /** Beats written this run (monotonic; a stalled count means a stalled loop). */
@@ -743,7 +776,7 @@ export class GatewayRegistry {
743
776
  * channel (a second message while a run is active queues behind it).
744
777
  * Every message is recorded in the inbox (P2) with its disposition.
745
778
  */
746
- async handleInbound(msg) {
779
+ async handleInbound(msg, opts = {}) {
747
780
  // Auto-learn Telegram chat IDs: when a message arrives from a Telegram
748
781
  // user, update any contact/alias that used a phone number format.
749
782
  // Also auto-registers new users with status: 'pending' for admin approval.
@@ -876,7 +909,11 @@ export class GatewayRegistry {
876
909
  // must never break handling).
877
910
  const historyKey = `${msg.platform}:${msg.channelId}`;
878
911
  try {
879
- this.chatStore.recordInbound(historyKey, msg.text);
912
+ // A deferred RETRY or an audit RE-ROUTE must not re-record the ask: the
913
+ // original arrival already wrote that user turn, and a second copy would
914
+ // make the model see the same request twice in the thread.
915
+ if (!opts.retryOf && !opts.reentry)
916
+ this.chatStore.recordInbound(historyKey, msg.text);
880
917
  }
881
918
  catch {
882
919
  /* best-effort */
@@ -895,6 +932,19 @@ export class GatewayRegistry {
895
932
  record('clarified', pendingAnswerLine);
896
933
  return pendingAnswerLine;
897
934
  }
935
+ // DEFERRED RETRY ANSWER — an earlier failure told the sender "Reply *yes*
936
+ // and I will keep trying", so a pending task may be waiting for exactly
937
+ // this message. Placed AFTER the pending-question gate (a turn ACTIVELY
938
+ // holding for an answer wins — it is mid-run, the queue is not) and BEFORE
939
+ // every routing decision, because the reply belongs to the OFFER rather
940
+ // than to the NLU. A message that is neither yes nor no falls through
941
+ // unchanged, so a real new request is never swallowed.
942
+ const retryLine = this.consumeRetryAnswer(msg);
943
+ if (retryLine) {
944
+ await replyTo(retryLine);
945
+ record('clarified', retryLine);
946
+ return retryLine;
947
+ }
898
948
  // ── Local-CLI asks ── "run nuvira gateway status" names a command for the
899
949
  // OPERATOR's terminal. Observed live: it was dispatched to the multi-agent
900
950
  // pipeline as a create intent, burned 112s, failed, and wrote an approval
@@ -937,8 +987,8 @@ export class GatewayRegistry {
937
987
  // `parsed.action.run`, so the two surfaces disagreed on the same ask —
938
988
  // "how do I add JWT auth to the app?" got prose here while chat/execute
939
989
  // did the work, and a question phrased like a task still burned a run.
940
- const askKind = resolveAskKind(msg.text, parsed);
941
- logger.debug(`gateway: route ${askKind} (intent ${parsed.intent} @ ${parsed.confidence.toFixed(2)}, coding=${hasCodingAction(msg.text)})`);
990
+ const askKind = opts.forceKind ?? resolveAskKind(msg.text, parsed);
991
+ logger.debug(`gateway: route ${askKind}${opts.forceKind ? ' (audit override)' : ''} (intent ${parsed.intent} @ ${parsed.confidence.toFixed(2)}, coding=${hasCodingAction(msg.text)})`);
942
992
  // Chat ask: a REAL chat answer through the same engine as the dashboard
943
993
  // console (ChatCommand.answerOnce) — so "write a poem and send it to Alex"
944
994
  // on WhatsApp actually writes the poem, and the model's toolset includes
@@ -951,6 +1001,10 @@ export class GatewayRegistry {
951
1001
  if (answer && answer.content.trim() && !answer.generationFailed) {
952
1002
  await replyTo(answer.content);
953
1003
  record('chat', answer.content);
1004
+ // A retry that produced a real answer is DONE — the queue entry must go,
1005
+ // or the drain would keep re-running a fulfilled ask.
1006
+ if (opts.retryOf)
1007
+ removeDeferredTask(opts.retryOf);
954
1008
  return answer.content;
955
1009
  }
956
1010
  // Generation failed. The sender gets NO parsed intent/confidence (that
@@ -968,11 +1022,37 @@ export class GatewayRegistry {
968
1022
  // A failure the sender cannot act on is a failure twice over. When a
969
1023
  // model IS configured, say which models were tried, which are parked and
970
1024
  // why, and offer to keep checking — instead of "couldn't get an answer".
1025
+ // BEFORE reporting the failure: ask the model whether a WRITTEN ANSWER is
1026
+ // even the right route (see `auditAndReroute`). A chat ask that keeps
1027
+ // failing is often a coding task the NLU misread — in which case the
1028
+ // honest answer is to do the work, not to report a model problem.
1029
+ const audit = await this.auditAndReroute(msg, 'chat', opts);
1030
+ if (audit?.rerouted) {
1031
+ // The corrected route already replied through its own branch.
1032
+ return audit.reply;
1033
+ }
971
1034
  const line = this.generationFailureLine();
972
- const detail = this.hasConfiguredModel()
973
- ? renderModelBreadthReport(modelBreadthReport(attemptMark, this.configManager), { task: msg.text })
1035
+ const breadth = this.hasConfiguredModel()
1036
+ ? modelBreadthReport(attemptMark, this.configManager)
974
1037
  : undefined;
975
- const full = detail ? `${line}\n\n${detail}` : line;
1038
+ // CHAT branch: no answer was generated at all, which is a MODEL-layer
1039
+ // failure by construction — so a retry is the right remedy even when the
1040
+ // walk recorded no attempt (it can throw before recording anything).
1041
+ const detail = breadth
1042
+ ? renderModelBreadthReport(breadth, { task: msg.text, modelLayerFailure: true })
1043
+ : undefined;
1044
+ // The offer above is only honest if something ENFORCES it: queue the ask
1045
+ // and retry it when a model is back (see drainDeferredTasks).
1046
+ //
1047
+ // G9: and only when a retry could actually help. When the pool was fine
1048
+ // and the failure was the ask itself (a wrong plan, an artifact the
1049
+ // writer cannot emit), the report no longer offers a retry — so queueing
1050
+ // one would retry SILENTLY for 6 hours and fail identically every time.
1051
+ // The renderer and the queue must agree, so both ask this one predicate.
1052
+ if (detail && reportWarrantsRetry(breadth, { modelLayerFailure: true })) {
1053
+ this.deferFailedAsk(msg, 'chat', breadth?.nextFreeInMs, line);
1054
+ }
1055
+ const full = [line, detail, audit?.note].filter(Boolean).join('\n\n');
976
1056
  await replyTo(full);
977
1057
  record('chat', full);
978
1058
  return full;
@@ -985,19 +1065,25 @@ export class GatewayRegistry {
985
1065
  // task with no delivery ask stays on the fast direct-orchestrator path.
986
1066
  if (hasDeliveryAsk(msg.text)) {
987
1067
  const answer = await this.runInboundChat(msg);
988
- const line = answer && answer.content.trim() && !answer.generationFailed
989
- ? answer.content
990
- : this.generationFailureLine();
1068
+ const answered = !!answer && answer.content.trim().length > 0 && !answer.generationFailed;
1069
+ const line = answered ? answer.content : this.generationFailureLine();
991
1070
  await replyTo(line);
992
1071
  record('pipeline', line);
1072
+ if (answered && opts.retryOf)
1073
+ removeDeferredTask(opts.retryOf);
993
1074
  // Status recipients: a pipeline task (even one routed through the loop
994
1075
  // for delivery) forwards its outcome to the configured contacts.
995
1076
  await this.notifyStatusRecipients(line);
996
1077
  return line;
997
1078
  }
998
- // Pure pipeline intent — run serialized. Deliberately does NOT touch
1079
+ // Pure pipeline intent — the RUN is serialized. Deliberately does NOT touch
999
1080
  // activeChannel before the gate: a chat message arriving while a pipeline
1000
1081
  // run is in flight must not redirect the run's board events.
1082
+ //
1083
+ // Only the WORK sits inside the chain. Composing and sending the reply used
1084
+ // to live there too, which is why a failed run could not be re-examined:
1085
+ // the intent audit re-enters `handleInbound`, and doing that from inside the
1086
+ // chain would wait on the chain it is itself blocking (a deadlock).
1001
1087
  await replyTo(`✅ Got it — running the ${parsed.action.name} pipeline…`);
1002
1088
  const pipelineAttemptMark = markFailoverAttempts();
1003
1089
  const run = this.runChain.then(async () => {
@@ -1006,47 +1092,196 @@ export class GatewayRegistry {
1006
1092
  // P2 — origin context: the pipeline model knows who it's talking to and
1007
1093
  // where (so it can reply/forward to the right place via gateway_send).
1008
1094
  const origin = `${PLATFORM_LABELS[msg.platform]} ${msg.isGroup ? 'group' : 'chat'} ${msg.from ?? msg.senderId ?? msg.channelId}`;
1009
- const result = await runPipelineTool(msg.text, this.configManager, {
1095
+ return runPipelineTool(msg.text, this.configManager, {
1010
1096
  board: false,
1011
1097
  mode: parsed.mode,
1012
1098
  taskIntentHint: parsed.action.taskIntent,
1013
1099
  origin,
1014
1100
  });
1015
- const reply = composePipelineReply({
1016
- success: result.success,
1017
- summary: result.summary,
1018
- structured: result.result
1019
- ? {
1020
- tasksCompleted: result.result.tasksCompleted,
1021
- tasksTotal: result.result.tasksTotal,
1022
- agentResults: result.result.agentResults,
1023
- }
1024
- : null,
1101
+ });
1102
+ this.runChain = run.catch(() => undefined);
1103
+ const result = await run;
1104
+ // G11 — a run that ended with unfinished long work is CONTINUED by the
1105
+ // unattended drain, not by asking the sender to reply "continue". Scheduled
1106
+ // before the reply so the sender is told the work is ongoing: unexplained
1107
+ // silence followed by chapters arriving hours later reads as a bug.
1108
+ const continued = scheduleFromPendingWork(result.result?.pendingWork, ref);
1109
+ const reply = composePipelineReply({
1110
+ success: result.success,
1111
+ summary: result.summary,
1112
+ structured: result.result
1113
+ ? {
1114
+ tasksCompleted: result.result.tasksCompleted,
1115
+ tasksTotal: result.result.tasksTotal,
1116
+ agentResults: result.result.agentResults,
1117
+ }
1118
+ : null,
1119
+ });
1120
+ // A FAILED pipeline is where the routing itself deserves one audit: six
1121
+ // agent steps were just spent on something the rules may have misread, and
1122
+ // the sender's alternative is to rephrase and hope. When the confirmation
1123
+ // says the ask was NOT a coding task, the right move is to answer it rather
1124
+ // than to report the failure — so the failure report is deliberately not
1125
+ // sent at all in that path (one reply, not two).
1126
+ let intentNote;
1127
+ if (!result.success) {
1128
+ const audit = await this.auditAndReroute(msg, 'pipeline', opts);
1129
+ if (audit?.rerouted) {
1130
+ logGatewayEvent('pipeline.completed', {
1131
+ platform: msg.platform,
1132
+ channelId: msg.channelId,
1133
+ success: false,
1134
+ summary: result.summary,
1135
+ tools: parsed.action.name,
1136
+ reroutedTo: 'chat',
1137
+ }, 'warn');
1138
+ // Status recipients still hear the outcome — the ask was answered, just
1139
+ // by the other route.
1140
+ await this.notifyStatusRecipients(audit.reply);
1141
+ return audit.reply;
1142
+ }
1143
+ intentNote = audit?.note;
1144
+ }
1145
+ // On failure, name the models that were tried and the ones parked, so the
1146
+ // sender learns WHY rather than just that it failed.
1147
+ const report = result.success
1148
+ ? undefined
1149
+ : modelBreadthReport(pipelineAttemptMark, this.configManager);
1150
+ // PIPELINE branch: the tasks may have run and failed. With a healthy pool
1151
+ // that is a task-shape failure (live: 6 identical runs in 32 minutes), so
1152
+ // no `modelLayerFailure` — the report decides honestly instead.
1153
+ const breadth = report ? renderModelBreadthReport(report, { task: msg.text }) : undefined;
1154
+ // Queue the ask so the retry offer in that report is actually enforced —
1155
+ // but only when the report actually OFFERED a retry (G9). A pipeline
1156
+ // failure with a healthy pool is a task-shape failure: 40 queued attempts
1157
+ // would each reproduce it, so nothing is queued and the reply says so.
1158
+ // Never BOTH: when the work is already scheduled to continue, a queued retry
1159
+ // would re-run the same ask on a different timer and race it on the same
1160
+ // files. Continuation wins — it is the mechanism that can actually finish.
1161
+ if (breadth && reportWarrantsRetry(report) && !continued) {
1162
+ this.deferFailedAsk(msg, 'pipeline', report?.nextFreeInMs, reply);
1163
+ }
1164
+ const continuationLine = continued
1165
+ ? '🤖 This is bigger than one pass — I\'m continuing automatically and will report back here when it\'s done. No reply needed.'
1166
+ : undefined;
1167
+ const finalReply = [reply, continuationLine, breadth, intentNote].filter(Boolean).join('\n\n');
1168
+ await replyTo(finalReply);
1169
+ // Only a SUCCESSFUL run settles the retry: on failure the branch above has
1170
+ // already re-queued this same task with a fresh wait.
1171
+ if (result.success && opts.retryOf)
1172
+ removeDeferredTask(opts.retryOf);
1173
+ logGatewayEvent('pipeline.completed', {
1174
+ platform: msg.platform,
1175
+ channelId: msg.channelId,
1176
+ success: result.success,
1177
+ summary: result.summary,
1178
+ tools: parsed.action.name,
1179
+ }, result.success ? 'info' : 'warn');
1180
+ record('pipeline', finalReply);
1181
+ // Status recipients: ALWAYS forward the completion summary to the
1182
+ // configured contacts/groups, whoever triggered it.
1183
+ await this.notifyStatusRecipients(finalReply);
1184
+ return finalReply;
1185
+ }
1186
+ /**
1187
+ * On a REPEATED failure, ask the model what this ask really needs, and act on
1188
+ * the answer.
1189
+ *
1190
+ * THE PROBLEM IT SOLVES. A failure says nothing about whether the request was
1191
+ * read correctly — but a failure that happens TWICE (a queued retry, an audit
1192
+ * re-route, or a spent six-step pipeline) is expensive enough that the routing
1193
+ * itself is worth one question. Without it, a misreading is permanent: the same
1194
+ * ask runs down the same wrong path forever, and the user's only recourse is to
1195
+ * rephrase. This is the self-correction half of routing; `learnings.ts` is the
1196
+ * memory half, so the NEXT identical ask routes right the first time.
1197
+ *
1198
+ * Runs at most ONCE per inbound message (the `audited` flag), which is what
1199
+ * stops a misreading ping-ponging between the two routes.
1200
+ *
1201
+ * @returns null when no audit ran; `{rerouted: true}` when the corrected route
1202
+ * was run (its reply already delivered); `{rerouted: false, note}` when the
1203
+ * model CONFIRMED the reading, so the caller keeps its failure report and
1204
+ * adds the confirmation line.
1205
+ */
1206
+ async auditAndReroute(msg, routed, opts) {
1207
+ if (opts.audited || opts.reentry || opts.forceKind)
1208
+ return null;
1209
+ // REPEATED failure only. A single failure mostly describes the world, and
1210
+ // auditing every one would spend a quota-limited model to second-guess a
1211
+ // decision that was usually right.
1212
+ const queued = getPendingTask(msg.platform, msg.channelId);
1213
+ const repeated = !!opts.retryOf || (queued?.attempts ?? 0) >= 1;
1214
+ if (!repeated)
1215
+ return null;
1216
+ if (!this.hasConfiguredModel())
1217
+ return null;
1218
+ let verdict;
1219
+ try {
1220
+ const callLLM = createResilientCallLLM(this.configManager, {
1221
+ // A throwaway task identity on purpose: this probe must not inherit the
1222
+ // failover state of the turn that just died, and it must not write
1223
+ // cross-pipeline exclusions from a question about intent.
1224
+ task: { agentType: 'chat', description: msg.text },
1225
+ crossPipelineMemory: false,
1226
+ });
1227
+ verdict = await confirmRoutedIntent({
1228
+ ask: msg.text,
1229
+ routed,
1230
+ callLLM: (prompt) => callLLM(prompt, { maxTokens: 160, temperature: 0 }),
1025
1231
  });
1026
- // On failure, name the models that were tried and the ones parked, so the
1027
- // sender learns WHY rather than just that it failed.
1028
- const breadth = result.success
1029
- ? undefined
1030
- : renderModelBreadthReport(modelBreadthReport(pipelineAttemptMark, this.configManager), {
1031
- task: msg.text,
1032
- });
1033
- const finalReply = breadth ? `${reply}\n\n${breadth}` : reply;
1034
- await replyTo(finalReply);
1035
- logGatewayEvent('pipeline.completed', {
1232
+ }
1233
+ catch {
1234
+ return null;
1235
+ }
1236
+ if (verdict.agreed) {
1237
+ logGatewayEvent('intent.confirmed', {
1036
1238
  platform: msg.platform,
1037
1239
  channelId: msg.channelId,
1038
- success: result.success,
1039
- summary: result.summary,
1040
- tools: parsed.action.name,
1041
- }, result.success ? 'info' : 'warn');
1042
- record('pipeline', finalReply);
1043
- // Status recipients: ALWAYS forward the completion summary to the
1044
- // configured contacts/groups, whoever triggered it.
1045
- await this.notifyStatusRecipients(finalReply);
1046
- return finalReply;
1047
- });
1048
- this.runChain = run.catch(() => undefined);
1049
- return run;
1240
+ routed,
1241
+ probed: !verdict.failed,
1242
+ reason: verdict.reason,
1243
+ }, 'info');
1244
+ // The audit ran and confirmed the reading: say so, because "it failed" and
1245
+ // "it failed and I checked that I understood you" are different promises.
1246
+ return { rerouted: false, ...(verdict.failed ? {} : { note: intentConfirmedNote(routed) }) };
1247
+ }
1248
+ logger.warn(`gateway: intent audit corrected ${routed} → ${verdict.kind} for "${msg.text.slice(0, 70)}"${verdict.reason ? ` (${verdict.reason})` : ''}`);
1249
+ logGatewayEvent('intent.corrected', {
1250
+ platform: msg.platform,
1251
+ channelId: msg.channelId,
1252
+ from: routed,
1253
+ to: verdict.kind,
1254
+ reason: verdict.reason,
1255
+ learningId: verdict.learning?.id,
1256
+ }, 'warn');
1257
+ const corrected = verdict.kind;
1258
+ const line = intentCorrectedNote(corrected, verdict.reason);
1259
+ try {
1260
+ await this.sendToRef({ platform: msg.platform, channelId: msg.channelId }, line);
1261
+ }
1262
+ catch {
1263
+ /* best-effort — the corrected route still runs */
1264
+ }
1265
+ try {
1266
+ // Re-enter through the SAME entry point: the corrected route gets the
1267
+ // same gates, the same reply path and its own ledger rows, so it cannot
1268
+ // drift from a first-time ask. Fresh message id, `reentry` (the ask is
1269
+ // already in history), `audited` (never ping-pong the audit).
1270
+ const reply = await this.handleInbound({ ...msg, messageId: `${msg.messageId ?? 'ask'}-audit-${corrected}` }, {
1271
+ reentry: true,
1272
+ audited: true,
1273
+ forceKind: corrected,
1274
+ // Carry the retry identity through: the corrected route IS this
1275
+ // task's attempt, so its success must settle the queue entry rather
1276
+ // than leave a fulfilled ask waiting to be retried again.
1277
+ ...(opts.retryOf ? { retryOf: opts.retryOf } : {}),
1278
+ });
1279
+ return { rerouted: true, reply, to: corrected };
1280
+ }
1281
+ catch (err) {
1282
+ logger.warn(`gateway: audit re-route to ${corrected} failed: ${err instanceof Error ? err.message : String(err)}`);
1283
+ return { rerouted: false, note: line };
1284
+ }
1050
1285
  }
1051
1286
  /**
1052
1287
  * Chat-intent answer (write/explain/ask): run ONE tool-loop turn through the
@@ -1156,17 +1391,26 @@ export class GatewayRegistry {
1156
1391
  const continuation = isSuggestedFollowup(msg.text, lastFollowups);
1157
1392
  // P2 — origin context: the chat model knows who it's talking to, so its
1158
1393
  // gateway_send calls target the right contact/channel.
1159
- // The response format rules ensure the user gets a clean, direct answer
1160
- // without leaked internal reasoning or planning.
1161
- const prompt = [
1162
- `[Origin: ${origin} — this message was sent from a messaging app (WhatsApp/Telegram/etc). Your text response is automatically delivered back to the sender — do NOT call gateway_send for this conversation unless you need to send to a DIFFERENT target.]`,
1163
- '',
1394
+ //
1395
+ // Session 3 — CHANNEL POLICY MOVED TO THE STABLE LAYER. The response-
1396
+ // format rules are byte-identical on EVERY inbound message, so prepending
1397
+ // them to the user turn re-injected ~330 chars as volatile content every
1398
+ // message (an audit of the last 30 traces found they were ~80% of the
1399
+ // visible user turn on WhatsApp turns) and buried the ask. They now ride
1400
+ // in the system prompt via `systemPolicy` — prompt-cacheable, and visible
1401
+ // in the layered trace's `systemDigest`.
1402
+ const CHANNEL_POLICY = [
1164
1403
  'RESPONSE FORMAT (non-negotiable for messaging app replies):',
1165
- '- If you need to think or plan, put your reasoning inside <think> and </think> tags. Only the text OUTSIDE these tags is shown to the user.',
1404
+ '- If you need to think or plan, put your reasoning inside thinking and </think> tags. Only the text OUTSIDE these tags is shown to the user.',
1166
1405
  '- Deliver your answer DIRECTLY. No preamble, no "I would be happy to...", no restating the request.',
1167
- '- Do NOT include planning, constraints, self-evaluation, or tool deliberation in your visible response. Put ALL reasoning in <think> tags.',
1406
+ '- Do NOT include planning, constraints, self-evaluation, or tool deliberation in your visible response. Put ALL reasoning in thinking tags.',
1168
1407
  '- For creative tasks (poems, stories, messages): just write the content. No meta-commentary about how you wrote it.',
1169
1408
  '- End with suggest_followups (3 suggestions) — but NEVER include the suggest_followups JSON in your response text; use the tool call.',
1409
+ ].join('\n');
1410
+ // The open user turn now carries ONLY what genuinely varies per message:
1411
+ // who it came from, and the ask itself.
1412
+ const prompt = [
1413
+ `[Origin: ${origin} — this message was sent from a messaging app (WhatsApp/Telegram/etc). Your text response is automatically delivered back to the sender — do NOT call gateway_send for this conversation unless you need to send to a DIFFERENT target.]`,
1170
1414
  '',
1171
1415
  msg.text,
1172
1416
  ].join('\n');
@@ -1196,6 +1440,8 @@ export class GatewayRegistry {
1196
1440
  provider: useAuto ? 'auto' : providerType,
1197
1441
  model: useAuto ? 'auto' : providerConfig.model,
1198
1442
  history,
1443
+ // Session 3 — the channel rules go in the STABLE layer, not the ask.
1444
+ systemPolicy: CHANNEL_POLICY,
1199
1445
  // P5 — a replied followup carries the continuation marker into the
1200
1446
  // model thread (the previous answer is already in `history`).
1201
1447
  ...(continuation ? { continuation: true } : {}),
@@ -1401,6 +1647,19 @@ export class GatewayRegistry {
1401
1647
  this.deliveryTimer = setInterval(() => {
1402
1648
  void this.drainDelivery().catch(() => undefined);
1403
1649
  }, DELIVERY_DRAIN_INTERVAL_MS);
1650
+ // Deferred retries: the queue is PERSISTED, so this also resumes asks that
1651
+ // were still waiting when the gateway restarted — the normal case for a
1652
+ // quota wait, which routinely outlives the process that offered it.
1653
+ this.retryTimer = setInterval(() => {
1654
+ void this.drainDeferredTasks().catch(() => undefined);
1655
+ }, RETRY_DRAIN_INTERVAL_MS);
1656
+ void this.drainDeferredTasks().catch(() => undefined);
1657
+ // G11: keep unfinished LONG work going. Persisted too, so a book that was
1658
+ // half-written when the gateway restarted continues instead of being lost.
1659
+ this.unattendedTimer = setInterval(() => {
1660
+ void this.drainUnattendedJobs().catch(() => undefined);
1661
+ }, RETRY_DRAIN_INTERVAL_MS);
1662
+ void this.drainUnattendedJobs().catch(() => undefined);
1404
1663
  }
1405
1664
  /**
1406
1665
  * Auto-learn Telegram chat IDs: when a message arrives from a Telegram user,
@@ -1666,6 +1925,256 @@ export class GatewayRegistry {
1666
1925
  entry.resolve(match);
1667
1926
  return `👍 Got it — using "${match.answer}".`;
1668
1927
  }
1928
+ // ─── Deferred retries (the "Reply *yes*" offer, enforced) ────────────────
1929
+ /**
1930
+ * Resolve a reply against a pending retry offer. Returns a user-facing line
1931
+ * when the message WAS an answer to that offer, or `null` to let it fall
1932
+ * through to normal handling.
1933
+ *
1934
+ * Failing OPEN is the invariant (same as the pending-question machinery): a
1935
+ * message that is not clearly yes/no is never swallowed, and the queued task
1936
+ * simply keeps its place — so the sender can still accept or cancel later.
1937
+ */
1938
+ consumeRetryAnswer(msg) {
1939
+ const task = getPendingTask(msg.platform, msg.channelId);
1940
+ if (!task)
1941
+ return null;
1942
+ if (isRetryDecline(msg.text)) {
1943
+ const cancelled = cancelTasksFor(msg.platform, msg.channelId);
1944
+ logger.info(`gateway: sender declined the retry offer — ${cancelled} queued task(s) cancelled`);
1945
+ logGatewayEvent('retry.cancelled', {
1946
+ platform: msg.platform,
1947
+ channelId: msg.channelId,
1948
+ tasks: cancelled,
1949
+ }, 'info');
1950
+ return "👍 Okay — I've stopped retrying that. Ask me anything else whenever you're ready.";
1951
+ }
1952
+ if (!isRetryAcceptance(msg.text))
1953
+ return null;
1954
+ // The task was queued the moment the turn failed, so "yes" CONFIRMS it
1955
+ // rather than creating a second one — and confirmation is what buys the
1956
+ // long horizon (see `confirmTask`).
1957
+ const confirmed = confirmTask(task.id) ?? task;
1958
+ logger.info(`gateway: retry offer accepted (attempt ${confirmed.attempts}, next in ${Math.max(0, confirmed.notBefore - Date.now())}ms)`);
1959
+ logGatewayEvent('retry.accepted', {
1960
+ platform: msg.platform,
1961
+ channelId: msg.channelId,
1962
+ taskId: confirmed.id,
1963
+ attempts: confirmed.attempts,
1964
+ }, 'info');
1965
+ return acceptedLine(confirmed);
1966
+ }
1967
+ /**
1968
+ * Queue a failed ask for retry. Called from BOTH failure branches (chat and
1969
+ * pipeline) at the moment the breadth report — and therefore the "Reply
1970
+ * *yes*" offer — is being sent to the sender.
1971
+ *
1972
+ * Never throws and never awaits: a queue write must not delay or break the
1973
+ * failure reply the sender is owed.
1974
+ */
1975
+ deferFailedAsk(msg, kind, nextFreeInMs, lastError) {
1976
+ try {
1977
+ const { task, created } = deferTask({
1978
+ platform: msg.platform,
1979
+ channelId: msg.channelId,
1980
+ text: msg.text,
1981
+ kind,
1982
+ from: msg.from,
1983
+ senderId: msg.senderId,
1984
+ isGroup: msg.isGroup,
1985
+ nextFreeInMs,
1986
+ lastError,
1987
+ });
1988
+ logger.info(`gateway: ${created ? 'queued' : 're-queued'} deferred ask (${kind}) for ${msg.platform}:${msg.channelId} — next attempt in ${Math.max(0, task.notBefore - Date.now())}ms`);
1989
+ }
1990
+ catch (err) {
1991
+ logger.warn(`gateway: could not queue the failed ask for retry: ${err instanceof Error ? err.message : String(err)}`);
1992
+ }
1993
+ }
1994
+ /**
1995
+ * G11 — advance every unfinished LONG job whose turn is due.
1996
+ *
1997
+ * The other half of the story fix. `drainDeferredTasks` retries a turn that
1998
+ * FAILED; this keeps going a turn that SUCCEEDED but is 4 units into a
1999
+ * 39-unit deliverable. Without it the sender was asked to type "continue" ten
2000
+ * times for a book they had already asked for once.
2001
+ *
2002
+ * Ownership is filtered by platform like the retry drain (the dashboard owns
2003
+ * its own sessions), and a job belongs to the chat it was asked in, so the
2004
+ * progress lands back where the ask came from.
2005
+ */
2006
+ async drainUnattendedJobs() {
2007
+ const runner = new UnattendedRunner({
2008
+ owns: (job) => job.surface.platform !== 'dashboard',
2009
+ runBatch: (job) => this.runUnattendedBatch(job),
2010
+ notify: async (job, line) => {
2011
+ const ref = {
2012
+ platform: job.surface.platform,
2013
+ channelId: job.surface.channelId,
2014
+ };
2015
+ await this.sendToRef(ref, line);
2016
+ },
2017
+ });
2018
+ await runner.drain();
2019
+ }
2020
+ /**
2021
+ * Run ONE continuation batch for a job and report what it MEASURED.
2022
+ *
2023
+ * Serialized through `runChain` exactly like a first-time pipeline run: two
2024
+ * batches must never build the same chapters concurrently, and the board
2025
+ * events have to stream to the right chat.
2026
+ */
2027
+ async runUnattendedBatch(job) {
2028
+ const ref = {
2029
+ platform: job.surface.platform,
2030
+ channelId: job.surface.channelId,
2031
+ };
2032
+ const origin = `${PLATFORM_LABELS[ref.platform] ?? ref.platform} continuation of a larger task`;
2033
+ const run = this.runChain.then(async () => {
2034
+ this.activeChannel = ref;
2035
+ return runPipelineTool(job.continuationPrompt, this.configManager, { board: false, origin });
2036
+ });
2037
+ this.runChain = run.catch(() => undefined);
2038
+ let outcome;
2039
+ try {
2040
+ outcome = await run;
2041
+ }
2042
+ catch (err) {
2043
+ return { error: err instanceof Error ? err.message : String(err) };
2044
+ }
2045
+ const measured = measureUnattendedProgress(job);
2046
+ // Refresh the schedule from the newest snapshot: a composite job only knows
2047
+ // its full artifact list once its phases have been planned.
2048
+ scheduleFromPendingWork(outcome.result?.pendingWork, job.surface);
2049
+ // Measured, not claimed (see unattended-job.ts): a failed batch that still
2050
+ // wrote chapters is progress; a failed batch that moved nothing is a real
2051
+ // failure and counts toward the failure cap.
2052
+ const moved = (measured.progress ?? 0) > job.progress;
2053
+ if (!outcome.success && !measured.finished && !moved) {
2054
+ return { ...measured, error: outcome.error || outcome.summary || 'continuation batch failed' };
2055
+ }
2056
+ return measured;
2057
+ }
2058
+ /**
2059
+ * Run every deferred ask whose wait is over (and report the ones that ran
2060
+ * out of road).
2061
+ *
2062
+ * Serialized through `retryChain`: two drains must never run the same ask
2063
+ * twice, and a retry is a full chat/pipeline turn that can take minutes.
2064
+ */
2065
+ async drainDeferredTasks() {
2066
+ // The queue is SHARED with the dashboard server, which serves its own
2067
+ // `dashboard` tasks. This drain owns every OTHER platform (the messaging
2068
+ // channels this process actually has adapters for); claiming a dashboard
2069
+ // session id here would try to deliver it to a messaging platform.
2070
+ const owns = (task) => task.platform !== 'dashboard';
2071
+ // Expiry is reported FIRST: a sender who was told "I'll keep trying" must
2072
+ // hear the outcome even when the queue is busy with something else.
2073
+ for (const task of expiredTasks(Date.now(), owns)) {
2074
+ removeDeferredTask(task.id);
2075
+ const ref = { platform: task.platform, channelId: task.channelId };
2076
+ try {
2077
+ await this.sendToRef(ref, abandonedLine(task));
2078
+ }
2079
+ catch {
2080
+ /* best-effort — the report must not stop the drain */
2081
+ }
2082
+ logGatewayEvent('retry.abandoned', {
2083
+ platform: task.platform,
2084
+ channelId: task.channelId,
2085
+ attempts: task.attempts,
2086
+ confirmed: !!task.confirmed,
2087
+ }, 'warn');
2088
+ }
2089
+ const due = dueTasks(Date.now(), owns);
2090
+ if (due.length === 0)
2091
+ return;
2092
+ this.retryChain = this.retryChain
2093
+ .then(async () => {
2094
+ for (const task of due)
2095
+ await this.runDeferredTask(task);
2096
+ })
2097
+ .catch(() => undefined);
2098
+ await this.retryChain;
2099
+ }
2100
+ /**
2101
+ * Replay ONE deferred ask through the ordinary inbound path.
2102
+ *
2103
+ * Replaying the original message (rather than calling an engine directly) is
2104
+ * deliberate: the retry gets the SAME authorization, the SAME routing
2105
+ * verdict, the SAME pipeline serialization and the SAME reply path as a
2106
+ * first-time ask, so a retry can never behave differently from the request
2107
+ * it is retrying.
2108
+ */
2109
+ async runDeferredTask(task) {
2110
+ const runStart = Date.now();
2111
+ const ref = { platform: task.platform, channelId: task.channelId };
2112
+ const attempt = task.attempts + 1;
2113
+ updateDeferredTask(task.id, {
2114
+ status: 'running',
2115
+ attempts: attempt,
2116
+ lastAttemptAt: runStart,
2117
+ });
2118
+ logger.info(`gateway: retrying deferred ask (attempt ${attempt}) for ${task.platform}:${task.channelId} — ${task.text.slice(0, 60)}`);
2119
+ logGatewayEvent('retry.started', {
2120
+ platform: task.platform,
2121
+ channelId: task.channelId,
2122
+ attempt,
2123
+ }, 'info');
2124
+ // Say so BEFORE the result: a failure report arriving out of nowhere, or an
2125
+ // answer to a request the sender had given up on, both read as a bug. The
2126
+ // line also explains the task in hand, since the original ask may be hours old.
2127
+ try {
2128
+ await this.sendToRef(ref, retryingLine({ ...task, attempts: attempt }));
2129
+ }
2130
+ catch {
2131
+ /* best-effort */
2132
+ }
2133
+ let outcome = '';
2134
+ try {
2135
+ outcome = await this.handleInbound({
2136
+ platform: task.platform,
2137
+ channelId: task.channelId,
2138
+ text: task.text,
2139
+ ...(task.from ? { from: task.from } : {}),
2140
+ ...(task.senderId ? { senderId: task.senderId } : {}),
2141
+ ...(task.isGroup !== undefined ? { isGroup: task.isGroup } : {}),
2142
+ // A FRESH transport id: the dedup ledger consumed the original one, so
2143
+ // reusing it would classify the retry as a duplicate and silently
2144
+ // drop it — the exact "nothing happened" failure this queue exists to fix.
2145
+ messageId: `${task.id}-${attempt}`,
2146
+ }, { retryOf: task.id });
2147
+ }
2148
+ catch (err) {
2149
+ logger.warn(`gateway: deferred retry threw: ${err instanceof Error ? err.message : String(err)}`);
2150
+ }
2151
+ // A policy change can deauthorize the sender between the offer and the
2152
+ // retry. Retrying a refused sender forever would be harassment, so the task
2153
+ // is dropped (silently, matching the policy gate's own contract).
2154
+ if (outcome === 'refused') {
2155
+ removeDeferredTask(task.id);
2156
+ logger.info('gateway: deferred ask dropped — sender no longer authorized.');
2157
+ return;
2158
+ }
2159
+ // Did the replay FAIL again? Then the failure branch re-queued this same
2160
+ // task with a fresh wait, which is the ONLY thing that moves `notBefore`
2161
+ // past the moment we started. Anything else means the ask was answered.
2162
+ const after = getPendingTask(task.platform, task.channelId);
2163
+ if (after && after.id === task.id && after.notBefore > runStart)
2164
+ return;
2165
+ if (after && after.id === task.id) {
2166
+ // No decision was reached (the turn threw before its failure branch):
2167
+ // keep the task with a short backoff rather than dropping a real request.
2168
+ updateDeferredTask(task.id, { status: 'pending', notBefore: Date.now() + 60_000 });
2169
+ return;
2170
+ }
2171
+ removeDeferredTask(task.id);
2172
+ logGatewayEvent('retry.succeeded', {
2173
+ platform: task.platform,
2174
+ channelId: task.channelId,
2175
+ attempts: attempt,
2176
+ }, 'info');
2177
+ }
1669
2178
  /** Stop adapters + unsubscribe + stop the delivery drain. Idempotent. */
1670
2179
  async stop() {
1671
2180
  // Release every awaiting question FIRST, and before the `started` guard: a
@@ -1690,6 +2199,20 @@ export class GatewayRegistry {
1690
2199
  clearInterval(this.deliveryTimer);
1691
2200
  this.deliveryTimer = null;
1692
2201
  }
2202
+ // Deferred tasks are NOT cleared here: they are persisted precisely so a
2203
+ // restart resumes them (and `stop()` is also called on registries that were
2204
+ // never started).
2205
+ if (this.retryTimer) {
2206
+ clearInterval(this.retryTimer);
2207
+ this.retryTimer = null;
2208
+ }
2209
+ // Unattended jobs are NOT cleared either: they live in their own persisted
2210
+ // store so a restart resumes them (and `stop()` runs on registries that
2211
+ // were never started).
2212
+ if (this.unattendedTimer) {
2213
+ clearInterval(this.unattendedTimer);
2214
+ this.unattendedTimer = null;
2215
+ }
1693
2216
  for (const adapter of this.adapters.values()) {
1694
2217
  try {
1695
2218
  await adapter.stop();