@rei-standard/amsg-server 2.6.0-next.2 → 2.6.0-next.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
- import { validateAvatarUrl, toUint8, concatBytes, base64UrlToBytes, readReasoningContent, stripReasoningTags, buildReasoningPush, buildContentPush, normalizeVapidSubject } from '@rei-standard/amsg-shared';
1
+ import { validateAvatarUrl, toUint8, concatBytes, base64UrlToBytes, extractAssistantMessage, buildSessionContext, assertValidDecision, extractToolCallsFromDecision, readReasoningContent, stripReasoningTags, buildReasoningPush, buildContentPush, normalizeVapidSubject } from '@rei-standard/amsg-shared';
2
2
 
3
3
  /**
4
4
  * Validation utility library (SDK version)
@@ -421,7 +421,7 @@ const TEXT_ENCODER = new TextEncoder();
421
421
  const TEXT_DECODER = new TextDecoder('utf-8', { fatal: false });
422
422
 
423
423
  /** UTF-8 encode a string into a Uint8Array. */
424
- function utf8(str) {
424
+ function utf8$1(str) {
425
425
  return TEXT_ENCODER.encode(String(str));
426
426
  }
427
427
 
@@ -471,7 +471,7 @@ function hexToBytes(hex) {
471
471
 
472
472
  /** Encode a JSON-serializable value as base64url (UTF-8 JSON). */
473
473
  function jsonToBase64Url(value) {
474
- return bytesToBase64Url(utf8(JSON.stringify(value)));
474
+ return bytesToBase64Url(utf8$1(JSON.stringify(value)));
475
475
  }
476
476
 
477
477
  /** `crypto.randomUUID()`. The Node adapter polyfills `globalThis.crypto`. */
@@ -558,7 +558,7 @@ async function aesGcmOpen(hexKey, iv, ciphertext, authTag) {
558
558
  * @returns {Promise<string>} 64-char hex key.
559
559
  */
560
560
  async function deriveUserEncryptionKey(userId, masterKey) {
561
- const digest = await globalThis.crypto.subtle.digest('SHA-256', utf8(masterKey + userId));
561
+ const digest = await globalThis.crypto.subtle.digest('SHA-256', utf8$1(masterKey + userId));
562
562
  return bytesToHex(new Uint8Array(digest)).slice(0, 64);
563
563
  }
564
564
 
@@ -590,7 +590,7 @@ async function decryptPayload(encryptedPayload, encryptionKey) {
590
590
  async function encryptPayload(payload, encryptionKey) {
591
591
  const plaintext = typeof payload === 'string' ? payload : JSON.stringify(payload);
592
592
  const iv = randomBytes(12);
593
- const { ciphertext, authTag } = await aesGcmSeal(encryptionKey, iv, utf8(plaintext));
593
+ const { ciphertext, authTag } = await aesGcmSeal(encryptionKey, iv, utf8$1(plaintext));
594
594
 
595
595
  return {
596
596
  iv: bytesToBase64(iv),
@@ -608,7 +608,7 @@ async function encryptPayload(payload, encryptionKey) {
608
608
  */
609
609
  async function encryptForStorage(text, encryptionKey) {
610
610
  const iv = randomBytes(16);
611
- const { ciphertext, authTag } = await aesGcmSeal(encryptionKey, iv, utf8(text));
611
+ const { ciphertext, authTag } = await aesGcmSeal(encryptionKey, iv, utf8$1(text));
612
612
  return `${bytesToHex(iv)}:${bytesToHex(authTag)}:${bytesToHex(ciphertext)}`;
613
613
  }
614
614
 
@@ -699,6 +699,460 @@ function isUniqueViolation(error) {
699
699
  return message.includes('duplicate key') || message.includes('unique constraint');
700
700
  }
701
701
 
702
+ /**
703
+ * OpenAI-compatible LLM call for the server-side fire chain.
704
+ *
705
+ * Extracted from message-processor.js so the agentic fire loop
706
+ * (lib/agentic-fire.js) can call the LLM without a circular import.
707
+ * The legacy single-shot path (processSingleMessage) and the multi-round
708
+ * loop share this one function.
709
+ */
710
+
711
+ /**
712
+ * Call an OpenAI-compatible API.
713
+ *
714
+ * Returns the full response object alongside the extracted (trimmed)
715
+ * `content` string. Callers that only need the text can ignore
716
+ * `response`; callers that want `reasoning_content` / `tool_calls`
717
+ * read from `response.choices[0].message`.
718
+ *
719
+ * @param {Object} payload
720
+ * @param {{ requireContent?: boolean, timeoutMs?: number }} [options]
721
+ * requireContent defaults to true (legacy single-shot behavior:
722
+ * throw when the response carries no content). Tool rounds legitimately
723
+ * return no content (pure tool_calls), so the agentic loop passes
724
+ * `{ requireContent: false }` — same precedent as amsg-instant's
725
+ * callLlmRaw.
726
+ * timeoutMs defaults to 300000 (the legacy per-call ceiling). The
727
+ * agentic loop passes its remaining wall-time budget so a hung LLM
728
+ * request cannot outlive totalTimeoutMs.
729
+ * @returns {Promise<{ response: unknown, content: string }>}
730
+ */
731
+ async function callLlm(payload, options = {}) {
732
+ const requireContent = options.requireContent !== false;
733
+ const timeoutMs = typeof options.timeoutMs === 'number' && Number.isFinite(options.timeoutMs) && options.timeoutMs > 0
734
+ ? options.timeoutMs
735
+ : 300000;
736
+ const normalizedApiUrl = normalizeAiApiUrl(payload.apiUrl);
737
+ const requestBody = buildAiRequestBody(payload);
738
+
739
+ const aiResponse = await fetch(normalizedApiUrl, {
740
+ method: 'POST',
741
+ headers: {
742
+ 'Content-Type': 'application/json',
743
+ 'Authorization': `Bearer ${payload.apiKey}`
744
+ },
745
+ body: JSON.stringify(requestBody),
746
+ signal: AbortSignal.timeout(timeoutMs)
747
+ });
748
+
749
+ if (!aiResponse.ok) {
750
+ if (aiResponse.status === 405) {
751
+ throw new Error(
752
+ `AI API error: 405 Method Not Allowed. ` +
753
+ `apiUrl must point to a full chat endpoint (for example: /chat/completions). ` +
754
+ `Received: ${normalizedApiUrl}`
755
+ );
756
+ }
757
+
758
+ throw new Error(
759
+ `AI API error: ${aiResponse.status} ${aiResponse.statusText || 'Unknown Error'}. ` +
760
+ `Request URL: ${normalizedApiUrl}`
761
+ );
762
+ }
763
+
764
+ const aiData = await aiResponse.json();
765
+ const rawContent = aiData?.choices?.[0]?.message?.content;
766
+ if (requireContent && (typeof rawContent !== 'string' || !rawContent.trim())) {
767
+ throw new Error('AI API error: response missing choices[0].message.content');
768
+ }
769
+
770
+ return { response: aiData, content: typeof rawContent === 'string' ? rawContent.trim() : '' };
771
+ }
772
+
773
+ /**
774
+ * Build OpenAI-compatible request body.
775
+ *
776
+ * `max_tokens` is optional:
777
+ * - include it only when payload.maxTokens is provided
778
+ * - omit it when payload.maxTokens is undefined / null
779
+ *
780
+ * @param {Object} payload
781
+ * @returns {Object}
782
+ */
783
+ function buildAiRequestBody(payload) {
784
+ // messages mode (added in v2.2.0): forward the caller's OpenAI-style array
785
+ // verbatim — same contract as @rei-standard/amsg-instant 0.5.0+. No auto
786
+ // role injection, no concatenation back to a single user message. Lets
787
+ // the upstream app preserve system / multi-turn context byte-for-byte
788
+ // across the schedule-message path.
789
+ const llmMessages = Array.isArray(payload.messages) && payload.messages.length > 0
790
+ ? payload.messages
791
+ : [{ role: 'user', content: payload.completePrompt }];
792
+
793
+ const requestBody = {
794
+ model: payload.primaryModel,
795
+ messages: llmMessages,
796
+ };
797
+
798
+ // Match the instant package's behavior: only inject default temperature
799
+ // for the legacy completePrompt path; messages mode forwards whatever the
800
+ // upstream app set (or nothing) so behavior matches their main chat path.
801
+ if (payload.temperature !== undefined && payload.temperature !== null) {
802
+ requestBody.temperature = payload.temperature;
803
+ } else if (!Array.isArray(payload.messages)) {
804
+ requestBody.temperature = 0.8;
805
+ }
806
+
807
+ if (payload.maxTokens === undefined || payload.maxTokens === null) {
808
+ return requestBody;
809
+ }
810
+
811
+ if (!Number.isInteger(payload.maxTokens) || payload.maxTokens <= 0) {
812
+ throw new Error('Invalid maxTokens: maxTokens must be a positive integer when provided.');
813
+ }
814
+
815
+ requestBody.max_tokens = payload.maxTokens;
816
+ return requestBody;
817
+ }
818
+
819
+ /**
820
+ * Normalize AI API URL for OpenAI-compatible chat endpoints.
821
+ *
822
+ * **Keep in sync** with `@rei-standard/amsg-instant`'s
823
+ * `src/message-processor.js` `normalizeAiApiUrl` — same rules, same
824
+ * tests. The two packages share this logic but each carry their own copy
825
+ * to avoid an architectural dependency (server should not depend on the
826
+ * stateless worker package).
827
+ *
828
+ * @param {string} apiUrl
829
+ * @returns {string}
830
+ */
831
+ function normalizeAiApiUrl(apiUrl) {
832
+ if (typeof apiUrl !== 'string' || !apiUrl.trim()) {
833
+ throw new Error(
834
+ 'Invalid apiUrl: apiUrl is required. ' +
835
+ 'Please provide a chat endpoint URL ' +
836
+ '(for example: https://api.openai.com or https://api.openai.com/v1/chat/completions).'
837
+ );
838
+ }
839
+
840
+ const trimmedApiUrl = apiUrl.trim();
841
+ let parsedUrl;
842
+
843
+ try {
844
+ parsedUrl = new URL(trimmedApiUrl);
845
+ } catch {
846
+ throw new Error(
847
+ `Invalid apiUrl: "${apiUrl}". Please provide a valid absolute URL.`
848
+ );
849
+ }
850
+
851
+ let path = parsedUrl.pathname.replace(/\/+$/, '') || '/';
852
+
853
+ if (/\/chat\/completions$/.test(path)) ; else if (path === '/') {
854
+ // Bare host → assume OpenAI shape.
855
+ path = '/v1/chat/completions';
856
+ } else if (/\/v\d+$/.test(path)) {
857
+ // Path ends in `/v1`, `/v2`, … — caller already versioned the URL.
858
+ // Append only `/chat/completions`; never re-add `/v1`.
859
+ path = `${path}/chat/completions`;
860
+ }
861
+ // Any other custom path is left untouched on purpose.
862
+
863
+ parsedUrl.pathname = path;
864
+ return parsedUrl.toString();
865
+ }
866
+
867
+ /**
868
+ * Server-side agentic fire loop.
869
+ *
870
+ * When the host configures fire-time hooks, an LLM task stops replaying
871
+ * the completePrompt frozen at schedule time. At fire time instead:
872
+ *
873
+ * onBeforeFire(fireCtx) → fresh messages (may read client_state)
874
+ * → callLlm → onLLMOutput(sessionCtx) → decision
875
+ * ├─ 'finish' → push decision.pushPayloads, done
876
+ * ├─ 'skip-push' → record, done (task counts as delivered)
877
+ * ├─ 'continue' → replace history, next round
878
+ * └─ 'tool-request' → executeToolCalls IN the worker — the client
879
+ * is offline at fire time, so unlike
880
+ * amsg-instant nothing is pushed back for the
881
+ * client to execute — then append the
882
+ * assistant turn + tool results, next round
883
+ *
884
+ * The decision contract is shared with @rei-standard/amsg-instant
885
+ * (assertValidDecision / buildSessionContext live in
886
+ * @rei-standard/amsg-shared), so a classifier written for instant's
887
+ * onLLMOutput drops in unchanged: 'tool-request' may carry `toolCalls`
888
+ * directly, or tool_request pushPayloads that embed them — both work.
889
+ *
890
+ * Credential hiding: hook ctx objects never contain apiKey /
891
+ * pushSubscription / vapid / masterKey (same rationale as instant's
892
+ * SessionContext — a console.log(ctx) in a hook must not leak keys).
893
+ *
894
+ * Budget guards, both factory-level (ctx.maxToolIterations /
895
+ * ctx.totalTimeoutMs) and per-fire (onBeforeFire may return
896
+ * { messages, maxToolIterations?, totalTimeoutMs? } to override for one
897
+ * task — e.g. a task the host knows runs slow tools): maxToolIterations
898
+ * caps LLM rounds; totalTimeoutMs is a wall-time ceiling checked before
899
+ * each round so a cron tick can never hang forever. Both exhaustions
900
+ * surface as ordinary task failures → the existing retry / mark-failed
901
+ * semantics apply.
902
+ */
903
+
904
+
905
+ const DEFAULT_MAX_TOOL_ITERATIONS = 5;
906
+ const DEFAULT_TOTAL_TIMEOUT_MS = 240_000;
907
+
908
+ // Same pacing as the legacy path / amsg-instant.
909
+ const SLEEP_BETWEEN_MESSAGES_MS$1 = 1500;
910
+
911
+ // Task-payload fields that must never reach hook code.
912
+ const CREDENTIAL_PAYLOAD_KEYS = new Set(['apiKey', 'pushSubscription']);
913
+
914
+ const defaultSleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
915
+
916
+ /**
917
+ * Does this task need the LLM at fire time? Fixed text never does, so it
918
+ * always stays on the legacy path regardless of hooks.
919
+ *
920
+ * @param {Object} decryptedPayload
921
+ * @returns {boolean}
922
+ */
923
+ function taskNeedsLlm(decryptedPayload) {
924
+ const type = decryptedPayload.messageType;
925
+ if (type === 'prompted' || type === 'auto') return true;
926
+ if (type === 'instant') {
927
+ return !!(decryptedPayload.apiUrl && decryptedPayload.apiKey && decryptedPayload.primaryModel);
928
+ }
929
+ return false;
930
+ }
931
+
932
+ /** Frozen, credential-free view of the task for hook authors. */
933
+ function buildHookTask(task, decryptedPayload) {
934
+ const safe = {};
935
+ for (const [key, value] of Object.entries(decryptedPayload)) {
936
+ if (!CREDENTIAL_PAYLOAD_KEYS.has(key)) safe[key] = value;
937
+ }
938
+ return Object.freeze({
939
+ ...safe,
940
+ id: task.id ?? null,
941
+ uuid: task.uuid ?? null,
942
+ nextSendAt: task.next_send_at ?? null,
943
+ retryCount: task.retry_count ?? 0,
944
+ });
945
+ }
946
+
947
+ function normalizeBeforeFireResult(result) {
948
+ if (Array.isArray(result)) {
949
+ return { messages: result };
950
+ }
951
+ if (result && typeof result === 'object' && Array.isArray(result.messages)) {
952
+ return {
953
+ messages: result.messages,
954
+ maxToolIterations: result.maxToolIterations,
955
+ totalTimeoutMs: result.totalTimeoutMs,
956
+ };
957
+ }
958
+ throw new TypeError(
959
+ 'AGENTIC_BAD_BEFORE_FIRE: onBeforeFire must return ChatMessage[] | { messages, maxToolIterations?, totalTimeoutMs? } | null'
960
+ );
961
+ }
962
+
963
+ function firstPositiveInt(values, fallback) {
964
+ for (const v of values) {
965
+ if (Number.isInteger(v) && v > 0) return v;
966
+ }
967
+ return fallback;
968
+ }
969
+
970
+ function firstPositiveNumber(values, fallback) {
971
+ for (const v of values) {
972
+ if (typeof v === 'number' && Number.isFinite(v) && v > 0) return v;
973
+ }
974
+ return fallback;
975
+ }
976
+
977
+ /**
978
+ * Try the hook-driven fire path for one task.
979
+ *
980
+ * @param {Object} args
981
+ * @param {import('../adapters/interface.js').TaskRow} args.task
982
+ * @param {Object} args.decryptedPayload - decrypted task payload (has credentials; they stop here)
983
+ * @param {string} args.userKey - per-user storage key (for readState decryption)
984
+ * @param {Object} args.ctx - processor ctx ({ db, webpush, vapid, hooks, maxToolIterations, totalTimeoutMs })
985
+ * @returns {Promise<{ handled: false } | { handled: true, result: { success: true, messagesSent: number, status: 'finished'|'skipped', iterations: number } }>}
986
+ * `handled: false` → caller falls back to the legacy frozen-prompt path.
987
+ * Failures (timeout / loop exceeded / config errors) throw — the caller's
988
+ * existing error handling turns them into task retry/failure.
989
+ */
990
+ async function runAgenticFire({ task, decryptedPayload, userKey, ctx }) {
991
+ const hooks = ctx.hooks;
992
+ if (typeof hooks.onLLMOutput !== 'function') {
993
+ throw new Error('AGENTIC_CONFIG_ERROR: hooks.onBeforeFire requires hooks.onLLMOutput to classify LLM rounds');
994
+ }
995
+
996
+ // Injectable seams for tests only (fake clock / no real 1500ms pacing).
997
+ const nowFn = typeof ctx._agenticNow === 'function' ? ctx._agenticNow : Date.now;
998
+ const sleep = typeof ctx._agenticSleep === 'function' ? ctx._agenticSleep : defaultSleep;
999
+
1000
+ const readState = async (namespace) => {
1001
+ if (typeof namespace !== 'string' || !namespace.trim()) {
1002
+ throw new TypeError('readState(namespace) requires a non-empty string');
1003
+ }
1004
+ if (!ctx.db || typeof ctx.db.getClientState !== 'function') return [];
1005
+ const rows = await ctx.db.getClientState(task.user_id, namespace);
1006
+ return Promise.all(rows.map(async (row) => ({
1007
+ namespace: row.namespace,
1008
+ key: row.key,
1009
+ value: await decryptFromStorage(row.value, userKey),
1010
+ updatedAt: row.updated_at,
1011
+ })));
1012
+ };
1013
+
1014
+ const fireCtx = Object.freeze({
1015
+ task: buildHookTask(task, decryptedPayload),
1016
+ userId: task.user_id,
1017
+ readState,
1018
+ now: new Date(nowFn()),
1019
+ });
1020
+
1021
+ const before = await hooks.onBeforeFire(fireCtx);
1022
+ if (before == null) return { handled: false };
1023
+
1024
+ const normalized = normalizeBeforeFireResult(before);
1025
+ const maxToolIterations = firstPositiveInt(
1026
+ [normalized.maxToolIterations, ctx.maxToolIterations],
1027
+ DEFAULT_MAX_TOOL_ITERATIONS
1028
+ );
1029
+ const totalTimeoutMs = firstPositiveNumber(
1030
+ [normalized.totalTimeoutMs, ctx.totalTimeoutMs],
1031
+ DEFAULT_TOTAL_TIMEOUT_MS
1032
+ );
1033
+ const deadline = nowFn() + totalTimeoutMs;
1034
+
1035
+ // Same sessionId scheme as the legacy path: pinned to the task id so a
1036
+ // retried task reuses the same session and clients can group/dedupe.
1037
+ const sessionId = task.id != null ? `sess_task_${task.id}` : `sess_${randomUUID()}`;
1038
+ let messages = normalized.messages.slice();
1039
+
1040
+ for (let iteration = 0; iteration < maxToolIterations; iteration++) {
1041
+ if (nowFn() >= deadline) {
1042
+ throw new Error(`AGENTIC_TOTAL_TIMEOUT: fire chain exceeded ${totalTimeoutMs}ms after ${iteration} LLM round(s)`);
1043
+ }
1044
+
1045
+ // Shrink each round's fetch timeout to the remaining wall-time budget
1046
+ // (capped at the legacy 300s single-call ceiling) — a hung LLM request
1047
+ // must not outlive totalTimeoutMs waiting for its own 300s abort.
1048
+ const roundTimeoutMs = Math.max(1, Math.min(300_000, deadline - nowFn()));
1049
+ const { response: llmResponse } = await callLlm(
1050
+ { ...decryptedPayload, messages },
1051
+ { requireContent: false, timeoutMs: roundTimeoutMs }
1052
+ );
1053
+
1054
+ const assistantMessage = extractAssistantMessage(llmResponse);
1055
+ messages = [...messages, assistantMessage];
1056
+
1057
+ const sessionCtx = buildSessionContext({
1058
+ sessionId,
1059
+ messages,
1060
+ llmResponse,
1061
+ iteration,
1062
+ contactName: decryptedPayload.contactName,
1063
+ avatarUrl: decryptedPayload.avatarUrl || undefined,
1064
+ charId: decryptedPayload.charId,
1065
+ metadata: decryptedPayload.metadata,
1066
+ });
1067
+
1068
+ const decision = await hooks.onLLMOutput(sessionCtx);
1069
+ assertValidDecision(decision, { inlineToolCalls: true });
1070
+
1071
+ if (decision.decision === 'continue') {
1072
+ messages = decision.nextHistory.slice();
1073
+ continue;
1074
+ }
1075
+
1076
+ if (decision.decision === 'skip-push') {
1077
+ return { handled: true, result: { success: true, messagesSent: 0, status: 'skipped', iterations: iteration + 1 } };
1078
+ }
1079
+
1080
+ if (decision.decision === 'finish') {
1081
+ const messagesSent = await sendHookPushPayloads(decision.pushPayloads, decryptedPayload, ctx, sessionId, task, sleep);
1082
+ return { handled: true, result: { success: true, messagesSent, status: 'finished', iterations: iteration + 1 } };
1083
+ }
1084
+
1085
+ // 'tool-request' — execute right here in the worker.
1086
+ const toolCalls = extractToolCallsFromDecision(decision);
1087
+ if (toolCalls.length === 0) {
1088
+ throw new Error('AGENTIC_EMPTY_TOOL_REQUEST: tool-request decision carried no toolCalls (neither decision.toolCalls nor pushPayloads[].toolCalls)');
1089
+ }
1090
+ if (typeof hooks.executeToolCalls !== 'function') {
1091
+ throw new Error('AGENTIC_CONFIG_ERROR: onLLMOutput returned tool-request but hooks.executeToolCalls is not configured');
1092
+ }
1093
+ if (iteration === maxToolIterations - 1) {
1094
+ // No LLM round left to consume the results — executing tools now
1095
+ // would only burn external calls. Fall straight to the exceeded error.
1096
+ break;
1097
+ }
1098
+
1099
+ let toolResults;
1100
+ try {
1101
+ toolResults = await hooks.executeToolCalls(toolCalls, sessionCtx);
1102
+ if (!Array.isArray(toolResults)) {
1103
+ throw new TypeError('executeToolCalls must resolve to an array of { tool_call_id, role: "tool", content }');
1104
+ }
1105
+ } catch (error) {
1106
+ // Feed the failure back as tool results and let the LLM talk its way
1107
+ // out, instead of failing the whole fire.
1108
+ toolResults = toolCalls.map((toolCall) => ({
1109
+ tool_call_id: toolCall && typeof toolCall === 'object' && typeof toolCall.id === 'string' ? toolCall.id : '',
1110
+ role: 'tool',
1111
+ content: `Tool execution failed: ${error?.message ?? String(error)}`,
1112
+ }));
1113
+ }
1114
+
1115
+ // Text-protocol classifiers synthesize toolCalls the raw assistant
1116
+ // message doesn't carry; stamp them on so the appended role:'tool'
1117
+ // results stay valid for OpenAI-compatible APIs.
1118
+ const assistantWithTools = Array.isArray(assistantMessage.tool_calls) && assistantMessage.tool_calls.length > 0
1119
+ ? assistantMessage
1120
+ : { ...assistantMessage, tool_calls: toolCalls };
1121
+ messages = [...messages.slice(0, -1), assistantWithTools, ...toolResults];
1122
+ }
1123
+
1124
+ throw new Error(`AGENTIC_LOOP_EXCEEDED: no finish/skip-push decision within ${maxToolIterations} LLM round(s)`);
1125
+ }
1126
+
1127
+ /**
1128
+ * Deliver the hook's pushPayloads sequentially. Mirrors instant's
1129
+ * sendPushesSequentially: force-overwrite messageIndex/totalMessages,
1130
+ * stamp missing ids, pace with the same 1500ms spacing. Ids are
1131
+ * deterministic per (task, index) so a retried task reuses the same ids
1132
+ * and clients can dedupe.
1133
+ */
1134
+ async function sendHookPushPayloads(pushPayloads, decryptedPayload, ctx, sessionId, task, sleep) {
1135
+ if (!ctx.vapid || !ctx.vapid.email || !ctx.vapid.publicKey || !ctx.vapid.privateKey) {
1136
+ throw new Error('VAPID configuration missing - push notifications cannot be sent');
1137
+ }
1138
+ const pushSubscription = decryptedPayload.pushSubscription;
1139
+ const total = pushPayloads.length;
1140
+ const messageIdBase = task.id != null ? `msg_task_${task.id}` : `msg_${randomUUID()}`;
1141
+
1142
+ for (let i = 0; i < total; i++) {
1143
+ const push = { ...pushPayloads[i] };
1144
+ if (typeof push.messageId !== 'string' || !push.messageId) push.messageId = `${messageIdBase}_hook_${i}`;
1145
+ if (typeof push.sessionId !== 'string' || !push.sessionId) push.sessionId = sessionId;
1146
+ if (typeof push.timestamp !== 'string' || !push.timestamp) push.timestamp = new Date().toISOString();
1147
+ push.messageIndex = i + 1;
1148
+ push.totalMessages = total;
1149
+
1150
+ await ctx.webpush.sendNotification(pushSubscription, JSON.stringify(push));
1151
+ if (i < total - 1) await sleep(SLEEP_BETWEEN_MESSAGES_MS$1);
1152
+ }
1153
+ return total;
1154
+ }
1155
+
702
1156
  /**
703
1157
  * Message Processor (SDK version)
704
1158
  * ReiStandard amsg-server v2.4.0
@@ -799,6 +1253,16 @@ async function processSingleMessage(task, ctx, providedMasterKey) {
799
1253
  const userKey = await deriveUserEncryptionKey(task.user_id, masterKey);
800
1254
  const decryptedPayload = JSON.parse(await decryptFromStorage(task.encrypted_payload, userKey));
801
1255
 
1256
+ // Fire-time hooks: when the host configured onBeforeFire and the task
1257
+ // needs the LLM, offer the agentic path first. onBeforeFire → null
1258
+ // falls straight through to the frozen-prompt chain below, and
1259
+ // deployments without hooks never enter this branch — legacy behavior
1260
+ // is byte-identical.
1261
+ if (ctx.hooks && typeof ctx.hooks.onBeforeFire === 'function' && taskNeedsLlm(decryptedPayload)) {
1262
+ const agentic = await runAgenticFire({ task, decryptedPayload, userKey, ctx });
1263
+ if (agentic.handled) return agentic.result;
1264
+ }
1265
+
802
1266
  let messageContent;
803
1267
  /** @type {unknown} */
804
1268
  let llmResponse = null;
@@ -810,7 +1274,7 @@ async function processSingleMessage(task, ctx, providedMasterKey) {
810
1274
  const hasPrompt = !!decryptedPayload.completePrompt
811
1275
  || (Array.isArray(decryptedPayload.messages) && decryptedPayload.messages.length > 0);
812
1276
  if (hasPrompt && decryptedPayload.apiUrl && decryptedPayload.apiKey && decryptedPayload.primaryModel) {
813
- const aiResult = await _callAI(decryptedPayload);
1277
+ const aiResult = await callLlm(decryptedPayload);
814
1278
  messageContent = aiResult.content;
815
1279
  llmResponse = aiResult.response;
816
1280
  } else if (decryptedPayload.userMessage) {
@@ -820,7 +1284,7 @@ async function processSingleMessage(task, ctx, providedMasterKey) {
820
1284
  }
821
1285
 
822
1286
  } else if (decryptedPayload.messageType === 'prompted' || decryptedPayload.messageType === 'auto') {
823
- const aiResult = await _callAI(decryptedPayload);
1287
+ const aiResult = await callLlm(decryptedPayload);
824
1288
  messageContent = aiResult.content;
825
1289
  llmResponse = aiResult.response;
826
1290
  } else {
@@ -1010,150 +1474,6 @@ async function processMessagesByUuid(uuid, ctx, maxRetries = 2, userId, provided
1010
1474
  }
1011
1475
  }
1012
1476
 
1013
- /**
1014
- * Call an OpenAI-compatible API.
1015
- *
1016
- * Returns the full response object alongside the extracted (trimmed)
1017
- * `content` string. Callers that only need the text can ignore
1018
- * `response`; callers that want `reasoning_content` / `tool_calls`
1019
- * read from `response.choices[0].message`.
1020
- *
1021
- * @private
1022
- * @param {Object} payload
1023
- * @returns {Promise<{ response: unknown, content: string }>}
1024
- */
1025
- async function _callAI(payload) {
1026
- const normalizedApiUrl = normalizeAiApiUrl(payload.apiUrl);
1027
- const requestBody = buildAiRequestBody(payload);
1028
-
1029
- const aiResponse = await fetch(normalizedApiUrl, {
1030
- method: 'POST',
1031
- headers: {
1032
- 'Content-Type': 'application/json',
1033
- 'Authorization': `Bearer ${payload.apiKey}`
1034
- },
1035
- body: JSON.stringify(requestBody),
1036
- signal: AbortSignal.timeout(300000)
1037
- });
1038
-
1039
- if (!aiResponse.ok) {
1040
- if (aiResponse.status === 405) {
1041
- throw new Error(
1042
- `AI API error: 405 Method Not Allowed. ` +
1043
- `apiUrl must point to a full chat endpoint (for example: /chat/completions). ` +
1044
- `Received: ${normalizedApiUrl}`
1045
- );
1046
- }
1047
-
1048
- throw new Error(
1049
- `AI API error: ${aiResponse.status} ${aiResponse.statusText || 'Unknown Error'}. ` +
1050
- `Request URL: ${normalizedApiUrl}`
1051
- );
1052
- }
1053
-
1054
- const aiData = await aiResponse.json();
1055
- const content = aiData?.choices?.[0]?.message?.content;
1056
- if (typeof content !== 'string' || !content.trim()) {
1057
- throw new Error('AI API error: response missing choices[0].message.content');
1058
- }
1059
-
1060
- return { response: aiData, content: content.trim() };
1061
- }
1062
-
1063
- /**
1064
- * Build OpenAI-compatible request body.
1065
- *
1066
- * `max_tokens` is optional:
1067
- * - include it only when payload.maxTokens is provided
1068
- * - omit it when payload.maxTokens is undefined / null
1069
- *
1070
- * @param {Object} payload
1071
- * @returns {Object}
1072
- */
1073
- function buildAiRequestBody(payload) {
1074
- // messages mode (added in v2.2.0): forward the caller's OpenAI-style array
1075
- // verbatim — same contract as @rei-standard/amsg-instant 0.5.0+. No auto
1076
- // role injection, no concatenation back to a single user message. Lets
1077
- // the upstream app preserve system / multi-turn context byte-for-byte
1078
- // across the schedule-message path.
1079
- const llmMessages = Array.isArray(payload.messages) && payload.messages.length > 0
1080
- ? payload.messages
1081
- : [{ role: 'user', content: payload.completePrompt }];
1082
-
1083
- const requestBody = {
1084
- model: payload.primaryModel,
1085
- messages: llmMessages,
1086
- };
1087
-
1088
- // Match the instant package's behavior: only inject default temperature
1089
- // for the legacy completePrompt path; messages mode forwards whatever the
1090
- // upstream app set (or nothing) so behavior matches their main chat path.
1091
- if (payload.temperature !== undefined && payload.temperature !== null) {
1092
- requestBody.temperature = payload.temperature;
1093
- } else if (!Array.isArray(payload.messages)) {
1094
- requestBody.temperature = 0.8;
1095
- }
1096
-
1097
- if (payload.maxTokens === undefined || payload.maxTokens === null) {
1098
- return requestBody;
1099
- }
1100
-
1101
- if (!Number.isInteger(payload.maxTokens) || payload.maxTokens <= 0) {
1102
- throw new Error('Invalid maxTokens: maxTokens must be a positive integer when provided.');
1103
- }
1104
-
1105
- requestBody.max_tokens = payload.maxTokens;
1106
- return requestBody;
1107
- }
1108
-
1109
- /**
1110
- * Normalize AI API URL for OpenAI-compatible chat endpoints.
1111
- *
1112
- * **Keep in sync** with `@rei-standard/amsg-instant`'s
1113
- * `src/message-processor.js` `normalizeAiApiUrl` — same rules, same
1114
- * tests. The two packages share this logic but each carry their own copy
1115
- * to avoid an architectural dependency (server should not depend on the
1116
- * stateless worker package).
1117
- *
1118
- * @param {string} apiUrl
1119
- * @returns {string}
1120
- */
1121
- function normalizeAiApiUrl(apiUrl) {
1122
- if (typeof apiUrl !== 'string' || !apiUrl.trim()) {
1123
- throw new Error(
1124
- 'Invalid apiUrl: apiUrl is required. ' +
1125
- 'Please provide a chat endpoint URL ' +
1126
- '(for example: https://api.openai.com or https://api.openai.com/v1/chat/completions).'
1127
- );
1128
- }
1129
-
1130
- const trimmedApiUrl = apiUrl.trim();
1131
- let parsedUrl;
1132
-
1133
- try {
1134
- parsedUrl = new URL(trimmedApiUrl);
1135
- } catch {
1136
- throw new Error(
1137
- `Invalid apiUrl: "${apiUrl}". Please provide a valid absolute URL.`
1138
- );
1139
- }
1140
-
1141
- let path = parsedUrl.pathname.replace(/\/+$/, '') || '/';
1142
-
1143
- if (/\/chat\/completions$/.test(path)) ; else if (path === '/') {
1144
- // Bare host → assume OpenAI shape.
1145
- path = '/v1/chat/completions';
1146
- } else if (/\/v\d+$/.test(path)) {
1147
- // Path ends in `/v1`, `/v2`, … — caller already versioned the URL.
1148
- // Append only `/chat/completions`; never re-add `/v1`.
1149
- path = `${path}/chat/completions`;
1150
- }
1151
- // Any other custom path is left untouched on purpose.
1152
-
1153
- parsedUrl.pathname = path;
1154
- return parsedUrl.toString();
1155
- }
1156
-
1157
1477
  /**
1158
1478
  * Handler: schedule-message
1159
1479
  * ReiStandard SDK v2.0.1
@@ -1924,6 +2244,24 @@ const SQLITE_INDEXES = [
1924
2244
  }
1925
2245
  ];
1926
2246
 
2247
+ // client_state: cloud mirror of client-side state for the single-user
2248
+ // deployment. One live copy per (user, namespace, key) — not per-task
2249
+ // snapshots. The client is the only writer (batch upsert, last-write-wins
2250
+ // on updated_at); fire-time hooks are the reader. `value` holds
2251
+ // encryptForStorage ciphertext. `updated_at` is a caller-supplied epoch-ms
2252
+ // INTEGER (unlike scheduled_messages' ISO TEXT) so conflict resolution
2253
+ // compares without parsing. Single-user/SQLite only — no Postgres mirror.
2254
+ const CLIENT_STATE_TABLE_SQL = `
2255
+ CREATE TABLE IF NOT EXISTS client_state (
2256
+ user_id TEXT NOT NULL,
2257
+ namespace TEXT NOT NULL,
2258
+ key TEXT NOT NULL,
2259
+ value TEXT NOT NULL,
2260
+ updated_at INTEGER NOT NULL,
2261
+ PRIMARY KEY (user_id, namespace, key)
2262
+ )
2263
+ `;
2264
+
1927
2265
  /**
1928
2266
  * Cloudflare D1 (SQLite) Database Adapter.
1929
2267
  *
@@ -1966,6 +2304,7 @@ class D1Adapter {
1966
2304
 
1967
2305
  async initSchema() {
1968
2306
  await this._db.prepare(SQLITE_TABLE_SQL).run();
2307
+ await this._db.prepare(CLIENT_STATE_TABLE_SQL).run();
1969
2308
 
1970
2309
  const indexResults = [];
1971
2310
  for (const index of SQLITE_INDEXES) {
@@ -1997,6 +2336,7 @@ class D1Adapter {
1997
2336
 
1998
2337
  async dropSchema() {
1999
2338
  await this._db.prepare('DROP TABLE IF EXISTS scheduled_messages').run();
2339
+ await this._db.prepare('DROP TABLE IF EXISTS client_state').run();
2000
2340
  }
2001
2341
 
2002
2342
  async createTask(params) {
@@ -2145,6 +2485,83 @@ class D1Adapter {
2145
2485
  ).bind(uuid, userId).first();
2146
2486
  return row ? row.status : null;
2147
2487
  }
2488
+
2489
+ // ── client_state (single-user cloud state mirror) ──────────────────────
2490
+
2491
+ /**
2492
+ * Batch upsert. Last-write-wins per (namespace, key): an entry older
2493
+ * than the stored row (updatedAt strictly lower) is skipped; equal or
2494
+ * newer overwrites. Values arrive pre-encrypted (the handler encrypts).
2495
+ *
2496
+ * Uses D1's batch() — one network round trip for the whole set (implicit
2497
+ * transaction). The client calls this endpoint inside its few-seconds
2498
+ * background window, so N sequential round trips could eat the whole
2499
+ * window. Bindings without batch() (e.g. the sqlite test shim, custom
2500
+ * adapters) fall back to a sequential loop.
2501
+ *
2502
+ * @param {string} userId
2503
+ * @param {Array<{ namespace: string, key: string, value: string, updatedAt: number }>} entries
2504
+ * @returns {Promise<{ upserted: number, skipped: number }>}
2505
+ */
2506
+ async upsertClientState(userId, entries) {
2507
+ const UPSERT_SQL =
2508
+ `INSERT INTO client_state (user_id, namespace, key, value, updated_at)
2509
+ VALUES (?, ?, ?, ?, ?)
2510
+ ON CONFLICT (user_id, namespace, key) DO UPDATE SET
2511
+ value = excluded.value,
2512
+ updated_at = excluded.updated_at
2513
+ WHERE excluded.updated_at >= client_state.updated_at`;
2514
+
2515
+ let results;
2516
+ if (typeof this._db.batch === 'function') {
2517
+ const statements = entries.map((entry) =>
2518
+ this._db.prepare(UPSERT_SQL).bind(userId, entry.namespace, entry.key, entry.value, entry.updatedAt)
2519
+ );
2520
+ results = await this._db.batch(statements);
2521
+ } else {
2522
+ results = [];
2523
+ for (const entry of entries) {
2524
+ results.push(
2525
+ await this._db.prepare(UPSERT_SQL).bind(userId, entry.namespace, entry.key, entry.value, entry.updatedAt).run()
2526
+ );
2527
+ }
2528
+ }
2529
+
2530
+ let upserted = 0;
2531
+ let skipped = 0;
2532
+ for (const res of results) {
2533
+ if (res.meta.changes > 0) upserted++; else skipped++;
2534
+ }
2535
+ return { upserted, skipped };
2536
+ }
2537
+
2538
+ /**
2539
+ * All entries of one namespace (values still encrypted).
2540
+ *
2541
+ * @param {string} userId
2542
+ * @param {string} namespace
2543
+ * @returns {Promise<Array<{ namespace: string, key: string, value: string, updated_at: number }>>}
2544
+ */
2545
+ async getClientState(userId, namespace) {
2546
+ const res = await this._db.prepare(
2547
+ `SELECT namespace, key, value, updated_at
2548
+ FROM client_state
2549
+ WHERE user_id = ? AND namespace = ?
2550
+ ORDER BY key ASC`
2551
+ ).bind(userId, namespace).all();
2552
+ return res.results || [];
2553
+ }
2554
+
2555
+ /**
2556
+ * Wipe every entry of this user.
2557
+ *
2558
+ * @param {string} userId
2559
+ * @returns {Promise<number>} rows deleted
2560
+ */
2561
+ async clearClientState(userId) {
2562
+ const res = await this._db.prepare('DELETE FROM client_state WHERE user_id = ?').bind(userId).run();
2563
+ return res.meta.changes || 0;
2564
+ }
2148
2565
  }
2149
2566
 
2150
2567
  /**
@@ -2319,6 +2736,179 @@ function createVapidPublicKeyHandler(ctx) {
2319
2736
  return { GET };
2320
2737
  }
2321
2738
 
2739
+ /**
2740
+ * Handler: client-state
2741
+ *
2742
+ * Cloud mirror of client-side state for the single-user deployment. The
2743
+ * client batch-syncs entries up (PUT) whenever convenient — e.g. in the
2744
+ * few-seconds window before iOS backgrounds the page — and fire-time
2745
+ * hooks read them back via ctx.readState(namespace). One live copy per
2746
+ * (user, namespace, key); the client is the only writer.
2747
+ *
2748
+ * PUT /client-state batch upsert, last-write-wins on updatedAt
2749
+ * GET /client-state?namespace=<ns> one namespace's entries (decrypted, response re-encrypted)
2750
+ * DELETE /client-state wipe every entry of this user
2751
+ *
2752
+ * Auth & crypto follow the existing endpoints exactly: X-Client-Token is
2753
+ * all-or-nothing via resolveTenant, PUT bodies must be encrypted
2754
+ * (X-Payload-Encrypted / X-Encryption-Version), values are stored as
2755
+ * encryptForStorage ciphertext under the per-user key, and GET responses
2756
+ * ride the existing encrypted-response envelope.
2757
+ */
2758
+
2759
+
2760
+ // One value may hold a serialized state chunk, but must stay well under
2761
+ // D1's per-row limits — reject early with a clear error instead of an
2762
+ // opaque DB failure.
2763
+ const MAX_STATE_VALUE_BYTES = 200 * 1024;
2764
+ // "a few dozen entries in one background-window request" is the design
2765
+ // load; 200 bounds a single request with generous headroom.
2766
+ const MAX_STATE_ENTRIES_PER_REQUEST = 200;
2767
+ const MAX_NAMESPACE_CHARS = 128;
2768
+ const MAX_KEY_CHARS = 256;
2769
+
2770
+ const utf8 = new TextEncoder();
2771
+
2772
+ function err(status, code, message, details) {
2773
+ const error = details === undefined ? { code, message } : { code, message, details };
2774
+ return { status, body: { success: false, error } };
2775
+ }
2776
+
2777
+ function requireUserId(headers) {
2778
+ const userId = getHeader(headers, 'x-user-id');
2779
+ if (!userId) return { error: err(400, 'USER_ID_REQUIRED', '缺少用户标识符') };
2780
+ if (!isValidUUIDv4(userId)) return { error: err(400, 'INVALID_USER_ID_FORMAT', 'X-User-Id 必须是 UUID v4 格式') };
2781
+ return { userId };
2782
+ }
2783
+
2784
+ function validateEntry(entry, index) {
2785
+ if (!isPlainObject(entry)) return err(400, 'INVALID_STATE_ENTRY', `entries[${index}] 必须是对象`);
2786
+ if (typeof entry.namespace !== 'string' || !entry.namespace.trim() || entry.namespace.length > MAX_NAMESPACE_CHARS) {
2787
+ return err(400, 'INVALID_STATE_NAMESPACE', `entries[${index}].namespace 必须是 1-${MAX_NAMESPACE_CHARS} 字符的字符串`);
2788
+ }
2789
+ if (typeof entry.key !== 'string' || !entry.key.trim() || entry.key.length > MAX_KEY_CHARS) {
2790
+ return err(400, 'INVALID_STATE_KEY', `entries[${index}].key 必须是 1-${MAX_KEY_CHARS} 字符的字符串`);
2791
+ }
2792
+ if (typeof entry.value !== 'string') {
2793
+ return err(400, 'INVALID_STATE_VALUE', `entries[${index}].value 必须是字符串(宿主自行序列化)`);
2794
+ }
2795
+ const bytes = utf8.encode(entry.value).length;
2796
+ if (bytes > MAX_STATE_VALUE_BYTES) {
2797
+ return err(413, 'STATE_VALUE_TOO_LARGE', `entries[${index}].value 超过单条上限`, { index, key: entry.key, bytes, maxBytes: MAX_STATE_VALUE_BYTES });
2798
+ }
2799
+ if (!Number.isInteger(entry.updatedAt) || entry.updatedAt <= 0) {
2800
+ return err(400, 'INVALID_STATE_UPDATED_AT', `entries[${index}].updatedAt 必须是正整数(epoch 毫秒)`);
2801
+ }
2802
+ return null;
2803
+ }
2804
+
2805
+ function createClientStateHandler(ctx) {
2806
+ async function PUT(headers, body) {
2807
+ const tenantResult = await ctx.tenantManager.resolveTenant(headers);
2808
+ if (!tenantResult.ok) return tenantResult.error;
2809
+ const { db, masterKey } = tenantResult.context;
2810
+
2811
+ if (getHeader(headers, 'x-payload-encrypted') !== 'true') {
2812
+ return err(400, 'ENCRYPTION_REQUIRED', '请求体必须加密');
2813
+ }
2814
+ const gate = requireUserId(headers);
2815
+ if (gate.error) return gate.error;
2816
+ const { userId } = gate;
2817
+ if (getHeader(headers, 'x-encryption-version') !== '1') {
2818
+ return err(400, 'UNSUPPORTED_ENCRYPTION_VERSION', '加密版本不支持');
2819
+ }
2820
+
2821
+ const parsedBody = parseEncryptedBody(body);
2822
+ if (!parsedBody.ok) return { status: 400, body: { success: false, error: parsedBody.error } };
2823
+
2824
+ const userKey = await deriveUserEncryptionKey(userId, masterKey);
2825
+ let payload;
2826
+ try {
2827
+ payload = await decryptPayload(parsedBody.data, userKey);
2828
+ } catch (error) {
2829
+ if (error instanceof SyntaxError) {
2830
+ return err(400, 'INVALID_PAYLOAD_FORMAT', '解密后的数据不是有效 JSON');
2831
+ }
2832
+ return err(400, 'DECRYPTION_FAILED', '请求体解密失败');
2833
+ }
2834
+ if (!isPlainObject(payload)) return err(400, 'INVALID_PAYLOAD_FORMAT', '解密后的数据必须是 JSON 对象');
2835
+
2836
+ const entries = payload.entries;
2837
+ if (!Array.isArray(entries) || entries.length === 0) {
2838
+ return err(400, 'INVALID_STATE_ENTRIES', 'entries 必须是非空数组');
2839
+ }
2840
+ if (entries.length > MAX_STATE_ENTRIES_PER_REQUEST) {
2841
+ return err(400, 'TOO_MANY_STATE_ENTRIES', `单次最多 ${MAX_STATE_ENTRIES_PER_REQUEST} 条`, { count: entries.length });
2842
+ }
2843
+ for (let i = 0; i < entries.length; i++) {
2844
+ const invalid = validateEntry(entries[i], i);
2845
+ if (invalid) return invalid;
2846
+ }
2847
+
2848
+ if (typeof db.upsertClientState !== 'function') {
2849
+ return err(501, 'CLIENT_STATE_NOT_SUPPORTED', '当前数据库适配器不支持 client_state');
2850
+ }
2851
+
2852
+ const encryptedEntries = await Promise.all(entries.map(async (entry) => ({
2853
+ namespace: entry.namespace,
2854
+ key: entry.key,
2855
+ value: await encryptForStorage(entry.value, userKey),
2856
+ updatedAt: entry.updatedAt,
2857
+ })));
2858
+
2859
+ const { upserted, skipped } = await db.upsertClientState(userId, encryptedEntries);
2860
+ return { status: 200, body: { success: true, data: { upserted, skipped } } };
2861
+ }
2862
+
2863
+ async function GET(url, headers) {
2864
+ const tenantResult = await ctx.tenantManager.resolveTenant(headers);
2865
+ if (!tenantResult.ok) return tenantResult.error;
2866
+ const { db, masterKey } = tenantResult.context;
2867
+
2868
+ const gate = requireUserId(headers);
2869
+ if (gate.error) return gate.error;
2870
+ const { userId } = gate;
2871
+
2872
+ const namespace = new URL(url, 'https://dummy').searchParams.get('namespace') || '';
2873
+ if (!namespace.trim()) return err(400, 'NAMESPACE_REQUIRED', '必须提供 namespace 查询参数');
2874
+
2875
+ if (typeof db.getClientState !== 'function') {
2876
+ return err(501, 'CLIENT_STATE_NOT_SUPPORTED', '当前数据库适配器不支持 client_state');
2877
+ }
2878
+
2879
+ const userKey = await deriveUserEncryptionKey(userId, masterKey);
2880
+ const rows = await db.getClientState(userId, namespace);
2881
+ const decrypted = await Promise.all(rows.map(async (row) => ({
2882
+ namespace: row.namespace,
2883
+ key: row.key,
2884
+ value: await decryptFromStorage(row.value, userKey),
2885
+ updatedAt: row.updated_at,
2886
+ })));
2887
+
2888
+ const encryptedResponse = await encryptPayload({ namespace, entries: decrypted }, userKey);
2889
+ return { status: 200, body: { success: true, encrypted: true, version: 1, data: encryptedResponse } };
2890
+ }
2891
+
2892
+ async function DELETE(url, headers) {
2893
+ const tenantResult = await ctx.tenantManager.resolveTenant(headers);
2894
+ if (!tenantResult.ok) return tenantResult.error;
2895
+ const { db } = tenantResult.context;
2896
+
2897
+ const gate = requireUserId(headers);
2898
+ if (gate.error) return gate.error;
2899
+ const { userId } = gate;
2900
+
2901
+ if (typeof db.clearClientState !== 'function') {
2902
+ return err(501, 'CLIENT_STATE_NOT_SUPPORTED', '当前数据库适配器不支持 client_state');
2903
+ }
2904
+
2905
+ const deleted = await db.clearClientState(userId);
2906
+ return { status: 200, body: { success: true, data: { deleted } } };
2907
+ }
2908
+
2909
+ return { PUT, GET, DELETE };
2910
+ }
2911
+
2322
2912
  /**
2323
2913
  * Single-user ReiStandard server assembly.
2324
2914
  *
@@ -2334,6 +2924,11 @@ function createVapidPublicKeyHandler(ctx) {
2334
2924
  * @param {string} [config.serverToken] - optional shared secret (X-Client-Token)
2335
2925
  * @param {{ email?: string, publicKey?: string, privateKey?: string }} [config.vapid]
2336
2926
  * @param {{ sendNotification: function }} [config.webpush] - web-push-compatible sender
2927
+ * @param {Object} [config.hooks] - optional fire-time hooks (see lib/agentic-fire.js):
2928
+ * { onBeforeFire, onLLMOutput, executeToolCalls }. When omitted, AI tasks
2929
+ * replay the schedule-time frozen prompt (legacy behavior, unchanged).
2930
+ * @param {number} [config.maxToolIterations] - factory default LLM-round cap for the agentic loop (default 5).
2931
+ * @param {number} [config.totalTimeoutMs] - factory default wall-time ceiling for the agentic loop (default 240000).
2337
2932
  * @returns {{ handlers: Object, ctx: Object }}
2338
2933
  */
2339
2934
 
@@ -2356,7 +2951,13 @@ function createSingleUserServer(config) {
2356
2951
  privateKey: vapid.privateKey || ''
2357
2952
  },
2358
2953
  webpush: config.webpush || null,
2359
- tenantManager
2954
+ tenantManager,
2955
+ // Fire-time hooks (optional): the in-server instant path fires through
2956
+ // processMessagesByUuid with this ctx, so instant-type tasks can take
2957
+ // the agentic path too.
2958
+ hooks: config.hooks || null,
2959
+ maxToolIterations: config.maxToolIterations,
2960
+ totalTimeoutMs: config.totalTimeoutMs
2360
2961
  };
2361
2962
 
2362
2963
  return {
@@ -2368,7 +2969,8 @@ function createSingleUserServer(config) {
2368
2969
  updateMessage: createUpdateMessageHandler(ctx),
2369
2970
  cancelMessage: createCancelMessageHandler(ctx),
2370
2971
  messages: createMessagesHandler(ctx),
2371
- vapidPublicKey: createVapidPublicKeyHandler(ctx)
2972
+ vapidPublicKey: createVapidPublicKeyHandler(ctx),
2973
+ clientState: createClientStateHandler(ctx)
2372
2974
  }
2373
2975
  };
2374
2976
  }
@@ -2389,9 +2991,9 @@ function createSingleUserServer(config) {
2389
2991
 
2390
2992
 
2391
2993
  // RFC 8291 fixed labels (each followed by a NUL byte per HKDF "info" framing).
2392
- const KEY_INFO_PREFIX = utf8('WebPush: info\0');
2393
- const CEK_INFO = utf8('Content-Encoding: aes128gcm\0');
2394
- const NONCE_INFO = utf8('Content-Encoding: nonce\0');
2994
+ const KEY_INFO_PREFIX = utf8$1('WebPush: info\0');
2995
+ const CEK_INFO = utf8$1('Content-Encoding: aes128gcm\0');
2996
+ const NONCE_INFO = utf8$1('Content-Encoding: nonce\0');
2395
2997
 
2396
2998
  const VAPID_DEFAULT_TTL = 60; // seconds — short, matches single-shot instant.
2397
2999
  const VAPID_TOKEN_LIFETIME = 12 * 3600; // 12h — comfortably under the 24h RFC 8292 cap.
@@ -2433,7 +3035,7 @@ async function sendWebPush({ subscription, payload, vapid, ttl, fetch: fetchImpl
2433
3035
  }
2434
3036
 
2435
3037
  const encryptedBody = await encryptPushPayload({
2436
- plaintext: utf8(payload),
3038
+ plaintext: utf8$1(payload),
2437
3039
  uaPublicKey: base64UrlToBytes(subscriptionKeys.p256dh),
2438
3040
  authSecret: base64UrlToBytes(subscriptionKeys.auth),
2439
3041
  });
@@ -2606,7 +3208,7 @@ async function buildVapidJwt({ audience, subject, publicKey, privateKey }) {
2606
3208
  sub: subject,
2607
3209
  });
2608
3210
 
2609
- const signingInput = utf8(`${header}.${payload}`);
3211
+ const signingInput = utf8$1(`${header}.${payload}`);
2610
3212
 
2611
3213
  const pubBytes = base64UrlToBytes(publicKey);
2612
3214
  const privBytes = base64UrlToBytes(privateKey);
@@ -2717,10 +3319,19 @@ function createWebCryptoWebPush(vapid = {}, { ttl = SCHEDULED_DEFAULT_TTL } = {}
2717
3319
  * DELETE /cancel-message → delete
2718
3320
  * GET /vapid-public-key → this worker's VAPID public key (for the frontend's
2719
3321
  * Web Push subscription); 503 if VAPID_PUBLIC_KEY unset
3322
+ * PUT /client-state → batch upsert client state (last-write-wins on updatedAt)
3323
+ * GET /client-state → read one namespace's entries (?namespace=<ns>)
3324
+ * DELETE /client-state → wipe this user's client state
2720
3325
  *
2721
3326
  * CORS is opt-in: pass `cors: { origin }` in the config (a fixed origin, '*', or
2722
3327
  * an (origin) => allowedOrigin function) to answer OPTIONS preflights and echo
2723
3328
  * Access-Control-* on responses. With no `cors` the Worker stays same-origin.
3329
+ *
3330
+ * Fire-time hooks are opt-in too: pass `hooks: { onBeforeFire, onLLMOutput,
3331
+ * executeToolCalls }` (+ optional `maxToolIterations` / `totalTimeoutMs`) in the
3332
+ * config to let scheduled AI tasks assemble their prompt and run a server-side
3333
+ * tool loop at fire time. Omit them and AI tasks replay the schedule-time frozen
3334
+ * prompt exactly as before. See lib/agentic-fire.js.
2724
3335
  */
2725
3336
 
2726
3337
 
@@ -2815,6 +3426,12 @@ function createSingleUserCloudflareWorker(buildConfig) {
2815
3426
  result = await server.handlers.cancelMessage.DELETE(url, headers);
2816
3427
  } else if (method === 'GET' && pathname.endsWith('/vapid-public-key')) {
2817
3428
  result = await server.handlers.vapidPublicKey.GET(url, headers);
3429
+ } else if (method === 'PUT' && pathname.endsWith('/client-state')) {
3430
+ result = await server.handlers.clientState.PUT(headers, await request.text());
3431
+ } else if (method === 'GET' && pathname.endsWith('/client-state')) {
3432
+ result = await server.handlers.clientState.GET(url, headers);
3433
+ } else if (method === 'DELETE' && pathname.endsWith('/client-state')) {
3434
+ result = await server.handlers.clientState.DELETE(url, headers);
2818
3435
  } else {
2819
3436
  result = { status: 404, body: { success: false, error: { code: 'NOT_FOUND', message: 'Unknown route' } } };
2820
3437
  }
@@ -2836,7 +3453,17 @@ function createSingleUserCloudflareWorker(buildConfig) {
2836
3453
  // Swallow tick failures: pending tasks stay pending, so the next cron tick
2837
3454
  // retries them. Logging keeps the failure visible in the tail log.
2838
3455
  try {
2839
- await runScheduledTick({ db: cfg.db, masterKey: cfg.masterKey, vapid, webpush: cfg.webpush });
3456
+ await runScheduledTick({
3457
+ db: cfg.db,
3458
+ masterKey: cfg.masterKey,
3459
+ vapid,
3460
+ webpush: cfg.webpush,
3461
+ // Fire-time hooks (optional; see lib/agentic-fire.js). runScheduledTick
3462
+ // spreads its ctx into processSingleMessage, so these ride along.
3463
+ hooks: cfg.hooks || null,
3464
+ maxToolIterations: cfg.maxToolIterations,
3465
+ totalTimeoutMs: cfg.totalTimeoutMs
3466
+ });
2840
3467
  } catch (error) {
2841
3468
  console.error('[amsg single-user] scheduled(): tick failed:', error && error.message);
2842
3469
  }