@volter/twin-deepseek 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +198 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/deepseek-budget.d.ts +51 -0
  6. package/dist/src/deepseek-budget.js +152 -0
  7. package/dist/src/deepseek-cache.d.ts +56 -0
  8. package/dist/src/deepseek-cache.js +151 -0
  9. package/dist/src/deepseek-capabilities.d.ts +4 -0
  10. package/dist/src/deepseek-capabilities.js +1520 -0
  11. package/dist/src/deepseek-conformance.d.ts +14 -0
  12. package/dist/src/deepseek-conformance.js +473 -0
  13. package/dist/src/deepseek-connector.d.ts +168 -0
  14. package/dist/src/deepseek-connector.js +386 -0
  15. package/dist/src/deepseek-models.d.ts +30 -0
  16. package/dist/src/deepseek-models.js +38 -0
  17. package/dist/src/deepseek-scenario.d.ts +55 -0
  18. package/dist/src/deepseek-scenario.js +170 -0
  19. package/dist/src/deepseek-server.d.ts +16 -0
  20. package/dist/src/deepseek-server.js +191 -0
  21. package/dist/src/deepseek-stub.d.ts +75 -0
  22. package/dist/src/deepseek-stub.js +191 -0
  23. package/dist/src/deepseek-twin.d.ts +77 -0
  24. package/dist/src/deepseek-twin.js +1103 -0
  25. package/dist/src/deepseek-types.d.ts +172 -0
  26. package/dist/src/deepseek-types.js +26 -0
  27. package/dist/src/index.d.ts +15 -0
  28. package/dist/src/index.js +93 -0
  29. package/package.json +68 -0
  30. package/src/cli.ts +27 -0
  31. package/src/deepseek-budget.ts +178 -0
  32. package/src/deepseek-cache.ts +159 -0
  33. package/src/deepseek-capabilities.ts +1443 -0
  34. package/src/deepseek-conformance.ts +512 -0
  35. package/src/deepseek-connector.ts +440 -0
  36. package/src/deepseek-models.ts +65 -0
  37. package/src/deepseek-scenario.ts +188 -0
  38. package/src/deepseek-server.ts +201 -0
  39. package/src/deepseek-stub.ts +200 -0
  40. package/src/deepseek-twin.ts +1163 -0
  41. package/src/deepseek-types.ts +201 -0
  42. package/src/index.ts +133 -0
@@ -0,0 +1,1103 @@
1
+ // DeepSeek twin REQUEST HANDLER — the canonical DeepSeek Platform API surface for the twin.
2
+ // Contract: handleDeepSeekTwinRequest({method, path, body}) -> {status, body}. It is the faithful
3
+ // DeepSeek API the real `@ai-sdk/deepseek` (pointed at this baseURL) talks to UNMODIFIED.
4
+ //
5
+ // THE HONEST DESIGN: the twin cannot run the model, so `POST /chat/completions` and
6
+ // `POST /beta/completions` return DETERMINISTIC STUB output (deepseek-stub.ts) clearly labeled a
7
+ // twin stub — never pretending to be real inference. But the ENTIRE PROTOCOL ENVELOPE is
8
+ // vendor-faithful: response shapes, SSE chunk sequence, tool_calls, `reasoning_content`,
9
+ // finish_reason, and DeepSeek's KV-cache-bearing `usage`. The genuinely stateful surface is real:
10
+ // • GET /models — the published catalog (+ anything a pull observed)
11
+ // • GET /user/balance — kernel state; drives the real 402 path
12
+ // • POST/GET/DELETE /files — stateful (kernel action log)
13
+ // • the context-cache prefix ledger — stateful (deepseek-cache.ts), which is what makes
14
+ // `prompt_cache_hit_tokens` a measured fact rather than a fabricated number
15
+ //
16
+ // ═══ DEEPSEEK IS OPENAI-COMPATIBLE, WHICH IS EXACTLY WHY THE REJECTIONS ARE THE SURFACE ═══
17
+ // ADDING_A_TWIN.md §0 names this hazard by name: copying an OpenAI-shaped exemplar's PERMISSIVE
18
+ // validation is the bug, because what distinguishes the real vendor is what it REFUSES. Every
19
+ // refusal below is grounded in a first-party source, cited inline. The headline divergences:
20
+ //
21
+ // • THERE IS NO `/v1` PREFIX. `base_url` is `https://api.deepseek.com` and the path is
22
+ // `/chat/completions` (api-docs.deepseek.com, "Your First API Call", read 2026-08-31). A twin
23
+ // that answered on `/v1/chat/completions` would be serving a route the vendor does not.
24
+ // • DEEPSEEK'S DOCUMENTED ERROR TABLE IS {400, 401, 402, 422, 429, 500, 503} — no 404, and it
25
+ // includes two OpenAI does not use: 402 Insufficient Balance and 422 Invalid Parameters
26
+ // (api-docs.deepseek.com/quick_start/error_codes). OpenAI 400s a bad parameter; DeepSeek 422s
27
+ // it. 400 is reserved for "Invalid request body format".
28
+ // • `n`, `seed`, `logit_bias`, `top_k`, `user`, `max_completion_tokens`, `service_tier`,
29
+ // `parallel_tool_calls`, `functions` are NOT DeepSeek parameters. The accepted set is a closed
30
+ // documented table (api-docs.deepseek.com/api/create-chat-completion), which §6 licenses as a
31
+ // literal-allowlist oracle — so anything outside CHAT_PARAMS is 422 by name.
32
+ // • `response_format` accepts `text` and `json_object` ONLY. `json_schema` is an OpenAI feature;
33
+ // `@ai-sdk/deepseek` confirms it by INJECTING the schema into a system message instead of
34
+ // sending `response_format.json_schema` (src/chat/convert-to-deepseek-chat-messages.ts).
35
+ // • `user_id`, not `user` — with a regex and a 512-character cap the SDK enforces client-side and
36
+ // the vendor documents server-side.
37
+ // • Assistant-prefix completion and strict tool calls exist ONLY under the `/beta` base URL.
38
+ // • `frequency_penalty` / `presence_penalty` are DEPRECATED but MUST NOT ERROR, and
39
+ // `temperature` / `top_p` are IGNORED (not rejected) while thinking is enabled: "setting these
40
+ // parameters will not trigger an error but will also have no effect"
41
+ // (api-docs.deepseek.com/guides/thinking_mode). Rejecting them would be the mirror-image
42
+ // infidelity of the permissiveness this file exists to avoid.
43
+ // • With `tools` in the request, previous assistant turns MUST carry `reasoning_content` back or
44
+ // "the API will return a 400 error" (same page).
45
+ //
46
+ // State lives in the kernel action log (D1): all writes are local actions, reads are the
47
+ // projection. No real DeepSeek is ever called from this path (D4). Streaming uses an INJECTED sink
48
+ // — no real sockets / setTimeout (D5 verify is offline + deterministic).
49
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
50
+ import { cacheHitTokens, recordCachePrefixes } from "./deepseek-cache.js";
51
+ import { DEEPSEEK_MODELS, FIM_MODELS, RETIRED_MODEL_IDS, findModel, isThinkingModel } from "./deepseek-models.js";
52
+ import { buildUsage, countPromptTokens, estimateTokens, fnv1a, stubAssistantText, stubFimText, stubFingerprint, stubJsonObject, stubReasoningText, stubToolCall, } from "./deepseek-stub.js";
53
+ import { realizeDeepSeekRespond } from "./deepseek-scenario.js";
54
+ const SERVICE = 'deepseek';
55
+ /** DeepSeek's beta base URL suffix. `@ai-sdk/deepseek` turns prefix-completion and strict tool
56
+ * calls on iff `baseURL.endsWith('/beta')` (src/deepseek-provider.ts), so the twin gates the same
57
+ * two features on the same path segment. */
58
+ export const DEEPSEEK_BETA_PREFIX = '/beta';
59
+ /** DeepSeek's Anthropic-compatible surface lives here (api-docs.deepseek.com/guides/anthropic_api).
60
+ * It is real vendor surface this twin does not model yet — filed as `deepseek.anthropic.messages`
61
+ * and answered like any other unmodeled operation, never faked. */
62
+ export const DEEPSEEK_ANTHROPIC_PREFIX = '/anthropic';
63
+ // ── vendor-shaped errors ──────────────────────────────────────────────────────────────
64
+ /**
65
+ * DeepSeek's error envelope. `message` is the only key the vendor's own decoder requires; `type`,
66
+ * `param` and `code` are nullish (`deepSeekErrorSchema`, `@ai-sdk/deepseek@3.0.37`
67
+ * src/chat/deepseek-chat-api-types.ts — an SDK-generated schema, which §6's precedence order ranks
68
+ * above a rendered docs example).
69
+ *
70
+ * THE `type` STRINGS ARE NOT INVENTED AND NOT GUESSED. DeepSeek publishes STATUS codes and their
71
+ * causes, not envelope `type` values. Where `@ai-sdk/deepseek`'s own discriminator table
72
+ * (`getDeepSeekStreamErrorMetadata`) names a type that maps to the same status DeepSeek documents,
73
+ * the twin uses that string — the SDK wrote that table for THIS provider, so it is first-party
74
+ * evidence about what DeepSeek emits. Where it names none that agrees (402 has no entry at all),
75
+ * the twin OMITS `type` rather than making one up, and the capability asserts status + message
76
+ * only. The `message` strings are DeepSeek's own documented causes
77
+ * (api-docs.deepseek.com/quick_start/error_codes) plus a specific detail.
78
+ */
79
+ function errBody(message, type) {
80
+ return { error: { message, ...(type !== undefined ? { type } : {}) } };
81
+ }
82
+ /** 400 — "Invalid request body format". Reserved for a body the vendor cannot parse or a
83
+ * structurally impossible message list, NOT for a bad parameter value (that is 422). */
84
+ function invalidFormat(detail) {
85
+ return { status: 400, body: errBody(`Invalid request body format: ${detail}`, 'invalid_request_error') };
86
+ }
87
+ /** 422 — "Your request contains invalid parameters". DeepSeek's status for a well-formed body with
88
+ * a parameter the API refuses; this is the single most common OpenAI divergence in this file. */
89
+ function invalidParameters(detail) {
90
+ // NO `type`, deliberately — and this is the file's own stated rule applied to itself (see
91
+ // errBody). `@ai-sdk/deepseek`'s discriminator table has NO 422 entry at all, and the
92
+ // `invalid_request_error` this used to emit is mapped there to `statusCode: 400` — so a real
93
+ // client resolving a streamed 422 would have been told 400 by the twin's own discriminator.
94
+ // Where nothing first-party names a type for a status, the twin omits the field rather than
95
+ // inventing one, exactly as it does for 402. (§9 round one, SHOULD-FIX 3.)
96
+ return { status: 422, body: errBody(`Your request contains invalid parameters: ${detail}`) };
97
+ }
98
+ /** 401 — "Authentication fails due to the wrong API key". */
99
+ function authError() {
100
+ return { status: 401, body: errBody('Authentication fails due to the wrong API key', 'authentication_error') };
101
+ }
102
+ /** 402 — "You have run out of balance". DeepSeek's own status; OpenAI has no 402. No `type` is
103
+ * emitted: nothing first-party names one for this status (see errBody). */
104
+ function insufficientBalance() {
105
+ return { status: 402, body: errBody('You have run out of balance') };
106
+ }
107
+ /** 429 — "You are sending requests too quickly". */
108
+ function rateLimitError() {
109
+ return { status: 429, body: errBody('You are sending requests too quickly', 'rate_limit_exceeded') };
110
+ }
111
+ /**
112
+ * The reply to a route DeepSeek does not serve, or one this twin does not model yet.
113
+ *
114
+ * HONESTY NOTE: DeepSeek's published error table (400/401/402/422/429/500/503) covers REQUEST
115
+ * errors and contains no 404, and the vendor publishes nothing about its unknown-route envelope.
116
+ * The twin answers 404 with the vendor's envelope SHAPE because an unmodeled operation must fail
117
+ * rather than fake a success (docs/contributing/architecture.md); the exact status/body a real gateway returns for an
118
+ * unrouted path is filed as `deepseek.errors.unknown_route_envelope` (todo), and no capability
119
+ * asserts a published fact it does not have.
120
+ */
121
+ function notFound(message) {
122
+ return { status: 404, body: errBody(message, 'not_found_error') };
123
+ }
124
+ // ── modeled authentication (401) ────────────────────────────────────────────────────────
125
+ // Real DeepSeek requires a bearer credential on every request and returns 401 when it is missing or
126
+ // wrong. The twin can't validate against real keys, so it models the CHECKABLE failures: a missing
127
+ // credential, and a reserved sentinel for the invalid-key path. Any other non-empty key is
128
+ // accepted. Trusted in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated.
129
+ function checkAuth(req) {
130
+ const auth = req.headers?.['authorization'];
131
+ const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
132
+ const key = (req.apiKey ?? '').trim() || bearer;
133
+ if (!key)
134
+ return authError();
135
+ if (key === 'sk_invalid' || key === 'invalid')
136
+ return authError();
137
+ return null;
138
+ }
139
+ function triggered(req, header) {
140
+ const v = req.headers?.[header];
141
+ return v === '1' || v === 'true';
142
+ }
143
+ function nowEpoch(occurredAt) {
144
+ return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
145
+ }
146
+ // ── kernel helpers ──────────────────────────────────────────────────────────────────────
147
+ function rows(type, root) {
148
+ return projectResources(SERVICE, root).filter((r) => r.type === type);
149
+ }
150
+ /**
151
+ * Mint the next local file id. Derived from the ID SET ALREADY IN STATE (a scan of the projection),
152
+ * never a row count — a count-mint silently clobbers a pulled vendor id sitting in a gap above the
153
+ * count (ADDING_A_TWIN.md §5). Two further properties matter:
154
+ * • the `twin` infix namespaces LOCAL mints, so a pulled DeepSeek id (`file-api-a1b2c3d4e5f6g7h8`)
155
+ * is not matched by this regex and therefore is not re-minted. "Never" would be over-stated: a
156
+ * vendor id of literally `twin000000000123` has the same 16-character width and would match.
157
+ * That is astronomically unlikely rather than impossible, and saying so is the honest form
158
+ * (§9 round one, NIT 19);
159
+ * • the scan includes TOMBSTONED rows (a soft-deleted file keeps its projection row), so the
160
+ * counter RATCHETS across delete→recreate and a deleted id is never handed out twice.
161
+ * The shape matches DeepSeek's own `file-api-xxxxxxxxxxxxxxxx` (16 characters after the prefix).
162
+ */
163
+ const LOCAL_FILE_ID = /^file-api-twin(\d{12})$/;
164
+ function nextFileId(root) {
165
+ let max = 0;
166
+ for (const r of rows('file', root)) {
167
+ const m = LOCAL_FILE_ID.exec(String(r.id));
168
+ if (m)
169
+ max = Math.max(max, Number(m[1]));
170
+ }
171
+ return `file-api-twin${String(max + 1).padStart(12, '0')}`;
172
+ }
173
+ /** Models observed by a connector pull, reshaped into the served three-key model object. */
174
+ function pulledModels(root) {
175
+ return rows('model', root)
176
+ .filter((r) => !r._deleted)
177
+ .map((r) => ({ id: r.id, object: 'model', owned_by: r.owned_by ?? 'deepseek' }));
178
+ }
179
+ /** The catalog a request sees: the published table, with any PULLED row of the same id overriding
180
+ * it — otherwise a pulled account's model list is folded into the log and then never served. */
181
+ function servedModels(root) {
182
+ const byId = new Map();
183
+ for (const m of DEEPSEEK_MODELS)
184
+ byId.set(m.id, m);
185
+ for (const m of pulledModels(root))
186
+ byId.set(String(m.id), m);
187
+ return [...byId.values()];
188
+ }
189
+ function getRow(type, id, root) {
190
+ return rows(type, root).find((r) => r.id === id);
191
+ }
192
+ /** The account balance the twin serves and enforces. DeepSeek's own example response is the
193
+ * default until a connector pull or a twin-only seed replaces it. */
194
+ const DEFAULT_BALANCE = {
195
+ is_available: true,
196
+ balance_infos: [{ currency: 'CNY', total_balance: '110.00', granted_balance: '10.00', topped_up_balance: '100.00' }],
197
+ };
198
+ function servedBalance(root) {
199
+ const row = getRow('balance', 'account', root);
200
+ if (!row)
201
+ return DEFAULT_BALANCE;
202
+ return {
203
+ is_available: row.is_available !== false,
204
+ balance_infos: Array.isArray(row.balance_infos) ? row.balance_infos : DEFAULT_BALANCE.balance_infos,
205
+ };
206
+ }
207
+ // ── request parsing ─────────────────────────────────────────────────────────────────────
208
+ /** Parse the body, distinguishing "no body" from "unparseable body" — the latter is DeepSeek's
209
+ * 400 "Invalid request body format", which is the ONLY thing 400 is for. */
210
+ function parseJson(body) {
211
+ // THE LINE BETWEEN 400 AND 422, drawn once and applied everywhere: 400 is for a body this API
212
+ // cannot parse as a JSON object at all — absent, malformed, or an array/scalar. EVERYTHING about
213
+ // the parsed CONTENT (a missing required parameter, a bad type, a value outside a closed set) is
214
+ // 422 "Your request contains invalid parameters". The first cut had these mixed — a missing
215
+ // `messages` answered 400 while a missing `model` answered 422, which is the same class of error
216
+ // reported two different ways.
217
+ if (!body || !body.trim())
218
+ return { error: invalidFormat('a JSON request body is required') };
219
+ try {
220
+ const v = JSON.parse(body);
221
+ if (!v || typeof v !== 'object' || Array.isArray(v))
222
+ return { error: invalidFormat('the request body must be a JSON object') };
223
+ return { params: v };
224
+ }
225
+ catch {
226
+ return { error: invalidFormat('the request body is not valid JSON') };
227
+ }
228
+ }
229
+ /**
230
+ * THE CLOSED PARAMETER TABLE for `POST /chat/completions`
231
+ * (api-docs.deepseek.com/api/create-chat-completion, read 2026-08-31) plus the two streaming keys
232
+ * and the top-level `reasoning_effort` `@ai-sdk/deepseek` actually sends.
233
+ *
234
+ * §6 licenses a literal allowlist as an ORACLE precisely when the vendor documents a closed set,
235
+ * and this is one — which is what gives `n`, `seed`, `logit_bias`, `top_k`, `user`,
236
+ * `max_completion_tokens`, `service_tier` and `parallel_tool_calls` a *checkable* refusal instead
237
+ * of the silent acceptance an OpenAI-copied twin gives them.
238
+ *
239
+ * DOC CONFLICT, recorded rather than resolved by guess: the reference table puts reasoning effort
240
+ * at `thinking.reasoning_effort`, while `@ai-sdk/deepseek` sends a TOP-LEVEL `reasoning_effort`
241
+ * (src/chat/deepseek-chat-language-model.ts). Both are first-party. The twin accepts BOTH placements
242
+ * rather than rejecting a shape a first-party client demonstrably sends; it never invents a third.
243
+ */
244
+ const CHAT_PARAMS = new Set([
245
+ 'messages', 'model', 'thinking', 'max_tokens', 'response_format', 'stop', 'stream',
246
+ 'stream_options', 'temperature', 'top_p', 'tools', 'tool_choice', 'logprobs', 'top_logprobs',
247
+ 'user_id', 'frequency_penalty', 'presence_penalty', 'reasoning_effort',
248
+ ]);
249
+ /** The closed table for `POST /beta/completions` (api-docs.deepseek.com/api/create-completion). */
250
+ const FIM_PARAMS = new Set([
251
+ 'model', 'prompt', 'suffix', 'echo', 'max_tokens', 'temperature', 'top_p', 'logprobs', 'stop',
252
+ 'stream', 'stream_options', 'frequency_penalty', 'presence_penalty',
253
+ ]);
254
+ const MESSAGE_ROLES = new Set(['system', 'user', 'assistant', 'tool']);
255
+ /** `thinking.type` — a closed set of two (`enabled`, `disabled`). */
256
+ const THINKING_TYPES = new Set(['enabled', 'disabled']);
257
+ /** The canonical reasoning-effort values. */
258
+ const REASONING_EFFORTS = new Set(['low', 'high', 'max']);
259
+ /**
260
+ * Legacy values DeepSeek maps rather than refuses: "`medium` and `xhigh` both map to actual `high`
261
+ * effort" (api-docs.deepseek.com/guides/thinking_mode). Recorded as a MAP, not silently swallowed,
262
+ * so the twin normalizes exactly the two values the vendor says it normalizes and 422s the rest.
263
+ */
264
+ // A SECOND recorded conflict, alongside the placement one above: the vendor's mapping table says
265
+ // BOTH `medium` and `xhigh` become `high`, while `@ai-sdk/deepseek` maps `xhigh` to `max`
266
+ // client-side (`mapDeepSeekProviderReasoningEffort`; docs/30-deepseek.mdx: "reasoningEffort 'xhigh'
267
+ // becomes 'max'"). The twin follows the VENDOR page because this is server-side behaviour and the
268
+ // SDK's mapping happens before the request is sent — so the server never sees `xhigh` from that
269
+ // client anyway. (§9 round one, NIT 14.)
270
+ const LEGACY_EFFORTS = { medium: 'high', xhigh: 'high' };
271
+ /** "Up to 16 sequences where the API will stop" (api-docs.deepseek.com/api/create-chat-completion). */
272
+ const MAX_STOP_SEQUENCES = 16;
273
+ /** "Nullable; max 128 functions" (same page). */
274
+ const MAX_TOOLS = 128;
275
+ /** `user_id`: "a-zA-Z0-9, hyphens, underscores", "max 512 chars" (same page; enforced client-side by
276
+ * `@ai-sdk/deepseek`'s `deepseekLanguageModelChatOptions` with the identical regex and cap). */
277
+ const USER_ID_RE = /^[a-zA-Z0-9_-]+$/;
278
+ const USER_ID_MAX = 512;
279
+ function asStringArray(v) {
280
+ if (typeof v === 'string')
281
+ return [v];
282
+ if (Array.isArray(v) && v.every((s) => typeof s === 'string'))
283
+ return v;
284
+ return null;
285
+ }
286
+ function validateChat(params, beta) {
287
+ // (1) The closed parameter table. A key outside it is 422 BY NAME — the flagship
288
+ // OpenAI-divergence check (see CHAT_PARAMS).
289
+ for (const key of Object.keys(params)) {
290
+ if (!CHAT_PARAMS.has(key))
291
+ return { error: invalidParameters(`'${key}' is not a parameter of the DeepSeek chat completions API`) };
292
+ }
293
+ // (2) Structure. A missing or ill-typed `messages` is a complaint about parsed CONTENT and is
294
+ // therefore 422, not 400 — see `parseJson`, which owns the whole of the 400 side. (§9 round
295
+ // two, SHOULD-FIX 7: this comment still stated the pre-fix rule the code had stopped
296
+ // following, which is exactly the class round one caught elsewhere.)
297
+ if (!Array.isArray(params.messages))
298
+ return { error: invalidParameters("'messages' is a required array") };
299
+ if (params.messages.length === 0)
300
+ return { error: invalidParameters("'messages' must contain at least 1 item") };
301
+ const messages = params.messages;
302
+ for (const m of messages) {
303
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string')
304
+ return { error: invalidParameters("each entry of 'messages' must be an object with a 'role'") };
305
+ if (!MESSAGE_ROLES.has(m.role))
306
+ return { error: invalidParameters(`'messages[].role' must be one of ${[...MESSAGE_ROLES].map((r) => `'${r}'`).join(', ')}`) };
307
+ // MESSAGE FIELD TYPES, brought under the same 422 rule as everything else about parsed content.
308
+ // §9 round two, SHOULD-FIX 2 + 3: these were unchecked, and both failure modes were worse than a
309
+ // wrong status. A non-string `name` reached `estimateTokens` and produced NaN, which serialises
310
+ // as `null` — so the twin answered 200 with `prompt_tokens: null`, breaking the cache invariant
311
+ // its own capability asserts and handing its own fidelity client a body the SDK's usage schema
312
+ // cannot decode. A non-array `tool_calls` threw a TypeError that surfaced as a 500 with
313
+ // `internal_server_error`, which the SDK marks RETRYABLE — so a client would retry a permanently
314
+ // invalid request forever.
315
+ for (const field of ['name', 'reasoning_content', 'tool_call_id']) {
316
+ const v = m[field];
317
+ if (v !== undefined && v !== null && typeof v !== 'string') {
318
+ return { error: invalidParameters(`'messages[].${field}' must be a string`) };
319
+ }
320
+ }
321
+ if (m.tool_calls !== undefined && m.tool_calls !== null && !Array.isArray(m.tool_calls)) {
322
+ return { error: invalidParameters("'messages[].tool_calls' must be an array") };
323
+ }
324
+ for (const tc of m.tool_calls ?? []) {
325
+ if (!tc || typeof tc !== 'object' || typeof tc.function?.name !== 'string' || typeof tc.function?.arguments !== 'string') {
326
+ return { error: invalidParameters("each entry of 'messages[].tool_calls' must be { id, type: 'function', function: { name, arguments } }") };
327
+ }
328
+ }
329
+ }
330
+ // (3) Model. The catalog is closed and `deepseek-chat` / `deepseek-reasoner` were RETIRED on
331
+ // 2026-07-24 (`@ai-sdk/deepseek` docs/30-deepseek.mdx). A retired alias gets a message that
332
+ // says so instead of a generic refusal.
333
+ if (typeof params.model !== 'string' || !params.model)
334
+ return { error: invalidParameters("'model' is a required string") };
335
+ const model = params.model;
336
+ if (!findModel(model)) {
337
+ if (RETIRED_MODEL_IDS.has(model)) {
338
+ return { error: invalidParameters(`'model' "${model}" was retired on 2026-07-24; use 'deepseek-v4-flash' or 'deepseek-v4-pro'`) };
339
+ }
340
+ return { error: invalidParameters(`'model' "${model}" does not exist`) };
341
+ }
342
+ // (4) Thinking. `thinking.type` is a closed set; effort is closed with two legacy values the
343
+ // vendor documents as MAPPED (never a third behaviour invented here).
344
+ let thinkingEnabled = isThinkingModel(model); // "thinking mode is enabled" by default on V4
345
+ let effortRaw = params.reasoning_effort;
346
+ const thinking = params.thinking;
347
+ if (thinking !== undefined && thinking !== null) {
348
+ if (typeof thinking !== 'object' || Array.isArray(thinking))
349
+ return { error: invalidParameters("'thinking' must be an object") };
350
+ const t = thinking;
351
+ if (t.type !== undefined && t.type !== null) {
352
+ if (typeof t.type !== 'string' || !THINKING_TYPES.has(t.type)) {
353
+ return { error: invalidParameters(`'thinking.type' must be one of ${[...THINKING_TYPES].map((v) => `'${v}'`).join(', ')}`) };
354
+ }
355
+ thinkingEnabled = t.type === 'enabled';
356
+ }
357
+ if (t.reasoning_effort !== undefined && t.reasoning_effort !== null)
358
+ effortRaw = t.reasoning_effort;
359
+ }
360
+ // "Default: … effort level set to `high`" (api-docs.deepseek.com/guides/thinking_mode).
361
+ let reasoningEffort = 'high';
362
+ if (effortRaw !== undefined && effortRaw !== null) {
363
+ const raw = typeof effortRaw === 'string' ? effortRaw : '';
364
+ if (REASONING_EFFORTS.has(raw))
365
+ reasoningEffort = raw;
366
+ else if (raw in LEGACY_EFFORTS)
367
+ reasoningEffort = LEGACY_EFFORTS[raw];
368
+ else
369
+ return { error: invalidParameters(`'reasoning_effort' must be one of ${[...REASONING_EFFORTS].map((v) => `'${v}'`).join(', ')}`) };
370
+ }
371
+ // (5) Sampling. `temperature` 0–2, `top_p` 0–1 (documented ranges). NOTE what is NOT here:
372
+ // `frequency_penalty` / `presence_penalty` are accepted and ignored (deprecated, "no
373
+ // effect"), and temperature/top_p are IGNORED — not rejected — while thinking is enabled.
374
+ if (params.temperature !== undefined && params.temperature !== null) {
375
+ const t = params.temperature;
376
+ if (typeof t !== 'number' || Number.isNaN(t) || t < 0 || t > 2)
377
+ return { error: invalidParameters("'temperature' must be a number between 0 and 2") };
378
+ }
379
+ if (params.top_p !== undefined && params.top_p !== null) {
380
+ const p = params.top_p;
381
+ if (typeof p !== 'number' || Number.isNaN(p) || p <= 0 || p > 1)
382
+ return { error: invalidParameters("'top_p' must be a number in (0, 1]") };
383
+ }
384
+ // (6) max_tokens.
385
+ let maxTokens;
386
+ if (params.max_tokens !== undefined && params.max_tokens !== null) {
387
+ const v = params.max_tokens;
388
+ if (typeof v !== 'number' || !Number.isInteger(v) || v < 1)
389
+ return { error: invalidParameters("'max_tokens' must be an integer >= 1") };
390
+ maxTokens = v;
391
+ }
392
+ // (7) stop — "Up to 16 sequences".
393
+ let stop;
394
+ if (params.stop !== undefined && params.stop !== null) {
395
+ const s = asStringArray(params.stop);
396
+ if (s === null)
397
+ return { error: invalidParameters("'stop' must be a string or an array of strings") };
398
+ if (s.length > MAX_STOP_SEQUENCES)
399
+ return { error: invalidParameters(`'stop' supports up to ${MAX_STOP_SEQUENCES} sequences`) };
400
+ stop = s;
401
+ }
402
+ // (8) response_format — TEXT or JSON_OBJECT ONLY. `json_schema` is the OpenAI feature DeepSeek
403
+ // does not have on this endpoint, and it is the single most likely thing an OpenAI-copied
404
+ // twin would happily serve.
405
+ let responseFormat = 'text';
406
+ const rf = params.response_format;
407
+ if (rf !== undefined && rf !== null) {
408
+ if (typeof rf !== 'object' || Array.isArray(rf))
409
+ return { error: invalidParameters("'response_format' must be an object") };
410
+ const type = rf.type;
411
+ if (type !== undefined && type !== null) {
412
+ if (type === 'json_object')
413
+ responseFormat = 'json_object';
414
+ else if (type === 'text')
415
+ responseFormat = 'text';
416
+ else
417
+ return { error: invalidParameters("'response_format.type' must be one of 'text', 'json_object'") };
418
+ }
419
+ }
420
+ // (9) logprobs — a BOOLEAN here (the beta FIM endpoint's `logprobs` is an integer instead, a
421
+ // divergence modelled in validateFim). `top_logprobs` is 0–20 and "requires logprobs: true".
422
+ if (params.logprobs !== undefined && params.logprobs !== null && typeof params.logprobs !== 'boolean') {
423
+ return { error: invalidParameters("'logprobs' must be a boolean") };
424
+ }
425
+ const logprobs = params.logprobs === true;
426
+ let topLogprobs;
427
+ if (params.top_logprobs !== undefined && params.top_logprobs !== null) {
428
+ const v = params.top_logprobs;
429
+ if (typeof v !== 'number' || !Number.isInteger(v) || v < 0 || v > 20)
430
+ return { error: invalidParameters("'top_logprobs' must be an integer between 0 and 20") };
431
+ if (!logprobs)
432
+ return { error: invalidParameters("'top_logprobs' requires 'logprobs' to be true") };
433
+ topLogprobs = v;
434
+ }
435
+ // (10) user_id — DeepSeek's end-user identifier. NOT OpenAI's `user` (which CHAT_PARAMS refuses).
436
+ let userId;
437
+ if (params.user_id !== undefined && params.user_id !== null) {
438
+ const v = params.user_id;
439
+ if (typeof v !== 'string' || !USER_ID_RE.test(v))
440
+ return { error: invalidParameters("'user_id' must match /^[a-zA-Z0-9_-]+$/") };
441
+ if (v.length > USER_ID_MAX)
442
+ return { error: invalidParameters(`'user_id' must be at most ${USER_ID_MAX} characters long`) };
443
+ userId = v;
444
+ }
445
+ // (11) Tools — "max 128 functions"; strict mode is BETA-ONLY and all-or-nothing.
446
+ let tools;
447
+ if (params.tools !== undefined && params.tools !== null) {
448
+ if (!Array.isArray(params.tools))
449
+ return { error: invalidParameters("'tools' must be an array") };
450
+ if (params.tools.length > MAX_TOOLS)
451
+ return { error: invalidParameters(`'tools' supports at most ${MAX_TOOLS} functions`) };
452
+ for (const t of params.tools) {
453
+ const tool = t;
454
+ if (!tool || typeof tool !== 'object' || tool.type !== 'function' || typeof tool.function?.name !== 'string') {
455
+ return { error: invalidParameters("each entry of 'tools' must be { type: 'function', function: { name } }") };
456
+ }
457
+ }
458
+ const strictFlags = params.tools.map((t) => t.function?.strict === true);
459
+ if (strictFlags.some(Boolean)) {
460
+ // "DeepSeek strict tool calls require a beta base URL ending in `/beta`" and "DeepSeek strict
461
+ // mode requires every function tool in the request to set `strict: true`"
462
+ // (@ai-sdk/deepseek src/chat/deepseek-prepare-tools.ts — the SDK refuses both locally, which
463
+ // is first-party evidence of the vendor rule the twin reproduces server-side).
464
+ if (!beta)
465
+ return { error: invalidParameters("strict tool calls require the beta base URL (append '/beta' to your base_url)") };
466
+ if (!strictFlags.every(Boolean))
467
+ return { error: invalidParameters("strict mode requires every function tool to set 'strict': true") };
468
+ }
469
+ tools = params.tools;
470
+ }
471
+ let toolChoice;
472
+ const tcRaw = params.tool_choice;
473
+ if (tcRaw !== undefined && tcRaw !== null) {
474
+ if (typeof tcRaw === 'string') {
475
+ if (!['auto', 'none', 'required'].includes(tcRaw))
476
+ return { error: invalidParameters("'tool_choice' must be one of 'none', 'auto', 'required' or a named function") };
477
+ toolChoice = tcRaw;
478
+ }
479
+ else if (typeof tcRaw === 'object' && !Array.isArray(tcRaw)) {
480
+ const name = tcRaw.function?.name;
481
+ if (typeof name !== 'string' || !name)
482
+ return { error: invalidParameters("'tool_choice.function.name' is required for a named tool choice") };
483
+ toolChoice = { name };
484
+ }
485
+ else {
486
+ return { error: invalidParameters("'tool_choice' must be a string or an object") };
487
+ }
488
+ }
489
+ // (12) THE reasoning_content HAND-BACK RULE. "If the request carries the tools parameter: the
490
+ // reasoning_content of all previous turns should be passed back … the API will return a 400
491
+ // error" (api-docs.deepseek.com/guides/thinking_mode). Scoped to thinking models, exactly as
492
+ // `@ai-sdk/deepseek` scopes its own back-fill (`isDeepSeekV4`), so client and twin can never
493
+ // disagree about which turn owes the field. An EMPTY STRING counts as passed back — that is
494
+ // literally what the SDK sends for a turn with no reasoning.
495
+ if (tools !== undefined && isThinkingModel(model)) {
496
+ for (const m of messages) {
497
+ if (m.role !== 'assistant')
498
+ continue;
499
+ if (typeof m.reasoning_content !== 'string') {
500
+ // The ONE content-level 400 on this vendor, and it is 400 because DeepSeek documents it
501
+ // as one by name: "the API will return a 400 error" when a tools request omits a previous
502
+ // turn's reasoning_content (api-docs.deepseek.com/guides/thinking_mode). Everything else
503
+ // about parsed content is 422 — this is the deliberate exception, not a leak of the rule.
504
+ return { error: { status: 400, body: errBody("Invalid request body format: when 'tools' is set, every previous assistant message must carry back its 'reasoning_content'", 'invalid_request_error') } };
505
+ }
506
+ }
507
+ }
508
+ // (13) Assistant prefix completion — BETA ONLY, and only on the FINAL message.
509
+ const prefixIndexes = messages.map((m, i) => (m.prefix === true ? i : -1)).filter((i) => i >= 0);
510
+ if (prefixIndexes.length > 0) {
511
+ for (const i of prefixIndexes) {
512
+ if (messages[i].role !== 'assistant')
513
+ return { error: invalidParameters("'prefix' is only valid on an assistant message") };
514
+ }
515
+ if (!beta)
516
+ return { error: invalidParameters("chat prefix completion requires the beta base URL (append '/beta' to your base_url)") };
517
+ if (prefixIndexes.length > 1 || prefixIndexes[0] !== messages.length - 1) {
518
+ return { error: invalidParameters("the message carrying 'prefix' must be the final message") };
519
+ }
520
+ }
521
+ // (14) Streaming.
522
+ if (params.stream !== undefined && params.stream !== null && typeof params.stream !== 'boolean') {
523
+ return { error: invalidParameters("'stream' must be a boolean") };
524
+ }
525
+ const stream = params.stream === true;
526
+ let includeUsage = false;
527
+ if (params.stream_options !== undefined && params.stream_options !== null) {
528
+ if (typeof params.stream_options !== 'object' || Array.isArray(params.stream_options))
529
+ return { error: invalidParameters("'stream_options' must be an object") };
530
+ // ACCEPTED WITHOUT `stream`, deliberately. The reference table says only "set when stream: true"
531
+ // — guidance, not a documented refusal — and refusing it would be the over-strictness that is
532
+ // the mirror image of inherited permissiveness. It simply has no effect on a unary request.
533
+ // The vendor's actual behaviour here is unobserved and filed as
534
+ // `deepseek.streaming.stream_options_without_stream`. (§9 round one, NIT 13.)
535
+ includeUsage = stream && params.stream_options.include_usage === true;
536
+ }
537
+ return {
538
+ args: {
539
+ model,
540
+ messages,
541
+ ...(tools !== undefined ? { tools } : {}),
542
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
543
+ ...(stop !== undefined ? { stop } : {}),
544
+ stream,
545
+ includeUsage,
546
+ ...(toolChoice !== undefined ? { toolChoice } : {}),
547
+ responseFormat,
548
+ thinkingEnabled,
549
+ reasoningEffort,
550
+ logprobs,
551
+ ...(topLogprobs !== undefined ? { topLogprobs } : {}),
552
+ ...(userId !== undefined ? { userId } : {}),
553
+ beta,
554
+ },
555
+ };
556
+ }
557
+ /** Build ONE deterministic stub choice. */
558
+ function buildChoice(args) {
559
+ const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
560
+ const forbidTools = args.toolChoice === 'none';
561
+ const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
562
+ // Thinking is on by default for every V4 model, so the stub emits `reasoning_content` unless the
563
+ // caller turned it off — matching what a real V4 turn looks like on the wire.
564
+ const reasoning = args.thinkingEnabled ? stubReasoningText(args.messages, args.model, args.reasoningEffort) : undefined;
565
+ const reasoningTokens = reasoning ? estimateTokens(reasoning) : 0;
566
+ if (hasTools && !forbidTools) {
567
+ const list = args.tools;
568
+ const calls = [];
569
+ if (forcedName) {
570
+ const tc = stubToolCall(list, 1, forcedName);
571
+ if (tc)
572
+ calls.push(tc);
573
+ }
574
+ else {
575
+ for (let t = 0; t < list.length; t++) {
576
+ const tc = stubToolCall([list[t]], t + 1);
577
+ if (tc)
578
+ calls.push(tc);
579
+ }
580
+ }
581
+ if (calls.length) {
582
+ const message = { role: 'assistant', content: null, tool_calls: calls };
583
+ if (reasoning)
584
+ message.reasoning_content = reasoning;
585
+ return {
586
+ choice: { index: 0, message, logprobs: null, finish_reason: 'tool_calls' },
587
+ completionTokens: estimateTokens(JSON.stringify(calls)) + reasoningTokens,
588
+ reasoningTokens,
589
+ };
590
+ }
591
+ }
592
+ // Beta prefix completion: DeepSeek CONTINUES the final assistant message's content rather than
593
+ // starting a new turn, so the stub is appended to what the caller already wrote.
594
+ const prefixMessage = args.beta && args.messages[args.messages.length - 1]?.prefix === true ? args.messages[args.messages.length - 1] : null;
595
+ const prefixText = prefixMessage ? String(prefixMessage.content ?? '') : '';
596
+ let text = args.responseFormat === 'json_object'
597
+ ? stubJsonObject(args.messages, args.model)
598
+ : `${prefixText}${stubAssistantText(args.messages, args.model)}`;
599
+ let finish = 'stop';
600
+ // Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
601
+ let stopAt = -1;
602
+ for (const s of args.stop ?? []) {
603
+ if (!s)
604
+ continue;
605
+ const i = text.indexOf(s);
606
+ if (i >= 0 && (stopAt < 0 || i < stopAt))
607
+ stopAt = i;
608
+ }
609
+ if (stopAt >= 0)
610
+ text = text.slice(0, stopAt);
611
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
612
+ text = text.slice(0, args.maxTokens * 4);
613
+ finish = 'length';
614
+ }
615
+ const message = { role: 'assistant', content: text };
616
+ if (reasoning)
617
+ message.reasoning_content = reasoning;
618
+ return {
619
+ choice: { index: 0, message, logprobs: args.logprobs ? stubLogprobs(text, reasoning, args.topLogprobs) : null, finish_reason: finish },
620
+ completionTokens: estimateTokens(text) + reasoningTokens,
621
+ reasoningTokens,
622
+ };
623
+ }
624
+ /**
625
+ * DeepSeek's `logprobs` block, which — unlike OpenAI's — has TWO channels: `content` and
626
+ * `reasoning_content` (`deepseekChatLogprobsSchema`, `@ai-sdk/deepseek`
627
+ * src/chat/deepseek-chat-api-types.ts). The VALUES are a labeled deterministic derivation, not real
628
+ * model probabilities — the twin runs no model.
629
+ */
630
+ function stubLogprobs(text, reasoning, topLogprobs) {
631
+ const entry = (token) => {
632
+ const h = fnv1a(token);
633
+ const logprob = -Math.round(((h % 5000) / 1000) * 1e6) / 1e6;
634
+ const top = [];
635
+ for (let i = 0; i < (topLogprobs ?? 0); i++) {
636
+ top.push({ token: `${token}#${i}`, logprob: Math.round((logprob - i * 0.5) * 1e6) / 1e6, bytes: null });
637
+ }
638
+ return { token, logprob, bytes: [...new TextEncoder().encode(token)], top_logprobs: top };
639
+ };
640
+ const split = (s) => s.split(/(?<=\s)/).slice(0, 8).filter(Boolean);
641
+ const content = split(text).map(entry);
642
+ return {
643
+ ...(content.length ? { content } : { content: null }),
644
+ ...(reasoning ? { reasoning_content: split(reasoning).map(entry) } : {}),
645
+ };
646
+ }
647
+ /** A deterministic id suffix from the request (so ids are stable + assertable). */
648
+ function stableSuffix(args) {
649
+ return fnv1a(JSON.stringify(args.messages) + args.model + args.reasoningEffort).toString(36);
650
+ }
651
+ export function buildChatCompletion(args, occurredAt, scenarioEngine, root) {
652
+ const promptTokens = countPromptTokens(args.messages);
653
+ const hitTokens = cacheHitTokens(args.messages, root);
654
+ let scripted = null;
655
+ let missTeach = '';
656
+ if (scenarioEngine) {
657
+ const decision = scenarioEngine.next({ model: args.model, messages: args.messages, tools: args.tools, thinking: args.thinkingEnabled ? 'enabled' : 'disabled' });
658
+ if (decision.kind === 'handler') {
659
+ const respond = decision.respond;
660
+ if (respond.error)
661
+ return scriptedError(respond.error);
662
+ scripted = realizeDeepSeekRespond(respond);
663
+ }
664
+ else {
665
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/deepseek.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
666
+ }
667
+ }
668
+ let choice;
669
+ let completionTokens;
670
+ let reasoningTokens = 0;
671
+ if (scripted) {
672
+ const message = scripted.toolCalls.length
673
+ ? { role: 'assistant', content: scripted.text, tool_calls: scripted.toolCalls }
674
+ : { role: 'assistant', content: scripted.text ?? '' };
675
+ if (scripted.reasoning !== null) {
676
+ message.reasoning_content = scripted.reasoning;
677
+ reasoningTokens = estimateTokens(scripted.reasoning);
678
+ }
679
+ choice = { index: 0, message, logprobs: null, finish_reason: scripted.finishReason };
680
+ completionTokens = estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? '')) + reasoningTokens;
681
+ }
682
+ else {
683
+ const built = buildChoice(args);
684
+ choice = built.choice;
685
+ completionTokens = built.completionTokens;
686
+ reasoningTokens = built.reasoningTokens;
687
+ if (missTeach && typeof choice.message.content === 'string')
688
+ choice.message.content += missTeach;
689
+ }
690
+ const usage = buildUsage(promptTokens, completionTokens, hitTokens, reasoningTokens);
691
+ return {
692
+ id: `chatcmpl-twin-${stableSuffix(args)}`,
693
+ object: 'chat.completion',
694
+ created: nowEpoch(occurredAt),
695
+ model: args.model,
696
+ choices: [choice],
697
+ usage,
698
+ system_fingerprint: stubFingerprint(`${args.model}|${args.reasoningEffort}`),
699
+ };
700
+ }
701
+ /** Map a scripted scenario failure onto DeepSeek's real status + envelope. */
702
+ function scriptedError(err) {
703
+ if (err.type === 'rate_limit_exceeded') {
704
+ const base = rateLimitError();
705
+ return err.message ? { ...base, body: errBody(err.message, 'rate_limit_exceeded') } : base;
706
+ }
707
+ if (err.type === 'insufficient_balance') {
708
+ const base = insufficientBalance();
709
+ return err.message ? { ...base, body: errBody(err.message) } : base;
710
+ }
711
+ if (err.type === 'server_overloaded') {
712
+ return { status: 503, body: errBody(err.message ?? 'The server is overloaded due to high traffic', 'service_unavailable') };
713
+ }
714
+ return { status: 500, body: errBody(err.message ?? 'Our server encounters an issue', 'server_error') };
715
+ }
716
+ const isEnvelope = (v) => typeof v.status === 'number' && 'body' in v;
717
+ /** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
718
+ function chunkText(text) {
719
+ if (!text)
720
+ return [];
721
+ const out = [];
722
+ for (let i = 0; i < text.length; i += 20)
723
+ out.push(text.slice(i, i + 20));
724
+ return out;
725
+ }
726
+ /**
727
+ * Emit the vendor-faithful DeepSeek streaming sequence into the injected sink (NO sockets, NO
728
+ * setTimeout). The order: a first chunk with `delta:{role:'assistant', content:''}`, then
729
+ * `reasoning_content` deltas, then `content` deltas (or `tool_calls` deltas), then a chunk carrying
730
+ * `finish_reason`, and — when `stream_options.include_usage` was set — a FINAL chunk with an EMPTY
731
+ * `choices` array whose `usage` holds the completion's usage. `@ai-sdk/deepseek`'s `doStream`
732
+ * always sends `stream_options: { include_usage: true }`, so that tail chunk is what makes its
733
+ * `finish` part carry real token counts.
734
+ *
735
+ * REASONING BEFORE TEXT is not cosmetic: the SDK's transform closes its `reasoning-0` part the
736
+ * moment the first `content` delta arrives, so emitting them interleaved would produce a different
737
+ * part sequence than the vendor's.
738
+ */
739
+ export function streamChat(args, sink, occurredAt, scenarioEngine, root) {
740
+ const built = buildChatCompletion(args, occurredAt, scenarioEngine, root);
741
+ if (isEnvelope(built))
742
+ return built;
743
+ const full = built;
744
+ const base = { id: full.id, object: 'chat.completion.chunk', created: full.created, model: full.model, system_fingerprint: full.system_fingerprint };
745
+ const choice = full.choices[0];
746
+ sink({ data: { ...base, choices: [{ index: 0, delta: { role: 'assistant', content: '' }, logprobs: null, finish_reason: null }] } });
747
+ for (const piece of chunkText(choice.message.reasoning_content ?? '')) {
748
+ sink({ data: { ...base, choices: [{ index: 0, delta: { reasoning_content: piece }, logprobs: null, finish_reason: null }] } });
749
+ }
750
+ if (choice.message.tool_calls && choice.message.tool_calls.length) {
751
+ choice.message.tool_calls.forEach((tc, tIdx) => {
752
+ sink({ data: { ...base, choices: [{ index: 0, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, logprobs: null, finish_reason: null }] } });
753
+ sink({ data: { ...base, choices: [{ index: 0, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, logprobs: null, finish_reason: null }] } });
754
+ });
755
+ }
756
+ else {
757
+ for (const piece of chunkText(choice.message.content ?? '')) {
758
+ sink({ data: { ...base, choices: [{ index: 0, delta: { content: piece }, logprobs: null, finish_reason: null }] } });
759
+ }
760
+ }
761
+ sink({ data: { ...base, choices: [{ index: 0, delta: {}, logprobs: choice.logprobs, finish_reason: choice.finish_reason }] } });
762
+ // The usage tail is OPT-IN, exactly as the vendor's `stream_options.include_usage` documents.
763
+ if (args.includeUsage)
764
+ sink({ data: { ...base, choices: [], usage: full.usage } });
765
+ sink({ done: true });
766
+ return full;
767
+ }
768
+ function validateFim(params) {
769
+ for (const key of Object.keys(params)) {
770
+ if (!FIM_PARAMS.has(key))
771
+ return { error: invalidParameters(`'${key}' is not a parameter of the DeepSeek FIM completions API`) };
772
+ }
773
+ if (typeof params.model !== 'string' || !params.model)
774
+ return { error: invalidParameters("'model' is a required string") };
775
+ // "Only value: `deepseek-v4-pro`" — a closed set of one.
776
+ if (!FIM_MODELS.has(params.model)) {
777
+ return { error: invalidParameters(`'model' must be one of ${[...FIM_MODELS].map((m) => `'${m}'`).join(', ')} for FIM completion`) };
778
+ }
779
+ if (typeof params.prompt !== 'string' || !params.prompt)
780
+ return { error: invalidParameters("'prompt' is a required string") };
781
+ if (params.suffix !== undefined && params.suffix !== null && typeof params.suffix !== 'string')
782
+ return { error: invalidParameters("'suffix' must be a string") };
783
+ // THE DIVERGENCE FROM CHAT: here `logprobs` is an INTEGER ("Return log probabilities for top N
784
+ // tokens (max: 20)"), where the chat endpoint's `logprobs` is a boolean. Same key name, different
785
+ // type, on the same vendor — modelled rather than smoothed over.
786
+ let logprobs;
787
+ if (params.logprobs !== undefined && params.logprobs !== null) {
788
+ const v = params.logprobs;
789
+ if (typeof v !== 'number' || !Number.isInteger(v) || v < 0 || v > 20)
790
+ return { error: invalidParameters("'logprobs' must be an integer between 0 and 20") };
791
+ logprobs = v;
792
+ }
793
+ let maxTokens;
794
+ if (params.max_tokens !== undefined && params.max_tokens !== null) {
795
+ const v = params.max_tokens;
796
+ if (typeof v !== 'number' || !Number.isInteger(v) || v < 1)
797
+ return { error: invalidParameters("'max_tokens' must be an integer >= 1") };
798
+ maxTokens = v;
799
+ }
800
+ let stop;
801
+ if (params.stop !== undefined && params.stop !== null) {
802
+ const s = asStringArray(params.stop);
803
+ if (s === null)
804
+ return { error: invalidParameters("'stop' must be a string or an array of strings") };
805
+ if (s.length > MAX_STOP_SEQUENCES)
806
+ return { error: invalidParameters(`'stop' supports up to ${MAX_STOP_SEQUENCES} sequences`) };
807
+ stop = s;
808
+ }
809
+ if (params.echo !== undefined && params.echo !== null && typeof params.echo !== 'boolean')
810
+ return { error: invalidParameters("'echo' must be a boolean") };
811
+ // UNMODELED, NOT FAKED. DeepSeek's FIM endpoint really does declare `stream` / `stream_options`,
812
+ // and the twin does not model its SSE shape yet (`deepseek.completions.streaming`, todo). Serving
813
+ // a unary text_completion to a caller who asked for a stream is the fake success the bar forbids
814
+ // — a client reading `text/event-stream` would hang or mis-parse — so the twin refuses BY NAME
815
+ // and says it is the twin's gap, not the vendor's.
816
+ if (params.stream === true) {
817
+ // NO `type` — this is a 422 and the rule at `invalidParameters` is universal, not a property of
818
+ // one helper. §9 round two, BLOCKER 2: this envelope was written before that fix landed and kept
819
+ // emitting `invalid_request_error`, which the SDK's discriminator table resolves to 400.
820
+ return { error: { status: 422, body: errBody('the DeepSeek twin does not model streaming on /beta/completions yet (deepseek.completions.streaming); omit `stream` or use /chat/completions') } };
821
+ }
822
+ if (params.stream !== undefined && params.stream !== null && typeof params.stream !== 'boolean')
823
+ return { error: invalidParameters("'stream' must be a boolean") };
824
+ return {
825
+ args: {
826
+ model: params.model,
827
+ prompt: params.prompt,
828
+ ...(typeof params.suffix === 'string' ? { suffix: params.suffix } : {}),
829
+ echo: params.echo === true,
830
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
831
+ ...(stop !== undefined ? { stop } : {}),
832
+ ...(logprobs !== undefined ? { logprobs } : {}),
833
+ },
834
+ };
835
+ }
836
+ function buildFimCompletion(args, hitTokens, occurredAt) {
837
+ let text = stubFimText(args.model, args.prompt, args.suffix);
838
+ let finish = 'stop';
839
+ let stopAt = -1;
840
+ for (const s of args.stop ?? []) {
841
+ if (!s)
842
+ continue;
843
+ const i = text.indexOf(s);
844
+ if (i >= 0 && (stopAt < 0 || i < stopAt))
845
+ stopAt = i;
846
+ }
847
+ if (stopAt >= 0)
848
+ text = text.slice(0, stopAt);
849
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
850
+ text = text.slice(0, args.maxTokens * 4);
851
+ finish = 'length';
852
+ }
853
+ // "`echo`: Return prompt with completion" — so the prompt is PREPENDED to the returned text.
854
+ const out = args.echo ? `${args.prompt}${text}` : text;
855
+ const promptTokens = estimateTokens(args.prompt) + estimateTokens(args.suffix ?? '');
856
+ const completionTokens = estimateTokens(text);
857
+ return {
858
+ id: `cmpl-twin-${fnv1a(args.prompt + (args.suffix ?? '')).toString(36)}`,
859
+ object: 'text_completion',
860
+ created: nowEpoch(occurredAt),
861
+ model: args.model,
862
+ choices: [{
863
+ index: 0,
864
+ text: out,
865
+ finish_reason: finish,
866
+ logprobs: args.logprobs === undefined ? null : { tokens: [], token_logprobs: [], text_offset: [], top_logprobs: [] },
867
+ }],
868
+ usage: buildUsage(promptTokens, completionTokens, hitTokens),
869
+ system_fingerprint: stubFingerprint(`${args.model}|fim`),
870
+ };
871
+ }
872
+ // ── Files (stateful) ────────────────────────────────────────────────────────────────────
873
+ /** "`purpose`: Must be `user_data`" — a closed set of one (api-docs.deepseek.com/guides/files_api). */
874
+ const FILE_PURPOSES = new Set(['user_data']);
875
+ /** "Supported formats: JPEG, PNG, GIF, and WebP" (same page; `@ai-sdk/deepseek`'s
876
+ * `supportedMediaTypes` / `supportedFilenameExtensions` enforce the identical sets client-side). */
877
+ const FILE_MEDIA_TYPES = new Set(['image/gif', 'image/jpeg', 'image/jpg', 'image/png', 'image/webp']);
878
+ const FILE_EXTENSIONS = new Set(['gif', 'jpeg', 'jpg', 'png', 'webp']);
879
+ /** "Max upload size: 64 MiB"; "Max filename: 512 characters". */
880
+ const MAX_FILE_BYTES = 64 * 1024 * 1024;
881
+ const MAX_FILENAME_LENGTH = 512;
882
+ /** "Lifetime range of 3600-2592000 seconds (1 hour to 30 days)". */
883
+ const MIN_EXPIRES_SECONDS = 3600;
884
+ const MAX_EXPIRES_SECONDS = 2_592_000;
885
+ /** "`limit`: 1-1000 files per request". */
886
+ const MAX_FILE_LIST_LIMIT = 1000;
887
+ function fileExtension(filename) {
888
+ const i = filename.lastIndexOf('.');
889
+ return i === -1 ? '' : filename.slice(i + 1).toLowerCase();
890
+ }
891
+ async function createFile(params, req) {
892
+ const purpose = typeof params.purpose === 'string' ? params.purpose : '';
893
+ if (!purpose)
894
+ return invalidParameters("'purpose' is a required property");
895
+ if (!FILE_PURPOSES.has(purpose))
896
+ return invalidParameters(`'purpose' must be 'user_data' (got '${purpose}')`);
897
+ const filename = typeof params.filename === 'string' && params.filename ? params.filename : '';
898
+ if (!filename)
899
+ return invalidParameters("'file' is a required property");
900
+ if ([...filename].length > MAX_FILENAME_LENGTH)
901
+ return invalidParameters(`'filename' must be at most ${MAX_FILENAME_LENGTH} characters`);
902
+ const mediaType = typeof params.media_type === 'string' ? params.media_type.split(';', 1)[0].trim().toLowerCase() : '';
903
+ // DeepSeek's Files API is IMAGE-ONLY. A caller uploading a .jsonl batch file (the OpenAI habit)
904
+ // must be refused, not stored: this vendor has no batch surface here at all.
905
+ if (!FILE_MEDIA_TYPES.has(mediaType) && !FILE_EXTENSIONS.has(fileExtension(filename))) {
906
+ return invalidParameters('DeepSeek file uploads support JPEG, PNG, GIF, and WebP images');
907
+ }
908
+ const content = typeof params.content === 'string' ? params.content : '';
909
+ const bytes = typeof params.bytes === 'number' ? params.bytes : content.length;
910
+ if (bytes > MAX_FILE_BYTES)
911
+ return invalidParameters(`DeepSeek file uploads must not exceed ${MAX_FILE_BYTES} bytes`);
912
+ // `expires_after[anchor]` / `expires_after[seconds]` arrive as the flat form-field names the
913
+ // vendor documents and `@ai-sdk/deepseek` appends verbatim (src/files/deepseek-files.ts).
914
+ const anchor = params['expires_after[anchor]'];
915
+ const secondsRaw = params['expires_after[seconds]'];
916
+ let expiresAt;
917
+ if (anchor !== undefined || secondsRaw !== undefined) {
918
+ if (anchor !== undefined && anchor !== 'created_at')
919
+ return invalidParameters("'expires_after[anchor]' must be 'created_at'");
920
+ const seconds = Number(secondsRaw);
921
+ if (!Number.isInteger(seconds) || seconds < MIN_EXPIRES_SECONDS || seconds > MAX_EXPIRES_SECONDS) {
922
+ return invalidParameters(`'expires_after[seconds]' must be an integer between ${MIN_EXPIRES_SECONDS} and ${MAX_EXPIRES_SECONDS}`);
923
+ }
924
+ expiresAt = nowEpoch(req.occurredAt) + seconds;
925
+ }
926
+ const id = nextFileId(req.root);
927
+ await applyTwinWrite(SERVICE, {
928
+ operation: 'file.create',
929
+ subjectType: 'file',
930
+ subjectId: id,
931
+ fields: {
932
+ object: 'file', bytes, created_at: nowEpoch(req.occurredAt), filename, purpose,
933
+ ...(expiresAt !== undefined ? { expires_at: expiresAt } : {}),
934
+ _content: content,
935
+ },
936
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
937
+ actor: { kind: 'agent' },
938
+ }, req.root);
939
+ return { status: 200, body: fileView(getRow('file', id, req.root) ?? {}) };
940
+ }
941
+ function fileView(r) {
942
+ return {
943
+ id: r.id, object: 'file', bytes: r.bytes, created_at: r.created_at, filename: r.filename, purpose: r.purpose,
944
+ ...(r.expires_at !== undefined ? { expires_at: r.expires_at } : {}),
945
+ };
946
+ }
947
+ function listFiles(query, root) {
948
+ const purpose = query.get('purpose');
949
+ if (purpose !== null && !FILE_PURPOSES.has(purpose))
950
+ return invalidParameters("'purpose' must be 'user_data'");
951
+ const order = query.get('order');
952
+ if (order !== null && order !== 'asc' && order !== 'desc')
953
+ return invalidParameters("'order' must be 'asc' or 'desc'");
954
+ const limitRaw = query.get('limit');
955
+ let limit = MAX_FILE_LIST_LIMIT;
956
+ if (limitRaw !== null) {
957
+ const v = Number(limitRaw);
958
+ if (!Number.isInteger(v) || v < 1 || v > MAX_FILE_LIST_LIMIT)
959
+ return invalidParameters(`'limit' must be an integer between 1 and ${MAX_FILE_LIST_LIMIT}`);
960
+ limit = v;
961
+ }
962
+ let data = rows('file', root).filter((r) => !r._deleted).map(fileView);
963
+ data.sort((a, b) => Number(a.created_at) - Number(b.created_at) || String(a.id).localeCompare(String(b.id)));
964
+ if (order === 'desc')
965
+ data.reverse();
966
+ const after = query.get('after');
967
+ if (after !== null) {
968
+ const idx = data.findIndex((f) => f.id === after);
969
+ if (idx === -1)
970
+ return invalidParameters(`'after' cursor "${after}" is not a known file id`);
971
+ data = data.slice(idx + 1);
972
+ }
973
+ const has_more = data.length > limit;
974
+ const page = data.slice(0, limit);
975
+ // DeepSeek's own list example carries `first_id` / `last_id` alongside `has_more`
976
+ // (api-docs.deepseek.com/guides/files_api) — the cursor bookends a caller pages with. They are
977
+ // omitted on an empty page rather than emitted as null, because the vendor's example shows file
978
+ // ids there and a null would be a shape this twin cannot source.
979
+ return {
980
+ status: 200,
981
+ body: {
982
+ object: 'list',
983
+ data: page,
984
+ ...(page.length > 0 ? { first_id: page[0].id, last_id: page[page.length - 1].id } : {}),
985
+ has_more,
986
+ },
987
+ };
988
+ }
989
+ // ── public entry: cross-cutting protocol (auth / rate-limit) then route ──────────────────
990
+ export async function handleDeepSeekTwinRequest(req) {
991
+ const method = req.method.toUpperCase();
992
+ if (req.headers !== undefined || req.apiKey !== undefined) {
993
+ const authErr = checkAuth(req);
994
+ if (authErr)
995
+ return authErr;
996
+ }
997
+ // 429 is non-deterministic in production, so the twin exposes a DETERMINISTIC opt-in trigger.
998
+ if (triggered(req, 'x-twin-force-rate-limit'))
999
+ return rateLimitError();
1000
+ return routeDeepSeek(req, method);
1001
+ }
1002
+ // ── router ──────────────────────────────────────────────────────────────────────────────
1003
+ async function routeDeepSeek(req, method) {
1004
+ const [rawPath, rawQuery] = req.path.split('?');
1005
+ const path = (rawPath ?? '/').replace(/\/+$/, '') || '/';
1006
+ const query = new URLSearchParams(rawQuery ?? '');
1007
+ const dec = (s) => decodeURIComponent(s);
1008
+ // D3: a read-only twin rejects any mutation with a vendor-shaped error.
1009
+ if (req.readOnly && method !== 'GET') {
1010
+ // Same rule again (§9 round two, NIT 6): nothing first-party names a `type` for a 405 on this
1011
+ // vendor, and `invalid_request_error` would resolve to 400 in the SDK's own table.
1012
+ return { status: 405, body: errBody('twin is read-only; omit readOnly to accept writes') };
1013
+ }
1014
+ // DeepSeek's Anthropic-compatible surface is REAL vendor surface this twin does not model yet.
1015
+ // It fails like an unmodeled operation rather than silently answering as if it were the
1016
+ // OpenAI-shaped endpoint (`deepseek.anthropic.messages`, todo).
1017
+ if (path === DEEPSEEK_ANTHROPIC_PREFIX || path.startsWith(`${DEEPSEEK_ANTHROPIC_PREFIX}/`)) {
1018
+ return notFound(`The DeepSeek twin does not model the Anthropic-compatible surface (${method} ${path}) yet.`);
1019
+ }
1020
+ // THE BETA BASE URL. `https://api.deepseek.com/beta` unlocks assistant-prefix completion and
1021
+ // strict tool calls, and is where FIM completion lives — nothing else changes.
1022
+ const beta = path === DEEPSEEK_BETA_PREFIX || path.startsWith(`${DEEPSEEK_BETA_PREFIX}/`);
1023
+ const rest = beta ? (path.slice(DEEPSEEK_BETA_PREFIX.length) || '/') : path;
1024
+ const seg = rest.replace(/^\/+/, '').split('/').filter(Boolean);
1025
+ // A GET carries no body by definition, so it never goes through the body-format gate. A DELETE
1026
+ // likewise: `DELETE /files/{id}` addresses its subject by path.
1027
+ const parsed = method === 'GET' || method === 'DELETE' ? { params: {} } : parseJson(req.body);
1028
+ if ('error' in parsed)
1029
+ return parsed.error;
1030
+ const params = parsed.params;
1031
+ // ---- models (published catalog + anything a pull observed) ----
1032
+ if (seg[0] === 'models' && seg.length === 1 && method === 'GET') {
1033
+ return { status: 200, body: { object: 'list', data: servedModels(req.root) } };
1034
+ }
1035
+ // ---- user balance ----
1036
+ if (seg[0] === 'user' && seg[1] === 'balance' && seg.length === 2 && method === 'GET') {
1037
+ return { status: 200, body: servedBalance(req.root) };
1038
+ }
1039
+ // ---- chat completions (the generative stub; envelope is faithful) ----
1040
+ if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
1041
+ const validated = validateChat(params, beta);
1042
+ if ('error' in validated)
1043
+ return validated.error;
1044
+ const args = validated.args;
1045
+ // 402 is a real DeepSeek status with no OpenAI counterpart, and it is STATE-DRIVEN here: an
1046
+ // account whose balance says `is_available:false` cannot complete. That makes the 402 path a
1047
+ // fact about the twin's projection rather than a header trick.
1048
+ if (!servedBalance(req.root).is_available)
1049
+ return insufficientBalance();
1050
+ // §9 ROUND TWO, SHOULD-FIX 1: this used to fall through to the UNARY builder whenever a sink was
1051
+ // absent, so a `stream: true` request could be answered with a JSON blob a client reading
1052
+ // `text/event-stream` cannot parse — the same fake success `deepseek.completions.refuses_unmodeled_streaming`
1053
+ // exists to forbid on the FIM endpoint, live on the primary one. A streaming request with no
1054
+ // sink is now a loud refusal rather than a silent downgrade.
1055
+ if (args.stream && !req.sseSink) {
1056
+ return { status: 422, body: errBody("'stream' was requested but this caller supplied no SSE sink; the twin refuses to answer a streaming request with a unary body") };
1057
+ }
1058
+ const result = args.stream && req.sseSink
1059
+ ? streamChat(args, req.sseSink, req.occurredAt, req.scenarioEngine, req.root)
1060
+ : buildChatCompletion(args, req.occurredAt, req.scenarioEngine, req.root);
1061
+ if (isEnvelope(result))
1062
+ return result;
1063
+ // Record the context-cache prefix units this completion creates (deepseek-cache.ts) so a
1064
+ // follow-up turn genuinely reports `prompt_cache_hit_tokens`.
1065
+ await recordCachePrefixes(args.messages, { role: 'assistant', content: result.choices[0].message.content ?? '', ...(result.choices[0].message.tool_calls ? { tool_calls: result.choices[0].message.tool_calls } : {}) }, req.occurredAt, req.root);
1066
+ return { status: 200, body: result };
1067
+ }
1068
+ // ---- FIM completions (BETA ONLY) ----
1069
+ if (seg[0] === 'completions' && seg.length === 1 && method === 'POST') {
1070
+ // FIM lives at `https://api.deepseek.com/beta/completions`. On the non-beta base URL there is
1071
+ // no such endpoint, so it must fail rather than quietly work.
1072
+ if (!beta)
1073
+ return notFound("FIM completion requires the beta base URL (append '/beta' to your base_url).");
1074
+ const validated = validateFim(params);
1075
+ if ('error' in validated)
1076
+ return validated.error;
1077
+ if (!servedBalance(req.root).is_available)
1078
+ return insufficientBalance();
1079
+ return { status: 200, body: buildFimCompletion(validated.args, 0, req.occurredAt) };
1080
+ }
1081
+ // ---- files (stateful) ----
1082
+ if (seg[0] === 'files' && seg.length === 1 && method === 'POST')
1083
+ return createFile(params, req);
1084
+ if (seg[0] === 'files' && seg.length === 1 && method === 'GET')
1085
+ return listFiles(query, req.root);
1086
+ if (seg[0] === 'files' && seg.length === 2 && method === 'GET') {
1087
+ const f = getRow('file', dec(seg[1]), req.root);
1088
+ return f && !f._deleted ? { status: 200, body: fileView(f) } : notFound(`No such File object: ${dec(seg[1])}`);
1089
+ }
1090
+ if (seg[0] === 'files' && seg.length === 2 && method === 'DELETE') {
1091
+ const fid = dec(seg[1]);
1092
+ const f = getRow('file', fid, req.root);
1093
+ if (!f || f._deleted)
1094
+ return notFound(`No such File object: ${fid}`);
1095
+ await applyTwinWrite(SERVICE, {
1096
+ operation: 'file.delete', subjectType: 'file', subjectId: fid, fields: { _deleted: true, object: 'file' },
1097
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1098
+ }, req.root);
1099
+ return { status: 200, body: { id: fid, object: 'file', deleted: true } };
1100
+ }
1101
+ // Unmodeled operation → fail like the vendor (never a fake success).
1102
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1103
+ }