@volter/twin-deepseek 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +198 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/deepseek-budget.d.ts +51 -0
  6. package/dist/src/deepseek-budget.js +152 -0
  7. package/dist/src/deepseek-cache.d.ts +56 -0
  8. package/dist/src/deepseek-cache.js +151 -0
  9. package/dist/src/deepseek-capabilities.d.ts +4 -0
  10. package/dist/src/deepseek-capabilities.js +1520 -0
  11. package/dist/src/deepseek-conformance.d.ts +14 -0
  12. package/dist/src/deepseek-conformance.js +473 -0
  13. package/dist/src/deepseek-connector.d.ts +168 -0
  14. package/dist/src/deepseek-connector.js +386 -0
  15. package/dist/src/deepseek-models.d.ts +30 -0
  16. package/dist/src/deepseek-models.js +38 -0
  17. package/dist/src/deepseek-scenario.d.ts +55 -0
  18. package/dist/src/deepseek-scenario.js +170 -0
  19. package/dist/src/deepseek-server.d.ts +16 -0
  20. package/dist/src/deepseek-server.js +191 -0
  21. package/dist/src/deepseek-stub.d.ts +75 -0
  22. package/dist/src/deepseek-stub.js +191 -0
  23. package/dist/src/deepseek-twin.d.ts +77 -0
  24. package/dist/src/deepseek-twin.js +1103 -0
  25. package/dist/src/deepseek-types.d.ts +172 -0
  26. package/dist/src/deepseek-types.js +26 -0
  27. package/dist/src/index.d.ts +15 -0
  28. package/dist/src/index.js +93 -0
  29. package/package.json +68 -0
  30. package/src/cli.ts +27 -0
  31. package/src/deepseek-budget.ts +178 -0
  32. package/src/deepseek-cache.ts +159 -0
  33. package/src/deepseek-capabilities.ts +1443 -0
  34. package/src/deepseek-conformance.ts +512 -0
  35. package/src/deepseek-connector.ts +440 -0
  36. package/src/deepseek-models.ts +65 -0
  37. package/src/deepseek-scenario.ts +188 -0
  38. package/src/deepseek-server.ts +201 -0
  39. package/src/deepseek-stub.ts +200 -0
  40. package/src/deepseek-twin.ts +1163 -0
  41. package/src/deepseek-types.ts +201 -0
  42. package/src/index.ts +133 -0
@@ -0,0 +1,1443 @@
1
+ // DeepSeek capability manifest — the EXPECTED REAL-PRODUCT SURFACE (the target), authored top-down
2
+ // from what the DeepSeek Platform API actually does — NOT from what this twin has built. The
3
+ // denominator was enumerated from TWO first-party sources, both read on 2026-08-31:
4
+ // • api-docs.deepseek.com, walked as a docs NAV: quick_start/{error_codes,rate_limit,pricing},
5
+ // api/{create-chat-completion,create-completion,list-models,get-user-balance}, and
6
+ // guides/{kv_cache,thinking_mode,files_api,anthropic_api,chat_prefix_completion};
7
+ // • `@ai-sdk/deepseek@3.0.37` — the only first-party npm client of this surface — read as SOURCE:
8
+ // the zod schemas it encodes/decodes with (deepseek-chat-api-types.ts, files/deepseek-files-api.ts),
9
+ // the request it builds (deepseek-chat-language-model.ts), the refusals it raises locally
10
+ // (deepseek-prepare-tools.ts, convert-to-deepseek-chat-messages.ts), and the provider doc page
11
+ // shipped inside its tarball (docs/30-deepseek.mdx).
12
+ // DeepSeek publishes no OpenAPI document, so where those two conflict the SDK's generated schema
13
+ // wins over a rendered docs example (ADDING_A_TWIN.md §6, the precedence order), and where only
14
+ // prose exists the manifest says so rather than asserting a closed set.
15
+ //
16
+ // ═══ THE DENOMINATOR IS DELIBERATELY WEIGHTED TOWARD REFUSALS ═══
17
+ // DeepSeek is OPENAI-COMPATIBLE. Its response shapes are the easy half; what makes a DeepSeek twin
18
+ // a DeepSeek twin rather than a relabelled OpenAI twin is what it REFUSES — 422 where OpenAI 400s,
19
+ // 402 for a drained balance, no `/v1` segment, no `json_schema`/`n`/`seed`/`logit_bias`/`user`,
20
+ // beta-only prefix completion and strict tools, and a documented 400 when `tools` is sent without
21
+ // prior-turn `reasoning_content`. Those are first-class capabilities here, each with a failable
22
+ // negative verify, because ADDING_A_TWIN.md §0 records that inherited PERMISSIVENESS is how the
23
+ // nearest sibling pack shipped two false-greens.
24
+ //
25
+ // There are NO carve-outs: every entry here is done or todo. The twin returns DETERMINISTIC labeled
26
+ // stubs where the vendor runs a model — those stubs ARE its answer — while the protocol envelope is
27
+ // faithful, and it does not simulate properties of the vendor's own serving fleet (concurrency
28
+ // ceilings). `deepseek.usage.real_tokenizer` and `deepseek.cache.expiry` are honest todos: a BPE
29
+ // tokenizer is an offline data file and the kernel's world clock is deterministic, so both are work
30
+ // not yet done rather than anything unreachable.
31
+ //
32
+ // TIERING (§6 rule 3): `core` = "first-week-of-every-integration". The overwhelming majority of
33
+ // week-one DeepSeek integrations are chat completions and nothing else, so `core` is the
34
+ // chat/streaming/tools/reasoning/errors/auth spine plus the conformance and connector-pull
35
+ // backbone. The Files API (images only, and only usable by one vision model), the beta FIM
36
+ // endpoint, the beta prefix/strict features and the Anthropic-compatible surface are specialist and
37
+ // are tiered `common` or `niche`.
38
+ //
39
+ // (DeepSeek is an API-first vendor — platform.deepseek.com is a keys/billing/usage console, not
40
+ // where the work happens — so this pack ships NO mirror and has NO UI capabilities.)
41
+ import { mkdtempSync, rmSync } from 'node:fs';
42
+ import { tmpdir } from 'node:os';
43
+ import { join } from 'node:path';
44
+ import { checkCapabilities, type CapabilityReport, type CapabilitySpec, verifyBoundary, isInfrastructureError, harnessError } from '@volter/world-tooling';
45
+ import { applyTwinWrite, pendingActions, projectResources } from '@volter/world-core';
46
+ import { handleDeepSeekTwinRequest, type DeepSeekResponseEnvelope } from './deepseek-twin.ts';
47
+ import {
48
+ fullSyncDeepSeek,
49
+ deepseekRequestForAction,
50
+ externalIdFor,
51
+ liveDeepSeekExecute,
52
+ pullDeepSeekState,
53
+ pushPendingDeepSeekActions,
54
+ syncDeepSeekFromReal,
55
+ unpushableReason,
56
+ type DeepSeekExecute,
57
+ } from './deepseek-connector.ts';
58
+ import { createDeepSeekScenarioEngine } from './deepseek-scenario.ts';
59
+ import type { SseEvent } from './deepseek-types.ts';
60
+
61
+ // ── API verify: drive REAL requests against a fresh temp root, then assert status/shape ──
62
+ type Step = { m: string; p: string; b?: unknown };
63
+ type Body = Record<string, any>;
64
+
65
+ /** Run a sequence of real DeepSeek requests against an isolated root; return all responses. */
66
+ async function withRoot(steps: (h: (s: Step) => Promise<DeepSeekResponseEnvelope>, root: string) => Promise<boolean>): Promise<boolean> {
67
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
68
+ const h = (s: Step) => handleDeepSeekTwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : JSON.stringify(s.b), root, occurredAt: OCCURRED_AT });
69
+ try {
70
+ // `root` is handed to the steps too, so a verify can inspect the LOG (projectResources /
71
+ // pendingActions) and not merely the responses — the difference between proving "the reply did
72
+ // not change" and proving "nothing was written".
73
+ return await verifyBoundary('deepseek.withRoot', () => steps(h, root));
74
+ } finally {
75
+ rmSync(root, { recursive: true, force: true });
76
+ }
77
+ }
78
+
79
+ /** Like withRoot, but the request helper passes request HEADERS through (for auth and the
80
+ * deterministic 429 trigger, which the trusted no-headers helper never fires). */
81
+ type StepH = Step & { headers?: Record<string, string> };
82
+ async function withRootH(steps: (h: (s: StepH) => Promise<DeepSeekResponseEnvelope>) => Promise<boolean>): Promise<boolean> {
83
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
84
+ const h = (s: StepH) => handleDeepSeekTwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : JSON.stringify(s.b), root, occurredAt: OCCURRED_AT, ...(s.headers ? { headers: s.headers } : {}) });
85
+ try {
86
+ return await verifyBoundary('deepseek.withRootH', () => steps(h));
87
+ } finally {
88
+ rmSync(root, { recursive: true, force: true });
89
+ }
90
+ }
91
+
92
+ /** Collect the streaming SSE events for a chat request against an isolated root. */
93
+ function withStream(body: unknown, fn: (events: SseEvent[], final: DeepSeekResponseEnvelope) => boolean, path = CHAT_PATH): Promise<boolean> {
94
+ return new Promise<boolean>((resolve, reject) => {
95
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
96
+ const events: SseEvent[] = [];
97
+ handleDeepSeekTwinRequest({ method: 'POST', path, body: JSON.stringify(body), root, occurredAt: OCCURRED_AT, sseSink: (e) => events.push(e) })
98
+ .then((final) => resolve(fn(events, final)))
99
+ .catch((err) => { if (isInfrastructureError(err)) reject(harnessError('deepseek.withStream', err)); else resolve(false); })
100
+ .finally(() => rmSync(root, { recursive: true, force: true }));
101
+ });
102
+ }
103
+
104
+ /** A connector verify against an isolated root, with the injected fake executor the verify builds. */
105
+ async function withConnectorRoot(id: string, fn: (root: string) => Promise<boolean>): Promise<boolean> {
106
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
107
+ try {
108
+ return await fn(root);
109
+ } catch (err) {
110
+ if (isInfrastructureError(err)) throw harnessError(id, err);
111
+ return false;
112
+ } finally {
113
+ rmSync(root, { recursive: true, force: true });
114
+ }
115
+ }
116
+
117
+ const ok = (r: DeepSeekResponseEnvelope) => r.status >= 200 && r.status < 300;
118
+ /** Refused as a CLIENT error with the vendor's envelope. Used wherever the twin's exact status is
119
+ * its own unverified choice — DeepSeek's published error table has no 404 at all — so asserting a
120
+ * specific code would claim a vendor fact this pack states it does not have (§9 round one). */
121
+ const refused = (r: DeepSeekResponseEnvelope) => r.status >= 400 && r.status < 500 && typeof (r.body as Body)?.error?.message === 'string';
122
+ const body = (r: DeepSeekResponseEnvelope) => r.body as Body;
123
+ const msg = (r: DeepSeekResponseEnvelope) => String((r.body as Body)?.error?.message ?? '');
124
+ const choice0 = (r: DeepSeekResponseEnvelope) => ((r.body as Body)?.choices as Body[])?.[0];
125
+ const text0 = (r: DeepSeekResponseEnvelope) => String(choice0(r)?.message?.content ?? '');
126
+
127
+ // ── shorthands ──
128
+ const done = (id: string, area: string, title: string, dimension: CapabilitySpec['dimension'], tier: CapabilitySpec['tier'], verify: CapabilitySpec['verify']): CapabilitySpec => ({ id, area, title, dimension, tier, expected: 'done', verify });
129
+ const todo = (id: string, area: string, title: string, dimension: CapabilitySpec['dimension'], tier: CapabilitySpec['tier']): CapabilitySpec => ({ id, area, title, dimension, tier, expected: 'todo' });
130
+
131
+ // DeepSeek's real paths. NOTE the absence of `/v1` — that is not an omission, it is the vendor's
132
+ // own base_url shape and `deepseek.chat.no_v1_prefix` asserts a twin serving `/v1` would be wrong.
133
+ const CHAT_PATH = '/chat/completions';
134
+ const BETA_CHAT = '/beta/chat/completions';
135
+ const FIM_PATH = '/beta/completions';
136
+ const FILES = '/files';
137
+ const MODELS = '/models';
138
+ const BALANCE = '/user/balance';
139
+
140
+ /** A pinned timestamp so ids, `created` and cache-ledger writes are deterministic. Where a verify
141
+ * REPEATS an identical transition it pins a DIFFERENT value deliberately (the kernel dedupes by
142
+ * content + millisecond — ADDING_A_TWIN.md §6). */
143
+ const OCCURRED_AT = '2026-08-31T12:00:00.000Z';
144
+
145
+ const MODEL = 'deepseek-v4-flash';
146
+ const PRO = 'deepseek-v4-pro';
147
+ const CHAT = (extra: Record<string, unknown> = {}) => ({ model: MODEL, messages: [{ role: 'user', content: 'hello twin' }], ...extra });
148
+ const WEATHER_TOOL = { type: 'function', function: { name: 'get_weather', parameters: { type: 'object', properties: { city: { type: 'string' }, days: { type: 'integer' } } } } };
149
+ const TIME_TOOL = { type: 'function', function: { name: 'get_time', parameters: { type: 'object', properties: { tz: { type: 'string' } } } } };
150
+ /** A valid PNG upload as the server's multipart adapter hands it to the handler. */
151
+ const UPLOAD = (extra: Record<string, unknown> = {}) => ({ purpose: 'user_data', filename: 'shot.png', media_type: 'image/png', content: 'AAA=', bytes: 3, ...extra });
152
+ const FILE_1 = 'file-api-twin000000000001';
153
+ const FILE_2 = 'file-api-twin000000000002';
154
+
155
+ /** A fake live executor over an in-memory account. Records every (method, path) so a connector
156
+ * verify can assert WHICH calls went out, not merely that something did. */
157
+ function fakeExecute(account: {
158
+ models?: Body[]; files?: Body[]; balance?: Body; fail?: string;
159
+ }, calls: Array<{ method: string; path: string; body?: unknown }> = []): DeepSeekExecute {
160
+ return async (method, path, reqBody) => {
161
+ calls.push({ method, path, ...(reqBody === undefined ? {} : { body: reqBody }) });
162
+ if (account.fail === path) return { error: { message: 'Authentication fails due to the wrong API key' } };
163
+ if (path === '/models') return { object: 'list', data: account.models ?? [] };
164
+ if (path === '/files') return { object: 'list', data: account.files ?? [] };
165
+ if (path === '/user/balance') return (account.balance ?? { is_available: true, balance_infos: [] }) as Body;
166
+ if (method === 'DELETE' && path.startsWith('/files/')) return { id: path.slice('/files/'.length), object: 'file', deleted: true };
167
+ return {};
168
+ };
169
+ }
170
+
171
+ export const DEEPSEEK_CAPABILITIES: CapabilitySpec[] = [
172
+ // ══ chat completions ═════════════════════════════════════════════════════════════════════
173
+ done('deepseek.chat.completion', 'chat', 'POST /chat/completions returns a faithful chat.completion envelope', 'api', 'core', () => withRoot(async (h) => {
174
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
175
+ if (!ok(r)) return false;
176
+ const b = body(r);
177
+ return b.object === 'chat.completion' && b.model === MODEL && typeof b.id === 'string' && b.id.startsWith('chatcmpl-')
178
+ && typeof b.system_fingerprint === 'string' && b.system_fingerprint.startsWith('fp_')
179
+ && Array.isArray(b.choices) && b.choices.length === 1
180
+ && choice0(r).index === 0 && choice0(r).message.role === 'assistant'
181
+ && text0(r).includes('[twin-stub:deepseek-v4-flash]') && text0(r).includes('hello twin')
182
+ && choice0(r).finish_reason === 'stop'
183
+ // …and DeepSeek's assistant message has NO `refusal` (OpenAI's does). Serving one would be
184
+ // the inverse false-green.
185
+ && !('refusal' in choice0(r).message);
186
+ })),
187
+
188
+ done('deepseek.chat.system_and_multi_turn', 'chat', 'System messages and multi-turn history are accepted and echoed from the last user turn', 'api', 'core', () => withRoot(async (h) => {
189
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [
190
+ { role: 'system', content: 'be terse' },
191
+ { role: 'user', content: 'first question' },
192
+ { role: 'assistant', content: 'first answer', reasoning_content: '' },
193
+ { role: 'user', content: 'second question' },
194
+ ] } });
195
+ if (!ok(r)) return false;
196
+ // The echo proves the twin read the LAST user turn, not the first — a handler returning a
197
+ // fixed string would fail this.
198
+ if (!text0(r).includes('second question') || text0(r).includes('first question')) return false;
199
+ // Negative: an empty message list is refused with DeepSeek's parameter status.
200
+ const empty = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [] } });
201
+ // Negative: an unknown role is refused too.
202
+ const badRole = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'developer', content: 'x' }] } });
203
+ return empty.status === 422 && badRole.status === 422;
204
+ })),
205
+
206
+ done('deepseek.chat.max_tokens_truncation', 'chat', "max_tokens truncates and reports finish_reason 'length'", 'api', 'core', () => withRoot(async (h) => {
207
+ const full = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
208
+ const cut = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ max_tokens: 4 }) });
209
+ if (!ok(full) || !ok(cut)) return false;
210
+ if (choice0(cut).finish_reason !== 'length' || choice0(full).finish_reason !== 'stop') return false;
211
+ if (text0(cut).length >= text0(full).length) return false;
212
+ // Negative: DeepSeek's documented bound is an integer >= 1.
213
+ const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ max_tokens: 0 }) });
214
+ return bad.status === 422;
215
+ })),
216
+
217
+ done('deepseek.chat.stop_sequences', 'chat', 'stop truncates at the earliest match; more than 16 sequences is refused', 'api', 'common', () => withRoot(async (h) => {
218
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ stop: ['DeepSeek twin', 'deterministic'] }) });
219
+ if (!ok(r)) return false;
220
+ // 'deterministic' occurs EARLIER in the stub text than 'DeepSeek twin', so truncation must
221
+ // happen there — an implementation that took `stop[0]` would keep 'deterministic' and fail here.
222
+ if (!text0(r).includes('This is a ') || text0(r).includes('deterministic') || text0(r).includes('DeepSeek twin')) return false;
223
+ const many = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ stop: Array.from({ length: 17 }, (_, i) => `s${i}`) }) });
224
+ const sixteen = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ stop: Array.from({ length: 16 }, (_, i) => `s${i}`) }) });
225
+ return many.status === 422 && ok(sixteen);
226
+ })),
227
+
228
+ done('deepseek.chat.rejects_openai_only_params', 'chat', "The OpenAI parameters DeepSeek's closed table does not declare (n, seed, logit_bias, top_k, user, max_completion_tokens, service_tier, parallel_tool_calls, functions) are refused BY NAME with 422", 'api', 'core', () => withRoot(async (h) => {
229
+ // THE flagship OpenAI-divergence check. A twin copied from an OpenAI-shaped exemplar serves
230
+ // every one of these happily, which is the inherited-permissiveness false-green
231
+ // ADDING_A_TWIN.md §0 names. Each must be 422 — DeepSeek's "Invalid Parameters" — not 400.
232
+ for (const [key, value] of [
233
+ ['n', 2], ['seed', 42], ['logit_bias', { '1': 1 }], ['top_k', 5], ['user', 'u1'],
234
+ ['max_completion_tokens', 32], ['service_tier', 'flex'], ['parallel_tool_calls', false],
235
+ ['functions', [{ name: 'f' }]], ['store', true], ['metadata', { a: 'b' }],
236
+ ] as Array<[string, unknown]>) {
237
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ [key]: value }) });
238
+ if (r.status !== 422) return false;
239
+ // The refusal must NAME the offending key, so a caller can act on it.
240
+ if (!msg(r).includes(`'${key}'`)) return false;
241
+ }
242
+ return true;
243
+ })),
244
+
245
+ done('deepseek.chat.no_v1_prefix', 'chat', "DeepSeek's base_url carries no version segment: /v1/... is not served", 'api', 'core', () => withRoot(async (h) => {
246
+ // api-docs.deepseek.com's own quick start uses base_url = https://api.deepseek.com with the
247
+ // endpoint at /chat/completions. A twin answering on the OpenAI-shaped /v1 prefix would be
248
+ // asserting a route the vendor does not have.
249
+ const served = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
250
+ const v1chat = await h({ m: 'POST', p: '/v1/chat/completions', b: CHAT() });
251
+ const v1models = await h({ m: 'GET', p: '/v1/models' });
252
+ // The claim is that the route is REFUSED, not that the code is exactly 404: DeepSeek's
253
+ // published error table is {400,401,402,422,429,500,503} and contains no 404 at all, so the
254
+ // twin's choice of status for an unrouted path is unverified against any first-party source
255
+ // (filed as `deepseek.errors.unknown_route_envelope`). Asserting 404 exactly would be claiming
256
+ // a vendor fact this pack explicitly says it does not have (§9 round one, NIT 11).
257
+ return ok(served) && refused(v1chat) && refused(v1models)
258
+ && msg(v1chat).includes('Unknown request URL');
259
+ })),
260
+
261
+ done('deepseek.chat.deprecated_params_accepted', 'chat', 'frequency_penalty / presence_penalty are DEPRECATED but accepted with no effect — never an error', 'api', 'common', () => withRoot(async (h) => {
262
+ // The mirror image of the rejection surface, and just as load-bearing: DeepSeek documents these
263
+ // as having "no effect", not as errors. A twin that 4xx'd them would be refusing surface the
264
+ // vendor has. "No effect" is asserted, not assumed: the content must be byte-identical.
265
+ const plain = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
266
+ const penalised = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ frequency_penalty: 1.5, presence_penalty: -1.2 }) });
267
+ if (!ok(plain) || !ok(penalised)) return false;
268
+ // MUTATION-GATE FINDING: "the two answers are equal" is satisfied by a dead twin answering
269
+ // {} twice. The equality only means something once each side is proven to be a REAL completion,
270
+ // so the content is asserted absolutely before it is compared.
271
+ if (!text0(penalised).includes('[twin-stub:deepseek-v4-flash]') || !text0(penalised).includes('hello twin')) return false;
272
+ if (body(penalised).object !== 'chat.completion' || choice0(penalised).finish_reason !== 'stop') return false;
273
+ return text0(plain) === text0(penalised);
274
+ })),
275
+
276
+ done('deepseek.chat.temperature_ignored_while_thinking', 'chat', 'temperature / top_p are ignored (not refused) while thinking is enabled, and bounded when checked', 'api', 'common', () => withRoot(async (h) => {
277
+ // "setting these parameters will not trigger an error but will also have no effect"
278
+ // (api-docs.deepseek.com/guides/thinking_mode). Thinking is on by default for every V4 model.
279
+ const plain = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
280
+ const hot = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ temperature: 1.9, top_p: 0.1 }) });
281
+ if (!ok(plain) || !ok(hot)) return false;
282
+ // Same shape as the deprecated-params cell: prove each side is a real completion before the
283
+ // equality is allowed to mean anything (a dead twin answers {} twice, and {} === {}).
284
+ if (!text0(hot).includes('[twin-stub:deepseek-v4-flash]') || body(hot).object !== 'chat.completion') return false;
285
+ if (text0(plain) !== text0(hot)) return false;
286
+ // …but the documented RANGES are still enforced: temperature 0–2, top_p in (0, 1].
287
+ const tooHot = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ temperature: 2.5 }) });
288
+ const badP = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ top_p: 1.5 }) });
289
+ return tooHot.status === 422 && badP.status === 422;
290
+ })),
291
+
292
+ done('deepseek.chat.user_id', 'chat', "user_id (not OpenAI's `user`) is accepted, charset-checked and length-capped at 512", 'api', 'common', () => withRoot(async (h) => {
293
+ const good = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'tenant_123-user' }) });
294
+ const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'tenant 123' }) });
295
+ const long = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'a'.repeat(513) }) });
296
+ const atCap = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'a'.repeat(512) }) });
297
+ return ok(good) && bad.status === 422 && long.status === 422 && ok(atCap);
298
+ })),
299
+
300
+ done('deepseek.chat.message_name', 'chat', 'Participant `name` is accepted on system/user/assistant turns and counted into prompt tokens', 'api', 'niche', () => withRoot(async (h) => {
301
+ const named = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi', name: 'customer_alpha' }] } });
302
+ const plain = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi' }] } });
303
+ if (!ok(named) || !ok(plain)) return false;
304
+ // A name is real payload, so it must move the prompt-token count — a handler that dropped the
305
+ // field entirely would report the same number.
306
+ if (body(named).usage.prompt_tokens <= body(plain).usage.prompt_tokens) return false;
307
+ // …and a NON-STRING name is a 422, not a 200. §9 round two, SHOULD-FIX 2: it used to reach
308
+ // `estimateTokens`, produce NaN and serialise as `null`, so the twin answered 200 with
309
+ // `prompt_tokens: null` — breaking the cache invariant its own capability asserts and handing
310
+ // the SDK a body its usage schema cannot decode. The revert matrix showed this cell was HOLLOW
311
+ // without the assertion.
312
+ const badName = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi', name: 12345 }] } });
313
+ if (badName.status !== 422) return false;
314
+ const badReasoning = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'assistant', content: 'x', reasoning_content: 7 }] } });
315
+ // …and a non-array tool_calls is a 422 too, not the retryable 500 a TypeError used to produce.
316
+ const badCalls = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'assistant', content: 'x', tool_calls: 5 }] } });
317
+ return badReasoning.status === 422 && badCalls.status === 422;
318
+ })),
319
+
320
+ done('deepseek.chat.logprobs_envelope', 'chat', "logprobs returns DeepSeek's TWO-channel block (content + reasoning_content); top_logprobs is bounded and requires logprobs", 'api', 'niche', () => withRoot(async (h) => {
321
+ const off = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
322
+ if (!ok(off) || choice0(off).logprobs !== null) return false;
323
+ const on = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: true, top_logprobs: 3 }) });
324
+ if (!ok(on)) return false;
325
+ const lp = choice0(on).logprobs as Body;
326
+ if (!lp || !Array.isArray(lp.content) || lp.content.length === 0) return false;
327
+ if (typeof lp.content[0].token !== 'string' || typeof lp.content[0].logprob !== 'number') return false;
328
+ if (lp.content[0].top_logprobs.length !== 3) return false;
329
+ // The reasoning channel is the DeepSeek-specific half — OpenAI's logprobs has no such key.
330
+ if (!Array.isArray(lp.reasoning_content) || lp.reasoning_content.length === 0) return false;
331
+ const tooMany = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: true, top_logprobs: 21 }) });
332
+ const orphan = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ top_logprobs: 2 }) });
333
+ return tooMany.status === 422 && orphan.status === 422;
334
+ })),
335
+
336
+ todo('deepseek.chat.content_filter_finish', 'chat', "finish_reason 'content_filter' when DeepSeek's safety layer stops a generation", 'api', 'common'),
337
+ todo('deepseek.chat.insufficient_system_resource_finish', 'chat', "finish_reason 'insufficient_system_resource' — DeepSeek's own value, mapped by the SDK to the unified `error` reason", 'api', 'niche'),
338
+ todo('deepseek.chat.name_on_tool_message', 'chat', "How DeepSeek answers a `name` on a role:tool turn — it supports names on system/user/assistant only, and @ai-sdk/deepseek strips it client-side with a warning, so the server's own behaviour is unobserved", 'api', 'niche'),
339
+ // CORE, not common: every integration that grows a conversation hits the context ceiling in its
340
+ // first week, and the vendor's refusal is what a client must handle (§9 round one, NIT 16 — a
341
+ // manifest with ZERO core todos is a tiering smell, not an achievement).
342
+ todo('deepseek.chat.context_length_exceeded', 'chat', 'A prompt beyond the model context window is refused with the vendor-documented error', 'api', 'core'),
343
+
344
+ // ══ FIM completions (beta) ═══════════════════════════════════════════════════════════════
345
+ done('deepseek.completions.fim', 'completions', 'POST /beta/completions returns a faithful text_completion envelope', 'api', 'common', () => withRoot(async (h) => {
346
+ const r = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'def fib(n):\n ' } });
347
+ if (!ok(r)) return false;
348
+ const b = body(r);
349
+ return b.object === 'text_completion' && b.model === PRO && String(b.id).startsWith('cmpl-')
350
+ && Array.isArray(b.choices) && typeof b.choices[0].text === 'string'
351
+ && b.choices[0].text.includes('[twin-stub:deepseek-v4-pro]')
352
+ && b.choices[0].finish_reason === 'stop' && b.choices[0].logprobs === null
353
+ // FIM's response carries the same KV-cache usage split as chat.
354
+ && typeof b.usage.prompt_cache_miss_tokens === 'number';
355
+ })),
356
+
357
+ done('deepseek.completions.suffix_and_echo', 'completions', 'suffix is honoured and echo prepends the prompt to the returned text', 'api', 'common', () => withRoot(async (h) => {
358
+ const withSuffix = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'a=', suffix: 'return a' } });
359
+ const noSuffix = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'a=' } });
360
+ if (!ok(withSuffix) || !ok(noSuffix)) return false;
361
+ if (!String(body(withSuffix).choices[0].text).includes('return a')) return false;
362
+ if (String(noSuffix.body && body(noSuffix).choices[0].text).includes('return a')) return false;
363
+ const echoed = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'PROMPT_HEAD', echo: true } });
364
+ const plain = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'PROMPT_HEAD' } });
365
+ return ok(echoed) && String(body(echoed).choices[0].text).startsWith('PROMPT_HEAD')
366
+ && !String(body(plain).choices[0].text).startsWith('PROMPT_HEAD');
367
+ })),
368
+
369
+ done('deepseek.completions.beta_only_and_closed_model', 'completions', 'FIM lives only under /beta and accepts only deepseek-v4-pro', 'api', 'common', () => withRoot(async (h) => {
370
+ const offBeta = await h({ m: 'POST', p: '/completions', b: { model: PRO, prompt: 'x' } });
371
+ const wrongModel = await h({ m: 'POST', p: FIM_PATH, b: { model: MODEL, prompt: 'x' } });
372
+ const noPrompt = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO } });
373
+ return refused(offBeta) && msg(offBeta).includes('beta base URL')
374
+ && wrongModel.status === 422 && noPrompt.status === 422;
375
+ })),
376
+
377
+ done('deepseek.completions.logprobs_is_an_integer', 'completions', "The FIM endpoint's `logprobs` is an INTEGER (0..20), unlike the chat endpoint's boolean of the same name", 'api', 'niche', () => withRoot(async (h) => {
378
+ // Same key name, different type, on the same vendor. An OpenAI-shaped twin that shared one
379
+ // validator across both endpoints would accept the wrong type on one of them.
380
+ const good = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', logprobs: 5 } });
381
+ const boolean = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', logprobs: true } });
382
+ const tooMany = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', logprobs: 21 } });
383
+ // …and the chat endpoint is the exact inverse.
384
+ const chatBool = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: true }) });
385
+ const chatInt = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: 5 }) });
386
+ return ok(good) && boolean.status === 422 && tooMany.status === 422 && ok(chatBool) && chatInt.status === 422;
387
+ })),
388
+
389
+ done('deepseek.completions.refuses_unmodeled_streaming', 'completions', 'A FIM request asking to stream is REFUSED by name rather than answered with a unary body', 'api', 'common', () => withRoot(async (h) => {
390
+ // The endpoint really does declare `stream`, so serving a unary text_completion to a caller who
391
+ // asked for SSE would be a fake success — the exact failure mode "unmodeled ops fail like the
392
+ // vendor" exists to prevent. The refusal names the filed gap so the message is actionable.
393
+ const streamed = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', stream: true } });
394
+ if (streamed.status !== 422 || !msg(streamed).includes('deepseek.completions.streaming')) return false;
395
+ // …and the non-streaming path is unaffected, or the refusal would just be a broken endpoint.
396
+ const unary = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', stream: false } });
397
+ return ok(unary) && body(unary).object === 'text_completion';
398
+ })),
399
+ todo('deepseek.completions.cache_accounting', 'completions', 'FIM requests report prompt_cache_hit_tokens as a hard zero and record no prefix unit — the field is served in the faithful shape but is not measured on this endpoint the way it is on /chat/completions (§9 round one, NIT 18)', 'api', 'niche'),
400
+ todo('deepseek.completions.streaming', 'completions', 'FIM completion streaming (stream + stream_options.include_usage on /beta/completions) — currently refused by name rather than modeled', 'api', 'niche'),
401
+
402
+ // ══ models ═══════════════════════════════════════════════════════════════════════════════
403
+ done('deepseek.models.list', 'models', 'GET /models lists the published catalog', 'api', 'core', () => withRoot(async (h) => {
404
+ const r = await h({ m: 'GET', p: MODELS });
405
+ if (!ok(r)) return false;
406
+ const b = body(r);
407
+ const ids = (b.data as Body[]).map((m) => m.id);
408
+ return b.object === 'list' && ids.includes('deepseek-v4-flash') && ids.includes('deepseek-v4-pro')
409
+ && ids.includes('deepseek-v4-flash-vision-exp')
410
+ // The retired aliases must NOT be listed.
411
+ && !ids.includes('deepseek-chat') && !ids.includes('deepseek-reasoner');
412
+ })),
413
+
414
+ done('deepseek.models.row_shape', 'models', "A model row is exactly {id, object, owned_by} — DeepSeek's row has no `created` (OpenAI's does)", 'api', 'common', () => withRoot(async (h) => {
415
+ const r = await h({ m: 'GET', p: MODELS });
416
+ if (!ok(r)) return false;
417
+ // A LITERAL expected key set, licensed by §6 because DeepSeek's own list-models example
418
+ // response is a closed three-key object. Inventing `created` to look OpenAI-shaped is exactly
419
+ // the inverse false-green.
420
+ return (body(r).data as Body[]).every((m) => Object.keys(m).sort().join(',') === 'id,object,owned_by' && m.object === 'model' && m.owned_by === 'deepseek');
421
+ })),
422
+
423
+ done('deepseek.models.retired_and_unknown_ids', 'models', 'A retired alias is refused BY NAME with its retirement date; an unknown id is refused too', 'api', 'core', () => withRoot(async (h) => {
424
+ for (const retired of ['deepseek-chat', 'deepseek-reasoner']) {
425
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ model: retired }) });
426
+ if (r.status !== 422 || !msg(r).includes('2026-07-24')) return false;
427
+ }
428
+ const unknown = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ model: 'gpt-4o' }) });
429
+ return unknown.status === 422 && msg(unknown).includes('does not exist');
430
+ })),
431
+
432
+ done('deepseek.models.pulled_rows_are_served', 'models', "A model observed by a connector pull is served by GET /models, not merely folded into the log", 'api', 'niche', () => withConnectorRoot('deepseek.models.pulled_rows_are_served', async (root) => {
433
+ const execute = fakeExecute({ models: [{ id: 'deepseek-v4-preview-internal', object: 'model', owned_by: 'deepseek' }] });
434
+ await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
435
+ const r = await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, root });
436
+ const ids = (body(r).data as Body[]).map((m) => m.id);
437
+ // Both the pulled row AND the published catalog must survive: an override that dropped either
438
+ // side makes the twin disagree with itself.
439
+ return ok(r) && ids.includes('deepseek-v4-preview-internal') && ids.includes('deepseek-v4-flash');
440
+ })),
441
+
442
+ todo('deepseek.models.per_model_metadata', 'models', 'Per-model metadata (context window, pricing tier, concurrency) — DeepSeek publishes these on its pricing page but not on the API row', 'api', 'niche'),
443
+
444
+ // ══ user balance ═════════════════════════════════════════════════════════════════════════
445
+ done('deepseek.balance.get', 'balance', 'GET /user/balance returns is_available plus per-currency balance_infos with STRING amounts', 'api', 'common', () => withRoot(async (h) => {
446
+ const r = await h({ m: 'GET', p: BALANCE });
447
+ if (!ok(r)) return false;
448
+ const b = body(r);
449
+ const info = (b.balance_infos as Body[])[0];
450
+ return b.is_available === true && Array.isArray(b.balance_infos) && b.balance_infos.length > 0
451
+ && ['CNY', 'USD'].includes(info.currency)
452
+ // Amounts are STRINGS on this vendor. A twin that emitted numbers would break any client
453
+ // decoding them as the documented type.
454
+ && typeof info.total_balance === 'string' && typeof info.granted_balance === 'string' && typeof info.topped_up_balance === 'string';
455
+ })),
456
+
457
+ done('deepseek.balance.insufficient_402', 'balance', 'A drained account refuses completions with 402 "You have run out of balance" — a status OpenAI does not have', 'api', 'core', () => withRoot(async (h, root) => {
458
+ const before = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
459
+ if (!ok(before)) return false;
460
+ // STATE-DRIVEN, not a header trick: the refusal reads `is_available` off the same projection
461
+ // `GET /user/balance` serves, so it is a fact about the twin's state.
462
+ await applyTwinWrite('deepseek', {
463
+ operation: 'balance.update', subjectType: 'balance', subjectId: 'account',
464
+ fields: { is_available: false, balance_infos: [{ currency: 'USD', total_balance: '0.00', granted_balance: '0.00', topped_up_balance: '0.00' }] },
465
+ occurredAt: OCCURRED_AT, actor: { kind: 'agent' },
466
+ }, root);
467
+ const after = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
468
+ const fim = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x' } });
469
+ const balance = await h({ m: 'GET', p: BALANCE });
470
+ return after.status === 402 && msg(after).includes('run out of balance')
471
+ && fim.status === 402
472
+ // Reads still work on a drained account — only generation is refused.
473
+ && ok(balance) && body(balance).is_available === false;
474
+ })),
475
+
476
+ todo('deepseek.balance.multi_currency', 'balance', 'An account holding both CNY and USD balances reports both rows', 'api', 'niche'),
477
+
478
+ // ══ context cache (KV cache) ═════════════════════════════════════════════════════════════
479
+ done('deepseek.cache.hit_miss_split', 'cache', 'usage carries prompt_cache_hit_tokens + prompt_cache_miss_tokens, summing exactly to prompt_tokens', 'api', 'core', () => withRoot(async (h) => {
480
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
481
+ if (!ok(r)) return false;
482
+ const u = body(r).usage as Body;
483
+ return typeof u.prompt_cache_hit_tokens === 'number' && typeof u.prompt_cache_miss_tokens === 'number'
484
+ && u.prompt_cache_hit_tokens + u.prompt_cache_miss_tokens === u.prompt_tokens
485
+ // …and the OpenAI-shaped mirror field agrees with the hit count, because the SDK's usage
486
+ // schema declares both and its `cacheRead` reads the DeepSeek one.
487
+ && (u.prompt_tokens_details as Body).cached_tokens === u.prompt_cache_hit_tokens;
488
+ })),
489
+
490
+ done('deepseek.cache.prefix_hit_on_continuation', 'cache', "A follow-up turn genuinely HITS the cache: the twin keeps a prefix-unit ledger rather than fabricating a split", 'api', 'core', () => withRoot(async (h) => {
491
+ const first = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'a long opening question about caching behaviour' }] } });
492
+ if (!ok(first)) return false;
493
+ // A brand-new conversation can hit nothing.
494
+ if ((body(first).usage as Body).prompt_cache_hit_tokens !== 0) return false;
495
+ const answer = text0(first);
496
+ const second = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [
497
+ { role: 'user', content: 'a long opening question about caching behaviour' },
498
+ { role: 'assistant', content: answer, reasoning_content: '' },
499
+ { role: 'user', content: 'and a follow-up' },
500
+ ] } });
501
+ if (!ok(second)) return false;
502
+ const u = body(second).usage as Body;
503
+ // The continuation's recorded prefix (opening + answer) is reported as a hit and the NEW turn
504
+ // as a miss. `miss > 0` is the assertion that matters: an implementation measuring the hit off
505
+ // the stored prefix text instead of this request's own messages over-counts, the clamp in
506
+ // `buildUsage` rescues the invariant, and the follow-up reports a 100% cache with the new turn
507
+ // silently absent from the accounting. That was a real defect this cell caught.
508
+ return u.prompt_cache_hit_tokens > 0 && u.prompt_cache_miss_tokens > 0
509
+ && u.prompt_cache_hit_tokens + u.prompt_cache_miss_tokens === u.prompt_tokens;
510
+ })),
511
+
512
+ done('deepseek.cache.unrelated_prompt_misses', 'cache', 'An unrelated conversation reports a full miss even after the ledger has entries', 'api', 'common', () => withRoot(async (h) => {
513
+ // The other direction, and the one a fabricated split would fail: having served one
514
+ // conversation, a DIFFERENT one must still be 100% miss. A twin that reported a fixed
515
+ // hit ratio would pass the hit test and fail this one.
516
+ await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'topic one' }] } });
517
+ const other = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'a completely different topic' }] } });
518
+ if (!ok(other)) return false;
519
+ const u = body(other).usage as Body;
520
+ return u.prompt_cache_hit_tokens === 0 && u.prompt_cache_miss_tokens === u.prompt_tokens && u.prompt_tokens > 0;
521
+ })),
522
+
523
+ done('deepseek.cache.identical_replay_hits', 'cache', 'A byte-identical repeat of a request FULLY MATCHES the user-input-end prefix unit the first call recorded, and reports a full cache hit', 'api', 'common', () => withRoot(async (h) => {
524
+ // The canonical KV-cache demonstration, and the one the twin used to get wrong: it stopped its
525
+ // prefix search one message short, so an identical replay reported 0 hit tokens. The vendor's
526
+ // rule is that a request hits when it "fully matches a cache prefix unit", and units form at
527
+ // "the end position of the user input" (api-docs.deepseek.com/guides/kv_cache).
528
+ const req = { model: MODEL, messages: [{ role: 'user', content: 'a repeatable single-turn question' }] };
529
+ const first = await h({ m: 'POST', p: CHAT_PATH, b: req });
530
+ if (!ok(first) || (body(first).usage as Body).prompt_cache_hit_tokens !== 0) return false;
531
+ const second = await h({ m: 'POST', p: CHAT_PATH, b: req });
532
+ if (!ok(second)) return false;
533
+ const u = body(second).usage as Body;
534
+ // A FULL match: every prompt token is a hit, and none is a miss.
535
+ if (u.prompt_cache_hit_tokens !== u.prompt_tokens || u.prompt_cache_miss_tokens !== 0 || u.prompt_tokens === 0) return false;
536
+ // …and this is not a "second call always hits" rule: a DIFFERENT request in the same warmed
537
+ // root still misses completely.
538
+ const other = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'an entirely different question' }] } });
539
+ return ok(other) && (body(other).usage as Body).prompt_cache_hit_tokens === 0;
540
+ })),
541
+
542
+ done('deepseek.cache.key_covers_every_charged_field', 'cache', 'A turn carrying a different reasoning_content, name or tool_call is a MISS — the prefix key covers EVERY field the token count charges for', 'api', 'common', () => withRoot(async (h) => {
543
+ // §9 round one, NIT 15: the key dropped `name` and `reasoning_content` while `messageTokens`
544
+ // counted them, so a caller could attach an arbitrarily large reasoning_content to a matching
545
+ // turn and have those tokens billed as a hit.
546
+ const opening = { model: MODEL, messages: [{ role: 'user', content: 'opening' }] };
547
+ const seeded = await h({ m: 'POST', p: CHAT_PATH, b: opening });
548
+ if (!ok(seeded)) return false;
549
+ const answer = text0(seeded);
550
+ const cont = (assistant: Record<string, unknown>) => ({ model: MODEL, messages: [{ role: 'user', content: 'opening' }, assistant, { role: 'user', content: 'next' }] });
551
+ // A SECOND recorded shape whose assistant turn carries a tool call. Without it the padded-id
552
+ // case below differs from every recorded unit in TWO ways (it has tool_calls at all, and the id
553
+ // is long), so its miss would be explained by the first difference and the cell would stay green
554
+ // with the id dropped from the key — which is exactly how the first version of this pin came out
555
+ // HOLLOW on the revert matrix. This baseline makes the id the ONLY difference.
556
+ const TOOL_TURN = { role: 'assistant', content: answer, tool_calls: [{ id: 'call_twin_1', type: 'function', function: { name: 'f', arguments: '{}' } }] };
557
+ const toolBaseline = await h({ m: 'POST', p: CHAT_PATH, b: cont(TOOL_TURN) });
558
+ if (!ok(toolBaseline) || (body(toolBaseline).usage as Body).prompt_cache_miss_tokens > 200) return false;
559
+ // An absent reasoning_content and an empty one are the SAME key — that is the shape
560
+ // @ai-sdk/deepseek produces, so a stricter key would make every real SDK continuation a miss.
561
+ const plain = await h({ m: 'POST', p: CHAT_PATH, b: cont({ role: 'assistant', content: answer }) });
562
+ const empty = await h({ m: 'POST', p: CHAT_PATH, b: cont({ role: 'assistant', content: answer, reasoning_content: '' }) });
563
+ if (!ok(plain) || !ok(empty)) return false;
564
+ if ((body(plain).usage as Body).prompt_cache_hit_tokens === 0) return false;
565
+ // `empty` runs AFTER `plain` in the same root, so if the two canonicalize identically it fully
566
+ // matches the 3-message unit `plain` just recorded — a zero miss. If they canonicalized
567
+ // differently it would have to fall back to the shorter prefix and report a miss, which is
568
+ // exactly what a key that ignored `reasoning_content` inconsistently would produce.
569
+ if ((body(empty).usage as Body).prompt_cache_miss_tokens !== 0) return false;
570
+ // …but a turn carrying a LARGE payload in ANY field the token count charges for is a different
571
+ // conversation, and must not be billed as a hit against the unit recorded for the plain one.
572
+ // THREE fields, one per charged term in `messageTokens` — §9 round two, BLOCKER 1: the first
573
+ // version varied only `reasoning_content`, so `tool_calls[].id` (which `messageTokens` charges
574
+ // through `JSON.stringify(tc)`) slipped through and 999 tokens of caller padding were billed as
575
+ // a hit. A cell that proves one charged field says nothing about the others.
576
+ const fatCases: Array<[string, Record<string, unknown>]> = [
577
+ ['reasoning_content', { role: 'assistant', content: answer, reasoning_content: 'x'.repeat(4000) }],
578
+ ['name', { role: 'assistant', content: answer, name: 'n'.repeat(4000) }],
579
+ // Identical to TOOL_TURN except for the id — so a key that omitted the id would FULLY match
580
+ // the unit `toolBaseline` just recorded and bill all 4000 characters as a hit.
581
+ ['tool_calls[].id', { ...TOOL_TURN, tool_calls: [{ id: `call_${'z'.repeat(4000)}`, type: 'function', function: { name: 'f', arguments: '{}' } }] }],
582
+ ];
583
+ for (const [, assistant] of fatCases) {
584
+ const fat = await h({ m: 'POST', p: CHAT_PATH, b: cont(assistant) });
585
+ if (!ok(fat)) return false;
586
+ const fu = body(fat).usage as Body;
587
+ if (fu.prompt_cache_hit_tokens >= (body(toolBaseline).usage as Body).prompt_tokens + 10) return false;
588
+ if (fu.prompt_cache_miss_tokens <= 900) return false;
589
+ if (fu.prompt_cache_hit_tokens + fu.prompt_cache_miss_tokens !== fu.prompt_tokens) return false;
590
+ }
591
+ return true;
592
+ })),
593
+
594
+ todo('deepseek.cache.interval_units', 'cache', 'Cache prefix units cut at "fixed token intervals" inside long inputs — DeepSeek publishes no interval, so the twin does not invent one', 'api', 'niche'),
595
+ todo('deepseek.cache.expiry', 'cache', 'Cache entries auto-clearing "within hours to days": the kernel\'s world clock is deterministic by construction and is explicitly the seam TTL/expiry logic reads at serve time, so this is unbuilt rather than impossible (§9 round one, SHOULD-FIX 5)', 'api', 'niche'),
596
+
597
+ // ══ reasoning / thinking mode ════════════════════════════════════════════════════════════
598
+ done('deepseek.reasoning.enabled_by_default', 'reasoning', 'V4 models think by default: the assistant turn carries reasoning_content and reasoning_tokens', 'api', 'core', () => withRoot(async (h) => {
599
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
600
+ if (!ok(r)) return false;
601
+ const rc = choice0(r).message.reasoning_content;
602
+ return typeof rc === 'string' && rc.includes('[twin-stub:') && rc.includes('reasoning_effort=high')
603
+ && (body(r).usage.completion_tokens_details as Body)?.reasoning_tokens > 0;
604
+ })),
605
+
606
+ done('deepseek.reasoning.disabled', 'reasoning', "thinking.type 'disabled' removes reasoning_content and its token accounting", 'api', 'core', () => withRoot(async (h) => {
607
+ const off = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ thinking: { type: 'disabled' } }) });
608
+ if (!ok(off)) return false;
609
+ if (choice0(off).message.reasoning_content !== undefined) return false;
610
+ if (body(off).usage.completion_tokens_details !== undefined) return false;
611
+ // …and the closed set is enforced: `adaptive` is a legacy SDK-side value the API does not take.
612
+ const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ thinking: { type: 'adaptive' } }) });
613
+ return bad.status === 422;
614
+ })),
615
+
616
+ done('deepseek.reasoning.effort', 'reasoning', "reasoning_effort is the closed set low|high|max, accepted at BOTH documented placements, with medium/xhigh mapped as the vendor documents", 'api', 'common', () => withRoot(async (h) => {
617
+ // Two first-party sources disagree on placement — the API reference puts it at
618
+ // `thinking.reasoning_effort`, `@ai-sdk/deepseek` sends a top-level `reasoning_effort`. The twin
619
+ // accepts BOTH rather than rejecting a shape a first-party client demonstrably sends.
620
+ const top = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ reasoning_effort: 'max' }) });
621
+ const nested = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ thinking: { type: 'enabled', reasoning_effort: 'max' } }) });
622
+ if (!ok(top) || !ok(nested)) return false;
623
+ // The effort must reach the turn — a handler that parsed and dropped it would serve `high`.
624
+ if (!String(choice0(top).message.reasoning_content).includes('reasoning_effort=max')) return false;
625
+ if (String(choice0(nested).message.reasoning_content) !== String(choice0(top).message.reasoning_content)) return false;
626
+ // Legacy values the vendor documents as MAPPED to high, not refused.
627
+ const legacy = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ reasoning_effort: 'medium' }) });
628
+ if (!ok(legacy) || !String(choice0(legacy).message.reasoning_content).includes('reasoning_effort=high')) return false;
629
+ const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ reasoning_effort: 'ultra' }) });
630
+ return bad.status === 422;
631
+ })),
632
+
633
+ done('deepseek.reasoning.tools_require_handback', 'reasoning', 'With `tools` set, a previous assistant turn missing reasoning_content is a documented 400 — not a 422 and not a silent success', 'api', 'core', () => withRoot(async (h) => {
634
+ // "If the request carries the tools parameter: the reasoning_content of all previous turns
635
+ // should be passed back … the API will return a 400 error"
636
+ // (api-docs.deepseek.com/guides/thinking_mode). This is the one refusal on this vendor that is
637
+ // a 400 rather than a 422, and no OpenAI-shaped twin has it at all.
638
+ const history = [{ role: 'user', content: 'q' }, { role: 'assistant', content: 'a' }, { role: 'user', content: 'q2' }];
639
+ const missing = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: history, tools: [WEATHER_TOOL] } });
640
+ if (missing.status !== 400) return false;
641
+ // An EMPTY STRING counts as passed back — that is literally what @ai-sdk/deepseek sends for a
642
+ // turn with no reasoning, so a stricter check would refuse the real SDK.
643
+ const empty = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, tools: [WEATHER_TOOL], messages: [
644
+ { role: 'user', content: 'q' }, { role: 'assistant', content: 'a', reasoning_content: '' }, { role: 'user', content: 'q2' },
645
+ ] } });
646
+ // …and WITHOUT tools the same history is fine: the rule is scoped to tool requests.
647
+ const noTools = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: history } });
648
+ return ok(empty) && ok(noTools);
649
+ })),
650
+
651
+
652
+ // ══ streaming ════════════════════════════════════════════════════════════════════════════
653
+ done('deepseek.streaming.chat', 'streaming', 'Streaming emits the vendor chunk sequence and terminates with [DONE]', 'api', 'core', () => withStream({ ...CHAT(), stream: true }, (events) => {
654
+ const data = events.filter((e) => !e.done).map((e) => e.data as Body);
655
+ if (!events.length || events[events.length - 1]!.done !== true) return false;
656
+ if (data[0]?.object !== 'chat.completion.chunk') return false;
657
+ if ((data[0]!.choices as Body[])[0].delta.role !== 'assistant') return false;
658
+ const text = data.map((d) => (d.choices as Body[])[0]?.delta?.content ?? '').join('');
659
+ const finish = data.find((d) => (d.choices as Body[])[0]?.finish_reason != null);
660
+ return text.includes('[twin-stub:deepseek-v4-flash]') && text.includes('hello twin')
661
+ && (finish!.choices as Body[])[0].finish_reason === 'stop';
662
+ })),
663
+
664
+ done('deepseek.streaming.reasoning_before_text', 'streaming', 'reasoning_content deltas arrive BEFORE the first content delta and never interleave', 'api', 'core', () => withStream({ ...CHAT(), stream: true }, (events) => {
665
+ const data = events.filter((e) => !e.done).map((e) => e.data as Body);
666
+ const kinds = data.map((d) => {
667
+ const delta = (d.choices as Body[])[0]?.delta as Body | undefined;
668
+ if (typeof delta?.reasoning_content === 'string') return 'r';
669
+ if (typeof delta?.content === 'string' && delta.content !== '') return 'c';
670
+ return '.';
671
+ }).join('');
672
+ const lastR = kinds.lastIndexOf('r');
673
+ const firstC = kinds.indexOf('c');
674
+ // The SDK's transform closes its reasoning part the moment the first content delta arrives, so
675
+ // interleaving would produce a different part sequence than the vendor's.
676
+ return lastR >= 0 && firstC >= 0 && lastR < firstC;
677
+ })),
678
+
679
+ done('deepseek.streaming.usage_tail_is_opt_in', 'streaming', 'The empty-choices usage tail chunk is emitted only when stream_options.include_usage is set', 'api', 'core', async () => {
680
+ const withUsage = await withStream({ ...CHAT(), stream: true, stream_options: { include_usage: true } }, (events) => {
681
+ const tail = events[events.length - 2]?.data as Body | undefined;
682
+ return !!tail && Array.isArray(tail.choices) && tail.choices.length === 0
683
+ && typeof (tail.usage as Body)?.prompt_cache_miss_tokens === 'number'
684
+ && (tail.usage as Body).total_tokens > 0;
685
+ });
686
+ if (!withUsage) return false;
687
+ // The negative half: without the option there must be NO usage chunk at all. A twin that always
688
+ // emitted it would look right to the SDK (which always asks for it) and be wrong for everyone else.
689
+ return withStream({ ...CHAT(), stream: true }, (events) => !events.some((e) => !e.done && (e.data as Body)?.usage !== undefined));
690
+ }),
691
+
692
+ done('deepseek.streaming.tool_call_deltas', 'streaming', 'Tool calls stream as indexed name-then-arguments deltas', 'api', 'common', () => withStream({ ...CHAT(), tools: [WEATHER_TOOL], stream: true }, (events) => {
693
+ const data = events.filter((e) => !e.done).map((e) => e.data as Body);
694
+ const deltas = data.flatMap((d) => ((d.choices as Body[])[0]?.delta?.tool_calls ?? []) as Body[]);
695
+ if (deltas.length < 2) return false;
696
+ const named = deltas.find((t) => t.function?.name === 'get_weather');
697
+ const args = deltas.filter((t) => typeof t.function?.arguments === 'string' && t.function.arguments !== '');
698
+ const finish = data.find((d) => (d.choices as Body[])[0]?.finish_reason != null);
699
+ return !!named && named.index === 0 && typeof named.id === 'string' && named.type === 'function'
700
+ && args.length > 0 && JSON.parse(args[args.length - 1]!.function.arguments).city === ''
701
+ && (finish!.choices as Body[])[0].finish_reason === 'tool_calls';
702
+ })),
703
+
704
+ todo('deepseek.streaming.stream_options_without_stream', 'streaming', "How DeepSeek answers `stream_options` on a NON-streaming request — the reference table says only 'set when stream: true', which is guidance rather than a documented refusal, so the twin accepts and ignores it rather than inventing a 422 (§9 round one, NIT 13)", 'api', 'niche'),
705
+
706
+ done('deepseek.streaming.never_degrades_to_a_unary_body', 'streaming', 'A `stream: true` request is never answered with a unary JSON body — not through a repeated-slash path spelling, and not when a caller supplies no sink', 'api', 'core', async () => {
707
+ // §9 ROUND TWO, SHOULD-FIX 1. Two independent holes, both fake successes of exactly the kind
708
+ // `deepseek.completions.refuses_unmodeled_streaming` forbids on the FIM endpoint:
709
+ // • the handler fell through to the unary builder whenever an sseSink was absent;
710
+ // • the SERVER's STREAMABLE lookup collapsed only a TRAILING slash while the router collapsed
711
+ // REPEATED ones, so `/beta//chat/completions` reached the streaming route, missed the
712
+ // streaming response path, and answered `application/json` to a client reading SSE.
713
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
714
+ const { createDeepSeekTwinServer } = await import('./deepseek-server.ts');
715
+ const server = await createDeepSeekTwinServer({ root });
716
+ try {
717
+ return await verifyBoundary('deepseek.streaming.never_degrades_to_a_unary_body', async () => {
718
+ // The handler half: a streaming request with no sink is refused, not downgraded.
719
+ const noSink = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify({ ...CHAT(), stream: true }), root, occurredAt: OCCURRED_AT });
720
+ if (noSink.status !== 422 || (body(noSink) as Body).object === 'chat.completion') return false;
721
+ // The server half: every spelling that REACHES the streaming route must stream.
722
+ const base = `http://127.0.0.1:${server.port}`;
723
+ for (const path of ['/chat/completions', '//chat/completions', '/beta/chat/completions', '/beta//chat/completions']) {
724
+ const res = await fetch(`${base}${path}`, {
725
+ method: 'POST',
726
+ headers: { 'content-type': 'application/json', authorization: 'Bearer sk-x' },
727
+ body: JSON.stringify({ ...CHAT(), stream: true }),
728
+ });
729
+ const ct = res.headers.get('content-type') ?? '';
730
+ const text = await res.text();
731
+ if (!ct.includes('text/event-stream')) return false;
732
+ if (!text.trimEnd().endsWith('data: [DONE]')) return false;
733
+ }
734
+ return true;
735
+ });
736
+ } finally { server.stop(); rmSync(root, { recursive: true, force: true }); }
737
+ }),
738
+
739
+ todo('deepseek.streaming.error_chunk', 'streaming', 'A mid-stream failure emits DeepSeek\'s error envelope as an SSE data event with the retryable metadata the SDK discriminates on', 'api', 'niche'),
740
+
741
+ // ══ tools / function calling ═════════════════════════════════════════════════════════════
742
+ done('deepseek.tools.function_calling', 'tools', 'Providing tools yields tool_calls with schema-shaped arguments and finish_reason tool_calls', 'api', 'core', () => withRoot(async (h) => {
743
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL] } });
744
+ if (!ok(r)) return false;
745
+ const calls = choice0(r).message.tool_calls as Body[];
746
+ if (!Array.isArray(calls) || calls.length !== 1) return false;
747
+ const args = JSON.parse(calls[0]!.function.arguments);
748
+ return calls[0]!.type === 'function' && calls[0]!.function.name === 'get_weather'
749
+ && typeof calls[0]!.id === 'string' && calls[0]!.id.length > 0
750
+ // Arguments must contain every declared property with a type-appropriate placeholder.
751
+ && args.city === '' && args.days === 0
752
+ && choice0(r).message.content === null && choice0(r).finish_reason === 'tool_calls';
753
+ })),
754
+
755
+ done('deepseek.tools.tool_choice', 'tools', "tool_choice none suppresses calls, a named function forces that one, and an unknown value is refused", 'api', 'common', () => withRoot(async (h) => {
756
+ const none = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL], tool_choice: 'none' } });
757
+ if (!ok(none) || none.body === undefined) return false;
758
+ if (choice0(none).message.tool_calls !== undefined || choice0(none).finish_reason !== 'stop') return false;
759
+ const named = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL, TIME_TOOL], tool_choice: { type: 'function', function: { name: 'get_time' } } } });
760
+ if (!ok(named)) return false;
761
+ const calls = choice0(named).message.tool_calls as Body[];
762
+ // Exactly the NAMED tool, not the first one — a handler ignoring tool_choice returns get_weather.
763
+ if (calls.length !== 1 || calls[0]!.function.name !== 'get_time') return false;
764
+ const bad = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL], tool_choice: 'any' } });
765
+ const badNamed = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL], tool_choice: { type: 'function' } } });
766
+ return bad.status === 422 && badNamed.status === 422;
767
+ })),
768
+
769
+ done('deepseek.tools.tool_result_turn', 'tools', 'A role:tool result turn is accepted and folded into the prompt', 'api', 'common', () => withRoot(async (h) => {
770
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [
771
+ { role: 'user', content: 'weather in Lisbon?' },
772
+ { role: 'assistant', content: null, reasoning_content: '', tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'get_weather', arguments: '{"city":"Lisbon"}' } }] },
773
+ { role: 'tool', tool_call_id: 'call_1', content: '{"temp":21}' },
774
+ ] } });
775
+ if (!ok(r)) return false;
776
+ // The tool turn is real payload: it must be counted, so a handler dropping it reports fewer tokens.
777
+ const withoutTool = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'weather in Lisbon?' }] } });
778
+ return body(r).usage.prompt_tokens > body(withoutTool).usage.prompt_tokens && choice0(r).finish_reason === 'stop';
779
+ })),
780
+
781
+ done('deepseek.tools.max_128', 'tools', 'More than 128 function tools is refused', 'api', 'niche', () => withRoot(async (h) => {
782
+ const tool = (n: number) => ({ type: 'function', function: { name: `f${n}`, parameters: { type: 'object', properties: {} } } });
783
+ const at = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: Array.from({ length: 128 }, (_, i) => tool(i)) } });
784
+ const over = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: Array.from({ length: 129 }, (_, i) => tool(i)) } });
785
+ const malformed = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [{ type: 'retrieval' }] } });
786
+ return ok(at) && over.status === 422 && malformed.status === 422;
787
+ })),
788
+
789
+ done('deepseek.tools.strict_is_beta_only_and_all_or_nothing', 'tools', 'strict:true requires the /beta base URL, and inside it every function tool must be strict', 'api', 'common', () => withRoot(async (h) => {
790
+ const strict = { type: 'function', function: { name: 'a', strict: true, parameters: { type: 'object', properties: {} } } };
791
+ const loose = { type: 'function', function: { name: 'b', parameters: { type: 'object', properties: {} } } };
792
+ const offBeta = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [strict] } });
793
+ const mixed = await h({ m: 'POST', p: BETA_CHAT, b: { ...CHAT(), tools: [strict, loose] } });
794
+ const allStrict = await h({ m: 'POST', p: BETA_CHAT, b: { ...CHAT(), tools: [strict] } });
795
+ // The non-strict path must still work on the standard base URL, or the check has proved nothing
796
+ // about `strict` in particular.
797
+ const plain = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [loose] } });
798
+ return offBeta.status === 422 && msg(offBeta).includes('beta base URL')
799
+ && mixed.status === 422 && ok(allStrict) && ok(plain);
800
+ })),
801
+
802
+ todo('deepseek.tools.strict_schema_enforcement', 'tools', 'Under beta strict mode, generated arguments are guaranteed to validate against the declared JSON schema', 'api', 'niche'),
803
+ todo('deepseek.beta.prefix_with_tools', 'beta', 'Prefix completion combined with tools in one beta request', 'api', 'niche'),
804
+ // CORE for the same reason: an agent integration meets parallel tool calls immediately.
805
+ todo('deepseek.tools.parallel_multi_tool', 'tools', 'A turn returning several tool_calls at once, as the vendor does for independent tools', 'api', 'core'),
806
+
807
+ // ══ structured outputs ═══════════════════════════════════════════════════════════════════
808
+ done('deepseek.structured_outputs.json_object', 'structured_outputs', "On /chat/completions, response_format accepts text and json_object only — json_schema is REFUSED", 'api', 'core', () => withRoot(async (h) => {
809
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'json_object' } }) });
810
+ if (!ok(r)) return false;
811
+ const parsed = JSON.parse(text0(r));
812
+ if (parsed._twin_stub !== true || parsed.model !== MODEL) return false;
813
+ // THE DIVERGENCE, and it is ENDPOINT-SCOPED — stating it as "DeepSeek has no json_schema" would
814
+ // be wrong: its Responses API's `text.format` does take one. What has no json_schema is
815
+ // /chat/completions, and `@ai-sdk/deepseek` confirms it from the other side by INJECTING the
816
+ // schema into a system message rather than sending `response_format.json_schema`
817
+ // (src/chat/convert-to-deepseek-chat-messages.ts). A twin that accepted it here would be
818
+ // serving OpenAI's surface on an endpoint that does not have it.
819
+ const schema = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'json_schema', json_schema: { name: 'x', schema: { type: 'object' } } } }) });
820
+ const bogus = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'xml' } }) });
821
+ const text = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'text' } }) });
822
+ return schema.status === 422 && bogus.status === 422 && ok(text) && !text0(text).startsWith('{');
823
+ })),
824
+
825
+ // ── Responses API (real surface, wholly unmodeled) ───────────────────────────────────────
826
+ // `POST /responses` is a first-class entry in DeepSeek's own API-reference nav alongside Chat
827
+ // Completions, FIM, Lists Models, Get User Balance and Files. It is a DIFFERENT protocol — an
828
+ // `input` item array rather than `messages`, an `output` item array rather than `choices`,
829
+ // semantic SSE events rather than delta chunks — so it is enumerated as its own area rather than
830
+ // folded into chat. The twin answers it like any other unmodeled operation, never a fake success.
831
+ todo('deepseek.responses.create', 'responses', 'POST /responses — the Responses-protocol envelope (input items in, output items out, its own usage block)', 'api', 'common'),
832
+ todo('deepseek.responses.instructions', 'responses', "The Responses API's top-level `instructions` system-level input", 'api', 'niche'),
833
+ todo('deepseek.responses.reasoning_items', 'responses', 'Chain of thought returned as `reasoning` OUTPUT ITEMS rather than a reasoning_content field', 'api', 'common'),
834
+ todo('deepseek.responses.reasoning_effort_set', 'responses', "The Responses API's WIDER effort set (none|minimal|low|medium|high|xhigh|max) — /chat/completions takes only low|high|max", 'api', 'niche'),
835
+ todo('deepseek.responses.text_format_json_schema', 'responses', '`text.format` accepting json_schema — the structured-output shape /chat/completions does NOT have', 'api', 'common'),
836
+ todo('deepseek.responses.streaming', 'responses', 'Semantic SSE with `event` + `sequence_number`, terminating in response.completed / .incomplete / .failed', 'api', 'common'),
837
+ todo('deepseek.responses.function_tools', 'responses', 'Function tools and tool_choice on the Responses protocol, returning function_call output items', 'api', 'common'),
838
+ todo('deepseek.responses.web_search_tool', 'responses', 'The built-in server-side `web_search` tool and its web_search_call output items', 'api', 'niche'),
839
+ todo('deepseek.responses.stateless', 'responses', 'The documented statelessness: responses are not stored, so a client resubmits history rather than passing a previous-response id', 'api', 'niche'),
840
+
841
+ todo('deepseek.structured_outputs.system_prompt_contract', 'structured_outputs', 'JSON mode\'s documented requirement that the prompt itself instruct the model to emit JSON, and its empty-content failure mode', 'api', 'common'),
842
+
843
+ // ══ beta: chat prefix completion ═════════════════════════════════════════════════════════
844
+ done('deepseek.beta.prefix_completion', 'beta', 'Under /beta, an assistant message with prefix:true is CONTINUED rather than answered', 'api', 'common', () => withRoot(async (h) => {
845
+ const r = await h({ m: 'POST', p: BETA_CHAT, b: { model: MODEL, messages: [
846
+ { role: 'user', content: 'describe the sky' },
847
+ { role: 'assistant', content: 'The sky is', prefix: true },
848
+ ] } });
849
+ if (!ok(r)) return false;
850
+ // Continuation, not a fresh turn: the caller's own text must lead the content.
851
+ return text0(r).startsWith('The sky is') && text0(r).length > 'The sky is'.length && text0(r).includes('[twin-stub:');
852
+ })),
853
+
854
+ done('deepseek.beta.prefix_placement_rules', 'beta', 'prefix:true is refused off /beta, on a non-assistant role, and when it is not the final message', 'api', 'common', () => withRoot(async (h) => {
855
+ const offBeta = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'x' }, { role: 'assistant', content: 'y', prefix: true }] } });
856
+ const notFinal = await h({ m: 'POST', p: BETA_CHAT, b: { model: MODEL, messages: [{ role: 'assistant', content: 'y', prefix: true }, { role: 'user', content: 'x' }] } });
857
+ const wrongRole = await h({ m: 'POST', p: BETA_CHAT, b: { model: MODEL, messages: [{ role: 'user', content: 'x', prefix: true }] } });
858
+ return offBeta.status === 422 && msg(offBeta).includes('beta base URL')
859
+ && notFinal.status === 422 && msg(notFinal).includes('final message')
860
+ && wrongRole.status === 422 && msg(wrongRole).includes('assistant message');
861
+ })),
862
+
863
+ done('deepseek.beta.standard_surface_unchanged', 'beta', 'The /beta base URL serves the SAME chat surface — it only unlocks prefix and strict tools', 'api', 'common', () => withRoot(async (h) => {
864
+ const std = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
865
+ const beta = await h({ m: 'POST', p: BETA_CHAT, b: CHAT() });
866
+ if (!ok(std) || !ok(beta)) return false;
867
+ // Same request, same answer: /beta is a feature-flag prefix, not a second API. (Usage differs
868
+ // because the first call seeds the cache ledger, so only the content is compared.)
869
+ return text0(std) === text0(beta) && body(beta).object === 'chat.completion'
870
+ // …and /beta still refuses what the standard surface refuses.
871
+ && (await h({ m: 'POST', p: BETA_CHAT, b: CHAT({ n: 2 }) })).status === 422;
872
+ })),
873
+
874
+ // ══ files (images only) ══════════════════════════════════════════════════════════════════
875
+ done('deepseek.files.upload', 'files', 'POST /files stores an image and returns the vendor file object', 'api', 'common', () => withRoot(async (h) => {
876
+ const r = await h({ m: 'POST', p: FILES, b: UPLOAD() });
877
+ if (!ok(r)) return false;
878
+ const b = body(r);
879
+ return b.object === 'file' && b.id === FILE_1 && b.filename === 'shot.png' && b.purpose === 'user_data'
880
+ && b.bytes === 3 && typeof b.created_at === 'number' && b.expires_at === undefined;
881
+ })),
882
+
883
+ done('deepseek.files.image_only', 'files', "The Files API accepts JPEG/PNG/GIF/WebP only, and purpose must be 'user_data'", 'api', 'common', () => withRoot(async (h) => {
884
+ // DeepSeek's Files API is for vision inputs. The OpenAI habit — uploading a .jsonl for a Batch
885
+ // API — has no counterpart here, and accepting it would fabricate a whole product.
886
+ const jsonl = await h({ m: 'POST', p: FILES, b: { purpose: 'user_data', filename: 'in.jsonl', media_type: 'application/jsonl', content: 'x', bytes: 1 } });
887
+ const batch = await h({ m: 'POST', p: FILES, b: UPLOAD({ purpose: 'batch' }) });
888
+ const pdf = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'doc.pdf', media_type: 'application/pdf' }) });
889
+ if (jsonl.status !== 422 || batch.status !== 422 || pdf.status !== 422) return false;
890
+ // Every accepted format must actually be accepted, or the allowlist is wrong in the other
891
+ // direction.
892
+ for (const [filename, media] of [['a.jpg', 'image/jpeg'], ['a.jpeg', 'image/jpeg'], ['a.png', 'image/png'], ['a.gif', 'image/gif'], ['a.webp', 'image/webp']]) {
893
+ if (!ok(await h({ m: 'POST', p: FILES, b: UPLOAD({ filename, media_type: media }) }))) return false;
894
+ }
895
+ return true;
896
+ })),
897
+
898
+ done('deepseek.files.expires_after', 'files', 'expires_after sets expires_at; the anchor is closed and the range is 3600..2592000', 'api', 'niche', () => withRoot(async (h) => {
899
+ const r = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 3600 }) });
900
+ if (!ok(r)) return false;
901
+ if (body(r).expires_at !== body(r).created_at + 3600) return false;
902
+ const tooShort = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 3599 }) });
903
+ const tooLong = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 2_592_001 }) });
904
+ const badAnchor = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'now', 'expires_after[seconds]': 7200 }) });
905
+ const atMax = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 2_592_000 }) });
906
+ return tooShort.status === 422 && tooLong.status === 422 && badAnchor.status === 422 && ok(atMax);
907
+ })),
908
+
909
+ done('deepseek.files.limits', 'files', 'A 64 MiB size cap and a 512-character filename cap are enforced', 'api', 'niche', () => withRoot(async (h) => {
910
+ const tooBig = await h({ m: 'POST', p: FILES, b: UPLOAD({ bytes: 64 * 1024 * 1024 + 1 }) });
911
+ const atCap = await h({ m: 'POST', p: FILES, b: UPLOAD({ bytes: 64 * 1024 * 1024 }) });
912
+ const longName = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: `${'a'.repeat(510)}.png` }) });
913
+ return tooBig.status === 422 && ok(atCap) && longName.status === 422;
914
+ })),
915
+
916
+ done('deepseek.files.list_and_retrieve', 'files', 'GET /files lists with after/limit/order, and GET /files/{id} retrieves one', 'api', 'common', () => withRoot(async (h) => {
917
+ await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'one.png' }) });
918
+ await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'two.png' }) });
919
+ const all = await h({ m: 'GET', p: FILES });
920
+ if (!ok(all) || (body(all).data as Body[]).length !== 2 || body(all).has_more !== false) return false;
921
+ const one = await h({ m: 'GET', p: `${FILES}/${FILE_1}` });
922
+ if (!ok(one) || body(one).filename !== 'one.png') return false;
923
+ // The documented list envelope carries the cursor bookends, not just `has_more`.
924
+ if (body(all).first_id !== FILE_1 || body(all).last_id !== FILE_2) return false;
925
+ const desc = await h({ m: 'GET', p: `${FILES}?order=desc` });
926
+ if ((body(desc).data as Body[])[0].id !== FILE_2) return false;
927
+ if (body(desc).first_id !== FILE_2 || body(desc).last_id !== FILE_1) return false;
928
+ const after = await h({ m: 'GET', p: `${FILES}?after=${FILE_1}` });
929
+ if ((body(after).data as Body[]).length !== 1 || (body(after).data as Body[])[0].id !== FILE_2) return false;
930
+ const capped = await h({ m: 'GET', p: `${FILES}?limit=1` });
931
+ if ((body(capped).data as Body[]).length !== 1 || body(capped).has_more !== true) return false;
932
+ const missing = await h({ m: 'GET', p: `${FILES}/file-api-nope` });
933
+ const badLimit = await h({ m: 'GET', p: `${FILES}?limit=1001` });
934
+ const badOrder = await h({ m: 'GET', p: `${FILES}?order=sideways` });
935
+ const badPurpose = await h({ m: 'GET', p: `${FILES}?purpose=batch` });
936
+ return missing.status === 404 && badLimit.status === 422 && badOrder.status === 422 && badPurpose.status === 422;
937
+ })),
938
+
939
+ done('deepseek.files.delete_then_recreate_ratchets_ids', 'files', 'DIRTY STATE: after delete→recreate, the new file gets a FRESH id — a deleted id is never handed out twice', 'api', 'common', () => withRoot(async (h) => {
940
+ // A fresh-root-only manifest structurally cannot catch this class. Three sibling packs' §9
941
+ // reviews each found an id-reuse or tombstone bug here, so it is verified deliberately over
942
+ // prior state rather than from nothing.
943
+ const first = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'first.png' }) });
944
+ if (!ok(first) || body(first).id !== FILE_1) return false;
945
+ const del = await h({ m: 'DELETE', p: `${FILES}/${FILE_1}` });
946
+ if (!ok(del) || body(del).deleted !== true) return false;
947
+ const gone = await h({ m: 'GET', p: `${FILES}/${FILE_1}` });
948
+ const doubleDelete = await h({ m: 'DELETE', p: `${FILES}/${FILE_1}` });
949
+ if (gone.status !== 404 || doubleDelete.status !== 404) return false;
950
+ const second = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'second.png' }) });
951
+ // The counter must RATCHET across the tombstone: reusing FILE_1 would resurrect the deleted
952
+ // file's identity and silently clobber it.
953
+ if (!ok(second) || body(second).id !== FILE_2) return false;
954
+ const list = await h({ m: 'GET', p: FILES });
955
+ const ids = (body(list).data as Body[]).map((f) => f.id);
956
+ return ids.length === 1 && ids[0] === FILE_2;
957
+ })),
958
+
959
+ todo('deepseek.files.expiry_removes_the_file', 'files', 'A file past its `expires_at` stops being listed and retrievable — the twin stores the timestamp faithfully but nothing ever expires (§9 round one, SHOULD-FIX 5)', 'api', 'niche'),
960
+ todo('deepseek.files.storage_quota', 'files', 'The account-level 25 GiB / 10,000-file storage limits', 'api', 'niche'),
961
+ todo('deepseek.files.reference_in_chat', 'files', 'Passing an uploaded file_id as a chat content part to deepseek-v4-flash-vision-exp', 'api', 'common'),
962
+ todo('deepseek.files.binary_content', 'files', 'Files: persist the uploaded bytes and serve them back (upload metadata and byte count are recorded today)', 'api', 'niche'),
963
+
964
+ // ══ vision ═══════════════════════════════════════════════════════════════════════════════
965
+ todo('deepseek.vision.image_url_part', 'vision', 'image_url content parts (with the low/high/original/auto detail option) on the vision model', 'api', 'common'),
966
+ todo('deepseek.vision.file_data_part', 'vision', "The `file` content part carrying inline `file_data`, which @ai-sdk/deepseek's fileData option produces", 'api', 'niche'),
967
+ todo('deepseek.vision.media_type_closed_set', 'vision', 'Rejecting an image media type outside JPEG/PNG/GIF/WebP, and a URL beyond 8192 characters', 'api', 'niche'),
968
+ todo('deepseek.vision.non_vision_model_refuses_images', 'vision', 'A non-vision model refuses an image part the way the vendor does', 'api', 'common'),
969
+
970
+ // ══ Anthropic-compatible surface ═════════════════════════════════════════════════════════
971
+ todo('deepseek.anthropic.messages', 'anthropic', 'POST /anthropic/v1/messages — the Anthropic-shaped surface DeepSeek serves alongside the OpenAI-shaped one', 'api', 'common'),
972
+ todo('deepseek.anthropic.model_mapping', 'anthropic', 'Claude model names mapped to DeepSeek models (Opus→v4-pro, Haiku/Sonnet→v4-flash, unmapped→v4-flash)', 'api', 'niche'),
973
+ todo('deepseek.anthropic.x_api_key_auth', 'anthropic', 'The Anthropic surface authenticates with x-api-key, not a bearer Authorization header', 'api', 'niche'),
974
+ todo('deepseek.anthropic.streaming', 'anthropic', "Server-sent events on the Anthropic-compatible surface (Anthropic's own event grammar, not the OpenAI chunk shape)", 'api', 'niche'),
975
+ todo('deepseek.anthropic.thinking_budget_ignored', 'anthropic', "The Anthropic surface accepts `thinking` but IGNORES budget_tokens — a documented accept-and-ignore, so it must not be an error", 'api', 'niche'),
976
+ todo('deepseek.anthropic.unsupported_content_types', 'anthropic', 'Refusing the Anthropic content types DeepSeek does not support (document, search result, MCP tools, container uploads)', 'api', 'niche'),
977
+
978
+ // ══ errors ═══════════════════════════════════════════════════════════════════════════════
979
+ done('deepseek.errors.status_split_400_vs_422', 'errors', "DeepSeek's 400 is BODY FORMAT and its 422 is INVALID PARAMETERS — the split OpenAI does not have", 'api', 'core', () => withRoot(async (h) => {
980
+ // The single most consequential divergence in this pack. An OpenAI-copied twin answers 400 for
981
+ // both, which passes a status>=400 check and is wrong on every parameter refusal.
982
+ // 400 is ONLY for a body this API cannot parse as a JSON object: absent, malformed, or the
983
+ // wrong JSON kind.
984
+ const absent = await h({ m: 'POST', p: CHAT_PATH, b: undefined });
985
+ const raw = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: '{not json' });
986
+ const notObject = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: '[1,2]' });
987
+ if (raw.status !== 400 || !String((raw.body as Body).error.message).includes('Invalid request body format')) return false;
988
+ if (notObject.status !== 400 || absent.status !== 400) return false;
989
+ // …and EVERY complaint about parsed content is 422, including the ones an OpenAI-shaped twin
990
+ // would 400: a missing required parameter is the same class as a bad one and must not be
991
+ // reported two different ways.
992
+ for (const b of [CHAT({ model: 'nope' }), { messages: [{ role: 'user', content: 'x' }] }, { model: MODEL }, { model: MODEL, messages: [null] }]) {
993
+ const r = await h({ m: 'POST', p: CHAT_PATH, b });
994
+ if (r.status !== 422 || !msg(r).includes('invalid parameters')) return false;
995
+ }
996
+ return true;
997
+ })),
998
+
999
+ done('deepseek.errors.envelope', 'errors', "Refusals carry DeepSeek's { error: { message, ... } } envelope and no key its decoder does not declare", 'api', 'core', () => withRoot(async (h) => {
1000
+ const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ model: 'nope' }) });
1001
+ const e = body(r).error as Body;
1002
+ // The key set is a LITERAL here, from @ai-sdk/deepseek's `deepSeekErrorSchema` — asserting
1003
+ // against the handler's own constant would be a tautology that could not catch it drifting.
1004
+ const declared = new Set(['message', 'type', 'param', 'code']);
1005
+ if (r.status !== 422 || !e || typeof e.message !== 'string' || !e.message.length) return false;
1006
+ if (!Object.keys(e).every((k) => declared.has(k))) return false;
1007
+ // …and the envelope really is DECODABLE by the vendor's own schema shape on every status the
1008
+ // twin emits, not just this one. §9 round two: a subset check alone cannot fail for anything the
1009
+ // twin currently produces, so it is paired with a sweep that asserts `message` is a non-empty
1010
+ // string and no undeclared key appears anywhere.
1011
+ const others = [
1012
+ await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, headers: {} }),
1013
+ await h({ m: 'GET', p: '/nope' }),
1014
+ await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: '{not json' }),
1015
+ await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), readOnly: true }),
1016
+ ];
1017
+ for (const o of others) {
1018
+ const oe = (o.body as Body)?.error as Body;
1019
+ if (o.status < 400 || !oe || typeof oe.message !== 'string' || !oe.message.length) return false;
1020
+ if (!Object.keys(oe).every((k) => declared.has(k))) return false;
1021
+ }
1022
+ return true;
1023
+ })),
1024
+
1025
+ done('deepseek.errors.envelope_omits_an_unsourced_type', 'errors', "A 422 and a 402 carry NO `type` — nothing first-party names one for those statuses, and the string a 422 used to emit resolves to 400 in the SDK's own discriminator table", 'api', 'core', () => withRoot(async (h, root) => {
1026
+ // §9 ROUND ONE, SHOULD-FIX 3, and the pin the fix itself lacked. `@ai-sdk/deepseek`'s
1027
+ // `getDeepSeekStreamErrorMetadata` has NO 422 case at all, and maps `invalid_request_error` to
1028
+ // `statusCode: 400` — so emitting that string on a 422 made a real client resolve the twin's
1029
+ // own refusal to the wrong status. The rule the twin states is: use a type only where something
1030
+ // first-party names one for that status, otherwise omit the field.
1031
+ // SWEEP every 422 the twin can produce, not one sample. §9 ROUND TWO, BLOCKER 2: the fix was
1032
+ // applied to the `invalidParameters` helper only, so the FIM streaming refusal — a 422 built
1033
+ // inline 700 lines away — kept emitting `invalid_request_error`, and this cell passed because it
1034
+ // happened to probe a different one. A universal claim needs a universal check.
1035
+ const fourTwentyTwos: Array<{ m: string; p: string; b?: unknown }> = [
1036
+ { m: 'POST', p: CHAT_PATH, b: CHAT({ model: 'nope' }) },
1037
+ { m: 'POST', p: CHAT_PATH, b: CHAT({ n: 2 }) },
1038
+ { m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'json_schema' } }) },
1039
+ { m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', stream: true } },
1040
+ { m: 'POST', p: FIM_PATH, b: { model: MODEL, prompt: 'x' } },
1041
+ { m: 'POST', p: FILES, b: UPLOAD({ purpose: 'batch' }) },
1042
+ { m: 'GET', p: `${FILES}?limit=5000` },
1043
+ ];
1044
+ for (const req of fourTwentyTwos) {
1045
+ const r = await h(req);
1046
+ if (r.status !== 422 || 'type' in (body(r).error as Body)) return false;
1047
+ }
1048
+ // The 405 is the OTHER status nothing first-party names a type for, and it is built in a third
1049
+ // place again — the revert matrix showed `errors.envelope` cannot catch it (a `type` is a
1050
+ // DECLARED key, so a subset check passes), so the universal has to sweep it here.
1051
+ const ro = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), readOnly: true });
1052
+ if (ro.status !== 405 || 'type' in (body(ro).error as Body)) return false;
1053
+ // 402 likewise carries none — and this DRIVES that path rather than asserting it in a comment.
1054
+ await applyTwinWrite('deepseek', {
1055
+ operation: 'balance.update', subjectType: 'balance', subjectId: 'account',
1056
+ fields: { is_available: false, balance_infos: [] }, occurredAt: OCCURRED_AT, actor: { kind: 'agent' },
1057
+ }, root);
1058
+ const drained = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
1059
+ if (drained.status !== 402 || 'type' in (body(drained).error as Body)) return false;
1060
+ // …but the statuses the SDK's table DOES name keep their discriminator, or the rule would just
1061
+ // be "never emit a type", which is a different (and equally unsourced) claim.
1062
+ const unauth = await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, headers: {} });
1063
+ const notFound = await h({ m: 'GET', p: '/nope' });
1064
+ return unauth.status === 401 && (body(unauth).error as Body).type === 'authentication_error'
1065
+ && refused(notFound) && (body(notFound).error as Body).type === 'not_found_error';
1066
+ })),
1067
+
1068
+ done('deepseek.errors.unmodeled_ops_fail', 'errors', 'Unmodeled operations and OpenAI-only endpoints fail rather than fake a success', 'api', 'core', () => withRoot(async (h) => {
1069
+ for (const [m, p] of [
1070
+ ['POST', '/embeddings'], ['POST', '/images/generations'], ['POST', '/audio/transcriptions'],
1071
+ ['POST', '/moderations'], ['POST', '/batches'], ['GET', '/models/deepseek-v4-flash'],
1072
+ ['POST', '/fine_tuning/jobs'],
1073
+ ]) {
1074
+ const r = await h({ m: m!, p: p!, b: {} });
1075
+ // Refused as a CLIENT error with the vendor envelope; the exact code is the twin's unverified
1076
+ // choice (see deepseek.errors.unknown_route_envelope), so it is not asserted (NIT 11).
1077
+ if (!refused(r)) return false;
1078
+ }
1079
+ // REAL vendor surface this twin does not model yet must fail too — and must not be confused
1080
+ // with the OpenAI-only endpoints above. The Anthropic surface says so by name; `/responses` is
1081
+ // a first-class entry in DeepSeek's own reference nav and falls through to the router's
1082
+ // not-found rather than being answered as if it were /chat/completions.
1083
+ const anthropic = await h({ m: 'POST', p: '/anthropic/v1/messages', b: {} });
1084
+ if (!refused(anthropic) || !msg(anthropic).includes('Anthropic-compatible')) return false;
1085
+ const responses = await h({ m: 'POST', p: '/responses', b: { model: MODEL, input: 'hi' } });
1086
+ const betaResponses = await h({ m: 'POST', p: '/beta/responses', b: { model: MODEL, input: 'hi' } });
1087
+ return refused(responses) && refused(betaResponses);
1088
+ })),
1089
+
1090
+ done('deepseek.errors.read_only_405', 'errors', 'A read-only twin refuses every write with 405 while reads keep working', 'api', 'common', async () => {
1091
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
1092
+ try {
1093
+ return await verifyBoundary('deepseek.errors.read_only_405', async () => {
1094
+ const chat = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), root, readOnly: true });
1095
+ const upload = await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD()), root, readOnly: true });
1096
+ const models = await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, root, readOnly: true });
1097
+ // And nothing was written: the projection must still be empty.
1098
+ const wrote = projectResources('deepseek', root).length > 0;
1099
+ return chat.status === 405 && upload.status === 405 && ok(models) && !wrote;
1100
+ });
1101
+ } finally { rmSync(root, { recursive: true, force: true }); }
1102
+ }),
1103
+
1104
+ todo('deepseek.errors.rate_limit_429', 'errors', "Answering DeepSeek's documented 429 under a real rate condition. The twin serves the faithful envelope, but only behind a twin-only `x-twin-force-rate-limit` header — scaffolding, which §6 keeps OUT of the coverage claim, so this is a todo rather than a done proven by its own trigger (§9 round one)", 'api', 'common'),
1105
+
1106
+ // (Scripted 500/503 failures used to be claimed here as a `done`. They are SCENARIO scaffolding,
1107
+ // which ADDING_A_TWIN.md §6 says to keep out of the capability manifest because counting it pads
1108
+ // the denominator — the behaviour is gated by deepseek-scenario.test.ts instead. §9 round one.)
1109
+ todo('deepseek.errors.server_5xx', 'errors', "Answering DeepSeek's documented 500 'Our server encounters an issue' and 503 'The server is overloaded' under real conditions rather than only when a scenario handler scripts them", 'api', 'niche'),
1110
+
1111
+ todo('deepseek.errors.unknown_route_envelope', 'errors', "The exact status/body a real DeepSeek gateway returns for an unrouted path — its published error table (400/401/402/422/429/500/503) has no 404 entry, so the twin's 404 shape is unverified against a first-party source", 'api', 'niche'),
1112
+ todo('deepseek.errors.retryable_metadata', 'errors', 'Error `code`/`type` discriminators the SDK maps to retryability (insufficient_quota, overloaded_error, timeout, …)', 'api', 'niche'),
1113
+
1114
+ // ══ auth ═════════════════════════════════════════════════════════════════════════════════
1115
+ done('deepseek.auth.bearer_required', 'auth', 'A request carrying an auth surface needs a bearer credential; a missing or sentinel-invalid one is 401', 'api', 'core', () => withRootH(async (h) => {
1116
+ const missing = await h({ m: 'GET', p: MODELS, headers: {} });
1117
+ const invalid = await h({ m: 'GET', p: MODELS, headers: { authorization: 'Bearer sk_invalid' } });
1118
+ const notBearer = await h({ m: 'GET', p: MODELS, headers: { authorization: 'Basic abc' } });
1119
+ const good = await h({ m: 'GET', p: MODELS, headers: { authorization: 'Bearer sk-anything' } });
1120
+ return missing.status === 401 && invalid.status === 401 && notBearer.status === 401 && ok(good)
1121
+ && msg(missing).includes('Authentication fails');
1122
+ })),
1123
+
1124
+ todo('deepseek.auth.gates_before_routing', 'auth', 'Whether DeepSeek authenticates BEFORE routing (401 rather than 404 on an unknown path with no credential). The twin does, but nothing first-party documents the ordering, so it is a twin design choice and not a proven vendor fact (§9 round one, NIT 12)', 'api', 'common'),
1125
+
1126
+ todo('deepseek.auth.x_api_key', 'auth', 'x-api-key authentication on the Anthropic-compatible surface', 'api', 'niche'),
1127
+
1128
+ // ══ rate limits / budget ═════════════════════════════════════════════════════════════════
1129
+ done('deepseek.rate_limits.budget_refuses_past_ceiling', 'rate_limits', 'The client-side budget THROWS instead of calling past its ceiling, with the vendor call count unchanged', 'connector', 'core', async () => {
1130
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
1131
+ try {
1132
+ return await verifyBoundary('deepseek.rate_limits.budget_refuses_past_ceiling', async () => {
1133
+ const { DeepSeekBudget } = await import('./deepseek-budget.ts');
1134
+ let calls = 0;
1135
+ const fetchImpl = (async () => { calls++; return new Response('{}', { status: 200, headers: { 'content-type': 'application/json' } }); }) as unknown as typeof fetch;
1136
+ const ledger = new DeepSeekBudget({ path: join(root, 'ledger.json') });
1137
+ const execute = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { fetchImpl, budget: ledger });
1138
+ // /models costs the default weight 2 against a ceiling of 60 → exactly 30 fit.
1139
+ for (let i = 0; i < 30; i++) await execute('GET', '/models');
1140
+ if (calls !== 30) return false;
1141
+ let threw = false;
1142
+ try { await execute('GET', '/models'); } catch { threw = true; }
1143
+ // "It threw" is not the proof — the unchanged COUNT is what shows nothing reached DeepSeek.
1144
+ return threw && calls === 30;
1145
+ });
1146
+ } finally { rmSync(root, { recursive: true, force: true }); }
1147
+ }),
1148
+
1149
+ done('deepseek.rate_limits.inference_costs_more', 'rate_limits', 'Token-billed inference endpoints are priced above the default weight, including under the /beta prefix', 'connector', 'common', async () => {
1150
+ const { deepseekCallWeight } = await import('./deepseek-budget.ts');
1151
+ // The weights are asserted as LITERALS: importing the weight table and comparing it with itself
1152
+ // would drift together and could never catch an inference call being priced as a read.
1153
+ return deepseekCallWeight('POST', '/chat/completions') === 6
1154
+ && deepseekCallWeight('POST', '/beta/chat/completions') === 6
1155
+ && deepseekCallWeight('POST', '/beta/completions') === 6
1156
+ && deepseekCallWeight('GET', '/models') === 2
1157
+ && deepseekCallWeight('GET', '/user/balance') === 2
1158
+ // …and the anchored rules survive the two evasions that would otherwise price a write as a read.
1159
+ && deepseekCallWeight('post', '/chat/completions') === 6
1160
+ && deepseekCallWeight('POST', '//chat/completions') === 6
1161
+ && deepseekCallWeight('POST', '/chat/completions/') === 6;
1162
+ }),
1163
+
1164
+ done('deepseek.rate_limits.ledger_survives_restart', 'rate_limits', 'The spend ledger is persistent: a fresh executor over the same ledger gets no fresh allowance', 'connector', 'common', async () => {
1165
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
1166
+ try {
1167
+ return await verifyBoundary('deepseek.rate_limits.ledger_survives_restart', async () => {
1168
+ const path = join(root, 'ledger.json');
1169
+ let calls = 0;
1170
+ const fetchImpl = (async () => { calls++; return new Response('{}', { status: 200, headers: { 'content-type': 'application/json' } }); }) as unknown as typeof fetch;
1171
+ const first = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { budgetOptions: { path }, fetchImpl });
1172
+ for (let i = 0; i < 30; i++) await first('GET', '/models');
1173
+ // A brand-new budget object over the SAME ledger file — the "restarted process" case.
1174
+ const second = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { budgetOptions: { path }, fetchImpl });
1175
+ let threw = false;
1176
+ try { await second('GET', '/models'); } catch { threw = true; }
1177
+ return threw && calls === 30;
1178
+ });
1179
+ } finally { rmSync(root, { recursive: true, force: true }); }
1180
+ }),
1181
+
1182
+ done('deepseek.rate_limits.executor_path_allowlist', 'rate_limits', 'The live executor refuses an unmodeled path BEFORE spending budget or touching the network', 'connector', 'common', async () => {
1183
+ const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
1184
+ try {
1185
+ return await verifyBoundary('deepseek.rate_limits.executor_path_allowlist', async () => {
1186
+ const { DeepSeekBudget } = await import('./deepseek-budget.ts');
1187
+ const counter = { n: 0 };
1188
+ const fetchImpl = (async () => { counter.n++; return new Response('{}', { status: 200, headers: { 'content-type': 'application/json' } }); }) as unknown as typeof fetch;
1189
+ const ledger = new DeepSeekBudget({ path: join(root, 'ledger.json') });
1190
+ const execute = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { fetchImpl, budget: ledger });
1191
+ // The OpenAI-shaped path is the one a careless caller reaches for, and it would evade both
1192
+ // the budget's anchored rules and DeepSeek's own routing.
1193
+ for (const p of ['/v1/models', '/chat/completions', '/anthropic/v1/messages', '/../etc']) {
1194
+ let threw = false;
1195
+ try { await execute('GET', p); } catch { threw = true; }
1196
+ if (!threw) return false;
1197
+ }
1198
+ if (counter.n !== 0) return false;
1199
+ // The budget must be UNSPENT: the refusal happens BEFORE `checkBudget`, so a modeled call
1200
+ // still has the full allowance afterwards.
1201
+ if (ledger.snapshot().spend !== 0) return false;
1202
+ await execute('GET', '/models');
1203
+ return counter.n > 0;
1204
+ });
1205
+ } finally { rmSync(root, { recursive: true, force: true }); }
1206
+ }),
1207
+
1208
+
1209
+ // ══ usage accounting ═════════════════════════════════════════════════════════════════════
1210
+ done('deepseek.usage.token_counts', 'usage', 'usage reports deterministic prompt/completion/total counts that respond to the actual payload', 'api', 'core', () => withRoot(async (h) => {
1211
+ const short = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi' }] } });
1212
+ const long = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi '.repeat(200) }] } });
1213
+ if (!ok(short) || !ok(long)) return false;
1214
+ const su = body(short).usage as Body;
1215
+ const lu = body(long).usage as Body;
1216
+ return su.prompt_tokens > 0 && lu.prompt_tokens > su.prompt_tokens
1217
+ && su.total_tokens === su.prompt_tokens + su.completion_tokens
1218
+ && lu.total_tokens === lu.prompt_tokens + lu.completion_tokens;
1219
+ })),
1220
+
1221
+ done('deepseek.usage.serve_path_determinism', 'usage', 'The same request against two independent roots yields a byte-identical response', 'api', 'core', async () => {
1222
+ const a = mkdtempSync(join(tmpdir(), 'deepseek-det-a-'));
1223
+ const b = mkdtempSync(join(tmpdir(), 'deepseek-det-b-'));
1224
+ try {
1225
+ return await verifyBoundary('deepseek.usage.serve_path_determinism', async () => {
1226
+ // Both roots are driven through the IDENTICAL history from here on, which is what makes the
1227
+ // later comparison a (request, state) equality rather than a coincidence (§9 round two, NIT 1
1228
+ // — the comment used to claim this while root `a` had received extra calls root `b` had not).
1229
+ const run = (root: string) => handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT({ logprobs: true, top_logprobs: 2 })), root, occurredAt: OCCURRED_AT });
1230
+ const [ra, rb] = await Promise.all([run(a), run(b)]);
1231
+ // MUTATION-GATE FINDING: byte-equality alone is satisfied by a dead twin answering {} on
1232
+ // both roots. Determinism is only a claim about a REAL response, so every derived value the
1233
+ // cell is about — the id, the fingerprint, the cache split, the logprob numbers — is
1234
+ // asserted to EXIST and be well-formed before the two are compared.
1235
+ const bodyA = ra.body as Body;
1236
+ if (!ok(ra) || bodyA.object !== 'chat.completion') return false;
1237
+ if (typeof bodyA.id !== 'string' || !bodyA.id.startsWith('chatcmpl-twin-')) return false;
1238
+ if (typeof bodyA.system_fingerprint !== 'string' || !bodyA.system_fingerprint.endsWith('_twin_stub_kvcache')) return false;
1239
+ if (typeof (bodyA.usage as Body).prompt_cache_miss_tokens !== 'number') return false;
1240
+ if (!Array.isArray((choice0(ra).logprobs as Body)?.content) || (choice0(ra).logprobs as Body).content.length === 0) return false;
1241
+ if (typeof (choice0(ra).logprobs as Body).content[0].logprob !== 'number') return false;
1242
+ // Every derived value must be a pure function of (request, stored state). A wall clock or an
1243
+ // entropy source anywhere on the serve path breaks this.
1244
+ if (JSON.stringify(ra.body) !== JSON.stringify(rb.body)) return false;
1245
+ // …and a REPLAY into the same root is identical too, once the cache ledger already holds
1246
+ // this request's own prefix (proving the ledger read is deterministic, not accumulating).
1247
+ // A replay into each root, kept in lockstep so the two histories stay identical.
1248
+ const againA = await run(a);
1249
+ const againB = await run(b);
1250
+ if ((againA.body as Body).object !== 'chat.completion') return false;
1251
+ if (JSON.stringify(againA.body) !== JSON.stringify(againB.body)) return false;
1252
+ // The single-turn request above never HITS the cache, so it cannot prove the ledger read is
1253
+ // deterministic on the path that matters. This does: a continuation that genuinely hits,
1254
+ // replayed against a root whose ledger has since grown, must answer byte-identically — a
1255
+ // ledger read that accumulated (or a hit measured off stored state rather than the request)
1256
+ // would drift on the second call.
1257
+ // The cache-bearing path, which the single-turn half above never reaches (a one-message
1258
+ // request reads the ledger but cannot match anything until it has been served once).
1259
+ //
1260
+ // The comparison is across two roots driven through the SAME history, NOT two calls into
1261
+ // one root: serving a request CHANGES the ledger, so a second call into the same root is a
1262
+ // different (request, state) pair and is legitimately allowed to answer differently. The
1263
+ // determinism claim is "same request + same stored state → same bytes", and this is what
1264
+ // that actually looks like.
1265
+ const opening = { model: MODEL, messages: [{ role: 'user', content: 'a determinism opening turn' }] };
1266
+ const post = (root: string, payload: unknown) => handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(payload), root, occurredAt: OCCURRED_AT });
1267
+ const seededA = await post(a, opening);
1268
+ const seededB = await post(b, opening);
1269
+ if (!ok(seededA) || !ok(seededB)) return false;
1270
+ const convo = {
1271
+ model: MODEL,
1272
+ messages: [
1273
+ { role: 'user', content: 'a determinism opening turn' },
1274
+ { role: 'assistant', content: text0(seededA), reasoning_content: '' },
1275
+ { role: 'user', content: 'and a follow-up' },
1276
+ ],
1277
+ };
1278
+ const hitA = await post(a, convo);
1279
+ const hitB = await post(b, convo);
1280
+ // It really HIT — otherwise this cell would be proving determinism on the same cache-free
1281
+ // path the single-turn half already covered.
1282
+ if (!ok(hitA) || (body(hitA).usage as Body).prompt_cache_hit_tokens === 0) return false;
1283
+ return JSON.stringify(hitA.body) === JSON.stringify(hitB.body);
1284
+ });
1285
+ } finally { rmSync(a, { recursive: true, force: true }); rmSync(b, { recursive: true, force: true }); }
1286
+ }),
1287
+
1288
+ todo('deepseek.usage.real_tokenizer', 'usage', "Token counts from DeepSeek's real tokenizer: a BPE tokenizer is a data file, runs offline and is deterministic, so nothing about the serve-path invariant forbids it (§9 round one, SHOULD-FIX 5). The twin's counts are a ~4-chars-per-token estimate — the right order of magnitude and stable for a fixed input, but not DeepSeek's vocabulary, and no capability asserts an exact vendor count", 'api', 'common'),
1289
+
1290
+ // ══ connector ════════════════════════════════════════════════════════════════════════════
1291
+ done('deepseek.connector.pull_models', 'connector', 'Pull maps real model rows into the twin projection', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_models', async (root) => {
1292
+ const calls: Array<{ method: string; path: string }> = [];
1293
+ const execute = fakeExecute({ models: [{ id: 'deepseek-v4-pro', object: 'model', owned_by: 'deepseek' }] }, calls);
1294
+ const res = await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
1295
+ const row = projectResources('deepseek', root).find((r) => r.type === 'model' && r.id === 'deepseek-v4-pro');
1296
+ return res.observed > 0 && !!row && row.owned_by === 'deepseek'
1297
+ // …and the paths it actually addressed are DeepSeek's, not OpenAI-shaped ones.
1298
+ && calls.some((c) => c.method === 'GET' && c.path === '/models');
1299
+ })),
1300
+
1301
+ done('deepseek.connector.pull_files_and_balance', 'connector', 'Pull maps real files and the singleton user balance into the projection', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_files_and_balance', async (root) => {
1302
+ const execute = fakeExecute({
1303
+ files: [{ id: 'file-api-a1b2c3d4e5f6g7h8', object: 'file', bytes: 1024, created_at: 1_700_000_000, filename: 'real.png', purpose: 'user_data', expires_at: 1_700_003_600 }],
1304
+ balance: { is_available: false, balance_infos: [{ currency: 'USD', total_balance: '0.00', granted_balance: '0.00', topped_up_balance: '0.00' }] },
1305
+ });
1306
+ await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
1307
+ const file = projectResources('deepseek', root).find((r) => r.type === 'file' && r.id === 'file-api-a1b2c3d4e5f6g7h8');
1308
+ const bal = projectResources('deepseek', root).find((r) => r.type === 'balance' && r.id === 'account');
1309
+ if (!file || file.filename !== 'real.png' || file.bytes !== 1024 || file.expires_at !== 1_700_003_600) return false;
1310
+ if (!bal || bal.is_available !== false) return false;
1311
+ // The pull is LOAD-BEARING, not decorative: a pulled drained balance makes the twin refuse
1312
+ // completions exactly as the real account would.
1313
+ const chat = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), root });
1314
+ const served = await handleDeepSeekTwinRequest({ method: 'GET', path: `${FILES}/file-api-a1b2c3d4e5f6g7h8`, root });
1315
+ return chat.status === 402 && ok(served) && body(served).filename === 'real.png';
1316
+ })),
1317
+
1318
+ done('deepseek.connector.pull_is_idempotent', 'connector', 'Re-pulling identical state appends no deltas', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_is_idempotent', async (root) => {
1319
+ const execute = fakeExecute({ files: [{ id: 'file-api-aaaa', object: 'file', bytes: 1, created_at: 1, filename: 'a.png', purpose: 'user_data' }] });
1320
+ const first = await syncDeepSeekFromReal(execute, { root, occurredAt: '2026-08-31T12:00:00.000Z' });
1321
+ // A MOVING timestamp on the second poll, deliberately: the kernel hashes an observed event over
1322
+ // (occurredAt + post-state), so a pinned poll time would make this pass for the wrong reason.
1323
+ const second = await syncDeepSeekFromReal(execute, { root, occurredAt: '2026-08-31T12:05:00.000Z' });
1324
+ const rows = projectResources('deepseek', root).filter((r) => r.type === 'file');
1325
+ return first.deltasAppended > 0 && second.deltasAppended === 0 && rows.length === 1;
1326
+ })),
1327
+
1328
+ done('deepseek.connector.pull_refuses_an_error_envelope', 'connector', 'A refused pull THROWS rather than folding an empty account over real observed state', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_refuses_an_error_envelope', async (root) => {
1329
+ const good = fakeExecute({ files: [{ id: 'file-api-keepme', object: 'file', bytes: 1, created_at: 1, filename: 'keep.png', purpose: 'user_data' }] });
1330
+ await syncDeepSeekFromReal(good, { root, occurredAt: OCCURRED_AT });
1331
+ const failing = fakeExecute({ files: [], fail: '/files' });
1332
+ let threw = false;
1333
+ // `pullDeepSeekState` is a PURE mapper over the injected executor — it takes no root and writes
1334
+ // nothing, which is precisely why the refusal must happen here rather than in `syncPull`. The
1335
+ // root-scoped half is the survival check below (§9 round two, NIT 7).
1336
+ try { await pullDeepSeekState(failing); } catch { threw = true; }
1337
+ if (!threw) return false;
1338
+ // …and the same refusal through the ROOT-SCOPED entry point, so nothing was folded either.
1339
+ let syncThrew = false;
1340
+ try { await syncDeepSeekFromReal(failing, { root, occurredAt: '2026-08-31T12:20:00.000Z' }); } catch { syncThrew = true; }
1341
+ if (!syncThrew) return false;
1342
+ // The previously observed file must SURVIVE — the whole point of throwing.
1343
+ const still = await handleDeepSeekTwinRequest({ method: 'GET', path: `${FILES}/file-api-keepme`, root });
1344
+ return ok(still) && body(still).filename === 'keep.png';
1345
+ })),
1346
+
1347
+ done('deepseek.connector.push_file_delete', 'connector', 'A local delete of a PULLED file is pushed to the real account and confirmed', 'connector', 'core', () => withConnectorRoot('deepseek.connector.push_file_delete', async (root) => {
1348
+ const calls: Array<{ method: string; path: string }> = [];
1349
+ const execute = fakeExecute({ files: [{ id: 'file-api-realone', object: 'file', bytes: 1, created_at: 1, filename: 'real.png', purpose: 'user_data' }] }, calls);
1350
+ await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
1351
+ const del = await handleDeepSeekTwinRequest({ method: 'DELETE', path: `${FILES}/file-api-realone`, root, occurredAt: OCCURRED_AT });
1352
+ if (!ok(del)) return false;
1353
+ if (pendingActions('deepseek', root).filter((a) => a.subject.type === 'file').length !== 1) return false;
1354
+ const res = await pushPendingDeepSeekActions(execute, { root, occurredAt: '2026-08-31T12:10:00.000Z' });
1355
+ return res.pushed === 1
1356
+ && calls.some((c) => c.method === 'DELETE' && c.path === '/files/file-api-realone')
1357
+ && pendingActions('deepseek', root).filter((a) => a.subject.type === 'file').length === 0;
1358
+ })),
1359
+
1360
+ done('deepseek.connector.refuses_the_twins_own_id', 'connector', "A locally-minted file's delete is REFUSED, never issued against the real account under the twin's own id", 'connector', 'core', () => withConnectorRoot('deepseek.connector.refuses_the_twins_own_id', async (root) => {
1361
+ const calls: Array<{ method: string; path: string }> = [];
1362
+ const execute = fakeExecute({}, calls);
1363
+ await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD()), root, occurredAt: OCCURRED_AT });
1364
+ await handleDeepSeekTwinRequest({ method: 'DELETE', path: `${FILES}/${FILE_1}`, root, occurredAt: OCCURRED_AT });
1365
+ const res = await pushPendingDeepSeekActions(execute, { root, occurredAt: '2026-08-31T12:10:00.000Z' });
1366
+ // Both actions must be refused — the create because the vendor endpoint is multipart, the
1367
+ // delete because no vendor id was ever recorded — and NOTHING may have gone out.
1368
+ if (res.pushed !== 0 || res.refused.length !== 2) return false;
1369
+ if (calls.length !== 0) return false;
1370
+ if (!res.refused.some((r) => r.operation === 'file.create' && r.reason.includes('multipart'))) return false;
1371
+ if (!res.refused.some((r) => r.operation === 'file.delete' && r.reason.includes('no vendor id recorded'))) return false;
1372
+ // And the path builder refuses the twin's mint outright, not merely by convention.
1373
+ if (externalIdFor('file', FILE_1, root) !== null) return false;
1374
+ let threw = false;
1375
+ try { deepseekRequestForAction({ operation: 'file.delete', subject: { type: 'file', id: FILE_1 } } as never); } catch { threw = true; }
1376
+ return threw;
1377
+ })),
1378
+
1379
+ done('deepseek.connector.skips_the_internal_cache_ledger', 'connector', 'The twin-only context-cache ledger is skipped BY NAME rather than pushed or silently dropped', 'connector', 'common', () => withConnectorRoot('deepseek.connector.skips_the_internal_cache_ledger', async (root) => {
1380
+ const calls: Array<{ method: string; path: string }> = [];
1381
+ const execute = fakeExecute({}, calls);
1382
+ await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), root, occurredAt: OCCURRED_AT });
1383
+ const pendingCache = pendingActions('deepseek', root).filter((a) => a.subject.type === 'cache_prefix');
1384
+ if (pendingCache.length === 0) return false; // the completion really did mint ledger actions
1385
+ const res = await pushPendingDeepSeekActions(execute, { root, occurredAt: '2026-08-31T12:10:00.000Z' });
1386
+ // Reported, not refused (they are not a gap) and not pushed (DeepSeek has no cache endpoint).
1387
+ return res.skippedInternal.length === pendingCache.length && res.refused.length === 0 && calls.length === 0
1388
+ && unpushableReason('cache_prefix.observe') !== null;
1389
+ })),
1390
+
1391
+ done('deepseek.connector.pull_after_local_create', 'connector', 'ID COLLISION, both directions: a pulled vendor id and a locally-minted one never collide', 'connector', 'common', () => withConnectorRoot('deepseek.connector.pull_after_local_create', async (root) => {
1392
+ // Direction 1 — local create, THEN pull: the pulled row must not overwrite the local one.
1393
+ await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD({ filename: 'local.png' })), root, occurredAt: OCCURRED_AT });
1394
+ const execute = fakeExecute({ files: [{ id: 'file-api-vendor01', object: 'file', bytes: 9, created_at: 2, filename: 'vendor.png', purpose: 'user_data' }] });
1395
+ await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
1396
+ const local = await handleDeepSeekTwinRequest({ method: 'GET', path: `${FILES}/${FILE_1}`, root });
1397
+ const vendor = await handleDeepSeekTwinRequest({ method: 'GET', path: '/files/file-api-vendor01', root });
1398
+ if (!ok(local) || body(local).filename !== 'local.png' || !ok(vendor)) return false;
1399
+ // Direction 2 — create AFTER the pull: the new mint must not land on the pulled id, and the
1400
+ // local namespace must ratchet from the local set only.
1401
+ const next = await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD({ filename: 'after.png' })), root, occurredAt: OCCURRED_AT });
1402
+ return ok(next) && body(next).id === FILE_2 && body(next).id !== 'file-api-vendor01';
1403
+ })),
1404
+
1405
+ done('deepseek.connector.full_sync', 'connector', 'fullSync pushes pending writes then pulls every modeled collection and the balance singleton', 'connector', 'common', () => withConnectorRoot('deepseek.connector.full_sync', async (root) => {
1406
+ const calls: Array<{ method: string; path: string }> = [];
1407
+ const execute = fakeExecute({
1408
+ models: [{ id: 'deepseek-v4-flash', object: 'model', owned_by: 'deepseek' }],
1409
+ files: [{ id: 'file-api-sync01', object: 'file', bytes: 3, created_at: 5, filename: 's.png', purpose: 'user_data' }],
1410
+ balance: { is_available: true, balance_infos: [{ currency: 'USD', total_balance: '5.00', granted_balance: '0.00', topped_up_balance: '5.00' }] },
1411
+ }, calls);
1412
+ const res = await fullSyncDeepSeek(execute, { root, occurredAt: OCCURRED_AT });
1413
+ const paths = calls.map((c) => `${c.method} ${c.path}`);
1414
+ return res.collections === 3 && res.observed === 3 && res.deltasAppended > 0
1415
+ && paths.includes('GET /models') && paths.includes('GET /files') && paths.includes('GET /user/balance');
1416
+ })),
1417
+
1418
+ todo('deepseek.connector.push_file_create', 'connector', "Pushing a local file create — real DeepSeek's POST /files is multipart/form-data with the image bytes, which the JSON executor cannot express and the twin has no real bytes for", 'connector', 'common'),
1419
+ todo('deepseek.connector.pull_pagination', 'connector', 'Following the Files API `after` cursor across more than one page during a pull', 'connector', 'niche'),
1420
+ todo('deepseek.connector.unpushable_actions_drain', 'connector', 'Acknowledging a permanently-unpushable action so `pendingActions` can reach empty again', 'connector', 'niche'),
1421
+
1422
+ // ══ conformance ══════════════════════════════════════════════════════════════════════════
1423
+ done('deepseek.conformance.probes', 'conformance', 'Offline conformance harness passes: probes, the router census, the must-not-serve list and the rejection table', 'api', 'core', async () => {
1424
+ const { checkDeepSeekConformance } = await import('./deepseek-conformance.ts');
1425
+ const report = await checkDeepSeekConformance();
1426
+ return report.ok && report.probes >= 11 && report.checksRun >= 60;
1427
+ }),
1428
+ ];
1429
+
1430
+ // TWIN-87 committed area census — DeepSeek's top-level API product areas, authored top-down from
1431
+ // the api-docs.deepseek.com nav (Quick Start, API Reference, and each Guide) independently of what a
1432
+ // manifest entry happens to already exist for. deepseek-capabilities.test.ts's area-census meta-test
1433
+ // (assertAreaCensus) fails the gate if a declared area has zero manifest entries, OR if a manifest
1434
+ // entry's `area` drifts outside this list — so a whole missing area can never hide invisibly.
1435
+ export const DEEPSEEK_AREAS = [
1436
+ 'anthropic', 'auth', 'balance', 'beta', 'cache', 'chat', 'completions', 'conformance',
1437
+ 'connector', 'errors', 'files', 'models', 'rate_limits', 'reasoning', 'responses',
1438
+ 'streaming', 'structured_outputs', 'tools', 'usage', 'vision',
1439
+ ] as const;
1440
+
1441
+ export function deepseekCapabilities(): Promise<CapabilityReport> {
1442
+ return checkCapabilities('deepseek', DEEPSEEK_CAPABILITIES);
1443
+ }