@volter/twin-deepseek 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +198 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/deepseek-budget.d.ts +51 -0
- package/dist/src/deepseek-budget.js +152 -0
- package/dist/src/deepseek-cache.d.ts +56 -0
- package/dist/src/deepseek-cache.js +151 -0
- package/dist/src/deepseek-capabilities.d.ts +4 -0
- package/dist/src/deepseek-capabilities.js +1520 -0
- package/dist/src/deepseek-conformance.d.ts +14 -0
- package/dist/src/deepseek-conformance.js +473 -0
- package/dist/src/deepseek-connector.d.ts +168 -0
- package/dist/src/deepseek-connector.js +386 -0
- package/dist/src/deepseek-models.d.ts +30 -0
- package/dist/src/deepseek-models.js +38 -0
- package/dist/src/deepseek-scenario.d.ts +55 -0
- package/dist/src/deepseek-scenario.js +170 -0
- package/dist/src/deepseek-server.d.ts +16 -0
- package/dist/src/deepseek-server.js +191 -0
- package/dist/src/deepseek-stub.d.ts +75 -0
- package/dist/src/deepseek-stub.js +191 -0
- package/dist/src/deepseek-twin.d.ts +77 -0
- package/dist/src/deepseek-twin.js +1103 -0
- package/dist/src/deepseek-types.d.ts +172 -0
- package/dist/src/deepseek-types.js +26 -0
- package/dist/src/index.d.ts +15 -0
- package/dist/src/index.js +93 -0
- package/package.json +68 -0
- package/src/cli.ts +27 -0
- package/src/deepseek-budget.ts +178 -0
- package/src/deepseek-cache.ts +159 -0
- package/src/deepseek-capabilities.ts +1443 -0
- package/src/deepseek-conformance.ts +512 -0
- package/src/deepseek-connector.ts +440 -0
- package/src/deepseek-models.ts +65 -0
- package/src/deepseek-scenario.ts +188 -0
- package/src/deepseek-server.ts +201 -0
- package/src/deepseek-stub.ts +200 -0
- package/src/deepseek-twin.ts +1163 -0
- package/src/deepseek-types.ts +201 -0
- package/src/index.ts +133 -0
|
@@ -0,0 +1,1443 @@
|
|
|
1
|
+
// DeepSeek capability manifest — the EXPECTED REAL-PRODUCT SURFACE (the target), authored top-down
|
|
2
|
+
// from what the DeepSeek Platform API actually does — NOT from what this twin has built. The
|
|
3
|
+
// denominator was enumerated from TWO first-party sources, both read on 2026-08-31:
|
|
4
|
+
// • api-docs.deepseek.com, walked as a docs NAV: quick_start/{error_codes,rate_limit,pricing},
|
|
5
|
+
// api/{create-chat-completion,create-completion,list-models,get-user-balance}, and
|
|
6
|
+
// guides/{kv_cache,thinking_mode,files_api,anthropic_api,chat_prefix_completion};
|
|
7
|
+
// • `@ai-sdk/deepseek@3.0.37` — the only first-party npm client of this surface — read as SOURCE:
|
|
8
|
+
// the zod schemas it encodes/decodes with (deepseek-chat-api-types.ts, files/deepseek-files-api.ts),
|
|
9
|
+
// the request it builds (deepseek-chat-language-model.ts), the refusals it raises locally
|
|
10
|
+
// (deepseek-prepare-tools.ts, convert-to-deepseek-chat-messages.ts), and the provider doc page
|
|
11
|
+
// shipped inside its tarball (docs/30-deepseek.mdx).
|
|
12
|
+
// DeepSeek publishes no OpenAPI document, so where those two conflict the SDK's generated schema
|
|
13
|
+
// wins over a rendered docs example (ADDING_A_TWIN.md §6, the precedence order), and where only
|
|
14
|
+
// prose exists the manifest says so rather than asserting a closed set.
|
|
15
|
+
//
|
|
16
|
+
// ═══ THE DENOMINATOR IS DELIBERATELY WEIGHTED TOWARD REFUSALS ═══
|
|
17
|
+
// DeepSeek is OPENAI-COMPATIBLE. Its response shapes are the easy half; what makes a DeepSeek twin
|
|
18
|
+
// a DeepSeek twin rather than a relabelled OpenAI twin is what it REFUSES — 422 where OpenAI 400s,
|
|
19
|
+
// 402 for a drained balance, no `/v1` segment, no `json_schema`/`n`/`seed`/`logit_bias`/`user`,
|
|
20
|
+
// beta-only prefix completion and strict tools, and a documented 400 when `tools` is sent without
|
|
21
|
+
// prior-turn `reasoning_content`. Those are first-class capabilities here, each with a failable
|
|
22
|
+
// negative verify, because ADDING_A_TWIN.md §0 records that inherited PERMISSIVENESS is how the
|
|
23
|
+
// nearest sibling pack shipped two false-greens.
|
|
24
|
+
//
|
|
25
|
+
// There are NO carve-outs: every entry here is done or todo. The twin returns DETERMINISTIC labeled
|
|
26
|
+
// stubs where the vendor runs a model — those stubs ARE its answer — while the protocol envelope is
|
|
27
|
+
// faithful, and it does not simulate properties of the vendor's own serving fleet (concurrency
|
|
28
|
+
// ceilings). `deepseek.usage.real_tokenizer` and `deepseek.cache.expiry` are honest todos: a BPE
|
|
29
|
+
// tokenizer is an offline data file and the kernel's world clock is deterministic, so both are work
|
|
30
|
+
// not yet done rather than anything unreachable.
|
|
31
|
+
//
|
|
32
|
+
// TIERING (§6 rule 3): `core` = "first-week-of-every-integration". The overwhelming majority of
|
|
33
|
+
// week-one DeepSeek integrations are chat completions and nothing else, so `core` is the
|
|
34
|
+
// chat/streaming/tools/reasoning/errors/auth spine plus the conformance and connector-pull
|
|
35
|
+
// backbone. The Files API (images only, and only usable by one vision model), the beta FIM
|
|
36
|
+
// endpoint, the beta prefix/strict features and the Anthropic-compatible surface are specialist and
|
|
37
|
+
// are tiered `common` or `niche`.
|
|
38
|
+
//
|
|
39
|
+
// (DeepSeek is an API-first vendor — platform.deepseek.com is a keys/billing/usage console, not
|
|
40
|
+
// where the work happens — so this pack ships NO mirror and has NO UI capabilities.)
|
|
41
|
+
import { mkdtempSync, rmSync } from 'node:fs';
|
|
42
|
+
import { tmpdir } from 'node:os';
|
|
43
|
+
import { join } from 'node:path';
|
|
44
|
+
import { checkCapabilities, type CapabilityReport, type CapabilitySpec, verifyBoundary, isInfrastructureError, harnessError } from '@volter/world-tooling';
|
|
45
|
+
import { applyTwinWrite, pendingActions, projectResources } from '@volter/world-core';
|
|
46
|
+
import { handleDeepSeekTwinRequest, type DeepSeekResponseEnvelope } from './deepseek-twin.ts';
|
|
47
|
+
import {
|
|
48
|
+
fullSyncDeepSeek,
|
|
49
|
+
deepseekRequestForAction,
|
|
50
|
+
externalIdFor,
|
|
51
|
+
liveDeepSeekExecute,
|
|
52
|
+
pullDeepSeekState,
|
|
53
|
+
pushPendingDeepSeekActions,
|
|
54
|
+
syncDeepSeekFromReal,
|
|
55
|
+
unpushableReason,
|
|
56
|
+
type DeepSeekExecute,
|
|
57
|
+
} from './deepseek-connector.ts';
|
|
58
|
+
import { createDeepSeekScenarioEngine } from './deepseek-scenario.ts';
|
|
59
|
+
import type { SseEvent } from './deepseek-types.ts';
|
|
60
|
+
|
|
61
|
+
// ── API verify: drive REAL requests against a fresh temp root, then assert status/shape ──
|
|
62
|
+
type Step = { m: string; p: string; b?: unknown };
|
|
63
|
+
type Body = Record<string, any>;
|
|
64
|
+
|
|
65
|
+
/** Run a sequence of real DeepSeek requests against an isolated root; return all responses. */
|
|
66
|
+
async function withRoot(steps: (h: (s: Step) => Promise<DeepSeekResponseEnvelope>, root: string) => Promise<boolean>): Promise<boolean> {
|
|
67
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
68
|
+
const h = (s: Step) => handleDeepSeekTwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : JSON.stringify(s.b), root, occurredAt: OCCURRED_AT });
|
|
69
|
+
try {
|
|
70
|
+
// `root` is handed to the steps too, so a verify can inspect the LOG (projectResources /
|
|
71
|
+
// pendingActions) and not merely the responses — the difference between proving "the reply did
|
|
72
|
+
// not change" and proving "nothing was written".
|
|
73
|
+
return await verifyBoundary('deepseek.withRoot', () => steps(h, root));
|
|
74
|
+
} finally {
|
|
75
|
+
rmSync(root, { recursive: true, force: true });
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** Like withRoot, but the request helper passes request HEADERS through (for auth and the
|
|
80
|
+
* deterministic 429 trigger, which the trusted no-headers helper never fires). */
|
|
81
|
+
type StepH = Step & { headers?: Record<string, string> };
|
|
82
|
+
async function withRootH(steps: (h: (s: StepH) => Promise<DeepSeekResponseEnvelope>) => Promise<boolean>): Promise<boolean> {
|
|
83
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
84
|
+
const h = (s: StepH) => handleDeepSeekTwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : JSON.stringify(s.b), root, occurredAt: OCCURRED_AT, ...(s.headers ? { headers: s.headers } : {}) });
|
|
85
|
+
try {
|
|
86
|
+
return await verifyBoundary('deepseek.withRootH', () => steps(h));
|
|
87
|
+
} finally {
|
|
88
|
+
rmSync(root, { recursive: true, force: true });
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Collect the streaming SSE events for a chat request against an isolated root. */
|
|
93
|
+
function withStream(body: unknown, fn: (events: SseEvent[], final: DeepSeekResponseEnvelope) => boolean, path = CHAT_PATH): Promise<boolean> {
|
|
94
|
+
return new Promise<boolean>((resolve, reject) => {
|
|
95
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
96
|
+
const events: SseEvent[] = [];
|
|
97
|
+
handleDeepSeekTwinRequest({ method: 'POST', path, body: JSON.stringify(body), root, occurredAt: OCCURRED_AT, sseSink: (e) => events.push(e) })
|
|
98
|
+
.then((final) => resolve(fn(events, final)))
|
|
99
|
+
.catch((err) => { if (isInfrastructureError(err)) reject(harnessError('deepseek.withStream', err)); else resolve(false); })
|
|
100
|
+
.finally(() => rmSync(root, { recursive: true, force: true }));
|
|
101
|
+
});
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/** A connector verify against an isolated root, with the injected fake executor the verify builds. */
|
|
105
|
+
async function withConnectorRoot(id: string, fn: (root: string) => Promise<boolean>): Promise<boolean> {
|
|
106
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
107
|
+
try {
|
|
108
|
+
return await fn(root);
|
|
109
|
+
} catch (err) {
|
|
110
|
+
if (isInfrastructureError(err)) throw harnessError(id, err);
|
|
111
|
+
return false;
|
|
112
|
+
} finally {
|
|
113
|
+
rmSync(root, { recursive: true, force: true });
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
const ok = (r: DeepSeekResponseEnvelope) => r.status >= 200 && r.status < 300;
|
|
118
|
+
/** Refused as a CLIENT error with the vendor's envelope. Used wherever the twin's exact status is
|
|
119
|
+
* its own unverified choice — DeepSeek's published error table has no 404 at all — so asserting a
|
|
120
|
+
* specific code would claim a vendor fact this pack states it does not have (§9 round one). */
|
|
121
|
+
const refused = (r: DeepSeekResponseEnvelope) => r.status >= 400 && r.status < 500 && typeof (r.body as Body)?.error?.message === 'string';
|
|
122
|
+
const body = (r: DeepSeekResponseEnvelope) => r.body as Body;
|
|
123
|
+
const msg = (r: DeepSeekResponseEnvelope) => String((r.body as Body)?.error?.message ?? '');
|
|
124
|
+
const choice0 = (r: DeepSeekResponseEnvelope) => ((r.body as Body)?.choices as Body[])?.[0];
|
|
125
|
+
const text0 = (r: DeepSeekResponseEnvelope) => String(choice0(r)?.message?.content ?? '');
|
|
126
|
+
|
|
127
|
+
// ── shorthands ──
|
|
128
|
+
const done = (id: string, area: string, title: string, dimension: CapabilitySpec['dimension'], tier: CapabilitySpec['tier'], verify: CapabilitySpec['verify']): CapabilitySpec => ({ id, area, title, dimension, tier, expected: 'done', verify });
|
|
129
|
+
const todo = (id: string, area: string, title: string, dimension: CapabilitySpec['dimension'], tier: CapabilitySpec['tier']): CapabilitySpec => ({ id, area, title, dimension, tier, expected: 'todo' });
|
|
130
|
+
|
|
131
|
+
// DeepSeek's real paths. NOTE the absence of `/v1` — that is not an omission, it is the vendor's
|
|
132
|
+
// own base_url shape and `deepseek.chat.no_v1_prefix` asserts a twin serving `/v1` would be wrong.
|
|
133
|
+
const CHAT_PATH = '/chat/completions';
|
|
134
|
+
const BETA_CHAT = '/beta/chat/completions';
|
|
135
|
+
const FIM_PATH = '/beta/completions';
|
|
136
|
+
const FILES = '/files';
|
|
137
|
+
const MODELS = '/models';
|
|
138
|
+
const BALANCE = '/user/balance';
|
|
139
|
+
|
|
140
|
+
/** A pinned timestamp so ids, `created` and cache-ledger writes are deterministic. Where a verify
|
|
141
|
+
* REPEATS an identical transition it pins a DIFFERENT value deliberately (the kernel dedupes by
|
|
142
|
+
* content + millisecond — ADDING_A_TWIN.md §6). */
|
|
143
|
+
const OCCURRED_AT = '2026-08-31T12:00:00.000Z';
|
|
144
|
+
|
|
145
|
+
const MODEL = 'deepseek-v4-flash';
|
|
146
|
+
const PRO = 'deepseek-v4-pro';
|
|
147
|
+
const CHAT = (extra: Record<string, unknown> = {}) => ({ model: MODEL, messages: [{ role: 'user', content: 'hello twin' }], ...extra });
|
|
148
|
+
const WEATHER_TOOL = { type: 'function', function: { name: 'get_weather', parameters: { type: 'object', properties: { city: { type: 'string' }, days: { type: 'integer' } } } } };
|
|
149
|
+
const TIME_TOOL = { type: 'function', function: { name: 'get_time', parameters: { type: 'object', properties: { tz: { type: 'string' } } } } };
|
|
150
|
+
/** A valid PNG upload as the server's multipart adapter hands it to the handler. */
|
|
151
|
+
const UPLOAD = (extra: Record<string, unknown> = {}) => ({ purpose: 'user_data', filename: 'shot.png', media_type: 'image/png', content: 'AAA=', bytes: 3, ...extra });
|
|
152
|
+
const FILE_1 = 'file-api-twin000000000001';
|
|
153
|
+
const FILE_2 = 'file-api-twin000000000002';
|
|
154
|
+
|
|
155
|
+
/** A fake live executor over an in-memory account. Records every (method, path) so a connector
|
|
156
|
+
* verify can assert WHICH calls went out, not merely that something did. */
|
|
157
|
+
function fakeExecute(account: {
|
|
158
|
+
models?: Body[]; files?: Body[]; balance?: Body; fail?: string;
|
|
159
|
+
}, calls: Array<{ method: string; path: string; body?: unknown }> = []): DeepSeekExecute {
|
|
160
|
+
return async (method, path, reqBody) => {
|
|
161
|
+
calls.push({ method, path, ...(reqBody === undefined ? {} : { body: reqBody }) });
|
|
162
|
+
if (account.fail === path) return { error: { message: 'Authentication fails due to the wrong API key' } };
|
|
163
|
+
if (path === '/models') return { object: 'list', data: account.models ?? [] };
|
|
164
|
+
if (path === '/files') return { object: 'list', data: account.files ?? [] };
|
|
165
|
+
if (path === '/user/balance') return (account.balance ?? { is_available: true, balance_infos: [] }) as Body;
|
|
166
|
+
if (method === 'DELETE' && path.startsWith('/files/')) return { id: path.slice('/files/'.length), object: 'file', deleted: true };
|
|
167
|
+
return {};
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
export const DEEPSEEK_CAPABILITIES: CapabilitySpec[] = [
|
|
172
|
+
// ══ chat completions ═════════════════════════════════════════════════════════════════════
|
|
173
|
+
done('deepseek.chat.completion', 'chat', 'POST /chat/completions returns a faithful chat.completion envelope', 'api', 'core', () => withRoot(async (h) => {
|
|
174
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
175
|
+
if (!ok(r)) return false;
|
|
176
|
+
const b = body(r);
|
|
177
|
+
return b.object === 'chat.completion' && b.model === MODEL && typeof b.id === 'string' && b.id.startsWith('chatcmpl-')
|
|
178
|
+
&& typeof b.system_fingerprint === 'string' && b.system_fingerprint.startsWith('fp_')
|
|
179
|
+
&& Array.isArray(b.choices) && b.choices.length === 1
|
|
180
|
+
&& choice0(r).index === 0 && choice0(r).message.role === 'assistant'
|
|
181
|
+
&& text0(r).includes('[twin-stub:deepseek-v4-flash]') && text0(r).includes('hello twin')
|
|
182
|
+
&& choice0(r).finish_reason === 'stop'
|
|
183
|
+
// …and DeepSeek's assistant message has NO `refusal` (OpenAI's does). Serving one would be
|
|
184
|
+
// the inverse false-green.
|
|
185
|
+
&& !('refusal' in choice0(r).message);
|
|
186
|
+
})),
|
|
187
|
+
|
|
188
|
+
done('deepseek.chat.system_and_multi_turn', 'chat', 'System messages and multi-turn history are accepted and echoed from the last user turn', 'api', 'core', () => withRoot(async (h) => {
|
|
189
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [
|
|
190
|
+
{ role: 'system', content: 'be terse' },
|
|
191
|
+
{ role: 'user', content: 'first question' },
|
|
192
|
+
{ role: 'assistant', content: 'first answer', reasoning_content: '' },
|
|
193
|
+
{ role: 'user', content: 'second question' },
|
|
194
|
+
] } });
|
|
195
|
+
if (!ok(r)) return false;
|
|
196
|
+
// The echo proves the twin read the LAST user turn, not the first — a handler returning a
|
|
197
|
+
// fixed string would fail this.
|
|
198
|
+
if (!text0(r).includes('second question') || text0(r).includes('first question')) return false;
|
|
199
|
+
// Negative: an empty message list is refused with DeepSeek's parameter status.
|
|
200
|
+
const empty = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [] } });
|
|
201
|
+
// Negative: an unknown role is refused too.
|
|
202
|
+
const badRole = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'developer', content: 'x' }] } });
|
|
203
|
+
return empty.status === 422 && badRole.status === 422;
|
|
204
|
+
})),
|
|
205
|
+
|
|
206
|
+
done('deepseek.chat.max_tokens_truncation', 'chat', "max_tokens truncates and reports finish_reason 'length'", 'api', 'core', () => withRoot(async (h) => {
|
|
207
|
+
const full = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
208
|
+
const cut = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ max_tokens: 4 }) });
|
|
209
|
+
if (!ok(full) || !ok(cut)) return false;
|
|
210
|
+
if (choice0(cut).finish_reason !== 'length' || choice0(full).finish_reason !== 'stop') return false;
|
|
211
|
+
if (text0(cut).length >= text0(full).length) return false;
|
|
212
|
+
// Negative: DeepSeek's documented bound is an integer >= 1.
|
|
213
|
+
const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ max_tokens: 0 }) });
|
|
214
|
+
return bad.status === 422;
|
|
215
|
+
})),
|
|
216
|
+
|
|
217
|
+
done('deepseek.chat.stop_sequences', 'chat', 'stop truncates at the earliest match; more than 16 sequences is refused', 'api', 'common', () => withRoot(async (h) => {
|
|
218
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ stop: ['DeepSeek twin', 'deterministic'] }) });
|
|
219
|
+
if (!ok(r)) return false;
|
|
220
|
+
// 'deterministic' occurs EARLIER in the stub text than 'DeepSeek twin', so truncation must
|
|
221
|
+
// happen there — an implementation that took `stop[0]` would keep 'deterministic' and fail here.
|
|
222
|
+
if (!text0(r).includes('This is a ') || text0(r).includes('deterministic') || text0(r).includes('DeepSeek twin')) return false;
|
|
223
|
+
const many = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ stop: Array.from({ length: 17 }, (_, i) => `s${i}`) }) });
|
|
224
|
+
const sixteen = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ stop: Array.from({ length: 16 }, (_, i) => `s${i}`) }) });
|
|
225
|
+
return many.status === 422 && ok(sixteen);
|
|
226
|
+
})),
|
|
227
|
+
|
|
228
|
+
done('deepseek.chat.rejects_openai_only_params', 'chat', "The OpenAI parameters DeepSeek's closed table does not declare (n, seed, logit_bias, top_k, user, max_completion_tokens, service_tier, parallel_tool_calls, functions) are refused BY NAME with 422", 'api', 'core', () => withRoot(async (h) => {
|
|
229
|
+
// THE flagship OpenAI-divergence check. A twin copied from an OpenAI-shaped exemplar serves
|
|
230
|
+
// every one of these happily, which is the inherited-permissiveness false-green
|
|
231
|
+
// ADDING_A_TWIN.md §0 names. Each must be 422 — DeepSeek's "Invalid Parameters" — not 400.
|
|
232
|
+
for (const [key, value] of [
|
|
233
|
+
['n', 2], ['seed', 42], ['logit_bias', { '1': 1 }], ['top_k', 5], ['user', 'u1'],
|
|
234
|
+
['max_completion_tokens', 32], ['service_tier', 'flex'], ['parallel_tool_calls', false],
|
|
235
|
+
['functions', [{ name: 'f' }]], ['store', true], ['metadata', { a: 'b' }],
|
|
236
|
+
] as Array<[string, unknown]>) {
|
|
237
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ [key]: value }) });
|
|
238
|
+
if (r.status !== 422) return false;
|
|
239
|
+
// The refusal must NAME the offending key, so a caller can act on it.
|
|
240
|
+
if (!msg(r).includes(`'${key}'`)) return false;
|
|
241
|
+
}
|
|
242
|
+
return true;
|
|
243
|
+
})),
|
|
244
|
+
|
|
245
|
+
done('deepseek.chat.no_v1_prefix', 'chat', "DeepSeek's base_url carries no version segment: /v1/... is not served", 'api', 'core', () => withRoot(async (h) => {
|
|
246
|
+
// api-docs.deepseek.com's own quick start uses base_url = https://api.deepseek.com with the
|
|
247
|
+
// endpoint at /chat/completions. A twin answering on the OpenAI-shaped /v1 prefix would be
|
|
248
|
+
// asserting a route the vendor does not have.
|
|
249
|
+
const served = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
250
|
+
const v1chat = await h({ m: 'POST', p: '/v1/chat/completions', b: CHAT() });
|
|
251
|
+
const v1models = await h({ m: 'GET', p: '/v1/models' });
|
|
252
|
+
// The claim is that the route is REFUSED, not that the code is exactly 404: DeepSeek's
|
|
253
|
+
// published error table is {400,401,402,422,429,500,503} and contains no 404 at all, so the
|
|
254
|
+
// twin's choice of status for an unrouted path is unverified against any first-party source
|
|
255
|
+
// (filed as `deepseek.errors.unknown_route_envelope`). Asserting 404 exactly would be claiming
|
|
256
|
+
// a vendor fact this pack explicitly says it does not have (§9 round one, NIT 11).
|
|
257
|
+
return ok(served) && refused(v1chat) && refused(v1models)
|
|
258
|
+
&& msg(v1chat).includes('Unknown request URL');
|
|
259
|
+
})),
|
|
260
|
+
|
|
261
|
+
done('deepseek.chat.deprecated_params_accepted', 'chat', 'frequency_penalty / presence_penalty are DEPRECATED but accepted with no effect — never an error', 'api', 'common', () => withRoot(async (h) => {
|
|
262
|
+
// The mirror image of the rejection surface, and just as load-bearing: DeepSeek documents these
|
|
263
|
+
// as having "no effect", not as errors. A twin that 4xx'd them would be refusing surface the
|
|
264
|
+
// vendor has. "No effect" is asserted, not assumed: the content must be byte-identical.
|
|
265
|
+
const plain = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
266
|
+
const penalised = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ frequency_penalty: 1.5, presence_penalty: -1.2 }) });
|
|
267
|
+
if (!ok(plain) || !ok(penalised)) return false;
|
|
268
|
+
// MUTATION-GATE FINDING: "the two answers are equal" is satisfied by a dead twin answering
|
|
269
|
+
// {} twice. The equality only means something once each side is proven to be a REAL completion,
|
|
270
|
+
// so the content is asserted absolutely before it is compared.
|
|
271
|
+
if (!text0(penalised).includes('[twin-stub:deepseek-v4-flash]') || !text0(penalised).includes('hello twin')) return false;
|
|
272
|
+
if (body(penalised).object !== 'chat.completion' || choice0(penalised).finish_reason !== 'stop') return false;
|
|
273
|
+
return text0(plain) === text0(penalised);
|
|
274
|
+
})),
|
|
275
|
+
|
|
276
|
+
done('deepseek.chat.temperature_ignored_while_thinking', 'chat', 'temperature / top_p are ignored (not refused) while thinking is enabled, and bounded when checked', 'api', 'common', () => withRoot(async (h) => {
|
|
277
|
+
// "setting these parameters will not trigger an error but will also have no effect"
|
|
278
|
+
// (api-docs.deepseek.com/guides/thinking_mode). Thinking is on by default for every V4 model.
|
|
279
|
+
const plain = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
280
|
+
const hot = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ temperature: 1.9, top_p: 0.1 }) });
|
|
281
|
+
if (!ok(plain) || !ok(hot)) return false;
|
|
282
|
+
// Same shape as the deprecated-params cell: prove each side is a real completion before the
|
|
283
|
+
// equality is allowed to mean anything (a dead twin answers {} twice, and {} === {}).
|
|
284
|
+
if (!text0(hot).includes('[twin-stub:deepseek-v4-flash]') || body(hot).object !== 'chat.completion') return false;
|
|
285
|
+
if (text0(plain) !== text0(hot)) return false;
|
|
286
|
+
// …but the documented RANGES are still enforced: temperature 0–2, top_p in (0, 1].
|
|
287
|
+
const tooHot = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ temperature: 2.5 }) });
|
|
288
|
+
const badP = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ top_p: 1.5 }) });
|
|
289
|
+
return tooHot.status === 422 && badP.status === 422;
|
|
290
|
+
})),
|
|
291
|
+
|
|
292
|
+
done('deepseek.chat.user_id', 'chat', "user_id (not OpenAI's `user`) is accepted, charset-checked and length-capped at 512", 'api', 'common', () => withRoot(async (h) => {
|
|
293
|
+
const good = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'tenant_123-user' }) });
|
|
294
|
+
const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'tenant 123' }) });
|
|
295
|
+
const long = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'a'.repeat(513) }) });
|
|
296
|
+
const atCap = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ user_id: 'a'.repeat(512) }) });
|
|
297
|
+
return ok(good) && bad.status === 422 && long.status === 422 && ok(atCap);
|
|
298
|
+
})),
|
|
299
|
+
|
|
300
|
+
done('deepseek.chat.message_name', 'chat', 'Participant `name` is accepted on system/user/assistant turns and counted into prompt tokens', 'api', 'niche', () => withRoot(async (h) => {
|
|
301
|
+
const named = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi', name: 'customer_alpha' }] } });
|
|
302
|
+
const plain = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi' }] } });
|
|
303
|
+
if (!ok(named) || !ok(plain)) return false;
|
|
304
|
+
// A name is real payload, so it must move the prompt-token count — a handler that dropped the
|
|
305
|
+
// field entirely would report the same number.
|
|
306
|
+
if (body(named).usage.prompt_tokens <= body(plain).usage.prompt_tokens) return false;
|
|
307
|
+
// …and a NON-STRING name is a 422, not a 200. §9 round two, SHOULD-FIX 2: it used to reach
|
|
308
|
+
// `estimateTokens`, produce NaN and serialise as `null`, so the twin answered 200 with
|
|
309
|
+
// `prompt_tokens: null` — breaking the cache invariant its own capability asserts and handing
|
|
310
|
+
// the SDK a body its usage schema cannot decode. The revert matrix showed this cell was HOLLOW
|
|
311
|
+
// without the assertion.
|
|
312
|
+
const badName = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi', name: 12345 }] } });
|
|
313
|
+
if (badName.status !== 422) return false;
|
|
314
|
+
const badReasoning = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'assistant', content: 'x', reasoning_content: 7 }] } });
|
|
315
|
+
// …and a non-array tool_calls is a 422 too, not the retryable 500 a TypeError used to produce.
|
|
316
|
+
const badCalls = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'assistant', content: 'x', tool_calls: 5 }] } });
|
|
317
|
+
return badReasoning.status === 422 && badCalls.status === 422;
|
|
318
|
+
})),
|
|
319
|
+
|
|
320
|
+
done('deepseek.chat.logprobs_envelope', 'chat', "logprobs returns DeepSeek's TWO-channel block (content + reasoning_content); top_logprobs is bounded and requires logprobs", 'api', 'niche', () => withRoot(async (h) => {
|
|
321
|
+
const off = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
322
|
+
if (!ok(off) || choice0(off).logprobs !== null) return false;
|
|
323
|
+
const on = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: true, top_logprobs: 3 }) });
|
|
324
|
+
if (!ok(on)) return false;
|
|
325
|
+
const lp = choice0(on).logprobs as Body;
|
|
326
|
+
if (!lp || !Array.isArray(lp.content) || lp.content.length === 0) return false;
|
|
327
|
+
if (typeof lp.content[0].token !== 'string' || typeof lp.content[0].logprob !== 'number') return false;
|
|
328
|
+
if (lp.content[0].top_logprobs.length !== 3) return false;
|
|
329
|
+
// The reasoning channel is the DeepSeek-specific half — OpenAI's logprobs has no such key.
|
|
330
|
+
if (!Array.isArray(lp.reasoning_content) || lp.reasoning_content.length === 0) return false;
|
|
331
|
+
const tooMany = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: true, top_logprobs: 21 }) });
|
|
332
|
+
const orphan = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ top_logprobs: 2 }) });
|
|
333
|
+
return tooMany.status === 422 && orphan.status === 422;
|
|
334
|
+
})),
|
|
335
|
+
|
|
336
|
+
todo('deepseek.chat.content_filter_finish', 'chat', "finish_reason 'content_filter' when DeepSeek's safety layer stops a generation", 'api', 'common'),
|
|
337
|
+
todo('deepseek.chat.insufficient_system_resource_finish', 'chat', "finish_reason 'insufficient_system_resource' — DeepSeek's own value, mapped by the SDK to the unified `error` reason", 'api', 'niche'),
|
|
338
|
+
todo('deepseek.chat.name_on_tool_message', 'chat', "How DeepSeek answers a `name` on a role:tool turn — it supports names on system/user/assistant only, and @ai-sdk/deepseek strips it client-side with a warning, so the server's own behaviour is unobserved", 'api', 'niche'),
|
|
339
|
+
// CORE, not common: every integration that grows a conversation hits the context ceiling in its
|
|
340
|
+
// first week, and the vendor's refusal is what a client must handle (§9 round one, NIT 16 — a
|
|
341
|
+
// manifest with ZERO core todos is a tiering smell, not an achievement).
|
|
342
|
+
todo('deepseek.chat.context_length_exceeded', 'chat', 'A prompt beyond the model context window is refused with the vendor-documented error', 'api', 'core'),
|
|
343
|
+
|
|
344
|
+
// ══ FIM completions (beta) ═══════════════════════════════════════════════════════════════
|
|
345
|
+
done('deepseek.completions.fim', 'completions', 'POST /beta/completions returns a faithful text_completion envelope', 'api', 'common', () => withRoot(async (h) => {
|
|
346
|
+
const r = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'def fib(n):\n ' } });
|
|
347
|
+
if (!ok(r)) return false;
|
|
348
|
+
const b = body(r);
|
|
349
|
+
return b.object === 'text_completion' && b.model === PRO && String(b.id).startsWith('cmpl-')
|
|
350
|
+
&& Array.isArray(b.choices) && typeof b.choices[0].text === 'string'
|
|
351
|
+
&& b.choices[0].text.includes('[twin-stub:deepseek-v4-pro]')
|
|
352
|
+
&& b.choices[0].finish_reason === 'stop' && b.choices[0].logprobs === null
|
|
353
|
+
// FIM's response carries the same KV-cache usage split as chat.
|
|
354
|
+
&& typeof b.usage.prompt_cache_miss_tokens === 'number';
|
|
355
|
+
})),
|
|
356
|
+
|
|
357
|
+
done('deepseek.completions.suffix_and_echo', 'completions', 'suffix is honoured and echo prepends the prompt to the returned text', 'api', 'common', () => withRoot(async (h) => {
|
|
358
|
+
const withSuffix = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'a=', suffix: 'return a' } });
|
|
359
|
+
const noSuffix = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'a=' } });
|
|
360
|
+
if (!ok(withSuffix) || !ok(noSuffix)) return false;
|
|
361
|
+
if (!String(body(withSuffix).choices[0].text).includes('return a')) return false;
|
|
362
|
+
if (String(noSuffix.body && body(noSuffix).choices[0].text).includes('return a')) return false;
|
|
363
|
+
const echoed = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'PROMPT_HEAD', echo: true } });
|
|
364
|
+
const plain = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'PROMPT_HEAD' } });
|
|
365
|
+
return ok(echoed) && String(body(echoed).choices[0].text).startsWith('PROMPT_HEAD')
|
|
366
|
+
&& !String(body(plain).choices[0].text).startsWith('PROMPT_HEAD');
|
|
367
|
+
})),
|
|
368
|
+
|
|
369
|
+
done('deepseek.completions.beta_only_and_closed_model', 'completions', 'FIM lives only under /beta and accepts only deepseek-v4-pro', 'api', 'common', () => withRoot(async (h) => {
|
|
370
|
+
const offBeta = await h({ m: 'POST', p: '/completions', b: { model: PRO, prompt: 'x' } });
|
|
371
|
+
const wrongModel = await h({ m: 'POST', p: FIM_PATH, b: { model: MODEL, prompt: 'x' } });
|
|
372
|
+
const noPrompt = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO } });
|
|
373
|
+
return refused(offBeta) && msg(offBeta).includes('beta base URL')
|
|
374
|
+
&& wrongModel.status === 422 && noPrompt.status === 422;
|
|
375
|
+
})),
|
|
376
|
+
|
|
377
|
+
done('deepseek.completions.logprobs_is_an_integer', 'completions', "The FIM endpoint's `logprobs` is an INTEGER (0..20), unlike the chat endpoint's boolean of the same name", 'api', 'niche', () => withRoot(async (h) => {
|
|
378
|
+
// Same key name, different type, on the same vendor. An OpenAI-shaped twin that shared one
|
|
379
|
+
// validator across both endpoints would accept the wrong type on one of them.
|
|
380
|
+
const good = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', logprobs: 5 } });
|
|
381
|
+
const boolean = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', logprobs: true } });
|
|
382
|
+
const tooMany = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', logprobs: 21 } });
|
|
383
|
+
// …and the chat endpoint is the exact inverse.
|
|
384
|
+
const chatBool = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: true }) });
|
|
385
|
+
const chatInt = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ logprobs: 5 }) });
|
|
386
|
+
return ok(good) && boolean.status === 422 && tooMany.status === 422 && ok(chatBool) && chatInt.status === 422;
|
|
387
|
+
})),
|
|
388
|
+
|
|
389
|
+
done('deepseek.completions.refuses_unmodeled_streaming', 'completions', 'A FIM request asking to stream is REFUSED by name rather than answered with a unary body', 'api', 'common', () => withRoot(async (h) => {
|
|
390
|
+
// The endpoint really does declare `stream`, so serving a unary text_completion to a caller who
|
|
391
|
+
// asked for SSE would be a fake success — the exact failure mode "unmodeled ops fail like the
|
|
392
|
+
// vendor" exists to prevent. The refusal names the filed gap so the message is actionable.
|
|
393
|
+
const streamed = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', stream: true } });
|
|
394
|
+
if (streamed.status !== 422 || !msg(streamed).includes('deepseek.completions.streaming')) return false;
|
|
395
|
+
// …and the non-streaming path is unaffected, or the refusal would just be a broken endpoint.
|
|
396
|
+
const unary = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', stream: false } });
|
|
397
|
+
return ok(unary) && body(unary).object === 'text_completion';
|
|
398
|
+
})),
|
|
399
|
+
todo('deepseek.completions.cache_accounting', 'completions', 'FIM requests report prompt_cache_hit_tokens as a hard zero and record no prefix unit — the field is served in the faithful shape but is not measured on this endpoint the way it is on /chat/completions (§9 round one, NIT 18)', 'api', 'niche'),
|
|
400
|
+
todo('deepseek.completions.streaming', 'completions', 'FIM completion streaming (stream + stream_options.include_usage on /beta/completions) — currently refused by name rather than modeled', 'api', 'niche'),
|
|
401
|
+
|
|
402
|
+
// ══ models ═══════════════════════════════════════════════════════════════════════════════
|
|
403
|
+
done('deepseek.models.list', 'models', 'GET /models lists the published catalog', 'api', 'core', () => withRoot(async (h) => {
|
|
404
|
+
const r = await h({ m: 'GET', p: MODELS });
|
|
405
|
+
if (!ok(r)) return false;
|
|
406
|
+
const b = body(r);
|
|
407
|
+
const ids = (b.data as Body[]).map((m) => m.id);
|
|
408
|
+
return b.object === 'list' && ids.includes('deepseek-v4-flash') && ids.includes('deepseek-v4-pro')
|
|
409
|
+
&& ids.includes('deepseek-v4-flash-vision-exp')
|
|
410
|
+
// The retired aliases must NOT be listed.
|
|
411
|
+
&& !ids.includes('deepseek-chat') && !ids.includes('deepseek-reasoner');
|
|
412
|
+
})),
|
|
413
|
+
|
|
414
|
+
done('deepseek.models.row_shape', 'models', "A model row is exactly {id, object, owned_by} — DeepSeek's row has no `created` (OpenAI's does)", 'api', 'common', () => withRoot(async (h) => {
|
|
415
|
+
const r = await h({ m: 'GET', p: MODELS });
|
|
416
|
+
if (!ok(r)) return false;
|
|
417
|
+
// A LITERAL expected key set, licensed by §6 because DeepSeek's own list-models example
|
|
418
|
+
// response is a closed three-key object. Inventing `created` to look OpenAI-shaped is exactly
|
|
419
|
+
// the inverse false-green.
|
|
420
|
+
return (body(r).data as Body[]).every((m) => Object.keys(m).sort().join(',') === 'id,object,owned_by' && m.object === 'model' && m.owned_by === 'deepseek');
|
|
421
|
+
})),
|
|
422
|
+
|
|
423
|
+
done('deepseek.models.retired_and_unknown_ids', 'models', 'A retired alias is refused BY NAME with its retirement date; an unknown id is refused too', 'api', 'core', () => withRoot(async (h) => {
|
|
424
|
+
for (const retired of ['deepseek-chat', 'deepseek-reasoner']) {
|
|
425
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ model: retired }) });
|
|
426
|
+
if (r.status !== 422 || !msg(r).includes('2026-07-24')) return false;
|
|
427
|
+
}
|
|
428
|
+
const unknown = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ model: 'gpt-4o' }) });
|
|
429
|
+
return unknown.status === 422 && msg(unknown).includes('does not exist');
|
|
430
|
+
})),
|
|
431
|
+
|
|
432
|
+
done('deepseek.models.pulled_rows_are_served', 'models', "A model observed by a connector pull is served by GET /models, not merely folded into the log", 'api', 'niche', () => withConnectorRoot('deepseek.models.pulled_rows_are_served', async (root) => {
|
|
433
|
+
const execute = fakeExecute({ models: [{ id: 'deepseek-v4-preview-internal', object: 'model', owned_by: 'deepseek' }] });
|
|
434
|
+
await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
|
|
435
|
+
const r = await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, root });
|
|
436
|
+
const ids = (body(r).data as Body[]).map((m) => m.id);
|
|
437
|
+
// Both the pulled row AND the published catalog must survive: an override that dropped either
|
|
438
|
+
// side makes the twin disagree with itself.
|
|
439
|
+
return ok(r) && ids.includes('deepseek-v4-preview-internal') && ids.includes('deepseek-v4-flash');
|
|
440
|
+
})),
|
|
441
|
+
|
|
442
|
+
todo('deepseek.models.per_model_metadata', 'models', 'Per-model metadata (context window, pricing tier, concurrency) — DeepSeek publishes these on its pricing page but not on the API row', 'api', 'niche'),
|
|
443
|
+
|
|
444
|
+
// ══ user balance ═════════════════════════════════════════════════════════════════════════
|
|
445
|
+
done('deepseek.balance.get', 'balance', 'GET /user/balance returns is_available plus per-currency balance_infos with STRING amounts', 'api', 'common', () => withRoot(async (h) => {
|
|
446
|
+
const r = await h({ m: 'GET', p: BALANCE });
|
|
447
|
+
if (!ok(r)) return false;
|
|
448
|
+
const b = body(r);
|
|
449
|
+
const info = (b.balance_infos as Body[])[0];
|
|
450
|
+
return b.is_available === true && Array.isArray(b.balance_infos) && b.balance_infos.length > 0
|
|
451
|
+
&& ['CNY', 'USD'].includes(info.currency)
|
|
452
|
+
// Amounts are STRINGS on this vendor. A twin that emitted numbers would break any client
|
|
453
|
+
// decoding them as the documented type.
|
|
454
|
+
&& typeof info.total_balance === 'string' && typeof info.granted_balance === 'string' && typeof info.topped_up_balance === 'string';
|
|
455
|
+
})),
|
|
456
|
+
|
|
457
|
+
done('deepseek.balance.insufficient_402', 'balance', 'A drained account refuses completions with 402 "You have run out of balance" — a status OpenAI does not have', 'api', 'core', () => withRoot(async (h, root) => {
|
|
458
|
+
const before = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
459
|
+
if (!ok(before)) return false;
|
|
460
|
+
// STATE-DRIVEN, not a header trick: the refusal reads `is_available` off the same projection
|
|
461
|
+
// `GET /user/balance` serves, so it is a fact about the twin's state.
|
|
462
|
+
await applyTwinWrite('deepseek', {
|
|
463
|
+
operation: 'balance.update', subjectType: 'balance', subjectId: 'account',
|
|
464
|
+
fields: { is_available: false, balance_infos: [{ currency: 'USD', total_balance: '0.00', granted_balance: '0.00', topped_up_balance: '0.00' }] },
|
|
465
|
+
occurredAt: OCCURRED_AT, actor: { kind: 'agent' },
|
|
466
|
+
}, root);
|
|
467
|
+
const after = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
468
|
+
const fim = await h({ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x' } });
|
|
469
|
+
const balance = await h({ m: 'GET', p: BALANCE });
|
|
470
|
+
return after.status === 402 && msg(after).includes('run out of balance')
|
|
471
|
+
&& fim.status === 402
|
|
472
|
+
// Reads still work on a drained account — only generation is refused.
|
|
473
|
+
&& ok(balance) && body(balance).is_available === false;
|
|
474
|
+
})),
|
|
475
|
+
|
|
476
|
+
todo('deepseek.balance.multi_currency', 'balance', 'An account holding both CNY and USD balances reports both rows', 'api', 'niche'),
|
|
477
|
+
|
|
478
|
+
// ══ context cache (KV cache) ═════════════════════════════════════════════════════════════
|
|
479
|
+
done('deepseek.cache.hit_miss_split', 'cache', 'usage carries prompt_cache_hit_tokens + prompt_cache_miss_tokens, summing exactly to prompt_tokens', 'api', 'core', () => withRoot(async (h) => {
|
|
480
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
481
|
+
if (!ok(r)) return false;
|
|
482
|
+
const u = body(r).usage as Body;
|
|
483
|
+
return typeof u.prompt_cache_hit_tokens === 'number' && typeof u.prompt_cache_miss_tokens === 'number'
|
|
484
|
+
&& u.prompt_cache_hit_tokens + u.prompt_cache_miss_tokens === u.prompt_tokens
|
|
485
|
+
// …and the OpenAI-shaped mirror field agrees with the hit count, because the SDK's usage
|
|
486
|
+
// schema declares both and its `cacheRead` reads the DeepSeek one.
|
|
487
|
+
&& (u.prompt_tokens_details as Body).cached_tokens === u.prompt_cache_hit_tokens;
|
|
488
|
+
})),
|
|
489
|
+
|
|
490
|
+
done('deepseek.cache.prefix_hit_on_continuation', 'cache', "A follow-up turn genuinely HITS the cache: the twin keeps a prefix-unit ledger rather than fabricating a split", 'api', 'core', () => withRoot(async (h) => {
|
|
491
|
+
const first = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'a long opening question about caching behaviour' }] } });
|
|
492
|
+
if (!ok(first)) return false;
|
|
493
|
+
// A brand-new conversation can hit nothing.
|
|
494
|
+
if ((body(first).usage as Body).prompt_cache_hit_tokens !== 0) return false;
|
|
495
|
+
const answer = text0(first);
|
|
496
|
+
const second = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [
|
|
497
|
+
{ role: 'user', content: 'a long opening question about caching behaviour' },
|
|
498
|
+
{ role: 'assistant', content: answer, reasoning_content: '' },
|
|
499
|
+
{ role: 'user', content: 'and a follow-up' },
|
|
500
|
+
] } });
|
|
501
|
+
if (!ok(second)) return false;
|
|
502
|
+
const u = body(second).usage as Body;
|
|
503
|
+
// The continuation's recorded prefix (opening + answer) is reported as a hit and the NEW turn
|
|
504
|
+
// as a miss. `miss > 0` is the assertion that matters: an implementation measuring the hit off
|
|
505
|
+
// the stored prefix text instead of this request's own messages over-counts, the clamp in
|
|
506
|
+
// `buildUsage` rescues the invariant, and the follow-up reports a 100% cache with the new turn
|
|
507
|
+
// silently absent from the accounting. That was a real defect this cell caught.
|
|
508
|
+
return u.prompt_cache_hit_tokens > 0 && u.prompt_cache_miss_tokens > 0
|
|
509
|
+
&& u.prompt_cache_hit_tokens + u.prompt_cache_miss_tokens === u.prompt_tokens;
|
|
510
|
+
})),
|
|
511
|
+
|
|
512
|
+
done('deepseek.cache.unrelated_prompt_misses', 'cache', 'An unrelated conversation reports a full miss even after the ledger has entries', 'api', 'common', () => withRoot(async (h) => {
|
|
513
|
+
// The other direction, and the one a fabricated split would fail: having served one
|
|
514
|
+
// conversation, a DIFFERENT one must still be 100% miss. A twin that reported a fixed
|
|
515
|
+
// hit ratio would pass the hit test and fail this one.
|
|
516
|
+
await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'topic one' }] } });
|
|
517
|
+
const other = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'a completely different topic' }] } });
|
|
518
|
+
if (!ok(other)) return false;
|
|
519
|
+
const u = body(other).usage as Body;
|
|
520
|
+
return u.prompt_cache_hit_tokens === 0 && u.prompt_cache_miss_tokens === u.prompt_tokens && u.prompt_tokens > 0;
|
|
521
|
+
})),
|
|
522
|
+
|
|
523
|
+
done('deepseek.cache.identical_replay_hits', 'cache', 'A byte-identical repeat of a request FULLY MATCHES the user-input-end prefix unit the first call recorded, and reports a full cache hit', 'api', 'common', () => withRoot(async (h) => {
|
|
524
|
+
// The canonical KV-cache demonstration, and the one the twin used to get wrong: it stopped its
|
|
525
|
+
// prefix search one message short, so an identical replay reported 0 hit tokens. The vendor's
|
|
526
|
+
// rule is that a request hits when it "fully matches a cache prefix unit", and units form at
|
|
527
|
+
// "the end position of the user input" (api-docs.deepseek.com/guides/kv_cache).
|
|
528
|
+
const req = { model: MODEL, messages: [{ role: 'user', content: 'a repeatable single-turn question' }] };
|
|
529
|
+
const first = await h({ m: 'POST', p: CHAT_PATH, b: req });
|
|
530
|
+
if (!ok(first) || (body(first).usage as Body).prompt_cache_hit_tokens !== 0) return false;
|
|
531
|
+
const second = await h({ m: 'POST', p: CHAT_PATH, b: req });
|
|
532
|
+
if (!ok(second)) return false;
|
|
533
|
+
const u = body(second).usage as Body;
|
|
534
|
+
// A FULL match: every prompt token is a hit, and none is a miss.
|
|
535
|
+
if (u.prompt_cache_hit_tokens !== u.prompt_tokens || u.prompt_cache_miss_tokens !== 0 || u.prompt_tokens === 0) return false;
|
|
536
|
+
// …and this is not a "second call always hits" rule: a DIFFERENT request in the same warmed
|
|
537
|
+
// root still misses completely.
|
|
538
|
+
const other = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'an entirely different question' }] } });
|
|
539
|
+
return ok(other) && (body(other).usage as Body).prompt_cache_hit_tokens === 0;
|
|
540
|
+
})),
|
|
541
|
+
|
|
542
|
+
done('deepseek.cache.key_covers_every_charged_field', 'cache', 'A turn carrying a different reasoning_content, name or tool_call is a MISS — the prefix key covers EVERY field the token count charges for', 'api', 'common', () => withRoot(async (h) => {
|
|
543
|
+
// §9 round one, NIT 15: the key dropped `name` and `reasoning_content` while `messageTokens`
|
|
544
|
+
// counted them, so a caller could attach an arbitrarily large reasoning_content to a matching
|
|
545
|
+
// turn and have those tokens billed as a hit.
|
|
546
|
+
const opening = { model: MODEL, messages: [{ role: 'user', content: 'opening' }] };
|
|
547
|
+
const seeded = await h({ m: 'POST', p: CHAT_PATH, b: opening });
|
|
548
|
+
if (!ok(seeded)) return false;
|
|
549
|
+
const answer = text0(seeded);
|
|
550
|
+
const cont = (assistant: Record<string, unknown>) => ({ model: MODEL, messages: [{ role: 'user', content: 'opening' }, assistant, { role: 'user', content: 'next' }] });
|
|
551
|
+
// A SECOND recorded shape whose assistant turn carries a tool call. Without it the padded-id
|
|
552
|
+
// case below differs from every recorded unit in TWO ways (it has tool_calls at all, and the id
|
|
553
|
+
// is long), so its miss would be explained by the first difference and the cell would stay green
|
|
554
|
+
// with the id dropped from the key — which is exactly how the first version of this pin came out
|
|
555
|
+
// HOLLOW on the revert matrix. This baseline makes the id the ONLY difference.
|
|
556
|
+
const TOOL_TURN = { role: 'assistant', content: answer, tool_calls: [{ id: 'call_twin_1', type: 'function', function: { name: 'f', arguments: '{}' } }] };
|
|
557
|
+
const toolBaseline = await h({ m: 'POST', p: CHAT_PATH, b: cont(TOOL_TURN) });
|
|
558
|
+
if (!ok(toolBaseline) || (body(toolBaseline).usage as Body).prompt_cache_miss_tokens > 200) return false;
|
|
559
|
+
// An absent reasoning_content and an empty one are the SAME key — that is the shape
|
|
560
|
+
// @ai-sdk/deepseek produces, so a stricter key would make every real SDK continuation a miss.
|
|
561
|
+
const plain = await h({ m: 'POST', p: CHAT_PATH, b: cont({ role: 'assistant', content: answer }) });
|
|
562
|
+
const empty = await h({ m: 'POST', p: CHAT_PATH, b: cont({ role: 'assistant', content: answer, reasoning_content: '' }) });
|
|
563
|
+
if (!ok(plain) || !ok(empty)) return false;
|
|
564
|
+
if ((body(plain).usage as Body).prompt_cache_hit_tokens === 0) return false;
|
|
565
|
+
// `empty` runs AFTER `plain` in the same root, so if the two canonicalize identically it fully
|
|
566
|
+
// matches the 3-message unit `plain` just recorded — a zero miss. If they canonicalized
|
|
567
|
+
// differently it would have to fall back to the shorter prefix and report a miss, which is
|
|
568
|
+
// exactly what a key that ignored `reasoning_content` inconsistently would produce.
|
|
569
|
+
if ((body(empty).usage as Body).prompt_cache_miss_tokens !== 0) return false;
|
|
570
|
+
// …but a turn carrying a LARGE payload in ANY field the token count charges for is a different
|
|
571
|
+
// conversation, and must not be billed as a hit against the unit recorded for the plain one.
|
|
572
|
+
// THREE fields, one per charged term in `messageTokens` — §9 round two, BLOCKER 1: the first
|
|
573
|
+
// version varied only `reasoning_content`, so `tool_calls[].id` (which `messageTokens` charges
|
|
574
|
+
// through `JSON.stringify(tc)`) slipped through and 999 tokens of caller padding were billed as
|
|
575
|
+
// a hit. A cell that proves one charged field says nothing about the others.
|
|
576
|
+
const fatCases: Array<[string, Record<string, unknown>]> = [
|
|
577
|
+
['reasoning_content', { role: 'assistant', content: answer, reasoning_content: 'x'.repeat(4000) }],
|
|
578
|
+
['name', { role: 'assistant', content: answer, name: 'n'.repeat(4000) }],
|
|
579
|
+
// Identical to TOOL_TURN except for the id — so a key that omitted the id would FULLY match
|
|
580
|
+
// the unit `toolBaseline` just recorded and bill all 4000 characters as a hit.
|
|
581
|
+
['tool_calls[].id', { ...TOOL_TURN, tool_calls: [{ id: `call_${'z'.repeat(4000)}`, type: 'function', function: { name: 'f', arguments: '{}' } }] }],
|
|
582
|
+
];
|
|
583
|
+
for (const [, assistant] of fatCases) {
|
|
584
|
+
const fat = await h({ m: 'POST', p: CHAT_PATH, b: cont(assistant) });
|
|
585
|
+
if (!ok(fat)) return false;
|
|
586
|
+
const fu = body(fat).usage as Body;
|
|
587
|
+
if (fu.prompt_cache_hit_tokens >= (body(toolBaseline).usage as Body).prompt_tokens + 10) return false;
|
|
588
|
+
if (fu.prompt_cache_miss_tokens <= 900) return false;
|
|
589
|
+
if (fu.prompt_cache_hit_tokens + fu.prompt_cache_miss_tokens !== fu.prompt_tokens) return false;
|
|
590
|
+
}
|
|
591
|
+
return true;
|
|
592
|
+
})),
|
|
593
|
+
|
|
594
|
+
todo('deepseek.cache.interval_units', 'cache', 'Cache prefix units cut at "fixed token intervals" inside long inputs — DeepSeek publishes no interval, so the twin does not invent one', 'api', 'niche'),
|
|
595
|
+
todo('deepseek.cache.expiry', 'cache', 'Cache entries auto-clearing "within hours to days": the kernel\'s world clock is deterministic by construction and is explicitly the seam TTL/expiry logic reads at serve time, so this is unbuilt rather than impossible (§9 round one, SHOULD-FIX 5)', 'api', 'niche'),
|
|
596
|
+
|
|
597
|
+
// ══ reasoning / thinking mode ════════════════════════════════════════════════════════════
|
|
598
|
+
done('deepseek.reasoning.enabled_by_default', 'reasoning', 'V4 models think by default: the assistant turn carries reasoning_content and reasoning_tokens', 'api', 'core', () => withRoot(async (h) => {
|
|
599
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
600
|
+
if (!ok(r)) return false;
|
|
601
|
+
const rc = choice0(r).message.reasoning_content;
|
|
602
|
+
return typeof rc === 'string' && rc.includes('[twin-stub:') && rc.includes('reasoning_effort=high')
|
|
603
|
+
&& (body(r).usage.completion_tokens_details as Body)?.reasoning_tokens > 0;
|
|
604
|
+
})),
|
|
605
|
+
|
|
606
|
+
done('deepseek.reasoning.disabled', 'reasoning', "thinking.type 'disabled' removes reasoning_content and its token accounting", 'api', 'core', () => withRoot(async (h) => {
|
|
607
|
+
const off = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ thinking: { type: 'disabled' } }) });
|
|
608
|
+
if (!ok(off)) return false;
|
|
609
|
+
if (choice0(off).message.reasoning_content !== undefined) return false;
|
|
610
|
+
if (body(off).usage.completion_tokens_details !== undefined) return false;
|
|
611
|
+
// …and the closed set is enforced: `adaptive` is a legacy SDK-side value the API does not take.
|
|
612
|
+
const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ thinking: { type: 'adaptive' } }) });
|
|
613
|
+
return bad.status === 422;
|
|
614
|
+
})),
|
|
615
|
+
|
|
616
|
+
done('deepseek.reasoning.effort', 'reasoning', "reasoning_effort is the closed set low|high|max, accepted at BOTH documented placements, with medium/xhigh mapped as the vendor documents", 'api', 'common', () => withRoot(async (h) => {
|
|
617
|
+
// Two first-party sources disagree on placement — the API reference puts it at
|
|
618
|
+
// `thinking.reasoning_effort`, `@ai-sdk/deepseek` sends a top-level `reasoning_effort`. The twin
|
|
619
|
+
// accepts BOTH rather than rejecting a shape a first-party client demonstrably sends.
|
|
620
|
+
const top = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ reasoning_effort: 'max' }) });
|
|
621
|
+
const nested = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ thinking: { type: 'enabled', reasoning_effort: 'max' } }) });
|
|
622
|
+
if (!ok(top) || !ok(nested)) return false;
|
|
623
|
+
// The effort must reach the turn — a handler that parsed and dropped it would serve `high`.
|
|
624
|
+
if (!String(choice0(top).message.reasoning_content).includes('reasoning_effort=max')) return false;
|
|
625
|
+
if (String(choice0(nested).message.reasoning_content) !== String(choice0(top).message.reasoning_content)) return false;
|
|
626
|
+
// Legacy values the vendor documents as MAPPED to high, not refused.
|
|
627
|
+
const legacy = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ reasoning_effort: 'medium' }) });
|
|
628
|
+
if (!ok(legacy) || !String(choice0(legacy).message.reasoning_content).includes('reasoning_effort=high')) return false;
|
|
629
|
+
const bad = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ reasoning_effort: 'ultra' }) });
|
|
630
|
+
return bad.status === 422;
|
|
631
|
+
})),
|
|
632
|
+
|
|
633
|
+
done('deepseek.reasoning.tools_require_handback', 'reasoning', 'With `tools` set, a previous assistant turn missing reasoning_content is a documented 400 — not a 422 and not a silent success', 'api', 'core', () => withRoot(async (h) => {
|
|
634
|
+
// "If the request carries the tools parameter: the reasoning_content of all previous turns
|
|
635
|
+
// should be passed back … the API will return a 400 error"
|
|
636
|
+
// (api-docs.deepseek.com/guides/thinking_mode). This is the one refusal on this vendor that is
|
|
637
|
+
// a 400 rather than a 422, and no OpenAI-shaped twin has it at all.
|
|
638
|
+
const history = [{ role: 'user', content: 'q' }, { role: 'assistant', content: 'a' }, { role: 'user', content: 'q2' }];
|
|
639
|
+
const missing = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: history, tools: [WEATHER_TOOL] } });
|
|
640
|
+
if (missing.status !== 400) return false;
|
|
641
|
+
// An EMPTY STRING counts as passed back — that is literally what @ai-sdk/deepseek sends for a
|
|
642
|
+
// turn with no reasoning, so a stricter check would refuse the real SDK.
|
|
643
|
+
const empty = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, tools: [WEATHER_TOOL], messages: [
|
|
644
|
+
{ role: 'user', content: 'q' }, { role: 'assistant', content: 'a', reasoning_content: '' }, { role: 'user', content: 'q2' },
|
|
645
|
+
] } });
|
|
646
|
+
// …and WITHOUT tools the same history is fine: the rule is scoped to tool requests.
|
|
647
|
+
const noTools = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: history } });
|
|
648
|
+
return ok(empty) && ok(noTools);
|
|
649
|
+
})),
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
// ══ streaming ════════════════════════════════════════════════════════════════════════════
|
|
653
|
+
done('deepseek.streaming.chat', 'streaming', 'Streaming emits the vendor chunk sequence and terminates with [DONE]', 'api', 'core', () => withStream({ ...CHAT(), stream: true }, (events) => {
|
|
654
|
+
const data = events.filter((e) => !e.done).map((e) => e.data as Body);
|
|
655
|
+
if (!events.length || events[events.length - 1]!.done !== true) return false;
|
|
656
|
+
if (data[0]?.object !== 'chat.completion.chunk') return false;
|
|
657
|
+
if ((data[0]!.choices as Body[])[0].delta.role !== 'assistant') return false;
|
|
658
|
+
const text = data.map((d) => (d.choices as Body[])[0]?.delta?.content ?? '').join('');
|
|
659
|
+
const finish = data.find((d) => (d.choices as Body[])[0]?.finish_reason != null);
|
|
660
|
+
return text.includes('[twin-stub:deepseek-v4-flash]') && text.includes('hello twin')
|
|
661
|
+
&& (finish!.choices as Body[])[0].finish_reason === 'stop';
|
|
662
|
+
})),
|
|
663
|
+
|
|
664
|
+
done('deepseek.streaming.reasoning_before_text', 'streaming', 'reasoning_content deltas arrive BEFORE the first content delta and never interleave', 'api', 'core', () => withStream({ ...CHAT(), stream: true }, (events) => {
|
|
665
|
+
const data = events.filter((e) => !e.done).map((e) => e.data as Body);
|
|
666
|
+
const kinds = data.map((d) => {
|
|
667
|
+
const delta = (d.choices as Body[])[0]?.delta as Body | undefined;
|
|
668
|
+
if (typeof delta?.reasoning_content === 'string') return 'r';
|
|
669
|
+
if (typeof delta?.content === 'string' && delta.content !== '') return 'c';
|
|
670
|
+
return '.';
|
|
671
|
+
}).join('');
|
|
672
|
+
const lastR = kinds.lastIndexOf('r');
|
|
673
|
+
const firstC = kinds.indexOf('c');
|
|
674
|
+
// The SDK's transform closes its reasoning part the moment the first content delta arrives, so
|
|
675
|
+
// interleaving would produce a different part sequence than the vendor's.
|
|
676
|
+
return lastR >= 0 && firstC >= 0 && lastR < firstC;
|
|
677
|
+
})),
|
|
678
|
+
|
|
679
|
+
done('deepseek.streaming.usage_tail_is_opt_in', 'streaming', 'The empty-choices usage tail chunk is emitted only when stream_options.include_usage is set', 'api', 'core', async () => {
|
|
680
|
+
const withUsage = await withStream({ ...CHAT(), stream: true, stream_options: { include_usage: true } }, (events) => {
|
|
681
|
+
const tail = events[events.length - 2]?.data as Body | undefined;
|
|
682
|
+
return !!tail && Array.isArray(tail.choices) && tail.choices.length === 0
|
|
683
|
+
&& typeof (tail.usage as Body)?.prompt_cache_miss_tokens === 'number'
|
|
684
|
+
&& (tail.usage as Body).total_tokens > 0;
|
|
685
|
+
});
|
|
686
|
+
if (!withUsage) return false;
|
|
687
|
+
// The negative half: without the option there must be NO usage chunk at all. A twin that always
|
|
688
|
+
// emitted it would look right to the SDK (which always asks for it) and be wrong for everyone else.
|
|
689
|
+
return withStream({ ...CHAT(), stream: true }, (events) => !events.some((e) => !e.done && (e.data as Body)?.usage !== undefined));
|
|
690
|
+
}),
|
|
691
|
+
|
|
692
|
+
done('deepseek.streaming.tool_call_deltas', 'streaming', 'Tool calls stream as indexed name-then-arguments deltas', 'api', 'common', () => withStream({ ...CHAT(), tools: [WEATHER_TOOL], stream: true }, (events) => {
|
|
693
|
+
const data = events.filter((e) => !e.done).map((e) => e.data as Body);
|
|
694
|
+
const deltas = data.flatMap((d) => ((d.choices as Body[])[0]?.delta?.tool_calls ?? []) as Body[]);
|
|
695
|
+
if (deltas.length < 2) return false;
|
|
696
|
+
const named = deltas.find((t) => t.function?.name === 'get_weather');
|
|
697
|
+
const args = deltas.filter((t) => typeof t.function?.arguments === 'string' && t.function.arguments !== '');
|
|
698
|
+
const finish = data.find((d) => (d.choices as Body[])[0]?.finish_reason != null);
|
|
699
|
+
return !!named && named.index === 0 && typeof named.id === 'string' && named.type === 'function'
|
|
700
|
+
&& args.length > 0 && JSON.parse(args[args.length - 1]!.function.arguments).city === ''
|
|
701
|
+
&& (finish!.choices as Body[])[0].finish_reason === 'tool_calls';
|
|
702
|
+
})),
|
|
703
|
+
|
|
704
|
+
todo('deepseek.streaming.stream_options_without_stream', 'streaming', "How DeepSeek answers `stream_options` on a NON-streaming request — the reference table says only 'set when stream: true', which is guidance rather than a documented refusal, so the twin accepts and ignores it rather than inventing a 422 (§9 round one, NIT 13)", 'api', 'niche'),
|
|
705
|
+
|
|
706
|
+
done('deepseek.streaming.never_degrades_to_a_unary_body', 'streaming', 'A `stream: true` request is never answered with a unary JSON body — not through a repeated-slash path spelling, and not when a caller supplies no sink', 'api', 'core', async () => {
|
|
707
|
+
// §9 ROUND TWO, SHOULD-FIX 1. Two independent holes, both fake successes of exactly the kind
|
|
708
|
+
// `deepseek.completions.refuses_unmodeled_streaming` forbids on the FIM endpoint:
|
|
709
|
+
// • the handler fell through to the unary builder whenever an sseSink was absent;
|
|
710
|
+
// • the SERVER's STREAMABLE lookup collapsed only a TRAILING slash while the router collapsed
|
|
711
|
+
// REPEATED ones, so `/beta//chat/completions` reached the streaming route, missed the
|
|
712
|
+
// streaming response path, and answered `application/json` to a client reading SSE.
|
|
713
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
714
|
+
const { createDeepSeekTwinServer } = await import('./deepseek-server.ts');
|
|
715
|
+
const server = await createDeepSeekTwinServer({ root });
|
|
716
|
+
try {
|
|
717
|
+
return await verifyBoundary('deepseek.streaming.never_degrades_to_a_unary_body', async () => {
|
|
718
|
+
// The handler half: a streaming request with no sink is refused, not downgraded.
|
|
719
|
+
const noSink = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify({ ...CHAT(), stream: true }), root, occurredAt: OCCURRED_AT });
|
|
720
|
+
if (noSink.status !== 422 || (body(noSink) as Body).object === 'chat.completion') return false;
|
|
721
|
+
// The server half: every spelling that REACHES the streaming route must stream.
|
|
722
|
+
const base = `http://127.0.0.1:${server.port}`;
|
|
723
|
+
for (const path of ['/chat/completions', '//chat/completions', '/beta/chat/completions', '/beta//chat/completions']) {
|
|
724
|
+
const res = await fetch(`${base}${path}`, {
|
|
725
|
+
method: 'POST',
|
|
726
|
+
headers: { 'content-type': 'application/json', authorization: 'Bearer sk-x' },
|
|
727
|
+
body: JSON.stringify({ ...CHAT(), stream: true }),
|
|
728
|
+
});
|
|
729
|
+
const ct = res.headers.get('content-type') ?? '';
|
|
730
|
+
const text = await res.text();
|
|
731
|
+
if (!ct.includes('text/event-stream')) return false;
|
|
732
|
+
if (!text.trimEnd().endsWith('data: [DONE]')) return false;
|
|
733
|
+
}
|
|
734
|
+
return true;
|
|
735
|
+
});
|
|
736
|
+
} finally { server.stop(); rmSync(root, { recursive: true, force: true }); }
|
|
737
|
+
}),
|
|
738
|
+
|
|
739
|
+
todo('deepseek.streaming.error_chunk', 'streaming', 'A mid-stream failure emits DeepSeek\'s error envelope as an SSE data event with the retryable metadata the SDK discriminates on', 'api', 'niche'),
|
|
740
|
+
|
|
741
|
+
// ══ tools / function calling ═════════════════════════════════════════════════════════════
|
|
742
|
+
done('deepseek.tools.function_calling', 'tools', 'Providing tools yields tool_calls with schema-shaped arguments and finish_reason tool_calls', 'api', 'core', () => withRoot(async (h) => {
|
|
743
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL] } });
|
|
744
|
+
if (!ok(r)) return false;
|
|
745
|
+
const calls = choice0(r).message.tool_calls as Body[];
|
|
746
|
+
if (!Array.isArray(calls) || calls.length !== 1) return false;
|
|
747
|
+
const args = JSON.parse(calls[0]!.function.arguments);
|
|
748
|
+
return calls[0]!.type === 'function' && calls[0]!.function.name === 'get_weather'
|
|
749
|
+
&& typeof calls[0]!.id === 'string' && calls[0]!.id.length > 0
|
|
750
|
+
// Arguments must contain every declared property with a type-appropriate placeholder.
|
|
751
|
+
&& args.city === '' && args.days === 0
|
|
752
|
+
&& choice0(r).message.content === null && choice0(r).finish_reason === 'tool_calls';
|
|
753
|
+
})),
|
|
754
|
+
|
|
755
|
+
done('deepseek.tools.tool_choice', 'tools', "tool_choice none suppresses calls, a named function forces that one, and an unknown value is refused", 'api', 'common', () => withRoot(async (h) => {
|
|
756
|
+
const none = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL], tool_choice: 'none' } });
|
|
757
|
+
if (!ok(none) || none.body === undefined) return false;
|
|
758
|
+
if (choice0(none).message.tool_calls !== undefined || choice0(none).finish_reason !== 'stop') return false;
|
|
759
|
+
const named = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL, TIME_TOOL], tool_choice: { type: 'function', function: { name: 'get_time' } } } });
|
|
760
|
+
if (!ok(named)) return false;
|
|
761
|
+
const calls = choice0(named).message.tool_calls as Body[];
|
|
762
|
+
// Exactly the NAMED tool, not the first one — a handler ignoring tool_choice returns get_weather.
|
|
763
|
+
if (calls.length !== 1 || calls[0]!.function.name !== 'get_time') return false;
|
|
764
|
+
const bad = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL], tool_choice: 'any' } });
|
|
765
|
+
const badNamed = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [WEATHER_TOOL], tool_choice: { type: 'function' } } });
|
|
766
|
+
return bad.status === 422 && badNamed.status === 422;
|
|
767
|
+
})),
|
|
768
|
+
|
|
769
|
+
done('deepseek.tools.tool_result_turn', 'tools', 'A role:tool result turn is accepted and folded into the prompt', 'api', 'common', () => withRoot(async (h) => {
|
|
770
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [
|
|
771
|
+
{ role: 'user', content: 'weather in Lisbon?' },
|
|
772
|
+
{ role: 'assistant', content: null, reasoning_content: '', tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'get_weather', arguments: '{"city":"Lisbon"}' } }] },
|
|
773
|
+
{ role: 'tool', tool_call_id: 'call_1', content: '{"temp":21}' },
|
|
774
|
+
] } });
|
|
775
|
+
if (!ok(r)) return false;
|
|
776
|
+
// The tool turn is real payload: it must be counted, so a handler dropping it reports fewer tokens.
|
|
777
|
+
const withoutTool = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'weather in Lisbon?' }] } });
|
|
778
|
+
return body(r).usage.prompt_tokens > body(withoutTool).usage.prompt_tokens && choice0(r).finish_reason === 'stop';
|
|
779
|
+
})),
|
|
780
|
+
|
|
781
|
+
done('deepseek.tools.max_128', 'tools', 'More than 128 function tools is refused', 'api', 'niche', () => withRoot(async (h) => {
|
|
782
|
+
const tool = (n: number) => ({ type: 'function', function: { name: `f${n}`, parameters: { type: 'object', properties: {} } } });
|
|
783
|
+
const at = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: Array.from({ length: 128 }, (_, i) => tool(i)) } });
|
|
784
|
+
const over = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: Array.from({ length: 129 }, (_, i) => tool(i)) } });
|
|
785
|
+
const malformed = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [{ type: 'retrieval' }] } });
|
|
786
|
+
return ok(at) && over.status === 422 && malformed.status === 422;
|
|
787
|
+
})),
|
|
788
|
+
|
|
789
|
+
done('deepseek.tools.strict_is_beta_only_and_all_or_nothing', 'tools', 'strict:true requires the /beta base URL, and inside it every function tool must be strict', 'api', 'common', () => withRoot(async (h) => {
|
|
790
|
+
const strict = { type: 'function', function: { name: 'a', strict: true, parameters: { type: 'object', properties: {} } } };
|
|
791
|
+
const loose = { type: 'function', function: { name: 'b', parameters: { type: 'object', properties: {} } } };
|
|
792
|
+
const offBeta = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [strict] } });
|
|
793
|
+
const mixed = await h({ m: 'POST', p: BETA_CHAT, b: { ...CHAT(), tools: [strict, loose] } });
|
|
794
|
+
const allStrict = await h({ m: 'POST', p: BETA_CHAT, b: { ...CHAT(), tools: [strict] } });
|
|
795
|
+
// The non-strict path must still work on the standard base URL, or the check has proved nothing
|
|
796
|
+
// about `strict` in particular.
|
|
797
|
+
const plain = await h({ m: 'POST', p: CHAT_PATH, b: { ...CHAT(), tools: [loose] } });
|
|
798
|
+
return offBeta.status === 422 && msg(offBeta).includes('beta base URL')
|
|
799
|
+
&& mixed.status === 422 && ok(allStrict) && ok(plain);
|
|
800
|
+
})),
|
|
801
|
+
|
|
802
|
+
todo('deepseek.tools.strict_schema_enforcement', 'tools', 'Under beta strict mode, generated arguments are guaranteed to validate against the declared JSON schema', 'api', 'niche'),
|
|
803
|
+
todo('deepseek.beta.prefix_with_tools', 'beta', 'Prefix completion combined with tools in one beta request', 'api', 'niche'),
|
|
804
|
+
// CORE for the same reason: an agent integration meets parallel tool calls immediately.
|
|
805
|
+
todo('deepseek.tools.parallel_multi_tool', 'tools', 'A turn returning several tool_calls at once, as the vendor does for independent tools', 'api', 'core'),
|
|
806
|
+
|
|
807
|
+
// ══ structured outputs ═══════════════════════════════════════════════════════════════════
|
|
808
|
+
done('deepseek.structured_outputs.json_object', 'structured_outputs', "On /chat/completions, response_format accepts text and json_object only — json_schema is REFUSED", 'api', 'core', () => withRoot(async (h) => {
|
|
809
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'json_object' } }) });
|
|
810
|
+
if (!ok(r)) return false;
|
|
811
|
+
const parsed = JSON.parse(text0(r));
|
|
812
|
+
if (parsed._twin_stub !== true || parsed.model !== MODEL) return false;
|
|
813
|
+
// THE DIVERGENCE, and it is ENDPOINT-SCOPED — stating it as "DeepSeek has no json_schema" would
|
|
814
|
+
// be wrong: its Responses API's `text.format` does take one. What has no json_schema is
|
|
815
|
+
// /chat/completions, and `@ai-sdk/deepseek` confirms it from the other side by INJECTING the
|
|
816
|
+
// schema into a system message rather than sending `response_format.json_schema`
|
|
817
|
+
// (src/chat/convert-to-deepseek-chat-messages.ts). A twin that accepted it here would be
|
|
818
|
+
// serving OpenAI's surface on an endpoint that does not have it.
|
|
819
|
+
const schema = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'json_schema', json_schema: { name: 'x', schema: { type: 'object' } } } }) });
|
|
820
|
+
const bogus = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'xml' } }) });
|
|
821
|
+
const text = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'text' } }) });
|
|
822
|
+
return schema.status === 422 && bogus.status === 422 && ok(text) && !text0(text).startsWith('{');
|
|
823
|
+
})),
|
|
824
|
+
|
|
825
|
+
// ── Responses API (real surface, wholly unmodeled) ───────────────────────────────────────
|
|
826
|
+
// `POST /responses` is a first-class entry in DeepSeek's own API-reference nav alongside Chat
|
|
827
|
+
// Completions, FIM, Lists Models, Get User Balance and Files. It is a DIFFERENT protocol — an
|
|
828
|
+
// `input` item array rather than `messages`, an `output` item array rather than `choices`,
|
|
829
|
+
// semantic SSE events rather than delta chunks — so it is enumerated as its own area rather than
|
|
830
|
+
// folded into chat. The twin answers it like any other unmodeled operation, never a fake success.
|
|
831
|
+
todo('deepseek.responses.create', 'responses', 'POST /responses — the Responses-protocol envelope (input items in, output items out, its own usage block)', 'api', 'common'),
|
|
832
|
+
todo('deepseek.responses.instructions', 'responses', "The Responses API's top-level `instructions` system-level input", 'api', 'niche'),
|
|
833
|
+
todo('deepseek.responses.reasoning_items', 'responses', 'Chain of thought returned as `reasoning` OUTPUT ITEMS rather than a reasoning_content field', 'api', 'common'),
|
|
834
|
+
todo('deepseek.responses.reasoning_effort_set', 'responses', "The Responses API's WIDER effort set (none|minimal|low|medium|high|xhigh|max) — /chat/completions takes only low|high|max", 'api', 'niche'),
|
|
835
|
+
todo('deepseek.responses.text_format_json_schema', 'responses', '`text.format` accepting json_schema — the structured-output shape /chat/completions does NOT have', 'api', 'common'),
|
|
836
|
+
todo('deepseek.responses.streaming', 'responses', 'Semantic SSE with `event` + `sequence_number`, terminating in response.completed / .incomplete / .failed', 'api', 'common'),
|
|
837
|
+
todo('deepseek.responses.function_tools', 'responses', 'Function tools and tool_choice on the Responses protocol, returning function_call output items', 'api', 'common'),
|
|
838
|
+
todo('deepseek.responses.web_search_tool', 'responses', 'The built-in server-side `web_search` tool and its web_search_call output items', 'api', 'niche'),
|
|
839
|
+
todo('deepseek.responses.stateless', 'responses', 'The documented statelessness: responses are not stored, so a client resubmits history rather than passing a previous-response id', 'api', 'niche'),
|
|
840
|
+
|
|
841
|
+
todo('deepseek.structured_outputs.system_prompt_contract', 'structured_outputs', 'JSON mode\'s documented requirement that the prompt itself instruct the model to emit JSON, and its empty-content failure mode', 'api', 'common'),
|
|
842
|
+
|
|
843
|
+
// ══ beta: chat prefix completion ═════════════════════════════════════════════════════════
|
|
844
|
+
done('deepseek.beta.prefix_completion', 'beta', 'Under /beta, an assistant message with prefix:true is CONTINUED rather than answered', 'api', 'common', () => withRoot(async (h) => {
|
|
845
|
+
const r = await h({ m: 'POST', p: BETA_CHAT, b: { model: MODEL, messages: [
|
|
846
|
+
{ role: 'user', content: 'describe the sky' },
|
|
847
|
+
{ role: 'assistant', content: 'The sky is', prefix: true },
|
|
848
|
+
] } });
|
|
849
|
+
if (!ok(r)) return false;
|
|
850
|
+
// Continuation, not a fresh turn: the caller's own text must lead the content.
|
|
851
|
+
return text0(r).startsWith('The sky is') && text0(r).length > 'The sky is'.length && text0(r).includes('[twin-stub:');
|
|
852
|
+
})),
|
|
853
|
+
|
|
854
|
+
done('deepseek.beta.prefix_placement_rules', 'beta', 'prefix:true is refused off /beta, on a non-assistant role, and when it is not the final message', 'api', 'common', () => withRoot(async (h) => {
|
|
855
|
+
const offBeta = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'x' }, { role: 'assistant', content: 'y', prefix: true }] } });
|
|
856
|
+
const notFinal = await h({ m: 'POST', p: BETA_CHAT, b: { model: MODEL, messages: [{ role: 'assistant', content: 'y', prefix: true }, { role: 'user', content: 'x' }] } });
|
|
857
|
+
const wrongRole = await h({ m: 'POST', p: BETA_CHAT, b: { model: MODEL, messages: [{ role: 'user', content: 'x', prefix: true }] } });
|
|
858
|
+
return offBeta.status === 422 && msg(offBeta).includes('beta base URL')
|
|
859
|
+
&& notFinal.status === 422 && msg(notFinal).includes('final message')
|
|
860
|
+
&& wrongRole.status === 422 && msg(wrongRole).includes('assistant message');
|
|
861
|
+
})),
|
|
862
|
+
|
|
863
|
+
done('deepseek.beta.standard_surface_unchanged', 'beta', 'The /beta base URL serves the SAME chat surface — it only unlocks prefix and strict tools', 'api', 'common', () => withRoot(async (h) => {
|
|
864
|
+
const std = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
865
|
+
const beta = await h({ m: 'POST', p: BETA_CHAT, b: CHAT() });
|
|
866
|
+
if (!ok(std) || !ok(beta)) return false;
|
|
867
|
+
// Same request, same answer: /beta is a feature-flag prefix, not a second API. (Usage differs
|
|
868
|
+
// because the first call seeds the cache ledger, so only the content is compared.)
|
|
869
|
+
return text0(std) === text0(beta) && body(beta).object === 'chat.completion'
|
|
870
|
+
// …and /beta still refuses what the standard surface refuses.
|
|
871
|
+
&& (await h({ m: 'POST', p: BETA_CHAT, b: CHAT({ n: 2 }) })).status === 422;
|
|
872
|
+
})),
|
|
873
|
+
|
|
874
|
+
// ══ files (images only) ══════════════════════════════════════════════════════════════════
|
|
875
|
+
done('deepseek.files.upload', 'files', 'POST /files stores an image and returns the vendor file object', 'api', 'common', () => withRoot(async (h) => {
|
|
876
|
+
const r = await h({ m: 'POST', p: FILES, b: UPLOAD() });
|
|
877
|
+
if (!ok(r)) return false;
|
|
878
|
+
const b = body(r);
|
|
879
|
+
return b.object === 'file' && b.id === FILE_1 && b.filename === 'shot.png' && b.purpose === 'user_data'
|
|
880
|
+
&& b.bytes === 3 && typeof b.created_at === 'number' && b.expires_at === undefined;
|
|
881
|
+
})),
|
|
882
|
+
|
|
883
|
+
done('deepseek.files.image_only', 'files', "The Files API accepts JPEG/PNG/GIF/WebP only, and purpose must be 'user_data'", 'api', 'common', () => withRoot(async (h) => {
|
|
884
|
+
// DeepSeek's Files API is for vision inputs. The OpenAI habit — uploading a .jsonl for a Batch
|
|
885
|
+
// API — has no counterpart here, and accepting it would fabricate a whole product.
|
|
886
|
+
const jsonl = await h({ m: 'POST', p: FILES, b: { purpose: 'user_data', filename: 'in.jsonl', media_type: 'application/jsonl', content: 'x', bytes: 1 } });
|
|
887
|
+
const batch = await h({ m: 'POST', p: FILES, b: UPLOAD({ purpose: 'batch' }) });
|
|
888
|
+
const pdf = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'doc.pdf', media_type: 'application/pdf' }) });
|
|
889
|
+
if (jsonl.status !== 422 || batch.status !== 422 || pdf.status !== 422) return false;
|
|
890
|
+
// Every accepted format must actually be accepted, or the allowlist is wrong in the other
|
|
891
|
+
// direction.
|
|
892
|
+
for (const [filename, media] of [['a.jpg', 'image/jpeg'], ['a.jpeg', 'image/jpeg'], ['a.png', 'image/png'], ['a.gif', 'image/gif'], ['a.webp', 'image/webp']]) {
|
|
893
|
+
if (!ok(await h({ m: 'POST', p: FILES, b: UPLOAD({ filename, media_type: media }) }))) return false;
|
|
894
|
+
}
|
|
895
|
+
return true;
|
|
896
|
+
})),
|
|
897
|
+
|
|
898
|
+
done('deepseek.files.expires_after', 'files', 'expires_after sets expires_at; the anchor is closed and the range is 3600..2592000', 'api', 'niche', () => withRoot(async (h) => {
|
|
899
|
+
const r = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 3600 }) });
|
|
900
|
+
if (!ok(r)) return false;
|
|
901
|
+
if (body(r).expires_at !== body(r).created_at + 3600) return false;
|
|
902
|
+
const tooShort = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 3599 }) });
|
|
903
|
+
const tooLong = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 2_592_001 }) });
|
|
904
|
+
const badAnchor = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'now', 'expires_after[seconds]': 7200 }) });
|
|
905
|
+
const atMax = await h({ m: 'POST', p: FILES, b: UPLOAD({ 'expires_after[anchor]': 'created_at', 'expires_after[seconds]': 2_592_000 }) });
|
|
906
|
+
return tooShort.status === 422 && tooLong.status === 422 && badAnchor.status === 422 && ok(atMax);
|
|
907
|
+
})),
|
|
908
|
+
|
|
909
|
+
done('deepseek.files.limits', 'files', 'A 64 MiB size cap and a 512-character filename cap are enforced', 'api', 'niche', () => withRoot(async (h) => {
|
|
910
|
+
const tooBig = await h({ m: 'POST', p: FILES, b: UPLOAD({ bytes: 64 * 1024 * 1024 + 1 }) });
|
|
911
|
+
const atCap = await h({ m: 'POST', p: FILES, b: UPLOAD({ bytes: 64 * 1024 * 1024 }) });
|
|
912
|
+
const longName = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: `${'a'.repeat(510)}.png` }) });
|
|
913
|
+
return tooBig.status === 422 && ok(atCap) && longName.status === 422;
|
|
914
|
+
})),
|
|
915
|
+
|
|
916
|
+
done('deepseek.files.list_and_retrieve', 'files', 'GET /files lists with after/limit/order, and GET /files/{id} retrieves one', 'api', 'common', () => withRoot(async (h) => {
|
|
917
|
+
await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'one.png' }) });
|
|
918
|
+
await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'two.png' }) });
|
|
919
|
+
const all = await h({ m: 'GET', p: FILES });
|
|
920
|
+
if (!ok(all) || (body(all).data as Body[]).length !== 2 || body(all).has_more !== false) return false;
|
|
921
|
+
const one = await h({ m: 'GET', p: `${FILES}/${FILE_1}` });
|
|
922
|
+
if (!ok(one) || body(one).filename !== 'one.png') return false;
|
|
923
|
+
// The documented list envelope carries the cursor bookends, not just `has_more`.
|
|
924
|
+
if (body(all).first_id !== FILE_1 || body(all).last_id !== FILE_2) return false;
|
|
925
|
+
const desc = await h({ m: 'GET', p: `${FILES}?order=desc` });
|
|
926
|
+
if ((body(desc).data as Body[])[0].id !== FILE_2) return false;
|
|
927
|
+
if (body(desc).first_id !== FILE_2 || body(desc).last_id !== FILE_1) return false;
|
|
928
|
+
const after = await h({ m: 'GET', p: `${FILES}?after=${FILE_1}` });
|
|
929
|
+
if ((body(after).data as Body[]).length !== 1 || (body(after).data as Body[])[0].id !== FILE_2) return false;
|
|
930
|
+
const capped = await h({ m: 'GET', p: `${FILES}?limit=1` });
|
|
931
|
+
if ((body(capped).data as Body[]).length !== 1 || body(capped).has_more !== true) return false;
|
|
932
|
+
const missing = await h({ m: 'GET', p: `${FILES}/file-api-nope` });
|
|
933
|
+
const badLimit = await h({ m: 'GET', p: `${FILES}?limit=1001` });
|
|
934
|
+
const badOrder = await h({ m: 'GET', p: `${FILES}?order=sideways` });
|
|
935
|
+
const badPurpose = await h({ m: 'GET', p: `${FILES}?purpose=batch` });
|
|
936
|
+
return missing.status === 404 && badLimit.status === 422 && badOrder.status === 422 && badPurpose.status === 422;
|
|
937
|
+
})),
|
|
938
|
+
|
|
939
|
+
done('deepseek.files.delete_then_recreate_ratchets_ids', 'files', 'DIRTY STATE: after delete→recreate, the new file gets a FRESH id — a deleted id is never handed out twice', 'api', 'common', () => withRoot(async (h) => {
|
|
940
|
+
// A fresh-root-only manifest structurally cannot catch this class. Three sibling packs' §9
|
|
941
|
+
// reviews each found an id-reuse or tombstone bug here, so it is verified deliberately over
|
|
942
|
+
// prior state rather than from nothing.
|
|
943
|
+
const first = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'first.png' }) });
|
|
944
|
+
if (!ok(first) || body(first).id !== FILE_1) return false;
|
|
945
|
+
const del = await h({ m: 'DELETE', p: `${FILES}/${FILE_1}` });
|
|
946
|
+
if (!ok(del) || body(del).deleted !== true) return false;
|
|
947
|
+
const gone = await h({ m: 'GET', p: `${FILES}/${FILE_1}` });
|
|
948
|
+
const doubleDelete = await h({ m: 'DELETE', p: `${FILES}/${FILE_1}` });
|
|
949
|
+
if (gone.status !== 404 || doubleDelete.status !== 404) return false;
|
|
950
|
+
const second = await h({ m: 'POST', p: FILES, b: UPLOAD({ filename: 'second.png' }) });
|
|
951
|
+
// The counter must RATCHET across the tombstone: reusing FILE_1 would resurrect the deleted
|
|
952
|
+
// file's identity and silently clobber it.
|
|
953
|
+
if (!ok(second) || body(second).id !== FILE_2) return false;
|
|
954
|
+
const list = await h({ m: 'GET', p: FILES });
|
|
955
|
+
const ids = (body(list).data as Body[]).map((f) => f.id);
|
|
956
|
+
return ids.length === 1 && ids[0] === FILE_2;
|
|
957
|
+
})),
|
|
958
|
+
|
|
959
|
+
todo('deepseek.files.expiry_removes_the_file', 'files', 'A file past its `expires_at` stops being listed and retrievable — the twin stores the timestamp faithfully but nothing ever expires (§9 round one, SHOULD-FIX 5)', 'api', 'niche'),
|
|
960
|
+
todo('deepseek.files.storage_quota', 'files', 'The account-level 25 GiB / 10,000-file storage limits', 'api', 'niche'),
|
|
961
|
+
todo('deepseek.files.reference_in_chat', 'files', 'Passing an uploaded file_id as a chat content part to deepseek-v4-flash-vision-exp', 'api', 'common'),
|
|
962
|
+
todo('deepseek.files.binary_content', 'files', 'Files: persist the uploaded bytes and serve them back (upload metadata and byte count are recorded today)', 'api', 'niche'),
|
|
963
|
+
|
|
964
|
+
// ══ vision ═══════════════════════════════════════════════════════════════════════════════
|
|
965
|
+
todo('deepseek.vision.image_url_part', 'vision', 'image_url content parts (with the low/high/original/auto detail option) on the vision model', 'api', 'common'),
|
|
966
|
+
todo('deepseek.vision.file_data_part', 'vision', "The `file` content part carrying inline `file_data`, which @ai-sdk/deepseek's fileData option produces", 'api', 'niche'),
|
|
967
|
+
todo('deepseek.vision.media_type_closed_set', 'vision', 'Rejecting an image media type outside JPEG/PNG/GIF/WebP, and a URL beyond 8192 characters', 'api', 'niche'),
|
|
968
|
+
todo('deepseek.vision.non_vision_model_refuses_images', 'vision', 'A non-vision model refuses an image part the way the vendor does', 'api', 'common'),
|
|
969
|
+
|
|
970
|
+
// ══ Anthropic-compatible surface ═════════════════════════════════════════════════════════
|
|
971
|
+
todo('deepseek.anthropic.messages', 'anthropic', 'POST /anthropic/v1/messages — the Anthropic-shaped surface DeepSeek serves alongside the OpenAI-shaped one', 'api', 'common'),
|
|
972
|
+
todo('deepseek.anthropic.model_mapping', 'anthropic', 'Claude model names mapped to DeepSeek models (Opus→v4-pro, Haiku/Sonnet→v4-flash, unmapped→v4-flash)', 'api', 'niche'),
|
|
973
|
+
todo('deepseek.anthropic.x_api_key_auth', 'anthropic', 'The Anthropic surface authenticates with x-api-key, not a bearer Authorization header', 'api', 'niche'),
|
|
974
|
+
todo('deepseek.anthropic.streaming', 'anthropic', "Server-sent events on the Anthropic-compatible surface (Anthropic's own event grammar, not the OpenAI chunk shape)", 'api', 'niche'),
|
|
975
|
+
todo('deepseek.anthropic.thinking_budget_ignored', 'anthropic', "The Anthropic surface accepts `thinking` but IGNORES budget_tokens — a documented accept-and-ignore, so it must not be an error", 'api', 'niche'),
|
|
976
|
+
todo('deepseek.anthropic.unsupported_content_types', 'anthropic', 'Refusing the Anthropic content types DeepSeek does not support (document, search result, MCP tools, container uploads)', 'api', 'niche'),
|
|
977
|
+
|
|
978
|
+
// ══ errors ═══════════════════════════════════════════════════════════════════════════════
|
|
979
|
+
done('deepseek.errors.status_split_400_vs_422', 'errors', "DeepSeek's 400 is BODY FORMAT and its 422 is INVALID PARAMETERS — the split OpenAI does not have", 'api', 'core', () => withRoot(async (h) => {
|
|
980
|
+
// The single most consequential divergence in this pack. An OpenAI-copied twin answers 400 for
|
|
981
|
+
// both, which passes a status>=400 check and is wrong on every parameter refusal.
|
|
982
|
+
// 400 is ONLY for a body this API cannot parse as a JSON object: absent, malformed, or the
|
|
983
|
+
// wrong JSON kind.
|
|
984
|
+
const absent = await h({ m: 'POST', p: CHAT_PATH, b: undefined });
|
|
985
|
+
const raw = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: '{not json' });
|
|
986
|
+
const notObject = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: '[1,2]' });
|
|
987
|
+
if (raw.status !== 400 || !String((raw.body as Body).error.message).includes('Invalid request body format')) return false;
|
|
988
|
+
if (notObject.status !== 400 || absent.status !== 400) return false;
|
|
989
|
+
// …and EVERY complaint about parsed content is 422, including the ones an OpenAI-shaped twin
|
|
990
|
+
// would 400: a missing required parameter is the same class as a bad one and must not be
|
|
991
|
+
// reported two different ways.
|
|
992
|
+
for (const b of [CHAT({ model: 'nope' }), { messages: [{ role: 'user', content: 'x' }] }, { model: MODEL }, { model: MODEL, messages: [null] }]) {
|
|
993
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b });
|
|
994
|
+
if (r.status !== 422 || !msg(r).includes('invalid parameters')) return false;
|
|
995
|
+
}
|
|
996
|
+
return true;
|
|
997
|
+
})),
|
|
998
|
+
|
|
999
|
+
done('deepseek.errors.envelope', 'errors', "Refusals carry DeepSeek's { error: { message, ... } } envelope and no key its decoder does not declare", 'api', 'core', () => withRoot(async (h) => {
|
|
1000
|
+
const r = await h({ m: 'POST', p: CHAT_PATH, b: CHAT({ model: 'nope' }) });
|
|
1001
|
+
const e = body(r).error as Body;
|
|
1002
|
+
// The key set is a LITERAL here, from @ai-sdk/deepseek's `deepSeekErrorSchema` — asserting
|
|
1003
|
+
// against the handler's own constant would be a tautology that could not catch it drifting.
|
|
1004
|
+
const declared = new Set(['message', 'type', 'param', 'code']);
|
|
1005
|
+
if (r.status !== 422 || !e || typeof e.message !== 'string' || !e.message.length) return false;
|
|
1006
|
+
if (!Object.keys(e).every((k) => declared.has(k))) return false;
|
|
1007
|
+
// …and the envelope really is DECODABLE by the vendor's own schema shape on every status the
|
|
1008
|
+
// twin emits, not just this one. §9 round two: a subset check alone cannot fail for anything the
|
|
1009
|
+
// twin currently produces, so it is paired with a sweep that asserts `message` is a non-empty
|
|
1010
|
+
// string and no undeclared key appears anywhere.
|
|
1011
|
+
const others = [
|
|
1012
|
+
await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, headers: {} }),
|
|
1013
|
+
await h({ m: 'GET', p: '/nope' }),
|
|
1014
|
+
await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: '{not json' }),
|
|
1015
|
+
await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), readOnly: true }),
|
|
1016
|
+
];
|
|
1017
|
+
for (const o of others) {
|
|
1018
|
+
const oe = (o.body as Body)?.error as Body;
|
|
1019
|
+
if (o.status < 400 || !oe || typeof oe.message !== 'string' || !oe.message.length) return false;
|
|
1020
|
+
if (!Object.keys(oe).every((k) => declared.has(k))) return false;
|
|
1021
|
+
}
|
|
1022
|
+
return true;
|
|
1023
|
+
})),
|
|
1024
|
+
|
|
1025
|
+
done('deepseek.errors.envelope_omits_an_unsourced_type', 'errors', "A 422 and a 402 carry NO `type` — nothing first-party names one for those statuses, and the string a 422 used to emit resolves to 400 in the SDK's own discriminator table", 'api', 'core', () => withRoot(async (h, root) => {
|
|
1026
|
+
// §9 ROUND ONE, SHOULD-FIX 3, and the pin the fix itself lacked. `@ai-sdk/deepseek`'s
|
|
1027
|
+
// `getDeepSeekStreamErrorMetadata` has NO 422 case at all, and maps `invalid_request_error` to
|
|
1028
|
+
// `statusCode: 400` — so emitting that string on a 422 made a real client resolve the twin's
|
|
1029
|
+
// own refusal to the wrong status. The rule the twin states is: use a type only where something
|
|
1030
|
+
// first-party names one for that status, otherwise omit the field.
|
|
1031
|
+
// SWEEP every 422 the twin can produce, not one sample. §9 ROUND TWO, BLOCKER 2: the fix was
|
|
1032
|
+
// applied to the `invalidParameters` helper only, so the FIM streaming refusal — a 422 built
|
|
1033
|
+
// inline 700 lines away — kept emitting `invalid_request_error`, and this cell passed because it
|
|
1034
|
+
// happened to probe a different one. A universal claim needs a universal check.
|
|
1035
|
+
const fourTwentyTwos: Array<{ m: string; p: string; b?: unknown }> = [
|
|
1036
|
+
{ m: 'POST', p: CHAT_PATH, b: CHAT({ model: 'nope' }) },
|
|
1037
|
+
{ m: 'POST', p: CHAT_PATH, b: CHAT({ n: 2 }) },
|
|
1038
|
+
{ m: 'POST', p: CHAT_PATH, b: CHAT({ response_format: { type: 'json_schema' } }) },
|
|
1039
|
+
{ m: 'POST', p: FIM_PATH, b: { model: PRO, prompt: 'x', stream: true } },
|
|
1040
|
+
{ m: 'POST', p: FIM_PATH, b: { model: MODEL, prompt: 'x' } },
|
|
1041
|
+
{ m: 'POST', p: FILES, b: UPLOAD({ purpose: 'batch' }) },
|
|
1042
|
+
{ m: 'GET', p: `${FILES}?limit=5000` },
|
|
1043
|
+
];
|
|
1044
|
+
for (const req of fourTwentyTwos) {
|
|
1045
|
+
const r = await h(req);
|
|
1046
|
+
if (r.status !== 422 || 'type' in (body(r).error as Body)) return false;
|
|
1047
|
+
}
|
|
1048
|
+
// The 405 is the OTHER status nothing first-party names a type for, and it is built in a third
|
|
1049
|
+
// place again — the revert matrix showed `errors.envelope` cannot catch it (a `type` is a
|
|
1050
|
+
// DECLARED key, so a subset check passes), so the universal has to sweep it here.
|
|
1051
|
+
const ro = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), readOnly: true });
|
|
1052
|
+
if (ro.status !== 405 || 'type' in (body(ro).error as Body)) return false;
|
|
1053
|
+
// 402 likewise carries none — and this DRIVES that path rather than asserting it in a comment.
|
|
1054
|
+
await applyTwinWrite('deepseek', {
|
|
1055
|
+
operation: 'balance.update', subjectType: 'balance', subjectId: 'account',
|
|
1056
|
+
fields: { is_available: false, balance_infos: [] }, occurredAt: OCCURRED_AT, actor: { kind: 'agent' },
|
|
1057
|
+
}, root);
|
|
1058
|
+
const drained = await h({ m: 'POST', p: CHAT_PATH, b: CHAT() });
|
|
1059
|
+
if (drained.status !== 402 || 'type' in (body(drained).error as Body)) return false;
|
|
1060
|
+
// …but the statuses the SDK's table DOES name keep their discriminator, or the rule would just
|
|
1061
|
+
// be "never emit a type", which is a different (and equally unsourced) claim.
|
|
1062
|
+
const unauth = await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, headers: {} });
|
|
1063
|
+
const notFound = await h({ m: 'GET', p: '/nope' });
|
|
1064
|
+
return unauth.status === 401 && (body(unauth).error as Body).type === 'authentication_error'
|
|
1065
|
+
&& refused(notFound) && (body(notFound).error as Body).type === 'not_found_error';
|
|
1066
|
+
})),
|
|
1067
|
+
|
|
1068
|
+
done('deepseek.errors.unmodeled_ops_fail', 'errors', 'Unmodeled operations and OpenAI-only endpoints fail rather than fake a success', 'api', 'core', () => withRoot(async (h) => {
|
|
1069
|
+
for (const [m, p] of [
|
|
1070
|
+
['POST', '/embeddings'], ['POST', '/images/generations'], ['POST', '/audio/transcriptions'],
|
|
1071
|
+
['POST', '/moderations'], ['POST', '/batches'], ['GET', '/models/deepseek-v4-flash'],
|
|
1072
|
+
['POST', '/fine_tuning/jobs'],
|
|
1073
|
+
]) {
|
|
1074
|
+
const r = await h({ m: m!, p: p!, b: {} });
|
|
1075
|
+
// Refused as a CLIENT error with the vendor envelope; the exact code is the twin's unverified
|
|
1076
|
+
// choice (see deepseek.errors.unknown_route_envelope), so it is not asserted (NIT 11).
|
|
1077
|
+
if (!refused(r)) return false;
|
|
1078
|
+
}
|
|
1079
|
+
// REAL vendor surface this twin does not model yet must fail too — and must not be confused
|
|
1080
|
+
// with the OpenAI-only endpoints above. The Anthropic surface says so by name; `/responses` is
|
|
1081
|
+
// a first-class entry in DeepSeek's own reference nav and falls through to the router's
|
|
1082
|
+
// not-found rather than being answered as if it were /chat/completions.
|
|
1083
|
+
const anthropic = await h({ m: 'POST', p: '/anthropic/v1/messages', b: {} });
|
|
1084
|
+
if (!refused(anthropic) || !msg(anthropic).includes('Anthropic-compatible')) return false;
|
|
1085
|
+
const responses = await h({ m: 'POST', p: '/responses', b: { model: MODEL, input: 'hi' } });
|
|
1086
|
+
const betaResponses = await h({ m: 'POST', p: '/beta/responses', b: { model: MODEL, input: 'hi' } });
|
|
1087
|
+
return refused(responses) && refused(betaResponses);
|
|
1088
|
+
})),
|
|
1089
|
+
|
|
1090
|
+
done('deepseek.errors.read_only_405', 'errors', 'A read-only twin refuses every write with 405 while reads keep working', 'api', 'common', async () => {
|
|
1091
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
1092
|
+
try {
|
|
1093
|
+
return await verifyBoundary('deepseek.errors.read_only_405', async () => {
|
|
1094
|
+
const chat = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), root, readOnly: true });
|
|
1095
|
+
const upload = await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD()), root, readOnly: true });
|
|
1096
|
+
const models = await handleDeepSeekTwinRequest({ method: 'GET', path: MODELS, root, readOnly: true });
|
|
1097
|
+
// And nothing was written: the projection must still be empty.
|
|
1098
|
+
const wrote = projectResources('deepseek', root).length > 0;
|
|
1099
|
+
return chat.status === 405 && upload.status === 405 && ok(models) && !wrote;
|
|
1100
|
+
});
|
|
1101
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
1102
|
+
}),
|
|
1103
|
+
|
|
1104
|
+
todo('deepseek.errors.rate_limit_429', 'errors', "Answering DeepSeek's documented 429 under a real rate condition. The twin serves the faithful envelope, but only behind a twin-only `x-twin-force-rate-limit` header — scaffolding, which §6 keeps OUT of the coverage claim, so this is a todo rather than a done proven by its own trigger (§9 round one)", 'api', 'common'),
|
|
1105
|
+
|
|
1106
|
+
// (Scripted 500/503 failures used to be claimed here as a `done`. They are SCENARIO scaffolding,
|
|
1107
|
+
// which ADDING_A_TWIN.md §6 says to keep out of the capability manifest because counting it pads
|
|
1108
|
+
// the denominator — the behaviour is gated by deepseek-scenario.test.ts instead. §9 round one.)
|
|
1109
|
+
todo('deepseek.errors.server_5xx', 'errors', "Answering DeepSeek's documented 500 'Our server encounters an issue' and 503 'The server is overloaded' under real conditions rather than only when a scenario handler scripts them", 'api', 'niche'),
|
|
1110
|
+
|
|
1111
|
+
todo('deepseek.errors.unknown_route_envelope', 'errors', "The exact status/body a real DeepSeek gateway returns for an unrouted path — its published error table (400/401/402/422/429/500/503) has no 404 entry, so the twin's 404 shape is unverified against a first-party source", 'api', 'niche'),
|
|
1112
|
+
todo('deepseek.errors.retryable_metadata', 'errors', 'Error `code`/`type` discriminators the SDK maps to retryability (insufficient_quota, overloaded_error, timeout, …)', 'api', 'niche'),
|
|
1113
|
+
|
|
1114
|
+
// ══ auth ═════════════════════════════════════════════════════════════════════════════════
|
|
1115
|
+
done('deepseek.auth.bearer_required', 'auth', 'A request carrying an auth surface needs a bearer credential; a missing or sentinel-invalid one is 401', 'api', 'core', () => withRootH(async (h) => {
|
|
1116
|
+
const missing = await h({ m: 'GET', p: MODELS, headers: {} });
|
|
1117
|
+
const invalid = await h({ m: 'GET', p: MODELS, headers: { authorization: 'Bearer sk_invalid' } });
|
|
1118
|
+
const notBearer = await h({ m: 'GET', p: MODELS, headers: { authorization: 'Basic abc' } });
|
|
1119
|
+
const good = await h({ m: 'GET', p: MODELS, headers: { authorization: 'Bearer sk-anything' } });
|
|
1120
|
+
return missing.status === 401 && invalid.status === 401 && notBearer.status === 401 && ok(good)
|
|
1121
|
+
&& msg(missing).includes('Authentication fails');
|
|
1122
|
+
})),
|
|
1123
|
+
|
|
1124
|
+
todo('deepseek.auth.gates_before_routing', 'auth', 'Whether DeepSeek authenticates BEFORE routing (401 rather than 404 on an unknown path with no credential). The twin does, but nothing first-party documents the ordering, so it is a twin design choice and not a proven vendor fact (§9 round one, NIT 12)', 'api', 'common'),
|
|
1125
|
+
|
|
1126
|
+
todo('deepseek.auth.x_api_key', 'auth', 'x-api-key authentication on the Anthropic-compatible surface', 'api', 'niche'),
|
|
1127
|
+
|
|
1128
|
+
// ══ rate limits / budget ═════════════════════════════════════════════════════════════════
|
|
1129
|
+
done('deepseek.rate_limits.budget_refuses_past_ceiling', 'rate_limits', 'The client-side budget THROWS instead of calling past its ceiling, with the vendor call count unchanged', 'connector', 'core', async () => {
|
|
1130
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
1131
|
+
try {
|
|
1132
|
+
return await verifyBoundary('deepseek.rate_limits.budget_refuses_past_ceiling', async () => {
|
|
1133
|
+
const { DeepSeekBudget } = await import('./deepseek-budget.ts');
|
|
1134
|
+
let calls = 0;
|
|
1135
|
+
const fetchImpl = (async () => { calls++; return new Response('{}', { status: 200, headers: { 'content-type': 'application/json' } }); }) as unknown as typeof fetch;
|
|
1136
|
+
const ledger = new DeepSeekBudget({ path: join(root, 'ledger.json') });
|
|
1137
|
+
const execute = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { fetchImpl, budget: ledger });
|
|
1138
|
+
// /models costs the default weight 2 against a ceiling of 60 → exactly 30 fit.
|
|
1139
|
+
for (let i = 0; i < 30; i++) await execute('GET', '/models');
|
|
1140
|
+
if (calls !== 30) return false;
|
|
1141
|
+
let threw = false;
|
|
1142
|
+
try { await execute('GET', '/models'); } catch { threw = true; }
|
|
1143
|
+
// "It threw" is not the proof — the unchanged COUNT is what shows nothing reached DeepSeek.
|
|
1144
|
+
return threw && calls === 30;
|
|
1145
|
+
});
|
|
1146
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
1147
|
+
}),
|
|
1148
|
+
|
|
1149
|
+
done('deepseek.rate_limits.inference_costs_more', 'rate_limits', 'Token-billed inference endpoints are priced above the default weight, including under the /beta prefix', 'connector', 'common', async () => {
|
|
1150
|
+
const { deepseekCallWeight } = await import('./deepseek-budget.ts');
|
|
1151
|
+
// The weights are asserted as LITERALS: importing the weight table and comparing it with itself
|
|
1152
|
+
// would drift together and could never catch an inference call being priced as a read.
|
|
1153
|
+
return deepseekCallWeight('POST', '/chat/completions') === 6
|
|
1154
|
+
&& deepseekCallWeight('POST', '/beta/chat/completions') === 6
|
|
1155
|
+
&& deepseekCallWeight('POST', '/beta/completions') === 6
|
|
1156
|
+
&& deepseekCallWeight('GET', '/models') === 2
|
|
1157
|
+
&& deepseekCallWeight('GET', '/user/balance') === 2
|
|
1158
|
+
// …and the anchored rules survive the two evasions that would otherwise price a write as a read.
|
|
1159
|
+
&& deepseekCallWeight('post', '/chat/completions') === 6
|
|
1160
|
+
&& deepseekCallWeight('POST', '//chat/completions') === 6
|
|
1161
|
+
&& deepseekCallWeight('POST', '/chat/completions/') === 6;
|
|
1162
|
+
}),
|
|
1163
|
+
|
|
1164
|
+
done('deepseek.rate_limits.ledger_survives_restart', 'rate_limits', 'The spend ledger is persistent: a fresh executor over the same ledger gets no fresh allowance', 'connector', 'common', async () => {
|
|
1165
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
1166
|
+
try {
|
|
1167
|
+
return await verifyBoundary('deepseek.rate_limits.ledger_survives_restart', async () => {
|
|
1168
|
+
const path = join(root, 'ledger.json');
|
|
1169
|
+
let calls = 0;
|
|
1170
|
+
const fetchImpl = (async () => { calls++; return new Response('{}', { status: 200, headers: { 'content-type': 'application/json' } }); }) as unknown as typeof fetch;
|
|
1171
|
+
const first = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { budgetOptions: { path }, fetchImpl });
|
|
1172
|
+
for (let i = 0; i < 30; i++) await first('GET', '/models');
|
|
1173
|
+
// A brand-new budget object over the SAME ledger file — the "restarted process" case.
|
|
1174
|
+
const second = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { budgetOptions: { path }, fetchImpl });
|
|
1175
|
+
let threw = false;
|
|
1176
|
+
try { await second('GET', '/models'); } catch { threw = true; }
|
|
1177
|
+
return threw && calls === 30;
|
|
1178
|
+
});
|
|
1179
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
1180
|
+
}),
|
|
1181
|
+
|
|
1182
|
+
done('deepseek.rate_limits.executor_path_allowlist', 'rate_limits', 'The live executor refuses an unmodeled path BEFORE spending budget or touching the network', 'connector', 'common', async () => {
|
|
1183
|
+
const root = mkdtempSync(join(tmpdir(), 'deepseek-cap-'));
|
|
1184
|
+
try {
|
|
1185
|
+
return await verifyBoundary('deepseek.rate_limits.executor_path_allowlist', async () => {
|
|
1186
|
+
const { DeepSeekBudget } = await import('./deepseek-budget.ts');
|
|
1187
|
+
const counter = { n: 0 };
|
|
1188
|
+
const fetchImpl = (async () => { counter.n++; return new Response('{}', { status: 200, headers: { 'content-type': 'application/json' } }); }) as unknown as typeof fetch;
|
|
1189
|
+
const ledger = new DeepSeekBudget({ path: join(root, 'ledger.json') });
|
|
1190
|
+
const execute = liveDeepSeekExecute('sk-twin', 'http://127.0.0.1:1', { fetchImpl, budget: ledger });
|
|
1191
|
+
// The OpenAI-shaped path is the one a careless caller reaches for, and it would evade both
|
|
1192
|
+
// the budget's anchored rules and DeepSeek's own routing.
|
|
1193
|
+
for (const p of ['/v1/models', '/chat/completions', '/anthropic/v1/messages', '/../etc']) {
|
|
1194
|
+
let threw = false;
|
|
1195
|
+
try { await execute('GET', p); } catch { threw = true; }
|
|
1196
|
+
if (!threw) return false;
|
|
1197
|
+
}
|
|
1198
|
+
if (counter.n !== 0) return false;
|
|
1199
|
+
// The budget must be UNSPENT: the refusal happens BEFORE `checkBudget`, so a modeled call
|
|
1200
|
+
// still has the full allowance afterwards.
|
|
1201
|
+
if (ledger.snapshot().spend !== 0) return false;
|
|
1202
|
+
await execute('GET', '/models');
|
|
1203
|
+
return counter.n > 0;
|
|
1204
|
+
});
|
|
1205
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
1206
|
+
}),
|
|
1207
|
+
|
|
1208
|
+
|
|
1209
|
+
// ══ usage accounting ═════════════════════════════════════════════════════════════════════
|
|
1210
|
+
done('deepseek.usage.token_counts', 'usage', 'usage reports deterministic prompt/completion/total counts that respond to the actual payload', 'api', 'core', () => withRoot(async (h) => {
|
|
1211
|
+
const short = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi' }] } });
|
|
1212
|
+
const long = await h({ m: 'POST', p: CHAT_PATH, b: { model: MODEL, messages: [{ role: 'user', content: 'hi '.repeat(200) }] } });
|
|
1213
|
+
if (!ok(short) || !ok(long)) return false;
|
|
1214
|
+
const su = body(short).usage as Body;
|
|
1215
|
+
const lu = body(long).usage as Body;
|
|
1216
|
+
return su.prompt_tokens > 0 && lu.prompt_tokens > su.prompt_tokens
|
|
1217
|
+
&& su.total_tokens === su.prompt_tokens + su.completion_tokens
|
|
1218
|
+
&& lu.total_tokens === lu.prompt_tokens + lu.completion_tokens;
|
|
1219
|
+
})),
|
|
1220
|
+
|
|
1221
|
+
done('deepseek.usage.serve_path_determinism', 'usage', 'The same request against two independent roots yields a byte-identical response', 'api', 'core', async () => {
|
|
1222
|
+
const a = mkdtempSync(join(tmpdir(), 'deepseek-det-a-'));
|
|
1223
|
+
const b = mkdtempSync(join(tmpdir(), 'deepseek-det-b-'));
|
|
1224
|
+
try {
|
|
1225
|
+
return await verifyBoundary('deepseek.usage.serve_path_determinism', async () => {
|
|
1226
|
+
// Both roots are driven through the IDENTICAL history from here on, which is what makes the
|
|
1227
|
+
// later comparison a (request, state) equality rather than a coincidence (§9 round two, NIT 1
|
|
1228
|
+
// — the comment used to claim this while root `a` had received extra calls root `b` had not).
|
|
1229
|
+
const run = (root: string) => handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT({ logprobs: true, top_logprobs: 2 })), root, occurredAt: OCCURRED_AT });
|
|
1230
|
+
const [ra, rb] = await Promise.all([run(a), run(b)]);
|
|
1231
|
+
// MUTATION-GATE FINDING: byte-equality alone is satisfied by a dead twin answering {} on
|
|
1232
|
+
// both roots. Determinism is only a claim about a REAL response, so every derived value the
|
|
1233
|
+
// cell is about — the id, the fingerprint, the cache split, the logprob numbers — is
|
|
1234
|
+
// asserted to EXIST and be well-formed before the two are compared.
|
|
1235
|
+
const bodyA = ra.body as Body;
|
|
1236
|
+
if (!ok(ra) || bodyA.object !== 'chat.completion') return false;
|
|
1237
|
+
if (typeof bodyA.id !== 'string' || !bodyA.id.startsWith('chatcmpl-twin-')) return false;
|
|
1238
|
+
if (typeof bodyA.system_fingerprint !== 'string' || !bodyA.system_fingerprint.endsWith('_twin_stub_kvcache')) return false;
|
|
1239
|
+
if (typeof (bodyA.usage as Body).prompt_cache_miss_tokens !== 'number') return false;
|
|
1240
|
+
if (!Array.isArray((choice0(ra).logprobs as Body)?.content) || (choice0(ra).logprobs as Body).content.length === 0) return false;
|
|
1241
|
+
if (typeof (choice0(ra).logprobs as Body).content[0].logprob !== 'number') return false;
|
|
1242
|
+
// Every derived value must be a pure function of (request, stored state). A wall clock or an
|
|
1243
|
+
// entropy source anywhere on the serve path breaks this.
|
|
1244
|
+
if (JSON.stringify(ra.body) !== JSON.stringify(rb.body)) return false;
|
|
1245
|
+
// …and a REPLAY into the same root is identical too, once the cache ledger already holds
|
|
1246
|
+
// this request's own prefix (proving the ledger read is deterministic, not accumulating).
|
|
1247
|
+
// A replay into each root, kept in lockstep so the two histories stay identical.
|
|
1248
|
+
const againA = await run(a);
|
|
1249
|
+
const againB = await run(b);
|
|
1250
|
+
if ((againA.body as Body).object !== 'chat.completion') return false;
|
|
1251
|
+
if (JSON.stringify(againA.body) !== JSON.stringify(againB.body)) return false;
|
|
1252
|
+
// The single-turn request above never HITS the cache, so it cannot prove the ledger read is
|
|
1253
|
+
// deterministic on the path that matters. This does: a continuation that genuinely hits,
|
|
1254
|
+
// replayed against a root whose ledger has since grown, must answer byte-identically — a
|
|
1255
|
+
// ledger read that accumulated (or a hit measured off stored state rather than the request)
|
|
1256
|
+
// would drift on the second call.
|
|
1257
|
+
// The cache-bearing path, which the single-turn half above never reaches (a one-message
|
|
1258
|
+
// request reads the ledger but cannot match anything until it has been served once).
|
|
1259
|
+
//
|
|
1260
|
+
// The comparison is across two roots driven through the SAME history, NOT two calls into
|
|
1261
|
+
// one root: serving a request CHANGES the ledger, so a second call into the same root is a
|
|
1262
|
+
// different (request, state) pair and is legitimately allowed to answer differently. The
|
|
1263
|
+
// determinism claim is "same request + same stored state → same bytes", and this is what
|
|
1264
|
+
// that actually looks like.
|
|
1265
|
+
const opening = { model: MODEL, messages: [{ role: 'user', content: 'a determinism opening turn' }] };
|
|
1266
|
+
const post = (root: string, payload: unknown) => handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(payload), root, occurredAt: OCCURRED_AT });
|
|
1267
|
+
const seededA = await post(a, opening);
|
|
1268
|
+
const seededB = await post(b, opening);
|
|
1269
|
+
if (!ok(seededA) || !ok(seededB)) return false;
|
|
1270
|
+
const convo = {
|
|
1271
|
+
model: MODEL,
|
|
1272
|
+
messages: [
|
|
1273
|
+
{ role: 'user', content: 'a determinism opening turn' },
|
|
1274
|
+
{ role: 'assistant', content: text0(seededA), reasoning_content: '' },
|
|
1275
|
+
{ role: 'user', content: 'and a follow-up' },
|
|
1276
|
+
],
|
|
1277
|
+
};
|
|
1278
|
+
const hitA = await post(a, convo);
|
|
1279
|
+
const hitB = await post(b, convo);
|
|
1280
|
+
// It really HIT — otherwise this cell would be proving determinism on the same cache-free
|
|
1281
|
+
// path the single-turn half already covered.
|
|
1282
|
+
if (!ok(hitA) || (body(hitA).usage as Body).prompt_cache_hit_tokens === 0) return false;
|
|
1283
|
+
return JSON.stringify(hitA.body) === JSON.stringify(hitB.body);
|
|
1284
|
+
});
|
|
1285
|
+
} finally { rmSync(a, { recursive: true, force: true }); rmSync(b, { recursive: true, force: true }); }
|
|
1286
|
+
}),
|
|
1287
|
+
|
|
1288
|
+
todo('deepseek.usage.real_tokenizer', 'usage', "Token counts from DeepSeek's real tokenizer: a BPE tokenizer is a data file, runs offline and is deterministic, so nothing about the serve-path invariant forbids it (§9 round one, SHOULD-FIX 5). The twin's counts are a ~4-chars-per-token estimate — the right order of magnitude and stable for a fixed input, but not DeepSeek's vocabulary, and no capability asserts an exact vendor count", 'api', 'common'),
|
|
1289
|
+
|
|
1290
|
+
// ══ connector ════════════════════════════════════════════════════════════════════════════
|
|
1291
|
+
done('deepseek.connector.pull_models', 'connector', 'Pull maps real model rows into the twin projection', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_models', async (root) => {
|
|
1292
|
+
const calls: Array<{ method: string; path: string }> = [];
|
|
1293
|
+
const execute = fakeExecute({ models: [{ id: 'deepseek-v4-pro', object: 'model', owned_by: 'deepseek' }] }, calls);
|
|
1294
|
+
const res = await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
|
|
1295
|
+
const row = projectResources('deepseek', root).find((r) => r.type === 'model' && r.id === 'deepseek-v4-pro');
|
|
1296
|
+
return res.observed > 0 && !!row && row.owned_by === 'deepseek'
|
|
1297
|
+
// …and the paths it actually addressed are DeepSeek's, not OpenAI-shaped ones.
|
|
1298
|
+
&& calls.some((c) => c.method === 'GET' && c.path === '/models');
|
|
1299
|
+
})),
|
|
1300
|
+
|
|
1301
|
+
done('deepseek.connector.pull_files_and_balance', 'connector', 'Pull maps real files and the singleton user balance into the projection', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_files_and_balance', async (root) => {
|
|
1302
|
+
const execute = fakeExecute({
|
|
1303
|
+
files: [{ id: 'file-api-a1b2c3d4e5f6g7h8', object: 'file', bytes: 1024, created_at: 1_700_000_000, filename: 'real.png', purpose: 'user_data', expires_at: 1_700_003_600 }],
|
|
1304
|
+
balance: { is_available: false, balance_infos: [{ currency: 'USD', total_balance: '0.00', granted_balance: '0.00', topped_up_balance: '0.00' }] },
|
|
1305
|
+
});
|
|
1306
|
+
await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
|
|
1307
|
+
const file = projectResources('deepseek', root).find((r) => r.type === 'file' && r.id === 'file-api-a1b2c3d4e5f6g7h8');
|
|
1308
|
+
const bal = projectResources('deepseek', root).find((r) => r.type === 'balance' && r.id === 'account');
|
|
1309
|
+
if (!file || file.filename !== 'real.png' || file.bytes !== 1024 || file.expires_at !== 1_700_003_600) return false;
|
|
1310
|
+
if (!bal || bal.is_available !== false) return false;
|
|
1311
|
+
// The pull is LOAD-BEARING, not decorative: a pulled drained balance makes the twin refuse
|
|
1312
|
+
// completions exactly as the real account would.
|
|
1313
|
+
const chat = await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), root });
|
|
1314
|
+
const served = await handleDeepSeekTwinRequest({ method: 'GET', path: `${FILES}/file-api-a1b2c3d4e5f6g7h8`, root });
|
|
1315
|
+
return chat.status === 402 && ok(served) && body(served).filename === 'real.png';
|
|
1316
|
+
})),
|
|
1317
|
+
|
|
1318
|
+
done('deepseek.connector.pull_is_idempotent', 'connector', 'Re-pulling identical state appends no deltas', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_is_idempotent', async (root) => {
|
|
1319
|
+
const execute = fakeExecute({ files: [{ id: 'file-api-aaaa', object: 'file', bytes: 1, created_at: 1, filename: 'a.png', purpose: 'user_data' }] });
|
|
1320
|
+
const first = await syncDeepSeekFromReal(execute, { root, occurredAt: '2026-08-31T12:00:00.000Z' });
|
|
1321
|
+
// A MOVING timestamp on the second poll, deliberately: the kernel hashes an observed event over
|
|
1322
|
+
// (occurredAt + post-state), so a pinned poll time would make this pass for the wrong reason.
|
|
1323
|
+
const second = await syncDeepSeekFromReal(execute, { root, occurredAt: '2026-08-31T12:05:00.000Z' });
|
|
1324
|
+
const rows = projectResources('deepseek', root).filter((r) => r.type === 'file');
|
|
1325
|
+
return first.deltasAppended > 0 && second.deltasAppended === 0 && rows.length === 1;
|
|
1326
|
+
})),
|
|
1327
|
+
|
|
1328
|
+
done('deepseek.connector.pull_refuses_an_error_envelope', 'connector', 'A refused pull THROWS rather than folding an empty account over real observed state', 'connector', 'core', () => withConnectorRoot('deepseek.connector.pull_refuses_an_error_envelope', async (root) => {
|
|
1329
|
+
const good = fakeExecute({ files: [{ id: 'file-api-keepme', object: 'file', bytes: 1, created_at: 1, filename: 'keep.png', purpose: 'user_data' }] });
|
|
1330
|
+
await syncDeepSeekFromReal(good, { root, occurredAt: OCCURRED_AT });
|
|
1331
|
+
const failing = fakeExecute({ files: [], fail: '/files' });
|
|
1332
|
+
let threw = false;
|
|
1333
|
+
// `pullDeepSeekState` is a PURE mapper over the injected executor — it takes no root and writes
|
|
1334
|
+
// nothing, which is precisely why the refusal must happen here rather than in `syncPull`. The
|
|
1335
|
+
// root-scoped half is the survival check below (§9 round two, NIT 7).
|
|
1336
|
+
try { await pullDeepSeekState(failing); } catch { threw = true; }
|
|
1337
|
+
if (!threw) return false;
|
|
1338
|
+
// …and the same refusal through the ROOT-SCOPED entry point, so nothing was folded either.
|
|
1339
|
+
let syncThrew = false;
|
|
1340
|
+
try { await syncDeepSeekFromReal(failing, { root, occurredAt: '2026-08-31T12:20:00.000Z' }); } catch { syncThrew = true; }
|
|
1341
|
+
if (!syncThrew) return false;
|
|
1342
|
+
// The previously observed file must SURVIVE — the whole point of throwing.
|
|
1343
|
+
const still = await handleDeepSeekTwinRequest({ method: 'GET', path: `${FILES}/file-api-keepme`, root });
|
|
1344
|
+
return ok(still) && body(still).filename === 'keep.png';
|
|
1345
|
+
})),
|
|
1346
|
+
|
|
1347
|
+
done('deepseek.connector.push_file_delete', 'connector', 'A local delete of a PULLED file is pushed to the real account and confirmed', 'connector', 'core', () => withConnectorRoot('deepseek.connector.push_file_delete', async (root) => {
|
|
1348
|
+
const calls: Array<{ method: string; path: string }> = [];
|
|
1349
|
+
const execute = fakeExecute({ files: [{ id: 'file-api-realone', object: 'file', bytes: 1, created_at: 1, filename: 'real.png', purpose: 'user_data' }] }, calls);
|
|
1350
|
+
await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
|
|
1351
|
+
const del = await handleDeepSeekTwinRequest({ method: 'DELETE', path: `${FILES}/file-api-realone`, root, occurredAt: OCCURRED_AT });
|
|
1352
|
+
if (!ok(del)) return false;
|
|
1353
|
+
if (pendingActions('deepseek', root).filter((a) => a.subject.type === 'file').length !== 1) return false;
|
|
1354
|
+
const res = await pushPendingDeepSeekActions(execute, { root, occurredAt: '2026-08-31T12:10:00.000Z' });
|
|
1355
|
+
return res.pushed === 1
|
|
1356
|
+
&& calls.some((c) => c.method === 'DELETE' && c.path === '/files/file-api-realone')
|
|
1357
|
+
&& pendingActions('deepseek', root).filter((a) => a.subject.type === 'file').length === 0;
|
|
1358
|
+
})),
|
|
1359
|
+
|
|
1360
|
+
done('deepseek.connector.refuses_the_twins_own_id', 'connector', "A locally-minted file's delete is REFUSED, never issued against the real account under the twin's own id", 'connector', 'core', () => withConnectorRoot('deepseek.connector.refuses_the_twins_own_id', async (root) => {
|
|
1361
|
+
const calls: Array<{ method: string; path: string }> = [];
|
|
1362
|
+
const execute = fakeExecute({}, calls);
|
|
1363
|
+
await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD()), root, occurredAt: OCCURRED_AT });
|
|
1364
|
+
await handleDeepSeekTwinRequest({ method: 'DELETE', path: `${FILES}/${FILE_1}`, root, occurredAt: OCCURRED_AT });
|
|
1365
|
+
const res = await pushPendingDeepSeekActions(execute, { root, occurredAt: '2026-08-31T12:10:00.000Z' });
|
|
1366
|
+
// Both actions must be refused — the create because the vendor endpoint is multipart, the
|
|
1367
|
+
// delete because no vendor id was ever recorded — and NOTHING may have gone out.
|
|
1368
|
+
if (res.pushed !== 0 || res.refused.length !== 2) return false;
|
|
1369
|
+
if (calls.length !== 0) return false;
|
|
1370
|
+
if (!res.refused.some((r) => r.operation === 'file.create' && r.reason.includes('multipart'))) return false;
|
|
1371
|
+
if (!res.refused.some((r) => r.operation === 'file.delete' && r.reason.includes('no vendor id recorded'))) return false;
|
|
1372
|
+
// And the path builder refuses the twin's mint outright, not merely by convention.
|
|
1373
|
+
if (externalIdFor('file', FILE_1, root) !== null) return false;
|
|
1374
|
+
let threw = false;
|
|
1375
|
+
try { deepseekRequestForAction({ operation: 'file.delete', subject: { type: 'file', id: FILE_1 } } as never); } catch { threw = true; }
|
|
1376
|
+
return threw;
|
|
1377
|
+
})),
|
|
1378
|
+
|
|
1379
|
+
done('deepseek.connector.skips_the_internal_cache_ledger', 'connector', 'The twin-only context-cache ledger is skipped BY NAME rather than pushed or silently dropped', 'connector', 'common', () => withConnectorRoot('deepseek.connector.skips_the_internal_cache_ledger', async (root) => {
|
|
1380
|
+
const calls: Array<{ method: string; path: string }> = [];
|
|
1381
|
+
const execute = fakeExecute({}, calls);
|
|
1382
|
+
await handleDeepSeekTwinRequest({ method: 'POST', path: CHAT_PATH, body: JSON.stringify(CHAT()), root, occurredAt: OCCURRED_AT });
|
|
1383
|
+
const pendingCache = pendingActions('deepseek', root).filter((a) => a.subject.type === 'cache_prefix');
|
|
1384
|
+
if (pendingCache.length === 0) return false; // the completion really did mint ledger actions
|
|
1385
|
+
const res = await pushPendingDeepSeekActions(execute, { root, occurredAt: '2026-08-31T12:10:00.000Z' });
|
|
1386
|
+
// Reported, not refused (they are not a gap) and not pushed (DeepSeek has no cache endpoint).
|
|
1387
|
+
return res.skippedInternal.length === pendingCache.length && res.refused.length === 0 && calls.length === 0
|
|
1388
|
+
&& unpushableReason('cache_prefix.observe') !== null;
|
|
1389
|
+
})),
|
|
1390
|
+
|
|
1391
|
+
done('deepseek.connector.pull_after_local_create', 'connector', 'ID COLLISION, both directions: a pulled vendor id and a locally-minted one never collide', 'connector', 'common', () => withConnectorRoot('deepseek.connector.pull_after_local_create', async (root) => {
|
|
1392
|
+
// Direction 1 — local create, THEN pull: the pulled row must not overwrite the local one.
|
|
1393
|
+
await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD({ filename: 'local.png' })), root, occurredAt: OCCURRED_AT });
|
|
1394
|
+
const execute = fakeExecute({ files: [{ id: 'file-api-vendor01', object: 'file', bytes: 9, created_at: 2, filename: 'vendor.png', purpose: 'user_data' }] });
|
|
1395
|
+
await syncDeepSeekFromReal(execute, { root, occurredAt: OCCURRED_AT });
|
|
1396
|
+
const local = await handleDeepSeekTwinRequest({ method: 'GET', path: `${FILES}/${FILE_1}`, root });
|
|
1397
|
+
const vendor = await handleDeepSeekTwinRequest({ method: 'GET', path: '/files/file-api-vendor01', root });
|
|
1398
|
+
if (!ok(local) || body(local).filename !== 'local.png' || !ok(vendor)) return false;
|
|
1399
|
+
// Direction 2 — create AFTER the pull: the new mint must not land on the pulled id, and the
|
|
1400
|
+
// local namespace must ratchet from the local set only.
|
|
1401
|
+
const next = await handleDeepSeekTwinRequest({ method: 'POST', path: FILES, body: JSON.stringify(UPLOAD({ filename: 'after.png' })), root, occurredAt: OCCURRED_AT });
|
|
1402
|
+
return ok(next) && body(next).id === FILE_2 && body(next).id !== 'file-api-vendor01';
|
|
1403
|
+
})),
|
|
1404
|
+
|
|
1405
|
+
done('deepseek.connector.full_sync', 'connector', 'fullSync pushes pending writes then pulls every modeled collection and the balance singleton', 'connector', 'common', () => withConnectorRoot('deepseek.connector.full_sync', async (root) => {
|
|
1406
|
+
const calls: Array<{ method: string; path: string }> = [];
|
|
1407
|
+
const execute = fakeExecute({
|
|
1408
|
+
models: [{ id: 'deepseek-v4-flash', object: 'model', owned_by: 'deepseek' }],
|
|
1409
|
+
files: [{ id: 'file-api-sync01', object: 'file', bytes: 3, created_at: 5, filename: 's.png', purpose: 'user_data' }],
|
|
1410
|
+
balance: { is_available: true, balance_infos: [{ currency: 'USD', total_balance: '5.00', granted_balance: '0.00', topped_up_balance: '5.00' }] },
|
|
1411
|
+
}, calls);
|
|
1412
|
+
const res = await fullSyncDeepSeek(execute, { root, occurredAt: OCCURRED_AT });
|
|
1413
|
+
const paths = calls.map((c) => `${c.method} ${c.path}`);
|
|
1414
|
+
return res.collections === 3 && res.observed === 3 && res.deltasAppended > 0
|
|
1415
|
+
&& paths.includes('GET /models') && paths.includes('GET /files') && paths.includes('GET /user/balance');
|
|
1416
|
+
})),
|
|
1417
|
+
|
|
1418
|
+
todo('deepseek.connector.push_file_create', 'connector', "Pushing a local file create — real DeepSeek's POST /files is multipart/form-data with the image bytes, which the JSON executor cannot express and the twin has no real bytes for", 'connector', 'common'),
|
|
1419
|
+
todo('deepseek.connector.pull_pagination', 'connector', 'Following the Files API `after` cursor across more than one page during a pull', 'connector', 'niche'),
|
|
1420
|
+
todo('deepseek.connector.unpushable_actions_drain', 'connector', 'Acknowledging a permanently-unpushable action so `pendingActions` can reach empty again', 'connector', 'niche'),
|
|
1421
|
+
|
|
1422
|
+
// ══ conformance ══════════════════════════════════════════════════════════════════════════
|
|
1423
|
+
done('deepseek.conformance.probes', 'conformance', 'Offline conformance harness passes: probes, the router census, the must-not-serve list and the rejection table', 'api', 'core', async () => {
|
|
1424
|
+
const { checkDeepSeekConformance } = await import('./deepseek-conformance.ts');
|
|
1425
|
+
const report = await checkDeepSeekConformance();
|
|
1426
|
+
return report.ok && report.probes >= 11 && report.checksRun >= 60;
|
|
1427
|
+
}),
|
|
1428
|
+
];
|
|
1429
|
+
|
|
1430
|
+
// TWIN-87 committed area census — DeepSeek's top-level API product areas, authored top-down from
|
|
1431
|
+
// the api-docs.deepseek.com nav (Quick Start, API Reference, and each Guide) independently of what a
|
|
1432
|
+
// manifest entry happens to already exist for. deepseek-capabilities.test.ts's area-census meta-test
|
|
1433
|
+
// (assertAreaCensus) fails the gate if a declared area has zero manifest entries, OR if a manifest
|
|
1434
|
+
// entry's `area` drifts outside this list — so a whole missing area can never hide invisibly.
|
|
1435
|
+
export const DEEPSEEK_AREAS = [
|
|
1436
|
+
'anthropic', 'auth', 'balance', 'beta', 'cache', 'chat', 'completions', 'conformance',
|
|
1437
|
+
'connector', 'errors', 'files', 'models', 'rate_limits', 'reasoning', 'responses',
|
|
1438
|
+
'streaming', 'structured_outputs', 'tools', 'usage', 'vision',
|
|
1439
|
+
] as const;
|
|
1440
|
+
|
|
1441
|
+
export function deepseekCapabilities(): Promise<CapabilityReport> {
|
|
1442
|
+
return checkCapabilities('deepseek', DEEPSEEK_CAPABILITIES);
|
|
1443
|
+
}
|