@volter/twin-deepseek 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +198 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/deepseek-budget.d.ts +51 -0
- package/dist/src/deepseek-budget.js +152 -0
- package/dist/src/deepseek-cache.d.ts +56 -0
- package/dist/src/deepseek-cache.js +151 -0
- package/dist/src/deepseek-capabilities.d.ts +4 -0
- package/dist/src/deepseek-capabilities.js +1520 -0
- package/dist/src/deepseek-conformance.d.ts +14 -0
- package/dist/src/deepseek-conformance.js +473 -0
- package/dist/src/deepseek-connector.d.ts +168 -0
- package/dist/src/deepseek-connector.js +386 -0
- package/dist/src/deepseek-models.d.ts +30 -0
- package/dist/src/deepseek-models.js +38 -0
- package/dist/src/deepseek-scenario.d.ts +55 -0
- package/dist/src/deepseek-scenario.js +170 -0
- package/dist/src/deepseek-server.d.ts +16 -0
- package/dist/src/deepseek-server.js +191 -0
- package/dist/src/deepseek-stub.d.ts +75 -0
- package/dist/src/deepseek-stub.js +191 -0
- package/dist/src/deepseek-twin.d.ts +77 -0
- package/dist/src/deepseek-twin.js +1103 -0
- package/dist/src/deepseek-types.d.ts +172 -0
- package/dist/src/deepseek-types.js +26 -0
- package/dist/src/index.d.ts +15 -0
- package/dist/src/index.js +93 -0
- package/package.json +68 -0
- package/src/cli.ts +27 -0
- package/src/deepseek-budget.ts +178 -0
- package/src/deepseek-cache.ts +159 -0
- package/src/deepseek-capabilities.ts +1443 -0
- package/src/deepseek-conformance.ts +512 -0
- package/src/deepseek-connector.ts +440 -0
- package/src/deepseek-models.ts +65 -0
- package/src/deepseek-scenario.ts +188 -0
- package/src/deepseek-server.ts +201 -0
- package/src/deepseek-stub.ts +200 -0
- package/src/deepseek-twin.ts +1163 -0
- package/src/deepseek-types.ts +201 -0
- package/src/index.ts +133 -0
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
// The deepseek pack's scenario system on the kernel's ONE engine (@volter/world-core scenario.ts):
|
|
2
|
+
// DeepSeek chat-completions vocabulary + the scripted-turn respond shape. THE KERNEL ENGINE IS NOT
|
|
3
|
+
// RE-IMPLEMENTED HERE — this file is only the pack's adapter (features, matchers, load
|
|
4
|
+
// validation) plus the realizer that turns a validated `respond` into DeepSeek's own envelope.
|
|
5
|
+
//
|
|
6
|
+
// The handler FILE (handlers/deepseek.json in a world dir) is the only write surface; the doors
|
|
7
|
+
// (`GET /twin`, `GET /twin/scenario`) are read-only. Scenario support is twin-only scaffolding
|
|
8
|
+
// for eval worlds, NOT vendor surface, so it is deliberately absent from the capability
|
|
9
|
+
// manifest (ADDING_A_TWIN.md §6, "Scenario scripting") and gated by deepseek-scenario.test.ts.
|
|
10
|
+
import { getActiveWorldStore, parseScenarioDocument, type PackScenarioAdapter, ScenarioError, type ScenarioDocument, ScenarioEngine, type ScenarioFeatures } from '@volter/world-core';
|
|
11
|
+
import { contentToText, lastUserText } from './deepseek-stub.ts';
|
|
12
|
+
import type { DeepSeekMessageParam, DeepSeekToolCall } from './deepseek-types.ts';
|
|
13
|
+
|
|
14
|
+
export type DeepSeekScenarioRequest = {
|
|
15
|
+
model: string;
|
|
16
|
+
messages: DeepSeekMessageParam[];
|
|
17
|
+
tools?: unknown;
|
|
18
|
+
/** DeepSeek-specific: 'enabled' | 'disabled'. Scriptable because thinking mode is what decides
|
|
19
|
+
* whether a turn carries `reasoning_content` at all. */
|
|
20
|
+
thinking?: string;
|
|
21
|
+
};
|
|
22
|
+
export type DeepSeekScenarioEngine = ScenarioEngine<DeepSeekScenarioRequest>;
|
|
23
|
+
|
|
24
|
+
export type ScenarioToolCall = { name: string; arguments: Record<string, unknown>; id?: string };
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* What a deepseek handler may script. `reasoning` becomes the assistant turn's
|
|
28
|
+
* `reasoning_content` (DeepSeek-specific — OpenAI has no such field); `error` scripts one of
|
|
29
|
+
* DeepSeek's own documented failure statuses rather than a success.
|
|
30
|
+
*/
|
|
31
|
+
export type DeepSeekScenarioRespond = {
|
|
32
|
+
text?: string;
|
|
33
|
+
reasoning?: string;
|
|
34
|
+
toolCalls?: ScenarioToolCall | ScenarioToolCall[];
|
|
35
|
+
finishReason?: 'stop' | 'length' | 'content_filter' | 'tool_calls' | 'insufficient_system_resource';
|
|
36
|
+
/** A scripted vendor failure. Every type here maps onto a status DeepSeek's own error table
|
|
37
|
+
* publishes (api-docs.deepseek.com/quick_start/error_codes): 429, 402, 500 and 503 respectively.
|
|
38
|
+
* `insufficient_balance` and `server_overloaded` have no OpenAI counterpart. */
|
|
39
|
+
error?: { type: 'rate_limit_exceeded' | 'insufficient_balance' | 'internal_server_error' | 'server_overloaded'; message?: string };
|
|
40
|
+
};
|
|
41
|
+
|
|
42
|
+
export type ScriptedResult = {
|
|
43
|
+
text: string | null;
|
|
44
|
+
reasoning: string | null;
|
|
45
|
+
toolCalls: DeepSeekToolCall[];
|
|
46
|
+
finishReason: 'stop' | 'length' | 'content_filter' | 'tool_calls' | 'insufficient_system_resource';
|
|
47
|
+
};
|
|
48
|
+
|
|
49
|
+
const RESPOND_KEYS = new Set(['text', 'reasoning', 'toolCalls', 'finishReason', 'error']);
|
|
50
|
+
// DeepSeek's own finish_reason set — note `insufficient_system_resource`, which has no OpenAI
|
|
51
|
+
// counterpart, and the absence of OpenAI's `function_call`.
|
|
52
|
+
const FINISH_REASONS = new Set(['stop', 'length', 'content_filter', 'tool_calls', 'insufficient_system_resource']);
|
|
53
|
+
const ERROR_TYPES = new Set(['rate_limit_exceeded', 'insufficient_balance', 'internal_server_error', 'server_overloaded']);
|
|
54
|
+
const THINKING_STATES = new Set(['enabled', 'disabled']);
|
|
55
|
+
const nonEmptyString = (cond: unknown): cond is string => typeof cond === 'string' && cond.length > 0;
|
|
56
|
+
|
|
57
|
+
function lastToolResultNames(messages: DeepSeekMessageParam[]): Set<string> {
|
|
58
|
+
const names = new Set<string>();
|
|
59
|
+
const last = messages[messages.length - 1] as { role?: string; tool_call_id?: unknown } | undefined;
|
|
60
|
+
if (!last || last.role !== 'tool' || typeof last.tool_call_id !== 'string') return names;
|
|
61
|
+
for (const m of messages) {
|
|
62
|
+
const am = m as { role?: string; tool_calls?: Array<{ id?: unknown; function?: { name?: unknown } }> };
|
|
63
|
+
if (am.role !== 'assistant' || !Array.isArray(am.tool_calls)) continue;
|
|
64
|
+
for (const tc of am.tool_calls) {
|
|
65
|
+
if (tc?.id === last.tool_call_id && typeof tc?.function?.name === 'string') names.add(tc.function.name);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
return names;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function toolNames(tools: unknown): string[] {
|
|
72
|
+
if (!Array.isArray(tools)) return [];
|
|
73
|
+
return (tools as Array<{ function?: { name?: unknown }; name?: unknown }>).map((t) =>
|
|
74
|
+
typeof t?.function?.name === 'string' ? t.function.name : typeof t?.name === 'string' ? t.name : null,
|
|
75
|
+
).filter((n): n is string => n !== null);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export const deepseekScenarioAdapter: PackScenarioAdapter<DeepSeekScenarioRequest> = {
|
|
79
|
+
vendor: 'deepseek',
|
|
80
|
+
features: (req): ScenarioFeatures => ({
|
|
81
|
+
model: req.model,
|
|
82
|
+
lastUserText: lastUserText(req.messages).slice(0, 300),
|
|
83
|
+
tools: toolNames(req.tools),
|
|
84
|
+
lastMessageIsToolResult: (req.messages[req.messages.length - 1] as { role?: string } | undefined)?.role === 'tool',
|
|
85
|
+
toolResultFor: [...lastToolResultNames(req.messages)],
|
|
86
|
+
thinking: req.thinking ?? 'enabled',
|
|
87
|
+
}),
|
|
88
|
+
matchers: {
|
|
89
|
+
modelEquals: (req, cond) => nonEmptyString(cond) && req.model === cond,
|
|
90
|
+
userTextIncludes: (req, cond) => nonEmptyString(cond) && lastUserText(req.messages).toLowerCase().includes(cond.toLowerCase()),
|
|
91
|
+
anyTextIncludes: (req, cond) => nonEmptyString(cond) && req.messages.map((m) => contentToText(m.content)).join('\n').toLowerCase().includes(cond.toLowerCase()),
|
|
92
|
+
lastMessageIsToolResult: (req, cond) => typeof cond === 'boolean' && ((req.messages[req.messages.length - 1] as { role?: string } | undefined)?.role === 'tool') === cond,
|
|
93
|
+
toolResultFor: (req, cond) => nonEmptyString(cond) && lastToolResultNames(req.messages).has(cond),
|
|
94
|
+
hasTool: (req, cond) => nonEmptyString(cond) && toolNames(req.tools).includes(cond),
|
|
95
|
+
// DeepSeek-specific: script behaviour by thinking mode (V4 models think by default).
|
|
96
|
+
thinkingEquals: (req, cond) => nonEmptyString(cond) && (req.thinking ?? 'enabled') === cond,
|
|
97
|
+
},
|
|
98
|
+
text: (req) => req.messages.map((m) => contentToText(m.content)).join('\n'),
|
|
99
|
+
validateOn: (on) => {
|
|
100
|
+
for (const k of ['modelEquals', 'userTextIncludes', 'anyTextIncludes', 'toolResultFor', 'hasTool'] as const) {
|
|
101
|
+
if (on[k] !== undefined && (typeof on[k] !== 'string' || !on[k])) return `on.${k} is a non-empty string`;
|
|
102
|
+
}
|
|
103
|
+
if (on.lastMessageIsToolResult !== undefined && typeof on.lastMessageIsToolResult !== 'boolean') return 'on.lastMessageIsToolResult is a boolean';
|
|
104
|
+
if (on.thinkingEquals !== undefined) {
|
|
105
|
+
if (typeof on.thinkingEquals !== 'string' || !THINKING_STATES.has(on.thinkingEquals)) {
|
|
106
|
+
return `on.thinkingEquals is one of ${[...THINKING_STATES].join(', ')}`;
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
return null;
|
|
110
|
+
},
|
|
111
|
+
validateRespond: (respond) => {
|
|
112
|
+
if (typeof respond !== 'object' || respond === null || Array.isArray(respond)) return 'respond is an object { text?, reasoning?, toolCalls?, finishReason?, error? }';
|
|
113
|
+
const r = respond as Record<string, unknown>;
|
|
114
|
+
for (const k of Object.keys(r)) if (!RESPOND_KEYS.has(k)) return `respond: unknown key "${k}" (valid: ${[...RESPOND_KEYS].join(', ')})`;
|
|
115
|
+
if (r.error !== undefined) {
|
|
116
|
+
const e = r.error as Record<string, unknown>;
|
|
117
|
+
if (!e || typeof e !== 'object' || Array.isArray(e)) return 'respond.error is an object { type, message? }';
|
|
118
|
+
if (typeof e.type !== 'string' || !ERROR_TYPES.has(e.type)) return `respond.error.type is one of ${[...ERROR_TYPES].join(', ')}`;
|
|
119
|
+
if (e.message !== undefined && typeof e.message !== 'string') return 'respond.error.message is a string';
|
|
120
|
+
// An error handler scripts a FAILURE — mixing it with success content is a mis-typed rule.
|
|
121
|
+
for (const k of ['text', 'reasoning', 'toolCalls', 'finishReason']) if (r[k] !== undefined) return `respond.error cannot be combined with respond.${k}`;
|
|
122
|
+
return null;
|
|
123
|
+
}
|
|
124
|
+
if (r.text !== undefined && typeof r.text !== 'string') return 'respond.text is a string';
|
|
125
|
+
if (r.reasoning !== undefined && typeof r.reasoning !== 'string') return 'respond.reasoning is a string';
|
|
126
|
+
if (r.finishReason !== undefined && (typeof r.finishReason !== 'string' || !FINISH_REASONS.has(r.finishReason))) return `respond.finishReason is one of ${[...FINISH_REASONS].join(', ')}`;
|
|
127
|
+
if (r.toolCalls !== undefined) {
|
|
128
|
+
for (const tc of Array.isArray(r.toolCalls) ? r.toolCalls : [r.toolCalls]) {
|
|
129
|
+
const t = tc as Record<string, unknown>;
|
|
130
|
+
if (!t || typeof t !== 'object' || Array.isArray(t)) return 'respond.toolCalls entries are objects';
|
|
131
|
+
if (typeof t.name !== 'string' || !t.name) return 'respond.toolCalls[].name is a non-empty string';
|
|
132
|
+
if (!t.arguments || typeof t.arguments !== 'object' || Array.isArray(t.arguments)) return 'respond.toolCalls[].arguments is an object';
|
|
133
|
+
if (t.id !== undefined && typeof t.id !== 'string') return 'respond.toolCalls[].id is a string';
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
if (r.text === undefined && r.toolCalls === undefined) return 'respond needs text or toolCalls (or an error)';
|
|
137
|
+
return null;
|
|
138
|
+
},
|
|
139
|
+
};
|
|
140
|
+
|
|
141
|
+
/** Load + STRICTLY validate a scenario document. A malformed file throws at load with what is
|
|
142
|
+
* wrong — never a silent ignore-and-stub (a mis-typed rule falling back is a fake success). */
|
|
143
|
+
export function loadDeepSeekScenarioDocument(path: string): ScenarioDocument {
|
|
144
|
+
let parsed: unknown;
|
|
145
|
+
try {
|
|
146
|
+
// Read through the ACTIVE WorldStore, never the filesystem directly (runtime contract
|
|
147
|
+
// R12b): the handlers document is WORLD STATE, so a MemoryWorldStore / DO-backed world
|
|
148
|
+
// serves ITS OWN scenario instead of whatever happens to sit on the host disk — and the
|
|
149
|
+
// serve path stays workerd-clean. A missing document keeps the historical ENOENT wording,
|
|
150
|
+
// so the loud load-time failure reads byte-identically to the read it replaces.
|
|
151
|
+
const raw = getActiveWorldStore().read(path);
|
|
152
|
+
if (raw === null) throw new Error(`ENOENT: no such file or directory, open '${path}'`);
|
|
153
|
+
parsed = JSON.parse(raw);
|
|
154
|
+
} catch (e) {
|
|
155
|
+
throw new ScenarioError(`deepseek scenario: cannot read/parse ${path}: ${e instanceof Error ? e.message : String(e)}`);
|
|
156
|
+
}
|
|
157
|
+
return parseScenarioDocument(parsed, deepseekScenarioAdapter);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
export function createDeepSeekScenarioEngine(document?: ScenarioDocument): DeepSeekScenarioEngine {
|
|
161
|
+
return new ScenarioEngine(deepseekScenarioAdapter, document);
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Turn a validated `respond` into the pack's own faithful assistant turn.
|
|
166
|
+
*
|
|
167
|
+
* §9 ROUND TWO: this used a MODULE-LEVEL counter, so two identical scripted requests in one
|
|
168
|
+
* process got different `tool_calls[].id`s and the same request answered differently after a
|
|
169
|
+
* restart — a serve-path determinism violation (CLAUDE.md: a served response is a pure function of
|
|
170
|
+
* (request, stored state)). The id is now derived from the scripted call's own content and its
|
|
171
|
+
* position in the handler, exactly as the non-scenario path derives its ids.
|
|
172
|
+
*/
|
|
173
|
+
export function realizeDeepSeekRespond(respond: DeepSeekScenarioRespond): ScriptedResult {
|
|
174
|
+
const toolCalls: DeepSeekToolCall[] = [];
|
|
175
|
+
const scripted = respond.toolCalls ? (Array.isArray(respond.toolCalls) ? respond.toolCalls : [respond.toolCalls]) : [];
|
|
176
|
+
scripted.forEach((tc, index) => {
|
|
177
|
+
const seed = `${index}:${tc.name}:${JSON.stringify(tc.arguments)}`;
|
|
178
|
+
let h = 0x811c9dc5;
|
|
179
|
+
for (let i = 0; i < seed.length; i++) { h ^= seed.charCodeAt(i); h = Math.imul(h, 0x01000193); }
|
|
180
|
+
toolCalls.push({ id: tc.id ?? `call_scripted_${(h >>> 0).toString(36)}`, type: 'function', function: { name: tc.name, arguments: JSON.stringify(tc.arguments) } });
|
|
181
|
+
});
|
|
182
|
+
return {
|
|
183
|
+
text: respond.text ?? (toolCalls.length ? null : ''),
|
|
184
|
+
reasoning: respond.reasoning ?? null,
|
|
185
|
+
toolCalls,
|
|
186
|
+
finishReason: respond.finishReason ?? (toolCalls.length ? 'tool_calls' : 'stop'),
|
|
187
|
+
};
|
|
188
|
+
}
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
// DeepSeek twin HTTP server — serve the full DeepSeek twin handler over HTTP so the real
|
|
2
|
+
// `@ai-sdk/deepseek` works UNMODIFIED: `createDeepSeek({ baseURL: 'http://127.0.0.1:<port>' })` for
|
|
3
|
+
// the standard surface and `.../beta` for prefix completion + strict tool calls, exactly as a real
|
|
4
|
+
// integrator configures it. There is NO `/v1` segment — DeepSeek's own base_url has none. JSON
|
|
5
|
+
// bodies pass straight through. Writable by default; pass readOnly to reject mutations (D3).
|
|
6
|
+
//
|
|
7
|
+
// Streaming: when the request body has `"stream": true`, the server constructs a REAL SSE
|
|
8
|
+
// response by feeding the handler an sseSink that writes each chunk in DeepSeek's `data: <json>\n\n`
|
|
9
|
+
// wire format, ending with `data: [DONE]\n\n`. (The handler itself stays socket-free — the sink
|
|
10
|
+
// is the only place a socket is touched, on the live HTTP path.)
|
|
11
|
+
//
|
|
12
|
+
// Multipart: POST /files takes `multipart/form-data`. The server
|
|
13
|
+
// parses the form into the handler's JSON contract so the handler stays a pure JSON function.
|
|
14
|
+
//
|
|
15
|
+
// FETCH-FIRST (runtime contract R12b): the surface is the plain `createDeepSeekTwinFetch` and the
|
|
16
|
+
// SERVER is one line of `Bun.serve` around it. This is a CUSTOM fetch, not the kernel adapter
|
|
17
|
+
// (`createTwinFetchFromHandler`): SSE streaming plus the multipart adaptation genuinely exceed the
|
|
18
|
+
// common shape — openai-server.ts is the reference for that lane.
|
|
19
|
+
import { serveHttp } from '@volter/world-core';
|
|
20
|
+
import { twinManifest, worldNow } from '@volter/world-core';
|
|
21
|
+
import { createDeepSeekScenarioEngine, type DeepSeekScenarioEngine, loadDeepSeekScenarioDocument } from './deepseek-scenario.ts';
|
|
22
|
+
import { DEEPSEEK_BETA_PREFIX, handleDeepSeekTwinRequest } from './deepseek-twin.ts';
|
|
23
|
+
import type { SseEvent } from './deepseek-types.ts';
|
|
24
|
+
|
|
25
|
+
function wantsStream(body: string): boolean {
|
|
26
|
+
if (!body) return false;
|
|
27
|
+
try {
|
|
28
|
+
return (JSON.parse(body) as { stream?: unknown })?.stream === true;
|
|
29
|
+
} catch {
|
|
30
|
+
return false;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function encodeSse(event: SseEvent): string {
|
|
35
|
+
if (event.done) return 'data: [DONE]\n\n';
|
|
36
|
+
return `data: ${JSON.stringify(event.data)}\n\n`;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Both the standard and the beta chat endpoints stream. (`/beta/completions` — FIM — declares a
|
|
40
|
+
// `stream` parameter too; the twin does not model its SSE shape yet and files it as
|
|
41
|
+
// `deepseek.completions.streaming`, so it is deliberately absent here rather than served wrongly.)
|
|
42
|
+
const STREAMABLE = new Set(['/chat/completions', `${DEEPSEEK_BETA_PREFIX}/chat/completions`]);
|
|
43
|
+
|
|
44
|
+
async function multipartToJson(request: Request): Promise<string> {
|
|
45
|
+
try {
|
|
46
|
+
const form = await request.formData();
|
|
47
|
+
const out: Record<string, unknown> = {};
|
|
48
|
+
// Forward every scalar field generically. On DeepSeek that is `purpose` and the two
|
|
49
|
+
// `expires_after[...]` fields the vendor documents and @ai-sdk/deepseek appends verbatim
|
|
50
|
+
// (src/files/deepseek-files.ts). (§9 round one, SHOULD-FIX 6: this comment used to list
|
|
51
|
+
// `response_format`, `language` and `url` — Groq/OpenAI AUDIO-transcription fields that
|
|
52
|
+
// survived the clone. DeepSeek publishes no audio endpoint at all.)
|
|
53
|
+
for (const [key, value] of form.entries()) {
|
|
54
|
+
if (typeof value === 'string') {
|
|
55
|
+
// A repeated scalar field would arrive one entry at a time; kept generic rather than
|
|
56
|
+
// named after a field DeepSeek does not have.
|
|
57
|
+
if (key.endsWith('[]')) {
|
|
58
|
+
const k = key.slice(0, -2);
|
|
59
|
+
const prior = out[k];
|
|
60
|
+
out[k] = Array.isArray(prior) ? [...prior, value] : [value];
|
|
61
|
+
} else {
|
|
62
|
+
out[key] = value;
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
const file = form.get('file');
|
|
67
|
+
if (file instanceof File) {
|
|
68
|
+
const bytes = new Uint8Array(await file.arrayBuffer());
|
|
69
|
+
out.file = file.name || 'upload';
|
|
70
|
+
out.filename = file.name || 'upload';
|
|
71
|
+
// Base64 so binary image bytes survive the JSON handler contract intact.
|
|
72
|
+
out.content = Buffer.from(bytes).toString('base64');
|
|
73
|
+
// The declared media type is what DeepSeek's image-only Files API checks first.
|
|
74
|
+
out.media_type = file.type || '';
|
|
75
|
+
out.bytes = bytes.length;
|
|
76
|
+
}
|
|
77
|
+
return JSON.stringify(out);
|
|
78
|
+
} catch {
|
|
79
|
+
return '{}';
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** Options every DeepSeek-twin HTTP surface needs, independent of who owns the socket. */
|
|
84
|
+
export interface DeepSeekTwinFetchOptions {
|
|
85
|
+
root?: string;
|
|
86
|
+
readOnly?: boolean;
|
|
87
|
+
scenarioPath?: string;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export function createDeepSeekTwinFetch(options: DeepSeekTwinFetchOptions): (request: Request) => Promise<Response> {
|
|
91
|
+
const readOnly = options.readOnly ?? false;
|
|
92
|
+
// Scenario scripting (deepseek-scenario.ts): a JSON scenario file — via the scenarioPath option or
|
|
93
|
+
// the TWIN_DEEPSEEK_SCENARIO env var — scripts chat completions. Loaded ONCE at startup (a
|
|
94
|
+
// malformed file fails loudly here, never silently).
|
|
95
|
+
const scenarioPath = options.scenarioPath ?? process.env.TWIN_DEEPSEEK_SCENARIO;
|
|
96
|
+
const scenarioEngine: DeepSeekScenarioEngine | undefined = scenarioPath ? createDeepSeekScenarioEngine(loadDeepSeekScenarioDocument(scenarioPath)) : undefined;
|
|
97
|
+
return async function deepSeekTwinFetch(request: Request): Promise<Response> {
|
|
98
|
+
const url = new URL(request.url);
|
|
99
|
+
// THE READ DOORS (TWIN-PROGRAMMING-MODEL): discovery + inspection, read-only.
|
|
100
|
+
if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin') {
|
|
101
|
+
return Response.json(twinManifest({
|
|
102
|
+
vendor: 'deepseek',
|
|
103
|
+
twinOf: 'DeepSeek Platform API (OpenAI-compatible; base_url has NO /v1 segment, and /beta unlocks prefix completion + strict tools)',
|
|
104
|
+
stateSentence: 'Seed image files through the ordinary API (POST /files, multipart, purpose=user_data) with any key; the context-cache ledger fills itself as you chat.',
|
|
105
|
+
behaviorSentence: "Chat completions are scripted by MSW-shaped handlers in the world dir (handlers/deepseek.json): {on:{userTextIncludes|anyTextIncludes|modelEquals|hasTool|toolResultFor|lastMessageIsToolResult|thinkingEquals|nthCall}, respond:{text|reasoning|toolCalls, finishReason?} | {error:{type:rate_limit_exceeded|insufficient_balance|internal_server_error|server_overloaded}}, once?, phase?}. Unmatched requests answer a labeled stub naming this door.",
|
|
106
|
+
exampleHandler: { on: { userTextIncludes: 'summarize', hasTool: 'search_docs' }, respond: { text: 'Scripted summary.' }, once: true },
|
|
107
|
+
engine: scenarioEngine as never,
|
|
108
|
+
}));
|
|
109
|
+
}
|
|
110
|
+
if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin/scenario') {
|
|
111
|
+
return Response.json(scenarioEngine ? scenarioEngine.status() : { vendor: 'deepseek', handlers: [], misses: 0, recentMisses: [] });
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const path = url.pathname + (url.search || '');
|
|
115
|
+
// Collapse REPEATED slashes as well as the trailing one, exactly as `routeDeepSeek` does when
|
|
116
|
+
// it builds `seg` (and as the budget's `splitQuery` does for its anchored rules). §9 ROUND
|
|
117
|
+
// TWO, SHOULD-FIX 1: the two normalizations disagreed, so `POST /beta//chat/completions` with
|
|
118
|
+
// `stream: true` reached the streaming ROUTE but missed this STREAMABLE set and was answered
|
|
119
|
+
// with a unary JSON body to a client reading text/event-stream.
|
|
120
|
+
const cleanPath = url.pathname.replace(/\/{2,}/g, '/').replace(/\/+$/, '');
|
|
121
|
+
const contentType = request.headers.get('content-type') ?? '';
|
|
122
|
+
|
|
123
|
+
const passHeaders: Record<string, string> = {};
|
|
124
|
+
for (const k of ['authorization', 'x-twin-force-rate-limit']) {
|
|
125
|
+
const v = request.headers.get(k);
|
|
126
|
+
if (v !== null) passHeaders[k] = v;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
let body = '';
|
|
130
|
+
if (request.method !== 'GET') {
|
|
131
|
+
body = contentType.includes('multipart/form-data') ? await multipartToJson(request) : await request.text();
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
// Streaming POST → a real text/event-stream response built from the sink.
|
|
135
|
+
//
|
|
136
|
+
// Events are COLLECTED first, then framed. The handler is synchronous-fast, so buffering
|
|
137
|
+
// costs nothing — and it is what lets a PRE-STREAM failure (a 422 for an unknown model,
|
|
138
|
+
// a 400 for a body with no `messages`, auth 401, rate-limit 429) answer with its REAL
|
|
139
|
+
// status and the vendor's JSON error envelope. Emitting the refusal as a lone `data:`
|
|
140
|
+
// frame inside a 200 text/event-stream was a FAKE SUCCESS the in-process path could not
|
|
141
|
+
// see: `handleDeepSeekTwinRequest` returns 422 for `model: "nope"` while the wire
|
|
142
|
+
// returned 200, so every verify asserting that 422 asserted a status the socket never
|
|
143
|
+
// carried. DeepSeek rejects a bad request BEFORE opening the event stream.
|
|
144
|
+
if (!readOnly && request.method.toUpperCase() === 'POST' && STREAMABLE.has(cleanPath) && wantsStream(body)) {
|
|
145
|
+
const events: SseEvent[] = [];
|
|
146
|
+
const { status, body: out, headers: errHeaders } = await handleDeepSeekTwinRequest({
|
|
147
|
+
...(scenarioEngine ? { scenarioEngine } : {}),
|
|
148
|
+
method: request.method, path, body, readOnly, occurredAt: worldNow(), headers: passHeaders,
|
|
149
|
+
...(options.root !== undefined ? { root: options.root } : {}), sseSink: (e: SseEvent) => events.push(e),
|
|
150
|
+
});
|
|
151
|
+
// No `x-request-id`: DeepSeek documents no such response header, and §6's figma precedent
|
|
152
|
+
// is that a twin declines to fabricate headers the vendor does not publish (§9 NIT 10).
|
|
153
|
+
if (status >= 400) {
|
|
154
|
+
return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', ...(errHeaders ?? {}) } });
|
|
155
|
+
}
|
|
156
|
+
const stream = new ReadableStream<Uint8Array>({
|
|
157
|
+
start(controller) {
|
|
158
|
+
const enc = new TextEncoder();
|
|
159
|
+
for (const e of events) controller.enqueue(enc.encode(encodeSse(e)));
|
|
160
|
+
controller.close();
|
|
161
|
+
},
|
|
162
|
+
});
|
|
163
|
+
return new Response(stream, { headers: { 'content-type': 'text/event-stream; charset=utf-8', 'cache-control': 'no-cache' } });
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
// Defensive: a handler throw must become a vendor-shaped 500, never an escaped exception
|
|
167
|
+
// (ADDING_A_TWIN.md §8, "In-process server harnesses … contain handler errors").
|
|
168
|
+
let status: number;
|
|
169
|
+
let out: unknown;
|
|
170
|
+
let outHeaders: Record<string, string> | undefined;
|
|
171
|
+
try {
|
|
172
|
+
const res = await handleDeepSeekTwinRequest({
|
|
173
|
+
...(scenarioEngine ? { scenarioEngine } : {}),
|
|
174
|
+
method: request.method, path, body, readOnly,
|
|
175
|
+
occurredAt: worldNow(), headers: passHeaders,
|
|
176
|
+
...(options.root !== undefined ? { root: options.root } : {}),
|
|
177
|
+
});
|
|
178
|
+
status = res.status; out = res.body; outHeaders = res.headers;
|
|
179
|
+
} catch (err) {
|
|
180
|
+
return new Response(JSON.stringify({ error: { message: `twin handler error: ${err instanceof Error ? err.message : String(err)}`, type: 'internal_server_error' } }), {
|
|
181
|
+
status: 500, headers: { 'content-type': 'application/json' },
|
|
182
|
+
});
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
// NOTE there is deliberately no raw/string-body branch. The clone this pack came from had one
|
|
186
|
+
// for file-content downloads, `response_format:'text'` transcripts and a TTS stub body —
|
|
187
|
+
// three surfaces DeepSeek does not have (its Files API is image upload only, and it publishes
|
|
188
|
+
// no audio endpoints). `handleDeepSeekTwinRequest` never returns a string body, so the branch
|
|
189
|
+
// was dead code whose comment asserted vendor surface that does not exist (§9, SHOULD-FIX 6).
|
|
190
|
+
return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', ...(outHeaders ?? {}) } });
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
export async function createDeepSeekTwinServer(options: { root?: string; port?: number; readOnly?: boolean; scenarioPath?: string }): Promise<{ port: number; stop: () => void }> {
|
|
195
|
+
const server = await serveHttp({
|
|
196
|
+
port: options.port ?? 0,
|
|
197
|
+
idleTimeout: 60,
|
|
198
|
+
fetch: createDeepSeekTwinFetch(options),
|
|
199
|
+
});
|
|
200
|
+
return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
|
|
201
|
+
}
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
// THE GENERATIVE-STUB CORE — the deterministic labeled answer this twin gives where the vendor
|
|
2
|
+
// runs a model.
|
|
3
|
+
//
|
|
4
|
+
// The twin CANNOT run the model — there are no weights here. So `POST /chat/completions` returns a
|
|
5
|
+
// DETERMINISTIC STUB completion that is CLEARLY a twin stub, NEVER pretending to be real model
|
|
6
|
+
// output, and the same is true of `reasoning_content` and the beta FIM endpoint. What IS faithful
|
|
7
|
+
// is the ENTIRE PROTOCOL ENVELOPE: the response shape, the SSE chunk sequence, the `tool_calls`
|
|
8
|
+
// shape, `finish_reason`, and DeepSeek's KV-cache-bearing `usage`.
|
|
9
|
+
//
|
|
10
|
+
// SERVE-PATH DETERMINISM (CLAUDE.md): every number below is a pure function of (request, stored
|
|
11
|
+
// state). Note what is NOT here: DeepSeek's cache-hit accounting is NOT a random split and NOT a
|
|
12
|
+
// clock reading — it is computed from the twin's own observed cache-prefix ledger
|
|
13
|
+
// (deepseek-cache.ts), so a replay is byte-identical and a genuine prefix reuse is genuinely
|
|
14
|
+
// reported as a hit.
|
|
15
|
+
//
|
|
16
|
+
// The stub IS the twin's answer: the manifest's chat, reasoning and FIM capabilities are claimed
|
|
17
|
+
// on the envelope and the determinism. See the README ## Coverage.
|
|
18
|
+
|
|
19
|
+
import type { DeepSeekMessageParam, DeepSeekUsage } from './deepseek-types.ts';
|
|
20
|
+
|
|
21
|
+
/** Deterministic token estimate for a string: ~1 token per 4 chars (faithful order of magnitude;
|
|
22
|
+
* deterministic so usage counts are assertable, like the vendor's tokenizer on a fixed input).
|
|
23
|
+
* Never zero for non-empty text. */
|
|
24
|
+
export function estimateTokens(text: string): number {
|
|
25
|
+
if (!text) return 0;
|
|
26
|
+
return Math.max(1, Math.ceil(text.length / 4));
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** Flatten a chat message's content (string OR content-part array) to its text for token counting
|
|
30
|
+
* / echo. Non-text parts contribute their JSON length so the count is deterministic and reflects
|
|
31
|
+
* payload size. */
|
|
32
|
+
export function contentToText(content: DeepSeekMessageParam['content']): string {
|
|
33
|
+
if (typeof content === 'string') return content;
|
|
34
|
+
if (content === null || content === undefined) return '';
|
|
35
|
+
if (!Array.isArray(content)) return '';
|
|
36
|
+
return content
|
|
37
|
+
.map((part) => {
|
|
38
|
+
if (part && typeof part === 'object' && (part as { type?: string }).type === 'text') {
|
|
39
|
+
return String((part as { text?: unknown }).text ?? '');
|
|
40
|
+
}
|
|
41
|
+
return JSON.stringify(part);
|
|
42
|
+
})
|
|
43
|
+
.join('\n');
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Deterministic prompt-token count for one message (the unit the cache ledger also counts in, so
|
|
47
|
+
* a prefix's hit tokens and the request's prompt tokens are measured the same way). */
|
|
48
|
+
export function messageTokens(m: DeepSeekMessageParam): number {
|
|
49
|
+
let total = estimateTokens(contentToText(m.content));
|
|
50
|
+
if (m.name) total += estimateTokens(m.name);
|
|
51
|
+
if (typeof m.reasoning_content === 'string') total += estimateTokens(m.reasoning_content);
|
|
52
|
+
for (const tc of m.tool_calls ?? []) total += estimateTokens(JSON.stringify(tc));
|
|
53
|
+
return total;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** Deterministic prompt-token count for the full set of messages. */
|
|
57
|
+
export function countPromptTokens(messages: DeepSeekMessageParam[]): number {
|
|
58
|
+
let total = 0;
|
|
59
|
+
for (const m of messages) total += messageTokens(m);
|
|
60
|
+
return total;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** The last user turn's text — the thing the stub echoes (deterministic, clearly labeled). */
|
|
64
|
+
export function lastUserText(messages: DeepSeekMessageParam[]): string {
|
|
65
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
66
|
+
if (messages[i]!.role === 'user') return contentToText(messages[i]!.content);
|
|
67
|
+
}
|
|
68
|
+
return messages.length ? contentToText(messages[messages.length - 1]!.content) : '';
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Build the deterministic stub ASSISTANT text. It is unmistakably a twin stub: it carries the
|
|
73
|
+
* `[twin-stub:<model>]` marker and echoes the prompt, so no caller can mistake it for real model
|
|
74
|
+
* output. Deterministic for a given prompt → assertable in tests.
|
|
75
|
+
*/
|
|
76
|
+
export function stubAssistantText(messages: DeepSeekMessageParam[], model: string): string {
|
|
77
|
+
const prompt = lastUserText(messages).trim();
|
|
78
|
+
const echo = prompt.length > 200 ? `${prompt.slice(0, 200)}…` : prompt;
|
|
79
|
+
return `[twin-stub:${model}] This is a deterministic stub from the DeepSeek twin (no model weights are run). Echoing your last message: ${echo || '(empty)'}`;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* The `reasoning_content` DeepSeek returns while thinking mode is enabled (the default for every
|
|
84
|
+
* V4 model). A labeled stub, like the content — never a real chain of thought.
|
|
85
|
+
*/
|
|
86
|
+
export function stubReasoningText(messages: DeepSeekMessageParam[], model: string, effort: string): string {
|
|
87
|
+
return `[twin-stub:${model}] deterministic stub reasoning_content (no model weights are run, reasoning_effort=${effort}) for: ${lastUserText(messages).trim().slice(0, 120) || '(empty)'}`;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** The deterministic FIM continuation for `POST /beta/completions`. Labeled, and it names the
|
|
91
|
+
* suffix it was asked to bridge to so the caller can see the twin honoured the parameter. */
|
|
92
|
+
export function stubFimText(model: string, prompt: string, suffix: string | undefined): string {
|
|
93
|
+
const bridge = suffix ? ` bridging to suffix "${suffix.slice(0, 60)}"` : '';
|
|
94
|
+
return `[twin-stub:${model}] deterministic fill-in-the-middle stub (no model weights are run)${bridge} after: ${prompt.trim().slice(0, 120) || '(empty)'}`;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Assemble DeepSeek's `usage`. The cache split is supplied by the caller (from the cache-prefix
|
|
99
|
+
* ledger) and the invariant DeepSeek's own docs state is enforced here rather than assumed:
|
|
100
|
+
* `prompt_cache_hit_tokens + prompt_cache_miss_tokens === prompt_tokens`
|
|
101
|
+
* (api-docs.deepseek.com/guides/kv_cache). `prompt_tokens_details.cached_tokens` mirrors the hit
|
|
102
|
+
* count — `@ai-sdk/deepseek`'s usage schema declares it and reads `prompt_cache_hit_tokens` for
|
|
103
|
+
* `cacheRead`, so the two must agree.
|
|
104
|
+
*/
|
|
105
|
+
export function buildUsage(
|
|
106
|
+
promptTokens: number,
|
|
107
|
+
completionTokens: number,
|
|
108
|
+
cacheHitTokens: number,
|
|
109
|
+
reasoningTokens = 0,
|
|
110
|
+
): DeepSeekUsage {
|
|
111
|
+
const hit = Math.max(0, Math.min(cacheHitTokens, promptTokens));
|
|
112
|
+
return {
|
|
113
|
+
prompt_tokens: promptTokens,
|
|
114
|
+
completion_tokens: completionTokens,
|
|
115
|
+
total_tokens: promptTokens + completionTokens,
|
|
116
|
+
prompt_cache_hit_tokens: hit,
|
|
117
|
+
prompt_cache_miss_tokens: promptTokens - hit,
|
|
118
|
+
prompt_tokens_details: { cached_tokens: hit },
|
|
119
|
+
...(reasoningTokens > 0 ? { completion_tokens_details: { reasoning_tokens: reasoningTokens } } : {}),
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/** Extract a tool's function name from a DeepSeek/OpenAI-shaped tool
|
|
124
|
+
* (`{ type:'function', function:{ name } }`). */
|
|
125
|
+
function toolName(t: unknown): string {
|
|
126
|
+
const o = t as { function?: { name?: unknown } } | undefined;
|
|
127
|
+
if (o?.function && typeof o.function.name === 'string') return o.function.name;
|
|
128
|
+
return 'unknown_function';
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function placeholderForSchema(def: unknown): unknown {
|
|
132
|
+
const d = def as { type?: unknown; enum?: unknown[] } | undefined;
|
|
133
|
+
if (Array.isArray(d?.enum) && d!.enum!.length) return d!.enum![0];
|
|
134
|
+
switch (d?.type) {
|
|
135
|
+
case 'number':
|
|
136
|
+
case 'integer': return 0;
|
|
137
|
+
case 'boolean': return false;
|
|
138
|
+
case 'array': return [];
|
|
139
|
+
case 'object': return {};
|
|
140
|
+
default: return '';
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Build a deterministic stub argument string for a tool. When the tool declares a JSON-schema
|
|
146
|
+
* `parameters` object, real models emit arguments that validate against the schema; the twin
|
|
147
|
+
* synthesizes a deterministic object containing every declared property with a type-appropriate
|
|
148
|
+
* placeholder so strict callers parse it cleanly.
|
|
149
|
+
*/
|
|
150
|
+
export function stubToolArguments(tool: unknown): string {
|
|
151
|
+
const o = tool as { function?: { parameters?: unknown } } | undefined;
|
|
152
|
+
const schema = o?.function?.parameters as { properties?: Record<string, unknown> } | undefined;
|
|
153
|
+
const props = schema && typeof schema === 'object' ? schema.properties : undefined;
|
|
154
|
+
if (!props || typeof props !== 'object') return '{}';
|
|
155
|
+
const out: Record<string, unknown> = {};
|
|
156
|
+
for (const [key, def] of Object.entries(props)) out[key] = placeholderForSchema(def);
|
|
157
|
+
return JSON.stringify(out);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* When tools are provided, real models may respond with `tool_calls` and
|
|
162
|
+
* `finish_reason:'tool_calls'`. The stub deterministically "calls" the tool selected by
|
|
163
|
+
* `forcedName` (a named tool_choice) or the FIRST provided tool. Returns null when no tools.
|
|
164
|
+
*/
|
|
165
|
+
export function stubToolCall(tools: unknown, seq: number, forcedName?: string): { id: string; type: 'function'; function: { name: string; arguments: string } } | null {
|
|
166
|
+
if (!Array.isArray(tools) || tools.length === 0) return null;
|
|
167
|
+
const chosen = forcedName ? (tools.find((t) => toolName(t) === forcedName) ?? tools[0]) : tools[0];
|
|
168
|
+
return { id: `call_twin_${seq}`, type: 'function', function: { name: toolName(chosen), arguments: stubToolArguments(chosen) } };
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* A deterministic JSON-object stub for `response_format: { type: 'json_object' }`. DeepSeek's JSON
|
|
173
|
+
* output mode guarantees only that the content parses as JSON — there is no schema on this vendor
|
|
174
|
+
* (`json_schema` is rejected, see deepseek-twin.ts), so the twin returns a clearly-labeled
|
|
175
|
+
* deterministic object rather than fabricating schema conformance it was never given.
|
|
176
|
+
*/
|
|
177
|
+
export function stubJsonObject(messages: DeepSeekMessageParam[], model: string): string {
|
|
178
|
+
return JSON.stringify({ _twin_stub: true, model, echo: lastUserText(messages).slice(0, 200) });
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// ── Deterministic hashing ───────────────────────────────────────────────────────────────
|
|
182
|
+
/** A small deterministic 32-bit hash (FNV-1a) of a string. */
|
|
183
|
+
export function fnv1a(text: string): number {
|
|
184
|
+
let h = 0x811c9dc5;
|
|
185
|
+
for (let i = 0; i < text.length; i++) {
|
|
186
|
+
h ^= text.charCodeAt(i);
|
|
187
|
+
h = Math.imul(h, 0x01000193);
|
|
188
|
+
}
|
|
189
|
+
return h >>> 0;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* DeepSeek's `system_fingerprint` is a backend-configuration string like
|
|
194
|
+
* `fp_eaab8d114b_prod0820_fp8_kvcache` (`@ai-sdk/deepseek` docs/30-deepseek.mdx, "Provider
|
|
195
|
+
* Metadata"). The twin serves a value in that SHAPE that is unmistakably a twin fingerprint and is
|
|
196
|
+
* derived from the request, so a replay is byte-identical.
|
|
197
|
+
*/
|
|
198
|
+
export function stubFingerprint(seed: string): string {
|
|
199
|
+
return `fp_${fnv1a(seed).toString(16).padStart(8, '0')}_twin_stub_kvcache`;
|
|
200
|
+
}
|