@volter/twin-deepseek 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +198 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/deepseek-budget.d.ts +51 -0
- package/dist/src/deepseek-budget.js +152 -0
- package/dist/src/deepseek-cache.d.ts +56 -0
- package/dist/src/deepseek-cache.js +151 -0
- package/dist/src/deepseek-capabilities.d.ts +4 -0
- package/dist/src/deepseek-capabilities.js +1520 -0
- package/dist/src/deepseek-conformance.d.ts +14 -0
- package/dist/src/deepseek-conformance.js +473 -0
- package/dist/src/deepseek-connector.d.ts +168 -0
- package/dist/src/deepseek-connector.js +386 -0
- package/dist/src/deepseek-models.d.ts +30 -0
- package/dist/src/deepseek-models.js +38 -0
- package/dist/src/deepseek-scenario.d.ts +55 -0
- package/dist/src/deepseek-scenario.js +170 -0
- package/dist/src/deepseek-server.d.ts +16 -0
- package/dist/src/deepseek-server.js +191 -0
- package/dist/src/deepseek-stub.d.ts +75 -0
- package/dist/src/deepseek-stub.js +191 -0
- package/dist/src/deepseek-twin.d.ts +77 -0
- package/dist/src/deepseek-twin.js +1103 -0
- package/dist/src/deepseek-types.d.ts +172 -0
- package/dist/src/deepseek-types.js +26 -0
- package/dist/src/index.d.ts +15 -0
- package/dist/src/index.js +93 -0
- package/package.json +68 -0
- package/src/cli.ts +27 -0
- package/src/deepseek-budget.ts +178 -0
- package/src/deepseek-cache.ts +159 -0
- package/src/deepseek-capabilities.ts +1443 -0
- package/src/deepseek-conformance.ts +512 -0
- package/src/deepseek-connector.ts +440 -0
- package/src/deepseek-models.ts +65 -0
- package/src/deepseek-scenario.ts +188 -0
- package/src/deepseek-server.ts +201 -0
- package/src/deepseek-stub.ts +200 -0
- package/src/deepseek-twin.ts +1163 -0
- package/src/deepseek-types.ts +201 -0
- package/src/index.ts +133 -0
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
// THE GENERATIVE-STUB CORE — the deterministic labeled answer this twin gives where the vendor
|
|
2
|
+
// runs a model.
|
|
3
|
+
//
|
|
4
|
+
// The twin CANNOT run the model — there are no weights here. So `POST /chat/completions` returns a
|
|
5
|
+
// DETERMINISTIC STUB completion that is CLEARLY a twin stub, NEVER pretending to be real model
|
|
6
|
+
// output, and the same is true of `reasoning_content` and the beta FIM endpoint. What IS faithful
|
|
7
|
+
// is the ENTIRE PROTOCOL ENVELOPE: the response shape, the SSE chunk sequence, the `tool_calls`
|
|
8
|
+
// shape, `finish_reason`, and DeepSeek's KV-cache-bearing `usage`.
|
|
9
|
+
//
|
|
10
|
+
// SERVE-PATH DETERMINISM (CLAUDE.md): every number below is a pure function of (request, stored
|
|
11
|
+
// state). Note what is NOT here: DeepSeek's cache-hit accounting is NOT a random split and NOT a
|
|
12
|
+
// clock reading — it is computed from the twin's own observed cache-prefix ledger
|
|
13
|
+
// (deepseek-cache.ts), so a replay is byte-identical and a genuine prefix reuse is genuinely
|
|
14
|
+
// reported as a hit.
|
|
15
|
+
//
|
|
16
|
+
// The stub IS the twin's answer: the manifest's chat, reasoning and FIM capabilities are claimed
|
|
17
|
+
// on the envelope and the determinism. See the README ## Coverage.
|
|
18
|
+
/** Deterministic token estimate for a string: ~1 token per 4 chars (faithful order of magnitude;
|
|
19
|
+
* deterministic so usage counts are assertable, like the vendor's tokenizer on a fixed input).
|
|
20
|
+
* Never zero for non-empty text. */
|
|
21
|
+
export function estimateTokens(text) {
|
|
22
|
+
if (!text)
|
|
23
|
+
return 0;
|
|
24
|
+
return Math.max(1, Math.ceil(text.length / 4));
|
|
25
|
+
}
|
|
26
|
+
/** Flatten a chat message's content (string OR content-part array) to its text for token counting
|
|
27
|
+
* / echo. Non-text parts contribute their JSON length so the count is deterministic and reflects
|
|
28
|
+
* payload size. */
|
|
29
|
+
export function contentToText(content) {
|
|
30
|
+
if (typeof content === 'string')
|
|
31
|
+
return content;
|
|
32
|
+
if (content === null || content === undefined)
|
|
33
|
+
return '';
|
|
34
|
+
if (!Array.isArray(content))
|
|
35
|
+
return '';
|
|
36
|
+
return content
|
|
37
|
+
.map((part) => {
|
|
38
|
+
if (part && typeof part === 'object' && part.type === 'text') {
|
|
39
|
+
return String(part.text ?? '');
|
|
40
|
+
}
|
|
41
|
+
return JSON.stringify(part);
|
|
42
|
+
})
|
|
43
|
+
.join('\n');
|
|
44
|
+
}
|
|
45
|
+
/** Deterministic prompt-token count for one message (the unit the cache ledger also counts in, so
|
|
46
|
+
* a prefix's hit tokens and the request's prompt tokens are measured the same way). */
|
|
47
|
+
export function messageTokens(m) {
|
|
48
|
+
let total = estimateTokens(contentToText(m.content));
|
|
49
|
+
if (m.name)
|
|
50
|
+
total += estimateTokens(m.name);
|
|
51
|
+
if (typeof m.reasoning_content === 'string')
|
|
52
|
+
total += estimateTokens(m.reasoning_content);
|
|
53
|
+
for (const tc of m.tool_calls ?? [])
|
|
54
|
+
total += estimateTokens(JSON.stringify(tc));
|
|
55
|
+
return total;
|
|
56
|
+
}
|
|
57
|
+
/** Deterministic prompt-token count for the full set of messages. */
|
|
58
|
+
export function countPromptTokens(messages) {
|
|
59
|
+
let total = 0;
|
|
60
|
+
for (const m of messages)
|
|
61
|
+
total += messageTokens(m);
|
|
62
|
+
return total;
|
|
63
|
+
}
|
|
64
|
+
/** The last user turn's text — the thing the stub echoes (deterministic, clearly labeled). */
|
|
65
|
+
export function lastUserText(messages) {
|
|
66
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
67
|
+
if (messages[i].role === 'user')
|
|
68
|
+
return contentToText(messages[i].content);
|
|
69
|
+
}
|
|
70
|
+
return messages.length ? contentToText(messages[messages.length - 1].content) : '';
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Build the deterministic stub ASSISTANT text. It is unmistakably a twin stub: it carries the
|
|
74
|
+
* `[twin-stub:<model>]` marker and echoes the prompt, so no caller can mistake it for real model
|
|
75
|
+
* output. Deterministic for a given prompt → assertable in tests.
|
|
76
|
+
*/
|
|
77
|
+
export function stubAssistantText(messages, model) {
|
|
78
|
+
const prompt = lastUserText(messages).trim();
|
|
79
|
+
const echo = prompt.length > 200 ? `${prompt.slice(0, 200)}…` : prompt;
|
|
80
|
+
return `[twin-stub:${model}] This is a deterministic stub from the DeepSeek twin (no model weights are run). Echoing your last message: ${echo || '(empty)'}`;
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* The `reasoning_content` DeepSeek returns while thinking mode is enabled (the default for every
|
|
84
|
+
* V4 model). A labeled stub, like the content — never a real chain of thought.
|
|
85
|
+
*/
|
|
86
|
+
export function stubReasoningText(messages, model, effort) {
|
|
87
|
+
return `[twin-stub:${model}] deterministic stub reasoning_content (no model weights are run, reasoning_effort=${effort}) for: ${lastUserText(messages).trim().slice(0, 120) || '(empty)'}`;
|
|
88
|
+
}
|
|
89
|
+
/** The deterministic FIM continuation for `POST /beta/completions`. Labeled, and it names the
|
|
90
|
+
* suffix it was asked to bridge to so the caller can see the twin honoured the parameter. */
|
|
91
|
+
export function stubFimText(model, prompt, suffix) {
|
|
92
|
+
const bridge = suffix ? ` bridging to suffix "${suffix.slice(0, 60)}"` : '';
|
|
93
|
+
return `[twin-stub:${model}] deterministic fill-in-the-middle stub (no model weights are run)${bridge} after: ${prompt.trim().slice(0, 120) || '(empty)'}`;
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* Assemble DeepSeek's `usage`. The cache split is supplied by the caller (from the cache-prefix
|
|
97
|
+
* ledger) and the invariant DeepSeek's own docs state is enforced here rather than assumed:
|
|
98
|
+
* `prompt_cache_hit_tokens + prompt_cache_miss_tokens === prompt_tokens`
|
|
99
|
+
* (api-docs.deepseek.com/guides/kv_cache). `prompt_tokens_details.cached_tokens` mirrors the hit
|
|
100
|
+
* count — `@ai-sdk/deepseek`'s usage schema declares it and reads `prompt_cache_hit_tokens` for
|
|
101
|
+
* `cacheRead`, so the two must agree.
|
|
102
|
+
*/
|
|
103
|
+
export function buildUsage(promptTokens, completionTokens, cacheHitTokens, reasoningTokens = 0) {
|
|
104
|
+
const hit = Math.max(0, Math.min(cacheHitTokens, promptTokens));
|
|
105
|
+
return {
|
|
106
|
+
prompt_tokens: promptTokens,
|
|
107
|
+
completion_tokens: completionTokens,
|
|
108
|
+
total_tokens: promptTokens + completionTokens,
|
|
109
|
+
prompt_cache_hit_tokens: hit,
|
|
110
|
+
prompt_cache_miss_tokens: promptTokens - hit,
|
|
111
|
+
prompt_tokens_details: { cached_tokens: hit },
|
|
112
|
+
...(reasoningTokens > 0 ? { completion_tokens_details: { reasoning_tokens: reasoningTokens } } : {}),
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
/** Extract a tool's function name from a DeepSeek/OpenAI-shaped tool
|
|
116
|
+
* (`{ type:'function', function:{ name } }`). */
|
|
117
|
+
function toolName(t) {
|
|
118
|
+
const o = t;
|
|
119
|
+
if (o?.function && typeof o.function.name === 'string')
|
|
120
|
+
return o.function.name;
|
|
121
|
+
return 'unknown_function';
|
|
122
|
+
}
|
|
123
|
+
function placeholderForSchema(def) {
|
|
124
|
+
const d = def;
|
|
125
|
+
if (Array.isArray(d?.enum) && d.enum.length)
|
|
126
|
+
return d.enum[0];
|
|
127
|
+
switch (d?.type) {
|
|
128
|
+
case 'number':
|
|
129
|
+
case 'integer': return 0;
|
|
130
|
+
case 'boolean': return false;
|
|
131
|
+
case 'array': return [];
|
|
132
|
+
case 'object': return {};
|
|
133
|
+
default: return '';
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
/**
|
|
137
|
+
* Build a deterministic stub argument string for a tool. When the tool declares a JSON-schema
|
|
138
|
+
* `parameters` object, real models emit arguments that validate against the schema; the twin
|
|
139
|
+
* synthesizes a deterministic object containing every declared property with a type-appropriate
|
|
140
|
+
* placeholder so strict callers parse it cleanly.
|
|
141
|
+
*/
|
|
142
|
+
export function stubToolArguments(tool) {
|
|
143
|
+
const o = tool;
|
|
144
|
+
const schema = o?.function?.parameters;
|
|
145
|
+
const props = schema && typeof schema === 'object' ? schema.properties : undefined;
|
|
146
|
+
if (!props || typeof props !== 'object')
|
|
147
|
+
return '{}';
|
|
148
|
+
const out = {};
|
|
149
|
+
for (const [key, def] of Object.entries(props))
|
|
150
|
+
out[key] = placeholderForSchema(def);
|
|
151
|
+
return JSON.stringify(out);
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* When tools are provided, real models may respond with `tool_calls` and
|
|
155
|
+
* `finish_reason:'tool_calls'`. The stub deterministically "calls" the tool selected by
|
|
156
|
+
* `forcedName` (a named tool_choice) or the FIRST provided tool. Returns null when no tools.
|
|
157
|
+
*/
|
|
158
|
+
export function stubToolCall(tools, seq, forcedName) {
|
|
159
|
+
if (!Array.isArray(tools) || tools.length === 0)
|
|
160
|
+
return null;
|
|
161
|
+
const chosen = forcedName ? (tools.find((t) => toolName(t) === forcedName) ?? tools[0]) : tools[0];
|
|
162
|
+
return { id: `call_twin_${seq}`, type: 'function', function: { name: toolName(chosen), arguments: stubToolArguments(chosen) } };
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* A deterministic JSON-object stub for `response_format: { type: 'json_object' }`. DeepSeek's JSON
|
|
166
|
+
* output mode guarantees only that the content parses as JSON — there is no schema on this vendor
|
|
167
|
+
* (`json_schema` is rejected, see deepseek-twin.ts), so the twin returns a clearly-labeled
|
|
168
|
+
* deterministic object rather than fabricating schema conformance it was never given.
|
|
169
|
+
*/
|
|
170
|
+
export function stubJsonObject(messages, model) {
|
|
171
|
+
return JSON.stringify({ _twin_stub: true, model, echo: lastUserText(messages).slice(0, 200) });
|
|
172
|
+
}
|
|
173
|
+
// ── Deterministic hashing ───────────────────────────────────────────────────────────────
|
|
174
|
+
/** A small deterministic 32-bit hash (FNV-1a) of a string. */
|
|
175
|
+
export function fnv1a(text) {
|
|
176
|
+
let h = 0x811c9dc5;
|
|
177
|
+
for (let i = 0; i < text.length; i++) {
|
|
178
|
+
h ^= text.charCodeAt(i);
|
|
179
|
+
h = Math.imul(h, 0x01000193);
|
|
180
|
+
}
|
|
181
|
+
return h >>> 0;
|
|
182
|
+
}
|
|
183
|
+
/**
|
|
184
|
+
* DeepSeek's `system_fingerprint` is a backend-configuration string like
|
|
185
|
+
* `fp_eaab8d114b_prod0820_fp8_kvcache` (`@ai-sdk/deepseek` docs/30-deepseek.mdx, "Provider
|
|
186
|
+
* Metadata"). The twin serves a value in that SHAPE that is unmistakably a twin fingerprint and is
|
|
187
|
+
* derived from the request, so a replay is byte-identical.
|
|
188
|
+
*/
|
|
189
|
+
export function stubFingerprint(seed) {
|
|
190
|
+
return `fp_${fnv1a(seed).toString(16).padStart(8, '0')}_twin_stub_kvcache`;
|
|
191
|
+
}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
import { type DeepSeekScenarioEngine } from './deepseek-scenario.js';
|
|
2
|
+
import type { DeepSeekChatCompletion, DeepSeekMessageParam, SseSink } from './deepseek-types.js';
|
|
3
|
+
/** DeepSeek's beta base URL suffix. `@ai-sdk/deepseek` turns prefix-completion and strict tool
|
|
4
|
+
* calls on iff `baseURL.endsWith('/beta')` (src/deepseek-provider.ts), so the twin gates the same
|
|
5
|
+
* two features on the same path segment. */
|
|
6
|
+
export declare const DEEPSEEK_BETA_PREFIX = "/beta";
|
|
7
|
+
/** DeepSeek's Anthropic-compatible surface lives here (api-docs.deepseek.com/guides/anthropic_api).
|
|
8
|
+
* It is real vendor surface this twin does not model yet — filed as `deepseek.anthropic.messages`
|
|
9
|
+
* and answered like any other unmodeled operation, never faked. */
|
|
10
|
+
export declare const DEEPSEEK_ANTHROPIC_PREFIX = "/anthropic";
|
|
11
|
+
export type DeepSeekRequest = {
|
|
12
|
+
/** The scenario engine (kernel grammar + this pack's vocabulary) — scripts chat turns. */
|
|
13
|
+
scenarioEngine?: DeepSeekScenarioEngine;
|
|
14
|
+
method: string;
|
|
15
|
+
path: string;
|
|
16
|
+
body?: string;
|
|
17
|
+
occurredAt?: string;
|
|
18
|
+
root?: string;
|
|
19
|
+
readOnly?: boolean;
|
|
20
|
+
/** The credential the caller presents (the SDK's bearer `Authorization` header). When a request
|
|
21
|
+
* carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the real
|
|
22
|
+
* vendor rule: a credential is required → 401 on missing/invalid. In-process trusted calls
|
|
23
|
+
* (capability verify, connector) omit BOTH and are not auth-gated. */
|
|
24
|
+
apiKey?: string;
|
|
25
|
+
/** Lower-cased request headers the HTTP server passes through so the handler can model auth
|
|
26
|
+
* (401) and the deterministic rate-limit trigger (429). */
|
|
27
|
+
headers?: Record<string, string>;
|
|
28
|
+
/** When set on a streaming POST, chunks are written here (no sockets). */
|
|
29
|
+
sseSink?: SseSink;
|
|
30
|
+
};
|
|
31
|
+
/** The handler response. `headers` (when present) are response headers the HTTP server should
|
|
32
|
+
* set. */
|
|
33
|
+
export type DeepSeekResponseEnvelope = {
|
|
34
|
+
status: number;
|
|
35
|
+
body: unknown;
|
|
36
|
+
headers?: Record<string, string>;
|
|
37
|
+
};
|
|
38
|
+
type ToolChoice = 'auto' | 'none' | 'required' | {
|
|
39
|
+
name: string;
|
|
40
|
+
};
|
|
41
|
+
type ResponseFormat = 'text' | 'json_object';
|
|
42
|
+
type ReasoningEffort = 'low' | 'high' | 'max';
|
|
43
|
+
type ChatArgs = {
|
|
44
|
+
model: string;
|
|
45
|
+
messages: DeepSeekMessageParam[];
|
|
46
|
+
tools?: unknown[];
|
|
47
|
+
maxTokens?: number;
|
|
48
|
+
stop?: string[];
|
|
49
|
+
stream: boolean;
|
|
50
|
+
includeUsage: boolean;
|
|
51
|
+
toolChoice?: ToolChoice;
|
|
52
|
+
responseFormat: ResponseFormat;
|
|
53
|
+
thinkingEnabled: boolean;
|
|
54
|
+
reasoningEffort: ReasoningEffort;
|
|
55
|
+
logprobs: boolean;
|
|
56
|
+
topLogprobs?: number;
|
|
57
|
+
userId?: string;
|
|
58
|
+
/** True when the request arrived under the `/beta` base URL. */
|
|
59
|
+
beta: boolean;
|
|
60
|
+
};
|
|
61
|
+
export declare function buildChatCompletion(args: ChatArgs, occurredAt?: string, scenarioEngine?: DeepSeekScenarioEngine, root?: string): DeepSeekChatCompletion | DeepSeekResponseEnvelope;
|
|
62
|
+
/**
|
|
63
|
+
* Emit the vendor-faithful DeepSeek streaming sequence into the injected sink (NO sockets, NO
|
|
64
|
+
* setTimeout). The order: a first chunk with `delta:{role:'assistant', content:''}`, then
|
|
65
|
+
* `reasoning_content` deltas, then `content` deltas (or `tool_calls` deltas), then a chunk carrying
|
|
66
|
+
* `finish_reason`, and — when `stream_options.include_usage` was set — a FINAL chunk with an EMPTY
|
|
67
|
+
* `choices` array whose `usage` holds the completion's usage. `@ai-sdk/deepseek`'s `doStream`
|
|
68
|
+
* always sends `stream_options: { include_usage: true }`, so that tail chunk is what makes its
|
|
69
|
+
* `finish` part carry real token counts.
|
|
70
|
+
*
|
|
71
|
+
* REASONING BEFORE TEXT is not cosmetic: the SDK's transform closes its `reasoning-0` part the
|
|
72
|
+
* moment the first `content` delta arrives, so emitting them interleaved would produce a different
|
|
73
|
+
* part sequence than the vendor's.
|
|
74
|
+
*/
|
|
75
|
+
export declare function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, scenarioEngine?: DeepSeekScenarioEngine, root?: string): DeepSeekChatCompletion | DeepSeekResponseEnvelope;
|
|
76
|
+
export declare function handleDeepSeekTwinRequest(req: DeepSeekRequest): Promise<DeepSeekResponseEnvelope>;
|
|
77
|
+
export {};
|