@volter/twin-deepseek 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +198 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/deepseek-budget.d.ts +51 -0
- package/dist/src/deepseek-budget.js +152 -0
- package/dist/src/deepseek-cache.d.ts +56 -0
- package/dist/src/deepseek-cache.js +151 -0
- package/dist/src/deepseek-capabilities.d.ts +4 -0
- package/dist/src/deepseek-capabilities.js +1520 -0
- package/dist/src/deepseek-conformance.d.ts +14 -0
- package/dist/src/deepseek-conformance.js +473 -0
- package/dist/src/deepseek-connector.d.ts +168 -0
- package/dist/src/deepseek-connector.js +386 -0
- package/dist/src/deepseek-models.d.ts +30 -0
- package/dist/src/deepseek-models.js +38 -0
- package/dist/src/deepseek-scenario.d.ts +55 -0
- package/dist/src/deepseek-scenario.js +170 -0
- package/dist/src/deepseek-server.d.ts +16 -0
- package/dist/src/deepseek-server.js +191 -0
- package/dist/src/deepseek-stub.d.ts +75 -0
- package/dist/src/deepseek-stub.js +191 -0
- package/dist/src/deepseek-twin.d.ts +77 -0
- package/dist/src/deepseek-twin.js +1103 -0
- package/dist/src/deepseek-types.d.ts +172 -0
- package/dist/src/deepseek-types.js +26 -0
- package/dist/src/index.d.ts +15 -0
- package/dist/src/index.js +93 -0
- package/package.json +68 -0
- package/src/cli.ts +27 -0
- package/src/deepseek-budget.ts +178 -0
- package/src/deepseek-cache.ts +159 -0
- package/src/deepseek-capabilities.ts +1443 -0
- package/src/deepseek-conformance.ts +512 -0
- package/src/deepseek-connector.ts +440 -0
- package/src/deepseek-models.ts +65 -0
- package/src/deepseek-scenario.ts +188 -0
- package/src/deepseek-server.ts +201 -0
- package/src/deepseek-stub.ts +200 -0
- package/src/deepseek-twin.ts +1163 -0
- package/src/deepseek-types.ts +201 -0
- package/src/index.ts +133 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import type { DeepSeekMessageParam } from './deepseek-types.js';
|
|
2
|
+
/**
|
|
3
|
+
* The canonical text of one message — see the file header for why this, not the JSON.
|
|
4
|
+
*
|
|
5
|
+
* §9 ROUND ONE, NIT 15: the key must cover EVERY field `messageTokens` charges for. It previously
|
|
6
|
+
* dropped `name` and `reasoning_content` while the token count included them, so a caller could
|
|
7
|
+
* attach an arbitrarily large `reasoning_content` to a replayed assistant turn, still match the
|
|
8
|
+
* recorded unit, and have those tokens billed as a HIT. Both are now normalized into the key
|
|
9
|
+
* (`?? ''`), which keeps an absent field and an empty string identical — the shape
|
|
10
|
+
* `@ai-sdk/deepseek` produces, since it back-fills `reasoning_content: ''` on every V4 assistant
|
|
11
|
+
* turn — while making a DIFFERENT reasoning payload a different key, i.e. a miss. Under-reporting
|
|
12
|
+
* is the safe direction here; over-reporting bills the caller for cache they did not earn.
|
|
13
|
+
*/
|
|
14
|
+
export declare function canonicalMessage(m: DeepSeekMessageParam): string;
|
|
15
|
+
/** The prefix key for the first `count` messages. */
|
|
16
|
+
export declare function prefixKey(messages: DeepSeekMessageParam[], count: number): string;
|
|
17
|
+
/** The deterministic subject id a prefix unit is stored under. */
|
|
18
|
+
export declare function prefixId(key: string): string;
|
|
19
|
+
/**
|
|
20
|
+
* How many of THIS request's prompt tokens hit the cache: the token count of the longest recorded
|
|
21
|
+
* prefix unit that fully matches the head of `messages` — INCLUDING the whole message list.
|
|
22
|
+
*
|
|
23
|
+
* §9 ROUND ONE, SHOULD-FIX 4: this used to stop one short (`count < messages.length`), on the
|
|
24
|
+
* stated grounds that counting the whole request "is not what the vendor's prefix unit rule says".
|
|
25
|
+
* The vendor's rule says the opposite. Verbatim: a request hits when it "fully matches a cache
|
|
26
|
+
* prefix unit", and units form at "the end position of the user input" — so a byte-identical repeat
|
|
27
|
+
* of a request fully matches the user-input-end unit the first call recorded. That is the canonical
|
|
28
|
+
* KV-cache demonstration, and the SDK's own docs example shows it (promptCacheHitTokens 1856 /
|
|
29
|
+
* MissTokens 5). The twin was answering 0/0 for exactly that case, and the comment justifying it
|
|
30
|
+
* misquoted the page it cited.
|
|
31
|
+
*
|
|
32
|
+
* TWO properties are load-bearing and were both defects in the first cut:
|
|
33
|
+
*
|
|
34
|
+
* • The count is taken over the CURRENT request's own messages, with the SAME function that
|
|
35
|
+
* computes its `prompt_tokens` (`countPromptTokens`). The ledger stores only HOW MANY messages
|
|
36
|
+
* the unit covers, never a token total of its own. Measuring the stored prefix text separately
|
|
37
|
+
* produced a hit figure larger than the request's whole prompt — which `buildUsage`'s clamp then
|
|
38
|
+
* silently hid, reporting a 100%-cache follow-up whose new turn had vanished from the
|
|
39
|
+
* accounting. Deriving both halves from one function makes `hit + miss === prompt_tokens` a
|
|
40
|
+
* structural fact rather than something a clamp rescues.
|
|
41
|
+
* (A second bullet used to stand here saying the full message list was deliberately excluded. That
|
|
42
|
+
* WAS the bug — see the paragraph above — and leaving the justification in place while the loop did
|
|
43
|
+
* the opposite is the same stale-comment class round one caught elsewhere. §9 round two,
|
|
44
|
+
* SHOULD-FIX 6.)
|
|
45
|
+
*/
|
|
46
|
+
export declare function cacheHitTokens(messages: DeepSeekMessageParam[], root?: string): number;
|
|
47
|
+
/**
|
|
48
|
+
* Record the cache prefix units a served completion creates: the caller's messages (user-input
|
|
49
|
+
* end) and those plus the assistant turn (model-output end).
|
|
50
|
+
*
|
|
51
|
+
* `occurredAt` is threaded so the write is deterministic. Re-recording an identical prefix is
|
|
52
|
+
* idempotent by construction — the subject id is derived from the prefix content, and the kernel
|
|
53
|
+
* folds a repeat into the same subject — so a replayed request neither grows the log nor changes
|
|
54
|
+
* what a later request reads.
|
|
55
|
+
*/
|
|
56
|
+
export declare function recordCachePrefixes(messages: DeepSeekMessageParam[], assistant: DeepSeekMessageParam | null, occurredAt?: string, root?: string): Promise<void>;
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
// DeepSeek's CONTEXT CACHE, modeled as real state rather than a made-up number.
|
|
2
|
+
//
|
|
3
|
+
// Every DeepSeek chat completion reports `prompt_cache_hit_tokens` / `prompt_cache_miss_tokens`
|
|
4
|
+
// (api-docs.deepseek.com/guides/kv_cache, read 2026-08-31). A twin has two ways to fill those in:
|
|
5
|
+
// fabricate a split (a lie the caller cannot detect), or actually keep the ledger. This file keeps
|
|
6
|
+
// the ledger, in the kernel's append-only action log like every other piece of twin state, so the
|
|
7
|
+
// number a caller reads is a fact about what this twin has actually served.
|
|
8
|
+
//
|
|
9
|
+
// THE VENDOR'S RULE, quoted: "A subsequent request can only hit the cache if it **fully matches** a
|
|
10
|
+
// **cache prefix unit**", and prefix units are created at request boundaries — "user input end and
|
|
11
|
+
// model output end". So:
|
|
12
|
+
// • after serving a request, the twin records TWO prefix units: the messages the caller sent
|
|
13
|
+
// (user-input end) and those messages plus the assistant turn it answered with (model-output
|
|
14
|
+
// end);
|
|
15
|
+
// • a later request hits on the LONGEST recorded unit that is a full prefix of its own message
|
|
16
|
+
// list; those tokens are `prompt_cache_hit_tokens`, the rest are misses.
|
|
17
|
+
//
|
|
18
|
+
// WHAT IS DELIBERATELY NOT MODELED (filed as todos, not faked):
|
|
19
|
+
// • the vendor also cuts units "at fixed token intervals" for long inputs — it publishes no
|
|
20
|
+
// interval, so this twin does not invent one (`deepseek.cache.interval_units`);
|
|
21
|
+
// • the vendor's cache is best-effort and "auto-clears within hours to days" — the twin does not
|
|
22
|
+
// expire entries yet (`deepseek.cache.expiry`, a TODO). It is deliberately NOT called
|
|
23
|
+
// impossible: the kernel's world clock is deterministic by construction and is explicitly the
|
|
24
|
+
// seam TTL/expiry logic reads at serve time, so this is unbuilt rather than unbuildable
|
|
25
|
+
// (§9 round one, SHOULD-FIX 5; the "must not have" wording that stood here was the claim that
|
|
26
|
+
// fix retracted).
|
|
27
|
+
//
|
|
28
|
+
// THE PREFIX KEY IS THE CONVERSATION TEXT, NOT THE RAW JSON. A caller replaying a turn rebuilds
|
|
29
|
+
// the assistant message from its own client state, so the JSON it sends back is never byte-equal to
|
|
30
|
+
// what the twin emitted (`@ai-sdk/deepseek`, for example, back-fills `reasoning_content: ''` on
|
|
31
|
+
// every V4 assistant turn). Keying on `role:text` per message is what makes a genuine continuation
|
|
32
|
+
// register as one; keying on the raw JSON would report a miss for every real conversation and the
|
|
33
|
+
// whole ledger would be decoration.
|
|
34
|
+
import { applyTwinWrite, projectResources } from '@volter/world-core';
|
|
35
|
+
import { contentToText, countPromptTokens, fnv1a } from "./deepseek-stub.js";
|
|
36
|
+
const SERVICE = 'deepseek';
|
|
37
|
+
/**
|
|
38
|
+
* The canonical text of one message — see the file header for why this, not the JSON.
|
|
39
|
+
*
|
|
40
|
+
* §9 ROUND ONE, NIT 15: the key must cover EVERY field `messageTokens` charges for. It previously
|
|
41
|
+
* dropped `name` and `reasoning_content` while the token count included them, so a caller could
|
|
42
|
+
* attach an arbitrarily large `reasoning_content` to a replayed assistant turn, still match the
|
|
43
|
+
* recorded unit, and have those tokens billed as a HIT. Both are now normalized into the key
|
|
44
|
+
* (`?? ''`), which keeps an absent field and an empty string identical — the shape
|
|
45
|
+
* `@ai-sdk/deepseek` produces, since it back-fills `reasoning_content: ''` on every V4 assistant
|
|
46
|
+
* turn — while making a DIFFERENT reasoning payload a different key, i.e. a miss. Under-reporting
|
|
47
|
+
* is the safe direction here; over-reporting bills the caller for cache they did not earn.
|
|
48
|
+
*/
|
|
49
|
+
export function canonicalMessage(m) {
|
|
50
|
+
// THE WHOLE tool call, not just name+arguments. §9 ROUND TWO, BLOCKER 1: `messageTokens` charges
|
|
51
|
+
// `estimateTokens(JSON.stringify(tc))` — id and type included — while this rendered only
|
|
52
|
+
// name(arguments). A caller could pad `tool_calls[0].id` with 4000 characters, still match the
|
|
53
|
+
// recorded unit, and have 999 tokens of its own padding billed as a cache HIT. That is the
|
|
54
|
+
// over-reporting direction this function's own header calls the unsafe one, and it was only
|
|
55
|
+
// runtime-reachable because the round-one fix to `cacheHitTokens` let a full replay match at all.
|
|
56
|
+
const calls = (m.tool_calls ?? []).map((tc) => JSON.stringify(tc)).join(',');
|
|
57
|
+
const name = m.name ?? '';
|
|
58
|
+
const reasoning = m.reasoning_content ?? '';
|
|
59
|
+
const toolCallId = m.tool_call_id ?? '';
|
|
60
|
+
return `${m.role}|${name}|${toolCallId}:${contentToText(m.content)}|r:${reasoning}${calls ? `|tools:${calls}` : ''}`;
|
|
61
|
+
}
|
|
62
|
+
/** The prefix key for the first `count` messages. */
|
|
63
|
+
export function prefixKey(messages, count) {
|
|
64
|
+
// The separator is a NUL between newlines, written as an ESCAPE so this source file stays pure
|
|
65
|
+
// ASCII text. A raw NUL byte here would make `file` classify the module as binary, and `grep`
|
|
66
|
+
// silently matches NOTHING in a file it thinks is binary — so every later verification grep
|
|
67
|
+
// over this tree would read as "no survivors" while checking nothing (ADDING_A_TWIN.md §0.5).
|
|
68
|
+
// NUL is the right SEPARATOR because no message text can contain it, so two different
|
|
69
|
+
// conversations can never join into the same key.
|
|
70
|
+
return messages.slice(0, count).map(canonicalMessage).join('\n\u0000\n');
|
|
71
|
+
}
|
|
72
|
+
/** The deterministic subject id a prefix unit is stored under. */
|
|
73
|
+
export function prefixId(key) {
|
|
74
|
+
return `cachepfx_twin_${fnv1a(key).toString(36)}_${key.length}`;
|
|
75
|
+
}
|
|
76
|
+
function prefixRows(root) {
|
|
77
|
+
return projectResources(SERVICE, root).filter((r) => r.type === 'cache_prefix');
|
|
78
|
+
}
|
|
79
|
+
/**
|
|
80
|
+
* How many of THIS request's prompt tokens hit the cache: the token count of the longest recorded
|
|
81
|
+
* prefix unit that fully matches the head of `messages` — INCLUDING the whole message list.
|
|
82
|
+
*
|
|
83
|
+
* §9 ROUND ONE, SHOULD-FIX 4: this used to stop one short (`count < messages.length`), on the
|
|
84
|
+
* stated grounds that counting the whole request "is not what the vendor's prefix unit rule says".
|
|
85
|
+
* The vendor's rule says the opposite. Verbatim: a request hits when it "fully matches a cache
|
|
86
|
+
* prefix unit", and units form at "the end position of the user input" — so a byte-identical repeat
|
|
87
|
+
* of a request fully matches the user-input-end unit the first call recorded. That is the canonical
|
|
88
|
+
* KV-cache demonstration, and the SDK's own docs example shows it (promptCacheHitTokens 1856 /
|
|
89
|
+
* MissTokens 5). The twin was answering 0/0 for exactly that case, and the comment justifying it
|
|
90
|
+
* misquoted the page it cited.
|
|
91
|
+
*
|
|
92
|
+
* TWO properties are load-bearing and were both defects in the first cut:
|
|
93
|
+
*
|
|
94
|
+
* • The count is taken over the CURRENT request's own messages, with the SAME function that
|
|
95
|
+
* computes its `prompt_tokens` (`countPromptTokens`). The ledger stores only HOW MANY messages
|
|
96
|
+
* the unit covers, never a token total of its own. Measuring the stored prefix text separately
|
|
97
|
+
* produced a hit figure larger than the request's whole prompt — which `buildUsage`'s clamp then
|
|
98
|
+
* silently hid, reporting a 100%-cache follow-up whose new turn had vanished from the
|
|
99
|
+
* accounting. Deriving both halves from one function makes `hit + miss === prompt_tokens` a
|
|
100
|
+
* structural fact rather than something a clamp rescues.
|
|
101
|
+
* (A second bullet used to stand here saying the full message list was deliberately excluded. That
|
|
102
|
+
* WAS the bug — see the paragraph above — and leaving the justification in place while the loop did
|
|
103
|
+
* the opposite is the same stale-comment class round one caught elsewhere. §9 round two,
|
|
104
|
+
* SHOULD-FIX 6.)
|
|
105
|
+
*/
|
|
106
|
+
export function cacheHitTokens(messages, root) {
|
|
107
|
+
if (messages.length < 1)
|
|
108
|
+
return 0;
|
|
109
|
+
const byId = new Map();
|
|
110
|
+
for (const r of prefixRows(root))
|
|
111
|
+
if (!r._deleted)
|
|
112
|
+
byId.set(String(r.id), r);
|
|
113
|
+
if (byId.size === 0)
|
|
114
|
+
return 0;
|
|
115
|
+
for (let count = messages.length; count >= 1; count--) {
|
|
116
|
+
if (byId.has(prefixId(prefixKey(messages, count))))
|
|
117
|
+
return countPromptTokens(messages.slice(0, count));
|
|
118
|
+
}
|
|
119
|
+
return 0;
|
|
120
|
+
}
|
|
121
|
+
/**
|
|
122
|
+
* Record the cache prefix units a served completion creates: the caller's messages (user-input
|
|
123
|
+
* end) and those plus the assistant turn (model-output end).
|
|
124
|
+
*
|
|
125
|
+
* `occurredAt` is threaded so the write is deterministic. Re-recording an identical prefix is
|
|
126
|
+
* idempotent by construction — the subject id is derived from the prefix content, and the kernel
|
|
127
|
+
* folds a repeat into the same subject — so a replayed request neither grows the log nor changes
|
|
128
|
+
* what a later request reads.
|
|
129
|
+
*/
|
|
130
|
+
export async function recordCachePrefixes(messages, assistant, occurredAt, root) {
|
|
131
|
+
const units = [messages];
|
|
132
|
+
if (assistant)
|
|
133
|
+
units.push([...messages, assistant]);
|
|
134
|
+
for (const unit of units) {
|
|
135
|
+
if (unit.length === 0)
|
|
136
|
+
continue;
|
|
137
|
+
const key = prefixKey(unit, unit.length);
|
|
138
|
+
const id = prefixId(key);
|
|
139
|
+
await applyTwinWrite(SERVICE, {
|
|
140
|
+
operation: 'cache_prefix.observe',
|
|
141
|
+
subjectType: 'cache_prefix',
|
|
142
|
+
subjectId: id,
|
|
143
|
+
// `key` is stored so an operator inspecting the log can see WHAT was cached; `messages` is
|
|
144
|
+
// how many turns the unit covers. Deliberately NO token total: the hit count is measured off
|
|
145
|
+
// the requesting message list (see `cacheHitTokens`), never off the recorded one.
|
|
146
|
+
fields: { key: key.slice(0, 512), messages: unit.length },
|
|
147
|
+
...(occurredAt ? { occurredAt } : {}),
|
|
148
|
+
actor: { kind: 'agent' },
|
|
149
|
+
}, root);
|
|
150
|
+
}
|
|
151
|
+
}
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import { type CapabilityReport, type CapabilitySpec } from '@volter/world-tooling';
|
|
2
|
+
export declare const DEEPSEEK_CAPABILITIES: CapabilitySpec[];
|
|
3
|
+
export declare const DEEPSEEK_AREAS: readonly ["anthropic", "auth", "balance", "beta", "cache", "chat", "completions", "conformance", "connector", "errors", "files", "models", "rate_limits", "reasoning", "responses", "streaming", "structured_outputs", "tools", "usage", "vision"];
|
|
4
|
+
export declare function deepseekCapabilities(): Promise<CapabilityReport>;
|