@volter/twin-deepseek 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +198 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/deepseek-budget.d.ts +51 -0
  6. package/dist/src/deepseek-budget.js +152 -0
  7. package/dist/src/deepseek-cache.d.ts +56 -0
  8. package/dist/src/deepseek-cache.js +151 -0
  9. package/dist/src/deepseek-capabilities.d.ts +4 -0
  10. package/dist/src/deepseek-capabilities.js +1520 -0
  11. package/dist/src/deepseek-conformance.d.ts +14 -0
  12. package/dist/src/deepseek-conformance.js +473 -0
  13. package/dist/src/deepseek-connector.d.ts +168 -0
  14. package/dist/src/deepseek-connector.js +386 -0
  15. package/dist/src/deepseek-models.d.ts +30 -0
  16. package/dist/src/deepseek-models.js +38 -0
  17. package/dist/src/deepseek-scenario.d.ts +55 -0
  18. package/dist/src/deepseek-scenario.js +170 -0
  19. package/dist/src/deepseek-server.d.ts +16 -0
  20. package/dist/src/deepseek-server.js +191 -0
  21. package/dist/src/deepseek-stub.d.ts +75 -0
  22. package/dist/src/deepseek-stub.js +191 -0
  23. package/dist/src/deepseek-twin.d.ts +77 -0
  24. package/dist/src/deepseek-twin.js +1103 -0
  25. package/dist/src/deepseek-types.d.ts +172 -0
  26. package/dist/src/deepseek-types.js +26 -0
  27. package/dist/src/index.d.ts +15 -0
  28. package/dist/src/index.js +93 -0
  29. package/package.json +68 -0
  30. package/src/cli.ts +27 -0
  31. package/src/deepseek-budget.ts +178 -0
  32. package/src/deepseek-cache.ts +159 -0
  33. package/src/deepseek-capabilities.ts +1443 -0
  34. package/src/deepseek-conformance.ts +512 -0
  35. package/src/deepseek-connector.ts +440 -0
  36. package/src/deepseek-models.ts +65 -0
  37. package/src/deepseek-scenario.ts +188 -0
  38. package/src/deepseek-server.ts +201 -0
  39. package/src/deepseek-stub.ts +200 -0
  40. package/src/deepseek-twin.ts +1163 -0
  41. package/src/deepseek-types.ts +201 -0
  42. package/src/index.ts +133 -0
@@ -0,0 +1,159 @@
1
+ // DeepSeek's CONTEXT CACHE, modeled as real state rather than a made-up number.
2
+ //
3
+ // Every DeepSeek chat completion reports `prompt_cache_hit_tokens` / `prompt_cache_miss_tokens`
4
+ // (api-docs.deepseek.com/guides/kv_cache, read 2026-08-31). A twin has two ways to fill those in:
5
+ // fabricate a split (a lie the caller cannot detect), or actually keep the ledger. This file keeps
6
+ // the ledger, in the kernel's append-only action log like every other piece of twin state, so the
7
+ // number a caller reads is a fact about what this twin has actually served.
8
+ //
9
+ // THE VENDOR'S RULE, quoted: "A subsequent request can only hit the cache if it **fully matches** a
10
+ // **cache prefix unit**", and prefix units are created at request boundaries — "user input end and
11
+ // model output end". So:
12
+ // • after serving a request, the twin records TWO prefix units: the messages the caller sent
13
+ // (user-input end) and those messages plus the assistant turn it answered with (model-output
14
+ // end);
15
+ // • a later request hits on the LONGEST recorded unit that is a full prefix of its own message
16
+ // list; those tokens are `prompt_cache_hit_tokens`, the rest are misses.
17
+ //
18
+ // WHAT IS DELIBERATELY NOT MODELED (filed as todos, not faked):
19
+ // • the vendor also cuts units "at fixed token intervals" for long inputs — it publishes no
20
+ // interval, so this twin does not invent one (`deepseek.cache.interval_units`);
21
+ // • the vendor's cache is best-effort and "auto-clears within hours to days" — the twin does not
22
+ // expire entries yet (`deepseek.cache.expiry`, a TODO). It is deliberately NOT called
23
+ // impossible: the kernel's world clock is deterministic by construction and is explicitly the
24
+ // seam TTL/expiry logic reads at serve time, so this is unbuilt rather than unbuildable
25
+ // (§9 round one, SHOULD-FIX 5; the "must not have" wording that stood here was the claim that
26
+ // fix retracted).
27
+ //
28
+ // THE PREFIX KEY IS THE CONVERSATION TEXT, NOT THE RAW JSON. A caller replaying a turn rebuilds
29
+ // the assistant message from its own client state, so the JSON it sends back is never byte-equal to
30
+ // what the twin emitted (`@ai-sdk/deepseek`, for example, back-fills `reasoning_content: ''` on
31
+ // every V4 assistant turn). Keying on `role:text` per message is what makes a genuine continuation
32
+ // register as one; keying on the raw JSON would report a miss for every real conversation and the
33
+ // whole ledger would be decoration.
34
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
35
+ import { contentToText, countPromptTokens, fnv1a } from './deepseek-stub.ts';
36
+ import type { DeepSeekMessageParam } from './deepseek-types.ts';
37
+
38
+ const SERVICE = 'deepseek';
39
+
40
+ /**
41
+ * The canonical text of one message — see the file header for why this, not the JSON.
42
+ *
43
+ * §9 ROUND ONE, NIT 15: the key must cover EVERY field `messageTokens` charges for. It previously
44
+ * dropped `name` and `reasoning_content` while the token count included them, so a caller could
45
+ * attach an arbitrarily large `reasoning_content` to a replayed assistant turn, still match the
46
+ * recorded unit, and have those tokens billed as a HIT. Both are now normalized into the key
47
+ * (`?? ''`), which keeps an absent field and an empty string identical — the shape
48
+ * `@ai-sdk/deepseek` produces, since it back-fills `reasoning_content: ''` on every V4 assistant
49
+ * turn — while making a DIFFERENT reasoning payload a different key, i.e. a miss. Under-reporting
50
+ * is the safe direction here; over-reporting bills the caller for cache they did not earn.
51
+ */
52
+ export function canonicalMessage(m: DeepSeekMessageParam): string {
53
+ // THE WHOLE tool call, not just name+arguments. §9 ROUND TWO, BLOCKER 1: `messageTokens` charges
54
+ // `estimateTokens(JSON.stringify(tc))` — id and type included — while this rendered only
55
+ // name(arguments). A caller could pad `tool_calls[0].id` with 4000 characters, still match the
56
+ // recorded unit, and have 999 tokens of its own padding billed as a cache HIT. That is the
57
+ // over-reporting direction this function's own header calls the unsafe one, and it was only
58
+ // runtime-reachable because the round-one fix to `cacheHitTokens` let a full replay match at all.
59
+ const calls = (m.tool_calls ?? []).map((tc) => JSON.stringify(tc)).join(',');
60
+ const name = m.name ?? '';
61
+ const reasoning = m.reasoning_content ?? '';
62
+ const toolCallId = m.tool_call_id ?? '';
63
+ return `${m.role}|${name}|${toolCallId}:${contentToText(m.content)}|r:${reasoning}${calls ? `|tools:${calls}` : ''}`;
64
+ }
65
+
66
+ /** The prefix key for the first `count` messages. */
67
+ export function prefixKey(messages: DeepSeekMessageParam[], count: number): string {
68
+ // The separator is a NUL between newlines, written as an ESCAPE so this source file stays pure
69
+ // ASCII text. A raw NUL byte here would make `file` classify the module as binary, and `grep`
70
+ // silently matches NOTHING in a file it thinks is binary — so every later verification grep
71
+ // over this tree would read as "no survivors" while checking nothing (ADDING_A_TWIN.md §0.5).
72
+ // NUL is the right SEPARATOR because no message text can contain it, so two different
73
+ // conversations can never join into the same key.
74
+ return messages.slice(0, count).map(canonicalMessage).join('\n\u0000\n');
75
+ }
76
+
77
+ /** The deterministic subject id a prefix unit is stored under. */
78
+ export function prefixId(key: string): string {
79
+ return `cachepfx_twin_${fnv1a(key).toString(36)}_${key.length}`;
80
+ }
81
+
82
+ type PrefixRow = { id: unknown; messages?: unknown; key?: unknown; _deleted?: unknown };
83
+
84
+ function prefixRows(root?: string): PrefixRow[] {
85
+ return projectResources(SERVICE, root).filter((r) => r.type === 'cache_prefix') as unknown as PrefixRow[];
86
+ }
87
+
88
+ /**
89
+ * How many of THIS request's prompt tokens hit the cache: the token count of the longest recorded
90
+ * prefix unit that fully matches the head of `messages` — INCLUDING the whole message list.
91
+ *
92
+ * §9 ROUND ONE, SHOULD-FIX 4: this used to stop one short (`count < messages.length`), on the
93
+ * stated grounds that counting the whole request "is not what the vendor's prefix unit rule says".
94
+ * The vendor's rule says the opposite. Verbatim: a request hits when it "fully matches a cache
95
+ * prefix unit", and units form at "the end position of the user input" — so a byte-identical repeat
96
+ * of a request fully matches the user-input-end unit the first call recorded. That is the canonical
97
+ * KV-cache demonstration, and the SDK's own docs example shows it (promptCacheHitTokens 1856 /
98
+ * MissTokens 5). The twin was answering 0/0 for exactly that case, and the comment justifying it
99
+ * misquoted the page it cited.
100
+ *
101
+ * TWO properties are load-bearing and were both defects in the first cut:
102
+ *
103
+ * • The count is taken over the CURRENT request's own messages, with the SAME function that
104
+ * computes its `prompt_tokens` (`countPromptTokens`). The ledger stores only HOW MANY messages
105
+ * the unit covers, never a token total of its own. Measuring the stored prefix text separately
106
+ * produced a hit figure larger than the request's whole prompt — which `buildUsage`'s clamp then
107
+ * silently hid, reporting a 100%-cache follow-up whose new turn had vanished from the
108
+ * accounting. Deriving both halves from one function makes `hit + miss === prompt_tokens` a
109
+ * structural fact rather than something a clamp rescues.
110
+ * (A second bullet used to stand here saying the full message list was deliberately excluded. That
111
+ * WAS the bug — see the paragraph above — and leaving the justification in place while the loop did
112
+ * the opposite is the same stale-comment class round one caught elsewhere. §9 round two,
113
+ * SHOULD-FIX 6.)
114
+ */
115
+ export function cacheHitTokens(messages: DeepSeekMessageParam[], root?: string): number {
116
+ if (messages.length < 1) return 0;
117
+ const byId = new Map<string, PrefixRow>();
118
+ for (const r of prefixRows(root)) if (!r._deleted) byId.set(String(r.id), r);
119
+ if (byId.size === 0) return 0;
120
+ for (let count = messages.length; count >= 1; count--) {
121
+ if (byId.has(prefixId(prefixKey(messages, count)))) return countPromptTokens(messages.slice(0, count));
122
+ }
123
+ return 0;
124
+ }
125
+
126
+ /**
127
+ * Record the cache prefix units a served completion creates: the caller's messages (user-input
128
+ * end) and those plus the assistant turn (model-output end).
129
+ *
130
+ * `occurredAt` is threaded so the write is deterministic. Re-recording an identical prefix is
131
+ * idempotent by construction — the subject id is derived from the prefix content, and the kernel
132
+ * folds a repeat into the same subject — so a replayed request neither grows the log nor changes
133
+ * what a later request reads.
134
+ */
135
+ export async function recordCachePrefixes(
136
+ messages: DeepSeekMessageParam[],
137
+ assistant: DeepSeekMessageParam | null,
138
+ occurredAt?: string,
139
+ root?: string,
140
+ ): Promise<void> {
141
+ const units: DeepSeekMessageParam[][] = [messages];
142
+ if (assistant) units.push([...messages, assistant]);
143
+ for (const unit of units) {
144
+ if (unit.length === 0) continue;
145
+ const key = prefixKey(unit, unit.length);
146
+ const id = prefixId(key);
147
+ await applyTwinWrite(SERVICE, {
148
+ operation: 'cache_prefix.observe',
149
+ subjectType: 'cache_prefix',
150
+ subjectId: id,
151
+ // `key` is stored so an operator inspecting the log can see WHAT was cached; `messages` is
152
+ // how many turns the unit covers. Deliberately NO token total: the hit count is measured off
153
+ // the requesting message list (see `cacheHitTokens`), never off the recorded one.
154
+ fields: { key: key.slice(0, 512), messages: unit.length },
155
+ ...(occurredAt ? { occurredAt } : {}),
156
+ actor: { kind: 'agent' },
157
+ }, root);
158
+ }
159
+ }