@volter/twin-moonshot 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +164 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +25 -0
  5. package/dist/src/index.d.ts +14 -0
  6. package/dist/src/index.js +86 -0
  7. package/dist/src/moonshot-budget.d.ts +57 -0
  8. package/dist/src/moonshot-budget.js +142 -0
  9. package/dist/src/moonshot-capabilities.d.ts +4 -0
  10. package/dist/src/moonshot-capabilities.js +1200 -0
  11. package/dist/src/moonshot-conformance.d.ts +14 -0
  12. package/dist/src/moonshot-conformance.js +405 -0
  13. package/dist/src/moonshot-connector.d.ts +168 -0
  14. package/dist/src/moonshot-connector.js +416 -0
  15. package/dist/src/moonshot-models.d.ts +36 -0
  16. package/dist/src/moonshot-models.js +37 -0
  17. package/dist/src/moonshot-scenario.d.ts +54 -0
  18. package/dist/src/moonshot-scenario.js +175 -0
  19. package/dist/src/moonshot-server.d.ts +13 -0
  20. package/dist/src/moonshot-server.js +202 -0
  21. package/dist/src/moonshot-stub.d.ts +70 -0
  22. package/dist/src/moonshot-stub.js +222 -0
  23. package/dist/src/moonshot-twin.d.ts +144 -0
  24. package/dist/src/moonshot-twin.js +1647 -0
  25. package/dist/src/moonshot-types.d.ts +251 -0
  26. package/dist/src/moonshot-types.js +19 -0
  27. package/package.json +53 -0
  28. package/src/cli.ts +25 -0
  29. package/src/index.ts +129 -0
  30. package/src/moonshot-budget.ts +163 -0
  31. package/src/moonshot-capabilities.ts +1220 -0
  32. package/src/moonshot-conformance.ts +416 -0
  33. package/src/moonshot-connector.ts +465 -0
  34. package/src/moonshot-models.ts +89 -0
  35. package/src/moonshot-scenario.ts +194 -0
  36. package/src/moonshot-server.ts +220 -0
  37. package/src/moonshot-stub.ts +230 -0
  38. package/src/moonshot-twin.ts +1670 -0
  39. package/src/moonshot-types.ts +225 -0
@@ -0,0 +1,222 @@
1
+ // THE GENERATIVE-STUB CORE — the deterministic labeled answer this twin gives where the vendor
2
+ // runs a model.
3
+ //
4
+ // The twin CANNOT run the model — there are no weights here. So `POST /v1/chat/completions`,
5
+ // `POST /v1/responses` and `POST /anthropic/v1/messages` return DETERMINISTIC STUB completions
6
+ // clearly labeled `[twin-stub:<model>]` — they NEVER pretend to be real model output. What IS
7
+ // faithful is the ENTIRE PROTOCOL ENVELOPE: the response shapes of all three protocols, the
8
+ // streaming chunk/event sequences, tool_calls / tool_use blocks, finish_reason / stop_reason,
9
+ // `reasoning_content` / thinking blocks, and Moonshot's cache-split `usage` objects.
10
+ //
11
+ // SERVE-PATH DETERMINISM (CLAUDE.md): every number below is a pure function of the request. The
12
+ // cache split in the usage object is DERIVED from the token counts by a fixed rule (Moonshot
13
+ // documents that a cache hit requires >256 prompt tokens), never from a wall clock — a replay is
14
+ // byte-identical.
15
+ //
16
+ // The stub IS the twin's answer: the manifest's chat/responses/messages capabilities are claimed
17
+ // on the envelope and the determinism, which is what a consumer binds to.
18
+ /** Deterministic token estimate for a string: ~1 token per 4 chars (faithful order of
19
+ * magnitude; deterministic so usage counts are assertable). Never zero for non-empty text. */
20
+ export function estimateTokens(text) {
21
+ if (!text)
22
+ return 0;
23
+ return Math.max(1, Math.ceil(text.length / 4));
24
+ }
25
+ /** Flatten a chat message's content (string OR content-part array) to its text for token
26
+ * counting / echo. Non-text parts contribute their JSON length so the count is deterministic
27
+ * and reflects payload size. */
28
+ export function contentToText(content) {
29
+ if (typeof content === 'string')
30
+ return content;
31
+ if (content === null || content === undefined)
32
+ return '';
33
+ if (!Array.isArray(content))
34
+ return '';
35
+ return content
36
+ .map((part) => {
37
+ if (part && typeof part === 'object' && part.type === 'text') {
38
+ return String(part.text ?? '');
39
+ }
40
+ return JSON.stringify(part);
41
+ })
42
+ .join('\n');
43
+ }
44
+ /** Deterministic prompt-token count for a set of chat messages. */
45
+ export function countPromptTokens(messages) {
46
+ let total = 0;
47
+ for (const m of messages) {
48
+ total += estimateTokens(contentToText(m.content));
49
+ if (m.name)
50
+ total += estimateTokens(m.name);
51
+ for (const tc of m.tool_calls ?? [])
52
+ total += estimateTokens(JSON.stringify(tc));
53
+ }
54
+ return total;
55
+ }
56
+ /** The last user turn's text — the thing the stub echoes (deterministic, clearly labeled). */
57
+ export function lastUserText(messages) {
58
+ for (let i = messages.length - 1; i >= 0; i--) {
59
+ if (messages[i].role === 'user')
60
+ return contentToText(messages[i].content);
61
+ }
62
+ return messages.length ? contentToText(messages[messages.length - 1].content) : '';
63
+ }
64
+ /**
65
+ * Build the deterministic stub ASSISTANT text. It is unmistakably a twin stub: it carries the
66
+ * `[twin-stub:<model>]` marker and echoes the prompt, so no caller can mistake it for real
67
+ * model output. Deterministic for a given prompt → assertable in tests.
68
+ */
69
+ export function stubAssistantText(messages, model) {
70
+ const prompt = lastUserText(messages).trim();
71
+ const echo = prompt.length > 200 ? `${prompt.slice(0, 200)}…` : prompt;
72
+ return `[twin-stub:${model}] This is a deterministic stub from the Moonshot twin (no model weights are run). Echoing your last message: ${echo || '(empty)'}`;
73
+ }
74
+ /**
75
+ * The `reasoning_content` string a thinking-mode model returns (kimi-k3 always; k2.6/k2.7-code
76
+ * when thinking is enabled). A labeled stub, like the content — never a real chain of thought.
77
+ */
78
+ export function stubReasoningContent(messages, model) {
79
+ return `[twin-stub:${model}] deterministic stub reasoning (no model weights are run) for: ${lastUserText(messages).trim().slice(0, 120) || '(empty)'}`;
80
+ }
81
+ /** A deterministic stub signature for a Messages `thinking` block (the vendor asks callers to
82
+ * pass it back unchanged; the twin's is a deterministic label). */
83
+ export function stubSignature(messages, model) {
84
+ return `sig_twin_${fnv1a(`${model}:${lastUserText(messages)}`).toString(36)}`;
85
+ }
86
+ // ── Usage assembly ──────────────────────────────────────────────────────────────────────
87
+ /**
88
+ * Moonshot's automatic context caching: a hit requires MORE THAN 256 prompt tokens
89
+ * (platform.kimi.ai/docs/guides/context-caching — automatic caching, read 2026-09-16). The twin
90
+ * derives the cached share deterministically from the token count: at or below the threshold the
91
+ * cache is cold (0); above it a fixed 25% of the prompt is served from cache. That fraction is a
92
+ * judgement call, not a vendor figure — what IS vendor-shaped is the FIELD (cached_tokens in
93
+ * chat usage; cached_tokens / cache_write_tokens in Responses usage) and the threshold rule.
94
+ */
95
+ export const CACHE_HIT_THRESHOLD = 256;
96
+ export function stubCachedTokens(promptTokens) {
97
+ return promptTokens > CACHE_HIT_THRESHOLD ? Math.floor(promptTokens * 0.25) : 0;
98
+ }
99
+ export function buildChatUsage(promptTokens, completionTokens, promptCacheKey) {
100
+ // The cache split is LOAD-BEARING on the key: Moonshot's context caching is keyed by
101
+ // `prompt_cache_key` (platform.kimi.ai/docs/guides/context-caching) — a request WITHOUT one has
102
+ // no cache namespace to hit, so `cached_tokens` is 0 however long the prompt is. With a key the
103
+ // threshold/fraction rule applies. (The threshold itself stays in stubCachedTokens.)
104
+ const cached = promptCacheKey !== undefined ? stubCachedTokens(promptTokens) : 0;
105
+ return {
106
+ prompt_tokens: promptTokens,
107
+ completion_tokens: completionTokens,
108
+ total_tokens: promptTokens + completionTokens,
109
+ cached_tokens: cached,
110
+ };
111
+ }
112
+ // ── Tool calls ──────────────────────────────────────────────────────────────────────────
113
+ /** Extract a tool's function name from an OpenAI-shaped tool (`{type:'function',function:{name}}`). */
114
+ function toolName(t) {
115
+ const o = t;
116
+ if (o?.function && typeof o.function.name === 'string')
117
+ return o.function.name;
118
+ if (typeof o?.name === 'string')
119
+ return o.name;
120
+ return 'unknown_function';
121
+ }
122
+ function placeholderForSchema(def) {
123
+ const d = def;
124
+ if (Array.isArray(d?.enum) && d.enum.length)
125
+ return d.enum[0];
126
+ switch (d?.type) {
127
+ case 'number':
128
+ case 'integer': return 0;
129
+ case 'boolean': return false;
130
+ case 'array': return [];
131
+ case 'object': return {};
132
+ default: return '';
133
+ }
134
+ }
135
+ /** Build a deterministic stub argument string for a tool (schema-typed placeholders). Reads the
136
+ * OpenAI shape (`function.parameters`), the bare `parameters`, AND the Anthropic Messages shape
137
+ * (`input_schema`) — the Messages surface stubs its tool_use input through the same helper. */
138
+ export function stubToolArguments(tool) {
139
+ const o = tool;
140
+ const schema = (o?.function?.parameters ?? o?.parameters ?? o?.input_schema);
141
+ const props = schema && typeof schema === 'object' ? schema.properties : undefined;
142
+ if (!props || typeof props !== 'object')
143
+ return '{}';
144
+ const out = {};
145
+ for (const [key, def] of Object.entries(props))
146
+ out[key] = placeholderForSchema(def);
147
+ return JSON.stringify(out);
148
+ }
149
+ /**
150
+ * When tools are provided, the stub deterministically "calls" the tool selected by `forcedName`
151
+ * (a named tool_choice) or the FIRST provided tool. Returns the tool_call, or null when no
152
+ * tools were provided.
153
+ */
154
+ export function stubToolCall(tools, seq, forcedName) {
155
+ if (!Array.isArray(tools) || tools.length === 0)
156
+ return null;
157
+ const chosen = forcedName ? (tools.find((t) => toolName(t) === forcedName) ?? tools[0]) : tools[0];
158
+ return { id: `call_twin_${seq}`, type: 'function', function: { name: toolName(chosen), arguments: stubToolArguments(chosen) } };
159
+ }
160
+ /** Build a deterministic JSON-object stub for `response_format` json_object / json_schema. */
161
+ export function stubJsonObject(messages, model, jsonSchema) {
162
+ const schema = jsonSchema;
163
+ const props = schema?.schema?.properties ?? schema?.properties;
164
+ if (props && typeof props === 'object') {
165
+ const out = {};
166
+ for (const [key, def] of Object.entries(props))
167
+ out[key] = placeholderForSchema(def);
168
+ return JSON.stringify(out);
169
+ }
170
+ return JSON.stringify({ _twin_stub: true, model, echo: lastUserText(messages).slice(0, 200) });
171
+ }
172
+ // ── Deterministic hashing / pseudo-content ──────────────────────────────────────────────
173
+ /** A small deterministic 32-bit hash (FNV-1a) of a string. */
174
+ export function fnv1a(text) {
175
+ let h = 0x811c9dc5;
176
+ for (let i = 0; i < text.length; i++) {
177
+ h ^= text.charCodeAt(i);
178
+ h = Math.imul(h, 0x01000193);
179
+ }
180
+ return h >>> 0;
181
+ }
182
+ /** A deterministic id suffix from a request (so ids are stable + assertable). */
183
+ export function stableSuffix(...parts) {
184
+ return fnv1a(parts.map((p) => JSON.stringify(p) ?? String(p)).join('|')).toString(36);
185
+ }
186
+ /**
187
+ * A deterministic stub web-search result set for POST /v1/tools/search{,_pro} and the Responses
188
+ * web_search tool. The twin cannot reach the web (D4 — no real network), so the results are a
189
+ * labeled, deterministic echo of the query: same query → same results.
190
+ */
191
+ export function stubSearchResults(textQuery, limit, pro, includeContent = false) {
192
+ const out = [];
193
+ for (let i = 0; i < limit; i++) {
194
+ const base = {
195
+ authority: 'twin',
196
+ date: '2026-01-01',
197
+ icon: '',
198
+ mime: 'text/html',
199
+ site_name: 'twin-stub',
200
+ snippet: `[twin-stub] deterministic search result ${i + 1}/${limit} for: ${textQuery}`,
201
+ title: `[twin-stub] result ${i + 1} for: ${textQuery}`,
202
+ url: `https://twin-stub.invalid/search?q=${encodeURIComponent(textQuery)}&i=${i}`,
203
+ };
204
+ if (pro) {
205
+ out.push({ ...base, chunks: [{ text: `[twin-stub] chunk for: ${textQuery}`, score: 0.5 }] });
206
+ }
207
+ else {
208
+ // `include_content: true` SERVES the content (the option's documented meaning): a labeled
209
+ // deterministic page stub, like tools/fetch's. Default (false) is the vendor's bare shape.
210
+ out.push({ ...base, text: includeContent ? `[twin-stub] content for: ${textQuery} (result ${i + 1}) — the twin performs no real network I/O (D4).` : '' });
211
+ }
212
+ }
213
+ return out;
214
+ }
215
+ /** A deterministic stub markdown page for POST /v1/tools/fetch. */
216
+ export function stubFetchedMarkdown(url) {
217
+ return {
218
+ url,
219
+ markdown: `[twin-stub] deterministic fetch of ${url} — the twin performs no real network I/O (D4).`,
220
+ title: `[twin-stub] ${url}`,
221
+ };
222
+ }
@@ -0,0 +1,144 @@
1
+ import { type ScenarioDecision } from '@volter/world-core';
2
+ import { type MoonshotScenarioEngine } from './moonshot-scenario.js';
3
+ import type { MessagesSseEvent, MoonshotChatCompletion, MoonshotMessageParam, MoonshotMessagesResponse, MoonshotResponsesResponse, SseSink } from './moonshot-types.js';
4
+ /** The base path Moonshot's OpenAI-compatible surface hangs off. The Anthropic-compatible
5
+ * surface hangs off MESSAGES_PREFIX; both are served by the same host. */
6
+ export declare const MOONSHOT_API_PREFIX = "/v1";
7
+ export declare const MESSAGES_PREFIX = "/anthropic/v1";
8
+ export type MoonshotRequest = {
9
+ /** The scenario engine (kernel grammar + this pack's vocabulary) — scripts the three
10
+ * inference endpoints. */
11
+ scenarioEngine?: MoonshotScenarioEngine;
12
+ method: string;
13
+ path: string;
14
+ body?: string;
15
+ occurredAt?: string;
16
+ root?: string;
17
+ readOnly?: boolean;
18
+ /** The credential the caller presents (the bearer `Authorization` header). When a request
19
+ * carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the
20
+ * real vendor rule: a credential is required → 401 on missing/invalid. In-process trusted
21
+ * calls (capability verify, connector) omit BOTH and are not auth-gated. */
22
+ apiKey?: string;
23
+ /** Lower-cased request headers the HTTP server passes through so the handler can model auth
24
+ * (401), the rate-limit trigger (429), and the signature headers the /v1/signatures/verify
25
+ * contract references. */
26
+ headers?: Record<string, string>;
27
+ /** When set on a streaming POST, chunks/events are written here (no sockets). */
28
+ sseSink?: SseSink;
29
+ /** When set on a streaming /anthropic/v1/messages POST, Anthropic-grammar events are written
30
+ * here (event name + data; no sockets). */
31
+ messagesSseSink?: MessagesSseSink;
32
+ };
33
+ /** The handler response. `headers` (when present) are response headers the HTTP server should
34
+ * set — e.g. `retry-after` + Moonshot's `x-ratelimit-*` family on a modeled 429. */
35
+ export type MoonshotResponseEnvelope = {
36
+ status: number;
37
+ body: unknown;
38
+ headers?: Record<string, string>;
39
+ };
40
+ type ToolChoice = 'auto' | 'none' | 'required' | {
41
+ name: string;
42
+ };
43
+ type ResponseFormat = {
44
+ kind: 'text';
45
+ } | {
46
+ kind: 'json_object';
47
+ } | {
48
+ kind: 'json_schema';
49
+ schema: unknown;
50
+ };
51
+ type Thinking = {
52
+ type: 'enabled' | 'disabled';
53
+ keep?: string | null;
54
+ };
55
+ type ChatArgs = {
56
+ model: string;
57
+ messages: MoonshotMessageParam[];
58
+ tools?: unknown[];
59
+ n: number;
60
+ maxTokens?: number;
61
+ stop?: string[];
62
+ stream: boolean;
63
+ streamOptions?: {
64
+ includeUsage: boolean;
65
+ };
66
+ toolChoice?: ToolChoice;
67
+ responseFormat: ResponseFormat;
68
+ logprobs: boolean;
69
+ topLogprobs?: number;
70
+ /** kimi-k3 only: 'low' | 'high' | 'max' (default 'max'). */
71
+ reasoningEffort?: string;
72
+ /** kimi-k2.6 / kimi-k2.7-code: the `thinking` object. */
73
+ thinking?: Thinking;
74
+ promptCacheKey?: string;
75
+ };
76
+ export declare function buildChatCompletion(args: ChatArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotChatCompletion | MoonshotResponseEnvelope;
77
+ /**
78
+ * Emit the vendor-faithful Moonshot streaming sequence into the injected sink (NO sockets, NO
79
+ * setTimeout). Moonshot's order: a first chunk with `delta:{role:'assistant'}`, then
80
+ * `delta:{content}` / `delta:{reasoning_content}` / tool_calls deltas, then a chunk carrying
81
+ * `finish_reason`, then — because Moonshot's ChatCompletionChunk.usage is "Object in the final
82
+ * chunk with usage, null in ordinary chunks" (the OpenAPI's own wording, the first-party
83
+ * denominator; the prose chat doc mentions the chunk under stream_options.include_usage, so the
84
+ * two sources disagree and the OpenAPI governs) — a FINAL chunk whose `choices:[]` and `usage`
85
+ * hold the whole usage object, then `data: [DONE]`. `stream_options` is ACCEPTED (validated as
86
+ * an object) but the tail is not gated on it. Deterministic + synchronous.
87
+ */
88
+ export declare function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotChatCompletion | MoonshotResponseEnvelope;
89
+ /** The deterministic signature the twin issues on inference responses and verifies here. */
90
+ export declare function messagesSignature(nonce: string, timestamp: number, model: string): string;
91
+ type MessagesArgs = {
92
+ model: string;
93
+ messages: Array<{
94
+ role: 'user' | 'assistant';
95
+ content: string | Array<Record<string, unknown>>;
96
+ }>;
97
+ maxTokens: number;
98
+ system?: string;
99
+ stopSequences?: string[];
100
+ stream: boolean;
101
+ /** Anthropic's MessagesTool shape: { name, input_schema } (nested inside type:'custom'). */
102
+ tools?: Array<Record<string, unknown>>;
103
+ /** Anthropic's ToolChoice: type 'tool' names one tool (`name` REQUIRED, validated against
104
+ * `tools` — the sibling pack's rule). */
105
+ toolChoice?: {
106
+ type: 'auto' | 'any' | 'tool' | 'none';
107
+ name?: string;
108
+ };
109
+ outputEffort?: string;
110
+ };
111
+ export declare function buildMessagesResponse(args: MessagesArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotMessagesResponse | MoonshotResponseEnvelope;
112
+ /**
113
+ * Emit the Anthropic-compatible streaming grammar into the injected sink: message_start →
114
+ * ping → content_block_start(thinking) → thinking_delta → signature_delta → content_block_stop →
115
+ * content_block_start(text) → text_delta → content_block_stop → message_delta(stop_reason) →
116
+ * message_stop. No [DONE] sentinel — the stream ends after message_stop.
117
+ *
118
+ * message_start carries the EMPTY message (content: [], stop_reason: null, output_tokens small)
119
+ * and the blocks stream in one at a time — the real grammar (and the sibling pack's frames). A
120
+ * message_start carrying the COMPLETE final message made the SDK's own accumulator double every
121
+ * block: it pushes one block per content_block_start ON TOP of what message_start already held
122
+ * (the round-three BLOCKER).
123
+ */
124
+ export declare function streamMessages(args: MessagesArgs, sink: MessagesSseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotMessagesResponse | MoonshotResponseEnvelope;
125
+ type ResponsesArgs = {
126
+ model: string;
127
+ input: string | Array<Record<string, unknown>>;
128
+ instructions?: string;
129
+ maxOutputTokens?: number;
130
+ reasoningEffort?: string;
131
+ stream: boolean;
132
+ };
133
+ export declare function buildResponsesResponse(args: ResponsesArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotResponsesResponse | MoonshotResponseEnvelope;
134
+ /**
135
+ * Emit the Responses SSE grammar into the injected sink: response.created →
136
+ * response.in_progress → output_item.added/done per item → response.completed. Each frame
137
+ * carries `event: <type>` and a monotonically increasing `sequence_number` from 0.
138
+ */
139
+ export declare function streamResponses(args: ResponsesArgs, sink: MessagesSseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotResponsesResponse | MoonshotResponseEnvelope;
140
+ export declare function handleMoonshotTwinRequest(req: MoonshotRequest): Promise<MoonshotResponseEnvelope>;
141
+ /** A sink the Messages/Responses streaming paths write events into. `event` is the SSE event
142
+ * name (Anthropic/Responses grammar); `data` the JSON payload; `done: true` ends the stream. */
143
+ export type MessagesSseSink = (event: MessagesSseEvent) => void;
144
+ export {};