@volter/twin-moonshot 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +164 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +25 -0
  5. package/dist/src/index.d.ts +14 -0
  6. package/dist/src/index.js +86 -0
  7. package/dist/src/moonshot-budget.d.ts +57 -0
  8. package/dist/src/moonshot-budget.js +142 -0
  9. package/dist/src/moonshot-capabilities.d.ts +4 -0
  10. package/dist/src/moonshot-capabilities.js +1200 -0
  11. package/dist/src/moonshot-conformance.d.ts +14 -0
  12. package/dist/src/moonshot-conformance.js +405 -0
  13. package/dist/src/moonshot-connector.d.ts +168 -0
  14. package/dist/src/moonshot-connector.js +416 -0
  15. package/dist/src/moonshot-models.d.ts +36 -0
  16. package/dist/src/moonshot-models.js +37 -0
  17. package/dist/src/moonshot-scenario.d.ts +54 -0
  18. package/dist/src/moonshot-scenario.js +175 -0
  19. package/dist/src/moonshot-server.d.ts +13 -0
  20. package/dist/src/moonshot-server.js +202 -0
  21. package/dist/src/moonshot-stub.d.ts +70 -0
  22. package/dist/src/moonshot-stub.js +222 -0
  23. package/dist/src/moonshot-twin.d.ts +144 -0
  24. package/dist/src/moonshot-twin.js +1647 -0
  25. package/dist/src/moonshot-types.d.ts +251 -0
  26. package/dist/src/moonshot-types.js +19 -0
  27. package/package.json +53 -0
  28. package/src/cli.ts +25 -0
  29. package/src/index.ts +129 -0
  30. package/src/moonshot-budget.ts +163 -0
  31. package/src/moonshot-capabilities.ts +1220 -0
  32. package/src/moonshot-conformance.ts +416 -0
  33. package/src/moonshot-connector.ts +465 -0
  34. package/src/moonshot-models.ts +89 -0
  35. package/src/moonshot-scenario.ts +194 -0
  36. package/src/moonshot-server.ts +220 -0
  37. package/src/moonshot-stub.ts +230 -0
  38. package/src/moonshot-twin.ts +1670 -0
  39. package/src/moonshot-types.ts +225 -0
@@ -0,0 +1,1670 @@
1
+ // Moonshot twin REQUEST HANDLER — the canonical Moonshot (Kimi) API surface for the twin.
2
+ // Contract: handleMoonshotTwinRequest({method, path, body}) -> {status, body}. It is the faithful
3
+ // Moonshot API the real clients (the standard `openai` SDK pointed at
4
+ // `https://api.moonshot.ai/v1`, the `anthropic` SDK pointed at `https://api.moonshot.ai/anthropic`,
5
+ // and plain HTTP callers) talk to UNMODIFIED — Moonshot ships no SDK of its own; its documented
6
+ // integration path is the standard OpenAI/Anthropic clients with a swapped base URL
7
+ // (platform.kimi.ai/docs/overview, read 2026-09-16).
8
+ //
9
+ // THE HONEST DESIGN: the twin cannot run the model, so the three inference endpoints
10
+ // (`POST /v1/chat/completions`, `POST /v1/responses`, `POST /anthropic/v1/messages`) return a
11
+ // DETERMINISTIC STUB completion (moonshot-stub.ts) clearly labeled a twin stub — it NEVER pretends
12
+ // to be real model output. But the ENTIRE PROTOCOL ENVELOPE is vendor-faithful: all three response
13
+ // shapes, all three streaming grammars (OpenAI chunks, Responses SSE events with
14
+ // `sequence_number`, Anthropic message events), tool_calls / tool_use, finish_reason /
15
+ // stop_reason, `reasoning_content` / thinking blocks, and Moonshot's cache-split usage. The
16
+ // genuinely stateful + static surface is real:
17
+ // • GET /v1/models — static catalog (moonshot-models.ts)
18
+ // • POST/GET/DELETE /v1/files (+ /content) — stateful (kernel action log)
19
+ // • POST/GET /v1/batches (+ /cancel) — stateful
20
+ // • GET /v1/users/me/balance — stateful (a mutable account balance)
21
+ // plus the deterministic stateless helpers: POST /v1/tokenizers/estimate-token-count,
22
+ // POST /v1/signatures/verify, POST /v1/tools/{search,search_pro,fetch}.
23
+ //
24
+ // TWO PATH PREFIXES, BOTH REAL: Moonshot serves its OpenAI-compatible surface under `/v1` and its
25
+ // Anthropic-compatible surface under `/anthropic/v1` on the same host (api.moonshot.ai). The
26
+ // Anthropic surface's error envelope is `{ type:'error', error:{type,message}, request_id? }` —
27
+ // a DIFFERENT envelope from `/v1`'s `{ error: { message, type, code? } }` — and its streaming
28
+ // grammar is Anthropic's event-name SSE, not OpenAI's `data:`-only frames.
29
+ //
30
+ // State lives in the kernel action log (D1): all writes are local actions, reads are the
31
+ // projection. No real Moonshot is ever called from this path (D4). Streaming uses an INJECTED
32
+ // sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
33
+ import { applyTwinWrite, projectResources, resolveSubjectId, type ScenarioDecision, worldNow } from '@volter/world-core';
34
+ import {
35
+ findModel,
36
+ K26_THINKING_TYPES,
37
+ K27_THINKING_TYPES,
38
+ MOONSHOT_MODELS,
39
+ REASONING_EFFORTS,
40
+ type ReasoningEffort,
41
+ } from './moonshot-models.ts';
42
+ import {
43
+ buildChatUsage,
44
+ contentToText,
45
+ countPromptTokens,
46
+ estimateTokens,
47
+ fnv1a,
48
+ lastUserText,
49
+ stableSuffix,
50
+ stubAssistantText,
51
+ stubCachedTokens,
52
+ stubFetchedMarkdown,
53
+ stubJsonObject,
54
+ stubReasoningContent,
55
+ stubSearchResults,
56
+ stubSignature,
57
+ stubToolArguments,
58
+ stubToolCall,
59
+ } from './moonshot-stub.ts';
60
+ import { type MoonshotScenarioEngine, type MoonshotScenarioRespond, realizeMoonshotRespond, type ScriptedResult } from './moonshot-scenario.ts';
61
+ import type {
62
+ MessagesSseEvent,
63
+ MoonshotAssistantMessage,
64
+ MoonshotChatCompletion,
65
+ MoonshotChoice,
66
+ MoonshotMessageParam,
67
+ MoonshotMessagesResponse,
68
+ MoonshotResponsesOutputItem,
69
+ MoonshotResponsesResponse,
70
+ MoonshotToolCall,
71
+ MoonshotUsage,
72
+ SseEvent,
73
+ SseSink,
74
+ } from './moonshot-types.ts';
75
+
76
+ const SERVICE = 'moonshot';
77
+
78
+ /** The base path Moonshot's OpenAI-compatible surface hangs off. The Anthropic-compatible
79
+ * surface hangs off MESSAGES_PREFIX; both are served by the same host. */
80
+ export const MOONSHOT_API_PREFIX = '/v1';
81
+ export const MESSAGES_PREFIX = '/anthropic/v1';
82
+
83
+ export type MoonshotRequest = {
84
+ /** The scenario engine (kernel grammar + this pack's vocabulary) — scripts the three
85
+ * inference endpoints. */
86
+ scenarioEngine?: MoonshotScenarioEngine;
87
+ method: string;
88
+ path: string;
89
+ body?: string;
90
+ occurredAt?: string;
91
+ root?: string;
92
+ readOnly?: boolean;
93
+ /** The credential the caller presents (the bearer `Authorization` header). When a request
94
+ * carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the
95
+ * real vendor rule: a credential is required → 401 on missing/invalid. In-process trusted
96
+ * calls (capability verify, connector) omit BOTH and are not auth-gated. */
97
+ apiKey?: string;
98
+ /** Lower-cased request headers the HTTP server passes through so the handler can model auth
99
+ * (401), the rate-limit trigger (429), and the signature headers the /v1/signatures/verify
100
+ * contract references. */
101
+ headers?: Record<string, string>;
102
+ /** When set on a streaming POST, chunks/events are written here (no sockets). */
103
+ sseSink?: SseSink;
104
+ /** When set on a streaming /anthropic/v1/messages POST, Anthropic-grammar events are written
105
+ * here (event name + data; no sockets). */
106
+ messagesSseSink?: MessagesSseSink;
107
+ };
108
+
109
+ /** The handler response. `headers` (when present) are response headers the HTTP server should
110
+ * set — e.g. `retry-after` + Moonshot's `x-ratelimit-*` family on a modeled 429. */
111
+ export type MoonshotResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
112
+
113
+ // ── vendor-shaped errors ──────────────────────────────────────────────────────────────
114
+ /**
115
+ * Moonshot's OpenAI-surface error envelope (ErrorResponse schema): `error.message` REQUIRED,
116
+ * `type` and `code` optional. The `type` strings below are Moonshot's OWN documented error-code
117
+ * page (platform.kimi.ai/docs/api/errors, read 2026-09-16) — a CLOSED published set:
118
+ * 400 invalid_request_error / content_filter
119
+ * 401 invalid_authentication_error / incorrect_api_key_error
120
+ * 403 permission_denied_error
121
+ * 404 resource_not_found_error
122
+ * 429 engine_overloaded_error / exceeded_current_quota_error / rate_limit_reached_error
123
+ * 499 client_closed_request
124
+ * 500 server_error / unexpected_output
125
+ * 503 server_unavailable
126
+ * 504 timeout_error? — the page names 504 but the twin models no timeout path; see the 429/503
127
+ * helpers below for the ones it does.
128
+ */
129
+ function errBody(type: string, message: string, code?: string) {
130
+ return { error: { message, ...(type !== undefined ? { type } : {}), ...(code !== undefined ? { code } : {}) } };
131
+ }
132
+ function invalidRequest(message: string, code?: string): MoonshotResponseEnvelope {
133
+ return { status: 400, body: errBody('invalid_request_error', message, code) };
134
+ }
135
+ function notFound(message: string): MoonshotResponseEnvelope {
136
+ return { status: 404, body: errBody('resource_not_found_error', message) };
137
+ }
138
+ function authError(message: string, type: string): MoonshotResponseEnvelope {
139
+ return { status: 401, body: errBody(type, message) };
140
+ }
141
+
142
+ // ── modeled authentication (401) ────────────────────────────────────────────────────────
143
+ // Real Moonshot requires a credential on every request and returns 401 when it is missing or
144
+ // invalid (platform.kimi.ai/docs/api/errors: 401 = invalid_authentication_error when the key
145
+ // is absent/malformed, incorrect_api_key_error when the key is wrong). The twin can't validate
146
+ // against real keys, so it models the CHECKABLE failures: a missing credential →
147
+ // invalid_authentication_error, and a reserved sentinel ('sk_invalid'/'invalid') for the
148
+ // wrong-key path → incorrect_api_key_error. Any other non-empty key is accepted. Trusted
149
+ // in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated; both real
150
+ // clients always send a key → they pass.
151
+ //
152
+ // THE KEY FOLLOWS THE PREFIX: the Anthropic-compatible surface's documented client is the
153
+ // unmodified `@anthropic-ai/sdk`, which authenticates with `x-api-key` (never a bearer) — a
154
+ // bearer-only check made the whole /anthropic surface 401-dead for it. On MESSAGES_PREFIX the
155
+ // key candidates and the error envelope are Anthropic's own grammar (x-api-key first,
156
+ // `authentication_error` in the {type:'error',error:{…}} envelope), exactly as the anthropic
157
+ // pack's checkAuth reads them; on /v1 the bearer leads and Moonshot's own error types apply.
158
+ function checkAuth(req: MoonshotRequest): MoonshotResponseEnvelope | null {
159
+ const onMessages = onMessagesPath(req.path);
160
+ const auth = req.headers?.['authorization'];
161
+ const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
162
+ const xApiKey = typeof req.headers?.['x-api-key'] === 'string' ? req.headers['x-api-key'].trim() : '';
163
+ const key = (req.apiKey ?? '').trim() || (onMessages ? xApiKey || bearer : bearer || xApiKey);
164
+ if (!key) {
165
+ return onMessages
166
+ ? messagesError(401, 'authentication_error', 'missing API key. Provide an x-api-key header (or Authorization: Bearer …).')
167
+ : authError('The API key is missing or malformed. Please check your API key.', 'invalid_authentication_error');
168
+ }
169
+ if (key === 'sk_invalid' || key === 'invalid') {
170
+ return onMessages
171
+ ? messagesError(401, 'authentication_error', 'invalid x-api-key.')
172
+ : authError('The API key is invalid. Please check your API key.', 'incorrect_api_key_error');
173
+ }
174
+ return null;
175
+ }
176
+
177
+ /** True when the request targets the Anthropic-compatible surface — the prefix that owns its
178
+ * OWN error envelope, auth header and error-type vocabulary (see the two-prefixes note). */
179
+ function onMessagesPath(path: string): boolean {
180
+ const bare = path.split('?')[0] ?? '';
181
+ return bare === MESSAGES_PREFIX || bare.startsWith(`${MESSAGES_PREFIX}/`);
182
+ }
183
+
184
+ // ── modeled rate limiting (429) ────────────────────────────────────────────────────────
185
+ // Non-deterministic in production, so the twin exposes a DETERMINISTIC opt-in trigger:
186
+ // `x-twin-force-rate-limit: 1` returns the faithful 429 envelope plus Moonshot's own documented
187
+ // header family (platform.kimi.ai/docs/pricing/limits: a 429 carries X-RateLimit-Limit /
188
+ // X-RateLimit-Remaining / X-RateLimit-Reset) and `retry-after`. The Tier-0 figures are the
189
+ // LOWEST published row (RPM 3, TPM 500,000) — the same grounding the budget declaration uses.
190
+ function rateLimitError(): MoonshotResponseEnvelope {
191
+ return {
192
+ status: 429,
193
+ body: errBody('rate_limit_reached_error', 'Request rate limit reached: 3 requests per minute (Tier 0). Please retry after 20 seconds.'),
194
+ headers: {
195
+ 'retry-after': '20',
196
+ 'x-ratelimit-limit': '3',
197
+ 'x-ratelimit-remaining': '0',
198
+ 'x-ratelimit-reset': '20s',
199
+ },
200
+ };
201
+ }
202
+ /** Moonshot documents 503 as `server_unavailable` (platform.kimi.ai/docs/api/errors). The twin
203
+ * exposes it as a deterministic trigger so a caller can script the vendor's outage shape. */
204
+ function serverUnavailable(): MoonshotResponseEnvelope {
205
+ return { status: 503, body: errBody('server_unavailable', 'The server is overloaded or not ready to handle the request. Please try again later.') };
206
+ }
207
+ function triggered(req: MoonshotRequest, header: string): boolean {
208
+ const v = req.headers?.[header];
209
+ return v === '1' || v === 'true';
210
+ }
211
+
212
+ function nowEpoch(occurredAt?: string): number {
213
+ return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
214
+ }
215
+
216
+ // ── kernel helpers ──────────────────────────────────────────────────────────────────────
217
+ function rows(type: string, root?: string): Array<Record<string, unknown>> {
218
+ return projectResources(SERVICE, root).filter((r) => r.type === type);
219
+ }
220
+ /**
221
+ * Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
222
+ * projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
223
+ * gap above the count (ADDING_A_TWIN.md §5). Two further properties matter:
224
+ * • the `_twin_` infix namespaces LOCAL mints, so a pulled Moonshot id can never be matched by
225
+ * this regex and therefore can never be re-minted;
226
+ * • the scan includes TOMBSTONED rows (a soft-deleted file keeps its projection row), so the
227
+ * counter RATCHETS across delete→recreate and a deleted id is never handed out twice.
228
+ */
229
+ function nextId(type: string, prefix: string, root?: string): string {
230
+ let max = 0;
231
+ for (const r of rows(type, root)) {
232
+ const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(String(r.id));
233
+ if (m) max = Math.max(max, Number(m[1]));
234
+ }
235
+ return `${prefix}_twin_${max + 1}`;
236
+ }
237
+ /** Models observed by a connector pull (mapModel), reshaped into the served model object. */
238
+ function pulledModels(root?: string): Array<Record<string, unknown>> {
239
+ return rows('model', root)
240
+ .filter((r) => !r._deleted)
241
+ .map((r) => ({ id: r.id, object: 'model', created: r.created, owned_by: r.owned_by }));
242
+ }
243
+ /** The catalog a request sees: the static table, with any PULLED row of the same id OVERRIDING it. */
244
+ function servedModels(root?: string): Array<Record<string, unknown>> {
245
+ const pulled = pulledModels(root);
246
+ const byId = new Map<string, Record<string, unknown>>();
247
+ for (const m of MOONSHOT_MODELS) byId.set(m.id, { id: m.id, object: 'model', created: m.created, owned_by: m.owned_by });
248
+ for (const m of pulled) byId.set(String(m.id), m);
249
+ return [...byId.values()];
250
+ }
251
+ function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined {
252
+ return rows(type, root).find((r) => r.id === id);
253
+ }
254
+ /** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
255
+ function strip(r: Record<string, unknown>): Record<string, unknown> {
256
+ const out: Record<string, unknown> = {};
257
+ for (const [k, v] of Object.entries(r)) {
258
+ if (k === 'type' || k === 'updatedAt' || k.startsWith('_')) continue;
259
+ out[k] = v;
260
+ }
261
+ return out;
262
+ }
263
+
264
+ // ── request parsing ─────────────────────────────────────────────────────────────────────
265
+ function parseJson(body?: string): Record<string, unknown> {
266
+ if (!body || !body.trim()) return {};
267
+ try {
268
+ const v = JSON.parse(body);
269
+ return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
270
+ } catch {
271
+ return {};
272
+ }
273
+ }
274
+
275
+ // ── chat completions: validate the request the way Moonshot does ────────────────────────
276
+ type ToolChoice = 'auto' | 'none' | 'required' | { name: string };
277
+ type ResponseFormat = { kind: 'text' } | { kind: 'json_object' } | { kind: 'json_schema'; schema: unknown };
278
+ type Thinking = { type: 'enabled' | 'disabled'; keep?: string | null };
279
+
280
+ /** Moonshot's documented tool-name regex (ToolDefinition / MessagesTool schemas). */
281
+ const TOOL_NAME_RE = /^[a-zA-Z_][a-zA-Z0-9-_]{0,127}$/;
282
+ /** `stop`: "A maximum of 5 strings is allowed, and each string must not exceed 32 bytes"
283
+ * (ChatRequestBase.stop description). */
284
+ const STOP_MAX_ITEMS = 5;
285
+ const STOP_MAX_BYTES = 32;
286
+
287
+ type ChatArgs = {
288
+ model: string;
289
+ messages: MoonshotMessageParam[];
290
+ tools?: unknown[];
291
+ n: number;
292
+ maxTokens?: number;
293
+ stop?: string[];
294
+ stream: boolean;
295
+ streamOptions?: { includeUsage: boolean };
296
+ toolChoice?: ToolChoice;
297
+ responseFormat: ResponseFormat;
298
+ logprobs: boolean;
299
+ topLogprobs?: number;
300
+ /** kimi-k3 only: 'low' | 'high' | 'max' (default 'max'). */
301
+ reasoningEffort?: string;
302
+ /** kimi-k2.6 / kimi-k2.7-code: the `thinking` object. */
303
+ thinking?: Thinking;
304
+ promptCacheKey?: string;
305
+ };
306
+
307
+ /** Validate `thinking` per model. Returns the parsed value or an error envelope. */
308
+ function validateThinking(model: string, raw: unknown): { value?: Thinking; error?: MoonshotResponseEnvelope } {
309
+ if (raw === undefined || raw === null) return {};
310
+ if (typeof raw !== 'object' || Array.isArray(raw)) return { error: invalidRequest("'thinking' must be an object") };
311
+ const o = raw as Record<string, unknown>;
312
+ const type = o.type;
313
+ if (typeof type !== 'string') return { error: invalidRequest("'thinking.type' is required") };
314
+ if (model === 'kimi-k2.6') {
315
+ if (!K26_THINKING_TYPES.includes(type as 'enabled' | 'disabled')) {
316
+ return { error: invalidRequest(`'thinking.type' must be one of ${K26_THINKING_TYPES.map((t) => `'${t}'`).join(', ')} for kimi-k2.6`) };
317
+ }
318
+ } else if (model === 'kimi-k2.7-code' || model === 'kimi-k2.7-code-highspeed') {
319
+ // Moonshot's OpenAPI: "For kimi-k2.7-code, only `\"enabled\"` is accepted; passing
320
+ // `\"disabled\"` returns an error. This differs from kimi-k2.6."
321
+ if (!K27_THINKING_TYPES.includes(type as 'enabled')) {
322
+ return { error: invalidRequest(`'thinking.type' must be 'enabled' for ${model} — 'disabled' is not supported and returns an error`) };
323
+ }
324
+ } else {
325
+ return { error: invalidRequest(`'thinking' is not a parameter of ${model}`) };
326
+ }
327
+ let keep: string | null | undefined;
328
+ if (o.keep !== undefined) {
329
+ if (o.keep !== null && o.keep !== 'all') {
330
+ // kimi-k2.7-code: only "all" (or null/omitted) is valid; any other value errors.
331
+ // kimi-k2.6: same closed set {all, null}.
332
+ return { error: invalidRequest(`'thinking.keep' must be 'all' or null`) };
333
+ }
334
+ keep = o.keep as string | null;
335
+ }
336
+ return { value: { type: type as 'enabled' | 'disabled', ...(keep !== undefined ? { keep } : {}) } };
337
+ }
338
+
339
+ function validateChat(params: Record<string, unknown>): { args: ChatArgs } | { error: MoonshotResponseEnvelope } {
340
+ if (params.model === undefined || params.model === '') return { error: invalidRequest("'model' is a required property") };
341
+ if (typeof params.model !== 'string') return { error: invalidRequest("'model' must be a string") };
342
+ const model = findModel(params.model);
343
+ if (!model) return { error: invalidRequest(`The model '${params.model}' does not exist or you do not have access to it.`) };
344
+ if (!Array.isArray(params.messages)) return { error: invalidRequest("'messages' is a required property") };
345
+ if (params.messages.length === 0) return { error: invalidRequest("[] is too short - 'messages'") };
346
+ const messages = params.messages as MoonshotMessageParam[];
347
+ for (const m of messages) {
348
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
349
+ return { error: invalidRequest("each message must have a valid 'role'") };
350
+ }
351
+ if (!['system', 'user', 'assistant', 'tool'].includes(m.role)) {
352
+ return { error: invalidRequest(`'${m.role}' is not one of ['system', 'user', 'assistant', 'tool']`) };
353
+ }
354
+ }
355
+ // kimi-k3's dynamic tool loading message: role 'system', `tools` present, NO content.
356
+ // Any OTHER message with `tools` and no content is malformed.
357
+ for (const m of messages) {
358
+ const hasTools = Array.isArray((m as { tools?: unknown }).tools);
359
+ if (m.role === 'system' && hasTools && m.content === undefined) continue; // the dynamic-tool shape
360
+ if (hasTools && m.content === undefined) {
361
+ return { error: invalidRequest("a dynamic tool message must use the 'system' role") };
362
+ }
363
+ }
364
+ // Vision: image_url parts are only valid on a vision model (kimi-k3, kimi-k2.6).
365
+ if (!model.supports.vision) {
366
+ for (const m of messages) {
367
+ const parts = m.content;
368
+ if (Array.isArray(parts) && parts.some((p) => (p as { type?: string })?.type === 'image_url')) {
369
+ return { error: invalidRequest(`The model '${params.model}' does not support image input`) };
370
+ }
371
+ }
372
+ }
373
+ // logprobs: Moonshot's OpenAPI ACCEPTS logprobs (boolean) + top_logprobs (0..20) — unlike
374
+ // Groq, this is real surface. The twin models acceptance (a logprobs echo on the choice) only
375
+ // as far as the shape goes: `top_logprobs` without `logprobs:true` contradicts the documented
376
+ // coupling ("logprobs must be set to true when this parameter is used").
377
+ const logprobs = params.logprobs === true;
378
+ let topLogprobs: number | undefined;
379
+ if (params.top_logprobs !== undefined && params.top_logprobs !== null) {
380
+ const n = Number(params.top_logprobs);
381
+ if (!Number.isInteger(n) || n < 0 || n > 20) return { error: invalidRequest("'top_logprobs' must be an integer between 0 and 20") };
382
+ if (!logprobs) return { error: invalidRequest("'logprobs' must be set to true when 'top_logprobs' is used") };
383
+ topLogprobs = n;
384
+ }
385
+ const maxRaw = params.max_completion_tokens ?? params.max_tokens;
386
+ let maxTokens: number | undefined;
387
+ if (maxRaw !== undefined) {
388
+ maxTokens = Number(maxRaw);
389
+ if (!Number.isInteger(maxTokens) || maxTokens < 1) return { error: invalidRequest("'max_completion_tokens' must be an integer >= 1") };
390
+ // "If input plus max_completion_tokens exceeds the model context window, the API returns
391
+ // invalid_request_error" (ChatRequestCommon).
392
+ const promptTokens = countPromptTokens(messages);
393
+ if (promptTokens + maxTokens > model.context_length) {
394
+ return { error: invalidRequest(`'max_completion_tokens' plus input tokens (${promptTokens + maxTokens}) exceeds the model context window (${model.context_length})`) };
395
+ }
396
+ }
397
+ let stop: string[] | undefined;
398
+ if (params.stop !== undefined && params.stop !== null) {
399
+ if (typeof params.stop === 'string') stop = [params.stop];
400
+ else if (Array.isArray(params.stop)) stop = params.stop as string[];
401
+ else return { error: invalidRequest("'stop' must be a string or an array of strings") };
402
+ if (stop.length > STOP_MAX_ITEMS) return { error: invalidRequest(`'stop' must not exceed ${STOP_MAX_ITEMS} strings`) };
403
+ for (const s of stop) {
404
+ if (typeof s !== 'string') return { error: invalidRequest("'stop' must be a string or an array of strings") };
405
+ if (Buffer.byteLength(s, 'utf8') > STOP_MAX_BYTES) return { error: invalidRequest(`'stop' entries must not exceed ${STOP_MAX_BYTES} bytes`) };
406
+ }
407
+ }
408
+ let toolChoice: ToolChoice | undefined;
409
+ const tcRaw = params.tool_choice;
410
+ if (tcRaw !== undefined && tcRaw !== null) {
411
+ if (typeof tcRaw === 'string') {
412
+ if (!['auto', 'none', 'required'].includes(tcRaw)) return { error: invalidRequest("'tool_choice' must be one of 'none', 'auto', 'required' or a named function") };
413
+ toolChoice = tcRaw as ToolChoice;
414
+ } else if (typeof tcRaw === 'object') {
415
+ const name = (tcRaw as { function?: { name?: unknown } }).function?.name;
416
+ if (typeof name !== 'string' || !name) return { error: invalidRequest("'tool_choice.function.name' is required for a named tool choice") };
417
+ toolChoice = { name };
418
+ }
419
+ }
420
+ let responseFormat: ResponseFormat = { kind: 'text' };
421
+ const rf = params.response_format as { type?: unknown; json_schema?: unknown } | undefined;
422
+ if (rf && typeof rf === 'object') {
423
+ if (rf.type === 'json_object') responseFormat = { kind: 'json_object' };
424
+ else if (rf.type === 'json_schema') {
425
+ // json_schema REQUIRES the json_schema object (name + schema required by the OpenAPI).
426
+ const js = rf.json_schema as { name?: unknown; schema?: unknown } | undefined;
427
+ if (!js || typeof js !== 'object') return { error: invalidRequest("'response_format.json_schema' is required when 'response_format.type' is 'json_schema'") };
428
+ if (typeof js.name !== 'string' || !js.name) return { error: invalidRequest("'response_format.json_schema.name' is required") };
429
+ if (!js.schema || typeof js.schema !== 'object') return { error: invalidRequest("'response_format.json_schema.schema' is required") };
430
+ responseFormat = { kind: 'json_schema', schema: js.schema };
431
+ }
432
+ else if (rf.type !== undefined && rf.type !== 'text') return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
433
+ }
434
+ // Tools: validate the documented name regex on every provided function tool.
435
+ if (params.tools !== undefined) {
436
+ if (!Array.isArray(params.tools)) return { error: invalidRequest("'tools' must be an array") };
437
+ for (const t of params.tools) {
438
+ const name = (t as { function?: { name?: unknown } })?.function?.name;
439
+ if (typeof name !== 'string' || !TOOL_NAME_RE.test(name)) {
440
+ return { error: invalidRequest(`'tools[].function.name' must match ${TOOL_NAME_RE.source}`) };
441
+ }
442
+ }
443
+ }
444
+ // kimi-k3's reasoning_effort: a CLOSED set, default 'max'.
445
+ let reasoningEffort: string | undefined;
446
+ if (params.reasoning_effort !== undefined && params.reasoning_effort !== null) {
447
+ if (params.model !== 'kimi-k3') return { error: invalidRequest(`'reasoning_effort' is not a parameter of ${params.model}`) };
448
+ if (typeof params.reasoning_effort !== 'string' || !REASONING_EFFORTS.includes(params.reasoning_effort as ReasoningEffort)) {
449
+ return { error: invalidRequest(`'reasoning_effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) };
450
+ }
451
+ reasoningEffort = params.reasoning_effort;
452
+ }
453
+ const thinking = validateThinking(params.model, params.thinking);
454
+ if (thinking.error) return { error: thinking.error };
455
+ const streamOptionsRaw = params.stream_options as { include_usage?: unknown } | undefined;
456
+ if (streamOptionsRaw !== undefined && (typeof streamOptionsRaw !== 'object' || streamOptionsRaw === null)) {
457
+ return { error: invalidRequest("'stream_options' must be an object") };
458
+ }
459
+ return {
460
+ args: {
461
+ model: params.model,
462
+ messages,
463
+ ...(params.tools !== undefined ? { tools: params.tools as unknown[] } : {}),
464
+ n: 1,
465
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
466
+ ...(stop !== undefined ? { stop } : {}),
467
+ stream: params.stream === true,
468
+ ...(streamOptionsRaw?.include_usage === true ? { streamOptions: { includeUsage: true } } : {}),
469
+ ...(toolChoice !== undefined ? { toolChoice } : {}),
470
+ responseFormat,
471
+ logprobs,
472
+ ...(topLogprobs !== undefined ? { topLogprobs } : {}),
473
+ ...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
474
+ ...(thinking.value !== undefined ? { thinking: thinking.value } : {}),
475
+ ...(typeof params.prompt_cache_key === 'string' ? { promptCacheKey: params.prompt_cache_key } : {}),
476
+ },
477
+ };
478
+ }
479
+
480
+ /** Build ONE deterministic stub choice. Thinking mode follows the model/params: kimi-k3 always
481
+ * thinks; k2.6 thinks unless `thinking.type:'disabled'`; k2.7-code always thinks. */
482
+ function buildChoice(args: ChatArgs, idx: number): { choice: MoonshotChoice; completionTokens: number } {
483
+ const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
484
+ const forbidTools = args.toolChoice === 'none';
485
+ const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
486
+ if (hasTools && !forbidTools) {
487
+ const list = args.tools as unknown[];
488
+ const calls: MoonshotToolCall[] = [];
489
+ if (forcedName) {
490
+ const tc = stubToolCall(args.tools, idx + 1, forcedName);
491
+ if (tc) calls.push(tc);
492
+ } else {
493
+ for (let t = 0; t < list.length; t++) {
494
+ const tc = stubToolCall([list[t]], idx * 100 + t + 1);
495
+ if (tc) calls.push(tc);
496
+ }
497
+ }
498
+ if (calls.length) {
499
+ return {
500
+ choice: { index: idx, message: { role: 'assistant', content: null, tool_calls: calls }, finish_reason: 'tool_calls' },
501
+ completionTokens: estimateTokens(JSON.stringify(calls)),
502
+ };
503
+ }
504
+ }
505
+ let text = args.responseFormat.kind === 'json_object'
506
+ ? stubJsonObject(args.messages, args.model)
507
+ : args.responseFormat.kind === 'json_schema'
508
+ ? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
509
+ : stubAssistantText(args.messages, args.model);
510
+ let finish: MoonshotChoice['finish_reason'] = 'stop';
511
+ // Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
512
+ let stopAt = -1;
513
+ for (const s of args.stop ?? []) {
514
+ if (!s) continue;
515
+ const i = text.indexOf(s);
516
+ if (i >= 0 && (stopAt < 0 || i < stopAt)) stopAt = i;
517
+ }
518
+ if (stopAt >= 0) text = text.slice(0, stopAt);
519
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
520
+ text = text.slice(0, args.maxTokens * 4);
521
+ finish = 'length';
522
+ }
523
+ const message: MoonshotAssistantMessage = { role: 'assistant', content: text };
524
+ // Thinking mode: kimi-k3 ALWAYS (Preserved Thinking); k2.7-code ALWAYS (type only accepts
525
+ // 'enabled'); k2.6 when thinking.type is not 'disabled'.
526
+ const thinkingOn = args.model === 'kimi-k3'
527
+ || args.model === 'kimi-k2.7-code' || args.model === 'kimi-k2.7-code-highspeed'
528
+ || (args.model === 'kimi-k2.6' && args.thinking?.type !== 'disabled');
529
+ if (thinkingOn) {
530
+ message.reasoning_content = stubReasoningContent(args.messages, args.model);
531
+ // The cap applies to EVERYTHING the model emits — reasoning is output tokens at the vendor
532
+ // too — so truncate it as well and let completion_tokens count only what was returned.
533
+ // finish_reason stays 'length': the cap is what truncated, whichever field hit it.
534
+ if (args.maxTokens !== undefined) {
535
+ const textTokens = estimateTokens(String(message.content ?? ''));
536
+ if (textTokens >= args.maxTokens) {
537
+ delete message.reasoning_content;
538
+ } else {
539
+ const budget = args.maxTokens - textTokens;
540
+ const reasoning = message.reasoning_content;
541
+ if (estimateTokens(reasoning) > budget) message.reasoning_content = reasoning.slice(0, budget * 4);
542
+ }
543
+ }
544
+ }
545
+ return {
546
+ choice: { index: idx, message, finish_reason: finish },
547
+ completionTokens: estimateTokens(String(message.content ?? '')) + estimateTokens(message.reasoning_content ?? ''),
548
+ };
549
+ }
550
+
551
+ export function buildChatCompletion(args: ChatArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotChatCompletion | MoonshotResponseEnvelope {
552
+ const promptTokens = countPromptTokens(args.messages);
553
+ let scripted: ScriptedResult | null = null;
554
+ let missTeach = '';
555
+ if (decision) {
556
+ // The route served the request through the engine (R15: serve() honors the handler's fault
557
+ // before any content exists); this realizer only sees the content decision.
558
+ if (decision.kind === 'handler') {
559
+ const respond = decision.respond as MoonshotScenarioRespond;
560
+ // A scripted FAILURE short-circuits into Moonshot's own error envelope + status.
561
+ if (respond.error) return scriptedError(respond.error);
562
+ scripted = realizeMoonshotRespond(respond);
563
+ } else {
564
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/moonshot.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
565
+ }
566
+ }
567
+ const choices: MoonshotChoice[] = [];
568
+ let completionTokens = 0;
569
+ for (let i = 0; i < args.n; i++) {
570
+ if (scripted) {
571
+ const message: MoonshotAssistantMessage = scripted.toolCalls.length
572
+ ? { role: 'assistant', content: scripted.text, tool_calls: scripted.toolCalls }
573
+ : { role: 'assistant', content: scripted.text ?? '' };
574
+ if (scripted.reasoning !== null) message.reasoning_content = scripted.reasoning;
575
+ choices.push({ index: i, message, finish_reason: scripted.finishReason });
576
+ completionTokens += estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
577
+ continue;
578
+ }
579
+ const { choice, completionTokens: ct } = buildChoice(args, i);
580
+ if (missTeach && typeof choice.message.content === 'string') choice.message.content += missTeach;
581
+ choices.push(choice);
582
+ completionTokens += ct;
583
+ }
584
+ const usage: MoonshotUsage = buildChatUsage(promptTokens, completionTokens, args.promptCacheKey);
585
+ const id = `chatcmpl-twin-${stableSuffix(args.messages, args.model)}`;
586
+ return {
587
+ id,
588
+ object: 'chat.completion',
589
+ created: nowEpoch(occurredAt),
590
+ model: args.model,
591
+ choices,
592
+ usage,
593
+ };
594
+ }
595
+
596
+ /** Map a scripted scenario failure onto Moonshot's real status + envelope. */
597
+ function scriptedError(err: NonNullable<MoonshotScenarioRespond['error']>): MoonshotResponseEnvelope {
598
+ if (err.type === 'rate_limit_reached_error') {
599
+ const base = rateLimitError();
600
+ return err.message ? { ...base, body: errBody('rate_limit_reached_error', err.message) } : base;
601
+ }
602
+ if (err.type === 'server_unavailable') return err.message ? { ...serverUnavailable(), body: errBody('server_unavailable', err.message) } : serverUnavailable();
603
+ return { status: 500, body: errBody('server_error', err.message ?? 'Internal Server Error') };
604
+ }
605
+
606
+ const isEnvelope = (v: object): v is MoonshotResponseEnvelope =>
607
+ typeof (v as MoonshotResponseEnvelope).status === 'number' && 'body' in v;
608
+
609
+ /** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
610
+ function chunkText(text: string): string[] {
611
+ if (!text) return [];
612
+ const out: string[] = [];
613
+ for (let i = 0; i < text.length; i += 20) out.push(text.slice(i, i + 20));
614
+ return out;
615
+ }
616
+
617
+ /**
618
+ * Emit the vendor-faithful Moonshot streaming sequence into the injected sink (NO sockets, NO
619
+ * setTimeout). Moonshot's order: a first chunk with `delta:{role:'assistant'}`, then
620
+ * `delta:{content}` / `delta:{reasoning_content}` / tool_calls deltas, then a chunk carrying
621
+ * `finish_reason`, then — because Moonshot's ChatCompletionChunk.usage is "Object in the final
622
+ * chunk with usage, null in ordinary chunks" (the OpenAPI's own wording, the first-party
623
+ * denominator; the prose chat doc mentions the chunk under stream_options.include_usage, so the
624
+ * two sources disagree and the OpenAPI governs) — a FINAL chunk whose `choices:[]` and `usage`
625
+ * hold the whole usage object, then `data: [DONE]`. `stream_options` is ACCEPTED (validated as
626
+ * an object) but the tail is not gated on it. Deterministic + synchronous.
627
+ */
628
+ export function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotChatCompletion | MoonshotResponseEnvelope {
629
+ const built = buildChatCompletion(args, occurredAt, decision);
630
+ if (isEnvelope(built)) return built;
631
+ const full = built;
632
+ const base = { id: full.id, object: 'chat.completion.chunk' as const, created: full.created, model: full.model };
633
+ for (const choice of full.choices) {
634
+ const idx = choice.index;
635
+ sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, finish_reason: null, usage: null }] } });
636
+ if (choice.message.reasoning_content) {
637
+ sink({ data: { ...base, choices: [{ index: idx, delta: { reasoning_content: choice.message.reasoning_content }, finish_reason: null, usage: null }] } });
638
+ }
639
+ if (choice.message.tool_calls && choice.message.tool_calls.length) {
640
+ choice.message.tool_calls.forEach((tc, tIdx) => {
641
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, finish_reason: null, usage: null }] } });
642
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, finish_reason: null, usage: null }] } });
643
+ });
644
+ } else {
645
+ for (const piece of chunkText(choice.message.content ?? '')) {
646
+ sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, finish_reason: null, usage: null }] } });
647
+ }
648
+ }
649
+ sink({ data: { ...base, choices: [{ index: idx, delta: {}, finish_reason: choice.finish_reason, usage: null }] } });
650
+ }
651
+ // Moonshot's usage tail: a final chunk with an EMPTY choices array carrying the usage object.
652
+ sink({ data: { ...base, choices: [], finish_reason: null, usage: full.usage } });
653
+ sink({ done: true });
654
+ return full;
655
+ }
656
+
657
+ // ── /v1/tokenizers/estimate-token-count (stateless, deterministic) ──────────────────────
658
+ function handleEstimateTokens(params: Record<string, unknown>): MoonshotResponseEnvelope {
659
+ if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
660
+ if (!findModel(params.model)) return invalidRequest(`The model '${params.model}' does not exist or you do not have access to it.`);
661
+ if (!Array.isArray(params.messages)) return { status: 400, body: errBody('invalid_request_error', "'messages' is a required property") };
662
+ const messages = params.messages as MoonshotMessageParam[];
663
+ for (const m of messages) {
664
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string') return invalidRequest("each message must have a valid 'role'");
665
+ // The OpenAPI: "content must not be empty".
666
+ if (m.content === undefined || m.content === null || (typeof m.content === 'string' && m.content === '')) {
667
+ return invalidRequest("'messages[].content' must not be empty");
668
+ }
669
+ }
670
+ return { status: 200, body: { data: { total_tokens: countPromptTokens(messages) } } };
671
+ }
672
+
673
+ // ── /v1/signatures/verify (stateless, deterministic) ────────────────────────────────────
674
+ // Moonshot's request-signing contract: a model call carries `X-Msh-Request-Nonce`; the response
675
+ // carries `Msh-Request-Timestamp` (unix ms) and `Msh-Request-Signature` (`reqsigv1_<opaque>`).
676
+ // POST /v1/signatures/verify checks a (nonce, timestamp, model, signature) tuple. The twin has
677
+ // no real signing key, so it models the CHECKABLE part: a well-formed tuple whose signature
678
+ // carries the documented `reqsigv1_` prefix AND whose signature recomputes exactly under the
679
+ // twin's own deterministic scheme (messagesSignature over nonce/timestamp/model) answers
680
+ // `valid:true`; anything else answers `valid:false`. The twin ALSO emits the two Msh- headers
681
+ // on its own streaming model responses WHEN the request carried a nonce (see the server) so the
682
+ // round-trip is exercisable end to end.
683
+ function handleSignatureVerify(params: Record<string, unknown>, req: MoonshotRequest): MoonshotResponseEnvelope {
684
+ for (const field of ['nonce', 'timestamp', 'model', 'signature'] as const) {
685
+ if (params[field] === undefined || params[field] === null || params[field] === '') {
686
+ return invalidRequest(`'${field}' is a required property`);
687
+ }
688
+ }
689
+ if (typeof params.nonce !== 'string' || typeof params.model !== 'string' || typeof params.signature !== 'string') {
690
+ return invalidRequest("'nonce', 'model' and 'signature' must be strings");
691
+ }
692
+ const ts = Number(params.timestamp);
693
+ if (!Number.isFinite(ts) || ts < 1) return invalidRequest("'timestamp' must be a positive integer");
694
+ const sig = String(params.signature);
695
+ if (!sig.startsWith('reqsigv1_')) return { status: 200, body: { valid: false } };
696
+ // The twin's own signatures are derived from the tuple (see messagesSignature); a signature
697
+ // that recomputes exactly answers valid:true.
698
+ const expected = messagesSignature(String(params.nonce), ts, String(params.model));
699
+ return { status: 200, body: { valid: sig === expected } };
700
+ }
701
+
702
+ /** The deterministic signature the twin issues on inference responses and verifies here. */
703
+ export function messagesSignature(nonce: string, timestamp: number, model: string): string {
704
+ return `reqsigv1_twin_${fnv1a(`${nonce}|${timestamp}|${model}`).toString(36)}`;
705
+ }
706
+
707
+ // ── Web-search tools (stateless, deterministic labeled stubs) ───────────────────────────
708
+ const TIME_WINDOW_RE = /^\d{4}(-\d{2})?(-\d{2})?$/;
709
+
710
+ function validateTimeWindow(raw: unknown): MoonshotResponseEnvelope | null {
711
+ if (raw === undefined) return null;
712
+ if (typeof raw !== 'object' || raw === null) return invalidRequest("'time_window' must be an object");
713
+ const o = raw as { start?: unknown; end?: unknown };
714
+ for (const k of ['start', 'end'] as const) {
715
+ if (o[k] !== undefined && (typeof o[k] !== 'string' || !TIME_WINDOW_RE.test(o[k]))) {
716
+ return invalidRequest(`'time_window.${k}' must be in YYYY, YYYY-MM or YYYY-MM-DD format`);
717
+ }
718
+ }
719
+ return null;
720
+ }
721
+
722
+ function handleToolsSearch(params: Record<string, unknown>, pro: boolean): MoonshotResponseEnvelope {
723
+ if (typeof params.text_query !== 'string' || !params.text_query) return invalidRequest("'text_query' is a required property and must not be empty");
724
+ if (params.limit !== undefined) {
725
+ const n = Number(params.limit);
726
+ if (!Number.isInteger(n) || n < 1 || n > 20) return invalidRequest("'limit' must be an integer between 1 and 20");
727
+ }
728
+ if (params.timeout_seconds !== undefined) {
729
+ const n = Number(params.timeout_seconds);
730
+ if (!Number.isInteger(n) || n < 1 || n > 60) return invalidRequest("'timeout_seconds' must be an integer between 1 and 60");
731
+ }
732
+ if (pro) {
733
+ if (params.sites !== undefined) {
734
+ if (!Array.isArray(params.sites) || params.sites.length > 5) return invalidRequest("'sites' must be an array of at most 5 strings");
735
+ for (const s of params.sites) {
736
+ if (typeof s !== 'string' || !s || /\s|\(|\)/.test(s)) return invalidRequest("'sites' entries must be non-empty and contain no whitespace or parentheses");
737
+ }
738
+ }
739
+ const tw = validateTimeWindow(params.time_window);
740
+ if (tw) return tw;
741
+ } else {
742
+ // search (not search_pro) takes include_content; search_pro does not declare it.
743
+ if (params.include_content !== undefined && typeof params.include_content !== 'boolean') {
744
+ return invalidRequest("'include_content' must be a boolean");
745
+ }
746
+ }
747
+ const limit = params.limit === undefined ? 5 : Number(params.limit);
748
+ // `include_content: true` SERVES content (the option's documented meaning): the results carry
749
+ // a labeled deterministic page stub. Default (false) is the vendor's bare shape — `text: ''`.
750
+ // A no-op 200 that accepted the flag and always served '' was the round-two finding.
751
+ const includeContent = pro ? false : params.include_content === true;
752
+ const results = stubSearchResults(String(params.text_query), limit, pro, includeContent);
753
+ return { status: 200, body: { search_results: results } };
754
+ }
755
+
756
+ function handleToolsFetch(params: Record<string, unknown>): MoonshotResponseEnvelope {
757
+ if (typeof params.url !== 'string' || !params.url) return invalidRequest("'url' is a required property");
758
+ if (!/^https?:\/\//.test(params.url)) return invalidRequest("'url' must be an http or https URL");
759
+ return { status: 200, body: stubFetchedMarkdown(params.url) };
760
+ }
761
+
762
+ // ── Files (stateful) ────────────────────────────────────────────────────────────────────
763
+ /** FileObject.purpose — a CLOSED documented set: file-extract, image, video, batch. */
764
+ const FILE_CREATE_PURPOSES = new Set(['file-extract', 'image', 'video', 'batch']);
765
+
766
+ async function createFile(params: Record<string, unknown>, req: MoonshotRequest): Promise<MoonshotResponseEnvelope> {
767
+ const purpose = String(params.purpose ?? '');
768
+ if (!purpose) return invalidRequest("'purpose' is a required property");
769
+ if (!FILE_CREATE_PURPOSES.has(purpose)) return invalidRequest(`'purpose' must be one of ${[...FILE_CREATE_PURPOSES].map((p) => `'${p}'`).join(', ')} (got '${purpose}')`);
770
+ // The multipart adapter marks a form with no `file` part (§9 round two, F3): the vendor's
771
+ // Upload File requires the file body, so the twin refuses rather than minting an empty
772
+ // 'ready' file.
773
+ if (params._multipart_missing_file === true) return invalidRequest("the multipart form carries no 'file' part");
774
+ const filename = String(params.filename ?? params.file ?? 'upload');
775
+ const content = typeof params.content === 'string' ? params.content : '';
776
+ // `binary_content` is the SERVER's multipart-adapter marker (moonshot-server.ts): the JSON
777
+ // contract cannot carry raw bytes, so an uploaded binary travels base64-encoded under it and
778
+ // is stored with the marker; GET /content decodes before serving. A JSON-door create (plain
779
+ // text) stores the text as-is with no marker. A caller forging the marker through the JSON
780
+ // door gets strict validation: the content must actually be base64 (§9 round two, F6).
781
+ const binary = params.binary_content === true;
782
+ if (binary && !/^[A-Za-z0-9+/]*={0,2}$/.test(content) || (binary && content.length % 4 !== 0)) {
783
+ return invalidRequest("'binary_content' content must be base64-encoded");
784
+ }
785
+ // `bytes` is the vendor's byte count — a text file's UTF-8 length, not the JS string's
786
+ // UTF-16 code-unit count (a §9-round-two finding: the two doors disagreed on the same field;
787
+ // the multipart door already counted real bytes).
788
+ const bytes = typeof params.bytes === 'number' ? params.bytes : binary ? Buffer.from(content, 'base64').length : Buffer.byteLength(content, 'utf8');
789
+ const id = nextId('file', 'file', req.root);
790
+ await applyTwinWrite(SERVICE, {
791
+ operation: 'file.create',
792
+ subjectType: 'file',
793
+ subjectId: id,
794
+ fields: { object: 'file', bytes, created_at: nowEpoch(req.occurredAt), filename, purpose, status: 'ready', _content: content, ...(binary ? { _content_encoding: 'base64' } : {}) },
795
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
796
+ actor: { kind: 'agent' },
797
+ }, req.root);
798
+ const row = getRow('file', id, req.root);
799
+ return { status: 200, body: fileView(row ?? {}) };
800
+ }
801
+ function fileView(r: Record<string, unknown>): Record<string, unknown> {
802
+ const s = strip(r);
803
+ return { id: r.id, object: 'file', bytes: s.bytes, created_at: s.created_at, filename: s.filename, purpose: s.purpose, status: s.status ?? 'ready' };
804
+ }
805
+
806
+ // ── Batches (stateful) ──────────────────────────────────────────────────────────────────
807
+ /** BatchCreateRequest.endpoint — a CLOSED documented set: only /v1/chat/completions. */
808
+ const BATCH_ENDPOINTS = new Set(['/v1/chat/completions']);
809
+ /** "supports formats like 12h, 1d, 3d, minimum 12h, maximum 7d" (BatchCreateRequest). */
810
+ const COMPLETION_WINDOW = /^(\d+)(h|d)$/;
811
+ function completionWindowHours(w: string): number | null {
812
+ const m = COMPLETION_WINDOW.exec(w);
813
+ if (!m) return null;
814
+ const n = Number(m[1]);
815
+ return m[2] === 'd' ? n * 24 : n;
816
+ }
817
+
818
+ async function createBatch(params: Record<string, unknown>, req: MoonshotRequest): Promise<MoonshotResponseEnvelope> {
819
+ const inputFileId = params.input_file_id;
820
+ if (typeof inputFileId !== 'string' || !inputFileId) return invalidRequest("'input_file_id' is a required property");
821
+ const endpoint = params.endpoint;
822
+ if (typeof endpoint !== 'string' || !BATCH_ENDPOINTS.has(endpoint)) {
823
+ return invalidRequest(`'endpoint' must be one of ${[...BATCH_ENDPOINTS].map((e) => `'${e}'`).join(', ')}`);
824
+ }
825
+ const window = params.completion_window;
826
+ if (typeof window !== 'string') return invalidRequest("'completion_window' is a required property");
827
+ const hours = completionWindowHours(window);
828
+ if (hours === null || hours < 12 || hours > 168) return invalidRequest("'completion_window' must be a duration from '12h' to '7d'");
829
+ if (params.metadata !== undefined) {
830
+ const md = params.metadata;
831
+ if (!md || typeof md !== 'object' || Array.isArray(md)) return invalidRequest("'metadata' must be an object");
832
+ const entries = Object.entries(md as Record<string, unknown>);
833
+ if (entries.length > 16) return invalidRequest("'metadata' must not exceed 16 key-value pairs");
834
+ for (const [k, v] of entries) {
835
+ if (k.length > 64) return invalidRequest("'metadata' keys must not exceed 64 characters");
836
+ if (typeof v !== 'string' || v.length > 512) return invalidRequest("'metadata' values must be strings of at most 512 characters");
837
+ }
838
+ }
839
+ const file = getRow('file', inputFileId, req.root);
840
+ if (!file || file._deleted) return notFound(`No such File object: ${inputFileId}`);
841
+ if (file.purpose !== 'batch') return invalidRequest(`File ${inputFileId} must have purpose 'batch' (has '${String(file.purpose)}')`);
842
+ const created = nowEpoch(req.occurredAt);
843
+ const id = nextId('batch', 'batch', req.root);
844
+ await applyTwinWrite(SERVICE, {
845
+ operation: 'batch.create',
846
+ subjectType: 'batch',
847
+ subjectId: id,
848
+ fields: {
849
+ object: 'batch',
850
+ endpoint,
851
+ input_file_id: inputFileId,
852
+ completion_window: window,
853
+ status: 'validating',
854
+ output_file_id: null,
855
+ error_file_id: null,
856
+ created_at: created,
857
+ in_progress_at: null,
858
+ expires_at: created + hours * 3600,
859
+ finalizing_at: null,
860
+ completed_at: null,
861
+ failed_at: null,
862
+ cancelling_at: null,
863
+ cancelled_at: null,
864
+ request_counts: { total: 0, completed: 0, failed: 0 },
865
+ metadata: params.metadata ?? null,
866
+ },
867
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
868
+ actor: { kind: 'agent' },
869
+ }, req.root);
870
+ return { status: 200, body: batchView(getRow('batch', id, req.root) ?? {}) };
871
+ }
872
+ function batchView(r: Record<string, unknown>): Record<string, unknown> {
873
+ return { id: r.id, ...strip(r) };
874
+ }
875
+
876
+ // ── Balance (stateful) ──────────────────────────────────────────────────────────────────
877
+ // GET /v1/users/me/balance returns Moonshot's own envelope ({code, data:{available_balance,
878
+ // voucher_balance, cash_balance}, scode, status}) — NOT the OpenAI envelope. The balance is
879
+ // STATEFUL: the connector's pull can fold a real account's balance into the log (type 'balance',
880
+ // id 'me'), and the served figure is that row when present, else a deterministic positive stub.
881
+ function balanceView(root?: string): MoonshotResponseEnvelope {
882
+ const row = getRow('balance', 'me', root);
883
+ const b = row && !row._deleted
884
+ ? { available_balance: Number(row.available_balance), voucher_balance: Number(row.voucher_balance), cash_balance: Number(row.cash_balance) }
885
+ : { available_balance: 49.58894, voucher_balance: 46.58893, cash_balance: 3.00001 };
886
+ return {
887
+ status: 200,
888
+ body: { code: 0, data: b, scode: '0x0', status: true },
889
+ };
890
+ }
891
+
892
+ // ── Anthropic-compatible Messages (/anthropic/v1/messages) ──────────────────────────────
893
+ type MessagesArgs = {
894
+ model: string;
895
+ messages: Array<{ role: 'user' | 'assistant'; content: string | Array<Record<string, unknown>> }>;
896
+ maxTokens: number;
897
+ system?: string;
898
+ stopSequences?: string[];
899
+ stream: boolean;
900
+ /** Anthropic's MessagesTool shape: { name, input_schema } (nested inside type:'custom'). */
901
+ tools?: Array<Record<string, unknown>>;
902
+ /** Anthropic's ToolChoice: type 'tool' names one tool (`name` REQUIRED, validated against
903
+ * `tools` — the sibling pack's rule). */
904
+ toolChoice?: { type: 'auto' | 'any' | 'tool' | 'none'; name?: string };
905
+ outputEffort?: string;
906
+ };
907
+
908
+ /** Extract the tool name from an Anthropic Messages tool definition ({type:'custom', name, …}). */
909
+ function messagesToolName(t: unknown): string {
910
+ const o = t as { name?: unknown } | undefined;
911
+ return typeof o?.name === 'string' ? o.name : '';
912
+ }
913
+
914
+ /** Validate the Anthropic-compatible request. kimi-k3 only; max_tokens REQUIRED. */
915
+ function validateMessages(params: Record<string, unknown>): { args: MessagesArgs } | { error: MoonshotResponseEnvelope } {
916
+ if (params.model === undefined || params.model === '') return { error: messagesError(400, 'invalid_request_error', "'model' is a required property") }
917
+ // MessagesRequest['model'] enum (kimi-k3 only) — read from the catalog's own supports flag.
918
+ if (!findModel(String(params.model))?.supports.messages) {
919
+ return { error: messagesError(400, 'invalid_request_error', `The endpoint currently supports 'kimi-k3' only (got '${String(params.model)}')`) }
920
+ }
921
+ if (!Array.isArray(params.messages)) return { error: messagesError(400, 'invalid_request_error', "'messages' is a required property") }
922
+ if (params.messages.length === 0) return { error: messagesError(400, 'invalid_request_error', "'messages' must not be empty") }
923
+ const messages: Array<{ role: 'user' | 'assistant'; content: string | Array<Record<string, unknown>> }> = [];
924
+ for (const m of params.messages as Array<Record<string, unknown>>) {
925
+ if (!m || typeof m !== 'object') return { error: messagesError(400, 'invalid_request_error', 'each message must be an object') }
926
+ if (m.role !== 'user' && m.role !== 'assistant') {
927
+ // Anthropic's grammar: the top-level `system` field owns the system prompt.
928
+ return { error: messagesError(400, 'invalid_request_error', `'${String(m.role)}' is not one of ['user', 'assistant'] — use the top-level 'system' field for the system prompt`) }
929
+ }
930
+ messages.push({ role: m.role, content: m.content as string | Array<Record<string, unknown>> });
931
+ }
932
+ if (params.max_tokens === undefined || params.max_tokens === null) return { error: messagesError(400, 'invalid_request_error', "'max_tokens' is a required property") }
933
+ const maxTokens = Number(params.max_tokens);
934
+ if (!Number.isInteger(maxTokens) || maxTokens < 1) return { error: messagesError(400, 'invalid_request_error', "'max_tokens' must be an integer >= 1") }
935
+ let system: string | undefined;
936
+ if (params.system !== undefined) {
937
+ if (typeof params.system === 'string') system = params.system;
938
+ else if (Array.isArray(params.system)) system = params.system.map((b) => String((b as { text?: unknown })?.text ?? '')).join('\n');
939
+ else return { error: messagesError(400, 'invalid_request_error', "'system' must be a string or an array of text blocks") }
940
+ }
941
+ let stopSequences: string[] | undefined;
942
+ if (params.stop_sequences !== undefined) {
943
+ if (!Array.isArray(params.stop_sequences)) return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' must be an array of strings") }
944
+ if (params.stop_sequences.length > 5) return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' must not exceed 5 entries") }
945
+ for (const s of params.stop_sequences) {
946
+ if (typeof s !== 'string') return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' entries must be strings") }
947
+ if (Buffer.byteLength(s, 'utf8') > 32) return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' entries must not exceed 32 bytes") }
948
+ }
949
+ stopSequences = params.stop_sequences;
950
+ }
951
+ // Tools: Anthropic's MessagesTool shape, with the SAME documented name regex the /v1 surface
952
+ // enforces (a bad tool name is a 400 here too — never a silent acceptance the stub then
953
+ // ignores).
954
+ let tools: Array<Record<string, unknown>> | undefined;
955
+ if (params.tools !== undefined) {
956
+ if (!Array.isArray(params.tools)) return { error: messagesError(400, 'invalid_request_error', "'tools' must be an array") }
957
+ for (const t of params.tools) {
958
+ if (!t || typeof t !== 'object' || Array.isArray(t)) return { error: messagesError(400, 'invalid_request_error', 'each tool must be an object') }
959
+ const name = messagesToolName(t);
960
+ if (!name || !TOOL_NAME_RE.test(name)) {
961
+ return { error: messagesError(400, 'invalid_request_error', `'tools[].name' must match ${TOOL_NAME_RE.source}`) }
962
+ }
963
+ const schema = (t as { input_schema?: unknown }).input_schema;
964
+ if (schema !== undefined && (!schema || typeof schema !== 'object' || Array.isArray(schema))) {
965
+ return { error: messagesError(400, 'invalid_request_error', "'tools[].input_schema' must be an object") }
966
+ }
967
+ }
968
+ tools = params.tools as Array<Record<string, unknown>>;
969
+ }
970
+ // tool_choice: Anthropic's closed type set INCLUDING type:'tool' with its required `name`
971
+ // (the sibling pack's rule: a named choice must name a tool that is actually in `tools`).
972
+ let toolChoice: MessagesArgs['toolChoice'] | undefined;
973
+ if (params.tool_choice !== undefined) {
974
+ const t = (params.tool_choice as { type?: unknown })?.type;
975
+ if (typeof t !== 'string' || !['auto', 'any', 'tool', 'none'].includes(t)) return { error: messagesError(400, 'invalid_request_error', "'tool_choice.type' must be one of 'auto', 'any', 'tool', 'none'") }
976
+ // A tool_choice WITHOUT tools is a contradiction — Anthropic's own surface refuses it.
977
+ if (!tools) return { error: messagesError(400, 'invalid_request_error', "'tool_choice' requires 'tools' to be set") }
978
+ if (t === 'tool') {
979
+ const name = (params.tool_choice as { name?: unknown }).name;
980
+ if (typeof name !== 'string' || !tools.some((tool) => messagesToolName(tool) === name)) {
981
+ return { error: messagesError(400, 'invalid_request_error', `'tool_choice' names tool '${typeof name === 'string' ? name : String(name)}', which is not in tools`) }
982
+ }
983
+ toolChoice = { type: 'tool', name };
984
+ } else {
985
+ toolChoice = { type: t as 'auto' | 'any' | 'none' };
986
+ }
987
+ }
988
+ let outputEffort: string | undefined;
989
+ const oc = params.output_config as { effort?: unknown } | undefined;
990
+ if (oc && typeof oc === 'object' && oc.effort !== undefined) {
991
+ if (typeof oc.effort !== 'string' || !REASONING_EFFORTS.includes(oc.effort as ReasoningEffort)) {
992
+ return { error: messagesError(400, 'invalid_request_error', `'output_config.effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) }
993
+ }
994
+ outputEffort = oc.effort;
995
+ }
996
+ return {
997
+ args: {
998
+ model: String(params.model),
999
+ messages,
1000
+ maxTokens,
1001
+ ...(system !== undefined ? { system } : {}),
1002
+ ...(stopSequences !== undefined ? { stopSequences } : {}),
1003
+ stream: params.stream === true,
1004
+ ...(tools !== undefined ? { tools } : {}),
1005
+ ...(toolChoice !== undefined ? { toolChoice } : {}),
1006
+ ...(outputEffort !== undefined ? { outputEffort } : {}),
1007
+ },
1008
+ };
1009
+ }
1010
+
1011
+ /** The Messages surface's OWN error envelope (MessagesErrorResponse schema) — different from /v1. */
1012
+ function messagesError(status: number, type: string, message: string): MoonshotResponseEnvelope {
1013
+ return { status, body: { type: 'error', error: { type, message } } };
1014
+ }
1015
+
1016
+ /** Re-wear a scripted fault's Moonshot-shaped body in the Anthropic envelope for the /anthropic
1017
+ * surface: the fault result carries the Moonshot error body the adapter rendered; this surface
1018
+ * serves {type:'error', error:{type,message}} with the equivalent Anthropic type. */
1019
+ function messagesErrorBody(status: number, moonshotBody: unknown): unknown {
1020
+ const e = (moonshotBody as { error?: { message?: string; type?: string } })?.error;
1021
+ const message = e?.message ?? `The request was refused by a scripted fault (${status}).`;
1022
+ const type = e?.type === 'rate_limit_reached_error' ? 'rate_limit_error'
1023
+ : e?.type === 'server_unavailable' || e?.type === 'server_error' ? 'api_error'
1024
+ : 'invalid_request_error';
1025
+ return { type: 'error', error: { type, message } };
1026
+ }
1027
+
1028
+ export function buildMessagesResponse(args: MessagesArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotMessagesResponse | MoonshotResponseEnvelope {
1029
+ let scripted: ScriptedResult | null = null;
1030
+ let missTeach = '';
1031
+ if (decision) {
1032
+ // The route served the request through the engine (R15); tools/thinking rode into the
1033
+ // match there — a handler keyed on `hasTool`/`toolResultFor` fires on this surface exactly
1034
+ // as it fires on /v1 (the sibling pack wires the same request features through on its
1035
+ // Messages surface).
1036
+ if (decision.kind === 'handler') {
1037
+ const respond = decision.respond as MoonshotScenarioRespond;
1038
+ if (respond.error) return scriptedMessagesError(respond.error);
1039
+ scripted = realizeMoonshotRespond(respond);
1040
+ } else {
1041
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in handlers/moonshot.json.]`;
1042
+ }
1043
+ }
1044
+ const inputTokens = countPromptTokens(args.messages.map((m) => ({ role: m.role, content: m.content as never })) as MoonshotMessageParam[]);
1045
+ let content: MoonshotMessagesResponse['content'];
1046
+ let stopReason: MoonshotMessagesResponse['stop_reason'];
1047
+ let outputTokens: number;
1048
+ if (scripted) {
1049
+ content = [];
1050
+ if (scripted.reasoning !== null) content.push({ type: 'thinking', thinking: scripted.reasoning, signature: stubSignature(args.messages, args.model) });
1051
+ if (scripted.text) content.push({ type: 'text', text: scripted.text + (missTeach && scripted.text ? missTeach : '') });
1052
+ for (const tc of scripted.toolCalls) {
1053
+ content.push({ type: 'tool_use', id: tc.id, name: tc.function.name, input: JSON.parse(tc.function.arguments) as Record<string, unknown> });
1054
+ }
1055
+ stopReason = scripted.finishReason === 'tool_calls' ? 'tool_use' : scripted.finishReason === 'length' ? 'max_tokens' : 'end_turn';
1056
+ } else {
1057
+ // Tools on the Messages surface are REAL surface: tool_choice 'any'/'auto' (with tools
1058
+ // present) yields a tool_use block the SDK's own decoder reads; a named 'tool' choice calls
1059
+ // THAT tool; 'none' suppresses. The tautology that silently answered text regardless (the
1060
+ // round-two finding) is gone.
1061
+ const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
1062
+ const choice = args.toolChoice?.type ?? 'auto';
1063
+ const suppress = choice === 'none';
1064
+ const forcedName = choice === 'tool' ? args.toolChoice?.name : undefined;
1065
+ // A tool_result turn is ANSWERED, not re-tooled: once the client returns the tool_result the
1066
+ // model's next turn is text with stop_reason end_turn — deciding tool_use from tools.length
1067
+ // alone made an agent loop emit tool_use forever (the round-three finding). An EXPLICIT
1068
+ // forced choice ('any'/'tool') still calls, because the caller demanded one.
1069
+ const last = args.messages[args.messages.length - 1];
1070
+ const toolResultTurn = last?.role === 'user' && Array.isArray(last.content)
1071
+ && last.content.some((b) => (b as { type?: string })?.type === 'tool_result');
1072
+ const forceTool = choice === 'any' || choice === 'tool';
1073
+ const toolCalls = hasTools && !suppress && (forceTool || !toolResultTurn)
1074
+ ? [stubToolCall(args.tools, 1, forcedName)].filter((tc): tc is MoonshotToolCall => tc !== null)
1075
+ : [];
1076
+ // kimi-k3 ALWAYS thinks (Preserved Thinking — the same rule the /v1 surface applies), and
1077
+ // Moonshot's Messages doc has NO thinking parameter: reasoning is tuned through
1078
+ // output_config.effort (default max) and the content array is documented
1079
+ // "ordered thinking → text → tool_use" (platform.kimi.ai/docs/api/messages, read
1080
+ // 2026-09-16) — so the thinking block leads unconditionally, like the vendor's.
1081
+ content = [
1082
+ { type: 'thinking', thinking: stubReasoningContent(args.messages.map((m) => ({ role: m.role, content: m.content })) as MoonshotMessageParam[], args.model), signature: stubSignature(args.messages, args.model) },
1083
+ ];
1084
+ if (toolCalls.length) {
1085
+ toolCalls.forEach((tc, seq) => {
1086
+ // Anthropic's tool_use id prefix is `toolu_` (the sibling pack's grammar —
1087
+ // `toolu_twin_${seq}`), NOT the `call_` prefix the OpenAI-compatible /v1 surface uses.
1088
+ content.push({ type: 'tool_use', id: `toolu_twin_${seq + 1}`, name: tc.function.name, input: JSON.parse(tc.function.arguments) as Record<string, unknown> });
1089
+ });
1090
+ } else {
1091
+ content.push({ type: 'text', text: stubAssistantText(args.messages.map((m) => ({ role: m.role, content: m.content })) as MoonshotMessageParam[], args.model) + (missTeach || '') });
1092
+ }
1093
+ stopReason = toolCalls.length ? 'tool_use' : 'end_turn';
1094
+ }
1095
+ // max_tokens caps the OUTPUT ITSELF, not merely the report: truncate the emitted blocks to the
1096
+ // cap and count usage off what remains (the vendor stops emitting at the cap — usage.output_tokens
1097
+ // is what was emitted, never a fabricated agreement with a text block it did not produce).
1098
+ // Tokens are counted per-block off the block's own payload (thinking/text/tool input), NOT off
1099
+ // JSON.stringify of the whole array — JSON scaffolding would keep a capped answer above its cap.
1100
+ const blockTokens = (b: MoonshotMessagesResponse['content'][number]): number =>
1101
+ b.type === 'thinking' ? estimateTokens(b.thinking)
1102
+ : b.type === 'text' ? estimateTokens(b.text)
1103
+ : estimateTokens(JSON.stringify(b.input));
1104
+ const totalTokens = () => content.reduce((n, b) => n + blockTokens(b), 0);
1105
+ if (args.maxTokens !== undefined && totalTokens() > args.maxTokens) {
1106
+ // THE CAP WALKS THE BLOCKS IN ORDER (the sibling pack's method): the emitted content is a
1107
+ // PREFIX of what the stub would have said. Each block is kept only as far as the remaining
1108
+ // budget allows — text and thinking truncate mid-string, a tool_use that does not fit ends
1109
+ // the turn (everything after it is never emitted). Anthropic NEVER returns a zero-block
1110
+ // assistant message: the FIRST block is kept truncated to at least one token, so
1111
+ // output_tokens > 0 and content is non-empty at any cap >= 1. The walk is BOUNDED — one
1112
+ // pass, one truncation — so no estimate can spin it.
1113
+ let spent = 0;
1114
+ let cut = false;
1115
+ for (let i = 0; i < content.length; i++) {
1116
+ const block = content[i]!;
1117
+ const cost = blockTokens(block);
1118
+ if (spent + cost <= args.maxTokens) {
1119
+ spent += cost;
1120
+ continue;
1121
+ }
1122
+ const budget = args.maxTokens - spent;
1123
+ cut = true;
1124
+ if ((block.type === 'text' || block.type === 'thinking') && (budget >= 1 || i === 0)) {
1125
+ // Truncate to the remaining budget (~4 chars/token), keeping at least one token so the
1126
+ // message never loses its last block — the floor a zero-block message is forbidden by.
1127
+ const keep = Math.max(1, budget);
1128
+ if (block.type === 'text') block.text = block.text.slice(0, keep * 4);
1129
+ else block.thinking = block.thinking.slice(0, keep * 4);
1130
+ spent += blockTokens(block);
1131
+ content.length = i + 1;
1132
+ break;
1133
+ }
1134
+ // The budget ran out exactly at this block's boundary, or the block is a tool_use that
1135
+ // cannot be cut: the block was never emitted. A retained whole block here shipped 57
1136
+ // tokens under max_tokens 23 (round-four review) — the cap is a prefix, never a rounding.
1137
+ content.length = i;
1138
+ if (content.length === 0) content.push({ type: 'text', text: '' });
1139
+ break;
1140
+ }
1141
+ outputTokens = spent;
1142
+ if (cut) stopReason = 'max_tokens';
1143
+ } else {
1144
+ outputTokens = totalTokens();
1145
+ }
1146
+ // A stop_sequence match truncates the text at the hit and reports stop_reason 'stop_sequence'
1147
+ // with the matched string (Anthropic's documented shape — 'end_turn' would misreport WHY the
1148
+ // turn ended). The RECOUNT happens AFTER the truncation: usage.output_tokens is what was
1149
+ // emitted, so a 113-char answer reported as 54 tokens (counted before the cut) contradicted
1150
+ // its own usage (the round-three finding).
1151
+ let stopSequence: string | null = null;
1152
+ if (args.stopSequences?.length) {
1153
+ const textBlock = content.find((b) => b.type === 'text');
1154
+ if (textBlock && textBlock.type === 'text') {
1155
+ for (const s of args.stopSequences) {
1156
+ const i = textBlock.text.indexOf(s);
1157
+ if (i >= 0) {
1158
+ textBlock.text = textBlock.text.slice(0, i);
1159
+ stopSequence = s;
1160
+ stopReason = 'stop_sequence';
1161
+ break;
1162
+ }
1163
+ }
1164
+ outputTokens = totalTokens();
1165
+ }
1166
+ }
1167
+ return {
1168
+ id: `msg_twin_${stableSuffix(args.messages, args.model)}`,
1169
+ type: 'message',
1170
+ role: 'assistant',
1171
+ model: args.model,
1172
+ content,
1173
+ stop_reason: stopReason,
1174
+ stop_sequence: stopSequence,
1175
+ usage: {
1176
+ input_tokens: inputTokens,
1177
+ output_tokens: outputTokens,
1178
+ cache_read_input_tokens: stubCachedTokens(inputTokens),
1179
+ cache_creation_input_tokens: 0,
1180
+ },
1181
+ };
1182
+ }
1183
+
1184
+ function scriptedMessagesError(err: NonNullable<MoonshotScenarioRespond['error']>): MoonshotResponseEnvelope {
1185
+ if (err.type === 'rate_limit_reached_error') return messagesError(429, 'rate_limit_reached_error', err.message ?? 'Request rate limit reached');
1186
+ if (err.type === 'server_unavailable') return messagesError(503, 'server_unavailable', err.message ?? 'The server is overloaded');
1187
+ return messagesError(500, 'server_error', err.message ?? 'Internal Server Error');
1188
+ }
1189
+
1190
+ /**
1191
+ * Emit the Anthropic-compatible streaming grammar into the injected sink: message_start →
1192
+ * ping → content_block_start(thinking) → thinking_delta → signature_delta → content_block_stop →
1193
+ * content_block_start(text) → text_delta → content_block_stop → message_delta(stop_reason) →
1194
+ * message_stop. No [DONE] sentinel — the stream ends after message_stop.
1195
+ *
1196
+ * message_start carries the EMPTY message (content: [], stop_reason: null, output_tokens small)
1197
+ * and the blocks stream in one at a time — the real grammar (and the sibling pack's frames). A
1198
+ * message_start carrying the COMPLETE final message made the SDK's own accumulator double every
1199
+ * block: it pushes one block per content_block_start ON TOP of what message_start already held
1200
+ * (the round-three BLOCKER).
1201
+ */
1202
+ export function streamMessages(args: MessagesArgs, sink: MessagesSseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotMessagesResponse | MoonshotResponseEnvelope {
1203
+ const built = buildMessagesResponse(args, occurredAt, decision);
1204
+ if (isEnvelope(built)) return built;
1205
+ const full = built;
1206
+ sink({
1207
+ event: 'message_start',
1208
+ data: {
1209
+ type: 'message_start',
1210
+ message: {
1211
+ id: full.id, type: 'message', role: 'assistant', model: full.model,
1212
+ content: [], stop_reason: null, stop_sequence: null,
1213
+ usage: { input_tokens: full.usage.input_tokens, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 },
1214
+ },
1215
+ },
1216
+ });
1217
+ // The heartbeat the real API interleaves (and the sibling pack emits); the SDK's SSE decoder
1218
+ // skips `event: ping` frames by name, so it is inert to every consumer.
1219
+ sink({ event: 'ping', data: { type: 'ping' } });
1220
+ full.content.forEach((block, index) => {
1221
+ if (block.type === 'thinking') {
1222
+ sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'thinking', thinking: '' } } });
1223
+ for (const piece of chunkText(block.thinking)) sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'thinking_delta', thinking: piece } } });
1224
+ sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'signature_delta', signature: block.signature ?? '' } } });
1225
+ sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
1226
+ } else if (block.type === 'text') {
1227
+ sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'text', text: '' } } });
1228
+ for (const piece of chunkText(block.text)) sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'text_delta', text: piece } } });
1229
+ sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
1230
+ } else {
1231
+ sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'tool_use', id: block.id, name: block.name, input: {} } } });
1232
+ const json = JSON.stringify(block.input);
1233
+ sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'input_json_delta', partial_json: json } } });
1234
+ sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
1235
+ }
1236
+ });
1237
+ sink({ event: 'message_delta', data: { type: 'message_delta', delta: { stop_reason: full.stop_reason, stop_sequence: full.stop_sequence }, usage: { output_tokens: full.usage.output_tokens } } });
1238
+ sink({ event: 'message_stop', data: { type: 'message_stop' } });
1239
+ sink({ done: true });
1240
+ return full;
1241
+ }
1242
+
1243
+ // ── Responses (/v1/responses) — kimi-k3 only ────────────────────────────────────────────
1244
+ type ResponsesArgs = {
1245
+ model: string;
1246
+ input: string | Array<Record<string, unknown>>;
1247
+ instructions?: string;
1248
+ maxOutputTokens?: number;
1249
+ reasoningEffort?: string;
1250
+ stream: boolean;
1251
+ };
1252
+
1253
+ function validateResponses(params: Record<string, unknown>): { args: ResponsesArgs } | { error: MoonshotResponseEnvelope } {
1254
+ if (params.model === undefined || params.model === '') return { error: invalidRequest("'model' is a required property") }
1255
+ // "This endpoint currently supports `kimi-k3`" (ResponsesRequest.model) — read from the
1256
+ // catalog's own supports flag, so the enum and the catalog cannot drift apart.
1257
+ const responsesModel = findModel(String(params.model));
1258
+ if (!responsesModel?.supports.responses) return { error: invalidRequest(`This endpoint currently supports 'kimi-k3' only (got '${String(params.model)}')`) }
1259
+ if (params.input === undefined || params.input === null) return { error: invalidRequest("'input' is a required property") }
1260
+ if (typeof params.input !== 'string' && !Array.isArray(params.input)) return { error: invalidRequest("'input' must be a string or an array of items") }
1261
+ let reasoningEffort: string | undefined;
1262
+ const r = params.reasoning as { effort?: unknown } | undefined;
1263
+ if (r && typeof r === 'object' && r.effort !== undefined) {
1264
+ if (typeof r.effort !== 'string' || !REASONING_EFFORTS.includes(r.effort as ReasoningEffort)) {
1265
+ return { error: invalidRequest(`'reasoning.effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) }
1266
+ }
1267
+ reasoningEffort = r.effort;
1268
+ }
1269
+ let maxOutputTokens: number | undefined;
1270
+ if (params.max_output_tokens !== undefined) {
1271
+ maxOutputTokens = Number(params.max_output_tokens);
1272
+ if (!Number.isInteger(maxOutputTokens) || maxOutputTokens < 1) return { error: invalidRequest("'max_output_tokens' must be an integer >= 1") }
1273
+ }
1274
+ return {
1275
+ args: {
1276
+ model: String(params.model),
1277
+ input: params.input as string | Array<Record<string, unknown>>,
1278
+ ...(typeof params.instructions === 'string' ? { instructions: params.instructions } : {}),
1279
+ ...(maxOutputTokens !== undefined ? { maxOutputTokens } : {}),
1280
+ ...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
1281
+ stream: params.stream === true,
1282
+ },
1283
+ };
1284
+ }
1285
+
1286
+ function responsesInputText(input: string | Array<Record<string, unknown>>): string {
1287
+ if (typeof input === 'string') return input;
1288
+ return input.map((item) => {
1289
+ if ((item as { type?: string }).type === 'message' || (item as { type?: string }).type === undefined) {
1290
+ const content = (item as { content?: unknown }).content;
1291
+ return contentToText(content);
1292
+ }
1293
+ return JSON.stringify(item);
1294
+ }).join('\n');
1295
+ }
1296
+
1297
+ export function buildResponsesResponse(args: ResponsesArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotResponsesResponse | MoonshotResponseEnvelope {
1298
+ let scripted: ScriptedResult | null = null;
1299
+ let missTeach = '';
1300
+ if (decision) {
1301
+ // The route served the request through the engine (R15); this realizer only sees the
1302
+ // content decision.
1303
+ if (decision.kind === 'handler') {
1304
+ const respond = decision.respond as MoonshotScenarioRespond;
1305
+ if (respond.error) return scriptedError(respond.error);
1306
+ scripted = realizeMoonshotRespond(respond);
1307
+ } else {
1308
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in handlers/moonshot.json.]`;
1309
+ }
1310
+ }
1311
+ const inputText = responsesInputText(args.input);
1312
+ const asMessages: MoonshotMessageParam[] = [{ role: 'user', content: inputText }];
1313
+ const inputTokens = countPromptTokens(asMessages);
1314
+ const output: MoonshotResponsesOutputItem[] = [];
1315
+ let outputTokens = 0;
1316
+ if (scripted) {
1317
+ if (scripted.reasoning !== null) {
1318
+ output.push({ type: 'reasoning', id: `rs_twin_${stableSuffix(args.input, 'reasoning')}`, summary: [{ type: 'summary_text', text: scripted.reasoning }], status: 'completed' });
1319
+ outputTokens += estimateTokens(scripted.reasoning);
1320
+ }
1321
+ if (scripted.toolCalls.length) {
1322
+ for (const tc of scripted.toolCalls) {
1323
+ output.push({ type: 'function_call', id: `fc_twin_${stableSuffix(tc)}`, call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments, status: 'completed' });
1324
+ outputTokens += estimateTokens(tc.function.arguments);
1325
+ }
1326
+ } else {
1327
+ const text = (scripted.text ?? '') + (missTeach || '');
1328
+ output.push({ type: 'message', id: `msg_twin_${stableSuffix(args.input, 'message')}`, role: 'assistant', status: 'completed', content: [{ type: 'output_text', text, annotations: [] }] });
1329
+ outputTokens += estimateTokens(text);
1330
+ }
1331
+ } else {
1332
+ output.push({ type: 'reasoning', id: `rs_twin_${stableSuffix(inputText, 'reasoning')}`, summary: [{ type: 'summary_text', text: stubReasoningContent(asMessages, args.model) }], status: 'completed' });
1333
+ const text = stubAssistantText(asMessages, args.model) + (missTeach || '');
1334
+ output.push({ type: 'message', id: `msg_twin_${stableSuffix(inputText, 'message')}`, role: 'assistant', status: 'completed', content: [{ type: 'output_text', text, annotations: [] }] });
1335
+ outputTokens = estimateTokens(stubReasoningContent(asMessages, args.model)) + estimateTokens(text);
1336
+ }
1337
+ let status: MoonshotResponsesResponse['status'] = 'completed';
1338
+ let incompleteDetails: MoonshotResponsesResponse['incomplete_details'] = null;
1339
+ // max_output_tokens caps the OUTPUT ITSELF: the vendor stops emitting at the cap, so truncate
1340
+ // the emitted items (dropping whole items from the tail — the Responses API emits complete
1341
+ // items) and let usage.output_tokens count what remains. Reporting status 'incomplete' with
1342
+ // usage.output_tokens still above the cap was the round-two finding: the envelope contradicted
1343
+ // its own usage.
1344
+ if (args.maxOutputTokens !== undefined && outputTokens > args.maxOutputTokens) {
1345
+ // THE CAP IS A PREFIX, LIKE THE MESSAGES SURFACE'S: items are emitted in order until the
1346
+ // budget is exhausted; a text-carrying item truncates to the remaining budget rather than
1347
+ // vanishing, so a tight cap still yields a non-empty output with output_tokens > 0 — a
1348
+ // zero-item `output: []` (the round-three finding at max_output_tokens 2) is a message the
1349
+ // vendor never sends. One bounded pass; nothing after the cut is emitted.
1350
+ let spent = 0;
1351
+ let cut = false;
1352
+ for (let i = 0; i < output.length; i++) {
1353
+ const item = output[i]!;
1354
+ const cost = estimateTokens(
1355
+ item.type === 'message' ? (item.content as Array<{ text?: string }>).map((c) => c.text ?? '').join('')
1356
+ : item.type === 'reasoning' ? (item.summary as Array<{ text?: string }>).map((s) => s.text ?? '').join('')
1357
+ : item.type === 'function_call' ? item.arguments : item.type === 'custom_tool_call' ? item.input : '',
1358
+ );
1359
+ if (spent + cost <= args.maxOutputTokens) {
1360
+ spent += cost;
1361
+ continue;
1362
+ }
1363
+ const budget = args.maxOutputTokens - spent;
1364
+ const keep = Math.max(1, budget) * 4;
1365
+ if ((item.type === 'message' || item.type === 'reasoning') && budget >= 1) {
1366
+ if (item.type === 'message') item.content[0]!.text = (item.content[0]!.text ?? '').slice(0, keep);
1367
+ else item.summary[0]!.text = (item.summary[0]!.text ?? '').slice(0, keep);
1368
+ spent += estimateTokens(item.type === 'message' ? item.content[0]!.text ?? '' : item.summary[0]!.text ?? '');
1369
+ }
1370
+ output.length = i + 1;
1371
+ cut = true;
1372
+ break;
1373
+ }
1374
+ outputTokens = spent;
1375
+ if (cut) {
1376
+ status = 'incomplete';
1377
+ incompleteDetails = { reason: 'max_output_tokens' };
1378
+ }
1379
+ }
1380
+ return {
1381
+ id: `resp_twin_${stableSuffix(inputText, args.model)}`,
1382
+ object: 'response',
1383
+ created_at: nowEpoch(occurredAt),
1384
+ completed_at: status === 'completed' || status === 'incomplete' ? nowEpoch(occurredAt) : null,
1385
+ status,
1386
+ model: args.model,
1387
+ output,
1388
+ usage: {
1389
+ input_tokens: inputTokens,
1390
+ input_tokens_details: { cached_tokens: stubCachedTokens(inputTokens), cache_write_tokens: 0 },
1391
+ output_tokens: outputTokens,
1392
+ // reasoning_tokens counts the REASONING items only (the /anthropic thinking_tokens path
1393
+ // does the same): reporting the whole output_tokens as reasoning claimed every text token
1394
+ // was reasoning (the round-three finding).
1395
+ output_tokens_details: {
1396
+ reasoning_tokens: output
1397
+ .filter((o) => o.type === 'reasoning')
1398
+ .reduce((n, o) => n + estimateTokens((o.summary as Array<{ text?: string }>).map((s) => s.text ?? '').join('')), 0),
1399
+ },
1400
+ total_tokens: inputTokens + outputTokens,
1401
+ },
1402
+ incomplete_details: incompleteDetails,
1403
+ error: null,
1404
+ store: false,
1405
+ };
1406
+ }
1407
+
1408
+ /**
1409
+ * Emit the Responses SSE grammar into the injected sink: response.created →
1410
+ * response.in_progress → output_item.added/done per item → response.completed. Each frame
1411
+ * carries `event: <type>` and a monotonically increasing `sequence_number` from 0.
1412
+ */
1413
+ export function streamResponses(args: ResponsesArgs, sink: MessagesSseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotResponsesResponse | MoonshotResponseEnvelope {
1414
+ const built = buildResponsesResponse(args, occurredAt, decision);
1415
+ if (isEnvelope(built)) return built;
1416
+ const full = built;
1417
+ let seq = 0;
1418
+ const emit = (type: string, data: Record<string, unknown>) => sink({ event: type, data: { type, sequence_number: seq++, ...data } });
1419
+ emit('response.created', { response: { ...full, status: 'in_progress', output: [], usage: null } });
1420
+ emit('response.in_progress', { response: { ...full, status: 'in_progress', output: [], usage: null } });
1421
+ for (const item of full.output) {
1422
+ emit('response.output_item.added', { output_index: full.output.indexOf(item), output_item: item });
1423
+ emit('response.output_item.done', { output_index: full.output.indexOf(item), output_item: item });
1424
+ }
1425
+ emit(full.status === 'completed' ? 'response.completed' : 'response.incomplete', { response: full });
1426
+ sink({ done: true });
1427
+ return full;
1428
+ }
1429
+
1430
+ // ── public entry: cross-cutting protocol (auth / rate-limit) then route ─────────────────
1431
+ export async function handleMoonshotTwinRequest(req: MoonshotRequest): Promise<MoonshotResponseEnvelope> {
1432
+ const method = req.method.toUpperCase();
1433
+ // EVERY cross-cutting failure answers in the prefix's OWN envelope: the /anthropic surface's
1434
+ // documented client decodes Anthropic's {type:'error',error:{…}} grammar, so a 401/429/503
1435
+ // there carrying the /v1 envelope would be unparseable to it (the clone's lie, cross-cutting
1436
+ // edition). checkAuth applies this itself; the two deterministic fault triggers follow.
1437
+ const onMessages = onMessagesPath(req.path);
1438
+ if (req.headers !== undefined || req.apiKey !== undefined) {
1439
+ const authErr = checkAuth(req);
1440
+ if (authErr) return authErr;
1441
+ }
1442
+ if (triggered(req, 'x-twin-force-rate-limit')) {
1443
+ const base = rateLimitError();
1444
+ // The HEADERS ride along on the re-enveloped 429: retry-after and the X-RateLimit-* family
1445
+ // are part of the refusal a real client reads, whatever envelope the body wears — dropping
1446
+ // them on /anthropic left the documented SDK blind to the back-off (the round-three finding).
1447
+ return onMessages
1448
+ ? { ...messagesError(429, 'rate_limit_error', (base.body as { error: { message: string } }).error.message), headers: base.headers }
1449
+ : base;
1450
+ }
1451
+ if (triggered(req, 'x-twin-force-server-unavailable')) {
1452
+ const base = serverUnavailable();
1453
+ return onMessages ? messagesError(503, 'overloaded_error', (base.body as { error: { message: string } }).error.message) : base;
1454
+ }
1455
+ const res = await routeMoonshot(req, method);
1456
+ // Moonshot's request-signing contract, in the HANDLER so every surface (server, fetch
1457
+ // adapter, in-process callers) signs identically (§9 round two, F8): a STREAMED model call
1458
+ // that carries `X-Msh-Request-Nonce` gets `Msh-Request-Timestamp` + `Msh-Request-Signature`
1459
+ // response headers; a nonce-less call proceeds unsigned. The server forwards these headers
1460
+ // on the SSE response it frames.
1461
+ const nonce = req.headers?.['x-msh-request-nonce'];
1462
+ if (nonce && req.sseSink && req.method.toUpperCase() === 'POST' && (req.path === `${MOONSHOT_API_PREFIX}/chat/completions`)) {
1463
+ const model = (() => { try { return typeof (JSON.parse(req.body ?? '') as { model?: unknown }).model === 'string' ? (JSON.parse(req.body ?? '') as { model: string }).model : null; } catch { return null; } })();
1464
+ if (model) {
1465
+ const ts = Date.parse(req.occurredAt ?? worldNow());
1466
+ return { ...res, headers: { ...(res.headers ?? {}), 'msh-request-timestamp': String(ts), 'msh-request-signature': messagesSignature(nonce, ts, model) } };
1467
+ }
1468
+ }
1469
+ return res;
1470
+ }
1471
+
1472
+ // ── router ──────────────────────────────────────────────────────────────────────────────
1473
+ async function routeMoonshot(req: MoonshotRequest, method: string): Promise<MoonshotResponseEnvelope> {
1474
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
1475
+ const params = parseJson(req.body);
1476
+ const dec = (s: string) => decodeURIComponent(s);
1477
+
1478
+ const onOpenai = path === MOONSHOT_API_PREFIX || path.startsWith(`${MOONSHOT_API_PREFIX}/`);
1479
+ const onMessages = path === MESSAGES_PREFIX || path.startsWith(`${MESSAGES_PREFIX}/`);
1480
+ if (!onOpenai && !onMessages) {
1481
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1482
+ }
1483
+
1484
+ // D3: a read-only twin rejects any mutation with a vendor-shaped error — in the prefix's OWN
1485
+ // envelope (the Messages surface decodes Anthropic's grammar, not /v1's), so a read-only twin
1486
+ // never hands the anthropic SDK an envelope it cannot parse. Computed AFTER the prefix so the
1487
+ // envelope follows the path, not the call order.
1488
+ if (req.readOnly && method !== 'GET') {
1489
+ const message = 'twin is read-only; omit readOnly to accept writes';
1490
+ return onMessages ? messagesError(405, 'invalid_request_error', message) : { status: 405, body: errBody('invalid_request_error', message) };
1491
+ }
1492
+
1493
+ // ---- Anthropic-compatible Messages surface ----
1494
+ if (onMessages) {
1495
+ if (path !== `${MESSAGES_PREFIX}/messages`) return messagesError(404, 'not_found_error', `Unknown request URL: ${method} ${path}.`);
1496
+ if (method !== 'POST') return messagesError(405, 'invalid_request_error', `${method} is not supported on /anthropic/v1/messages`);
1497
+ const validated = validateMessages(params);
1498
+ if ('error' in validated) return validated.error;
1499
+ const args = validated.args;
1500
+ // R15 — the scenario engine decides AND honors a fault here, before any message exists:
1501
+ // a `status` fault is this vendor's own refusal envelope re-worn in the Anthropic shape,
1502
+ // a `slow` has already held the answer, a `drop` never returns.
1503
+ let messagesDecision: ScenarioDecision | undefined;
1504
+ if (req.scenarioEngine) {
1505
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages.map((m) => ({ role: m.role, content: m.content })) as MoonshotMessageParam[], tools: args.tools, thinking: undefined });
1506
+ if (served.kind === 'fault') return { status: served.result.status, body: messagesErrorBody(served.result.status, served.result.body), headers: served.result.headers };
1507
+ messagesDecision = served;
1508
+ }
1509
+ const result = args.stream && req.messagesSseSink
1510
+ ? streamMessages(args, req.messagesSseSink, req.occurredAt, messagesDecision)
1511
+ : buildMessagesResponse(args, req.occurredAt, messagesDecision);
1512
+ if (isEnvelope(result)) return result;
1513
+ return { status: 200, body: result };
1514
+ }
1515
+
1516
+ const seg = path.slice(MOONSHOT_API_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean); // ["chat","completions"]
1517
+ // protocol 2: after a landing adopted Moonshot's id for a file or a batch, a caller on a branch
1518
+ // may still address it by the local id — resolved through the alias map once, here at the boundary
1519
+ if (seg[0] === 'files' && seg[1]) seg[1] = resolveSubjectId(SERVICE, 'file', dec(seg[1]!), req.root);
1520
+ if (seg[0] === 'batches' && seg[1]) seg[1] = resolveSubjectId(SERVICE, 'batch', dec(seg[1]!), req.root);
1521
+
1522
+ // ---- models (static catalog) ----
1523
+ // LIST only: Moonshot's own OpenAPI declares GET /v1/models and NO retrieve-by-id operation
1524
+ // (the fixture's 19 operations have no /v1/models/{model}) — serving a 200 retrieve here was a
1525
+ // 200 for an unmodeled route. An unknown/unmodeled sub-path falls through to the router's
1526
+ // vendor-shaped not-found below.
1527
+ if (seg[0] === 'models' && seg.length === 1 && method === 'GET') {
1528
+ return { status: 200, body: { object: 'list', data: servedModels(req.root) } };
1529
+ }
1530
+
1531
+ // ---- chat completions (the generative stub; envelope is faithful) ----
1532
+ if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
1533
+ const validated = validateChat(params);
1534
+ if ('error' in validated) return validated.error;
1535
+ const args = validated.args;
1536
+ // R15 — the scenario engine decides AND honors a fault here, before any completion exists:
1537
+ // a `status` fault is this vendor's own refusal envelope, a `slow` has already held the
1538
+ // answer, a `drop` never returns. The realizers below only see a content decision.
1539
+ let chatDecision: ScenarioDecision | undefined;
1540
+ if (req.scenarioEngine) {
1541
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages, tools: args.tools, thinking: args.thinking });
1542
+ if (served.kind === 'fault') return { status: served.result.status, body: served.result.body, headers: served.result.headers };
1543
+ chatDecision = served;
1544
+ }
1545
+ const result = args.stream && req.sseSink
1546
+ ? streamChat(args, req.sseSink, req.occurredAt, chatDecision)
1547
+ : buildChatCompletion(args, req.occurredAt, chatDecision);
1548
+ if (isEnvelope(result)) return result;
1549
+ return { status: 200, body: result };
1550
+ }
1551
+
1552
+ // ---- responses (kimi-k3 only) ----
1553
+ if (seg[0] === 'responses' && seg.length === 1 && method === 'POST') {
1554
+ const validated = validateResponses(params);
1555
+ if ('error' in validated) return validated.error;
1556
+ const args = validated.args;
1557
+ // R15 — the same serve-and-honor as chat completions: the Responses door serves the same
1558
+ // scripted brain from the same handlers document.
1559
+ let responsesDecision: ScenarioDecision | undefined;
1560
+ if (req.scenarioEngine) {
1561
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: [{ role: 'user', content: responsesInputText(args.input) }], tools: undefined, thinking: undefined });
1562
+ if (served.kind === 'fault') return { status: served.result.status, body: served.result.body, headers: served.result.headers };
1563
+ responsesDecision = served;
1564
+ }
1565
+ const result = args.stream && req.messagesSseSink
1566
+ ? streamResponses(args, req.messagesSseSink, req.occurredAt, responsesDecision)
1567
+ : buildResponsesResponse(args, req.occurredAt, responsesDecision);
1568
+ if (isEnvelope(result)) return result;
1569
+ return { status: 200, body: result };
1570
+ }
1571
+
1572
+ // ---- token counting ----
1573
+ if (seg[0] === 'tokenizers' && seg[1] === 'estimate-token-count' && seg.length === 2 && method === 'POST') {
1574
+ return handleEstimateTokens(params);
1575
+ }
1576
+
1577
+ // ---- signatures ----
1578
+ if (seg[0] === 'signatures' && seg[1] === 'verify' && seg.length === 2 && method === 'POST') {
1579
+ return handleSignatureVerify(params, req);
1580
+ }
1581
+
1582
+ // ---- web-search tools ----
1583
+ if (seg[0] === 'tools' && seg[1] === 'search' && seg.length === 2 && method === 'POST') return handleToolsSearch(params, false);
1584
+ if (seg[0] === 'tools' && seg[1] === 'search_pro' && seg.length === 2 && method === 'POST') return handleToolsSearch(params, true);
1585
+ if (seg[0] === 'tools' && seg[1] === 'fetch' && seg.length === 2 && method === 'POST') return handleToolsFetch(params);
1586
+
1587
+ // ---- balance (Moonshot's own envelope) ----
1588
+ if (seg[0] === 'users' && seg[1] === 'me' && seg[2] === 'balance' && seg.length === 3 && method === 'GET') {
1589
+ return balanceView(req.root);
1590
+ }
1591
+
1592
+ // ---- files (stateful) ----
1593
+ if (seg[0] === 'files' && seg.length === 1 && method === 'POST') return createFile(params, req);
1594
+ if (seg[0] === 'files' && seg.length === 1 && method === 'GET') {
1595
+ const data = rows('file', req.root).filter((r) => !r._deleted).map(fileView);
1596
+ return { status: 200, body: { object: 'list', data } };
1597
+ }
1598
+ if (seg[0] === 'files' && seg.length === 2 && method === 'GET') {
1599
+ const f = getRow('file', dec(seg[1]!), req.root);
1600
+ return f && !f._deleted ? { status: 200, body: fileView(f) } : notFound(`No such File object: ${dec(seg[1]!)}`);
1601
+ }
1602
+ if (seg[0] === 'files' && seg.length === 3 && seg[2] === 'content' && method === 'GET') {
1603
+ const f = getRow('file', dec(seg[1]!), req.root);
1604
+ if (!f || f._deleted) return notFound(`No such File object: ${dec(seg[1]!)}`);
1605
+ // A PULLED file (the connector's mapFile) carries metadata only — the real list endpoint
1606
+ // returns no content, so the twin holds none. Serving an empty 200 here was a §9-round-two
1607
+ // finding (F2): a fake success for bytes the twin does not have. The honest answer is a
1608
+ // vendor-shaped refusal naming the gap.
1609
+ if (f._content === undefined && Number(f.bytes) > 0) {
1610
+ return { status: 501, body: errBody('server_error', `the twin holds no content for pulled file ${dec(seg[1]!)} (the vendor's list endpoint does not return file content; re-create the file locally to read it back)`) };
1611
+ }
1612
+ // The server turns the body STRING into bytes; this header tells it how (§9 round two, F1):
1613
+ // a binary upload (the multipart adapter's `binary_content` marker → `_content_encoding`)
1614
+ // is decoded from its stored base64 and carried latin1 (every byte value intact through a
1615
+ // JS string, re-materialized by the server); anything else is TEXT and travels as the
1616
+ // string itself — the server writes it UTF-8, so a non-ASCII text file round-trips.
1617
+ if (f._content_encoding === 'base64') {
1618
+ const raw = Buffer.from(String(f._content ?? ''), 'base64').toString('latin1');
1619
+ return { status: 200, body: raw, headers: { 'content-type': 'application/octet-stream', 'x-twin-content-binary': '1' } };
1620
+ }
1621
+ return { status: 200, body: String(f._content ?? ''), headers: { 'content-type': 'application/octet-stream' } };
1622
+ }
1623
+ if (seg[0] === 'files' && seg.length === 2 && method === 'DELETE') {
1624
+ const fid = dec(seg[1]!);
1625
+ const f = getRow('file', fid, req.root);
1626
+ if (!f || f._deleted) return notFound(`No such File object: ${fid}`);
1627
+ await applyTwinWrite(SERVICE, {
1628
+ operation: 'file.delete', subjectType: 'file', subjectId: fid, fields: { _deleted: true, object: 'file' },
1629
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1630
+ }, req.root);
1631
+ return { status: 200, body: { id: fid, object: 'file', deleted: true } };
1632
+ }
1633
+
1634
+ // ---- batches (stateful) ----
1635
+ if (seg[0] === 'batches' && seg.length === 1 && method === 'POST') return createBatch(params, req);
1636
+ if (seg[0] === 'batches' && seg.length === 1 && method === 'GET') {
1637
+ return { status: 200, body: { object: 'list', data: rows('batch', req.root).map(batchView), has_more: false } };
1638
+ }
1639
+ if (seg[0] === 'batches' && seg.length === 2 && method === 'GET') {
1640
+ const b = getRow('batch', dec(seg[1]!), req.root);
1641
+ if (!b) return notFound(`No such Batch object: ${dec(seg[1]!)}`);
1642
+ return { status: 200, body: batchView(b) };
1643
+ }
1644
+ if (seg[0] === 'batches' && seg.length === 3 && seg[2] === 'cancel' && method === 'POST') {
1645
+ const bid = dec(seg[1]!);
1646
+ const b = getRow('batch', bid, req.root);
1647
+ if (!b) return notFound(`No such Batch object: ${bid}`);
1648
+ if (b.status === 'cancelling' || b.status === 'cancelled') return invalidRequest(`Cannot cancel a batch with status '${String(b.status)}'.`);
1649
+ // NOTE: the twin does not simulate the asynchronous cancelling→cancelled settlement (the
1650
+ // groq pack's §9 finding applies identically here: settling on a READ would break the
1651
+ // read-only contract). The terminal transition is filed as
1652
+ // `moonshot.batches.cancellation_settles` (todo).
1653
+ await applyTwinWrite(SERVICE, {
1654
+ operation: 'batch.cancel', subjectType: 'batch', subjectId: bid,
1655
+ fields: { status: 'cancelling', cancelling_at: nowEpoch(req.occurredAt) },
1656
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1657
+ }, req.root);
1658
+ return { status: 200, body: batchView(getRow('batch', bid, req.root) ?? {}) };
1659
+ }
1660
+
1661
+ // Unmodeled operation → fail like the vendor (never a fake success). A /anthropic path that
1662
+ // is not /messages was already refused in the Messages branch; anything falling through here
1663
+ // is on the OpenAI-compatible surface, whose 404 envelope this is.
1664
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1665
+ }
1666
+
1667
+ // ── streaming sink types (exported for the server) ─────────────────────────────────────
1668
+ /** A sink the Messages/Responses streaming paths write events into. `event` is the SSE event
1669
+ * name (Anthropic/Responses grammar); `data` the JSON payload; `done: true` ends the stream. */
1670
+ export type MessagesSseSink = (event: MessagesSseEvent) => void;