@volter/twin-moonshot 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +164 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +25 -0
  5. package/dist/src/index.d.ts +14 -0
  6. package/dist/src/index.js +86 -0
  7. package/dist/src/moonshot-budget.d.ts +57 -0
  8. package/dist/src/moonshot-budget.js +142 -0
  9. package/dist/src/moonshot-capabilities.d.ts +4 -0
  10. package/dist/src/moonshot-capabilities.js +1200 -0
  11. package/dist/src/moonshot-conformance.d.ts +14 -0
  12. package/dist/src/moonshot-conformance.js +405 -0
  13. package/dist/src/moonshot-connector.d.ts +168 -0
  14. package/dist/src/moonshot-connector.js +416 -0
  15. package/dist/src/moonshot-models.d.ts +36 -0
  16. package/dist/src/moonshot-models.js +37 -0
  17. package/dist/src/moonshot-scenario.d.ts +54 -0
  18. package/dist/src/moonshot-scenario.js +175 -0
  19. package/dist/src/moonshot-server.d.ts +13 -0
  20. package/dist/src/moonshot-server.js +202 -0
  21. package/dist/src/moonshot-stub.d.ts +70 -0
  22. package/dist/src/moonshot-stub.js +222 -0
  23. package/dist/src/moonshot-twin.d.ts +144 -0
  24. package/dist/src/moonshot-twin.js +1647 -0
  25. package/dist/src/moonshot-types.d.ts +251 -0
  26. package/dist/src/moonshot-types.js +19 -0
  27. package/package.json +53 -0
  28. package/src/cli.ts +25 -0
  29. package/src/index.ts +129 -0
  30. package/src/moonshot-budget.ts +163 -0
  31. package/src/moonshot-capabilities.ts +1220 -0
  32. package/src/moonshot-conformance.ts +416 -0
  33. package/src/moonshot-connector.ts +465 -0
  34. package/src/moonshot-models.ts +89 -0
  35. package/src/moonshot-scenario.ts +194 -0
  36. package/src/moonshot-server.ts +220 -0
  37. package/src/moonshot-stub.ts +230 -0
  38. package/src/moonshot-twin.ts +1670 -0
  39. package/src/moonshot-types.ts +225 -0
@@ -0,0 +1,1647 @@
1
+ // Moonshot twin REQUEST HANDLER — the canonical Moonshot (Kimi) API surface for the twin.
2
+ // Contract: handleMoonshotTwinRequest({method, path, body}) -> {status, body}. It is the faithful
3
+ // Moonshot API the real clients (the standard `openai` SDK pointed at
4
+ // `https://api.moonshot.ai/v1`, the `anthropic` SDK pointed at `https://api.moonshot.ai/anthropic`,
5
+ // and plain HTTP callers) talk to UNMODIFIED — Moonshot ships no SDK of its own; its documented
6
+ // integration path is the standard OpenAI/Anthropic clients with a swapped base URL
7
+ // (platform.kimi.ai/docs/overview, read 2026-09-16).
8
+ //
9
+ // THE HONEST DESIGN: the twin cannot run the model, so the three inference endpoints
10
+ // (`POST /v1/chat/completions`, `POST /v1/responses`, `POST /anthropic/v1/messages`) return a
11
+ // DETERMINISTIC STUB completion (moonshot-stub.ts) clearly labeled a twin stub — it NEVER pretends
12
+ // to be real model output. But the ENTIRE PROTOCOL ENVELOPE is vendor-faithful: all three response
13
+ // shapes, all three streaming grammars (OpenAI chunks, Responses SSE events with
14
+ // `sequence_number`, Anthropic message events), tool_calls / tool_use, finish_reason /
15
+ // stop_reason, `reasoning_content` / thinking blocks, and Moonshot's cache-split usage. The
16
+ // genuinely stateful + static surface is real:
17
+ // • GET /v1/models — static catalog (moonshot-models.ts)
18
+ // • POST/GET/DELETE /v1/files (+ /content) — stateful (kernel action log)
19
+ // • POST/GET /v1/batches (+ /cancel) — stateful
20
+ // • GET /v1/users/me/balance — stateful (a mutable account balance)
21
+ // plus the deterministic stateless helpers: POST /v1/tokenizers/estimate-token-count,
22
+ // POST /v1/signatures/verify, POST /v1/tools/{search,search_pro,fetch}.
23
+ //
24
+ // TWO PATH PREFIXES, BOTH REAL: Moonshot serves its OpenAI-compatible surface under `/v1` and its
25
+ // Anthropic-compatible surface under `/anthropic/v1` on the same host (api.moonshot.ai). The
26
+ // Anthropic surface's error envelope is `{ type:'error', error:{type,message}, request_id? }` —
27
+ // a DIFFERENT envelope from `/v1`'s `{ error: { message, type, code? } }` — and its streaming
28
+ // grammar is Anthropic's event-name SSE, not OpenAI's `data:`-only frames.
29
+ //
30
+ // State lives in the kernel action log (D1): all writes are local actions, reads are the
31
+ // projection. No real Moonshot is ever called from this path (D4). Streaming uses an INJECTED
32
+ // sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
33
+ import { applyTwinWrite, projectResources, resolveSubjectId, worldNow } from '@volter/world-core';
34
+ import { findModel, K26_THINKING_TYPES, K27_THINKING_TYPES, MOONSHOT_MODELS, REASONING_EFFORTS, } from "./moonshot-models.js";
35
+ import { buildChatUsage, contentToText, countPromptTokens, estimateTokens, fnv1a, stableSuffix, stubAssistantText, stubCachedTokens, stubFetchedMarkdown, stubJsonObject, stubReasoningContent, stubSearchResults, stubSignature, stubToolCall, } from "./moonshot-stub.js";
36
+ import { realizeMoonshotRespond } from "./moonshot-scenario.js";
37
+ const SERVICE = 'moonshot';
38
+ /** The base path Moonshot's OpenAI-compatible surface hangs off. The Anthropic-compatible
39
+ * surface hangs off MESSAGES_PREFIX; both are served by the same host. */
40
+ export const MOONSHOT_API_PREFIX = '/v1';
41
+ export const MESSAGES_PREFIX = '/anthropic/v1';
42
+ // ── vendor-shaped errors ──────────────────────────────────────────────────────────────
43
+ /**
44
+ * Moonshot's OpenAI-surface error envelope (ErrorResponse schema): `error.message` REQUIRED,
45
+ * `type` and `code` optional. The `type` strings below are Moonshot's OWN documented error-code
46
+ * page (platform.kimi.ai/docs/api/errors, read 2026-09-16) — a CLOSED published set:
47
+ * 400 invalid_request_error / content_filter
48
+ * 401 invalid_authentication_error / incorrect_api_key_error
49
+ * 403 permission_denied_error
50
+ * 404 resource_not_found_error
51
+ * 429 engine_overloaded_error / exceeded_current_quota_error / rate_limit_reached_error
52
+ * 499 client_closed_request
53
+ * 500 server_error / unexpected_output
54
+ * 503 server_unavailable
55
+ * 504 timeout_error? — the page names 504 but the twin models no timeout path; see the 429/503
56
+ * helpers below for the ones it does.
57
+ */
58
+ function errBody(type, message, code) {
59
+ return { error: { message, ...(type !== undefined ? { type } : {}), ...(code !== undefined ? { code } : {}) } };
60
+ }
61
+ function invalidRequest(message, code) {
62
+ return { status: 400, body: errBody('invalid_request_error', message, code) };
63
+ }
64
+ function notFound(message) {
65
+ return { status: 404, body: errBody('resource_not_found_error', message) };
66
+ }
67
+ function authError(message, type) {
68
+ return { status: 401, body: errBody(type, message) };
69
+ }
70
+ // ── modeled authentication (401) ────────────────────────────────────────────────────────
71
+ // Real Moonshot requires a credential on every request and returns 401 when it is missing or
72
+ // invalid (platform.kimi.ai/docs/api/errors: 401 = invalid_authentication_error when the key
73
+ // is absent/malformed, incorrect_api_key_error when the key is wrong). The twin can't validate
74
+ // against real keys, so it models the CHECKABLE failures: a missing credential →
75
+ // invalid_authentication_error, and a reserved sentinel ('sk_invalid'/'invalid') for the
76
+ // wrong-key path → incorrect_api_key_error. Any other non-empty key is accepted. Trusted
77
+ // in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated; both real
78
+ // clients always send a key → they pass.
79
+ //
80
+ // THE KEY FOLLOWS THE PREFIX: the Anthropic-compatible surface's documented client is the
81
+ // unmodified `@anthropic-ai/sdk`, which authenticates with `x-api-key` (never a bearer) — a
82
+ // bearer-only check made the whole /anthropic surface 401-dead for it. On MESSAGES_PREFIX the
83
+ // key candidates and the error envelope are Anthropic's own grammar (x-api-key first,
84
+ // `authentication_error` in the {type:'error',error:{…}} envelope), exactly as the anthropic
85
+ // pack's checkAuth reads them; on /v1 the bearer leads and Moonshot's own error types apply.
86
+ function checkAuth(req) {
87
+ const onMessages = onMessagesPath(req.path);
88
+ const auth = req.headers?.['authorization'];
89
+ const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
90
+ const xApiKey = typeof req.headers?.['x-api-key'] === 'string' ? req.headers['x-api-key'].trim() : '';
91
+ const key = (req.apiKey ?? '').trim() || (onMessages ? xApiKey || bearer : bearer || xApiKey);
92
+ if (!key) {
93
+ return onMessages
94
+ ? messagesError(401, 'authentication_error', 'missing API key. Provide an x-api-key header (or Authorization: Bearer …).')
95
+ : authError('The API key is missing or malformed. Please check your API key.', 'invalid_authentication_error');
96
+ }
97
+ if (key === 'sk_invalid' || key === 'invalid') {
98
+ return onMessages
99
+ ? messagesError(401, 'authentication_error', 'invalid x-api-key.')
100
+ : authError('The API key is invalid. Please check your API key.', 'incorrect_api_key_error');
101
+ }
102
+ return null;
103
+ }
104
+ /** True when the request targets the Anthropic-compatible surface — the prefix that owns its
105
+ * OWN error envelope, auth header and error-type vocabulary (see the two-prefixes note). */
106
+ function onMessagesPath(path) {
107
+ const bare = path.split('?')[0] ?? '';
108
+ return bare === MESSAGES_PREFIX || bare.startsWith(`${MESSAGES_PREFIX}/`);
109
+ }
110
+ // ── modeled rate limiting (429) ────────────────────────────────────────────────────────
111
+ // Non-deterministic in production, so the twin exposes a DETERMINISTIC opt-in trigger:
112
+ // `x-twin-force-rate-limit: 1` returns the faithful 429 envelope plus Moonshot's own documented
113
+ // header family (platform.kimi.ai/docs/pricing/limits: a 429 carries X-RateLimit-Limit /
114
+ // X-RateLimit-Remaining / X-RateLimit-Reset) and `retry-after`. The Tier-0 figures are the
115
+ // LOWEST published row (RPM 3, TPM 500,000) — the same grounding the budget declaration uses.
116
+ function rateLimitError() {
117
+ return {
118
+ status: 429,
119
+ body: errBody('rate_limit_reached_error', 'Request rate limit reached: 3 requests per minute (Tier 0). Please retry after 20 seconds.'),
120
+ headers: {
121
+ 'retry-after': '20',
122
+ 'x-ratelimit-limit': '3',
123
+ 'x-ratelimit-remaining': '0',
124
+ 'x-ratelimit-reset': '20s',
125
+ },
126
+ };
127
+ }
128
+ /** Moonshot documents 503 as `server_unavailable` (platform.kimi.ai/docs/api/errors). The twin
129
+ * exposes it as a deterministic trigger so a caller can script the vendor's outage shape. */
130
+ function serverUnavailable() {
131
+ return { status: 503, body: errBody('server_unavailable', 'The server is overloaded or not ready to handle the request. Please try again later.') };
132
+ }
133
+ function triggered(req, header) {
134
+ const v = req.headers?.[header];
135
+ return v === '1' || v === 'true';
136
+ }
137
+ function nowEpoch(occurredAt) {
138
+ return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
139
+ }
140
+ // ── kernel helpers ──────────────────────────────────────────────────────────────────────
141
+ function rows(type, root) {
142
+ return projectResources(SERVICE, root).filter((r) => r.type === type);
143
+ }
144
+ /**
145
+ * Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
146
+ * projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
147
+ * gap above the count (ADDING_A_TWIN.md §5). Two further properties matter:
148
+ * • the `_twin_` infix namespaces LOCAL mints, so a pulled Moonshot id can never be matched by
149
+ * this regex and therefore can never be re-minted;
150
+ * • the scan includes TOMBSTONED rows (a soft-deleted file keeps its projection row), so the
151
+ * counter RATCHETS across delete→recreate and a deleted id is never handed out twice.
152
+ */
153
+ function nextId(type, prefix, root) {
154
+ let max = 0;
155
+ for (const r of rows(type, root)) {
156
+ const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(String(r.id));
157
+ if (m)
158
+ max = Math.max(max, Number(m[1]));
159
+ }
160
+ return `${prefix}_twin_${max + 1}`;
161
+ }
162
+ /** Models observed by a connector pull (mapModel), reshaped into the served model object. */
163
+ function pulledModels(root) {
164
+ return rows('model', root)
165
+ .filter((r) => !r._deleted)
166
+ .map((r) => ({ id: r.id, object: 'model', created: r.created, owned_by: r.owned_by }));
167
+ }
168
+ /** The catalog a request sees: the static table, with any PULLED row of the same id OVERRIDING it. */
169
+ function servedModels(root) {
170
+ const pulled = pulledModels(root);
171
+ const byId = new Map();
172
+ for (const m of MOONSHOT_MODELS)
173
+ byId.set(m.id, { id: m.id, object: 'model', created: m.created, owned_by: m.owned_by });
174
+ for (const m of pulled)
175
+ byId.set(String(m.id), m);
176
+ return [...byId.values()];
177
+ }
178
+ function getRow(type, id, root) {
179
+ return rows(type, root).find((r) => r.id === id);
180
+ }
181
+ /** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
182
+ function strip(r) {
183
+ const out = {};
184
+ for (const [k, v] of Object.entries(r)) {
185
+ if (k === 'type' || k === 'updatedAt' || k.startsWith('_'))
186
+ continue;
187
+ out[k] = v;
188
+ }
189
+ return out;
190
+ }
191
+ // ── request parsing ─────────────────────────────────────────────────────────────────────
192
+ function parseJson(body) {
193
+ if (!body || !body.trim())
194
+ return {};
195
+ try {
196
+ const v = JSON.parse(body);
197
+ return v && typeof v === 'object' ? v : {};
198
+ }
199
+ catch {
200
+ return {};
201
+ }
202
+ }
203
+ /** Moonshot's documented tool-name regex (ToolDefinition / MessagesTool schemas). */
204
+ const TOOL_NAME_RE = /^[a-zA-Z_][a-zA-Z0-9-_]{0,127}$/;
205
+ /** `stop`: "A maximum of 5 strings is allowed, and each string must not exceed 32 bytes"
206
+ * (ChatRequestBase.stop description). */
207
+ const STOP_MAX_ITEMS = 5;
208
+ const STOP_MAX_BYTES = 32;
209
+ /** Validate `thinking` per model. Returns the parsed value or an error envelope. */
210
+ function validateThinking(model, raw) {
211
+ if (raw === undefined || raw === null)
212
+ return {};
213
+ if (typeof raw !== 'object' || Array.isArray(raw))
214
+ return { error: invalidRequest("'thinking' must be an object") };
215
+ const o = raw;
216
+ const type = o.type;
217
+ if (typeof type !== 'string')
218
+ return { error: invalidRequest("'thinking.type' is required") };
219
+ if (model === 'kimi-k2.6') {
220
+ if (!K26_THINKING_TYPES.includes(type)) {
221
+ return { error: invalidRequest(`'thinking.type' must be one of ${K26_THINKING_TYPES.map((t) => `'${t}'`).join(', ')} for kimi-k2.6`) };
222
+ }
223
+ }
224
+ else if (model === 'kimi-k2.7-code' || model === 'kimi-k2.7-code-highspeed') {
225
+ // Moonshot's OpenAPI: "For kimi-k2.7-code, only `\"enabled\"` is accepted; passing
226
+ // `\"disabled\"` returns an error. This differs from kimi-k2.6."
227
+ if (!K27_THINKING_TYPES.includes(type)) {
228
+ return { error: invalidRequest(`'thinking.type' must be 'enabled' for ${model} — 'disabled' is not supported and returns an error`) };
229
+ }
230
+ }
231
+ else {
232
+ return { error: invalidRequest(`'thinking' is not a parameter of ${model}`) };
233
+ }
234
+ let keep;
235
+ if (o.keep !== undefined) {
236
+ if (o.keep !== null && o.keep !== 'all') {
237
+ // kimi-k2.7-code: only "all" (or null/omitted) is valid; any other value errors.
238
+ // kimi-k2.6: same closed set {all, null}.
239
+ return { error: invalidRequest(`'thinking.keep' must be 'all' or null`) };
240
+ }
241
+ keep = o.keep;
242
+ }
243
+ return { value: { type: type, ...(keep !== undefined ? { keep } : {}) } };
244
+ }
245
+ function validateChat(params) {
246
+ if (params.model === undefined || params.model === '')
247
+ return { error: invalidRequest("'model' is a required property") };
248
+ if (typeof params.model !== 'string')
249
+ return { error: invalidRequest("'model' must be a string") };
250
+ const model = findModel(params.model);
251
+ if (!model)
252
+ return { error: invalidRequest(`The model '${params.model}' does not exist or you do not have access to it.`) };
253
+ if (!Array.isArray(params.messages))
254
+ return { error: invalidRequest("'messages' is a required property") };
255
+ if (params.messages.length === 0)
256
+ return { error: invalidRequest("[] is too short - 'messages'") };
257
+ const messages = params.messages;
258
+ for (const m of messages) {
259
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
260
+ return { error: invalidRequest("each message must have a valid 'role'") };
261
+ }
262
+ if (!['system', 'user', 'assistant', 'tool'].includes(m.role)) {
263
+ return { error: invalidRequest(`'${m.role}' is not one of ['system', 'user', 'assistant', 'tool']`) };
264
+ }
265
+ }
266
+ // kimi-k3's dynamic tool loading message: role 'system', `tools` present, NO content.
267
+ // Any OTHER message with `tools` and no content is malformed.
268
+ for (const m of messages) {
269
+ const hasTools = Array.isArray(m.tools);
270
+ if (m.role === 'system' && hasTools && m.content === undefined)
271
+ continue; // the dynamic-tool shape
272
+ if (hasTools && m.content === undefined) {
273
+ return { error: invalidRequest("a dynamic tool message must use the 'system' role") };
274
+ }
275
+ }
276
+ // Vision: image_url parts are only valid on a vision model (kimi-k3, kimi-k2.6).
277
+ if (!model.supports.vision) {
278
+ for (const m of messages) {
279
+ const parts = m.content;
280
+ if (Array.isArray(parts) && parts.some((p) => p?.type === 'image_url')) {
281
+ return { error: invalidRequest(`The model '${params.model}' does not support image input`) };
282
+ }
283
+ }
284
+ }
285
+ // logprobs: Moonshot's OpenAPI ACCEPTS logprobs (boolean) + top_logprobs (0..20) — unlike
286
+ // Groq, this is real surface. The twin models acceptance (a logprobs echo on the choice) only
287
+ // as far as the shape goes: `top_logprobs` without `logprobs:true` contradicts the documented
288
+ // coupling ("logprobs must be set to true when this parameter is used").
289
+ const logprobs = params.logprobs === true;
290
+ let topLogprobs;
291
+ if (params.top_logprobs !== undefined && params.top_logprobs !== null) {
292
+ const n = Number(params.top_logprobs);
293
+ if (!Number.isInteger(n) || n < 0 || n > 20)
294
+ return { error: invalidRequest("'top_logprobs' must be an integer between 0 and 20") };
295
+ if (!logprobs)
296
+ return { error: invalidRequest("'logprobs' must be set to true when 'top_logprobs' is used") };
297
+ topLogprobs = n;
298
+ }
299
+ const maxRaw = params.max_completion_tokens ?? params.max_tokens;
300
+ let maxTokens;
301
+ if (maxRaw !== undefined) {
302
+ maxTokens = Number(maxRaw);
303
+ if (!Number.isInteger(maxTokens) || maxTokens < 1)
304
+ return { error: invalidRequest("'max_completion_tokens' must be an integer >= 1") };
305
+ // "If input plus max_completion_tokens exceeds the model context window, the API returns
306
+ // invalid_request_error" (ChatRequestCommon).
307
+ const promptTokens = countPromptTokens(messages);
308
+ if (promptTokens + maxTokens > model.context_length) {
309
+ return { error: invalidRequest(`'max_completion_tokens' plus input tokens (${promptTokens + maxTokens}) exceeds the model context window (${model.context_length})`) };
310
+ }
311
+ }
312
+ let stop;
313
+ if (params.stop !== undefined && params.stop !== null) {
314
+ if (typeof params.stop === 'string')
315
+ stop = [params.stop];
316
+ else if (Array.isArray(params.stop))
317
+ stop = params.stop;
318
+ else
319
+ return { error: invalidRequest("'stop' must be a string or an array of strings") };
320
+ if (stop.length > STOP_MAX_ITEMS)
321
+ return { error: invalidRequest(`'stop' must not exceed ${STOP_MAX_ITEMS} strings`) };
322
+ for (const s of stop) {
323
+ if (typeof s !== 'string')
324
+ return { error: invalidRequest("'stop' must be a string or an array of strings") };
325
+ if (Buffer.byteLength(s, 'utf8') > STOP_MAX_BYTES)
326
+ return { error: invalidRequest(`'stop' entries must not exceed ${STOP_MAX_BYTES} bytes`) };
327
+ }
328
+ }
329
+ let toolChoice;
330
+ const tcRaw = params.tool_choice;
331
+ if (tcRaw !== undefined && tcRaw !== null) {
332
+ if (typeof tcRaw === 'string') {
333
+ if (!['auto', 'none', 'required'].includes(tcRaw))
334
+ return { error: invalidRequest("'tool_choice' must be one of 'none', 'auto', 'required' or a named function") };
335
+ toolChoice = tcRaw;
336
+ }
337
+ else if (typeof tcRaw === 'object') {
338
+ const name = tcRaw.function?.name;
339
+ if (typeof name !== 'string' || !name)
340
+ return { error: invalidRequest("'tool_choice.function.name' is required for a named tool choice") };
341
+ toolChoice = { name };
342
+ }
343
+ }
344
+ let responseFormat = { kind: 'text' };
345
+ const rf = params.response_format;
346
+ if (rf && typeof rf === 'object') {
347
+ if (rf.type === 'json_object')
348
+ responseFormat = { kind: 'json_object' };
349
+ else if (rf.type === 'json_schema') {
350
+ // json_schema REQUIRES the json_schema object (name + schema required by the OpenAPI).
351
+ const js = rf.json_schema;
352
+ if (!js || typeof js !== 'object')
353
+ return { error: invalidRequest("'response_format.json_schema' is required when 'response_format.type' is 'json_schema'") };
354
+ if (typeof js.name !== 'string' || !js.name)
355
+ return { error: invalidRequest("'response_format.json_schema.name' is required") };
356
+ if (!js.schema || typeof js.schema !== 'object')
357
+ return { error: invalidRequest("'response_format.json_schema.schema' is required") };
358
+ responseFormat = { kind: 'json_schema', schema: js.schema };
359
+ }
360
+ else if (rf.type !== undefined && rf.type !== 'text')
361
+ return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
362
+ }
363
+ // Tools: validate the documented name regex on every provided function tool.
364
+ if (params.tools !== undefined) {
365
+ if (!Array.isArray(params.tools))
366
+ return { error: invalidRequest("'tools' must be an array") };
367
+ for (const t of params.tools) {
368
+ const name = t?.function?.name;
369
+ if (typeof name !== 'string' || !TOOL_NAME_RE.test(name)) {
370
+ return { error: invalidRequest(`'tools[].function.name' must match ${TOOL_NAME_RE.source}`) };
371
+ }
372
+ }
373
+ }
374
+ // kimi-k3's reasoning_effort: a CLOSED set, default 'max'.
375
+ let reasoningEffort;
376
+ if (params.reasoning_effort !== undefined && params.reasoning_effort !== null) {
377
+ if (params.model !== 'kimi-k3')
378
+ return { error: invalidRequest(`'reasoning_effort' is not a parameter of ${params.model}`) };
379
+ if (typeof params.reasoning_effort !== 'string' || !REASONING_EFFORTS.includes(params.reasoning_effort)) {
380
+ return { error: invalidRequest(`'reasoning_effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) };
381
+ }
382
+ reasoningEffort = params.reasoning_effort;
383
+ }
384
+ const thinking = validateThinking(params.model, params.thinking);
385
+ if (thinking.error)
386
+ return { error: thinking.error };
387
+ const streamOptionsRaw = params.stream_options;
388
+ if (streamOptionsRaw !== undefined && (typeof streamOptionsRaw !== 'object' || streamOptionsRaw === null)) {
389
+ return { error: invalidRequest("'stream_options' must be an object") };
390
+ }
391
+ return {
392
+ args: {
393
+ model: params.model,
394
+ messages,
395
+ ...(params.tools !== undefined ? { tools: params.tools } : {}),
396
+ n: 1,
397
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
398
+ ...(stop !== undefined ? { stop } : {}),
399
+ stream: params.stream === true,
400
+ ...(streamOptionsRaw?.include_usage === true ? { streamOptions: { includeUsage: true } } : {}),
401
+ ...(toolChoice !== undefined ? { toolChoice } : {}),
402
+ responseFormat,
403
+ logprobs,
404
+ ...(topLogprobs !== undefined ? { topLogprobs } : {}),
405
+ ...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
406
+ ...(thinking.value !== undefined ? { thinking: thinking.value } : {}),
407
+ ...(typeof params.prompt_cache_key === 'string' ? { promptCacheKey: params.prompt_cache_key } : {}),
408
+ },
409
+ };
410
+ }
411
+ /** Build ONE deterministic stub choice. Thinking mode follows the model/params: kimi-k3 always
412
+ * thinks; k2.6 thinks unless `thinking.type:'disabled'`; k2.7-code always thinks. */
413
+ function buildChoice(args, idx) {
414
+ const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
415
+ const forbidTools = args.toolChoice === 'none';
416
+ const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
417
+ if (hasTools && !forbidTools) {
418
+ const list = args.tools;
419
+ const calls = [];
420
+ if (forcedName) {
421
+ const tc = stubToolCall(args.tools, idx + 1, forcedName);
422
+ if (tc)
423
+ calls.push(tc);
424
+ }
425
+ else {
426
+ for (let t = 0; t < list.length; t++) {
427
+ const tc = stubToolCall([list[t]], idx * 100 + t + 1);
428
+ if (tc)
429
+ calls.push(tc);
430
+ }
431
+ }
432
+ if (calls.length) {
433
+ return {
434
+ choice: { index: idx, message: { role: 'assistant', content: null, tool_calls: calls }, finish_reason: 'tool_calls' },
435
+ completionTokens: estimateTokens(JSON.stringify(calls)),
436
+ };
437
+ }
438
+ }
439
+ let text = args.responseFormat.kind === 'json_object'
440
+ ? stubJsonObject(args.messages, args.model)
441
+ : args.responseFormat.kind === 'json_schema'
442
+ ? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
443
+ : stubAssistantText(args.messages, args.model);
444
+ let finish = 'stop';
445
+ // Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
446
+ let stopAt = -1;
447
+ for (const s of args.stop ?? []) {
448
+ if (!s)
449
+ continue;
450
+ const i = text.indexOf(s);
451
+ if (i >= 0 && (stopAt < 0 || i < stopAt))
452
+ stopAt = i;
453
+ }
454
+ if (stopAt >= 0)
455
+ text = text.slice(0, stopAt);
456
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
457
+ text = text.slice(0, args.maxTokens * 4);
458
+ finish = 'length';
459
+ }
460
+ const message = { role: 'assistant', content: text };
461
+ // Thinking mode: kimi-k3 ALWAYS (Preserved Thinking); k2.7-code ALWAYS (type only accepts
462
+ // 'enabled'); k2.6 when thinking.type is not 'disabled'.
463
+ const thinkingOn = args.model === 'kimi-k3'
464
+ || args.model === 'kimi-k2.7-code' || args.model === 'kimi-k2.7-code-highspeed'
465
+ || (args.model === 'kimi-k2.6' && args.thinking?.type !== 'disabled');
466
+ if (thinkingOn) {
467
+ message.reasoning_content = stubReasoningContent(args.messages, args.model);
468
+ // The cap applies to EVERYTHING the model emits — reasoning is output tokens at the vendor
469
+ // too — so truncate it as well and let completion_tokens count only what was returned.
470
+ // finish_reason stays 'length': the cap is what truncated, whichever field hit it.
471
+ if (args.maxTokens !== undefined) {
472
+ const textTokens = estimateTokens(String(message.content ?? ''));
473
+ if (textTokens >= args.maxTokens) {
474
+ delete message.reasoning_content;
475
+ }
476
+ else {
477
+ const budget = args.maxTokens - textTokens;
478
+ const reasoning = message.reasoning_content;
479
+ if (estimateTokens(reasoning) > budget)
480
+ message.reasoning_content = reasoning.slice(0, budget * 4);
481
+ }
482
+ }
483
+ }
484
+ return {
485
+ choice: { index: idx, message, finish_reason: finish },
486
+ completionTokens: estimateTokens(String(message.content ?? '')) + estimateTokens(message.reasoning_content ?? ''),
487
+ };
488
+ }
489
+ export function buildChatCompletion(args, occurredAt, decision) {
490
+ const promptTokens = countPromptTokens(args.messages);
491
+ let scripted = null;
492
+ let missTeach = '';
493
+ if (decision) {
494
+ // The route served the request through the engine (R15: serve() honors the handler's fault
495
+ // before any content exists); this realizer only sees the content decision.
496
+ if (decision.kind === 'handler') {
497
+ const respond = decision.respond;
498
+ // A scripted FAILURE short-circuits into Moonshot's own error envelope + status.
499
+ if (respond.error)
500
+ return scriptedError(respond.error);
501
+ scripted = realizeMoonshotRespond(respond);
502
+ }
503
+ else {
504
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/moonshot.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
505
+ }
506
+ }
507
+ const choices = [];
508
+ let completionTokens = 0;
509
+ for (let i = 0; i < args.n; i++) {
510
+ if (scripted) {
511
+ const message = scripted.toolCalls.length
512
+ ? { role: 'assistant', content: scripted.text, tool_calls: scripted.toolCalls }
513
+ : { role: 'assistant', content: scripted.text ?? '' };
514
+ if (scripted.reasoning !== null)
515
+ message.reasoning_content = scripted.reasoning;
516
+ choices.push({ index: i, message, finish_reason: scripted.finishReason });
517
+ completionTokens += estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
518
+ continue;
519
+ }
520
+ const { choice, completionTokens: ct } = buildChoice(args, i);
521
+ if (missTeach && typeof choice.message.content === 'string')
522
+ choice.message.content += missTeach;
523
+ choices.push(choice);
524
+ completionTokens += ct;
525
+ }
526
+ const usage = buildChatUsage(promptTokens, completionTokens, args.promptCacheKey);
527
+ const id = `chatcmpl-twin-${stableSuffix(args.messages, args.model)}`;
528
+ return {
529
+ id,
530
+ object: 'chat.completion',
531
+ created: nowEpoch(occurredAt),
532
+ model: args.model,
533
+ choices,
534
+ usage,
535
+ };
536
+ }
537
+ /** Map a scripted scenario failure onto Moonshot's real status + envelope. */
538
+ function scriptedError(err) {
539
+ if (err.type === 'rate_limit_reached_error') {
540
+ const base = rateLimitError();
541
+ return err.message ? { ...base, body: errBody('rate_limit_reached_error', err.message) } : base;
542
+ }
543
+ if (err.type === 'server_unavailable')
544
+ return err.message ? { ...serverUnavailable(), body: errBody('server_unavailable', err.message) } : serverUnavailable();
545
+ return { status: 500, body: errBody('server_error', err.message ?? 'Internal Server Error') };
546
+ }
547
+ const isEnvelope = (v) => typeof v.status === 'number' && 'body' in v;
548
+ /** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
549
+ function chunkText(text) {
550
+ if (!text)
551
+ return [];
552
+ const out = [];
553
+ for (let i = 0; i < text.length; i += 20)
554
+ out.push(text.slice(i, i + 20));
555
+ return out;
556
+ }
557
+ /**
558
+ * Emit the vendor-faithful Moonshot streaming sequence into the injected sink (NO sockets, NO
559
+ * setTimeout). Moonshot's order: a first chunk with `delta:{role:'assistant'}`, then
560
+ * `delta:{content}` / `delta:{reasoning_content}` / tool_calls deltas, then a chunk carrying
561
+ * `finish_reason`, then — because Moonshot's ChatCompletionChunk.usage is "Object in the final
562
+ * chunk with usage, null in ordinary chunks" (the OpenAPI's own wording, the first-party
563
+ * denominator; the prose chat doc mentions the chunk under stream_options.include_usage, so the
564
+ * two sources disagree and the OpenAPI governs) — a FINAL chunk whose `choices:[]` and `usage`
565
+ * hold the whole usage object, then `data: [DONE]`. `stream_options` is ACCEPTED (validated as
566
+ * an object) but the tail is not gated on it. Deterministic + synchronous.
567
+ */
568
+ export function streamChat(args, sink, occurredAt, decision) {
569
+ const built = buildChatCompletion(args, occurredAt, decision);
570
+ if (isEnvelope(built))
571
+ return built;
572
+ const full = built;
573
+ const base = { id: full.id, object: 'chat.completion.chunk', created: full.created, model: full.model };
574
+ for (const choice of full.choices) {
575
+ const idx = choice.index;
576
+ sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, finish_reason: null, usage: null }] } });
577
+ if (choice.message.reasoning_content) {
578
+ sink({ data: { ...base, choices: [{ index: idx, delta: { reasoning_content: choice.message.reasoning_content }, finish_reason: null, usage: null }] } });
579
+ }
580
+ if (choice.message.tool_calls && choice.message.tool_calls.length) {
581
+ choice.message.tool_calls.forEach((tc, tIdx) => {
582
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, finish_reason: null, usage: null }] } });
583
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, finish_reason: null, usage: null }] } });
584
+ });
585
+ }
586
+ else {
587
+ for (const piece of chunkText(choice.message.content ?? '')) {
588
+ sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, finish_reason: null, usage: null }] } });
589
+ }
590
+ }
591
+ sink({ data: { ...base, choices: [{ index: idx, delta: {}, finish_reason: choice.finish_reason, usage: null }] } });
592
+ }
593
+ // Moonshot's usage tail: a final chunk with an EMPTY choices array carrying the usage object.
594
+ sink({ data: { ...base, choices: [], finish_reason: null, usage: full.usage } });
595
+ sink({ done: true });
596
+ return full;
597
+ }
598
+ // ── /v1/tokenizers/estimate-token-count (stateless, deterministic) ──────────────────────
599
+ function handleEstimateTokens(params) {
600
+ if (params.model === undefined || typeof params.model !== 'string' || !params.model)
601
+ return invalidRequest("'model' is a required property");
602
+ if (!findModel(params.model))
603
+ return invalidRequest(`The model '${params.model}' does not exist or you do not have access to it.`);
604
+ if (!Array.isArray(params.messages))
605
+ return { status: 400, body: errBody('invalid_request_error', "'messages' is a required property") };
606
+ const messages = params.messages;
607
+ for (const m of messages) {
608
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string')
609
+ return invalidRequest("each message must have a valid 'role'");
610
+ // The OpenAPI: "content must not be empty".
611
+ if (m.content === undefined || m.content === null || (typeof m.content === 'string' && m.content === '')) {
612
+ return invalidRequest("'messages[].content' must not be empty");
613
+ }
614
+ }
615
+ return { status: 200, body: { data: { total_tokens: countPromptTokens(messages) } } };
616
+ }
617
+ // ── /v1/signatures/verify (stateless, deterministic) ────────────────────────────────────
618
+ // Moonshot's request-signing contract: a model call carries `X-Msh-Request-Nonce`; the response
619
+ // carries `Msh-Request-Timestamp` (unix ms) and `Msh-Request-Signature` (`reqsigv1_<opaque>`).
620
+ // POST /v1/signatures/verify checks a (nonce, timestamp, model, signature) tuple. The twin has
621
+ // no real signing key, so it models the CHECKABLE part: a well-formed tuple whose signature
622
+ // carries the documented `reqsigv1_` prefix AND whose signature recomputes exactly under the
623
+ // twin's own deterministic scheme (messagesSignature over nonce/timestamp/model) answers
624
+ // `valid:true`; anything else answers `valid:false`. The twin ALSO emits the two Msh- headers
625
+ // on its own streaming model responses WHEN the request carried a nonce (see the server) so the
626
+ // round-trip is exercisable end to end.
627
+ function handleSignatureVerify(params, req) {
628
+ for (const field of ['nonce', 'timestamp', 'model', 'signature']) {
629
+ if (params[field] === undefined || params[field] === null || params[field] === '') {
630
+ return invalidRequest(`'${field}' is a required property`);
631
+ }
632
+ }
633
+ if (typeof params.nonce !== 'string' || typeof params.model !== 'string' || typeof params.signature !== 'string') {
634
+ return invalidRequest("'nonce', 'model' and 'signature' must be strings");
635
+ }
636
+ const ts = Number(params.timestamp);
637
+ if (!Number.isFinite(ts) || ts < 1)
638
+ return invalidRequest("'timestamp' must be a positive integer");
639
+ const sig = String(params.signature);
640
+ if (!sig.startsWith('reqsigv1_'))
641
+ return { status: 200, body: { valid: false } };
642
+ // The twin's own signatures are derived from the tuple (see messagesSignature); a signature
643
+ // that recomputes exactly answers valid:true.
644
+ const expected = messagesSignature(String(params.nonce), ts, String(params.model));
645
+ return { status: 200, body: { valid: sig === expected } };
646
+ }
647
+ /** The deterministic signature the twin issues on inference responses and verifies here. */
648
+ export function messagesSignature(nonce, timestamp, model) {
649
+ return `reqsigv1_twin_${fnv1a(`${nonce}|${timestamp}|${model}`).toString(36)}`;
650
+ }
651
+ // ── Web-search tools (stateless, deterministic labeled stubs) ───────────────────────────
652
+ const TIME_WINDOW_RE = /^\d{4}(-\d{2})?(-\d{2})?$/;
653
+ function validateTimeWindow(raw) {
654
+ if (raw === undefined)
655
+ return null;
656
+ if (typeof raw !== 'object' || raw === null)
657
+ return invalidRequest("'time_window' must be an object");
658
+ const o = raw;
659
+ for (const k of ['start', 'end']) {
660
+ if (o[k] !== undefined && (typeof o[k] !== 'string' || !TIME_WINDOW_RE.test(o[k]))) {
661
+ return invalidRequest(`'time_window.${k}' must be in YYYY, YYYY-MM or YYYY-MM-DD format`);
662
+ }
663
+ }
664
+ return null;
665
+ }
666
+ function handleToolsSearch(params, pro) {
667
+ if (typeof params.text_query !== 'string' || !params.text_query)
668
+ return invalidRequest("'text_query' is a required property and must not be empty");
669
+ if (params.limit !== undefined) {
670
+ const n = Number(params.limit);
671
+ if (!Number.isInteger(n) || n < 1 || n > 20)
672
+ return invalidRequest("'limit' must be an integer between 1 and 20");
673
+ }
674
+ if (params.timeout_seconds !== undefined) {
675
+ const n = Number(params.timeout_seconds);
676
+ if (!Number.isInteger(n) || n < 1 || n > 60)
677
+ return invalidRequest("'timeout_seconds' must be an integer between 1 and 60");
678
+ }
679
+ if (pro) {
680
+ if (params.sites !== undefined) {
681
+ if (!Array.isArray(params.sites) || params.sites.length > 5)
682
+ return invalidRequest("'sites' must be an array of at most 5 strings");
683
+ for (const s of params.sites) {
684
+ if (typeof s !== 'string' || !s || /\s|\(|\)/.test(s))
685
+ return invalidRequest("'sites' entries must be non-empty and contain no whitespace or parentheses");
686
+ }
687
+ }
688
+ const tw = validateTimeWindow(params.time_window);
689
+ if (tw)
690
+ return tw;
691
+ }
692
+ else {
693
+ // search (not search_pro) takes include_content; search_pro does not declare it.
694
+ if (params.include_content !== undefined && typeof params.include_content !== 'boolean') {
695
+ return invalidRequest("'include_content' must be a boolean");
696
+ }
697
+ }
698
+ const limit = params.limit === undefined ? 5 : Number(params.limit);
699
+ // `include_content: true` SERVES content (the option's documented meaning): the results carry
700
+ // a labeled deterministic page stub. Default (false) is the vendor's bare shape — `text: ''`.
701
+ // A no-op 200 that accepted the flag and always served '' was the round-two finding.
702
+ const includeContent = pro ? false : params.include_content === true;
703
+ const results = stubSearchResults(String(params.text_query), limit, pro, includeContent);
704
+ return { status: 200, body: { search_results: results } };
705
+ }
706
+ function handleToolsFetch(params) {
707
+ if (typeof params.url !== 'string' || !params.url)
708
+ return invalidRequest("'url' is a required property");
709
+ if (!/^https?:\/\//.test(params.url))
710
+ return invalidRequest("'url' must be an http or https URL");
711
+ return { status: 200, body: stubFetchedMarkdown(params.url) };
712
+ }
713
+ // ── Files (stateful) ────────────────────────────────────────────────────────────────────
714
+ /** FileObject.purpose — a CLOSED documented set: file-extract, image, video, batch. */
715
+ const FILE_CREATE_PURPOSES = new Set(['file-extract', 'image', 'video', 'batch']);
716
+ async function createFile(params, req) {
717
+ const purpose = String(params.purpose ?? '');
718
+ if (!purpose)
719
+ return invalidRequest("'purpose' is a required property");
720
+ if (!FILE_CREATE_PURPOSES.has(purpose))
721
+ return invalidRequest(`'purpose' must be one of ${[...FILE_CREATE_PURPOSES].map((p) => `'${p}'`).join(', ')} (got '${purpose}')`);
722
+ // The multipart adapter marks a form with no `file` part (§9 round two, F3): the vendor's
723
+ // Upload File requires the file body, so the twin refuses rather than minting an empty
724
+ // 'ready' file.
725
+ if (params._multipart_missing_file === true)
726
+ return invalidRequest("the multipart form carries no 'file' part");
727
+ const filename = String(params.filename ?? params.file ?? 'upload');
728
+ const content = typeof params.content === 'string' ? params.content : '';
729
+ // `binary_content` is the SERVER's multipart-adapter marker (moonshot-server.ts): the JSON
730
+ // contract cannot carry raw bytes, so an uploaded binary travels base64-encoded under it and
731
+ // is stored with the marker; GET /content decodes before serving. A JSON-door create (plain
732
+ // text) stores the text as-is with no marker. A caller forging the marker through the JSON
733
+ // door gets strict validation: the content must actually be base64 (§9 round two, F6).
734
+ const binary = params.binary_content === true;
735
+ if (binary && !/^[A-Za-z0-9+/]*={0,2}$/.test(content) || (binary && content.length % 4 !== 0)) {
736
+ return invalidRequest("'binary_content' content must be base64-encoded");
737
+ }
738
+ // `bytes` is the vendor's byte count — a text file's UTF-8 length, not the JS string's
739
+ // UTF-16 code-unit count (a §9-round-two finding: the two doors disagreed on the same field;
740
+ // the multipart door already counted real bytes).
741
+ const bytes = typeof params.bytes === 'number' ? params.bytes : binary ? Buffer.from(content, 'base64').length : Buffer.byteLength(content, 'utf8');
742
+ const id = nextId('file', 'file', req.root);
743
+ await applyTwinWrite(SERVICE, {
744
+ operation: 'file.create',
745
+ subjectType: 'file',
746
+ subjectId: id,
747
+ fields: { object: 'file', bytes, created_at: nowEpoch(req.occurredAt), filename, purpose, status: 'ready', _content: content, ...(binary ? { _content_encoding: 'base64' } : {}) },
748
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
749
+ actor: { kind: 'agent' },
750
+ }, req.root);
751
+ const row = getRow('file', id, req.root);
752
+ return { status: 200, body: fileView(row ?? {}) };
753
+ }
754
+ function fileView(r) {
755
+ const s = strip(r);
756
+ return { id: r.id, object: 'file', bytes: s.bytes, created_at: s.created_at, filename: s.filename, purpose: s.purpose, status: s.status ?? 'ready' };
757
+ }
758
+ // ── Batches (stateful) ──────────────────────────────────────────────────────────────────
759
+ /** BatchCreateRequest.endpoint — a CLOSED documented set: only /v1/chat/completions. */
760
+ const BATCH_ENDPOINTS = new Set(['/v1/chat/completions']);
761
+ /** "supports formats like 12h, 1d, 3d, minimum 12h, maximum 7d" (BatchCreateRequest). */
762
+ const COMPLETION_WINDOW = /^(\d+)(h|d)$/;
763
+ function completionWindowHours(w) {
764
+ const m = COMPLETION_WINDOW.exec(w);
765
+ if (!m)
766
+ return null;
767
+ const n = Number(m[1]);
768
+ return m[2] === 'd' ? n * 24 : n;
769
+ }
770
+ async function createBatch(params, req) {
771
+ const inputFileId = params.input_file_id;
772
+ if (typeof inputFileId !== 'string' || !inputFileId)
773
+ return invalidRequest("'input_file_id' is a required property");
774
+ const endpoint = params.endpoint;
775
+ if (typeof endpoint !== 'string' || !BATCH_ENDPOINTS.has(endpoint)) {
776
+ return invalidRequest(`'endpoint' must be one of ${[...BATCH_ENDPOINTS].map((e) => `'${e}'`).join(', ')}`);
777
+ }
778
+ const window = params.completion_window;
779
+ if (typeof window !== 'string')
780
+ return invalidRequest("'completion_window' is a required property");
781
+ const hours = completionWindowHours(window);
782
+ if (hours === null || hours < 12 || hours > 168)
783
+ return invalidRequest("'completion_window' must be a duration from '12h' to '7d'");
784
+ if (params.metadata !== undefined) {
785
+ const md = params.metadata;
786
+ if (!md || typeof md !== 'object' || Array.isArray(md))
787
+ return invalidRequest("'metadata' must be an object");
788
+ const entries = Object.entries(md);
789
+ if (entries.length > 16)
790
+ return invalidRequest("'metadata' must not exceed 16 key-value pairs");
791
+ for (const [k, v] of entries) {
792
+ if (k.length > 64)
793
+ return invalidRequest("'metadata' keys must not exceed 64 characters");
794
+ if (typeof v !== 'string' || v.length > 512)
795
+ return invalidRequest("'metadata' values must be strings of at most 512 characters");
796
+ }
797
+ }
798
+ const file = getRow('file', inputFileId, req.root);
799
+ if (!file || file._deleted)
800
+ return notFound(`No such File object: ${inputFileId}`);
801
+ if (file.purpose !== 'batch')
802
+ return invalidRequest(`File ${inputFileId} must have purpose 'batch' (has '${String(file.purpose)}')`);
803
+ const created = nowEpoch(req.occurredAt);
804
+ const id = nextId('batch', 'batch', req.root);
805
+ await applyTwinWrite(SERVICE, {
806
+ operation: 'batch.create',
807
+ subjectType: 'batch',
808
+ subjectId: id,
809
+ fields: {
810
+ object: 'batch',
811
+ endpoint,
812
+ input_file_id: inputFileId,
813
+ completion_window: window,
814
+ status: 'validating',
815
+ output_file_id: null,
816
+ error_file_id: null,
817
+ created_at: created,
818
+ in_progress_at: null,
819
+ expires_at: created + hours * 3600,
820
+ finalizing_at: null,
821
+ completed_at: null,
822
+ failed_at: null,
823
+ cancelling_at: null,
824
+ cancelled_at: null,
825
+ request_counts: { total: 0, completed: 0, failed: 0 },
826
+ metadata: params.metadata ?? null,
827
+ },
828
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
829
+ actor: { kind: 'agent' },
830
+ }, req.root);
831
+ return { status: 200, body: batchView(getRow('batch', id, req.root) ?? {}) };
832
+ }
833
+ function batchView(r) {
834
+ return { id: r.id, ...strip(r) };
835
+ }
836
+ // ── Balance (stateful) ──────────────────────────────────────────────────────────────────
837
+ // GET /v1/users/me/balance returns Moonshot's own envelope ({code, data:{available_balance,
838
+ // voucher_balance, cash_balance}, scode, status}) — NOT the OpenAI envelope. The balance is
839
+ // STATEFUL: the connector's pull can fold a real account's balance into the log (type 'balance',
840
+ // id 'me'), and the served figure is that row when present, else a deterministic positive stub.
841
+ function balanceView(root) {
842
+ const row = getRow('balance', 'me', root);
843
+ const b = row && !row._deleted
844
+ ? { available_balance: Number(row.available_balance), voucher_balance: Number(row.voucher_balance), cash_balance: Number(row.cash_balance) }
845
+ : { available_balance: 49.58894, voucher_balance: 46.58893, cash_balance: 3.00001 };
846
+ return {
847
+ status: 200,
848
+ body: { code: 0, data: b, scode: '0x0', status: true },
849
+ };
850
+ }
851
+ /** Extract the tool name from an Anthropic Messages tool definition ({type:'custom', name, …}). */
852
+ function messagesToolName(t) {
853
+ const o = t;
854
+ return typeof o?.name === 'string' ? o.name : '';
855
+ }
856
+ /** Validate the Anthropic-compatible request. kimi-k3 only; max_tokens REQUIRED. */
857
+ function validateMessages(params) {
858
+ if (params.model === undefined || params.model === '')
859
+ return { error: messagesError(400, 'invalid_request_error', "'model' is a required property") };
860
+ // MessagesRequest['model'] enum (kimi-k3 only) — read from the catalog's own supports flag.
861
+ if (!findModel(String(params.model))?.supports.messages) {
862
+ return { error: messagesError(400, 'invalid_request_error', `The endpoint currently supports 'kimi-k3' only (got '${String(params.model)}')`) };
863
+ }
864
+ if (!Array.isArray(params.messages))
865
+ return { error: messagesError(400, 'invalid_request_error', "'messages' is a required property") };
866
+ if (params.messages.length === 0)
867
+ return { error: messagesError(400, 'invalid_request_error', "'messages' must not be empty") };
868
+ const messages = [];
869
+ for (const m of params.messages) {
870
+ if (!m || typeof m !== 'object')
871
+ return { error: messagesError(400, 'invalid_request_error', 'each message must be an object') };
872
+ if (m.role !== 'user' && m.role !== 'assistant') {
873
+ // Anthropic's grammar: the top-level `system` field owns the system prompt.
874
+ return { error: messagesError(400, 'invalid_request_error', `'${String(m.role)}' is not one of ['user', 'assistant'] — use the top-level 'system' field for the system prompt`) };
875
+ }
876
+ messages.push({ role: m.role, content: m.content });
877
+ }
878
+ if (params.max_tokens === undefined || params.max_tokens === null)
879
+ return { error: messagesError(400, 'invalid_request_error', "'max_tokens' is a required property") };
880
+ const maxTokens = Number(params.max_tokens);
881
+ if (!Number.isInteger(maxTokens) || maxTokens < 1)
882
+ return { error: messagesError(400, 'invalid_request_error', "'max_tokens' must be an integer >= 1") };
883
+ let system;
884
+ if (params.system !== undefined) {
885
+ if (typeof params.system === 'string')
886
+ system = params.system;
887
+ else if (Array.isArray(params.system))
888
+ system = params.system.map((b) => String(b?.text ?? '')).join('\n');
889
+ else
890
+ return { error: messagesError(400, 'invalid_request_error', "'system' must be a string or an array of text blocks") };
891
+ }
892
+ let stopSequences;
893
+ if (params.stop_sequences !== undefined) {
894
+ if (!Array.isArray(params.stop_sequences))
895
+ return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' must be an array of strings") };
896
+ if (params.stop_sequences.length > 5)
897
+ return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' must not exceed 5 entries") };
898
+ for (const s of params.stop_sequences) {
899
+ if (typeof s !== 'string')
900
+ return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' entries must be strings") };
901
+ if (Buffer.byteLength(s, 'utf8') > 32)
902
+ return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' entries must not exceed 32 bytes") };
903
+ }
904
+ stopSequences = params.stop_sequences;
905
+ }
906
+ // Tools: Anthropic's MessagesTool shape, with the SAME documented name regex the /v1 surface
907
+ // enforces (a bad tool name is a 400 here too — never a silent acceptance the stub then
908
+ // ignores).
909
+ let tools;
910
+ if (params.tools !== undefined) {
911
+ if (!Array.isArray(params.tools))
912
+ return { error: messagesError(400, 'invalid_request_error', "'tools' must be an array") };
913
+ for (const t of params.tools) {
914
+ if (!t || typeof t !== 'object' || Array.isArray(t))
915
+ return { error: messagesError(400, 'invalid_request_error', 'each tool must be an object') };
916
+ const name = messagesToolName(t);
917
+ if (!name || !TOOL_NAME_RE.test(name)) {
918
+ return { error: messagesError(400, 'invalid_request_error', `'tools[].name' must match ${TOOL_NAME_RE.source}`) };
919
+ }
920
+ const schema = t.input_schema;
921
+ if (schema !== undefined && (!schema || typeof schema !== 'object' || Array.isArray(schema))) {
922
+ return { error: messagesError(400, 'invalid_request_error', "'tools[].input_schema' must be an object") };
923
+ }
924
+ }
925
+ tools = params.tools;
926
+ }
927
+ // tool_choice: Anthropic's closed type set INCLUDING type:'tool' with its required `name`
928
+ // (the sibling pack's rule: a named choice must name a tool that is actually in `tools`).
929
+ let toolChoice;
930
+ if (params.tool_choice !== undefined) {
931
+ const t = params.tool_choice?.type;
932
+ if (typeof t !== 'string' || !['auto', 'any', 'tool', 'none'].includes(t))
933
+ return { error: messagesError(400, 'invalid_request_error', "'tool_choice.type' must be one of 'auto', 'any', 'tool', 'none'") };
934
+ // A tool_choice WITHOUT tools is a contradiction — Anthropic's own surface refuses it.
935
+ if (!tools)
936
+ return { error: messagesError(400, 'invalid_request_error', "'tool_choice' requires 'tools' to be set") };
937
+ if (t === 'tool') {
938
+ const name = params.tool_choice.name;
939
+ if (typeof name !== 'string' || !tools.some((tool) => messagesToolName(tool) === name)) {
940
+ return { error: messagesError(400, 'invalid_request_error', `'tool_choice' names tool '${typeof name === 'string' ? name : String(name)}', which is not in tools`) };
941
+ }
942
+ toolChoice = { type: 'tool', name };
943
+ }
944
+ else {
945
+ toolChoice = { type: t };
946
+ }
947
+ }
948
+ let outputEffort;
949
+ const oc = params.output_config;
950
+ if (oc && typeof oc === 'object' && oc.effort !== undefined) {
951
+ if (typeof oc.effort !== 'string' || !REASONING_EFFORTS.includes(oc.effort)) {
952
+ return { error: messagesError(400, 'invalid_request_error', `'output_config.effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) };
953
+ }
954
+ outputEffort = oc.effort;
955
+ }
956
+ return {
957
+ args: {
958
+ model: String(params.model),
959
+ messages,
960
+ maxTokens,
961
+ ...(system !== undefined ? { system } : {}),
962
+ ...(stopSequences !== undefined ? { stopSequences } : {}),
963
+ stream: params.stream === true,
964
+ ...(tools !== undefined ? { tools } : {}),
965
+ ...(toolChoice !== undefined ? { toolChoice } : {}),
966
+ ...(outputEffort !== undefined ? { outputEffort } : {}),
967
+ },
968
+ };
969
+ }
970
+ /** The Messages surface's OWN error envelope (MessagesErrorResponse schema) — different from /v1. */
971
+ function messagesError(status, type, message) {
972
+ return { status, body: { type: 'error', error: { type, message } } };
973
+ }
974
+ /** Re-wear a scripted fault's Moonshot-shaped body in the Anthropic envelope for the /anthropic
975
+ * surface: the fault result carries the Moonshot error body the adapter rendered; this surface
976
+ * serves {type:'error', error:{type,message}} with the equivalent Anthropic type. */
977
+ function messagesErrorBody(status, moonshotBody) {
978
+ const e = moonshotBody?.error;
979
+ const message = e?.message ?? `The request was refused by a scripted fault (${status}).`;
980
+ const type = e?.type === 'rate_limit_reached_error' ? 'rate_limit_error'
981
+ : e?.type === 'server_unavailable' || e?.type === 'server_error' ? 'api_error'
982
+ : 'invalid_request_error';
983
+ return { type: 'error', error: { type, message } };
984
+ }
985
+ export function buildMessagesResponse(args, occurredAt, decision) {
986
+ let scripted = null;
987
+ let missTeach = '';
988
+ if (decision) {
989
+ // The route served the request through the engine (R15); tools/thinking rode into the
990
+ // match there — a handler keyed on `hasTool`/`toolResultFor` fires on this surface exactly
991
+ // as it fires on /v1 (the sibling pack wires the same request features through on its
992
+ // Messages surface).
993
+ if (decision.kind === 'handler') {
994
+ const respond = decision.respond;
995
+ if (respond.error)
996
+ return scriptedMessagesError(respond.error);
997
+ scripted = realizeMoonshotRespond(respond);
998
+ }
999
+ else {
1000
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in handlers/moonshot.json.]`;
1001
+ }
1002
+ }
1003
+ const inputTokens = countPromptTokens(args.messages.map((m) => ({ role: m.role, content: m.content })));
1004
+ let content;
1005
+ let stopReason;
1006
+ let outputTokens;
1007
+ if (scripted) {
1008
+ content = [];
1009
+ if (scripted.reasoning !== null)
1010
+ content.push({ type: 'thinking', thinking: scripted.reasoning, signature: stubSignature(args.messages, args.model) });
1011
+ if (scripted.text)
1012
+ content.push({ type: 'text', text: scripted.text + (missTeach && scripted.text ? missTeach : '') });
1013
+ for (const tc of scripted.toolCalls) {
1014
+ content.push({ type: 'tool_use', id: tc.id, name: tc.function.name, input: JSON.parse(tc.function.arguments) });
1015
+ }
1016
+ stopReason = scripted.finishReason === 'tool_calls' ? 'tool_use' : scripted.finishReason === 'length' ? 'max_tokens' : 'end_turn';
1017
+ }
1018
+ else {
1019
+ // Tools on the Messages surface are REAL surface: tool_choice 'any'/'auto' (with tools
1020
+ // present) yields a tool_use block the SDK's own decoder reads; a named 'tool' choice calls
1021
+ // THAT tool; 'none' suppresses. The tautology that silently answered text regardless (the
1022
+ // round-two finding) is gone.
1023
+ const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
1024
+ const choice = args.toolChoice?.type ?? 'auto';
1025
+ const suppress = choice === 'none';
1026
+ const forcedName = choice === 'tool' ? args.toolChoice?.name : undefined;
1027
+ // A tool_result turn is ANSWERED, not re-tooled: once the client returns the tool_result the
1028
+ // model's next turn is text with stop_reason end_turn — deciding tool_use from tools.length
1029
+ // alone made an agent loop emit tool_use forever (the round-three finding). An EXPLICIT
1030
+ // forced choice ('any'/'tool') still calls, because the caller demanded one.
1031
+ const last = args.messages[args.messages.length - 1];
1032
+ const toolResultTurn = last?.role === 'user' && Array.isArray(last.content)
1033
+ && last.content.some((b) => b?.type === 'tool_result');
1034
+ const forceTool = choice === 'any' || choice === 'tool';
1035
+ const toolCalls = hasTools && !suppress && (forceTool || !toolResultTurn)
1036
+ ? [stubToolCall(args.tools, 1, forcedName)].filter((tc) => tc !== null)
1037
+ : [];
1038
+ // kimi-k3 ALWAYS thinks (Preserved Thinking — the same rule the /v1 surface applies), and
1039
+ // Moonshot's Messages doc has NO thinking parameter: reasoning is tuned through
1040
+ // output_config.effort (default max) and the content array is documented
1041
+ // "ordered thinking → text → tool_use" (platform.kimi.ai/docs/api/messages, read
1042
+ // 2026-09-16) — so the thinking block leads unconditionally, like the vendor's.
1043
+ content = [
1044
+ { type: 'thinking', thinking: stubReasoningContent(args.messages.map((m) => ({ role: m.role, content: m.content })), args.model), signature: stubSignature(args.messages, args.model) },
1045
+ ];
1046
+ if (toolCalls.length) {
1047
+ toolCalls.forEach((tc, seq) => {
1048
+ // Anthropic's tool_use id prefix is `toolu_` (the sibling pack's grammar —
1049
+ // `toolu_twin_${seq}`), NOT the `call_` prefix the OpenAI-compatible /v1 surface uses.
1050
+ content.push({ type: 'tool_use', id: `toolu_twin_${seq + 1}`, name: tc.function.name, input: JSON.parse(tc.function.arguments) });
1051
+ });
1052
+ }
1053
+ else {
1054
+ content.push({ type: 'text', text: stubAssistantText(args.messages.map((m) => ({ role: m.role, content: m.content })), args.model) + (missTeach || '') });
1055
+ }
1056
+ stopReason = toolCalls.length ? 'tool_use' : 'end_turn';
1057
+ }
1058
+ // max_tokens caps the OUTPUT ITSELF, not merely the report: truncate the emitted blocks to the
1059
+ // cap and count usage off what remains (the vendor stops emitting at the cap — usage.output_tokens
1060
+ // is what was emitted, never a fabricated agreement with a text block it did not produce).
1061
+ // Tokens are counted per-block off the block's own payload (thinking/text/tool input), NOT off
1062
+ // JSON.stringify of the whole array — JSON scaffolding would keep a capped answer above its cap.
1063
+ const blockTokens = (b) => b.type === 'thinking' ? estimateTokens(b.thinking)
1064
+ : b.type === 'text' ? estimateTokens(b.text)
1065
+ : estimateTokens(JSON.stringify(b.input));
1066
+ const totalTokens = () => content.reduce((n, b) => n + blockTokens(b), 0);
1067
+ if (args.maxTokens !== undefined && totalTokens() > args.maxTokens) {
1068
+ // THE CAP WALKS THE BLOCKS IN ORDER (the sibling pack's method): the emitted content is a
1069
+ // PREFIX of what the stub would have said. Each block is kept only as far as the remaining
1070
+ // budget allows — text and thinking truncate mid-string, a tool_use that does not fit ends
1071
+ // the turn (everything after it is never emitted). Anthropic NEVER returns a zero-block
1072
+ // assistant message: the FIRST block is kept truncated to at least one token, so
1073
+ // output_tokens > 0 and content is non-empty at any cap >= 1. The walk is BOUNDED — one
1074
+ // pass, one truncation — so no estimate can spin it.
1075
+ let spent = 0;
1076
+ let cut = false;
1077
+ for (let i = 0; i < content.length; i++) {
1078
+ const block = content[i];
1079
+ const cost = blockTokens(block);
1080
+ if (spent + cost <= args.maxTokens) {
1081
+ spent += cost;
1082
+ continue;
1083
+ }
1084
+ const budget = args.maxTokens - spent;
1085
+ cut = true;
1086
+ if ((block.type === 'text' || block.type === 'thinking') && (budget >= 1 || i === 0)) {
1087
+ // Truncate to the remaining budget (~4 chars/token), keeping at least one token so the
1088
+ // message never loses its last block — the floor a zero-block message is forbidden by.
1089
+ const keep = Math.max(1, budget);
1090
+ if (block.type === 'text')
1091
+ block.text = block.text.slice(0, keep * 4);
1092
+ else
1093
+ block.thinking = block.thinking.slice(0, keep * 4);
1094
+ spent += blockTokens(block);
1095
+ content.length = i + 1;
1096
+ break;
1097
+ }
1098
+ // The budget ran out exactly at this block's boundary, or the block is a tool_use that
1099
+ // cannot be cut: the block was never emitted. A retained whole block here shipped 57
1100
+ // tokens under max_tokens 23 (round-four review) — the cap is a prefix, never a rounding.
1101
+ content.length = i;
1102
+ if (content.length === 0)
1103
+ content.push({ type: 'text', text: '' });
1104
+ break;
1105
+ }
1106
+ outputTokens = spent;
1107
+ if (cut)
1108
+ stopReason = 'max_tokens';
1109
+ }
1110
+ else {
1111
+ outputTokens = totalTokens();
1112
+ }
1113
+ // A stop_sequence match truncates the text at the hit and reports stop_reason 'stop_sequence'
1114
+ // with the matched string (Anthropic's documented shape — 'end_turn' would misreport WHY the
1115
+ // turn ended). The RECOUNT happens AFTER the truncation: usage.output_tokens is what was
1116
+ // emitted, so a 113-char answer reported as 54 tokens (counted before the cut) contradicted
1117
+ // its own usage (the round-three finding).
1118
+ let stopSequence = null;
1119
+ if (args.stopSequences?.length) {
1120
+ const textBlock = content.find((b) => b.type === 'text');
1121
+ if (textBlock && textBlock.type === 'text') {
1122
+ for (const s of args.stopSequences) {
1123
+ const i = textBlock.text.indexOf(s);
1124
+ if (i >= 0) {
1125
+ textBlock.text = textBlock.text.slice(0, i);
1126
+ stopSequence = s;
1127
+ stopReason = 'stop_sequence';
1128
+ break;
1129
+ }
1130
+ }
1131
+ outputTokens = totalTokens();
1132
+ }
1133
+ }
1134
+ return {
1135
+ id: `msg_twin_${stableSuffix(args.messages, args.model)}`,
1136
+ type: 'message',
1137
+ role: 'assistant',
1138
+ model: args.model,
1139
+ content,
1140
+ stop_reason: stopReason,
1141
+ stop_sequence: stopSequence,
1142
+ usage: {
1143
+ input_tokens: inputTokens,
1144
+ output_tokens: outputTokens,
1145
+ cache_read_input_tokens: stubCachedTokens(inputTokens),
1146
+ cache_creation_input_tokens: 0,
1147
+ },
1148
+ };
1149
+ }
1150
+ function scriptedMessagesError(err) {
1151
+ if (err.type === 'rate_limit_reached_error')
1152
+ return messagesError(429, 'rate_limit_reached_error', err.message ?? 'Request rate limit reached');
1153
+ if (err.type === 'server_unavailable')
1154
+ return messagesError(503, 'server_unavailable', err.message ?? 'The server is overloaded');
1155
+ return messagesError(500, 'server_error', err.message ?? 'Internal Server Error');
1156
+ }
1157
+ /**
1158
+ * Emit the Anthropic-compatible streaming grammar into the injected sink: message_start →
1159
+ * ping → content_block_start(thinking) → thinking_delta → signature_delta → content_block_stop →
1160
+ * content_block_start(text) → text_delta → content_block_stop → message_delta(stop_reason) →
1161
+ * message_stop. No [DONE] sentinel — the stream ends after message_stop.
1162
+ *
1163
+ * message_start carries the EMPTY message (content: [], stop_reason: null, output_tokens small)
1164
+ * and the blocks stream in one at a time — the real grammar (and the sibling pack's frames). A
1165
+ * message_start carrying the COMPLETE final message made the SDK's own accumulator double every
1166
+ * block: it pushes one block per content_block_start ON TOP of what message_start already held
1167
+ * (the round-three BLOCKER).
1168
+ */
1169
+ export function streamMessages(args, sink, occurredAt, decision) {
1170
+ const built = buildMessagesResponse(args, occurredAt, decision);
1171
+ if (isEnvelope(built))
1172
+ return built;
1173
+ const full = built;
1174
+ sink({
1175
+ event: 'message_start',
1176
+ data: {
1177
+ type: 'message_start',
1178
+ message: {
1179
+ id: full.id, type: 'message', role: 'assistant', model: full.model,
1180
+ content: [], stop_reason: null, stop_sequence: null,
1181
+ usage: { input_tokens: full.usage.input_tokens, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 },
1182
+ },
1183
+ },
1184
+ });
1185
+ // The heartbeat the real API interleaves (and the sibling pack emits); the SDK's SSE decoder
1186
+ // skips `event: ping` frames by name, so it is inert to every consumer.
1187
+ sink({ event: 'ping', data: { type: 'ping' } });
1188
+ full.content.forEach((block, index) => {
1189
+ if (block.type === 'thinking') {
1190
+ sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'thinking', thinking: '' } } });
1191
+ for (const piece of chunkText(block.thinking))
1192
+ sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'thinking_delta', thinking: piece } } });
1193
+ sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'signature_delta', signature: block.signature ?? '' } } });
1194
+ sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
1195
+ }
1196
+ else if (block.type === 'text') {
1197
+ sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'text', text: '' } } });
1198
+ for (const piece of chunkText(block.text))
1199
+ sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'text_delta', text: piece } } });
1200
+ sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
1201
+ }
1202
+ else {
1203
+ sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'tool_use', id: block.id, name: block.name, input: {} } } });
1204
+ const json = JSON.stringify(block.input);
1205
+ sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'input_json_delta', partial_json: json } } });
1206
+ sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
1207
+ }
1208
+ });
1209
+ sink({ event: 'message_delta', data: { type: 'message_delta', delta: { stop_reason: full.stop_reason, stop_sequence: full.stop_sequence }, usage: { output_tokens: full.usage.output_tokens } } });
1210
+ sink({ event: 'message_stop', data: { type: 'message_stop' } });
1211
+ sink({ done: true });
1212
+ return full;
1213
+ }
1214
+ function validateResponses(params) {
1215
+ if (params.model === undefined || params.model === '')
1216
+ return { error: invalidRequest("'model' is a required property") };
1217
+ // "This endpoint currently supports `kimi-k3`" (ResponsesRequest.model) — read from the
1218
+ // catalog's own supports flag, so the enum and the catalog cannot drift apart.
1219
+ const responsesModel = findModel(String(params.model));
1220
+ if (!responsesModel?.supports.responses)
1221
+ return { error: invalidRequest(`This endpoint currently supports 'kimi-k3' only (got '${String(params.model)}')`) };
1222
+ if (params.input === undefined || params.input === null)
1223
+ return { error: invalidRequest("'input' is a required property") };
1224
+ if (typeof params.input !== 'string' && !Array.isArray(params.input))
1225
+ return { error: invalidRequest("'input' must be a string or an array of items") };
1226
+ let reasoningEffort;
1227
+ const r = params.reasoning;
1228
+ if (r && typeof r === 'object' && r.effort !== undefined) {
1229
+ if (typeof r.effort !== 'string' || !REASONING_EFFORTS.includes(r.effort)) {
1230
+ return { error: invalidRequest(`'reasoning.effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) };
1231
+ }
1232
+ reasoningEffort = r.effort;
1233
+ }
1234
+ let maxOutputTokens;
1235
+ if (params.max_output_tokens !== undefined) {
1236
+ maxOutputTokens = Number(params.max_output_tokens);
1237
+ if (!Number.isInteger(maxOutputTokens) || maxOutputTokens < 1)
1238
+ return { error: invalidRequest("'max_output_tokens' must be an integer >= 1") };
1239
+ }
1240
+ return {
1241
+ args: {
1242
+ model: String(params.model),
1243
+ input: params.input,
1244
+ ...(typeof params.instructions === 'string' ? { instructions: params.instructions } : {}),
1245
+ ...(maxOutputTokens !== undefined ? { maxOutputTokens } : {}),
1246
+ ...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
1247
+ stream: params.stream === true,
1248
+ },
1249
+ };
1250
+ }
1251
+ function responsesInputText(input) {
1252
+ if (typeof input === 'string')
1253
+ return input;
1254
+ return input.map((item) => {
1255
+ if (item.type === 'message' || item.type === undefined) {
1256
+ const content = item.content;
1257
+ return contentToText(content);
1258
+ }
1259
+ return JSON.stringify(item);
1260
+ }).join('\n');
1261
+ }
1262
+ export function buildResponsesResponse(args, occurredAt, decision) {
1263
+ let scripted = null;
1264
+ let missTeach = '';
1265
+ if (decision) {
1266
+ // The route served the request through the engine (R15); this realizer only sees the
1267
+ // content decision.
1268
+ if (decision.kind === 'handler') {
1269
+ const respond = decision.respond;
1270
+ if (respond.error)
1271
+ return scriptedError(respond.error);
1272
+ scripted = realizeMoonshotRespond(respond);
1273
+ }
1274
+ else {
1275
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in handlers/moonshot.json.]`;
1276
+ }
1277
+ }
1278
+ const inputText = responsesInputText(args.input);
1279
+ const asMessages = [{ role: 'user', content: inputText }];
1280
+ const inputTokens = countPromptTokens(asMessages);
1281
+ const output = [];
1282
+ let outputTokens = 0;
1283
+ if (scripted) {
1284
+ if (scripted.reasoning !== null) {
1285
+ output.push({ type: 'reasoning', id: `rs_twin_${stableSuffix(args.input, 'reasoning')}`, summary: [{ type: 'summary_text', text: scripted.reasoning }], status: 'completed' });
1286
+ outputTokens += estimateTokens(scripted.reasoning);
1287
+ }
1288
+ if (scripted.toolCalls.length) {
1289
+ for (const tc of scripted.toolCalls) {
1290
+ output.push({ type: 'function_call', id: `fc_twin_${stableSuffix(tc)}`, call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments, status: 'completed' });
1291
+ outputTokens += estimateTokens(tc.function.arguments);
1292
+ }
1293
+ }
1294
+ else {
1295
+ const text = (scripted.text ?? '') + (missTeach || '');
1296
+ output.push({ type: 'message', id: `msg_twin_${stableSuffix(args.input, 'message')}`, role: 'assistant', status: 'completed', content: [{ type: 'output_text', text, annotations: [] }] });
1297
+ outputTokens += estimateTokens(text);
1298
+ }
1299
+ }
1300
+ else {
1301
+ output.push({ type: 'reasoning', id: `rs_twin_${stableSuffix(inputText, 'reasoning')}`, summary: [{ type: 'summary_text', text: stubReasoningContent(asMessages, args.model) }], status: 'completed' });
1302
+ const text = stubAssistantText(asMessages, args.model) + (missTeach || '');
1303
+ output.push({ type: 'message', id: `msg_twin_${stableSuffix(inputText, 'message')}`, role: 'assistant', status: 'completed', content: [{ type: 'output_text', text, annotations: [] }] });
1304
+ outputTokens = estimateTokens(stubReasoningContent(asMessages, args.model)) + estimateTokens(text);
1305
+ }
1306
+ let status = 'completed';
1307
+ let incompleteDetails = null;
1308
+ // max_output_tokens caps the OUTPUT ITSELF: the vendor stops emitting at the cap, so truncate
1309
+ // the emitted items (dropping whole items from the tail — the Responses API emits complete
1310
+ // items) and let usage.output_tokens count what remains. Reporting status 'incomplete' with
1311
+ // usage.output_tokens still above the cap was the round-two finding: the envelope contradicted
1312
+ // its own usage.
1313
+ if (args.maxOutputTokens !== undefined && outputTokens > args.maxOutputTokens) {
1314
+ // THE CAP IS A PREFIX, LIKE THE MESSAGES SURFACE'S: items are emitted in order until the
1315
+ // budget is exhausted; a text-carrying item truncates to the remaining budget rather than
1316
+ // vanishing, so a tight cap still yields a non-empty output with output_tokens > 0 — a
1317
+ // zero-item `output: []` (the round-three finding at max_output_tokens 2) is a message the
1318
+ // vendor never sends. One bounded pass; nothing after the cut is emitted.
1319
+ let spent = 0;
1320
+ let cut = false;
1321
+ for (let i = 0; i < output.length; i++) {
1322
+ const item = output[i];
1323
+ const cost = estimateTokens(item.type === 'message' ? item.content.map((c) => c.text ?? '').join('')
1324
+ : item.type === 'reasoning' ? item.summary.map((s) => s.text ?? '').join('')
1325
+ : item.type === 'function_call' ? item.arguments : item.type === 'custom_tool_call' ? item.input : '');
1326
+ if (spent + cost <= args.maxOutputTokens) {
1327
+ spent += cost;
1328
+ continue;
1329
+ }
1330
+ const budget = args.maxOutputTokens - spent;
1331
+ const keep = Math.max(1, budget) * 4;
1332
+ if ((item.type === 'message' || item.type === 'reasoning') && budget >= 1) {
1333
+ if (item.type === 'message')
1334
+ item.content[0].text = (item.content[0].text ?? '').slice(0, keep);
1335
+ else
1336
+ item.summary[0].text = (item.summary[0].text ?? '').slice(0, keep);
1337
+ spent += estimateTokens(item.type === 'message' ? item.content[0].text ?? '' : item.summary[0].text ?? '');
1338
+ }
1339
+ output.length = i + 1;
1340
+ cut = true;
1341
+ break;
1342
+ }
1343
+ outputTokens = spent;
1344
+ if (cut) {
1345
+ status = 'incomplete';
1346
+ incompleteDetails = { reason: 'max_output_tokens' };
1347
+ }
1348
+ }
1349
+ return {
1350
+ id: `resp_twin_${stableSuffix(inputText, args.model)}`,
1351
+ object: 'response',
1352
+ created_at: nowEpoch(occurredAt),
1353
+ completed_at: status === 'completed' || status === 'incomplete' ? nowEpoch(occurredAt) : null,
1354
+ status,
1355
+ model: args.model,
1356
+ output,
1357
+ usage: {
1358
+ input_tokens: inputTokens,
1359
+ input_tokens_details: { cached_tokens: stubCachedTokens(inputTokens), cache_write_tokens: 0 },
1360
+ output_tokens: outputTokens,
1361
+ // reasoning_tokens counts the REASONING items only (the /anthropic thinking_tokens path
1362
+ // does the same): reporting the whole output_tokens as reasoning claimed every text token
1363
+ // was reasoning (the round-three finding).
1364
+ output_tokens_details: {
1365
+ reasoning_tokens: output
1366
+ .filter((o) => o.type === 'reasoning')
1367
+ .reduce((n, o) => n + estimateTokens(o.summary.map((s) => s.text ?? '').join('')), 0),
1368
+ },
1369
+ total_tokens: inputTokens + outputTokens,
1370
+ },
1371
+ incomplete_details: incompleteDetails,
1372
+ error: null,
1373
+ store: false,
1374
+ };
1375
+ }
1376
+ /**
1377
+ * Emit the Responses SSE grammar into the injected sink: response.created →
1378
+ * response.in_progress → output_item.added/done per item → response.completed. Each frame
1379
+ * carries `event: <type>` and a monotonically increasing `sequence_number` from 0.
1380
+ */
1381
+ export function streamResponses(args, sink, occurredAt, decision) {
1382
+ const built = buildResponsesResponse(args, occurredAt, decision);
1383
+ if (isEnvelope(built))
1384
+ return built;
1385
+ const full = built;
1386
+ let seq = 0;
1387
+ const emit = (type, data) => sink({ event: type, data: { type, sequence_number: seq++, ...data } });
1388
+ emit('response.created', { response: { ...full, status: 'in_progress', output: [], usage: null } });
1389
+ emit('response.in_progress', { response: { ...full, status: 'in_progress', output: [], usage: null } });
1390
+ for (const item of full.output) {
1391
+ emit('response.output_item.added', { output_index: full.output.indexOf(item), output_item: item });
1392
+ emit('response.output_item.done', { output_index: full.output.indexOf(item), output_item: item });
1393
+ }
1394
+ emit(full.status === 'completed' ? 'response.completed' : 'response.incomplete', { response: full });
1395
+ sink({ done: true });
1396
+ return full;
1397
+ }
1398
+ // ── public entry: cross-cutting protocol (auth / rate-limit) then route ─────────────────
1399
+ export async function handleMoonshotTwinRequest(req) {
1400
+ const method = req.method.toUpperCase();
1401
+ // EVERY cross-cutting failure answers in the prefix's OWN envelope: the /anthropic surface's
1402
+ // documented client decodes Anthropic's {type:'error',error:{…}} grammar, so a 401/429/503
1403
+ // there carrying the /v1 envelope would be unparseable to it (the clone's lie, cross-cutting
1404
+ // edition). checkAuth applies this itself; the two deterministic fault triggers follow.
1405
+ const onMessages = onMessagesPath(req.path);
1406
+ if (req.headers !== undefined || req.apiKey !== undefined) {
1407
+ const authErr = checkAuth(req);
1408
+ if (authErr)
1409
+ return authErr;
1410
+ }
1411
+ if (triggered(req, 'x-twin-force-rate-limit')) {
1412
+ const base = rateLimitError();
1413
+ // The HEADERS ride along on the re-enveloped 429: retry-after and the X-RateLimit-* family
1414
+ // are part of the refusal a real client reads, whatever envelope the body wears — dropping
1415
+ // them on /anthropic left the documented SDK blind to the back-off (the round-three finding).
1416
+ return onMessages
1417
+ ? { ...messagesError(429, 'rate_limit_error', base.body.error.message), headers: base.headers }
1418
+ : base;
1419
+ }
1420
+ if (triggered(req, 'x-twin-force-server-unavailable')) {
1421
+ const base = serverUnavailable();
1422
+ return onMessages ? messagesError(503, 'overloaded_error', base.body.error.message) : base;
1423
+ }
1424
+ const res = await routeMoonshot(req, method);
1425
+ // Moonshot's request-signing contract, in the HANDLER so every surface (server, fetch
1426
+ // adapter, in-process callers) signs identically (§9 round two, F8): a STREAMED model call
1427
+ // that carries `X-Msh-Request-Nonce` gets `Msh-Request-Timestamp` + `Msh-Request-Signature`
1428
+ // response headers; a nonce-less call proceeds unsigned. The server forwards these headers
1429
+ // on the SSE response it frames.
1430
+ const nonce = req.headers?.['x-msh-request-nonce'];
1431
+ if (nonce && req.sseSink && req.method.toUpperCase() === 'POST' && (req.path === `${MOONSHOT_API_PREFIX}/chat/completions`)) {
1432
+ const model = (() => { try {
1433
+ return typeof JSON.parse(req.body ?? '').model === 'string' ? JSON.parse(req.body ?? '').model : null;
1434
+ }
1435
+ catch {
1436
+ return null;
1437
+ } })();
1438
+ if (model) {
1439
+ const ts = Date.parse(req.occurredAt ?? worldNow());
1440
+ return { ...res, headers: { ...(res.headers ?? {}), 'msh-request-timestamp': String(ts), 'msh-request-signature': messagesSignature(nonce, ts, model) } };
1441
+ }
1442
+ }
1443
+ return res;
1444
+ }
1445
+ // ── router ──────────────────────────────────────────────────────────────────────────────
1446
+ async function routeMoonshot(req, method) {
1447
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
1448
+ const params = parseJson(req.body);
1449
+ const dec = (s) => decodeURIComponent(s);
1450
+ const onOpenai = path === MOONSHOT_API_PREFIX || path.startsWith(`${MOONSHOT_API_PREFIX}/`);
1451
+ const onMessages = path === MESSAGES_PREFIX || path.startsWith(`${MESSAGES_PREFIX}/`);
1452
+ if (!onOpenai && !onMessages) {
1453
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1454
+ }
1455
+ // D3: a read-only twin rejects any mutation with a vendor-shaped error — in the prefix's OWN
1456
+ // envelope (the Messages surface decodes Anthropic's grammar, not /v1's), so a read-only twin
1457
+ // never hands the anthropic SDK an envelope it cannot parse. Computed AFTER the prefix so the
1458
+ // envelope follows the path, not the call order.
1459
+ if (req.readOnly && method !== 'GET') {
1460
+ const message = 'twin is read-only; omit readOnly to accept writes';
1461
+ return onMessages ? messagesError(405, 'invalid_request_error', message) : { status: 405, body: errBody('invalid_request_error', message) };
1462
+ }
1463
+ // ---- Anthropic-compatible Messages surface ----
1464
+ if (onMessages) {
1465
+ if (path !== `${MESSAGES_PREFIX}/messages`)
1466
+ return messagesError(404, 'not_found_error', `Unknown request URL: ${method} ${path}.`);
1467
+ if (method !== 'POST')
1468
+ return messagesError(405, 'invalid_request_error', `${method} is not supported on /anthropic/v1/messages`);
1469
+ const validated = validateMessages(params);
1470
+ if ('error' in validated)
1471
+ return validated.error;
1472
+ const args = validated.args;
1473
+ // R15 — the scenario engine decides AND honors a fault here, before any message exists:
1474
+ // a `status` fault is this vendor's own refusal envelope re-worn in the Anthropic shape,
1475
+ // a `slow` has already held the answer, a `drop` never returns.
1476
+ let messagesDecision;
1477
+ if (req.scenarioEngine) {
1478
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages.map((m) => ({ role: m.role, content: m.content })), tools: args.tools, thinking: undefined });
1479
+ if (served.kind === 'fault')
1480
+ return { status: served.result.status, body: messagesErrorBody(served.result.status, served.result.body), headers: served.result.headers };
1481
+ messagesDecision = served;
1482
+ }
1483
+ const result = args.stream && req.messagesSseSink
1484
+ ? streamMessages(args, req.messagesSseSink, req.occurredAt, messagesDecision)
1485
+ : buildMessagesResponse(args, req.occurredAt, messagesDecision);
1486
+ if (isEnvelope(result))
1487
+ return result;
1488
+ return { status: 200, body: result };
1489
+ }
1490
+ const seg = path.slice(MOONSHOT_API_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean); // ["chat","completions"]
1491
+ // protocol 2: after a landing adopted Moonshot's id for a file or a batch, a caller on a branch
1492
+ // may still address it by the local id — resolved through the alias map once, here at the boundary
1493
+ if (seg[0] === 'files' && seg[1])
1494
+ seg[1] = resolveSubjectId(SERVICE, 'file', dec(seg[1]), req.root);
1495
+ if (seg[0] === 'batches' && seg[1])
1496
+ seg[1] = resolveSubjectId(SERVICE, 'batch', dec(seg[1]), req.root);
1497
+ // ---- models (static catalog) ----
1498
+ // LIST only: Moonshot's own OpenAPI declares GET /v1/models and NO retrieve-by-id operation
1499
+ // (the fixture's 19 operations have no /v1/models/{model}) — serving a 200 retrieve here was a
1500
+ // 200 for an unmodeled route. An unknown/unmodeled sub-path falls through to the router's
1501
+ // vendor-shaped not-found below.
1502
+ if (seg[0] === 'models' && seg.length === 1 && method === 'GET') {
1503
+ return { status: 200, body: { object: 'list', data: servedModels(req.root) } };
1504
+ }
1505
+ // ---- chat completions (the generative stub; envelope is faithful) ----
1506
+ if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
1507
+ const validated = validateChat(params);
1508
+ if ('error' in validated)
1509
+ return validated.error;
1510
+ const args = validated.args;
1511
+ // R15 — the scenario engine decides AND honors a fault here, before any completion exists:
1512
+ // a `status` fault is this vendor's own refusal envelope, a `slow` has already held the
1513
+ // answer, a `drop` never returns. The realizers below only see a content decision.
1514
+ let chatDecision;
1515
+ if (req.scenarioEngine) {
1516
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages, tools: args.tools, thinking: args.thinking });
1517
+ if (served.kind === 'fault')
1518
+ return { status: served.result.status, body: served.result.body, headers: served.result.headers };
1519
+ chatDecision = served;
1520
+ }
1521
+ const result = args.stream && req.sseSink
1522
+ ? streamChat(args, req.sseSink, req.occurredAt, chatDecision)
1523
+ : buildChatCompletion(args, req.occurredAt, chatDecision);
1524
+ if (isEnvelope(result))
1525
+ return result;
1526
+ return { status: 200, body: result };
1527
+ }
1528
+ // ---- responses (kimi-k3 only) ----
1529
+ if (seg[0] === 'responses' && seg.length === 1 && method === 'POST') {
1530
+ const validated = validateResponses(params);
1531
+ if ('error' in validated)
1532
+ return validated.error;
1533
+ const args = validated.args;
1534
+ // R15 — the same serve-and-honor as chat completions: the Responses door serves the same
1535
+ // scripted brain from the same handlers document.
1536
+ let responsesDecision;
1537
+ if (req.scenarioEngine) {
1538
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: [{ role: 'user', content: responsesInputText(args.input) }], tools: undefined, thinking: undefined });
1539
+ if (served.kind === 'fault')
1540
+ return { status: served.result.status, body: served.result.body, headers: served.result.headers };
1541
+ responsesDecision = served;
1542
+ }
1543
+ const result = args.stream && req.messagesSseSink
1544
+ ? streamResponses(args, req.messagesSseSink, req.occurredAt, responsesDecision)
1545
+ : buildResponsesResponse(args, req.occurredAt, responsesDecision);
1546
+ if (isEnvelope(result))
1547
+ return result;
1548
+ return { status: 200, body: result };
1549
+ }
1550
+ // ---- token counting ----
1551
+ if (seg[0] === 'tokenizers' && seg[1] === 'estimate-token-count' && seg.length === 2 && method === 'POST') {
1552
+ return handleEstimateTokens(params);
1553
+ }
1554
+ // ---- signatures ----
1555
+ if (seg[0] === 'signatures' && seg[1] === 'verify' && seg.length === 2 && method === 'POST') {
1556
+ return handleSignatureVerify(params, req);
1557
+ }
1558
+ // ---- web-search tools ----
1559
+ if (seg[0] === 'tools' && seg[1] === 'search' && seg.length === 2 && method === 'POST')
1560
+ return handleToolsSearch(params, false);
1561
+ if (seg[0] === 'tools' && seg[1] === 'search_pro' && seg.length === 2 && method === 'POST')
1562
+ return handleToolsSearch(params, true);
1563
+ if (seg[0] === 'tools' && seg[1] === 'fetch' && seg.length === 2 && method === 'POST')
1564
+ return handleToolsFetch(params);
1565
+ // ---- balance (Moonshot's own envelope) ----
1566
+ if (seg[0] === 'users' && seg[1] === 'me' && seg[2] === 'balance' && seg.length === 3 && method === 'GET') {
1567
+ return balanceView(req.root);
1568
+ }
1569
+ // ---- files (stateful) ----
1570
+ if (seg[0] === 'files' && seg.length === 1 && method === 'POST')
1571
+ return createFile(params, req);
1572
+ if (seg[0] === 'files' && seg.length === 1 && method === 'GET') {
1573
+ const data = rows('file', req.root).filter((r) => !r._deleted).map(fileView);
1574
+ return { status: 200, body: { object: 'list', data } };
1575
+ }
1576
+ if (seg[0] === 'files' && seg.length === 2 && method === 'GET') {
1577
+ const f = getRow('file', dec(seg[1]), req.root);
1578
+ return f && !f._deleted ? { status: 200, body: fileView(f) } : notFound(`No such File object: ${dec(seg[1])}`);
1579
+ }
1580
+ if (seg[0] === 'files' && seg.length === 3 && seg[2] === 'content' && method === 'GET') {
1581
+ const f = getRow('file', dec(seg[1]), req.root);
1582
+ if (!f || f._deleted)
1583
+ return notFound(`No such File object: ${dec(seg[1])}`);
1584
+ // A PULLED file (the connector's mapFile) carries metadata only — the real list endpoint
1585
+ // returns no content, so the twin holds none. Serving an empty 200 here was a §9-round-two
1586
+ // finding (F2): a fake success for bytes the twin does not have. The honest answer is a
1587
+ // vendor-shaped refusal naming the gap.
1588
+ if (f._content === undefined && Number(f.bytes) > 0) {
1589
+ return { status: 501, body: errBody('server_error', `the twin holds no content for pulled file ${dec(seg[1])} (the vendor's list endpoint does not return file content; re-create the file locally to read it back)`) };
1590
+ }
1591
+ // The server turns the body STRING into bytes; this header tells it how (§9 round two, F1):
1592
+ // a binary upload (the multipart adapter's `binary_content` marker → `_content_encoding`)
1593
+ // is decoded from its stored base64 and carried latin1 (every byte value intact through a
1594
+ // JS string, re-materialized by the server); anything else is TEXT and travels as the
1595
+ // string itself — the server writes it UTF-8, so a non-ASCII text file round-trips.
1596
+ if (f._content_encoding === 'base64') {
1597
+ const raw = Buffer.from(String(f._content ?? ''), 'base64').toString('latin1');
1598
+ return { status: 200, body: raw, headers: { 'content-type': 'application/octet-stream', 'x-twin-content-binary': '1' } };
1599
+ }
1600
+ return { status: 200, body: String(f._content ?? ''), headers: { 'content-type': 'application/octet-stream' } };
1601
+ }
1602
+ if (seg[0] === 'files' && seg.length === 2 && method === 'DELETE') {
1603
+ const fid = dec(seg[1]);
1604
+ const f = getRow('file', fid, req.root);
1605
+ if (!f || f._deleted)
1606
+ return notFound(`No such File object: ${fid}`);
1607
+ await applyTwinWrite(SERVICE, {
1608
+ operation: 'file.delete', subjectType: 'file', subjectId: fid, fields: { _deleted: true, object: 'file' },
1609
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1610
+ }, req.root);
1611
+ return { status: 200, body: { id: fid, object: 'file', deleted: true } };
1612
+ }
1613
+ // ---- batches (stateful) ----
1614
+ if (seg[0] === 'batches' && seg.length === 1 && method === 'POST')
1615
+ return createBatch(params, req);
1616
+ if (seg[0] === 'batches' && seg.length === 1 && method === 'GET') {
1617
+ return { status: 200, body: { object: 'list', data: rows('batch', req.root).map(batchView), has_more: false } };
1618
+ }
1619
+ if (seg[0] === 'batches' && seg.length === 2 && method === 'GET') {
1620
+ const b = getRow('batch', dec(seg[1]), req.root);
1621
+ if (!b)
1622
+ return notFound(`No such Batch object: ${dec(seg[1])}`);
1623
+ return { status: 200, body: batchView(b) };
1624
+ }
1625
+ if (seg[0] === 'batches' && seg.length === 3 && seg[2] === 'cancel' && method === 'POST') {
1626
+ const bid = dec(seg[1]);
1627
+ const b = getRow('batch', bid, req.root);
1628
+ if (!b)
1629
+ return notFound(`No such Batch object: ${bid}`);
1630
+ if (b.status === 'cancelling' || b.status === 'cancelled')
1631
+ return invalidRequest(`Cannot cancel a batch with status '${String(b.status)}'.`);
1632
+ // NOTE: the twin does not simulate the asynchronous cancelling→cancelled settlement (the
1633
+ // groq pack's §9 finding applies identically here: settling on a READ would break the
1634
+ // read-only contract). The terminal transition is filed as
1635
+ // `moonshot.batches.cancellation_settles` (todo).
1636
+ await applyTwinWrite(SERVICE, {
1637
+ operation: 'batch.cancel', subjectType: 'batch', subjectId: bid,
1638
+ fields: { status: 'cancelling', cancelling_at: nowEpoch(req.occurredAt) },
1639
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1640
+ }, req.root);
1641
+ return { status: 200, body: batchView(getRow('batch', bid, req.root) ?? {}) };
1642
+ }
1643
+ // Unmodeled operation → fail like the vendor (never a fake success). A /anthropic path that
1644
+ // is not /messages was already refused in the Messages branch; anything falling through here
1645
+ // is on the OpenAI-compatible surface, whose 404 envelope this is.
1646
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1647
+ }