@volter/twin-moonshot 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +164 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +25 -0
- package/dist/src/index.d.ts +14 -0
- package/dist/src/index.js +86 -0
- package/dist/src/moonshot-budget.d.ts +57 -0
- package/dist/src/moonshot-budget.js +142 -0
- package/dist/src/moonshot-capabilities.d.ts +4 -0
- package/dist/src/moonshot-capabilities.js +1200 -0
- package/dist/src/moonshot-conformance.d.ts +14 -0
- package/dist/src/moonshot-conformance.js +405 -0
- package/dist/src/moonshot-connector.d.ts +168 -0
- package/dist/src/moonshot-connector.js +416 -0
- package/dist/src/moonshot-models.d.ts +36 -0
- package/dist/src/moonshot-models.js +37 -0
- package/dist/src/moonshot-scenario.d.ts +54 -0
- package/dist/src/moonshot-scenario.js +175 -0
- package/dist/src/moonshot-server.d.ts +13 -0
- package/dist/src/moonshot-server.js +202 -0
- package/dist/src/moonshot-stub.d.ts +70 -0
- package/dist/src/moonshot-stub.js +222 -0
- package/dist/src/moonshot-twin.d.ts +144 -0
- package/dist/src/moonshot-twin.js +1647 -0
- package/dist/src/moonshot-types.d.ts +251 -0
- package/dist/src/moonshot-types.js +19 -0
- package/package.json +53 -0
- package/src/cli.ts +25 -0
- package/src/index.ts +129 -0
- package/src/moonshot-budget.ts +163 -0
- package/src/moonshot-capabilities.ts +1220 -0
- package/src/moonshot-conformance.ts +416 -0
- package/src/moonshot-connector.ts +465 -0
- package/src/moonshot-models.ts +89 -0
- package/src/moonshot-scenario.ts +194 -0
- package/src/moonshot-server.ts +220 -0
- package/src/moonshot-stub.ts +230 -0
- package/src/moonshot-twin.ts +1670 -0
- package/src/moonshot-types.ts +225 -0
|
@@ -0,0 +1,1670 @@
|
|
|
1
|
+
// Moonshot twin REQUEST HANDLER — the canonical Moonshot (Kimi) API surface for the twin.
|
|
2
|
+
// Contract: handleMoonshotTwinRequest({method, path, body}) -> {status, body}. It is the faithful
|
|
3
|
+
// Moonshot API the real clients (the standard `openai` SDK pointed at
|
|
4
|
+
// `https://api.moonshot.ai/v1`, the `anthropic` SDK pointed at `https://api.moonshot.ai/anthropic`,
|
|
5
|
+
// and plain HTTP callers) talk to UNMODIFIED — Moonshot ships no SDK of its own; its documented
|
|
6
|
+
// integration path is the standard OpenAI/Anthropic clients with a swapped base URL
|
|
7
|
+
// (platform.kimi.ai/docs/overview, read 2026-09-16).
|
|
8
|
+
//
|
|
9
|
+
// THE HONEST DESIGN: the twin cannot run the model, so the three inference endpoints
|
|
10
|
+
// (`POST /v1/chat/completions`, `POST /v1/responses`, `POST /anthropic/v1/messages`) return a
|
|
11
|
+
// DETERMINISTIC STUB completion (moonshot-stub.ts) clearly labeled a twin stub — it NEVER pretends
|
|
12
|
+
// to be real model output. But the ENTIRE PROTOCOL ENVELOPE is vendor-faithful: all three response
|
|
13
|
+
// shapes, all three streaming grammars (OpenAI chunks, Responses SSE events with
|
|
14
|
+
// `sequence_number`, Anthropic message events), tool_calls / tool_use, finish_reason /
|
|
15
|
+
// stop_reason, `reasoning_content` / thinking blocks, and Moonshot's cache-split usage. The
|
|
16
|
+
// genuinely stateful + static surface is real:
|
|
17
|
+
// • GET /v1/models — static catalog (moonshot-models.ts)
|
|
18
|
+
// • POST/GET/DELETE /v1/files (+ /content) — stateful (kernel action log)
|
|
19
|
+
// • POST/GET /v1/batches (+ /cancel) — stateful
|
|
20
|
+
// • GET /v1/users/me/balance — stateful (a mutable account balance)
|
|
21
|
+
// plus the deterministic stateless helpers: POST /v1/tokenizers/estimate-token-count,
|
|
22
|
+
// POST /v1/signatures/verify, POST /v1/tools/{search,search_pro,fetch}.
|
|
23
|
+
//
|
|
24
|
+
// TWO PATH PREFIXES, BOTH REAL: Moonshot serves its OpenAI-compatible surface under `/v1` and its
|
|
25
|
+
// Anthropic-compatible surface under `/anthropic/v1` on the same host (api.moonshot.ai). The
|
|
26
|
+
// Anthropic surface's error envelope is `{ type:'error', error:{type,message}, request_id? }` —
|
|
27
|
+
// a DIFFERENT envelope from `/v1`'s `{ error: { message, type, code? } }` — and its streaming
|
|
28
|
+
// grammar is Anthropic's event-name SSE, not OpenAI's `data:`-only frames.
|
|
29
|
+
//
|
|
30
|
+
// State lives in the kernel action log (D1): all writes are local actions, reads are the
|
|
31
|
+
// projection. No real Moonshot is ever called from this path (D4). Streaming uses an INJECTED
|
|
32
|
+
// sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
|
|
33
|
+
import { applyTwinWrite, projectResources, resolveSubjectId, type ScenarioDecision, worldNow } from '@volter/world-core';
|
|
34
|
+
import {
|
|
35
|
+
findModel,
|
|
36
|
+
K26_THINKING_TYPES,
|
|
37
|
+
K27_THINKING_TYPES,
|
|
38
|
+
MOONSHOT_MODELS,
|
|
39
|
+
REASONING_EFFORTS,
|
|
40
|
+
type ReasoningEffort,
|
|
41
|
+
} from './moonshot-models.ts';
|
|
42
|
+
import {
|
|
43
|
+
buildChatUsage,
|
|
44
|
+
contentToText,
|
|
45
|
+
countPromptTokens,
|
|
46
|
+
estimateTokens,
|
|
47
|
+
fnv1a,
|
|
48
|
+
lastUserText,
|
|
49
|
+
stableSuffix,
|
|
50
|
+
stubAssistantText,
|
|
51
|
+
stubCachedTokens,
|
|
52
|
+
stubFetchedMarkdown,
|
|
53
|
+
stubJsonObject,
|
|
54
|
+
stubReasoningContent,
|
|
55
|
+
stubSearchResults,
|
|
56
|
+
stubSignature,
|
|
57
|
+
stubToolArguments,
|
|
58
|
+
stubToolCall,
|
|
59
|
+
} from './moonshot-stub.ts';
|
|
60
|
+
import { type MoonshotScenarioEngine, type MoonshotScenarioRespond, realizeMoonshotRespond, type ScriptedResult } from './moonshot-scenario.ts';
|
|
61
|
+
import type {
|
|
62
|
+
MessagesSseEvent,
|
|
63
|
+
MoonshotAssistantMessage,
|
|
64
|
+
MoonshotChatCompletion,
|
|
65
|
+
MoonshotChoice,
|
|
66
|
+
MoonshotMessageParam,
|
|
67
|
+
MoonshotMessagesResponse,
|
|
68
|
+
MoonshotResponsesOutputItem,
|
|
69
|
+
MoonshotResponsesResponse,
|
|
70
|
+
MoonshotToolCall,
|
|
71
|
+
MoonshotUsage,
|
|
72
|
+
SseEvent,
|
|
73
|
+
SseSink,
|
|
74
|
+
} from './moonshot-types.ts';
|
|
75
|
+
|
|
76
|
+
const SERVICE = 'moonshot';
|
|
77
|
+
|
|
78
|
+
/** The base path Moonshot's OpenAI-compatible surface hangs off. The Anthropic-compatible
|
|
79
|
+
* surface hangs off MESSAGES_PREFIX; both are served by the same host. */
|
|
80
|
+
export const MOONSHOT_API_PREFIX = '/v1';
|
|
81
|
+
export const MESSAGES_PREFIX = '/anthropic/v1';
|
|
82
|
+
|
|
83
|
+
export type MoonshotRequest = {
|
|
84
|
+
/** The scenario engine (kernel grammar + this pack's vocabulary) — scripts the three
|
|
85
|
+
* inference endpoints. */
|
|
86
|
+
scenarioEngine?: MoonshotScenarioEngine;
|
|
87
|
+
method: string;
|
|
88
|
+
path: string;
|
|
89
|
+
body?: string;
|
|
90
|
+
occurredAt?: string;
|
|
91
|
+
root?: string;
|
|
92
|
+
readOnly?: boolean;
|
|
93
|
+
/** The credential the caller presents (the bearer `Authorization` header). When a request
|
|
94
|
+
* carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the
|
|
95
|
+
* real vendor rule: a credential is required → 401 on missing/invalid. In-process trusted
|
|
96
|
+
* calls (capability verify, connector) omit BOTH and are not auth-gated. */
|
|
97
|
+
apiKey?: string;
|
|
98
|
+
/** Lower-cased request headers the HTTP server passes through so the handler can model auth
|
|
99
|
+
* (401), the rate-limit trigger (429), and the signature headers the /v1/signatures/verify
|
|
100
|
+
* contract references. */
|
|
101
|
+
headers?: Record<string, string>;
|
|
102
|
+
/** When set on a streaming POST, chunks/events are written here (no sockets). */
|
|
103
|
+
sseSink?: SseSink;
|
|
104
|
+
/** When set on a streaming /anthropic/v1/messages POST, Anthropic-grammar events are written
|
|
105
|
+
* here (event name + data; no sockets). */
|
|
106
|
+
messagesSseSink?: MessagesSseSink;
|
|
107
|
+
};
|
|
108
|
+
|
|
109
|
+
/** The handler response. `headers` (when present) are response headers the HTTP server should
|
|
110
|
+
* set — e.g. `retry-after` + Moonshot's `x-ratelimit-*` family on a modeled 429. */
|
|
111
|
+
export type MoonshotResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
|
|
112
|
+
|
|
113
|
+
// ── vendor-shaped errors ──────────────────────────────────────────────────────────────
|
|
114
|
+
/**
|
|
115
|
+
* Moonshot's OpenAI-surface error envelope (ErrorResponse schema): `error.message` REQUIRED,
|
|
116
|
+
* `type` and `code` optional. The `type` strings below are Moonshot's OWN documented error-code
|
|
117
|
+
* page (platform.kimi.ai/docs/api/errors, read 2026-09-16) — a CLOSED published set:
|
|
118
|
+
* 400 invalid_request_error / content_filter
|
|
119
|
+
* 401 invalid_authentication_error / incorrect_api_key_error
|
|
120
|
+
* 403 permission_denied_error
|
|
121
|
+
* 404 resource_not_found_error
|
|
122
|
+
* 429 engine_overloaded_error / exceeded_current_quota_error / rate_limit_reached_error
|
|
123
|
+
* 499 client_closed_request
|
|
124
|
+
* 500 server_error / unexpected_output
|
|
125
|
+
* 503 server_unavailable
|
|
126
|
+
* 504 timeout_error? — the page names 504 but the twin models no timeout path; see the 429/503
|
|
127
|
+
* helpers below for the ones it does.
|
|
128
|
+
*/
|
|
129
|
+
function errBody(type: string, message: string, code?: string) {
|
|
130
|
+
return { error: { message, ...(type !== undefined ? { type } : {}), ...(code !== undefined ? { code } : {}) } };
|
|
131
|
+
}
|
|
132
|
+
function invalidRequest(message: string, code?: string): MoonshotResponseEnvelope {
|
|
133
|
+
return { status: 400, body: errBody('invalid_request_error', message, code) };
|
|
134
|
+
}
|
|
135
|
+
function notFound(message: string): MoonshotResponseEnvelope {
|
|
136
|
+
return { status: 404, body: errBody('resource_not_found_error', message) };
|
|
137
|
+
}
|
|
138
|
+
function authError(message: string, type: string): MoonshotResponseEnvelope {
|
|
139
|
+
return { status: 401, body: errBody(type, message) };
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
// ── modeled authentication (401) ────────────────────────────────────────────────────────
|
|
143
|
+
// Real Moonshot requires a credential on every request and returns 401 when it is missing or
|
|
144
|
+
// invalid (platform.kimi.ai/docs/api/errors: 401 = invalid_authentication_error when the key
|
|
145
|
+
// is absent/malformed, incorrect_api_key_error when the key is wrong). The twin can't validate
|
|
146
|
+
// against real keys, so it models the CHECKABLE failures: a missing credential →
|
|
147
|
+
// invalid_authentication_error, and a reserved sentinel ('sk_invalid'/'invalid') for the
|
|
148
|
+
// wrong-key path → incorrect_api_key_error. Any other non-empty key is accepted. Trusted
|
|
149
|
+
// in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated; both real
|
|
150
|
+
// clients always send a key → they pass.
|
|
151
|
+
//
|
|
152
|
+
// THE KEY FOLLOWS THE PREFIX: the Anthropic-compatible surface's documented client is the
|
|
153
|
+
// unmodified `@anthropic-ai/sdk`, which authenticates with `x-api-key` (never a bearer) — a
|
|
154
|
+
// bearer-only check made the whole /anthropic surface 401-dead for it. On MESSAGES_PREFIX the
|
|
155
|
+
// key candidates and the error envelope are Anthropic's own grammar (x-api-key first,
|
|
156
|
+
// `authentication_error` in the {type:'error',error:{…}} envelope), exactly as the anthropic
|
|
157
|
+
// pack's checkAuth reads them; on /v1 the bearer leads and Moonshot's own error types apply.
|
|
158
|
+
function checkAuth(req: MoonshotRequest): MoonshotResponseEnvelope | null {
|
|
159
|
+
const onMessages = onMessagesPath(req.path);
|
|
160
|
+
const auth = req.headers?.['authorization'];
|
|
161
|
+
const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
|
|
162
|
+
const xApiKey = typeof req.headers?.['x-api-key'] === 'string' ? req.headers['x-api-key'].trim() : '';
|
|
163
|
+
const key = (req.apiKey ?? '').trim() || (onMessages ? xApiKey || bearer : bearer || xApiKey);
|
|
164
|
+
if (!key) {
|
|
165
|
+
return onMessages
|
|
166
|
+
? messagesError(401, 'authentication_error', 'missing API key. Provide an x-api-key header (or Authorization: Bearer …).')
|
|
167
|
+
: authError('The API key is missing or malformed. Please check your API key.', 'invalid_authentication_error');
|
|
168
|
+
}
|
|
169
|
+
if (key === 'sk_invalid' || key === 'invalid') {
|
|
170
|
+
return onMessages
|
|
171
|
+
? messagesError(401, 'authentication_error', 'invalid x-api-key.')
|
|
172
|
+
: authError('The API key is invalid. Please check your API key.', 'incorrect_api_key_error');
|
|
173
|
+
}
|
|
174
|
+
return null;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/** True when the request targets the Anthropic-compatible surface — the prefix that owns its
|
|
178
|
+
* OWN error envelope, auth header and error-type vocabulary (see the two-prefixes note). */
|
|
179
|
+
function onMessagesPath(path: string): boolean {
|
|
180
|
+
const bare = path.split('?')[0] ?? '';
|
|
181
|
+
return bare === MESSAGES_PREFIX || bare.startsWith(`${MESSAGES_PREFIX}/`);
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
// ── modeled rate limiting (429) ────────────────────────────────────────────────────────
|
|
185
|
+
// Non-deterministic in production, so the twin exposes a DETERMINISTIC opt-in trigger:
|
|
186
|
+
// `x-twin-force-rate-limit: 1` returns the faithful 429 envelope plus Moonshot's own documented
|
|
187
|
+
// header family (platform.kimi.ai/docs/pricing/limits: a 429 carries X-RateLimit-Limit /
|
|
188
|
+
// X-RateLimit-Remaining / X-RateLimit-Reset) and `retry-after`. The Tier-0 figures are the
|
|
189
|
+
// LOWEST published row (RPM 3, TPM 500,000) — the same grounding the budget declaration uses.
|
|
190
|
+
function rateLimitError(): MoonshotResponseEnvelope {
|
|
191
|
+
return {
|
|
192
|
+
status: 429,
|
|
193
|
+
body: errBody('rate_limit_reached_error', 'Request rate limit reached: 3 requests per minute (Tier 0). Please retry after 20 seconds.'),
|
|
194
|
+
headers: {
|
|
195
|
+
'retry-after': '20',
|
|
196
|
+
'x-ratelimit-limit': '3',
|
|
197
|
+
'x-ratelimit-remaining': '0',
|
|
198
|
+
'x-ratelimit-reset': '20s',
|
|
199
|
+
},
|
|
200
|
+
};
|
|
201
|
+
}
|
|
202
|
+
/** Moonshot documents 503 as `server_unavailable` (platform.kimi.ai/docs/api/errors). The twin
|
|
203
|
+
* exposes it as a deterministic trigger so a caller can script the vendor's outage shape. */
|
|
204
|
+
function serverUnavailable(): MoonshotResponseEnvelope {
|
|
205
|
+
return { status: 503, body: errBody('server_unavailable', 'The server is overloaded or not ready to handle the request. Please try again later.') };
|
|
206
|
+
}
|
|
207
|
+
function triggered(req: MoonshotRequest, header: string): boolean {
|
|
208
|
+
const v = req.headers?.[header];
|
|
209
|
+
return v === '1' || v === 'true';
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
function nowEpoch(occurredAt?: string): number {
|
|
213
|
+
return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
// ── kernel helpers ──────────────────────────────────────────────────────────────────────
|
|
217
|
+
function rows(type: string, root?: string): Array<Record<string, unknown>> {
|
|
218
|
+
return projectResources(SERVICE, root).filter((r) => r.type === type);
|
|
219
|
+
}
|
|
220
|
+
/**
|
|
221
|
+
* Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
|
|
222
|
+
* projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
|
|
223
|
+
* gap above the count (ADDING_A_TWIN.md §5). Two further properties matter:
|
|
224
|
+
* • the `_twin_` infix namespaces LOCAL mints, so a pulled Moonshot id can never be matched by
|
|
225
|
+
* this regex and therefore can never be re-minted;
|
|
226
|
+
* • the scan includes TOMBSTONED rows (a soft-deleted file keeps its projection row), so the
|
|
227
|
+
* counter RATCHETS across delete→recreate and a deleted id is never handed out twice.
|
|
228
|
+
*/
|
|
229
|
+
function nextId(type: string, prefix: string, root?: string): string {
|
|
230
|
+
let max = 0;
|
|
231
|
+
for (const r of rows(type, root)) {
|
|
232
|
+
const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(String(r.id));
|
|
233
|
+
if (m) max = Math.max(max, Number(m[1]));
|
|
234
|
+
}
|
|
235
|
+
return `${prefix}_twin_${max + 1}`;
|
|
236
|
+
}
|
|
237
|
+
/** Models observed by a connector pull (mapModel), reshaped into the served model object. */
|
|
238
|
+
function pulledModels(root?: string): Array<Record<string, unknown>> {
|
|
239
|
+
return rows('model', root)
|
|
240
|
+
.filter((r) => !r._deleted)
|
|
241
|
+
.map((r) => ({ id: r.id, object: 'model', created: r.created, owned_by: r.owned_by }));
|
|
242
|
+
}
|
|
243
|
+
/** The catalog a request sees: the static table, with any PULLED row of the same id OVERRIDING it. */
|
|
244
|
+
function servedModels(root?: string): Array<Record<string, unknown>> {
|
|
245
|
+
const pulled = pulledModels(root);
|
|
246
|
+
const byId = new Map<string, Record<string, unknown>>();
|
|
247
|
+
for (const m of MOONSHOT_MODELS) byId.set(m.id, { id: m.id, object: 'model', created: m.created, owned_by: m.owned_by });
|
|
248
|
+
for (const m of pulled) byId.set(String(m.id), m);
|
|
249
|
+
return [...byId.values()];
|
|
250
|
+
}
|
|
251
|
+
function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined {
|
|
252
|
+
return rows(type, root).find((r) => r.id === id);
|
|
253
|
+
}
|
|
254
|
+
/** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
|
|
255
|
+
function strip(r: Record<string, unknown>): Record<string, unknown> {
|
|
256
|
+
const out: Record<string, unknown> = {};
|
|
257
|
+
for (const [k, v] of Object.entries(r)) {
|
|
258
|
+
if (k === 'type' || k === 'updatedAt' || k.startsWith('_')) continue;
|
|
259
|
+
out[k] = v;
|
|
260
|
+
}
|
|
261
|
+
return out;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
// ── request parsing ─────────────────────────────────────────────────────────────────────
|
|
265
|
+
function parseJson(body?: string): Record<string, unknown> {
|
|
266
|
+
if (!body || !body.trim()) return {};
|
|
267
|
+
try {
|
|
268
|
+
const v = JSON.parse(body);
|
|
269
|
+
return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
|
|
270
|
+
} catch {
|
|
271
|
+
return {};
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
// ── chat completions: validate the request the way Moonshot does ────────────────────────
|
|
276
|
+
type ToolChoice = 'auto' | 'none' | 'required' | { name: string };
|
|
277
|
+
type ResponseFormat = { kind: 'text' } | { kind: 'json_object' } | { kind: 'json_schema'; schema: unknown };
|
|
278
|
+
type Thinking = { type: 'enabled' | 'disabled'; keep?: string | null };
|
|
279
|
+
|
|
280
|
+
/** Moonshot's documented tool-name regex (ToolDefinition / MessagesTool schemas). */
|
|
281
|
+
const TOOL_NAME_RE = /^[a-zA-Z_][a-zA-Z0-9-_]{0,127}$/;
|
|
282
|
+
/** `stop`: "A maximum of 5 strings is allowed, and each string must not exceed 32 bytes"
|
|
283
|
+
* (ChatRequestBase.stop description). */
|
|
284
|
+
const STOP_MAX_ITEMS = 5;
|
|
285
|
+
const STOP_MAX_BYTES = 32;
|
|
286
|
+
|
|
287
|
+
type ChatArgs = {
|
|
288
|
+
model: string;
|
|
289
|
+
messages: MoonshotMessageParam[];
|
|
290
|
+
tools?: unknown[];
|
|
291
|
+
n: number;
|
|
292
|
+
maxTokens?: number;
|
|
293
|
+
stop?: string[];
|
|
294
|
+
stream: boolean;
|
|
295
|
+
streamOptions?: { includeUsage: boolean };
|
|
296
|
+
toolChoice?: ToolChoice;
|
|
297
|
+
responseFormat: ResponseFormat;
|
|
298
|
+
logprobs: boolean;
|
|
299
|
+
topLogprobs?: number;
|
|
300
|
+
/** kimi-k3 only: 'low' | 'high' | 'max' (default 'max'). */
|
|
301
|
+
reasoningEffort?: string;
|
|
302
|
+
/** kimi-k2.6 / kimi-k2.7-code: the `thinking` object. */
|
|
303
|
+
thinking?: Thinking;
|
|
304
|
+
promptCacheKey?: string;
|
|
305
|
+
};
|
|
306
|
+
|
|
307
|
+
/** Validate `thinking` per model. Returns the parsed value or an error envelope. */
|
|
308
|
+
function validateThinking(model: string, raw: unknown): { value?: Thinking; error?: MoonshotResponseEnvelope } {
|
|
309
|
+
if (raw === undefined || raw === null) return {};
|
|
310
|
+
if (typeof raw !== 'object' || Array.isArray(raw)) return { error: invalidRequest("'thinking' must be an object") };
|
|
311
|
+
const o = raw as Record<string, unknown>;
|
|
312
|
+
const type = o.type;
|
|
313
|
+
if (typeof type !== 'string') return { error: invalidRequest("'thinking.type' is required") };
|
|
314
|
+
if (model === 'kimi-k2.6') {
|
|
315
|
+
if (!K26_THINKING_TYPES.includes(type as 'enabled' | 'disabled')) {
|
|
316
|
+
return { error: invalidRequest(`'thinking.type' must be one of ${K26_THINKING_TYPES.map((t) => `'${t}'`).join(', ')} for kimi-k2.6`) };
|
|
317
|
+
}
|
|
318
|
+
} else if (model === 'kimi-k2.7-code' || model === 'kimi-k2.7-code-highspeed') {
|
|
319
|
+
// Moonshot's OpenAPI: "For kimi-k2.7-code, only `\"enabled\"` is accepted; passing
|
|
320
|
+
// `\"disabled\"` returns an error. This differs from kimi-k2.6."
|
|
321
|
+
if (!K27_THINKING_TYPES.includes(type as 'enabled')) {
|
|
322
|
+
return { error: invalidRequest(`'thinking.type' must be 'enabled' for ${model} — 'disabled' is not supported and returns an error`) };
|
|
323
|
+
}
|
|
324
|
+
} else {
|
|
325
|
+
return { error: invalidRequest(`'thinking' is not a parameter of ${model}`) };
|
|
326
|
+
}
|
|
327
|
+
let keep: string | null | undefined;
|
|
328
|
+
if (o.keep !== undefined) {
|
|
329
|
+
if (o.keep !== null && o.keep !== 'all') {
|
|
330
|
+
// kimi-k2.7-code: only "all" (or null/omitted) is valid; any other value errors.
|
|
331
|
+
// kimi-k2.6: same closed set {all, null}.
|
|
332
|
+
return { error: invalidRequest(`'thinking.keep' must be 'all' or null`) };
|
|
333
|
+
}
|
|
334
|
+
keep = o.keep as string | null;
|
|
335
|
+
}
|
|
336
|
+
return { value: { type: type as 'enabled' | 'disabled', ...(keep !== undefined ? { keep } : {}) } };
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
function validateChat(params: Record<string, unknown>): { args: ChatArgs } | { error: MoonshotResponseEnvelope } {
|
|
340
|
+
if (params.model === undefined || params.model === '') return { error: invalidRequest("'model' is a required property") };
|
|
341
|
+
if (typeof params.model !== 'string') return { error: invalidRequest("'model' must be a string") };
|
|
342
|
+
const model = findModel(params.model);
|
|
343
|
+
if (!model) return { error: invalidRequest(`The model '${params.model}' does not exist or you do not have access to it.`) };
|
|
344
|
+
if (!Array.isArray(params.messages)) return { error: invalidRequest("'messages' is a required property") };
|
|
345
|
+
if (params.messages.length === 0) return { error: invalidRequest("[] is too short - 'messages'") };
|
|
346
|
+
const messages = params.messages as MoonshotMessageParam[];
|
|
347
|
+
for (const m of messages) {
|
|
348
|
+
if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
|
|
349
|
+
return { error: invalidRequest("each message must have a valid 'role'") };
|
|
350
|
+
}
|
|
351
|
+
if (!['system', 'user', 'assistant', 'tool'].includes(m.role)) {
|
|
352
|
+
return { error: invalidRequest(`'${m.role}' is not one of ['system', 'user', 'assistant', 'tool']`) };
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
// kimi-k3's dynamic tool loading message: role 'system', `tools` present, NO content.
|
|
356
|
+
// Any OTHER message with `tools` and no content is malformed.
|
|
357
|
+
for (const m of messages) {
|
|
358
|
+
const hasTools = Array.isArray((m as { tools?: unknown }).tools);
|
|
359
|
+
if (m.role === 'system' && hasTools && m.content === undefined) continue; // the dynamic-tool shape
|
|
360
|
+
if (hasTools && m.content === undefined) {
|
|
361
|
+
return { error: invalidRequest("a dynamic tool message must use the 'system' role") };
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
// Vision: image_url parts are only valid on a vision model (kimi-k3, kimi-k2.6).
|
|
365
|
+
if (!model.supports.vision) {
|
|
366
|
+
for (const m of messages) {
|
|
367
|
+
const parts = m.content;
|
|
368
|
+
if (Array.isArray(parts) && parts.some((p) => (p as { type?: string })?.type === 'image_url')) {
|
|
369
|
+
return { error: invalidRequest(`The model '${params.model}' does not support image input`) };
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
// logprobs: Moonshot's OpenAPI ACCEPTS logprobs (boolean) + top_logprobs (0..20) — unlike
|
|
374
|
+
// Groq, this is real surface. The twin models acceptance (a logprobs echo on the choice) only
|
|
375
|
+
// as far as the shape goes: `top_logprobs` without `logprobs:true` contradicts the documented
|
|
376
|
+
// coupling ("logprobs must be set to true when this parameter is used").
|
|
377
|
+
const logprobs = params.logprobs === true;
|
|
378
|
+
let topLogprobs: number | undefined;
|
|
379
|
+
if (params.top_logprobs !== undefined && params.top_logprobs !== null) {
|
|
380
|
+
const n = Number(params.top_logprobs);
|
|
381
|
+
if (!Number.isInteger(n) || n < 0 || n > 20) return { error: invalidRequest("'top_logprobs' must be an integer between 0 and 20") };
|
|
382
|
+
if (!logprobs) return { error: invalidRequest("'logprobs' must be set to true when 'top_logprobs' is used") };
|
|
383
|
+
topLogprobs = n;
|
|
384
|
+
}
|
|
385
|
+
const maxRaw = params.max_completion_tokens ?? params.max_tokens;
|
|
386
|
+
let maxTokens: number | undefined;
|
|
387
|
+
if (maxRaw !== undefined) {
|
|
388
|
+
maxTokens = Number(maxRaw);
|
|
389
|
+
if (!Number.isInteger(maxTokens) || maxTokens < 1) return { error: invalidRequest("'max_completion_tokens' must be an integer >= 1") };
|
|
390
|
+
// "If input plus max_completion_tokens exceeds the model context window, the API returns
|
|
391
|
+
// invalid_request_error" (ChatRequestCommon).
|
|
392
|
+
const promptTokens = countPromptTokens(messages);
|
|
393
|
+
if (promptTokens + maxTokens > model.context_length) {
|
|
394
|
+
return { error: invalidRequest(`'max_completion_tokens' plus input tokens (${promptTokens + maxTokens}) exceeds the model context window (${model.context_length})`) };
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
let stop: string[] | undefined;
|
|
398
|
+
if (params.stop !== undefined && params.stop !== null) {
|
|
399
|
+
if (typeof params.stop === 'string') stop = [params.stop];
|
|
400
|
+
else if (Array.isArray(params.stop)) stop = params.stop as string[];
|
|
401
|
+
else return { error: invalidRequest("'stop' must be a string or an array of strings") };
|
|
402
|
+
if (stop.length > STOP_MAX_ITEMS) return { error: invalidRequest(`'stop' must not exceed ${STOP_MAX_ITEMS} strings`) };
|
|
403
|
+
for (const s of stop) {
|
|
404
|
+
if (typeof s !== 'string') return { error: invalidRequest("'stop' must be a string or an array of strings") };
|
|
405
|
+
if (Buffer.byteLength(s, 'utf8') > STOP_MAX_BYTES) return { error: invalidRequest(`'stop' entries must not exceed ${STOP_MAX_BYTES} bytes`) };
|
|
406
|
+
}
|
|
407
|
+
}
|
|
408
|
+
let toolChoice: ToolChoice | undefined;
|
|
409
|
+
const tcRaw = params.tool_choice;
|
|
410
|
+
if (tcRaw !== undefined && tcRaw !== null) {
|
|
411
|
+
if (typeof tcRaw === 'string') {
|
|
412
|
+
if (!['auto', 'none', 'required'].includes(tcRaw)) return { error: invalidRequest("'tool_choice' must be one of 'none', 'auto', 'required' or a named function") };
|
|
413
|
+
toolChoice = tcRaw as ToolChoice;
|
|
414
|
+
} else if (typeof tcRaw === 'object') {
|
|
415
|
+
const name = (tcRaw as { function?: { name?: unknown } }).function?.name;
|
|
416
|
+
if (typeof name !== 'string' || !name) return { error: invalidRequest("'tool_choice.function.name' is required for a named tool choice") };
|
|
417
|
+
toolChoice = { name };
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
let responseFormat: ResponseFormat = { kind: 'text' };
|
|
421
|
+
const rf = params.response_format as { type?: unknown; json_schema?: unknown } | undefined;
|
|
422
|
+
if (rf && typeof rf === 'object') {
|
|
423
|
+
if (rf.type === 'json_object') responseFormat = { kind: 'json_object' };
|
|
424
|
+
else if (rf.type === 'json_schema') {
|
|
425
|
+
// json_schema REQUIRES the json_schema object (name + schema required by the OpenAPI).
|
|
426
|
+
const js = rf.json_schema as { name?: unknown; schema?: unknown } | undefined;
|
|
427
|
+
if (!js || typeof js !== 'object') return { error: invalidRequest("'response_format.json_schema' is required when 'response_format.type' is 'json_schema'") };
|
|
428
|
+
if (typeof js.name !== 'string' || !js.name) return { error: invalidRequest("'response_format.json_schema.name' is required") };
|
|
429
|
+
if (!js.schema || typeof js.schema !== 'object') return { error: invalidRequest("'response_format.json_schema.schema' is required") };
|
|
430
|
+
responseFormat = { kind: 'json_schema', schema: js.schema };
|
|
431
|
+
}
|
|
432
|
+
else if (rf.type !== undefined && rf.type !== 'text') return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
|
|
433
|
+
}
|
|
434
|
+
// Tools: validate the documented name regex on every provided function tool.
|
|
435
|
+
if (params.tools !== undefined) {
|
|
436
|
+
if (!Array.isArray(params.tools)) return { error: invalidRequest("'tools' must be an array") };
|
|
437
|
+
for (const t of params.tools) {
|
|
438
|
+
const name = (t as { function?: { name?: unknown } })?.function?.name;
|
|
439
|
+
if (typeof name !== 'string' || !TOOL_NAME_RE.test(name)) {
|
|
440
|
+
return { error: invalidRequest(`'tools[].function.name' must match ${TOOL_NAME_RE.source}`) };
|
|
441
|
+
}
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
// kimi-k3's reasoning_effort: a CLOSED set, default 'max'.
|
|
445
|
+
let reasoningEffort: string | undefined;
|
|
446
|
+
if (params.reasoning_effort !== undefined && params.reasoning_effort !== null) {
|
|
447
|
+
if (params.model !== 'kimi-k3') return { error: invalidRequest(`'reasoning_effort' is not a parameter of ${params.model}`) };
|
|
448
|
+
if (typeof params.reasoning_effort !== 'string' || !REASONING_EFFORTS.includes(params.reasoning_effort as ReasoningEffort)) {
|
|
449
|
+
return { error: invalidRequest(`'reasoning_effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) };
|
|
450
|
+
}
|
|
451
|
+
reasoningEffort = params.reasoning_effort;
|
|
452
|
+
}
|
|
453
|
+
const thinking = validateThinking(params.model, params.thinking);
|
|
454
|
+
if (thinking.error) return { error: thinking.error };
|
|
455
|
+
const streamOptionsRaw = params.stream_options as { include_usage?: unknown } | undefined;
|
|
456
|
+
if (streamOptionsRaw !== undefined && (typeof streamOptionsRaw !== 'object' || streamOptionsRaw === null)) {
|
|
457
|
+
return { error: invalidRequest("'stream_options' must be an object") };
|
|
458
|
+
}
|
|
459
|
+
return {
|
|
460
|
+
args: {
|
|
461
|
+
model: params.model,
|
|
462
|
+
messages,
|
|
463
|
+
...(params.tools !== undefined ? { tools: params.tools as unknown[] } : {}),
|
|
464
|
+
n: 1,
|
|
465
|
+
...(maxTokens !== undefined ? { maxTokens } : {}),
|
|
466
|
+
...(stop !== undefined ? { stop } : {}),
|
|
467
|
+
stream: params.stream === true,
|
|
468
|
+
...(streamOptionsRaw?.include_usage === true ? { streamOptions: { includeUsage: true } } : {}),
|
|
469
|
+
...(toolChoice !== undefined ? { toolChoice } : {}),
|
|
470
|
+
responseFormat,
|
|
471
|
+
logprobs,
|
|
472
|
+
...(topLogprobs !== undefined ? { topLogprobs } : {}),
|
|
473
|
+
...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
|
|
474
|
+
...(thinking.value !== undefined ? { thinking: thinking.value } : {}),
|
|
475
|
+
...(typeof params.prompt_cache_key === 'string' ? { promptCacheKey: params.prompt_cache_key } : {}),
|
|
476
|
+
},
|
|
477
|
+
};
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
/** Build ONE deterministic stub choice. Thinking mode follows the model/params: kimi-k3 always
|
|
481
|
+
* thinks; k2.6 thinks unless `thinking.type:'disabled'`; k2.7-code always thinks. */
|
|
482
|
+
function buildChoice(args: ChatArgs, idx: number): { choice: MoonshotChoice; completionTokens: number } {
|
|
483
|
+
const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
|
|
484
|
+
const forbidTools = args.toolChoice === 'none';
|
|
485
|
+
const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
|
|
486
|
+
if (hasTools && !forbidTools) {
|
|
487
|
+
const list = args.tools as unknown[];
|
|
488
|
+
const calls: MoonshotToolCall[] = [];
|
|
489
|
+
if (forcedName) {
|
|
490
|
+
const tc = stubToolCall(args.tools, idx + 1, forcedName);
|
|
491
|
+
if (tc) calls.push(tc);
|
|
492
|
+
} else {
|
|
493
|
+
for (let t = 0; t < list.length; t++) {
|
|
494
|
+
const tc = stubToolCall([list[t]], idx * 100 + t + 1);
|
|
495
|
+
if (tc) calls.push(tc);
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
if (calls.length) {
|
|
499
|
+
return {
|
|
500
|
+
choice: { index: idx, message: { role: 'assistant', content: null, tool_calls: calls }, finish_reason: 'tool_calls' },
|
|
501
|
+
completionTokens: estimateTokens(JSON.stringify(calls)),
|
|
502
|
+
};
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
let text = args.responseFormat.kind === 'json_object'
|
|
506
|
+
? stubJsonObject(args.messages, args.model)
|
|
507
|
+
: args.responseFormat.kind === 'json_schema'
|
|
508
|
+
? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
|
|
509
|
+
: stubAssistantText(args.messages, args.model);
|
|
510
|
+
let finish: MoonshotChoice['finish_reason'] = 'stop';
|
|
511
|
+
// Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
|
|
512
|
+
let stopAt = -1;
|
|
513
|
+
for (const s of args.stop ?? []) {
|
|
514
|
+
if (!s) continue;
|
|
515
|
+
const i = text.indexOf(s);
|
|
516
|
+
if (i >= 0 && (stopAt < 0 || i < stopAt)) stopAt = i;
|
|
517
|
+
}
|
|
518
|
+
if (stopAt >= 0) text = text.slice(0, stopAt);
|
|
519
|
+
if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
|
|
520
|
+
text = text.slice(0, args.maxTokens * 4);
|
|
521
|
+
finish = 'length';
|
|
522
|
+
}
|
|
523
|
+
const message: MoonshotAssistantMessage = { role: 'assistant', content: text };
|
|
524
|
+
// Thinking mode: kimi-k3 ALWAYS (Preserved Thinking); k2.7-code ALWAYS (type only accepts
|
|
525
|
+
// 'enabled'); k2.6 when thinking.type is not 'disabled'.
|
|
526
|
+
const thinkingOn = args.model === 'kimi-k3'
|
|
527
|
+
|| args.model === 'kimi-k2.7-code' || args.model === 'kimi-k2.7-code-highspeed'
|
|
528
|
+
|| (args.model === 'kimi-k2.6' && args.thinking?.type !== 'disabled');
|
|
529
|
+
if (thinkingOn) {
|
|
530
|
+
message.reasoning_content = stubReasoningContent(args.messages, args.model);
|
|
531
|
+
// The cap applies to EVERYTHING the model emits — reasoning is output tokens at the vendor
|
|
532
|
+
// too — so truncate it as well and let completion_tokens count only what was returned.
|
|
533
|
+
// finish_reason stays 'length': the cap is what truncated, whichever field hit it.
|
|
534
|
+
if (args.maxTokens !== undefined) {
|
|
535
|
+
const textTokens = estimateTokens(String(message.content ?? ''));
|
|
536
|
+
if (textTokens >= args.maxTokens) {
|
|
537
|
+
delete message.reasoning_content;
|
|
538
|
+
} else {
|
|
539
|
+
const budget = args.maxTokens - textTokens;
|
|
540
|
+
const reasoning = message.reasoning_content;
|
|
541
|
+
if (estimateTokens(reasoning) > budget) message.reasoning_content = reasoning.slice(0, budget * 4);
|
|
542
|
+
}
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
return {
|
|
546
|
+
choice: { index: idx, message, finish_reason: finish },
|
|
547
|
+
completionTokens: estimateTokens(String(message.content ?? '')) + estimateTokens(message.reasoning_content ?? ''),
|
|
548
|
+
};
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
export function buildChatCompletion(args: ChatArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotChatCompletion | MoonshotResponseEnvelope {
|
|
552
|
+
const promptTokens = countPromptTokens(args.messages);
|
|
553
|
+
let scripted: ScriptedResult | null = null;
|
|
554
|
+
let missTeach = '';
|
|
555
|
+
if (decision) {
|
|
556
|
+
// The route served the request through the engine (R15: serve() honors the handler's fault
|
|
557
|
+
// before any content exists); this realizer only sees the content decision.
|
|
558
|
+
if (decision.kind === 'handler') {
|
|
559
|
+
const respond = decision.respond as MoonshotScenarioRespond;
|
|
560
|
+
// A scripted FAILURE short-circuits into Moonshot's own error envelope + status.
|
|
561
|
+
if (respond.error) return scriptedError(respond.error);
|
|
562
|
+
scripted = realizeMoonshotRespond(respond);
|
|
563
|
+
} else {
|
|
564
|
+
missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/moonshot.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
|
|
565
|
+
}
|
|
566
|
+
}
|
|
567
|
+
const choices: MoonshotChoice[] = [];
|
|
568
|
+
let completionTokens = 0;
|
|
569
|
+
for (let i = 0; i < args.n; i++) {
|
|
570
|
+
if (scripted) {
|
|
571
|
+
const message: MoonshotAssistantMessage = scripted.toolCalls.length
|
|
572
|
+
? { role: 'assistant', content: scripted.text, tool_calls: scripted.toolCalls }
|
|
573
|
+
: { role: 'assistant', content: scripted.text ?? '' };
|
|
574
|
+
if (scripted.reasoning !== null) message.reasoning_content = scripted.reasoning;
|
|
575
|
+
choices.push({ index: i, message, finish_reason: scripted.finishReason });
|
|
576
|
+
completionTokens += estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
|
|
577
|
+
continue;
|
|
578
|
+
}
|
|
579
|
+
const { choice, completionTokens: ct } = buildChoice(args, i);
|
|
580
|
+
if (missTeach && typeof choice.message.content === 'string') choice.message.content += missTeach;
|
|
581
|
+
choices.push(choice);
|
|
582
|
+
completionTokens += ct;
|
|
583
|
+
}
|
|
584
|
+
const usage: MoonshotUsage = buildChatUsage(promptTokens, completionTokens, args.promptCacheKey);
|
|
585
|
+
const id = `chatcmpl-twin-${stableSuffix(args.messages, args.model)}`;
|
|
586
|
+
return {
|
|
587
|
+
id,
|
|
588
|
+
object: 'chat.completion',
|
|
589
|
+
created: nowEpoch(occurredAt),
|
|
590
|
+
model: args.model,
|
|
591
|
+
choices,
|
|
592
|
+
usage,
|
|
593
|
+
};
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
/** Map a scripted scenario failure onto Moonshot's real status + envelope. */
|
|
597
|
+
function scriptedError(err: NonNullable<MoonshotScenarioRespond['error']>): MoonshotResponseEnvelope {
|
|
598
|
+
if (err.type === 'rate_limit_reached_error') {
|
|
599
|
+
const base = rateLimitError();
|
|
600
|
+
return err.message ? { ...base, body: errBody('rate_limit_reached_error', err.message) } : base;
|
|
601
|
+
}
|
|
602
|
+
if (err.type === 'server_unavailable') return err.message ? { ...serverUnavailable(), body: errBody('server_unavailable', err.message) } : serverUnavailable();
|
|
603
|
+
return { status: 500, body: errBody('server_error', err.message ?? 'Internal Server Error') };
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
const isEnvelope = (v: object): v is MoonshotResponseEnvelope =>
|
|
607
|
+
typeof (v as MoonshotResponseEnvelope).status === 'number' && 'body' in v;
|
|
608
|
+
|
|
609
|
+
/** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
|
|
610
|
+
function chunkText(text: string): string[] {
|
|
611
|
+
if (!text) return [];
|
|
612
|
+
const out: string[] = [];
|
|
613
|
+
for (let i = 0; i < text.length; i += 20) out.push(text.slice(i, i + 20));
|
|
614
|
+
return out;
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
/**
|
|
618
|
+
* Emit the vendor-faithful Moonshot streaming sequence into the injected sink (NO sockets, NO
|
|
619
|
+
* setTimeout). Moonshot's order: a first chunk with `delta:{role:'assistant'}`, then
|
|
620
|
+
* `delta:{content}` / `delta:{reasoning_content}` / tool_calls deltas, then a chunk carrying
|
|
621
|
+
* `finish_reason`, then — because Moonshot's ChatCompletionChunk.usage is "Object in the final
|
|
622
|
+
* chunk with usage, null in ordinary chunks" (the OpenAPI's own wording, the first-party
|
|
623
|
+
* denominator; the prose chat doc mentions the chunk under stream_options.include_usage, so the
|
|
624
|
+
* two sources disagree and the OpenAPI governs) — a FINAL chunk whose `choices:[]` and `usage`
|
|
625
|
+
* hold the whole usage object, then `data: [DONE]`. `stream_options` is ACCEPTED (validated as
|
|
626
|
+
* an object) but the tail is not gated on it. Deterministic + synchronous.
|
|
627
|
+
*/
|
|
628
|
+
export function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotChatCompletion | MoonshotResponseEnvelope {
|
|
629
|
+
const built = buildChatCompletion(args, occurredAt, decision);
|
|
630
|
+
if (isEnvelope(built)) return built;
|
|
631
|
+
const full = built;
|
|
632
|
+
const base = { id: full.id, object: 'chat.completion.chunk' as const, created: full.created, model: full.model };
|
|
633
|
+
for (const choice of full.choices) {
|
|
634
|
+
const idx = choice.index;
|
|
635
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, finish_reason: null, usage: null }] } });
|
|
636
|
+
if (choice.message.reasoning_content) {
|
|
637
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { reasoning_content: choice.message.reasoning_content }, finish_reason: null, usage: null }] } });
|
|
638
|
+
}
|
|
639
|
+
if (choice.message.tool_calls && choice.message.tool_calls.length) {
|
|
640
|
+
choice.message.tool_calls.forEach((tc, tIdx) => {
|
|
641
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, finish_reason: null, usage: null }] } });
|
|
642
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, finish_reason: null, usage: null }] } });
|
|
643
|
+
});
|
|
644
|
+
} else {
|
|
645
|
+
for (const piece of chunkText(choice.message.content ?? '')) {
|
|
646
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, finish_reason: null, usage: null }] } });
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: {}, finish_reason: choice.finish_reason, usage: null }] } });
|
|
650
|
+
}
|
|
651
|
+
// Moonshot's usage tail: a final chunk with an EMPTY choices array carrying the usage object.
|
|
652
|
+
sink({ data: { ...base, choices: [], finish_reason: null, usage: full.usage } });
|
|
653
|
+
sink({ done: true });
|
|
654
|
+
return full;
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
// ── /v1/tokenizers/estimate-token-count (stateless, deterministic) ──────────────────────
|
|
658
|
+
function handleEstimateTokens(params: Record<string, unknown>): MoonshotResponseEnvelope {
|
|
659
|
+
if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
|
|
660
|
+
if (!findModel(params.model)) return invalidRequest(`The model '${params.model}' does not exist or you do not have access to it.`);
|
|
661
|
+
if (!Array.isArray(params.messages)) return { status: 400, body: errBody('invalid_request_error', "'messages' is a required property") };
|
|
662
|
+
const messages = params.messages as MoonshotMessageParam[];
|
|
663
|
+
for (const m of messages) {
|
|
664
|
+
if (!m || typeof m !== 'object' || typeof m.role !== 'string') return invalidRequest("each message must have a valid 'role'");
|
|
665
|
+
// The OpenAPI: "content must not be empty".
|
|
666
|
+
if (m.content === undefined || m.content === null || (typeof m.content === 'string' && m.content === '')) {
|
|
667
|
+
return invalidRequest("'messages[].content' must not be empty");
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
return { status: 200, body: { data: { total_tokens: countPromptTokens(messages) } } };
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
// ── /v1/signatures/verify (stateless, deterministic) ────────────────────────────────────
|
|
674
|
+
// Moonshot's request-signing contract: a model call carries `X-Msh-Request-Nonce`; the response
|
|
675
|
+
// carries `Msh-Request-Timestamp` (unix ms) and `Msh-Request-Signature` (`reqsigv1_<opaque>`).
|
|
676
|
+
// POST /v1/signatures/verify checks a (nonce, timestamp, model, signature) tuple. The twin has
|
|
677
|
+
// no real signing key, so it models the CHECKABLE part: a well-formed tuple whose signature
|
|
678
|
+
// carries the documented `reqsigv1_` prefix AND whose signature recomputes exactly under the
|
|
679
|
+
// twin's own deterministic scheme (messagesSignature over nonce/timestamp/model) answers
|
|
680
|
+
// `valid:true`; anything else answers `valid:false`. The twin ALSO emits the two Msh- headers
|
|
681
|
+
// on its own streaming model responses WHEN the request carried a nonce (see the server) so the
|
|
682
|
+
// round-trip is exercisable end to end.
|
|
683
|
+
function handleSignatureVerify(params: Record<string, unknown>, req: MoonshotRequest): MoonshotResponseEnvelope {
|
|
684
|
+
for (const field of ['nonce', 'timestamp', 'model', 'signature'] as const) {
|
|
685
|
+
if (params[field] === undefined || params[field] === null || params[field] === '') {
|
|
686
|
+
return invalidRequest(`'${field}' is a required property`);
|
|
687
|
+
}
|
|
688
|
+
}
|
|
689
|
+
if (typeof params.nonce !== 'string' || typeof params.model !== 'string' || typeof params.signature !== 'string') {
|
|
690
|
+
return invalidRequest("'nonce', 'model' and 'signature' must be strings");
|
|
691
|
+
}
|
|
692
|
+
const ts = Number(params.timestamp);
|
|
693
|
+
if (!Number.isFinite(ts) || ts < 1) return invalidRequest("'timestamp' must be a positive integer");
|
|
694
|
+
const sig = String(params.signature);
|
|
695
|
+
if (!sig.startsWith('reqsigv1_')) return { status: 200, body: { valid: false } };
|
|
696
|
+
// The twin's own signatures are derived from the tuple (see messagesSignature); a signature
|
|
697
|
+
// that recomputes exactly answers valid:true.
|
|
698
|
+
const expected = messagesSignature(String(params.nonce), ts, String(params.model));
|
|
699
|
+
return { status: 200, body: { valid: sig === expected } };
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
/** The deterministic signature the twin issues on inference responses and verifies here. */
|
|
703
|
+
export function messagesSignature(nonce: string, timestamp: number, model: string): string {
|
|
704
|
+
return `reqsigv1_twin_${fnv1a(`${nonce}|${timestamp}|${model}`).toString(36)}`;
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
// ── Web-search tools (stateless, deterministic labeled stubs) ───────────────────────────
|
|
708
|
+
const TIME_WINDOW_RE = /^\d{4}(-\d{2})?(-\d{2})?$/;
|
|
709
|
+
|
|
710
|
+
function validateTimeWindow(raw: unknown): MoonshotResponseEnvelope | null {
|
|
711
|
+
if (raw === undefined) return null;
|
|
712
|
+
if (typeof raw !== 'object' || raw === null) return invalidRequest("'time_window' must be an object");
|
|
713
|
+
const o = raw as { start?: unknown; end?: unknown };
|
|
714
|
+
for (const k of ['start', 'end'] as const) {
|
|
715
|
+
if (o[k] !== undefined && (typeof o[k] !== 'string' || !TIME_WINDOW_RE.test(o[k]))) {
|
|
716
|
+
return invalidRequest(`'time_window.${k}' must be in YYYY, YYYY-MM or YYYY-MM-DD format`);
|
|
717
|
+
}
|
|
718
|
+
}
|
|
719
|
+
return null;
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
function handleToolsSearch(params: Record<string, unknown>, pro: boolean): MoonshotResponseEnvelope {
|
|
723
|
+
if (typeof params.text_query !== 'string' || !params.text_query) return invalidRequest("'text_query' is a required property and must not be empty");
|
|
724
|
+
if (params.limit !== undefined) {
|
|
725
|
+
const n = Number(params.limit);
|
|
726
|
+
if (!Number.isInteger(n) || n < 1 || n > 20) return invalidRequest("'limit' must be an integer between 1 and 20");
|
|
727
|
+
}
|
|
728
|
+
if (params.timeout_seconds !== undefined) {
|
|
729
|
+
const n = Number(params.timeout_seconds);
|
|
730
|
+
if (!Number.isInteger(n) || n < 1 || n > 60) return invalidRequest("'timeout_seconds' must be an integer between 1 and 60");
|
|
731
|
+
}
|
|
732
|
+
if (pro) {
|
|
733
|
+
if (params.sites !== undefined) {
|
|
734
|
+
if (!Array.isArray(params.sites) || params.sites.length > 5) return invalidRequest("'sites' must be an array of at most 5 strings");
|
|
735
|
+
for (const s of params.sites) {
|
|
736
|
+
if (typeof s !== 'string' || !s || /\s|\(|\)/.test(s)) return invalidRequest("'sites' entries must be non-empty and contain no whitespace or parentheses");
|
|
737
|
+
}
|
|
738
|
+
}
|
|
739
|
+
const tw = validateTimeWindow(params.time_window);
|
|
740
|
+
if (tw) return tw;
|
|
741
|
+
} else {
|
|
742
|
+
// search (not search_pro) takes include_content; search_pro does not declare it.
|
|
743
|
+
if (params.include_content !== undefined && typeof params.include_content !== 'boolean') {
|
|
744
|
+
return invalidRequest("'include_content' must be a boolean");
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
const limit = params.limit === undefined ? 5 : Number(params.limit);
|
|
748
|
+
// `include_content: true` SERVES content (the option's documented meaning): the results carry
|
|
749
|
+
// a labeled deterministic page stub. Default (false) is the vendor's bare shape — `text: ''`.
|
|
750
|
+
// A no-op 200 that accepted the flag and always served '' was the round-two finding.
|
|
751
|
+
const includeContent = pro ? false : params.include_content === true;
|
|
752
|
+
const results = stubSearchResults(String(params.text_query), limit, pro, includeContent);
|
|
753
|
+
return { status: 200, body: { search_results: results } };
|
|
754
|
+
}
|
|
755
|
+
|
|
756
|
+
function handleToolsFetch(params: Record<string, unknown>): MoonshotResponseEnvelope {
|
|
757
|
+
if (typeof params.url !== 'string' || !params.url) return invalidRequest("'url' is a required property");
|
|
758
|
+
if (!/^https?:\/\//.test(params.url)) return invalidRequest("'url' must be an http or https URL");
|
|
759
|
+
return { status: 200, body: stubFetchedMarkdown(params.url) };
|
|
760
|
+
}
|
|
761
|
+
|
|
762
|
+
// ── Files (stateful) ────────────────────────────────────────────────────────────────────
|
|
763
|
+
/** FileObject.purpose — a CLOSED documented set: file-extract, image, video, batch. */
|
|
764
|
+
const FILE_CREATE_PURPOSES = new Set(['file-extract', 'image', 'video', 'batch']);
|
|
765
|
+
|
|
766
|
+
async function createFile(params: Record<string, unknown>, req: MoonshotRequest): Promise<MoonshotResponseEnvelope> {
|
|
767
|
+
const purpose = String(params.purpose ?? '');
|
|
768
|
+
if (!purpose) return invalidRequest("'purpose' is a required property");
|
|
769
|
+
if (!FILE_CREATE_PURPOSES.has(purpose)) return invalidRequest(`'purpose' must be one of ${[...FILE_CREATE_PURPOSES].map((p) => `'${p}'`).join(', ')} (got '${purpose}')`);
|
|
770
|
+
// The multipart adapter marks a form with no `file` part (§9 round two, F3): the vendor's
|
|
771
|
+
// Upload File requires the file body, so the twin refuses rather than minting an empty
|
|
772
|
+
// 'ready' file.
|
|
773
|
+
if (params._multipart_missing_file === true) return invalidRequest("the multipart form carries no 'file' part");
|
|
774
|
+
const filename = String(params.filename ?? params.file ?? 'upload');
|
|
775
|
+
const content = typeof params.content === 'string' ? params.content : '';
|
|
776
|
+
// `binary_content` is the SERVER's multipart-adapter marker (moonshot-server.ts): the JSON
|
|
777
|
+
// contract cannot carry raw bytes, so an uploaded binary travels base64-encoded under it and
|
|
778
|
+
// is stored with the marker; GET /content decodes before serving. A JSON-door create (plain
|
|
779
|
+
// text) stores the text as-is with no marker. A caller forging the marker through the JSON
|
|
780
|
+
// door gets strict validation: the content must actually be base64 (§9 round two, F6).
|
|
781
|
+
const binary = params.binary_content === true;
|
|
782
|
+
if (binary && !/^[A-Za-z0-9+/]*={0,2}$/.test(content) || (binary && content.length % 4 !== 0)) {
|
|
783
|
+
return invalidRequest("'binary_content' content must be base64-encoded");
|
|
784
|
+
}
|
|
785
|
+
// `bytes` is the vendor's byte count — a text file's UTF-8 length, not the JS string's
|
|
786
|
+
// UTF-16 code-unit count (a §9-round-two finding: the two doors disagreed on the same field;
|
|
787
|
+
// the multipart door already counted real bytes).
|
|
788
|
+
const bytes = typeof params.bytes === 'number' ? params.bytes : binary ? Buffer.from(content, 'base64').length : Buffer.byteLength(content, 'utf8');
|
|
789
|
+
const id = nextId('file', 'file', req.root);
|
|
790
|
+
await applyTwinWrite(SERVICE, {
|
|
791
|
+
operation: 'file.create',
|
|
792
|
+
subjectType: 'file',
|
|
793
|
+
subjectId: id,
|
|
794
|
+
fields: { object: 'file', bytes, created_at: nowEpoch(req.occurredAt), filename, purpose, status: 'ready', _content: content, ...(binary ? { _content_encoding: 'base64' } : {}) },
|
|
795
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
796
|
+
actor: { kind: 'agent' },
|
|
797
|
+
}, req.root);
|
|
798
|
+
const row = getRow('file', id, req.root);
|
|
799
|
+
return { status: 200, body: fileView(row ?? {}) };
|
|
800
|
+
}
|
|
801
|
+
function fileView(r: Record<string, unknown>): Record<string, unknown> {
|
|
802
|
+
const s = strip(r);
|
|
803
|
+
return { id: r.id, object: 'file', bytes: s.bytes, created_at: s.created_at, filename: s.filename, purpose: s.purpose, status: s.status ?? 'ready' };
|
|
804
|
+
}
|
|
805
|
+
|
|
806
|
+
// ── Batches (stateful) ──────────────────────────────────────────────────────────────────
|
|
807
|
+
/** BatchCreateRequest.endpoint — a CLOSED documented set: only /v1/chat/completions. */
|
|
808
|
+
const BATCH_ENDPOINTS = new Set(['/v1/chat/completions']);
|
|
809
|
+
/** "supports formats like 12h, 1d, 3d, minimum 12h, maximum 7d" (BatchCreateRequest). */
|
|
810
|
+
const COMPLETION_WINDOW = /^(\d+)(h|d)$/;
|
|
811
|
+
function completionWindowHours(w: string): number | null {
|
|
812
|
+
const m = COMPLETION_WINDOW.exec(w);
|
|
813
|
+
if (!m) return null;
|
|
814
|
+
const n = Number(m[1]);
|
|
815
|
+
return m[2] === 'd' ? n * 24 : n;
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
async function createBatch(params: Record<string, unknown>, req: MoonshotRequest): Promise<MoonshotResponseEnvelope> {
|
|
819
|
+
const inputFileId = params.input_file_id;
|
|
820
|
+
if (typeof inputFileId !== 'string' || !inputFileId) return invalidRequest("'input_file_id' is a required property");
|
|
821
|
+
const endpoint = params.endpoint;
|
|
822
|
+
if (typeof endpoint !== 'string' || !BATCH_ENDPOINTS.has(endpoint)) {
|
|
823
|
+
return invalidRequest(`'endpoint' must be one of ${[...BATCH_ENDPOINTS].map((e) => `'${e}'`).join(', ')}`);
|
|
824
|
+
}
|
|
825
|
+
const window = params.completion_window;
|
|
826
|
+
if (typeof window !== 'string') return invalidRequest("'completion_window' is a required property");
|
|
827
|
+
const hours = completionWindowHours(window);
|
|
828
|
+
if (hours === null || hours < 12 || hours > 168) return invalidRequest("'completion_window' must be a duration from '12h' to '7d'");
|
|
829
|
+
if (params.metadata !== undefined) {
|
|
830
|
+
const md = params.metadata;
|
|
831
|
+
if (!md || typeof md !== 'object' || Array.isArray(md)) return invalidRequest("'metadata' must be an object");
|
|
832
|
+
const entries = Object.entries(md as Record<string, unknown>);
|
|
833
|
+
if (entries.length > 16) return invalidRequest("'metadata' must not exceed 16 key-value pairs");
|
|
834
|
+
for (const [k, v] of entries) {
|
|
835
|
+
if (k.length > 64) return invalidRequest("'metadata' keys must not exceed 64 characters");
|
|
836
|
+
if (typeof v !== 'string' || v.length > 512) return invalidRequest("'metadata' values must be strings of at most 512 characters");
|
|
837
|
+
}
|
|
838
|
+
}
|
|
839
|
+
const file = getRow('file', inputFileId, req.root);
|
|
840
|
+
if (!file || file._deleted) return notFound(`No such File object: ${inputFileId}`);
|
|
841
|
+
if (file.purpose !== 'batch') return invalidRequest(`File ${inputFileId} must have purpose 'batch' (has '${String(file.purpose)}')`);
|
|
842
|
+
const created = nowEpoch(req.occurredAt);
|
|
843
|
+
const id = nextId('batch', 'batch', req.root);
|
|
844
|
+
await applyTwinWrite(SERVICE, {
|
|
845
|
+
operation: 'batch.create',
|
|
846
|
+
subjectType: 'batch',
|
|
847
|
+
subjectId: id,
|
|
848
|
+
fields: {
|
|
849
|
+
object: 'batch',
|
|
850
|
+
endpoint,
|
|
851
|
+
input_file_id: inputFileId,
|
|
852
|
+
completion_window: window,
|
|
853
|
+
status: 'validating',
|
|
854
|
+
output_file_id: null,
|
|
855
|
+
error_file_id: null,
|
|
856
|
+
created_at: created,
|
|
857
|
+
in_progress_at: null,
|
|
858
|
+
expires_at: created + hours * 3600,
|
|
859
|
+
finalizing_at: null,
|
|
860
|
+
completed_at: null,
|
|
861
|
+
failed_at: null,
|
|
862
|
+
cancelling_at: null,
|
|
863
|
+
cancelled_at: null,
|
|
864
|
+
request_counts: { total: 0, completed: 0, failed: 0 },
|
|
865
|
+
metadata: params.metadata ?? null,
|
|
866
|
+
},
|
|
867
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
868
|
+
actor: { kind: 'agent' },
|
|
869
|
+
}, req.root);
|
|
870
|
+
return { status: 200, body: batchView(getRow('batch', id, req.root) ?? {}) };
|
|
871
|
+
}
|
|
872
|
+
function batchView(r: Record<string, unknown>): Record<string, unknown> {
|
|
873
|
+
return { id: r.id, ...strip(r) };
|
|
874
|
+
}
|
|
875
|
+
|
|
876
|
+
// ── Balance (stateful) ──────────────────────────────────────────────────────────────────
|
|
877
|
+
// GET /v1/users/me/balance returns Moonshot's own envelope ({code, data:{available_balance,
|
|
878
|
+
// voucher_balance, cash_balance}, scode, status}) — NOT the OpenAI envelope. The balance is
|
|
879
|
+
// STATEFUL: the connector's pull can fold a real account's balance into the log (type 'balance',
|
|
880
|
+
// id 'me'), and the served figure is that row when present, else a deterministic positive stub.
|
|
881
|
+
function balanceView(root?: string): MoonshotResponseEnvelope {
|
|
882
|
+
const row = getRow('balance', 'me', root);
|
|
883
|
+
const b = row && !row._deleted
|
|
884
|
+
? { available_balance: Number(row.available_balance), voucher_balance: Number(row.voucher_balance), cash_balance: Number(row.cash_balance) }
|
|
885
|
+
: { available_balance: 49.58894, voucher_balance: 46.58893, cash_balance: 3.00001 };
|
|
886
|
+
return {
|
|
887
|
+
status: 200,
|
|
888
|
+
body: { code: 0, data: b, scode: '0x0', status: true },
|
|
889
|
+
};
|
|
890
|
+
}
|
|
891
|
+
|
|
892
|
+
// ── Anthropic-compatible Messages (/anthropic/v1/messages) ──────────────────────────────
|
|
893
|
+
type MessagesArgs = {
|
|
894
|
+
model: string;
|
|
895
|
+
messages: Array<{ role: 'user' | 'assistant'; content: string | Array<Record<string, unknown>> }>;
|
|
896
|
+
maxTokens: number;
|
|
897
|
+
system?: string;
|
|
898
|
+
stopSequences?: string[];
|
|
899
|
+
stream: boolean;
|
|
900
|
+
/** Anthropic's MessagesTool shape: { name, input_schema } (nested inside type:'custom'). */
|
|
901
|
+
tools?: Array<Record<string, unknown>>;
|
|
902
|
+
/** Anthropic's ToolChoice: type 'tool' names one tool (`name` REQUIRED, validated against
|
|
903
|
+
* `tools` — the sibling pack's rule). */
|
|
904
|
+
toolChoice?: { type: 'auto' | 'any' | 'tool' | 'none'; name?: string };
|
|
905
|
+
outputEffort?: string;
|
|
906
|
+
};
|
|
907
|
+
|
|
908
|
+
/** Extract the tool name from an Anthropic Messages tool definition ({type:'custom', name, …}). */
|
|
909
|
+
function messagesToolName(t: unknown): string {
|
|
910
|
+
const o = t as { name?: unknown } | undefined;
|
|
911
|
+
return typeof o?.name === 'string' ? o.name : '';
|
|
912
|
+
}
|
|
913
|
+
|
|
914
|
+
/** Validate the Anthropic-compatible request. kimi-k3 only; max_tokens REQUIRED. */
|
|
915
|
+
function validateMessages(params: Record<string, unknown>): { args: MessagesArgs } | { error: MoonshotResponseEnvelope } {
|
|
916
|
+
if (params.model === undefined || params.model === '') return { error: messagesError(400, 'invalid_request_error', "'model' is a required property") }
|
|
917
|
+
// MessagesRequest['model'] enum (kimi-k3 only) — read from the catalog's own supports flag.
|
|
918
|
+
if (!findModel(String(params.model))?.supports.messages) {
|
|
919
|
+
return { error: messagesError(400, 'invalid_request_error', `The endpoint currently supports 'kimi-k3' only (got '${String(params.model)}')`) }
|
|
920
|
+
}
|
|
921
|
+
if (!Array.isArray(params.messages)) return { error: messagesError(400, 'invalid_request_error', "'messages' is a required property") }
|
|
922
|
+
if (params.messages.length === 0) return { error: messagesError(400, 'invalid_request_error', "'messages' must not be empty") }
|
|
923
|
+
const messages: Array<{ role: 'user' | 'assistant'; content: string | Array<Record<string, unknown>> }> = [];
|
|
924
|
+
for (const m of params.messages as Array<Record<string, unknown>>) {
|
|
925
|
+
if (!m || typeof m !== 'object') return { error: messagesError(400, 'invalid_request_error', 'each message must be an object') }
|
|
926
|
+
if (m.role !== 'user' && m.role !== 'assistant') {
|
|
927
|
+
// Anthropic's grammar: the top-level `system` field owns the system prompt.
|
|
928
|
+
return { error: messagesError(400, 'invalid_request_error', `'${String(m.role)}' is not one of ['user', 'assistant'] — use the top-level 'system' field for the system prompt`) }
|
|
929
|
+
}
|
|
930
|
+
messages.push({ role: m.role, content: m.content as string | Array<Record<string, unknown>> });
|
|
931
|
+
}
|
|
932
|
+
if (params.max_tokens === undefined || params.max_tokens === null) return { error: messagesError(400, 'invalid_request_error', "'max_tokens' is a required property") }
|
|
933
|
+
const maxTokens = Number(params.max_tokens);
|
|
934
|
+
if (!Number.isInteger(maxTokens) || maxTokens < 1) return { error: messagesError(400, 'invalid_request_error', "'max_tokens' must be an integer >= 1") }
|
|
935
|
+
let system: string | undefined;
|
|
936
|
+
if (params.system !== undefined) {
|
|
937
|
+
if (typeof params.system === 'string') system = params.system;
|
|
938
|
+
else if (Array.isArray(params.system)) system = params.system.map((b) => String((b as { text?: unknown })?.text ?? '')).join('\n');
|
|
939
|
+
else return { error: messagesError(400, 'invalid_request_error', "'system' must be a string or an array of text blocks") }
|
|
940
|
+
}
|
|
941
|
+
let stopSequences: string[] | undefined;
|
|
942
|
+
if (params.stop_sequences !== undefined) {
|
|
943
|
+
if (!Array.isArray(params.stop_sequences)) return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' must be an array of strings") }
|
|
944
|
+
if (params.stop_sequences.length > 5) return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' must not exceed 5 entries") }
|
|
945
|
+
for (const s of params.stop_sequences) {
|
|
946
|
+
if (typeof s !== 'string') return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' entries must be strings") }
|
|
947
|
+
if (Buffer.byteLength(s, 'utf8') > 32) return { error: messagesError(400, 'invalid_request_error', "'stop_sequences' entries must not exceed 32 bytes") }
|
|
948
|
+
}
|
|
949
|
+
stopSequences = params.stop_sequences;
|
|
950
|
+
}
|
|
951
|
+
// Tools: Anthropic's MessagesTool shape, with the SAME documented name regex the /v1 surface
|
|
952
|
+
// enforces (a bad tool name is a 400 here too — never a silent acceptance the stub then
|
|
953
|
+
// ignores).
|
|
954
|
+
let tools: Array<Record<string, unknown>> | undefined;
|
|
955
|
+
if (params.tools !== undefined) {
|
|
956
|
+
if (!Array.isArray(params.tools)) return { error: messagesError(400, 'invalid_request_error', "'tools' must be an array") }
|
|
957
|
+
for (const t of params.tools) {
|
|
958
|
+
if (!t || typeof t !== 'object' || Array.isArray(t)) return { error: messagesError(400, 'invalid_request_error', 'each tool must be an object') }
|
|
959
|
+
const name = messagesToolName(t);
|
|
960
|
+
if (!name || !TOOL_NAME_RE.test(name)) {
|
|
961
|
+
return { error: messagesError(400, 'invalid_request_error', `'tools[].name' must match ${TOOL_NAME_RE.source}`) }
|
|
962
|
+
}
|
|
963
|
+
const schema = (t as { input_schema?: unknown }).input_schema;
|
|
964
|
+
if (schema !== undefined && (!schema || typeof schema !== 'object' || Array.isArray(schema))) {
|
|
965
|
+
return { error: messagesError(400, 'invalid_request_error', "'tools[].input_schema' must be an object") }
|
|
966
|
+
}
|
|
967
|
+
}
|
|
968
|
+
tools = params.tools as Array<Record<string, unknown>>;
|
|
969
|
+
}
|
|
970
|
+
// tool_choice: Anthropic's closed type set INCLUDING type:'tool' with its required `name`
|
|
971
|
+
// (the sibling pack's rule: a named choice must name a tool that is actually in `tools`).
|
|
972
|
+
let toolChoice: MessagesArgs['toolChoice'] | undefined;
|
|
973
|
+
if (params.tool_choice !== undefined) {
|
|
974
|
+
const t = (params.tool_choice as { type?: unknown })?.type;
|
|
975
|
+
if (typeof t !== 'string' || !['auto', 'any', 'tool', 'none'].includes(t)) return { error: messagesError(400, 'invalid_request_error', "'tool_choice.type' must be one of 'auto', 'any', 'tool', 'none'") }
|
|
976
|
+
// A tool_choice WITHOUT tools is a contradiction — Anthropic's own surface refuses it.
|
|
977
|
+
if (!tools) return { error: messagesError(400, 'invalid_request_error', "'tool_choice' requires 'tools' to be set") }
|
|
978
|
+
if (t === 'tool') {
|
|
979
|
+
const name = (params.tool_choice as { name?: unknown }).name;
|
|
980
|
+
if (typeof name !== 'string' || !tools.some((tool) => messagesToolName(tool) === name)) {
|
|
981
|
+
return { error: messagesError(400, 'invalid_request_error', `'tool_choice' names tool '${typeof name === 'string' ? name : String(name)}', which is not in tools`) }
|
|
982
|
+
}
|
|
983
|
+
toolChoice = { type: 'tool', name };
|
|
984
|
+
} else {
|
|
985
|
+
toolChoice = { type: t as 'auto' | 'any' | 'none' };
|
|
986
|
+
}
|
|
987
|
+
}
|
|
988
|
+
let outputEffort: string | undefined;
|
|
989
|
+
const oc = params.output_config as { effort?: unknown } | undefined;
|
|
990
|
+
if (oc && typeof oc === 'object' && oc.effort !== undefined) {
|
|
991
|
+
if (typeof oc.effort !== 'string' || !REASONING_EFFORTS.includes(oc.effort as ReasoningEffort)) {
|
|
992
|
+
return { error: messagesError(400, 'invalid_request_error', `'output_config.effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) }
|
|
993
|
+
}
|
|
994
|
+
outputEffort = oc.effort;
|
|
995
|
+
}
|
|
996
|
+
return {
|
|
997
|
+
args: {
|
|
998
|
+
model: String(params.model),
|
|
999
|
+
messages,
|
|
1000
|
+
maxTokens,
|
|
1001
|
+
...(system !== undefined ? { system } : {}),
|
|
1002
|
+
...(stopSequences !== undefined ? { stopSequences } : {}),
|
|
1003
|
+
stream: params.stream === true,
|
|
1004
|
+
...(tools !== undefined ? { tools } : {}),
|
|
1005
|
+
...(toolChoice !== undefined ? { toolChoice } : {}),
|
|
1006
|
+
...(outputEffort !== undefined ? { outputEffort } : {}),
|
|
1007
|
+
},
|
|
1008
|
+
};
|
|
1009
|
+
}
|
|
1010
|
+
|
|
1011
|
+
/** The Messages surface's OWN error envelope (MessagesErrorResponse schema) — different from /v1. */
|
|
1012
|
+
function messagesError(status: number, type: string, message: string): MoonshotResponseEnvelope {
|
|
1013
|
+
return { status, body: { type: 'error', error: { type, message } } };
|
|
1014
|
+
}
|
|
1015
|
+
|
|
1016
|
+
/** Re-wear a scripted fault's Moonshot-shaped body in the Anthropic envelope for the /anthropic
|
|
1017
|
+
* surface: the fault result carries the Moonshot error body the adapter rendered; this surface
|
|
1018
|
+
* serves {type:'error', error:{type,message}} with the equivalent Anthropic type. */
|
|
1019
|
+
function messagesErrorBody(status: number, moonshotBody: unknown): unknown {
|
|
1020
|
+
const e = (moonshotBody as { error?: { message?: string; type?: string } })?.error;
|
|
1021
|
+
const message = e?.message ?? `The request was refused by a scripted fault (${status}).`;
|
|
1022
|
+
const type = e?.type === 'rate_limit_reached_error' ? 'rate_limit_error'
|
|
1023
|
+
: e?.type === 'server_unavailable' || e?.type === 'server_error' ? 'api_error'
|
|
1024
|
+
: 'invalid_request_error';
|
|
1025
|
+
return { type: 'error', error: { type, message } };
|
|
1026
|
+
}
|
|
1027
|
+
|
|
1028
|
+
export function buildMessagesResponse(args: MessagesArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotMessagesResponse | MoonshotResponseEnvelope {
|
|
1029
|
+
let scripted: ScriptedResult | null = null;
|
|
1030
|
+
let missTeach = '';
|
|
1031
|
+
if (decision) {
|
|
1032
|
+
// The route served the request through the engine (R15); tools/thinking rode into the
|
|
1033
|
+
// match there — a handler keyed on `hasTool`/`toolResultFor` fires on this surface exactly
|
|
1034
|
+
// as it fires on /v1 (the sibling pack wires the same request features through on its
|
|
1035
|
+
// Messages surface).
|
|
1036
|
+
if (decision.kind === 'handler') {
|
|
1037
|
+
const respond = decision.respond as MoonshotScenarioRespond;
|
|
1038
|
+
if (respond.error) return scriptedMessagesError(respond.error);
|
|
1039
|
+
scripted = realizeMoonshotRespond(respond);
|
|
1040
|
+
} else {
|
|
1041
|
+
missTeach = `\n[twin-scenario miss — no handler matched. Author one in handlers/moonshot.json.]`;
|
|
1042
|
+
}
|
|
1043
|
+
}
|
|
1044
|
+
const inputTokens = countPromptTokens(args.messages.map((m) => ({ role: m.role, content: m.content as never })) as MoonshotMessageParam[]);
|
|
1045
|
+
let content: MoonshotMessagesResponse['content'];
|
|
1046
|
+
let stopReason: MoonshotMessagesResponse['stop_reason'];
|
|
1047
|
+
let outputTokens: number;
|
|
1048
|
+
if (scripted) {
|
|
1049
|
+
content = [];
|
|
1050
|
+
if (scripted.reasoning !== null) content.push({ type: 'thinking', thinking: scripted.reasoning, signature: stubSignature(args.messages, args.model) });
|
|
1051
|
+
if (scripted.text) content.push({ type: 'text', text: scripted.text + (missTeach && scripted.text ? missTeach : '') });
|
|
1052
|
+
for (const tc of scripted.toolCalls) {
|
|
1053
|
+
content.push({ type: 'tool_use', id: tc.id, name: tc.function.name, input: JSON.parse(tc.function.arguments) as Record<string, unknown> });
|
|
1054
|
+
}
|
|
1055
|
+
stopReason = scripted.finishReason === 'tool_calls' ? 'tool_use' : scripted.finishReason === 'length' ? 'max_tokens' : 'end_turn';
|
|
1056
|
+
} else {
|
|
1057
|
+
// Tools on the Messages surface are REAL surface: tool_choice 'any'/'auto' (with tools
|
|
1058
|
+
// present) yields a tool_use block the SDK's own decoder reads; a named 'tool' choice calls
|
|
1059
|
+
// THAT tool; 'none' suppresses. The tautology that silently answered text regardless (the
|
|
1060
|
+
// round-two finding) is gone.
|
|
1061
|
+
const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
|
|
1062
|
+
const choice = args.toolChoice?.type ?? 'auto';
|
|
1063
|
+
const suppress = choice === 'none';
|
|
1064
|
+
const forcedName = choice === 'tool' ? args.toolChoice?.name : undefined;
|
|
1065
|
+
// A tool_result turn is ANSWERED, not re-tooled: once the client returns the tool_result the
|
|
1066
|
+
// model's next turn is text with stop_reason end_turn — deciding tool_use from tools.length
|
|
1067
|
+
// alone made an agent loop emit tool_use forever (the round-three finding). An EXPLICIT
|
|
1068
|
+
// forced choice ('any'/'tool') still calls, because the caller demanded one.
|
|
1069
|
+
const last = args.messages[args.messages.length - 1];
|
|
1070
|
+
const toolResultTurn = last?.role === 'user' && Array.isArray(last.content)
|
|
1071
|
+
&& last.content.some((b) => (b as { type?: string })?.type === 'tool_result');
|
|
1072
|
+
const forceTool = choice === 'any' || choice === 'tool';
|
|
1073
|
+
const toolCalls = hasTools && !suppress && (forceTool || !toolResultTurn)
|
|
1074
|
+
? [stubToolCall(args.tools, 1, forcedName)].filter((tc): tc is MoonshotToolCall => tc !== null)
|
|
1075
|
+
: [];
|
|
1076
|
+
// kimi-k3 ALWAYS thinks (Preserved Thinking — the same rule the /v1 surface applies), and
|
|
1077
|
+
// Moonshot's Messages doc has NO thinking parameter: reasoning is tuned through
|
|
1078
|
+
// output_config.effort (default max) and the content array is documented
|
|
1079
|
+
// "ordered thinking → text → tool_use" (platform.kimi.ai/docs/api/messages, read
|
|
1080
|
+
// 2026-09-16) — so the thinking block leads unconditionally, like the vendor's.
|
|
1081
|
+
content = [
|
|
1082
|
+
{ type: 'thinking', thinking: stubReasoningContent(args.messages.map((m) => ({ role: m.role, content: m.content })) as MoonshotMessageParam[], args.model), signature: stubSignature(args.messages, args.model) },
|
|
1083
|
+
];
|
|
1084
|
+
if (toolCalls.length) {
|
|
1085
|
+
toolCalls.forEach((tc, seq) => {
|
|
1086
|
+
// Anthropic's tool_use id prefix is `toolu_` (the sibling pack's grammar —
|
|
1087
|
+
// `toolu_twin_${seq}`), NOT the `call_` prefix the OpenAI-compatible /v1 surface uses.
|
|
1088
|
+
content.push({ type: 'tool_use', id: `toolu_twin_${seq + 1}`, name: tc.function.name, input: JSON.parse(tc.function.arguments) as Record<string, unknown> });
|
|
1089
|
+
});
|
|
1090
|
+
} else {
|
|
1091
|
+
content.push({ type: 'text', text: stubAssistantText(args.messages.map((m) => ({ role: m.role, content: m.content })) as MoonshotMessageParam[], args.model) + (missTeach || '') });
|
|
1092
|
+
}
|
|
1093
|
+
stopReason = toolCalls.length ? 'tool_use' : 'end_turn';
|
|
1094
|
+
}
|
|
1095
|
+
// max_tokens caps the OUTPUT ITSELF, not merely the report: truncate the emitted blocks to the
|
|
1096
|
+
// cap and count usage off what remains (the vendor stops emitting at the cap — usage.output_tokens
|
|
1097
|
+
// is what was emitted, never a fabricated agreement with a text block it did not produce).
|
|
1098
|
+
// Tokens are counted per-block off the block's own payload (thinking/text/tool input), NOT off
|
|
1099
|
+
// JSON.stringify of the whole array — JSON scaffolding would keep a capped answer above its cap.
|
|
1100
|
+
const blockTokens = (b: MoonshotMessagesResponse['content'][number]): number =>
|
|
1101
|
+
b.type === 'thinking' ? estimateTokens(b.thinking)
|
|
1102
|
+
: b.type === 'text' ? estimateTokens(b.text)
|
|
1103
|
+
: estimateTokens(JSON.stringify(b.input));
|
|
1104
|
+
const totalTokens = () => content.reduce((n, b) => n + blockTokens(b), 0);
|
|
1105
|
+
if (args.maxTokens !== undefined && totalTokens() > args.maxTokens) {
|
|
1106
|
+
// THE CAP WALKS THE BLOCKS IN ORDER (the sibling pack's method): the emitted content is a
|
|
1107
|
+
// PREFIX of what the stub would have said. Each block is kept only as far as the remaining
|
|
1108
|
+
// budget allows — text and thinking truncate mid-string, a tool_use that does not fit ends
|
|
1109
|
+
// the turn (everything after it is never emitted). Anthropic NEVER returns a zero-block
|
|
1110
|
+
// assistant message: the FIRST block is kept truncated to at least one token, so
|
|
1111
|
+
// output_tokens > 0 and content is non-empty at any cap >= 1. The walk is BOUNDED — one
|
|
1112
|
+
// pass, one truncation — so no estimate can spin it.
|
|
1113
|
+
let spent = 0;
|
|
1114
|
+
let cut = false;
|
|
1115
|
+
for (let i = 0; i < content.length; i++) {
|
|
1116
|
+
const block = content[i]!;
|
|
1117
|
+
const cost = blockTokens(block);
|
|
1118
|
+
if (spent + cost <= args.maxTokens) {
|
|
1119
|
+
spent += cost;
|
|
1120
|
+
continue;
|
|
1121
|
+
}
|
|
1122
|
+
const budget = args.maxTokens - spent;
|
|
1123
|
+
cut = true;
|
|
1124
|
+
if ((block.type === 'text' || block.type === 'thinking') && (budget >= 1 || i === 0)) {
|
|
1125
|
+
// Truncate to the remaining budget (~4 chars/token), keeping at least one token so the
|
|
1126
|
+
// message never loses its last block — the floor a zero-block message is forbidden by.
|
|
1127
|
+
const keep = Math.max(1, budget);
|
|
1128
|
+
if (block.type === 'text') block.text = block.text.slice(0, keep * 4);
|
|
1129
|
+
else block.thinking = block.thinking.slice(0, keep * 4);
|
|
1130
|
+
spent += blockTokens(block);
|
|
1131
|
+
content.length = i + 1;
|
|
1132
|
+
break;
|
|
1133
|
+
}
|
|
1134
|
+
// The budget ran out exactly at this block's boundary, or the block is a tool_use that
|
|
1135
|
+
// cannot be cut: the block was never emitted. A retained whole block here shipped 57
|
|
1136
|
+
// tokens under max_tokens 23 (round-four review) — the cap is a prefix, never a rounding.
|
|
1137
|
+
content.length = i;
|
|
1138
|
+
if (content.length === 0) content.push({ type: 'text', text: '' });
|
|
1139
|
+
break;
|
|
1140
|
+
}
|
|
1141
|
+
outputTokens = spent;
|
|
1142
|
+
if (cut) stopReason = 'max_tokens';
|
|
1143
|
+
} else {
|
|
1144
|
+
outputTokens = totalTokens();
|
|
1145
|
+
}
|
|
1146
|
+
// A stop_sequence match truncates the text at the hit and reports stop_reason 'stop_sequence'
|
|
1147
|
+
// with the matched string (Anthropic's documented shape — 'end_turn' would misreport WHY the
|
|
1148
|
+
// turn ended). The RECOUNT happens AFTER the truncation: usage.output_tokens is what was
|
|
1149
|
+
// emitted, so a 113-char answer reported as 54 tokens (counted before the cut) contradicted
|
|
1150
|
+
// its own usage (the round-three finding).
|
|
1151
|
+
let stopSequence: string | null = null;
|
|
1152
|
+
if (args.stopSequences?.length) {
|
|
1153
|
+
const textBlock = content.find((b) => b.type === 'text');
|
|
1154
|
+
if (textBlock && textBlock.type === 'text') {
|
|
1155
|
+
for (const s of args.stopSequences) {
|
|
1156
|
+
const i = textBlock.text.indexOf(s);
|
|
1157
|
+
if (i >= 0) {
|
|
1158
|
+
textBlock.text = textBlock.text.slice(0, i);
|
|
1159
|
+
stopSequence = s;
|
|
1160
|
+
stopReason = 'stop_sequence';
|
|
1161
|
+
break;
|
|
1162
|
+
}
|
|
1163
|
+
}
|
|
1164
|
+
outputTokens = totalTokens();
|
|
1165
|
+
}
|
|
1166
|
+
}
|
|
1167
|
+
return {
|
|
1168
|
+
id: `msg_twin_${stableSuffix(args.messages, args.model)}`,
|
|
1169
|
+
type: 'message',
|
|
1170
|
+
role: 'assistant',
|
|
1171
|
+
model: args.model,
|
|
1172
|
+
content,
|
|
1173
|
+
stop_reason: stopReason,
|
|
1174
|
+
stop_sequence: stopSequence,
|
|
1175
|
+
usage: {
|
|
1176
|
+
input_tokens: inputTokens,
|
|
1177
|
+
output_tokens: outputTokens,
|
|
1178
|
+
cache_read_input_tokens: stubCachedTokens(inputTokens),
|
|
1179
|
+
cache_creation_input_tokens: 0,
|
|
1180
|
+
},
|
|
1181
|
+
};
|
|
1182
|
+
}
|
|
1183
|
+
|
|
1184
|
+
function scriptedMessagesError(err: NonNullable<MoonshotScenarioRespond['error']>): MoonshotResponseEnvelope {
|
|
1185
|
+
if (err.type === 'rate_limit_reached_error') return messagesError(429, 'rate_limit_reached_error', err.message ?? 'Request rate limit reached');
|
|
1186
|
+
if (err.type === 'server_unavailable') return messagesError(503, 'server_unavailable', err.message ?? 'The server is overloaded');
|
|
1187
|
+
return messagesError(500, 'server_error', err.message ?? 'Internal Server Error');
|
|
1188
|
+
}
|
|
1189
|
+
|
|
1190
|
+
/**
|
|
1191
|
+
* Emit the Anthropic-compatible streaming grammar into the injected sink: message_start →
|
|
1192
|
+
* ping → content_block_start(thinking) → thinking_delta → signature_delta → content_block_stop →
|
|
1193
|
+
* content_block_start(text) → text_delta → content_block_stop → message_delta(stop_reason) →
|
|
1194
|
+
* message_stop. No [DONE] sentinel — the stream ends after message_stop.
|
|
1195
|
+
*
|
|
1196
|
+
* message_start carries the EMPTY message (content: [], stop_reason: null, output_tokens small)
|
|
1197
|
+
* and the blocks stream in one at a time — the real grammar (and the sibling pack's frames). A
|
|
1198
|
+
* message_start carrying the COMPLETE final message made the SDK's own accumulator double every
|
|
1199
|
+
* block: it pushes one block per content_block_start ON TOP of what message_start already held
|
|
1200
|
+
* (the round-three BLOCKER).
|
|
1201
|
+
*/
|
|
1202
|
+
export function streamMessages(args: MessagesArgs, sink: MessagesSseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotMessagesResponse | MoonshotResponseEnvelope {
|
|
1203
|
+
const built = buildMessagesResponse(args, occurredAt, decision);
|
|
1204
|
+
if (isEnvelope(built)) return built;
|
|
1205
|
+
const full = built;
|
|
1206
|
+
sink({
|
|
1207
|
+
event: 'message_start',
|
|
1208
|
+
data: {
|
|
1209
|
+
type: 'message_start',
|
|
1210
|
+
message: {
|
|
1211
|
+
id: full.id, type: 'message', role: 'assistant', model: full.model,
|
|
1212
|
+
content: [], stop_reason: null, stop_sequence: null,
|
|
1213
|
+
usage: { input_tokens: full.usage.input_tokens, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 },
|
|
1214
|
+
},
|
|
1215
|
+
},
|
|
1216
|
+
});
|
|
1217
|
+
// The heartbeat the real API interleaves (and the sibling pack emits); the SDK's SSE decoder
|
|
1218
|
+
// skips `event: ping` frames by name, so it is inert to every consumer.
|
|
1219
|
+
sink({ event: 'ping', data: { type: 'ping' } });
|
|
1220
|
+
full.content.forEach((block, index) => {
|
|
1221
|
+
if (block.type === 'thinking') {
|
|
1222
|
+
sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'thinking', thinking: '' } } });
|
|
1223
|
+
for (const piece of chunkText(block.thinking)) sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'thinking_delta', thinking: piece } } });
|
|
1224
|
+
sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'signature_delta', signature: block.signature ?? '' } } });
|
|
1225
|
+
sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
|
|
1226
|
+
} else if (block.type === 'text') {
|
|
1227
|
+
sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'text', text: '' } } });
|
|
1228
|
+
for (const piece of chunkText(block.text)) sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'text_delta', text: piece } } });
|
|
1229
|
+
sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
|
|
1230
|
+
} else {
|
|
1231
|
+
sink({ event: 'content_block_start', data: { type: 'content_block_start', index, content_block: { type: 'tool_use', id: block.id, name: block.name, input: {} } } });
|
|
1232
|
+
const json = JSON.stringify(block.input);
|
|
1233
|
+
sink({ event: 'content_block_delta', data: { type: 'content_block_delta', index, delta: { type: 'input_json_delta', partial_json: json } } });
|
|
1234
|
+
sink({ event: 'content_block_stop', data: { type: 'content_block_stop', index } });
|
|
1235
|
+
}
|
|
1236
|
+
});
|
|
1237
|
+
sink({ event: 'message_delta', data: { type: 'message_delta', delta: { stop_reason: full.stop_reason, stop_sequence: full.stop_sequence }, usage: { output_tokens: full.usage.output_tokens } } });
|
|
1238
|
+
sink({ event: 'message_stop', data: { type: 'message_stop' } });
|
|
1239
|
+
sink({ done: true });
|
|
1240
|
+
return full;
|
|
1241
|
+
}
|
|
1242
|
+
|
|
1243
|
+
// ── Responses (/v1/responses) — kimi-k3 only ────────────────────────────────────────────
|
|
1244
|
+
type ResponsesArgs = {
|
|
1245
|
+
model: string;
|
|
1246
|
+
input: string | Array<Record<string, unknown>>;
|
|
1247
|
+
instructions?: string;
|
|
1248
|
+
maxOutputTokens?: number;
|
|
1249
|
+
reasoningEffort?: string;
|
|
1250
|
+
stream: boolean;
|
|
1251
|
+
};
|
|
1252
|
+
|
|
1253
|
+
function validateResponses(params: Record<string, unknown>): { args: ResponsesArgs } | { error: MoonshotResponseEnvelope } {
|
|
1254
|
+
if (params.model === undefined || params.model === '') return { error: invalidRequest("'model' is a required property") }
|
|
1255
|
+
// "This endpoint currently supports `kimi-k3`" (ResponsesRequest.model) — read from the
|
|
1256
|
+
// catalog's own supports flag, so the enum and the catalog cannot drift apart.
|
|
1257
|
+
const responsesModel = findModel(String(params.model));
|
|
1258
|
+
if (!responsesModel?.supports.responses) return { error: invalidRequest(`This endpoint currently supports 'kimi-k3' only (got '${String(params.model)}')`) }
|
|
1259
|
+
if (params.input === undefined || params.input === null) return { error: invalidRequest("'input' is a required property") }
|
|
1260
|
+
if (typeof params.input !== 'string' && !Array.isArray(params.input)) return { error: invalidRequest("'input' must be a string or an array of items") }
|
|
1261
|
+
let reasoningEffort: string | undefined;
|
|
1262
|
+
const r = params.reasoning as { effort?: unknown } | undefined;
|
|
1263
|
+
if (r && typeof r === 'object' && r.effort !== undefined) {
|
|
1264
|
+
if (typeof r.effort !== 'string' || !REASONING_EFFORTS.includes(r.effort as ReasoningEffort)) {
|
|
1265
|
+
return { error: invalidRequest(`'reasoning.effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) }
|
|
1266
|
+
}
|
|
1267
|
+
reasoningEffort = r.effort;
|
|
1268
|
+
}
|
|
1269
|
+
let maxOutputTokens: number | undefined;
|
|
1270
|
+
if (params.max_output_tokens !== undefined) {
|
|
1271
|
+
maxOutputTokens = Number(params.max_output_tokens);
|
|
1272
|
+
if (!Number.isInteger(maxOutputTokens) || maxOutputTokens < 1) return { error: invalidRequest("'max_output_tokens' must be an integer >= 1") }
|
|
1273
|
+
}
|
|
1274
|
+
return {
|
|
1275
|
+
args: {
|
|
1276
|
+
model: String(params.model),
|
|
1277
|
+
input: params.input as string | Array<Record<string, unknown>>,
|
|
1278
|
+
...(typeof params.instructions === 'string' ? { instructions: params.instructions } : {}),
|
|
1279
|
+
...(maxOutputTokens !== undefined ? { maxOutputTokens } : {}),
|
|
1280
|
+
...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
|
|
1281
|
+
stream: params.stream === true,
|
|
1282
|
+
},
|
|
1283
|
+
};
|
|
1284
|
+
}
|
|
1285
|
+
|
|
1286
|
+
function responsesInputText(input: string | Array<Record<string, unknown>>): string {
|
|
1287
|
+
if (typeof input === 'string') return input;
|
|
1288
|
+
return input.map((item) => {
|
|
1289
|
+
if ((item as { type?: string }).type === 'message' || (item as { type?: string }).type === undefined) {
|
|
1290
|
+
const content = (item as { content?: unknown }).content;
|
|
1291
|
+
return contentToText(content);
|
|
1292
|
+
}
|
|
1293
|
+
return JSON.stringify(item);
|
|
1294
|
+
}).join('\n');
|
|
1295
|
+
}
|
|
1296
|
+
|
|
1297
|
+
export function buildResponsesResponse(args: ResponsesArgs, occurredAt?: string, decision?: ScenarioDecision): MoonshotResponsesResponse | MoonshotResponseEnvelope {
|
|
1298
|
+
let scripted: ScriptedResult | null = null;
|
|
1299
|
+
let missTeach = '';
|
|
1300
|
+
if (decision) {
|
|
1301
|
+
// The route served the request through the engine (R15); this realizer only sees the
|
|
1302
|
+
// content decision.
|
|
1303
|
+
if (decision.kind === 'handler') {
|
|
1304
|
+
const respond = decision.respond as MoonshotScenarioRespond;
|
|
1305
|
+
if (respond.error) return scriptedError(respond.error);
|
|
1306
|
+
scripted = realizeMoonshotRespond(respond);
|
|
1307
|
+
} else {
|
|
1308
|
+
missTeach = `\n[twin-scenario miss — no handler matched. Author one in handlers/moonshot.json.]`;
|
|
1309
|
+
}
|
|
1310
|
+
}
|
|
1311
|
+
const inputText = responsesInputText(args.input);
|
|
1312
|
+
const asMessages: MoonshotMessageParam[] = [{ role: 'user', content: inputText }];
|
|
1313
|
+
const inputTokens = countPromptTokens(asMessages);
|
|
1314
|
+
const output: MoonshotResponsesOutputItem[] = [];
|
|
1315
|
+
let outputTokens = 0;
|
|
1316
|
+
if (scripted) {
|
|
1317
|
+
if (scripted.reasoning !== null) {
|
|
1318
|
+
output.push({ type: 'reasoning', id: `rs_twin_${stableSuffix(args.input, 'reasoning')}`, summary: [{ type: 'summary_text', text: scripted.reasoning }], status: 'completed' });
|
|
1319
|
+
outputTokens += estimateTokens(scripted.reasoning);
|
|
1320
|
+
}
|
|
1321
|
+
if (scripted.toolCalls.length) {
|
|
1322
|
+
for (const tc of scripted.toolCalls) {
|
|
1323
|
+
output.push({ type: 'function_call', id: `fc_twin_${stableSuffix(tc)}`, call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments, status: 'completed' });
|
|
1324
|
+
outputTokens += estimateTokens(tc.function.arguments);
|
|
1325
|
+
}
|
|
1326
|
+
} else {
|
|
1327
|
+
const text = (scripted.text ?? '') + (missTeach || '');
|
|
1328
|
+
output.push({ type: 'message', id: `msg_twin_${stableSuffix(args.input, 'message')}`, role: 'assistant', status: 'completed', content: [{ type: 'output_text', text, annotations: [] }] });
|
|
1329
|
+
outputTokens += estimateTokens(text);
|
|
1330
|
+
}
|
|
1331
|
+
} else {
|
|
1332
|
+
output.push({ type: 'reasoning', id: `rs_twin_${stableSuffix(inputText, 'reasoning')}`, summary: [{ type: 'summary_text', text: stubReasoningContent(asMessages, args.model) }], status: 'completed' });
|
|
1333
|
+
const text = stubAssistantText(asMessages, args.model) + (missTeach || '');
|
|
1334
|
+
output.push({ type: 'message', id: `msg_twin_${stableSuffix(inputText, 'message')}`, role: 'assistant', status: 'completed', content: [{ type: 'output_text', text, annotations: [] }] });
|
|
1335
|
+
outputTokens = estimateTokens(stubReasoningContent(asMessages, args.model)) + estimateTokens(text);
|
|
1336
|
+
}
|
|
1337
|
+
let status: MoonshotResponsesResponse['status'] = 'completed';
|
|
1338
|
+
let incompleteDetails: MoonshotResponsesResponse['incomplete_details'] = null;
|
|
1339
|
+
// max_output_tokens caps the OUTPUT ITSELF: the vendor stops emitting at the cap, so truncate
|
|
1340
|
+
// the emitted items (dropping whole items from the tail — the Responses API emits complete
|
|
1341
|
+
// items) and let usage.output_tokens count what remains. Reporting status 'incomplete' with
|
|
1342
|
+
// usage.output_tokens still above the cap was the round-two finding: the envelope contradicted
|
|
1343
|
+
// its own usage.
|
|
1344
|
+
if (args.maxOutputTokens !== undefined && outputTokens > args.maxOutputTokens) {
|
|
1345
|
+
// THE CAP IS A PREFIX, LIKE THE MESSAGES SURFACE'S: items are emitted in order until the
|
|
1346
|
+
// budget is exhausted; a text-carrying item truncates to the remaining budget rather than
|
|
1347
|
+
// vanishing, so a tight cap still yields a non-empty output with output_tokens > 0 — a
|
|
1348
|
+
// zero-item `output: []` (the round-three finding at max_output_tokens 2) is a message the
|
|
1349
|
+
// vendor never sends. One bounded pass; nothing after the cut is emitted.
|
|
1350
|
+
let spent = 0;
|
|
1351
|
+
let cut = false;
|
|
1352
|
+
for (let i = 0; i < output.length; i++) {
|
|
1353
|
+
const item = output[i]!;
|
|
1354
|
+
const cost = estimateTokens(
|
|
1355
|
+
item.type === 'message' ? (item.content as Array<{ text?: string }>).map((c) => c.text ?? '').join('')
|
|
1356
|
+
: item.type === 'reasoning' ? (item.summary as Array<{ text?: string }>).map((s) => s.text ?? '').join('')
|
|
1357
|
+
: item.type === 'function_call' ? item.arguments : item.type === 'custom_tool_call' ? item.input : '',
|
|
1358
|
+
);
|
|
1359
|
+
if (spent + cost <= args.maxOutputTokens) {
|
|
1360
|
+
spent += cost;
|
|
1361
|
+
continue;
|
|
1362
|
+
}
|
|
1363
|
+
const budget = args.maxOutputTokens - spent;
|
|
1364
|
+
const keep = Math.max(1, budget) * 4;
|
|
1365
|
+
if ((item.type === 'message' || item.type === 'reasoning') && budget >= 1) {
|
|
1366
|
+
if (item.type === 'message') item.content[0]!.text = (item.content[0]!.text ?? '').slice(0, keep);
|
|
1367
|
+
else item.summary[0]!.text = (item.summary[0]!.text ?? '').slice(0, keep);
|
|
1368
|
+
spent += estimateTokens(item.type === 'message' ? item.content[0]!.text ?? '' : item.summary[0]!.text ?? '');
|
|
1369
|
+
}
|
|
1370
|
+
output.length = i + 1;
|
|
1371
|
+
cut = true;
|
|
1372
|
+
break;
|
|
1373
|
+
}
|
|
1374
|
+
outputTokens = spent;
|
|
1375
|
+
if (cut) {
|
|
1376
|
+
status = 'incomplete';
|
|
1377
|
+
incompleteDetails = { reason: 'max_output_tokens' };
|
|
1378
|
+
}
|
|
1379
|
+
}
|
|
1380
|
+
return {
|
|
1381
|
+
id: `resp_twin_${stableSuffix(inputText, args.model)}`,
|
|
1382
|
+
object: 'response',
|
|
1383
|
+
created_at: nowEpoch(occurredAt),
|
|
1384
|
+
completed_at: status === 'completed' || status === 'incomplete' ? nowEpoch(occurredAt) : null,
|
|
1385
|
+
status,
|
|
1386
|
+
model: args.model,
|
|
1387
|
+
output,
|
|
1388
|
+
usage: {
|
|
1389
|
+
input_tokens: inputTokens,
|
|
1390
|
+
input_tokens_details: { cached_tokens: stubCachedTokens(inputTokens), cache_write_tokens: 0 },
|
|
1391
|
+
output_tokens: outputTokens,
|
|
1392
|
+
// reasoning_tokens counts the REASONING items only (the /anthropic thinking_tokens path
|
|
1393
|
+
// does the same): reporting the whole output_tokens as reasoning claimed every text token
|
|
1394
|
+
// was reasoning (the round-three finding).
|
|
1395
|
+
output_tokens_details: {
|
|
1396
|
+
reasoning_tokens: output
|
|
1397
|
+
.filter((o) => o.type === 'reasoning')
|
|
1398
|
+
.reduce((n, o) => n + estimateTokens((o.summary as Array<{ text?: string }>).map((s) => s.text ?? '').join('')), 0),
|
|
1399
|
+
},
|
|
1400
|
+
total_tokens: inputTokens + outputTokens,
|
|
1401
|
+
},
|
|
1402
|
+
incomplete_details: incompleteDetails,
|
|
1403
|
+
error: null,
|
|
1404
|
+
store: false,
|
|
1405
|
+
};
|
|
1406
|
+
}
|
|
1407
|
+
|
|
1408
|
+
/**
|
|
1409
|
+
* Emit the Responses SSE grammar into the injected sink: response.created →
|
|
1410
|
+
* response.in_progress → output_item.added/done per item → response.completed. Each frame
|
|
1411
|
+
* carries `event: <type>` and a monotonically increasing `sequence_number` from 0.
|
|
1412
|
+
*/
|
|
1413
|
+
export function streamResponses(args: ResponsesArgs, sink: MessagesSseSink, occurredAt?: string, decision?: ScenarioDecision): MoonshotResponsesResponse | MoonshotResponseEnvelope {
|
|
1414
|
+
const built = buildResponsesResponse(args, occurredAt, decision);
|
|
1415
|
+
if (isEnvelope(built)) return built;
|
|
1416
|
+
const full = built;
|
|
1417
|
+
let seq = 0;
|
|
1418
|
+
const emit = (type: string, data: Record<string, unknown>) => sink({ event: type, data: { type, sequence_number: seq++, ...data } });
|
|
1419
|
+
emit('response.created', { response: { ...full, status: 'in_progress', output: [], usage: null } });
|
|
1420
|
+
emit('response.in_progress', { response: { ...full, status: 'in_progress', output: [], usage: null } });
|
|
1421
|
+
for (const item of full.output) {
|
|
1422
|
+
emit('response.output_item.added', { output_index: full.output.indexOf(item), output_item: item });
|
|
1423
|
+
emit('response.output_item.done', { output_index: full.output.indexOf(item), output_item: item });
|
|
1424
|
+
}
|
|
1425
|
+
emit(full.status === 'completed' ? 'response.completed' : 'response.incomplete', { response: full });
|
|
1426
|
+
sink({ done: true });
|
|
1427
|
+
return full;
|
|
1428
|
+
}
|
|
1429
|
+
|
|
1430
|
+
// ── public entry: cross-cutting protocol (auth / rate-limit) then route ─────────────────
|
|
1431
|
+
export async function handleMoonshotTwinRequest(req: MoonshotRequest): Promise<MoonshotResponseEnvelope> {
|
|
1432
|
+
const method = req.method.toUpperCase();
|
|
1433
|
+
// EVERY cross-cutting failure answers in the prefix's OWN envelope: the /anthropic surface's
|
|
1434
|
+
// documented client decodes Anthropic's {type:'error',error:{…}} grammar, so a 401/429/503
|
|
1435
|
+
// there carrying the /v1 envelope would be unparseable to it (the clone's lie, cross-cutting
|
|
1436
|
+
// edition). checkAuth applies this itself; the two deterministic fault triggers follow.
|
|
1437
|
+
const onMessages = onMessagesPath(req.path);
|
|
1438
|
+
if (req.headers !== undefined || req.apiKey !== undefined) {
|
|
1439
|
+
const authErr = checkAuth(req);
|
|
1440
|
+
if (authErr) return authErr;
|
|
1441
|
+
}
|
|
1442
|
+
if (triggered(req, 'x-twin-force-rate-limit')) {
|
|
1443
|
+
const base = rateLimitError();
|
|
1444
|
+
// The HEADERS ride along on the re-enveloped 429: retry-after and the X-RateLimit-* family
|
|
1445
|
+
// are part of the refusal a real client reads, whatever envelope the body wears — dropping
|
|
1446
|
+
// them on /anthropic left the documented SDK blind to the back-off (the round-three finding).
|
|
1447
|
+
return onMessages
|
|
1448
|
+
? { ...messagesError(429, 'rate_limit_error', (base.body as { error: { message: string } }).error.message), headers: base.headers }
|
|
1449
|
+
: base;
|
|
1450
|
+
}
|
|
1451
|
+
if (triggered(req, 'x-twin-force-server-unavailable')) {
|
|
1452
|
+
const base = serverUnavailable();
|
|
1453
|
+
return onMessages ? messagesError(503, 'overloaded_error', (base.body as { error: { message: string } }).error.message) : base;
|
|
1454
|
+
}
|
|
1455
|
+
const res = await routeMoonshot(req, method);
|
|
1456
|
+
// Moonshot's request-signing contract, in the HANDLER so every surface (server, fetch
|
|
1457
|
+
// adapter, in-process callers) signs identically (§9 round two, F8): a STREAMED model call
|
|
1458
|
+
// that carries `X-Msh-Request-Nonce` gets `Msh-Request-Timestamp` + `Msh-Request-Signature`
|
|
1459
|
+
// response headers; a nonce-less call proceeds unsigned. The server forwards these headers
|
|
1460
|
+
// on the SSE response it frames.
|
|
1461
|
+
const nonce = req.headers?.['x-msh-request-nonce'];
|
|
1462
|
+
if (nonce && req.sseSink && req.method.toUpperCase() === 'POST' && (req.path === `${MOONSHOT_API_PREFIX}/chat/completions`)) {
|
|
1463
|
+
const model = (() => { try { return typeof (JSON.parse(req.body ?? '') as { model?: unknown }).model === 'string' ? (JSON.parse(req.body ?? '') as { model: string }).model : null; } catch { return null; } })();
|
|
1464
|
+
if (model) {
|
|
1465
|
+
const ts = Date.parse(req.occurredAt ?? worldNow());
|
|
1466
|
+
return { ...res, headers: { ...(res.headers ?? {}), 'msh-request-timestamp': String(ts), 'msh-request-signature': messagesSignature(nonce, ts, model) } };
|
|
1467
|
+
}
|
|
1468
|
+
}
|
|
1469
|
+
return res;
|
|
1470
|
+
}
|
|
1471
|
+
|
|
1472
|
+
// ── router ──────────────────────────────────────────────────────────────────────────────
|
|
1473
|
+
async function routeMoonshot(req: MoonshotRequest, method: string): Promise<MoonshotResponseEnvelope> {
|
|
1474
|
+
const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
|
|
1475
|
+
const params = parseJson(req.body);
|
|
1476
|
+
const dec = (s: string) => decodeURIComponent(s);
|
|
1477
|
+
|
|
1478
|
+
const onOpenai = path === MOONSHOT_API_PREFIX || path.startsWith(`${MOONSHOT_API_PREFIX}/`);
|
|
1479
|
+
const onMessages = path === MESSAGES_PREFIX || path.startsWith(`${MESSAGES_PREFIX}/`);
|
|
1480
|
+
if (!onOpenai && !onMessages) {
|
|
1481
|
+
return notFound(`Unknown request URL: ${method} ${path}.`);
|
|
1482
|
+
}
|
|
1483
|
+
|
|
1484
|
+
// D3: a read-only twin rejects any mutation with a vendor-shaped error — in the prefix's OWN
|
|
1485
|
+
// envelope (the Messages surface decodes Anthropic's grammar, not /v1's), so a read-only twin
|
|
1486
|
+
// never hands the anthropic SDK an envelope it cannot parse. Computed AFTER the prefix so the
|
|
1487
|
+
// envelope follows the path, not the call order.
|
|
1488
|
+
if (req.readOnly && method !== 'GET') {
|
|
1489
|
+
const message = 'twin is read-only; omit readOnly to accept writes';
|
|
1490
|
+
return onMessages ? messagesError(405, 'invalid_request_error', message) : { status: 405, body: errBody('invalid_request_error', message) };
|
|
1491
|
+
}
|
|
1492
|
+
|
|
1493
|
+
// ---- Anthropic-compatible Messages surface ----
|
|
1494
|
+
if (onMessages) {
|
|
1495
|
+
if (path !== `${MESSAGES_PREFIX}/messages`) return messagesError(404, 'not_found_error', `Unknown request URL: ${method} ${path}.`);
|
|
1496
|
+
if (method !== 'POST') return messagesError(405, 'invalid_request_error', `${method} is not supported on /anthropic/v1/messages`);
|
|
1497
|
+
const validated = validateMessages(params);
|
|
1498
|
+
if ('error' in validated) return validated.error;
|
|
1499
|
+
const args = validated.args;
|
|
1500
|
+
// R15 — the scenario engine decides AND honors a fault here, before any message exists:
|
|
1501
|
+
// a `status` fault is this vendor's own refusal envelope re-worn in the Anthropic shape,
|
|
1502
|
+
// a `slow` has already held the answer, a `drop` never returns.
|
|
1503
|
+
let messagesDecision: ScenarioDecision | undefined;
|
|
1504
|
+
if (req.scenarioEngine) {
|
|
1505
|
+
const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages.map((m) => ({ role: m.role, content: m.content })) as MoonshotMessageParam[], tools: args.tools, thinking: undefined });
|
|
1506
|
+
if (served.kind === 'fault') return { status: served.result.status, body: messagesErrorBody(served.result.status, served.result.body), headers: served.result.headers };
|
|
1507
|
+
messagesDecision = served;
|
|
1508
|
+
}
|
|
1509
|
+
const result = args.stream && req.messagesSseSink
|
|
1510
|
+
? streamMessages(args, req.messagesSseSink, req.occurredAt, messagesDecision)
|
|
1511
|
+
: buildMessagesResponse(args, req.occurredAt, messagesDecision);
|
|
1512
|
+
if (isEnvelope(result)) return result;
|
|
1513
|
+
return { status: 200, body: result };
|
|
1514
|
+
}
|
|
1515
|
+
|
|
1516
|
+
const seg = path.slice(MOONSHOT_API_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean); // ["chat","completions"]
|
|
1517
|
+
// protocol 2: after a landing adopted Moonshot's id for a file or a batch, a caller on a branch
|
|
1518
|
+
// may still address it by the local id — resolved through the alias map once, here at the boundary
|
|
1519
|
+
if (seg[0] === 'files' && seg[1]) seg[1] = resolveSubjectId(SERVICE, 'file', dec(seg[1]!), req.root);
|
|
1520
|
+
if (seg[0] === 'batches' && seg[1]) seg[1] = resolveSubjectId(SERVICE, 'batch', dec(seg[1]!), req.root);
|
|
1521
|
+
|
|
1522
|
+
// ---- models (static catalog) ----
|
|
1523
|
+
// LIST only: Moonshot's own OpenAPI declares GET /v1/models and NO retrieve-by-id operation
|
|
1524
|
+
// (the fixture's 19 operations have no /v1/models/{model}) — serving a 200 retrieve here was a
|
|
1525
|
+
// 200 for an unmodeled route. An unknown/unmodeled sub-path falls through to the router's
|
|
1526
|
+
// vendor-shaped not-found below.
|
|
1527
|
+
if (seg[0] === 'models' && seg.length === 1 && method === 'GET') {
|
|
1528
|
+
return { status: 200, body: { object: 'list', data: servedModels(req.root) } };
|
|
1529
|
+
}
|
|
1530
|
+
|
|
1531
|
+
// ---- chat completions (the generative stub; envelope is faithful) ----
|
|
1532
|
+
if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
|
|
1533
|
+
const validated = validateChat(params);
|
|
1534
|
+
if ('error' in validated) return validated.error;
|
|
1535
|
+
const args = validated.args;
|
|
1536
|
+
// R15 — the scenario engine decides AND honors a fault here, before any completion exists:
|
|
1537
|
+
// a `status` fault is this vendor's own refusal envelope, a `slow` has already held the
|
|
1538
|
+
// answer, a `drop` never returns. The realizers below only see a content decision.
|
|
1539
|
+
let chatDecision: ScenarioDecision | undefined;
|
|
1540
|
+
if (req.scenarioEngine) {
|
|
1541
|
+
const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages, tools: args.tools, thinking: args.thinking });
|
|
1542
|
+
if (served.kind === 'fault') return { status: served.result.status, body: served.result.body, headers: served.result.headers };
|
|
1543
|
+
chatDecision = served;
|
|
1544
|
+
}
|
|
1545
|
+
const result = args.stream && req.sseSink
|
|
1546
|
+
? streamChat(args, req.sseSink, req.occurredAt, chatDecision)
|
|
1547
|
+
: buildChatCompletion(args, req.occurredAt, chatDecision);
|
|
1548
|
+
if (isEnvelope(result)) return result;
|
|
1549
|
+
return { status: 200, body: result };
|
|
1550
|
+
}
|
|
1551
|
+
|
|
1552
|
+
// ---- responses (kimi-k3 only) ----
|
|
1553
|
+
if (seg[0] === 'responses' && seg.length === 1 && method === 'POST') {
|
|
1554
|
+
const validated = validateResponses(params);
|
|
1555
|
+
if ('error' in validated) return validated.error;
|
|
1556
|
+
const args = validated.args;
|
|
1557
|
+
// R15 — the same serve-and-honor as chat completions: the Responses door serves the same
|
|
1558
|
+
// scripted brain from the same handlers document.
|
|
1559
|
+
let responsesDecision: ScenarioDecision | undefined;
|
|
1560
|
+
if (req.scenarioEngine) {
|
|
1561
|
+
const served = await req.scenarioEngine.serve({ model: args.model, messages: [{ role: 'user', content: responsesInputText(args.input) }], tools: undefined, thinking: undefined });
|
|
1562
|
+
if (served.kind === 'fault') return { status: served.result.status, body: served.result.body, headers: served.result.headers };
|
|
1563
|
+
responsesDecision = served;
|
|
1564
|
+
}
|
|
1565
|
+
const result = args.stream && req.messagesSseSink
|
|
1566
|
+
? streamResponses(args, req.messagesSseSink, req.occurredAt, responsesDecision)
|
|
1567
|
+
: buildResponsesResponse(args, req.occurredAt, responsesDecision);
|
|
1568
|
+
if (isEnvelope(result)) return result;
|
|
1569
|
+
return { status: 200, body: result };
|
|
1570
|
+
}
|
|
1571
|
+
|
|
1572
|
+
// ---- token counting ----
|
|
1573
|
+
if (seg[0] === 'tokenizers' && seg[1] === 'estimate-token-count' && seg.length === 2 && method === 'POST') {
|
|
1574
|
+
return handleEstimateTokens(params);
|
|
1575
|
+
}
|
|
1576
|
+
|
|
1577
|
+
// ---- signatures ----
|
|
1578
|
+
if (seg[0] === 'signatures' && seg[1] === 'verify' && seg.length === 2 && method === 'POST') {
|
|
1579
|
+
return handleSignatureVerify(params, req);
|
|
1580
|
+
}
|
|
1581
|
+
|
|
1582
|
+
// ---- web-search tools ----
|
|
1583
|
+
if (seg[0] === 'tools' && seg[1] === 'search' && seg.length === 2 && method === 'POST') return handleToolsSearch(params, false);
|
|
1584
|
+
if (seg[0] === 'tools' && seg[1] === 'search_pro' && seg.length === 2 && method === 'POST') return handleToolsSearch(params, true);
|
|
1585
|
+
if (seg[0] === 'tools' && seg[1] === 'fetch' && seg.length === 2 && method === 'POST') return handleToolsFetch(params);
|
|
1586
|
+
|
|
1587
|
+
// ---- balance (Moonshot's own envelope) ----
|
|
1588
|
+
if (seg[0] === 'users' && seg[1] === 'me' && seg[2] === 'balance' && seg.length === 3 && method === 'GET') {
|
|
1589
|
+
return balanceView(req.root);
|
|
1590
|
+
}
|
|
1591
|
+
|
|
1592
|
+
// ---- files (stateful) ----
|
|
1593
|
+
if (seg[0] === 'files' && seg.length === 1 && method === 'POST') return createFile(params, req);
|
|
1594
|
+
if (seg[0] === 'files' && seg.length === 1 && method === 'GET') {
|
|
1595
|
+
const data = rows('file', req.root).filter((r) => !r._deleted).map(fileView);
|
|
1596
|
+
return { status: 200, body: { object: 'list', data } };
|
|
1597
|
+
}
|
|
1598
|
+
if (seg[0] === 'files' && seg.length === 2 && method === 'GET') {
|
|
1599
|
+
const f = getRow('file', dec(seg[1]!), req.root);
|
|
1600
|
+
return f && !f._deleted ? { status: 200, body: fileView(f) } : notFound(`No such File object: ${dec(seg[1]!)}`);
|
|
1601
|
+
}
|
|
1602
|
+
if (seg[0] === 'files' && seg.length === 3 && seg[2] === 'content' && method === 'GET') {
|
|
1603
|
+
const f = getRow('file', dec(seg[1]!), req.root);
|
|
1604
|
+
if (!f || f._deleted) return notFound(`No such File object: ${dec(seg[1]!)}`);
|
|
1605
|
+
// A PULLED file (the connector's mapFile) carries metadata only — the real list endpoint
|
|
1606
|
+
// returns no content, so the twin holds none. Serving an empty 200 here was a §9-round-two
|
|
1607
|
+
// finding (F2): a fake success for bytes the twin does not have. The honest answer is a
|
|
1608
|
+
// vendor-shaped refusal naming the gap.
|
|
1609
|
+
if (f._content === undefined && Number(f.bytes) > 0) {
|
|
1610
|
+
return { status: 501, body: errBody('server_error', `the twin holds no content for pulled file ${dec(seg[1]!)} (the vendor's list endpoint does not return file content; re-create the file locally to read it back)`) };
|
|
1611
|
+
}
|
|
1612
|
+
// The server turns the body STRING into bytes; this header tells it how (§9 round two, F1):
|
|
1613
|
+
// a binary upload (the multipart adapter's `binary_content` marker → `_content_encoding`)
|
|
1614
|
+
// is decoded from its stored base64 and carried latin1 (every byte value intact through a
|
|
1615
|
+
// JS string, re-materialized by the server); anything else is TEXT and travels as the
|
|
1616
|
+
// string itself — the server writes it UTF-8, so a non-ASCII text file round-trips.
|
|
1617
|
+
if (f._content_encoding === 'base64') {
|
|
1618
|
+
const raw = Buffer.from(String(f._content ?? ''), 'base64').toString('latin1');
|
|
1619
|
+
return { status: 200, body: raw, headers: { 'content-type': 'application/octet-stream', 'x-twin-content-binary': '1' } };
|
|
1620
|
+
}
|
|
1621
|
+
return { status: 200, body: String(f._content ?? ''), headers: { 'content-type': 'application/octet-stream' } };
|
|
1622
|
+
}
|
|
1623
|
+
if (seg[0] === 'files' && seg.length === 2 && method === 'DELETE') {
|
|
1624
|
+
const fid = dec(seg[1]!);
|
|
1625
|
+
const f = getRow('file', fid, req.root);
|
|
1626
|
+
if (!f || f._deleted) return notFound(`No such File object: ${fid}`);
|
|
1627
|
+
await applyTwinWrite(SERVICE, {
|
|
1628
|
+
operation: 'file.delete', subjectType: 'file', subjectId: fid, fields: { _deleted: true, object: 'file' },
|
|
1629
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1630
|
+
}, req.root);
|
|
1631
|
+
return { status: 200, body: { id: fid, object: 'file', deleted: true } };
|
|
1632
|
+
}
|
|
1633
|
+
|
|
1634
|
+
// ---- batches (stateful) ----
|
|
1635
|
+
if (seg[0] === 'batches' && seg.length === 1 && method === 'POST') return createBatch(params, req);
|
|
1636
|
+
if (seg[0] === 'batches' && seg.length === 1 && method === 'GET') {
|
|
1637
|
+
return { status: 200, body: { object: 'list', data: rows('batch', req.root).map(batchView), has_more: false } };
|
|
1638
|
+
}
|
|
1639
|
+
if (seg[0] === 'batches' && seg.length === 2 && method === 'GET') {
|
|
1640
|
+
const b = getRow('batch', dec(seg[1]!), req.root);
|
|
1641
|
+
if (!b) return notFound(`No such Batch object: ${dec(seg[1]!)}`);
|
|
1642
|
+
return { status: 200, body: batchView(b) };
|
|
1643
|
+
}
|
|
1644
|
+
if (seg[0] === 'batches' && seg.length === 3 && seg[2] === 'cancel' && method === 'POST') {
|
|
1645
|
+
const bid = dec(seg[1]!);
|
|
1646
|
+
const b = getRow('batch', bid, req.root);
|
|
1647
|
+
if (!b) return notFound(`No such Batch object: ${bid}`);
|
|
1648
|
+
if (b.status === 'cancelling' || b.status === 'cancelled') return invalidRequest(`Cannot cancel a batch with status '${String(b.status)}'.`);
|
|
1649
|
+
// NOTE: the twin does not simulate the asynchronous cancelling→cancelled settlement (the
|
|
1650
|
+
// groq pack's §9 finding applies identically here: settling on a READ would break the
|
|
1651
|
+
// read-only contract). The terminal transition is filed as
|
|
1652
|
+
// `moonshot.batches.cancellation_settles` (todo).
|
|
1653
|
+
await applyTwinWrite(SERVICE, {
|
|
1654
|
+
operation: 'batch.cancel', subjectType: 'batch', subjectId: bid,
|
|
1655
|
+
fields: { status: 'cancelling', cancelling_at: nowEpoch(req.occurredAt) },
|
|
1656
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1657
|
+
}, req.root);
|
|
1658
|
+
return { status: 200, body: batchView(getRow('batch', bid, req.root) ?? {}) };
|
|
1659
|
+
}
|
|
1660
|
+
|
|
1661
|
+
// Unmodeled operation → fail like the vendor (never a fake success). A /anthropic path that
|
|
1662
|
+
// is not /messages was already refused in the Messages branch; anything falling through here
|
|
1663
|
+
// is on the OpenAI-compatible surface, whose 404 envelope this is.
|
|
1664
|
+
return notFound(`Unknown request URL: ${method} ${path}.`);
|
|
1665
|
+
}
|
|
1666
|
+
|
|
1667
|
+
// ── streaming sink types (exported for the server) ─────────────────────────────────────
|
|
1668
|
+
/** A sink the Messages/Responses streaming paths write events into. `event` is the SSE event
|
|
1669
|
+
* name (Anthropic/Responses grammar); `data` the JSON payload; `done: true` ends the stream. */
|
|
1670
|
+
export type MessagesSseSink = (event: MessagesSseEvent) => void;
|