@volter/twin-togetherai 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +147 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/index.d.ts +14 -0
  6. package/dist/src/index.js +79 -0
  7. package/dist/src/togetherai-budget.d.ts +52 -0
  8. package/dist/src/togetherai-budget.js +130 -0
  9. package/dist/src/togetherai-capabilities.d.ts +4 -0
  10. package/dist/src/togetherai-capabilities.js +1428 -0
  11. package/dist/src/togetherai-conformance.d.ts +14 -0
  12. package/dist/src/togetherai-conformance.js +452 -0
  13. package/dist/src/togetherai-connector.d.ts +164 -0
  14. package/dist/src/togetherai-connector.js +457 -0
  15. package/dist/src/togetherai-models.d.ts +19 -0
  16. package/dist/src/togetherai-models.js +49 -0
  17. package/dist/src/togetherai-scenario.d.ts +52 -0
  18. package/dist/src/togetherai-scenario.js +168 -0
  19. package/dist/src/togetherai-server.d.ts +16 -0
  20. package/dist/src/togetherai-server.js +187 -0
  21. package/dist/src/togetherai-stub.d.ts +59 -0
  22. package/dist/src/togetherai-stub.js +195 -0
  23. package/dist/src/togetherai-twin.d.ts +83 -0
  24. package/dist/src/togetherai-twin.js +1419 -0
  25. package/dist/src/togetherai-types.d.ts +207 -0
  26. package/dist/src/togetherai-types.js +26 -0
  27. package/package.json +52 -0
  28. package/src/cli.ts +27 -0
  29. package/src/index.ts +118 -0
  30. package/src/togetherai-budget.ts +156 -0
  31. package/src/togetherai-capabilities.ts +1315 -0
  32. package/src/togetherai-conformance.ts +459 -0
  33. package/src/togetherai-connector.ts +496 -0
  34. package/src/togetherai-models.ts +74 -0
  35. package/src/togetherai-scenario.ts +185 -0
  36. package/src/togetherai-server.ts +199 -0
  37. package/src/togetherai-stub.ts +197 -0
  38. package/src/togetherai-twin.ts +1448 -0
  39. package/src/togetherai-types.ts +222 -0
@@ -0,0 +1,1448 @@
1
+ // Together AI twin REQUEST HANDLER — the canonical Together AI surface for the twin.
2
+ // Contract: handleTogetheraiTwinRequest({method, path, body}) -> {status, body}. It is the
3
+ // faithful Together AI API the real `together-ai` SDK (pointed at this baseURL) talks to
4
+ // UNMODIFIED — with ONE documented exception: the SDK's `files.upload()` (its custom 302-redirect
5
+ // upload) reads `TOGETHER_API_BASE_URL` at MODULE LOAD and ignores the client's `baseURL`
6
+ // (together-ai@0.53.0 lib/upload.js:12), so it must be pointed at the twin explicitly with
7
+ // `TOGETHER_API_BASE_URL=<twin>/v1` — unset, it egresses to https://api.together.xyz with the
8
+ // caller's real key. Every other SDK call rides `baseURL` (or the injector's api.together.ai
9
+ // interception) and needs nothing else.
10
+ //
11
+ // THE HONEST DESIGN: the twin cannot run the model, so `POST /v1/chat/completions` returns a
12
+ // DETERMINISTIC STUB completion (togetherai-stub.ts) clearly labeled a twin stub — it NEVER
13
+ // pretends to be real model output. `/v1/embeddings` returns DETERMINISTIC pseudo-vectors and
14
+ // the audio endpoints return DETERMINISTIC labeled stubs. But the ENTIRE PROTOCOL ENVELOPE is
15
+ // vendor-faithful: response shapes (including Together's REQUIRED `prompt` array and its `eos`
16
+ // finish_reason), streaming SSE chunks with the nullable per-chunk `usage`/`warnings`,
17
+ // tool_calls, and Together's OWN status table (402 spending limit, 403 context length, 503
18
+ // engine overloaded). The genuinely stateful + static surface is real:
19
+ // • GET /v1/models — static catalog (togetherai-models.ts)
20
+ // • GET /v1/whoami — static identity for the presented key
21
+ // • POST /v1/files/upload + GET/DELETE /v1/files{,/:id,/content} — stateful (kernel action
22
+ // log), BOTH upload flows: the spec's multipart POST /v1/files/upload AND the together-ai
23
+ // SDK's own redirect dance
24
+ // (POST /v1/files?<params> → 302 + x-together-file-id → PUT the bytes).
25
+ // There is NO JSON create at POST /v1/files — the spec lists GET only there; a bodyless or
26
+ // JSON POST is a 400 naming the real doors.
27
+ // • POST/GET /v1/batches (+ cancel) — stateful, Together-native shapes (201
28
+ // BatchJobWithWarning, bare-array list, `{error: string}` failures)
29
+ // • POST/GET/DELETE /v1/fine-tunes (+ cancel/events) — stateful, Together-native
30
+ //
31
+ // THE OPENAI-COMPATIBILITY LINE (docs.together.ai/docs/inference/openai-compatibility, read
32
+ // 2026-09-16): Together's OpenAI-compatible surface is the INFERENCE endpoints only. Assistants/
33
+ // Threads/Runs are NOT implemented; `moderations.create` is NOT implemented; OpenAI-shaped
34
+ // Batch/Files/Fine-tuning APIs are NOT supported (Together ships its own native equivalents,
35
+ // modeled here); `service_tier`/`store`/`metadata`/`prediction` are ACCEPTED BUT IGNORED. A twin
36
+ // that served the OpenAI shapes on those stateful resources would be surface the vendor does not
37
+ // have — the inverse false-green (ADDING_A_TWIN.md §6).
38
+ //
39
+ // State lives in the kernel action log (D1): all writes are local actions, reads are the
40
+ // projection. No real Together API is ever called from this path (D4). Streaming uses an
41
+ // INJECTED sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
42
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
43
+ import {
44
+ EMBEDDING_DIMENSIONS,
45
+ EMBEDDING_MODELS,
46
+ RERANK_MODELS,
47
+ SPEECH_MODELS,
48
+ TOGETHERAI_MODELS,
49
+ findModel,
50
+ } from './togetherai-models.ts';
51
+ import {
52
+ buildUsage,
53
+ contentToText,
54
+ countPromptTokens,
55
+ estimateTokens,
56
+ fnv1a,
57
+ pseudoEmbedding,
58
+ stubAssistantText,
59
+ stubAudioSeconds,
60
+ stubJsonObject,
61
+ stubReasoningText,
62
+ stubToolCall,
63
+ stubTranscript,
64
+ } from './togetherai-stub.ts';
65
+ import { type TogetheraiScenarioEngine, type TogetheraiScenarioRespond, realizeTogetheraiRespond, type ScriptedResult } from './togetherai-scenario.ts';
66
+ import type {
67
+ TogetheraiAssistantMessage,
68
+ TogetheraiBatch,
69
+ TogetheraiBatchEndpoint,
70
+ TogetheraiBatchStatus,
71
+ TogetheraiChatCompletion,
72
+ TogetheraiChoice,
73
+ TogetheraiEmbedding,
74
+ TogetheraiEmbeddingResponse,
75
+ TogetheraiError,
76
+ TogetheraiFile,
77
+ TogetheraiFilePurpose,
78
+ TogetheraiFileType,
79
+ TogetheraiFinishReason,
80
+ TogetheraiMessageParam,
81
+ TogetheraiModel,
82
+ TogetheraiRerankResponse,
83
+ TogetheraiToolCall,
84
+ TogetheraiUsage,
85
+ SseSink,
86
+ } from './togetherai-types.ts';
87
+
88
+ const SERVICE = 'togetherai';
89
+
90
+ /** The ONE base path Together's inference API serves. Everything the twin routes hangs off this. */
91
+ export const TOGETHERAI_API_PREFIX = '/v1';
92
+
93
+ export type TogetheraiRequest = {
94
+ /** The scenario engine (kernel grammar + this pack's vocabulary) — scripts chat turns. */
95
+ scenarioEngine?: TogetheraiScenarioEngine;
96
+ method: string;
97
+ path: string;
98
+ body?: string;
99
+ occurredAt?: string;
100
+ root?: string;
101
+ readOnly?: boolean;
102
+ /** The credential the caller presents (the SDK's bearer `Authorization` header). When a request
103
+ * carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the real
104
+ * vendor rule: a credential is required → 401 on missing/invalid. In-process trusted calls
105
+ * (capability verify, connector) omit BOTH and are not auth-gated. */
106
+ apiKey?: string;
107
+ /** Lower-cased request headers the HTTP server passes through so the handler can model auth
108
+ * (401) and the deterministic fault triggers (429 / 402 / 503). */
109
+ headers?: Record<string, string>;
110
+ /** When set on a streaming POST, chunks are written here (no sockets). */
111
+ sseSink?: SseSink;
112
+ };
113
+
114
+ /** The handler response. `headers` (when present) are response headers the HTTP server should set
115
+ * — e.g. `x-ratelimit-reset` on a modeled 429, or the redirect pair on the SDK's upload flow. */
116
+ export type TogetheraiResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
117
+
118
+ // ── vendor-shaped errors ──────────────────────────────────────────────────────────────
119
+ /**
120
+ * Together's error envelope (`ErrorData`): `{ error: { message, type, param, code } }` with
121
+ * `message` + `type` REQUIRED and `param`/`code` nullable with default null. The twin emits all
122
+ * four keys so the served shape matches the vendor's REQUIRED+DEFAULT shape exactly.
123
+ */
124
+ function errBody(type: string, message: string, param: string | null = null, code: string | null = null): TogetheraiError {
125
+ return { error: { message, type, param, code } };
126
+ }
127
+ function invalidRequest(message: string): TogetheraiResponseEnvelope {
128
+ return { status: 400, body: errBody('invalid_request_error', message) };
129
+ }
130
+ function notFound(message: string): TogetheraiResponseEnvelope {
131
+ return { status: 404, body: errBody('invalid_request_error', message) };
132
+ }
133
+ function authError(message: string): TogetheraiResponseEnvelope {
134
+ return { status: 401, body: errBody('invalid_request_error', message) };
135
+ }
136
+ /** Together's 403 is NOT a permission denial — docs.together.ai/docs/error-codes: 403 means
137
+ * "the sum of input tokens plus max_tokens exceeds the context length of the model". */
138
+ function contextLengthError(model: string, needed: number, limit: number): TogetheraiResponseEnvelope {
139
+ return {
140
+ status: 403,
141
+ body: errBody('invalid_request_error', `This model's maximum context length is ${limit} tokens. However, you requested ${needed} tokens (${needed - limit} in the messages, Please reduce the length of the messages or max_tokens.`),
142
+ };
143
+ }
144
+
145
+ // ── modeled authentication (401) ────────────────────────────────────────────────────────
146
+ // Real Together requires a bearer credential on every request and returns 401 when it is missing
147
+ // or invalid (docs.together.ai/docs/error-codes: 401 = "A missing or invalid API key"). The twin
148
+ // can't validate against real keys, so it models the CHECKABLE failures: a missing credential,
149
+ // and a reserved sentinel for the invalid-key path. Any other non-empty key is accepted. Trusted
150
+ // in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated; both official
151
+ // clients always send a key → they pass.
152
+ function checkAuth(req: TogetheraiRequest): TogetheraiResponseEnvelope | null {
153
+ const auth = req.headers?.['authorization'];
154
+ const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
155
+ const key = (req.apiKey ?? '').trim() || bearer;
156
+ if (!key) return authError('Missing or invalid API Key');
157
+ if (key === 'twin_invalid' || key === 'invalid') return authError('Missing or invalid API Key');
158
+ return null;
159
+ }
160
+
161
+ // ── modeled rate limiting (429), spending limit (402), overload (503) ───────────────────
162
+ // All three are non-deterministic in production, so the twin exposes DETERMINISTIC opt-in
163
+ // triggers. Together's documented behaviors (docs.together.ai/docs/error-codes +
164
+ // /docs/serverless/rate-limits, read 2026-09-16):
165
+ // • 429 — serverless rate limit exceeded; error types `dynamic_request_limited` /
166
+ // `dynamic_token_limited`; carries `x-ratelimit-reset` (seconds until reset). Success
167
+ // responses carry NO rate-limit headers.
168
+ // • 402 — the account hit its monthly spending limit ("Payment Required").
169
+ // • 503 — "Engine Overloaded: servers are under heavy traffic".
170
+ function rateLimitError(kind: 'dynamic_request_limited' | 'dynamic_token_limited' = 'dynamic_request_limited'): TogetheraiResponseEnvelope {
171
+ return {
172
+ status: 429,
173
+ body: errBody(kind, 'Rate limit exceeded: dynamic request limit reached for this model. Please retry after the reset window.'),
174
+ headers: { 'x-ratelimit-reset': '60' },
175
+ };
176
+ }
177
+ function spendingLimitError(): TogetheraiResponseEnvelope {
178
+ return { status: 402, body: errBody('insufficient_quota', 'The account associated with the API key has reached its maximum allowed monthly spending limit.') };
179
+ }
180
+ function engineOverloadedError(): TogetheraiResponseEnvelope {
181
+ return { status: 503, body: errBody('engine_overloaded', 'Engine overloaded: servers are under heavy traffic. Please retry after a short wait.') };
182
+ }
183
+ function triggered(req: TogetheraiRequest, header: string): boolean {
184
+ const v = req.headers?.[header];
185
+ return v === '1' || v === 'true';
186
+ }
187
+
188
+ function nowEpoch(occurredAt?: string): number {
189
+ return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
190
+ }
191
+
192
+ // ── kernel helpers ──────────────────────────────────────────────────────────────────────
193
+ function rows(type: string, root?: string): Array<Record<string, unknown>> {
194
+ return projectResources(SERVICE, root).filter((r) => r.type === type);
195
+ }
196
+ /**
197
+ * Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
198
+ * projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
199
+ * gap above the count (ADDING_A-TWIN.md §5). Two further properties matter:
200
+ * • the `_twin_` infix namespaces LOCAL mints, so a pulled Together id can never be matched by
201
+ * this regex and therefore can never be re-minted;
202
+ * • the scan includes TOMBSTONED rows (a soft-deleted file keeps its projection row), so the
203
+ * counter RATCHETS across delete→recreate and a deleted id is never handed out twice.
204
+ */
205
+ function nextId(type: string, prefix: string, root?: string): string {
206
+ let max = 0;
207
+ for (const r of rows(type, root)) {
208
+ const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(String(r.id));
209
+ if (m) max = Math.max(max, Number(m[1]));
210
+ }
211
+ return `${prefix}_twin_${max + 1}`;
212
+ }
213
+ /** Models observed by a connector pull (mapModel), reshaped into the served model object. */
214
+ function pulledModels(root?: string): TogetheraiModel[] {
215
+ return rows('model', root)
216
+ .filter((r) => !r._deleted)
217
+ .map((r) => ({ id: String(r.id), object: 'model', created: Number(r.created ?? 0), type: (r.type as TogetheraiModel['type']) ?? 'language' }));
218
+ }
219
+ /** The catalog a request sees: the static table, with any PULLED row of the same id OVERRIDING it
220
+ * (the groq pack's §9-round-two lesson: a pulled row must stay served, not shadowed). */
221
+ function servedModels(root?: string): TogetheraiModel[] {
222
+ const pulled = pulledModels(root);
223
+ const byId = new Map<string, TogetheraiModel>();
224
+ for (const m of TOGETHERAI_MODELS) byId.set(m.id, m);
225
+ for (const m of pulled) byId.set(m.id, m);
226
+ return [...byId.values()];
227
+ }
228
+ function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined {
229
+ return rows(type, root).find((r) => r.id === id);
230
+ }
231
+ /** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
232
+ function strip(r: Record<string, unknown>): Record<string, unknown> {
233
+ const out: Record<string, unknown> = {};
234
+ for (const [k, v] of Object.entries(r)) {
235
+ if (k === 'type' || k === 'updatedAt' || k.startsWith('_')) continue;
236
+ out[k] = v;
237
+ }
238
+ return out;
239
+ }
240
+
241
+ // ── request parsing ─────────────────────────────────────────────────────────────────────
242
+ function parseJson(body?: string): Record<string, unknown> {
243
+ if (!body || !body.trim()) return {};
244
+ try {
245
+ const v = JSON.parse(body);
246
+ return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
247
+ } catch {
248
+ return {};
249
+ }
250
+ }
251
+
252
+ // ── chat completions: validate the request the way Together does ────────────────────────
253
+ type ToolChoice = 'auto' | 'none' | 'required' | { name: string };
254
+ type ResponseFormat = { kind: 'text' } | { kind: 'json_object' } | { kind: 'json_schema'; schema: unknown };
255
+
256
+ /** Together's `ChatCompletionRequest.context_length_exceeded_behavior` — a CLOSED documented
257
+ * enum with default `error`. */
258
+ const CONTEXT_LENGTH_BEHAVIORS = ['truncate', 'error'] as const;
259
+ type ContextLengthBehavior = (typeof CONTEXT_LENGTH_BEHAVIORS)[number];
260
+ /** Together's `reasoning_effort` — a CLOSED documented enum. */
261
+ const REASONING_EFFORTS = ['low', 'medium', 'high'] as const;
262
+ type ReasoningEffort = (typeof REASONING_EFFORTS)[number];
263
+
264
+ type ChatArgs = {
265
+ model: string;
266
+ messages: TogetheraiMessageParam[];
267
+ tools?: unknown;
268
+ n: number;
269
+ maxTokens?: number;
270
+ stop?: string[];
271
+ stream: boolean;
272
+ toolChoice?: ToolChoice;
273
+ parallelToolCalls: boolean;
274
+ legacyFunctions: boolean;
275
+ responseFormat: ResponseFormat;
276
+ seed?: number;
277
+ echo: boolean;
278
+ logprobs?: number;
279
+ reasoningEffort?: ReasoningEffort;
280
+ contextBehavior: ContextLengthBehavior;
281
+ };
282
+
283
+ function validateChat(params: Record<string, unknown>): { args: ChatArgs } | { error: TogetheraiResponseEnvelope } {
284
+ if (params.model === undefined || params.model === '') return { error: invalidRequest("'model' is a required property") };
285
+ if (typeof params.model !== 'string') return { error: invalidRequest("'model' must be a string") };
286
+ if (!Array.isArray(params.messages)) return { error: invalidRequest("'messages' is a required property") };
287
+ if (params.messages.length === 0) return { error: invalidRequest("[] is too short - 'messages'") };
288
+ const messages = params.messages as TogetheraiMessageParam[];
289
+ for (const m of messages) {
290
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
291
+ return { error: invalidRequest("each message must have a valid 'role'") };
292
+ }
293
+ }
294
+ // Together's audio/embedding/rerank/image models are not chat models — asking one to chat is
295
+ // a 404 (Together's docs: 404 = "An invalid endpoint URL or model name"), not a stub. A model
296
+ // id the catalog does not know AT ALL (an OpenAI-style flat id like `gpt-4o`) is the same 404:
297
+ // Together serves only its slash-namespaced catalog.
298
+ if (EMBEDDING_MODELS.has(params.model) || RERANK_MODELS.has(params.model) || SPEECH_MODELS.has(params.model)
299
+ || !findModel(params.model) || findModel(params.model)?.type === 'image') {
300
+ return { error: notFound(`Model ${params.model} does not exist or is not a chat model.`) };
301
+ }
302
+ // `n`: Together's schema pins minimum 1, maximum 128 — an OpenAI-compatible freedom Groq does
303
+ // not have. A number outside the documented range is a 400.
304
+ if (params.n !== undefined && params.n !== null) {
305
+ if (typeof params.n !== 'number' || !Number.isInteger(params.n) || params.n < 1 || params.n > 128) {
306
+ return { error: invalidRequest("'n' must be an integer between 1 and 128") };
307
+ }
308
+ }
309
+ // `logprobs` is an INTEGER 0–20 on Together (OpenAI's boolean; Groq 400s the field outright).
310
+ if (params.logprobs !== undefined && params.logprobs !== null) {
311
+ if (typeof params.logprobs !== 'number' || !Number.isInteger(params.logprobs) || params.logprobs < 0 || params.logprobs > 20) {
312
+ return { error: invalidRequest("'logprobs' must be an integer between 0 and 20") };
313
+ }
314
+ }
315
+ // `context_length_exceeded_behavior` — a CLOSED documented enum, default 'error'.
316
+ let contextBehavior: ContextLengthBehavior = 'error';
317
+ if (params.context_length_exceeded_behavior !== undefined && params.context_length_exceeded_behavior !== null) {
318
+ if (typeof params.context_length_exceeded_behavior !== 'string' || !CONTEXT_LENGTH_BEHAVIORS.includes(params.context_length_exceeded_behavior as ContextLengthBehavior)) {
319
+ return { error: invalidRequest(`'context_length_exceeded_behavior' must be one of ${CONTEXT_LENGTH_BEHAVIORS.map((t) => `'${t}'`).join(', ')}`) };
320
+ }
321
+ contextBehavior = params.context_length_exceeded_behavior as ContextLengthBehavior;
322
+ }
323
+ // `reasoning_effort` — a CLOSED documented enum.
324
+ let reasoningEffort: ReasoningEffort | undefined;
325
+ if (params.reasoning_effort !== undefined && params.reasoning_effort !== null) {
326
+ if (typeof params.reasoning_effort !== 'string' || !REASONING_EFFORTS.includes(params.reasoning_effort as ReasoningEffort)) {
327
+ return { error: invalidRequest(`'reasoning_effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) };
328
+ }
329
+ reasoningEffort = params.reasoning_effort as ReasoningEffort;
330
+ }
331
+ // `compliance` is a CONST ('hipaa') on Together's schema — any other value is not surface.
332
+ if (params.compliance !== undefined && params.compliance !== null && params.compliance !== 'hipaa') {
333
+ return { error: invalidRequest("'compliance' must be 'hipaa'") };
334
+ }
335
+ // `response_format`: text | json_object | json_schema. Together's ResponseFormatJsonSchema
336
+ // REQUIRES `json_schema.name` (a-z A-Z 0-9 underscore/dash, ≤64) — an omission Together's
337
+ // Structured Outputs documents as invalid.
338
+ let responseFormat: ResponseFormat = { kind: 'text' };
339
+ const rf = params.response_format as { type?: unknown; json_schema?: { name?: unknown; schema?: unknown } } | undefined;
340
+ if (rf && typeof rf === 'object') {
341
+ if (rf.type === 'json_object') responseFormat = { kind: 'json_object' };
342
+ else if (rf.type === 'json_schema') {
343
+ const name = rf.json_schema?.name;
344
+ if (typeof name !== 'string' || !name) return { error: invalidRequest("'response_format.json_schema.name' is a required property") };
345
+ if (name.length > 64 || !/^[a-zA-Z0-9_-]+$/.test(name)) {
346
+ return { error: invalidRequest("'response_format.json_schema.name' must be a-z, A-Z, 0-9, or contain underscores and dashes, with a maximum length of 64") };
347
+ }
348
+ responseFormat = { kind: 'json_schema', schema: rf.json_schema };
349
+ } else if (rf.type !== undefined && rf.type !== 'text') {
350
+ return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
351
+ }
352
+ }
353
+ let toolChoice: ToolChoice | undefined;
354
+ const tcRaw = params.tool_choice;
355
+ if (tcRaw !== undefined && tcRaw !== null) {
356
+ if (typeof tcRaw === 'string') {
357
+ if (!['auto', 'none', 'required'].includes(tcRaw)) return { error: invalidRequest("'tool_choice' must be one of 'none', 'auto', 'required' or a named function") };
358
+ toolChoice = tcRaw as ToolChoice;
359
+ } else if (typeof tcRaw === 'object') {
360
+ const name = (tcRaw as { function?: { name?: unknown } }).function?.name;
361
+ if (typeof name !== 'string' || !name) return { error: invalidRequest("'tool_choice.function.name' is required for a named tool choice") };
362
+ toolChoice = { name };
363
+ }
364
+ }
365
+ const maxRaw = params.max_tokens;
366
+ let maxTokens: number | undefined;
367
+ if (maxRaw !== undefined && maxRaw !== null) {
368
+ maxTokens = Number(maxRaw);
369
+ if (!Number.isInteger(maxTokens) || maxTokens < 1) return { error: invalidRequest("'max_tokens' must be an integer >= 1") };
370
+ }
371
+ let stop: string[] | undefined;
372
+ if (params.stop !== undefined && params.stop !== null) {
373
+ if (typeof params.stop === 'string') stop = [params.stop];
374
+ else if (Array.isArray(params.stop)) stop = params.stop as string[];
375
+ else return { error: invalidRequest("'stop' must be a string or an array of strings") };
376
+ }
377
+ return {
378
+ args: {
379
+ model: params.model,
380
+ messages,
381
+ ...(params.tools !== undefined ? { tools: params.tools } : (params.functions !== undefined ? { tools: params.functions } : {})),
382
+ // The DEPRECATED `functions` request parameter has a deprecated RESPONSE shape too:
383
+ // Together's ChatCompletionMessage carries `function_call` (not `tool_calls`) and
384
+ // `FinishReason` includes 'function_call'.
385
+ legacyFunctions: params.tools === undefined && params.functions !== undefined,
386
+ n: typeof params.n === 'number' ? params.n : 1,
387
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
388
+ ...(stop !== undefined ? { stop } : {}),
389
+ stream: params.stream === true,
390
+ ...(toolChoice !== undefined ? { toolChoice } : {}),
391
+ parallelToolCalls: params.parallel_tool_calls !== false,
392
+ responseFormat,
393
+ ...(typeof params.seed === 'number' ? { seed: params.seed } : {}),
394
+ echo: params.echo === true,
395
+ ...(typeof params.logprobs === 'number' ? { logprobs: params.logprobs } : {}),
396
+ ...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
397
+ contextBehavior,
398
+ },
399
+ };
400
+ }
401
+
402
+ /** Build ONE deterministic stub choice (index `idx`). */
403
+ function buildChoice(args: ChatArgs, idx: number): { choice: TogetheraiChoice; completionTokens: number } {
404
+ const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
405
+ const forbidTools = args.toolChoice === 'none';
406
+ const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
407
+ if (hasTools && !forbidTools) {
408
+ const list = args.tools as unknown[];
409
+ const calls: TogetheraiToolCall[] = [];
410
+ if (forcedName || args.parallelToolCalls === false) {
411
+ const tc = stubToolCall(args.tools, idx + 1, forcedName);
412
+ if (tc) calls.push(tc);
413
+ } else {
414
+ for (let t = 0; t < list.length; t++) {
415
+ const tc = stubToolCall([list[t]], idx * 100 + t + 1);
416
+ if (tc) calls.push(tc);
417
+ }
418
+ }
419
+ if (calls.length) {
420
+ if (args.legacyFunctions) {
421
+ const fc = calls[0]!.function;
422
+ return {
423
+ choice: { index: idx, message: { role: 'assistant', content: null, function_call: { name: fc.name, arguments: fc.arguments } }, finish_reason: 'function_call' },
424
+ completionTokens: estimateTokens(JSON.stringify(fc)),
425
+ };
426
+ }
427
+ return {
428
+ choice: { index: idx, message: { role: 'assistant', content: null, tool_calls: calls }, finish_reason: 'tool_calls' },
429
+ completionTokens: estimateTokens(JSON.stringify(calls)),
430
+ };
431
+ }
432
+ }
433
+ let text = args.responseFormat.kind === 'json_object'
434
+ ? stubJsonObject(args.messages, args.model)
435
+ : args.responseFormat.kind === 'json_schema'
436
+ ? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
437
+ : stubAssistantText(args.messages, args.model);
438
+ let finish: TogetheraiFinishReason = 'stop';
439
+ // Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
440
+ let stopAt = -1;
441
+ for (const s of args.stop ?? []) {
442
+ if (!s) continue;
443
+ const i = text.indexOf(s);
444
+ if (i >= 0 && (stopAt < 0 || i < stopAt)) stopAt = i;
445
+ }
446
+ if (stopAt >= 0) text = text.slice(0, stopAt);
447
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
448
+ text = text.slice(0, args.maxTokens * 4);
449
+ finish = 'length';
450
+ }
451
+ const message: TogetheraiAssistantMessage = { role: 'assistant', content: text };
452
+ // `reasoning_effort` present → Together's reasoning models surface the reasoning on the
453
+ // assistant message's own `reasoning` field (varies by model; the stub always populates it
454
+ // when effort is requested so the field's presence is observable).
455
+ if (args.reasoningEffort !== undefined) message.reasoning = stubReasoningText(args.messages, args.model);
456
+ return {
457
+ choice: { index: idx, message, finish_reason: finish },
458
+ completionTokens: estimateTokens(String(message.content ?? '')) + estimateTokens(message.reasoning ?? ''),
459
+ };
460
+ }
461
+
462
+ /** A deterministic id suffix from the request (so ids are stable + assertable). */
463
+ function stableSuffix(args: ChatArgs): string {
464
+ const s = JSON.stringify(args.messages) + args.model + (args.seed !== undefined ? `|seed=${args.seed}` : '');
465
+ return fnv1a(s).toString(36);
466
+ }
467
+
468
+ export function buildChatCompletion(args: ChatArgs, occurredAt?: string, scenarioEngine?: TogetheraiScenarioEngine): TogetheraiChatCompletion | TogetheraiResponseEnvelope {
469
+ // THE 403: Together's documented context-length refusal. Input tokens + max_tokens beyond the
470
+ // model's context length answers 403 (NOT 400, NOT 413) — unless
471
+ // `context_length_exceeded_behavior:'truncate'` overrides max_tokens down to the window.
472
+ const info = findModel(args.model);
473
+ const promptTokens = countPromptTokens(args.messages);
474
+ if (info?.context_length && info.context_length > 0) {
475
+ const needed = promptTokens + (args.maxTokens ?? 0);
476
+ if (needed > info.context_length) {
477
+ if (args.contextBehavior === 'truncate') {
478
+ args = { ...args, maxTokens: Math.max(1, info.context_length - promptTokens) };
479
+ } else {
480
+ return contextLengthError(args.model, needed, info.context_length);
481
+ }
482
+ }
483
+ }
484
+ let scripted: ScriptedResult | null = null;
485
+ let missTeach = '';
486
+ if (scenarioEngine) {
487
+ const decision = scenarioEngine.next({ model: args.model, messages: args.messages, tools: args.tools, reasoningEffort: args.reasoningEffort });
488
+ if (decision.kind === 'handler') {
489
+ const respond = decision.respond as TogetheraiScenarioRespond;
490
+ // A scripted FAILURE short-circuits into Together's own error envelope + status.
491
+ if (respond.error) return scriptedError(respond.error);
492
+ scripted = realizeTogetheraiRespond(respond);
493
+ } else {
494
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/togetherai.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
495
+ }
496
+ }
497
+ const choices: TogetheraiChoice[] = [];
498
+ let completionTokens = 0;
499
+ for (let i = 0; i < args.n; i++) {
500
+ if (scripted) {
501
+ const message: TogetheraiAssistantMessage = scripted.toolCalls.length
502
+ ? { role: 'assistant', content: scripted.text, tool_calls: scripted.toolCalls }
503
+ : { role: 'assistant', content: scripted.text ?? '' };
504
+ if (scripted.reasoning !== null) message.reasoning = scripted.reasoning;
505
+ choices.push({ index: i, message, finish_reason: scripted.finishReason });
506
+ completionTokens += estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
507
+ continue;
508
+ }
509
+ const { choice, completionTokens: ct } = buildChoice(args, i);
510
+ if (missTeach && typeof choice.message?.content === 'string') choice.message.content += missTeach;
511
+ choices.push(choice);
512
+ completionTokens += ct;
513
+ }
514
+ const usage: TogetheraiUsage = buildUsage(promptTokens, completionTokens);
515
+ const id = `chatcmpl-twin-${stableSuffix(args)}`;
516
+ // Together's REQUIRED `prompt` array: the echoed prompt when `echo:true` (the vendor's schema
517
+ // documents it as "When echo is true, the prompt is included in the response"). REQUIRED on
518
+ // the schema, so it is served even when empty.
519
+ const prompt = args.echo ? [{ text: args.messages.map((m) => contentToText(m.content)).join('\n') }] : [];
520
+ return {
521
+ id,
522
+ object: 'chat.completion',
523
+ created: nowEpoch(occurredAt),
524
+ model: args.model,
525
+ choices,
526
+ prompt,
527
+ usage,
528
+ };
529
+ }
530
+
531
+ /** Map a scripted scenario failure onto Together's real status + envelope. */
532
+ function scriptedError(err: NonNullable<TogetheraiScenarioRespond['error']>): TogetheraiResponseEnvelope {
533
+ if (err.type === 'rate_limit_exceeded') {
534
+ const base = rateLimitError();
535
+ return err.message ? { ...base, body: errBody('dynamic_request_limited', err.message) } : base;
536
+ }
537
+ if (err.type === 'engine_overloaded') {
538
+ const base = engineOverloadedError();
539
+ return err.message ? { ...base, body: errBody('engine_overloaded', err.message) } : base;
540
+ }
541
+ return { status: 500, body: errBody('internal_server_error', err.message ?? 'Internal Server Error') };
542
+ }
543
+
544
+ const isEnvelope = (v: TogetheraiChatCompletion | TogetheraiResponseEnvelope): v is TogetheraiResponseEnvelope =>
545
+ typeof (v as TogetheraiResponseEnvelope).status === 'number' && 'body' in v;
546
+
547
+ /** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
548
+ function chunkText(text: string): string[] {
549
+ if (!text) return [];
550
+ const out: string[] = [];
551
+ for (let i = 0; i < text.length; i += 20) out.push(text.slice(i, i + 20));
552
+ return out;
553
+ }
554
+
555
+ /**
556
+ * Emit the vendor-faithful Together streaming sequence into the injected sink (NO sockets, NO
557
+ * setTimeout). Together's order: a first chunk with `delta:{role:'assistant'}`, then
558
+ * `delta:{content}` chunks (or tool_calls deltas), then a chunk carrying `finish_reason`. Every
559
+ * chunk carries Together's nullable `usage` + `warnings` (the schema declares both on
560
+ * ChatCompletionChunk itself — OpenAI puts usage only in an opt-in tail chunk). Ends with
561
+ * `data: [DONE]`. Deterministic + synchronous.
562
+ */
563
+ export function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, scenarioEngine?: TogetheraiScenarioEngine): TogetheraiChatCompletion | TogetheraiResponseEnvelope {
564
+ const built = buildChatCompletion(args, occurredAt, scenarioEngine);
565
+ if (isEnvelope(built)) return built;
566
+ const full = built;
567
+ const base = { id: full.id, object: 'chat.completion.chunk' as const, created: full.created, model: full.model };
568
+ for (const choice of full.choices) {
569
+ const idx = choice.index;
570
+ sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, finish_reason: null }], usage: null, warnings: [] } });
571
+ if (choice.message?.reasoning) {
572
+ sink({ data: { ...base, choices: [{ index: idx, delta: { reasoning: choice.message.reasoning }, finish_reason: null }], usage: null, warnings: [] } });
573
+ }
574
+ if (choice.message?.tool_calls && choice.message.tool_calls.length) {
575
+ choice.message.tool_calls.forEach((tc, tIdx) => {
576
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, finish_reason: null }], usage: null, warnings: [] } });
577
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, finish_reason: null }], usage: null, warnings: [] } });
578
+ });
579
+ } else {
580
+ for (const piece of chunkText(choice.message?.content ?? '')) {
581
+ sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, finish_reason: null }], usage: null, warnings: [] } });
582
+ }
583
+ }
584
+ sink({ data: { ...base, choices: [{ index: idx, delta: {}, finish_reason: choice.finish_reason }], usage: null, warnings: [] } });
585
+ }
586
+ // Together carries usage on the chunks themselves; the FINAL chunk reports the real counts.
587
+ sink({ data: { ...base, choices: [], usage: full.usage, warnings: [] } });
588
+ sink({ done: true });
589
+ return full;
590
+ }
591
+
592
+ // ── Embeddings (deterministic pseudo-vectors) ───────────────────────────────────────────
593
+ function handleEmbeddings(params: Record<string, unknown>): TogetheraiResponseEnvelope {
594
+ if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
595
+ if (params.input === undefined) return invalidRequest("'input' is a required property");
596
+ const model = params.model;
597
+ if (!EMBEDDING_MODELS.has(model)) return notFound(`Model ${model} does not exist or is not an embedding model.`);
598
+ const inputs = Array.isArray(params.input) ? (params.input as unknown[]).map(String) : [String(params.input)];
599
+ if (inputs.some((s) => s.length === 0)) return invalidRequest("'input' must not be an empty string");
600
+ const dims = EMBEDDING_DIMENSIONS[model] ?? 768;
601
+ const data: TogetheraiEmbedding[] = inputs.map((text, index) => {
602
+ const vec = pseudoEmbedding(text, dims);
603
+ return { object: 'embedding', index, embedding: vec };
604
+ });
605
+ // Together's EmbeddingsResponse REQUIRED keys are object/model/data — NO usage key on the
606
+ // schema (Together does not document one on this endpoint).
607
+ const body: TogetheraiEmbeddingResponse = { object: 'list', model, data };
608
+ return { status: 200, body };
609
+ }
610
+
611
+ // ── Rerank (Together-native) ────────────────────────────────────────────────────────────
612
+ /**
613
+ * Deterministic rerank: score each document by a pure function of (query, document text) so the
614
+ * ORDER is stable and assertable, then return the top `top_n` with Together's
615
+ * `{object:'rerank', model, results:[{index, relevance_score, document}]}` shape. NOT a real
616
+ * ranker — the values carry no semantic meaning; only the SHAPE and DETERMINISM are faithful.
617
+ */
618
+ function handleRerank(params: Record<string, unknown>): TogetheraiResponseEnvelope {
619
+ if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
620
+ if (!RERANK_MODELS.has(params.model)) return notFound(`Model ${params.model} does not exist or is not a rerank model.`);
621
+ if (typeof params.query !== 'string' || !params.query) return invalidRequest("'query' is a required property");
622
+ if (!Array.isArray(params.documents)) return invalidRequest("'documents' is a required property");
623
+ if (params.documents.length === 0) return invalidRequest("'documents' must not be empty");
624
+ // Documents may be strings OR objects (Together's oneOf); an object is ranked by its text-ish
625
+ // fields (`rank_fields` names the keys, defaulting to all of them).
626
+ const rankFields = Array.isArray(params.rank_fields) ? (params.rank_fields as unknown[]).map(String) : null;
627
+ const textOf = (doc: unknown): string => {
628
+ if (typeof doc === 'string') return doc;
629
+ if (doc && typeof doc === 'object') {
630
+ const o = doc as Record<string, unknown>;
631
+ const keys = rankFields ?? Object.keys(o);
632
+ return keys.map((k) => String(o[k] ?? '')).join(' ');
633
+ }
634
+ return String(doc ?? '');
635
+ };
636
+ const query = params.query as string;
637
+ const scored = (params.documents as unknown[]).map((doc, index) => {
638
+ const text = textOf(doc);
639
+ // A deterministic pseudo-score in (0,1): hash(query + text) — same inputs, same order.
640
+ const score = (fnv1a(`${query}\0${text}`) % 10_000) / 10_000;
641
+ return { index, score, doc };
642
+ });
643
+ scored.sort((a, b) => b.score - a.score || a.index - b.index);
644
+ const topN = typeof params.top_n === 'number' && Number.isInteger(params.top_n) && params.top_n > 0 ? params.top_n : scored.length;
645
+ const returnDocuments = params.return_documents === undefined ? true : params.return_documents === true;
646
+ const usageTokens = countPromptTokens([{ role: 'user', content: query }, ...((params.documents as unknown[]).map((d) => ({ role: 'user' as const, content: textOf(d) })))]);
647
+ const body: TogetheraiRerankResponse = {
648
+ object: 'rerank',
649
+ id: `rrank-twin-${fnv1a(`${query}|${JSON.stringify(params.documents)}`).toString(36)}`,
650
+ model: params.model,
651
+ results: scored.slice(0, topN).map((s) => ({
652
+ index: s.index,
653
+ relevance_score: s.score,
654
+ ...(returnDocuments ? { document: { text: typeof s.doc === 'string' ? s.doc : JSON.stringify(s.doc) } } : {}),
655
+ })),
656
+ usage: buildUsage(usageTokens, 0),
657
+ };
658
+ return { status: 200, body };
659
+ }
660
+
661
+ // ── Images (deterministic stub) ─────────────────────────────────────────────────────────
662
+ function handleImages(params: Record<string, unknown>): TogetheraiResponseEnvelope {
663
+ if (typeof params.prompt !== 'string' || !params.prompt) return invalidRequest("'prompt' is a required property");
664
+ if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
665
+ if (findModel(params.model)?.type !== 'image') return notFound(`Model ${params.model} does not exist or is not an image model.`);
666
+ // The SDK documents the default as `jpeg` ("Defaults to `jpeg`", images.d.ts:80) — NOT png.
667
+ const format = params.output_format === undefined ? 'jpeg' : params.output_format;
668
+ if (format !== 'jpeg' && format !== 'png') return invalidRequest("'output_format' must be one of 'jpeg', 'png'");
669
+ const responseFormat = params.response_format === undefined ? 'url' : params.response_format;
670
+ // Together's enum is 'base64' | 'url' (together-ai@0.53.0 images.d.ts:95) — NOT OpenAI's
671
+ // 'b64_json'. Serving OpenAI's value here would be exactly the shape contamination the pack
672
+ // refuses elsewhere (the OpenAI-compatibility line: Together's image API is its own).
673
+ if (responseFormat !== 'url' && responseFormat !== 'base64') return invalidRequest("'response_format' must be one of 'url', 'base64'");
674
+ const n = params.n === undefined ? 1 : params.n;
675
+ if (typeof n !== 'number' || !Number.isInteger(n) || n < 1) return invalidRequest("'n' must be a positive integer");
676
+ const steps = params.steps === undefined ? 20 : params.steps;
677
+ if (typeof steps !== 'number' || !Number.isInteger(steps) || steps < 1) return invalidRequest("'steps' must be a positive integer");
678
+ // No diffusion here: each image is a labeled stub — a data: URL carrying the deterministic
679
+ // marker (so `type:'url'` still carries retrievable bytes) or a base64 body.
680
+ const marker = fnv1a(`${params.prompt}|${params.model}|${params.seed ?? ''}`);
681
+ const data = Array.from({ length: n }, (_, index) => {
682
+ // The format is named IN the label so the (documented) default is observable on the wire.
683
+ const stub = `[twin-stub:${params.model}] deterministic ${format} image stub (no diffusion is run) for prompt "${String(params.prompt).slice(0, 80)}" #${index} (${marker.toString(36)})`;
684
+ if (responseFormat === 'base64') return { index, b64_json: Buffer.from(stub).toString('base64'), type: 'b64_json' as const };
685
+ return { index, url: `data:text/plain;base64,${Buffer.from(stub).toString('base64')}`, type: 'url' as const };
686
+ });
687
+ return {
688
+ status: 200,
689
+ body: { id: `img-twin-${marker.toString(36)}`, model: params.model, object: 'list', data },
690
+ };
691
+ }
692
+
693
+ // ── Audio (deterministic labeled stubs stand in for acoustic model output) ──────────────
694
+ const TRANSCRIPTION_FORMATS = new Set(['json', 'text', 'verbose_json']);
695
+
696
+ function handleTranscription(params: Record<string, unknown>, translate: boolean): TogetheraiResponseEnvelope {
697
+ const model = params.model;
698
+ if (typeof model !== 'string' || !model) return invalidRequest("'model' is a required property");
699
+ if (findModel(model)?.type !== 'language' || !SPEECH_MODELS.has(model)) return notFound(`Model ${model} does not exist or is not an audio model.`);
700
+ // "Either a file or a URL must be provided" (together-ai TranscriptionCreateParams).
701
+ const source = typeof params.file === 'string' && params.file ? params.file : (typeof params.url === 'string' && params.url ? params.url : '');
702
+ if (!source) return invalidRequest("Either 'file' or 'url' must be provided");
703
+ const format = params.response_format === undefined ? 'json' : params.response_format;
704
+ if (typeof format !== 'string' || !TRANSCRIPTION_FORMATS.has(format)) {
705
+ return invalidRequest(`'response_format' must be one of ${[...TRANSCRIPTION_FORMATS].map((f) => `'${f}'`).join(', ')}`);
706
+ }
707
+ const text = stubTranscript(model, source, translate);
708
+ // `response_format:'text'` returns the bare transcript as plain text, not a JSON envelope.
709
+ if (format === 'text') return { status: 200, body: text, headers: { 'content-type': 'text/plain; charset=utf-8' } };
710
+ if (format === 'json') return { status: 200, body: { text } };
711
+ const duration = stubAudioSeconds(source);
712
+ return {
713
+ status: 200,
714
+ body: {
715
+ task: translate ? 'translate' : 'transcribe',
716
+ language: 'english',
717
+ duration,
718
+ text,
719
+ segments: [{ id: 0, seek: 0, start: 0, end: duration, text, tokens: [], temperature: Number(params.temperature ?? 0), avg_logprob: -0.25, compression_ratio: 1.2, no_speech_prob: 0.01 }],
720
+ },
721
+ };
722
+ }
723
+
724
+ function handleSpeech(params: Record<string, unknown>): TogetheraiResponseEnvelope {
725
+ const model = params.model;
726
+ if (typeof model !== 'string' || !model) return invalidRequest("'model' is a required property");
727
+ if (!SPEECH_MODELS.has(model)) return notFound(`Model ${model} does not exist or is not a speech model.`);
728
+ if (typeof params.input !== 'string' || !params.input) return invalidRequest("'input' is a required property");
729
+ if (typeof params.voice !== 'string' || !params.voice) return invalidRequest("'voice' is a required property");
730
+ const format = params.response_format === undefined ? 'mp3' : params.response_format;
731
+ if (typeof format !== 'string' || !['mp3', 'wav', 'opus', 'flac'].includes(format)) {
732
+ return invalidRequest("'response_format' must be one of 'mp3', 'wav', 'opus', 'flac'");
733
+ }
734
+ // No vocoder here: the body is a clearly-labeled deterministic stub, and the server serves it
735
+ // with the audio content-type the vendor would use.
736
+ return {
737
+ status: 200,
738
+ body: `[twin-stub:${model}] no real speech synthesis — voice=${params.voice} format=${format}; text: ${params.input}`,
739
+ headers: { 'content-type': `audio/${format}` },
740
+ };
741
+ }
742
+
743
+ // ── Files (stateful; BOTH upload flows) ─────────────────────────────────────────────────
744
+ /** Together's `FilePurpose` — a CLOSED documented set (NOT OpenAI's). */
745
+ const FILE_PURPOSES = new Set<TogetheraiFilePurpose>(['fine-tune', 'eval', 'batch-api']);
746
+ /** Together's `FileType` — a CLOSED documented set. */
747
+ const FILE_TYPES = new Set<TogetheraiFileType>(['csv', 'jsonl', 'parquet']);
748
+
749
+ /** Build the vendor-faithful FileResponse view of a stored row. */
750
+ function fileView(r: Record<string, unknown>): Record<string, unknown> {
751
+ const s = strip(r);
752
+ return {
753
+ id: r.id,
754
+ object: 'file',
755
+ created_at: s.created_at,
756
+ filename: s.filename,
757
+ bytes: s.bytes,
758
+ purpose: s.purpose,
759
+ Processed: s.Processed ?? true,
760
+ FileType: s.FileType,
761
+ ...(s.processing_status !== undefined ? { processing_status: s.processing_status } : {}),
762
+ ...(s.validation_report !== undefined ? { validation_report: s.validation_report } : {}),
763
+ };
764
+ }
765
+
766
+ /** Validate + derive the stored fields for a file create (shared by both upload flows).
767
+ *
768
+ * The REQUIRED parts are the vendor's, not ours: the SDK types `purpose: FilePurpose` as
769
+ * REQUIRED (files.d.ts:48, `upload(file: string, purpose: FilePurpose, …)`) and the multipart
770
+ * form's `file` part names the upload (`filename`); the 302 flow's query carries `file_name` +
771
+ * `purpose` (lib/upload.js:40). A request missing one answers the vendor's 400 and stores
772
+ * NOTHING — a defaulted `fine-tune`/`upload.jsonl` minted a file from a bodyless POST (the §9
773
+ * round-three blocker; the class fix, not the one instance). */
774
+ function fileFieldsFrom(params: Record<string, unknown>): { fields: Record<string, unknown> } | { error: TogetheraiResponseEnvelope } {
775
+ if (params.purpose === undefined || params.purpose === null || params.purpose === '') {
776
+ return { error: invalidRequest("'purpose' is a required property") };
777
+ }
778
+ const purpose = params.purpose;
779
+ if (typeof purpose !== 'string' || !FILE_PURPOSES.has(purpose as TogetheraiFilePurpose)) {
780
+ return { error: invalidRequest(`'purpose' must be one of ${[...FILE_PURPOSES].map((p) => `'${p}'`).join(', ')} (got '${String(purpose)}')`) };
781
+ }
782
+ const rawName = params.filename ?? params.file_name;
783
+ if (rawName === undefined || rawName === null || String(rawName).trim() === '') {
784
+ return { error: invalidRequest("'filename' is a required property") };
785
+ }
786
+ const filename = String(rawName);
787
+ // An EXPLICIT file_type wins; otherwise it is derived from the filename's extension (§9 round
788
+ // two, R2-D3: the old OR discarded an explicit file_type whenever FileType was absent, and no
789
+ // value was ever validated against the closed set).
790
+ const fileType = params.file_type !== undefined ? String(params.file_type)
791
+ : params.FileType !== undefined ? String(params.FileType)
792
+ : (filename.includes('.') ? String(filename.split('.').pop()).toLowerCase() : 'jsonl');
793
+ if (!FILE_TYPES.has(fileType as TogetheraiFileType)) {
794
+ return { error: invalidRequest(`'file_type' must be one of ${[...FILE_TYPES].map((t) => `'${t}'`).join(', ')} (got '${fileType}')`) };
795
+ }
796
+ const content = typeof params.content === 'string' ? params.content : '';
797
+ const bytes = typeof params.bytes === 'number' ? params.bytes : content.length;
798
+ // Files for non-`fine-tune` purposes SKIP validation (Together's own schema wording); a
799
+ // fine-tune file enters the pipeline as PENDING.
800
+ const processing_status = purpose === 'fine-tune' ? 'PENDING' : undefined;
801
+ return {
802
+ fields: {
803
+ object: 'file', bytes, created_at: 0, filename, purpose, Processed: true, FileType: fileType,
804
+ ...(processing_status ? { processing_status } : {}),
805
+ _content: content,
806
+ },
807
+ };
808
+ }
809
+
810
+ async function createFileMultipart(params: Record<string, unknown>, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
811
+ const built = fileFieldsFrom(params);
812
+ if ('error' in built) return built.error;
813
+ const id = nextId('file', 'file', req.root);
814
+ await applyTwinWrite(SERVICE, {
815
+ operation: 'file.create',
816
+ subjectType: 'file',
817
+ subjectId: id,
818
+ fields: { ...built.fields, created_at: nowEpoch(req.occurredAt) },
819
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
820
+ actor: { kind: 'agent' },
821
+ }, req.root);
822
+ return { status: 200, body: fileView(getRow('file', id, req.root) ?? {}) };
823
+ }
824
+
825
+ /**
826
+ * The together-ai SDK's OWN upload flow (lib/upload.js, together-ai@0.53.0) — NOT the spec's
827
+ * multipart form: it POSTs `/files?file_name=…&file_type=…&purpose=…` as
828
+ * `application/x-www-form-urlencoded`, requires a **302** with a `Location` upload URL and an
829
+ * `x-together-file-id` header, PUTs the bytes there, then confirms with `files.retrieve`. The
830
+ * twin answers the 302 with a twin-local upload URL and mints the file row immediately; the PUT
831
+ * below fills its content.
832
+ */
833
+ async function createFileSdkRedirect(params: Record<string, unknown>, req: TogetheraiRequest, query: URLSearchParams): Promise<TogetheraiResponseEnvelope> {
834
+ const built = fileFieldsFrom(params);
835
+ if ('error' in built) return built.error;
836
+ const id = nextId('file', 'file', req.root);
837
+ await applyTwinWrite(SERVICE, {
838
+ operation: 'file.create',
839
+ subjectType: 'file',
840
+ subjectId: id,
841
+ fields: { ...built.fields, created_at: nowEpoch(req.occurredAt) },
842
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
843
+ actor: { kind: 'agent' },
844
+ }, req.root);
845
+ return {
846
+ status: 302,
847
+ body: '',
848
+ headers: {
849
+ // Twin-LOCAL on purpose: the handler knows only paths. The HTTP server absolutizes it
850
+ // against the twin's public base — the real SDK fetches `Location` with no base URL.
851
+ location: `/twin/upload/${id}`,
852
+ 'x-together-file-id': id,
853
+ },
854
+ };
855
+ }
856
+
857
+ async function storeUploadBytes(fileId: string, content: string, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
858
+ const f = getRow('file', fileId, req.root);
859
+ if (!f || f._deleted) return notFound(`No such File object: ${fileId}`);
860
+ await applyTwinWrite(SERVICE, {
861
+ operation: 'file.update',
862
+ subjectType: 'file', subjectId: fileId,
863
+ fields: { _content: content, bytes: content.length },
864
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
865
+ }, req.root);
866
+ return { status: 200, body: { ok: true } };
867
+ }
868
+
869
+ // ── Batches (Together-native shapes) ────────────────────────────────────────────────────
870
+ /** Together's `CreateBatchRequest.endpoint` — the THREE documented endpoints. */
871
+ const BATCH_ENDPOINTS = new Set<TogetheraiBatchEndpoint>(['/v1/chat/completions', '/v1/audio/transcriptions', '/v1/audio/translations']);
872
+
873
+ /** The batch view of a stored row. `error` stays a bare STRING (Together's BatchJob.error). */
874
+ function batchView(r: Record<string, unknown>): Record<string, unknown> {
875
+ const s = strip(r);
876
+ return {
877
+ id: r.id,
878
+ user_id: s.user_id,
879
+ input_file_id: s.input_file_id,
880
+ file_size_bytes: s.file_size_bytes,
881
+ status: s.status,
882
+ job_deadline: s.job_deadline ?? null,
883
+ created_at: s.created_at,
884
+ endpoint: s.endpoint,
885
+ progress: s.progress ?? 0,
886
+ ...(s.model_id !== undefined ? { model_id: s.model_id } : {}),
887
+ output_file_id: s.output_file_id ?? null,
888
+ error_file_id: s.error_file_id ?? null,
889
+ error: s.error ?? null,
890
+ completed_at: s.completed_at ?? null,
891
+ };
892
+ }
893
+
894
+ async function createBatch(params: Record<string, unknown>, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
895
+ const inputFileId = params.input_file_id;
896
+ if (typeof inputFileId !== 'string' || !inputFileId) return invalidRequest("'input_file_id' is a required property");
897
+ const endpoint = params.endpoint;
898
+ if (typeof endpoint !== 'string' || !BATCH_ENDPOINTS.has(endpoint as TogetheraiBatchEndpoint)) {
899
+ return invalidRequest(`'endpoint' must be one of ${[...BATCH_ENDPOINTS].map((e) => `'${e}'`).join(', ')}`);
900
+ }
901
+ const file = getRow('file', inputFileId, req.root);
902
+ if (!file || file._deleted) return notFound(`No such File object: ${inputFileId}`);
903
+ const created = nowEpoch(req.occurredAt);
904
+ const id = nextId('batch', 'batch', req.root);
905
+ await applyTwinWrite(SERVICE, {
906
+ operation: 'batch.create',
907
+ subjectType: 'batch',
908
+ subjectId: id,
909
+ fields: {
910
+ object: 'batch',
911
+ endpoint,
912
+ input_file_id: inputFileId,
913
+ status: 'VALIDATING' as TogetheraiBatchStatus,
914
+ created_at: new Date(created * 1000).toISOString(),
915
+ file_size_bytes: typeof file.bytes === 'number' ? file.bytes : 0,
916
+ progress: 0,
917
+ ...(params.completion_window !== undefined ? { completion_window: params.completion_window } : {}),
918
+ ...(params.priority !== undefined ? { priority: params.priority } : {}),
919
+ ...(params.model_id !== undefined ? { model_id: params.model_id } : {}),
920
+ },
921
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
922
+ actor: { kind: 'agent' },
923
+ }, req.root);
924
+ // Together's create answers **201** with `BatchJobWithWarning { job, warning? }` — NOT an
925
+ // OpenAI-style 200 batch object.
926
+ return { status: 201, body: { job: batchView(getRow('batch', id, req.root) ?? {}) } };
927
+ }
928
+
929
+ // ── Fine-tunes (Together-native shapes) ─────────────────────────────────────────────────
930
+ /** Together's `FinetuneJobStatus` enum. */
931
+ const FINETUNE_STATUSES = ['pending', 'queued', 'running', 'compressing', 'uploading', 'cancel_requested', 'cancelled', 'error', 'completed'] as const;
932
+
933
+ function finetuneView(r: Record<string, unknown>): Record<string, unknown> {
934
+ const s = strip(r);
935
+ return {
936
+ id: r.id,
937
+ status: s.status,
938
+ user_id: s.user_id,
939
+ ...(s.training_file !== undefined ? { training_file: s.training_file } : {}),
940
+ ...(s.validation_file !== undefined ? { validation_file: s.validation_file } : {}),
941
+ ...(s.model !== undefined ? { model: s.model } : {}),
942
+ ...(s.model_output_name !== undefined ? { model_output_name: s.model_output_name } : {}),
943
+ ...(s.created_at !== undefined ? { created_at: s.created_at } : {}),
944
+ ...(s.updated_at !== undefined ? { updated_at: s.updated_at } : {}),
945
+ ...(s.n_epochs !== undefined ? { n_epochs: s.n_epochs } : {}),
946
+ };
947
+ }
948
+
949
+ async function createFinetune(params: Record<string, unknown>, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
950
+ if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
951
+ if (typeof params.training_file !== 'string' || !params.training_file) return invalidRequest("'training_file' is a required property");
952
+ const base = servedModels(req.root).find((row) => row.id === params.model);
953
+ if (!base || base.type !== 'chat') return notFound(`Model ${params.model} does not exist or is not a fine-tunable model.`);
954
+ const training = getRow('file', params.training_file, req.root);
955
+ if (!training || training._deleted) return notFound(`No such File object: ${params.training_file}`);
956
+ if (training.purpose !== 'fine-tune') return invalidRequest(`File ${params.training_file} must have purpose 'fine-tune' (has '${String(training.purpose)}')`);
957
+ if (params.validation_file !== undefined && params.validation_file !== null) {
958
+ const validation = getRow('file', String(params.validation_file), req.root);
959
+ if (!validation || validation._deleted) return notFound(`No such File object: ${String(params.validation_file)}`);
960
+ if (validation.purpose !== 'fine-tune') return invalidRequest(`File ${params.validation_file} must have purpose 'fine-tune' (has '${String(validation.purpose)}')`);
961
+ }
962
+ const id = nextId('finetune', 'ft', req.root);
963
+ const at = new Date(nowEpoch(req.occurredAt) * 1000).toISOString();
964
+ await applyTwinWrite(SERVICE, {
965
+ operation: 'finetune.create',
966
+ subjectType: 'finetune',
967
+ subjectId: id,
968
+ fields: {
969
+ object: 'finetune',
970
+ status: 'pending',
971
+ user_id: 'user_twin',
972
+ model: params.model,
973
+ training_file: params.training_file,
974
+ ...(params.validation_file !== undefined && params.validation_file !== null ? { validation_file: params.validation_file } : {}),
975
+ created_at: at,
976
+ updated_at: at,
977
+ ...(typeof params.n_epochs === 'number' ? { n_epochs: params.n_epochs } : {}),
978
+ },
979
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
980
+ actor: { kind: 'agent' },
981
+ }, req.root);
982
+ return { status: 200, body: finetuneView(getRow('finetune', id, req.root) ?? {}) };
983
+ }
984
+
985
+ // ── Fine-tune aux endpoints (modeled against Together's documented shapes) ──────────────
986
+ /** The content of a fine-tune-purpose file row, JSONL-decoded into per-row objects. */
987
+ function jsonlRows(content: unknown): Array<Record<string, unknown>> {
988
+ if (typeof content !== 'string' || !content.trim()) return [];
989
+ return content.split('\n').filter((l) => l.trim()).map((l) => {
990
+ try {
991
+ const v = JSON.parse(l);
992
+ return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
993
+ } catch {
994
+ return {};
995
+ }
996
+ });
997
+ }
998
+
999
+ /** GET /v1/fine-tunes/models/limits — Together's `FinetuneModelLimits` for ONE model
1000
+ * (together-ai@0.53.0 fine-tuning.d.ts:270, `model_name` REQUIRED — fine-tuning.d.ts:1798).
1001
+ * A model the catalog does not know fails like the vendor. The numbers are the twin's
1002
+ * deterministic stand-ins for the vendor's per-model training limits; the SHAPE is the vendor's
1003
+ * REQUIRED+OPTIONAL contract. */
1004
+ function finetuneModelLimits(req: TogetheraiRequest): TogetheraiResponseEnvelope {
1005
+ const search = req.path.includes('?') ? new URLSearchParams(req.path.slice(req.path.indexOf('?') + 1)) : new URLSearchParams();
1006
+ const modelName = search.get('model_name');
1007
+ if (!modelName) return invalidRequest("'model_name' is a required property");
1008
+ const m = servedModels(req.root).find((pm) => pm.id === modelName);
1009
+ if (!m || m.type !== 'chat') return notFound(`Model ${modelName} does not exist or is not a fine-tunable model.`);
1010
+ const seq = m.context_length || 8_192;
1011
+ const seed = fnv1a(modelName);
1012
+ return {
1013
+ status: 200,
1014
+ body: {
1015
+ model_name: m.id,
1016
+ default_gradient_accumulation_steps: 16,
1017
+ lora_training: {
1018
+ max_batch_size: 8,
1019
+ max_batch_size_dpo: 4,
1020
+ max_rank: 64,
1021
+ min_batch_size: 1,
1022
+ target_modules: ['q_proj', 'k_proj', 'v_proj', 'o_proj', 'gate_proj', 'up_proj', 'down_proj'],
1023
+ },
1024
+ max_learning_rate: 0.0002,
1025
+ max_num_checkpoints: 10,
1026
+ max_num_epochs: 36,
1027
+ max_num_evals: 36,
1028
+ max_seq_length_dpo: seq,
1029
+ max_seq_length_sft: seq,
1030
+ merge_output_lora: true,
1031
+ min_learning_rate: 0.000001,
1032
+ min_max_seq_length: 128,
1033
+ supports_full_training: false,
1034
+ supports_reasoning: false,
1035
+ supports_tools: true,
1036
+ supports_vision: false,
1037
+ // deterministic per-model jitter so two models never answer identical limits
1038
+ ...(seed % 2 === 0 ? { full_training: { max_batch_size: 4, max_batch_size_dpo: 2, min_batch_size: 1 } } : {}),
1039
+ },
1040
+ };
1041
+ }
1042
+
1043
+ /** POST /v1/fine-tunes/estimate-price — Together's DISCRIMINATED union
1044
+ * (together-ai@0.53.0 fine-tuning.d.ts:1323 — AvailableEstimate | UnavailableEstimate on
1045
+ * `estimation_available`). A training_file the twin has seen and validated resolves to an
1046
+ * AvailableEstimate whose token counts are DETERMINISTIC functions of the file's stored
1047
+ * content; an unknown id is `false` with the vendor's `train_file_invalid` reason. */
1048
+ function estimateFinetunePrice(params: Record<string, unknown>, req: TogetheraiRequest): TogetheraiResponseEnvelope {
1049
+ // together-ai@0.53.0 fine-tuning.d.ts:1688-1691: training_file is REQUIRED, model optional. A
1050
+ // request without the file has nothing to estimate over; the vendor refuses it, never invents 64 tokens.
1051
+ if (params.training_file === undefined || typeof params.training_file !== 'string' || !params.training_file) return invalidRequest("'training_file' is a required property");
1052
+ const file = params.training_file !== undefined ? getRow('file', String(params.training_file), req.root) : undefined;
1053
+ if (params.training_file !== undefined && (!file || file._deleted)) {
1054
+ return { status: 200, body: { estimation_available: false, unavailable_reason: 'train_file_invalid' } };
1055
+ }
1056
+ const dataset = jsonlRows(file?._content);
1057
+ const totalTokens = dataset.reduce((n, row) => n + estimateTokens(JSON.stringify(row)), 0) || 64;
1058
+ const epochs = typeof params.n_epochs === 'number' && params.n_epochs > 0 ? params.n_epochs : 1;
1059
+ return {
1060
+ status: 200,
1061
+ body: {
1062
+ estimation_available: true,
1063
+ allowed_to_proceed: true,
1064
+ estimated_total_price: Math.round(totalTokens * epochs * 0.000002 * 1_000_000) / 1_000_000,
1065
+ estimated_train_token_count: totalTokens * epochs,
1066
+ estimated_eval_token_count: 0,
1067
+ user_limit: 100,
1068
+ },
1069
+ };
1070
+ }
1071
+
1072
+ /** POST /v1/fine-tunes/preview — Together's `FineTunePreviewResponse` (fine-tuning.d.ts:160):
1073
+ * tokenized preview rows over the SAMPLED training file, not a prose string. The token ids are
1074
+ * the twin's deterministic stand-ins; the shape and the per-row contract (input_ids/labels/
1075
+ * num_tokens/num_trained_tokens/tokens/trained_spans/truncated) are the vendor's. An unknown
1076
+ * training_file fails like the vendor. */
1077
+ function previewFinetuneTokenization(params: Record<string, unknown>, req: TogetheraiRequest): TogetheraiResponseEnvelope {
1078
+ if (typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
1079
+ if (typeof params.training_file !== 'string' || !params.training_file) return invalidRequest("'training_file' is a required property");
1080
+ const m = servedModels(req.root).find((pm) => pm.id === params.model);
1081
+ if (!m || m.type !== 'chat') return notFound(`Model ${params.model} does not exist or is not a fine-tunable model.`);
1082
+ const file = getRow('file', params.training_file, req.root);
1083
+ if (!file || file._deleted) return notFound(`No such File object: ${params.training_file}`);
1084
+ const maxSeq = m.context_length || 8_192;
1085
+ const topK = typeof params.top_k === 'number' && Number.isInteger(params.top_k) && params.top_k > 0 ? Math.min(params.top_k, 100) : 5;
1086
+ const trainOnInputs = params.train_on_inputs === undefined ? true : params.train_on_inputs === true;
1087
+ const sampled = jsonlRows(file._content).slice(0, topK);
1088
+ // The SDK documents dataset_format as DETECTED per sampled rows ("Detected SFT dataset format
1089
+ // for the sampled rows", fine-tuning.d.ts:158) — the check the vendor's own uploader runs
1090
+ // (lib/check-file.mjs: JSONL_REQUIRED_COLUMNS_MAP: general=['text'], conversation=['messages'],
1091
+ // instruction=['prompt','completion']). A file whose rows carry no recognizable column set
1092
+ // answers 'general', the format a bare text row is.
1093
+ const datasetFormat = (() => {
1094
+ const first = sampled[0];
1095
+ if (!first) return 'general';
1096
+ if ('messages' in first) return 'conversation';
1097
+ if ('prompt' in first && 'completion' in first) return 'instruction';
1098
+ return 'general';
1099
+ })();
1100
+ const rows = (sampled.length ? sampled : [{}]).map((row) => {
1101
+ const text = JSON.stringify(row);
1102
+ // Deterministic pseudo-token ids: one per ~4 chars (fnv1a-seeded), same input → same ids.
1103
+ const words = text.length > 0 ? Math.max(1, Math.ceil(text.length / 4)) : 1;
1104
+ const inputIds = Array.from({ length: words }, (_, i) => fnv1a(`${text}:${i}`) % 50_000);
1105
+ const labels = trainOnInputs ? [...inputIds] : inputIds.map(() => -100);
1106
+ const numTrained = labels.filter((l) => l !== -100).length;
1107
+ const truncated = words > maxSeq;
1108
+ return {
1109
+ input_ids: inputIds.slice(0, maxSeq),
1110
+ labels: labels.slice(0, maxSeq),
1111
+ num_tokens: Math.min(words, maxSeq),
1112
+ num_trained_tokens: Math.min(numTrained, maxSeq),
1113
+ tokens: inputIds.map((id) => `tok_${id.toString(36)}`).slice(0, maxSeq),
1114
+ trained_spans: trainOnInputs && numTrained > 0 ? [[0, Math.min(numTrained, maxSeq)]] : [],
1115
+ truncated,
1116
+ };
1117
+ });
1118
+ return {
1119
+ status: 200,
1120
+ body: {
1121
+ dataset_format: datasetFormat,
1122
+ max_seq_length: maxSeq,
1123
+ model: m.id,
1124
+ rows,
1125
+ train_on_inputs: trainOnInputs,
1126
+ },
1127
+ };
1128
+ }
1129
+
1130
+ /** GET /v1/fine-tunes/{id}/events — Together's `FinetuneEvent` list (fine-tuning.d.ts:234).
1131
+ * A job still `pending` has NO events yet (the vendor's job has not started); a job the twin
1132
+ * has seen move (cancel_requested or beyond) carries the events its history grounds — the
1133
+ * lifecycle events the vendor's own enum names, derived from the STORED row, never invented
1134
+ * progress. */
1135
+ function finetuneEvents(ft: Record<string, unknown>): Array<Record<string, unknown>> {
1136
+ const status = String(ft.status ?? 'pending');
1137
+ const at = (ft.created_at as string) ?? new Date(0).toISOString();
1138
+ const events: Array<Record<string, unknown>> = [];
1139
+ if (status !== 'pending') {
1140
+ events.push({ object: 'fine-tune-event', created_at: at, message: 'Job started', type: 'job_start' });
1141
+ }
1142
+ if (status === 'cancel_requested') {
1143
+ events.push({ object: 'fine-tune-event', created_at: at, message: 'Cancel requested', type: 'cancel_requested' });
1144
+ }
1145
+ return events;
1146
+ }
1147
+
1148
+ /** GET /v1/fine-tunes/{id}/checkpoints — Together's `FineTuningListCheckpointsResponse`
1149
+ * (fine-tuning.d.ts:1362). Checkpoints are artifacts a COMPLETED training run produced; the
1150
+ * twin runs no training and its jobs never progress past the states above, so an honest answer
1151
+ * over stored state is an EMPTY list — the vendor's shape, not a fabricated artifact. */
1152
+ function finetuneCheckpoints(ft: Record<string, unknown>): Array<Record<string, unknown>> {
1153
+ return String(ft.status) === 'completed'
1154
+ ? [{ checkpoint_type: 'final', created_at: String(ft.updated_at ?? ft.created_at ?? new Date(0).toISOString()), path: `twin://finetune/${String(ft.id)}/final`, step: 1 }]
1155
+ : [];
1156
+ }
1157
+
1158
+ // ── public entry: cross-cutting protocol (auth / fault triggers) then route ─────────────
1159
+ export async function handleTogetheraiTwinRequest(req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
1160
+ const method = req.method.toUpperCase();
1161
+ if (req.headers !== undefined || req.apiKey !== undefined) {
1162
+ const authErr = checkAuth(req);
1163
+ if (authErr) return authErr;
1164
+ }
1165
+ if (triggered(req, 'x-twin-force-rate-limit')) return rateLimitError();
1166
+ if (triggered(req, 'x-twin-force-spending-limit')) return spendingLimitError();
1167
+ if (triggered(req, 'x-twin-force-engine-overloaded')) return engineOverloadedError();
1168
+ return routeTogetherai(req, method);
1169
+ }
1170
+
1171
+ // ── router ──────────────────────────────────────────────────────────────────────────────
1172
+ async function routeTogetherai(req: TogetheraiRequest, method: string): Promise<TogetheraiResponseEnvelope> {
1173
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
1174
+ const search = req.path.includes('?') ? req.path.slice(req.path.indexOf('?') + 1) : '';
1175
+ const params = parseJson(req.body);
1176
+ // A percent-decode that never throws: `GET /v1/files/%zz` is a malformed escape, and the vendor
1177
+ // answers it like any other unknown object (404), not an internal error. decodeURIComponent
1178
+ // throws URIError on it — the same trap the upload door's decSafe guards.
1179
+ const dec = decSafe;
1180
+
1181
+ // D3: a read-only twin rejects any mutation with a vendor-shaped error.
1182
+ if (req.readOnly && method !== 'GET') {
1183
+ return { status: 405, body: errBody('invalid_request_error', 'twin is read-only; omit readOnly to accept writes') };
1184
+ }
1185
+
1186
+ // Everything Together's inference API serves hangs off /v1. A request outside it is a 404 like
1187
+ // any other unknown path.
1188
+ if (path !== TOGETHERAI_API_PREFIX && !path.startsWith(`${TOGETHERAI_API_PREFIX}/`)) {
1189
+ return notFound(`Unknown request URL: ${method} ${path}. Together's inference API is served under ${TOGETHERAI_API_PREFIX}/.`);
1190
+ }
1191
+ const seg = path.slice(TOGETHERAI_API_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean);
1192
+
1193
+ // ---- whoami (static identity for the presented key) ----
1194
+ if (seg[0] === 'whoami' && seg.length === 1 && method === 'GET') {
1195
+ return {
1196
+ status: 200,
1197
+ body: {
1198
+ api_key_id: 'key_twin',
1199
+ project_id: 'proj_twin',
1200
+ project_name: 'Twin Project',
1201
+ project_slug: 'twin-project',
1202
+ organization_id: 'org_twin',
1203
+ organization_name: 'Twin Organization',
1204
+ user_id: 'user_twin',
1205
+ },
1206
+ };
1207
+ }
1208
+
1209
+ // ---- models (static catalog) ----
1210
+ if (seg[0] === 'models' && seg.length === 1 && method === 'GET') {
1211
+ // Together answers a BARE ARRAY of ModelInfo (ModelInfoList), not an OpenAI list envelope.
1212
+ return { status: 200, body: servedModels(req.root) };
1213
+ }
1214
+ if (seg[0] === 'models' && seg.length >= 2 && method === 'GET') {
1215
+ // Together model ids contain slashes (`meta-llama/Llama-3.3-70B-Instruct-Turbo`), so the id
1216
+ // is EVERY remaining segment joined.
1217
+ const mid = seg.slice(1).map(dec).join('/');
1218
+ const m = servedModels(req.root).find((pm) => pm.id === mid);
1219
+ return m ? { status: 200, body: m } : notFound(`Model ${mid} does not exist.`);
1220
+ }
1221
+
1222
+ // ---- chat completions (the generative stub; envelope is faithful) ----
1223
+ if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
1224
+ const validated = validateChat(params);
1225
+ if ('error' in validated) return validated.error;
1226
+ const args = validated.args;
1227
+ const result = args.stream && req.sseSink
1228
+ ? streamChat(args, req.sseSink, req.occurredAt, req.scenarioEngine)
1229
+ : buildChatCompletion(args, req.occurredAt, req.scenarioEngine);
1230
+ if (isEnvelope(result)) return result;
1231
+ return { status: 200, body: result };
1232
+ }
1233
+
1234
+ // ---- completions (legacy text) ----
1235
+ if (seg[0] === 'completions' && seg.length === 1 && method === 'POST') {
1236
+ if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
1237
+ if (typeof params.prompt !== 'string') return invalidRequest("'prompt' is a required property");
1238
+ const info = findModel(params.model);
1239
+ if (!info || info.type !== 'chat' && info.type !== 'language' && info.type !== 'code') {
1240
+ return notFound(`Model ${params.model} does not exist or is not a completion model.`);
1241
+ }
1242
+ const prompt = params.prompt;
1243
+ let text = `[twin-stub:${params.model}] deterministic completion stub (no model weights are run) continuing: ${prompt.slice(0, 120) || '(empty)'}`;
1244
+ let finish: TogetheraiFinishReason = 'stop';
1245
+ const maxRaw = params.max_tokens;
1246
+ if (maxRaw !== undefined && maxRaw !== null) {
1247
+ const maxTokens = Number(maxRaw);
1248
+ if (!Number.isInteger(maxTokens) || maxTokens < 1) return invalidRequest("'max_tokens' must be an integer >= 1");
1249
+ if (estimateTokens(text) > maxTokens) {
1250
+ text = text.slice(0, maxTokens * 4);
1251
+ finish = 'length';
1252
+ }
1253
+ }
1254
+ const completion = {
1255
+ id: `cmpl-twin-${fnv1a(prompt + params.model).toString(36)}`,
1256
+ object: 'text.completion' as const,
1257
+ created: nowEpoch(req.occurredAt),
1258
+ model: params.model,
1259
+ prompt: [{ text: prompt }],
1260
+ choices: [{ text, index: 0, finish_reason: finish }],
1261
+ usage: buildUsage(estimateTokens(prompt), estimateTokens(text)),
1262
+ };
1263
+ // stream:true streams Together's CompletionChunk sequence (the legacy API streams too — the
1264
+ // SDK types it `Stream<CompletionChunk>`): a `token` chunk per piece, then the finish chunk.
1265
+ if (params.stream === true && req.sseSink) {
1266
+ const base = { id: completion.id, object: 'completion.chunk' as const, created: completion.created, model: params.model };
1267
+ for (const piece of chunkText(text)) {
1268
+ req.sseSink({ data: { ...base, token: { id: 0, logprob: 0, special: false, text: piece }, choices: [{ index: 0, text: piece }], finish_reason: null, usage: null } });
1269
+ }
1270
+ req.sseSink({ data: { ...base, token: { id: 0, logprob: 0, special: true, text: '' }, choices: [{ index: 0 }], finish_reason: finish, usage: completion.usage } });
1271
+ req.sseSink({ done: true });
1272
+ return { status: 200, body: completion };
1273
+ }
1274
+ return { status: 200, body: completion };
1275
+ }
1276
+
1277
+ // ---- embeddings ----
1278
+ if (seg[0] === 'embeddings' && seg.length === 1 && method === 'POST') return handleEmbeddings(params);
1279
+
1280
+ // ---- rerank (Together-native) ----
1281
+ if (seg[0] === 'rerank' && seg.length === 1 && method === 'POST') return handleRerank(params);
1282
+
1283
+ // ---- images ----
1284
+ if (seg[0] === 'images' && seg[1] === 'generations' && seg.length === 2 && method === 'POST') return handleImages(params);
1285
+
1286
+ // ---- audio ----
1287
+ if (seg[0] === 'audio' && seg[1] === 'transcriptions' && seg.length === 2 && method === 'POST') return handleTranscription(params, false);
1288
+ if (seg[0] === 'audio' && seg[1] === 'translations' && seg.length === 2 && method === 'POST') return handleTranscription(params, true);
1289
+ if (seg[0] === 'audio' && seg[1] === 'speech' && seg.length === 2 && method === 'POST') return handleSpeech(params);
1290
+
1291
+ // ---- files (stateful; BOTH upload flows) ----
1292
+ if (seg[0] === 'files' && seg.length === 1 && method === 'POST') {
1293
+ // The together-ai SDK's redirect upload flow addresses POST /files with its params in the
1294
+ // QUERY STRING and a form-urlencoded body (lib/upload.js:41). That query-param shape is the
1295
+ // ONLY create on this path: Together's spec inventory has no JSON create on /files
1296
+ // (test-fixtures/togetherai-openapi-operations.json — GET only), and a JSON or EMPTY body
1297
+ // mints nothing. Serving a create here would be an invented door (§ round two, defect 1);
1298
+ // the vendor's own table answers a misconfigured request 400 invalid_request_error
1299
+ // (docs.together.ai/docs/error-codes). The flow is recognized by PARSING the query for the
1300
+ // params it sends (file_name/purpose — lib/upload.js:40), never by substring-matching the
1301
+ // raw string: `?file_type=jsonl` alone must NOT be mistaken for the flow.
1302
+ const q = new URLSearchParams(search);
1303
+ const hasFile = q.has('file_name');
1304
+ const hasPurpose = q.has('purpose');
1305
+ if (hasFile || hasPurpose) {
1306
+ return createFileSdkRedirect({
1307
+ ...(hasPurpose ? { purpose: q.get('purpose') ?? undefined } : {}),
1308
+ ...(hasFile ? { file_name: q.get('file_name') ?? undefined } : {}),
1309
+ ...(q.has('file_type') ? { file_type: q.get('file_type') ?? undefined } : {}),
1310
+ ...(typeof params.content === 'string' ? { content: params.content } : {}),
1311
+ }, req, q);
1312
+ }
1313
+ return invalidRequest("POST /v1/files takes the upload flow's urlencoded query parameters (?file_name=&file_type=&purpose=); the create itself is POST /v1/files/upload (multipart)");
1314
+ }
1315
+ if (seg[0] === 'files' && seg.length === 1 && method === 'GET') {
1316
+ // Together answers `{ data: [...] }` (FileList), NOT OpenAI's `{object:'list',data}`.
1317
+ return { status: 200, body: { data: rows('file', req.root).filter((r) => !r._deleted).map(fileView) } };
1318
+ }
1319
+ if (seg[0] === 'files' && seg[1] === 'upload' && seg.length === 2 && method === 'POST') {
1320
+ // The spec's multipart form upload (the server adapts multipart → JSON before the handler).
1321
+ return createFileMultipart(params, req);
1322
+ }
1323
+ if (seg[0] === 'files' && seg.length === 2 && method === 'GET') {
1324
+ const f = getRow('file', dec(seg[1]!), req.root);
1325
+ return f && !f._deleted ? { status: 200, body: fileView(f) } : notFound(`No such File object: ${dec(seg[1]!)}`);
1326
+ }
1327
+ if (seg[0] === 'files' && seg.length === 3 && seg[2] === 'content' && method === 'GET') {
1328
+ const f = getRow('file', dec(seg[1]!), req.root);
1329
+ if (!f || f._deleted) return notFound(`No such File object: ${dec(seg[1]!)}`);
1330
+ return { status: 200, body: String(f._content ?? ''), headers: { 'content-type': 'application/octet-stream' } };
1331
+ }
1332
+ if (seg[0] === 'files' && seg.length === 2 && method === 'DELETE') {
1333
+ const fid = dec(seg[1]!);
1334
+ const f = getRow('file', fid, req.root);
1335
+ if (!f || f._deleted) return notFound(`No such File object: ${fid}`);
1336
+ await applyTwinWrite(SERVICE, {
1337
+ operation: 'file.delete', subjectType: 'file', subjectId: fid, fields: { _deleted: true, object: 'file' },
1338
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1339
+ }, req.root);
1340
+ return { status: 200, body: { id: fid, deleted: true } };
1341
+ }
1342
+
1343
+ // ---- batches (stateful, Together-native shapes) ----
1344
+ if (seg[0] === 'batches' && seg.length === 1 && method === 'POST') return createBatch(params, req);
1345
+ if (seg[0] === 'batches' && seg.length === 1 && method === 'GET') {
1346
+ // Together answers a BARE ARRAY of BatchJob (its own published schema), NOT an OpenAI
1347
+ // `{object:'list',data}` envelope.
1348
+ return { status: 200, body: rows('batch', req.root).map(batchView) };
1349
+ }
1350
+ if (seg[0] === 'batches' && seg.length === 2 && method === 'GET') {
1351
+ const b = getRow('batch', dec(seg[1]!), req.root);
1352
+ if (!b) return notFound(`No such Batch object: ${dec(seg[1]!)}`);
1353
+ return { status: 200, body: batchView(b) };
1354
+ }
1355
+ if (seg[0] === 'batches' && seg.length === 3 && seg[2] === 'cancel' && method === 'POST') {
1356
+ const bid = dec(seg[1]!);
1357
+ const b = getRow('batch', bid, req.root);
1358
+ if (!b) return notFound(`No such Batch object: ${bid}`);
1359
+ if (b.status === 'CANCELLED') return invalidRequest(`Cannot cancel a batch with status '${String(b.status)}'.`);
1360
+ // NOTE: the twin does not simulate the asynchronous VALIDATING→IN_PROGRESS→COMPLETED
1361
+ // progression. Doing it on a READ made a GET write (breaking the read-only contract) and
1362
+ // minted actions the connector has no vendor endpoint to push (the groq pack's §9 round-one
1363
+ // findings 4 + 5). The terminal transitions are filed as todos.
1364
+ await applyTwinWrite(SERVICE, {
1365
+ operation: 'batch.cancel', subjectType: 'batch', subjectId: bid,
1366
+ fields: { status: 'CANCELLED' as TogetheraiBatchStatus },
1367
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1368
+ }, req.root);
1369
+ return { status: 200, body: batchView(getRow('batch', bid, req.root) ?? {}) };
1370
+ }
1371
+
1372
+ // ---- fine-tunes (stateful, Together-native shapes) ----
1373
+ if (seg[0] === 'fine-tunes' && seg.length === 1 && method === 'POST') return createFinetune(params, req);
1374
+ if (seg[0] === 'fine-tunes' && seg.length === 1 && method === 'GET') {
1375
+ return { status: 200, body: rows('finetune', req.root).filter((r) => !r._deleted).map(finetuneView) };
1376
+ }
1377
+ if (seg[0] === 'fine-tunes' && seg[1] === 'models' && seg[2] === 'limits' && seg.length === 3 && method === 'GET') {
1378
+ return finetuneModelLimits(req);
1379
+ }
1380
+ if (seg[0] === 'fine-tunes' && seg[1] === 'estimate-price' && seg.length === 2 && method === 'POST') {
1381
+ return estimateFinetunePrice(params, req);
1382
+ }
1383
+ if (seg[0] === 'fine-tunes' && seg[1] === 'preview' && seg.length === 2 && method === 'POST') {
1384
+ return previewFinetuneTokenization(params, req);
1385
+ }
1386
+ if (seg[0] === 'fine-tunes' && seg.length === 2 && method === 'GET') {
1387
+ const ft = getRow('finetune', dec(seg[1]!), req.root);
1388
+ return ft && !ft._deleted ? { status: 200, body: finetuneView(ft) } : notFound(`No such Fine-tune object: ${dec(seg[1]!)}`);
1389
+ }
1390
+ if (seg[0] === 'fine-tunes' && seg.length === 3 && seg[2] === 'cancel' && method === 'POST') {
1391
+ const fid = dec(seg[1]!);
1392
+ const ft = getRow('finetune', fid, req.root);
1393
+ if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${fid}`);
1394
+ if (ft.status === 'cancelled' || ft.status === 'cancel_requested' || ft.status === 'completed' || ft.status === 'error') {
1395
+ return invalidRequest(`Cannot cancel a fine-tune with status '${String(ft.status)}'.`);
1396
+ }
1397
+ await applyTwinWrite(SERVICE, {
1398
+ operation: 'finetune.cancel', subjectType: 'finetune', subjectId: fid,
1399
+ fields: { status: 'cancel_requested' },
1400
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1401
+ }, req.root);
1402
+ return { status: 200, body: finetuneView(getRow('finetune', fid, req.root) ?? {}) };
1403
+ }
1404
+ if (seg[0] === 'fine-tunes' && seg.length === 3 && seg[2] === 'events' && method === 'GET') {
1405
+ const ft = getRow('finetune', dec(seg[1]!), req.root);
1406
+ if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${dec(seg[1]!)}`);
1407
+ return { status: 200, body: { data: finetuneEvents(ft) } };
1408
+ }
1409
+ if (seg[0] === 'fine-tunes' && seg.length === 3 && seg[2] === 'checkpoints' && method === 'GET') {
1410
+ const ft = getRow('finetune', dec(seg[1]!), req.root);
1411
+ if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${dec(seg[1]!)}`);
1412
+ return { status: 200, body: { data: finetuneCheckpoints(ft) } };
1413
+ }
1414
+ if (seg[0] === 'fine-tunes' && seg.length === 2 && method === 'DELETE') {
1415
+ const fid = dec(seg[1]!);
1416
+ const ft = getRow('finetune', fid, req.root);
1417
+ if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${fid}`);
1418
+ await applyTwinWrite(SERVICE, {
1419
+ operation: 'finetune.delete', subjectType: 'finetune', subjectId: fid, fields: { _deleted: true, object: 'finetune' },
1420
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1421
+ }, req.root);
1422
+ // Together's fine-tune delete answers {message} (together-ai@0.53.0 fine-tuning.d.ts:1073) —
1423
+ // NOT OpenAI's {id, deleted} (which is the FILE delete shape, files.d.ts:168, served above).
1424
+ return { status: 200, body: { message: `Fine-tune ${fid} deleted.` } };
1425
+ }
1426
+
1427
+ // Unmodeled operation → fail like the vendor (never a fake success).
1428
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1429
+ }
1430
+
1431
+ /** The twin-only door the SDK redirect flow uploads bytes through (kept OUT of the manifest:
1432
+ * twin-only scaffolding, not vendor surface — the same rule the `/twin` doors follow). */
1433
+ export const TOGETHERAI_UPLOAD_DOOR = '/twin/upload';
1434
+ export async function handleTogetheraiUploadDoor(req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
1435
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '');
1436
+ const m = new RegExp(`^${TOGETHERAI_UPLOAD_DOOR}/([^/]+)$`).exec(path);
1437
+ if (!m || req.method.toUpperCase() !== 'PUT') return notFound(`Unknown request URL: ${req.method} ${path}.`);
1438
+ // D3: the door WRITES (file.update), so a read-only twin refuses it with the same vendor-shaped
1439
+ // 405 every other write door answers — the server forwards readOnly here, and ignoring it let
1440
+ // `PUT /twin/upload/<id>` write on a --read-only server (§ round three, defect 2).
1441
+ if (req.readOnly) {
1442
+ return { status: 405, body: errBody('invalid_request_error', 'twin is read-only; omit readOnly to accept writes') };
1443
+ }
1444
+ return storeUploadBytes(decSafe(m[1]!), typeof req.body === 'string' ? req.body : '', req);
1445
+ }
1446
+ function decSafe(s: string): string {
1447
+ try { return decodeURIComponent(s); } catch { return s; }
1448
+ }