@volter/twin-togetherai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +147 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/index.d.ts +14 -0
- package/dist/src/index.js +79 -0
- package/dist/src/togetherai-budget.d.ts +52 -0
- package/dist/src/togetherai-budget.js +130 -0
- package/dist/src/togetherai-capabilities.d.ts +4 -0
- package/dist/src/togetherai-capabilities.js +1428 -0
- package/dist/src/togetherai-conformance.d.ts +14 -0
- package/dist/src/togetherai-conformance.js +452 -0
- package/dist/src/togetherai-connector.d.ts +164 -0
- package/dist/src/togetherai-connector.js +457 -0
- package/dist/src/togetherai-models.d.ts +19 -0
- package/dist/src/togetherai-models.js +49 -0
- package/dist/src/togetherai-scenario.d.ts +52 -0
- package/dist/src/togetherai-scenario.js +168 -0
- package/dist/src/togetherai-server.d.ts +16 -0
- package/dist/src/togetherai-server.js +187 -0
- package/dist/src/togetherai-stub.d.ts +59 -0
- package/dist/src/togetherai-stub.js +195 -0
- package/dist/src/togetherai-twin.d.ts +83 -0
- package/dist/src/togetherai-twin.js +1419 -0
- package/dist/src/togetherai-types.d.ts +207 -0
- package/dist/src/togetherai-types.js +26 -0
- package/package.json +52 -0
- package/src/cli.ts +27 -0
- package/src/index.ts +118 -0
- package/src/togetherai-budget.ts +156 -0
- package/src/togetherai-capabilities.ts +1315 -0
- package/src/togetherai-conformance.ts +459 -0
- package/src/togetherai-connector.ts +496 -0
- package/src/togetherai-models.ts +74 -0
- package/src/togetherai-scenario.ts +185 -0
- package/src/togetherai-server.ts +199 -0
- package/src/togetherai-stub.ts +197 -0
- package/src/togetherai-twin.ts +1448 -0
- package/src/togetherai-types.ts +222 -0
|
@@ -0,0 +1,1448 @@
|
|
|
1
|
+
// Together AI twin REQUEST HANDLER — the canonical Together AI surface for the twin.
|
|
2
|
+
// Contract: handleTogetheraiTwinRequest({method, path, body}) -> {status, body}. It is the
|
|
3
|
+
// faithful Together AI API the real `together-ai` SDK (pointed at this baseURL) talks to
|
|
4
|
+
// UNMODIFIED — with ONE documented exception: the SDK's `files.upload()` (its custom 302-redirect
|
|
5
|
+
// upload) reads `TOGETHER_API_BASE_URL` at MODULE LOAD and ignores the client's `baseURL`
|
|
6
|
+
// (together-ai@0.53.0 lib/upload.js:12), so it must be pointed at the twin explicitly with
|
|
7
|
+
// `TOGETHER_API_BASE_URL=<twin>/v1` — unset, it egresses to https://api.together.xyz with the
|
|
8
|
+
// caller's real key. Every other SDK call rides `baseURL` (or the injector's api.together.ai
|
|
9
|
+
// interception) and needs nothing else.
|
|
10
|
+
//
|
|
11
|
+
// THE HONEST DESIGN: the twin cannot run the model, so `POST /v1/chat/completions` returns a
|
|
12
|
+
// DETERMINISTIC STUB completion (togetherai-stub.ts) clearly labeled a twin stub — it NEVER
|
|
13
|
+
// pretends to be real model output. `/v1/embeddings` returns DETERMINISTIC pseudo-vectors and
|
|
14
|
+
// the audio endpoints return DETERMINISTIC labeled stubs. But the ENTIRE PROTOCOL ENVELOPE is
|
|
15
|
+
// vendor-faithful: response shapes (including Together's REQUIRED `prompt` array and its `eos`
|
|
16
|
+
// finish_reason), streaming SSE chunks with the nullable per-chunk `usage`/`warnings`,
|
|
17
|
+
// tool_calls, and Together's OWN status table (402 spending limit, 403 context length, 503
|
|
18
|
+
// engine overloaded). The genuinely stateful + static surface is real:
|
|
19
|
+
// • GET /v1/models — static catalog (togetherai-models.ts)
|
|
20
|
+
// • GET /v1/whoami — static identity for the presented key
|
|
21
|
+
// • POST /v1/files/upload + GET/DELETE /v1/files{,/:id,/content} — stateful (kernel action
|
|
22
|
+
// log), BOTH upload flows: the spec's multipart POST /v1/files/upload AND the together-ai
|
|
23
|
+
// SDK's own redirect dance
|
|
24
|
+
// (POST /v1/files?<params> → 302 + x-together-file-id → PUT the bytes).
|
|
25
|
+
// There is NO JSON create at POST /v1/files — the spec lists GET only there; a bodyless or
|
|
26
|
+
// JSON POST is a 400 naming the real doors.
|
|
27
|
+
// • POST/GET /v1/batches (+ cancel) — stateful, Together-native shapes (201
|
|
28
|
+
// BatchJobWithWarning, bare-array list, `{error: string}` failures)
|
|
29
|
+
// • POST/GET/DELETE /v1/fine-tunes (+ cancel/events) — stateful, Together-native
|
|
30
|
+
//
|
|
31
|
+
// THE OPENAI-COMPATIBILITY LINE (docs.together.ai/docs/inference/openai-compatibility, read
|
|
32
|
+
// 2026-09-16): Together's OpenAI-compatible surface is the INFERENCE endpoints only. Assistants/
|
|
33
|
+
// Threads/Runs are NOT implemented; `moderations.create` is NOT implemented; OpenAI-shaped
|
|
34
|
+
// Batch/Files/Fine-tuning APIs are NOT supported (Together ships its own native equivalents,
|
|
35
|
+
// modeled here); `service_tier`/`store`/`metadata`/`prediction` are ACCEPTED BUT IGNORED. A twin
|
|
36
|
+
// that served the OpenAI shapes on those stateful resources would be surface the vendor does not
|
|
37
|
+
// have — the inverse false-green (ADDING_A_TWIN.md §6).
|
|
38
|
+
//
|
|
39
|
+
// State lives in the kernel action log (D1): all writes are local actions, reads are the
|
|
40
|
+
// projection. No real Together API is ever called from this path (D4). Streaming uses an
|
|
41
|
+
// INJECTED sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
|
|
42
|
+
import { applyTwinWrite, projectResources } from '@volter/world-core';
|
|
43
|
+
import {
|
|
44
|
+
EMBEDDING_DIMENSIONS,
|
|
45
|
+
EMBEDDING_MODELS,
|
|
46
|
+
RERANK_MODELS,
|
|
47
|
+
SPEECH_MODELS,
|
|
48
|
+
TOGETHERAI_MODELS,
|
|
49
|
+
findModel,
|
|
50
|
+
} from './togetherai-models.ts';
|
|
51
|
+
import {
|
|
52
|
+
buildUsage,
|
|
53
|
+
contentToText,
|
|
54
|
+
countPromptTokens,
|
|
55
|
+
estimateTokens,
|
|
56
|
+
fnv1a,
|
|
57
|
+
pseudoEmbedding,
|
|
58
|
+
stubAssistantText,
|
|
59
|
+
stubAudioSeconds,
|
|
60
|
+
stubJsonObject,
|
|
61
|
+
stubReasoningText,
|
|
62
|
+
stubToolCall,
|
|
63
|
+
stubTranscript,
|
|
64
|
+
} from './togetherai-stub.ts';
|
|
65
|
+
import { type TogetheraiScenarioEngine, type TogetheraiScenarioRespond, realizeTogetheraiRespond, type ScriptedResult } from './togetherai-scenario.ts';
|
|
66
|
+
import type {
|
|
67
|
+
TogetheraiAssistantMessage,
|
|
68
|
+
TogetheraiBatch,
|
|
69
|
+
TogetheraiBatchEndpoint,
|
|
70
|
+
TogetheraiBatchStatus,
|
|
71
|
+
TogetheraiChatCompletion,
|
|
72
|
+
TogetheraiChoice,
|
|
73
|
+
TogetheraiEmbedding,
|
|
74
|
+
TogetheraiEmbeddingResponse,
|
|
75
|
+
TogetheraiError,
|
|
76
|
+
TogetheraiFile,
|
|
77
|
+
TogetheraiFilePurpose,
|
|
78
|
+
TogetheraiFileType,
|
|
79
|
+
TogetheraiFinishReason,
|
|
80
|
+
TogetheraiMessageParam,
|
|
81
|
+
TogetheraiModel,
|
|
82
|
+
TogetheraiRerankResponse,
|
|
83
|
+
TogetheraiToolCall,
|
|
84
|
+
TogetheraiUsage,
|
|
85
|
+
SseSink,
|
|
86
|
+
} from './togetherai-types.ts';
|
|
87
|
+
|
|
88
|
+
const SERVICE = 'togetherai';
|
|
89
|
+
|
|
90
|
+
/** The ONE base path Together's inference API serves. Everything the twin routes hangs off this. */
|
|
91
|
+
export const TOGETHERAI_API_PREFIX = '/v1';
|
|
92
|
+
|
|
93
|
+
export type TogetheraiRequest = {
|
|
94
|
+
/** The scenario engine (kernel grammar + this pack's vocabulary) — scripts chat turns. */
|
|
95
|
+
scenarioEngine?: TogetheraiScenarioEngine;
|
|
96
|
+
method: string;
|
|
97
|
+
path: string;
|
|
98
|
+
body?: string;
|
|
99
|
+
occurredAt?: string;
|
|
100
|
+
root?: string;
|
|
101
|
+
readOnly?: boolean;
|
|
102
|
+
/** The credential the caller presents (the SDK's bearer `Authorization` header). When a request
|
|
103
|
+
* carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the real
|
|
104
|
+
* vendor rule: a credential is required → 401 on missing/invalid. In-process trusted calls
|
|
105
|
+
* (capability verify, connector) omit BOTH and are not auth-gated. */
|
|
106
|
+
apiKey?: string;
|
|
107
|
+
/** Lower-cased request headers the HTTP server passes through so the handler can model auth
|
|
108
|
+
* (401) and the deterministic fault triggers (429 / 402 / 503). */
|
|
109
|
+
headers?: Record<string, string>;
|
|
110
|
+
/** When set on a streaming POST, chunks are written here (no sockets). */
|
|
111
|
+
sseSink?: SseSink;
|
|
112
|
+
};
|
|
113
|
+
|
|
114
|
+
/** The handler response. `headers` (when present) are response headers the HTTP server should set
|
|
115
|
+
* — e.g. `x-ratelimit-reset` on a modeled 429, or the redirect pair on the SDK's upload flow. */
|
|
116
|
+
export type TogetheraiResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
|
|
117
|
+
|
|
118
|
+
// ── vendor-shaped errors ──────────────────────────────────────────────────────────────
|
|
119
|
+
/**
|
|
120
|
+
* Together's error envelope (`ErrorData`): `{ error: { message, type, param, code } }` with
|
|
121
|
+
* `message` + `type` REQUIRED and `param`/`code` nullable with default null. The twin emits all
|
|
122
|
+
* four keys so the served shape matches the vendor's REQUIRED+DEFAULT shape exactly.
|
|
123
|
+
*/
|
|
124
|
+
function errBody(type: string, message: string, param: string | null = null, code: string | null = null): TogetheraiError {
|
|
125
|
+
return { error: { message, type, param, code } };
|
|
126
|
+
}
|
|
127
|
+
function invalidRequest(message: string): TogetheraiResponseEnvelope {
|
|
128
|
+
return { status: 400, body: errBody('invalid_request_error', message) };
|
|
129
|
+
}
|
|
130
|
+
function notFound(message: string): TogetheraiResponseEnvelope {
|
|
131
|
+
return { status: 404, body: errBody('invalid_request_error', message) };
|
|
132
|
+
}
|
|
133
|
+
function authError(message: string): TogetheraiResponseEnvelope {
|
|
134
|
+
return { status: 401, body: errBody('invalid_request_error', message) };
|
|
135
|
+
}
|
|
136
|
+
/** Together's 403 is NOT a permission denial — docs.together.ai/docs/error-codes: 403 means
|
|
137
|
+
* "the sum of input tokens plus max_tokens exceeds the context length of the model". */
|
|
138
|
+
function contextLengthError(model: string, needed: number, limit: number): TogetheraiResponseEnvelope {
|
|
139
|
+
return {
|
|
140
|
+
status: 403,
|
|
141
|
+
body: errBody('invalid_request_error', `This model's maximum context length is ${limit} tokens. However, you requested ${needed} tokens (${needed - limit} in the messages, Please reduce the length of the messages or max_tokens.`),
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
// ── modeled authentication (401) ────────────────────────────────────────────────────────
|
|
146
|
+
// Real Together requires a bearer credential on every request and returns 401 when it is missing
|
|
147
|
+
// or invalid (docs.together.ai/docs/error-codes: 401 = "A missing or invalid API key"). The twin
|
|
148
|
+
// can't validate against real keys, so it models the CHECKABLE failures: a missing credential,
|
|
149
|
+
// and a reserved sentinel for the invalid-key path. Any other non-empty key is accepted. Trusted
|
|
150
|
+
// in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated; both official
|
|
151
|
+
// clients always send a key → they pass.
|
|
152
|
+
function checkAuth(req: TogetheraiRequest): TogetheraiResponseEnvelope | null {
|
|
153
|
+
const auth = req.headers?.['authorization'];
|
|
154
|
+
const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
|
|
155
|
+
const key = (req.apiKey ?? '').trim() || bearer;
|
|
156
|
+
if (!key) return authError('Missing or invalid API Key');
|
|
157
|
+
if (key === 'twin_invalid' || key === 'invalid') return authError('Missing or invalid API Key');
|
|
158
|
+
return null;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
// ── modeled rate limiting (429), spending limit (402), overload (503) ───────────────────
|
|
162
|
+
// All three are non-deterministic in production, so the twin exposes DETERMINISTIC opt-in
|
|
163
|
+
// triggers. Together's documented behaviors (docs.together.ai/docs/error-codes +
|
|
164
|
+
// /docs/serverless/rate-limits, read 2026-09-16):
|
|
165
|
+
// • 429 — serverless rate limit exceeded; error types `dynamic_request_limited` /
|
|
166
|
+
// `dynamic_token_limited`; carries `x-ratelimit-reset` (seconds until reset). Success
|
|
167
|
+
// responses carry NO rate-limit headers.
|
|
168
|
+
// • 402 — the account hit its monthly spending limit ("Payment Required").
|
|
169
|
+
// • 503 — "Engine Overloaded: servers are under heavy traffic".
|
|
170
|
+
function rateLimitError(kind: 'dynamic_request_limited' | 'dynamic_token_limited' = 'dynamic_request_limited'): TogetheraiResponseEnvelope {
|
|
171
|
+
return {
|
|
172
|
+
status: 429,
|
|
173
|
+
body: errBody(kind, 'Rate limit exceeded: dynamic request limit reached for this model. Please retry after the reset window.'),
|
|
174
|
+
headers: { 'x-ratelimit-reset': '60' },
|
|
175
|
+
};
|
|
176
|
+
}
|
|
177
|
+
function spendingLimitError(): TogetheraiResponseEnvelope {
|
|
178
|
+
return { status: 402, body: errBody('insufficient_quota', 'The account associated with the API key has reached its maximum allowed monthly spending limit.') };
|
|
179
|
+
}
|
|
180
|
+
function engineOverloadedError(): TogetheraiResponseEnvelope {
|
|
181
|
+
return { status: 503, body: errBody('engine_overloaded', 'Engine overloaded: servers are under heavy traffic. Please retry after a short wait.') };
|
|
182
|
+
}
|
|
183
|
+
function triggered(req: TogetheraiRequest, header: string): boolean {
|
|
184
|
+
const v = req.headers?.[header];
|
|
185
|
+
return v === '1' || v === 'true';
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
function nowEpoch(occurredAt?: string): number {
|
|
189
|
+
return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
// ── kernel helpers ──────────────────────────────────────────────────────────────────────
|
|
193
|
+
function rows(type: string, root?: string): Array<Record<string, unknown>> {
|
|
194
|
+
return projectResources(SERVICE, root).filter((r) => r.type === type);
|
|
195
|
+
}
|
|
196
|
+
/**
|
|
197
|
+
* Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
|
|
198
|
+
* projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
|
|
199
|
+
* gap above the count (ADDING_A-TWIN.md §5). Two further properties matter:
|
|
200
|
+
* • the `_twin_` infix namespaces LOCAL mints, so a pulled Together id can never be matched by
|
|
201
|
+
* this regex and therefore can never be re-minted;
|
|
202
|
+
* • the scan includes TOMBSTONED rows (a soft-deleted file keeps its projection row), so the
|
|
203
|
+
* counter RATCHETS across delete→recreate and a deleted id is never handed out twice.
|
|
204
|
+
*/
|
|
205
|
+
function nextId(type: string, prefix: string, root?: string): string {
|
|
206
|
+
let max = 0;
|
|
207
|
+
for (const r of rows(type, root)) {
|
|
208
|
+
const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(String(r.id));
|
|
209
|
+
if (m) max = Math.max(max, Number(m[1]));
|
|
210
|
+
}
|
|
211
|
+
return `${prefix}_twin_${max + 1}`;
|
|
212
|
+
}
|
|
213
|
+
/** Models observed by a connector pull (mapModel), reshaped into the served model object. */
|
|
214
|
+
function pulledModels(root?: string): TogetheraiModel[] {
|
|
215
|
+
return rows('model', root)
|
|
216
|
+
.filter((r) => !r._deleted)
|
|
217
|
+
.map((r) => ({ id: String(r.id), object: 'model', created: Number(r.created ?? 0), type: (r.type as TogetheraiModel['type']) ?? 'language' }));
|
|
218
|
+
}
|
|
219
|
+
/** The catalog a request sees: the static table, with any PULLED row of the same id OVERRIDING it
|
|
220
|
+
* (the groq pack's §9-round-two lesson: a pulled row must stay served, not shadowed). */
|
|
221
|
+
function servedModels(root?: string): TogetheraiModel[] {
|
|
222
|
+
const pulled = pulledModels(root);
|
|
223
|
+
const byId = new Map<string, TogetheraiModel>();
|
|
224
|
+
for (const m of TOGETHERAI_MODELS) byId.set(m.id, m);
|
|
225
|
+
for (const m of pulled) byId.set(m.id, m);
|
|
226
|
+
return [...byId.values()];
|
|
227
|
+
}
|
|
228
|
+
function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined {
|
|
229
|
+
return rows(type, root).find((r) => r.id === id);
|
|
230
|
+
}
|
|
231
|
+
/** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
|
|
232
|
+
function strip(r: Record<string, unknown>): Record<string, unknown> {
|
|
233
|
+
const out: Record<string, unknown> = {};
|
|
234
|
+
for (const [k, v] of Object.entries(r)) {
|
|
235
|
+
if (k === 'type' || k === 'updatedAt' || k.startsWith('_')) continue;
|
|
236
|
+
out[k] = v;
|
|
237
|
+
}
|
|
238
|
+
return out;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
// ── request parsing ─────────────────────────────────────────────────────────────────────
|
|
242
|
+
function parseJson(body?: string): Record<string, unknown> {
|
|
243
|
+
if (!body || !body.trim()) return {};
|
|
244
|
+
try {
|
|
245
|
+
const v = JSON.parse(body);
|
|
246
|
+
return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
|
|
247
|
+
} catch {
|
|
248
|
+
return {};
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
// ── chat completions: validate the request the way Together does ────────────────────────
|
|
253
|
+
type ToolChoice = 'auto' | 'none' | 'required' | { name: string };
|
|
254
|
+
type ResponseFormat = { kind: 'text' } | { kind: 'json_object' } | { kind: 'json_schema'; schema: unknown };
|
|
255
|
+
|
|
256
|
+
/** Together's `ChatCompletionRequest.context_length_exceeded_behavior` — a CLOSED documented
|
|
257
|
+
* enum with default `error`. */
|
|
258
|
+
const CONTEXT_LENGTH_BEHAVIORS = ['truncate', 'error'] as const;
|
|
259
|
+
type ContextLengthBehavior = (typeof CONTEXT_LENGTH_BEHAVIORS)[number];
|
|
260
|
+
/** Together's `reasoning_effort` — a CLOSED documented enum. */
|
|
261
|
+
const REASONING_EFFORTS = ['low', 'medium', 'high'] as const;
|
|
262
|
+
type ReasoningEffort = (typeof REASONING_EFFORTS)[number];
|
|
263
|
+
|
|
264
|
+
type ChatArgs = {
|
|
265
|
+
model: string;
|
|
266
|
+
messages: TogetheraiMessageParam[];
|
|
267
|
+
tools?: unknown;
|
|
268
|
+
n: number;
|
|
269
|
+
maxTokens?: number;
|
|
270
|
+
stop?: string[];
|
|
271
|
+
stream: boolean;
|
|
272
|
+
toolChoice?: ToolChoice;
|
|
273
|
+
parallelToolCalls: boolean;
|
|
274
|
+
legacyFunctions: boolean;
|
|
275
|
+
responseFormat: ResponseFormat;
|
|
276
|
+
seed?: number;
|
|
277
|
+
echo: boolean;
|
|
278
|
+
logprobs?: number;
|
|
279
|
+
reasoningEffort?: ReasoningEffort;
|
|
280
|
+
contextBehavior: ContextLengthBehavior;
|
|
281
|
+
};
|
|
282
|
+
|
|
283
|
+
function validateChat(params: Record<string, unknown>): { args: ChatArgs } | { error: TogetheraiResponseEnvelope } {
|
|
284
|
+
if (params.model === undefined || params.model === '') return { error: invalidRequest("'model' is a required property") };
|
|
285
|
+
if (typeof params.model !== 'string') return { error: invalidRequest("'model' must be a string") };
|
|
286
|
+
if (!Array.isArray(params.messages)) return { error: invalidRequest("'messages' is a required property") };
|
|
287
|
+
if (params.messages.length === 0) return { error: invalidRequest("[] is too short - 'messages'") };
|
|
288
|
+
const messages = params.messages as TogetheraiMessageParam[];
|
|
289
|
+
for (const m of messages) {
|
|
290
|
+
if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
|
|
291
|
+
return { error: invalidRequest("each message must have a valid 'role'") };
|
|
292
|
+
}
|
|
293
|
+
}
|
|
294
|
+
// Together's audio/embedding/rerank/image models are not chat models — asking one to chat is
|
|
295
|
+
// a 404 (Together's docs: 404 = "An invalid endpoint URL or model name"), not a stub. A model
|
|
296
|
+
// id the catalog does not know AT ALL (an OpenAI-style flat id like `gpt-4o`) is the same 404:
|
|
297
|
+
// Together serves only its slash-namespaced catalog.
|
|
298
|
+
if (EMBEDDING_MODELS.has(params.model) || RERANK_MODELS.has(params.model) || SPEECH_MODELS.has(params.model)
|
|
299
|
+
|| !findModel(params.model) || findModel(params.model)?.type === 'image') {
|
|
300
|
+
return { error: notFound(`Model ${params.model} does not exist or is not a chat model.`) };
|
|
301
|
+
}
|
|
302
|
+
// `n`: Together's schema pins minimum 1, maximum 128 — an OpenAI-compatible freedom Groq does
|
|
303
|
+
// not have. A number outside the documented range is a 400.
|
|
304
|
+
if (params.n !== undefined && params.n !== null) {
|
|
305
|
+
if (typeof params.n !== 'number' || !Number.isInteger(params.n) || params.n < 1 || params.n > 128) {
|
|
306
|
+
return { error: invalidRequest("'n' must be an integer between 1 and 128") };
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
// `logprobs` is an INTEGER 0–20 on Together (OpenAI's boolean; Groq 400s the field outright).
|
|
310
|
+
if (params.logprobs !== undefined && params.logprobs !== null) {
|
|
311
|
+
if (typeof params.logprobs !== 'number' || !Number.isInteger(params.logprobs) || params.logprobs < 0 || params.logprobs > 20) {
|
|
312
|
+
return { error: invalidRequest("'logprobs' must be an integer between 0 and 20") };
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
// `context_length_exceeded_behavior` — a CLOSED documented enum, default 'error'.
|
|
316
|
+
let contextBehavior: ContextLengthBehavior = 'error';
|
|
317
|
+
if (params.context_length_exceeded_behavior !== undefined && params.context_length_exceeded_behavior !== null) {
|
|
318
|
+
if (typeof params.context_length_exceeded_behavior !== 'string' || !CONTEXT_LENGTH_BEHAVIORS.includes(params.context_length_exceeded_behavior as ContextLengthBehavior)) {
|
|
319
|
+
return { error: invalidRequest(`'context_length_exceeded_behavior' must be one of ${CONTEXT_LENGTH_BEHAVIORS.map((t) => `'${t}'`).join(', ')}`) };
|
|
320
|
+
}
|
|
321
|
+
contextBehavior = params.context_length_exceeded_behavior as ContextLengthBehavior;
|
|
322
|
+
}
|
|
323
|
+
// `reasoning_effort` — a CLOSED documented enum.
|
|
324
|
+
let reasoningEffort: ReasoningEffort | undefined;
|
|
325
|
+
if (params.reasoning_effort !== undefined && params.reasoning_effort !== null) {
|
|
326
|
+
if (typeof params.reasoning_effort !== 'string' || !REASONING_EFFORTS.includes(params.reasoning_effort as ReasoningEffort)) {
|
|
327
|
+
return { error: invalidRequest(`'reasoning_effort' must be one of ${REASONING_EFFORTS.map((t) => `'${t}'`).join(', ')}`) };
|
|
328
|
+
}
|
|
329
|
+
reasoningEffort = params.reasoning_effort as ReasoningEffort;
|
|
330
|
+
}
|
|
331
|
+
// `compliance` is a CONST ('hipaa') on Together's schema — any other value is not surface.
|
|
332
|
+
if (params.compliance !== undefined && params.compliance !== null && params.compliance !== 'hipaa') {
|
|
333
|
+
return { error: invalidRequest("'compliance' must be 'hipaa'") };
|
|
334
|
+
}
|
|
335
|
+
// `response_format`: text | json_object | json_schema. Together's ResponseFormatJsonSchema
|
|
336
|
+
// REQUIRES `json_schema.name` (a-z A-Z 0-9 underscore/dash, ≤64) — an omission Together's
|
|
337
|
+
// Structured Outputs documents as invalid.
|
|
338
|
+
let responseFormat: ResponseFormat = { kind: 'text' };
|
|
339
|
+
const rf = params.response_format as { type?: unknown; json_schema?: { name?: unknown; schema?: unknown } } | undefined;
|
|
340
|
+
if (rf && typeof rf === 'object') {
|
|
341
|
+
if (rf.type === 'json_object') responseFormat = { kind: 'json_object' };
|
|
342
|
+
else if (rf.type === 'json_schema') {
|
|
343
|
+
const name = rf.json_schema?.name;
|
|
344
|
+
if (typeof name !== 'string' || !name) return { error: invalidRequest("'response_format.json_schema.name' is a required property") };
|
|
345
|
+
if (name.length > 64 || !/^[a-zA-Z0-9_-]+$/.test(name)) {
|
|
346
|
+
return { error: invalidRequest("'response_format.json_schema.name' must be a-z, A-Z, 0-9, or contain underscores and dashes, with a maximum length of 64") };
|
|
347
|
+
}
|
|
348
|
+
responseFormat = { kind: 'json_schema', schema: rf.json_schema };
|
|
349
|
+
} else if (rf.type !== undefined && rf.type !== 'text') {
|
|
350
|
+
return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
let toolChoice: ToolChoice | undefined;
|
|
354
|
+
const tcRaw = params.tool_choice;
|
|
355
|
+
if (tcRaw !== undefined && tcRaw !== null) {
|
|
356
|
+
if (typeof tcRaw === 'string') {
|
|
357
|
+
if (!['auto', 'none', 'required'].includes(tcRaw)) return { error: invalidRequest("'tool_choice' must be one of 'none', 'auto', 'required' or a named function") };
|
|
358
|
+
toolChoice = tcRaw as ToolChoice;
|
|
359
|
+
} else if (typeof tcRaw === 'object') {
|
|
360
|
+
const name = (tcRaw as { function?: { name?: unknown } }).function?.name;
|
|
361
|
+
if (typeof name !== 'string' || !name) return { error: invalidRequest("'tool_choice.function.name' is required for a named tool choice") };
|
|
362
|
+
toolChoice = { name };
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
const maxRaw = params.max_tokens;
|
|
366
|
+
let maxTokens: number | undefined;
|
|
367
|
+
if (maxRaw !== undefined && maxRaw !== null) {
|
|
368
|
+
maxTokens = Number(maxRaw);
|
|
369
|
+
if (!Number.isInteger(maxTokens) || maxTokens < 1) return { error: invalidRequest("'max_tokens' must be an integer >= 1") };
|
|
370
|
+
}
|
|
371
|
+
let stop: string[] | undefined;
|
|
372
|
+
if (params.stop !== undefined && params.stop !== null) {
|
|
373
|
+
if (typeof params.stop === 'string') stop = [params.stop];
|
|
374
|
+
else if (Array.isArray(params.stop)) stop = params.stop as string[];
|
|
375
|
+
else return { error: invalidRequest("'stop' must be a string or an array of strings") };
|
|
376
|
+
}
|
|
377
|
+
return {
|
|
378
|
+
args: {
|
|
379
|
+
model: params.model,
|
|
380
|
+
messages,
|
|
381
|
+
...(params.tools !== undefined ? { tools: params.tools } : (params.functions !== undefined ? { tools: params.functions } : {})),
|
|
382
|
+
// The DEPRECATED `functions` request parameter has a deprecated RESPONSE shape too:
|
|
383
|
+
// Together's ChatCompletionMessage carries `function_call` (not `tool_calls`) and
|
|
384
|
+
// `FinishReason` includes 'function_call'.
|
|
385
|
+
legacyFunctions: params.tools === undefined && params.functions !== undefined,
|
|
386
|
+
n: typeof params.n === 'number' ? params.n : 1,
|
|
387
|
+
...(maxTokens !== undefined ? { maxTokens } : {}),
|
|
388
|
+
...(stop !== undefined ? { stop } : {}),
|
|
389
|
+
stream: params.stream === true,
|
|
390
|
+
...(toolChoice !== undefined ? { toolChoice } : {}),
|
|
391
|
+
parallelToolCalls: params.parallel_tool_calls !== false,
|
|
392
|
+
responseFormat,
|
|
393
|
+
...(typeof params.seed === 'number' ? { seed: params.seed } : {}),
|
|
394
|
+
echo: params.echo === true,
|
|
395
|
+
...(typeof params.logprobs === 'number' ? { logprobs: params.logprobs } : {}),
|
|
396
|
+
...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
|
|
397
|
+
contextBehavior,
|
|
398
|
+
},
|
|
399
|
+
};
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
/** Build ONE deterministic stub choice (index `idx`). */
|
|
403
|
+
function buildChoice(args: ChatArgs, idx: number): { choice: TogetheraiChoice; completionTokens: number } {
|
|
404
|
+
const hasTools = Array.isArray(args.tools) && args.tools.length > 0;
|
|
405
|
+
const forbidTools = args.toolChoice === 'none';
|
|
406
|
+
const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
|
|
407
|
+
if (hasTools && !forbidTools) {
|
|
408
|
+
const list = args.tools as unknown[];
|
|
409
|
+
const calls: TogetheraiToolCall[] = [];
|
|
410
|
+
if (forcedName || args.parallelToolCalls === false) {
|
|
411
|
+
const tc = stubToolCall(args.tools, idx + 1, forcedName);
|
|
412
|
+
if (tc) calls.push(tc);
|
|
413
|
+
} else {
|
|
414
|
+
for (let t = 0; t < list.length; t++) {
|
|
415
|
+
const tc = stubToolCall([list[t]], idx * 100 + t + 1);
|
|
416
|
+
if (tc) calls.push(tc);
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
if (calls.length) {
|
|
420
|
+
if (args.legacyFunctions) {
|
|
421
|
+
const fc = calls[0]!.function;
|
|
422
|
+
return {
|
|
423
|
+
choice: { index: idx, message: { role: 'assistant', content: null, function_call: { name: fc.name, arguments: fc.arguments } }, finish_reason: 'function_call' },
|
|
424
|
+
completionTokens: estimateTokens(JSON.stringify(fc)),
|
|
425
|
+
};
|
|
426
|
+
}
|
|
427
|
+
return {
|
|
428
|
+
choice: { index: idx, message: { role: 'assistant', content: null, tool_calls: calls }, finish_reason: 'tool_calls' },
|
|
429
|
+
completionTokens: estimateTokens(JSON.stringify(calls)),
|
|
430
|
+
};
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
let text = args.responseFormat.kind === 'json_object'
|
|
434
|
+
? stubJsonObject(args.messages, args.model)
|
|
435
|
+
: args.responseFormat.kind === 'json_schema'
|
|
436
|
+
? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
|
|
437
|
+
: stubAssistantText(args.messages, args.model);
|
|
438
|
+
let finish: TogetheraiFinishReason = 'stop';
|
|
439
|
+
// Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
|
|
440
|
+
let stopAt = -1;
|
|
441
|
+
for (const s of args.stop ?? []) {
|
|
442
|
+
if (!s) continue;
|
|
443
|
+
const i = text.indexOf(s);
|
|
444
|
+
if (i >= 0 && (stopAt < 0 || i < stopAt)) stopAt = i;
|
|
445
|
+
}
|
|
446
|
+
if (stopAt >= 0) text = text.slice(0, stopAt);
|
|
447
|
+
if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
|
|
448
|
+
text = text.slice(0, args.maxTokens * 4);
|
|
449
|
+
finish = 'length';
|
|
450
|
+
}
|
|
451
|
+
const message: TogetheraiAssistantMessage = { role: 'assistant', content: text };
|
|
452
|
+
// `reasoning_effort` present → Together's reasoning models surface the reasoning on the
|
|
453
|
+
// assistant message's own `reasoning` field (varies by model; the stub always populates it
|
|
454
|
+
// when effort is requested so the field's presence is observable).
|
|
455
|
+
if (args.reasoningEffort !== undefined) message.reasoning = stubReasoningText(args.messages, args.model);
|
|
456
|
+
return {
|
|
457
|
+
choice: { index: idx, message, finish_reason: finish },
|
|
458
|
+
completionTokens: estimateTokens(String(message.content ?? '')) + estimateTokens(message.reasoning ?? ''),
|
|
459
|
+
};
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
/** A deterministic id suffix from the request (so ids are stable + assertable). */
|
|
463
|
+
function stableSuffix(args: ChatArgs): string {
|
|
464
|
+
const s = JSON.stringify(args.messages) + args.model + (args.seed !== undefined ? `|seed=${args.seed}` : '');
|
|
465
|
+
return fnv1a(s).toString(36);
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
export function buildChatCompletion(args: ChatArgs, occurredAt?: string, scenarioEngine?: TogetheraiScenarioEngine): TogetheraiChatCompletion | TogetheraiResponseEnvelope {
|
|
469
|
+
// THE 403: Together's documented context-length refusal. Input tokens + max_tokens beyond the
|
|
470
|
+
// model's context length answers 403 (NOT 400, NOT 413) — unless
|
|
471
|
+
// `context_length_exceeded_behavior:'truncate'` overrides max_tokens down to the window.
|
|
472
|
+
const info = findModel(args.model);
|
|
473
|
+
const promptTokens = countPromptTokens(args.messages);
|
|
474
|
+
if (info?.context_length && info.context_length > 0) {
|
|
475
|
+
const needed = promptTokens + (args.maxTokens ?? 0);
|
|
476
|
+
if (needed > info.context_length) {
|
|
477
|
+
if (args.contextBehavior === 'truncate') {
|
|
478
|
+
args = { ...args, maxTokens: Math.max(1, info.context_length - promptTokens) };
|
|
479
|
+
} else {
|
|
480
|
+
return contextLengthError(args.model, needed, info.context_length);
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
let scripted: ScriptedResult | null = null;
|
|
485
|
+
let missTeach = '';
|
|
486
|
+
if (scenarioEngine) {
|
|
487
|
+
const decision = scenarioEngine.next({ model: args.model, messages: args.messages, tools: args.tools, reasoningEffort: args.reasoningEffort });
|
|
488
|
+
if (decision.kind === 'handler') {
|
|
489
|
+
const respond = decision.respond as TogetheraiScenarioRespond;
|
|
490
|
+
// A scripted FAILURE short-circuits into Together's own error envelope + status.
|
|
491
|
+
if (respond.error) return scriptedError(respond.error);
|
|
492
|
+
scripted = realizeTogetheraiRespond(respond);
|
|
493
|
+
} else {
|
|
494
|
+
missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/togetherai.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
const choices: TogetheraiChoice[] = [];
|
|
498
|
+
let completionTokens = 0;
|
|
499
|
+
for (let i = 0; i < args.n; i++) {
|
|
500
|
+
if (scripted) {
|
|
501
|
+
const message: TogetheraiAssistantMessage = scripted.toolCalls.length
|
|
502
|
+
? { role: 'assistant', content: scripted.text, tool_calls: scripted.toolCalls }
|
|
503
|
+
: { role: 'assistant', content: scripted.text ?? '' };
|
|
504
|
+
if (scripted.reasoning !== null) message.reasoning = scripted.reasoning;
|
|
505
|
+
choices.push({ index: i, message, finish_reason: scripted.finishReason });
|
|
506
|
+
completionTokens += estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
|
|
507
|
+
continue;
|
|
508
|
+
}
|
|
509
|
+
const { choice, completionTokens: ct } = buildChoice(args, i);
|
|
510
|
+
if (missTeach && typeof choice.message?.content === 'string') choice.message.content += missTeach;
|
|
511
|
+
choices.push(choice);
|
|
512
|
+
completionTokens += ct;
|
|
513
|
+
}
|
|
514
|
+
const usage: TogetheraiUsage = buildUsage(promptTokens, completionTokens);
|
|
515
|
+
const id = `chatcmpl-twin-${stableSuffix(args)}`;
|
|
516
|
+
// Together's REQUIRED `prompt` array: the echoed prompt when `echo:true` (the vendor's schema
|
|
517
|
+
// documents it as "When echo is true, the prompt is included in the response"). REQUIRED on
|
|
518
|
+
// the schema, so it is served even when empty.
|
|
519
|
+
const prompt = args.echo ? [{ text: args.messages.map((m) => contentToText(m.content)).join('\n') }] : [];
|
|
520
|
+
return {
|
|
521
|
+
id,
|
|
522
|
+
object: 'chat.completion',
|
|
523
|
+
created: nowEpoch(occurredAt),
|
|
524
|
+
model: args.model,
|
|
525
|
+
choices,
|
|
526
|
+
prompt,
|
|
527
|
+
usage,
|
|
528
|
+
};
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
/** Map a scripted scenario failure onto Together's real status + envelope. */
|
|
532
|
+
function scriptedError(err: NonNullable<TogetheraiScenarioRespond['error']>): TogetheraiResponseEnvelope {
|
|
533
|
+
if (err.type === 'rate_limit_exceeded') {
|
|
534
|
+
const base = rateLimitError();
|
|
535
|
+
return err.message ? { ...base, body: errBody('dynamic_request_limited', err.message) } : base;
|
|
536
|
+
}
|
|
537
|
+
if (err.type === 'engine_overloaded') {
|
|
538
|
+
const base = engineOverloadedError();
|
|
539
|
+
return err.message ? { ...base, body: errBody('engine_overloaded', err.message) } : base;
|
|
540
|
+
}
|
|
541
|
+
return { status: 500, body: errBody('internal_server_error', err.message ?? 'Internal Server Error') };
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
const isEnvelope = (v: TogetheraiChatCompletion | TogetheraiResponseEnvelope): v is TogetheraiResponseEnvelope =>
|
|
545
|
+
typeof (v as TogetheraiResponseEnvelope).status === 'number' && 'body' in v;
|
|
546
|
+
|
|
547
|
+
/** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
|
|
548
|
+
function chunkText(text: string): string[] {
|
|
549
|
+
if (!text) return [];
|
|
550
|
+
const out: string[] = [];
|
|
551
|
+
for (let i = 0; i < text.length; i += 20) out.push(text.slice(i, i + 20));
|
|
552
|
+
return out;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
/**
|
|
556
|
+
* Emit the vendor-faithful Together streaming sequence into the injected sink (NO sockets, NO
|
|
557
|
+
* setTimeout). Together's order: a first chunk with `delta:{role:'assistant'}`, then
|
|
558
|
+
* `delta:{content}` chunks (or tool_calls deltas), then a chunk carrying `finish_reason`. Every
|
|
559
|
+
* chunk carries Together's nullable `usage` + `warnings` (the schema declares both on
|
|
560
|
+
* ChatCompletionChunk itself — OpenAI puts usage only in an opt-in tail chunk). Ends with
|
|
561
|
+
* `data: [DONE]`. Deterministic + synchronous.
|
|
562
|
+
*/
|
|
563
|
+
export function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, scenarioEngine?: TogetheraiScenarioEngine): TogetheraiChatCompletion | TogetheraiResponseEnvelope {
|
|
564
|
+
const built = buildChatCompletion(args, occurredAt, scenarioEngine);
|
|
565
|
+
if (isEnvelope(built)) return built;
|
|
566
|
+
const full = built;
|
|
567
|
+
const base = { id: full.id, object: 'chat.completion.chunk' as const, created: full.created, model: full.model };
|
|
568
|
+
for (const choice of full.choices) {
|
|
569
|
+
const idx = choice.index;
|
|
570
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, finish_reason: null }], usage: null, warnings: [] } });
|
|
571
|
+
if (choice.message?.reasoning) {
|
|
572
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { reasoning: choice.message.reasoning }, finish_reason: null }], usage: null, warnings: [] } });
|
|
573
|
+
}
|
|
574
|
+
if (choice.message?.tool_calls && choice.message.tool_calls.length) {
|
|
575
|
+
choice.message.tool_calls.forEach((tc, tIdx) => {
|
|
576
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, finish_reason: null }], usage: null, warnings: [] } });
|
|
577
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, finish_reason: null }], usage: null, warnings: [] } });
|
|
578
|
+
});
|
|
579
|
+
} else {
|
|
580
|
+
for (const piece of chunkText(choice.message?.content ?? '')) {
|
|
581
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, finish_reason: null }], usage: null, warnings: [] } });
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: {}, finish_reason: choice.finish_reason }], usage: null, warnings: [] } });
|
|
585
|
+
}
|
|
586
|
+
// Together carries usage on the chunks themselves; the FINAL chunk reports the real counts.
|
|
587
|
+
sink({ data: { ...base, choices: [], usage: full.usage, warnings: [] } });
|
|
588
|
+
sink({ done: true });
|
|
589
|
+
return full;
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
// ── Embeddings (deterministic pseudo-vectors) ───────────────────────────────────────────
|
|
593
|
+
function handleEmbeddings(params: Record<string, unknown>): TogetheraiResponseEnvelope {
|
|
594
|
+
if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
|
|
595
|
+
if (params.input === undefined) return invalidRequest("'input' is a required property");
|
|
596
|
+
const model = params.model;
|
|
597
|
+
if (!EMBEDDING_MODELS.has(model)) return notFound(`Model ${model} does not exist or is not an embedding model.`);
|
|
598
|
+
const inputs = Array.isArray(params.input) ? (params.input as unknown[]).map(String) : [String(params.input)];
|
|
599
|
+
if (inputs.some((s) => s.length === 0)) return invalidRequest("'input' must not be an empty string");
|
|
600
|
+
const dims = EMBEDDING_DIMENSIONS[model] ?? 768;
|
|
601
|
+
const data: TogetheraiEmbedding[] = inputs.map((text, index) => {
|
|
602
|
+
const vec = pseudoEmbedding(text, dims);
|
|
603
|
+
return { object: 'embedding', index, embedding: vec };
|
|
604
|
+
});
|
|
605
|
+
// Together's EmbeddingsResponse REQUIRED keys are object/model/data — NO usage key on the
|
|
606
|
+
// schema (Together does not document one on this endpoint).
|
|
607
|
+
const body: TogetheraiEmbeddingResponse = { object: 'list', model, data };
|
|
608
|
+
return { status: 200, body };
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
// ── Rerank (Together-native) ────────────────────────────────────────────────────────────
|
|
612
|
+
/**
|
|
613
|
+
* Deterministic rerank: score each document by a pure function of (query, document text) so the
|
|
614
|
+
* ORDER is stable and assertable, then return the top `top_n` with Together's
|
|
615
|
+
* `{object:'rerank', model, results:[{index, relevance_score, document}]}` shape. NOT a real
|
|
616
|
+
* ranker — the values carry no semantic meaning; only the SHAPE and DETERMINISM are faithful.
|
|
617
|
+
*/
|
|
618
|
+
function handleRerank(params: Record<string, unknown>): TogetheraiResponseEnvelope {
|
|
619
|
+
if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
|
|
620
|
+
if (!RERANK_MODELS.has(params.model)) return notFound(`Model ${params.model} does not exist or is not a rerank model.`);
|
|
621
|
+
if (typeof params.query !== 'string' || !params.query) return invalidRequest("'query' is a required property");
|
|
622
|
+
if (!Array.isArray(params.documents)) return invalidRequest("'documents' is a required property");
|
|
623
|
+
if (params.documents.length === 0) return invalidRequest("'documents' must not be empty");
|
|
624
|
+
// Documents may be strings OR objects (Together's oneOf); an object is ranked by its text-ish
|
|
625
|
+
// fields (`rank_fields` names the keys, defaulting to all of them).
|
|
626
|
+
const rankFields = Array.isArray(params.rank_fields) ? (params.rank_fields as unknown[]).map(String) : null;
|
|
627
|
+
const textOf = (doc: unknown): string => {
|
|
628
|
+
if (typeof doc === 'string') return doc;
|
|
629
|
+
if (doc && typeof doc === 'object') {
|
|
630
|
+
const o = doc as Record<string, unknown>;
|
|
631
|
+
const keys = rankFields ?? Object.keys(o);
|
|
632
|
+
return keys.map((k) => String(o[k] ?? '')).join(' ');
|
|
633
|
+
}
|
|
634
|
+
return String(doc ?? '');
|
|
635
|
+
};
|
|
636
|
+
const query = params.query as string;
|
|
637
|
+
const scored = (params.documents as unknown[]).map((doc, index) => {
|
|
638
|
+
const text = textOf(doc);
|
|
639
|
+
// A deterministic pseudo-score in (0,1): hash(query + text) — same inputs, same order.
|
|
640
|
+
const score = (fnv1a(`${query}\0${text}`) % 10_000) / 10_000;
|
|
641
|
+
return { index, score, doc };
|
|
642
|
+
});
|
|
643
|
+
scored.sort((a, b) => b.score - a.score || a.index - b.index);
|
|
644
|
+
const topN = typeof params.top_n === 'number' && Number.isInteger(params.top_n) && params.top_n > 0 ? params.top_n : scored.length;
|
|
645
|
+
const returnDocuments = params.return_documents === undefined ? true : params.return_documents === true;
|
|
646
|
+
const usageTokens = countPromptTokens([{ role: 'user', content: query }, ...((params.documents as unknown[]).map((d) => ({ role: 'user' as const, content: textOf(d) })))]);
|
|
647
|
+
const body: TogetheraiRerankResponse = {
|
|
648
|
+
object: 'rerank',
|
|
649
|
+
id: `rrank-twin-${fnv1a(`${query}|${JSON.stringify(params.documents)}`).toString(36)}`,
|
|
650
|
+
model: params.model,
|
|
651
|
+
results: scored.slice(0, topN).map((s) => ({
|
|
652
|
+
index: s.index,
|
|
653
|
+
relevance_score: s.score,
|
|
654
|
+
...(returnDocuments ? { document: { text: typeof s.doc === 'string' ? s.doc : JSON.stringify(s.doc) } } : {}),
|
|
655
|
+
})),
|
|
656
|
+
usage: buildUsage(usageTokens, 0),
|
|
657
|
+
};
|
|
658
|
+
return { status: 200, body };
|
|
659
|
+
}
|
|
660
|
+
|
|
661
|
+
// ── Images (deterministic stub) ─────────────────────────────────────────────────────────
|
|
662
|
+
function handleImages(params: Record<string, unknown>): TogetheraiResponseEnvelope {
|
|
663
|
+
if (typeof params.prompt !== 'string' || !params.prompt) return invalidRequest("'prompt' is a required property");
|
|
664
|
+
if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
|
|
665
|
+
if (findModel(params.model)?.type !== 'image') return notFound(`Model ${params.model} does not exist or is not an image model.`);
|
|
666
|
+
// The SDK documents the default as `jpeg` ("Defaults to `jpeg`", images.d.ts:80) — NOT png.
|
|
667
|
+
const format = params.output_format === undefined ? 'jpeg' : params.output_format;
|
|
668
|
+
if (format !== 'jpeg' && format !== 'png') return invalidRequest("'output_format' must be one of 'jpeg', 'png'");
|
|
669
|
+
const responseFormat = params.response_format === undefined ? 'url' : params.response_format;
|
|
670
|
+
// Together's enum is 'base64' | 'url' (together-ai@0.53.0 images.d.ts:95) — NOT OpenAI's
|
|
671
|
+
// 'b64_json'. Serving OpenAI's value here would be exactly the shape contamination the pack
|
|
672
|
+
// refuses elsewhere (the OpenAI-compatibility line: Together's image API is its own).
|
|
673
|
+
if (responseFormat !== 'url' && responseFormat !== 'base64') return invalidRequest("'response_format' must be one of 'url', 'base64'");
|
|
674
|
+
const n = params.n === undefined ? 1 : params.n;
|
|
675
|
+
if (typeof n !== 'number' || !Number.isInteger(n) || n < 1) return invalidRequest("'n' must be a positive integer");
|
|
676
|
+
const steps = params.steps === undefined ? 20 : params.steps;
|
|
677
|
+
if (typeof steps !== 'number' || !Number.isInteger(steps) || steps < 1) return invalidRequest("'steps' must be a positive integer");
|
|
678
|
+
// No diffusion here: each image is a labeled stub — a data: URL carrying the deterministic
|
|
679
|
+
// marker (so `type:'url'` still carries retrievable bytes) or a base64 body.
|
|
680
|
+
const marker = fnv1a(`${params.prompt}|${params.model}|${params.seed ?? ''}`);
|
|
681
|
+
const data = Array.from({ length: n }, (_, index) => {
|
|
682
|
+
// The format is named IN the label so the (documented) default is observable on the wire.
|
|
683
|
+
const stub = `[twin-stub:${params.model}] deterministic ${format} image stub (no diffusion is run) for prompt "${String(params.prompt).slice(0, 80)}" #${index} (${marker.toString(36)})`;
|
|
684
|
+
if (responseFormat === 'base64') return { index, b64_json: Buffer.from(stub).toString('base64'), type: 'b64_json' as const };
|
|
685
|
+
return { index, url: `data:text/plain;base64,${Buffer.from(stub).toString('base64')}`, type: 'url' as const };
|
|
686
|
+
});
|
|
687
|
+
return {
|
|
688
|
+
status: 200,
|
|
689
|
+
body: { id: `img-twin-${marker.toString(36)}`, model: params.model, object: 'list', data },
|
|
690
|
+
};
|
|
691
|
+
}
|
|
692
|
+
|
|
693
|
+
// ── Audio (deterministic labeled stubs stand in for acoustic model output) ──────────────
|
|
694
|
+
const TRANSCRIPTION_FORMATS = new Set(['json', 'text', 'verbose_json']);
|
|
695
|
+
|
|
696
|
+
function handleTranscription(params: Record<string, unknown>, translate: boolean): TogetheraiResponseEnvelope {
|
|
697
|
+
const model = params.model;
|
|
698
|
+
if (typeof model !== 'string' || !model) return invalidRequest("'model' is a required property");
|
|
699
|
+
if (findModel(model)?.type !== 'language' || !SPEECH_MODELS.has(model)) return notFound(`Model ${model} does not exist or is not an audio model.`);
|
|
700
|
+
// "Either a file or a URL must be provided" (together-ai TranscriptionCreateParams).
|
|
701
|
+
const source = typeof params.file === 'string' && params.file ? params.file : (typeof params.url === 'string' && params.url ? params.url : '');
|
|
702
|
+
if (!source) return invalidRequest("Either 'file' or 'url' must be provided");
|
|
703
|
+
const format = params.response_format === undefined ? 'json' : params.response_format;
|
|
704
|
+
if (typeof format !== 'string' || !TRANSCRIPTION_FORMATS.has(format)) {
|
|
705
|
+
return invalidRequest(`'response_format' must be one of ${[...TRANSCRIPTION_FORMATS].map((f) => `'${f}'`).join(', ')}`);
|
|
706
|
+
}
|
|
707
|
+
const text = stubTranscript(model, source, translate);
|
|
708
|
+
// `response_format:'text'` returns the bare transcript as plain text, not a JSON envelope.
|
|
709
|
+
if (format === 'text') return { status: 200, body: text, headers: { 'content-type': 'text/plain; charset=utf-8' } };
|
|
710
|
+
if (format === 'json') return { status: 200, body: { text } };
|
|
711
|
+
const duration = stubAudioSeconds(source);
|
|
712
|
+
return {
|
|
713
|
+
status: 200,
|
|
714
|
+
body: {
|
|
715
|
+
task: translate ? 'translate' : 'transcribe',
|
|
716
|
+
language: 'english',
|
|
717
|
+
duration,
|
|
718
|
+
text,
|
|
719
|
+
segments: [{ id: 0, seek: 0, start: 0, end: duration, text, tokens: [], temperature: Number(params.temperature ?? 0), avg_logprob: -0.25, compression_ratio: 1.2, no_speech_prob: 0.01 }],
|
|
720
|
+
},
|
|
721
|
+
};
|
|
722
|
+
}
|
|
723
|
+
|
|
724
|
+
function handleSpeech(params: Record<string, unknown>): TogetheraiResponseEnvelope {
|
|
725
|
+
const model = params.model;
|
|
726
|
+
if (typeof model !== 'string' || !model) return invalidRequest("'model' is a required property");
|
|
727
|
+
if (!SPEECH_MODELS.has(model)) return notFound(`Model ${model} does not exist or is not a speech model.`);
|
|
728
|
+
if (typeof params.input !== 'string' || !params.input) return invalidRequest("'input' is a required property");
|
|
729
|
+
if (typeof params.voice !== 'string' || !params.voice) return invalidRequest("'voice' is a required property");
|
|
730
|
+
const format = params.response_format === undefined ? 'mp3' : params.response_format;
|
|
731
|
+
if (typeof format !== 'string' || !['mp3', 'wav', 'opus', 'flac'].includes(format)) {
|
|
732
|
+
return invalidRequest("'response_format' must be one of 'mp3', 'wav', 'opus', 'flac'");
|
|
733
|
+
}
|
|
734
|
+
// No vocoder here: the body is a clearly-labeled deterministic stub, and the server serves it
|
|
735
|
+
// with the audio content-type the vendor would use.
|
|
736
|
+
return {
|
|
737
|
+
status: 200,
|
|
738
|
+
body: `[twin-stub:${model}] no real speech synthesis — voice=${params.voice} format=${format}; text: ${params.input}`,
|
|
739
|
+
headers: { 'content-type': `audio/${format}` },
|
|
740
|
+
};
|
|
741
|
+
}
|
|
742
|
+
|
|
743
|
+
// ── Files (stateful; BOTH upload flows) ─────────────────────────────────────────────────
|
|
744
|
+
/** Together's `FilePurpose` — a CLOSED documented set (NOT OpenAI's). */
|
|
745
|
+
const FILE_PURPOSES = new Set<TogetheraiFilePurpose>(['fine-tune', 'eval', 'batch-api']);
|
|
746
|
+
/** Together's `FileType` — a CLOSED documented set. */
|
|
747
|
+
const FILE_TYPES = new Set<TogetheraiFileType>(['csv', 'jsonl', 'parquet']);
|
|
748
|
+
|
|
749
|
+
/** Build the vendor-faithful FileResponse view of a stored row. */
|
|
750
|
+
function fileView(r: Record<string, unknown>): Record<string, unknown> {
|
|
751
|
+
const s = strip(r);
|
|
752
|
+
return {
|
|
753
|
+
id: r.id,
|
|
754
|
+
object: 'file',
|
|
755
|
+
created_at: s.created_at,
|
|
756
|
+
filename: s.filename,
|
|
757
|
+
bytes: s.bytes,
|
|
758
|
+
purpose: s.purpose,
|
|
759
|
+
Processed: s.Processed ?? true,
|
|
760
|
+
FileType: s.FileType,
|
|
761
|
+
...(s.processing_status !== undefined ? { processing_status: s.processing_status } : {}),
|
|
762
|
+
...(s.validation_report !== undefined ? { validation_report: s.validation_report } : {}),
|
|
763
|
+
};
|
|
764
|
+
}
|
|
765
|
+
|
|
766
|
+
/** Validate + derive the stored fields for a file create (shared by both upload flows).
|
|
767
|
+
*
|
|
768
|
+
* The REQUIRED parts are the vendor's, not ours: the SDK types `purpose: FilePurpose` as
|
|
769
|
+
* REQUIRED (files.d.ts:48, `upload(file: string, purpose: FilePurpose, …)`) and the multipart
|
|
770
|
+
* form's `file` part names the upload (`filename`); the 302 flow's query carries `file_name` +
|
|
771
|
+
* `purpose` (lib/upload.js:40). A request missing one answers the vendor's 400 and stores
|
|
772
|
+
* NOTHING — a defaulted `fine-tune`/`upload.jsonl` minted a file from a bodyless POST (the §9
|
|
773
|
+
* round-three blocker; the class fix, not the one instance). */
|
|
774
|
+
function fileFieldsFrom(params: Record<string, unknown>): { fields: Record<string, unknown> } | { error: TogetheraiResponseEnvelope } {
|
|
775
|
+
if (params.purpose === undefined || params.purpose === null || params.purpose === '') {
|
|
776
|
+
return { error: invalidRequest("'purpose' is a required property") };
|
|
777
|
+
}
|
|
778
|
+
const purpose = params.purpose;
|
|
779
|
+
if (typeof purpose !== 'string' || !FILE_PURPOSES.has(purpose as TogetheraiFilePurpose)) {
|
|
780
|
+
return { error: invalidRequest(`'purpose' must be one of ${[...FILE_PURPOSES].map((p) => `'${p}'`).join(', ')} (got '${String(purpose)}')`) };
|
|
781
|
+
}
|
|
782
|
+
const rawName = params.filename ?? params.file_name;
|
|
783
|
+
if (rawName === undefined || rawName === null || String(rawName).trim() === '') {
|
|
784
|
+
return { error: invalidRequest("'filename' is a required property") };
|
|
785
|
+
}
|
|
786
|
+
const filename = String(rawName);
|
|
787
|
+
// An EXPLICIT file_type wins; otherwise it is derived from the filename's extension (§9 round
|
|
788
|
+
// two, R2-D3: the old OR discarded an explicit file_type whenever FileType was absent, and no
|
|
789
|
+
// value was ever validated against the closed set).
|
|
790
|
+
const fileType = params.file_type !== undefined ? String(params.file_type)
|
|
791
|
+
: params.FileType !== undefined ? String(params.FileType)
|
|
792
|
+
: (filename.includes('.') ? String(filename.split('.').pop()).toLowerCase() : 'jsonl');
|
|
793
|
+
if (!FILE_TYPES.has(fileType as TogetheraiFileType)) {
|
|
794
|
+
return { error: invalidRequest(`'file_type' must be one of ${[...FILE_TYPES].map((t) => `'${t}'`).join(', ')} (got '${fileType}')`) };
|
|
795
|
+
}
|
|
796
|
+
const content = typeof params.content === 'string' ? params.content : '';
|
|
797
|
+
const bytes = typeof params.bytes === 'number' ? params.bytes : content.length;
|
|
798
|
+
// Files for non-`fine-tune` purposes SKIP validation (Together's own schema wording); a
|
|
799
|
+
// fine-tune file enters the pipeline as PENDING.
|
|
800
|
+
const processing_status = purpose === 'fine-tune' ? 'PENDING' : undefined;
|
|
801
|
+
return {
|
|
802
|
+
fields: {
|
|
803
|
+
object: 'file', bytes, created_at: 0, filename, purpose, Processed: true, FileType: fileType,
|
|
804
|
+
...(processing_status ? { processing_status } : {}),
|
|
805
|
+
_content: content,
|
|
806
|
+
},
|
|
807
|
+
};
|
|
808
|
+
}
|
|
809
|
+
|
|
810
|
+
async function createFileMultipart(params: Record<string, unknown>, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
|
|
811
|
+
const built = fileFieldsFrom(params);
|
|
812
|
+
if ('error' in built) return built.error;
|
|
813
|
+
const id = nextId('file', 'file', req.root);
|
|
814
|
+
await applyTwinWrite(SERVICE, {
|
|
815
|
+
operation: 'file.create',
|
|
816
|
+
subjectType: 'file',
|
|
817
|
+
subjectId: id,
|
|
818
|
+
fields: { ...built.fields, created_at: nowEpoch(req.occurredAt) },
|
|
819
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
820
|
+
actor: { kind: 'agent' },
|
|
821
|
+
}, req.root);
|
|
822
|
+
return { status: 200, body: fileView(getRow('file', id, req.root) ?? {}) };
|
|
823
|
+
}
|
|
824
|
+
|
|
825
|
+
/**
|
|
826
|
+
* The together-ai SDK's OWN upload flow (lib/upload.js, together-ai@0.53.0) — NOT the spec's
|
|
827
|
+
* multipart form: it POSTs `/files?file_name=…&file_type=…&purpose=…` as
|
|
828
|
+
* `application/x-www-form-urlencoded`, requires a **302** with a `Location` upload URL and an
|
|
829
|
+
* `x-together-file-id` header, PUTs the bytes there, then confirms with `files.retrieve`. The
|
|
830
|
+
* twin answers the 302 with a twin-local upload URL and mints the file row immediately; the PUT
|
|
831
|
+
* below fills its content.
|
|
832
|
+
*/
|
|
833
|
+
async function createFileSdkRedirect(params: Record<string, unknown>, req: TogetheraiRequest, query: URLSearchParams): Promise<TogetheraiResponseEnvelope> {
|
|
834
|
+
const built = fileFieldsFrom(params);
|
|
835
|
+
if ('error' in built) return built.error;
|
|
836
|
+
const id = nextId('file', 'file', req.root);
|
|
837
|
+
await applyTwinWrite(SERVICE, {
|
|
838
|
+
operation: 'file.create',
|
|
839
|
+
subjectType: 'file',
|
|
840
|
+
subjectId: id,
|
|
841
|
+
fields: { ...built.fields, created_at: nowEpoch(req.occurredAt) },
|
|
842
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
843
|
+
actor: { kind: 'agent' },
|
|
844
|
+
}, req.root);
|
|
845
|
+
return {
|
|
846
|
+
status: 302,
|
|
847
|
+
body: '',
|
|
848
|
+
headers: {
|
|
849
|
+
// Twin-LOCAL on purpose: the handler knows only paths. The HTTP server absolutizes it
|
|
850
|
+
// against the twin's public base — the real SDK fetches `Location` with no base URL.
|
|
851
|
+
location: `/twin/upload/${id}`,
|
|
852
|
+
'x-together-file-id': id,
|
|
853
|
+
},
|
|
854
|
+
};
|
|
855
|
+
}
|
|
856
|
+
|
|
857
|
+
async function storeUploadBytes(fileId: string, content: string, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
|
|
858
|
+
const f = getRow('file', fileId, req.root);
|
|
859
|
+
if (!f || f._deleted) return notFound(`No such File object: ${fileId}`);
|
|
860
|
+
await applyTwinWrite(SERVICE, {
|
|
861
|
+
operation: 'file.update',
|
|
862
|
+
subjectType: 'file', subjectId: fileId,
|
|
863
|
+
fields: { _content: content, bytes: content.length },
|
|
864
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
865
|
+
}, req.root);
|
|
866
|
+
return { status: 200, body: { ok: true } };
|
|
867
|
+
}
|
|
868
|
+
|
|
869
|
+
// ── Batches (Together-native shapes) ────────────────────────────────────────────────────
|
|
870
|
+
/** Together's `CreateBatchRequest.endpoint` — the THREE documented endpoints. */
|
|
871
|
+
const BATCH_ENDPOINTS = new Set<TogetheraiBatchEndpoint>(['/v1/chat/completions', '/v1/audio/transcriptions', '/v1/audio/translations']);
|
|
872
|
+
|
|
873
|
+
/** The batch view of a stored row. `error` stays a bare STRING (Together's BatchJob.error). */
|
|
874
|
+
function batchView(r: Record<string, unknown>): Record<string, unknown> {
|
|
875
|
+
const s = strip(r);
|
|
876
|
+
return {
|
|
877
|
+
id: r.id,
|
|
878
|
+
user_id: s.user_id,
|
|
879
|
+
input_file_id: s.input_file_id,
|
|
880
|
+
file_size_bytes: s.file_size_bytes,
|
|
881
|
+
status: s.status,
|
|
882
|
+
job_deadline: s.job_deadline ?? null,
|
|
883
|
+
created_at: s.created_at,
|
|
884
|
+
endpoint: s.endpoint,
|
|
885
|
+
progress: s.progress ?? 0,
|
|
886
|
+
...(s.model_id !== undefined ? { model_id: s.model_id } : {}),
|
|
887
|
+
output_file_id: s.output_file_id ?? null,
|
|
888
|
+
error_file_id: s.error_file_id ?? null,
|
|
889
|
+
error: s.error ?? null,
|
|
890
|
+
completed_at: s.completed_at ?? null,
|
|
891
|
+
};
|
|
892
|
+
}
|
|
893
|
+
|
|
894
|
+
async function createBatch(params: Record<string, unknown>, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
|
|
895
|
+
const inputFileId = params.input_file_id;
|
|
896
|
+
if (typeof inputFileId !== 'string' || !inputFileId) return invalidRequest("'input_file_id' is a required property");
|
|
897
|
+
const endpoint = params.endpoint;
|
|
898
|
+
if (typeof endpoint !== 'string' || !BATCH_ENDPOINTS.has(endpoint as TogetheraiBatchEndpoint)) {
|
|
899
|
+
return invalidRequest(`'endpoint' must be one of ${[...BATCH_ENDPOINTS].map((e) => `'${e}'`).join(', ')}`);
|
|
900
|
+
}
|
|
901
|
+
const file = getRow('file', inputFileId, req.root);
|
|
902
|
+
if (!file || file._deleted) return notFound(`No such File object: ${inputFileId}`);
|
|
903
|
+
const created = nowEpoch(req.occurredAt);
|
|
904
|
+
const id = nextId('batch', 'batch', req.root);
|
|
905
|
+
await applyTwinWrite(SERVICE, {
|
|
906
|
+
operation: 'batch.create',
|
|
907
|
+
subjectType: 'batch',
|
|
908
|
+
subjectId: id,
|
|
909
|
+
fields: {
|
|
910
|
+
object: 'batch',
|
|
911
|
+
endpoint,
|
|
912
|
+
input_file_id: inputFileId,
|
|
913
|
+
status: 'VALIDATING' as TogetheraiBatchStatus,
|
|
914
|
+
created_at: new Date(created * 1000).toISOString(),
|
|
915
|
+
file_size_bytes: typeof file.bytes === 'number' ? file.bytes : 0,
|
|
916
|
+
progress: 0,
|
|
917
|
+
...(params.completion_window !== undefined ? { completion_window: params.completion_window } : {}),
|
|
918
|
+
...(params.priority !== undefined ? { priority: params.priority } : {}),
|
|
919
|
+
...(params.model_id !== undefined ? { model_id: params.model_id } : {}),
|
|
920
|
+
},
|
|
921
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
922
|
+
actor: { kind: 'agent' },
|
|
923
|
+
}, req.root);
|
|
924
|
+
// Together's create answers **201** with `BatchJobWithWarning { job, warning? }` — NOT an
|
|
925
|
+
// OpenAI-style 200 batch object.
|
|
926
|
+
return { status: 201, body: { job: batchView(getRow('batch', id, req.root) ?? {}) } };
|
|
927
|
+
}
|
|
928
|
+
|
|
929
|
+
// ── Fine-tunes (Together-native shapes) ─────────────────────────────────────────────────
|
|
930
|
+
/** Together's `FinetuneJobStatus` enum. */
|
|
931
|
+
const FINETUNE_STATUSES = ['pending', 'queued', 'running', 'compressing', 'uploading', 'cancel_requested', 'cancelled', 'error', 'completed'] as const;
|
|
932
|
+
|
|
933
|
+
function finetuneView(r: Record<string, unknown>): Record<string, unknown> {
|
|
934
|
+
const s = strip(r);
|
|
935
|
+
return {
|
|
936
|
+
id: r.id,
|
|
937
|
+
status: s.status,
|
|
938
|
+
user_id: s.user_id,
|
|
939
|
+
...(s.training_file !== undefined ? { training_file: s.training_file } : {}),
|
|
940
|
+
...(s.validation_file !== undefined ? { validation_file: s.validation_file } : {}),
|
|
941
|
+
...(s.model !== undefined ? { model: s.model } : {}),
|
|
942
|
+
...(s.model_output_name !== undefined ? { model_output_name: s.model_output_name } : {}),
|
|
943
|
+
...(s.created_at !== undefined ? { created_at: s.created_at } : {}),
|
|
944
|
+
...(s.updated_at !== undefined ? { updated_at: s.updated_at } : {}),
|
|
945
|
+
...(s.n_epochs !== undefined ? { n_epochs: s.n_epochs } : {}),
|
|
946
|
+
};
|
|
947
|
+
}
|
|
948
|
+
|
|
949
|
+
async function createFinetune(params: Record<string, unknown>, req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
|
|
950
|
+
if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
|
|
951
|
+
if (typeof params.training_file !== 'string' || !params.training_file) return invalidRequest("'training_file' is a required property");
|
|
952
|
+
const base = servedModels(req.root).find((row) => row.id === params.model);
|
|
953
|
+
if (!base || base.type !== 'chat') return notFound(`Model ${params.model} does not exist or is not a fine-tunable model.`);
|
|
954
|
+
const training = getRow('file', params.training_file, req.root);
|
|
955
|
+
if (!training || training._deleted) return notFound(`No such File object: ${params.training_file}`);
|
|
956
|
+
if (training.purpose !== 'fine-tune') return invalidRequest(`File ${params.training_file} must have purpose 'fine-tune' (has '${String(training.purpose)}')`);
|
|
957
|
+
if (params.validation_file !== undefined && params.validation_file !== null) {
|
|
958
|
+
const validation = getRow('file', String(params.validation_file), req.root);
|
|
959
|
+
if (!validation || validation._deleted) return notFound(`No such File object: ${String(params.validation_file)}`);
|
|
960
|
+
if (validation.purpose !== 'fine-tune') return invalidRequest(`File ${params.validation_file} must have purpose 'fine-tune' (has '${String(validation.purpose)}')`);
|
|
961
|
+
}
|
|
962
|
+
const id = nextId('finetune', 'ft', req.root);
|
|
963
|
+
const at = new Date(nowEpoch(req.occurredAt) * 1000).toISOString();
|
|
964
|
+
await applyTwinWrite(SERVICE, {
|
|
965
|
+
operation: 'finetune.create',
|
|
966
|
+
subjectType: 'finetune',
|
|
967
|
+
subjectId: id,
|
|
968
|
+
fields: {
|
|
969
|
+
object: 'finetune',
|
|
970
|
+
status: 'pending',
|
|
971
|
+
user_id: 'user_twin',
|
|
972
|
+
model: params.model,
|
|
973
|
+
training_file: params.training_file,
|
|
974
|
+
...(params.validation_file !== undefined && params.validation_file !== null ? { validation_file: params.validation_file } : {}),
|
|
975
|
+
created_at: at,
|
|
976
|
+
updated_at: at,
|
|
977
|
+
...(typeof params.n_epochs === 'number' ? { n_epochs: params.n_epochs } : {}),
|
|
978
|
+
},
|
|
979
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
980
|
+
actor: { kind: 'agent' },
|
|
981
|
+
}, req.root);
|
|
982
|
+
return { status: 200, body: finetuneView(getRow('finetune', id, req.root) ?? {}) };
|
|
983
|
+
}
|
|
984
|
+
|
|
985
|
+
// ── Fine-tune aux endpoints (modeled against Together's documented shapes) ──────────────
|
|
986
|
+
/** The content of a fine-tune-purpose file row, JSONL-decoded into per-row objects. */
|
|
987
|
+
function jsonlRows(content: unknown): Array<Record<string, unknown>> {
|
|
988
|
+
if (typeof content !== 'string' || !content.trim()) return [];
|
|
989
|
+
return content.split('\n').filter((l) => l.trim()).map((l) => {
|
|
990
|
+
try {
|
|
991
|
+
const v = JSON.parse(l);
|
|
992
|
+
return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
|
|
993
|
+
} catch {
|
|
994
|
+
return {};
|
|
995
|
+
}
|
|
996
|
+
});
|
|
997
|
+
}
|
|
998
|
+
|
|
999
|
+
/** GET /v1/fine-tunes/models/limits — Together's `FinetuneModelLimits` for ONE model
|
|
1000
|
+
* (together-ai@0.53.0 fine-tuning.d.ts:270, `model_name` REQUIRED — fine-tuning.d.ts:1798).
|
|
1001
|
+
* A model the catalog does not know fails like the vendor. The numbers are the twin's
|
|
1002
|
+
* deterministic stand-ins for the vendor's per-model training limits; the SHAPE is the vendor's
|
|
1003
|
+
* REQUIRED+OPTIONAL contract. */
|
|
1004
|
+
function finetuneModelLimits(req: TogetheraiRequest): TogetheraiResponseEnvelope {
|
|
1005
|
+
const search = req.path.includes('?') ? new URLSearchParams(req.path.slice(req.path.indexOf('?') + 1)) : new URLSearchParams();
|
|
1006
|
+
const modelName = search.get('model_name');
|
|
1007
|
+
if (!modelName) return invalidRequest("'model_name' is a required property");
|
|
1008
|
+
const m = servedModels(req.root).find((pm) => pm.id === modelName);
|
|
1009
|
+
if (!m || m.type !== 'chat') return notFound(`Model ${modelName} does not exist or is not a fine-tunable model.`);
|
|
1010
|
+
const seq = m.context_length || 8_192;
|
|
1011
|
+
const seed = fnv1a(modelName);
|
|
1012
|
+
return {
|
|
1013
|
+
status: 200,
|
|
1014
|
+
body: {
|
|
1015
|
+
model_name: m.id,
|
|
1016
|
+
default_gradient_accumulation_steps: 16,
|
|
1017
|
+
lora_training: {
|
|
1018
|
+
max_batch_size: 8,
|
|
1019
|
+
max_batch_size_dpo: 4,
|
|
1020
|
+
max_rank: 64,
|
|
1021
|
+
min_batch_size: 1,
|
|
1022
|
+
target_modules: ['q_proj', 'k_proj', 'v_proj', 'o_proj', 'gate_proj', 'up_proj', 'down_proj'],
|
|
1023
|
+
},
|
|
1024
|
+
max_learning_rate: 0.0002,
|
|
1025
|
+
max_num_checkpoints: 10,
|
|
1026
|
+
max_num_epochs: 36,
|
|
1027
|
+
max_num_evals: 36,
|
|
1028
|
+
max_seq_length_dpo: seq,
|
|
1029
|
+
max_seq_length_sft: seq,
|
|
1030
|
+
merge_output_lora: true,
|
|
1031
|
+
min_learning_rate: 0.000001,
|
|
1032
|
+
min_max_seq_length: 128,
|
|
1033
|
+
supports_full_training: false,
|
|
1034
|
+
supports_reasoning: false,
|
|
1035
|
+
supports_tools: true,
|
|
1036
|
+
supports_vision: false,
|
|
1037
|
+
// deterministic per-model jitter so two models never answer identical limits
|
|
1038
|
+
...(seed % 2 === 0 ? { full_training: { max_batch_size: 4, max_batch_size_dpo: 2, min_batch_size: 1 } } : {}),
|
|
1039
|
+
},
|
|
1040
|
+
};
|
|
1041
|
+
}
|
|
1042
|
+
|
|
1043
|
+
/** POST /v1/fine-tunes/estimate-price — Together's DISCRIMINATED union
|
|
1044
|
+
* (together-ai@0.53.0 fine-tuning.d.ts:1323 — AvailableEstimate | UnavailableEstimate on
|
|
1045
|
+
* `estimation_available`). A training_file the twin has seen and validated resolves to an
|
|
1046
|
+
* AvailableEstimate whose token counts are DETERMINISTIC functions of the file's stored
|
|
1047
|
+
* content; an unknown id is `false` with the vendor's `train_file_invalid` reason. */
|
|
1048
|
+
function estimateFinetunePrice(params: Record<string, unknown>, req: TogetheraiRequest): TogetheraiResponseEnvelope {
|
|
1049
|
+
// together-ai@0.53.0 fine-tuning.d.ts:1688-1691: training_file is REQUIRED, model optional. A
|
|
1050
|
+
// request without the file has nothing to estimate over; the vendor refuses it, never invents 64 tokens.
|
|
1051
|
+
if (params.training_file === undefined || typeof params.training_file !== 'string' || !params.training_file) return invalidRequest("'training_file' is a required property");
|
|
1052
|
+
const file = params.training_file !== undefined ? getRow('file', String(params.training_file), req.root) : undefined;
|
|
1053
|
+
if (params.training_file !== undefined && (!file || file._deleted)) {
|
|
1054
|
+
return { status: 200, body: { estimation_available: false, unavailable_reason: 'train_file_invalid' } };
|
|
1055
|
+
}
|
|
1056
|
+
const dataset = jsonlRows(file?._content);
|
|
1057
|
+
const totalTokens = dataset.reduce((n, row) => n + estimateTokens(JSON.stringify(row)), 0) || 64;
|
|
1058
|
+
const epochs = typeof params.n_epochs === 'number' && params.n_epochs > 0 ? params.n_epochs : 1;
|
|
1059
|
+
return {
|
|
1060
|
+
status: 200,
|
|
1061
|
+
body: {
|
|
1062
|
+
estimation_available: true,
|
|
1063
|
+
allowed_to_proceed: true,
|
|
1064
|
+
estimated_total_price: Math.round(totalTokens * epochs * 0.000002 * 1_000_000) / 1_000_000,
|
|
1065
|
+
estimated_train_token_count: totalTokens * epochs,
|
|
1066
|
+
estimated_eval_token_count: 0,
|
|
1067
|
+
user_limit: 100,
|
|
1068
|
+
},
|
|
1069
|
+
};
|
|
1070
|
+
}
|
|
1071
|
+
|
|
1072
|
+
/** POST /v1/fine-tunes/preview — Together's `FineTunePreviewResponse` (fine-tuning.d.ts:160):
|
|
1073
|
+
* tokenized preview rows over the SAMPLED training file, not a prose string. The token ids are
|
|
1074
|
+
* the twin's deterministic stand-ins; the shape and the per-row contract (input_ids/labels/
|
|
1075
|
+
* num_tokens/num_trained_tokens/tokens/trained_spans/truncated) are the vendor's. An unknown
|
|
1076
|
+
* training_file fails like the vendor. */
|
|
1077
|
+
function previewFinetuneTokenization(params: Record<string, unknown>, req: TogetheraiRequest): TogetheraiResponseEnvelope {
|
|
1078
|
+
if (typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
|
|
1079
|
+
if (typeof params.training_file !== 'string' || !params.training_file) return invalidRequest("'training_file' is a required property");
|
|
1080
|
+
const m = servedModels(req.root).find((pm) => pm.id === params.model);
|
|
1081
|
+
if (!m || m.type !== 'chat') return notFound(`Model ${params.model} does not exist or is not a fine-tunable model.`);
|
|
1082
|
+
const file = getRow('file', params.training_file, req.root);
|
|
1083
|
+
if (!file || file._deleted) return notFound(`No such File object: ${params.training_file}`);
|
|
1084
|
+
const maxSeq = m.context_length || 8_192;
|
|
1085
|
+
const topK = typeof params.top_k === 'number' && Number.isInteger(params.top_k) && params.top_k > 0 ? Math.min(params.top_k, 100) : 5;
|
|
1086
|
+
const trainOnInputs = params.train_on_inputs === undefined ? true : params.train_on_inputs === true;
|
|
1087
|
+
const sampled = jsonlRows(file._content).slice(0, topK);
|
|
1088
|
+
// The SDK documents dataset_format as DETECTED per sampled rows ("Detected SFT dataset format
|
|
1089
|
+
// for the sampled rows", fine-tuning.d.ts:158) — the check the vendor's own uploader runs
|
|
1090
|
+
// (lib/check-file.mjs: JSONL_REQUIRED_COLUMNS_MAP: general=['text'], conversation=['messages'],
|
|
1091
|
+
// instruction=['prompt','completion']). A file whose rows carry no recognizable column set
|
|
1092
|
+
// answers 'general', the format a bare text row is.
|
|
1093
|
+
const datasetFormat = (() => {
|
|
1094
|
+
const first = sampled[0];
|
|
1095
|
+
if (!first) return 'general';
|
|
1096
|
+
if ('messages' in first) return 'conversation';
|
|
1097
|
+
if ('prompt' in first && 'completion' in first) return 'instruction';
|
|
1098
|
+
return 'general';
|
|
1099
|
+
})();
|
|
1100
|
+
const rows = (sampled.length ? sampled : [{}]).map((row) => {
|
|
1101
|
+
const text = JSON.stringify(row);
|
|
1102
|
+
// Deterministic pseudo-token ids: one per ~4 chars (fnv1a-seeded), same input → same ids.
|
|
1103
|
+
const words = text.length > 0 ? Math.max(1, Math.ceil(text.length / 4)) : 1;
|
|
1104
|
+
const inputIds = Array.from({ length: words }, (_, i) => fnv1a(`${text}:${i}`) % 50_000);
|
|
1105
|
+
const labels = trainOnInputs ? [...inputIds] : inputIds.map(() => -100);
|
|
1106
|
+
const numTrained = labels.filter((l) => l !== -100).length;
|
|
1107
|
+
const truncated = words > maxSeq;
|
|
1108
|
+
return {
|
|
1109
|
+
input_ids: inputIds.slice(0, maxSeq),
|
|
1110
|
+
labels: labels.slice(0, maxSeq),
|
|
1111
|
+
num_tokens: Math.min(words, maxSeq),
|
|
1112
|
+
num_trained_tokens: Math.min(numTrained, maxSeq),
|
|
1113
|
+
tokens: inputIds.map((id) => `tok_${id.toString(36)}`).slice(0, maxSeq),
|
|
1114
|
+
trained_spans: trainOnInputs && numTrained > 0 ? [[0, Math.min(numTrained, maxSeq)]] : [],
|
|
1115
|
+
truncated,
|
|
1116
|
+
};
|
|
1117
|
+
});
|
|
1118
|
+
return {
|
|
1119
|
+
status: 200,
|
|
1120
|
+
body: {
|
|
1121
|
+
dataset_format: datasetFormat,
|
|
1122
|
+
max_seq_length: maxSeq,
|
|
1123
|
+
model: m.id,
|
|
1124
|
+
rows,
|
|
1125
|
+
train_on_inputs: trainOnInputs,
|
|
1126
|
+
},
|
|
1127
|
+
};
|
|
1128
|
+
}
|
|
1129
|
+
|
|
1130
|
+
/** GET /v1/fine-tunes/{id}/events — Together's `FinetuneEvent` list (fine-tuning.d.ts:234).
|
|
1131
|
+
* A job still `pending` has NO events yet (the vendor's job has not started); a job the twin
|
|
1132
|
+
* has seen move (cancel_requested or beyond) carries the events its history grounds — the
|
|
1133
|
+
* lifecycle events the vendor's own enum names, derived from the STORED row, never invented
|
|
1134
|
+
* progress. */
|
|
1135
|
+
function finetuneEvents(ft: Record<string, unknown>): Array<Record<string, unknown>> {
|
|
1136
|
+
const status = String(ft.status ?? 'pending');
|
|
1137
|
+
const at = (ft.created_at as string) ?? new Date(0).toISOString();
|
|
1138
|
+
const events: Array<Record<string, unknown>> = [];
|
|
1139
|
+
if (status !== 'pending') {
|
|
1140
|
+
events.push({ object: 'fine-tune-event', created_at: at, message: 'Job started', type: 'job_start' });
|
|
1141
|
+
}
|
|
1142
|
+
if (status === 'cancel_requested') {
|
|
1143
|
+
events.push({ object: 'fine-tune-event', created_at: at, message: 'Cancel requested', type: 'cancel_requested' });
|
|
1144
|
+
}
|
|
1145
|
+
return events;
|
|
1146
|
+
}
|
|
1147
|
+
|
|
1148
|
+
/** GET /v1/fine-tunes/{id}/checkpoints — Together's `FineTuningListCheckpointsResponse`
|
|
1149
|
+
* (fine-tuning.d.ts:1362). Checkpoints are artifacts a COMPLETED training run produced; the
|
|
1150
|
+
* twin runs no training and its jobs never progress past the states above, so an honest answer
|
|
1151
|
+
* over stored state is an EMPTY list — the vendor's shape, not a fabricated artifact. */
|
|
1152
|
+
function finetuneCheckpoints(ft: Record<string, unknown>): Array<Record<string, unknown>> {
|
|
1153
|
+
return String(ft.status) === 'completed'
|
|
1154
|
+
? [{ checkpoint_type: 'final', created_at: String(ft.updated_at ?? ft.created_at ?? new Date(0).toISOString()), path: `twin://finetune/${String(ft.id)}/final`, step: 1 }]
|
|
1155
|
+
: [];
|
|
1156
|
+
}
|
|
1157
|
+
|
|
1158
|
+
// ── public entry: cross-cutting protocol (auth / fault triggers) then route ─────────────
|
|
1159
|
+
export async function handleTogetheraiTwinRequest(req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
|
|
1160
|
+
const method = req.method.toUpperCase();
|
|
1161
|
+
if (req.headers !== undefined || req.apiKey !== undefined) {
|
|
1162
|
+
const authErr = checkAuth(req);
|
|
1163
|
+
if (authErr) return authErr;
|
|
1164
|
+
}
|
|
1165
|
+
if (triggered(req, 'x-twin-force-rate-limit')) return rateLimitError();
|
|
1166
|
+
if (triggered(req, 'x-twin-force-spending-limit')) return spendingLimitError();
|
|
1167
|
+
if (triggered(req, 'x-twin-force-engine-overloaded')) return engineOverloadedError();
|
|
1168
|
+
return routeTogetherai(req, method);
|
|
1169
|
+
}
|
|
1170
|
+
|
|
1171
|
+
// ── router ──────────────────────────────────────────────────────────────────────────────
|
|
1172
|
+
async function routeTogetherai(req: TogetheraiRequest, method: string): Promise<TogetheraiResponseEnvelope> {
|
|
1173
|
+
const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
|
|
1174
|
+
const search = req.path.includes('?') ? req.path.slice(req.path.indexOf('?') + 1) : '';
|
|
1175
|
+
const params = parseJson(req.body);
|
|
1176
|
+
// A percent-decode that never throws: `GET /v1/files/%zz` is a malformed escape, and the vendor
|
|
1177
|
+
// answers it like any other unknown object (404), not an internal error. decodeURIComponent
|
|
1178
|
+
// throws URIError on it — the same trap the upload door's decSafe guards.
|
|
1179
|
+
const dec = decSafe;
|
|
1180
|
+
|
|
1181
|
+
// D3: a read-only twin rejects any mutation with a vendor-shaped error.
|
|
1182
|
+
if (req.readOnly && method !== 'GET') {
|
|
1183
|
+
return { status: 405, body: errBody('invalid_request_error', 'twin is read-only; omit readOnly to accept writes') };
|
|
1184
|
+
}
|
|
1185
|
+
|
|
1186
|
+
// Everything Together's inference API serves hangs off /v1. A request outside it is a 404 like
|
|
1187
|
+
// any other unknown path.
|
|
1188
|
+
if (path !== TOGETHERAI_API_PREFIX && !path.startsWith(`${TOGETHERAI_API_PREFIX}/`)) {
|
|
1189
|
+
return notFound(`Unknown request URL: ${method} ${path}. Together's inference API is served under ${TOGETHERAI_API_PREFIX}/.`);
|
|
1190
|
+
}
|
|
1191
|
+
const seg = path.slice(TOGETHERAI_API_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean);
|
|
1192
|
+
|
|
1193
|
+
// ---- whoami (static identity for the presented key) ----
|
|
1194
|
+
if (seg[0] === 'whoami' && seg.length === 1 && method === 'GET') {
|
|
1195
|
+
return {
|
|
1196
|
+
status: 200,
|
|
1197
|
+
body: {
|
|
1198
|
+
api_key_id: 'key_twin',
|
|
1199
|
+
project_id: 'proj_twin',
|
|
1200
|
+
project_name: 'Twin Project',
|
|
1201
|
+
project_slug: 'twin-project',
|
|
1202
|
+
organization_id: 'org_twin',
|
|
1203
|
+
organization_name: 'Twin Organization',
|
|
1204
|
+
user_id: 'user_twin',
|
|
1205
|
+
},
|
|
1206
|
+
};
|
|
1207
|
+
}
|
|
1208
|
+
|
|
1209
|
+
// ---- models (static catalog) ----
|
|
1210
|
+
if (seg[0] === 'models' && seg.length === 1 && method === 'GET') {
|
|
1211
|
+
// Together answers a BARE ARRAY of ModelInfo (ModelInfoList), not an OpenAI list envelope.
|
|
1212
|
+
return { status: 200, body: servedModels(req.root) };
|
|
1213
|
+
}
|
|
1214
|
+
if (seg[0] === 'models' && seg.length >= 2 && method === 'GET') {
|
|
1215
|
+
// Together model ids contain slashes (`meta-llama/Llama-3.3-70B-Instruct-Turbo`), so the id
|
|
1216
|
+
// is EVERY remaining segment joined.
|
|
1217
|
+
const mid = seg.slice(1).map(dec).join('/');
|
|
1218
|
+
const m = servedModels(req.root).find((pm) => pm.id === mid);
|
|
1219
|
+
return m ? { status: 200, body: m } : notFound(`Model ${mid} does not exist.`);
|
|
1220
|
+
}
|
|
1221
|
+
|
|
1222
|
+
// ---- chat completions (the generative stub; envelope is faithful) ----
|
|
1223
|
+
if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
|
|
1224
|
+
const validated = validateChat(params);
|
|
1225
|
+
if ('error' in validated) return validated.error;
|
|
1226
|
+
const args = validated.args;
|
|
1227
|
+
const result = args.stream && req.sseSink
|
|
1228
|
+
? streamChat(args, req.sseSink, req.occurredAt, req.scenarioEngine)
|
|
1229
|
+
: buildChatCompletion(args, req.occurredAt, req.scenarioEngine);
|
|
1230
|
+
if (isEnvelope(result)) return result;
|
|
1231
|
+
return { status: 200, body: result };
|
|
1232
|
+
}
|
|
1233
|
+
|
|
1234
|
+
// ---- completions (legacy text) ----
|
|
1235
|
+
if (seg[0] === 'completions' && seg.length === 1 && method === 'POST') {
|
|
1236
|
+
if (params.model === undefined || typeof params.model !== 'string' || !params.model) return invalidRequest("'model' is a required property");
|
|
1237
|
+
if (typeof params.prompt !== 'string') return invalidRequest("'prompt' is a required property");
|
|
1238
|
+
const info = findModel(params.model);
|
|
1239
|
+
if (!info || info.type !== 'chat' && info.type !== 'language' && info.type !== 'code') {
|
|
1240
|
+
return notFound(`Model ${params.model} does not exist or is not a completion model.`);
|
|
1241
|
+
}
|
|
1242
|
+
const prompt = params.prompt;
|
|
1243
|
+
let text = `[twin-stub:${params.model}] deterministic completion stub (no model weights are run) continuing: ${prompt.slice(0, 120) || '(empty)'}`;
|
|
1244
|
+
let finish: TogetheraiFinishReason = 'stop';
|
|
1245
|
+
const maxRaw = params.max_tokens;
|
|
1246
|
+
if (maxRaw !== undefined && maxRaw !== null) {
|
|
1247
|
+
const maxTokens = Number(maxRaw);
|
|
1248
|
+
if (!Number.isInteger(maxTokens) || maxTokens < 1) return invalidRequest("'max_tokens' must be an integer >= 1");
|
|
1249
|
+
if (estimateTokens(text) > maxTokens) {
|
|
1250
|
+
text = text.slice(0, maxTokens * 4);
|
|
1251
|
+
finish = 'length';
|
|
1252
|
+
}
|
|
1253
|
+
}
|
|
1254
|
+
const completion = {
|
|
1255
|
+
id: `cmpl-twin-${fnv1a(prompt + params.model).toString(36)}`,
|
|
1256
|
+
object: 'text.completion' as const,
|
|
1257
|
+
created: nowEpoch(req.occurredAt),
|
|
1258
|
+
model: params.model,
|
|
1259
|
+
prompt: [{ text: prompt }],
|
|
1260
|
+
choices: [{ text, index: 0, finish_reason: finish }],
|
|
1261
|
+
usage: buildUsage(estimateTokens(prompt), estimateTokens(text)),
|
|
1262
|
+
};
|
|
1263
|
+
// stream:true streams Together's CompletionChunk sequence (the legacy API streams too — the
|
|
1264
|
+
// SDK types it `Stream<CompletionChunk>`): a `token` chunk per piece, then the finish chunk.
|
|
1265
|
+
if (params.stream === true && req.sseSink) {
|
|
1266
|
+
const base = { id: completion.id, object: 'completion.chunk' as const, created: completion.created, model: params.model };
|
|
1267
|
+
for (const piece of chunkText(text)) {
|
|
1268
|
+
req.sseSink({ data: { ...base, token: { id: 0, logprob: 0, special: false, text: piece }, choices: [{ index: 0, text: piece }], finish_reason: null, usage: null } });
|
|
1269
|
+
}
|
|
1270
|
+
req.sseSink({ data: { ...base, token: { id: 0, logprob: 0, special: true, text: '' }, choices: [{ index: 0 }], finish_reason: finish, usage: completion.usage } });
|
|
1271
|
+
req.sseSink({ done: true });
|
|
1272
|
+
return { status: 200, body: completion };
|
|
1273
|
+
}
|
|
1274
|
+
return { status: 200, body: completion };
|
|
1275
|
+
}
|
|
1276
|
+
|
|
1277
|
+
// ---- embeddings ----
|
|
1278
|
+
if (seg[0] === 'embeddings' && seg.length === 1 && method === 'POST') return handleEmbeddings(params);
|
|
1279
|
+
|
|
1280
|
+
// ---- rerank (Together-native) ----
|
|
1281
|
+
if (seg[0] === 'rerank' && seg.length === 1 && method === 'POST') return handleRerank(params);
|
|
1282
|
+
|
|
1283
|
+
// ---- images ----
|
|
1284
|
+
if (seg[0] === 'images' && seg[1] === 'generations' && seg.length === 2 && method === 'POST') return handleImages(params);
|
|
1285
|
+
|
|
1286
|
+
// ---- audio ----
|
|
1287
|
+
if (seg[0] === 'audio' && seg[1] === 'transcriptions' && seg.length === 2 && method === 'POST') return handleTranscription(params, false);
|
|
1288
|
+
if (seg[0] === 'audio' && seg[1] === 'translations' && seg.length === 2 && method === 'POST') return handleTranscription(params, true);
|
|
1289
|
+
if (seg[0] === 'audio' && seg[1] === 'speech' && seg.length === 2 && method === 'POST') return handleSpeech(params);
|
|
1290
|
+
|
|
1291
|
+
// ---- files (stateful; BOTH upload flows) ----
|
|
1292
|
+
if (seg[0] === 'files' && seg.length === 1 && method === 'POST') {
|
|
1293
|
+
// The together-ai SDK's redirect upload flow addresses POST /files with its params in the
|
|
1294
|
+
// QUERY STRING and a form-urlencoded body (lib/upload.js:41). That query-param shape is the
|
|
1295
|
+
// ONLY create on this path: Together's spec inventory has no JSON create on /files
|
|
1296
|
+
// (test-fixtures/togetherai-openapi-operations.json — GET only), and a JSON or EMPTY body
|
|
1297
|
+
// mints nothing. Serving a create here would be an invented door (§ round two, defect 1);
|
|
1298
|
+
// the vendor's own table answers a misconfigured request 400 invalid_request_error
|
|
1299
|
+
// (docs.together.ai/docs/error-codes). The flow is recognized by PARSING the query for the
|
|
1300
|
+
// params it sends (file_name/purpose — lib/upload.js:40), never by substring-matching the
|
|
1301
|
+
// raw string: `?file_type=jsonl` alone must NOT be mistaken for the flow.
|
|
1302
|
+
const q = new URLSearchParams(search);
|
|
1303
|
+
const hasFile = q.has('file_name');
|
|
1304
|
+
const hasPurpose = q.has('purpose');
|
|
1305
|
+
if (hasFile || hasPurpose) {
|
|
1306
|
+
return createFileSdkRedirect({
|
|
1307
|
+
...(hasPurpose ? { purpose: q.get('purpose') ?? undefined } : {}),
|
|
1308
|
+
...(hasFile ? { file_name: q.get('file_name') ?? undefined } : {}),
|
|
1309
|
+
...(q.has('file_type') ? { file_type: q.get('file_type') ?? undefined } : {}),
|
|
1310
|
+
...(typeof params.content === 'string' ? { content: params.content } : {}),
|
|
1311
|
+
}, req, q);
|
|
1312
|
+
}
|
|
1313
|
+
return invalidRequest("POST /v1/files takes the upload flow's urlencoded query parameters (?file_name=&file_type=&purpose=); the create itself is POST /v1/files/upload (multipart)");
|
|
1314
|
+
}
|
|
1315
|
+
if (seg[0] === 'files' && seg.length === 1 && method === 'GET') {
|
|
1316
|
+
// Together answers `{ data: [...] }` (FileList), NOT OpenAI's `{object:'list',data}`.
|
|
1317
|
+
return { status: 200, body: { data: rows('file', req.root).filter((r) => !r._deleted).map(fileView) } };
|
|
1318
|
+
}
|
|
1319
|
+
if (seg[0] === 'files' && seg[1] === 'upload' && seg.length === 2 && method === 'POST') {
|
|
1320
|
+
// The spec's multipart form upload (the server adapts multipart → JSON before the handler).
|
|
1321
|
+
return createFileMultipart(params, req);
|
|
1322
|
+
}
|
|
1323
|
+
if (seg[0] === 'files' && seg.length === 2 && method === 'GET') {
|
|
1324
|
+
const f = getRow('file', dec(seg[1]!), req.root);
|
|
1325
|
+
return f && !f._deleted ? { status: 200, body: fileView(f) } : notFound(`No such File object: ${dec(seg[1]!)}`);
|
|
1326
|
+
}
|
|
1327
|
+
if (seg[0] === 'files' && seg.length === 3 && seg[2] === 'content' && method === 'GET') {
|
|
1328
|
+
const f = getRow('file', dec(seg[1]!), req.root);
|
|
1329
|
+
if (!f || f._deleted) return notFound(`No such File object: ${dec(seg[1]!)}`);
|
|
1330
|
+
return { status: 200, body: String(f._content ?? ''), headers: { 'content-type': 'application/octet-stream' } };
|
|
1331
|
+
}
|
|
1332
|
+
if (seg[0] === 'files' && seg.length === 2 && method === 'DELETE') {
|
|
1333
|
+
const fid = dec(seg[1]!);
|
|
1334
|
+
const f = getRow('file', fid, req.root);
|
|
1335
|
+
if (!f || f._deleted) return notFound(`No such File object: ${fid}`);
|
|
1336
|
+
await applyTwinWrite(SERVICE, {
|
|
1337
|
+
operation: 'file.delete', subjectType: 'file', subjectId: fid, fields: { _deleted: true, object: 'file' },
|
|
1338
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1339
|
+
}, req.root);
|
|
1340
|
+
return { status: 200, body: { id: fid, deleted: true } };
|
|
1341
|
+
}
|
|
1342
|
+
|
|
1343
|
+
// ---- batches (stateful, Together-native shapes) ----
|
|
1344
|
+
if (seg[0] === 'batches' && seg.length === 1 && method === 'POST') return createBatch(params, req);
|
|
1345
|
+
if (seg[0] === 'batches' && seg.length === 1 && method === 'GET') {
|
|
1346
|
+
// Together answers a BARE ARRAY of BatchJob (its own published schema), NOT an OpenAI
|
|
1347
|
+
// `{object:'list',data}` envelope.
|
|
1348
|
+
return { status: 200, body: rows('batch', req.root).map(batchView) };
|
|
1349
|
+
}
|
|
1350
|
+
if (seg[0] === 'batches' && seg.length === 2 && method === 'GET') {
|
|
1351
|
+
const b = getRow('batch', dec(seg[1]!), req.root);
|
|
1352
|
+
if (!b) return notFound(`No such Batch object: ${dec(seg[1]!)}`);
|
|
1353
|
+
return { status: 200, body: batchView(b) };
|
|
1354
|
+
}
|
|
1355
|
+
if (seg[0] === 'batches' && seg.length === 3 && seg[2] === 'cancel' && method === 'POST') {
|
|
1356
|
+
const bid = dec(seg[1]!);
|
|
1357
|
+
const b = getRow('batch', bid, req.root);
|
|
1358
|
+
if (!b) return notFound(`No such Batch object: ${bid}`);
|
|
1359
|
+
if (b.status === 'CANCELLED') return invalidRequest(`Cannot cancel a batch with status '${String(b.status)}'.`);
|
|
1360
|
+
// NOTE: the twin does not simulate the asynchronous VALIDATING→IN_PROGRESS→COMPLETED
|
|
1361
|
+
// progression. Doing it on a READ made a GET write (breaking the read-only contract) and
|
|
1362
|
+
// minted actions the connector has no vendor endpoint to push (the groq pack's §9 round-one
|
|
1363
|
+
// findings 4 + 5). The terminal transitions are filed as todos.
|
|
1364
|
+
await applyTwinWrite(SERVICE, {
|
|
1365
|
+
operation: 'batch.cancel', subjectType: 'batch', subjectId: bid,
|
|
1366
|
+
fields: { status: 'CANCELLED' as TogetheraiBatchStatus },
|
|
1367
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1368
|
+
}, req.root);
|
|
1369
|
+
return { status: 200, body: batchView(getRow('batch', bid, req.root) ?? {}) };
|
|
1370
|
+
}
|
|
1371
|
+
|
|
1372
|
+
// ---- fine-tunes (stateful, Together-native shapes) ----
|
|
1373
|
+
if (seg[0] === 'fine-tunes' && seg.length === 1 && method === 'POST') return createFinetune(params, req);
|
|
1374
|
+
if (seg[0] === 'fine-tunes' && seg.length === 1 && method === 'GET') {
|
|
1375
|
+
return { status: 200, body: rows('finetune', req.root).filter((r) => !r._deleted).map(finetuneView) };
|
|
1376
|
+
}
|
|
1377
|
+
if (seg[0] === 'fine-tunes' && seg[1] === 'models' && seg[2] === 'limits' && seg.length === 3 && method === 'GET') {
|
|
1378
|
+
return finetuneModelLimits(req);
|
|
1379
|
+
}
|
|
1380
|
+
if (seg[0] === 'fine-tunes' && seg[1] === 'estimate-price' && seg.length === 2 && method === 'POST') {
|
|
1381
|
+
return estimateFinetunePrice(params, req);
|
|
1382
|
+
}
|
|
1383
|
+
if (seg[0] === 'fine-tunes' && seg[1] === 'preview' && seg.length === 2 && method === 'POST') {
|
|
1384
|
+
return previewFinetuneTokenization(params, req);
|
|
1385
|
+
}
|
|
1386
|
+
if (seg[0] === 'fine-tunes' && seg.length === 2 && method === 'GET') {
|
|
1387
|
+
const ft = getRow('finetune', dec(seg[1]!), req.root);
|
|
1388
|
+
return ft && !ft._deleted ? { status: 200, body: finetuneView(ft) } : notFound(`No such Fine-tune object: ${dec(seg[1]!)}`);
|
|
1389
|
+
}
|
|
1390
|
+
if (seg[0] === 'fine-tunes' && seg.length === 3 && seg[2] === 'cancel' && method === 'POST') {
|
|
1391
|
+
const fid = dec(seg[1]!);
|
|
1392
|
+
const ft = getRow('finetune', fid, req.root);
|
|
1393
|
+
if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${fid}`);
|
|
1394
|
+
if (ft.status === 'cancelled' || ft.status === 'cancel_requested' || ft.status === 'completed' || ft.status === 'error') {
|
|
1395
|
+
return invalidRequest(`Cannot cancel a fine-tune with status '${String(ft.status)}'.`);
|
|
1396
|
+
}
|
|
1397
|
+
await applyTwinWrite(SERVICE, {
|
|
1398
|
+
operation: 'finetune.cancel', subjectType: 'finetune', subjectId: fid,
|
|
1399
|
+
fields: { status: 'cancel_requested' },
|
|
1400
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1401
|
+
}, req.root);
|
|
1402
|
+
return { status: 200, body: finetuneView(getRow('finetune', fid, req.root) ?? {}) };
|
|
1403
|
+
}
|
|
1404
|
+
if (seg[0] === 'fine-tunes' && seg.length === 3 && seg[2] === 'events' && method === 'GET') {
|
|
1405
|
+
const ft = getRow('finetune', dec(seg[1]!), req.root);
|
|
1406
|
+
if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${dec(seg[1]!)}`);
|
|
1407
|
+
return { status: 200, body: { data: finetuneEvents(ft) } };
|
|
1408
|
+
}
|
|
1409
|
+
if (seg[0] === 'fine-tunes' && seg.length === 3 && seg[2] === 'checkpoints' && method === 'GET') {
|
|
1410
|
+
const ft = getRow('finetune', dec(seg[1]!), req.root);
|
|
1411
|
+
if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${dec(seg[1]!)}`);
|
|
1412
|
+
return { status: 200, body: { data: finetuneCheckpoints(ft) } };
|
|
1413
|
+
}
|
|
1414
|
+
if (seg[0] === 'fine-tunes' && seg.length === 2 && method === 'DELETE') {
|
|
1415
|
+
const fid = dec(seg[1]!);
|
|
1416
|
+
const ft = getRow('finetune', fid, req.root);
|
|
1417
|
+
if (!ft || ft._deleted) return notFound(`No such Fine-tune object: ${fid}`);
|
|
1418
|
+
await applyTwinWrite(SERVICE, {
|
|
1419
|
+
operation: 'finetune.delete', subjectType: 'finetune', subjectId: fid, fields: { _deleted: true, object: 'finetune' },
|
|
1420
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1421
|
+
}, req.root);
|
|
1422
|
+
// Together's fine-tune delete answers {message} (together-ai@0.53.0 fine-tuning.d.ts:1073) —
|
|
1423
|
+
// NOT OpenAI's {id, deleted} (which is the FILE delete shape, files.d.ts:168, served above).
|
|
1424
|
+
return { status: 200, body: { message: `Fine-tune ${fid} deleted.` } };
|
|
1425
|
+
}
|
|
1426
|
+
|
|
1427
|
+
// Unmodeled operation → fail like the vendor (never a fake success).
|
|
1428
|
+
return notFound(`Unknown request URL: ${method} ${path}.`);
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1431
|
+
/** The twin-only door the SDK redirect flow uploads bytes through (kept OUT of the manifest:
|
|
1432
|
+
* twin-only scaffolding, not vendor surface — the same rule the `/twin` doors follow). */
|
|
1433
|
+
export const TOGETHERAI_UPLOAD_DOOR = '/twin/upload';
|
|
1434
|
+
export async function handleTogetheraiUploadDoor(req: TogetheraiRequest): Promise<TogetheraiResponseEnvelope> {
|
|
1435
|
+
const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '');
|
|
1436
|
+
const m = new RegExp(`^${TOGETHERAI_UPLOAD_DOOR}/([^/]+)$`).exec(path);
|
|
1437
|
+
if (!m || req.method.toUpperCase() !== 'PUT') return notFound(`Unknown request URL: ${req.method} ${path}.`);
|
|
1438
|
+
// D3: the door WRITES (file.update), so a read-only twin refuses it with the same vendor-shaped
|
|
1439
|
+
// 405 every other write door answers — the server forwards readOnly here, and ignoring it let
|
|
1440
|
+
// `PUT /twin/upload/<id>` write on a --read-only server (§ round three, defect 2).
|
|
1441
|
+
if (req.readOnly) {
|
|
1442
|
+
return { status: 405, body: errBody('invalid_request_error', 'twin is read-only; omit readOnly to accept writes') };
|
|
1443
|
+
}
|
|
1444
|
+
return storeUploadBytes(decSafe(m[1]!), typeof req.body === 'string' ? req.body : '', req);
|
|
1445
|
+
}
|
|
1446
|
+
function decSafe(s: string): string {
|
|
1447
|
+
try { return decodeURIComponent(s); } catch { return s; }
|
|
1448
|
+
}
|