@volter/twin-togetherai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +147 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/index.d.ts +14 -0
- package/dist/src/index.js +79 -0
- package/dist/src/togetherai-budget.d.ts +52 -0
- package/dist/src/togetherai-budget.js +130 -0
- package/dist/src/togetherai-capabilities.d.ts +4 -0
- package/dist/src/togetherai-capabilities.js +1428 -0
- package/dist/src/togetherai-conformance.d.ts +14 -0
- package/dist/src/togetherai-conformance.js +452 -0
- package/dist/src/togetherai-connector.d.ts +164 -0
- package/dist/src/togetherai-connector.js +457 -0
- package/dist/src/togetherai-models.d.ts +19 -0
- package/dist/src/togetherai-models.js +49 -0
- package/dist/src/togetherai-scenario.d.ts +52 -0
- package/dist/src/togetherai-scenario.js +168 -0
- package/dist/src/togetherai-server.d.ts +16 -0
- package/dist/src/togetherai-server.js +187 -0
- package/dist/src/togetherai-stub.d.ts +59 -0
- package/dist/src/togetherai-stub.js +195 -0
- package/dist/src/togetherai-twin.d.ts +83 -0
- package/dist/src/togetherai-twin.js +1419 -0
- package/dist/src/togetherai-types.d.ts +207 -0
- package/dist/src/togetherai-types.js +26 -0
- package/package.json +52 -0
- package/src/cli.ts +27 -0
- package/src/index.ts +118 -0
- package/src/togetherai-budget.ts +156 -0
- package/src/togetherai-capabilities.ts +1315 -0
- package/src/togetherai-conformance.ts +459 -0
- package/src/togetherai-connector.ts +496 -0
- package/src/togetherai-models.ts +74 -0
- package/src/togetherai-scenario.ts +185 -0
- package/src/togetherai-server.ts +199 -0
- package/src/togetherai-stub.ts +197 -0
- package/src/togetherai-twin.ts +1448 -0
- package/src/togetherai-types.ts +222 -0
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
/** A chat message param as the caller sends it. Together's `ChatCompletionMessageParam` oneOf
|
|
2
|
+
* covers system/user/assistant/tool/function; content is a string (Together's schema types
|
|
3
|
+
* every param's content as `string | null` — no content-part arrays). */
|
|
4
|
+
export type TogetheraiMessageParam = {
|
|
5
|
+
role: 'system' | 'user' | 'assistant' | 'tool' | 'function';
|
|
6
|
+
content?: string | null;
|
|
7
|
+
name?: string;
|
|
8
|
+
tool_calls?: TogetheraiToolCall[];
|
|
9
|
+
tool_call_id?: string;
|
|
10
|
+
reasoning?: string | null;
|
|
11
|
+
reasoning_content?: string | null;
|
|
12
|
+
};
|
|
13
|
+
/** A function tool_call inside an assistant message (Together's `ToolChoice`). */
|
|
14
|
+
export type TogetheraiToolCall = {
|
|
15
|
+
id: string;
|
|
16
|
+
type: 'function';
|
|
17
|
+
function: {
|
|
18
|
+
name: string;
|
|
19
|
+
arguments: string;
|
|
20
|
+
};
|
|
21
|
+
};
|
|
22
|
+
/** Together's `UsageData`: the three token counts, REQUIRED and NULLABLE (Together reports
|
|
23
|
+
* `null` usage on some streamed/failed turns). No timing fields — that is Groq's block. */
|
|
24
|
+
export type TogetheraiUsage = {
|
|
25
|
+
prompt_tokens: number | null;
|
|
26
|
+
completion_tokens: number | null;
|
|
27
|
+
total_tokens: number | null;
|
|
28
|
+
/** Reasoning models nest the detail blocks (Together's docs: cached/reasoning token splits). */
|
|
29
|
+
prompt_tokens_details?: {
|
|
30
|
+
cached_tokens: number;
|
|
31
|
+
} | null;
|
|
32
|
+
completion_tokens_details?: {
|
|
33
|
+
reasoning_tokens: number;
|
|
34
|
+
} | null;
|
|
35
|
+
};
|
|
36
|
+
/** Together's assistant message. `reasoning`/`reasoning_content` vary by model (Together's docs:
|
|
37
|
+
* "varies by model" — both are declared on the schema). No `refusal` field. */
|
|
38
|
+
export type TogetheraiAssistantMessage = {
|
|
39
|
+
role: 'assistant';
|
|
40
|
+
content: string | null;
|
|
41
|
+
tool_calls?: TogetheraiToolCall[];
|
|
42
|
+
function_call?: {
|
|
43
|
+
name: string;
|
|
44
|
+
arguments: string;
|
|
45
|
+
};
|
|
46
|
+
reasoning?: string | null;
|
|
47
|
+
reasoning_content?: string | null;
|
|
48
|
+
};
|
|
49
|
+
/** Together's `FinishReason`: adds **`eos`** to the OpenAI set. */
|
|
50
|
+
export type TogetheraiFinishReason = 'stop' | 'eos' | 'length' | 'tool_calls' | 'function_call';
|
|
51
|
+
export type TogetheraiChoice = {
|
|
52
|
+
index: number;
|
|
53
|
+
message?: TogetheraiAssistantMessage;
|
|
54
|
+
text?: string;
|
|
55
|
+
seed?: number;
|
|
56
|
+
logprobs?: unknown | null;
|
|
57
|
+
finish_reason: TogetheraiFinishReason | null;
|
|
58
|
+
};
|
|
59
|
+
/** The unary chat.completion response. `prompt` is REQUIRED on Together's schema — the echoed
|
|
60
|
+
* prompt parts (empty unless `echo:true`); `warnings` is Together's inference-warnings array. */
|
|
61
|
+
export type TogetheraiChatCompletion = {
|
|
62
|
+
id: string;
|
|
63
|
+
object: 'chat.completion';
|
|
64
|
+
created: number;
|
|
65
|
+
model: string;
|
|
66
|
+
choices: TogetheraiChoice[];
|
|
67
|
+
prompt: Array<{
|
|
68
|
+
text?: string;
|
|
69
|
+
logprobs?: unknown;
|
|
70
|
+
}>;
|
|
71
|
+
usage?: TogetheraiUsage | null;
|
|
72
|
+
warnings?: unknown[];
|
|
73
|
+
};
|
|
74
|
+
/** A single Server-Sent Event the streaming path emits (collected, never socketed in tests).
|
|
75
|
+
* `data` is the JSON payload; `[DONE]` is signalled with `done: true` (no data object). */
|
|
76
|
+
export type SseEvent = {
|
|
77
|
+
data?: Record<string, unknown>;
|
|
78
|
+
done?: boolean;
|
|
79
|
+
};
|
|
80
|
+
/** A sink the streaming path writes events into (an injected collector in tests / a real
|
|
81
|
+
* HTTP SSE writer in the server). NO real sockets or setTimeout in the handler. */
|
|
82
|
+
export type SseSink = (event: SseEvent) => void;
|
|
83
|
+
/** A streaming chunk. Together's `ChatCompletionChunk` carries `usage` (nullable) and
|
|
84
|
+
* `warnings` on EVERY chunk (OpenAI puts usage only in an opt-in tail chunk). */
|
|
85
|
+
export type TogetheraiChunk = {
|
|
86
|
+
id: string;
|
|
87
|
+
object: 'chat.completion.chunk';
|
|
88
|
+
created: number;
|
|
89
|
+
model: string;
|
|
90
|
+
system_fingerprint?: string;
|
|
91
|
+
choices: Array<{
|
|
92
|
+
index: number;
|
|
93
|
+
delta: Record<string, unknown>;
|
|
94
|
+
logprobs?: unknown | null;
|
|
95
|
+
finish_reason: TogetheraiFinishReason | null;
|
|
96
|
+
}>;
|
|
97
|
+
usage?: TogetheraiUsage | null;
|
|
98
|
+
warnings?: unknown[];
|
|
99
|
+
};
|
|
100
|
+
/** Together's `ModelInfo`: REQUIRED keys are id/object/created/type — `type` is Together's OWN
|
|
101
|
+
* enum (chat|language|code|image|embedding|moderation|rerank), not OpenAI's flat model object. */
|
|
102
|
+
export type TogetheraiModel = {
|
|
103
|
+
id: string;
|
|
104
|
+
object: 'model';
|
|
105
|
+
created: number;
|
|
106
|
+
type: 'chat' | 'language' | 'code' | 'image' | 'embedding' | 'moderation' | 'rerank';
|
|
107
|
+
display_name?: string;
|
|
108
|
+
organization?: string;
|
|
109
|
+
link?: string;
|
|
110
|
+
license?: string;
|
|
111
|
+
context_length?: number;
|
|
112
|
+
pricing?: Record<string, unknown>;
|
|
113
|
+
};
|
|
114
|
+
export type TogetheraiEmbedding = {
|
|
115
|
+
object: 'embedding';
|
|
116
|
+
index: number;
|
|
117
|
+
embedding: number[];
|
|
118
|
+
};
|
|
119
|
+
/** Together's `EmbeddingsResponse`: REQUIRED object/model/data — and NO `usage` key on the
|
|
120
|
+
* schema (Together does not document one on this endpoint). */
|
|
121
|
+
export type TogetheraiEmbeddingResponse = {
|
|
122
|
+
object: 'list';
|
|
123
|
+
model: string;
|
|
124
|
+
data: TogetheraiEmbedding[];
|
|
125
|
+
};
|
|
126
|
+
/** Together's `RerankResponse` — Together-native (NOT an OpenAI endpoint at all). */
|
|
127
|
+
export type TogetheraiRerankResponse = {
|
|
128
|
+
object: 'rerank';
|
|
129
|
+
id?: string;
|
|
130
|
+
model: string;
|
|
131
|
+
results: Array<{
|
|
132
|
+
index: number;
|
|
133
|
+
relevance_score: number;
|
|
134
|
+
document?: {
|
|
135
|
+
text?: string | null;
|
|
136
|
+
};
|
|
137
|
+
}>;
|
|
138
|
+
usage?: TogetheraiUsage | null;
|
|
139
|
+
};
|
|
140
|
+
/** Together's `ImageResponse`: `object:'list'` with a discriminated data union
|
|
141
|
+
* (`type:'url'` | `type:'b64_json'`). */
|
|
142
|
+
export type TogetheraiImageResponse = {
|
|
143
|
+
id: string;
|
|
144
|
+
model: string;
|
|
145
|
+
object: 'list';
|
|
146
|
+
data: Array<{
|
|
147
|
+
index: number;
|
|
148
|
+
url?: string;
|
|
149
|
+
b64_json?: string;
|
|
150
|
+
type: 'url' | 'b64_json';
|
|
151
|
+
}>;
|
|
152
|
+
};
|
|
153
|
+
/** Together's `FilePurpose` — a CLOSED set (`fine-tune` | `eval` | `batch-api`), NOT OpenAI's. */
|
|
154
|
+
export type TogetheraiFilePurpose = 'fine-tune' | 'eval' | 'batch-api';
|
|
155
|
+
/** Together's `FileType` — csv|jsonl|parquet. */
|
|
156
|
+
export type TogetheraiFileType = 'csv' | 'jsonl' | 'parquet';
|
|
157
|
+
/** The validation pipeline's lifecycle state (fine-tune files only). */
|
|
158
|
+
export type TogetheraiFileProcessingStatus = 'PENDING' | 'QUEUED' | 'RUNNING' | 'COMPLETED' | 'FAILED' | 'INVALID_FORMAT';
|
|
159
|
+
/** Together's `FileResponse`. REQUIRED keys include Together's own `FileType` (capital F) and
|
|
160
|
+
* the deprecated `Processed` boolean. */
|
|
161
|
+
export type TogetheraiFile = {
|
|
162
|
+
id: string;
|
|
163
|
+
object: 'file';
|
|
164
|
+
created_at: number;
|
|
165
|
+
filename: string;
|
|
166
|
+
bytes: number;
|
|
167
|
+
purpose: TogetheraiFilePurpose;
|
|
168
|
+
Processed: boolean;
|
|
169
|
+
FileType: TogetheraiFileType;
|
|
170
|
+
processing_status?: TogetheraiFileProcessingStatus;
|
|
171
|
+
validation_report?: {
|
|
172
|
+
valid: boolean;
|
|
173
|
+
[k: string]: unknown;
|
|
174
|
+
};
|
|
175
|
+
};
|
|
176
|
+
/** Together's `BatchJobStatus` — UPPER-CASE, not OpenAI's lowercase set. */
|
|
177
|
+
export type TogetheraiBatchStatus = 'VALIDATING' | 'IN_PROGRESS' | 'COMPLETED' | 'FAILED' | 'EXPIRED' | 'CANCELLED';
|
|
178
|
+
/** Together's `CreateBatchRequest.endpoint` — the three documented endpoints. */
|
|
179
|
+
export type TogetheraiBatchEndpoint = '/v1/chat/completions' | '/v1/audio/transcriptions' | '/v1/audio/translations';
|
|
180
|
+
/** Together's `BatchJob`. `error` is a bare STRING (not an ErrorData object), and the
|
|
181
|
+
* timestamps are ISO date-time STRINGS (not OpenAI's unix integers). */
|
|
182
|
+
export type TogetheraiBatch = {
|
|
183
|
+
id: string;
|
|
184
|
+
user_id?: string;
|
|
185
|
+
input_file_id: string;
|
|
186
|
+
file_size_bytes?: number;
|
|
187
|
+
status: TogetheraiBatchStatus;
|
|
188
|
+
job_deadline?: string | null;
|
|
189
|
+
created_at: string;
|
|
190
|
+
endpoint: string;
|
|
191
|
+
progress?: number;
|
|
192
|
+
model_id?: string;
|
|
193
|
+
output_file_id?: string | null;
|
|
194
|
+
error_file_id?: string | null;
|
|
195
|
+
error?: string | null;
|
|
196
|
+
completed_at?: string | null;
|
|
197
|
+
};
|
|
198
|
+
/** Together's `ErrorData`: `{ error: { message, type, param, code } }` with `message`+`type`
|
|
199
|
+
* REQUIRED and `param`/`code` nullable with default null. */
|
|
200
|
+
export type TogetheraiError = {
|
|
201
|
+
error: {
|
|
202
|
+
message: string;
|
|
203
|
+
type: string;
|
|
204
|
+
param?: string | null;
|
|
205
|
+
code?: string | null;
|
|
206
|
+
};
|
|
207
|
+
};
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
// Shared wire-shape types for the Together AI surface. These mirror the REAL vendor JSON shapes
|
|
2
|
+
// as published by Together's own OpenAPI (https://docs.together.ai/openapi.yaml, fetched
|
|
3
|
+
// 2026-09-16) and the Stainless-generated `together-ai@0.53.0` SDK — NOT the SDK's internal
|
|
4
|
+
// types (the twin never imports the SDK at runtime; the SDK is exercised only in *.test.ts).
|
|
5
|
+
//
|
|
6
|
+
// TOGETHER IS NOT OPENAI, and the differences below are load-bearing (they are what a
|
|
7
|
+
// "just copy the openai pack" twin gets wrong):
|
|
8
|
+
// • `n` spans 1–128 (OpenAI-compatible; Groq pins it at 1) and `logprobs` is an INTEGER 0–20
|
|
9
|
+
// (OpenAI's boolean) — Together serves both;
|
|
10
|
+
// • `finish_reason` adds **`eos`** (Together's end-of-sequence token) alongside stop/length;
|
|
11
|
+
// • the response carries a REQUIRED `prompt: []` array (the echoed prompt when `echo:true`);
|
|
12
|
+
// • the error envelope is `{ error: { message, type, param, code } }` with `param`/`code`
|
|
13
|
+
// nullable, and Together's STATUS TABLE is its own: 402 = monthly spending limit,
|
|
14
|
+
// **403 = context length exceeded** (a BAD REQUEST, not a permission denial),
|
|
15
|
+
// 503 = engine overloaded, 524 = Cloudflare timeout, 529 = server error;
|
|
16
|
+
// • the batch surface is Together-NATIVE: `GET /v1/batches` answers a bare ARRAY (not an
|
|
17
|
+
// OpenAI `{object:'list',data}` envelope), `POST /v1/batches` answers 201 with
|
|
18
|
+
// `BatchJobWithWarning { job, warning? }`, and every batch error is `{ error: string }`
|
|
19
|
+
// (a bare string, NOT an ErrorData object);
|
|
20
|
+
// • files carry Together's own `FileType` (csv|jsonl|parquet) and the validation pipeline's
|
|
21
|
+
// `processing_status` / `validation_report`;
|
|
22
|
+
// • model ids are slash-namespaced (`meta-llama/Llama-3.3-70B-Instruct-Turbo`) and an
|
|
23
|
+
// OpenAI-style flat id like `gpt-4o` is a 404;
|
|
24
|
+
// • `service_tier` / `store` / `metadata` / `prediction` are ACCEPTED BUT IGNORED (the
|
|
25
|
+
// OpenAI-compat page's own wording) — accepted here, never echoed into the response.
|
|
26
|
+
export {};
|
package/package.json
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@volter/twin-togetherai",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Local Together AI twin — a faithful, stateful local Together API your real `together-ai` SDK talks to unmodified. The model is stubbed (deterministic) and embeddings/rerank/images/transcripts are deterministic stubs, but the protocol envelope (chat completions/streaming/tool_calls, Together's prompt array + eos finish reasons + 402/403/503 status table, models, files, batches, fine-tunes, audio) is vendor-faithful. Built on @volter/world-core.",
|
|
5
|
+
"author": "Volter (https://github.com/volter-ai)",
|
|
6
|
+
"license": "Apache-2.0",
|
|
7
|
+
"files": [
|
|
8
|
+
"src",
|
|
9
|
+
"README.md",
|
|
10
|
+
"LICENSE",
|
|
11
|
+
"!**/*.test.ts",
|
|
12
|
+
"!**/*.test.tsx",
|
|
13
|
+
"dist"
|
|
14
|
+
],
|
|
15
|
+
"repository": {
|
|
16
|
+
"type": "git",
|
|
17
|
+
"url": "git+https://github.com/volter-ai/twin.git",
|
|
18
|
+
"directory": "packages/twin/togetherai"
|
|
19
|
+
},
|
|
20
|
+
"homepage": "https://github.com/volter-ai/twin/tree/main/packages/twin/togetherai#readme",
|
|
21
|
+
"type": "module",
|
|
22
|
+
"exports": {
|
|
23
|
+
".": {
|
|
24
|
+
"types": "./dist/src/index.d.ts",
|
|
25
|
+
"default": "./dist/src/index.js"
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
"bin": {
|
|
29
|
+
"world-togetherai": "dist/src/cli.js"
|
|
30
|
+
},
|
|
31
|
+
"scripts": {
|
|
32
|
+
"test": "bun test src/*.test.ts",
|
|
33
|
+
"typecheck": "tsc --noEmit",
|
|
34
|
+
"build": "node ../../../scripts/publish/build.mjs",
|
|
35
|
+
"prepack": "node ../../../scripts/publish/prepare-publish.mjs prepack",
|
|
36
|
+
"postpack": "node ../../../scripts/publish/prepare-publish.mjs postpack"
|
|
37
|
+
},
|
|
38
|
+
"peerDependencies": {
|
|
39
|
+
"@volter/world-core": "2.0.0"
|
|
40
|
+
},
|
|
41
|
+
"devDependencies": {
|
|
42
|
+
"@types/bun": "^1.2.20",
|
|
43
|
+
"@types/node": "^24.0.0",
|
|
44
|
+
"@volter/world-core": "2.0.0",
|
|
45
|
+
"@volter/world-tooling": "0.1.0",
|
|
46
|
+
"together-ai": "0.53.0",
|
|
47
|
+
"typescript": "^5.9.0"
|
|
48
|
+
},
|
|
49
|
+
"engines": {
|
|
50
|
+
"node": ">=22.3"
|
|
51
|
+
}
|
|
52
|
+
}
|
package/src/cli.ts
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { keepProcessAlive } from '@volter/world-core/lifecycle';
|
|
3
|
+
// world-togetherai CLI: serve the Together API twin or run conformance. Together is an API-first
|
|
4
|
+
// vendor — app.together.ai is a console, not where the work happens
|
|
5
|
+
// (docs/contributing/architecture.md C1b) — so this pack ships no mirror.
|
|
6
|
+
import { hasFlag, optionValue } from '@volter/world-core/args';
|
|
7
|
+
import { createTogetheraiTwinServer } from './togetherai-server.ts';
|
|
8
|
+
|
|
9
|
+
const [cmd, ...rest] = process.argv.slice(2);
|
|
10
|
+
const port = Number(optionValue(rest, '--port', '0')) || undefined;
|
|
11
|
+
const root = optionValue(rest, '--root') || undefined;
|
|
12
|
+
const readOnly = hasFlag(rest, '--read-only'); // a twin accepts writes unless started read-only
|
|
13
|
+
const scenario = optionValue(rest, '--scenario') || undefined; // scripted completions (JSON file)
|
|
14
|
+
|
|
15
|
+
if (cmd === 'serve') {
|
|
16
|
+
const s = await createTogetheraiTwinServer({ readOnly, ...(root ? { root } : {}), ...(port ? { port } : {}), ...(scenario ? { scenarioPath: scenario } : {}) });
|
|
17
|
+
process.stdout.write(`togetherai twin (Together API at /v1; model output is a deterministic stub)${readOnly ? ' [read-only]' : ''}${scenario ? ` [scenario: ${scenario}]` : ''} at http://127.0.0.1:${s.port}\n`);
|
|
18
|
+
await keepProcessAlive();
|
|
19
|
+
} else if (cmd === 'conformance') {
|
|
20
|
+
// dev-only; lazy so the bin runs without @volter/world-tooling
|
|
21
|
+
const { checkTogetheraiConformance } = await import('./togetherai-conformance.ts');
|
|
22
|
+
const report = await checkTogetheraiConformance({ ...(root ? { root } : {}) });
|
|
23
|
+
process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
|
|
24
|
+
if (!report.ok) process.exitCode = 1;
|
|
25
|
+
} else {
|
|
26
|
+
process.stdout.write('Usage: world-togetherai serve|conformance [--port N] [--root DIR] [--read-only] [--scenario FILE]\n');
|
|
27
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
// @volter/twin-togetherai — the Together AI twin (one vendor, one package), built on the shared
|
|
2
|
+
// @volter/world-core kernel. The Together API: a vendor-faithful protocol envelope (chat
|
|
3
|
+
// completions/streaming/tool_calls with Together's REQUIRED `prompt` array and per-chunk
|
|
4
|
+
// usage/warnings, `eos` finish reasons, Together's OWN status table — 402 spending limit, 403
|
|
5
|
+
// context length, 503 engine overloaded), a static models catalog, deterministic embeddings/
|
|
6
|
+
// rerank/image/audio stubs, and STATEFUL files + batches + fine-tunes — over an event/action log.
|
|
7
|
+
// (Together is an API-first vendor: app.together.ai is a console, not where the work happens, so
|
|
8
|
+
// this pack ships no mirror.)
|
|
9
|
+
//
|
|
10
|
+
// THE DETERMINISTIC STUB IS THE ANSWER: the twin runs no model, so POST /v1/chat/completions
|
|
11
|
+
// returns a DETERMINISTIC STUB completion (clearly labeled), /v1/embeddings returns DETERMINISTIC
|
|
12
|
+
// pseudo-vectors, and the audio endpoints return DETERMINISTIC labeled stubs — never pretending to
|
|
13
|
+
// be real inference. Everything around them — the wire protocol — is faithful. (Conformance +
|
|
14
|
+
// capability tooling live in @volter/world-tooling, a dev dependency — NOT re-exported here, per
|
|
15
|
+
// E2.)
|
|
16
|
+
export { handleTogetheraiTwinRequest, streamChat, buildChatCompletion, TOGETHERAI_API_PREFIX, TOGETHERAI_UPLOAD_DOOR, handleTogetheraiUploadDoor } from './togetherai-twin.ts';
|
|
17
|
+
export type { TogetheraiRequest, TogetheraiResponseEnvelope } from './togetherai-twin.ts';
|
|
18
|
+
export { createTogetheraiTwinFetch, createTogetheraiTwinServer, type TogetheraiTwinFetchOptions } from './togetherai-server.ts';
|
|
19
|
+
export { TOGETHERAI_MODELS, findModel, SPEECH_MODELS, EMBEDDING_MODELS, EMBEDDING_DIMENSIONS, RERANK_MODELS } from './togetherai-models.ts';
|
|
20
|
+
export {
|
|
21
|
+
buildUsage, contentToText, countPromptTokens, estimateTokens,
|
|
22
|
+
fnv1a, lastUserText, pseudoEmbedding, stubAssistantText, stubAudioSeconds, stubJsonObject,
|
|
23
|
+
stubReasoningText, stubToolArguments, stubToolCall, stubTranscript,
|
|
24
|
+
} from './togetherai-stub.ts';
|
|
25
|
+
export type {
|
|
26
|
+
TogetheraiAssistantMessage, TogetheraiBatch, TogetheraiBatchEndpoint, TogetheraiBatchStatus,
|
|
27
|
+
TogetheraiChatCompletion, TogetheraiChoice, TogetheraiEmbedding, TogetheraiEmbeddingResponse,
|
|
28
|
+
TogetheraiError, TogetheraiFile, TogetheraiFilePurpose, TogetheraiFileType,
|
|
29
|
+
TogetheraiFinishReason, TogetheraiMessageParam, TogetheraiModel, TogetheraiRerankResponse,
|
|
30
|
+
TogetheraiToolCall, TogetheraiUsage, SseEvent, SseSink,
|
|
31
|
+
} from './togetherai-types.ts';
|
|
32
|
+
export { createTogetheraiScenarioEngine, togetheraiScenarioAdapter, loadTogetheraiScenarioDocument, realizeTogetheraiRespond } from './togetherai-scenario.ts';
|
|
33
|
+
export type { TogetheraiScenarioEngine, TogetheraiScenarioRequest, TogetheraiScenarioRespond, ScenarioToolCall, ScriptedResult } from './togetherai-scenario.ts';
|
|
34
|
+
export {
|
|
35
|
+
fullSyncTogetherai,
|
|
36
|
+
externalIdFor,
|
|
37
|
+
togetheraiRequestForAction,
|
|
38
|
+
liveTogetheraiExecute,
|
|
39
|
+
mapBatch,
|
|
40
|
+
mapFile,
|
|
41
|
+
mapModel,
|
|
42
|
+
pullTogetheraiState,
|
|
43
|
+
pushTogetheraiAction,
|
|
44
|
+
pushPendingTogetheraiActions,
|
|
45
|
+
syncTogetheraiFromReal,
|
|
46
|
+
unpushableReason,
|
|
47
|
+
} from './togetherai-connector.ts';
|
|
48
|
+
export type { TogetheraiExecute, LiveTogetheraiOptions } from './togetherai-connector.ts';
|
|
49
|
+
// The client-side rate budget — the fail-closed backstop `liveTogetheraiExecute` routes every live
|
|
50
|
+
// request through. The MECHANISM is the kernel's shared, vendor-agnostic `RateBudget`; what lives
|
|
51
|
+
// here is Together's DECLARATION (window/ceiling/per-endpoint weights) plus the vendor-bound
|
|
52
|
+
// bindings. Exported so an operator can inspect spend (`snapshot`) and so a caller can catch
|
|
53
|
+
// `TogetheraiBudgetError` by type; there is deliberately no export that disables the guard.
|
|
54
|
+
export {
|
|
55
|
+
TOGETHERAI_BUDGET_CEILING,
|
|
56
|
+
TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S,
|
|
57
|
+
TOGETHERAI_BUDGET_WINDOW_MS,
|
|
58
|
+
TOGETHERAI_CALL_WEIGHTS,
|
|
59
|
+
TOGETHERAI_RATE_BUDGET,
|
|
60
|
+
TogetheraiBudget,
|
|
61
|
+
TogetheraiBudgetError,
|
|
62
|
+
togetheraiBudgetPath,
|
|
63
|
+
togetheraiCallWeight,
|
|
64
|
+
} from './togetherai-budget.ts';
|
|
65
|
+
export type { TogetheraiBudgetErrorKind, TogetheraiBudgetOptions, TogetheraiBudgetReservation, TogetheraiBudgetSnapshot } from './togetherai-budget.ts';
|
|
66
|
+
|
|
67
|
+
// Registry descriptor: the pack self-describes so tooling can discover it.
|
|
68
|
+
import { registerPack, type TwinPack } from '@volter/world-core';
|
|
69
|
+
import { TOGETHERAI_RATE_BUDGET as RATE_BUDGET } from './togetherai-budget.ts';
|
|
70
|
+
import { performTogetheraiAction, syncTogetheraiFromRemote } from './togetherai-connector.ts';
|
|
71
|
+
export const pack: TwinPack = {
|
|
72
|
+
vendor: 'togetherai',
|
|
73
|
+
// PROTOCOL 2 (docs/contributing/architecture.md#protocol-2-the-pack-is-a-plugin): the pack is a plugin — its wire, its tree, and its half of
|
|
74
|
+
// the real state system: perform one entry against Together, refresh the root from it.
|
|
75
|
+
protocol: '2',
|
|
76
|
+
stateSystem: { perform: performTogetheraiAction, refresh: syncTogetheraiFromRemote },
|
|
77
|
+
// the round trip: a file create — the one thing this API mints an id for and keeps, a fresh id
|
|
78
|
+
// each time, so it repeats cleanly on a branch (the spec's own create door, POST
|
|
79
|
+
// /v1/files/upload; Together's purposes are a CLOSED set: fine-tune | eval | batch-api)
|
|
80
|
+
roundTrip: { method: 'POST', path: '/v1/files/upload', body: { purpose: 'batch-api', file_name: 'round-trip.jsonl', content: '' }, headers: { authorization: 'Bearer rt' } },
|
|
81
|
+
// The SAME object togetherai-budget.ts declares at module load — one source of truth, so
|
|
82
|
+
// registering the pack and importing the connector can never arm two different ceilings.
|
|
83
|
+
rateBudget: RATE_BUDGET,
|
|
84
|
+
transport: 'rest',
|
|
85
|
+
archetype: 'generative',
|
|
86
|
+
bin: 'world-togetherai',
|
|
87
|
+
// R2 adopted-as-debt, legibly: declared types no replay can create, each with its reason.
|
|
88
|
+
resourcesUnreachable: {
|
|
89
|
+
'model': 'the vendor catalog',
|
|
90
|
+
},
|
|
91
|
+
resources: ['file', 'batch', 'finetune', 'model'],
|
|
92
|
+
specSource: 'Together API — enumerated from Together\'s own published OpenAPI spec (api.together.xyz/openapi.json, captured to test-fixtures/togetherai-openapi.yaml: 141 paths / 156 operations — the /v1 inference half AND the v2 management half, of which this pack models the inference half) + together-ai@0.53.0 (Stainless-generated from Together\'s spec: the model unions, the SDK-only 302-redirect upload flow, whoami, the TOGETHER_BASE_URL/TOGETHER_API_KEY/TOGETHER_PROJECT_ID env contract) + docs.together.ai/{error-codes,serverless/rate-limits,inference/openai-compatibility} (all read 2026-09-16). Envelope-faithful; model output, embedding values, rerank scores, images and audio are labeled deterministic stubs.',
|
|
93
|
+
description: 'Together AI API twin — faithful protocol envelope (chat completions/streaming/tool_calls, Together\'s prompt array and eos finish reasons, Together\'s own 402/403/503 status table), models, stateful files/batches/fine-tunes (both upload flows, incl. the SDK\'s 302 redirect), embeddings/rerank/images/audio; generative output is a labeled stub.',
|
|
94
|
+
// ADOPTION (adding-a-twin.md §3): how an app repo betrays that it talks to Together. Declared
|
|
95
|
+
// HERE, not in world-runtime's central SDK_TWINS/ENV_STEM_VENDORS maps — declaring in both
|
|
96
|
+
// throws. `together-ai` is Together's own official client (npm).
|
|
97
|
+
adoption: {
|
|
98
|
+
sdks: ['together-ai'], envStems: ['TOGETHER'],
|
|
99
|
+
},
|
|
100
|
+
// INTERCEPTION: the host the official client addresses by default
|
|
101
|
+
// (together-ai@0.53.0 client.js: TOGETHER_BASE_URL defaults to https://api.together.ai/v1).
|
|
102
|
+
hosts: [{ host: 'api.together.ai' }],
|
|
103
|
+
// App-read endpoint templates RIDE ALONGSIDE the injector (the PLANETSCALE_DATABASE_URL
|
|
104
|
+
// precedent, init.ts wiringFor): `together-ai`'s custom `files.upload()` reads
|
|
105
|
+
// `TOGETHER_API_BASE_URL` at MODULE LOAD and IGNORES the client's `baseURL`
|
|
106
|
+
// (lib/upload.js:12 — it then POSTs `${baseURL}/files?…` itself), so interception of
|
|
107
|
+
// api.together.ai does NOT cover it: unset, `files.upload()` egresses to the real
|
|
108
|
+
// https://api.together.xyz with the caller's key. The template points that var at the twin's
|
|
109
|
+
// `/v1` (the SDK POSTs `/files?…` onto it directly), so the one flow the injector cannot reach
|
|
110
|
+
// is wired too, while ordinary SDK traffic still rides the injector.
|
|
111
|
+
endpointEnv: {
|
|
112
|
+
name: 'TOGETHER_TWIN_URL',
|
|
113
|
+
templates: { TOGETHER_API_BASE_URL: '${url}/v1' },
|
|
114
|
+
note: 'the injector covers api.together.ai, but together-ai\'s files.upload() reads TOGETHER_API_BASE_URL at module load and ignores client.baseURL (lib/upload.js:12) — the template points it at the twin\'s /v1 so the SDK upload flow does not egress to the real vendor.',
|
|
115
|
+
},
|
|
116
|
+
};
|
|
117
|
+
// registered at import: the kernel learns the pack's state system (protocol 2)
|
|
118
|
+
registerPack(pack);
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
// Together AI's CLIENT-SIDE RATE BUDGET — the pack's DECLARATION (the numbers) plus the thin
|
|
2
|
+
// typed bindings `liveTogetheraiExecute` uses. The MECHANISM — the durable token-keyed ledger,
|
|
3
|
+
// the rolling window, reserve-under-lock, the `Retry-After`/429 cooldown, fail-CLOSED on a
|
|
4
|
+
// corrupt ledger — lives ONCE in the vendor-agnostic kernel (`@volter/world-core` →
|
|
5
|
+
// `rateBudget.ts`). Read that module's header for the full rationale AND for the honest list of
|
|
6
|
+
// what the guard does not guarantee.
|
|
7
|
+
//
|
|
8
|
+
// ── WHY THIS EXISTS ─────────────────────────────────────────────────────────────────────────
|
|
9
|
+
// A real ~4.5-DAY vendor lockout (Figma, 2026-07-25) happened because raw API calls were made
|
|
10
|
+
// outside the pack's connector — no cache, no batching, no ceiling. Discipline only binds the code
|
|
11
|
+
// that follows it; a BUDGET binds the code that does not.
|
|
12
|
+
//
|
|
13
|
+
// ── HOW THE CEILING WAS CHOSEN (live-read, 2026-09-16) ──────────────────────────────────────
|
|
14
|
+
// https://docs.together.ai/serverless/rate-limits (read 2026-09-16) publishes NO scalar a client
|
|
15
|
+
// could be bound by: limits are "dynamic" and PER ORGANIZATION and PER MODEL, and the page's own
|
|
16
|
+
// wording is "there are no fixed per-model limits published" — the account's dashboard is the only
|
|
17
|
+
// place the real numbers live. Success responses carry NO rate-limit headers; a 429 carries only
|
|
18
|
+
// `x-ratelimit-reset` (seconds until the window resets), and its error types are
|
|
19
|
+
// `dynamic_request_limited` / `dynamic_token_limited`; a 503 means the dynamic rate is at/below
|
|
20
|
+
// capacity.
|
|
21
|
+
//
|
|
22
|
+
// So this declaration does not model Together's limit, and nothing here is more permissive than
|
|
23
|
+
// the kernel's undeclared fallback: window 60s, ceiling 60, defaultWeight 2 — i.e. 30
|
|
24
|
+
// calls/minute, EXACTLY `DEFAULT_RATE_BUDGET`, with no endpoint priced cheaper than the fallback
|
|
25
|
+
// would price it. Being stricter than the fallback needs no vendor justification; being looser
|
|
26
|
+
// would, and there is none to have.
|
|
27
|
+
//
|
|
28
|
+
// It bounds the 60s AVERAGE; it does not pace (the kernel refuses, it never sleeps). The backstop
|
|
29
|
+
// for a sub-second burst is the cooldown: Together's `x-ratelimit-reset` (documented on the 429)
|
|
30
|
+
// is read off the response and turns into a persisted refusal.
|
|
31
|
+
import {
|
|
32
|
+
declareRateBudget,
|
|
33
|
+
rateBudgetPath,
|
|
34
|
+
rateBudgetWeight,
|
|
35
|
+
RateBudget,
|
|
36
|
+
type RateBudgetDeclaration,
|
|
37
|
+
type RateBudgetOptions,
|
|
38
|
+
type RateBudgetReservation,
|
|
39
|
+
type RateBudgetSnapshot,
|
|
40
|
+
} from '@volter/world-core';
|
|
41
|
+
|
|
42
|
+
const VENDOR = 'togetherai';
|
|
43
|
+
|
|
44
|
+
/** Rolling window, in ms. Spend older than this is pruned. */
|
|
45
|
+
export const TOGETHERAI_BUDGET_WINDOW_MS = 60_000;
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Weighted units allowed inside one window. 60/60s at `defaultWeight` 2 = 30 calls a minute —
|
|
49
|
+
* EXACTLY the kernel's undeclared fallback, because Together publishes no scalar that would
|
|
50
|
+
* justify more. See the header.
|
|
51
|
+
*/
|
|
52
|
+
export const TOGETHERAI_BUDGET_CEILING = 60;
|
|
53
|
+
|
|
54
|
+
/** Seconds. A `x-ratelimit-reset` above this means the key is throttled hard — fail loudly, don't sleep. */
|
|
55
|
+
export const TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S = 300;
|
|
56
|
+
|
|
57
|
+
/** Per-call cost, keyed by `"<METHOD> <path>"`. See the header for what is documented vs. judged. */
|
|
58
|
+
export const TOGETHERAI_CALL_WEIGHTS = {
|
|
59
|
+
/** `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/rerank`,
|
|
60
|
+
* `/v1/images/generations`, `/v1/audio/*` — token-metered, where Together's DYNAMIC token
|
|
61
|
+
* budget (`dynamic_token_limited`) rather than its request budget is usually what binds. */
|
|
62
|
+
inference: 6,
|
|
63
|
+
/** Everything else: files, batches, models, whoami, fine-tunes, endpoints. */
|
|
64
|
+
other: 2,
|
|
65
|
+
} as const;
|
|
66
|
+
|
|
67
|
+
/** THE PACK'S DECLARATION — pure data, the only Together-specific thing in the whole budget. */
|
|
68
|
+
export const TOGETHERAI_RATE_BUDGET: RateBudgetDeclaration = {
|
|
69
|
+
windowMs: TOGETHERAI_BUDGET_WINDOW_MS,
|
|
70
|
+
ceiling: TOGETHERAI_BUDGET_CEILING,
|
|
71
|
+
defaultWeight: TOGETHERAI_CALL_WEIGHTS.other,
|
|
72
|
+
maxRetryAfterSeconds: TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S,
|
|
73
|
+
rules: [
|
|
74
|
+
{ match: '^POST /v1/(chat/completions|completions|embeddings|rerank|images/generations)$', weight: TOGETHERAI_CALL_WEIGHTS.inference },
|
|
75
|
+
{ match: '^POST /v1/audio/', weight: TOGETHERAI_CALL_WEIGHTS.inference },
|
|
76
|
+
],
|
|
77
|
+
reason:
|
|
78
|
+
'Together publishes NO fixed request limit (docs.together.ai/serverless/rate-limits, read ' +
|
|
79
|
+
'2026-09-16): limits are DYNAMIC and per organization per model — the page states "there are ' +
|
|
80
|
+
'no fixed per-model limits published", the real numbers live only on the account dashboard, ' +
|
|
81
|
+
'and success responses carry no rate-limit headers at all. A 429 carries only ' +
|
|
82
|
+
'`x-ratelimit-reset` (seconds) with error types `dynamic_request_limited` / ' +
|
|
83
|
+
'`dynamic_token_limited`; a 503 signals the dynamic rate is at/below capacity. Because no ' +
|
|
84
|
+
'published figure justifies going higher, the ceiling is pinned at the kernel fallback in ' +
|
|
85
|
+
'EVERY dimension — 60 units / 60s at defaultWeight 2 = 30 calls/min — and the declaration ' +
|
|
86
|
+
'buys resolution DOWNWARD, never headroom: the inference endpoints cost 6, so at most 10 land ' +
|
|
87
|
+
'in a window. The window bounds the 60s AVERAGE and does not pace; the `x-ratelimit-reset` ' +
|
|
88
|
+
'cooldown is the backstop for a sub-second burst.',
|
|
89
|
+
};
|
|
90
|
+
|
|
91
|
+
// Declared at module load, so merely importing this module (which `togetherai-connector.ts`
|
|
92
|
+
// does) is enough to arm the real ceiling.
|
|
93
|
+
declareRateBudget(VENDOR, TOGETHERAI_RATE_BUDGET);
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Price one call. The key is `"<METHOD> <path>"` with the query string split off, so a rule can
|
|
97
|
+
* price by method (a write is not a read) without the kernel knowing anything about Together. An
|
|
98
|
+
* unclassified endpoint still costs `defaultWeight` — nothing is ever free.
|
|
99
|
+
*/
|
|
100
|
+
export function togetheraiCallWeight(method: string, path: string): number {
|
|
101
|
+
const { bare, query } = splitQuery(path);
|
|
102
|
+
// UPPER-CASE the method: `fetch` normalizes a known lowercase method before sending, so
|
|
103
|
+
// `execute('post', …)` really does issue a POST and must be priced as one.
|
|
104
|
+
return rateBudgetWeight(VENDOR, `${String(method).toUpperCase()} ${bare}`, query);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* `/v1/x?a=1` -> `{ bare: '/v1/x', query: { a: '1' } }`. Rules match the path; NORMALIZED,
|
|
109
|
+
* because the anchored rules are otherwise trivially evaded: `fetch` upper-cases a known method
|
|
110
|
+
* before sending, so `execute('post', …)` issues a real WRITE that a `^POST ` rule would price as
|
|
111
|
+
* a read; and a trailing slash makes a path miss a `$` anchor while most routers treat it as the
|
|
112
|
+
* same endpoint.
|
|
113
|
+
*/
|
|
114
|
+
function splitQuery(path: string): { bare: string; query: Record<string, string> } {
|
|
115
|
+
const at = path.indexOf('?');
|
|
116
|
+
const query: Record<string, string> = {};
|
|
117
|
+
if (at !== -1) for (const [k, v] of new URLSearchParams(path.slice(at + 1))) query[k] = v;
|
|
118
|
+
// Collapse REPEATED slashes as well as a trailing one: `/v1//chat/completions` reaches the same
|
|
119
|
+
// endpoint on most routers but misses a `^POST /v1/(chat/completions|…)$` rule, which would
|
|
120
|
+
// price an inference call as a 2-unit read (the groq pack's §9 round one, NIT 14).
|
|
121
|
+
const raw = (at === -1 ? path : path.slice(0, at)).replace(/\/{2,}/g, '/');
|
|
122
|
+
const bare = raw.length > 1 && raw.endsWith('/') ? raw.replace(/\/+$/, '') : raw;
|
|
123
|
+
return { bare, query };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** Where Together's ledger lives. Token-keyed and cwd-independent by default (Together's limits
|
|
127
|
+
* are per ORGANIZATION, i.e. per key, so a cwd-scoped ledger would hand the same key a fresh
|
|
128
|
+
* allowance in every checkout, worktree and CI matrix leg); pass `root` for world-scoped
|
|
129
|
+
* accounting. */
|
|
130
|
+
export function togetheraiBudgetPath(opts: { root?: string; token?: string } | string = {}): string {
|
|
131
|
+
const o = typeof opts === 'string' ? { root: opts } : opts;
|
|
132
|
+
// VENDOR spread LAST: a loosely-typed `{ vendor: 'other', … }` slipping through (TypeScript's
|
|
133
|
+
// excess-property check only catches object literals) must not redirect this pack's ledger.
|
|
134
|
+
return rateBudgetPath({ ...o, vendor: VENDOR });
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/** Construction options for Together's budget. The vendor is fixed; everything else may only TIGHTEN. */
|
|
138
|
+
export type TogetheraiBudgetOptions = Omit<RateBudgetOptions, 'vendor'>;
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Together's budget — the shared kernel guard bound to this vendor's declaration. A real
|
|
142
|
+
* subclass, not an alias, so `budget instanceof TogetheraiBudget` in `liveTogetheraiExecute`
|
|
143
|
+
* means "a budget that accounts against TOGETHER's ledger under TOGETHER's ceiling".
|
|
144
|
+
*/
|
|
145
|
+
export class TogetheraiBudget extends RateBudget {
|
|
146
|
+
constructor(opts: TogetheraiBudgetOptions = {}) {
|
|
147
|
+
super({ ...opts, vendor: VENDOR });
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/** The typed refusal. One error class shared with every other vendor's budget; `err.vendor` says
|
|
152
|
+
* which one refused, and `err.kind` says why. */
|
|
153
|
+
export { RateBudgetError as TogetheraiBudgetError } from '@volter/world-core';
|
|
154
|
+
export type { RateBudgetErrorKind as TogetheraiBudgetErrorKind } from '@volter/world-core';
|
|
155
|
+
export type TogetheraiBudgetReservation = RateBudgetReservation;
|
|
156
|
+
export type TogetheraiBudgetSnapshot = RateBudgetSnapshot;
|