@volter/twin-togetherai 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +147 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/index.d.ts +14 -0
  6. package/dist/src/index.js +79 -0
  7. package/dist/src/togetherai-budget.d.ts +52 -0
  8. package/dist/src/togetherai-budget.js +130 -0
  9. package/dist/src/togetherai-capabilities.d.ts +4 -0
  10. package/dist/src/togetherai-capabilities.js +1428 -0
  11. package/dist/src/togetherai-conformance.d.ts +14 -0
  12. package/dist/src/togetherai-conformance.js +452 -0
  13. package/dist/src/togetherai-connector.d.ts +164 -0
  14. package/dist/src/togetherai-connector.js +457 -0
  15. package/dist/src/togetherai-models.d.ts +19 -0
  16. package/dist/src/togetherai-models.js +49 -0
  17. package/dist/src/togetherai-scenario.d.ts +52 -0
  18. package/dist/src/togetherai-scenario.js +168 -0
  19. package/dist/src/togetherai-server.d.ts +16 -0
  20. package/dist/src/togetherai-server.js +187 -0
  21. package/dist/src/togetherai-stub.d.ts +59 -0
  22. package/dist/src/togetherai-stub.js +195 -0
  23. package/dist/src/togetherai-twin.d.ts +83 -0
  24. package/dist/src/togetherai-twin.js +1419 -0
  25. package/dist/src/togetherai-types.d.ts +207 -0
  26. package/dist/src/togetherai-types.js +26 -0
  27. package/package.json +52 -0
  28. package/src/cli.ts +27 -0
  29. package/src/index.ts +118 -0
  30. package/src/togetherai-budget.ts +156 -0
  31. package/src/togetherai-capabilities.ts +1315 -0
  32. package/src/togetherai-conformance.ts +459 -0
  33. package/src/togetherai-connector.ts +496 -0
  34. package/src/togetherai-models.ts +74 -0
  35. package/src/togetherai-scenario.ts +185 -0
  36. package/src/togetherai-server.ts +199 -0
  37. package/src/togetherai-stub.ts +197 -0
  38. package/src/togetherai-twin.ts +1448 -0
  39. package/src/togetherai-types.ts +222 -0
@@ -0,0 +1,207 @@
1
+ /** A chat message param as the caller sends it. Together's `ChatCompletionMessageParam` oneOf
2
+ * covers system/user/assistant/tool/function; content is a string (Together's schema types
3
+ * every param's content as `string | null` — no content-part arrays). */
4
+ export type TogetheraiMessageParam = {
5
+ role: 'system' | 'user' | 'assistant' | 'tool' | 'function';
6
+ content?: string | null;
7
+ name?: string;
8
+ tool_calls?: TogetheraiToolCall[];
9
+ tool_call_id?: string;
10
+ reasoning?: string | null;
11
+ reasoning_content?: string | null;
12
+ };
13
+ /** A function tool_call inside an assistant message (Together's `ToolChoice`). */
14
+ export type TogetheraiToolCall = {
15
+ id: string;
16
+ type: 'function';
17
+ function: {
18
+ name: string;
19
+ arguments: string;
20
+ };
21
+ };
22
+ /** Together's `UsageData`: the three token counts, REQUIRED and NULLABLE (Together reports
23
+ * `null` usage on some streamed/failed turns). No timing fields — that is Groq's block. */
24
+ export type TogetheraiUsage = {
25
+ prompt_tokens: number | null;
26
+ completion_tokens: number | null;
27
+ total_tokens: number | null;
28
+ /** Reasoning models nest the detail blocks (Together's docs: cached/reasoning token splits). */
29
+ prompt_tokens_details?: {
30
+ cached_tokens: number;
31
+ } | null;
32
+ completion_tokens_details?: {
33
+ reasoning_tokens: number;
34
+ } | null;
35
+ };
36
+ /** Together's assistant message. `reasoning`/`reasoning_content` vary by model (Together's docs:
37
+ * "varies by model" — both are declared on the schema). No `refusal` field. */
38
+ export type TogetheraiAssistantMessage = {
39
+ role: 'assistant';
40
+ content: string | null;
41
+ tool_calls?: TogetheraiToolCall[];
42
+ function_call?: {
43
+ name: string;
44
+ arguments: string;
45
+ };
46
+ reasoning?: string | null;
47
+ reasoning_content?: string | null;
48
+ };
49
+ /** Together's `FinishReason`: adds **`eos`** to the OpenAI set. */
50
+ export type TogetheraiFinishReason = 'stop' | 'eos' | 'length' | 'tool_calls' | 'function_call';
51
+ export type TogetheraiChoice = {
52
+ index: number;
53
+ message?: TogetheraiAssistantMessage;
54
+ text?: string;
55
+ seed?: number;
56
+ logprobs?: unknown | null;
57
+ finish_reason: TogetheraiFinishReason | null;
58
+ };
59
+ /** The unary chat.completion response. `prompt` is REQUIRED on Together's schema — the echoed
60
+ * prompt parts (empty unless `echo:true`); `warnings` is Together's inference-warnings array. */
61
+ export type TogetheraiChatCompletion = {
62
+ id: string;
63
+ object: 'chat.completion';
64
+ created: number;
65
+ model: string;
66
+ choices: TogetheraiChoice[];
67
+ prompt: Array<{
68
+ text?: string;
69
+ logprobs?: unknown;
70
+ }>;
71
+ usage?: TogetheraiUsage | null;
72
+ warnings?: unknown[];
73
+ };
74
+ /** A single Server-Sent Event the streaming path emits (collected, never socketed in tests).
75
+ * `data` is the JSON payload; `[DONE]` is signalled with `done: true` (no data object). */
76
+ export type SseEvent = {
77
+ data?: Record<string, unknown>;
78
+ done?: boolean;
79
+ };
80
+ /** A sink the streaming path writes events into (an injected collector in tests / a real
81
+ * HTTP SSE writer in the server). NO real sockets or setTimeout in the handler. */
82
+ export type SseSink = (event: SseEvent) => void;
83
+ /** A streaming chunk. Together's `ChatCompletionChunk` carries `usage` (nullable) and
84
+ * `warnings` on EVERY chunk (OpenAI puts usage only in an opt-in tail chunk). */
85
+ export type TogetheraiChunk = {
86
+ id: string;
87
+ object: 'chat.completion.chunk';
88
+ created: number;
89
+ model: string;
90
+ system_fingerprint?: string;
91
+ choices: Array<{
92
+ index: number;
93
+ delta: Record<string, unknown>;
94
+ logprobs?: unknown | null;
95
+ finish_reason: TogetheraiFinishReason | null;
96
+ }>;
97
+ usage?: TogetheraiUsage | null;
98
+ warnings?: unknown[];
99
+ };
100
+ /** Together's `ModelInfo`: REQUIRED keys are id/object/created/type — `type` is Together's OWN
101
+ * enum (chat|language|code|image|embedding|moderation|rerank), not OpenAI's flat model object. */
102
+ export type TogetheraiModel = {
103
+ id: string;
104
+ object: 'model';
105
+ created: number;
106
+ type: 'chat' | 'language' | 'code' | 'image' | 'embedding' | 'moderation' | 'rerank';
107
+ display_name?: string;
108
+ organization?: string;
109
+ link?: string;
110
+ license?: string;
111
+ context_length?: number;
112
+ pricing?: Record<string, unknown>;
113
+ };
114
+ export type TogetheraiEmbedding = {
115
+ object: 'embedding';
116
+ index: number;
117
+ embedding: number[];
118
+ };
119
+ /** Together's `EmbeddingsResponse`: REQUIRED object/model/data — and NO `usage` key on the
120
+ * schema (Together does not document one on this endpoint). */
121
+ export type TogetheraiEmbeddingResponse = {
122
+ object: 'list';
123
+ model: string;
124
+ data: TogetheraiEmbedding[];
125
+ };
126
+ /** Together's `RerankResponse` — Together-native (NOT an OpenAI endpoint at all). */
127
+ export type TogetheraiRerankResponse = {
128
+ object: 'rerank';
129
+ id?: string;
130
+ model: string;
131
+ results: Array<{
132
+ index: number;
133
+ relevance_score: number;
134
+ document?: {
135
+ text?: string | null;
136
+ };
137
+ }>;
138
+ usage?: TogetheraiUsage | null;
139
+ };
140
+ /** Together's `ImageResponse`: `object:'list'` with a discriminated data union
141
+ * (`type:'url'` | `type:'b64_json'`). */
142
+ export type TogetheraiImageResponse = {
143
+ id: string;
144
+ model: string;
145
+ object: 'list';
146
+ data: Array<{
147
+ index: number;
148
+ url?: string;
149
+ b64_json?: string;
150
+ type: 'url' | 'b64_json';
151
+ }>;
152
+ };
153
+ /** Together's `FilePurpose` — a CLOSED set (`fine-tune` | `eval` | `batch-api`), NOT OpenAI's. */
154
+ export type TogetheraiFilePurpose = 'fine-tune' | 'eval' | 'batch-api';
155
+ /** Together's `FileType` — csv|jsonl|parquet. */
156
+ export type TogetheraiFileType = 'csv' | 'jsonl' | 'parquet';
157
+ /** The validation pipeline's lifecycle state (fine-tune files only). */
158
+ export type TogetheraiFileProcessingStatus = 'PENDING' | 'QUEUED' | 'RUNNING' | 'COMPLETED' | 'FAILED' | 'INVALID_FORMAT';
159
+ /** Together's `FileResponse`. REQUIRED keys include Together's own `FileType` (capital F) and
160
+ * the deprecated `Processed` boolean. */
161
+ export type TogetheraiFile = {
162
+ id: string;
163
+ object: 'file';
164
+ created_at: number;
165
+ filename: string;
166
+ bytes: number;
167
+ purpose: TogetheraiFilePurpose;
168
+ Processed: boolean;
169
+ FileType: TogetheraiFileType;
170
+ processing_status?: TogetheraiFileProcessingStatus;
171
+ validation_report?: {
172
+ valid: boolean;
173
+ [k: string]: unknown;
174
+ };
175
+ };
176
+ /** Together's `BatchJobStatus` — UPPER-CASE, not OpenAI's lowercase set. */
177
+ export type TogetheraiBatchStatus = 'VALIDATING' | 'IN_PROGRESS' | 'COMPLETED' | 'FAILED' | 'EXPIRED' | 'CANCELLED';
178
+ /** Together's `CreateBatchRequest.endpoint` — the three documented endpoints. */
179
+ export type TogetheraiBatchEndpoint = '/v1/chat/completions' | '/v1/audio/transcriptions' | '/v1/audio/translations';
180
+ /** Together's `BatchJob`. `error` is a bare STRING (not an ErrorData object), and the
181
+ * timestamps are ISO date-time STRINGS (not OpenAI's unix integers). */
182
+ export type TogetheraiBatch = {
183
+ id: string;
184
+ user_id?: string;
185
+ input_file_id: string;
186
+ file_size_bytes?: number;
187
+ status: TogetheraiBatchStatus;
188
+ job_deadline?: string | null;
189
+ created_at: string;
190
+ endpoint: string;
191
+ progress?: number;
192
+ model_id?: string;
193
+ output_file_id?: string | null;
194
+ error_file_id?: string | null;
195
+ error?: string | null;
196
+ completed_at?: string | null;
197
+ };
198
+ /** Together's `ErrorData`: `{ error: { message, type, param, code } }` with `message`+`type`
199
+ * REQUIRED and `param`/`code` nullable with default null. */
200
+ export type TogetheraiError = {
201
+ error: {
202
+ message: string;
203
+ type: string;
204
+ param?: string | null;
205
+ code?: string | null;
206
+ };
207
+ };
@@ -0,0 +1,26 @@
1
+ // Shared wire-shape types for the Together AI surface. These mirror the REAL vendor JSON shapes
2
+ // as published by Together's own OpenAPI (https://docs.together.ai/openapi.yaml, fetched
3
+ // 2026-09-16) and the Stainless-generated `together-ai@0.53.0` SDK — NOT the SDK's internal
4
+ // types (the twin never imports the SDK at runtime; the SDK is exercised only in *.test.ts).
5
+ //
6
+ // TOGETHER IS NOT OPENAI, and the differences below are load-bearing (they are what a
7
+ // "just copy the openai pack" twin gets wrong):
8
+ // • `n` spans 1–128 (OpenAI-compatible; Groq pins it at 1) and `logprobs` is an INTEGER 0–20
9
+ // (OpenAI's boolean) — Together serves both;
10
+ // • `finish_reason` adds **`eos`** (Together's end-of-sequence token) alongside stop/length;
11
+ // • the response carries a REQUIRED `prompt: []` array (the echoed prompt when `echo:true`);
12
+ // • the error envelope is `{ error: { message, type, param, code } }` with `param`/`code`
13
+ // nullable, and Together's STATUS TABLE is its own: 402 = monthly spending limit,
14
+ // **403 = context length exceeded** (a BAD REQUEST, not a permission denial),
15
+ // 503 = engine overloaded, 524 = Cloudflare timeout, 529 = server error;
16
+ // • the batch surface is Together-NATIVE: `GET /v1/batches` answers a bare ARRAY (not an
17
+ // OpenAI `{object:'list',data}` envelope), `POST /v1/batches` answers 201 with
18
+ // `BatchJobWithWarning { job, warning? }`, and every batch error is `{ error: string }`
19
+ // (a bare string, NOT an ErrorData object);
20
+ // • files carry Together's own `FileType` (csv|jsonl|parquet) and the validation pipeline's
21
+ // `processing_status` / `validation_report`;
22
+ // • model ids are slash-namespaced (`meta-llama/Llama-3.3-70B-Instruct-Turbo`) and an
23
+ // OpenAI-style flat id like `gpt-4o` is a 404;
24
+ // • `service_tier` / `store` / `metadata` / `prediction` are ACCEPTED BUT IGNORED (the
25
+ // OpenAI-compat page's own wording) — accepted here, never echoed into the response.
26
+ export {};
package/package.json ADDED
@@ -0,0 +1,52 @@
1
+ {
2
+ "name": "@volter/twin-togetherai",
3
+ "version": "0.1.0",
4
+ "description": "Local Together AI twin — a faithful, stateful local Together API your real `together-ai` SDK talks to unmodified. The model is stubbed (deterministic) and embeddings/rerank/images/transcripts are deterministic stubs, but the protocol envelope (chat completions/streaming/tool_calls, Together's prompt array + eos finish reasons + 402/403/503 status table, models, files, batches, fine-tunes, audio) is vendor-faithful. Built on @volter/world-core.",
5
+ "author": "Volter (https://github.com/volter-ai)",
6
+ "license": "Apache-2.0",
7
+ "files": [
8
+ "src",
9
+ "README.md",
10
+ "LICENSE",
11
+ "!**/*.test.ts",
12
+ "!**/*.test.tsx",
13
+ "dist"
14
+ ],
15
+ "repository": {
16
+ "type": "git",
17
+ "url": "git+https://github.com/volter-ai/twin.git",
18
+ "directory": "packages/twin/togetherai"
19
+ },
20
+ "homepage": "https://github.com/volter-ai/twin/tree/main/packages/twin/togetherai#readme",
21
+ "type": "module",
22
+ "exports": {
23
+ ".": {
24
+ "types": "./dist/src/index.d.ts",
25
+ "default": "./dist/src/index.js"
26
+ }
27
+ },
28
+ "bin": {
29
+ "world-togetherai": "dist/src/cli.js"
30
+ },
31
+ "scripts": {
32
+ "test": "bun test src/*.test.ts",
33
+ "typecheck": "tsc --noEmit",
34
+ "build": "node ../../../scripts/publish/build.mjs",
35
+ "prepack": "node ../../../scripts/publish/prepare-publish.mjs prepack",
36
+ "postpack": "node ../../../scripts/publish/prepare-publish.mjs postpack"
37
+ },
38
+ "peerDependencies": {
39
+ "@volter/world-core": "2.0.0"
40
+ },
41
+ "devDependencies": {
42
+ "@types/bun": "^1.2.20",
43
+ "@types/node": "^24.0.0",
44
+ "@volter/world-core": "2.0.0",
45
+ "@volter/world-tooling": "0.1.0",
46
+ "together-ai": "0.53.0",
47
+ "typescript": "^5.9.0"
48
+ },
49
+ "engines": {
50
+ "node": ">=22.3"
51
+ }
52
+ }
package/src/cli.ts ADDED
@@ -0,0 +1,27 @@
1
+ #!/usr/bin/env node
2
+ import { keepProcessAlive } from '@volter/world-core/lifecycle';
3
+ // world-togetherai CLI: serve the Together API twin or run conformance. Together is an API-first
4
+ // vendor — app.together.ai is a console, not where the work happens
5
+ // (docs/contributing/architecture.md C1b) — so this pack ships no mirror.
6
+ import { hasFlag, optionValue } from '@volter/world-core/args';
7
+ import { createTogetheraiTwinServer } from './togetherai-server.ts';
8
+
9
+ const [cmd, ...rest] = process.argv.slice(2);
10
+ const port = Number(optionValue(rest, '--port', '0')) || undefined;
11
+ const root = optionValue(rest, '--root') || undefined;
12
+ const readOnly = hasFlag(rest, '--read-only'); // a twin accepts writes unless started read-only
13
+ const scenario = optionValue(rest, '--scenario') || undefined; // scripted completions (JSON file)
14
+
15
+ if (cmd === 'serve') {
16
+ const s = await createTogetheraiTwinServer({ readOnly, ...(root ? { root } : {}), ...(port ? { port } : {}), ...(scenario ? { scenarioPath: scenario } : {}) });
17
+ process.stdout.write(`togetherai twin (Together API at /v1; model output is a deterministic stub)${readOnly ? ' [read-only]' : ''}${scenario ? ` [scenario: ${scenario}]` : ''} at http://127.0.0.1:${s.port}\n`);
18
+ await keepProcessAlive();
19
+ } else if (cmd === 'conformance') {
20
+ // dev-only; lazy so the bin runs without @volter/world-tooling
21
+ const { checkTogetheraiConformance } = await import('./togetherai-conformance.ts');
22
+ const report = await checkTogetheraiConformance({ ...(root ? { root } : {}) });
23
+ process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
24
+ if (!report.ok) process.exitCode = 1;
25
+ } else {
26
+ process.stdout.write('Usage: world-togetherai serve|conformance [--port N] [--root DIR] [--read-only] [--scenario FILE]\n');
27
+ }
package/src/index.ts ADDED
@@ -0,0 +1,118 @@
1
+ // @volter/twin-togetherai — the Together AI twin (one vendor, one package), built on the shared
2
+ // @volter/world-core kernel. The Together API: a vendor-faithful protocol envelope (chat
3
+ // completions/streaming/tool_calls with Together's REQUIRED `prompt` array and per-chunk
4
+ // usage/warnings, `eos` finish reasons, Together's OWN status table — 402 spending limit, 403
5
+ // context length, 503 engine overloaded), a static models catalog, deterministic embeddings/
6
+ // rerank/image/audio stubs, and STATEFUL files + batches + fine-tunes — over an event/action log.
7
+ // (Together is an API-first vendor: app.together.ai is a console, not where the work happens, so
8
+ // this pack ships no mirror.)
9
+ //
10
+ // THE DETERMINISTIC STUB IS THE ANSWER: the twin runs no model, so POST /v1/chat/completions
11
+ // returns a DETERMINISTIC STUB completion (clearly labeled), /v1/embeddings returns DETERMINISTIC
12
+ // pseudo-vectors, and the audio endpoints return DETERMINISTIC labeled stubs — never pretending to
13
+ // be real inference. Everything around them — the wire protocol — is faithful. (Conformance +
14
+ // capability tooling live in @volter/world-tooling, a dev dependency — NOT re-exported here, per
15
+ // E2.)
16
+ export { handleTogetheraiTwinRequest, streamChat, buildChatCompletion, TOGETHERAI_API_PREFIX, TOGETHERAI_UPLOAD_DOOR, handleTogetheraiUploadDoor } from './togetherai-twin.ts';
17
+ export type { TogetheraiRequest, TogetheraiResponseEnvelope } from './togetherai-twin.ts';
18
+ export { createTogetheraiTwinFetch, createTogetheraiTwinServer, type TogetheraiTwinFetchOptions } from './togetherai-server.ts';
19
+ export { TOGETHERAI_MODELS, findModel, SPEECH_MODELS, EMBEDDING_MODELS, EMBEDDING_DIMENSIONS, RERANK_MODELS } from './togetherai-models.ts';
20
+ export {
21
+ buildUsage, contentToText, countPromptTokens, estimateTokens,
22
+ fnv1a, lastUserText, pseudoEmbedding, stubAssistantText, stubAudioSeconds, stubJsonObject,
23
+ stubReasoningText, stubToolArguments, stubToolCall, stubTranscript,
24
+ } from './togetherai-stub.ts';
25
+ export type {
26
+ TogetheraiAssistantMessage, TogetheraiBatch, TogetheraiBatchEndpoint, TogetheraiBatchStatus,
27
+ TogetheraiChatCompletion, TogetheraiChoice, TogetheraiEmbedding, TogetheraiEmbeddingResponse,
28
+ TogetheraiError, TogetheraiFile, TogetheraiFilePurpose, TogetheraiFileType,
29
+ TogetheraiFinishReason, TogetheraiMessageParam, TogetheraiModel, TogetheraiRerankResponse,
30
+ TogetheraiToolCall, TogetheraiUsage, SseEvent, SseSink,
31
+ } from './togetherai-types.ts';
32
+ export { createTogetheraiScenarioEngine, togetheraiScenarioAdapter, loadTogetheraiScenarioDocument, realizeTogetheraiRespond } from './togetherai-scenario.ts';
33
+ export type { TogetheraiScenarioEngine, TogetheraiScenarioRequest, TogetheraiScenarioRespond, ScenarioToolCall, ScriptedResult } from './togetherai-scenario.ts';
34
+ export {
35
+ fullSyncTogetherai,
36
+ externalIdFor,
37
+ togetheraiRequestForAction,
38
+ liveTogetheraiExecute,
39
+ mapBatch,
40
+ mapFile,
41
+ mapModel,
42
+ pullTogetheraiState,
43
+ pushTogetheraiAction,
44
+ pushPendingTogetheraiActions,
45
+ syncTogetheraiFromReal,
46
+ unpushableReason,
47
+ } from './togetherai-connector.ts';
48
+ export type { TogetheraiExecute, LiveTogetheraiOptions } from './togetherai-connector.ts';
49
+ // The client-side rate budget — the fail-closed backstop `liveTogetheraiExecute` routes every live
50
+ // request through. The MECHANISM is the kernel's shared, vendor-agnostic `RateBudget`; what lives
51
+ // here is Together's DECLARATION (window/ceiling/per-endpoint weights) plus the vendor-bound
52
+ // bindings. Exported so an operator can inspect spend (`snapshot`) and so a caller can catch
53
+ // `TogetheraiBudgetError` by type; there is deliberately no export that disables the guard.
54
+ export {
55
+ TOGETHERAI_BUDGET_CEILING,
56
+ TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S,
57
+ TOGETHERAI_BUDGET_WINDOW_MS,
58
+ TOGETHERAI_CALL_WEIGHTS,
59
+ TOGETHERAI_RATE_BUDGET,
60
+ TogetheraiBudget,
61
+ TogetheraiBudgetError,
62
+ togetheraiBudgetPath,
63
+ togetheraiCallWeight,
64
+ } from './togetherai-budget.ts';
65
+ export type { TogetheraiBudgetErrorKind, TogetheraiBudgetOptions, TogetheraiBudgetReservation, TogetheraiBudgetSnapshot } from './togetherai-budget.ts';
66
+
67
+ // Registry descriptor: the pack self-describes so tooling can discover it.
68
+ import { registerPack, type TwinPack } from '@volter/world-core';
69
+ import { TOGETHERAI_RATE_BUDGET as RATE_BUDGET } from './togetherai-budget.ts';
70
+ import { performTogetheraiAction, syncTogetheraiFromRemote } from './togetherai-connector.ts';
71
+ export const pack: TwinPack = {
72
+ vendor: 'togetherai',
73
+ // PROTOCOL 2 (docs/contributing/architecture.md#protocol-2-the-pack-is-a-plugin): the pack is a plugin — its wire, its tree, and its half of
74
+ // the real state system: perform one entry against Together, refresh the root from it.
75
+ protocol: '2',
76
+ stateSystem: { perform: performTogetheraiAction, refresh: syncTogetheraiFromRemote },
77
+ // the round trip: a file create — the one thing this API mints an id for and keeps, a fresh id
78
+ // each time, so it repeats cleanly on a branch (the spec's own create door, POST
79
+ // /v1/files/upload; Together's purposes are a CLOSED set: fine-tune | eval | batch-api)
80
+ roundTrip: { method: 'POST', path: '/v1/files/upload', body: { purpose: 'batch-api', file_name: 'round-trip.jsonl', content: '' }, headers: { authorization: 'Bearer rt' } },
81
+ // The SAME object togetherai-budget.ts declares at module load — one source of truth, so
82
+ // registering the pack and importing the connector can never arm two different ceilings.
83
+ rateBudget: RATE_BUDGET,
84
+ transport: 'rest',
85
+ archetype: 'generative',
86
+ bin: 'world-togetherai',
87
+ // R2 adopted-as-debt, legibly: declared types no replay can create, each with its reason.
88
+ resourcesUnreachable: {
89
+ 'model': 'the vendor catalog',
90
+ },
91
+ resources: ['file', 'batch', 'finetune', 'model'],
92
+ specSource: 'Together API — enumerated from Together\'s own published OpenAPI spec (api.together.xyz/openapi.json, captured to test-fixtures/togetherai-openapi.yaml: 141 paths / 156 operations — the /v1 inference half AND the v2 management half, of which this pack models the inference half) + together-ai@0.53.0 (Stainless-generated from Together\'s spec: the model unions, the SDK-only 302-redirect upload flow, whoami, the TOGETHER_BASE_URL/TOGETHER_API_KEY/TOGETHER_PROJECT_ID env contract) + docs.together.ai/{error-codes,serverless/rate-limits,inference/openai-compatibility} (all read 2026-09-16). Envelope-faithful; model output, embedding values, rerank scores, images and audio are labeled deterministic stubs.',
93
+ description: 'Together AI API twin — faithful protocol envelope (chat completions/streaming/tool_calls, Together\'s prompt array and eos finish reasons, Together\'s own 402/403/503 status table), models, stateful files/batches/fine-tunes (both upload flows, incl. the SDK\'s 302 redirect), embeddings/rerank/images/audio; generative output is a labeled stub.',
94
+ // ADOPTION (adding-a-twin.md §3): how an app repo betrays that it talks to Together. Declared
95
+ // HERE, not in world-runtime's central SDK_TWINS/ENV_STEM_VENDORS maps — declaring in both
96
+ // throws. `together-ai` is Together's own official client (npm).
97
+ adoption: {
98
+ sdks: ['together-ai'], envStems: ['TOGETHER'],
99
+ },
100
+ // INTERCEPTION: the host the official client addresses by default
101
+ // (together-ai@0.53.0 client.js: TOGETHER_BASE_URL defaults to https://api.together.ai/v1).
102
+ hosts: [{ host: 'api.together.ai' }],
103
+ // App-read endpoint templates RIDE ALONGSIDE the injector (the PLANETSCALE_DATABASE_URL
104
+ // precedent, init.ts wiringFor): `together-ai`'s custom `files.upload()` reads
105
+ // `TOGETHER_API_BASE_URL` at MODULE LOAD and IGNORES the client's `baseURL`
106
+ // (lib/upload.js:12 — it then POSTs `${baseURL}/files?…` itself), so interception of
107
+ // api.together.ai does NOT cover it: unset, `files.upload()` egresses to the real
108
+ // https://api.together.xyz with the caller's key. The template points that var at the twin's
109
+ // `/v1` (the SDK POSTs `/files?…` onto it directly), so the one flow the injector cannot reach
110
+ // is wired too, while ordinary SDK traffic still rides the injector.
111
+ endpointEnv: {
112
+ name: 'TOGETHER_TWIN_URL',
113
+ templates: { TOGETHER_API_BASE_URL: '${url}/v1' },
114
+ note: 'the injector covers api.together.ai, but together-ai\'s files.upload() reads TOGETHER_API_BASE_URL at module load and ignores client.baseURL (lib/upload.js:12) — the template points it at the twin\'s /v1 so the SDK upload flow does not egress to the real vendor.',
115
+ },
116
+ };
117
+ // registered at import: the kernel learns the pack's state system (protocol 2)
118
+ registerPack(pack);
@@ -0,0 +1,156 @@
1
+ // Together AI's CLIENT-SIDE RATE BUDGET — the pack's DECLARATION (the numbers) plus the thin
2
+ // typed bindings `liveTogetheraiExecute` uses. The MECHANISM — the durable token-keyed ledger,
3
+ // the rolling window, reserve-under-lock, the `Retry-After`/429 cooldown, fail-CLOSED on a
4
+ // corrupt ledger — lives ONCE in the vendor-agnostic kernel (`@volter/world-core` →
5
+ // `rateBudget.ts`). Read that module's header for the full rationale AND for the honest list of
6
+ // what the guard does not guarantee.
7
+ //
8
+ // ── WHY THIS EXISTS ─────────────────────────────────────────────────────────────────────────
9
+ // A real ~4.5-DAY vendor lockout (Figma, 2026-07-25) happened because raw API calls were made
10
+ // outside the pack's connector — no cache, no batching, no ceiling. Discipline only binds the code
11
+ // that follows it; a BUDGET binds the code that does not.
12
+ //
13
+ // ── HOW THE CEILING WAS CHOSEN (live-read, 2026-09-16) ──────────────────────────────────────
14
+ // https://docs.together.ai/serverless/rate-limits (read 2026-09-16) publishes NO scalar a client
15
+ // could be bound by: limits are "dynamic" and PER ORGANIZATION and PER MODEL, and the page's own
16
+ // wording is "there are no fixed per-model limits published" — the account's dashboard is the only
17
+ // place the real numbers live. Success responses carry NO rate-limit headers; a 429 carries only
18
+ // `x-ratelimit-reset` (seconds until the window resets), and its error types are
19
+ // `dynamic_request_limited` / `dynamic_token_limited`; a 503 means the dynamic rate is at/below
20
+ // capacity.
21
+ //
22
+ // So this declaration does not model Together's limit, and nothing here is more permissive than
23
+ // the kernel's undeclared fallback: window 60s, ceiling 60, defaultWeight 2 — i.e. 30
24
+ // calls/minute, EXACTLY `DEFAULT_RATE_BUDGET`, with no endpoint priced cheaper than the fallback
25
+ // would price it. Being stricter than the fallback needs no vendor justification; being looser
26
+ // would, and there is none to have.
27
+ //
28
+ // It bounds the 60s AVERAGE; it does not pace (the kernel refuses, it never sleeps). The backstop
29
+ // for a sub-second burst is the cooldown: Together's `x-ratelimit-reset` (documented on the 429)
30
+ // is read off the response and turns into a persisted refusal.
31
+ import {
32
+ declareRateBudget,
33
+ rateBudgetPath,
34
+ rateBudgetWeight,
35
+ RateBudget,
36
+ type RateBudgetDeclaration,
37
+ type RateBudgetOptions,
38
+ type RateBudgetReservation,
39
+ type RateBudgetSnapshot,
40
+ } from '@volter/world-core';
41
+
42
+ const VENDOR = 'togetherai';
43
+
44
+ /** Rolling window, in ms. Spend older than this is pruned. */
45
+ export const TOGETHERAI_BUDGET_WINDOW_MS = 60_000;
46
+
47
+ /**
48
+ * Weighted units allowed inside one window. 60/60s at `defaultWeight` 2 = 30 calls a minute —
49
+ * EXACTLY the kernel's undeclared fallback, because Together publishes no scalar that would
50
+ * justify more. See the header.
51
+ */
52
+ export const TOGETHERAI_BUDGET_CEILING = 60;
53
+
54
+ /** Seconds. A `x-ratelimit-reset` above this means the key is throttled hard — fail loudly, don't sleep. */
55
+ export const TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S = 300;
56
+
57
+ /** Per-call cost, keyed by `"<METHOD> <path>"`. See the header for what is documented vs. judged. */
58
+ export const TOGETHERAI_CALL_WEIGHTS = {
59
+ /** `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/rerank`,
60
+ * `/v1/images/generations`, `/v1/audio/*` — token-metered, where Together's DYNAMIC token
61
+ * budget (`dynamic_token_limited`) rather than its request budget is usually what binds. */
62
+ inference: 6,
63
+ /** Everything else: files, batches, models, whoami, fine-tunes, endpoints. */
64
+ other: 2,
65
+ } as const;
66
+
67
+ /** THE PACK'S DECLARATION — pure data, the only Together-specific thing in the whole budget. */
68
+ export const TOGETHERAI_RATE_BUDGET: RateBudgetDeclaration = {
69
+ windowMs: TOGETHERAI_BUDGET_WINDOW_MS,
70
+ ceiling: TOGETHERAI_BUDGET_CEILING,
71
+ defaultWeight: TOGETHERAI_CALL_WEIGHTS.other,
72
+ maxRetryAfterSeconds: TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S,
73
+ rules: [
74
+ { match: '^POST /v1/(chat/completions|completions|embeddings|rerank|images/generations)$', weight: TOGETHERAI_CALL_WEIGHTS.inference },
75
+ { match: '^POST /v1/audio/', weight: TOGETHERAI_CALL_WEIGHTS.inference },
76
+ ],
77
+ reason:
78
+ 'Together publishes NO fixed request limit (docs.together.ai/serverless/rate-limits, read ' +
79
+ '2026-09-16): limits are DYNAMIC and per organization per model — the page states "there are ' +
80
+ 'no fixed per-model limits published", the real numbers live only on the account dashboard, ' +
81
+ 'and success responses carry no rate-limit headers at all. A 429 carries only ' +
82
+ '`x-ratelimit-reset` (seconds) with error types `dynamic_request_limited` / ' +
83
+ '`dynamic_token_limited`; a 503 signals the dynamic rate is at/below capacity. Because no ' +
84
+ 'published figure justifies going higher, the ceiling is pinned at the kernel fallback in ' +
85
+ 'EVERY dimension — 60 units / 60s at defaultWeight 2 = 30 calls/min — and the declaration ' +
86
+ 'buys resolution DOWNWARD, never headroom: the inference endpoints cost 6, so at most 10 land ' +
87
+ 'in a window. The window bounds the 60s AVERAGE and does not pace; the `x-ratelimit-reset` ' +
88
+ 'cooldown is the backstop for a sub-second burst.',
89
+ };
90
+
91
+ // Declared at module load, so merely importing this module (which `togetherai-connector.ts`
92
+ // does) is enough to arm the real ceiling.
93
+ declareRateBudget(VENDOR, TOGETHERAI_RATE_BUDGET);
94
+
95
+ /**
96
+ * Price one call. The key is `"<METHOD> <path>"` with the query string split off, so a rule can
97
+ * price by method (a write is not a read) without the kernel knowing anything about Together. An
98
+ * unclassified endpoint still costs `defaultWeight` — nothing is ever free.
99
+ */
100
+ export function togetheraiCallWeight(method: string, path: string): number {
101
+ const { bare, query } = splitQuery(path);
102
+ // UPPER-CASE the method: `fetch` normalizes a known lowercase method before sending, so
103
+ // `execute('post', …)` really does issue a POST and must be priced as one.
104
+ return rateBudgetWeight(VENDOR, `${String(method).toUpperCase()} ${bare}`, query);
105
+ }
106
+
107
+ /**
108
+ * `/v1/x?a=1` -> `{ bare: '/v1/x', query: { a: '1' } }`. Rules match the path; NORMALIZED,
109
+ * because the anchored rules are otherwise trivially evaded: `fetch` upper-cases a known method
110
+ * before sending, so `execute('post', …)` issues a real WRITE that a `^POST ` rule would price as
111
+ * a read; and a trailing slash makes a path miss a `$` anchor while most routers treat it as the
112
+ * same endpoint.
113
+ */
114
+ function splitQuery(path: string): { bare: string; query: Record<string, string> } {
115
+ const at = path.indexOf('?');
116
+ const query: Record<string, string> = {};
117
+ if (at !== -1) for (const [k, v] of new URLSearchParams(path.slice(at + 1))) query[k] = v;
118
+ // Collapse REPEATED slashes as well as a trailing one: `/v1//chat/completions` reaches the same
119
+ // endpoint on most routers but misses a `^POST /v1/(chat/completions|…)$` rule, which would
120
+ // price an inference call as a 2-unit read (the groq pack's §9 round one, NIT 14).
121
+ const raw = (at === -1 ? path : path.slice(0, at)).replace(/\/{2,}/g, '/');
122
+ const bare = raw.length > 1 && raw.endsWith('/') ? raw.replace(/\/+$/, '') : raw;
123
+ return { bare, query };
124
+ }
125
+
126
+ /** Where Together's ledger lives. Token-keyed and cwd-independent by default (Together's limits
127
+ * are per ORGANIZATION, i.e. per key, so a cwd-scoped ledger would hand the same key a fresh
128
+ * allowance in every checkout, worktree and CI matrix leg); pass `root` for world-scoped
129
+ * accounting. */
130
+ export function togetheraiBudgetPath(opts: { root?: string; token?: string } | string = {}): string {
131
+ const o = typeof opts === 'string' ? { root: opts } : opts;
132
+ // VENDOR spread LAST: a loosely-typed `{ vendor: 'other', … }` slipping through (TypeScript's
133
+ // excess-property check only catches object literals) must not redirect this pack's ledger.
134
+ return rateBudgetPath({ ...o, vendor: VENDOR });
135
+ }
136
+
137
+ /** Construction options for Together's budget. The vendor is fixed; everything else may only TIGHTEN. */
138
+ export type TogetheraiBudgetOptions = Omit<RateBudgetOptions, 'vendor'>;
139
+
140
+ /**
141
+ * Together's budget — the shared kernel guard bound to this vendor's declaration. A real
142
+ * subclass, not an alias, so `budget instanceof TogetheraiBudget` in `liveTogetheraiExecute`
143
+ * means "a budget that accounts against TOGETHER's ledger under TOGETHER's ceiling".
144
+ */
145
+ export class TogetheraiBudget extends RateBudget {
146
+ constructor(opts: TogetheraiBudgetOptions = {}) {
147
+ super({ ...opts, vendor: VENDOR });
148
+ }
149
+ }
150
+
151
+ /** The typed refusal. One error class shared with every other vendor's budget; `err.vendor` says
152
+ * which one refused, and `err.kind` says why. */
153
+ export { RateBudgetError as TogetheraiBudgetError } from '@volter/world-core';
154
+ export type { RateBudgetErrorKind as TogetheraiBudgetErrorKind } from '@volter/world-core';
155
+ export type TogetheraiBudgetReservation = RateBudgetReservation;
156
+ export type TogetheraiBudgetSnapshot = RateBudgetSnapshot;