@volter/twin-togetherai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +147 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/index.d.ts +14 -0
- package/dist/src/index.js +79 -0
- package/dist/src/togetherai-budget.d.ts +52 -0
- package/dist/src/togetherai-budget.js +130 -0
- package/dist/src/togetherai-capabilities.d.ts +4 -0
- package/dist/src/togetherai-capabilities.js +1428 -0
- package/dist/src/togetherai-conformance.d.ts +14 -0
- package/dist/src/togetherai-conformance.js +452 -0
- package/dist/src/togetherai-connector.d.ts +164 -0
- package/dist/src/togetherai-connector.js +457 -0
- package/dist/src/togetherai-models.d.ts +19 -0
- package/dist/src/togetherai-models.js +49 -0
- package/dist/src/togetherai-scenario.d.ts +52 -0
- package/dist/src/togetherai-scenario.js +168 -0
- package/dist/src/togetherai-server.d.ts +16 -0
- package/dist/src/togetherai-server.js +187 -0
- package/dist/src/togetherai-stub.d.ts +59 -0
- package/dist/src/togetherai-stub.js +195 -0
- package/dist/src/togetherai-twin.d.ts +83 -0
- package/dist/src/togetherai-twin.js +1419 -0
- package/dist/src/togetherai-types.d.ts +207 -0
- package/dist/src/togetherai-types.js +26 -0
- package/package.json +52 -0
- package/src/cli.ts +27 -0
- package/src/index.ts +118 -0
- package/src/togetherai-budget.ts +156 -0
- package/src/togetherai-capabilities.ts +1315 -0
- package/src/togetherai-conformance.ts +459 -0
- package/src/togetherai-connector.ts +496 -0
- package/src/togetherai-models.ts +74 -0
- package/src/togetherai-scenario.ts +185 -0
- package/src/togetherai-server.ts +199 -0
- package/src/togetherai-stub.ts +197 -0
- package/src/togetherai-twin.ts +1448 -0
- package/src/togetherai-types.ts +222 -0
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
// Together AI's CLIENT-SIDE RATE BUDGET — the pack's DECLARATION (the numbers) plus the thin
|
|
2
|
+
// typed bindings `liveTogetheraiExecute` uses. The MECHANISM — the durable token-keyed ledger,
|
|
3
|
+
// the rolling window, reserve-under-lock, the `Retry-After`/429 cooldown, fail-CLOSED on a
|
|
4
|
+
// corrupt ledger — lives ONCE in the vendor-agnostic kernel (`@volter/world-core` →
|
|
5
|
+
// `rateBudget.ts`). Read that module's header for the full rationale AND for the honest list of
|
|
6
|
+
// what the guard does not guarantee.
|
|
7
|
+
//
|
|
8
|
+
// ── WHY THIS EXISTS ─────────────────────────────────────────────────────────────────────────
|
|
9
|
+
// A real ~4.5-DAY vendor lockout (Figma, 2026-07-25) happened because raw API calls were made
|
|
10
|
+
// outside the pack's connector — no cache, no batching, no ceiling. Discipline only binds the code
|
|
11
|
+
// that follows it; a BUDGET binds the code that does not.
|
|
12
|
+
//
|
|
13
|
+
// ── HOW THE CEILING WAS CHOSEN (live-read, 2026-09-16) ──────────────────────────────────────
|
|
14
|
+
// https://docs.together.ai/serverless/rate-limits (read 2026-09-16) publishes NO scalar a client
|
|
15
|
+
// could be bound by: limits are "dynamic" and PER ORGANIZATION and PER MODEL, and the page's own
|
|
16
|
+
// wording is "there are no fixed per-model limits published" — the account's dashboard is the only
|
|
17
|
+
// place the real numbers live. Success responses carry NO rate-limit headers; a 429 carries only
|
|
18
|
+
// `x-ratelimit-reset` (seconds until the window resets), and its error types are
|
|
19
|
+
// `dynamic_request_limited` / `dynamic_token_limited`; a 503 means the dynamic rate is at/below
|
|
20
|
+
// capacity.
|
|
21
|
+
//
|
|
22
|
+
// So this declaration does not model Together's limit, and nothing here is more permissive than
|
|
23
|
+
// the kernel's undeclared fallback: window 60s, ceiling 60, defaultWeight 2 — i.e. 30
|
|
24
|
+
// calls/minute, EXACTLY `DEFAULT_RATE_BUDGET`, with no endpoint priced cheaper than the fallback
|
|
25
|
+
// would price it. Being stricter than the fallback needs no vendor justification; being looser
|
|
26
|
+
// would, and there is none to have.
|
|
27
|
+
//
|
|
28
|
+
// It bounds the 60s AVERAGE; it does not pace (the kernel refuses, it never sleeps). The backstop
|
|
29
|
+
// for a sub-second burst is the cooldown: Together's `x-ratelimit-reset` (documented on the 429)
|
|
30
|
+
// is read off the response and turns into a persisted refusal.
|
|
31
|
+
import { declareRateBudget, rateBudgetPath, rateBudgetWeight, RateBudget, } from '@volter/world-core';
|
|
32
|
+
const VENDOR = 'togetherai';
|
|
33
|
+
/** Rolling window, in ms. Spend older than this is pruned. */
|
|
34
|
+
export const TOGETHERAI_BUDGET_WINDOW_MS = 60_000;
|
|
35
|
+
/**
|
|
36
|
+
* Weighted units allowed inside one window. 60/60s at `defaultWeight` 2 = 30 calls a minute —
|
|
37
|
+
* EXACTLY the kernel's undeclared fallback, because Together publishes no scalar that would
|
|
38
|
+
* justify more. See the header.
|
|
39
|
+
*/
|
|
40
|
+
export const TOGETHERAI_BUDGET_CEILING = 60;
|
|
41
|
+
/** Seconds. A `x-ratelimit-reset` above this means the key is throttled hard — fail loudly, don't sleep. */
|
|
42
|
+
export const TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S = 300;
|
|
43
|
+
/** Per-call cost, keyed by `"<METHOD> <path>"`. See the header for what is documented vs. judged. */
|
|
44
|
+
export const TOGETHERAI_CALL_WEIGHTS = {
|
|
45
|
+
/** `/v1/chat/completions`, `/v1/completions`, `/v1/embeddings`, `/v1/rerank`,
|
|
46
|
+
* `/v1/images/generations`, `/v1/audio/*` — token-metered, where Together's DYNAMIC token
|
|
47
|
+
* budget (`dynamic_token_limited`) rather than its request budget is usually what binds. */
|
|
48
|
+
inference: 6,
|
|
49
|
+
/** Everything else: files, batches, models, whoami, fine-tunes, endpoints. */
|
|
50
|
+
other: 2,
|
|
51
|
+
};
|
|
52
|
+
/** THE PACK'S DECLARATION — pure data, the only Together-specific thing in the whole budget. */
|
|
53
|
+
export const TOGETHERAI_RATE_BUDGET = {
|
|
54
|
+
windowMs: TOGETHERAI_BUDGET_WINDOW_MS,
|
|
55
|
+
ceiling: TOGETHERAI_BUDGET_CEILING,
|
|
56
|
+
defaultWeight: TOGETHERAI_CALL_WEIGHTS.other,
|
|
57
|
+
maxRetryAfterSeconds: TOGETHERAI_BUDGET_MAX_RETRY_AFTER_S,
|
|
58
|
+
rules: [
|
|
59
|
+
{ match: '^POST /v1/(chat/completions|completions|embeddings|rerank|images/generations)$', weight: TOGETHERAI_CALL_WEIGHTS.inference },
|
|
60
|
+
{ match: '^POST /v1/audio/', weight: TOGETHERAI_CALL_WEIGHTS.inference },
|
|
61
|
+
],
|
|
62
|
+
reason: 'Together publishes NO fixed request limit (docs.together.ai/serverless/rate-limits, read ' +
|
|
63
|
+
'2026-09-16): limits are DYNAMIC and per organization per model — the page states "there are ' +
|
|
64
|
+
'no fixed per-model limits published", the real numbers live only on the account dashboard, ' +
|
|
65
|
+
'and success responses carry no rate-limit headers at all. A 429 carries only ' +
|
|
66
|
+
'`x-ratelimit-reset` (seconds) with error types `dynamic_request_limited` / ' +
|
|
67
|
+
'`dynamic_token_limited`; a 503 signals the dynamic rate is at/below capacity. Because no ' +
|
|
68
|
+
'published figure justifies going higher, the ceiling is pinned at the kernel fallback in ' +
|
|
69
|
+
'EVERY dimension — 60 units / 60s at defaultWeight 2 = 30 calls/min — and the declaration ' +
|
|
70
|
+
'buys resolution DOWNWARD, never headroom: the inference endpoints cost 6, so at most 10 land ' +
|
|
71
|
+
'in a window. The window bounds the 60s AVERAGE and does not pace; the `x-ratelimit-reset` ' +
|
|
72
|
+
'cooldown is the backstop for a sub-second burst.',
|
|
73
|
+
};
|
|
74
|
+
// Declared at module load, so merely importing this module (which `togetherai-connector.ts`
|
|
75
|
+
// does) is enough to arm the real ceiling.
|
|
76
|
+
declareRateBudget(VENDOR, TOGETHERAI_RATE_BUDGET);
|
|
77
|
+
/**
|
|
78
|
+
* Price one call. The key is `"<METHOD> <path>"` with the query string split off, so a rule can
|
|
79
|
+
* price by method (a write is not a read) without the kernel knowing anything about Together. An
|
|
80
|
+
* unclassified endpoint still costs `defaultWeight` — nothing is ever free.
|
|
81
|
+
*/
|
|
82
|
+
export function togetheraiCallWeight(method, path) {
|
|
83
|
+
const { bare, query } = splitQuery(path);
|
|
84
|
+
// UPPER-CASE the method: `fetch` normalizes a known lowercase method before sending, so
|
|
85
|
+
// `execute('post', …)` really does issue a POST and must be priced as one.
|
|
86
|
+
return rateBudgetWeight(VENDOR, `${String(method).toUpperCase()} ${bare}`, query);
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* `/v1/x?a=1` -> `{ bare: '/v1/x', query: { a: '1' } }`. Rules match the path; NORMALIZED,
|
|
90
|
+
* because the anchored rules are otherwise trivially evaded: `fetch` upper-cases a known method
|
|
91
|
+
* before sending, so `execute('post', …)` issues a real WRITE that a `^POST ` rule would price as
|
|
92
|
+
* a read; and a trailing slash makes a path miss a `$` anchor while most routers treat it as the
|
|
93
|
+
* same endpoint.
|
|
94
|
+
*/
|
|
95
|
+
function splitQuery(path) {
|
|
96
|
+
const at = path.indexOf('?');
|
|
97
|
+
const query = {};
|
|
98
|
+
if (at !== -1)
|
|
99
|
+
for (const [k, v] of new URLSearchParams(path.slice(at + 1)))
|
|
100
|
+
query[k] = v;
|
|
101
|
+
// Collapse REPEATED slashes as well as a trailing one: `/v1//chat/completions` reaches the same
|
|
102
|
+
// endpoint on most routers but misses a `^POST /v1/(chat/completions|…)$` rule, which would
|
|
103
|
+
// price an inference call as a 2-unit read (the groq pack's §9 round one, NIT 14).
|
|
104
|
+
const raw = (at === -1 ? path : path.slice(0, at)).replace(/\/{2,}/g, '/');
|
|
105
|
+
const bare = raw.length > 1 && raw.endsWith('/') ? raw.replace(/\/+$/, '') : raw;
|
|
106
|
+
return { bare, query };
|
|
107
|
+
}
|
|
108
|
+
/** Where Together's ledger lives. Token-keyed and cwd-independent by default (Together's limits
|
|
109
|
+
* are per ORGANIZATION, i.e. per key, so a cwd-scoped ledger would hand the same key a fresh
|
|
110
|
+
* allowance in every checkout, worktree and CI matrix leg); pass `root` for world-scoped
|
|
111
|
+
* accounting. */
|
|
112
|
+
export function togetheraiBudgetPath(opts = {}) {
|
|
113
|
+
const o = typeof opts === 'string' ? { root: opts } : opts;
|
|
114
|
+
// VENDOR spread LAST: a loosely-typed `{ vendor: 'other', … }` slipping through (TypeScript's
|
|
115
|
+
// excess-property check only catches object literals) must not redirect this pack's ledger.
|
|
116
|
+
return rateBudgetPath({ ...o, vendor: VENDOR });
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* Together's budget — the shared kernel guard bound to this vendor's declaration. A real
|
|
120
|
+
* subclass, not an alias, so `budget instanceof TogetheraiBudget` in `liveTogetheraiExecute`
|
|
121
|
+
* means "a budget that accounts against TOGETHER's ledger under TOGETHER's ceiling".
|
|
122
|
+
*/
|
|
123
|
+
export class TogetheraiBudget extends RateBudget {
|
|
124
|
+
constructor(opts = {}) {
|
|
125
|
+
super({ ...opts, vendor: VENDOR });
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
/** The typed refusal. One error class shared with every other vendor's budget; `err.vendor` says
|
|
129
|
+
* which one refused, and `err.kind` says why. */
|
|
130
|
+
export { RateBudgetError as TogetheraiBudgetError } from '@volter/world-core';
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import { type CapabilityReport, type CapabilitySpec } from '@volter/world-tooling';
|
|
2
|
+
export declare const TOGETHERAI_CAPABILITIES: CapabilitySpec[];
|
|
3
|
+
export declare const TOGETHERAI_AREAS: readonly ["audio", "auth", "batches", "chat", "compute", "conformance", "connector", "embeddings", "endpoints", "errors", "evaluation", "files", "fine_tuning", "images", "models", "protocol", "queue", "rate_limits", "rerank", "streaming", "structured_outputs", "tools", "tci", "usage", "videos", "whoami"];
|
|
4
|
+
export declare function togetheraiCapabilities(): Promise<CapabilityReport>;
|