@volter/twin-cohere 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +224 -0
  2. package/defaults/handlers.json +26 -0
  3. package/dist/defaults/handlers.json +26 -0
  4. package/dist/src/cli.d.ts +2 -0
  5. package/dist/src/cli.js +31 -0
  6. package/dist/src/cohere-budget.d.ts +55 -0
  7. package/dist/src/cohere-budget.js +171 -0
  8. package/dist/src/cohere-capabilities.d.ts +14 -0
  9. package/dist/src/cohere-capabilities.js +1852 -0
  10. package/dist/src/cohere-conformance.d.ts +17 -0
  11. package/dist/src/cohere-conformance.js +464 -0
  12. package/dist/src/cohere-connector.d.ts +150 -0
  13. package/dist/src/cohere-connector.js +625 -0
  14. package/dist/src/cohere-models.d.ts +21 -0
  15. package/dist/src/cohere-models.js +73 -0
  16. package/dist/src/cohere-scenario.d.ts +57 -0
  17. package/dist/src/cohere-scenario.js +176 -0
  18. package/dist/src/cohere-server.d.ts +16 -0
  19. package/dist/src/cohere-server.js +184 -0
  20. package/dist/src/cohere-stub.d.ts +119 -0
  21. package/dist/src/cohere-stub.js +321 -0
  22. package/dist/src/cohere-twin.d.ts +82 -0
  23. package/dist/src/cohere-twin.js +1243 -0
  24. package/dist/src/cohere-types.d.ts +226 -0
  25. package/dist/src/cohere-types.js +40 -0
  26. package/dist/src/index.d.ts +15 -0
  27. package/dist/src/index.js +84 -0
  28. package/package.json +71 -0
  29. package/src/cli.ts +30 -0
  30. package/src/cohere-budget.ts +197 -0
  31. package/src/cohere-capabilities.ts +1855 -0
  32. package/src/cohere-conformance.ts +489 -0
  33. package/src/cohere-connector.ts +709 -0
  34. package/src/cohere-models.ts +79 -0
  35. package/src/cohere-scenario.ts +194 -0
  36. package/src/cohere-server.ts +195 -0
  37. package/src/cohere-stub.ts +337 -0
  38. package/src/cohere-twin.ts +1290 -0
  39. package/src/cohere-types.ts +231 -0
  40. package/src/index.ts +159 -0
@@ -0,0 +1,1290 @@
1
+ // Cohere twin REQUEST HANDLER — the canonical Cohere API surface for the twin. Contract:
2
+ // handleCohereTwinRequest({method, path, body}) -> {status, body}. It is the faithful Cohere API
3
+ // that the real `cohere-ai` client (constructed with `environment: 'http://127.0.0.1:<port>'`) and
4
+ // the real `@ai-sdk/cohere` provider (`createCohere({ baseURL: '…/v2' })`) talk to UNMODIFIED.
5
+ //
6
+ // ── COHERE IS ITS OWN DIALECT, AND THE REFUSALS ARE THE FIDELITY SURFACE ────────────────
7
+ // This is NOT an OpenAI-compatible API and copying an OpenAI-shaped pack's permissiveness is
8
+ // precisely the bug (ADDING_A_TWIN.md §0). What distinguishes Cohere is what it REFUSES and what
9
+ // its envelopes look like:
10
+ // • the error body is a BARE `{ "message": "…" }` — no `error` wrapper, no `type`, no `code`
11
+ // (@ai-sdk/cohere's `cohereErrorDataSchema` is literally `z.object({ message: z.string() })`);
12
+ // • `finish_reason` is UPPER-CASE from Cohere's own set (COMPLETE / TOOL_CALL / MAX_TOKENS /
13
+ // STOP_SEQUENCE / ERROR / TIMEOUT) — `stop` and `tool_calls` are not Cohere values;
14
+ // • a v2 chat response has NO `choices` array: one `message`, one `finish_reason`, one `usage`
15
+ // with `billed_units` AND `tokens` sub-objects;
16
+ // • v1 and v2 are DIFFERENT PROTOCOLS on one host, not a versioned alias — v1 chat takes a
17
+ // single `message` STRING and answers a flat `text`; v2 takes `messages[]` and answers a
18
+ // structured assistant message;
19
+ // • `POST /v2/embed` REQUIRES `input_type`; `POST /v1/embed` does not, but refuses a v3/v4 embed
20
+ // model without one;
21
+ // • v1 embed defaults to `embeddings_floats` (a flat `number[][]`) while v2 always answers
22
+ // `embeddings_by_type` (keyed by `float`/`int8`/…);
23
+ // • v2 rerank `documents` must be STRINGS; v1 accepts objects and has `return_documents`;
24
+ // • `POST /v1/check-api-key` is a POST, not the GET its name suggests;
25
+ // • the enums are UPPER-CASE (`truncate: NONE|START|END`, `tool_choice: REQUIRED|NONE`,
26
+ // `safety_mode: CONTEXTUAL|STRICT|OFF`) where most vendors' are lower.
27
+ // Every one of those is asserted by a manifest verify, so an accidental drift toward the
28
+ // OpenAI shape reddens by name.
29
+ //
30
+ // ── THE HONEST DESIGN ──────────────────────────────────────────────────────────────────
31
+ // The twin cannot run the model, so `/v2/chat`, `/v1/chat`, `/v2/embed`, `/v1/embed`,
32
+ // `/v2/rerank`, `/v1/rerank`, `/v1/classify` and `/v1/tokenize` return DETERMINISTIC STUBS
33
+ // (cohere-stub.ts) clearly labeled as such — never real model output. But the ENTIRE PROTOCOL
34
+ // ENVELOPE is vendor-faithful, including the SSE event sequence terminated by `data: [DONE]`.
35
+ //
36
+ // The genuinely stateful + static surface is real, not stubbed:
37
+ // • GET /v1/models (+ /{name}) — the static real catalog (cohere-models.ts)
38
+ // • Datasets — POST/GET/DELETE /v1/datasets (+ /usage, /{id}), stateful
39
+ // • Connectors — POST/GET/PATCH/DELETE /v1/connectors (+ /{id}/oauth/authorize), stateful
40
+ // • Embed jobs — POST/GET /v1/embed-jobs (+ /{id}, /{id}/cancel), stateful, cross-referencing
41
+ // a real dataset id
42
+ // • The tokenizer VOCABULARY — `/v1/tokenize` observes (segment → id) pairs into the log, and
43
+ // `/v1/detokenize` folds that projection. That is what makes detokenize honest rather than a
44
+ // fabrication: the twin can only detokenize what it has actually seen, and says so otherwise.
45
+ //
46
+ // State lives in the kernel action log (D1): all writes are local actions, reads are the
47
+ // projection. No real Cohere is ever called from this path (D4). Streaming uses an INJECTED sink
48
+ // — no real sockets / setTimeout (D5 verify is offline + deterministic).
49
+ //
50
+ // ID MINTING: deterministic, and the ordinal is scanned from the id-SET INCLUDING TOMBSTONES
51
+ // (`_deleted` rows are retained in the projection and filtered out of reads) so a
52
+ // delete-then-create can never re-issue a live id — the failure mode three packs' §9 reviews each
53
+ // found in a count-mint.
54
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
55
+ import { COHERE_MODELS, findModel, modelServes } from './cohere-models.ts';
56
+ import {
57
+ base64Embedding,
58
+ classifyText,
59
+ COHERE_OUTPUT_DIMENSIONS,
60
+ cohereId,
61
+ contentToText,
62
+ countInputTokens,
63
+ embedDimensions,
64
+ estimateTokens,
65
+ fnv1a,
66
+ lastUserText,
67
+ preferredTokenId,
68
+ pseudoEmbedding,
69
+ quantizeEmbedding,
70
+ rerankScore,
71
+ segmentText,
72
+ stubAssistantText,
73
+ stubToolCall,
74
+ stubToolPlan,
75
+ stubV1Text,
76
+ toolNames,
77
+ } from './cohere-stub.ts';
78
+ import { type CohereScenarioEngine, type CohereScenarioRespond, realizeCohereRespond, type ScriptedResult } from './cohere-scenario.ts';
79
+ import {
80
+ COHERE_EMBEDDING_TYPES,
81
+ COHERE_INPUT_TYPES,
82
+ COHERE_TRUNCATE,
83
+ type CohereApiMeta,
84
+ type CohereAssistantContentItem,
85
+ type CohereChatV2Response,
86
+ type CohereEmbeddingType,
87
+ type CohereFinishReason,
88
+ type CohereV1FinishReason,
89
+ type CohereMessageV2,
90
+ type CohereToolCallV2,
91
+ type CohereUsage,
92
+ type SseSink,
93
+ } from './cohere-types.ts';
94
+
95
+ const SERVICE = 'cohere';
96
+
97
+ /** The API version string Cohere stamps into every `meta.api_version`. */
98
+ const API_VERSION = '1';
99
+
100
+ export type CohereRequest = {
101
+ /** The scenario engine (kernel grammar + this pack's vocabulary) — scripts chat turns. */
102
+ scenarioEngine?: CohereScenarioEngine;
103
+ method: string;
104
+ path: string;
105
+ body?: string;
106
+ occurredAt?: string;
107
+ root?: string;
108
+ readOnly?: boolean;
109
+ /** Lower-cased request headers (e.g. `authorization`) the HTTP server passes through so the
110
+ * handler can model auth (401). In-process trusted calls (capability verifies, the connector)
111
+ * omit them and are not auth-gated — the twin cannot validate against real keys, so the
112
+ * modeled failure is the CHECKABLE missing/sentinel-invalid case. */
113
+ headers?: Record<string, string>;
114
+ /** When set on a streaming POST, chunks are written here (no sockets). */
115
+ sseSink?: SseSink;
116
+ };
117
+
118
+ /** The handler response. `headers` (when present) are response headers the HTTP server should
119
+ * set — e.g. `Retry-After` on a scripted 429. */
120
+ export type CohereResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
121
+
122
+ // ── vendor-shaped errors ────────────────────────────────────────────────────────────────
123
+ /**
124
+ * THE error envelope. One key, `message`, and nothing else — the shape
125
+ * @ai-sdk/cohere@4.0.35 pins with `cohereErrorDataSchema = z.object({ message: z.string() })`,
126
+ * and the shape cohere-ai@8.1.0 hands to every typed error class (`new Cohere.BadRequestError(
127
+ * _response.error.body)`) without unwrapping.
128
+ *
129
+ * The STATUS codes below are the vendor's; the message TEXT for the twin's own refusals follows
130
+ * Cohere's `invalid request: …` idiom but is authored here — a twin cannot know the vendor's exact
131
+ * prose for every input, and inventing a `type`/`code` field to look more official would be
132
+ * serving surface the vendor does not have.
133
+ */
134
+ function apiError(status: number, message: string): CohereResponseEnvelope {
135
+ return { status, body: { message } };
136
+ }
137
+
138
+ /** `invalid request: …` — Cohere's 400 idiom for a malformed body. */
139
+ function invalidRequest(detail: string): CohereResponseEnvelope {
140
+ return apiError(400, `invalid request: ${detail}`);
141
+ }
142
+
143
+ /**
144
+ * The vendor's 404 for a model the endpoint cannot serve. Cohere answers the same way for a model
145
+ * that does not exist AND for one that exists but is not available on the endpoint being called —
146
+ * from the endpoint's point of view the model is simply not found — so the twin does not invent a
147
+ * separate "incompatible model" status.
148
+ */
149
+ function modelNotFound(model: string): CohereResponseEnvelope {
150
+ return apiError(404, `model '${model}' not found, make sure the correct model ID was used and that you have access to the model.`);
151
+ }
152
+
153
+ function notFound(message: string): CohereResponseEnvelope {
154
+ return apiError(404, message);
155
+ }
156
+
157
+ /** The 404 an unrouted path answers. Written as one function so conformance can assert the
158
+ * envelope as a literal predicate without importing this. */
159
+ function routeNotFound(method: string, path: string): CohereResponseEnvelope {
160
+ return apiError(404, `not found: ${method} ${path}`);
161
+ }
162
+
163
+ /** Read-only mode refuses every mutation (D3). Cohere has no 405 idiom of its own, so the twin
164
+ * answers its own `{message}` envelope with the standard 405 status. */
165
+ function readOnlyRefusal(): CohereResponseEnvelope {
166
+ return apiError(405, 'this twin is running read-only; writes are refused.');
167
+ }
168
+
169
+ // ── modeled authentication (401) ────────────────────────────────────────────────────────
170
+ /** The obvious-invalid sentinel a caller can use to exercise the 401 path deterministically. */
171
+ const INVALID_KEY_SENTINEL = 'invalid';
172
+
173
+ /**
174
+ * 401 when a request carries an auth SURFACE (headers present) but no usable bearer token. The
175
+ * twin cannot validate against real Cohere keys, so the modeled failure is the CHECKABLE
176
+ * missing / sentinel-invalid case; any other non-empty token is accepted (auth is faked, D3).
177
+ *
178
+ * cohere-ai sends `Authorization: Bearer <token>`, sourcing the token from the constructor's
179
+ * `token` option or the `CO_API_KEY` env var (`auth/BearerAuthProvider.js`, `const ENV_TOKEN =
180
+ * "CO_API_KEY"`); @ai-sdk/cohere sends the same header from `COHERE_API_KEY`.
181
+ */
182
+ function checkAuth(req: CohereRequest): CohereResponseEnvelope | null {
183
+ if (req.headers === undefined) return null;
184
+ const raw = req.headers.authorization ?? '';
185
+ const token = raw.toLowerCase().startsWith('bearer ') ? raw.slice(7).trim() : '';
186
+ if (token === '' || token === INVALID_KEY_SENTINEL) {
187
+ return apiError(401, 'invalid api token');
188
+ }
189
+ return null;
190
+ }
191
+
192
+ // ── time ────────────────────────────────────────────────────────────────────────────────
193
+ /** A FIXED default instant. The served response must be a pure function of (request, stored
194
+ * state) — reading the wall clock here would make replay non-byte-identical. Callers that want a
195
+ * moving clock pass `occurredAt` explicitly. */
196
+ const PINNED_EPOCH_MS = 1_800_000_000_000;
197
+ function nowIso(occurredAt?: string): string {
198
+ return occurredAt ?? new Date(PINNED_EPOCH_MS).toISOString();
199
+ }
200
+
201
+ // ── meta ────────────────────────────────────────────────────────────────────────────────
202
+ function meta(billed: CohereApiMeta['billed_units'], tokens?: CohereApiMeta['tokens'], warnings?: string[]): CohereApiMeta {
203
+ return {
204
+ api_version: { version: API_VERSION },
205
+ billed_units: billed,
206
+ ...(tokens ? { tokens } : {}),
207
+ ...(warnings && warnings.length ? { warnings } : {}),
208
+ };
209
+ }
210
+
211
+ // ── kernel helpers ──────────────────────────────────────────────────────────────────────
212
+ /** EVERY row of a type, tombstones included — the id-mint denominator. */
213
+ function allRows(type: string, root?: string): Array<Record<string, unknown>> {
214
+ return projectResources(SERVICE, root).filter((r) => r.type === type);
215
+ }
216
+ /** The LIVE rows of a type — what reads serve (a `_deleted` tombstone is invisible). */
217
+ function rows(type: string, root?: string): Array<Record<string, unknown>> {
218
+ return allRows(type, root).filter((r) => r._deleted !== true);
219
+ }
220
+ function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined {
221
+ return rows(type, root).find((r) => r.id === id);
222
+ }
223
+
224
+ /**
225
+ * Mint the next id for `type`. The ordinal is the max already seen ACROSS TOMBSTONES, so
226
+ * delete-then-create never re-issues a live id, and it is scanned from the id SET rather than
227
+ * derived from a row count or a module-level counter — a count-mint collides the moment a pulled
228
+ * vendor id sits in a gap above the row count.
229
+ *
230
+ * Cohere's own ids are UUIDs, so the twin's are UUID-SHAPED too (vendor-faithful ids are part of
231
+ * the surface) but derived from a namespaced, ordinal-bearing seed. The `-twin-` marker in the
232
+ * seed is what keeps a locally minted id from ever colliding with a pulled vendor id in either
233
+ * direction: the twin's ids live in a namespace no Cohere UUID occupies.
234
+ */
235
+ function nextId(type: string, root?: string): string {
236
+ const seen = new Set<string>();
237
+ for (const r of allRows(type, root)) seen.add(String(r.id));
238
+ let ordinal = seen.size + 1;
239
+ // Probe upward until the derived id is genuinely unused. Bounded, deterministic, and it cannot
240
+ // spin: each iteration tries a different ordinal and the set is finite.
241
+ for (let guard = 0; guard <= seen.size + 1; guard++, ordinal++) {
242
+ const candidate = cohereId(`cohere-twin-${type}-${ordinal}`);
243
+ if (!seen.has(candidate)) return candidate;
244
+ }
245
+ /* c8 ignore next */
246
+ throw new Error(`cohere: could not mint a free ${type} id`);
247
+ }
248
+
249
+ /** Drop the kernel housekeeping fields (`type`/`updatedAt` are RESERVED by the projection and
250
+ * never survive it) and the twin's private underscore-prefixed fields, so the served view is
251
+ * exactly the vendor's. */
252
+ function strip(r: Record<string, unknown>): Record<string, unknown> {
253
+ const out: Record<string, unknown> = {};
254
+ for (const [k, v] of Object.entries(r)) {
255
+ if (k === 'type' || k === 'updatedAt' || k.startsWith('_')) continue;
256
+ out[k] = v;
257
+ }
258
+ return out;
259
+ }
260
+
261
+ async function write(req: CohereRequest, operation: string, subjectType: string, subjectId: string, fields: Record<string, unknown>): Promise<void> {
262
+ await applyTwinWrite(SERVICE, {
263
+ operation,
264
+ subjectType,
265
+ subjectId,
266
+ fields,
267
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
268
+ actor: { kind: 'agent' },
269
+ }, req.root);
270
+ }
271
+
272
+ /**
273
+ * The next per-subject write ordinal — the kernel's content+millisecond dedupe hazard, closed.
274
+ *
275
+ * `applyTwinWrite`'s action id is (content + `occurredAt` millisecond), so a write that returns a
276
+ * subject to a value it PREVIOUSLY HELD at the same instant collides with the earlier action and is
277
+ * silently dropped as `replayed`: the reply reports the new value while the projection keeps the
278
+ * old one. That is not theoretical here — a world running under a FROZEN clock
279
+ * (`TWIN_WORLD_CLOCK_FILE`) gives every request in it one `occurredAt`, so
280
+ * `PATCH name:'A'` → `PATCH name:'B'` → `PATCH name:'A'` makes the third action byte-identical to
281
+ * the first, and the handler answers 200 carrying `'B'` — the value the caller did NOT ask for
282
+ * (§9 round 1, finding 4).
283
+ *
284
+ * Folding a monotonically increasing ordinal into the write's fields makes every genuine write
285
+ * distinct by content, so no real write can be mistaken for a replay. It is underscore-prefixed and
286
+ * therefore stripped from every served view. (`upstash-store.ts`'s `rev` is the precedent.)
287
+ *
288
+ * Only revisitable paths need it: a CREATE mints a fresh id and a DELETE is one-way, so
289
+ * `connector.update` is the single write in this twin whose (subject, fields) can revisit a prior
290
+ * value. `token.observe` deliberately does NOT get one — an (id → segment) pair is immutable, so a
291
+ * repeat genuinely IS a replay and the kernel's dedupe is the correct behaviour there.
292
+ */
293
+ function nextRev(type: string, id: string, root?: string): number {
294
+ const row = getRow(type, id, root);
295
+ const rev = row?._rev;
296
+ return (typeof rev === 'number' ? rev : 0) + 1;
297
+ }
298
+
299
+ function parseJson(body?: string): Record<string, unknown> {
300
+ if (!body) return {};
301
+ try {
302
+ const v = JSON.parse(body) as unknown;
303
+ return v && typeof v === 'object' && !Array.isArray(v) ? (v as Record<string, unknown>) : {};
304
+ } catch {
305
+ return {};
306
+ }
307
+ }
308
+
309
+ // ════════════════════════════════════════════════════════════════════════════════════════
310
+ // CHAT v2
311
+ // ════════════════════════════════════════════════════════════════════════════════════════
312
+
313
+ export type ChatV2Args = {
314
+ model: string;
315
+ messages: CohereMessageV2[];
316
+ tools?: unknown;
317
+ toolChoice?: 'REQUIRED' | 'NONE';
318
+ maxTokens?: number;
319
+ stream: boolean;
320
+ thinking: boolean;
321
+ };
322
+
323
+ const V2_ROLES = new Set(['user', 'assistant', 'system', 'tool']);
324
+ const TOOL_CHOICES = new Set(['REQUIRED', 'NONE']);
325
+ const SAFETY_MODES = new Set(['CONTEXTUAL', 'STRICT', 'OFF']);
326
+
327
+ /**
328
+ * Validate a v2 chat body. Every rejection below is a CLOSED SET the SDK itself declares, so the
329
+ * check is an oracle rather than a guess: `ChatMessageV2` is a four-member union discriminated on
330
+ * `role`, `V2ChatRequestToolChoice` is exactly {REQUIRED, NONE}, and `V2ChatRequestSafetyMode` is
331
+ * exactly {CONTEXTUAL, STRICT, OFF}.
332
+ */
333
+ function validateChatV2(params: Record<string, unknown>): { args: ChatV2Args } | { error: CohereResponseEnvelope } {
334
+ const model = params.model;
335
+ if (typeof model !== 'string' || model === '') return { error: invalidRequest('model is required') };
336
+ const messages = params.messages;
337
+ if (!Array.isArray(messages)) return { error: invalidRequest('messages is required') };
338
+ if (messages.length === 0) return { error: invalidRequest('messages must not be empty') };
339
+ for (const [i, m] of messages.entries()) {
340
+ const role = (m as { role?: unknown })?.role;
341
+ if (typeof role !== 'string' || !V2_ROLES.has(role)) {
342
+ return { error: invalidRequest(`messages[${i}].role must be one of ${[...V2_ROLES].join(', ')}`) };
343
+ }
344
+ }
345
+ if (params.tool_choice !== undefined && (typeof params.tool_choice !== 'string' || !TOOL_CHOICES.has(params.tool_choice))) {
346
+ return { error: invalidRequest(`tool_choice must be one of ${[...TOOL_CHOICES].join(', ')}`) };
347
+ }
348
+ if (params.safety_mode !== undefined && (typeof params.safety_mode !== 'string' || !SAFETY_MODES.has(params.safety_mode))) {
349
+ return { error: invalidRequest(`safety_mode must be one of ${[...SAFETY_MODES].join(', ')}`) };
350
+ }
351
+ if (!modelServes(model, 'chat')) return { error: modelNotFound(model) };
352
+ const thinking = (params.thinking as { type?: unknown } | undefined)?.type === 'enabled';
353
+ return {
354
+ args: {
355
+ model,
356
+ messages: messages as CohereMessageV2[],
357
+ ...(params.tools !== undefined ? { tools: params.tools } : {}),
358
+ ...(typeof params.tool_choice === 'string' ? { toolChoice: params.tool_choice as 'REQUIRED' | 'NONE' } : {}),
359
+ ...(typeof params.max_tokens === 'number' ? { maxTokens: params.max_tokens } : {}),
360
+ stream: params.stream === true,
361
+ thinking,
362
+ },
363
+ };
364
+ }
365
+
366
+ /** The outcome of ONE engine advance. `missTeach` is appended to the stub so an in-world agent
367
+ * that never matched a handler is told what features it would have to match on. */
368
+ type ScenarioOutcome = { scripted?: ScriptedResult; error?: CohereResponseEnvelope; missTeach?: string };
369
+
370
+ /** Ask the scenario engine (if any) what this turn should say. ONE advance per request — the
371
+ * kernel's `next()` mutates scope state (call counter, `once`, `phase`) exactly like serving, so
372
+ * calling it twice for one request would double-count every handler. */
373
+ function scenarioDecision(args: ChatV2Args, engine: CohereScenarioEngine): ScenarioOutcome {
374
+ const decision = engine.next({
375
+ model: args.model,
376
+ messages: args.messages,
377
+ ...(args.tools !== undefined ? { tools: args.tools } : {}),
378
+ });
379
+ if (decision.kind !== 'handler') {
380
+ return { missTeach: `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/cohere.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]` };
381
+ }
382
+ const respond = decision.respond as CohereScenarioRespond;
383
+ if (respond.error) {
384
+ // A scripted API error is served INSTEAD of a completion — the vendor's own status and
385
+ // `{message}` envelope, on both the unary and the streaming path. Deciding it here is what
386
+ // lets it be a real HTTP status rather than a 200 carrying an error-shaped body.
387
+ const kind = respond.error.type;
388
+ if (kind === 'rate_limit') {
389
+ const retry = respond.error.retryAfter ?? 1;
390
+ return { error: { ...apiError(429, 'too many requests'), headers: { 'retry-after': String(retry) } } };
391
+ }
392
+ if (kind === 'service_unavailable') return { error: apiError(503, 'service unavailable') };
393
+ return { error: apiError(500, 'internal server error') };
394
+ }
395
+ return { scripted: realizeCohereRespond(respond) };
396
+ }
397
+
398
+ /** The assistant turn (scripted or stubbed) plus the finish reason it implies. */
399
+ function assistantTurn(args: ChatV2Args, scripted?: ScriptedResult, missTeach = ''): { content: CohereAssistantContentItem[]; toolCalls: CohereToolCallV2[]; toolPlan?: string; finish: CohereFinishReason } {
400
+ if (scripted) {
401
+ return {
402
+ content: scripted.content,
403
+ toolCalls: scripted.toolCalls,
404
+ ...(scripted.toolPlan !== undefined ? { toolPlan: scripted.toolPlan } : {}),
405
+ finish: scripted.finishReason,
406
+ };
407
+ }
408
+ const names = toolNames(args.tools);
409
+ // `tool_choice: 'NONE'` forbids a tool call even when tools are declared — the vendor honours it
410
+ // and so must the twin, or a caller testing the NONE path gets a tool call it explicitly banned.
411
+ const wantsTool = names.length > 0 && args.toolChoice !== 'NONE';
412
+ const seed = JSON.stringify(args.messages);
413
+ if (wantsTool) {
414
+ const call = stubToolCall(args.tools, seed, 0);
415
+ return { content: [], toolCalls: call ? [call] : [], toolPlan: stubToolPlan(names), finish: 'TOOL_CALL' };
416
+ }
417
+ const content: CohereAssistantContentItem[] = [];
418
+ if (args.thinking) content.push({ type: 'thinking', thinking: `[twin-stub] deterministic reasoning trace for ${args.model}.` });
419
+ content.push({ type: 'text', text: stubAssistantText(args.messages, args.model) + missTeach });
420
+ return { content, toolCalls: [], finish: 'COMPLETE' };
421
+ }
422
+
423
+ function usageFor(args: ChatV2Args, outputText: string): CohereUsage {
424
+ const input = countInputTokens(args.messages);
425
+ const output = estimateTokens(outputText);
426
+ // Cohere reports the SAME counts twice, under `billed_units` and `tokens`. They are distinct
427
+ // objects on the wire (a caller may read either), so both are emitted.
428
+ return { billed_units: { input_tokens: input, output_tokens: output }, tokens: { input_tokens: input, output_tokens: output } };
429
+ }
430
+
431
+ function outputTextOf(content: CohereAssistantContentItem[], toolCalls: CohereToolCallV2[], toolPlan?: string): string {
432
+ const parts = content.map((c) => (c.type === 'text' ? c.text : c.thinking));
433
+ if (toolPlan) parts.push(toolPlan);
434
+ for (const tc of toolCalls) parts.push(tc.function.arguments);
435
+ return parts.join('');
436
+ }
437
+
438
+ /** Build the unary `POST /v2/chat` envelope. */
439
+ export function buildChatV2(args: ChatV2Args, outcome: ScenarioOutcome = {}): CohereChatV2Response {
440
+ const turn = assistantTurn(args, outcome.scripted, outcome.missTeach ?? '');
441
+ const text = outputTextOf(turn.content, turn.toolCalls, turn.toolPlan);
442
+ const message: CohereChatV2Response['message'] = { role: 'assistant' };
443
+ // Cohere OMITS the empty halves rather than sending nulls: an assistant turn that made a tool
444
+ // call carries `tool_calls` + `tool_plan` and no `content`, and a plain turn carries `content`
445
+ // and neither of the others.
446
+ if (turn.content.length > 0) message.content = turn.content;
447
+ if (turn.toolPlan !== undefined) message.tool_plan = turn.toolPlan;
448
+ if (turn.toolCalls.length > 0) message.tool_calls = turn.toolCalls;
449
+ return {
450
+ id: cohereId(`chat-v2|${args.model}|${JSON.stringify(args.messages)}`),
451
+ finish_reason: turn.finish,
452
+ message,
453
+ usage: usageFor(args, text),
454
+ };
455
+ }
456
+
457
+ /**
458
+ * Stream `POST /v2/chat` into the injected sink and return the same body the unary path would.
459
+ *
460
+ * The event sequence is Cohere's own, and it is asserted key-for-key against
461
+ * @ai-sdk/cohere's `cohereChatChunkSchema` — a `z.discriminatedUnion('type', …)`, so a wrong
462
+ * `type` or a missing `delta.message.content` is a hard parse failure in the real provider, not a
463
+ * soft mismatch. Note the shapes differ between `content-start` (a full content BLOCK under
464
+ * `delta.message.content`) and `content-delta` (just `{ text }`), which is exactly the sort of
465
+ * detail a hand-written stream gets wrong.
466
+ */
467
+ export function streamChatV2(args: ChatV2Args, sink: SseSink, outcome: ScenarioOutcome = {}): CohereChatV2Response {
468
+ const full = buildChatV2(args, outcome);
469
+ sink({ data: { type: 'message-start', id: full.id, delta: { message: { role: 'assistant' } } } });
470
+
471
+ const toolPlan = full.message.tool_plan;
472
+ if (toolPlan !== undefined) {
473
+ for (const piece of chunkText(toolPlan)) {
474
+ sink({ data: { type: 'tool-plan-delta', delta: { message: { tool_plan: piece } } } });
475
+ }
476
+ }
477
+
478
+ const blocks = full.message.content ?? [];
479
+ for (const [index, block] of blocks.entries()) {
480
+ const kind = block.type;
481
+ const whole = kind === 'text' ? block.text : block.thinking;
482
+ sink({ data: { type: 'content-start', index, delta: { message: { content: kind === 'text' ? { type: 'text', text: '' } : { type: 'thinking', thinking: '' } } } } });
483
+ for (const piece of chunkText(whole)) {
484
+ sink({ data: { type: 'content-delta', index, delta: { message: { content: kind === 'text' ? { text: piece } : { thinking: piece } } } } });
485
+ }
486
+ sink({ data: { type: 'content-end', index } });
487
+ }
488
+
489
+ for (const [index, call] of (full.message.tool_calls ?? []).entries()) {
490
+ sink({ data: { type: 'tool-call-start', index, delta: { message: { tool_calls: { id: call.id, type: 'function', function: { name: call.function.name, arguments: '' } } } } } });
491
+ for (const piece of chunkText(call.function.arguments)) {
492
+ sink({ data: { type: 'tool-call-delta', index, delta: { message: { tool_calls: { function: { arguments: piece } } } } } });
493
+ }
494
+ sink({ data: { type: 'tool-call-end', index } });
495
+ }
496
+
497
+ sink({ data: { type: 'message-end', id: full.id, delta: { finish_reason: full.finish_reason, usage: full.usage } } });
498
+ sink({ done: true });
499
+ return full;
500
+ }
501
+
502
+ /** Split text into deterministic streaming pieces. Deterministic chunking is what makes a
503
+ * streamed replay byte-identical to the previous one. */
504
+ function chunkText(text: string): string[] {
505
+ if (text === '') return [];
506
+ const out: string[] = [];
507
+ for (let i = 0; i < text.length; i += 24) out.push(text.slice(i, i + 24));
508
+ return out;
509
+ }
510
+
511
+ // ════════════════════════════════════════════════════════════════════════════════════════
512
+ // CHAT v1 — a DIFFERENT protocol, not a versioned alias
513
+ // ════════════════════════════════════════════════════════════════════════════════════════
514
+
515
+ function handleChatV1(params: Record<string, unknown>, req: CohereRequest): CohereResponseEnvelope {
516
+ const message = params.message;
517
+ // v1's required field is `message` (a STRING) — a caller that sends v2's `messages` array here
518
+ // gets this refusal, which is the honest answer: it is talking the wrong protocol version.
519
+ if (typeof message !== 'string' || message === '') return invalidRequest('message is required');
520
+ const model = typeof params.model === 'string' && params.model !== '' ? params.model : 'command-a-03-2025';
521
+ if (!modelServes(model, 'chat')) return modelNotFound(model);
522
+ const history = Array.isArray(params.chat_history) ? (params.chat_history as Array<Record<string, unknown>>) : [];
523
+ const text = stubV1Text(message, model);
524
+ const inputTokens = estimateTokens(message) + history.reduce((a, h) => a + estimateTokens(String(h.message ?? '')), 0);
525
+ const outputTokens = estimateTokens(text);
526
+ const seed = `chat-v1|${model}|${message}|${JSON.stringify(history)}`;
527
+ if (params.stream === true) {
528
+ if (!req.sseSink) return invalidRequest('stream:true requires a streaming transport');
529
+ return { status: 200, body: streamChatV1(text, seed, req.sseSink, inputTokens, outputTokens, model, message, history) };
530
+ }
531
+ return { status: 200, body: v1ChatBody(text, seed, inputTokens, outputTokens, message, history) };
532
+ }
533
+
534
+ function v1ChatBody(text: string, seed: string, inputTokens: number, outputTokens: number, message: string, history: Array<Record<string, unknown>>): Record<string, unknown> {
535
+ return {
536
+ text,
537
+ generation_id: cohereId(`${seed}|gen`),
538
+ response_id: cohereId(`${seed}|res`),
539
+ finish_reason: 'COMPLETE' satisfies CohereV1FinishReason, // valid in v1's set, v2's, and the stream-end set
540
+ chat_history: [
541
+ ...history.map((h) => ({ role: String(h.role ?? 'USER'), message: String(h.message ?? '') })),
542
+ { role: 'USER', message },
543
+ { role: 'CHATBOT', message: text },
544
+ ],
545
+ meta: meta({ input_tokens: inputTokens, output_tokens: outputTokens }, { input_tokens: inputTokens, output_tokens: outputTokens }),
546
+ };
547
+ }
548
+
549
+ /**
550
+ * v1's stream is a DIFFERENT wire from v2's: newline-delimited JSON objects discriminated on
551
+ * `event_type` (`stream-start` / `text-generation` / `stream-end`), not v2's `type`-tagged SSE.
552
+ * The twin emits it through the same sink; the server frames v1 as NDJSON and v2 as SSE, which is
553
+ * what the two halves of cohere-ai actually decode.
554
+ */
555
+ function streamChatV1(text: string, seed: string, sink: SseSink, inputTokens: number, outputTokens: number, model: string, message: string, history: Array<Record<string, unknown>>): Record<string, unknown> {
556
+ const full = v1ChatBody(text, seed, inputTokens, outputTokens, message, history);
557
+ void model;
558
+ sink({ data: { is_finished: false, event_type: 'stream-start', generation_id: full.generation_id } });
559
+ for (const piece of chunkText(text)) {
560
+ sink({ data: { is_finished: false, event_type: 'text-generation', text: piece } });
561
+ }
562
+ sink({ data: { is_finished: true, event_type: 'stream-end', finish_reason: 'COMPLETE', response: full } });
563
+ return full;
564
+ }
565
+
566
+ // ════════════════════════════════════════════════════════════════════════════════════════
567
+ // EMBED
568
+ // ════════════════════════════════════════════════════════════════════════════════════════
569
+
570
+ const EMBEDDING_TYPES = new Set<string>(COHERE_EMBEDDING_TYPES);
571
+ const INPUT_TYPES = new Set<string>(COHERE_INPUT_TYPES);
572
+ const TRUNCATE = new Set<string>(COHERE_TRUNCATE);
573
+
574
+ /** Which embed models require an explicit `input_type` on v1. Cohere's v3+ embed models do; the
575
+ * legacy v2 ones did not, which is exactly why the v1 endpoint keeps the field optional and
576
+ * refuses at request time instead. */
577
+ function requiresInputType(model: string): boolean {
578
+ return /-v[34](\.\d+)?$|v3\.0$|^embed-v4\.0$/.test(model) || model.includes('v3.0') || model === 'embed-v4.0';
579
+ }
580
+
581
+ function embedVectors(texts: string[], model: string, outputDimension?: number): number[][] {
582
+ const dim = outputDimension ?? embedDimensions(model);
583
+ return texts.map((t) => pseudoEmbedding(t, dim));
584
+ }
585
+
586
+ /** Build the `embeddings` object keyed by the requested types. */
587
+ function embeddingsByType(floats: number[][], types: CohereEmbeddingType[]): Record<string, unknown> {
588
+ const out: Record<string, unknown> = {};
589
+ for (const t of types) {
590
+ if (t === 'float') out.float = floats;
591
+ else if (t === 'base64') out.base64 = floats.map(base64Embedding);
592
+ else out[t] = floats.map((v) => quantizeEmbedding(v, t));
593
+ }
594
+ return out;
595
+ }
596
+
597
+ function collectEmbedInputs(params: Record<string, unknown>): string[] | null {
598
+ const texts = params.texts;
599
+ if (Array.isArray(texts)) return texts.map((t) => String(t));
600
+ // v2 also accepts `inputs: EmbedInput[]`, each carrying a `content` part array.
601
+ const inputs = params.inputs;
602
+ if (Array.isArray(inputs)) {
603
+ return inputs.map((i) => contentToText((i as { content?: CohereMessageV2['content'] })?.content ?? ''));
604
+ }
605
+ const images = params.images;
606
+ if (Array.isArray(images)) return images.map((i) => String(i));
607
+ return null;
608
+ }
609
+
610
+ function validateEmbedShared(params: Record<string, unknown>): CohereResponseEnvelope | null {
611
+ const types = params.embedding_types;
612
+ if (types !== undefined) {
613
+ if (!Array.isArray(types) || types.length === 0) return invalidRequest('embedding_types must be a non-empty list');
614
+ for (const t of types) {
615
+ if (typeof t !== 'string' || !EMBEDDING_TYPES.has(t)) {
616
+ return invalidRequest(`embedding_types must be a subset of ${[...EMBEDDING_TYPES].join(', ')}`);
617
+ }
618
+ }
619
+ }
620
+ if (params.input_type !== undefined && (typeof params.input_type !== 'string' || !INPUT_TYPES.has(params.input_type))) {
621
+ return invalidRequest(`input_type must be one of ${[...INPUT_TYPES].join(', ')}`);
622
+ }
623
+ if (params.truncate !== undefined && (typeof params.truncate !== 'string' || !TRUNCATE.has(params.truncate))) {
624
+ return invalidRequest(`truncate must be one of ${[...TRUNCATE].join(', ')}`);
625
+ }
626
+ if (params.output_dimension !== undefined) {
627
+ const d = params.output_dimension;
628
+ if (typeof d !== 'number' || !(COHERE_OUTPUT_DIMENSIONS as readonly number[]).includes(d)) {
629
+ return invalidRequest(`output_dimension must be one of ${COHERE_OUTPUT_DIMENSIONS.join(', ')}`);
630
+ }
631
+ }
632
+ return null;
633
+ }
634
+
635
+ /** `POST /v2/embed`. `model` AND `input_type` are both REQUIRED here — the v2 request type
636
+ * declares them non-optional, and this is the sharpest v1-vs-v2 difference. */
637
+ function handleEmbedV2(params: Record<string, unknown>): CohereResponseEnvelope {
638
+ const model = params.model;
639
+ if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
640
+ if (typeof params.input_type !== 'string' || params.input_type === '') return invalidRequest('input_type is required');
641
+ const shared = validateEmbedShared(params);
642
+ if (shared) return shared;
643
+ if (!modelServes(model, 'embed')) return modelNotFound(model);
644
+ const texts = collectEmbedInputs(params);
645
+ if (texts === null || texts.length === 0) return invalidRequest('one of texts, images or inputs is required');
646
+ const types = (Array.isArray(params.embedding_types) ? params.embedding_types : ['float']) as CohereEmbeddingType[];
647
+ const floats = embedVectors(texts, model, typeof params.output_dimension === 'number' ? params.output_dimension : undefined);
648
+ const inputTokens = texts.reduce((a, t) => a + estimateTokens(t), 0);
649
+ return {
650
+ status: 200,
651
+ body: {
652
+ id: cohereId(`embed-v2|${model}|${JSON.stringify(texts)}|${types.join(',')}`),
653
+ embeddings: embeddingsByType(floats, types),
654
+ texts,
655
+ response_type: 'embeddings_by_type',
656
+ // NO `output_tokens`: an embed call has no output side, and Cohere's embed `billed_units`
657
+ // carries only `input_tokens` (plus images/image_tokens for image inputs). Emitting a zero
658
+ // would be an invented key in a pack whose whole thesis is that it invents none.
659
+ meta: meta({ input_tokens: inputTokens }, { input_tokens: inputTokens }),
660
+ },
661
+ };
662
+ }
663
+
664
+ /**
665
+ * `POST /v1/embed`. Two things differ from v2 and BOTH are load-bearing:
666
+ * • `input_type` is optional in the schema — but a v3/v4 embed model still refuses without one,
667
+ * so the check moves from the schema to the handler;
668
+ * • the DEFAULT response is `embeddings_floats` — a FLAT `number[][]` at `embeddings`. Only when
669
+ * the caller asks for `embedding_types` does v1 answer the by-type shape. A twin that always
670
+ * answered by-type would break every default v1 caller.
671
+ */
672
+ function handleEmbedV1(params: Record<string, unknown>): CohereResponseEnvelope {
673
+ const model = typeof params.model === 'string' && params.model !== '' ? params.model : 'embed-english-v3.0';
674
+ const shared = validateEmbedShared(params);
675
+ if (shared) return shared;
676
+ if (!modelServes(model, 'embed')) return modelNotFound(model);
677
+ if (requiresInputType(model) && params.input_type === undefined) {
678
+ return invalidRequest(`input_type is required for model '${model}'`);
679
+ }
680
+ const texts = collectEmbedInputs(params);
681
+ if (texts === null || texts.length === 0) return invalidRequest('one of texts or images is required');
682
+ const floats = embedVectors(texts, model, typeof params.output_dimension === 'number' ? params.output_dimension : undefined);
683
+ const inputTokens = texts.reduce((a, t) => a + estimateTokens(t), 0);
684
+ const id = cohereId(`embed-v1|${model}|${JSON.stringify(texts)}`);
685
+ const m = meta({ input_tokens: inputTokens }, { input_tokens: inputTokens }); // embed has no output side
686
+ if (params.embedding_types === undefined) {
687
+ return { status: 200, body: { id, embeddings: floats, texts, response_type: 'embeddings_floats', meta: m } };
688
+ }
689
+ const types = params.embedding_types as CohereEmbeddingType[];
690
+ return { status: 200, body: { id, embeddings: embeddingsByType(floats, types), texts, response_type: 'embeddings_by_type', meta: m } };
691
+ }
692
+
693
+ // ════════════════════════════════════════════════════════════════════════════════════════
694
+ // RERANK
695
+ // ════════════════════════════════════════════════════════════════════════════════════════
696
+
697
+ function rankAll(query: string, docs: string[]): Array<{ index: number; relevance_score: number }> {
698
+ return docs
699
+ .map((d, index) => ({ index, relevance_score: rerankScore(query, d) }))
700
+ .sort((a, b) => b.relevance_score - a.relevance_score);
701
+ }
702
+
703
+ /** `POST /v2/rerank`. `documents` is `string[]` in v2 — an object element is refused. */
704
+ function handleRerankV2(params: Record<string, unknown>): CohereResponseEnvelope {
705
+ const model = params.model;
706
+ if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
707
+ const query = params.query;
708
+ if (typeof query !== 'string' || query === '') return invalidRequest('query is required');
709
+ const documents = params.documents;
710
+ if (!Array.isArray(documents) || documents.length === 0) return invalidRequest('documents is required');
711
+ for (const [i, d] of documents.entries()) {
712
+ if (typeof d !== 'string') return invalidRequest(`documents[${i}] must be a string`);
713
+ }
714
+ if (!modelServes(model, 'rerank')) return modelNotFound(model);
715
+ const docs = documents as string[];
716
+ const topN = typeof params.top_n === 'number' && params.top_n > 0 ? Math.min(params.top_n, docs.length) : docs.length;
717
+ const results = rankAll(query, docs).slice(0, topN);
718
+ return {
719
+ status: 200,
720
+ body: {
721
+ id: cohereId(`rerank-v2|${model}|${query}|${JSON.stringify(docs)}`),
722
+ results,
723
+ // Rerank is billed in SEARCH UNITS, not tokens — one per (query, up to 100 documents) call.
724
+ meta: meta({ search_units: 1 }),
725
+ },
726
+ };
727
+ }
728
+
729
+ /** `POST /v1/rerank`. Accepts object documents and honours `return_documents`, neither of which
730
+ * exists in v2. */
731
+ function handleRerankV1(params: Record<string, unknown>): CohereResponseEnvelope {
732
+ const query = params.query;
733
+ if (typeof query !== 'string' || query === '') return invalidRequest('query is required');
734
+ const documents = params.documents;
735
+ if (!Array.isArray(documents) || documents.length === 0) return invalidRequest('documents is required');
736
+ const model = typeof params.model === 'string' && params.model !== '' ? params.model : 'rerank-v3.5';
737
+ if (!modelServes(model, 'rerank')) return modelNotFound(model);
738
+ const rankFields = Array.isArray(params.rank_fields) ? (params.rank_fields as string[]) : undefined;
739
+ const texts = documents.map((d) => {
740
+ if (typeof d === 'string') return d;
741
+ const o = d as Record<string, unknown>;
742
+ if (rankFields) return rankFields.map((f) => String(o[f] ?? '')).join(' ');
743
+ if (typeof o.text === 'string') return o.text;
744
+ return JSON.stringify(o);
745
+ });
746
+ const topN = typeof params.top_n === 'number' && params.top_n > 0 ? Math.min(params.top_n, texts.length) : texts.length;
747
+ const ranked = rankAll(query, texts).slice(0, topN);
748
+ const returnDocuments = params.return_documents === true;
749
+ return {
750
+ status: 200,
751
+ body: {
752
+ id: cohereId(`rerank-v1|${model}|${query}|${JSON.stringify(texts)}`),
753
+ results: ranked.map((r) => (returnDocuments ? { ...r, document: { text: texts[r.index]! } } : r)),
754
+ meta: meta({ search_units: 1 }),
755
+ },
756
+ };
757
+ }
758
+
759
+ // ════════════════════════════════════════════════════════════════════════════════════════
760
+ // CLASSIFY / TOKENIZE / DETOKENIZE / CHECK-API-KEY
761
+ // ════════════════════════════════════════════════════════════════════════════════════════
762
+
763
+ function handleClassify(params: Record<string, unknown>): CohereResponseEnvelope {
764
+ const inputs = params.inputs;
765
+ if (!Array.isArray(inputs) || inputs.length === 0) return invalidRequest('inputs is required');
766
+ const examples = params.examples;
767
+ const preset = params.preset;
768
+ // Cohere's classify needs a way to know the label space: either training `examples` or a saved
769
+ // `preset`. With neither there is nothing to classify against, and it refuses.
770
+ if (!Array.isArray(examples) && typeof preset !== 'string') {
771
+ return invalidRequest('one of examples or preset is required');
772
+ }
773
+ let labels: string[];
774
+ if (Array.isArray(examples)) {
775
+ const seen: string[] = [];
776
+ for (const e of examples) {
777
+ const label = (e as { label?: unknown })?.label;
778
+ if (typeof label !== 'string' || label === '') return invalidRequest('each example requires a label');
779
+ if (!seen.includes(label)) seen.push(label);
780
+ }
781
+ // The vendor's documented minimum: a classifier needs at least two distinct classes.
782
+ if (seen.length < 2) return invalidRequest('at least 2 unique labels are required');
783
+ labels = seen;
784
+ } else {
785
+ // NEVER A FAKE SUCCESS. An earlier version invented `['positive','negative']` as the label space
786
+ // for ANY preset string, returning confident-looking classifications against labels the caller
787
+ // never named — the exact class the detokenize path treats as a hard rule and that
788
+ // `cohere.errors.unmodeled_ops_404` claims pack-wide. Presets are a filed todo
789
+ // (`cohere.classify.preset`); until they are modeled, an unmodeled op fails like the vendor.
790
+ return notFound(`preset '${String(preset)}' not found — this twin does not model saved classify presets`);
791
+ }
792
+ const texts = inputs.map((i) => String(i));
793
+ const classifications = texts.map((input, i) => {
794
+ const { prediction, confidences } = classifyText(input, labels);
795
+ const labelMap: Record<string, { confidence: number }> = {};
796
+ labels.forEach((l, j) => { labelMap[l] = { confidence: confidences[j]! }; });
797
+ return {
798
+ id: cohereId(`classify|${input}|${i}`),
799
+ input,
800
+ prediction,
801
+ predictions: [prediction],
802
+ confidence: Math.max(...confidences),
803
+ confidences: [Math.max(...confidences)],
804
+ labels: labelMap,
805
+ classification_type: 'single-label' as const,
806
+ };
807
+ });
808
+ return {
809
+ status: 200,
810
+ body: {
811
+ id: cohereId(`classify-call|${JSON.stringify(texts)}|${labels.join(',')}`),
812
+ classifications,
813
+ // Classify is billed per CLASSIFICATION, not per token.
814
+ meta: meta({ classifications: texts.length }),
815
+ },
816
+ };
817
+ }
818
+
819
+ /**
820
+ * `POST /v1/tokenize` — and the place the twin's vocabulary is LEARNED.
821
+ *
822
+ * The twin has no BPE vocabulary yet (`cohere.tokenize.bpe_vocabulary` is a todo), so it
823
+ * cannot invent a reverse mapping for `detokenize` out of thin air without fabricating. Instead
824
+ * tokenize OBSERVES each (id → segment) pair into the kernel log; detokenize folds that
825
+ * projection. The result is a twin that can only detokenize what it has genuinely seen — and says
826
+ * so, with a 400, otherwise.
827
+ */
828
+ async function handleTokenize(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
829
+ const text = params.text;
830
+ // Both fields are REQUIRED in `TokenizeRequest` — `model` is not optional here, unlike on
831
+ // /v1/embed and /v1/rerank.
832
+ if (typeof text !== 'string' || text === '') return invalidRequest('text is required');
833
+ const model = params.model;
834
+ if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
835
+ if (findModel(model) === undefined) return modelNotFound(model);
836
+
837
+ const segments = segmentText(text);
838
+ // Read the vocabulary ONCE, then resolve every segment against the same in-memory view. Reading
839
+ // per segment would make an id assigned earlier in THIS request invisible to a later one, so a
840
+ // text repeating a colliding pair would mint two ids for one string.
841
+ const vocab = new Map<string, number>();
842
+ const taken = new Map<number, string>();
843
+ for (const r of rows('token', req.root)) {
844
+ const seg = String(r.segment);
845
+ const tid = Number(r.token_id);
846
+ vocab.set(seg, tid);
847
+ taken.set(tid, seg);
848
+ }
849
+ const tokens: number[] = [];
850
+ const fresh: Array<{ id: number; segment: string }> = [];
851
+ for (const segment of segments) {
852
+ const existing = vocab.get(segment);
853
+ if (existing !== undefined) { tokens.push(existing); continue; }
854
+ const id = assignToken(segment, taken);
855
+ vocab.set(segment, id);
856
+ taken.set(id, segment);
857
+ fresh.push({ id, segment });
858
+ tokens.push(id);
859
+ }
860
+ for (const f of fresh) {
861
+ // One row per token, keyed by the id — the subject id IS the token id, so a re-observation of
862
+ // the same pair dedupes in the kernel rather than growing the log.
863
+ await write(req, 'token.observe', 'token', `tok_${f.id}`, { token_id: f.id, segment: f.segment });
864
+ }
865
+ const inputTokens = estimateTokens(text);
866
+ return {
867
+ status: 200,
868
+ body: { tokens, token_strings: segments, meta: meta({ input_tokens: inputTokens }, { input_tokens: inputTokens }) },
869
+ };
870
+ }
871
+
872
+ /**
873
+ * Resolve a segment's token id. The preferred id is a pure hash of the segment; when a DIFFERENT
874
+ * segment already holds it, probe upward until a free id is found.
875
+ *
876
+ * Without the probe the second segment would silently overwrite the first's row and `detokenize`
877
+ * would then return the WRONG text for every earlier caller — a dirty-state bug invisible to any
878
+ * fresh-root verify, which is exactly why `cohere.tokenize.collision_probe` builds the collision
879
+ * up deliberately.
880
+ */
881
+ function assignToken(segment: string, taken: Map<number, string>): number {
882
+ let id = preferredTokenId(segment);
883
+ // BOUNDED, like `nextId`. The id space is [1, 250_000]; an unbounded probe would spin FOREVER
884
+ // once the learned vocabulary filled it, hanging the server's fetch callback rather than failing
885
+ // (§9 round 1, finding 5). The bound is the occupied-set size + 1, so it can only be reached when
886
+ // every id genuinely is taken — and then it says so loudly.
887
+ for (let guard = 0; guard <= taken.size; guard++) {
888
+ if (!taken.has(id) || taken.get(id) === segment) return id;
889
+ id = (id % 250_000) + 1;
890
+ }
891
+ throw new Error('cohere: token vocabulary exhausted — every id in [1, 250000] is assigned to a different segment');
892
+ }
893
+
894
+ async function handleDetokenize(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
895
+ const tokens = params.tokens;
896
+ if (!Array.isArray(tokens)) return invalidRequest('tokens is required');
897
+ const model = params.model;
898
+ if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
899
+ if (findModel(model) === undefined) return modelNotFound(model);
900
+ const vocab = new Map<number, string>();
901
+ for (const r of rows('token', req.root)) vocab.set(Number(r.token_id), String(r.segment));
902
+ const parts: string[] = [];
903
+ for (const t of tokens) {
904
+ const seg = vocab.get(Number(t));
905
+ // NEVER a fake success. The twin's vocabulary is what it has observed; an id it has never
906
+ // issued is not something it can honestly decode, and silently dropping it (or emitting a
907
+ // placeholder) would hand the caller text the twin invented.
908
+ if (seg === undefined) return invalidRequest(`unknown token id ${String(t)} — this twin detokenizes only ids it has issued via POST /v1/tokenize`);
909
+ parts.push(seg);
910
+ }
911
+ const text = parts.join('');
912
+ return { status: 200, body: { text, meta: meta({ input_tokens: tokens.length }, { input_tokens: tokens.length }) } };
913
+ }
914
+
915
+ // ════════════════════════════════════════════════════════════════════════════════════════
916
+ // DATASETS (stateful)
917
+ // ════════════════════════════════════════════════════════════════════════════════════════
918
+
919
+ /** `DatasetType` — the vendor's CLOSED set (`api/types/DatasetType.d.ts`). A literal allowlist is
920
+ * an ORACLE here: the accepted values must biject exactly with this documented set. */
921
+ const DATASET_TYPES = new Set([
922
+ 'embed-input', 'embed-result', 'cluster-result', 'cluster-outliers',
923
+ 'reranker-finetune-input', 'single-label-classification-finetune-input',
924
+ 'chat-finetune-input', 'multi-label-classification-finetune-input',
925
+ 'batch-chat-input', 'batch-openai-chat-input', 'batch-embed-v2-input', 'batch-chat-v2-input',
926
+ ]);
927
+
928
+ /** `DatasetValidationStatus` — the vendor's closed set. */
929
+ const VALIDATION_STATUSES = new Set(['unknown', 'queued', 'processing', 'failed', 'validated', 'skipped']);
930
+
931
+ async function createDataset(params: Record<string, unknown>, query: URLSearchParams, req: CohereRequest): Promise<CohereResponseEnvelope> {
932
+ // `name` and `type` travel as QUERY parameters on this endpoint — the body is the multipart
933
+ // file. The server folds both into the handler's JSON contract, and the handler accepts either
934
+ // home so an in-process caller need not synthesize a query string.
935
+ const name = query.get('name') ?? (typeof params.name === 'string' ? params.name : '');
936
+ const type = query.get('type') ?? (typeof params.type === 'string' ? params.type : '');
937
+ if (!name) return invalidRequest('name is required');
938
+ if (!type) return invalidRequest('type is required');
939
+ if (!DATASET_TYPES.has(type)) return invalidRequest(`type must be one of ${[...DATASET_TYPES].join(', ')}`);
940
+ // The `data` file is REQUIRED: `datasets.create(data, evalData, request)` unconditionally does
941
+ // `_body.appendFile("data", data)` in cohere-ai@8.1.0, so the vendor never sees a fileless
942
+ // create. Accepting one would be an unmodeled input NOT failing like the vendor (§9 round 1,
943
+ // finding 6) — and it would have let the closed-DatasetType oracle be built on the lenient path.
944
+ const content = params.content;
945
+ if (typeof content !== 'string' || content === '') return invalidRequest('the data file is required');
946
+ const id = nextId('dataset', req.root);
947
+ const at = nowIso(req.occurredAt);
948
+ await write(req, 'dataset.create', 'dataset', id, {
949
+ // THE TWIN MINTED THIS ID, and the vendor has never heard of it. The connector reads this
950
+ // marker to refuse pushing an update/cancel/delete against a subject that has no vendor
951
+ // identity — without it a local delete fires the twin's own UUID at the REAL account
952
+ // (§9 round 1, connector B1/M1). It is underscore-prefixed, so no served view carries it, and
953
+ // the push-confirm strips it when it rebuilds the row under the vendor's id.
954
+ _twin_minted: true,
955
+ name,
956
+ dataset_type: type,
957
+ created_at: at,
958
+ updated_at: at,
959
+ validation_status: 'validated',
960
+ validation_error: null,
961
+ validation_warnings: [],
962
+ required_fields: [],
963
+ preserve_fields: [],
964
+ dataset_parts: [],
965
+ _content: content,
966
+ });
967
+ // Cohere's create answers ONLY `{ id }` — not the dataset. A twin returning the whole object
968
+ // here would be more "helpful" and less faithful.
969
+ return { status: 200, body: { id } };
970
+ }
971
+
972
+ function datasetView(r: Record<string, unknown>): Record<string, unknown> {
973
+ return { id: r.id, ...strip(r) };
974
+ }
975
+
976
+ // ════════════════════════════════════════════════════════════════════════════════════════
977
+ // CONNECTORS (stateful)
978
+ // ════════════════════════════════════════════════════════════════════════════════════════
979
+
980
+ async function createConnector(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
981
+ const name = params.name;
982
+ if (typeof name !== 'string' || name === '') return invalidRequest('name is required');
983
+ const url = params.url;
984
+ if (typeof url !== 'string' || url === '') return invalidRequest('url is required');
985
+ const id = nextId('connector', req.root);
986
+ const at = nowIso(req.occurredAt);
987
+ const oauth = params.oauth as Record<string, unknown> | undefined;
988
+ await write(req, 'connector.create', 'connector', id, {
989
+ _twin_minted: true, // see the note on dataset.create — the connector refuses to push a
990
+ _rev: 1, // twin-minted subject's later mutations at a real account.
991
+ organization_id: 'org_twin',
992
+ name,
993
+ description: typeof params.description === 'string' ? params.description : null,
994
+ url,
995
+ created_at: at,
996
+ updated_at: at,
997
+ excludes: Array.isArray(params.excludes) ? params.excludes : [],
998
+ // `auth_type` is derived, not caller-supplied — it reports HOW the connector authenticates.
999
+ auth_type: oauth ? 'oauth' : params.service_auth ? 'service_auth' : 'none',
1000
+ oauth: oauth ? { client_id: String(oauth.client_id ?? ''), authorize_url: String(oauth.authorize_url ?? ''), token_url: String(oauth.token_url ?? ''), scope: oauth.scope ?? null } : null,
1001
+ auth_status: oauth ? 'expired' : 'valid',
1002
+ active: params.active === undefined ? true : params.active === true,
1003
+ continue_on_failure: params.continue_on_failure === true,
1004
+ });
1005
+ const row = getRow('connector', id, req.root)!;
1006
+ return { status: 200, body: { connector: connectorView(row) } };
1007
+ }
1008
+
1009
+ function connectorView(r: Record<string, unknown>): Record<string, unknown> {
1010
+ return { id: r.id, ...strip(r) };
1011
+ }
1012
+
1013
+ // ════════════════════════════════════════════════════════════════════════════════════════
1014
+ // EMBED JOBS (stateful)
1015
+ // ════════════════════════════════════════════════════════════════════════════════════════
1016
+
1017
+ const EMBED_JOB_STATUSES = new Set(['processing', 'complete', 'cancelling', 'cancelled', 'failed']);
1018
+
1019
+ async function createEmbedJob(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
1020
+ const model = params.model;
1021
+ if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
1022
+ const datasetId = params.dataset_id;
1023
+ if (typeof datasetId !== 'string' || datasetId === '') return invalidRequest('dataset_id is required');
1024
+ const inputType = params.input_type;
1025
+ if (typeof inputType !== 'string' || inputType === '') return invalidRequest('input_type is required');
1026
+ if (!INPUT_TYPES.has(inputType)) return invalidRequest(`input_type must be one of ${[...INPUT_TYPES].join(', ')}`);
1027
+ if (params.truncate !== undefined && (typeof params.truncate !== 'string' || !TRUNCATE.has(params.truncate))) {
1028
+ return invalidRequest(`truncate must be one of ${[...TRUNCATE].join(', ')}`);
1029
+ }
1030
+ if (!modelServes(model, 'embed')) return modelNotFound(model);
1031
+ // The job's input must be a dataset that actually exists — a cross-resource invariant the
1032
+ // vendor enforces and a twin that skipped it would let a caller queue work over nothing.
1033
+ if (getRow('dataset', datasetId, req.root) === undefined) return notFound(`dataset '${datasetId}' not found`);
1034
+ const id = nextId('embed_job', req.root);
1035
+ await write(req, 'embed_job.create', 'embed_job', id, {
1036
+ _twin_minted: true, // see the note on dataset.create
1037
+ job_id: id,
1038
+ name: typeof params.name === 'string' ? params.name : null,
1039
+ status: 'processing',
1040
+ created_at: nowIso(req.occurredAt),
1041
+ input_dataset_id: datasetId,
1042
+ output_dataset_id: null,
1043
+ model,
1044
+ truncate: typeof params.truncate === 'string' ? params.truncate : 'END',
1045
+ });
1046
+ // Cohere's create answers `{ job_id, meta }` — not the job object.
1047
+ return { status: 200, body: { job_id: id, meta: meta({}) } };
1048
+ }
1049
+
1050
+ function embedJobView(r: Record<string, unknown>): Record<string, unknown> {
1051
+ const v = strip(r);
1052
+ delete v.id; // the vendor's key for an embed job is `job_id`, and there is no `id` alongside it.
1053
+ return v;
1054
+ }
1055
+
1056
+ // ════════════════════════════════════════════════════════════════════════════════════════
1057
+ // THE ROUTER CENSUS
1058
+ // ════════════════════════════════════════════════════════════════════════════════════════
1059
+
1060
+ /**
1061
+ * Every method/path pair the dispatch below branches on. HAND-AUTHORED (the honest limit): a
1062
+ * branch added to the handler and to neither this list nor the conformance snapshot is invisible
1063
+ * to the conformance check's third direction. The mechanical backstop for that lives in
1064
+ * `cohere-twin.test.ts` ("covers every literal path the handler dispatches on"), which reads this
1065
+ * module's own source; segment-matched sub-routes still rest on review.
1066
+ *
1067
+ * `{id}` stands for one path segment.
1068
+ */
1069
+ export const COHERE_ROUTER_SURFACE: ReadonlyArray<{ method: string; path: string }> = [
1070
+ { method: 'POST', path: '/v2/chat' },
1071
+ { method: 'POST', path: '/v2/embed' },
1072
+ { method: 'POST', path: '/v2/rerank' },
1073
+ { method: 'POST', path: '/v1/chat' },
1074
+ { method: 'POST', path: '/v1/embed' },
1075
+ { method: 'POST', path: '/v1/rerank' },
1076
+ { method: 'POST', path: '/v1/classify' },
1077
+ { method: 'POST', path: '/v1/tokenize' },
1078
+ { method: 'POST', path: '/v1/detokenize' },
1079
+ { method: 'POST', path: '/v1/check-api-key' },
1080
+ { method: 'GET', path: '/v1/models' },
1081
+ { method: 'GET', path: '/v1/models/{id}' },
1082
+ { method: 'POST', path: '/v1/datasets' },
1083
+ { method: 'GET', path: '/v1/datasets' },
1084
+ { method: 'GET', path: '/v1/datasets/usage' },
1085
+ { method: 'GET', path: '/v1/datasets/{id}' },
1086
+ { method: 'DELETE', path: '/v1/datasets/{id}' },
1087
+ { method: 'POST', path: '/v1/connectors' },
1088
+ { method: 'GET', path: '/v1/connectors' },
1089
+ { method: 'GET', path: '/v1/connectors/{id}' },
1090
+ { method: 'PATCH', path: '/v1/connectors/{id}' },
1091
+ { method: 'DELETE', path: '/v1/connectors/{id}' },
1092
+ { method: 'POST', path: '/v1/connectors/{id}/oauth/authorize' },
1093
+ { method: 'POST', path: '/v1/embed-jobs' },
1094
+ { method: 'GET', path: '/v1/embed-jobs' },
1095
+ { method: 'GET', path: '/v1/embed-jobs/{id}' },
1096
+ { method: 'POST', path: '/v1/embed-jobs/{id}/cancel' },
1097
+ ];
1098
+
1099
+ // ════════════════════════════════════════════════════════════════════════════════════════
1100
+ // DISPATCH
1101
+ // ════════════════════════════════════════════════════════════════════════════════════════
1102
+
1103
+ export async function handleCohereTwinRequest(req: CohereRequest): Promise<CohereResponseEnvelope> {
1104
+ const method = req.method.toUpperCase();
1105
+ const [rawPath, rawQuery = ''] = req.path.split('?');
1106
+ const path = (rawPath ?? '').replace(/\/+$/, '') || '/';
1107
+ const query = new URLSearchParams(rawQuery);
1108
+ const seg = path.split('/').filter(Boolean);
1109
+ const params = parseJson(req.body);
1110
+ const readOnly = req.readOnly === true;
1111
+
1112
+ const authFailure = checkAuth(req);
1113
+ if (authFailure) return authFailure;
1114
+
1115
+ const isWrite = method === 'POST' || method === 'DELETE' || method === 'PATCH' || method === 'PUT';
1116
+ if (readOnly && isWrite) return readOnlyRefusal();
1117
+
1118
+ // ── chat ──────────────────────────────────────────────────────────────────────────────
1119
+ if (method === 'POST' && path === '/v2/chat') {
1120
+ const validated = validateChatV2(params);
1121
+ if ('error' in validated) return validated.error;
1122
+ const args = validated.args;
1123
+ // ONE engine advance, here, before anything branches on it.
1124
+ const outcome = req.scenarioEngine ? scenarioDecision(args, req.scenarioEngine) : {};
1125
+ if (outcome.error) return outcome.error;
1126
+ if (args.stream) {
1127
+ if (!req.sseSink) return invalidRequest('stream:true requires a streaming transport');
1128
+ return { status: 200, body: streamChatV2(args, req.sseSink, outcome) };
1129
+ }
1130
+ return { status: 200, body: buildChatV2(args, outcome) };
1131
+ }
1132
+ if (method === 'POST' && path === '/v1/chat') return handleChatV1(params, req);
1133
+
1134
+ // ── embed / rerank / classify ─────────────────────────────────────────────────────────
1135
+ if (method === 'POST' && path === '/v2/embed') return handleEmbedV2(params);
1136
+ if (method === 'POST' && path === '/v1/embed') return handleEmbedV1(params);
1137
+ if (method === 'POST' && path === '/v2/rerank') return handleRerankV2(params);
1138
+ if (method === 'POST' && path === '/v1/rerank') return handleRerankV1(params);
1139
+ if (method === 'POST' && path === '/v1/classify') return handleClassify(params);
1140
+
1141
+ // ── tokenizer ─────────────────────────────────────────────────────────────────────────
1142
+ if (method === 'POST' && path === '/v1/tokenize') return handleTokenize(params, req);
1143
+ if (method === 'POST' && path === '/v1/detokenize') return handleDetokenize(params, req);
1144
+
1145
+ // ── auth probe. A POST despite the name — `Client.js` sends `method: "POST"`. ──────────
1146
+ if (method === 'POST' && path === '/v1/check-api-key') {
1147
+ return { status: 200, body: { valid: true, organization_id: 'org_twin', owner_id: 'owner_twin' } };
1148
+ }
1149
+
1150
+ // ── models ────────────────────────────────────────────────────────────────────────────
1151
+ if (method === 'GET' && path === '/v1/models') {
1152
+ const endpoint = query.get('endpoint');
1153
+ let models = COHERE_MODELS;
1154
+ if (endpoint) models = models.filter((m) => m.endpoints.includes(endpoint as never));
1155
+ const pageSize = Number(query.get('page_size') ?? '0');
1156
+ if (Number.isFinite(pageSize) && pageSize > 0) models = models.slice(0, pageSize);
1157
+ return { status: 200, body: { models, next_page_token: null } };
1158
+ }
1159
+ if (method === 'GET' && seg[0] === 'v1' && seg[1] === 'models' && seg.length === 3) {
1160
+ const name = decodeURIComponent(seg[2]!);
1161
+ const card = findModel(name);
1162
+ return card ? { status: 200, body: card } : modelNotFound(name);
1163
+ }
1164
+
1165
+ // ── datasets ──────────────────────────────────────────────────────────────────────────
1166
+ if (method === 'POST' && path === '/v1/datasets') return createDataset(params, query, req);
1167
+ if (method === 'GET' && path === '/v1/datasets') {
1168
+ let items = rows('dataset', req.root);
1169
+ const dt = query.get('datasetType');
1170
+ if (dt) {
1171
+ // The SAME closed set the create path enforces. Real Cohere serializes this parameter through
1172
+ // `serializers.DatasetType` and 400s on a miss; answering `200 {datasets:[]}` would report an
1173
+ // empty account for what is actually a rejected request (§9 round two, MINOR 9).
1174
+ if (!DATASET_TYPES.has(dt)) return invalidRequest(`datasetType must be one of ${[...DATASET_TYPES].join(', ')}`);
1175
+ items = items.filter((r) => r.dataset_type === dt);
1176
+ }
1177
+ const vs = query.get('validationStatus');
1178
+ if (vs) {
1179
+ if (!VALIDATION_STATUSES.has(vs)) return invalidRequest(`validationStatus must be one of ${[...VALIDATION_STATUSES].join(', ')}`);
1180
+ items = items.filter((r) => r.validation_status === vs);
1181
+ }
1182
+ const limit = Number(query.get('limit') ?? '0');
1183
+ const offset = Number(query.get('offset') ?? '0');
1184
+ if (Number.isFinite(offset) && offset > 0) items = items.slice(offset);
1185
+ if (Number.isFinite(limit) && limit > 0) items = items.slice(0, limit);
1186
+ return { status: 200, body: { datasets: items.map(datasetView) } };
1187
+ }
1188
+ // `/usage` is a LITERAL sibling of `/{id}` and must be matched first, or a dataset could never
1189
+ // be named `usage` and the usage endpoint would be shadowed by the id lookup.
1190
+ if (method === 'GET' && path === '/v1/datasets/usage') {
1191
+ const used = rows('dataset', req.root).reduce((a, r) => a + String(r._content ?? '').length, 0);
1192
+ return { status: 200, body: { organization_usage: used } };
1193
+ }
1194
+ if (seg[0] === 'v1' && seg[1] === 'datasets' && seg.length === 3) {
1195
+ const id = decodeURIComponent(seg[2]!);
1196
+ const row = getRow('dataset', id, req.root);
1197
+ if (method === 'GET') return row ? { status: 200, body: { dataset: datasetView(row) } } : notFound(`dataset '${id}' not found`);
1198
+ if (method === 'DELETE') {
1199
+ if (!row) return notFound(`dataset '${id}' not found`);
1200
+ await write(req, 'dataset.delete', 'dataset', id, { _deleted: true });
1201
+ // Cohere's delete answers an EMPTY object (`Record<string, unknown>` in the SDK), not the
1202
+ // `{deleted:true}` envelope most vendors send.
1203
+ return { status: 200, body: {} };
1204
+ }
1205
+ }
1206
+
1207
+ // ── connectors ────────────────────────────────────────────────────────────────────────
1208
+ if (method === 'POST' && path === '/v1/connectors') return createConnector(params, req);
1209
+ if (method === 'GET' && path === '/v1/connectors') {
1210
+ let items = rows('connector', req.root);
1211
+ const total = items.length;
1212
+ const offset = Number(query.get('offset') ?? '0');
1213
+ const limit = Number(query.get('limit') ?? '0');
1214
+ if (Number.isFinite(offset) && offset > 0) items = items.slice(offset);
1215
+ if (Number.isFinite(limit) && limit > 0) items = items.slice(0, limit);
1216
+ return { status: 200, body: { connectors: items.map(connectorView), total_count: total } };
1217
+ }
1218
+ if (seg[0] === 'v1' && seg[1] === 'connectors' && seg.length === 3) {
1219
+ const id = decodeURIComponent(seg[2]!);
1220
+ const row = getRow('connector', id, req.root);
1221
+ if (method === 'GET') return row ? { status: 200, body: { connector: connectorView(row) } } : notFound(`connector '${id}' not found`);
1222
+ if (method === 'PATCH') {
1223
+ if (!row) return notFound(`connector '${id}' not found`);
1224
+ const patch: Record<string, unknown> = { updated_at: nowIso(req.occurredAt), _rev: nextRev('connector', id, req.root) };
1225
+ for (const k of ['name', 'url', 'excludes', 'active', 'continue_on_failure'] as const) {
1226
+ if (params[k] !== undefined) patch[k] = params[k];
1227
+ }
1228
+ await write(req, 'connector.update', 'connector', id, patch);
1229
+ return { status: 200, body: { connector: connectorView(getRow('connector', id, req.root)!) } };
1230
+ }
1231
+ if (method === 'DELETE') {
1232
+ if (!row) return notFound(`connector '${id}' not found`);
1233
+ await write(req, 'connector.delete', 'connector', id, { _deleted: true });
1234
+ return { status: 200, body: {} };
1235
+ }
1236
+ }
1237
+ if (method === 'POST' && seg[0] === 'v1' && seg[1] === 'connectors' && seg.length === 5 && seg[3] === 'oauth' && seg[4] === 'authorize') {
1238
+ const id = decodeURIComponent(seg[2]!);
1239
+ const row = getRow('connector', id, req.root);
1240
+ if (!row) return notFound(`connector '${id}' not found`);
1241
+ // A connector with no OAuth configuration has nothing to authorize. Answering a fabricated
1242
+ // redirect URL would be a fake success on a path the vendor refuses.
1243
+ if (row.auth_type !== 'oauth') return invalidRequest(`connector '${id}' is not configured for oauth`);
1244
+ const oauth = row.oauth as { authorize_url?: string; client_id?: string } | null;
1245
+ const after = query.get('after_token_redirect');
1246
+ const u = new URL(String(oauth?.authorize_url || 'https://auth.example.test/authorize'));
1247
+ u.searchParams.set('client_id', String(oauth?.client_id ?? ''));
1248
+ u.searchParams.set('state', cohereId(`oauth|${id}`));
1249
+ if (after) u.searchParams.set('after_token_redirect', after);
1250
+ return { status: 200, body: { redirect_url: u.toString() } };
1251
+ }
1252
+
1253
+ // ── embed jobs ────────────────────────────────────────────────────────────────────────
1254
+ if (method === 'POST' && path === '/v1/embed-jobs') return createEmbedJob(params, req);
1255
+ if (method === 'GET' && path === '/v1/embed-jobs') {
1256
+ return { status: 200, body: { embed_jobs: rows('embed_job', req.root).map(embedJobView) } };
1257
+ }
1258
+ if (method === 'GET' && seg[0] === 'v1' && seg[1] === 'embed-jobs' && seg.length === 3) {
1259
+ const id = decodeURIComponent(seg[2]!);
1260
+ const row = getRow('embed_job', id, req.root);
1261
+ return row ? { status: 200, body: embedJobView(row) } : notFound(`embed job '${id}' not found`);
1262
+ }
1263
+ if (method === 'POST' && seg[0] === 'v1' && seg[1] === 'embed-jobs' && seg.length === 4 && seg[3] === 'cancel') {
1264
+ const id = decodeURIComponent(seg[2]!);
1265
+ const row = getRow('embed_job', id, req.root);
1266
+ if (!row) return notFound(`embed job '${id}' not found`);
1267
+ // The vendor's state machine: only a job still in flight can be cancelled. A terminal job
1268
+ // answers a 400 rather than pretending the cancel took.
1269
+ if (row.status !== 'processing') return invalidRequest(`embed job '${id}' is ${String(row.status)} and cannot be cancelled`);
1270
+ await write(req, 'embed_job.cancel', 'embed_job', id, { status: 'cancelling' });
1271
+ // `embedJobs.cancel` is declared `-> void` in the SDK: an empty 200 body.
1272
+ return { status: 200, body: {} };
1273
+ }
1274
+
1275
+ return routeNotFound(method, path);
1276
+ }
1277
+
1278
+ /** Re-exported so tests can assert the twin's closed sets against the vendor's without importing
1279
+ * the private constants by name. */
1280
+ export const COHERE_CLOSED_SETS = {
1281
+ datasetTypes: [...DATASET_TYPES],
1282
+ validationStatuses: [...VALIDATION_STATUSES],
1283
+ embedJobStatuses: [...EMBED_JOB_STATUSES],
1284
+ toolChoices: [...TOOL_CHOICES],
1285
+ safetyModes: [...SAFETY_MODES],
1286
+ v2Roles: [...V2_ROLES],
1287
+ } as const;
1288
+
1289
+ /** Exported for the connector's shared FNV seed and for tests that need the twin's id derivation. */
1290
+ export { fnv1a };