@volter/twin-cohere 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +224 -0
- package/defaults/handlers.json +26 -0
- package/dist/defaults/handlers.json +26 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +31 -0
- package/dist/src/cohere-budget.d.ts +55 -0
- package/dist/src/cohere-budget.js +171 -0
- package/dist/src/cohere-capabilities.d.ts +14 -0
- package/dist/src/cohere-capabilities.js +1852 -0
- package/dist/src/cohere-conformance.d.ts +17 -0
- package/dist/src/cohere-conformance.js +464 -0
- package/dist/src/cohere-connector.d.ts +150 -0
- package/dist/src/cohere-connector.js +625 -0
- package/dist/src/cohere-models.d.ts +21 -0
- package/dist/src/cohere-models.js +73 -0
- package/dist/src/cohere-scenario.d.ts +57 -0
- package/dist/src/cohere-scenario.js +176 -0
- package/dist/src/cohere-server.d.ts +16 -0
- package/dist/src/cohere-server.js +184 -0
- package/dist/src/cohere-stub.d.ts +119 -0
- package/dist/src/cohere-stub.js +321 -0
- package/dist/src/cohere-twin.d.ts +82 -0
- package/dist/src/cohere-twin.js +1243 -0
- package/dist/src/cohere-types.d.ts +226 -0
- package/dist/src/cohere-types.js +40 -0
- package/dist/src/index.d.ts +15 -0
- package/dist/src/index.js +84 -0
- package/package.json +71 -0
- package/src/cli.ts +30 -0
- package/src/cohere-budget.ts +197 -0
- package/src/cohere-capabilities.ts +1855 -0
- package/src/cohere-conformance.ts +489 -0
- package/src/cohere-connector.ts +709 -0
- package/src/cohere-models.ts +79 -0
- package/src/cohere-scenario.ts +194 -0
- package/src/cohere-server.ts +195 -0
- package/src/cohere-stub.ts +337 -0
- package/src/cohere-twin.ts +1290 -0
- package/src/cohere-types.ts +231 -0
- package/src/index.ts +159 -0
|
@@ -0,0 +1,1290 @@
|
|
|
1
|
+
// Cohere twin REQUEST HANDLER — the canonical Cohere API surface for the twin. Contract:
|
|
2
|
+
// handleCohereTwinRequest({method, path, body}) -> {status, body}. It is the faithful Cohere API
|
|
3
|
+
// that the real `cohere-ai` client (constructed with `environment: 'http://127.0.0.1:<port>'`) and
|
|
4
|
+
// the real `@ai-sdk/cohere` provider (`createCohere({ baseURL: '…/v2' })`) talk to UNMODIFIED.
|
|
5
|
+
//
|
|
6
|
+
// ── COHERE IS ITS OWN DIALECT, AND THE REFUSALS ARE THE FIDELITY SURFACE ────────────────
|
|
7
|
+
// This is NOT an OpenAI-compatible API and copying an OpenAI-shaped pack's permissiveness is
|
|
8
|
+
// precisely the bug (ADDING_A_TWIN.md §0). What distinguishes Cohere is what it REFUSES and what
|
|
9
|
+
// its envelopes look like:
|
|
10
|
+
// • the error body is a BARE `{ "message": "…" }` — no `error` wrapper, no `type`, no `code`
|
|
11
|
+
// (@ai-sdk/cohere's `cohereErrorDataSchema` is literally `z.object({ message: z.string() })`);
|
|
12
|
+
// • `finish_reason` is UPPER-CASE from Cohere's own set (COMPLETE / TOOL_CALL / MAX_TOKENS /
|
|
13
|
+
// STOP_SEQUENCE / ERROR / TIMEOUT) — `stop` and `tool_calls` are not Cohere values;
|
|
14
|
+
// • a v2 chat response has NO `choices` array: one `message`, one `finish_reason`, one `usage`
|
|
15
|
+
// with `billed_units` AND `tokens` sub-objects;
|
|
16
|
+
// • v1 and v2 are DIFFERENT PROTOCOLS on one host, not a versioned alias — v1 chat takes a
|
|
17
|
+
// single `message` STRING and answers a flat `text`; v2 takes `messages[]` and answers a
|
|
18
|
+
// structured assistant message;
|
|
19
|
+
// • `POST /v2/embed` REQUIRES `input_type`; `POST /v1/embed` does not, but refuses a v3/v4 embed
|
|
20
|
+
// model without one;
|
|
21
|
+
// • v1 embed defaults to `embeddings_floats` (a flat `number[][]`) while v2 always answers
|
|
22
|
+
// `embeddings_by_type` (keyed by `float`/`int8`/…);
|
|
23
|
+
// • v2 rerank `documents` must be STRINGS; v1 accepts objects and has `return_documents`;
|
|
24
|
+
// • `POST /v1/check-api-key` is a POST, not the GET its name suggests;
|
|
25
|
+
// • the enums are UPPER-CASE (`truncate: NONE|START|END`, `tool_choice: REQUIRED|NONE`,
|
|
26
|
+
// `safety_mode: CONTEXTUAL|STRICT|OFF`) where most vendors' are lower.
|
|
27
|
+
// Every one of those is asserted by a manifest verify, so an accidental drift toward the
|
|
28
|
+
// OpenAI shape reddens by name.
|
|
29
|
+
//
|
|
30
|
+
// ── THE HONEST DESIGN ──────────────────────────────────────────────────────────────────
|
|
31
|
+
// The twin cannot run the model, so `/v2/chat`, `/v1/chat`, `/v2/embed`, `/v1/embed`,
|
|
32
|
+
// `/v2/rerank`, `/v1/rerank`, `/v1/classify` and `/v1/tokenize` return DETERMINISTIC STUBS
|
|
33
|
+
// (cohere-stub.ts) clearly labeled as such — never real model output. But the ENTIRE PROTOCOL
|
|
34
|
+
// ENVELOPE is vendor-faithful, including the SSE event sequence terminated by `data: [DONE]`.
|
|
35
|
+
//
|
|
36
|
+
// The genuinely stateful + static surface is real, not stubbed:
|
|
37
|
+
// • GET /v1/models (+ /{name}) — the static real catalog (cohere-models.ts)
|
|
38
|
+
// • Datasets — POST/GET/DELETE /v1/datasets (+ /usage, /{id}), stateful
|
|
39
|
+
// • Connectors — POST/GET/PATCH/DELETE /v1/connectors (+ /{id}/oauth/authorize), stateful
|
|
40
|
+
// • Embed jobs — POST/GET /v1/embed-jobs (+ /{id}, /{id}/cancel), stateful, cross-referencing
|
|
41
|
+
// a real dataset id
|
|
42
|
+
// • The tokenizer VOCABULARY — `/v1/tokenize` observes (segment → id) pairs into the log, and
|
|
43
|
+
// `/v1/detokenize` folds that projection. That is what makes detokenize honest rather than a
|
|
44
|
+
// fabrication: the twin can only detokenize what it has actually seen, and says so otherwise.
|
|
45
|
+
//
|
|
46
|
+
// State lives in the kernel action log (D1): all writes are local actions, reads are the
|
|
47
|
+
// projection. No real Cohere is ever called from this path (D4). Streaming uses an INJECTED sink
|
|
48
|
+
// — no real sockets / setTimeout (D5 verify is offline + deterministic).
|
|
49
|
+
//
|
|
50
|
+
// ID MINTING: deterministic, and the ordinal is scanned from the id-SET INCLUDING TOMBSTONES
|
|
51
|
+
// (`_deleted` rows are retained in the projection and filtered out of reads) so a
|
|
52
|
+
// delete-then-create can never re-issue a live id — the failure mode three packs' §9 reviews each
|
|
53
|
+
// found in a count-mint.
|
|
54
|
+
import { applyTwinWrite, projectResources } from '@volter/world-core';
|
|
55
|
+
import { COHERE_MODELS, findModel, modelServes } from './cohere-models.ts';
|
|
56
|
+
import {
|
|
57
|
+
base64Embedding,
|
|
58
|
+
classifyText,
|
|
59
|
+
COHERE_OUTPUT_DIMENSIONS,
|
|
60
|
+
cohereId,
|
|
61
|
+
contentToText,
|
|
62
|
+
countInputTokens,
|
|
63
|
+
embedDimensions,
|
|
64
|
+
estimateTokens,
|
|
65
|
+
fnv1a,
|
|
66
|
+
lastUserText,
|
|
67
|
+
preferredTokenId,
|
|
68
|
+
pseudoEmbedding,
|
|
69
|
+
quantizeEmbedding,
|
|
70
|
+
rerankScore,
|
|
71
|
+
segmentText,
|
|
72
|
+
stubAssistantText,
|
|
73
|
+
stubToolCall,
|
|
74
|
+
stubToolPlan,
|
|
75
|
+
stubV1Text,
|
|
76
|
+
toolNames,
|
|
77
|
+
} from './cohere-stub.ts';
|
|
78
|
+
import { type CohereScenarioEngine, type CohereScenarioRespond, realizeCohereRespond, type ScriptedResult } from './cohere-scenario.ts';
|
|
79
|
+
import {
|
|
80
|
+
COHERE_EMBEDDING_TYPES,
|
|
81
|
+
COHERE_INPUT_TYPES,
|
|
82
|
+
COHERE_TRUNCATE,
|
|
83
|
+
type CohereApiMeta,
|
|
84
|
+
type CohereAssistantContentItem,
|
|
85
|
+
type CohereChatV2Response,
|
|
86
|
+
type CohereEmbeddingType,
|
|
87
|
+
type CohereFinishReason,
|
|
88
|
+
type CohereV1FinishReason,
|
|
89
|
+
type CohereMessageV2,
|
|
90
|
+
type CohereToolCallV2,
|
|
91
|
+
type CohereUsage,
|
|
92
|
+
type SseSink,
|
|
93
|
+
} from './cohere-types.ts';
|
|
94
|
+
|
|
95
|
+
const SERVICE = 'cohere';
|
|
96
|
+
|
|
97
|
+
/** The API version string Cohere stamps into every `meta.api_version`. */
|
|
98
|
+
const API_VERSION = '1';
|
|
99
|
+
|
|
100
|
+
export type CohereRequest = {
|
|
101
|
+
/** The scenario engine (kernel grammar + this pack's vocabulary) — scripts chat turns. */
|
|
102
|
+
scenarioEngine?: CohereScenarioEngine;
|
|
103
|
+
method: string;
|
|
104
|
+
path: string;
|
|
105
|
+
body?: string;
|
|
106
|
+
occurredAt?: string;
|
|
107
|
+
root?: string;
|
|
108
|
+
readOnly?: boolean;
|
|
109
|
+
/** Lower-cased request headers (e.g. `authorization`) the HTTP server passes through so the
|
|
110
|
+
* handler can model auth (401). In-process trusted calls (capability verifies, the connector)
|
|
111
|
+
* omit them and are not auth-gated — the twin cannot validate against real keys, so the
|
|
112
|
+
* modeled failure is the CHECKABLE missing/sentinel-invalid case. */
|
|
113
|
+
headers?: Record<string, string>;
|
|
114
|
+
/** When set on a streaming POST, chunks are written here (no sockets). */
|
|
115
|
+
sseSink?: SseSink;
|
|
116
|
+
};
|
|
117
|
+
|
|
118
|
+
/** The handler response. `headers` (when present) are response headers the HTTP server should
|
|
119
|
+
* set — e.g. `Retry-After` on a scripted 429. */
|
|
120
|
+
export type CohereResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
|
|
121
|
+
|
|
122
|
+
// ── vendor-shaped errors ────────────────────────────────────────────────────────────────
|
|
123
|
+
/**
|
|
124
|
+
* THE error envelope. One key, `message`, and nothing else — the shape
|
|
125
|
+
* @ai-sdk/cohere@4.0.35 pins with `cohereErrorDataSchema = z.object({ message: z.string() })`,
|
|
126
|
+
* and the shape cohere-ai@8.1.0 hands to every typed error class (`new Cohere.BadRequestError(
|
|
127
|
+
* _response.error.body)`) without unwrapping.
|
|
128
|
+
*
|
|
129
|
+
* The STATUS codes below are the vendor's; the message TEXT for the twin's own refusals follows
|
|
130
|
+
* Cohere's `invalid request: …` idiom but is authored here — a twin cannot know the vendor's exact
|
|
131
|
+
* prose for every input, and inventing a `type`/`code` field to look more official would be
|
|
132
|
+
* serving surface the vendor does not have.
|
|
133
|
+
*/
|
|
134
|
+
function apiError(status: number, message: string): CohereResponseEnvelope {
|
|
135
|
+
return { status, body: { message } };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** `invalid request: …` — Cohere's 400 idiom for a malformed body. */
|
|
139
|
+
function invalidRequest(detail: string): CohereResponseEnvelope {
|
|
140
|
+
return apiError(400, `invalid request: ${detail}`);
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* The vendor's 404 for a model the endpoint cannot serve. Cohere answers the same way for a model
|
|
145
|
+
* that does not exist AND for one that exists but is not available on the endpoint being called —
|
|
146
|
+
* from the endpoint's point of view the model is simply not found — so the twin does not invent a
|
|
147
|
+
* separate "incompatible model" status.
|
|
148
|
+
*/
|
|
149
|
+
function modelNotFound(model: string): CohereResponseEnvelope {
|
|
150
|
+
return apiError(404, `model '${model}' not found, make sure the correct model ID was used and that you have access to the model.`);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
function notFound(message: string): CohereResponseEnvelope {
|
|
154
|
+
return apiError(404, message);
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/** The 404 an unrouted path answers. Written as one function so conformance can assert the
|
|
158
|
+
* envelope as a literal predicate without importing this. */
|
|
159
|
+
function routeNotFound(method: string, path: string): CohereResponseEnvelope {
|
|
160
|
+
return apiError(404, `not found: ${method} ${path}`);
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/** Read-only mode refuses every mutation (D3). Cohere has no 405 idiom of its own, so the twin
|
|
164
|
+
* answers its own `{message}` envelope with the standard 405 status. */
|
|
165
|
+
function readOnlyRefusal(): CohereResponseEnvelope {
|
|
166
|
+
return apiError(405, 'this twin is running read-only; writes are refused.');
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// ── modeled authentication (401) ────────────────────────────────────────────────────────
|
|
170
|
+
/** The obvious-invalid sentinel a caller can use to exercise the 401 path deterministically. */
|
|
171
|
+
const INVALID_KEY_SENTINEL = 'invalid';
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* 401 when a request carries an auth SURFACE (headers present) but no usable bearer token. The
|
|
175
|
+
* twin cannot validate against real Cohere keys, so the modeled failure is the CHECKABLE
|
|
176
|
+
* missing / sentinel-invalid case; any other non-empty token is accepted (auth is faked, D3).
|
|
177
|
+
*
|
|
178
|
+
* cohere-ai sends `Authorization: Bearer <token>`, sourcing the token from the constructor's
|
|
179
|
+
* `token` option or the `CO_API_KEY` env var (`auth/BearerAuthProvider.js`, `const ENV_TOKEN =
|
|
180
|
+
* "CO_API_KEY"`); @ai-sdk/cohere sends the same header from `COHERE_API_KEY`.
|
|
181
|
+
*/
|
|
182
|
+
function checkAuth(req: CohereRequest): CohereResponseEnvelope | null {
|
|
183
|
+
if (req.headers === undefined) return null;
|
|
184
|
+
const raw = req.headers.authorization ?? '';
|
|
185
|
+
const token = raw.toLowerCase().startsWith('bearer ') ? raw.slice(7).trim() : '';
|
|
186
|
+
if (token === '' || token === INVALID_KEY_SENTINEL) {
|
|
187
|
+
return apiError(401, 'invalid api token');
|
|
188
|
+
}
|
|
189
|
+
return null;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
// ── time ────────────────────────────────────────────────────────────────────────────────
|
|
193
|
+
/** A FIXED default instant. The served response must be a pure function of (request, stored
|
|
194
|
+
* state) — reading the wall clock here would make replay non-byte-identical. Callers that want a
|
|
195
|
+
* moving clock pass `occurredAt` explicitly. */
|
|
196
|
+
const PINNED_EPOCH_MS = 1_800_000_000_000;
|
|
197
|
+
function nowIso(occurredAt?: string): string {
|
|
198
|
+
return occurredAt ?? new Date(PINNED_EPOCH_MS).toISOString();
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// ── meta ────────────────────────────────────────────────────────────────────────────────
|
|
202
|
+
function meta(billed: CohereApiMeta['billed_units'], tokens?: CohereApiMeta['tokens'], warnings?: string[]): CohereApiMeta {
|
|
203
|
+
return {
|
|
204
|
+
api_version: { version: API_VERSION },
|
|
205
|
+
billed_units: billed,
|
|
206
|
+
...(tokens ? { tokens } : {}),
|
|
207
|
+
...(warnings && warnings.length ? { warnings } : {}),
|
|
208
|
+
};
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
// ── kernel helpers ──────────────────────────────────────────────────────────────────────
|
|
212
|
+
/** EVERY row of a type, tombstones included — the id-mint denominator. */
|
|
213
|
+
function allRows(type: string, root?: string): Array<Record<string, unknown>> {
|
|
214
|
+
return projectResources(SERVICE, root).filter((r) => r.type === type);
|
|
215
|
+
}
|
|
216
|
+
/** The LIVE rows of a type — what reads serve (a `_deleted` tombstone is invisible). */
|
|
217
|
+
function rows(type: string, root?: string): Array<Record<string, unknown>> {
|
|
218
|
+
return allRows(type, root).filter((r) => r._deleted !== true);
|
|
219
|
+
}
|
|
220
|
+
function getRow(type: string, id: string, root?: string): Record<string, unknown> | undefined {
|
|
221
|
+
return rows(type, root).find((r) => r.id === id);
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* Mint the next id for `type`. The ordinal is the max already seen ACROSS TOMBSTONES, so
|
|
226
|
+
* delete-then-create never re-issues a live id, and it is scanned from the id SET rather than
|
|
227
|
+
* derived from a row count or a module-level counter — a count-mint collides the moment a pulled
|
|
228
|
+
* vendor id sits in a gap above the row count.
|
|
229
|
+
*
|
|
230
|
+
* Cohere's own ids are UUIDs, so the twin's are UUID-SHAPED too (vendor-faithful ids are part of
|
|
231
|
+
* the surface) but derived from a namespaced, ordinal-bearing seed. The `-twin-` marker in the
|
|
232
|
+
* seed is what keeps a locally minted id from ever colliding with a pulled vendor id in either
|
|
233
|
+
* direction: the twin's ids live in a namespace no Cohere UUID occupies.
|
|
234
|
+
*/
|
|
235
|
+
function nextId(type: string, root?: string): string {
|
|
236
|
+
const seen = new Set<string>();
|
|
237
|
+
for (const r of allRows(type, root)) seen.add(String(r.id));
|
|
238
|
+
let ordinal = seen.size + 1;
|
|
239
|
+
// Probe upward until the derived id is genuinely unused. Bounded, deterministic, and it cannot
|
|
240
|
+
// spin: each iteration tries a different ordinal and the set is finite.
|
|
241
|
+
for (let guard = 0; guard <= seen.size + 1; guard++, ordinal++) {
|
|
242
|
+
const candidate = cohereId(`cohere-twin-${type}-${ordinal}`);
|
|
243
|
+
if (!seen.has(candidate)) return candidate;
|
|
244
|
+
}
|
|
245
|
+
/* c8 ignore next */
|
|
246
|
+
throw new Error(`cohere: could not mint a free ${type} id`);
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
/** Drop the kernel housekeeping fields (`type`/`updatedAt` are RESERVED by the projection and
|
|
250
|
+
* never survive it) and the twin's private underscore-prefixed fields, so the served view is
|
|
251
|
+
* exactly the vendor's. */
|
|
252
|
+
function strip(r: Record<string, unknown>): Record<string, unknown> {
|
|
253
|
+
const out: Record<string, unknown> = {};
|
|
254
|
+
for (const [k, v] of Object.entries(r)) {
|
|
255
|
+
if (k === 'type' || k === 'updatedAt' || k.startsWith('_')) continue;
|
|
256
|
+
out[k] = v;
|
|
257
|
+
}
|
|
258
|
+
return out;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
async function write(req: CohereRequest, operation: string, subjectType: string, subjectId: string, fields: Record<string, unknown>): Promise<void> {
|
|
262
|
+
await applyTwinWrite(SERVICE, {
|
|
263
|
+
operation,
|
|
264
|
+
subjectType,
|
|
265
|
+
subjectId,
|
|
266
|
+
fields,
|
|
267
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
268
|
+
actor: { kind: 'agent' },
|
|
269
|
+
}, req.root);
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* The next per-subject write ordinal — the kernel's content+millisecond dedupe hazard, closed.
|
|
274
|
+
*
|
|
275
|
+
* `applyTwinWrite`'s action id is (content + `occurredAt` millisecond), so a write that returns a
|
|
276
|
+
* subject to a value it PREVIOUSLY HELD at the same instant collides with the earlier action and is
|
|
277
|
+
* silently dropped as `replayed`: the reply reports the new value while the projection keeps the
|
|
278
|
+
* old one. That is not theoretical here — a world running under a FROZEN clock
|
|
279
|
+
* (`TWIN_WORLD_CLOCK_FILE`) gives every request in it one `occurredAt`, so
|
|
280
|
+
* `PATCH name:'A'` → `PATCH name:'B'` → `PATCH name:'A'` makes the third action byte-identical to
|
|
281
|
+
* the first, and the handler answers 200 carrying `'B'` — the value the caller did NOT ask for
|
|
282
|
+
* (§9 round 1, finding 4).
|
|
283
|
+
*
|
|
284
|
+
* Folding a monotonically increasing ordinal into the write's fields makes every genuine write
|
|
285
|
+
* distinct by content, so no real write can be mistaken for a replay. It is underscore-prefixed and
|
|
286
|
+
* therefore stripped from every served view. (`upstash-store.ts`'s `rev` is the precedent.)
|
|
287
|
+
*
|
|
288
|
+
* Only revisitable paths need it: a CREATE mints a fresh id and a DELETE is one-way, so
|
|
289
|
+
* `connector.update` is the single write in this twin whose (subject, fields) can revisit a prior
|
|
290
|
+
* value. `token.observe` deliberately does NOT get one — an (id → segment) pair is immutable, so a
|
|
291
|
+
* repeat genuinely IS a replay and the kernel's dedupe is the correct behaviour there.
|
|
292
|
+
*/
|
|
293
|
+
function nextRev(type: string, id: string, root?: string): number {
|
|
294
|
+
const row = getRow(type, id, root);
|
|
295
|
+
const rev = row?._rev;
|
|
296
|
+
return (typeof rev === 'number' ? rev : 0) + 1;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
function parseJson(body?: string): Record<string, unknown> {
|
|
300
|
+
if (!body) return {};
|
|
301
|
+
try {
|
|
302
|
+
const v = JSON.parse(body) as unknown;
|
|
303
|
+
return v && typeof v === 'object' && !Array.isArray(v) ? (v as Record<string, unknown>) : {};
|
|
304
|
+
} catch {
|
|
305
|
+
return {};
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
310
|
+
// CHAT v2
|
|
311
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
312
|
+
|
|
313
|
+
export type ChatV2Args = {
|
|
314
|
+
model: string;
|
|
315
|
+
messages: CohereMessageV2[];
|
|
316
|
+
tools?: unknown;
|
|
317
|
+
toolChoice?: 'REQUIRED' | 'NONE';
|
|
318
|
+
maxTokens?: number;
|
|
319
|
+
stream: boolean;
|
|
320
|
+
thinking: boolean;
|
|
321
|
+
};
|
|
322
|
+
|
|
323
|
+
const V2_ROLES = new Set(['user', 'assistant', 'system', 'tool']);
|
|
324
|
+
const TOOL_CHOICES = new Set(['REQUIRED', 'NONE']);
|
|
325
|
+
const SAFETY_MODES = new Set(['CONTEXTUAL', 'STRICT', 'OFF']);
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* Validate a v2 chat body. Every rejection below is a CLOSED SET the SDK itself declares, so the
|
|
329
|
+
* check is an oracle rather than a guess: `ChatMessageV2` is a four-member union discriminated on
|
|
330
|
+
* `role`, `V2ChatRequestToolChoice` is exactly {REQUIRED, NONE}, and `V2ChatRequestSafetyMode` is
|
|
331
|
+
* exactly {CONTEXTUAL, STRICT, OFF}.
|
|
332
|
+
*/
|
|
333
|
+
function validateChatV2(params: Record<string, unknown>): { args: ChatV2Args } | { error: CohereResponseEnvelope } {
|
|
334
|
+
const model = params.model;
|
|
335
|
+
if (typeof model !== 'string' || model === '') return { error: invalidRequest('model is required') };
|
|
336
|
+
const messages = params.messages;
|
|
337
|
+
if (!Array.isArray(messages)) return { error: invalidRequest('messages is required') };
|
|
338
|
+
if (messages.length === 0) return { error: invalidRequest('messages must not be empty') };
|
|
339
|
+
for (const [i, m] of messages.entries()) {
|
|
340
|
+
const role = (m as { role?: unknown })?.role;
|
|
341
|
+
if (typeof role !== 'string' || !V2_ROLES.has(role)) {
|
|
342
|
+
return { error: invalidRequest(`messages[${i}].role must be one of ${[...V2_ROLES].join(', ')}`) };
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
if (params.tool_choice !== undefined && (typeof params.tool_choice !== 'string' || !TOOL_CHOICES.has(params.tool_choice))) {
|
|
346
|
+
return { error: invalidRequest(`tool_choice must be one of ${[...TOOL_CHOICES].join(', ')}`) };
|
|
347
|
+
}
|
|
348
|
+
if (params.safety_mode !== undefined && (typeof params.safety_mode !== 'string' || !SAFETY_MODES.has(params.safety_mode))) {
|
|
349
|
+
return { error: invalidRequest(`safety_mode must be one of ${[...SAFETY_MODES].join(', ')}`) };
|
|
350
|
+
}
|
|
351
|
+
if (!modelServes(model, 'chat')) return { error: modelNotFound(model) };
|
|
352
|
+
const thinking = (params.thinking as { type?: unknown } | undefined)?.type === 'enabled';
|
|
353
|
+
return {
|
|
354
|
+
args: {
|
|
355
|
+
model,
|
|
356
|
+
messages: messages as CohereMessageV2[],
|
|
357
|
+
...(params.tools !== undefined ? { tools: params.tools } : {}),
|
|
358
|
+
...(typeof params.tool_choice === 'string' ? { toolChoice: params.tool_choice as 'REQUIRED' | 'NONE' } : {}),
|
|
359
|
+
...(typeof params.max_tokens === 'number' ? { maxTokens: params.max_tokens } : {}),
|
|
360
|
+
stream: params.stream === true,
|
|
361
|
+
thinking,
|
|
362
|
+
},
|
|
363
|
+
};
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/** The outcome of ONE engine advance. `missTeach` is appended to the stub so an in-world agent
|
|
367
|
+
* that never matched a handler is told what features it would have to match on. */
|
|
368
|
+
type ScenarioOutcome = { scripted?: ScriptedResult; error?: CohereResponseEnvelope; missTeach?: string };
|
|
369
|
+
|
|
370
|
+
/** Ask the scenario engine (if any) what this turn should say. ONE advance per request — the
|
|
371
|
+
* kernel's `next()` mutates scope state (call counter, `once`, `phase`) exactly like serving, so
|
|
372
|
+
* calling it twice for one request would double-count every handler. */
|
|
373
|
+
function scenarioDecision(args: ChatV2Args, engine: CohereScenarioEngine): ScenarioOutcome {
|
|
374
|
+
const decision = engine.next({
|
|
375
|
+
model: args.model,
|
|
376
|
+
messages: args.messages,
|
|
377
|
+
...(args.tools !== undefined ? { tools: args.tools } : {}),
|
|
378
|
+
});
|
|
379
|
+
if (decision.kind !== 'handler') {
|
|
380
|
+
return { missTeach: `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/cohere.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]` };
|
|
381
|
+
}
|
|
382
|
+
const respond = decision.respond as CohereScenarioRespond;
|
|
383
|
+
if (respond.error) {
|
|
384
|
+
// A scripted API error is served INSTEAD of a completion — the vendor's own status and
|
|
385
|
+
// `{message}` envelope, on both the unary and the streaming path. Deciding it here is what
|
|
386
|
+
// lets it be a real HTTP status rather than a 200 carrying an error-shaped body.
|
|
387
|
+
const kind = respond.error.type;
|
|
388
|
+
if (kind === 'rate_limit') {
|
|
389
|
+
const retry = respond.error.retryAfter ?? 1;
|
|
390
|
+
return { error: { ...apiError(429, 'too many requests'), headers: { 'retry-after': String(retry) } } };
|
|
391
|
+
}
|
|
392
|
+
if (kind === 'service_unavailable') return { error: apiError(503, 'service unavailable') };
|
|
393
|
+
return { error: apiError(500, 'internal server error') };
|
|
394
|
+
}
|
|
395
|
+
return { scripted: realizeCohereRespond(respond) };
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
/** The assistant turn (scripted or stubbed) plus the finish reason it implies. */
|
|
399
|
+
function assistantTurn(args: ChatV2Args, scripted?: ScriptedResult, missTeach = ''): { content: CohereAssistantContentItem[]; toolCalls: CohereToolCallV2[]; toolPlan?: string; finish: CohereFinishReason } {
|
|
400
|
+
if (scripted) {
|
|
401
|
+
return {
|
|
402
|
+
content: scripted.content,
|
|
403
|
+
toolCalls: scripted.toolCalls,
|
|
404
|
+
...(scripted.toolPlan !== undefined ? { toolPlan: scripted.toolPlan } : {}),
|
|
405
|
+
finish: scripted.finishReason,
|
|
406
|
+
};
|
|
407
|
+
}
|
|
408
|
+
const names = toolNames(args.tools);
|
|
409
|
+
// `tool_choice: 'NONE'` forbids a tool call even when tools are declared — the vendor honours it
|
|
410
|
+
// and so must the twin, or a caller testing the NONE path gets a tool call it explicitly banned.
|
|
411
|
+
const wantsTool = names.length > 0 && args.toolChoice !== 'NONE';
|
|
412
|
+
const seed = JSON.stringify(args.messages);
|
|
413
|
+
if (wantsTool) {
|
|
414
|
+
const call = stubToolCall(args.tools, seed, 0);
|
|
415
|
+
return { content: [], toolCalls: call ? [call] : [], toolPlan: stubToolPlan(names), finish: 'TOOL_CALL' };
|
|
416
|
+
}
|
|
417
|
+
const content: CohereAssistantContentItem[] = [];
|
|
418
|
+
if (args.thinking) content.push({ type: 'thinking', thinking: `[twin-stub] deterministic reasoning trace for ${args.model}.` });
|
|
419
|
+
content.push({ type: 'text', text: stubAssistantText(args.messages, args.model) + missTeach });
|
|
420
|
+
return { content, toolCalls: [], finish: 'COMPLETE' };
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
function usageFor(args: ChatV2Args, outputText: string): CohereUsage {
|
|
424
|
+
const input = countInputTokens(args.messages);
|
|
425
|
+
const output = estimateTokens(outputText);
|
|
426
|
+
// Cohere reports the SAME counts twice, under `billed_units` and `tokens`. They are distinct
|
|
427
|
+
// objects on the wire (a caller may read either), so both are emitted.
|
|
428
|
+
return { billed_units: { input_tokens: input, output_tokens: output }, tokens: { input_tokens: input, output_tokens: output } };
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
function outputTextOf(content: CohereAssistantContentItem[], toolCalls: CohereToolCallV2[], toolPlan?: string): string {
|
|
432
|
+
const parts = content.map((c) => (c.type === 'text' ? c.text : c.thinking));
|
|
433
|
+
if (toolPlan) parts.push(toolPlan);
|
|
434
|
+
for (const tc of toolCalls) parts.push(tc.function.arguments);
|
|
435
|
+
return parts.join('');
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
/** Build the unary `POST /v2/chat` envelope. */
|
|
439
|
+
export function buildChatV2(args: ChatV2Args, outcome: ScenarioOutcome = {}): CohereChatV2Response {
|
|
440
|
+
const turn = assistantTurn(args, outcome.scripted, outcome.missTeach ?? '');
|
|
441
|
+
const text = outputTextOf(turn.content, turn.toolCalls, turn.toolPlan);
|
|
442
|
+
const message: CohereChatV2Response['message'] = { role: 'assistant' };
|
|
443
|
+
// Cohere OMITS the empty halves rather than sending nulls: an assistant turn that made a tool
|
|
444
|
+
// call carries `tool_calls` + `tool_plan` and no `content`, and a plain turn carries `content`
|
|
445
|
+
// and neither of the others.
|
|
446
|
+
if (turn.content.length > 0) message.content = turn.content;
|
|
447
|
+
if (turn.toolPlan !== undefined) message.tool_plan = turn.toolPlan;
|
|
448
|
+
if (turn.toolCalls.length > 0) message.tool_calls = turn.toolCalls;
|
|
449
|
+
return {
|
|
450
|
+
id: cohereId(`chat-v2|${args.model}|${JSON.stringify(args.messages)}`),
|
|
451
|
+
finish_reason: turn.finish,
|
|
452
|
+
message,
|
|
453
|
+
usage: usageFor(args, text),
|
|
454
|
+
};
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
/**
|
|
458
|
+
* Stream `POST /v2/chat` into the injected sink and return the same body the unary path would.
|
|
459
|
+
*
|
|
460
|
+
* The event sequence is Cohere's own, and it is asserted key-for-key against
|
|
461
|
+
* @ai-sdk/cohere's `cohereChatChunkSchema` — a `z.discriminatedUnion('type', …)`, so a wrong
|
|
462
|
+
* `type` or a missing `delta.message.content` is a hard parse failure in the real provider, not a
|
|
463
|
+
* soft mismatch. Note the shapes differ between `content-start` (a full content BLOCK under
|
|
464
|
+
* `delta.message.content`) and `content-delta` (just `{ text }`), which is exactly the sort of
|
|
465
|
+
* detail a hand-written stream gets wrong.
|
|
466
|
+
*/
|
|
467
|
+
export function streamChatV2(args: ChatV2Args, sink: SseSink, outcome: ScenarioOutcome = {}): CohereChatV2Response {
|
|
468
|
+
const full = buildChatV2(args, outcome);
|
|
469
|
+
sink({ data: { type: 'message-start', id: full.id, delta: { message: { role: 'assistant' } } } });
|
|
470
|
+
|
|
471
|
+
const toolPlan = full.message.tool_plan;
|
|
472
|
+
if (toolPlan !== undefined) {
|
|
473
|
+
for (const piece of chunkText(toolPlan)) {
|
|
474
|
+
sink({ data: { type: 'tool-plan-delta', delta: { message: { tool_plan: piece } } } });
|
|
475
|
+
}
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
const blocks = full.message.content ?? [];
|
|
479
|
+
for (const [index, block] of blocks.entries()) {
|
|
480
|
+
const kind = block.type;
|
|
481
|
+
const whole = kind === 'text' ? block.text : block.thinking;
|
|
482
|
+
sink({ data: { type: 'content-start', index, delta: { message: { content: kind === 'text' ? { type: 'text', text: '' } : { type: 'thinking', thinking: '' } } } } });
|
|
483
|
+
for (const piece of chunkText(whole)) {
|
|
484
|
+
sink({ data: { type: 'content-delta', index, delta: { message: { content: kind === 'text' ? { text: piece } : { thinking: piece } } } } });
|
|
485
|
+
}
|
|
486
|
+
sink({ data: { type: 'content-end', index } });
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
for (const [index, call] of (full.message.tool_calls ?? []).entries()) {
|
|
490
|
+
sink({ data: { type: 'tool-call-start', index, delta: { message: { tool_calls: { id: call.id, type: 'function', function: { name: call.function.name, arguments: '' } } } } } });
|
|
491
|
+
for (const piece of chunkText(call.function.arguments)) {
|
|
492
|
+
sink({ data: { type: 'tool-call-delta', index, delta: { message: { tool_calls: { function: { arguments: piece } } } } } });
|
|
493
|
+
}
|
|
494
|
+
sink({ data: { type: 'tool-call-end', index } });
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
sink({ data: { type: 'message-end', id: full.id, delta: { finish_reason: full.finish_reason, usage: full.usage } } });
|
|
498
|
+
sink({ done: true });
|
|
499
|
+
return full;
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
/** Split text into deterministic streaming pieces. Deterministic chunking is what makes a
|
|
503
|
+
* streamed replay byte-identical to the previous one. */
|
|
504
|
+
function chunkText(text: string): string[] {
|
|
505
|
+
if (text === '') return [];
|
|
506
|
+
const out: string[] = [];
|
|
507
|
+
for (let i = 0; i < text.length; i += 24) out.push(text.slice(i, i + 24));
|
|
508
|
+
return out;
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
512
|
+
// CHAT v1 — a DIFFERENT protocol, not a versioned alias
|
|
513
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
514
|
+
|
|
515
|
+
function handleChatV1(params: Record<string, unknown>, req: CohereRequest): CohereResponseEnvelope {
|
|
516
|
+
const message = params.message;
|
|
517
|
+
// v1's required field is `message` (a STRING) — a caller that sends v2's `messages` array here
|
|
518
|
+
// gets this refusal, which is the honest answer: it is talking the wrong protocol version.
|
|
519
|
+
if (typeof message !== 'string' || message === '') return invalidRequest('message is required');
|
|
520
|
+
const model = typeof params.model === 'string' && params.model !== '' ? params.model : 'command-a-03-2025';
|
|
521
|
+
if (!modelServes(model, 'chat')) return modelNotFound(model);
|
|
522
|
+
const history = Array.isArray(params.chat_history) ? (params.chat_history as Array<Record<string, unknown>>) : [];
|
|
523
|
+
const text = stubV1Text(message, model);
|
|
524
|
+
const inputTokens = estimateTokens(message) + history.reduce((a, h) => a + estimateTokens(String(h.message ?? '')), 0);
|
|
525
|
+
const outputTokens = estimateTokens(text);
|
|
526
|
+
const seed = `chat-v1|${model}|${message}|${JSON.stringify(history)}`;
|
|
527
|
+
if (params.stream === true) {
|
|
528
|
+
if (!req.sseSink) return invalidRequest('stream:true requires a streaming transport');
|
|
529
|
+
return { status: 200, body: streamChatV1(text, seed, req.sseSink, inputTokens, outputTokens, model, message, history) };
|
|
530
|
+
}
|
|
531
|
+
return { status: 200, body: v1ChatBody(text, seed, inputTokens, outputTokens, message, history) };
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
function v1ChatBody(text: string, seed: string, inputTokens: number, outputTokens: number, message: string, history: Array<Record<string, unknown>>): Record<string, unknown> {
|
|
535
|
+
return {
|
|
536
|
+
text,
|
|
537
|
+
generation_id: cohereId(`${seed}|gen`),
|
|
538
|
+
response_id: cohereId(`${seed}|res`),
|
|
539
|
+
finish_reason: 'COMPLETE' satisfies CohereV1FinishReason, // valid in v1's set, v2's, and the stream-end set
|
|
540
|
+
chat_history: [
|
|
541
|
+
...history.map((h) => ({ role: String(h.role ?? 'USER'), message: String(h.message ?? '') })),
|
|
542
|
+
{ role: 'USER', message },
|
|
543
|
+
{ role: 'CHATBOT', message: text },
|
|
544
|
+
],
|
|
545
|
+
meta: meta({ input_tokens: inputTokens, output_tokens: outputTokens }, { input_tokens: inputTokens, output_tokens: outputTokens }),
|
|
546
|
+
};
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
/**
|
|
550
|
+
* v1's stream is a DIFFERENT wire from v2's: newline-delimited JSON objects discriminated on
|
|
551
|
+
* `event_type` (`stream-start` / `text-generation` / `stream-end`), not v2's `type`-tagged SSE.
|
|
552
|
+
* The twin emits it through the same sink; the server frames v1 as NDJSON and v2 as SSE, which is
|
|
553
|
+
* what the two halves of cohere-ai actually decode.
|
|
554
|
+
*/
|
|
555
|
+
function streamChatV1(text: string, seed: string, sink: SseSink, inputTokens: number, outputTokens: number, model: string, message: string, history: Array<Record<string, unknown>>): Record<string, unknown> {
|
|
556
|
+
const full = v1ChatBody(text, seed, inputTokens, outputTokens, message, history);
|
|
557
|
+
void model;
|
|
558
|
+
sink({ data: { is_finished: false, event_type: 'stream-start', generation_id: full.generation_id } });
|
|
559
|
+
for (const piece of chunkText(text)) {
|
|
560
|
+
sink({ data: { is_finished: false, event_type: 'text-generation', text: piece } });
|
|
561
|
+
}
|
|
562
|
+
sink({ data: { is_finished: true, event_type: 'stream-end', finish_reason: 'COMPLETE', response: full } });
|
|
563
|
+
return full;
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
567
|
+
// EMBED
|
|
568
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
569
|
+
|
|
570
|
+
const EMBEDDING_TYPES = new Set<string>(COHERE_EMBEDDING_TYPES);
|
|
571
|
+
const INPUT_TYPES = new Set<string>(COHERE_INPUT_TYPES);
|
|
572
|
+
const TRUNCATE = new Set<string>(COHERE_TRUNCATE);
|
|
573
|
+
|
|
574
|
+
/** Which embed models require an explicit `input_type` on v1. Cohere's v3+ embed models do; the
|
|
575
|
+
* legacy v2 ones did not, which is exactly why the v1 endpoint keeps the field optional and
|
|
576
|
+
* refuses at request time instead. */
|
|
577
|
+
function requiresInputType(model: string): boolean {
|
|
578
|
+
return /-v[34](\.\d+)?$|v3\.0$|^embed-v4\.0$/.test(model) || model.includes('v3.0') || model === 'embed-v4.0';
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
function embedVectors(texts: string[], model: string, outputDimension?: number): number[][] {
|
|
582
|
+
const dim = outputDimension ?? embedDimensions(model);
|
|
583
|
+
return texts.map((t) => pseudoEmbedding(t, dim));
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
/** Build the `embeddings` object keyed by the requested types. */
|
|
587
|
+
function embeddingsByType(floats: number[][], types: CohereEmbeddingType[]): Record<string, unknown> {
|
|
588
|
+
const out: Record<string, unknown> = {};
|
|
589
|
+
for (const t of types) {
|
|
590
|
+
if (t === 'float') out.float = floats;
|
|
591
|
+
else if (t === 'base64') out.base64 = floats.map(base64Embedding);
|
|
592
|
+
else out[t] = floats.map((v) => quantizeEmbedding(v, t));
|
|
593
|
+
}
|
|
594
|
+
return out;
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
function collectEmbedInputs(params: Record<string, unknown>): string[] | null {
|
|
598
|
+
const texts = params.texts;
|
|
599
|
+
if (Array.isArray(texts)) return texts.map((t) => String(t));
|
|
600
|
+
// v2 also accepts `inputs: EmbedInput[]`, each carrying a `content` part array.
|
|
601
|
+
const inputs = params.inputs;
|
|
602
|
+
if (Array.isArray(inputs)) {
|
|
603
|
+
return inputs.map((i) => contentToText((i as { content?: CohereMessageV2['content'] })?.content ?? ''));
|
|
604
|
+
}
|
|
605
|
+
const images = params.images;
|
|
606
|
+
if (Array.isArray(images)) return images.map((i) => String(i));
|
|
607
|
+
return null;
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
function validateEmbedShared(params: Record<string, unknown>): CohereResponseEnvelope | null {
|
|
611
|
+
const types = params.embedding_types;
|
|
612
|
+
if (types !== undefined) {
|
|
613
|
+
if (!Array.isArray(types) || types.length === 0) return invalidRequest('embedding_types must be a non-empty list');
|
|
614
|
+
for (const t of types) {
|
|
615
|
+
if (typeof t !== 'string' || !EMBEDDING_TYPES.has(t)) {
|
|
616
|
+
return invalidRequest(`embedding_types must be a subset of ${[...EMBEDDING_TYPES].join(', ')}`);
|
|
617
|
+
}
|
|
618
|
+
}
|
|
619
|
+
}
|
|
620
|
+
if (params.input_type !== undefined && (typeof params.input_type !== 'string' || !INPUT_TYPES.has(params.input_type))) {
|
|
621
|
+
return invalidRequest(`input_type must be one of ${[...INPUT_TYPES].join(', ')}`);
|
|
622
|
+
}
|
|
623
|
+
if (params.truncate !== undefined && (typeof params.truncate !== 'string' || !TRUNCATE.has(params.truncate))) {
|
|
624
|
+
return invalidRequest(`truncate must be one of ${[...TRUNCATE].join(', ')}`);
|
|
625
|
+
}
|
|
626
|
+
if (params.output_dimension !== undefined) {
|
|
627
|
+
const d = params.output_dimension;
|
|
628
|
+
if (typeof d !== 'number' || !(COHERE_OUTPUT_DIMENSIONS as readonly number[]).includes(d)) {
|
|
629
|
+
return invalidRequest(`output_dimension must be one of ${COHERE_OUTPUT_DIMENSIONS.join(', ')}`);
|
|
630
|
+
}
|
|
631
|
+
}
|
|
632
|
+
return null;
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
/** `POST /v2/embed`. `model` AND `input_type` are both REQUIRED here — the v2 request type
|
|
636
|
+
* declares them non-optional, and this is the sharpest v1-vs-v2 difference. */
|
|
637
|
+
function handleEmbedV2(params: Record<string, unknown>): CohereResponseEnvelope {
|
|
638
|
+
const model = params.model;
|
|
639
|
+
if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
|
|
640
|
+
if (typeof params.input_type !== 'string' || params.input_type === '') return invalidRequest('input_type is required');
|
|
641
|
+
const shared = validateEmbedShared(params);
|
|
642
|
+
if (shared) return shared;
|
|
643
|
+
if (!modelServes(model, 'embed')) return modelNotFound(model);
|
|
644
|
+
const texts = collectEmbedInputs(params);
|
|
645
|
+
if (texts === null || texts.length === 0) return invalidRequest('one of texts, images or inputs is required');
|
|
646
|
+
const types = (Array.isArray(params.embedding_types) ? params.embedding_types : ['float']) as CohereEmbeddingType[];
|
|
647
|
+
const floats = embedVectors(texts, model, typeof params.output_dimension === 'number' ? params.output_dimension : undefined);
|
|
648
|
+
const inputTokens = texts.reduce((a, t) => a + estimateTokens(t), 0);
|
|
649
|
+
return {
|
|
650
|
+
status: 200,
|
|
651
|
+
body: {
|
|
652
|
+
id: cohereId(`embed-v2|${model}|${JSON.stringify(texts)}|${types.join(',')}`),
|
|
653
|
+
embeddings: embeddingsByType(floats, types),
|
|
654
|
+
texts,
|
|
655
|
+
response_type: 'embeddings_by_type',
|
|
656
|
+
// NO `output_tokens`: an embed call has no output side, and Cohere's embed `billed_units`
|
|
657
|
+
// carries only `input_tokens` (plus images/image_tokens for image inputs). Emitting a zero
|
|
658
|
+
// would be an invented key in a pack whose whole thesis is that it invents none.
|
|
659
|
+
meta: meta({ input_tokens: inputTokens }, { input_tokens: inputTokens }),
|
|
660
|
+
},
|
|
661
|
+
};
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
/**
|
|
665
|
+
* `POST /v1/embed`. Two things differ from v2 and BOTH are load-bearing:
|
|
666
|
+
* • `input_type` is optional in the schema — but a v3/v4 embed model still refuses without one,
|
|
667
|
+
* so the check moves from the schema to the handler;
|
|
668
|
+
* • the DEFAULT response is `embeddings_floats` — a FLAT `number[][]` at `embeddings`. Only when
|
|
669
|
+
* the caller asks for `embedding_types` does v1 answer the by-type shape. A twin that always
|
|
670
|
+
* answered by-type would break every default v1 caller.
|
|
671
|
+
*/
|
|
672
|
+
function handleEmbedV1(params: Record<string, unknown>): CohereResponseEnvelope {
|
|
673
|
+
const model = typeof params.model === 'string' && params.model !== '' ? params.model : 'embed-english-v3.0';
|
|
674
|
+
const shared = validateEmbedShared(params);
|
|
675
|
+
if (shared) return shared;
|
|
676
|
+
if (!modelServes(model, 'embed')) return modelNotFound(model);
|
|
677
|
+
if (requiresInputType(model) && params.input_type === undefined) {
|
|
678
|
+
return invalidRequest(`input_type is required for model '${model}'`);
|
|
679
|
+
}
|
|
680
|
+
const texts = collectEmbedInputs(params);
|
|
681
|
+
if (texts === null || texts.length === 0) return invalidRequest('one of texts or images is required');
|
|
682
|
+
const floats = embedVectors(texts, model, typeof params.output_dimension === 'number' ? params.output_dimension : undefined);
|
|
683
|
+
const inputTokens = texts.reduce((a, t) => a + estimateTokens(t), 0);
|
|
684
|
+
const id = cohereId(`embed-v1|${model}|${JSON.stringify(texts)}`);
|
|
685
|
+
const m = meta({ input_tokens: inputTokens }, { input_tokens: inputTokens }); // embed has no output side
|
|
686
|
+
if (params.embedding_types === undefined) {
|
|
687
|
+
return { status: 200, body: { id, embeddings: floats, texts, response_type: 'embeddings_floats', meta: m } };
|
|
688
|
+
}
|
|
689
|
+
const types = params.embedding_types as CohereEmbeddingType[];
|
|
690
|
+
return { status: 200, body: { id, embeddings: embeddingsByType(floats, types), texts, response_type: 'embeddings_by_type', meta: m } };
|
|
691
|
+
}
|
|
692
|
+
|
|
693
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
694
|
+
// RERANK
|
|
695
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
696
|
+
|
|
697
|
+
function rankAll(query: string, docs: string[]): Array<{ index: number; relevance_score: number }> {
|
|
698
|
+
return docs
|
|
699
|
+
.map((d, index) => ({ index, relevance_score: rerankScore(query, d) }))
|
|
700
|
+
.sort((a, b) => b.relevance_score - a.relevance_score);
|
|
701
|
+
}
|
|
702
|
+
|
|
703
|
+
/** `POST /v2/rerank`. `documents` is `string[]` in v2 — an object element is refused. */
|
|
704
|
+
function handleRerankV2(params: Record<string, unknown>): CohereResponseEnvelope {
|
|
705
|
+
const model = params.model;
|
|
706
|
+
if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
|
|
707
|
+
const query = params.query;
|
|
708
|
+
if (typeof query !== 'string' || query === '') return invalidRequest('query is required');
|
|
709
|
+
const documents = params.documents;
|
|
710
|
+
if (!Array.isArray(documents) || documents.length === 0) return invalidRequest('documents is required');
|
|
711
|
+
for (const [i, d] of documents.entries()) {
|
|
712
|
+
if (typeof d !== 'string') return invalidRequest(`documents[${i}] must be a string`);
|
|
713
|
+
}
|
|
714
|
+
if (!modelServes(model, 'rerank')) return modelNotFound(model);
|
|
715
|
+
const docs = documents as string[];
|
|
716
|
+
const topN = typeof params.top_n === 'number' && params.top_n > 0 ? Math.min(params.top_n, docs.length) : docs.length;
|
|
717
|
+
const results = rankAll(query, docs).slice(0, topN);
|
|
718
|
+
return {
|
|
719
|
+
status: 200,
|
|
720
|
+
body: {
|
|
721
|
+
id: cohereId(`rerank-v2|${model}|${query}|${JSON.stringify(docs)}`),
|
|
722
|
+
results,
|
|
723
|
+
// Rerank is billed in SEARCH UNITS, not tokens — one per (query, up to 100 documents) call.
|
|
724
|
+
meta: meta({ search_units: 1 }),
|
|
725
|
+
},
|
|
726
|
+
};
|
|
727
|
+
}
|
|
728
|
+
|
|
729
|
+
/** `POST /v1/rerank`. Accepts object documents and honours `return_documents`, neither of which
|
|
730
|
+
* exists in v2. */
|
|
731
|
+
function handleRerankV1(params: Record<string, unknown>): CohereResponseEnvelope {
|
|
732
|
+
const query = params.query;
|
|
733
|
+
if (typeof query !== 'string' || query === '') return invalidRequest('query is required');
|
|
734
|
+
const documents = params.documents;
|
|
735
|
+
if (!Array.isArray(documents) || documents.length === 0) return invalidRequest('documents is required');
|
|
736
|
+
const model = typeof params.model === 'string' && params.model !== '' ? params.model : 'rerank-v3.5';
|
|
737
|
+
if (!modelServes(model, 'rerank')) return modelNotFound(model);
|
|
738
|
+
const rankFields = Array.isArray(params.rank_fields) ? (params.rank_fields as string[]) : undefined;
|
|
739
|
+
const texts = documents.map((d) => {
|
|
740
|
+
if (typeof d === 'string') return d;
|
|
741
|
+
const o = d as Record<string, unknown>;
|
|
742
|
+
if (rankFields) return rankFields.map((f) => String(o[f] ?? '')).join(' ');
|
|
743
|
+
if (typeof o.text === 'string') return o.text;
|
|
744
|
+
return JSON.stringify(o);
|
|
745
|
+
});
|
|
746
|
+
const topN = typeof params.top_n === 'number' && params.top_n > 0 ? Math.min(params.top_n, texts.length) : texts.length;
|
|
747
|
+
const ranked = rankAll(query, texts).slice(0, topN);
|
|
748
|
+
const returnDocuments = params.return_documents === true;
|
|
749
|
+
return {
|
|
750
|
+
status: 200,
|
|
751
|
+
body: {
|
|
752
|
+
id: cohereId(`rerank-v1|${model}|${query}|${JSON.stringify(texts)}`),
|
|
753
|
+
results: ranked.map((r) => (returnDocuments ? { ...r, document: { text: texts[r.index]! } } : r)),
|
|
754
|
+
meta: meta({ search_units: 1 }),
|
|
755
|
+
},
|
|
756
|
+
};
|
|
757
|
+
}
|
|
758
|
+
|
|
759
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
760
|
+
// CLASSIFY / TOKENIZE / DETOKENIZE / CHECK-API-KEY
|
|
761
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
762
|
+
|
|
763
|
+
function handleClassify(params: Record<string, unknown>): CohereResponseEnvelope {
|
|
764
|
+
const inputs = params.inputs;
|
|
765
|
+
if (!Array.isArray(inputs) || inputs.length === 0) return invalidRequest('inputs is required');
|
|
766
|
+
const examples = params.examples;
|
|
767
|
+
const preset = params.preset;
|
|
768
|
+
// Cohere's classify needs a way to know the label space: either training `examples` or a saved
|
|
769
|
+
// `preset`. With neither there is nothing to classify against, and it refuses.
|
|
770
|
+
if (!Array.isArray(examples) && typeof preset !== 'string') {
|
|
771
|
+
return invalidRequest('one of examples or preset is required');
|
|
772
|
+
}
|
|
773
|
+
let labels: string[];
|
|
774
|
+
if (Array.isArray(examples)) {
|
|
775
|
+
const seen: string[] = [];
|
|
776
|
+
for (const e of examples) {
|
|
777
|
+
const label = (e as { label?: unknown })?.label;
|
|
778
|
+
if (typeof label !== 'string' || label === '') return invalidRequest('each example requires a label');
|
|
779
|
+
if (!seen.includes(label)) seen.push(label);
|
|
780
|
+
}
|
|
781
|
+
// The vendor's documented minimum: a classifier needs at least two distinct classes.
|
|
782
|
+
if (seen.length < 2) return invalidRequest('at least 2 unique labels are required');
|
|
783
|
+
labels = seen;
|
|
784
|
+
} else {
|
|
785
|
+
// NEVER A FAKE SUCCESS. An earlier version invented `['positive','negative']` as the label space
|
|
786
|
+
// for ANY preset string, returning confident-looking classifications against labels the caller
|
|
787
|
+
// never named — the exact class the detokenize path treats as a hard rule and that
|
|
788
|
+
// `cohere.errors.unmodeled_ops_404` claims pack-wide. Presets are a filed todo
|
|
789
|
+
// (`cohere.classify.preset`); until they are modeled, an unmodeled op fails like the vendor.
|
|
790
|
+
return notFound(`preset '${String(preset)}' not found — this twin does not model saved classify presets`);
|
|
791
|
+
}
|
|
792
|
+
const texts = inputs.map((i) => String(i));
|
|
793
|
+
const classifications = texts.map((input, i) => {
|
|
794
|
+
const { prediction, confidences } = classifyText(input, labels);
|
|
795
|
+
const labelMap: Record<string, { confidence: number }> = {};
|
|
796
|
+
labels.forEach((l, j) => { labelMap[l] = { confidence: confidences[j]! }; });
|
|
797
|
+
return {
|
|
798
|
+
id: cohereId(`classify|${input}|${i}`),
|
|
799
|
+
input,
|
|
800
|
+
prediction,
|
|
801
|
+
predictions: [prediction],
|
|
802
|
+
confidence: Math.max(...confidences),
|
|
803
|
+
confidences: [Math.max(...confidences)],
|
|
804
|
+
labels: labelMap,
|
|
805
|
+
classification_type: 'single-label' as const,
|
|
806
|
+
};
|
|
807
|
+
});
|
|
808
|
+
return {
|
|
809
|
+
status: 200,
|
|
810
|
+
body: {
|
|
811
|
+
id: cohereId(`classify-call|${JSON.stringify(texts)}|${labels.join(',')}`),
|
|
812
|
+
classifications,
|
|
813
|
+
// Classify is billed per CLASSIFICATION, not per token.
|
|
814
|
+
meta: meta({ classifications: texts.length }),
|
|
815
|
+
},
|
|
816
|
+
};
|
|
817
|
+
}
|
|
818
|
+
|
|
819
|
+
/**
|
|
820
|
+
* `POST /v1/tokenize` — and the place the twin's vocabulary is LEARNED.
|
|
821
|
+
*
|
|
822
|
+
* The twin has no BPE vocabulary yet (`cohere.tokenize.bpe_vocabulary` is a todo), so it
|
|
823
|
+
* cannot invent a reverse mapping for `detokenize` out of thin air without fabricating. Instead
|
|
824
|
+
* tokenize OBSERVES each (id → segment) pair into the kernel log; detokenize folds that
|
|
825
|
+
* projection. The result is a twin that can only detokenize what it has genuinely seen — and says
|
|
826
|
+
* so, with a 400, otherwise.
|
|
827
|
+
*/
|
|
828
|
+
async function handleTokenize(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
|
|
829
|
+
const text = params.text;
|
|
830
|
+
// Both fields are REQUIRED in `TokenizeRequest` — `model` is not optional here, unlike on
|
|
831
|
+
// /v1/embed and /v1/rerank.
|
|
832
|
+
if (typeof text !== 'string' || text === '') return invalidRequest('text is required');
|
|
833
|
+
const model = params.model;
|
|
834
|
+
if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
|
|
835
|
+
if (findModel(model) === undefined) return modelNotFound(model);
|
|
836
|
+
|
|
837
|
+
const segments = segmentText(text);
|
|
838
|
+
// Read the vocabulary ONCE, then resolve every segment against the same in-memory view. Reading
|
|
839
|
+
// per segment would make an id assigned earlier in THIS request invisible to a later one, so a
|
|
840
|
+
// text repeating a colliding pair would mint two ids for one string.
|
|
841
|
+
const vocab = new Map<string, number>();
|
|
842
|
+
const taken = new Map<number, string>();
|
|
843
|
+
for (const r of rows('token', req.root)) {
|
|
844
|
+
const seg = String(r.segment);
|
|
845
|
+
const tid = Number(r.token_id);
|
|
846
|
+
vocab.set(seg, tid);
|
|
847
|
+
taken.set(tid, seg);
|
|
848
|
+
}
|
|
849
|
+
const tokens: number[] = [];
|
|
850
|
+
const fresh: Array<{ id: number; segment: string }> = [];
|
|
851
|
+
for (const segment of segments) {
|
|
852
|
+
const existing = vocab.get(segment);
|
|
853
|
+
if (existing !== undefined) { tokens.push(existing); continue; }
|
|
854
|
+
const id = assignToken(segment, taken);
|
|
855
|
+
vocab.set(segment, id);
|
|
856
|
+
taken.set(id, segment);
|
|
857
|
+
fresh.push({ id, segment });
|
|
858
|
+
tokens.push(id);
|
|
859
|
+
}
|
|
860
|
+
for (const f of fresh) {
|
|
861
|
+
// One row per token, keyed by the id — the subject id IS the token id, so a re-observation of
|
|
862
|
+
// the same pair dedupes in the kernel rather than growing the log.
|
|
863
|
+
await write(req, 'token.observe', 'token', `tok_${f.id}`, { token_id: f.id, segment: f.segment });
|
|
864
|
+
}
|
|
865
|
+
const inputTokens = estimateTokens(text);
|
|
866
|
+
return {
|
|
867
|
+
status: 200,
|
|
868
|
+
body: { tokens, token_strings: segments, meta: meta({ input_tokens: inputTokens }, { input_tokens: inputTokens }) },
|
|
869
|
+
};
|
|
870
|
+
}
|
|
871
|
+
|
|
872
|
+
/**
|
|
873
|
+
* Resolve a segment's token id. The preferred id is a pure hash of the segment; when a DIFFERENT
|
|
874
|
+
* segment already holds it, probe upward until a free id is found.
|
|
875
|
+
*
|
|
876
|
+
* Without the probe the second segment would silently overwrite the first's row and `detokenize`
|
|
877
|
+
* would then return the WRONG text for every earlier caller — a dirty-state bug invisible to any
|
|
878
|
+
* fresh-root verify, which is exactly why `cohere.tokenize.collision_probe` builds the collision
|
|
879
|
+
* up deliberately.
|
|
880
|
+
*/
|
|
881
|
+
function assignToken(segment: string, taken: Map<number, string>): number {
|
|
882
|
+
let id = preferredTokenId(segment);
|
|
883
|
+
// BOUNDED, like `nextId`. The id space is [1, 250_000]; an unbounded probe would spin FOREVER
|
|
884
|
+
// once the learned vocabulary filled it, hanging the server's fetch callback rather than failing
|
|
885
|
+
// (§9 round 1, finding 5). The bound is the occupied-set size + 1, so it can only be reached when
|
|
886
|
+
// every id genuinely is taken — and then it says so loudly.
|
|
887
|
+
for (let guard = 0; guard <= taken.size; guard++) {
|
|
888
|
+
if (!taken.has(id) || taken.get(id) === segment) return id;
|
|
889
|
+
id = (id % 250_000) + 1;
|
|
890
|
+
}
|
|
891
|
+
throw new Error('cohere: token vocabulary exhausted — every id in [1, 250000] is assigned to a different segment');
|
|
892
|
+
}
|
|
893
|
+
|
|
894
|
+
async function handleDetokenize(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
|
|
895
|
+
const tokens = params.tokens;
|
|
896
|
+
if (!Array.isArray(tokens)) return invalidRequest('tokens is required');
|
|
897
|
+
const model = params.model;
|
|
898
|
+
if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
|
|
899
|
+
if (findModel(model) === undefined) return modelNotFound(model);
|
|
900
|
+
const vocab = new Map<number, string>();
|
|
901
|
+
for (const r of rows('token', req.root)) vocab.set(Number(r.token_id), String(r.segment));
|
|
902
|
+
const parts: string[] = [];
|
|
903
|
+
for (const t of tokens) {
|
|
904
|
+
const seg = vocab.get(Number(t));
|
|
905
|
+
// NEVER a fake success. The twin's vocabulary is what it has observed; an id it has never
|
|
906
|
+
// issued is not something it can honestly decode, and silently dropping it (or emitting a
|
|
907
|
+
// placeholder) would hand the caller text the twin invented.
|
|
908
|
+
if (seg === undefined) return invalidRequest(`unknown token id ${String(t)} — this twin detokenizes only ids it has issued via POST /v1/tokenize`);
|
|
909
|
+
parts.push(seg);
|
|
910
|
+
}
|
|
911
|
+
const text = parts.join('');
|
|
912
|
+
return { status: 200, body: { text, meta: meta({ input_tokens: tokens.length }, { input_tokens: tokens.length }) } };
|
|
913
|
+
}
|
|
914
|
+
|
|
915
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
916
|
+
// DATASETS (stateful)
|
|
917
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
918
|
+
|
|
919
|
+
/** `DatasetType` — the vendor's CLOSED set (`api/types/DatasetType.d.ts`). A literal allowlist is
|
|
920
|
+
* an ORACLE here: the accepted values must biject exactly with this documented set. */
|
|
921
|
+
const DATASET_TYPES = new Set([
|
|
922
|
+
'embed-input', 'embed-result', 'cluster-result', 'cluster-outliers',
|
|
923
|
+
'reranker-finetune-input', 'single-label-classification-finetune-input',
|
|
924
|
+
'chat-finetune-input', 'multi-label-classification-finetune-input',
|
|
925
|
+
'batch-chat-input', 'batch-openai-chat-input', 'batch-embed-v2-input', 'batch-chat-v2-input',
|
|
926
|
+
]);
|
|
927
|
+
|
|
928
|
+
/** `DatasetValidationStatus` — the vendor's closed set. */
|
|
929
|
+
const VALIDATION_STATUSES = new Set(['unknown', 'queued', 'processing', 'failed', 'validated', 'skipped']);
|
|
930
|
+
|
|
931
|
+
async function createDataset(params: Record<string, unknown>, query: URLSearchParams, req: CohereRequest): Promise<CohereResponseEnvelope> {
|
|
932
|
+
// `name` and `type` travel as QUERY parameters on this endpoint — the body is the multipart
|
|
933
|
+
// file. The server folds both into the handler's JSON contract, and the handler accepts either
|
|
934
|
+
// home so an in-process caller need not synthesize a query string.
|
|
935
|
+
const name = query.get('name') ?? (typeof params.name === 'string' ? params.name : '');
|
|
936
|
+
const type = query.get('type') ?? (typeof params.type === 'string' ? params.type : '');
|
|
937
|
+
if (!name) return invalidRequest('name is required');
|
|
938
|
+
if (!type) return invalidRequest('type is required');
|
|
939
|
+
if (!DATASET_TYPES.has(type)) return invalidRequest(`type must be one of ${[...DATASET_TYPES].join(', ')}`);
|
|
940
|
+
// The `data` file is REQUIRED: `datasets.create(data, evalData, request)` unconditionally does
|
|
941
|
+
// `_body.appendFile("data", data)` in cohere-ai@8.1.0, so the vendor never sees a fileless
|
|
942
|
+
// create. Accepting one would be an unmodeled input NOT failing like the vendor (§9 round 1,
|
|
943
|
+
// finding 6) — and it would have let the closed-DatasetType oracle be built on the lenient path.
|
|
944
|
+
const content = params.content;
|
|
945
|
+
if (typeof content !== 'string' || content === '') return invalidRequest('the data file is required');
|
|
946
|
+
const id = nextId('dataset', req.root);
|
|
947
|
+
const at = nowIso(req.occurredAt);
|
|
948
|
+
await write(req, 'dataset.create', 'dataset', id, {
|
|
949
|
+
// THE TWIN MINTED THIS ID, and the vendor has never heard of it. The connector reads this
|
|
950
|
+
// marker to refuse pushing an update/cancel/delete against a subject that has no vendor
|
|
951
|
+
// identity — without it a local delete fires the twin's own UUID at the REAL account
|
|
952
|
+
// (§9 round 1, connector B1/M1). It is underscore-prefixed, so no served view carries it, and
|
|
953
|
+
// the push-confirm strips it when it rebuilds the row under the vendor's id.
|
|
954
|
+
_twin_minted: true,
|
|
955
|
+
name,
|
|
956
|
+
dataset_type: type,
|
|
957
|
+
created_at: at,
|
|
958
|
+
updated_at: at,
|
|
959
|
+
validation_status: 'validated',
|
|
960
|
+
validation_error: null,
|
|
961
|
+
validation_warnings: [],
|
|
962
|
+
required_fields: [],
|
|
963
|
+
preserve_fields: [],
|
|
964
|
+
dataset_parts: [],
|
|
965
|
+
_content: content,
|
|
966
|
+
});
|
|
967
|
+
// Cohere's create answers ONLY `{ id }` — not the dataset. A twin returning the whole object
|
|
968
|
+
// here would be more "helpful" and less faithful.
|
|
969
|
+
return { status: 200, body: { id } };
|
|
970
|
+
}
|
|
971
|
+
|
|
972
|
+
function datasetView(r: Record<string, unknown>): Record<string, unknown> {
|
|
973
|
+
return { id: r.id, ...strip(r) };
|
|
974
|
+
}
|
|
975
|
+
|
|
976
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
977
|
+
// CONNECTORS (stateful)
|
|
978
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
979
|
+
|
|
980
|
+
async function createConnector(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
|
|
981
|
+
const name = params.name;
|
|
982
|
+
if (typeof name !== 'string' || name === '') return invalidRequest('name is required');
|
|
983
|
+
const url = params.url;
|
|
984
|
+
if (typeof url !== 'string' || url === '') return invalidRequest('url is required');
|
|
985
|
+
const id = nextId('connector', req.root);
|
|
986
|
+
const at = nowIso(req.occurredAt);
|
|
987
|
+
const oauth = params.oauth as Record<string, unknown> | undefined;
|
|
988
|
+
await write(req, 'connector.create', 'connector', id, {
|
|
989
|
+
_twin_minted: true, // see the note on dataset.create — the connector refuses to push a
|
|
990
|
+
_rev: 1, // twin-minted subject's later mutations at a real account.
|
|
991
|
+
organization_id: 'org_twin',
|
|
992
|
+
name,
|
|
993
|
+
description: typeof params.description === 'string' ? params.description : null,
|
|
994
|
+
url,
|
|
995
|
+
created_at: at,
|
|
996
|
+
updated_at: at,
|
|
997
|
+
excludes: Array.isArray(params.excludes) ? params.excludes : [],
|
|
998
|
+
// `auth_type` is derived, not caller-supplied — it reports HOW the connector authenticates.
|
|
999
|
+
auth_type: oauth ? 'oauth' : params.service_auth ? 'service_auth' : 'none',
|
|
1000
|
+
oauth: oauth ? { client_id: String(oauth.client_id ?? ''), authorize_url: String(oauth.authorize_url ?? ''), token_url: String(oauth.token_url ?? ''), scope: oauth.scope ?? null } : null,
|
|
1001
|
+
auth_status: oauth ? 'expired' : 'valid',
|
|
1002
|
+
active: params.active === undefined ? true : params.active === true,
|
|
1003
|
+
continue_on_failure: params.continue_on_failure === true,
|
|
1004
|
+
});
|
|
1005
|
+
const row = getRow('connector', id, req.root)!;
|
|
1006
|
+
return { status: 200, body: { connector: connectorView(row) } };
|
|
1007
|
+
}
|
|
1008
|
+
|
|
1009
|
+
function connectorView(r: Record<string, unknown>): Record<string, unknown> {
|
|
1010
|
+
return { id: r.id, ...strip(r) };
|
|
1011
|
+
}
|
|
1012
|
+
|
|
1013
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
1014
|
+
// EMBED JOBS (stateful)
|
|
1015
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
1016
|
+
|
|
1017
|
+
const EMBED_JOB_STATUSES = new Set(['processing', 'complete', 'cancelling', 'cancelled', 'failed']);
|
|
1018
|
+
|
|
1019
|
+
async function createEmbedJob(params: Record<string, unknown>, req: CohereRequest): Promise<CohereResponseEnvelope> {
|
|
1020
|
+
const model = params.model;
|
|
1021
|
+
if (typeof model !== 'string' || model === '') return invalidRequest('model is required');
|
|
1022
|
+
const datasetId = params.dataset_id;
|
|
1023
|
+
if (typeof datasetId !== 'string' || datasetId === '') return invalidRequest('dataset_id is required');
|
|
1024
|
+
const inputType = params.input_type;
|
|
1025
|
+
if (typeof inputType !== 'string' || inputType === '') return invalidRequest('input_type is required');
|
|
1026
|
+
if (!INPUT_TYPES.has(inputType)) return invalidRequest(`input_type must be one of ${[...INPUT_TYPES].join(', ')}`);
|
|
1027
|
+
if (params.truncate !== undefined && (typeof params.truncate !== 'string' || !TRUNCATE.has(params.truncate))) {
|
|
1028
|
+
return invalidRequest(`truncate must be one of ${[...TRUNCATE].join(', ')}`);
|
|
1029
|
+
}
|
|
1030
|
+
if (!modelServes(model, 'embed')) return modelNotFound(model);
|
|
1031
|
+
// The job's input must be a dataset that actually exists — a cross-resource invariant the
|
|
1032
|
+
// vendor enforces and a twin that skipped it would let a caller queue work over nothing.
|
|
1033
|
+
if (getRow('dataset', datasetId, req.root) === undefined) return notFound(`dataset '${datasetId}' not found`);
|
|
1034
|
+
const id = nextId('embed_job', req.root);
|
|
1035
|
+
await write(req, 'embed_job.create', 'embed_job', id, {
|
|
1036
|
+
_twin_minted: true, // see the note on dataset.create
|
|
1037
|
+
job_id: id,
|
|
1038
|
+
name: typeof params.name === 'string' ? params.name : null,
|
|
1039
|
+
status: 'processing',
|
|
1040
|
+
created_at: nowIso(req.occurredAt),
|
|
1041
|
+
input_dataset_id: datasetId,
|
|
1042
|
+
output_dataset_id: null,
|
|
1043
|
+
model,
|
|
1044
|
+
truncate: typeof params.truncate === 'string' ? params.truncate : 'END',
|
|
1045
|
+
});
|
|
1046
|
+
// Cohere's create answers `{ job_id, meta }` — not the job object.
|
|
1047
|
+
return { status: 200, body: { job_id: id, meta: meta({}) } };
|
|
1048
|
+
}
|
|
1049
|
+
|
|
1050
|
+
function embedJobView(r: Record<string, unknown>): Record<string, unknown> {
|
|
1051
|
+
const v = strip(r);
|
|
1052
|
+
delete v.id; // the vendor's key for an embed job is `job_id`, and there is no `id` alongside it.
|
|
1053
|
+
return v;
|
|
1054
|
+
}
|
|
1055
|
+
|
|
1056
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
1057
|
+
// THE ROUTER CENSUS
|
|
1058
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
1059
|
+
|
|
1060
|
+
/**
|
|
1061
|
+
* Every method/path pair the dispatch below branches on. HAND-AUTHORED (the honest limit): a
|
|
1062
|
+
* branch added to the handler and to neither this list nor the conformance snapshot is invisible
|
|
1063
|
+
* to the conformance check's third direction. The mechanical backstop for that lives in
|
|
1064
|
+
* `cohere-twin.test.ts` ("covers every literal path the handler dispatches on"), which reads this
|
|
1065
|
+
* module's own source; segment-matched sub-routes still rest on review.
|
|
1066
|
+
*
|
|
1067
|
+
* `{id}` stands for one path segment.
|
|
1068
|
+
*/
|
|
1069
|
+
export const COHERE_ROUTER_SURFACE: ReadonlyArray<{ method: string; path: string }> = [
|
|
1070
|
+
{ method: 'POST', path: '/v2/chat' },
|
|
1071
|
+
{ method: 'POST', path: '/v2/embed' },
|
|
1072
|
+
{ method: 'POST', path: '/v2/rerank' },
|
|
1073
|
+
{ method: 'POST', path: '/v1/chat' },
|
|
1074
|
+
{ method: 'POST', path: '/v1/embed' },
|
|
1075
|
+
{ method: 'POST', path: '/v1/rerank' },
|
|
1076
|
+
{ method: 'POST', path: '/v1/classify' },
|
|
1077
|
+
{ method: 'POST', path: '/v1/tokenize' },
|
|
1078
|
+
{ method: 'POST', path: '/v1/detokenize' },
|
|
1079
|
+
{ method: 'POST', path: '/v1/check-api-key' },
|
|
1080
|
+
{ method: 'GET', path: '/v1/models' },
|
|
1081
|
+
{ method: 'GET', path: '/v1/models/{id}' },
|
|
1082
|
+
{ method: 'POST', path: '/v1/datasets' },
|
|
1083
|
+
{ method: 'GET', path: '/v1/datasets' },
|
|
1084
|
+
{ method: 'GET', path: '/v1/datasets/usage' },
|
|
1085
|
+
{ method: 'GET', path: '/v1/datasets/{id}' },
|
|
1086
|
+
{ method: 'DELETE', path: '/v1/datasets/{id}' },
|
|
1087
|
+
{ method: 'POST', path: '/v1/connectors' },
|
|
1088
|
+
{ method: 'GET', path: '/v1/connectors' },
|
|
1089
|
+
{ method: 'GET', path: '/v1/connectors/{id}' },
|
|
1090
|
+
{ method: 'PATCH', path: '/v1/connectors/{id}' },
|
|
1091
|
+
{ method: 'DELETE', path: '/v1/connectors/{id}' },
|
|
1092
|
+
{ method: 'POST', path: '/v1/connectors/{id}/oauth/authorize' },
|
|
1093
|
+
{ method: 'POST', path: '/v1/embed-jobs' },
|
|
1094
|
+
{ method: 'GET', path: '/v1/embed-jobs' },
|
|
1095
|
+
{ method: 'GET', path: '/v1/embed-jobs/{id}' },
|
|
1096
|
+
{ method: 'POST', path: '/v1/embed-jobs/{id}/cancel' },
|
|
1097
|
+
];
|
|
1098
|
+
|
|
1099
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
1100
|
+
// DISPATCH
|
|
1101
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
1102
|
+
|
|
1103
|
+
export async function handleCohereTwinRequest(req: CohereRequest): Promise<CohereResponseEnvelope> {
|
|
1104
|
+
const method = req.method.toUpperCase();
|
|
1105
|
+
const [rawPath, rawQuery = ''] = req.path.split('?');
|
|
1106
|
+
const path = (rawPath ?? '').replace(/\/+$/, '') || '/';
|
|
1107
|
+
const query = new URLSearchParams(rawQuery);
|
|
1108
|
+
const seg = path.split('/').filter(Boolean);
|
|
1109
|
+
const params = parseJson(req.body);
|
|
1110
|
+
const readOnly = req.readOnly === true;
|
|
1111
|
+
|
|
1112
|
+
const authFailure = checkAuth(req);
|
|
1113
|
+
if (authFailure) return authFailure;
|
|
1114
|
+
|
|
1115
|
+
const isWrite = method === 'POST' || method === 'DELETE' || method === 'PATCH' || method === 'PUT';
|
|
1116
|
+
if (readOnly && isWrite) return readOnlyRefusal();
|
|
1117
|
+
|
|
1118
|
+
// ── chat ──────────────────────────────────────────────────────────────────────────────
|
|
1119
|
+
if (method === 'POST' && path === '/v2/chat') {
|
|
1120
|
+
const validated = validateChatV2(params);
|
|
1121
|
+
if ('error' in validated) return validated.error;
|
|
1122
|
+
const args = validated.args;
|
|
1123
|
+
// ONE engine advance, here, before anything branches on it.
|
|
1124
|
+
const outcome = req.scenarioEngine ? scenarioDecision(args, req.scenarioEngine) : {};
|
|
1125
|
+
if (outcome.error) return outcome.error;
|
|
1126
|
+
if (args.stream) {
|
|
1127
|
+
if (!req.sseSink) return invalidRequest('stream:true requires a streaming transport');
|
|
1128
|
+
return { status: 200, body: streamChatV2(args, req.sseSink, outcome) };
|
|
1129
|
+
}
|
|
1130
|
+
return { status: 200, body: buildChatV2(args, outcome) };
|
|
1131
|
+
}
|
|
1132
|
+
if (method === 'POST' && path === '/v1/chat') return handleChatV1(params, req);
|
|
1133
|
+
|
|
1134
|
+
// ── embed / rerank / classify ─────────────────────────────────────────────────────────
|
|
1135
|
+
if (method === 'POST' && path === '/v2/embed') return handleEmbedV2(params);
|
|
1136
|
+
if (method === 'POST' && path === '/v1/embed') return handleEmbedV1(params);
|
|
1137
|
+
if (method === 'POST' && path === '/v2/rerank') return handleRerankV2(params);
|
|
1138
|
+
if (method === 'POST' && path === '/v1/rerank') return handleRerankV1(params);
|
|
1139
|
+
if (method === 'POST' && path === '/v1/classify') return handleClassify(params);
|
|
1140
|
+
|
|
1141
|
+
// ── tokenizer ─────────────────────────────────────────────────────────────────────────
|
|
1142
|
+
if (method === 'POST' && path === '/v1/tokenize') return handleTokenize(params, req);
|
|
1143
|
+
if (method === 'POST' && path === '/v1/detokenize') return handleDetokenize(params, req);
|
|
1144
|
+
|
|
1145
|
+
// ── auth probe. A POST despite the name — `Client.js` sends `method: "POST"`. ──────────
|
|
1146
|
+
if (method === 'POST' && path === '/v1/check-api-key') {
|
|
1147
|
+
return { status: 200, body: { valid: true, organization_id: 'org_twin', owner_id: 'owner_twin' } };
|
|
1148
|
+
}
|
|
1149
|
+
|
|
1150
|
+
// ── models ────────────────────────────────────────────────────────────────────────────
|
|
1151
|
+
if (method === 'GET' && path === '/v1/models') {
|
|
1152
|
+
const endpoint = query.get('endpoint');
|
|
1153
|
+
let models = COHERE_MODELS;
|
|
1154
|
+
if (endpoint) models = models.filter((m) => m.endpoints.includes(endpoint as never));
|
|
1155
|
+
const pageSize = Number(query.get('page_size') ?? '0');
|
|
1156
|
+
if (Number.isFinite(pageSize) && pageSize > 0) models = models.slice(0, pageSize);
|
|
1157
|
+
return { status: 200, body: { models, next_page_token: null } };
|
|
1158
|
+
}
|
|
1159
|
+
if (method === 'GET' && seg[0] === 'v1' && seg[1] === 'models' && seg.length === 3) {
|
|
1160
|
+
const name = decodeURIComponent(seg[2]!);
|
|
1161
|
+
const card = findModel(name);
|
|
1162
|
+
return card ? { status: 200, body: card } : modelNotFound(name);
|
|
1163
|
+
}
|
|
1164
|
+
|
|
1165
|
+
// ── datasets ──────────────────────────────────────────────────────────────────────────
|
|
1166
|
+
if (method === 'POST' && path === '/v1/datasets') return createDataset(params, query, req);
|
|
1167
|
+
if (method === 'GET' && path === '/v1/datasets') {
|
|
1168
|
+
let items = rows('dataset', req.root);
|
|
1169
|
+
const dt = query.get('datasetType');
|
|
1170
|
+
if (dt) {
|
|
1171
|
+
// The SAME closed set the create path enforces. Real Cohere serializes this parameter through
|
|
1172
|
+
// `serializers.DatasetType` and 400s on a miss; answering `200 {datasets:[]}` would report an
|
|
1173
|
+
// empty account for what is actually a rejected request (§9 round two, MINOR 9).
|
|
1174
|
+
if (!DATASET_TYPES.has(dt)) return invalidRequest(`datasetType must be one of ${[...DATASET_TYPES].join(', ')}`);
|
|
1175
|
+
items = items.filter((r) => r.dataset_type === dt);
|
|
1176
|
+
}
|
|
1177
|
+
const vs = query.get('validationStatus');
|
|
1178
|
+
if (vs) {
|
|
1179
|
+
if (!VALIDATION_STATUSES.has(vs)) return invalidRequest(`validationStatus must be one of ${[...VALIDATION_STATUSES].join(', ')}`);
|
|
1180
|
+
items = items.filter((r) => r.validation_status === vs);
|
|
1181
|
+
}
|
|
1182
|
+
const limit = Number(query.get('limit') ?? '0');
|
|
1183
|
+
const offset = Number(query.get('offset') ?? '0');
|
|
1184
|
+
if (Number.isFinite(offset) && offset > 0) items = items.slice(offset);
|
|
1185
|
+
if (Number.isFinite(limit) && limit > 0) items = items.slice(0, limit);
|
|
1186
|
+
return { status: 200, body: { datasets: items.map(datasetView) } };
|
|
1187
|
+
}
|
|
1188
|
+
// `/usage` is a LITERAL sibling of `/{id}` and must be matched first, or a dataset could never
|
|
1189
|
+
// be named `usage` and the usage endpoint would be shadowed by the id lookup.
|
|
1190
|
+
if (method === 'GET' && path === '/v1/datasets/usage') {
|
|
1191
|
+
const used = rows('dataset', req.root).reduce((a, r) => a + String(r._content ?? '').length, 0);
|
|
1192
|
+
return { status: 200, body: { organization_usage: used } };
|
|
1193
|
+
}
|
|
1194
|
+
if (seg[0] === 'v1' && seg[1] === 'datasets' && seg.length === 3) {
|
|
1195
|
+
const id = decodeURIComponent(seg[2]!);
|
|
1196
|
+
const row = getRow('dataset', id, req.root);
|
|
1197
|
+
if (method === 'GET') return row ? { status: 200, body: { dataset: datasetView(row) } } : notFound(`dataset '${id}' not found`);
|
|
1198
|
+
if (method === 'DELETE') {
|
|
1199
|
+
if (!row) return notFound(`dataset '${id}' not found`);
|
|
1200
|
+
await write(req, 'dataset.delete', 'dataset', id, { _deleted: true });
|
|
1201
|
+
// Cohere's delete answers an EMPTY object (`Record<string, unknown>` in the SDK), not the
|
|
1202
|
+
// `{deleted:true}` envelope most vendors send.
|
|
1203
|
+
return { status: 200, body: {} };
|
|
1204
|
+
}
|
|
1205
|
+
}
|
|
1206
|
+
|
|
1207
|
+
// ── connectors ────────────────────────────────────────────────────────────────────────
|
|
1208
|
+
if (method === 'POST' && path === '/v1/connectors') return createConnector(params, req);
|
|
1209
|
+
if (method === 'GET' && path === '/v1/connectors') {
|
|
1210
|
+
let items = rows('connector', req.root);
|
|
1211
|
+
const total = items.length;
|
|
1212
|
+
const offset = Number(query.get('offset') ?? '0');
|
|
1213
|
+
const limit = Number(query.get('limit') ?? '0');
|
|
1214
|
+
if (Number.isFinite(offset) && offset > 0) items = items.slice(offset);
|
|
1215
|
+
if (Number.isFinite(limit) && limit > 0) items = items.slice(0, limit);
|
|
1216
|
+
return { status: 200, body: { connectors: items.map(connectorView), total_count: total } };
|
|
1217
|
+
}
|
|
1218
|
+
if (seg[0] === 'v1' && seg[1] === 'connectors' && seg.length === 3) {
|
|
1219
|
+
const id = decodeURIComponent(seg[2]!);
|
|
1220
|
+
const row = getRow('connector', id, req.root);
|
|
1221
|
+
if (method === 'GET') return row ? { status: 200, body: { connector: connectorView(row) } } : notFound(`connector '${id}' not found`);
|
|
1222
|
+
if (method === 'PATCH') {
|
|
1223
|
+
if (!row) return notFound(`connector '${id}' not found`);
|
|
1224
|
+
const patch: Record<string, unknown> = { updated_at: nowIso(req.occurredAt), _rev: nextRev('connector', id, req.root) };
|
|
1225
|
+
for (const k of ['name', 'url', 'excludes', 'active', 'continue_on_failure'] as const) {
|
|
1226
|
+
if (params[k] !== undefined) patch[k] = params[k];
|
|
1227
|
+
}
|
|
1228
|
+
await write(req, 'connector.update', 'connector', id, patch);
|
|
1229
|
+
return { status: 200, body: { connector: connectorView(getRow('connector', id, req.root)!) } };
|
|
1230
|
+
}
|
|
1231
|
+
if (method === 'DELETE') {
|
|
1232
|
+
if (!row) return notFound(`connector '${id}' not found`);
|
|
1233
|
+
await write(req, 'connector.delete', 'connector', id, { _deleted: true });
|
|
1234
|
+
return { status: 200, body: {} };
|
|
1235
|
+
}
|
|
1236
|
+
}
|
|
1237
|
+
if (method === 'POST' && seg[0] === 'v1' && seg[1] === 'connectors' && seg.length === 5 && seg[3] === 'oauth' && seg[4] === 'authorize') {
|
|
1238
|
+
const id = decodeURIComponent(seg[2]!);
|
|
1239
|
+
const row = getRow('connector', id, req.root);
|
|
1240
|
+
if (!row) return notFound(`connector '${id}' not found`);
|
|
1241
|
+
// A connector with no OAuth configuration has nothing to authorize. Answering a fabricated
|
|
1242
|
+
// redirect URL would be a fake success on a path the vendor refuses.
|
|
1243
|
+
if (row.auth_type !== 'oauth') return invalidRequest(`connector '${id}' is not configured for oauth`);
|
|
1244
|
+
const oauth = row.oauth as { authorize_url?: string; client_id?: string } | null;
|
|
1245
|
+
const after = query.get('after_token_redirect');
|
|
1246
|
+
const u = new URL(String(oauth?.authorize_url || 'https://auth.example.test/authorize'));
|
|
1247
|
+
u.searchParams.set('client_id', String(oauth?.client_id ?? ''));
|
|
1248
|
+
u.searchParams.set('state', cohereId(`oauth|${id}`));
|
|
1249
|
+
if (after) u.searchParams.set('after_token_redirect', after);
|
|
1250
|
+
return { status: 200, body: { redirect_url: u.toString() } };
|
|
1251
|
+
}
|
|
1252
|
+
|
|
1253
|
+
// ── embed jobs ────────────────────────────────────────────────────────────────────────
|
|
1254
|
+
if (method === 'POST' && path === '/v1/embed-jobs') return createEmbedJob(params, req);
|
|
1255
|
+
if (method === 'GET' && path === '/v1/embed-jobs') {
|
|
1256
|
+
return { status: 200, body: { embed_jobs: rows('embed_job', req.root).map(embedJobView) } };
|
|
1257
|
+
}
|
|
1258
|
+
if (method === 'GET' && seg[0] === 'v1' && seg[1] === 'embed-jobs' && seg.length === 3) {
|
|
1259
|
+
const id = decodeURIComponent(seg[2]!);
|
|
1260
|
+
const row = getRow('embed_job', id, req.root);
|
|
1261
|
+
return row ? { status: 200, body: embedJobView(row) } : notFound(`embed job '${id}' not found`);
|
|
1262
|
+
}
|
|
1263
|
+
if (method === 'POST' && seg[0] === 'v1' && seg[1] === 'embed-jobs' && seg.length === 4 && seg[3] === 'cancel') {
|
|
1264
|
+
const id = decodeURIComponent(seg[2]!);
|
|
1265
|
+
const row = getRow('embed_job', id, req.root);
|
|
1266
|
+
if (!row) return notFound(`embed job '${id}' not found`);
|
|
1267
|
+
// The vendor's state machine: only a job still in flight can be cancelled. A terminal job
|
|
1268
|
+
// answers a 400 rather than pretending the cancel took.
|
|
1269
|
+
if (row.status !== 'processing') return invalidRequest(`embed job '${id}' is ${String(row.status)} and cannot be cancelled`);
|
|
1270
|
+
await write(req, 'embed_job.cancel', 'embed_job', id, { status: 'cancelling' });
|
|
1271
|
+
// `embedJobs.cancel` is declared `-> void` in the SDK: an empty 200 body.
|
|
1272
|
+
return { status: 200, body: {} };
|
|
1273
|
+
}
|
|
1274
|
+
|
|
1275
|
+
return routeNotFound(method, path);
|
|
1276
|
+
}
|
|
1277
|
+
|
|
1278
|
+
/** Re-exported so tests can assert the twin's closed sets against the vendor's without importing
|
|
1279
|
+
* the private constants by name. */
|
|
1280
|
+
export const COHERE_CLOSED_SETS = {
|
|
1281
|
+
datasetTypes: [...DATASET_TYPES],
|
|
1282
|
+
validationStatuses: [...VALIDATION_STATUSES],
|
|
1283
|
+
embedJobStatuses: [...EMBED_JOB_STATUSES],
|
|
1284
|
+
toolChoices: [...TOOL_CHOICES],
|
|
1285
|
+
safetyModes: [...SAFETY_MODES],
|
|
1286
|
+
v2Roles: [...V2_ROLES],
|
|
1287
|
+
} as const;
|
|
1288
|
+
|
|
1289
|
+
/** Exported for the connector's shared FNV seed and for tests that need the twin's id derivation. */
|
|
1290
|
+
export { fnv1a };
|