@volter/twin-cohere 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/README.md +224 -0
  2. package/defaults/handlers.json +26 -0
  3. package/dist/defaults/handlers.json +26 -0
  4. package/dist/src/cli.d.ts +2 -0
  5. package/dist/src/cli.js +31 -0
  6. package/dist/src/cohere-budget.d.ts +55 -0
  7. package/dist/src/cohere-budget.js +171 -0
  8. package/dist/src/cohere-capabilities.d.ts +14 -0
  9. package/dist/src/cohere-capabilities.js +1852 -0
  10. package/dist/src/cohere-conformance.d.ts +17 -0
  11. package/dist/src/cohere-conformance.js +464 -0
  12. package/dist/src/cohere-connector.d.ts +150 -0
  13. package/dist/src/cohere-connector.js +625 -0
  14. package/dist/src/cohere-models.d.ts +21 -0
  15. package/dist/src/cohere-models.js +73 -0
  16. package/dist/src/cohere-scenario.d.ts +57 -0
  17. package/dist/src/cohere-scenario.js +176 -0
  18. package/dist/src/cohere-server.d.ts +16 -0
  19. package/dist/src/cohere-server.js +184 -0
  20. package/dist/src/cohere-stub.d.ts +119 -0
  21. package/dist/src/cohere-stub.js +321 -0
  22. package/dist/src/cohere-twin.d.ts +82 -0
  23. package/dist/src/cohere-twin.js +1243 -0
  24. package/dist/src/cohere-types.d.ts +226 -0
  25. package/dist/src/cohere-types.js +40 -0
  26. package/dist/src/index.d.ts +15 -0
  27. package/dist/src/index.js +84 -0
  28. package/package.json +71 -0
  29. package/src/cli.ts +30 -0
  30. package/src/cohere-budget.ts +197 -0
  31. package/src/cohere-capabilities.ts +1855 -0
  32. package/src/cohere-conformance.ts +489 -0
  33. package/src/cohere-connector.ts +709 -0
  34. package/src/cohere-models.ts +79 -0
  35. package/src/cohere-scenario.ts +194 -0
  36. package/src/cohere-server.ts +195 -0
  37. package/src/cohere-stub.ts +337 -0
  38. package/src/cohere-twin.ts +1290 -0
  39. package/src/cohere-types.ts +231 -0
  40. package/src/index.ts +159 -0
@@ -0,0 +1,321 @@
1
+ // ── hashing ─────────────────────────────────────────────────────────────────────────────
2
+ /** A small deterministic 32-bit hash (FNV-1a) of a string — the seed for every stub below. */
3
+ export function fnv1a(text) {
4
+ let h = 0x811c9dc5;
5
+ for (let i = 0; i < text.length; i++) {
6
+ h ^= text.charCodeAt(i);
7
+ h = Math.imul(h, 0x01000193);
8
+ }
9
+ return h >>> 0;
10
+ }
11
+ /** A deterministic lowercase-hex id of `length` chars, seeded from `seed`. Cohere's response ids
12
+ * are UUID-shaped; `cohereId` below assembles one from this. */
13
+ function hexFrom(seed, length) {
14
+ let h = fnv1a(seed) || 1;
15
+ let out = '';
16
+ while (out.length < length) {
17
+ h ^= h << 13;
18
+ h >>>= 0;
19
+ h ^= h >> 17;
20
+ h ^= h << 5;
21
+ h >>>= 0;
22
+ out += h.toString(16).padStart(8, '0');
23
+ }
24
+ return out.slice(0, length);
25
+ }
26
+ /**
27
+ * A deterministic UUID-v4-SHAPED id. Cohere returns UUIDs for `id` / `generation_id` /
28
+ * `response_id`, so the twin returns something a caller parsing a UUID accepts — derived from the
29
+ * request rather than from `crypto.randomUUID`, because a random id would make the served response
30
+ * non-deterministic and break replay (CLAUDE.md, serve-path determinism).
31
+ */
32
+ export function cohereId(seed) {
33
+ const h = hexFrom(seed, 32);
34
+ // Set the version nibble to 4 and the variant nibble to 8 so the string is a well-formed UUIDv4.
35
+ return `${h.slice(0, 8)}-${h.slice(8, 12)}-4${h.slice(13, 16)}-8${h.slice(17, 20)}-${h.slice(20, 32)}`;
36
+ }
37
+ // ── token accounting ────────────────────────────────────────────────────────────────────
38
+ /** Deterministic token estimate: ~1 token per 4 chars. Never zero for non-empty text. */
39
+ export function estimateTokens(text) {
40
+ if (!text)
41
+ return 0;
42
+ return Math.max(1, Math.ceil(text.length / 4));
43
+ }
44
+ /** Flatten a v2 message's content (string OR content-part array) to its text. Cohere's parts are
45
+ * `{type:'text',text}`, `{type:'image_url',image_url}` and `{type:'document',document}`; only the
46
+ * first contributes literal text, the rest contribute their JSON length so the count stays
47
+ * deterministic and reflects payload size. */
48
+ export function contentToText(content) {
49
+ if (typeof content === 'string')
50
+ return content;
51
+ if (content === null || content === undefined)
52
+ return '';
53
+ if (!Array.isArray(content))
54
+ return '';
55
+ return content
56
+ .map((part) => {
57
+ if (part && typeof part === 'object' && part.type === 'text') {
58
+ return String(part.text ?? '');
59
+ }
60
+ return JSON.stringify(part);
61
+ })
62
+ .join('\n');
63
+ }
64
+ /** Deterministic input-token count for a full v2 message list. */
65
+ export function countInputTokens(messages) {
66
+ let total = 0;
67
+ for (const m of messages) {
68
+ total += estimateTokens(contentToText(m.content));
69
+ if (typeof m.tool_plan === 'string')
70
+ total += estimateTokens(m.tool_plan);
71
+ for (const tc of m.tool_calls ?? [])
72
+ total += estimateTokens(JSON.stringify(tc));
73
+ }
74
+ return total;
75
+ }
76
+ /** The last user turn's text — the thing the stub echoes. */
77
+ export function lastUserText(messages) {
78
+ for (let i = messages.length - 1; i >= 0; i--) {
79
+ if (messages[i].role === 'user')
80
+ return contentToText(messages[i].content);
81
+ }
82
+ return messages.length ? contentToText(messages[messages.length - 1].content) : '';
83
+ }
84
+ // ── chat stubs ──────────────────────────────────────────────────────────────────────────
85
+ /**
86
+ * The deterministic stub ASSISTANT text. Unmistakably a twin stub: it carries the
87
+ * `[twin-stub:<model>]` marker and echoes the prompt, so no caller can mistake it for real model
88
+ * output. Deterministic for a given prompt → assertable in tests.
89
+ */
90
+ export function stubAssistantText(messages, model) {
91
+ const prompt = lastUserText(messages).trim();
92
+ const echo = prompt.length > 200 ? `${prompt.slice(0, 200)}…` : prompt;
93
+ return `[twin-stub:${model}] This is a deterministic stub from the Cohere twin (no model weights are run). Echoing your last message: ${echo || '(empty)'}`;
94
+ }
95
+ /** The v1 chat stub — v1 takes a single `message` string rather than a message list. */
96
+ export function stubV1Text(message, model) {
97
+ const echo = message.length > 200 ? `${message.slice(0, 200)}…` : message;
98
+ return `[twin-stub:${model}] This is a deterministic stub from the Cohere twin (no model weights are run). Echoing your message: ${echo || '(empty)'}`;
99
+ }
100
+ /** The `tool_plan` a real Cohere model emits alongside tool calls — a chain-of-thought style
101
+ * reflection. Stubbed deterministically and labeled. */
102
+ export function stubToolPlan(toolNames) {
103
+ return `[twin-stub] I will call ${toolNames.length === 1 ? 'the tool' : 'the tools'} ${toolNames.join(', ')} to answer this.`;
104
+ }
105
+ function toolName(t) {
106
+ const o = t;
107
+ if (o?.function && typeof o.function.name === 'string')
108
+ return o.function.name;
109
+ if (typeof o?.name === 'string')
110
+ return o.name;
111
+ return 'unknown_function';
112
+ }
113
+ /** Every declared tool's name, in declaration order. */
114
+ export function toolNames(tools) {
115
+ if (!Array.isArray(tools))
116
+ return [];
117
+ return tools.map(toolName);
118
+ }
119
+ function placeholderForSchema(def) {
120
+ const d = def;
121
+ if (Array.isArray(d?.enum) && d.enum.length)
122
+ return d.enum[0];
123
+ switch (d?.type) {
124
+ case 'number':
125
+ case 'integer': return 0;
126
+ case 'boolean': return false;
127
+ case 'array': return [];
128
+ case 'object': return {};
129
+ default: return '';
130
+ }
131
+ }
132
+ /**
133
+ * A deterministic stub argument STRING for a tool. Real models emit arguments that validate
134
+ * against the tool's declared JSON schema, so the twin synthesizes a deterministic object
135
+ * containing every declared property with a type-appropriate placeholder. Cohere's `arguments` is
136
+ * a STRING on the wire (`ToolCallV2Function.arguments`), like OpenAI's.
137
+ */
138
+ export function stubToolArguments(tool) {
139
+ const o = tool;
140
+ const schema = (o?.function?.parameters ?? o?.parameters);
141
+ const props = schema && typeof schema === 'object' ? schema.properties : undefined;
142
+ if (!props || typeof props !== 'object')
143
+ return '{}';
144
+ const out = {};
145
+ for (const [key, def] of Object.entries(props))
146
+ out[key] = placeholderForSchema(def);
147
+ return JSON.stringify(out);
148
+ }
149
+ /**
150
+ * When tools are provided, a real Cohere model may answer with `tool_calls` and
151
+ * `finish_reason: 'TOOL_CALL'`. The stub deterministically "calls" the tool selected by
152
+ * `forcedName` (a named `tool_choice`) or the FIRST provided tool.
153
+ *
154
+ * The id is SEEDED FROM THE REQUEST, never from a module-level counter: a counter makes the id
155
+ * depend on how many completions the process happened to serve, which is not a property a replay
156
+ * can rely on, and hands two different turns the same id inside one process. Hashing the
157
+ * conversation keeps replay byte-identical while making ids differ between turns.
158
+ */
159
+ export function stubToolCall(tools, seed, index, forcedName) {
160
+ if (!Array.isArray(tools) || tools.length === 0)
161
+ return null;
162
+ const chosen = forcedName ? (tools.find((t) => toolName(t) === forcedName) ?? tools[0]) : tools[0];
163
+ const name = toolName(chosen);
164
+ return { id: toolCallId(`${seed}|${name}|${index}`), type: 'function', function: { name, arguments: stubToolArguments(chosen) } };
165
+ }
166
+ /** Cohere's tool-call ids are UUID-shaped, like its response ids. */
167
+ export function toolCallId(seed) {
168
+ return cohereId(`cohere-tool-${seed}`);
169
+ }
170
+ // ── embeddings: deterministic pseudo-vectors ────────────────────────────────────────────
171
+ /**
172
+ * A deterministic, reproducible L2-normalized pseudo-embedding for `text`. NOT a real embedding —
173
+ * the values carry no semantic meaning; only the SHAPE, the dimensionality and the DETERMINISM are
174
+ * faithful. Same text → same vector, in every process
175
+ * and every state root.
176
+ */
177
+ export function pseudoEmbedding(text, dimensions) {
178
+ const dim = Math.max(1, Math.floor(dimensions));
179
+ let state = fnv1a(text) || 1;
180
+ const raw = new Array(dim);
181
+ let norm = 0;
182
+ for (let i = 0; i < dim; i++) {
183
+ state ^= state << 13;
184
+ state >>>= 0;
185
+ state ^= state >> 17;
186
+ state ^= state << 5;
187
+ state >>>= 0;
188
+ const v = (state / 0xffffffff) * 2 - 1;
189
+ raw[i] = v;
190
+ norm += v * v;
191
+ }
192
+ norm = Math.sqrt(norm) || 1;
193
+ for (let i = 0; i < dim; i++)
194
+ raw[i] = raw[i] / norm;
195
+ return raw;
196
+ }
197
+ /**
198
+ * Quantize a float vector into one of Cohere's non-float `embedding_types`. The vendor returns the
199
+ * SAME vector under every requested type, quantized — so the twin derives all of them from the one
200
+ * pseudo-vector rather than hashing separately per type, which is what makes the types agree the
201
+ * way a real response's do.
202
+ * • int8 — signed 8-bit, [-128, 127]
203
+ * • uint8 — unsigned 8-bit, [0, 255]
204
+ * • binary/ubinary — one BIT per dimension, packed 8 to a byte (so the array is dim/8 long),
205
+ * signed and unsigned respectively; this packing is why a binary embedding is 1/32 the size.
206
+ */
207
+ export function quantizeEmbedding(floats, type) {
208
+ if (type === 'int8')
209
+ return floats.map((v) => clamp(Math.round(v * 127), -128, 127));
210
+ if (type === 'uint8')
211
+ return floats.map((v) => clamp(Math.round((v + 1) * 127.5), 0, 255));
212
+ // binary / ubinary: pack the SIGN of each dimension, 8 bits to a byte, most-significant first.
213
+ const bytes = [];
214
+ for (let i = 0; i < floats.length; i += 8) {
215
+ let b = 0;
216
+ for (let bit = 0; bit < 8; bit++) {
217
+ const v = floats[i + bit];
218
+ if (v !== undefined && v > 0)
219
+ b |= 1 << (7 - bit);
220
+ }
221
+ bytes.push(type === 'binary' ? (b > 127 ? b - 256 : b) : b);
222
+ }
223
+ return bytes;
224
+ }
225
+ /** Base64 of the float vector's little-endian float32 bytes — Cohere's `base64` embedding type. */
226
+ export function base64Embedding(floats) {
227
+ const buf = new ArrayBuffer(floats.length * 4);
228
+ const view = new DataView(buf);
229
+ for (let i = 0; i < floats.length; i++)
230
+ view.setFloat32(i * 4, floats[i], true);
231
+ const bytes = new Uint8Array(buf);
232
+ let binary = '';
233
+ for (const b of bytes)
234
+ binary += String.fromCharCode(b);
235
+ // btoa is available in Bun and in every runtime this pack targets.
236
+ return btoa(binary);
237
+ }
238
+ function clamp(n, lo, hi) {
239
+ return n < lo ? lo : n > hi ? hi : n;
240
+ }
241
+ /** The output dimensionality Cohere's embed models produce, by model id. `embed-v4.0` accepts an
242
+ * explicit `output_dimension` from the closed set {256, 512, 1024, 1536}; the v3 models are fixed
243
+ * (1024 for the full models, 384 for the `light` ones). */
244
+ export function embedDimensions(model) {
245
+ if (model.includes('light'))
246
+ return 384;
247
+ if (model === 'embed-v4.0')
248
+ return 1536;
249
+ return 1024;
250
+ }
251
+ /** `embed-v4.0`'s documented closed `output_dimension` set. A literal here, so the handler's
252
+ * rejection of anything outside it is checkable against the vendor's documented set. */
253
+ export const COHERE_OUTPUT_DIMENSIONS = [256, 512, 1024, 1536];
254
+ // ── rerank: a deterministic lexical heuristic ───────────────────────────────────────────
255
+ /**
256
+ * Score one document against a query. NOT the real cross-encoder — a deterministic
257
+ * token-overlap ratio, so the ORDERING is explainable and reproducible while carrying no learned
258
+ * relevance. Always in [0, 1], the range the vendor's `relevance_score` occupies.
259
+ *
260
+ * A tiny hash term breaks ties deterministically, so two documents with identical overlap still
261
+ * get a stable, total order rather than an order that depends on sort stability.
262
+ */
263
+ export function rerankScore(query, document) {
264
+ const q = new Set(tokens(query));
265
+ const d = tokens(document);
266
+ if (q.size === 0 || d.length === 0)
267
+ return 0;
268
+ let hits = 0;
269
+ for (const t of d)
270
+ if (q.has(t))
271
+ hits++;
272
+ const overlap = hits / Math.max(q.size, d.length);
273
+ const tiebreak = (fnv1a(`${query}|${document}`) % 1000) / 1_000_000; // < 0.001
274
+ return Math.min(1, Number((overlap + tiebreak).toFixed(6)));
275
+ }
276
+ function tokens(text) {
277
+ return text.toLowerCase().split(/[^a-z0-9]+/).filter(Boolean);
278
+ }
279
+ // ── classify: a deterministic label heuristic ───────────────────────────────────────────
280
+ /**
281
+ * Deterministically score `input` across `labels`. NOT the real classifier — the confidence is a
282
+ * hash-derived value per (input, label), normalized so the confidences SUM TO 1 the way a real
283
+ * single-label classification's do. The prediction is the argmax.
284
+ */
285
+ export function classifyText(input, labels) {
286
+ const raw = labels.map((l) => (fnv1a(`${input}::${l}`) % 10_000) + 1);
287
+ const total = raw.reduce((a, b) => a + b, 0);
288
+ const confidences = raw.map((r) => Number((r / total).toFixed(6)));
289
+ let best = 0;
290
+ for (let i = 1; i < confidences.length; i++)
291
+ if (confidences[i] > confidences[best])
292
+ best = i;
293
+ return { prediction: labels[best], confidences };
294
+ }
295
+ // ── tokenize / detokenize: a deterministic, LEARNED vocabulary ──────────────────────────
296
+ /**
297
+ * Segment text the way the twin's tokenizer does. NOT Cohere's BPE (vendoring the published
298
+ * tokenizer JSON is `cohere.tokenize.bpe_vocabulary`, a todo) — a deterministic word-with-trailing-whitespace
299
+ * split, chosen so that JOINING the segments restores the input EXACTLY, including runs of spaces
300
+ * and newlines. That exact-join property is what lets `detokenize(tokenize(t)) === t` hold, which
301
+ * is the property a caller actually depends on.
302
+ */
303
+ export function segmentText(text) {
304
+ if (text === '')
305
+ return [];
306
+ return text.match(/\S+\s*|\s+/g) ?? [text];
307
+ }
308
+ /**
309
+ * The PREFERRED token id for a segment — a pure function of the segment, so the same text
310
+ * tokenizes to the same ids in every process and every state root.
311
+ *
312
+ * This is only a preference, not the answer: two different segments can hash to the same value,
313
+ * and a vocabulary that silently let the second one overwrite the first would make `detokenize`
314
+ * return the WRONG text for the earlier one. The handler resolves that by probing upward over the
315
+ * vocabulary it has already observed (`cohere-twin.ts`, `assignToken`) — deterministic given the
316
+ * stored state, which is what a twin's determinism means. Ids start at 1: Cohere's token ids are
317
+ * positive, and 0 is reserved so an unset field can never read as a valid token.
318
+ */
319
+ export function preferredTokenId(segment) {
320
+ return (fnv1a(`cohere-tok:${segment}`) % 250_000) + 1;
321
+ }
@@ -0,0 +1,82 @@
1
+ import { fnv1a } from './cohere-stub.js';
2
+ import { type CohereScenarioEngine, type ScriptedResult } from './cohere-scenario.js';
3
+ import { type CohereChatV2Response, type CohereMessageV2, type SseSink } from './cohere-types.js';
4
+ export type CohereRequest = {
5
+ /** The scenario engine (kernel grammar + this pack's vocabulary) — scripts chat turns. */
6
+ scenarioEngine?: CohereScenarioEngine;
7
+ method: string;
8
+ path: string;
9
+ body?: string;
10
+ occurredAt?: string;
11
+ root?: string;
12
+ readOnly?: boolean;
13
+ /** Lower-cased request headers (e.g. `authorization`) the HTTP server passes through so the
14
+ * handler can model auth (401). In-process trusted calls (capability verifies, the connector)
15
+ * omit them and are not auth-gated — the twin cannot validate against real keys, so the
16
+ * modeled failure is the CHECKABLE missing/sentinel-invalid case. */
17
+ headers?: Record<string, string>;
18
+ /** When set on a streaming POST, chunks are written here (no sockets). */
19
+ sseSink?: SseSink;
20
+ };
21
+ /** The handler response. `headers` (when present) are response headers the HTTP server should
22
+ * set — e.g. `Retry-After` on a scripted 429. */
23
+ export type CohereResponseEnvelope = {
24
+ status: number;
25
+ body: unknown;
26
+ headers?: Record<string, string>;
27
+ };
28
+ export type ChatV2Args = {
29
+ model: string;
30
+ messages: CohereMessageV2[];
31
+ tools?: unknown;
32
+ toolChoice?: 'REQUIRED' | 'NONE';
33
+ maxTokens?: number;
34
+ stream: boolean;
35
+ thinking: boolean;
36
+ };
37
+ /** The outcome of ONE engine advance. `missTeach` is appended to the stub so an in-world agent
38
+ * that never matched a handler is told what features it would have to match on. */
39
+ type ScenarioOutcome = {
40
+ scripted?: ScriptedResult;
41
+ error?: CohereResponseEnvelope;
42
+ missTeach?: string;
43
+ };
44
+ /** Build the unary `POST /v2/chat` envelope. */
45
+ export declare function buildChatV2(args: ChatV2Args, outcome?: ScenarioOutcome): CohereChatV2Response;
46
+ /**
47
+ * Stream `POST /v2/chat` into the injected sink and return the same body the unary path would.
48
+ *
49
+ * The event sequence is Cohere's own, and it is asserted key-for-key against
50
+ * @ai-sdk/cohere's `cohereChatChunkSchema` — a `z.discriminatedUnion('type', …)`, so a wrong
51
+ * `type` or a missing `delta.message.content` is a hard parse failure in the real provider, not a
52
+ * soft mismatch. Note the shapes differ between `content-start` (a full content BLOCK under
53
+ * `delta.message.content`) and `content-delta` (just `{ text }`), which is exactly the sort of
54
+ * detail a hand-written stream gets wrong.
55
+ */
56
+ export declare function streamChatV2(args: ChatV2Args, sink: SseSink, outcome?: ScenarioOutcome): CohereChatV2Response;
57
+ /**
58
+ * Every method/path pair the dispatch below branches on. HAND-AUTHORED (the honest limit): a
59
+ * branch added to the handler and to neither this list nor the conformance snapshot is invisible
60
+ * to the conformance check's third direction. The mechanical backstop for that lives in
61
+ * `cohere-twin.test.ts` ("covers every literal path the handler dispatches on"), which reads this
62
+ * module's own source; segment-matched sub-routes still rest on review.
63
+ *
64
+ * `{id}` stands for one path segment.
65
+ */
66
+ export declare const COHERE_ROUTER_SURFACE: ReadonlyArray<{
67
+ method: string;
68
+ path: string;
69
+ }>;
70
+ export declare function handleCohereTwinRequest(req: CohereRequest): Promise<CohereResponseEnvelope>;
71
+ /** Re-exported so tests can assert the twin's closed sets against the vendor's without importing
72
+ * the private constants by name. */
73
+ export declare const COHERE_CLOSED_SETS: {
74
+ readonly datasetTypes: readonly string[];
75
+ readonly validationStatuses: readonly string[];
76
+ readonly embedJobStatuses: readonly string[];
77
+ readonly toolChoices: readonly string[];
78
+ readonly safetyModes: readonly string[];
79
+ readonly v2Roles: readonly string[];
80
+ };
81
+ /** Exported for the connector's shared FNV seed and for tests that need the twin's id derivation. */
82
+ export { fnv1a };