@volter/twin-cohere 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -4
- package/dist/src/cohere-capabilities.d.ts +1 -1
- package/dist/src/cohere-capabilities.js +106 -2
- package/dist/src/cohere-conformance.js +12 -0
- package/dist/src/cohere-server.js +5 -3
- package/dist/src/cohere-twin.d.ts +5 -0
- package/dist/src/cohere-twin.js +160 -2
- package/dist/src/index.js +8 -2
- package/package.json +4 -3
- package/src/cohere-capabilities.ts +108 -2
- package/src/cohere-conformance.ts +12 -0
- package/src/cohere-server.ts +5 -3
- package/src/cohere-twin.ts +152 -2
- package/src/index.ts +8 -2
package/README.md
CHANGED
|
@@ -54,6 +54,17 @@ modeled here and asserted by a named capability:
|
|
|
54
54
|
| `check-api-key` | a **POST**, despite the name | a GET |
|
|
55
55
|
| streaming | v2 is SSE terminated by `data: [DONE]`; v1 is newline-delimited JSON tagged `event_type` | one wire |
|
|
56
56
|
|
|
57
|
+
**The exception is the Compatibility API**, `https://api.cohere.ai/compatibility/v1`
|
|
58
|
+
([docs](https://docs.cohere.com/docs/compatibility-api)), where Cohere speaks OpenAI's wire, `choices` and all:
|
|
59
|
+
`POST /compatibility/v1/chat/completions` answers `chat.completion` (and streams `chat.completion.chunk` frames ending
|
|
60
|
+
`[DONE]`, with a usage chunk under `stream_options.include_usage`) for Cohere's models, so an application built on the
|
|
61
|
+
OpenAI SDK reaches Cohere by changing its base URL. The twin serves it from the same turn as `/v2/chat` (stub, tool
|
|
62
|
+
decision and handlers) in OpenAI's envelope; its embeddings, audio, `response_format` and `reasoning_effort` are
|
|
63
|
+
`todo`. Where the docs stop, the twin decides, and none of it is probed against the live API: the parameters the docs
|
|
64
|
+
name unsupported are ignored (the docs do not say whether Cohere refuses or ignores them), refusals keep the host's
|
|
65
|
+
bare `{message}` body, `ERROR` and `TIMEOUT` finish as `stop` (OpenAI's wire has no error finish), and a tool turn
|
|
66
|
+
carries no `tool_plan`.
|
|
67
|
+
|
|
57
68
|
## The generative-stub design (be honest)
|
|
58
69
|
|
|
59
70
|
The twin **cannot run the model** — there are no weights here. So every generative endpoint returns
|
|
@@ -106,8 +117,9 @@ top-down from `cohere-ai@8.1.0`'s own operation inventory — every `client.<sub
|
|
|
106
117
|
client module.
|
|
107
118
|
|
|
108
119
|
**One unit, throughout:** those **45 SDK methods** resolve to **42 distinct (method, path)
|
|
109
|
-
operations** over 32 paths. The twin serves **27** of the 42
|
|
110
|
-
|
|
120
|
+
operations** over 32 paths. The twin serves **27** of the 42. `COHERE_ROUTER_SURFACE` and the conformance snapshot hold **28**
|
|
121
|
+
pairs: those 27, and the Compatibility API's chat completions, which is outside `cohere-ai`'s denominator (it is
|
|
122
|
+
reached with the OpenAI SDK). The **15** it does not: `POST /v1/generate`,
|
|
111
123
|
`POST /v1/summarize`, `POST /v2/parse`, `POST /v2/audio/transcriptions`, the four `/v2/batches`
|
|
112
124
|
operations and the seven `/v1/finetuning/*` operations — every one enumerated as a `todo`, because a
|
|
113
125
|
denominator that omitted them would be a self-portrait rather than a measurement. Run
|
|
@@ -198,8 +210,10 @@ surface.
|
|
|
198
210
|
|
|
199
211
|
## Interception and world wiring
|
|
200
212
|
|
|
201
|
-
- **Hosts:**
|
|
202
|
-
|
|
213
|
+
- **Hosts:** `api.cohere.com`, the only entry in `cohere-ai`'s `CohereEnvironment` and
|
|
214
|
+
`@ai-sdk/cohere`'s default `baseURL` (plus `/v2`), and `api.cohere.ai`, Cohere's earlier host,
|
|
215
|
+
which its Compatibility API documents and applications speaking OpenAI's wire address
|
|
216
|
+
(LibreChat's own Cohere constant is `https://api.cohere.ai/v1`). Declared on
|
|
203
217
|
the pack descriptor, so `volter-world` routes an unmodified SDK's traffic here with zero edits.
|
|
204
218
|
- **Endpoint env: none, and that is a ruling rather than an omission.** Neither SDK reads a base-URL
|
|
205
219
|
environment variable — `cohere-ai` takes the override as the `baseUrl`/`environment` **constructor**
|
|
@@ -9,6 +9,6 @@ import { type CapabilityReport, type CapabilitySpec } from '@volter/world-toolin
|
|
|
9
9
|
* `connectors` is Cohere's PRODUCT (its RAG connector registry); `connector` is this twin's
|
|
10
10
|
* pull/push plane. Two different things that unfortunately share a word; both are real areas.
|
|
11
11
|
*/
|
|
12
|
-
export declare const COHERE_AREAS: readonly ["audio", "auth", "batches", "chat", "chat_v1", "classify", "conformance", "connector", "connectors", "datasets", "deployments", "determinism", "embed", "embed_jobs", "errors", "finetuning", "generate", "models", "parse", "rerank", "summarize", "tokenize"];
|
|
12
|
+
export declare const COHERE_AREAS: readonly ["audio", "auth", "batches", "chat", "chat_v1", "compat", "classify", "conformance", "connector", "connectors", "datasets", "deployments", "determinism", "embed", "embed_jobs", "errors", "finetuning", "generate", "models", "parse", "rerank", "summarize", "tokenize"];
|
|
13
13
|
export declare const COHERE_CAPABILITIES: CapabilitySpec[];
|
|
14
14
|
export declare function cohereCapabilities(): Promise<CapabilityReport>;
|
|
@@ -52,7 +52,7 @@ import { checkCohereConformance } from "./cohere-conformance.js";
|
|
|
52
52
|
import { cohereRequestForAction, fullSyncCohere, liveCohereExecute, mapConnector, mapDataset, mapEmbedJob, pollTimestamp, pushPendingCohereActions, syncCohereFromReal, unpushableReason, } from "./cohere-connector.js";
|
|
53
53
|
import { COHERE_ENDPOINTS } from "./cohere-models.js";
|
|
54
54
|
import { createCohereScenarioEngine } from "./cohere-scenario.js";
|
|
55
|
-
import { handleCohereTwinRequest } from "./cohere-twin.js";
|
|
55
|
+
import { compatMessages, handleCohereTwinRequest } from "./cohere-twin.js";
|
|
56
56
|
/** A FIXED `occurredAt` keeps ids/timestamps deterministic across runs. */
|
|
57
57
|
const PINNED_AT = '2026-08-31T12:00:00.000Z';
|
|
58
58
|
/** Run a sequence of real Cohere requests against an ISOLATED root; return all responses. The
|
|
@@ -178,6 +178,7 @@ export const COHERE_AREAS = [
|
|
|
178
178
|
'batches',
|
|
179
179
|
'chat',
|
|
180
180
|
'chat_v1',
|
|
181
|
+
'compat',
|
|
181
182
|
'classify',
|
|
182
183
|
'conformance',
|
|
183
184
|
'connector',
|
|
@@ -266,6 +267,21 @@ export const COHERE_CAPABILITIES = [
|
|
|
266
267
|
&& none.body.message?.tool_calls === undefined
|
|
267
268
|
&& typeof none.body.message?.content?.[0]?.text === 'string';
|
|
268
269
|
})),
|
|
270
|
+
done('cohere.chat.tool_result_answered', 'chat', 'v2 chat: a turn answering a tool result is words, not another tool call, unless tool_choice is REQUIRED', 'api', 'core', () => withRoot(async (h) => {
|
|
271
|
+
const first = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ tools: [TOOL] }) });
|
|
272
|
+
const call = first.body.message?.tool_calls?.[0];
|
|
273
|
+
if (!call)
|
|
274
|
+
return false;
|
|
275
|
+
const loop = [
|
|
276
|
+
{ role: 'user', content: 'weather?' },
|
|
277
|
+
{ role: 'assistant', tool_calls: [call], tool_plan: first.body.message.tool_plan },
|
|
278
|
+
{ role: 'tool', tool_call_id: call.id, content: [{ type: 'document', document: { data: '{"temp":20}' } }] },
|
|
279
|
+
];
|
|
280
|
+
const answered = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ tools: [TOOL], messages: loop }) });
|
|
281
|
+
const insisted = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ tools: [TOOL], messages: loop, tool_choice: 'REQUIRED' }) });
|
|
282
|
+
return answered.body.finish_reason === 'COMPLETE' && typeof answered.body.message?.content?.[0]?.text === 'string'
|
|
283
|
+
&& insisted.body.finish_reason === 'TOOL_CALL';
|
|
284
|
+
})),
|
|
269
285
|
done('cohere.chat.thinking', 'chat', 'v2 chat: `thinking: {type:enabled}` prepends a `thinking` content item', 'api', 'common', () => withRoot(async (h) => {
|
|
270
286
|
const off = await h({ m: 'POST', p: '/v2/chat', b: CHAT() });
|
|
271
287
|
const on = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ thinking: { type: 'enabled' } }) });
|
|
@@ -323,6 +339,94 @@ export const COHERE_CAPABILITIES = [
|
|
|
323
339
|
const end = events.find((e) => e.data?.type === 'message-end').data;
|
|
324
340
|
return end.delta.finish_reason === 'TOOL_CALL';
|
|
325
341
|
})),
|
|
342
|
+
// ══ THE COMPATIBILITY API — OpenAI's wire at api.cohere.ai/compatibility/v1 ════════════
|
|
343
|
+
done('cohere.compat.chat_completions', 'compat', 'Compatibility API: POST /compatibility/v1/chat/completions answers OpenAI\'s chat.completion envelope for a Cohere model', 'api', 'core', () => withRoot(async (h) => {
|
|
344
|
+
const r = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT() });
|
|
345
|
+
if (!ok(r))
|
|
346
|
+
return false;
|
|
347
|
+
const b = r.body;
|
|
348
|
+
// OpenAI's keys PRESENT, Cohere v2's ABSENT: the same turn in the other wire, not the v2 body relabelled.
|
|
349
|
+
if (b.object !== 'chat.completion' || typeof b.created !== 'number' || b.model !== 'command-a-03-2025')
|
|
350
|
+
return false;
|
|
351
|
+
if (b.message !== undefined || b.finish_reason !== undefined)
|
|
352
|
+
return false;
|
|
353
|
+
const c = b.choices?.[0];
|
|
354
|
+
if (b.choices.length !== 1 || c.index !== 0 || c.message?.role !== 'assistant' || c.finish_reason !== 'stop')
|
|
355
|
+
return false;
|
|
356
|
+
if (typeof c.message.content !== 'string' || !c.message.content.includes('[twin-stub:command-a-03-2025]') || !c.message.content.includes('hello twin'))
|
|
357
|
+
return false;
|
|
358
|
+
// The flat usage, whose counts move with the input (a constant would pass the shape check).
|
|
359
|
+
const long = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ messages: [{ role: 'user', content: 'x'.repeat(400) }] }) });
|
|
360
|
+
const u = b.usage;
|
|
361
|
+
const lu = long.body.usage;
|
|
362
|
+
return u.total_tokens === u.prompt_tokens + u.completion_tokens && lu.prompt_tokens > u.prompt_tokens * 5;
|
|
363
|
+
})),
|
|
364
|
+
done('cohere.compat.tool_calls', 'compat', 'Compatibility API: tools → OpenAI `tool_calls` with finish_reason tool_calls; tool_choice none forbids it; the tool result is answered in words', 'api', 'core', () => withRoot(async (h) => {
|
|
365
|
+
const call = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL] }) });
|
|
366
|
+
const none = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL], tool_choice: 'none' }) });
|
|
367
|
+
const c = call.body.choices?.[0];
|
|
368
|
+
if (c?.finish_reason !== 'tool_calls' || c.message?.content !== null)
|
|
369
|
+
return false;
|
|
370
|
+
const tc = c.message.tool_calls?.[0];
|
|
371
|
+
if (tc?.type !== 'function' || tc.function?.name !== 'get_weather' || typeof tc.function.arguments !== 'string')
|
|
372
|
+
return false;
|
|
373
|
+
if (none.body.choices?.[0]?.finish_reason !== 'stop' || none.body.choices[0].message.tool_calls !== undefined)
|
|
374
|
+
return false;
|
|
375
|
+
// OpenAI's tool loop: the assistant's tool call, then the `tool` result — the next turn is words, not another call.
|
|
376
|
+
const answered = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL], messages: [
|
|
377
|
+
{ role: 'user', content: 'weather?' },
|
|
378
|
+
{ role: 'assistant', content: null, tool_calls: [tc] },
|
|
379
|
+
{ role: 'tool', tool_call_id: tc.id, content: '{"temp":20}' },
|
|
380
|
+
] }) });
|
|
381
|
+
return ok(answered) && typeof answered.body.choices?.[0]?.message?.content === 'string';
|
|
382
|
+
})),
|
|
383
|
+
done('cohere.compat.stream', 'compat', 'Compatibility API streaming: chat.completion.chunk frames reassemble the unary answer, end with the finish reason, a usage chunk when stream_options asks, then [DONE]', 'api', 'core', () => withStream('/compatibility/v1/chat/completions', CHAT({ stream: true, stream_options: { include_usage: true } }), (events, final) => {
|
|
384
|
+
const frames = events.filter((e) => !e.done).map((e) => e.data);
|
|
385
|
+
if (events[events.length - 1]?.done !== true)
|
|
386
|
+
return false;
|
|
387
|
+
if (!frames.every((f) => f.object === 'chat.completion.chunk' && f.id === final.body.id))
|
|
388
|
+
return false;
|
|
389
|
+
if (frames[0].choices[0].delta.role !== 'assistant')
|
|
390
|
+
return false;
|
|
391
|
+
const last = frames[frames.length - 1];
|
|
392
|
+
// the usage-only chunk: no choices, the unary usage
|
|
393
|
+
if (last.choices.length !== 0 || last.usage?.total_tokens !== final.body.usage.total_tokens)
|
|
394
|
+
return false;
|
|
395
|
+
const finishing = frames.filter((f) => f.choices[0]?.finish_reason);
|
|
396
|
+
if (finishing.length !== 1 || finishing[0].choices[0].finish_reason !== 'stop')
|
|
397
|
+
return false;
|
|
398
|
+
const joined = frames.map((f) => f.choices[0]?.delta?.content ?? '').join('');
|
|
399
|
+
return joined === final.body.choices[0].message.content;
|
|
400
|
+
})),
|
|
401
|
+
done('cohere.compat.stream_no_usage_unasked', 'compat', 'Compatibility API streaming: no usage chunk unless stream_options.include_usage', 'api', 'common', () => withStream('/compatibility/v1/chat/completions', CHAT({ stream: true }), (events) => events.length > 2 && events.filter((e) => !e.done).every((e) => e.data.usage === undefined && e.data.choices.length === 1))),
|
|
402
|
+
done('cohere.compat.refusals', 'compat', 'Compatibility API: a model Cohere does not serve, a missing model, an empty messages array and an unknown tool_choice are refused; the parameters the docs name unsupported are ignored', 'api', 'common', () => withRoot(async (h) => {
|
|
403
|
+
const P = '/compatibility/v1/chat/completions';
|
|
404
|
+
const wrongModel = await h({ m: 'POST', p: P, b: CHAT({ model: 'gpt-4o' }) });
|
|
405
|
+
const noModel = await h({ m: 'POST', p: P, b: { messages: [{ role: 'user', content: 'x' }] } });
|
|
406
|
+
const empty = await h({ m: 'POST', p: P, b: CHAT({ messages: [] }) });
|
|
407
|
+
const badChoice = await h({ m: 'POST', p: P, b: CHAT({ tool_choice: 'sometimes' }) });
|
|
408
|
+
const unsupported = await h({ m: 'POST', p: P, b: CHAT({ n: 2, logit_bias: { 1: 1 }, store: true, parallel_tool_calls: false }) });
|
|
409
|
+
return wrongModel.status === 404 && bareError(noModel, 400) && bareError(empty, 400) && bareError(badChoice, 400) && ok(unsupported);
|
|
410
|
+
})),
|
|
411
|
+
done('cohere.compat.named_tool_choice', 'compat', 'Compatibility API: tool_choice naming a function calls that function, not the first declared', 'api', 'common', () => withRoot(async (h) => {
|
|
412
|
+
const OTHER = { type: 'function', function: { name: 'get_time', parameters: { type: 'object', properties: { zone: { type: 'string' } } } } };
|
|
413
|
+
const named = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL, OTHER], tool_choice: { type: 'function', function: { name: 'get_time' } } }) });
|
|
414
|
+
const unnamed = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL, OTHER] }) });
|
|
415
|
+
return named.body.choices?.[0]?.message?.tool_calls?.[0]?.function?.name === 'get_time'
|
|
416
|
+
&& unnamed.body.choices?.[0]?.message?.tool_calls?.[0]?.function?.name === 'get_weather';
|
|
417
|
+
})),
|
|
418
|
+
done('cohere.compat.developer_role', 'compat', 'Compatibility API: a `developer` message is a system message; an unknown role is refused', 'api', 'niche', () => withRoot(async (h) => {
|
|
419
|
+
const converted = compatMessages([{ role: 'developer', content: 'be brief' }, { role: 'user', content: 'hi' }]);
|
|
420
|
+
if (typeof converted === 'string' || converted[0]?.role !== 'system' || converted[0]?.content !== 'be brief')
|
|
421
|
+
return false;
|
|
422
|
+
const dev = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ messages: [{ role: 'developer', content: 'be brief' }, { role: 'user', content: 'hi' }] }) });
|
|
423
|
+
const bad = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ messages: [{ role: 'narrator', content: 'x' }] }) });
|
|
424
|
+
return ok(dev) && bareError(bad, 400);
|
|
425
|
+
})),
|
|
426
|
+
todo('cohere.compat.embeddings', 'compat', 'Compatibility API: POST /compatibility/v1/embeddings (input, model, encoding_format) in OpenAI\'s list-of-embeddings envelope', 'api', 'common'),
|
|
427
|
+
todo('cohere.compat.audio_transcriptions', 'compat', 'Compatibility API: POST /compatibility/v1/audio/transcriptions', 'api', 'niche'),
|
|
428
|
+
todo('cohere.compat.reasoning_effort', 'compat', 'Compatibility API: reasoning_effort (none / high) turns Cohere\'s thinking off and on', 'api', 'common'),
|
|
429
|
+
todo('cohere.compat.response_format', 'compat', 'Compatibility API: response_format json_object / json_schema answers JSON content', 'api', 'common'),
|
|
326
430
|
done('cohere.chat.stream_needs_transport', 'chat', 'v2 chat: `stream:true` with no streaming transport is refused, not silently unary', 'api', 'niche', () => withRoot(async (h) => {
|
|
327
431
|
const r = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ stream: true }) });
|
|
328
432
|
return bareError(r, 400) && message(r).includes('streaming transport');
|
|
@@ -1819,7 +1923,7 @@ export const COHERE_CAPABILITIES = [
|
|
|
1819
1923
|
// ══ CONFORMANCE ══════════════════════════════════════════════════════════════════════
|
|
1820
1924
|
done('cohere.conformance.probes', 'conformance', 'Conformance: every claimed endpoint is probed for a live OUTCOME, not just a dispatch', 'api', 'core', async () => verifyBoundary('cohere.conformance', async () => {
|
|
1821
1925
|
const report = await checkCohereConformance();
|
|
1822
|
-
return report.ok && report.claimed ===
|
|
1926
|
+
return report.ok && report.claimed === 28 && report.checksRun >= 60 && report.violations.length === 0;
|
|
1823
1927
|
})),
|
|
1824
1928
|
done('cohere.conformance.scenario_engine_is_the_kernel', 'conformance', 'Scenario scripting runs on THE kernel engine with a pack adapter (never a bespoke one)', 'api', 'common', () => verifyBoundary('cohere.scenario_kernel', async () => {
|
|
1825
1929
|
const engine = createCohereScenarioEngine();
|
|
@@ -39,6 +39,7 @@ export const COHERE_TWIN_SNAPSHOT = [
|
|
|
39
39
|
'POST /v2/embed',
|
|
40
40
|
'POST /v2/rerank',
|
|
41
41
|
'POST /v1/chat',
|
|
42
|
+
'POST /compatibility/v1/chat/completions',
|
|
42
43
|
'POST /v1/embed',
|
|
43
44
|
'POST /v1/rerank',
|
|
44
45
|
'POST /v1/classify',
|
|
@@ -87,6 +88,17 @@ const PROBES = [
|
|
|
87
88
|
&& b.usage?.billed_units?.input_tokens > 0 && b.usage?.tokens?.input_tokens === b.usage.billed_units.input_tokens
|
|
88
89
|
&& b.usage?.tokens?.output_tokens > 0,
|
|
89
90
|
},
|
|
91
|
+
{
|
|
92
|
+
claim: 'POST /compatibility/v1/chat/completions',
|
|
93
|
+
request: () => ({ method: 'POST', path: '/compatibility/v1/chat/completions', body: CHAT }),
|
|
94
|
+
status: [200],
|
|
95
|
+
// OpenAI's envelope on the Compatibility API — the very keys the v2 probe above asserts ABSENT.
|
|
96
|
+
expect: (b) => isObj(b) && b.object === 'chat.completion' && typeof b.created === 'number'
|
|
97
|
+
&& b.choices?.[0]?.message?.role === 'assistant'
|
|
98
|
+
&& String(b.choices[0].message.content).includes('[twin-stub:command-a-03-2025]')
|
|
99
|
+
&& b.choices[0].finish_reason === 'stop'
|
|
100
|
+
&& b.usage?.total_tokens === b.usage?.prompt_tokens + b.usage?.completion_tokens && b.usage.prompt_tokens > 0,
|
|
101
|
+
},
|
|
90
102
|
{
|
|
91
103
|
claim: 'POST /v1/chat',
|
|
92
104
|
request: () => ({ method: 'POST', path: '/v1/chat', body: { message: 'v1 probe', model: 'command-r-08-2024' } }),
|
|
@@ -123,7 +123,9 @@ export function createCohereTwinFetch(options) {
|
|
|
123
123
|
// before any body and returns a genuine 4xx here.
|
|
124
124
|
const isV2Stream = request.method.toUpperCase() === 'POST' && cleanPath === '/v2/chat' && wantsStream(body);
|
|
125
125
|
const isV1Stream = request.method.toUpperCase() === 'POST' && cleanPath === '/v1/chat' && wantsStream(body);
|
|
126
|
-
|
|
126
|
+
// The Compatibility API streams OpenAI's `chat.completion.chunk` frames over SSE, `[DONE]`-terminated.
|
|
127
|
+
const isCompatStream = request.method.toUpperCase() === 'POST' && cleanPath === '/compatibility/v1/chat/completions' && wantsStream(body);
|
|
128
|
+
if (!readOnly && (isV2Stream || isV1Stream || isCompatStream)) {
|
|
127
129
|
const events = [];
|
|
128
130
|
const { status, body: out, headers: extra } = await handleCohereTwinRequest({
|
|
129
131
|
method: request.method, path, body, readOnly, headers, occurredAt: worldNow(),
|
|
@@ -135,7 +137,7 @@ export function createCohereTwinFetch(options) {
|
|
|
135
137
|
return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', ...(extra ?? {}) } });
|
|
136
138
|
}
|
|
137
139
|
const enc = new TextEncoder();
|
|
138
|
-
const frame = isV2Stream ? encodeSse : encodeNdjson;
|
|
140
|
+
const frame = isV2Stream || isCompatStream ? encodeSse : encodeNdjson;
|
|
139
141
|
const stream = new ReadableStream({
|
|
140
142
|
start(controller) {
|
|
141
143
|
for (const e of events) {
|
|
@@ -149,7 +151,7 @@ export function createCohereTwinFetch(options) {
|
|
|
149
151
|
return new Response(stream, {
|
|
150
152
|
status,
|
|
151
153
|
headers: {
|
|
152
|
-
'content-type': isV2Stream ? 'text/event-stream; charset=utf-8' : 'application/stream+json; charset=utf-8',
|
|
154
|
+
'content-type': isV2Stream || isCompatStream ? 'text/event-stream; charset=utf-8' : 'application/stream+json; charset=utf-8',
|
|
153
155
|
'cache-control': 'no-cache',
|
|
154
156
|
...(extra ?? {}),
|
|
155
157
|
},
|
|
@@ -30,6 +30,8 @@ export type ChatV2Args = {
|
|
|
30
30
|
messages: CohereMessageV2[];
|
|
31
31
|
tools?: unknown;
|
|
32
32
|
toolChoice?: 'REQUIRED' | 'NONE';
|
|
33
|
+
/** The one tool a caller names (the Compatibility API's `tool_choice: {function:{name}}`; v2 cannot name one). */
|
|
34
|
+
forcedTool?: string;
|
|
33
35
|
maxTokens?: number;
|
|
34
36
|
stream: boolean;
|
|
35
37
|
thinking: boolean;
|
|
@@ -54,6 +56,9 @@ export declare function buildChatV2(args: ChatV2Args, outcome?: ScenarioOutcome)
|
|
|
54
56
|
* detail a hand-written stream gets wrong.
|
|
55
57
|
*/
|
|
56
58
|
export declare function streamChatV2(args: ChatV2Args, sink: SseSink, outcome?: ScenarioOutcome): CohereChatV2Response;
|
|
59
|
+
/** OpenAI's chat messages as the v2 turn reads them (OpenAI's text, image_url, tool and assistant tool_calls shapes are
|
|
60
|
+
* v2's own), or the refusal of the first message whose role neither wire has. */
|
|
61
|
+
export declare function compatMessages(messages: unknown[]): CohereMessageV2[] | string;
|
|
57
62
|
/**
|
|
58
63
|
* Every method/path pair the dispatch below branches on. HAND-AUTHORED (the honest limit): a
|
|
59
64
|
* branch added to the handler and to neither this list nor the conformance snapshot is invisible
|
package/dist/src/cohere-twin.js
CHANGED
|
@@ -322,10 +322,14 @@ function assistantTurn(args, scripted, missTeach = '') {
|
|
|
322
322
|
const names = toolNames(args.tools);
|
|
323
323
|
// `tool_choice: 'NONE'` forbids a tool call even when tools are declared — the vendor honours it
|
|
324
324
|
// and so must the twin, or a caller testing the NONE path gets a tool call it explicitly banned.
|
|
325
|
-
|
|
325
|
+
// A model answers a tool's result in words unless the caller insists on another call (REQUIRED):
|
|
326
|
+
// a stub that called a tool on every turn would never let an application's tool loop end (an
|
|
327
|
+
// agent runs until its recursion limit), where Cohere's model answers from what the tool returned.
|
|
328
|
+
const answeringTool = args.messages.at(-1)?.role === 'tool';
|
|
329
|
+
const wantsTool = names.length > 0 && args.toolChoice !== 'NONE' && (!answeringTool || args.toolChoice === 'REQUIRED');
|
|
326
330
|
const seed = JSON.stringify(args.messages);
|
|
327
331
|
if (wantsTool) {
|
|
328
|
-
const call = stubToolCall(args.tools, seed, 0);
|
|
332
|
+
const call = stubToolCall(args.tools, seed, 0, args.forcedTool);
|
|
329
333
|
return { content: [], toolCalls: call ? [call] : [], toolPlan: stubToolPlan(names), finish: 'TOOL_CALL' };
|
|
330
334
|
}
|
|
331
335
|
const content = [];
|
|
@@ -421,6 +425,157 @@ function chunkText(text) {
|
|
|
421
425
|
return out;
|
|
422
426
|
}
|
|
423
427
|
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
428
|
+
// THE COMPATIBILITY API — OpenAI's wire, Cohere's models (docs.cohere.com/docs/compatibility-api)
|
|
429
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
430
|
+
//
|
|
431
|
+
// Cohere serves OpenAI's Chat Completions wire at `https://api.cohere.ai/compatibility/v1` so an
|
|
432
|
+
// application built on the OpenAI SDK (or one that speaks OpenAI's wire to every provider, as
|
|
433
|
+
// LibreChat's custom endpoints do) reaches Cohere's models by changing base URL, key and model.
|
|
434
|
+
// The docs' code examples all address that base; the parameters they list for chat completions
|
|
435
|
+
// are exactly `model, messages, stream, reasoning_effort, response_format, tools, temperature,
|
|
436
|
+
// max_tokens, stop, seed, top_p, frequency_penalty, presence_penalty`, and they name `store`,
|
|
437
|
+
// `metadata`, `logit_bias`, `top_logprobs`, `n`, `modalities`, `prediction`, `audio`, `service_tier` and
|
|
438
|
+
// `parallel_tool_calls` unsupported.
|
|
439
|
+
//
|
|
440
|
+
// The turn itself is the v2 chat's (`assistantTurn`, scenario handlers included): the same stub,
|
|
441
|
+
// the same tool decision, and the same handler document script both wires. Only the envelope is
|
|
442
|
+
// OpenAI's: `object: 'chat.completion'`, one `choices[0]` with an assistant `message`, the
|
|
443
|
+
// lower-case finish reasons (`stop`, `tool_calls`, `length`), and a flat `usage`
|
|
444
|
+
// (`prompt_tokens`, `completion_tokens`, `total_tokens`); streaming is `chat.completion.chunk`
|
|
445
|
+
// frames over SSE ending `data: [DONE]`, with a usage-only chunk when `stream_options.include_usage`
|
|
446
|
+
// asks for it, as OpenAI's wire has it.
|
|
447
|
+
//
|
|
448
|
+
// Where the documentation stops and the twin decides (not probed against the live API, which
|
|
449
|
+
// needs a key): the parameters the docs name unsupported are ignored, since the docs do not say
|
|
450
|
+
// whether Cohere refuses or ignores them and a refusal would break a client that sends one; the
|
|
451
|
+
// refusals the twin does make (model, messages, role, tool_choice) keep Cohere's bare `{message}`
|
|
452
|
+
// envelope (the host's own); a named `tool_choice` function calls that function; a `developer`
|
|
453
|
+
// message is read as a system message, as OpenAI defines it; `ERROR` and `TIMEOUT` finish as
|
|
454
|
+
// `stop` (OpenAI's wire has no error finish reason); and a tool turn carries no `tool_plan` (the
|
|
455
|
+
// docs' examples show OpenAI's fields only). `reasoning_effort` is not modelled (a `todo`).
|
|
456
|
+
const COMPAT_ROLES = new Set(['system', 'developer', 'user', 'assistant', 'tool']);
|
|
457
|
+
const COMPAT_FINISH = {
|
|
458
|
+
COMPLETE: 'stop', STOP_SEQUENCE: 'stop', MAX_TOKENS: 'length', TOOL_CALL: 'tool_calls', ERROR: 'stop', TIMEOUT: 'stop',
|
|
459
|
+
};
|
|
460
|
+
/** OpenAI's chat messages as the v2 turn reads them (OpenAI's text, image_url, tool and assistant tool_calls shapes are
|
|
461
|
+
* v2's own), or the refusal of the first message whose role neither wire has. */
|
|
462
|
+
export function compatMessages(messages) {
|
|
463
|
+
const out = [];
|
|
464
|
+
for (const [i, raw] of messages.entries()) {
|
|
465
|
+
const m = raw;
|
|
466
|
+
const role = m?.role;
|
|
467
|
+
if (typeof role !== 'string' || !COMPAT_ROLES.has(role))
|
|
468
|
+
return `messages[${i}].role must be one of ${[...COMPAT_ROLES].join(', ')}`;
|
|
469
|
+
const msg = { role: role === 'developer' ? 'system' : role };
|
|
470
|
+
if (m.content !== undefined && m.content !== null)
|
|
471
|
+
msg.content = m.content;
|
|
472
|
+
if (Array.isArray(m.tool_calls) && m.tool_calls.length > 0)
|
|
473
|
+
msg.tool_calls = m.tool_calls;
|
|
474
|
+
if (typeof m.tool_call_id === 'string')
|
|
475
|
+
msg.tool_call_id = m.tool_call_id;
|
|
476
|
+
out.push(msg);
|
|
477
|
+
}
|
|
478
|
+
return out;
|
|
479
|
+
}
|
|
480
|
+
function validateCompatChat(params) {
|
|
481
|
+
const model = params.model;
|
|
482
|
+
if (typeof model !== 'string' || model === '')
|
|
483
|
+
return { error: invalidRequest('model is required') };
|
|
484
|
+
const messages = params.messages;
|
|
485
|
+
if (!Array.isArray(messages) || messages.length === 0)
|
|
486
|
+
return { error: invalidRequest('messages must be a non-empty array') };
|
|
487
|
+
const converted = compatMessages(messages);
|
|
488
|
+
if (typeof converted === 'string')
|
|
489
|
+
return { error: invalidRequest(converted) };
|
|
490
|
+
const v2Messages = converted;
|
|
491
|
+
const choice = params.tool_choice;
|
|
492
|
+
let toolChoice;
|
|
493
|
+
let forcedTool;
|
|
494
|
+
if (choice === 'none')
|
|
495
|
+
toolChoice = 'NONE';
|
|
496
|
+
else if (choice === 'required')
|
|
497
|
+
toolChoice = 'REQUIRED';
|
|
498
|
+
else if (choice !== null && typeof choice === 'object') {
|
|
499
|
+
toolChoice = 'REQUIRED';
|
|
500
|
+
const named = choice.function?.name;
|
|
501
|
+
if (typeof named === 'string')
|
|
502
|
+
forcedTool = named;
|
|
503
|
+
}
|
|
504
|
+
else if (choice !== undefined && choice !== 'auto')
|
|
505
|
+
return { error: invalidRequest('tool_choice must be none, auto, required or a named function') };
|
|
506
|
+
if (!modelServes(model, 'chat'))
|
|
507
|
+
return { error: modelNotFound(model) };
|
|
508
|
+
const maxTokens = typeof params.max_tokens === 'number' ? params.max_tokens : typeof params.max_completion_tokens === 'number' ? params.max_completion_tokens : undefined;
|
|
509
|
+
return {
|
|
510
|
+
args: {
|
|
511
|
+
v2: {
|
|
512
|
+
model,
|
|
513
|
+
messages: v2Messages,
|
|
514
|
+
...(params.tools !== undefined ? { tools: params.tools } : {}),
|
|
515
|
+
...(toolChoice !== undefined ? { toolChoice } : {}),
|
|
516
|
+
...(forcedTool !== undefined ? { forcedTool } : {}),
|
|
517
|
+
...(maxTokens !== undefined ? { maxTokens } : {}),
|
|
518
|
+
stream: params.stream === true,
|
|
519
|
+
thinking: false,
|
|
520
|
+
},
|
|
521
|
+
includeUsage: params.stream_options?.include_usage === true,
|
|
522
|
+
},
|
|
523
|
+
};
|
|
524
|
+
}
|
|
525
|
+
/** The OpenAI-shaped completion of one v2 turn. */
|
|
526
|
+
function compatCompletion(args, outcome, occurredAt) {
|
|
527
|
+
const full = buildChatV2(args, outcome);
|
|
528
|
+
const text = (full.message.content ?? []).filter((c) => c.type === 'text').map((c) => c.text).join('');
|
|
529
|
+
const toolCalls = full.message.tool_calls ?? [];
|
|
530
|
+
const usage = full.usage.tokens ?? { input_tokens: 0, output_tokens: 0 };
|
|
531
|
+
return {
|
|
532
|
+
id: full.id,
|
|
533
|
+
object: 'chat.completion',
|
|
534
|
+
created: Math.floor(Date.parse(nowIso(occurredAt)) / 1000),
|
|
535
|
+
model: args.model,
|
|
536
|
+
choices: [{
|
|
537
|
+
index: 0,
|
|
538
|
+
message: { role: 'assistant', content: toolCalls.length > 0 && text === '' ? null : text, ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}) },
|
|
539
|
+
finish_reason: COMPAT_FINISH[full.finish_reason],
|
|
540
|
+
}],
|
|
541
|
+
usage: { prompt_tokens: usage.input_tokens, completion_tokens: usage.output_tokens, total_tokens: usage.input_tokens + usage.output_tokens },
|
|
542
|
+
};
|
|
543
|
+
}
|
|
544
|
+
/** Stream the same completion as `chat.completion.chunk` frames and return the unary body. */
|
|
545
|
+
function streamCompat(completion, includeUsage, sink) {
|
|
546
|
+
const { id, created, model } = completion;
|
|
547
|
+
const choice = completion.choices[0];
|
|
548
|
+
const chunk = (delta, finish = null) => sink({ data: { id, object: 'chat.completion.chunk', created, model, choices: [{ index: 0, delta, finish_reason: finish }] } });
|
|
549
|
+
chunk({ role: 'assistant', content: '' });
|
|
550
|
+
for (const piece of chunkText(choice.message.content ?? ''))
|
|
551
|
+
chunk({ content: piece });
|
|
552
|
+
for (const [index, call] of (choice.message.tool_calls ?? []).entries()) {
|
|
553
|
+
chunk({ tool_calls: [{ index, id: call.id, type: 'function', function: { name: call.function.name, arguments: '' } }] });
|
|
554
|
+
for (const piece of chunkText(call.function.arguments))
|
|
555
|
+
chunk({ tool_calls: [{ index, function: { arguments: piece } }] });
|
|
556
|
+
}
|
|
557
|
+
chunk({}, choice.finish_reason);
|
|
558
|
+
if (includeUsage)
|
|
559
|
+
sink({ data: { id, object: 'chat.completion.chunk', created, model, choices: [], usage: completion.usage } });
|
|
560
|
+
sink({ done: true });
|
|
561
|
+
return completion;
|
|
562
|
+
}
|
|
563
|
+
function handleCompatChat(params, req) {
|
|
564
|
+
const validated = validateCompatChat(params);
|
|
565
|
+
if ('error' in validated)
|
|
566
|
+
return validated.error;
|
|
567
|
+
const { v2, includeUsage } = validated.args;
|
|
568
|
+
const outcome = req.scenarioEngine ? scenarioDecision(v2, req.scenarioEngine) : {};
|
|
569
|
+
if (outcome.error)
|
|
570
|
+
return outcome.error;
|
|
571
|
+
const completion = compatCompletion(v2, outcome, req.occurredAt);
|
|
572
|
+
if (!v2.stream)
|
|
573
|
+
return { status: 200, body: completion };
|
|
574
|
+
if (!req.sseSink)
|
|
575
|
+
return invalidRequest('stream:true requires a streaming transport');
|
|
576
|
+
return { status: 200, body: streamCompat(completion, includeUsage, req.sseSink) };
|
|
577
|
+
}
|
|
578
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
424
579
|
// CHAT v1 — a DIFFERENT protocol, not a versioned alias
|
|
425
580
|
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
426
581
|
function handleChatV1(params, req) {
|
|
@@ -1004,6 +1159,7 @@ export const COHERE_ROUTER_SURFACE = [
|
|
|
1004
1159
|
{ method: 'POST', path: '/v2/embed' },
|
|
1005
1160
|
{ method: 'POST', path: '/v2/rerank' },
|
|
1006
1161
|
{ method: 'POST', path: '/v1/chat' },
|
|
1162
|
+
{ method: 'POST', path: '/compatibility/v1/chat/completions' },
|
|
1007
1163
|
{ method: 'POST', path: '/v1/embed' },
|
|
1008
1164
|
{ method: 'POST', path: '/v1/rerank' },
|
|
1009
1165
|
{ method: 'POST', path: '/v1/classify' },
|
|
@@ -1064,6 +1220,8 @@ export async function handleCohereTwinRequest(req) {
|
|
|
1064
1220
|
}
|
|
1065
1221
|
if (method === 'POST' && path === '/v1/chat')
|
|
1066
1222
|
return handleChatV1(params, req);
|
|
1223
|
+
if (method === 'POST' && path === '/compatibility/v1/chat/completions')
|
|
1224
|
+
return handleCompatChat(params, req);
|
|
1067
1225
|
// ── embed / rerank / classify ─────────────────────────────────────────────────────────
|
|
1068
1226
|
if (method === 'POST' && path === '/v2/embed')
|
|
1069
1227
|
return handleEmbedV2(params);
|
package/dist/src/index.js
CHANGED
|
@@ -71,7 +71,13 @@ export const pack = {
|
|
|
71
71
|
// `/v2`. The Bedrock/SageMaker clients the SDK also ships (`AwsClient`, `BedrockClient`,
|
|
72
72
|
// `SagemakerClient`) address AWS hosts, which are a different vendor's plane and deliberately
|
|
73
73
|
// NOT claimed here.
|
|
74
|
-
|
|
74
|
+
//
|
|
75
|
+
// `api.cohere.ai` is Cohere's earlier host, still answering, and the one its Compatibility API
|
|
76
|
+
// documents (docs.cohere.com/docs/compatibility-api: every example's base URL is
|
|
77
|
+
// `https://api.cohere.ai/compatibility/v1`); applications that speak OpenAI's wire to Cohere
|
|
78
|
+
// address it there (LibreChat's own Cohere constant is `https://api.cohere.ai/v1`). The same
|
|
79
|
+
// twin answers both hosts, as the one API behind them.
|
|
80
|
+
hosts: [{ host: 'api.cohere.com' }, { host: 'api.cohere.ai' }],
|
|
75
81
|
// WORLD WIRING (§7 point 12) — the ruling is NONE, and it is grounded, not skipped.
|
|
76
82
|
// NEITHER SDK reads a base-URL environment variable. cohere-ai takes the override as a
|
|
77
83
|
// CONSTRUCTOR option (`baseUrl` / `environment`, resolved through `core.Supplier.get`) and the
|
|
@@ -80,5 +86,5 @@ export const pack = {
|
|
|
80
86
|
// Inventing a `COHERE_BASE_URL` would make `covers` report the world covered — any app-read env
|
|
81
87
|
// counts — while the app, reading no such var, still talked to the real vendor. Interception is
|
|
82
88
|
// therefore host-based, via the `hosts` entry above, which is the truth.
|
|
83
|
-
endpointEnvNone: 'neither official SDK reads a base-URL env var: cohere-ai@8.1.0 takes the override as the `baseUrl`/`environment` CONSTRUCTOR option and reads only CO_API_KEY from the environment (auth/BearerAuthProvider.js), and @ai-sdk/cohere@4.0.35 takes `options.baseURL` and reads only COHERE_API_KEY. Interception is host-based on api.cohere.com; inventing a COHERE_BASE_URL nothing reads would make `covers` claim a world it does not cover.',
|
|
89
|
+
endpointEnvNone: 'neither official SDK reads a base-URL env var: cohere-ai@8.1.0 takes the override as the `baseUrl`/`environment` CONSTRUCTOR option and reads only CO_API_KEY from the environment (auth/BearerAuthProvider.js), and @ai-sdk/cohere@4.0.35 takes `options.baseURL` and reads only COHERE_API_KEY. Interception is host-based on api.cohere.com and api.cohere.ai; inventing a COHERE_BASE_URL nothing reads would make `covers` claim a world it does not cover.',
|
|
84
90
|
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@volter/twin-cohere",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.1",
|
|
4
4
|
"description": "Local Cohere twin — a faithful, stateful local Cohere API your real cohere-ai / @ai-sdk/cohere client talks to unmodified. Both protocol versions are modeled (v1 chat/embed/rerank/classify/tokenize + v2 chat/embed/rerank), with deterministic stubs for model output and vendor-faithful envelopes, streaming, errors and refusals. Built on @volter/world-core.",
|
|
5
5
|
"author": "Volter (https://github.com/volter-ai)",
|
|
6
6
|
"license": "Apache-2.0",
|
|
@@ -37,15 +37,16 @@
|
|
|
37
37
|
"postpack": "node ../../../scripts/publish/prepare-publish.mjs postpack"
|
|
38
38
|
},
|
|
39
39
|
"peerDependencies": {
|
|
40
|
-
"@volter/world-core": "2.0.
|
|
40
|
+
"@volter/world-core": "2.0.1"
|
|
41
41
|
},
|
|
42
42
|
"devDependencies": {
|
|
43
43
|
"@ai-sdk/cohere": "^4.0.35",
|
|
44
44
|
"@types/bun": "^1.2.20",
|
|
45
45
|
"@types/node": "^24.0.0",
|
|
46
|
-
"@volter/world-core": "2.0.
|
|
46
|
+
"@volter/world-core": "2.0.1",
|
|
47
47
|
"@volter/world-tooling": "0.1.0",
|
|
48
48
|
"cohere-ai": "^8.1.0",
|
|
49
|
+
"openai": "^6.45.0",
|
|
49
50
|
"typescript": "^5.9.0"
|
|
50
51
|
},
|
|
51
52
|
"engines": {
|
|
@@ -64,7 +64,7 @@ import {
|
|
|
64
64
|
} from './cohere-connector.ts';
|
|
65
65
|
import { COHERE_MODELS, COHERE_ENDPOINTS } from './cohere-models.ts';
|
|
66
66
|
import { createCohereScenarioEngine } from './cohere-scenario.ts';
|
|
67
|
-
import { handleCohereTwinRequest, type CohereResponseEnvelope } from './cohere-twin.ts';
|
|
67
|
+
import { compatMessages, handleCohereTwinRequest, type CohereResponseEnvelope } from './cohere-twin.ts';
|
|
68
68
|
import type { SseEvent } from './cohere-types.ts';
|
|
69
69
|
|
|
70
70
|
// ── API verify: drive REAL requests against a fresh temp root, then assert status/shape ──
|
|
@@ -196,6 +196,7 @@ export const COHERE_AREAS = [
|
|
|
196
196
|
'batches',
|
|
197
197
|
'chat',
|
|
198
198
|
'chat_v1',
|
|
199
|
+
'compat',
|
|
199
200
|
'classify',
|
|
200
201
|
'conformance',
|
|
201
202
|
'connector',
|
|
@@ -284,6 +285,22 @@ export const COHERE_CAPABILITIES: CapabilitySpec[] = [
|
|
|
284
285
|
&& typeof (none.body as Body).message?.content?.[0]?.text === 'string';
|
|
285
286
|
}),
|
|
286
287
|
),
|
|
288
|
+
done('cohere.chat.tool_result_answered', 'chat', 'v2 chat: a turn answering a tool result is words, not another tool call, unless tool_choice is REQUIRED', 'api', 'core', () =>
|
|
289
|
+
withRoot(async (h) => {
|
|
290
|
+
const first = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ tools: [TOOL] }) });
|
|
291
|
+
const call = (first.body as Body).message?.tool_calls?.[0];
|
|
292
|
+
if (!call) return false;
|
|
293
|
+
const loop = [
|
|
294
|
+
{ role: 'user', content: 'weather?' },
|
|
295
|
+
{ role: 'assistant', tool_calls: [call], tool_plan: (first.body as Body).message.tool_plan },
|
|
296
|
+
{ role: 'tool', tool_call_id: call.id, content: [{ type: 'document', document: { data: '{"temp":20}' } }] },
|
|
297
|
+
];
|
|
298
|
+
const answered = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ tools: [TOOL], messages: loop }) });
|
|
299
|
+
const insisted = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ tools: [TOOL], messages: loop, tool_choice: 'REQUIRED' }) });
|
|
300
|
+
return (answered.body as Body).finish_reason === 'COMPLETE' && typeof (answered.body as Body).message?.content?.[0]?.text === 'string'
|
|
301
|
+
&& (insisted.body as Body).finish_reason === 'TOOL_CALL';
|
|
302
|
+
}),
|
|
303
|
+
),
|
|
287
304
|
done('cohere.chat.thinking', 'chat', 'v2 chat: `thinking: {type:enabled}` prepends a `thinking` content item', 'api', 'common', () =>
|
|
288
305
|
withRoot(async (h) => {
|
|
289
306
|
const off = await h({ m: 'POST', p: '/v2/chat', b: CHAT() });
|
|
@@ -338,6 +355,95 @@ export const COHERE_CAPABILITIES: CapabilitySpec[] = [
|
|
|
338
355
|
return end.delta.finish_reason === 'TOOL_CALL';
|
|
339
356
|
}),
|
|
340
357
|
),
|
|
358
|
+
// ══ THE COMPATIBILITY API — OpenAI's wire at api.cohere.ai/compatibility/v1 ════════════
|
|
359
|
+
done('cohere.compat.chat_completions', 'compat', 'Compatibility API: POST /compatibility/v1/chat/completions answers OpenAI\'s chat.completion envelope for a Cohere model', 'api', 'core', () =>
|
|
360
|
+
withRoot(async (h) => {
|
|
361
|
+
const r = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT() });
|
|
362
|
+
if (!ok(r)) return false;
|
|
363
|
+
const b = r.body as Body;
|
|
364
|
+
// OpenAI's keys PRESENT, Cohere v2's ABSENT: the same turn in the other wire, not the v2 body relabelled.
|
|
365
|
+
if (b.object !== 'chat.completion' || typeof b.created !== 'number' || b.model !== 'command-a-03-2025') return false;
|
|
366
|
+
if (b.message !== undefined || b.finish_reason !== undefined) return false;
|
|
367
|
+
const c = b.choices?.[0];
|
|
368
|
+
if (b.choices.length !== 1 || c.index !== 0 || c.message?.role !== 'assistant' || c.finish_reason !== 'stop') return false;
|
|
369
|
+
if (typeof c.message.content !== 'string' || !c.message.content.includes('[twin-stub:command-a-03-2025]') || !c.message.content.includes('hello twin')) return false;
|
|
370
|
+
// The flat usage, whose counts move with the input (a constant would pass the shape check).
|
|
371
|
+
const long = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ messages: [{ role: 'user', content: 'x'.repeat(400) }] }) });
|
|
372
|
+
const u = b.usage; const lu = (long.body as Body).usage;
|
|
373
|
+
return u.total_tokens === u.prompt_tokens + u.completion_tokens && lu.prompt_tokens > u.prompt_tokens * 5;
|
|
374
|
+
}),
|
|
375
|
+
),
|
|
376
|
+
done('cohere.compat.tool_calls', 'compat', 'Compatibility API: tools → OpenAI `tool_calls` with finish_reason tool_calls; tool_choice none forbids it; the tool result is answered in words', 'api', 'core', () =>
|
|
377
|
+
withRoot(async (h) => {
|
|
378
|
+
const call = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL] }) });
|
|
379
|
+
const none = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL], tool_choice: 'none' }) });
|
|
380
|
+
const c = (call.body as Body).choices?.[0];
|
|
381
|
+
if (c?.finish_reason !== 'tool_calls' || c.message?.content !== null) return false;
|
|
382
|
+
const tc = c.message.tool_calls?.[0];
|
|
383
|
+
if (tc?.type !== 'function' || tc.function?.name !== 'get_weather' || typeof tc.function.arguments !== 'string') return false;
|
|
384
|
+
if ((none.body as Body).choices?.[0]?.finish_reason !== 'stop' || (none.body as Body).choices[0].message.tool_calls !== undefined) return false;
|
|
385
|
+
// OpenAI's tool loop: the assistant's tool call, then the `tool` result — the next turn is words, not another call.
|
|
386
|
+
const answered = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL], messages: [
|
|
387
|
+
{ role: 'user', content: 'weather?' },
|
|
388
|
+
{ role: 'assistant', content: null, tool_calls: [tc] },
|
|
389
|
+
{ role: 'tool', tool_call_id: tc.id, content: '{"temp":20}' },
|
|
390
|
+
] }) });
|
|
391
|
+
return ok(answered) && typeof (answered.body as Body).choices?.[0]?.message?.content === 'string';
|
|
392
|
+
}),
|
|
393
|
+
),
|
|
394
|
+
done('cohere.compat.stream', 'compat', 'Compatibility API streaming: chat.completion.chunk frames reassemble the unary answer, end with the finish reason, a usage chunk when stream_options asks, then [DONE]', 'api', 'core', () =>
|
|
395
|
+
withStream('/compatibility/v1/chat/completions', CHAT({ stream: true, stream_options: { include_usage: true } }), (events, final) => {
|
|
396
|
+
const frames = events.filter((e) => !e.done).map((e) => e.data as Body);
|
|
397
|
+
if (events[events.length - 1]?.done !== true) return false;
|
|
398
|
+
if (!frames.every((f) => f.object === 'chat.completion.chunk' && f.id === (final.body as Body).id)) return false;
|
|
399
|
+
if (frames[0]!.choices[0].delta.role !== 'assistant') return false;
|
|
400
|
+
const last = frames[frames.length - 1]!;
|
|
401
|
+
// the usage-only chunk: no choices, the unary usage
|
|
402
|
+
if (last.choices.length !== 0 || last.usage?.total_tokens !== (final.body as Body).usage.total_tokens) return false;
|
|
403
|
+
const finishing = frames.filter((f) => f.choices[0]?.finish_reason);
|
|
404
|
+
if (finishing.length !== 1 || finishing[0]!.choices[0].finish_reason !== 'stop') return false;
|
|
405
|
+
const joined = frames.map((f) => f.choices[0]?.delta?.content ?? '').join('');
|
|
406
|
+
return joined === (final.body as Body).choices[0].message.content;
|
|
407
|
+
}),
|
|
408
|
+
),
|
|
409
|
+
done('cohere.compat.stream_no_usage_unasked', 'compat', 'Compatibility API streaming: no usage chunk unless stream_options.include_usage', 'api', 'common', () =>
|
|
410
|
+
withStream('/compatibility/v1/chat/completions', CHAT({ stream: true }), (events) =>
|
|
411
|
+
events.length > 2 && events.filter((e) => !e.done).every((e) => (e.data as Body).usage === undefined && (e.data as Body).choices.length === 1)),
|
|
412
|
+
),
|
|
413
|
+
done('cohere.compat.refusals', 'compat', 'Compatibility API: a model Cohere does not serve, a missing model, an empty messages array and an unknown tool_choice are refused; the parameters the docs name unsupported are ignored', 'api', 'common', () =>
|
|
414
|
+
withRoot(async (h) => {
|
|
415
|
+
const P = '/compatibility/v1/chat/completions';
|
|
416
|
+
const wrongModel = await h({ m: 'POST', p: P, b: CHAT({ model: 'gpt-4o' }) });
|
|
417
|
+
const noModel = await h({ m: 'POST', p: P, b: { messages: [{ role: 'user', content: 'x' }] } });
|
|
418
|
+
const empty = await h({ m: 'POST', p: P, b: CHAT({ messages: [] }) });
|
|
419
|
+
const badChoice = await h({ m: 'POST', p: P, b: CHAT({ tool_choice: 'sometimes' }) });
|
|
420
|
+
const unsupported = await h({ m: 'POST', p: P, b: CHAT({ n: 2, logit_bias: { 1: 1 }, store: true, parallel_tool_calls: false }) });
|
|
421
|
+
return wrongModel.status === 404 && bareError(noModel, 400) && bareError(empty, 400) && bareError(badChoice, 400) && ok(unsupported);
|
|
422
|
+
}),
|
|
423
|
+
),
|
|
424
|
+
done('cohere.compat.named_tool_choice', 'compat', 'Compatibility API: tool_choice naming a function calls that function, not the first declared', 'api', 'common', () =>
|
|
425
|
+
withRoot(async (h) => {
|
|
426
|
+
const OTHER = { type: 'function', function: { name: 'get_time', parameters: { type: 'object', properties: { zone: { type: 'string' } } } } };
|
|
427
|
+
const named = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL, OTHER], tool_choice: { type: 'function', function: { name: 'get_time' } } }) });
|
|
428
|
+
const unnamed = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ tools: [TOOL, OTHER] }) });
|
|
429
|
+
return (named.body as Body).choices?.[0]?.message?.tool_calls?.[0]?.function?.name === 'get_time'
|
|
430
|
+
&& (unnamed.body as Body).choices?.[0]?.message?.tool_calls?.[0]?.function?.name === 'get_weather';
|
|
431
|
+
}),
|
|
432
|
+
),
|
|
433
|
+
done('cohere.compat.developer_role', 'compat', 'Compatibility API: a `developer` message is a system message; an unknown role is refused', 'api', 'niche', () =>
|
|
434
|
+
withRoot(async (h) => {
|
|
435
|
+
const converted = compatMessages([{ role: 'developer', content: 'be brief' }, { role: 'user', content: 'hi' }]);
|
|
436
|
+
if (typeof converted === 'string' || converted[0]?.role !== 'system' || converted[0]?.content !== 'be brief') return false;
|
|
437
|
+
const dev = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ messages: [{ role: 'developer', content: 'be brief' }, { role: 'user', content: 'hi' }] }) });
|
|
438
|
+
const bad = await h({ m: 'POST', p: '/compatibility/v1/chat/completions', b: CHAT({ messages: [{ role: 'narrator', content: 'x' }] }) });
|
|
439
|
+
return ok(dev) && bareError(bad, 400);
|
|
440
|
+
}),
|
|
441
|
+
),
|
|
442
|
+
todo('cohere.compat.embeddings', 'compat', 'Compatibility API: POST /compatibility/v1/embeddings (input, model, encoding_format) in OpenAI\'s list-of-embeddings envelope', 'api', 'common'),
|
|
443
|
+
todo('cohere.compat.audio_transcriptions', 'compat', 'Compatibility API: POST /compatibility/v1/audio/transcriptions', 'api', 'niche'),
|
|
444
|
+
todo('cohere.compat.reasoning_effort', 'compat', 'Compatibility API: reasoning_effort (none / high) turns Cohere\'s thinking off and on', 'api', 'common'),
|
|
445
|
+
todo('cohere.compat.response_format', 'compat', 'Compatibility API: response_format json_object / json_schema answers JSON content', 'api', 'common'),
|
|
446
|
+
|
|
341
447
|
done('cohere.chat.stream_needs_transport', 'chat', 'v2 chat: `stream:true` with no streaming transport is refused, not silently unary', 'api', 'niche', () =>
|
|
342
448
|
withRoot(async (h) => {
|
|
343
449
|
const r = await h({ m: 'POST', p: '/v2/chat', b: CHAT({ stream: true }) });
|
|
@@ -1819,7 +1925,7 @@ export const COHERE_CAPABILITIES: CapabilitySpec[] = [
|
|
|
1819
1925
|
done('cohere.conformance.probes', 'conformance', 'Conformance: every claimed endpoint is probed for a live OUTCOME, not just a dispatch', 'api', 'core', async () =>
|
|
1820
1926
|
verifyBoundary('cohere.conformance', async () => {
|
|
1821
1927
|
const report = await checkCohereConformance();
|
|
1822
|
-
return report.ok && report.claimed ===
|
|
1928
|
+
return report.ok && report.claimed === 28 && report.checksRun >= 60 && report.violations.length === 0;
|
|
1823
1929
|
}),
|
|
1824
1930
|
),
|
|
1825
1931
|
done('cohere.conformance.scenario_engine_is_the_kernel', 'conformance', 'Scenario scripting runs on THE kernel engine with a pack adapter (never a bespoke one)', 'api', 'common', () =>
|
|
@@ -50,6 +50,7 @@ export const COHERE_TWIN_SNAPSHOT: ReadonlyArray<string> = [
|
|
|
50
50
|
'POST /v2/embed',
|
|
51
51
|
'POST /v2/rerank',
|
|
52
52
|
'POST /v1/chat',
|
|
53
|
+
'POST /compatibility/v1/chat/completions',
|
|
53
54
|
'POST /v1/embed',
|
|
54
55
|
'POST /v1/rerank',
|
|
55
56
|
'POST /v1/classify',
|
|
@@ -116,6 +117,17 @@ const PROBES: Probe[] = [
|
|
|
116
117
|
&& b.usage?.billed_units?.input_tokens > 0 && b.usage?.tokens?.input_tokens === b.usage.billed_units.input_tokens
|
|
117
118
|
&& b.usage?.tokens?.output_tokens > 0,
|
|
118
119
|
},
|
|
120
|
+
{
|
|
121
|
+
claim: 'POST /compatibility/v1/chat/completions',
|
|
122
|
+
request: () => ({ method: 'POST', path: '/compatibility/v1/chat/completions', body: CHAT }),
|
|
123
|
+
status: [200],
|
|
124
|
+
// OpenAI's envelope on the Compatibility API — the very keys the v2 probe above asserts ABSENT.
|
|
125
|
+
expect: (b) => isObj(b) && b.object === 'chat.completion' && typeof b.created === 'number'
|
|
126
|
+
&& b.choices?.[0]?.message?.role === 'assistant'
|
|
127
|
+
&& String(b.choices[0].message.content).includes('[twin-stub:command-a-03-2025]')
|
|
128
|
+
&& b.choices[0].finish_reason === 'stop'
|
|
129
|
+
&& b.usage?.total_tokens === b.usage?.prompt_tokens + b.usage?.completion_tokens && b.usage.prompt_tokens > 0,
|
|
130
|
+
},
|
|
119
131
|
{
|
|
120
132
|
claim: 'POST /v1/chat',
|
|
121
133
|
request: () => ({ method: 'POST', path: '/v1/chat', body: { message: 'v1 probe', model: 'command-r-08-2024' } }),
|
package/src/cohere-server.ts
CHANGED
|
@@ -134,7 +134,9 @@ export function createCohereTwinFetch(options: CohereTwinFetchOptions): (request
|
|
|
134
134
|
// before any body and returns a genuine 4xx here.
|
|
135
135
|
const isV2Stream = request.method.toUpperCase() === 'POST' && cleanPath === '/v2/chat' && wantsStream(body);
|
|
136
136
|
const isV1Stream = request.method.toUpperCase() === 'POST' && cleanPath === '/v1/chat' && wantsStream(body);
|
|
137
|
-
|
|
137
|
+
// The Compatibility API streams OpenAI's `chat.completion.chunk` frames over SSE, `[DONE]`-terminated.
|
|
138
|
+
const isCompatStream = request.method.toUpperCase() === 'POST' && cleanPath === '/compatibility/v1/chat/completions' && wantsStream(body);
|
|
139
|
+
if (!readOnly && (isV2Stream || isV1Stream || isCompatStream)) {
|
|
138
140
|
const events: SseEvent[] = [];
|
|
139
141
|
const { status, body: out, headers: extra } = await handleCohereTwinRequest({
|
|
140
142
|
method: request.method, path, body, readOnly, headers, occurredAt: worldNow(),
|
|
@@ -146,7 +148,7 @@ export function createCohereTwinFetch(options: CohereTwinFetchOptions): (request
|
|
|
146
148
|
return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', ...(extra ?? {}) } });
|
|
147
149
|
}
|
|
148
150
|
const enc = new TextEncoder();
|
|
149
|
-
const frame = isV2Stream ? encodeSse : encodeNdjson;
|
|
151
|
+
const frame = isV2Stream || isCompatStream ? encodeSse : encodeNdjson;
|
|
150
152
|
const stream = new ReadableStream<Uint8Array>({
|
|
151
153
|
start(controller) {
|
|
152
154
|
for (const e of events) {
|
|
@@ -159,7 +161,7 @@ export function createCohereTwinFetch(options: CohereTwinFetchOptions): (request
|
|
|
159
161
|
return new Response(stream, {
|
|
160
162
|
status,
|
|
161
163
|
headers: {
|
|
162
|
-
'content-type': isV2Stream ? 'text/event-stream; charset=utf-8' : 'application/stream+json; charset=utf-8',
|
|
164
|
+
'content-type': isV2Stream || isCompatStream ? 'text/event-stream; charset=utf-8' : 'application/stream+json; charset=utf-8',
|
|
163
165
|
'cache-control': 'no-cache',
|
|
164
166
|
...(extra ?? {}),
|
|
165
167
|
},
|
package/src/cohere-twin.ts
CHANGED
|
@@ -315,6 +315,8 @@ export type ChatV2Args = {
|
|
|
315
315
|
messages: CohereMessageV2[];
|
|
316
316
|
tools?: unknown;
|
|
317
317
|
toolChoice?: 'REQUIRED' | 'NONE';
|
|
318
|
+
/** The one tool a caller names (the Compatibility API's `tool_choice: {function:{name}}`; v2 cannot name one). */
|
|
319
|
+
forcedTool?: string;
|
|
318
320
|
maxTokens?: number;
|
|
319
321
|
stream: boolean;
|
|
320
322
|
thinking: boolean;
|
|
@@ -408,10 +410,14 @@ function assistantTurn(args: ChatV2Args, scripted?: ScriptedResult, missTeach =
|
|
|
408
410
|
const names = toolNames(args.tools);
|
|
409
411
|
// `tool_choice: 'NONE'` forbids a tool call even when tools are declared — the vendor honours it
|
|
410
412
|
// and so must the twin, or a caller testing the NONE path gets a tool call it explicitly banned.
|
|
411
|
-
|
|
413
|
+
// A model answers a tool's result in words unless the caller insists on another call (REQUIRED):
|
|
414
|
+
// a stub that called a tool on every turn would never let an application's tool loop end (an
|
|
415
|
+
// agent runs until its recursion limit), where Cohere's model answers from what the tool returned.
|
|
416
|
+
const answeringTool = args.messages.at(-1)?.role === 'tool';
|
|
417
|
+
const wantsTool = names.length > 0 && args.toolChoice !== 'NONE' && (!answeringTool || args.toolChoice === 'REQUIRED');
|
|
412
418
|
const seed = JSON.stringify(args.messages);
|
|
413
419
|
if (wantsTool) {
|
|
414
|
-
const call = stubToolCall(args.tools, seed, 0);
|
|
420
|
+
const call = stubToolCall(args.tools, seed, 0, args.forcedTool);
|
|
415
421
|
return { content: [], toolCalls: call ? [call] : [], toolPlan: stubToolPlan(names), finish: 'TOOL_CALL' };
|
|
416
422
|
}
|
|
417
423
|
const content: CohereAssistantContentItem[] = [];
|
|
@@ -508,6 +514,148 @@ function chunkText(text: string): string[] {
|
|
|
508
514
|
return out;
|
|
509
515
|
}
|
|
510
516
|
|
|
517
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
518
|
+
// THE COMPATIBILITY API — OpenAI's wire, Cohere's models (docs.cohere.com/docs/compatibility-api)
|
|
519
|
+
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
520
|
+
//
|
|
521
|
+
// Cohere serves OpenAI's Chat Completions wire at `https://api.cohere.ai/compatibility/v1` so an
|
|
522
|
+
// application built on the OpenAI SDK (or one that speaks OpenAI's wire to every provider, as
|
|
523
|
+
// LibreChat's custom endpoints do) reaches Cohere's models by changing base URL, key and model.
|
|
524
|
+
// The docs' code examples all address that base; the parameters they list for chat completions
|
|
525
|
+
// are exactly `model, messages, stream, reasoning_effort, response_format, tools, temperature,
|
|
526
|
+
// max_tokens, stop, seed, top_p, frequency_penalty, presence_penalty`, and they name `store`,
|
|
527
|
+
// `metadata`, `logit_bias`, `top_logprobs`, `n`, `modalities`, `prediction`, `audio`, `service_tier` and
|
|
528
|
+
// `parallel_tool_calls` unsupported.
|
|
529
|
+
//
|
|
530
|
+
// The turn itself is the v2 chat's (`assistantTurn`, scenario handlers included): the same stub,
|
|
531
|
+
// the same tool decision, and the same handler document script both wires. Only the envelope is
|
|
532
|
+
// OpenAI's: `object: 'chat.completion'`, one `choices[0]` with an assistant `message`, the
|
|
533
|
+
// lower-case finish reasons (`stop`, `tool_calls`, `length`), and a flat `usage`
|
|
534
|
+
// (`prompt_tokens`, `completion_tokens`, `total_tokens`); streaming is `chat.completion.chunk`
|
|
535
|
+
// frames over SSE ending `data: [DONE]`, with a usage-only chunk when `stream_options.include_usage`
|
|
536
|
+
// asks for it, as OpenAI's wire has it.
|
|
537
|
+
//
|
|
538
|
+
// Where the documentation stops and the twin decides (not probed against the live API, which
|
|
539
|
+
// needs a key): the parameters the docs name unsupported are ignored, since the docs do not say
|
|
540
|
+
// whether Cohere refuses or ignores them and a refusal would break a client that sends one; the
|
|
541
|
+
// refusals the twin does make (model, messages, role, tool_choice) keep Cohere's bare `{message}`
|
|
542
|
+
// envelope (the host's own); a named `tool_choice` function calls that function; a `developer`
|
|
543
|
+
// message is read as a system message, as OpenAI defines it; `ERROR` and `TIMEOUT` finish as
|
|
544
|
+
// `stop` (OpenAI's wire has no error finish reason); and a tool turn carries no `tool_plan` (the
|
|
545
|
+
// docs' examples show OpenAI's fields only). `reasoning_effort` is not modelled (a `todo`).
|
|
546
|
+
|
|
547
|
+
const COMPAT_ROLES = new Set(['system', 'developer', 'user', 'assistant', 'tool']);
|
|
548
|
+
const COMPAT_FINISH: Record<CohereFinishReason, string> = {
|
|
549
|
+
COMPLETE: 'stop', STOP_SEQUENCE: 'stop', MAX_TOKENS: 'length', TOOL_CALL: 'tool_calls', ERROR: 'stop', TIMEOUT: 'stop',
|
|
550
|
+
};
|
|
551
|
+
|
|
552
|
+
type CompatArgs = { v2: ChatV2Args; includeUsage: boolean };
|
|
553
|
+
|
|
554
|
+
/** OpenAI's chat messages as the v2 turn reads them (OpenAI's text, image_url, tool and assistant tool_calls shapes are
|
|
555
|
+
* v2's own), or the refusal of the first message whose role neither wire has. */
|
|
556
|
+
export function compatMessages(messages: unknown[]): CohereMessageV2[] | string {
|
|
557
|
+
const out: CohereMessageV2[] = [];
|
|
558
|
+
for (const [i, raw] of messages.entries()) {
|
|
559
|
+
const m = raw as Record<string, unknown> | null;
|
|
560
|
+
const role = m?.role;
|
|
561
|
+
if (typeof role !== 'string' || !COMPAT_ROLES.has(role)) return `messages[${i}].role must be one of ${[...COMPAT_ROLES].join(', ')}`;
|
|
562
|
+
const msg: CohereMessageV2 = { role: role === 'developer' ? 'system' : (role as CohereMessageV2['role']) };
|
|
563
|
+
if (m!.content !== undefined && m!.content !== null) msg.content = m!.content as CohereMessageV2['content'];
|
|
564
|
+
if (Array.isArray(m!.tool_calls) && m!.tool_calls.length > 0) msg.tool_calls = m!.tool_calls as CohereToolCallV2[];
|
|
565
|
+
if (typeof m!.tool_call_id === 'string') msg.tool_call_id = m!.tool_call_id;
|
|
566
|
+
out.push(msg);
|
|
567
|
+
}
|
|
568
|
+
return out;
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
function validateCompatChat(params: Record<string, unknown>): { args: CompatArgs } | { error: CohereResponseEnvelope } {
|
|
572
|
+
const model = params.model;
|
|
573
|
+
if (typeof model !== 'string' || model === '') return { error: invalidRequest('model is required') };
|
|
574
|
+
const messages = params.messages;
|
|
575
|
+
if (!Array.isArray(messages) || messages.length === 0) return { error: invalidRequest('messages must be a non-empty array') };
|
|
576
|
+
const converted = compatMessages(messages);
|
|
577
|
+
if (typeof converted === 'string') return { error: invalidRequest(converted) };
|
|
578
|
+
const v2Messages = converted;
|
|
579
|
+
const choice = params.tool_choice;
|
|
580
|
+
let toolChoice: ChatV2Args['toolChoice'];
|
|
581
|
+
let forcedTool: string | undefined;
|
|
582
|
+
if (choice === 'none') toolChoice = 'NONE';
|
|
583
|
+
else if (choice === 'required') toolChoice = 'REQUIRED';
|
|
584
|
+
else if (choice !== null && typeof choice === 'object') {
|
|
585
|
+
toolChoice = 'REQUIRED';
|
|
586
|
+
const named = (choice as { function?: { name?: unknown } }).function?.name;
|
|
587
|
+
if (typeof named === 'string') forcedTool = named;
|
|
588
|
+
}
|
|
589
|
+
else if (choice !== undefined && choice !== 'auto') return { error: invalidRequest('tool_choice must be none, auto, required or a named function') };
|
|
590
|
+
if (!modelServes(model, 'chat')) return { error: modelNotFound(model) };
|
|
591
|
+
const maxTokens = typeof params.max_tokens === 'number' ? params.max_tokens : typeof params.max_completion_tokens === 'number' ? params.max_completion_tokens : undefined;
|
|
592
|
+
return {
|
|
593
|
+
args: {
|
|
594
|
+
v2: {
|
|
595
|
+
model,
|
|
596
|
+
messages: v2Messages,
|
|
597
|
+
...(params.tools !== undefined ? { tools: params.tools } : {}),
|
|
598
|
+
...(toolChoice !== undefined ? { toolChoice } : {}),
|
|
599
|
+
...(forcedTool !== undefined ? { forcedTool } : {}),
|
|
600
|
+
...(maxTokens !== undefined ? { maxTokens } : {}),
|
|
601
|
+
stream: params.stream === true,
|
|
602
|
+
thinking: false,
|
|
603
|
+
},
|
|
604
|
+
includeUsage: (params.stream_options as { include_usage?: unknown } | undefined)?.include_usage === true,
|
|
605
|
+
},
|
|
606
|
+
};
|
|
607
|
+
}
|
|
608
|
+
|
|
609
|
+
/** The OpenAI-shaped completion of one v2 turn. */
|
|
610
|
+
function compatCompletion(args: ChatV2Args, outcome: ScenarioOutcome, occurredAt?: string): Record<string, unknown> {
|
|
611
|
+
const full = buildChatV2(args, outcome);
|
|
612
|
+
const text = (full.message.content ?? []).filter((c) => c.type === 'text').map((c) => (c as { text: string }).text).join('');
|
|
613
|
+
const toolCalls = full.message.tool_calls ?? [];
|
|
614
|
+
const usage = full.usage.tokens ?? { input_tokens: 0, output_tokens: 0 };
|
|
615
|
+
return {
|
|
616
|
+
id: full.id,
|
|
617
|
+
object: 'chat.completion',
|
|
618
|
+
created: Math.floor(Date.parse(nowIso(occurredAt)) / 1000),
|
|
619
|
+
model: args.model,
|
|
620
|
+
choices: [{
|
|
621
|
+
index: 0,
|
|
622
|
+
message: { role: 'assistant', content: toolCalls.length > 0 && text === '' ? null : text, ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}) },
|
|
623
|
+
finish_reason: COMPAT_FINISH[full.finish_reason],
|
|
624
|
+
}],
|
|
625
|
+
usage: { prompt_tokens: usage.input_tokens, completion_tokens: usage.output_tokens, total_tokens: usage.input_tokens + usage.output_tokens },
|
|
626
|
+
};
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
/** Stream the same completion as `chat.completion.chunk` frames and return the unary body. */
|
|
630
|
+
function streamCompat(completion: Record<string, unknown>, includeUsage: boolean, sink: SseSink): Record<string, unknown> {
|
|
631
|
+
const { id, created, model } = completion as { id: string; created: number; model: string };
|
|
632
|
+
const choice = (completion.choices as Array<{ message: { content: string | null; tool_calls?: CohereToolCallV2[] }; finish_reason: string }>)[0]!;
|
|
633
|
+
const chunk = (delta: Record<string, unknown>, finish: string | null = null): void =>
|
|
634
|
+
sink({ data: { id, object: 'chat.completion.chunk', created, model, choices: [{ index: 0, delta, finish_reason: finish }] } });
|
|
635
|
+
chunk({ role: 'assistant', content: '' });
|
|
636
|
+
for (const piece of chunkText(choice.message.content ?? '')) chunk({ content: piece });
|
|
637
|
+
for (const [index, call] of (choice.message.tool_calls ?? []).entries()) {
|
|
638
|
+
chunk({ tool_calls: [{ index, id: call.id, type: 'function', function: { name: call.function.name, arguments: '' } }] });
|
|
639
|
+
for (const piece of chunkText(call.function.arguments)) chunk({ tool_calls: [{ index, function: { arguments: piece } }] });
|
|
640
|
+
}
|
|
641
|
+
chunk({}, choice.finish_reason);
|
|
642
|
+
if (includeUsage) sink({ data: { id, object: 'chat.completion.chunk', created, model, choices: [], usage: completion.usage as Record<string, unknown> } });
|
|
643
|
+
sink({ done: true });
|
|
644
|
+
return completion;
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
function handleCompatChat(params: Record<string, unknown>, req: CohereRequest): CohereResponseEnvelope {
|
|
648
|
+
const validated = validateCompatChat(params);
|
|
649
|
+
if ('error' in validated) return validated.error;
|
|
650
|
+
const { v2, includeUsage } = validated.args;
|
|
651
|
+
const outcome = req.scenarioEngine ? scenarioDecision(v2, req.scenarioEngine) : {};
|
|
652
|
+
if (outcome.error) return outcome.error;
|
|
653
|
+
const completion = compatCompletion(v2, outcome, req.occurredAt);
|
|
654
|
+
if (!v2.stream) return { status: 200, body: completion };
|
|
655
|
+
if (!req.sseSink) return invalidRequest('stream:true requires a streaming transport');
|
|
656
|
+
return { status: 200, body: streamCompat(completion, includeUsage, req.sseSink) };
|
|
657
|
+
}
|
|
658
|
+
|
|
511
659
|
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
512
660
|
// CHAT v1 — a DIFFERENT protocol, not a versioned alias
|
|
513
661
|
// ════════════════════════════════════════════════════════════════════════════════════════
|
|
@@ -1071,6 +1219,7 @@ export const COHERE_ROUTER_SURFACE: ReadonlyArray<{ method: string; path: string
|
|
|
1071
1219
|
{ method: 'POST', path: '/v2/embed' },
|
|
1072
1220
|
{ method: 'POST', path: '/v2/rerank' },
|
|
1073
1221
|
{ method: 'POST', path: '/v1/chat' },
|
|
1222
|
+
{ method: 'POST', path: '/compatibility/v1/chat/completions' },
|
|
1074
1223
|
{ method: 'POST', path: '/v1/embed' },
|
|
1075
1224
|
{ method: 'POST', path: '/v1/rerank' },
|
|
1076
1225
|
{ method: 'POST', path: '/v1/classify' },
|
|
@@ -1130,6 +1279,7 @@ export async function handleCohereTwinRequest(req: CohereRequest): Promise<Coher
|
|
|
1130
1279
|
return { status: 200, body: buildChatV2(args, outcome) };
|
|
1131
1280
|
}
|
|
1132
1281
|
if (method === 'POST' && path === '/v1/chat') return handleChatV1(params, req);
|
|
1282
|
+
if (method === 'POST' && path === '/compatibility/v1/chat/completions') return handleCompatChat(params, req);
|
|
1133
1283
|
|
|
1134
1284
|
// ── embed / rerank / classify ─────────────────────────────────────────────────────────
|
|
1135
1285
|
if (method === 'POST' && path === '/v2/embed') return handleEmbedV2(params);
|
package/src/index.ts
CHANGED
|
@@ -145,7 +145,13 @@ export const pack: TwinPack = {
|
|
|
145
145
|
// `/v2`. The Bedrock/SageMaker clients the SDK also ships (`AwsClient`, `BedrockClient`,
|
|
146
146
|
// `SagemakerClient`) address AWS hosts, which are a different vendor's plane and deliberately
|
|
147
147
|
// NOT claimed here.
|
|
148
|
-
|
|
148
|
+
//
|
|
149
|
+
// `api.cohere.ai` is Cohere's earlier host, still answering, and the one its Compatibility API
|
|
150
|
+
// documents (docs.cohere.com/docs/compatibility-api: every example's base URL is
|
|
151
|
+
// `https://api.cohere.ai/compatibility/v1`); applications that speak OpenAI's wire to Cohere
|
|
152
|
+
// address it there (LibreChat's own Cohere constant is `https://api.cohere.ai/v1`). The same
|
|
153
|
+
// twin answers both hosts, as the one API behind them.
|
|
154
|
+
hosts: [{ host: 'api.cohere.com' }, { host: 'api.cohere.ai' }],
|
|
149
155
|
|
|
150
156
|
// WORLD WIRING (§7 point 12) — the ruling is NONE, and it is grounded, not skipped.
|
|
151
157
|
// NEITHER SDK reads a base-URL environment variable. cohere-ai takes the override as a
|
|
@@ -155,5 +161,5 @@ export const pack: TwinPack = {
|
|
|
155
161
|
// Inventing a `COHERE_BASE_URL` would make `covers` report the world covered — any app-read env
|
|
156
162
|
// counts — while the app, reading no such var, still talked to the real vendor. Interception is
|
|
157
163
|
// therefore host-based, via the `hosts` entry above, which is the truth.
|
|
158
|
-
endpointEnvNone: 'neither official SDK reads a base-URL env var: cohere-ai@8.1.0 takes the override as the `baseUrl`/`environment` CONSTRUCTOR option and reads only CO_API_KEY from the environment (auth/BearerAuthProvider.js), and @ai-sdk/cohere@4.0.35 takes `options.baseURL` and reads only COHERE_API_KEY. Interception is host-based on api.cohere.com; inventing a COHERE_BASE_URL nothing reads would make `covers` claim a world it does not cover.',
|
|
164
|
+
endpointEnvNone: 'neither official SDK reads a base-URL env var: cohere-ai@8.1.0 takes the override as the `baseUrl`/`environment` CONSTRUCTOR option and reads only CO_API_KEY from the environment (auth/BearerAuthProvider.js), and @ai-sdk/cohere@4.0.35 takes `options.baseURL` and reads only COHERE_API_KEY. Interception is host-based on api.cohere.com and api.cohere.ai; inventing a COHERE_BASE_URL nothing reads would make `covers` claim a world it does not cover.',
|
|
159
165
|
};
|