@volter/twin-cohere 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +224 -0
- package/defaults/handlers.json +26 -0
- package/dist/defaults/handlers.json +26 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +31 -0
- package/dist/src/cohere-budget.d.ts +55 -0
- package/dist/src/cohere-budget.js +171 -0
- package/dist/src/cohere-capabilities.d.ts +14 -0
- package/dist/src/cohere-capabilities.js +1852 -0
- package/dist/src/cohere-conformance.d.ts +17 -0
- package/dist/src/cohere-conformance.js +464 -0
- package/dist/src/cohere-connector.d.ts +150 -0
- package/dist/src/cohere-connector.js +625 -0
- package/dist/src/cohere-models.d.ts +21 -0
- package/dist/src/cohere-models.js +73 -0
- package/dist/src/cohere-scenario.d.ts +57 -0
- package/dist/src/cohere-scenario.js +176 -0
- package/dist/src/cohere-server.d.ts +16 -0
- package/dist/src/cohere-server.js +184 -0
- package/dist/src/cohere-stub.d.ts +119 -0
- package/dist/src/cohere-stub.js +321 -0
- package/dist/src/cohere-twin.d.ts +82 -0
- package/dist/src/cohere-twin.js +1243 -0
- package/dist/src/cohere-types.d.ts +226 -0
- package/dist/src/cohere-types.js +40 -0
- package/dist/src/index.d.ts +15 -0
- package/dist/src/index.js +84 -0
- package/package.json +71 -0
- package/src/cli.ts +30 -0
- package/src/cohere-budget.ts +197 -0
- package/src/cohere-capabilities.ts +1855 -0
- package/src/cohere-conformance.ts +489 -0
- package/src/cohere-connector.ts +709 -0
- package/src/cohere-models.ts +79 -0
- package/src/cohere-scenario.ts +194 -0
- package/src/cohere-server.ts +195 -0
- package/src/cohere-stub.ts +337 -0
- package/src/cohere-twin.ts +1290 -0
- package/src/cohere-types.ts +231 -0
- package/src/index.ts +159 -0
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
// The STATIC Cohere model catalog the twin serves at `GET /v1/models` and `GET /v1/models/{id}`.
|
|
2
|
+
//
|
|
3
|
+
// This is REAL vendor data, not a stub: the ids, context lengths and endpoint compatibility below
|
|
4
|
+
// were read from Cohere's own published models page (https://docs.cohere.com/docs/models, read
|
|
5
|
+
// 2026-08-31) and cross-checked against the model ids that appear in cohere-ai@8.1.0's own
|
|
6
|
+
// `reference.md` usage examples (`command-a-plus-05-2026`, `command-a-03-2025`,
|
|
7
|
+
// `command-a-vision-07-2025`, `embed-v4.0`, `rerank-v4.0-pro`).
|
|
8
|
+
//
|
|
9
|
+
// `endpoints` uses Cohere's own `CompatibleEndpoint` closed set (chat | embed | classify |
|
|
10
|
+
// summarize | rerank | rate | generate — `api/types/CompatibleEndpoint.d.ts`). That closed set is
|
|
11
|
+
// an ORACLE, not decoration: `cohere-twin.test.ts` asserts every catalog entry's endpoints are a
|
|
12
|
+
// subset of it, so an invented endpoint name cannot slip in.
|
|
13
|
+
import { COHERE_ENDPOINTS } from "./cohere-types.js";
|
|
14
|
+
/**
|
|
15
|
+
* The catalog. `tokenizer_url` is DERIVED from Cohere's published tokenizer URL pattern
|
|
16
|
+
* (`storage.googleapis.com/cohere-public/tokenizers/<model>.json`) and is NOT individually verified
|
|
17
|
+
* per model — several of these (the rerank and embed cards especially) may have no such artifact,
|
|
18
|
+
* and neither SDK carries the per-model URLs to check against. Saying so is the honest scope: an
|
|
19
|
+
* earlier note called this "Cohere's real published artifacts", which overstated what was read
|
|
20
|
+
* (§9 round two, m-R2.6). The twin never FETCHES it either way — that would be request-time egress
|
|
21
|
+
* and would break serve-path determinism.
|
|
22
|
+
*/
|
|
23
|
+
export const COHERE_MODELS = [
|
|
24
|
+
// ── chat ──────────────────────────────────────────────────────────────────────────────
|
|
25
|
+
card('command-a-plus-05-2026', ['chat'], 128_000),
|
|
26
|
+
card('command-a-03-2025', ['chat'], 256_000),
|
|
27
|
+
card('command-a-reasoning-08-2025', ['chat'], 256_000),
|
|
28
|
+
card('command-a-vision-07-2025', ['chat'], 128_000),
|
|
29
|
+
card('command-a-translate-08-2025', ['chat'], 8_000),
|
|
30
|
+
card('command-r7b-12-2024', ['chat'], 128_000),
|
|
31
|
+
card('command-r-plus-08-2024', ['chat'], 128_000),
|
|
32
|
+
card('command-r-08-2024', ['chat'], 128_000),
|
|
33
|
+
card('c4ai-aya-expanse-32b', ['chat'], 128_000),
|
|
34
|
+
card('c4ai-aya-vision-32b', ['chat'], 16_000),
|
|
35
|
+
// ── embed ─────────────────────────────────────────────────────────────────────────────
|
|
36
|
+
card('embed-v4.0', ['embed'], 128_000),
|
|
37
|
+
card('embed-english-v3.0', ['embed'], 512),
|
|
38
|
+
card('embed-english-light-v3.0', ['embed'], 512),
|
|
39
|
+
card('embed-multilingual-v3.0', ['embed'], 512),
|
|
40
|
+
card('embed-multilingual-light-v3.0', ['embed'], 512),
|
|
41
|
+
// ── rerank ────────────────────────────────────────────────────────────────────────────
|
|
42
|
+
card('rerank-v4.0-pro', ['rerank'], 32_000),
|
|
43
|
+
card('rerank-v4.0-fast', ['rerank'], 32_000),
|
|
44
|
+
card('rerank-v3.5', ['rerank'], 4_000),
|
|
45
|
+
card('rerank-english-v3.0', ['rerank'], 4_000),
|
|
46
|
+
card('rerank-multilingual-v3.0', ['rerank'], 4_000),
|
|
47
|
+
];
|
|
48
|
+
function card(name, endpoints, contextLength) {
|
|
49
|
+
return {
|
|
50
|
+
name,
|
|
51
|
+
endpoints,
|
|
52
|
+
finetuned: false,
|
|
53
|
+
context_length: contextLength,
|
|
54
|
+
tokenizer_url: `https://storage.googleapis.com/cohere-public/tokenizers/${name}.json`,
|
|
55
|
+
default_endpoints: [],
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
const BY_NAME = new Map(COHERE_MODELS.map((m) => [m.name, m]));
|
|
59
|
+
/** Look a model up by its exact id. Cohere has no alias resolution on this endpoint. */
|
|
60
|
+
export function findModel(name) {
|
|
61
|
+
return BY_NAME.get(name);
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Does `name` name a model that serves `endpoint`? Used to reject `POST /v2/chat` with an embed
|
|
65
|
+
* model (and vice versa) the way the vendor does, rather than happily generating from it — an
|
|
66
|
+
* unmodeled-op-fails-like-the-vendor property that is TESTABLE.
|
|
67
|
+
*/
|
|
68
|
+
export function modelServes(name, endpoint) {
|
|
69
|
+
const m = BY_NAME.get(name);
|
|
70
|
+
return m !== undefined && m.endpoints.includes(endpoint);
|
|
71
|
+
}
|
|
72
|
+
/** The closed `CompatibleEndpoint` set, re-exported so tests can assert against it by name. */
|
|
73
|
+
export { COHERE_ENDPOINTS };
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import { type PackScenarioAdapter, type ScenarioDocument, ScenarioEngine } from '@volter/world-core';
|
|
2
|
+
import { type CohereAssistantContentItem, type CohereFinishReason, type CohereMessageV2, type CohereToolCallV2 } from './cohere-types.js';
|
|
3
|
+
/** The request slice the scenario system sees — built by the `/v2/chat` route from validated
|
|
4
|
+
* args. */
|
|
5
|
+
export type CohereScenarioRequest = {
|
|
6
|
+
model: string;
|
|
7
|
+
messages: CohereMessageV2[];
|
|
8
|
+
tools?: unknown;
|
|
9
|
+
};
|
|
10
|
+
export type CohereScenarioEngine = ScenarioEngine<CohereScenarioRequest>;
|
|
11
|
+
/** A scripted tool call — the full argument payload is emitted verbatim (and streamed as a
|
|
12
|
+
* `tool-call-start` / `tool-call-delta` / `tool-call-end` triple on the streaming path). */
|
|
13
|
+
export type ScenarioToolCall = {
|
|
14
|
+
name: string;
|
|
15
|
+
arguments: Record<string, unknown>;
|
|
16
|
+
id?: string;
|
|
17
|
+
};
|
|
18
|
+
/** What the assistant says when a handler fires. */
|
|
19
|
+
export type CohereScenarioRespond = {
|
|
20
|
+
text?: string;
|
|
21
|
+
/** A `thinking` content item, emitted BEFORE the text item — Cohere's own block ordering. */
|
|
22
|
+
thinking?: string;
|
|
23
|
+
/** Cohere's `tool_plan`: the chain-of-thought reflection that accompanies tool calls. */
|
|
24
|
+
toolPlan?: string;
|
|
25
|
+
toolCalls?: ScenarioToolCall | ScenarioToolCall[];
|
|
26
|
+
/** Defaults to 'TOOL_CALL' when any tool call is present, else 'COMPLETE'. UPPER-CASE, from
|
|
27
|
+
* Cohere's own `ChatFinishReason` set — not OpenAI's lower-case values. */
|
|
28
|
+
finishReason?: CohereFinishReason;
|
|
29
|
+
/** Scripted API error instead of a body: 429 rate limit (with retry-after), 500 api_error,
|
|
30
|
+
* 503 service_unavailable. Exclusive of the content forms. */
|
|
31
|
+
error?: {
|
|
32
|
+
type: 'rate_limit' | 'api_error' | 'service_unavailable';
|
|
33
|
+
retryAfter?: number;
|
|
34
|
+
};
|
|
35
|
+
};
|
|
36
|
+
/** What a fired handler yields — the assistant turn for the envelope. */
|
|
37
|
+
export type ScriptedResult = {
|
|
38
|
+
content: CohereAssistantContentItem[];
|
|
39
|
+
toolCalls: CohereToolCallV2[];
|
|
40
|
+
toolPlan?: string;
|
|
41
|
+
finishReason: CohereFinishReason;
|
|
42
|
+
};
|
|
43
|
+
/** The pack's scenario vocabulary + feature bag. One adapter instance, shared. */
|
|
44
|
+
export declare const cohereScenarioAdapter: PackScenarioAdapter<CohereScenarioRequest>;
|
|
45
|
+
/** Load + strictly validate a handlers document (handlers/cohere.json). A broken file fails
|
|
46
|
+
* server startup loudly; it never falls back or misfires silently. */
|
|
47
|
+
export declare function loadCohereScenarioDocument(path: string): ScenarioDocument;
|
|
48
|
+
export declare function createCohereScenarioEngine(document?: ScenarioDocument): CohereScenarioEngine;
|
|
49
|
+
/**
|
|
50
|
+
* Realize a fired handler's respond payload into a vendor-faithful Cohere assistant turn.
|
|
51
|
+
*
|
|
52
|
+
* Scripted tool-call ids are derived from the CALL ITSELF (name + arguments + position), so they
|
|
53
|
+
* are stable across runs and distinct between different scripted calls. A module-level counter
|
|
54
|
+
* would make them depend on how many engines had run in the process — not a property a replay can
|
|
55
|
+
* rely on. An explicit `id` on the handler always wins.
|
|
56
|
+
*/
|
|
57
|
+
export declare function realizeCohereRespond(respond: CohereScenarioRespond): ScriptedResult;
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
// The cohere pack's HALF of the scenario system: vocabulary + realization. The GRAMMAR
|
|
2
|
+
// (ordering, once/scope/phase, extractors, strict parsing, miss records, twin.use) is the
|
|
3
|
+
// kernel's ONE engine — @volter/world-core scenario.ts; see company-repo
|
|
4
|
+
// BRIEFS/TWIN-PROGRAMMING-MODEL.md (LOCKED). There is deliberately NO bespoke engine here.
|
|
5
|
+
// This module declares:
|
|
6
|
+
// • the `on` vocabulary a handler may match (model/userText/toolResult/hasTool …),
|
|
7
|
+
// • what a `respond` payload may contain (text / thinking / toolPlan / toolCalls / error),
|
|
8
|
+
// • how a fired handler's respond REALIZES into a vendor-faithful Cohere assistant turn
|
|
9
|
+
// (content items, `tool_plan`, `tool_calls`, and Cohere's UPPER-CASE finish reason).
|
|
10
|
+
// It deliberately declares NO `scopeKey`: `/v2/chat` carries no end-user discriminator of the kind
|
|
11
|
+
// anthropic's `metadata.user_id` or mistral's `user` provides, so there is nothing honest to key a
|
|
12
|
+
// session on and every handler runs in the WORLD scope. The kernel enforces that — a handler
|
|
13
|
+
// written with `scope: "session"` is REFUSED at load rather than silently sharing world state.
|
|
14
|
+
// (§9 round 1 caught this header claiming the key anyway.)
|
|
15
|
+
// The handler FILE (handlers/cohere.json in a world dir) is the only write surface.
|
|
16
|
+
//
|
|
17
|
+
// Scenario support is TWIN-ONLY SCAFFOLDING for eval worlds, not vendor surface, so it is
|
|
18
|
+
// deliberately KEPT OUT of the capability manifest (the same rule as twin-only seed routes) and
|
|
19
|
+
// gated by cohere-scenario.test.ts instead.
|
|
20
|
+
import { getActiveWorldStore, parseScenarioDocument, ScenarioError, ScenarioEngine } from '@volter/world-core';
|
|
21
|
+
import { contentToText, lastUserText, toolCallId, toolNames } from "./cohere-stub.js";
|
|
22
|
+
import { COHERE_FINISH_REASONS } from "./cohere-types.js";
|
|
23
|
+
const RESPOND_KEYS = new Set(['text', 'thinking', 'toolPlan', 'toolCalls', 'finishReason', 'error']);
|
|
24
|
+
const FINISH_REASONS = new Set(COHERE_FINISH_REASONS);
|
|
25
|
+
const ERROR_TYPES = new Set(['rate_limit', 'api_error', 'service_unavailable']);
|
|
26
|
+
const nonEmptyString = (cond) => typeof cond === 'string' && cond.length > 0;
|
|
27
|
+
/**
|
|
28
|
+
* The tool NAMES the trailing `role:'tool'` messages answer — resolved through the
|
|
29
|
+
* `tool_call_id` of earlier assistant turns. Cohere's tool result is a `role:'tool'` message
|
|
30
|
+
* carrying `tool_call_id` (`ChatMessageV2.Tool`), and the id lives on the assistant turn's
|
|
31
|
+
* `tool_calls[].id`.
|
|
32
|
+
*/
|
|
33
|
+
function lastToolResultNames(messages) {
|
|
34
|
+
const names = new Set();
|
|
35
|
+
const last = messages[messages.length - 1];
|
|
36
|
+
if (!last || last.role !== 'tool' || typeof last.tool_call_id !== 'string')
|
|
37
|
+
return names;
|
|
38
|
+
for (const m of messages) {
|
|
39
|
+
if (m.role !== 'assistant' || !Array.isArray(m.tool_calls))
|
|
40
|
+
continue;
|
|
41
|
+
for (const tc of m.tool_calls) {
|
|
42
|
+
if (tc?.id === last.tool_call_id && typeof tc?.function?.name === 'string')
|
|
43
|
+
names.add(tc.function.name);
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
return names;
|
|
47
|
+
}
|
|
48
|
+
/** The pack's scenario vocabulary + feature bag. One adapter instance, shared. */
|
|
49
|
+
export const cohereScenarioAdapter = {
|
|
50
|
+
vendor: 'cohere',
|
|
51
|
+
features: (req) => ({
|
|
52
|
+
model: req.model,
|
|
53
|
+
lastUserText: lastUserText(req.messages).slice(0, 300),
|
|
54
|
+
tools: toolNames(req.tools),
|
|
55
|
+
lastMessageIsToolResult: (req.messages[req.messages.length - 1]?.role ?? '') === 'tool',
|
|
56
|
+
toolResultFor: [...lastToolResultNames(req.messages)],
|
|
57
|
+
}),
|
|
58
|
+
matchers: {
|
|
59
|
+
modelEquals: (req, cond) => nonEmptyString(cond) && req.model === cond,
|
|
60
|
+
userTextIncludes: (req, cond) => nonEmptyString(cond) && lastUserText(req.messages).toLowerCase().includes(cond.toLowerCase()),
|
|
61
|
+
anyTextIncludes: (req, cond) => nonEmptyString(cond) && req.messages.map((m) => contentToText(m.content)).join('\n').toLowerCase().includes(cond.toLowerCase()),
|
|
62
|
+
lastMessageIsToolResult: (req, cond) => typeof cond === 'boolean' && ((req.messages[req.messages.length - 1]?.role ?? '') === 'tool') === cond,
|
|
63
|
+
toolResultFor: (req, cond) => nonEmptyString(cond) && lastToolResultNames(req.messages).has(cond),
|
|
64
|
+
hasTool: (req, cond) => nonEmptyString(cond) && toolNames(req.tools).includes(cond),
|
|
65
|
+
},
|
|
66
|
+
text: (req) => req.messages.map((m) => contentToText(m.content)).join('\n'),
|
|
67
|
+
validateOn: (on) => {
|
|
68
|
+
for (const k of ['modelEquals', 'userTextIncludes', 'anyTextIncludes', 'toolResultFor', 'hasTool']) {
|
|
69
|
+
if (on[k] !== undefined && (typeof on[k] !== 'string' || !on[k]))
|
|
70
|
+
return `on.${k} is a non-empty string`;
|
|
71
|
+
}
|
|
72
|
+
if (on.lastMessageIsToolResult !== undefined && typeof on.lastMessageIsToolResult !== 'boolean')
|
|
73
|
+
return 'on.lastMessageIsToolResult is a boolean';
|
|
74
|
+
return null;
|
|
75
|
+
},
|
|
76
|
+
validateRespond: (respond) => {
|
|
77
|
+
if (typeof respond !== 'object' || respond === null || Array.isArray(respond))
|
|
78
|
+
return 'respond is an object { text?, thinking?, toolPlan?, toolCalls?, finishReason?, error? }';
|
|
79
|
+
const r = respond;
|
|
80
|
+
for (const k of Object.keys(r))
|
|
81
|
+
if (!RESPOND_KEYS.has(k))
|
|
82
|
+
return `respond: unknown key "${k}" (valid: ${[...RESPOND_KEYS].join(', ')})`;
|
|
83
|
+
if (r.text !== undefined && typeof r.text !== 'string')
|
|
84
|
+
return 'respond.text is a string';
|
|
85
|
+
if (r.thinking !== undefined && typeof r.thinking !== 'string')
|
|
86
|
+
return 'respond.thinking is a string';
|
|
87
|
+
if (r.toolPlan !== undefined && typeof r.toolPlan !== 'string')
|
|
88
|
+
return 'respond.toolPlan is a string';
|
|
89
|
+
if (r.finishReason !== undefined && (typeof r.finishReason !== 'string' || !FINISH_REASONS.has(r.finishReason))) {
|
|
90
|
+
return `respond.finishReason is one of ${[...FINISH_REASONS].join(', ')}`;
|
|
91
|
+
}
|
|
92
|
+
if (r.toolCalls !== undefined) {
|
|
93
|
+
for (const tc of Array.isArray(r.toolCalls) ? r.toolCalls : [r.toolCalls]) {
|
|
94
|
+
const t = tc;
|
|
95
|
+
if (!t || typeof t !== 'object' || Array.isArray(t))
|
|
96
|
+
return 'respond.toolCalls entries are objects';
|
|
97
|
+
if (typeof t.name !== 'string' || !t.name)
|
|
98
|
+
return 'respond.toolCalls[].name is a non-empty string';
|
|
99
|
+
if (!t.arguments || typeof t.arguments !== 'object' || Array.isArray(t.arguments))
|
|
100
|
+
return 'respond.toolCalls[].arguments is an object';
|
|
101
|
+
if (t.id !== undefined && typeof t.id !== 'string')
|
|
102
|
+
return 'respond.toolCalls[].id is a string';
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
if (r.error !== undefined) {
|
|
106
|
+
const e = r.error;
|
|
107
|
+
if (!e || typeof e !== 'object' || Array.isArray(e))
|
|
108
|
+
return 'respond.error is an object { type, retryAfter? }';
|
|
109
|
+
if (typeof e.type !== 'string' || !ERROR_TYPES.has(e.type))
|
|
110
|
+
return `respond.error.type is one of ${[...ERROR_TYPES].join(', ')}`;
|
|
111
|
+
if (e.retryAfter !== undefined && (typeof e.retryAfter !== 'number' || e.retryAfter <= 0))
|
|
112
|
+
return 'respond.error.retryAfter is a positive number of seconds';
|
|
113
|
+
for (const k of Object.keys(e))
|
|
114
|
+
if (!['type', 'retryAfter'].includes(k))
|
|
115
|
+
return `respond.error: unknown key "${k}"`;
|
|
116
|
+
}
|
|
117
|
+
const hasBody = r.text !== undefined || r.thinking !== undefined || r.toolPlan !== undefined || r.toolCalls !== undefined;
|
|
118
|
+
if (r.error !== undefined && hasBody)
|
|
119
|
+
return 'respond.error stands alone (no content forms beside it)';
|
|
120
|
+
if (!hasBody && r.error === undefined)
|
|
121
|
+
return 'respond needs text, thinking, toolPlan, toolCalls, or error';
|
|
122
|
+
return null;
|
|
123
|
+
},
|
|
124
|
+
};
|
|
125
|
+
/** Load + strictly validate a handlers document (handlers/cohere.json). A broken file fails
|
|
126
|
+
* server startup loudly; it never falls back or misfires silently. */
|
|
127
|
+
export function loadCohereScenarioDocument(path) {
|
|
128
|
+
let parsed;
|
|
129
|
+
try {
|
|
130
|
+
// Read through the ACTIVE WorldStore, never the filesystem directly (runtime contract
|
|
131
|
+
// R12b): the handlers document is WORLD STATE, so a MemoryWorldStore / DO-backed world
|
|
132
|
+
// serves ITS OWN scenario instead of whatever happens to sit on the host disk — and the
|
|
133
|
+
// serve path stays workerd-clean. A missing document keeps the historical ENOENT wording,
|
|
134
|
+
// so the loud load-time failure reads byte-identically to the read it replaces.
|
|
135
|
+
const raw = getActiveWorldStore().read(path);
|
|
136
|
+
if (raw === null)
|
|
137
|
+
throw new Error(`ENOENT: no such file or directory, open '${path}'`);
|
|
138
|
+
parsed = JSON.parse(raw);
|
|
139
|
+
}
|
|
140
|
+
catch (e) {
|
|
141
|
+
throw new ScenarioError(`cohere scenario: cannot read/parse ${path}: ${e instanceof Error ? e.message : String(e)}`);
|
|
142
|
+
}
|
|
143
|
+
return parseScenarioDocument(parsed, cohereScenarioAdapter);
|
|
144
|
+
}
|
|
145
|
+
export function createCohereScenarioEngine(document) {
|
|
146
|
+
return new ScenarioEngine(cohereScenarioAdapter, document);
|
|
147
|
+
}
|
|
148
|
+
/**
|
|
149
|
+
* Realize a fired handler's respond payload into a vendor-faithful Cohere assistant turn.
|
|
150
|
+
*
|
|
151
|
+
* Scripted tool-call ids are derived from the CALL ITSELF (name + arguments + position), so they
|
|
152
|
+
* are stable across runs and distinct between different scripted calls. A module-level counter
|
|
153
|
+
* would make them depend on how many engines had run in the process — not a property a replay can
|
|
154
|
+
* rely on. An explicit `id` on the handler always wins.
|
|
155
|
+
*/
|
|
156
|
+
export function realizeCohereRespond(respond) {
|
|
157
|
+
const toolCalls = [];
|
|
158
|
+
for (const tc of respond.toolCalls ? (Array.isArray(respond.toolCalls) ? respond.toolCalls : [respond.toolCalls]) : []) {
|
|
159
|
+
toolCalls.push({
|
|
160
|
+
id: tc.id ?? toolCallId(`scripted|${tc.name}|${JSON.stringify(tc.arguments)}|${toolCalls.length}`),
|
|
161
|
+
type: 'function',
|
|
162
|
+
function: { name: tc.name, arguments: JSON.stringify(tc.arguments) },
|
|
163
|
+
});
|
|
164
|
+
}
|
|
165
|
+
const content = [];
|
|
166
|
+
if (respond.thinking !== undefined)
|
|
167
|
+
content.push({ type: 'thinking', thinking: respond.thinking });
|
|
168
|
+
if (respond.text !== undefined)
|
|
169
|
+
content.push({ type: 'text', text: respond.text });
|
|
170
|
+
return {
|
|
171
|
+
content,
|
|
172
|
+
toolCalls,
|
|
173
|
+
...(respond.toolPlan !== undefined ? { toolPlan: respond.toolPlan } : {}),
|
|
174
|
+
finishReason: respond.finishReason ?? (toolCalls.length ? 'TOOL_CALL' : 'COMPLETE'),
|
|
175
|
+
};
|
|
176
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/** Options every Cohere-twin HTTP surface needs, independent of who owns the socket. */
|
|
2
|
+
export interface CohereTwinFetchOptions {
|
|
3
|
+
root?: string;
|
|
4
|
+
readOnly?: boolean;
|
|
5
|
+
scenarioPath?: string;
|
|
6
|
+
}
|
|
7
|
+
export declare function createCohereTwinFetch(options: CohereTwinFetchOptions): (request: Request) => Promise<Response>;
|
|
8
|
+
export declare function createCohereTwinServer(options: {
|
|
9
|
+
root?: string;
|
|
10
|
+
port?: number;
|
|
11
|
+
readOnly?: boolean;
|
|
12
|
+
scenarioPath?: string;
|
|
13
|
+
}): Promise<{
|
|
14
|
+
port: number;
|
|
15
|
+
stop: () => void;
|
|
16
|
+
}>;
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
// Cohere twin HTTP server — serve the full Cohere twin handler over HTTP so the real `cohere-ai`
|
|
2
|
+
// client (`new CohereClientV2({ token, environment: 'http://127.0.0.1:<port>' })`) and the real
|
|
3
|
+
// `@ai-sdk/cohere` provider (`createCohere({ baseURL: 'http://127.0.0.1:<port>/v2' })`) work
|
|
4
|
+
// UNMODIFIED. Writable by default; pass readOnly to reject mutations (D3).
|
|
5
|
+
//
|
|
6
|
+
// ── TWO STREAM WIRES, because Cohere has two ────────────────────────────────────────────
|
|
7
|
+
// `POST /v2/chat` with `stream:true` is SSE: `data: <json>\n\n` frames terminated by
|
|
8
|
+
// `data: [DONE]\n\n` — cohere-ai constructs its `core.Stream` for this endpoint with
|
|
9
|
+
// `responseType: "sse"` and `eventShape: { type: "sse", streamTerminator: "[DONE]" }`.
|
|
10
|
+
// `POST /v1/chat` with `stream:true` is NEWLINE-DELIMITED JSON — a different wire on the same
|
|
11
|
+
// host, which is why the framing is chosen per endpoint rather than globally.
|
|
12
|
+
//
|
|
13
|
+
// Multipart: `POST /v1/datasets` is `multipart/form-data` with the dataset name and type in the
|
|
14
|
+
// QUERY string. The server folds the form into the handler's JSON contract so the handler stays a
|
|
15
|
+
// pure JSON function (offline-testable).
|
|
16
|
+
//
|
|
17
|
+
// FETCH-FIRST (runtime contract R12b): the surface is the plain `createCohereTwinFetch` and the
|
|
18
|
+
// SERVER is one line of `Bun.serve` around it. This is a CUSTOM fetch, not the kernel adapter
|
|
19
|
+
// (`createTwinFetchFromHandler`): the two stream wires and the multipart adaptation genuinely
|
|
20
|
+
// exceed the common shape — openai-server.ts is the reference for that lane.
|
|
21
|
+
import { serveHttp } from '@volter/world-core';
|
|
22
|
+
import { twinManifest, worldNow } from '@volter/world-core';
|
|
23
|
+
import { createCohereScenarioEngine, loadCohereScenarioDocument } from "./cohere-scenario.js";
|
|
24
|
+
import { handleCohereTwinRequest } from "./cohere-twin.js";
|
|
25
|
+
function wantsStream(body) {
|
|
26
|
+
if (!body)
|
|
27
|
+
return false;
|
|
28
|
+
try {
|
|
29
|
+
return JSON.parse(body)?.stream === true;
|
|
30
|
+
}
|
|
31
|
+
catch {
|
|
32
|
+
return false;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
/** v2's SSE framing. */
|
|
36
|
+
function encodeSse(event) {
|
|
37
|
+
if (event.done)
|
|
38
|
+
return 'data: [DONE]\n\n';
|
|
39
|
+
return `data: ${JSON.stringify(event.data)}\n\n`;
|
|
40
|
+
}
|
|
41
|
+
/** v1's newline-delimited-JSON framing. There is no `[DONE]` sentinel on this wire — the final
|
|
42
|
+
* object's `is_finished: true` is the terminator. */
|
|
43
|
+
function encodeNdjson(event) {
|
|
44
|
+
if (event.done)
|
|
45
|
+
return '';
|
|
46
|
+
return `${JSON.stringify(event.data)}\n`;
|
|
47
|
+
}
|
|
48
|
+
async function multipartToJson(request) {
|
|
49
|
+
try {
|
|
50
|
+
const form = await request.formData();
|
|
51
|
+
const out = {};
|
|
52
|
+
for (const [key, value] of form.entries()) {
|
|
53
|
+
if (typeof value === 'string')
|
|
54
|
+
out[key] = value;
|
|
55
|
+
}
|
|
56
|
+
const file = form.get('data') ?? form.get('file');
|
|
57
|
+
if (file instanceof File) {
|
|
58
|
+
out.filename = file.name || 'upload';
|
|
59
|
+
out.content = await file.text();
|
|
60
|
+
}
|
|
61
|
+
else if (typeof file === 'string') {
|
|
62
|
+
// A multipart part is a `File` only when its Content-Disposition carries a `filename`.
|
|
63
|
+
// cohere-ai's `appendFile` omits it whenever the value is not a named value (a Buffer, a
|
|
64
|
+
// `fs.createReadStream`, an unnamed Blob), in which case Bun yields a plain STRING. Before
|
|
65
|
+
// the required-file rule that was harmless; after it, the twin would 400 "the data file is
|
|
66
|
+
// required" on a create the vendor accepts — a leniency turned into a false refusal
|
|
67
|
+
// (§9 round two, m-R2.4).
|
|
68
|
+
out.filename = 'upload';
|
|
69
|
+
out.content = file;
|
|
70
|
+
}
|
|
71
|
+
return JSON.stringify(out);
|
|
72
|
+
}
|
|
73
|
+
catch {
|
|
74
|
+
return '{}';
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
export function createCohereTwinFetch(options) {
|
|
78
|
+
const readOnly = options.readOnly ?? false;
|
|
79
|
+
// Scenario scripting (cohere-scenario.ts): a JSON handlers file — via the scenarioPath option or
|
|
80
|
+
// the TWIN_COHERE_SCENARIO env var — scripts `POST /v2/chat`. Loaded ONCE at construction (a
|
|
81
|
+
// malformed file fails loudly here, never silently); the session (call counter + fired `once`
|
|
82
|
+
// handlers) lives for the fetch's lifetime.
|
|
83
|
+
const scenarioPath = options.scenarioPath ?? process.env.TWIN_COHERE_SCENARIO;
|
|
84
|
+
const scenarioEngine = scenarioPath ? createCohereScenarioEngine(loadCohereScenarioDocument(scenarioPath)) : undefined;
|
|
85
|
+
return async function cohereTwinFetch(request) {
|
|
86
|
+
const url = new URL(request.url);
|
|
87
|
+
const cleanPath = url.pathname.replace(/\/+$/, '') || '/';
|
|
88
|
+
// THE READ DOORS (TWIN-PROGRAMMING-MODEL): discovery + inspection, unauthenticated —
|
|
89
|
+
// education ships inside the twin; an in-world agent has only HTTP. Read-only by design:
|
|
90
|
+
// the handler FILE is the only write surface, so a running world is never mutated.
|
|
91
|
+
if (request.method === 'GET' && cleanPath === '/twin') {
|
|
92
|
+
return Response.json(twinManifest({
|
|
93
|
+
vendor: 'cohere',
|
|
94
|
+
twinOf: 'Cohere API (v1 + v2)',
|
|
95
|
+
stateSentence: 'Seed datasets, connectors and embed jobs through the ordinary API with any bearer token; the tokenizer vocabulary is learned by POST /v1/tokenize.',
|
|
96
|
+
behaviorSentence: "Chat completions are scripted by MSW-shaped handlers in the world dir (handlers/cohere.json): {on:{userTextIncludes|anyTextIncludes|modelEquals|hasTool|toolResultFor|lastMessageIsToolResult|nthCall}, respond:{text|thinking|toolPlan|toolCalls, finishReason? | error:{type:rate_limit|api_error|service_unavailable}}, once?, scope?, phase?}. Unmatched requests answer a labeled stub and ledger the miss with its features.",
|
|
97
|
+
exampleHandler: { on: { userTextIncludes: 'refund', hasTool: 'create_job' }, respond: { toolCalls: { name: 'create_job', arguments: { title: 'Refund order' } } }, once: true },
|
|
98
|
+
engine: scenarioEngine,
|
|
99
|
+
}));
|
|
100
|
+
}
|
|
101
|
+
if (request.method === 'GET' && cleanPath === '/twin/scenario') {
|
|
102
|
+
return Response.json(scenarioEngine ? scenarioEngine.status() : { vendor: 'cohere', handlers: [], misses: 0, recentMisses: [] });
|
|
103
|
+
}
|
|
104
|
+
const path = url.pathname + (url.search || '');
|
|
105
|
+
const contentType = request.headers.get('content-type') ?? '';
|
|
106
|
+
// Pass the (lower-cased) request headers through so the handler can model auth (401)
|
|
107
|
+
// exactly as the real API sees it.
|
|
108
|
+
const headers = {};
|
|
109
|
+
request.headers.forEach((v, k) => { headers[k.toLowerCase()] = v; });
|
|
110
|
+
let body = '';
|
|
111
|
+
if (request.method !== 'GET') {
|
|
112
|
+
body = contentType.includes('multipart/form-data') ? await multipartToJson(request) : await request.text();
|
|
113
|
+
}
|
|
114
|
+
try {
|
|
115
|
+
// Streaming POST → RUN FIRST, THEN CHOOSE THE RESPONSE.
|
|
116
|
+
//
|
|
117
|
+
// The handler is deliberately socket-free (it writes into an injected sink), so buffering
|
|
118
|
+
// its events costs nothing and buys the thing that matters: the STATUS is known BEFORE the
|
|
119
|
+
// response is constructed. Opening a 200 `text/event-stream` and only then discovering a
|
|
120
|
+
// 400 would push the error body out as a lone SSE frame, and cohere-ai's `core.Stream`
|
|
121
|
+
// would try to parse a `{message}` error as a `V2ChatStreamResponse` instead of raising
|
|
122
|
+
// the typed `BadRequestError` the caller catches. Real Cohere sends response headers
|
|
123
|
+
// before any body and returns a genuine 4xx here.
|
|
124
|
+
const isV2Stream = request.method.toUpperCase() === 'POST' && cleanPath === '/v2/chat' && wantsStream(body);
|
|
125
|
+
const isV1Stream = request.method.toUpperCase() === 'POST' && cleanPath === '/v1/chat' && wantsStream(body);
|
|
126
|
+
if (!readOnly && (isV2Stream || isV1Stream)) {
|
|
127
|
+
const events = [];
|
|
128
|
+
const { status, body: out, headers: extra } = await handleCohereTwinRequest({
|
|
129
|
+
method: request.method, path, body, readOnly, headers, occurredAt: worldNow(),
|
|
130
|
+
...(options.root !== undefined ? { root: options.root } : {}), sseSink: (e) => events.push(e),
|
|
131
|
+
...(scenarioEngine ? { scenarioEngine } : {}),
|
|
132
|
+
});
|
|
133
|
+
if (status >= 400) {
|
|
134
|
+
// A real, catchable HTTP error — never a 200 stream carrying an error payload.
|
|
135
|
+
return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', ...(extra ?? {}) } });
|
|
136
|
+
}
|
|
137
|
+
const enc = new TextEncoder();
|
|
138
|
+
const frame = isV2Stream ? encodeSse : encodeNdjson;
|
|
139
|
+
const stream = new ReadableStream({
|
|
140
|
+
start(controller) {
|
|
141
|
+
for (const e of events) {
|
|
142
|
+
const text = frame(e);
|
|
143
|
+
if (text)
|
|
144
|
+
controller.enqueue(enc.encode(text));
|
|
145
|
+
}
|
|
146
|
+
controller.close();
|
|
147
|
+
},
|
|
148
|
+
});
|
|
149
|
+
return new Response(stream, {
|
|
150
|
+
status,
|
|
151
|
+
headers: {
|
|
152
|
+
'content-type': isV2Stream ? 'text/event-stream; charset=utf-8' : 'application/stream+json; charset=utf-8',
|
|
153
|
+
'cache-control': 'no-cache',
|
|
154
|
+
...(extra ?? {}),
|
|
155
|
+
},
|
|
156
|
+
});
|
|
157
|
+
}
|
|
158
|
+
const { status, body: out, headers: extra } = await handleCohereTwinRequest({
|
|
159
|
+
method: request.method, path, body, readOnly, headers,
|
|
160
|
+
occurredAt: worldNow(),
|
|
161
|
+
...(options.root !== undefined ? { root: options.root } : {}),
|
|
162
|
+
...(scenarioEngine ? { scenarioEngine } : {}),
|
|
163
|
+
});
|
|
164
|
+
return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', ...(extra ?? {}) } });
|
|
165
|
+
}
|
|
166
|
+
catch (error) {
|
|
167
|
+
// Never let an exception escape the fetch callback: under `bun test` a thrown handler error
|
|
168
|
+
// is reported as "unhandled between tests" and can prevent later test registration,
|
|
169
|
+
// obscuring the original defect. Answer the vendor's own error envelope.
|
|
170
|
+
return new Response(JSON.stringify({ message: `twin handler error: ${error instanceof Error ? error.message : String(error)}` }), {
|
|
171
|
+
status: 500,
|
|
172
|
+
headers: { 'content-type': 'application/json' },
|
|
173
|
+
});
|
|
174
|
+
}
|
|
175
|
+
};
|
|
176
|
+
}
|
|
177
|
+
export async function createCohereTwinServer(options) {
|
|
178
|
+
const server = await serveHttp({
|
|
179
|
+
port: options.port ?? 0,
|
|
180
|
+
idleTimeout: 60,
|
|
181
|
+
fetch: createCohereTwinFetch(options),
|
|
182
|
+
});
|
|
183
|
+
return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
|
|
184
|
+
}
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import type { CohereMessageV2, CohereToolCallV2 } from './cohere-types.js';
|
|
2
|
+
/** A small deterministic 32-bit hash (FNV-1a) of a string — the seed for every stub below. */
|
|
3
|
+
export declare function fnv1a(text: string): number;
|
|
4
|
+
/**
|
|
5
|
+
* A deterministic UUID-v4-SHAPED id. Cohere returns UUIDs for `id` / `generation_id` /
|
|
6
|
+
* `response_id`, so the twin returns something a caller parsing a UUID accepts — derived from the
|
|
7
|
+
* request rather than from `crypto.randomUUID`, because a random id would make the served response
|
|
8
|
+
* non-deterministic and break replay (CLAUDE.md, serve-path determinism).
|
|
9
|
+
*/
|
|
10
|
+
export declare function cohereId(seed: string): string;
|
|
11
|
+
/** Deterministic token estimate: ~1 token per 4 chars. Never zero for non-empty text. */
|
|
12
|
+
export declare function estimateTokens(text: string): number;
|
|
13
|
+
/** Flatten a v2 message's content (string OR content-part array) to its text. Cohere's parts are
|
|
14
|
+
* `{type:'text',text}`, `{type:'image_url',image_url}` and `{type:'document',document}`; only the
|
|
15
|
+
* first contributes literal text, the rest contribute their JSON length so the count stays
|
|
16
|
+
* deterministic and reflects payload size. */
|
|
17
|
+
export declare function contentToText(content: CohereMessageV2['content']): string;
|
|
18
|
+
/** Deterministic input-token count for a full v2 message list. */
|
|
19
|
+
export declare function countInputTokens(messages: CohereMessageV2[]): number;
|
|
20
|
+
/** The last user turn's text — the thing the stub echoes. */
|
|
21
|
+
export declare function lastUserText(messages: CohereMessageV2[]): string;
|
|
22
|
+
/**
|
|
23
|
+
* The deterministic stub ASSISTANT text. Unmistakably a twin stub: it carries the
|
|
24
|
+
* `[twin-stub:<model>]` marker and echoes the prompt, so no caller can mistake it for real model
|
|
25
|
+
* output. Deterministic for a given prompt → assertable in tests.
|
|
26
|
+
*/
|
|
27
|
+
export declare function stubAssistantText(messages: CohereMessageV2[], model: string): string;
|
|
28
|
+
/** The v1 chat stub — v1 takes a single `message` string rather than a message list. */
|
|
29
|
+
export declare function stubV1Text(message: string, model: string): string;
|
|
30
|
+
/** The `tool_plan` a real Cohere model emits alongside tool calls — a chain-of-thought style
|
|
31
|
+
* reflection. Stubbed deterministically and labeled. */
|
|
32
|
+
export declare function stubToolPlan(toolNames: string[]): string;
|
|
33
|
+
/** Every declared tool's name, in declaration order. */
|
|
34
|
+
export declare function toolNames(tools: unknown): string[];
|
|
35
|
+
/**
|
|
36
|
+
* A deterministic stub argument STRING for a tool. Real models emit arguments that validate
|
|
37
|
+
* against the tool's declared JSON schema, so the twin synthesizes a deterministic object
|
|
38
|
+
* containing every declared property with a type-appropriate placeholder. Cohere's `arguments` is
|
|
39
|
+
* a STRING on the wire (`ToolCallV2Function.arguments`), like OpenAI's.
|
|
40
|
+
*/
|
|
41
|
+
export declare function stubToolArguments(tool: unknown): string;
|
|
42
|
+
/**
|
|
43
|
+
* When tools are provided, a real Cohere model may answer with `tool_calls` and
|
|
44
|
+
* `finish_reason: 'TOOL_CALL'`. The stub deterministically "calls" the tool selected by
|
|
45
|
+
* `forcedName` (a named `tool_choice`) or the FIRST provided tool.
|
|
46
|
+
*
|
|
47
|
+
* The id is SEEDED FROM THE REQUEST, never from a module-level counter: a counter makes the id
|
|
48
|
+
* depend on how many completions the process happened to serve, which is not a property a replay
|
|
49
|
+
* can rely on, and hands two different turns the same id inside one process. Hashing the
|
|
50
|
+
* conversation keeps replay byte-identical while making ids differ between turns.
|
|
51
|
+
*/
|
|
52
|
+
export declare function stubToolCall(tools: unknown, seed: string, index: number, forcedName?: string): CohereToolCallV2 | null;
|
|
53
|
+
/** Cohere's tool-call ids are UUID-shaped, like its response ids. */
|
|
54
|
+
export declare function toolCallId(seed: string): string;
|
|
55
|
+
/**
|
|
56
|
+
* A deterministic, reproducible L2-normalized pseudo-embedding for `text`. NOT a real embedding —
|
|
57
|
+
* the values carry no semantic meaning; only the SHAPE, the dimensionality and the DETERMINISM are
|
|
58
|
+
* faithful. Same text → same vector, in every process
|
|
59
|
+
* and every state root.
|
|
60
|
+
*/
|
|
61
|
+
export declare function pseudoEmbedding(text: string, dimensions: number): number[];
|
|
62
|
+
/**
|
|
63
|
+
* Quantize a float vector into one of Cohere's non-float `embedding_types`. The vendor returns the
|
|
64
|
+
* SAME vector under every requested type, quantized — so the twin derives all of them from the one
|
|
65
|
+
* pseudo-vector rather than hashing separately per type, which is what makes the types agree the
|
|
66
|
+
* way a real response's do.
|
|
67
|
+
* • int8 — signed 8-bit, [-128, 127]
|
|
68
|
+
* • uint8 — unsigned 8-bit, [0, 255]
|
|
69
|
+
* • binary/ubinary — one BIT per dimension, packed 8 to a byte (so the array is dim/8 long),
|
|
70
|
+
* signed and unsigned respectively; this packing is why a binary embedding is 1/32 the size.
|
|
71
|
+
*/
|
|
72
|
+
export declare function quantizeEmbedding(floats: number[], type: 'int8' | 'uint8' | 'binary' | 'ubinary'): number[];
|
|
73
|
+
/** Base64 of the float vector's little-endian float32 bytes — Cohere's `base64` embedding type. */
|
|
74
|
+
export declare function base64Embedding(floats: number[]): string;
|
|
75
|
+
/** The output dimensionality Cohere's embed models produce, by model id. `embed-v4.0` accepts an
|
|
76
|
+
* explicit `output_dimension` from the closed set {256, 512, 1024, 1536}; the v3 models are fixed
|
|
77
|
+
* (1024 for the full models, 384 for the `light` ones). */
|
|
78
|
+
export declare function embedDimensions(model: string): number;
|
|
79
|
+
/** `embed-v4.0`'s documented closed `output_dimension` set. A literal here, so the handler's
|
|
80
|
+
* rejection of anything outside it is checkable against the vendor's documented set. */
|
|
81
|
+
export declare const COHERE_OUTPUT_DIMENSIONS: readonly [256, 512, 1024, 1536];
|
|
82
|
+
/**
|
|
83
|
+
* Score one document against a query. NOT the real cross-encoder — a deterministic
|
|
84
|
+
* token-overlap ratio, so the ORDERING is explainable and reproducible while carrying no learned
|
|
85
|
+
* relevance. Always in [0, 1], the range the vendor's `relevance_score` occupies.
|
|
86
|
+
*
|
|
87
|
+
* A tiny hash term breaks ties deterministically, so two documents with identical overlap still
|
|
88
|
+
* get a stable, total order rather than an order that depends on sort stability.
|
|
89
|
+
*/
|
|
90
|
+
export declare function rerankScore(query: string, document: string): number;
|
|
91
|
+
/**
|
|
92
|
+
* Deterministically score `input` across `labels`. NOT the real classifier — the confidence is a
|
|
93
|
+
* hash-derived value per (input, label), normalized so the confidences SUM TO 1 the way a real
|
|
94
|
+
* single-label classification's do. The prediction is the argmax.
|
|
95
|
+
*/
|
|
96
|
+
export declare function classifyText(input: string, labels: string[]): {
|
|
97
|
+
prediction: string;
|
|
98
|
+
confidences: number[];
|
|
99
|
+
};
|
|
100
|
+
/**
|
|
101
|
+
* Segment text the way the twin's tokenizer does. NOT Cohere's BPE (vendoring the published
|
|
102
|
+
* tokenizer JSON is `cohere.tokenize.bpe_vocabulary`, a todo) — a deterministic word-with-trailing-whitespace
|
|
103
|
+
* split, chosen so that JOINING the segments restores the input EXACTLY, including runs of spaces
|
|
104
|
+
* and newlines. That exact-join property is what lets `detokenize(tokenize(t)) === t` hold, which
|
|
105
|
+
* is the property a caller actually depends on.
|
|
106
|
+
*/
|
|
107
|
+
export declare function segmentText(text: string): string[];
|
|
108
|
+
/**
|
|
109
|
+
* The PREFERRED token id for a segment — a pure function of the segment, so the same text
|
|
110
|
+
* tokenizes to the same ids in every process and every state root.
|
|
111
|
+
*
|
|
112
|
+
* This is only a preference, not the answer: two different segments can hash to the same value,
|
|
113
|
+
* and a vocabulary that silently let the second one overwrite the first would make `detokenize`
|
|
114
|
+
* return the WRONG text for the earlier one. The handler resolves that by probing upward over the
|
|
115
|
+
* vocabulary it has already observed (`cohere-twin.ts`, `assignToken`) — deterministic given the
|
|
116
|
+
* stored state, which is what a twin's determinism means. Ids start at 1: Cohere's token ids are
|
|
117
|
+
* positive, and 0 is reserved so an unset field can never read as a valid token.
|
|
118
|
+
*/
|
|
119
|
+
export declare function preferredTokenId(segment: string): number;
|