@kindgi/adapter-model-openai-compat 0.1.4-rc.5 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,474 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+
4
+ /**
5
+ * The adapter's OpenAI Responses API path (`POST /v1/responses`). OpenAI's
6
+ * GPT-6 models call tools only here; Chat Completions stays the path for
7
+ * the OpenAI-compatible servers (`OpenAICompatProviderOptions.api`).
8
+ *
9
+ * Stateless: every call sends the whole conversation with `store: false`,
10
+ * so OpenAI keeps no conversation state for Kindgi's calls. A reasoning
11
+ * model's reasoning between its tool calls travels with the calls
12
+ * instead: the response's reasoning items (encrypted by OpenAI), and the
13
+ * ids and `phase` that tie them to the message and calls after them, go
14
+ * into the first call's `ModelToolCall.signature`. When the conversation
15
+ * continues with the same model, they're sent back in their order.
16
+ */
17
+
18
+ import type OpenAI from 'openai';
19
+
20
+ import type {
21
+ ModelCallInput,
22
+ ModelCallResult,
23
+ ModelCallWarning,
24
+ ModelInfo,
25
+ ModelMessage,
26
+ ModelToolCall,
27
+ ModelToolDefinition,
28
+ UsageCounters,
29
+ } from '@kindgi/capabilities';
30
+ import type { AttemptCounter } from '@kindgi/capabilities/attempts';
31
+
32
+ import { computeCost, decodeToolName, encodeToolName, parseArguments } from './wire.js';
33
+
34
+ type InputItem = OpenAI.Responses.ResponseInputItem;
35
+ type OutputItem = OpenAI.Responses.ResponseOutputItem;
36
+ type Phase = 'commentary' | 'final_answer';
37
+
38
+ /** Request fields the Responses path sets itself; `extraBody` can't override them. */
39
+ export const EXTRA_BODY_RESERVED_RESPONSES = [
40
+ 'model',
41
+ 'input',
42
+ 'tools',
43
+ 'text',
44
+ 'temperature',
45
+ 'max_output_tokens',
46
+ 'stream',
47
+ 'store',
48
+ ] as const;
49
+
50
+ /** One call on the Responses path, as `invoke` hands it over. */
51
+ export interface ResponsesCall {
52
+ readonly client: OpenAI;
53
+ readonly attempts: AttemptCounter;
54
+ readonly input: ModelCallInput;
55
+ readonly modelInfo: ModelInfo;
56
+ readonly providerId: string;
57
+ readonly extraBody: Readonly<Record<string, unknown>>;
58
+ /** The endpoint is a data-residency host: the model's uplift applies (`computeCost`). */
59
+ readonly dataResidency: boolean;
60
+ /** The temperature to send; absent sends none. */
61
+ readonly temperature?: number;
62
+ /**
63
+ * The reasoning effort to send (`reasoning.effort`): the model's lowest,
64
+ * when the call asks for as little thinking as it allows. Absent: the
65
+ * model's default, or a registration's `extraBody.reasoning`.
66
+ */
67
+ readonly reasoningEffort?: string;
68
+ /** Warnings about the call, for the answer's `warnings`. */
69
+ readonly warnings?: readonly ModelCallWarning[];
70
+ /** When the call started (`Date.now()`), for `durationMs`. */
71
+ readonly startedAt: number;
72
+ }
73
+
74
+ export async function invokeResponses(call: ResponsesCall): Promise<ModelCallResult> {
75
+ const { input } = call;
76
+ const tools = input.tools?.map(toTool);
77
+ const counted = await call.attempts.count(() =>
78
+ call.client.responses.create(
79
+ {
80
+ ...call.extraBody,
81
+ model: input.model,
82
+ input: toInput(input.messages, input.model),
83
+ ...(tools !== undefined && tools.length > 0 && { tools }),
84
+ ...(input.structuredOutput !== undefined && {
85
+ text: {
86
+ format: {
87
+ type: 'json_schema' as const,
88
+ name: input.structuredOutput.name,
89
+ schema: input.structuredOutput.schema,
90
+ strict: true,
91
+ },
92
+ },
93
+ }),
94
+ ...(call.temperature !== undefined && { temperature: call.temperature }),
95
+ ...(call.reasoningEffort !== undefined && {
96
+ reasoning: {
97
+ ...(isRecord(call.extraBody.reasoning) ? call.extraBody.reasoning : {}),
98
+ effort: call.reasoningEffort as OpenAI.ReasoningEffort,
99
+ },
100
+ }),
101
+ ...(input.maxOutputTokens !== undefined && { max_output_tokens: input.maxOutputTokens }),
102
+ store: false,
103
+ stream: false,
104
+ },
105
+ input.abortSignal !== undefined ? { signal: input.abortSignal } : undefined,
106
+ ),
107
+ );
108
+ const response = counted.value;
109
+ const durationMs = Date.now() - call.startedAt;
110
+ refuseUnfinished(response, call.providerId);
111
+
112
+ const output = readOutput(response.output, input.model);
113
+ const usage = toFrameworkUsage(response.usage);
114
+ const warnings = call.warnings ?? [];
115
+ return {
116
+ message: {
117
+ role: 'assistant',
118
+ content: output.content,
119
+ ...(output.toolCalls.length > 0 && { toolCalls: output.toolCalls }),
120
+ },
121
+ finishReason: finishReasonOf(response, output.toolCalls.length > 0),
122
+ usage,
123
+ costUsd: computeCost(call.modelInfo, usage, { dataResidency: call.dataResidency }),
124
+ durationMs,
125
+ provider: { id: call.providerId, model: input.model },
126
+ ...(typeof response.model === 'string' &&
127
+ response.model !== '' && { servedModel: response.model }),
128
+ ...(typeof response._request_id === 'string' && {
129
+ providerRequestId: response._request_id,
130
+ }),
131
+ // An injected client sends with its own fetch: nothing was counted.
132
+ ...(counted.attempts > 0 && { attempts: counted.attempts }),
133
+ ...(response.usage !== undefined && { rawUsage: { ...response.usage } }),
134
+ ...(warnings.length > 0 && { warnings: [...warnings] }),
135
+ };
136
+ }
137
+
138
+ // ---------------------------------------------------------------------------
139
+ // Request
140
+
141
+ /** The conversation as Responses input items, in order. */
142
+ function toInput(messages: readonly ModelMessage[], model: string): InputItem[] {
143
+ const items: InputItem[] = [];
144
+ for (const m of messages) {
145
+ switch (m.role) {
146
+ case 'system':
147
+ // The role OpenAI's Responses guides give an app's instructions.
148
+ items.push({ role: 'developer', content: m.content });
149
+ break;
150
+ case 'user':
151
+ items.push({ role: 'user', content: m.content });
152
+ break;
153
+ case 'tool':
154
+ items.push({
155
+ type: 'function_call_output',
156
+ call_id: m.toolCallId ?? '',
157
+ output: m.content,
158
+ });
159
+ break;
160
+ case 'assistant':
161
+ items.push(...assistantItems(m, model));
162
+ break;
163
+ }
164
+ }
165
+ return items;
166
+ }
167
+
168
+ /**
169
+ * An assistant turn. With tool calls this model made, and their carried
170
+ * output (`readCarried`), it goes back as the response gave it: reasoning,
171
+ * message and calls in their order, linked by their ids. Otherwise as its
172
+ * text and its calls, without ids: an item id names output the reasoning
173
+ * belongs with, so none is sent without it.
174
+ */
175
+ function assistantItems(m: ModelMessage, model: string): InputItem[] {
176
+ const calls = m.toolCalls ?? [];
177
+ const carried = readCarried(calls, model);
178
+ if (carried === undefined) {
179
+ return [
180
+ ...(m.content !== '' ? [{ role: 'assistant' as const, content: m.content }] : []),
181
+ ...calls.map((c) => functionCall(c, undefined)),
182
+ ];
183
+ }
184
+ // Item ids link reasoning to what follows it; without reasoning, none are sent.
185
+ const linked = carried.some((item) => item.type === 'reasoning');
186
+ const byId = new Map(calls.map((c) => [c.id, c] as const));
187
+ return carried.flatMap((item): InputItem[] => {
188
+ switch (item.type) {
189
+ case 'reasoning':
190
+ return [
191
+ {
192
+ type: 'reasoning',
193
+ id: item.id,
194
+ summary: item.summary.map((s) => ({ type: 'summary_text' as const, text: s.text })),
195
+ encrypted_content: item.encrypted_content,
196
+ },
197
+ ];
198
+ case 'message':
199
+ return m.content === '' ? [] : [carriedMessage(item, m.content, linked)];
200
+ case 'function_call':
201
+ // readCarried checked the turn has this call.
202
+ return [
203
+ functionCall(byId.get(item.call_id) as ModelToolCall, linked ? item.id : undefined),
204
+ ];
205
+ }
206
+ });
207
+ }
208
+
209
+ /** The turn's text, as the carried message it came in. */
210
+ function carriedMessage(
211
+ item: Extract<CarriedItem, { type: 'message' }>,
212
+ text: string,
213
+ linked: boolean,
214
+ ): InputItem {
215
+ const phase = item.phase !== undefined ? { phase: item.phase } : {};
216
+ return linked
217
+ ? {
218
+ type: 'message',
219
+ id: item.id,
220
+ role: 'assistant',
221
+ status: 'completed',
222
+ content: [{ type: 'output_text', text, annotations: [] }],
223
+ ...phase,
224
+ }
225
+ : { role: 'assistant', content: text, ...phase };
226
+ }
227
+
228
+ function functionCall(call: ModelToolCall, id: string | undefined): InputItem {
229
+ return {
230
+ type: 'function_call',
231
+ call_id: call.id,
232
+ // Encode when replaying prior assistant turns — see `encodeToolName`.
233
+ name: encodeToolName(call.name),
234
+ arguments: JSON.stringify(call.arguments),
235
+ ...(id !== undefined && { id }),
236
+ };
237
+ }
238
+
239
+ function toTool(tool: ModelToolDefinition): OpenAI.Responses.FunctionTool {
240
+ return {
241
+ type: 'function',
242
+ name: encodeToolName(tool.name),
243
+ description: tool.description,
244
+ parameters: tool.inputSchema as Record<string, unknown>,
245
+ // As on the Chat Completions path: tools' schemas needn't meet strict mode.
246
+ strict: false,
247
+ };
248
+ }
249
+
250
+ // ---------------------------------------------------------------------------
251
+ // Response
252
+
253
+ /** A `failed` (or never finished) response throws, naming OpenAI's error. */
254
+ function refuseUnfinished(response: OpenAI.Responses.Response, providerId: string): void {
255
+ const status = response.status ?? 'completed';
256
+ if (status === 'completed' || status === 'incomplete') return;
257
+ const error = response.error;
258
+ const why = error != null ? `${error.code}: ${error.message}` : `status "${status}"`;
259
+ throw new Error(
260
+ `@kindgi/adapter-model-openai-compat: provider "${providerId}": the response ${status === 'failed' ? 'failed' : `ended "${status}"`} (${why}).`,
261
+ );
262
+ }
263
+
264
+ interface Output {
265
+ readonly content: string;
266
+ readonly toolCalls: readonly ModelToolCall[];
267
+ }
268
+
269
+ /** What `readOutput` has read so far. */
270
+ interface Reading {
271
+ text: string;
272
+ refusal: string;
273
+ readonly calls: ModelToolCall[];
274
+ readonly carried: CarriedItem[];
275
+ reasoning: number;
276
+ messages: number;
277
+ phased: boolean;
278
+ /** Something the next call couldn't send back as it came. */
279
+ unlinkable: boolean;
280
+ }
281
+
282
+ /**
283
+ * The answer's text (its refusal when it has no text) and tool calls.
284
+ * The first call carries what the next call needs back.
285
+ */
286
+ function readOutput(output: readonly OutputItem[], model: string): Output {
287
+ const read: Reading = {
288
+ text: '',
289
+ refusal: '',
290
+ calls: [],
291
+ carried: [],
292
+ reasoning: 0,
293
+ messages: 0,
294
+ phased: false,
295
+ unlinkable: false,
296
+ };
297
+ for (const item of output) {
298
+ if (item.type === 'reasoning') readReasoning(item, read);
299
+ else if (item.type === 'message') readMessage(item, read);
300
+ else if (item.type === 'function_call') readFunctionCall(item, read);
301
+ }
302
+ // Carried when there's something to send back (reasoning, or a phase)
303
+ // and the next call can send it as it came: every reasoning item with
304
+ // its encrypted content, every call with its id, and at most one
305
+ // message (the turn keeps one text, so two can't be told apart).
306
+ const carry =
307
+ read.calls.length > 0 &&
308
+ (read.reasoning > 0 || read.phased) &&
309
+ !read.unlinkable &&
310
+ read.messages <= 1
311
+ ? writeSignature({ model, items: read.carried })
312
+ : undefined;
313
+ return {
314
+ content: read.text !== '' ? read.text : read.refusal,
315
+ toolCalls: read.calls.map((c, i) =>
316
+ i === 0 && carry !== undefined ? { ...c, signature: carry } : c,
317
+ ),
318
+ };
319
+ }
320
+
321
+ function readReasoning(item: OpenAI.Responses.ResponseReasoningItem, read: Reading): void {
322
+ read.reasoning += 1;
323
+ if (typeof item.encrypted_content !== 'string' || item.encrypted_content === '') {
324
+ read.unlinkable = true;
325
+ return;
326
+ }
327
+ read.carried.push({
328
+ type: 'reasoning',
329
+ id: item.id,
330
+ summary: item.summary.map((s) => ({ text: s.text })),
331
+ encrypted_content: item.encrypted_content,
332
+ });
333
+ }
334
+
335
+ function readMessage(item: OpenAI.Responses.ResponseOutputMessage, read: Reading): void {
336
+ read.messages += 1;
337
+ for (const part of item.content) {
338
+ if (part.type === 'output_text') read.text += part.text;
339
+ else if (part.type === 'refusal') read.refusal += part.refusal;
340
+ }
341
+ const phase = item.phase ?? undefined;
342
+ if (phase !== undefined) read.phased = true;
343
+ read.carried.push({ type: 'message', id: item.id, ...(phase !== undefined && { phase }) });
344
+ }
345
+
346
+ function readFunctionCall(item: OpenAI.Responses.ResponseFunctionToolCall, read: Reading): void {
347
+ // Decode: the wire returns the encoded name; the framework
348
+ // expects the canonical dotted form.
349
+ read.calls.push({
350
+ id: item.call_id,
351
+ name: decodeToolName(item.name),
352
+ arguments: parseArguments(item.arguments),
353
+ });
354
+ if (item.id === undefined) read.unlinkable = true;
355
+ else read.carried.push({ type: 'function_call', id: item.id, call_id: item.call_id });
356
+ }
357
+
358
+ function finishReasonOf(
359
+ response: OpenAI.Responses.Response,
360
+ hasToolCalls: boolean,
361
+ ): ModelCallResult['finishReason'] {
362
+ if (response.status === 'incomplete') {
363
+ switch (response.incomplete_details?.reason) {
364
+ case 'max_output_tokens':
365
+ return 'length';
366
+ case 'content_filter':
367
+ return 'content-filter';
368
+ }
369
+ }
370
+ return hasToolCalls ? 'tool-use' : 'stop';
371
+ }
372
+
373
+ /**
374
+ * The response's usage as the framework's counters. `input_tokens`
375
+ * include the cached ones (`input_tokens_details`), and `output_tokens`
376
+ * the reasoning ones (`output_tokens_details.reasoning_tokens`): reported
377
+ * apart when the response gives them.
378
+ */
379
+ function toFrameworkUsage(usage: OpenAI.Responses.ResponseUsage | undefined): UsageCounters {
380
+ const cacheRead = usage?.input_tokens_details?.cached_tokens;
381
+ const cacheWrite = usage?.input_tokens_details?.cache_write_tokens;
382
+ const reasoning = usage?.output_tokens_details?.reasoning_tokens;
383
+ return {
384
+ promptTokens: usage?.input_tokens ?? 0,
385
+ completionTokens: usage?.output_tokens ?? 0,
386
+ // A part the response reports is kept, 0 included.
387
+ ...(typeof cacheRead === 'number' && { cacheReadTokens: cacheRead }),
388
+ ...(typeof cacheWrite === 'number' && { cacheWriteTokens: cacheWrite }),
389
+ ...(typeof reasoning === 'number' && { reasoningTokens: reasoning }),
390
+ };
391
+ }
392
+
393
+ // ---------------------------------------------------------------------------
394
+ // The carried output (`ModelToolCall.signature`)
395
+
396
+ type CarriedItem =
397
+ | {
398
+ readonly type: 'reasoning';
399
+ readonly id: string;
400
+ readonly summary: readonly { readonly text: string }[];
401
+ readonly encrypted_content: string;
402
+ }
403
+ | { readonly type: 'message'; readonly id: string; readonly phase?: Phase }
404
+ | { readonly type: 'function_call'; readonly id: string; readonly call_id: string };
405
+
406
+ interface Carried {
407
+ /** The model that answered: only that model gets it back. */
408
+ readonly model: string;
409
+ /** The response's output, in order, without the text and arguments the turn keeps. */
410
+ readonly items: readonly CarriedItem[];
411
+ }
412
+
413
+ /** Versioned, so a later shape can tell this one apart. */
414
+ const SIGNATURE_PREFIX = 'oair1.';
415
+
416
+ function writeSignature(carried: Carried): string {
417
+ return SIGNATURE_PREFIX + Buffer.from(JSON.stringify(carried), 'utf8').toString('base64url');
418
+ }
419
+
420
+ /**
421
+ * The output carried on an assistant turn's first call, when it's this
422
+ * adapter's, for `model`, and names exactly the turn's calls in order.
423
+ * Anything else (another adapter's signature, another model's, one that
424
+ * doesn't parse) is ignored: the turn goes back without it.
425
+ */
426
+ function readCarried(
427
+ calls: readonly ModelToolCall[],
428
+ model: string,
429
+ ): readonly CarriedItem[] | undefined {
430
+ const signature = calls[0]?.signature;
431
+ if (signature === undefined || !signature.startsWith(SIGNATURE_PREFIX)) return undefined;
432
+ let parsed: unknown;
433
+ try {
434
+ parsed = JSON.parse(
435
+ Buffer.from(signature.slice(SIGNATURE_PREFIX.length), 'base64url').toString('utf8'),
436
+ );
437
+ } catch {
438
+ return undefined;
439
+ }
440
+ if (!isRecord(parsed) || parsed.model !== model || !Array.isArray(parsed.items)) {
441
+ return undefined;
442
+ }
443
+ const items = parsed.items as unknown[];
444
+ if (!items.every(isCarriedItem)) return undefined;
445
+ const callIds = items.flatMap((item) => (item.type === 'function_call' ? [item.call_id] : []));
446
+ if (callIds.length !== calls.length || callIds.some((id, i) => id !== calls[i]?.id)) {
447
+ return undefined;
448
+ }
449
+ return items;
450
+ }
451
+
452
+ function isCarriedItem(value: unknown): value is CarriedItem {
453
+ if (!isRecord(value) || typeof value.id !== 'string') return false;
454
+ switch (value.type) {
455
+ case 'reasoning':
456
+ return (
457
+ typeof value.encrypted_content === 'string' &&
458
+ Array.isArray(value.summary) &&
459
+ value.summary.every((s) => isRecord(s) && typeof s.text === 'string')
460
+ );
461
+ case 'message':
462
+ return (
463
+ value.phase === undefined || value.phase === 'commentary' || value.phase === 'final_answer'
464
+ );
465
+ case 'function_call':
466
+ return typeof value.call_id === 'string';
467
+ default:
468
+ return false;
469
+ }
470
+ }
471
+
472
+ function isRecord(value: unknown): value is Record<string, unknown> {
473
+ return typeof value === 'object' && value !== null && !Array.isArray(value);
474
+ }
package/src/wire.ts ADDED
@@ -0,0 +1,130 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Copyright (C) 2026 Kindgi Inc.
3
+
4
+ /**
5
+ * What both of the adapter's OpenAI APIs (Chat Completions and
6
+ * Responses) share: tool names on the wire, tool-call arguments, cost.
7
+ */
8
+
9
+ import type { ModelInfo, UsageCounters } from '@kindgi/capabilities';
10
+
11
+ /**
12
+ * OpenAI (and every downstream compat endpoint — Ollama, vLLM, Groq,
13
+ * OpenRouter, Together, Fireworks, LiteLLM, ...) constrains
14
+ * `function.name` to `^[a-zA-Z0-9_-]{1,128}$` — dots are rejected.
15
+ * The framework's tool id convention is `<pack>.<tool>`, so this
16
+ * adapter transparently encodes on send and decodes on receive.
17
+ *
18
+ * See the sibling comment in
19
+ * `packages/adapters/model-anthropic/src/translate.ts` — same
20
+ * substitution (`.` → `__`), same reversibility caveat (authors
21
+ * should not put literal `__` in tool ids).
22
+ */
23
+ export function encodeToolName(name: string): string {
24
+ return name.replace(/\./g, '__');
25
+ }
26
+
27
+ export function decodeToolName(name: string): string {
28
+ return name.replace(/__/g, '.');
29
+ }
30
+
31
+ /** A tool call's JSON arguments as an object; `{}` when they don't parse. */
32
+ export function parseArguments(raw: string): Record<string, unknown> {
33
+ try {
34
+ return JSON.parse(raw) as Record<string, unknown>;
35
+ } catch {
36
+ return {};
37
+ }
38
+ }
39
+
40
+ /**
41
+ * A model's rates, per 1K tokens, with the pricing extras an
42
+ * OpenAI-compatible endpoint may have. The names are the ones the Gemini
43
+ * and Anthropic adapters already price with.
44
+ */
45
+ export interface OpenAICompatCostRates {
46
+ readonly promptUsdPer1kTokens: number;
47
+ readonly completionUsdPer1kTokens: number;
48
+ /**
49
+ * Cached prompt tokens' share of the prompt rate (OpenAI's GPT-6: 0.1;
50
+ * GPT-6.1 Sol: 0.05). Absent: they bill at the prompt rate.
51
+ */
52
+ readonly cachedPromptMultiplier?: number;
53
+ /**
54
+ * Prompt tokens written to the cache, as a multiple of the prompt rate
55
+ * (OpenAI's GPT-6: 1.25). Absent: they bill at the prompt rate.
56
+ */
57
+ readonly promptCacheCreationMultiplier?: number;
58
+ /**
59
+ * Long-context pricing: when a call's prompt (its cached and cache-write
60
+ * tokens included) exceeds `thresholdTokens`, the whole call bills at
61
+ * these rates, the cache multipliers applying to the long prompt rate
62
+ * (OpenAI's GPT-6: twice the prompt and 1.5 times the completion rate
63
+ * past 272,000 input tokens).
64
+ */
65
+ readonly longContext?: {
66
+ readonly thresholdTokens: number;
67
+ readonly promptUsdPer1kTokens: number;
68
+ readonly completionUsdPer1kTokens: number;
69
+ };
70
+ /**
71
+ * The uplift on a whole call sent to a data-residency host
72
+ * (`eu.api.openai.com`: 1.1 for OpenAI models released on or after
73
+ * 2026-03-05). Applied only on such a host.
74
+ */
75
+ readonly dataResidencyMultiplier?: number;
76
+ }
77
+
78
+ /** `ModelInfo` whose `cost` takes the rates in `OpenAICompatCostRates`. */
79
+ export interface OpenAICompatModelInfo extends ModelInfo {
80
+ readonly cost: ModelInfo['cost'] & Omit<OpenAICompatCostRates, keyof ModelInfo['cost']>;
81
+ }
82
+
83
+ /**
84
+ * A call's cost from its usage and the model's rates: cached and
85
+ * cache-write prompt tokens at their multipliers, the whole call at the
86
+ * long-context rates past their threshold, and a data-residency host's
87
+ * uplift. A rate that isn't a finite, non-negative number is ignored (it
88
+ * came from a registration stored before rates were checked).
89
+ */
90
+ export function computeCost(
91
+ modelInfo: ModelInfo,
92
+ usage: UsageCounters,
93
+ options: { readonly dataResidency?: boolean } = {},
94
+ ): number {
95
+ const cost = modelInfo.cost as Partial<Record<keyof OpenAICompatCostRates, unknown>>;
96
+ const longContext = longContextOf(cost.longContext);
97
+ const long =
98
+ longContext !== undefined && usage.promptTokens > longContext.thresholdTokens
99
+ ? longContext
100
+ : undefined;
101
+ const promptRate = long?.promptUsdPer1kTokens ?? modelInfo.cost.promptUsdPer1kTokens;
102
+ const completionRate = long?.completionUsdPer1kTokens ?? modelInfo.cost.completionUsdPer1kTokens;
103
+ const cached = usage.cacheReadTokens ?? 0;
104
+ const written = usage.cacheWriteTokens ?? 0;
105
+ const fresh = Math.max(0, usage.promptTokens - cached - written);
106
+ const promptUnits =
107
+ fresh +
108
+ cached * (rateOf(cost.cachedPromptMultiplier) ?? 1) +
109
+ written * (rateOf(cost.promptCacheCreationMultiplier) ?? 1);
110
+ const total = (promptUnits * promptRate + usage.completionTokens * completionRate) / 1000;
111
+ const uplift = options.dataResidency === true ? rateOf(cost.dataResidencyMultiplier) : undefined;
112
+ return uplift !== undefined ? total * uplift : total;
113
+ }
114
+
115
+ function rateOf(value: unknown): number | undefined {
116
+ return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined;
117
+ }
118
+
119
+ function longContextOf(value: unknown): OpenAICompatCostRates['longContext'] {
120
+ if (typeof value !== 'object' || value === null) return undefined;
121
+ const tier = value as Record<string, unknown>;
122
+ const thresholdTokens = rateOf(tier.thresholdTokens);
123
+ const promptUsdPer1kTokens = rateOf(tier.promptUsdPer1kTokens);
124
+ const completionUsdPer1kTokens = rateOf(tier.completionUsdPer1kTokens);
125
+ return thresholdTokens !== undefined &&
126
+ promptUsdPer1kTokens !== undefined &&
127
+ completionUsdPer1kTokens !== undefined
128
+ ? { thresholdTokens, promptUsdPer1kTokens, completionUsdPer1kTokens }
129
+ : undefined;
130
+ }