@kindgi/adapter-model-openai-compat 0.1.4-rc.5 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/provider.ts CHANGED
@@ -17,27 +17,12 @@ import type {
17
17
  ProviderMetadata,
18
18
  UsageCounters,
19
19
  } from '@kindgi/capabilities';
20
+ import { samplingFor } from '@kindgi/capabilities';
20
21
  import { createAttemptCounter } from '@kindgi/capabilities/attempts';
22
+ import { nameToolsAsSent } from '@kindgi/capabilities/tool-names';
21
23
 
22
- /**
23
- * OpenAI (and every downstream compat endpoint — Ollama, vLLM, Groq,
24
- * OpenRouter, Together, Fireworks, LiteLLM, ...) constrains
25
- * `function.name` to `^[a-zA-Z0-9_-]{1,128}$` — dots are rejected.
26
- * The framework's tool id convention is `<pack>.<tool>`, so this
27
- * adapter transparently encodes on send and decodes on receive.
28
- *
29
- * See the sibling comment in
30
- * `packages/adapters/model-anthropic/src/translate.ts` — same
31
- * substitution (`.` → `__`), same reversibility caveat (authors
32
- * should not put literal `__` in tool ids).
33
- */
34
- function encodeToolName(name: string): string {
35
- return name.replace(/\./g, '__');
36
- }
37
-
38
- function decodeToolName(name: string): string {
39
- return name.replace(/__/g, '.');
40
- }
24
+ import { EXTRA_BODY_RESERVED_RESPONSES, invokeResponses } from './responses.js';
25
+ import { computeCost, decodeToolName, encodeToolName } from './wire.js';
41
26
 
42
27
  /**
43
28
  * Configuration for an OpenAI-compatible ModelProvider. `baseURL` is the
@@ -79,17 +64,58 @@ export interface OpenAICompatProviderOptions {
79
64
  'apiKey' | 'baseURL'
80
65
  >;
81
66
  /**
82
- * Fields merged into every Chat Completions request body: settings an
83
- * endpoint takes that the OpenAI format has no field for. A Qwen
84
- * thinking model served by vLLM, SGLang or llama-server answers with its
85
- * thinking first unless asked not to:
86
- * `{ chat_template_kwargs: { enable_thinking: false } }`. The fields the
87
- * adapter sets itself (`EXTRA_BODY_RESERVED`) are refused.
67
+ * Which OpenAI API the adapter speaks (`OPENAI_COMPAT_APIS`):
68
+ * - `'responses'`: OpenAI's Responses API, the one OpenAI's GPT-6
69
+ * models call tools through. Stateless: every call sends
70
+ * `store: false`, so OpenAI keeps no conversation state.
71
+ * - `'chat-completions'`: Chat Completions, the format the
72
+ * OpenAI-compatible servers (Ollama, vLLM, Groq, OpenRouter, …) speak.
73
+ * Absent: `'responses'` for a `baseURL` on `api.openai.com` (or one of
74
+ * its data-residency hosts, `eu.api.openai.com`), `'chat-completions'`
75
+ * for any other (`defaultOpenAICompatApi`).
76
+ */
77
+ readonly api?: OpenAICompatApi;
78
+ /**
79
+ * Fields merged into every request body: settings an endpoint takes
80
+ * that the OpenAI format has no field for. A Qwen thinking model served
81
+ * by vLLM, SGLang or llama-server answers with its thinking first unless
82
+ * asked not to: `{ chat_template_kwargs: { enable_thinking: false } }`.
83
+ * The fields the adapter sets itself are refused: `EXTRA_BODY_RESERVED`
84
+ * on Chat Completions, `EXTRA_BODY_RESERVED_RESPONSES` on Responses.
88
85
  */
89
86
  readonly extraBody?: Readonly<Record<string, unknown>>;
90
87
  }
91
88
 
92
- /** Request fields the adapter sets itself; `extraBody` can't override them. */
89
+ /** The OpenAI APIs the adapter speaks (`OpenAICompatProviderOptions.api`). */
90
+ export const OPENAI_COMPAT_APIS = ['responses', 'chat-completions'] as const;
91
+ export type OpenAICompatApi = (typeof OPENAI_COMPAT_APIS)[number];
92
+
93
+ /** The API a provider speaks when it doesn't say (`OpenAICompatProviderOptions.api`). */
94
+ export function defaultOpenAICompatApi(baseURL: string): OpenAICompatApi {
95
+ // OpenAI's own API, its data-residency hosts (`eu.api.openai.com`) included.
96
+ return openAIHostOf(baseURL) !== undefined ? 'responses' : 'chat-completions';
97
+ }
98
+
99
+ /**
100
+ * Whether a base URL is one of OpenAI's data-residency hosts
101
+ * (`eu.api.openai.com`), where a model's `dataResidencyMultiplier` applies.
102
+ */
103
+ export function isDataResidencyHost(baseURL: string): boolean {
104
+ return openAIHostOf(baseURL) === 'data-residency';
105
+ }
106
+
107
+ function openAIHostOf(baseURL: string): 'global' | 'data-residency' | undefined {
108
+ let host: string;
109
+ try {
110
+ host = new URL(baseURL).hostname;
111
+ } catch {
112
+ return undefined;
113
+ }
114
+ if (host === 'api.openai.com') return 'global';
115
+ return host.endsWith('.api.openai.com') ? 'data-residency' : undefined;
116
+ }
117
+
118
+ /** Request fields the adapter sets itself on Chat Completions; `extraBody` can't override them. */
93
119
  export const EXTRA_BODY_RESERVED = [
94
120
  'model',
95
121
  'messages',
@@ -109,7 +135,15 @@ export function createOpenAICompatModelProvider(
109
135
  options: OpenAICompatProviderOptions,
110
136
  ): ModelProvider {
111
137
  const metadata = options.metadata;
112
- const problem = extraBodyProblem(options.extraBody);
138
+ if (options.api !== undefined && !OPENAI_COMPAT_APIS.includes(options.api)) {
139
+ // A caller outside TypeScript can pass anything.
140
+ throw new Error(
141
+ `${OPENAI_COMPAT_ADAPTER_ID}: provider "${metadata.id}": api must be one of ${OPENAI_COMPAT_APIS.join(', ')}`,
142
+ );
143
+ }
144
+ const api = options.api ?? defaultOpenAICompatApi(options.baseURL);
145
+ const dataResidency = isDataResidencyHost(options.baseURL);
146
+ const problem = extraBodyProblem(options.extraBody, api);
113
147
  if (problem !== undefined) {
114
148
  throw new Error(`${OPENAI_COMPAT_ADAPTER_ID}: provider "${metadata.id}": ${problem}`);
115
149
  }
@@ -146,8 +180,37 @@ export function createOpenAICompatModelProvider(
146
180
  }
147
181
  const startedAt = Date.now();
148
182
  const openai = await clientForCall();
183
+ // The system prompt names the call's tools as they're sent (`acme__lookup_order`):
184
+ // a model told to call `acme.lookup_order` calls a name it wasn't given.
185
+ const toolIds = input.tools?.map((t) => t.name) ?? [];
186
+ const sent = input.messages.map((m) =>
187
+ m.role === 'system'
188
+ ? { ...m, content: nameToolsAsSent(m.content, toolIds, encodeToolName) }
189
+ : m,
190
+ );
191
+ const sampling = samplingFor(modelInfo, input);
192
+ // A caller that wants as little thinking as the model allows (a judge).
193
+ const lowestThinking =
194
+ input.thinking === 'lowest' && modelInfo.thinking !== undefined
195
+ ? modelInfo.thinking.lowest
196
+ : undefined;
197
+ if (api === 'responses') {
198
+ return invokeResponses({
199
+ client: openai,
200
+ attempts,
201
+ input: { ...input, messages: sent },
202
+ modelInfo,
203
+ providerId: metadata.id,
204
+ extraBody,
205
+ dataResidency,
206
+ ...(sampling.temperature !== undefined && { temperature: sampling.temperature }),
207
+ ...(lowestThinking !== undefined && { reasoningEffort: lowestThinking }),
208
+ warnings: sampling.warnings,
209
+ startedAt,
210
+ });
211
+ }
149
212
 
150
- const messages = input.messages.map(toOpenAiMessage);
213
+ const messages = sent.map(toOpenAiMessage);
151
214
  const tools = input.tools?.map(toOpenAiTool);
152
215
 
153
216
  const responseFormat = input.structuredOutput
@@ -169,7 +232,12 @@ export function createOpenAICompatModelProvider(
169
232
  messages,
170
233
  ...(tools !== undefined && tools.length > 0 && { tools }),
171
234
  ...(responseFormat !== undefined && { response_format: responseFormat }),
172
- ...(input.temperature !== undefined && { temperature: input.temperature }),
235
+ ...(sampling.temperature !== undefined && { temperature: sampling.temperature }),
236
+ ...(lowestThinking !== undefined && {
237
+ reasoning_effort: lowestThinking as NonNullable<
238
+ OpenAI.Chat.Completions.ChatCompletionCreateParamsNonStreaming['reasoning_effort']
239
+ >,
240
+ }),
173
241
  ...(input.maxOutputTokens !== undefined && { max_tokens: input.maxOutputTokens }),
174
242
  stream: false,
175
243
  },
@@ -194,7 +262,7 @@ export function createOpenAICompatModelProvider(
194
262
  message: responseMessage,
195
263
  finishReason: mapFinishReason(choice?.finish_reason),
196
264
  usage,
197
- costUsd: computeCost(modelInfo, usage.promptTokens, usage.completionTokens),
265
+ costUsd: computeCost(modelInfo, usage, { dataResidency }),
198
266
  durationMs,
199
267
  provider: { id: metadata.id, model: input.model },
200
268
  ...(completion.model !== undefined &&
@@ -207,6 +275,7 @@ export function createOpenAICompatModelProvider(
207
275
  // An injected client sends with its own fetch: nothing was counted.
208
276
  ...(counted.attempts > 0 && { attempts: counted.attempts }),
209
277
  ...(completion.usage !== undefined && { rawUsage: { ...completion.usage } }),
278
+ ...(sampling.warnings.length > 0 && { warnings: sampling.warnings }),
210
279
  };
211
280
  },
212
281
  };
@@ -288,12 +357,6 @@ function mapFinishReason(reason: string | null | undefined): ModelCallResult['fi
288
357
  }
289
358
  }
290
359
 
291
- function computeCost(modelInfo: ModelInfo, promptTokens: number, completionTokens: number): number {
292
- const promptCost = (promptTokens / 1000) * modelInfo.cost.promptUsdPer1kTokens;
293
- const completionCost = (completionTokens / 1000) * modelInfo.cost.completionUsdPer1kTokens;
294
- return promptCost + completionCost;
295
- }
296
-
297
360
  /**
298
361
  * Well-known base URLs — surfaced as constants so callers can import
299
362
  * without typos. Adding a new alias here doesn't lock anyone in; the
@@ -329,12 +392,14 @@ const NO_KEY = 'unused';
329
392
  * a placeholder key, as local runners (Ollama, vLLM, llama-server)
330
393
  * expect. Its `extraBody.*` keys (`EXTRA_BODY_PREFIX`) are extra request
331
394
  * fields, merged into every request (`OpenAICompatProviderOptions.extraBody`).
395
+ * Its `adapter_config.api` picks the OpenAI API (`openAICompatApi`).
332
396
  */
333
397
  export const openAICompatAdapterFactory: AdapterFactory = (input) => {
334
398
  const extraBody = openAICompatExtraBody(input);
335
399
  return createOpenAICompatModelProvider({
336
400
  metadata: input.metadata,
337
401
  baseURL: openAICompatBaseUrl(input),
402
+ api: openAICompatApi(input),
338
403
  apiKey: input.resolveApiKey ?? NO_KEY,
339
404
  ...(extraBody !== undefined && { extraBody }),
340
405
  // The registration chose the endpoint: the runtime's fetch decides
@@ -354,6 +419,26 @@ export function openAICompatBaseUrl(input: AdapterFactoryInput): string {
354
419
  return value;
355
420
  }
356
421
 
422
+ /**
423
+ * The API a registration speaks: its `adapter_config.api` (`OPENAI_COMPAT_APIS`),
424
+ * or without one, the default for its base URL (`defaultOpenAICompatApi`).
425
+ * Throws, naming the key, on an API the adapter doesn't speak.
426
+ */
427
+ export function openAICompatApi(input: AdapterFactoryInput): OpenAICompatApi {
428
+ const value = input.config?.api;
429
+ if (value === undefined) {
430
+ const baseURL = input.config?.baseURL;
431
+ return defaultOpenAICompatApi(typeof baseURL === 'string' ? baseURL : '');
432
+ }
433
+ const api = OPENAI_COMPAT_APIS.find((known) => known === value);
434
+ if (api === undefined) {
435
+ throw new Error(
436
+ `${OPENAI_COMPAT_ADAPTER_ID}: provider "${input.metadata.id}": adapter_config.api must be one of ${OPENAI_COMPAT_APIS.join(', ')}.`,
437
+ );
438
+ }
439
+ return api;
440
+ }
441
+
357
442
  /**
358
443
  * The `adapter_config` key prefix for extra request fields. `adapter_config`
359
444
  * is flat, so each field is its own key and dots nest:
@@ -390,7 +475,7 @@ export function openAICompatExtraBody(
390
475
  const problem = setField(body, key.slice(EXTRA_BODY_PREFIX.length).split('.'), value);
391
476
  if (problem !== undefined) fail(`adapter_config "${key}" ${problem}`);
392
477
  }
393
- const problem = extraBodyProblem(body);
478
+ const problem = extraBodyProblem(body, openAICompatApi(input));
394
479
  if (problem !== undefined) fail(`adapter_config ${problem}`);
395
480
  return body;
396
481
  }
@@ -419,12 +504,13 @@ function setField(body: Record<string, unknown>, path: readonly string[], value:
419
504
  return undefined;
420
505
  }
421
506
 
422
- function extraBodyProblem(value: unknown): string | undefined {
507
+ function extraBodyProblem(value: unknown, api: OpenAICompatApi): string | undefined {
423
508
  if (value === undefined) return undefined;
424
509
  if (value === null || typeof value !== 'object' || Array.isArray(value)) {
425
510
  return 'extraBody must be an object of request fields, e.g. { chat_template_kwargs: { enable_thinking: false } }';
426
511
  }
427
- const reserved = EXTRA_BODY_RESERVED.filter((key) => Object.hasOwn(value, key));
512
+ const fields = api === 'responses' ? EXTRA_BODY_RESERVED_RESPONSES : EXTRA_BODY_RESERVED;
513
+ const reserved = fields.filter((key) => Object.hasOwn(value, key));
428
514
  return reserved.length > 0
429
515
  ? `extraBody can't set ${reserved.join(', ')}: the adapter sets ${reserved.length === 1 ? 'it' : 'them'}`
430
516
  : undefined;