@plurnk/plurnk-providers 1.3.4 → 1.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/.env.defaults +41 -52
  2. package/README.md +44 -53
  3. package/SPEC.md +215 -354
  4. package/dist/AiSdkProvider.d.ts +78 -0
  5. package/dist/AiSdkProvider.d.ts.map +1 -0
  6. package/dist/AiSdkProvider.js +591 -0
  7. package/dist/AiSdkProvider.js.map +1 -0
  8. package/dist/Mock.d.ts +1 -1
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +1 -1
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/OpenAICompat.d.ts +3 -5
  13. package/dist/OpenAICompat.d.ts.map +1 -1
  14. package/dist/OpenAICompat.js +44 -133
  15. package/dist/OpenAICompat.js.map +1 -1
  16. package/dist/Pool.d.ts +1 -1
  17. package/dist/Pool.d.ts.map +1 -1
  18. package/dist/Pool.js +1 -1
  19. package/dist/Pool.js.map +1 -1
  20. package/dist/ProviderRegistry.d.ts.map +1 -1
  21. package/dist/ProviderRegistry.js +37 -24
  22. package/dist/ProviderRegistry.js.map +1 -1
  23. package/dist/aiSdkTransport.d.ts +52 -0
  24. package/dist/aiSdkTransport.d.ts.map +1 -0
  25. package/dist/aiSdkTransport.js +294 -0
  26. package/dist/aiSdkTransport.js.map +1 -0
  27. package/dist/catalogProvider.d.ts +15 -0
  28. package/dist/catalogProvider.d.ts.map +1 -0
  29. package/dist/catalogProvider.js +103 -0
  30. package/dist/catalogProvider.js.map +1 -0
  31. package/dist/compatibleProvider.d.ts +3 -0
  32. package/dist/compatibleProvider.d.ts.map +1 -0
  33. package/dist/compatibleProvider.js +146 -0
  34. package/dist/compatibleProvider.js.map +1 -0
  35. package/dist/discover.d.ts.map +1 -1
  36. package/dist/discover.js.map +1 -1
  37. package/dist/env.d.ts +1 -0
  38. package/dist/env.d.ts.map +1 -1
  39. package/dist/env.js +13 -6
  40. package/dist/env.js.map +1 -1
  41. package/dist/index.d.ts +5 -7
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +4 -8
  44. package/dist/index.js.map +1 -1
  45. package/dist/ollama.d.ts +3 -0
  46. package/dist/ollama.d.ts.map +1 -0
  47. package/dist/ollama.js +39 -0
  48. package/dist/ollama.js.map +1 -0
  49. package/dist/openai.d.ts +2 -4
  50. package/dist/openai.d.ts.map +1 -1
  51. package/dist/openai.js +1 -2
  52. package/dist/openai.js.map +1 -1
  53. package/dist/sdkModels.d.ts +13 -0
  54. package/dist/sdkModels.d.ts.map +1 -0
  55. package/dist/sdkModels.js +153 -0
  56. package/dist/sdkModels.js.map +1 -0
  57. package/dist/standardProviders.d.ts +0 -1
  58. package/dist/standardProviders.d.ts.map +1 -1
  59. package/dist/standardProviders.js +9 -11
  60. package/dist/standardProviders.js.map +1 -1
  61. package/dist/telemetry.d.ts.map +1 -1
  62. package/dist/telemetry.js +20 -9
  63. package/dist/telemetry.js.map +1 -1
  64. package/dist/types.d.ts +4 -3
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +1 -1
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +4 -2
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +19 -8
  71. package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +202 -177
  72. package/src/{OpenAICompat.ts → AiSdkProvider.ts} +89 -144
  73. package/src/Mock.test.ts +3 -3
  74. package/src/Mock.ts +1 -1
  75. package/src/Pool.test.ts +3 -3
  76. package/src/Pool.ts +1 -1
  77. package/src/ProviderRegistry.test.ts +40 -27
  78. package/src/ProviderRegistry.ts +35 -24
  79. package/src/aiSdkTransport.test.ts +253 -0
  80. package/src/aiSdkTransport.ts +369 -0
  81. package/src/boundaries.test.ts +2 -2
  82. package/src/catalogProvider.test.ts +100 -0
  83. package/src/catalogProvider.ts +151 -0
  84. package/src/compatibleProvider.test.ts +44 -0
  85. package/src/compatibleProvider.ts +205 -0
  86. package/src/discover.test.ts +12 -12
  87. package/src/discover.ts +3 -6
  88. package/src/env.ts +14 -6
  89. package/src/index.ts +7 -11
  90. package/src/ollama.ts +63 -0
  91. package/src/openai.ts +2 -8
  92. package/src/sdkModels.test.ts +47 -0
  93. package/src/sdkModels.ts +194 -0
  94. package/src/telemetry.test.ts +17 -10
  95. package/src/telemetry.ts +22 -14
  96. package/src/types.ts +10 -13
  97. package/src/usage.test.ts +8 -10
  98. package/src/usage.ts +7 -3
  99. package/src/openaiStream.ts +0 -310
  100. package/src/standardProviders.test.ts +0 -949
  101. package/src/standardProviders.ts +0 -635
@@ -0,0 +1,369 @@
1
+ import { createOpenAICompatible, type ProviderErrorStructure } from "@ai-sdk/openai-compatible";
2
+ import { generateText, streamText, type JSONValue, type LanguageModel, type LanguageModelUsage } from "ai";
3
+ import { z } from "zod/v4";
4
+ import type { ChatMessage, FinishReason, ProviderUsage, TokenLogprob } from "./types.ts";
5
+ import { normalizeUsage, type RawUsage } from "./usage.ts";
6
+ import { emitWarningOnce } from "./warnings.ts";
7
+
8
+ const errorSchema = z.object({
9
+ error: z.object({
10
+ message: z.string(),
11
+ type: z.string().nullish(),
12
+ param: z.unknown().nullish(),
13
+ code: z.union([z.string(), z.number()]).nullish(),
14
+ }).passthrough(),
15
+ }).passthrough();
16
+
17
+ const errorStructure: ProviderErrorStructure<z.infer<typeof errorSchema>> = {
18
+ errorSchema,
19
+ errorToMessage: ({ error }) => error.message,
20
+ isRetryable(response) {
21
+ const directive = response.headers.get("x-should-retry")?.trim().toLowerCase();
22
+ if (directive === "false") return false;
23
+ if (directive === "true") return true;
24
+ if (response.status >= 520 && response.status <= 527) return false;
25
+ return response.status === 408
26
+ || response.status === 409
27
+ || response.status === 429
28
+ || response.status >= 500;
29
+ },
30
+ };
31
+
32
+ const baseUrl = (completionUrl: string): string => {
33
+ const url = new URL(completionUrl);
34
+ if (!url.pathname.endsWith("/chat/completions")) {
35
+ throw new Error(`OpenAI-compatible URL must end in /chat/completions: ${completionUrl}`);
36
+ }
37
+ url.pathname = url.pathname.slice(0, -"/chat/completions".length);
38
+ return url.toString().replace(/\/$/, "");
39
+ };
40
+
41
+ const usageOf = (
42
+ usage: LanguageModelUsage,
43
+ reasoningText: string,
44
+ contentText: string,
45
+ ): ProviderUsage => normalizeUsage({
46
+ prompt_tokens: usage.inputTokens,
47
+ completion_tokens: usage.outputTokens,
48
+ total_tokens: usage.totalTokens,
49
+ prompt_tokens_details: { cached_tokens: usage.inputTokenDetails.cacheReadTokens },
50
+ completion_tokens_details: usage.outputTokenDetails.reasoningTokens !== undefined
51
+ ? { reasoning_tokens: usage.outputTokenDetails.reasoningTokens }
52
+ : undefined,
53
+ }, reasoningText, contentText);
54
+
55
+ const wireUsageOf = (
56
+ values: readonly unknown[],
57
+ reasoningText: string,
58
+ contentText: string,
59
+ ): ProviderUsage | null => {
60
+ for (let index = values.length - 1; index >= 0; index -= 1) {
61
+ const usage = recordOf(values[index])?.usage;
62
+ if (usage !== null && typeof usage === "object") {
63
+ return normalizeUsage(usage as RawUsage, reasoningText, contentText);
64
+ }
65
+ }
66
+ return null;
67
+ };
68
+
69
+ const finishReasonOf = (reason: string | undefined): FinishReason => {
70
+ switch (reason?.toLowerCase()) {
71
+ case "stop":
72
+ case "end_turn":
73
+ case "stop_sequence":
74
+ case "eos_token":
75
+ return "stop";
76
+ case "length":
77
+ case "max_tokens":
78
+ case "model_length":
79
+ case "max_completion_tokens":
80
+ return "length";
81
+ case "tool_calls":
82
+ case "tool_use":
83
+ return "tool_calls";
84
+ case "content_filter":
85
+ case "safety":
86
+ case "recitation":
87
+ return "content_filter";
88
+ default:
89
+ if (reason !== undefined && reason.length > 0) {
90
+ emitWarningOnce(
91
+ `unrecognized finish_reason "${reason}"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core's length-cap detection will miss it.`,
92
+ "PLURNK_FINISH_REASON_UNKNOWN",
93
+ );
94
+ }
95
+ return null;
96
+ }
97
+ };
98
+
99
+ const recordOf = (value: unknown): Record<string, unknown> | null =>
100
+ value !== null && typeof value === "object"
101
+ ? value as Record<string, unknown>
102
+ : null;
103
+
104
+ const metadataOf = (values: readonly unknown[]): Record<string, unknown> => {
105
+ const metadata: Record<string, unknown> = {};
106
+ for (const value of values) {
107
+ const record = recordOf(value);
108
+ if (record === null) continue;
109
+ for (const [key, item] of Object.entries(record)) {
110
+ if (key !== "choices" && key !== "usage") metadata[key] = item;
111
+ }
112
+ }
113
+ return metadata;
114
+ };
115
+
116
+ export type AiSdkTransportRequest = {
117
+ url: string;
118
+ model: string;
119
+ headers: Record<string, string>;
120
+ body: Record<string, unknown>;
121
+ messages: ChatMessage[];
122
+ signal?: AbortSignal;
123
+ fetch?: typeof globalThis.fetch;
124
+ fetchTimeoutMs: number;
125
+ streamIdleTimeoutMs?: number;
126
+ retryAttempts: number;
127
+ streaming: boolean;
128
+ captureRawBody: boolean;
129
+ };
130
+
131
+ export type AiSdkTransportResponse = {
132
+ model: string;
133
+ content: string;
134
+ reasoning: string;
135
+ finishReason: FinishReason;
136
+ usage: ProviderUsage;
137
+ metadata: Record<string, unknown>;
138
+ reasoningEncrypted: Array<{
139
+ id: string | null;
140
+ subtype: string;
141
+ encrypted: Array<{ data: string; format: string | null }>;
142
+ }>;
143
+ logprobs: TokenLogprob[];
144
+ rawBody?: unknown;
145
+ };
146
+
147
+ export type AiSdkModelRequest = Omit<AiSdkTransportRequest, "url" | "model" | "body" | "fetch"> & {
148
+ languageModel: LanguageModel;
149
+ providerOptions?: Record<string, Record<string, JSONValue | undefined>>;
150
+ temperature?: number;
151
+ topP?: number;
152
+ topK?: number;
153
+ presencePenalty?: number;
154
+ frequencyPenalty?: number;
155
+ stopSequences?: string[];
156
+ seed?: number;
157
+ maxOutputTokens?: number;
158
+ reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" | "none" | "provider-default";
159
+ };
160
+
161
+ const executeModel = async (
162
+ request: AiSdkModelRequest,
163
+ ): Promise<AiSdkTransportResponse> => {
164
+ const {
165
+ languageModel: model,
166
+ providerOptions,
167
+ temperature,
168
+ topP,
169
+ topK,
170
+ presencePenalty,
171
+ frequencyPenalty,
172
+ stopSequences,
173
+ seed,
174
+ maxOutputTokens,
175
+ reasoning,
176
+ } = request;
177
+ const settings = {
178
+ ...(temperature === undefined ? {} : { temperature }),
179
+ ...(topP === undefined ? {} : { topP }),
180
+ ...(topK === undefined ? {} : { topK }),
181
+ ...(presencePenalty === undefined ? {} : { presencePenalty }),
182
+ ...(frequencyPenalty === undefined ? {} : { frequencyPenalty }),
183
+ ...(stopSequences === undefined ? {} : { stopSequences }),
184
+ ...(seed === undefined ? {} : { seed }),
185
+ ...(maxOutputTokens === undefined ? {} : { maxOutputTokens }),
186
+ ...(reasoning === undefined ? {} : { reasoning }),
187
+ ...(providerOptions === undefined ? {} : { providerOptions }),
188
+ };
189
+ const firstNonSystem = request.messages.findIndex((message) => message.role !== "system");
190
+ const instructionCount = firstNonSystem === -1 ? request.messages.length : firstNonSystem;
191
+ if (request.messages.slice(instructionCount).some((message) => message.role === "system")) {
192
+ throw new Error("provider messages: system instructions must precede conversational messages");
193
+ }
194
+ const instructions = request.messages.slice(0, instructionCount).map(({ content }) => ({
195
+ role: "system" as const,
196
+ content,
197
+ }));
198
+ const messages = request.messages.slice(instructionCount);
199
+ const common = {
200
+ model,
201
+ ...(instructions.length === 0 ? {} : { instructions }),
202
+ messages: messages.length > 0
203
+ ? messages
204
+ : [{ role: "user" as const, content: "" }],
205
+ maxRetries: request.retryAttempts,
206
+ abortSignal: request.signal,
207
+ headers: request.headers,
208
+ timeout: {
209
+ totalMs: request.fetchTimeoutMs,
210
+ ...(request.streamIdleTimeoutMs !== undefined && request.streamIdleTimeoutMs > 0
211
+ ? { chunkMs: request.streamIdleTimeoutMs }
212
+ : {}),
213
+ },
214
+ ...settings,
215
+ } as const;
216
+
217
+ if (!request.streaming) {
218
+ const result = await generateText({
219
+ ...common,
220
+ include: { responseBody: true },
221
+ });
222
+ const rawBody = result.response.body;
223
+ const values = [rawBody];
224
+ const evidence = extractEvidence(values);
225
+ const reasoningText = evidence.reasoning || result.reasoningText || "";
226
+ return {
227
+ model: result.response.modelId,
228
+ content: result.text,
229
+ reasoning: reasoningText,
230
+ finishReason: finishReasonOf(result.rawFinishReason),
231
+ usage: wireUsageOf(values, reasoningText, result.text)
232
+ ?? usageOf(result.usage, reasoningText, result.text),
233
+ metadata: metadataOf(values),
234
+ reasoningEncrypted: evidence.reasoningEncrypted,
235
+ logprobs: evidence.logprobs,
236
+ ...(request.captureRawBody ? { rawBody } : {}),
237
+ };
238
+ }
239
+
240
+ const result = streamText({
241
+ ...common,
242
+ includeRawChunks: true,
243
+ onError: () => {},
244
+ });
245
+ const rawChunks: unknown[] = [];
246
+ let streamError: unknown;
247
+ for await (const part of result.fullStream) {
248
+ if (part.type === "raw") rawChunks.push(part.rawValue);
249
+ if (part.type === "error") streamError ??= part.error;
250
+ }
251
+ if (streamError !== undefined) throw streamError;
252
+ const evidence = extractEvidence(rawChunks);
253
+ const content = await result.text;
254
+ const reasoningText = evidence.reasoning || (await result.reasoningText) || "";
255
+ return {
256
+ model: (await result.response).modelId,
257
+ content,
258
+ reasoning: reasoningText,
259
+ finishReason: finishReasonOf(await result.rawFinishReason),
260
+ usage: wireUsageOf(rawChunks, reasoningText, content)
261
+ ?? usageOf(await result.usage, reasoningText, content),
262
+ metadata: metadataOf(rawChunks),
263
+ reasoningEncrypted: evidence.reasoningEncrypted,
264
+ logprobs: evidence.logprobs,
265
+ ...(request.captureRawBody ? { rawBody: rawChunks } : {}),
266
+ };
267
+ };
268
+
269
+ export const executeAiSdkModel = executeModel;
270
+
271
+ export const executeOpenAICompatible = async (
272
+ request: AiSdkTransportRequest,
273
+ ): Promise<AiSdkTransportResponse> => {
274
+ const provider = createOpenAICompatible({
275
+ name: "plurnk",
276
+ baseURL: baseUrl(request.url),
277
+ headers: request.headers,
278
+ fetch: request.fetch,
279
+ includeUsage: true,
280
+ transformRequestBody: (sdkBody) => ({
281
+ ...sdkBody,
282
+ ...request.body,
283
+ stream: sdkBody.stream,
284
+ ...(sdkBody.stream_options !== undefined
285
+ ? { stream_options: sdkBody.stream_options }
286
+ : {}),
287
+ }),
288
+ });
289
+ const model = provider.languageModel(request.model, { errorStructure });
290
+ return executeModel({
291
+ languageModel: model,
292
+ headers: {},
293
+ messages: request.messages,
294
+ signal: request.signal,
295
+ fetchTimeoutMs: request.fetchTimeoutMs,
296
+ streamIdleTimeoutMs: request.streamIdleTimeoutMs,
297
+ retryAttempts: request.retryAttempts,
298
+ streaming: request.streaming,
299
+ captureRawBody: request.captureRawBody,
300
+ });
301
+ };
302
+
303
+ const extractEvidence = (values: unknown[]): {
304
+ reasoningEncrypted: AiSdkTransportResponse["reasoningEncrypted"];
305
+ logprobs: TokenLogprob[];
306
+ reasoning: string;
307
+ } => {
308
+ const encrypted = new Map<string, AiSdkTransportResponse["reasoningEncrypted"][number]>();
309
+ const logprobs: TokenLogprob[] = [];
310
+ let reasoning = "";
311
+ let anonymous = 0;
312
+ for (const value of values) {
313
+ const choices = recordOf(value)?.choices;
314
+ if (!Array.isArray(choices)) continue;
315
+ const choice = recordOf(choices[0]);
316
+ if (choice === null) continue;
317
+ const logprobRecord = recordOf(choice.logprobs);
318
+ const entries = logprobRecord?.content;
319
+ if (Array.isArray(entries)) {
320
+ for (const value of entries) {
321
+ const entry = recordOf(value);
322
+ if (typeof entry?.token !== "string" || typeof entry.logprob !== "number") continue;
323
+ const top = Array.isArray(entry.top_logprobs)
324
+ ? entry.top_logprobs.flatMap((value) => {
325
+ const item = recordOf(value);
326
+ return typeof item?.token === "string" && typeof item.logprob === "number"
327
+ ? [{ token: item.token, logprob: item.logprob }]
328
+ : [];
329
+ })
330
+ : undefined;
331
+ logprobs.push(top === undefined
332
+ ? { token: entry.token, logprob: entry.logprob }
333
+ : { token: entry.token, logprob: entry.logprob, top });
334
+ }
335
+ }
336
+ const message = recordOf(choice.delta) ?? recordOf(choice.message) ?? {};
337
+ for (const key of ["reasoning_content", "reasoning", "thinking"]) { // lexicon-allow: backend wire fields
338
+ if (typeof message[key] === "string") reasoning += message[key];
339
+ }
340
+ if (!Array.isArray(message.reasoning_details)) continue;
341
+ for (const value of message.reasoning_details) {
342
+ const detail = recordOf(value);
343
+ if (detail?.type !== "reasoning.encrypted" || typeof detail.data !== "string") continue;
344
+ const id = typeof detail.id === "string" ? detail.id : null;
345
+ const key = typeof detail.index === "number"
346
+ ? `index:${detail.index}`
347
+ : id === null ? `anonymous:${anonymous++}` : `id:${id}`;
348
+ const item: AiSdkTransportResponse["reasoningEncrypted"][number] = encrypted.get(key) ?? {
349
+ id,
350
+ subtype: "message",
351
+ encrypted: [],
352
+ };
353
+ const format = typeof detail.format === "string" ? detail.format : null;
354
+ const prior = item.encrypted.at(-1);
355
+ if (prior !== undefined) {
356
+ prior.data += detail.data;
357
+ if (prior.format === null && format !== null) prior.format = format;
358
+ } else {
359
+ item.encrypted.push({ data: detail.data, format });
360
+ }
361
+ encrypted.set(key, item);
362
+ }
363
+ }
364
+ return {
365
+ reasoningEncrypted: [...encrypted.values()],
366
+ logprobs,
367
+ reasoning,
368
+ };
369
+ };
@@ -25,10 +25,10 @@ test("provider source does not import the PLURNK parser", () => {
25
25
 
26
26
  test("#608: the OpenAI-compatible entrypoint excludes Node-owned provider machinery", () => {
27
27
  const allowed = new Set([
28
- "OpenAICompat.ts",
28
+ "AiSdkProvider.ts",
29
+ "aiSdkTransport.ts",
29
30
  "env.ts",
30
31
  "openai.ts",
31
- "openaiStream.ts",
32
32
  "telemetry.ts",
33
33
  "types.ts",
34
34
  "usage.ts",
@@ -0,0 +1,100 @@
1
+ import test, { mock } from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { catalogProviderFromEnv } from "./catalogProvider.ts";
4
+ import { resetEmittedWarnings } from "./warnings.ts";
5
+
6
+ const env = {
7
+ OPENAI_API_KEY: "test-key",
8
+ OPENAI_BASE_URL: "https://api.openai.com/v1",
9
+ PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
10
+ PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
11
+ PLURNK_PROVIDERS_REASONING: "off",
12
+ PLURNK_PROVIDERS_TEMPERATURE: "0.2",
13
+ PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
14
+ PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
15
+ PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
16
+ PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
17
+ PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
18
+ PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
19
+ };
20
+
21
+ test.afterEach(() => {
22
+ mock.restoreAll();
23
+ resetEmittedWarnings();
24
+ });
25
+
26
+ test("catalog provider resolves model physics and Models.dev USD rates", () => {
27
+ const provider = catalogProviderFromEnv("openai", env, "gpt-4.1-mini");
28
+ assert.notEqual(provider, null);
29
+ assert.equal(provider?.model, "gpt-4.1-mini");
30
+ assert.equal(provider?.contextWindow, 1_047_576);
31
+ assert.equal(provider?.reasoningReserve, 16_384);
32
+ assert.equal(provider?.completionReserve, 32_768);
33
+ assert.ok((provider?.calculateCost({
34
+ prompt: 1_000_000,
35
+ completion: 1_000_000,
36
+ reasoning: 0,
37
+ cached: 0,
38
+ total: 2_000_000,
39
+ }) ?? 0) > 0);
40
+ });
41
+
42
+ test("official AI SDK provider owns the native request while PLURNK owns call settings", async () => {
43
+ const calls: Array<{ url: string; body: Record<string, unknown> }> = [];
44
+ mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
45
+ calls.push({
46
+ url: String(input),
47
+ body: JSON.parse(String(init?.body)) as Record<string, unknown>,
48
+ });
49
+ const chunks = [
50
+ `data: ${JSON.stringify({
51
+ id: "chatcmpl-test",
52
+ object: "chat.completion.chunk",
53
+ created: 1,
54
+ model: "gpt-4.1-mini",
55
+ choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
56
+ })}`,
57
+ `data: ${JSON.stringify({
58
+ id: "chatcmpl-test",
59
+ object: "chat.completion.chunk",
60
+ created: 2,
61
+ model: "gpt-4.1-mini",
62
+ choices: [],
63
+ usage: { prompt_tokens: 2, completion_tokens: 1, total_tokens: 3 },
64
+ })}`,
65
+ "data: [DONE]",
66
+ ].join("\n\n");
67
+ return new Response(chunks, {
68
+ status: 200,
69
+ headers: { "content-type": "text/event-stream" },
70
+ });
71
+ });
72
+
73
+ const provider = catalogProviderFromEnv("openai", env, "gpt-4.1-mini");
74
+ const result = await provider?.generate({
75
+ workerId: "worker",
76
+ messages: [{ role: "user", content: "hello" }],
77
+ maxTokens: 64,
78
+ sampling: { top_p: 0.8, seed: 7 },
79
+ });
80
+
81
+ assert.equal(result?.assistant.content, "done");
82
+ assert.equal(result?.assistant.usage.total, 3);
83
+ assert.equal(calls.length, 1);
84
+ assert.equal(calls[0]?.url, "https://api.openai.com/v1/chat/completions");
85
+ assert.equal(calls[0]?.body.model, "gpt-4.1-mini");
86
+ assert.equal(calls[0]?.body.temperature, 0.2);
87
+ assert.equal(calls[0]?.body.top_p, 0.8);
88
+ assert.equal(calls[0]?.body.seed, 7);
89
+ assert.equal(calls[0]?.body.max_tokens, 64);
90
+ });
91
+
92
+ test("cataloged unknown model fails unless its context is explicit", () => {
93
+ assert.equal(catalogProviderFromEnv("xai", env, "not-in-the-catalog"), null);
94
+ const provider = catalogProviderFromEnv("xai", {
95
+ ...env,
96
+ XAI_API_KEY: "test-key",
97
+ PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
98
+ }, "not-in-the-catalog");
99
+ assert.equal(provider?.contextWindow, 8192);
100
+ });
@@ -0,0 +1,151 @@
1
+ import { lookupProvider, resolveModel, type ModelInfo } from "@plurnk/plurnk-models";
2
+ import {
3
+ contextWindowFromEnv,
4
+ dataCaptureFromEnv,
5
+ envelopeFromEnv,
6
+ parseRequiredFloat,
7
+ parseRequiredInt,
8
+ promptCacheKeyFromEnv,
9
+ reasoningFromEnv,
10
+ resolveReserve,
11
+ type ReserveSpec,
12
+ } from "./env.ts";
13
+ import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
14
+ import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
15
+ import { providerSource } from "./telemetry.ts";
16
+ import type { Provider, ProviderUsage } from "./types.ts";
17
+ import { calculateCostUsd } from "./usage.ts";
18
+ import { emitWarningOnce } from "./warnings.ts";
19
+ import type { LanguageModel } from "ai";
20
+
21
+ const reasoningStyleFromEnv = (
22
+ env: NodeJS.ProcessEnv,
23
+ name: string,
24
+ ): ReasoningStyle | undefined => {
25
+ const prefix = name.replaceAll(/[^a-zA-Z0-9]/g, "_").toUpperCase();
26
+ const value = env[`PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_STYLE`];
27
+ if (value === undefined || value.length === 0) return undefined;
28
+ const styles: readonly ReasoningStyle[] = [
29
+ "none", "think", "include_reasoning", "effort",
30
+ "effort_explicit", "template", "anthropic",
31
+ ];
32
+ if (!styles.includes(value as ReasoningStyle)) {
33
+ throw new Error(`${name} provider: PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_STYLE has invalid value "${value}"`);
34
+ }
35
+ return value as ReasoningStyle;
36
+ };
37
+
38
+ export const providerFromSdkModel = ({
39
+ name,
40
+ env,
41
+ model,
42
+ languageModel,
43
+ url,
44
+ headers,
45
+ contextWindow,
46
+ info,
47
+ }: {
48
+ name: string;
49
+ env: NodeJS.ProcessEnv;
50
+ model: string;
51
+ languageModel?: LanguageModel;
52
+ url?: string;
53
+ headers?: Readonly<Record<string, string>>;
54
+ contextWindow: number;
55
+ info?: ModelInfo;
56
+ }): Provider => {
57
+ emitWarningOnce(
58
+ `${name} provider: countTokens is a chars/2 upper bound — exact counts come from the mimetypes tokenizer seam or tokenize()`,
59
+ "PLURNK_TOKENIZER_HEURISTIC",
60
+ );
61
+
62
+ const reasoning = reasoningFromEnv(env, name);
63
+ const { reasoningReserve: configuredReasoning, completionReserve: configuredCompletion } = envelopeFromEnv(env, name);
64
+ const completionReserve: ReserveSpec = "tokens" in configuredCompletion
65
+ ? configuredCompletion
66
+ : info?.maxOutput === undefined
67
+ ? configuredCompletion
68
+ : { tokens: Math.min(info.maxOutput, Math.round(configuredCompletion.percent * contextWindow)) };
69
+ const completionTokens = resolveReserve(completionReserve, contextWindow);
70
+ const reasoningReserve: ReserveSpec = "tokens" in configuredReasoning
71
+ ? configuredReasoning
72
+ : reasoning.budget !== null
73
+ ? { tokens: reasoning.budget }
74
+ : completionTokens === null
75
+ ? configuredReasoning
76
+ : { tokens: Math.round(completionTokens / 2) };
77
+
78
+ const cost = info?.cost;
79
+ const calculateCost = cost === undefined
80
+ ? undefined
81
+ : (usage: ProviderUsage): number => calculateCostUsd(usage, {
82
+ input: cost.inputPer1M,
83
+ output: cost.outputPer1M,
84
+ cached: cost.cacheReadPer1M ?? cost.inputPer1M,
85
+ });
86
+
87
+ return new AiSdkProvider({
88
+ model,
89
+ ...(languageModel === undefined ? {} : { languageModel }),
90
+ ...(url === undefined ? {} : { url }),
91
+ ...(headers === undefined ? {} : { headers: { ...headers } }),
92
+ contextWindow,
93
+ fetchTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
94
+ streamIdleTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
95
+ reasoning,
96
+ temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
97
+ repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
98
+ frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
99
+ reasoningReserve,
100
+ completionReserve,
101
+ retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
102
+ reasoningStyle: reasoningStyleFromEnv(env, name),
103
+ promptCacheKey: url === undefined ? false : promptCacheKeyFromEnv(env, name),
104
+ serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
105
+ calculateCost,
106
+ source: providerSource(name),
107
+ gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
108
+ && env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
109
+ && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
110
+ ...dataCaptureFromEnv(env, name),
111
+ });
112
+ };
113
+
114
+ export const catalogProviderFromEnv = (
115
+ name: string,
116
+ env: NodeJS.ProcessEnv,
117
+ model: string,
118
+ baseUrlOverride?: string,
119
+ ): Provider | null => {
120
+ const resolved = resolveModel(name, model);
121
+ const contextOverride = contextWindowFromEnv(env, name);
122
+ if (lookupProvider(name) === null && configuredProviderInfo(name, env) === null) return null;
123
+ if (name === "openai" && resolved === null) return null;
124
+ if (resolved === null && contextOverride === null) return null;
125
+ const wireModel = resolved?.id ?? model;
126
+ const sdk = createSdkModel(name, wireModel, env, baseUrlOverride);
127
+ if (sdk === null) return null;
128
+
129
+ emitWarningOnce(
130
+ `${name} provider: countTokens is a chars/2 upper bound — exact counts come from the mimetypes tokenizer seam or tokenize()`,
131
+ "PLURNK_TOKENIZER_HEURISTIC",
132
+ );
133
+
134
+ const info = resolved?.info;
135
+ const contextWindow = contextOverride ?? info?.contextWindow ?? null;
136
+ if (contextWindow === null) {
137
+ throw new Error(
138
+ `${name} provider: context window unresolved for "${wireModel}" — set PLURNK_PROVIDERS_CONTEXT_WINDOW or update the Models.dev snapshot`,
139
+ );
140
+ }
141
+ return providerFromSdkModel({
142
+ name,
143
+ env,
144
+ model: wireModel,
145
+ languageModel: sdk.languageModel,
146
+ url: sdk.compatible?.url,
147
+ headers: sdk.compatible?.headers,
148
+ contextWindow,
149
+ info,
150
+ });
151
+ };
@@ -0,0 +1,44 @@
1
+ import test, { mock } from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
4
+
5
+ const env = {
6
+ OPENAI_BASE_URL: "http://local.test/v1",
7
+ PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
8
+ PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
9
+ PLURNK_PROVIDERS_REASONING: "off",
10
+ PLURNK_PROVIDERS_TEMPERATURE: "0.2",
11
+ PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
12
+ PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
13
+ PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
14
+ PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
15
+ PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
16
+ PLURNK_PROVIDERS_PROBE_ATTEMPTS: "1",
17
+ PLURNK_PROVIDERS_PROBE_DELAY: "0",
18
+ PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
19
+ };
20
+
21
+ test.afterEach(() => mock.restoreAll());
22
+
23
+ test("compatible endpoints preserve configured prompt-cache affinity", async () => {
24
+ let body: Record<string, unknown> | undefined;
25
+ mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
26
+ if (String(input).endsWith("/models")) {
27
+ return new Response(JSON.stringify({ data: [{ id: "local", n_ctx: 8192 }] }));
28
+ }
29
+ body = JSON.parse(String(init?.body)) as Record<string, unknown>;
30
+ return new Response(JSON.stringify({
31
+ model: "local",
32
+ choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
33
+ usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
34
+ }), { headers: { "content-type": "application/json" } });
35
+ });
36
+
37
+ const provider = await compatibleProviderFromEnv("openai", env, "local");
38
+ await provider.generate({
39
+ workerId: "worker-affinity",
40
+ messages: [{ role: "user", content: "hello" }],
41
+ });
42
+
43
+ assert.equal(body?.prompt_cache_key, "worker-affinity");
44
+ });