@plurnk/plurnk-providers 1.3.5 → 1.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +35 -45
- package/README.md +44 -53
- package/SPEC.md +215 -371
- package/dist/AiSdkProvider.d.ts +78 -0
- package/dist/AiSdkProvider.d.ts.map +1 -0
- package/dist/AiSdkProvider.js +591 -0
- package/dist/AiSdkProvider.js.map +1 -0
- package/dist/OpenAICompat.d.ts +1 -2
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +39 -117
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +37 -24
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +52 -0
- package/dist/aiSdkTransport.d.ts.map +1 -0
- package/dist/aiSdkTransport.js +294 -0
- package/dist/aiSdkTransport.js.map +1 -0
- package/dist/catalogProvider.d.ts +15 -0
- package/dist/catalogProvider.d.ts.map +1 -0
- package/dist/catalogProvider.js +103 -0
- package/dist/catalogProvider.js.map +1 -0
- package/dist/compatibleProvider.d.ts +3 -0
- package/dist/compatibleProvider.d.ts.map +1 -0
- package/dist/compatibleProvider.js +146 -0
- package/dist/compatibleProvider.js.map +1 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +1 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +13 -6
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +4 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -7
- package/dist/index.js.map +1 -1
- package/dist/ollama.d.ts +3 -0
- package/dist/ollama.d.ts.map +1 -0
- package/dist/ollama.js +39 -0
- package/dist/ollama.js.map +1 -0
- package/dist/openai.d.ts +2 -4
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -2
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +13 -0
- package/dist/sdkModels.d.ts.map +1 -0
- package/dist/sdkModels.js +153 -0
- package/dist/sdkModels.js.map +1 -0
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +0 -1
- package/dist/standardProviders.js.map +1 -1
- package/dist/telemetry.d.ts.map +1 -1
- package/dist/telemetry.js +20 -9
- package/dist/telemetry.js.map +1 -1
- package/dist/types.d.ts +3 -2
- package/dist/types.d.ts.map +1 -1
- package/package.json +18 -10
- package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +193 -152
- package/src/{OpenAICompat.ts → AiSdkProvider.ts} +83 -127
- package/src/Mock.test.ts +1 -1
- package/src/ProviderRegistry.test.ts +40 -27
- package/src/ProviderRegistry.ts +35 -24
- package/src/aiSdkTransport.test.ts +253 -0
- package/src/aiSdkTransport.ts +369 -0
- package/src/boundaries.test.ts +2 -2
- package/src/catalogProvider.test.ts +100 -0
- package/src/catalogProvider.ts +151 -0
- package/src/compatibleProvider.test.ts +44 -0
- package/src/compatibleProvider.ts +205 -0
- package/src/discover.test.ts +12 -12
- package/src/discover.ts +3 -6
- package/src/env.ts +14 -6
- package/src/index.ts +6 -10
- package/src/ollama.ts +63 -0
- package/src/openai.ts +2 -8
- package/src/sdkModels.test.ts +47 -0
- package/src/sdkModels.ts +194 -0
- package/src/telemetry.test.ts +17 -10
- package/src/telemetry.ts +22 -14
- package/src/types.ts +5 -8
- package/src/aiSdkAdapter.spike.test.ts +0 -242
- package/src/openaiStream.ts +0 -310
- package/src/standardProviders.test.ts +0 -939
- package/src/standardProviders.ts +0 -631
|
@@ -1,242 +0,0 @@
|
|
|
1
|
-
import assert from "node:assert/strict";
|
|
2
|
-
import { describe, it } from "node:test";
|
|
3
|
-
import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
|
|
4
|
-
import { APICallError, streamText } from "ai";
|
|
5
|
-
|
|
6
|
-
const encoder = new TextEncoder();
|
|
7
|
-
|
|
8
|
-
const sseResponse = (...chunks: object[]): Response => new Response(
|
|
9
|
-
new ReadableStream({
|
|
10
|
-
start(controller) {
|
|
11
|
-
for (const chunk of chunks) {
|
|
12
|
-
controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
13
|
-
}
|
|
14
|
-
controller.enqueue(encoder.encode("data: [DONE]\n\n"));
|
|
15
|
-
controller.close();
|
|
16
|
-
},
|
|
17
|
-
}),
|
|
18
|
-
{ headers: { "content-type": "text/event-stream" } },
|
|
19
|
-
);
|
|
20
|
-
|
|
21
|
-
describe("AI SDK adapter spike", () => {
|
|
22
|
-
it("preserves PLURNK request extensions and complete stream evidence", async () => {
|
|
23
|
-
let requestBody: Record<string, unknown> | undefined;
|
|
24
|
-
const rawChunks = [
|
|
25
|
-
{
|
|
26
|
-
id: "response-1",
|
|
27
|
-
object: "chat.completion.chunk",
|
|
28
|
-
created: 1,
|
|
29
|
-
model: "test-model",
|
|
30
|
-
choices: [{
|
|
31
|
-
index: 0,
|
|
32
|
-
delta: { role: "assistant", reasoning_content: "because " },
|
|
33
|
-
finish_reason: null,
|
|
34
|
-
}],
|
|
35
|
-
},
|
|
36
|
-
{
|
|
37
|
-
id: "response-1",
|
|
38
|
-
object: "chat.completion.chunk",
|
|
39
|
-
created: 1,
|
|
40
|
-
model: "test-model",
|
|
41
|
-
choices: [{
|
|
42
|
-
index: 0,
|
|
43
|
-
delta: { content: "answer" },
|
|
44
|
-
finish_reason: null,
|
|
45
|
-
}],
|
|
46
|
-
},
|
|
47
|
-
{
|
|
48
|
-
id: "response-1",
|
|
49
|
-
object: "chat.completion.chunk",
|
|
50
|
-
created: 1,
|
|
51
|
-
model: "test-model",
|
|
52
|
-
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
|
53
|
-
usage: {
|
|
54
|
-
prompt_tokens: 3,
|
|
55
|
-
completion_tokens: 5,
|
|
56
|
-
total_tokens: 8,
|
|
57
|
-
completion_tokens_details: { reasoning_tokens: 2 },
|
|
58
|
-
},
|
|
59
|
-
},
|
|
60
|
-
];
|
|
61
|
-
const provider = createOpenAICompatible({
|
|
62
|
-
name: "spike",
|
|
63
|
-
baseURL: "https://example.test/v1",
|
|
64
|
-
apiKey: "test-key",
|
|
65
|
-
includeUsage: true,
|
|
66
|
-
fetch: async (_input, init) => {
|
|
67
|
-
requestBody = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
68
|
-
return sseResponse(...rawChunks);
|
|
69
|
-
},
|
|
70
|
-
});
|
|
71
|
-
|
|
72
|
-
const result = streamText({
|
|
73
|
-
model: provider("test-model"),
|
|
74
|
-
messages: [{ role: "user", content: "question" }],
|
|
75
|
-
maxOutputTokens: 64,
|
|
76
|
-
temperature: 0.25,
|
|
77
|
-
maxRetries: 0,
|
|
78
|
-
timeout: { totalMs: 1_000, chunkMs: 500 },
|
|
79
|
-
includeRawChunks: true,
|
|
80
|
-
providerOptions: {
|
|
81
|
-
spike: {
|
|
82
|
-
grammar: "root ::= \"answer\"",
|
|
83
|
-
id_slot: 2,
|
|
84
|
-
enable_thinking: true,
|
|
85
|
-
service_tier: "flex",
|
|
86
|
-
},
|
|
87
|
-
},
|
|
88
|
-
});
|
|
89
|
-
const parts = [];
|
|
90
|
-
for await (const part of result.fullStream) parts.push(part);
|
|
91
|
-
|
|
92
|
-
assert.equal(requestBody?.model, "test-model");
|
|
93
|
-
assert.equal(requestBody?.max_tokens, 64);
|
|
94
|
-
assert.equal(requestBody?.temperature, 0.25);
|
|
95
|
-
assert.equal(requestBody?.grammar, "root ::= \"answer\"");
|
|
96
|
-
assert.equal(requestBody?.id_slot, 2);
|
|
97
|
-
assert.equal(requestBody?.enable_thinking, true);
|
|
98
|
-
assert.equal(requestBody?.service_tier, "flex");
|
|
99
|
-
assert.equal(requestBody?.stream, true);
|
|
100
|
-
assert.deepEqual(requestBody?.stream_options, { include_usage: true });
|
|
101
|
-
|
|
102
|
-
assert.equal(await result.text, "answer");
|
|
103
|
-
assert.equal(await result.reasoningText, "because ");
|
|
104
|
-
assert.equal(await result.finishReason, "stop");
|
|
105
|
-
assert.equal(await result.rawFinishReason, "stop");
|
|
106
|
-
const usage = await result.usage;
|
|
107
|
-
assert.equal(usage.inputTokens, 3);
|
|
108
|
-
assert.equal(usage.outputTokens, 5);
|
|
109
|
-
assert.equal(usage.outputTokenDetails.reasoningTokens, 2);
|
|
110
|
-
assert.equal(usage.outputTokenDetails.textTokens, 3);
|
|
111
|
-
assert.equal(usage.totalTokens, 8);
|
|
112
|
-
assert.deepEqual(
|
|
113
|
-
parts.filter((part) => part.type === "raw").map((part) => part.rawValue),
|
|
114
|
-
rawChunks,
|
|
115
|
-
);
|
|
116
|
-
});
|
|
117
|
-
|
|
118
|
-
it("surfaces typed HTTP failures without retrying", async () => {
|
|
119
|
-
let requests = 0;
|
|
120
|
-
const provider = createOpenAICompatible({
|
|
121
|
-
name: "spike",
|
|
122
|
-
baseURL: "https://example.test/v1",
|
|
123
|
-
apiKey: "test-key",
|
|
124
|
-
fetch: async () => {
|
|
125
|
-
requests += 1;
|
|
126
|
-
return new Response(
|
|
127
|
-
JSON.stringify({ error: { message: "rate limited", type: "rate_limit" } }),
|
|
128
|
-
{
|
|
129
|
-
status: 429,
|
|
130
|
-
headers: {
|
|
131
|
-
"content-type": "application/json",
|
|
132
|
-
"retry-after": "3",
|
|
133
|
-
"x-request-id": "request-1",
|
|
134
|
-
},
|
|
135
|
-
},
|
|
136
|
-
);
|
|
137
|
-
},
|
|
138
|
-
});
|
|
139
|
-
const result = streamText({
|
|
140
|
-
model: provider("test-model"),
|
|
141
|
-
prompt: "question",
|
|
142
|
-
maxRetries: 0,
|
|
143
|
-
onError: () => {},
|
|
144
|
-
});
|
|
145
|
-
|
|
146
|
-
const parts = [];
|
|
147
|
-
for await (const part of result.fullStream) parts.push(part);
|
|
148
|
-
const errorPart = parts.find((part) => part.type === "error");
|
|
149
|
-
assert.ok(errorPart?.type === "error");
|
|
150
|
-
assert.ok(APICallError.isInstance(errorPart.error));
|
|
151
|
-
assert.equal(errorPart.error.statusCode, 429);
|
|
152
|
-
assert.equal(errorPart.error.isRetryable, true);
|
|
153
|
-
assert.equal(errorPart.error.responseHeaders?.["retry-after"], "3");
|
|
154
|
-
assert.equal(errorPart.error.responseHeaders?.["x-request-id"], "request-1");
|
|
155
|
-
assert.match(errorPart.error.responseBody ?? "", /rate limited/);
|
|
156
|
-
assert.equal(requests, 1);
|
|
157
|
-
});
|
|
158
|
-
|
|
159
|
-
it("enforces stream-idle timeout through the transport contract", async () => {
|
|
160
|
-
const provider = createOpenAICompatible({
|
|
161
|
-
name: "spike",
|
|
162
|
-
baseURL: "https://example.test/v1",
|
|
163
|
-
apiKey: "test-key",
|
|
164
|
-
fetch: async () => new Response(
|
|
165
|
-
new ReadableStream({
|
|
166
|
-
start(controller) {
|
|
167
|
-
controller.enqueue(encoder.encode(`data: ${JSON.stringify({
|
|
168
|
-
id: "response-1",
|
|
169
|
-
object: "chat.completion.chunk",
|
|
170
|
-
created: 1,
|
|
171
|
-
model: "test-model",
|
|
172
|
-
choices: [{
|
|
173
|
-
index: 0,
|
|
174
|
-
delta: { role: "assistant", content: "started" },
|
|
175
|
-
finish_reason: null,
|
|
176
|
-
}],
|
|
177
|
-
})}\n\n`));
|
|
178
|
-
setTimeout(() => controller.close(), 100);
|
|
179
|
-
},
|
|
180
|
-
}),
|
|
181
|
-
{ headers: { "content-type": "text/event-stream" } },
|
|
182
|
-
),
|
|
183
|
-
});
|
|
184
|
-
const result = streamText({
|
|
185
|
-
model: provider("test-model"),
|
|
186
|
-
prompt: "question",
|
|
187
|
-
maxRetries: 0,
|
|
188
|
-
timeout: { totalMs: 500, chunkMs: 25 },
|
|
189
|
-
onError: () => {},
|
|
190
|
-
});
|
|
191
|
-
|
|
192
|
-
await assert.rejects(
|
|
193
|
-
() => Promise.resolve(result.text),
|
|
194
|
-
(error: unknown) => {
|
|
195
|
-
assert.match(String(error), /timed out|timeout/i);
|
|
196
|
-
return true;
|
|
197
|
-
},
|
|
198
|
-
);
|
|
199
|
-
});
|
|
200
|
-
|
|
201
|
-
it("propagates caller cancellation into the injected transport", async () => {
|
|
202
|
-
const caller = new AbortController();
|
|
203
|
-
let transportAborted = false;
|
|
204
|
-
const provider = createOpenAICompatible({
|
|
205
|
-
name: "spike",
|
|
206
|
-
baseURL: "https://example.test/v1",
|
|
207
|
-
apiKey: "test-key",
|
|
208
|
-
fetch: async (_input, init) => {
|
|
209
|
-
const transportSignal = init?.signal as AbortSignal;
|
|
210
|
-
return await new Promise<Response>((_resolve, reject) => {
|
|
211
|
-
transportSignal.addEventListener(
|
|
212
|
-
"abort",
|
|
213
|
-
() => {
|
|
214
|
-
transportAborted = true;
|
|
215
|
-
reject(transportSignal.reason);
|
|
216
|
-
},
|
|
217
|
-
{ once: true },
|
|
218
|
-
);
|
|
219
|
-
});
|
|
220
|
-
},
|
|
221
|
-
});
|
|
222
|
-
const result = streamText({
|
|
223
|
-
model: provider("test-model"),
|
|
224
|
-
prompt: "question",
|
|
225
|
-
maxRetries: 0,
|
|
226
|
-
abortSignal: caller.signal,
|
|
227
|
-
onError: () => {},
|
|
228
|
-
});
|
|
229
|
-
const partsPromise = (async () => {
|
|
230
|
-
const parts = [];
|
|
231
|
-
for await (const part of result.fullStream) parts.push(part);
|
|
232
|
-
return parts;
|
|
233
|
-
})();
|
|
234
|
-
|
|
235
|
-
await new Promise((resolve) => setTimeout(resolve, 0));
|
|
236
|
-
caller.abort(new Error("caller stopped"));
|
|
237
|
-
const parts = await partsPromise;
|
|
238
|
-
|
|
239
|
-
assert.equal(transportAborted, true);
|
|
240
|
-
assert.ok(parts.some((part) => part.type === "abort"));
|
|
241
|
-
});
|
|
242
|
-
});
|
package/src/openaiStream.ts
DELETED
|
@@ -1,310 +0,0 @@
|
|
|
1
|
-
// SSE client for OpenAI-compatible /chat/completions. Streaming keeps long
|
|
2
|
-
// completions alive through CDN proxies; the aggregated result is returned as
|
|
3
|
-
// one StreamResponse (the Provider contract is atomic — no partial resolves).
|
|
4
|
-
// Adapted from rummy's proven implementation; previously copy-pasted byte-for-
|
|
5
|
-
// byte into every @plurnk/plurnk-providers-* sibling, now shared from here.
|
|
6
|
-
|
|
7
|
-
type StreamRequest = {
|
|
8
|
-
url: string;
|
|
9
|
-
headers: Record<string, string>;
|
|
10
|
-
body: Record<string, unknown>;
|
|
11
|
-
signal: AbortSignal;
|
|
12
|
-
fetch: ProviderFetch;
|
|
13
|
-
// #36: assemble the verbatim wire body onto StreamResponse.rawBody. Off by
|
|
14
|
-
// default so a serving turn never pays the reassembly/retention cost.
|
|
15
|
-
captureRawBody?: boolean;
|
|
16
|
-
// Maximum silence between streamed response-body chunks. Undefined/zero
|
|
17
|
-
// disables this clock; the caller's signal still owns the total deadline.
|
|
18
|
-
streamIdleTimeoutMs?: number;
|
|
19
|
-
};
|
|
20
|
-
|
|
21
|
-
import type { RawUsage } from "./usage.ts";
|
|
22
|
-
import type { TokenLogprob } from "./types.ts";
|
|
23
|
-
|
|
24
|
-
export type ProviderFetch = typeof globalThis.fetch;
|
|
25
|
-
|
|
26
|
-
// Sealed reasoning (#482, widened per client). A relay backend (OpenRouter
|
|
27
|
-
// fronting OpenAI o-series) returns the chain-of-thought ENCRYPTED as
|
|
28
|
-
// reasoning_details entries ({ type: "reasoning.encrypted", id, data, format,
|
|
29
|
-
// index }) while readable text still rides reasoning/reasoning_content (verified
|
|
30
|
-
// live: o4-mini via OpenRouter — reasoning null, one encrypted entry, format
|
|
31
|
-
// "openai-responses-v1"). The ITEM shape preserves the wire's `id` (a flat blob
|
|
32
|
-
// list dropped item identity — the widening's whole point) and a `subtype` from
|
|
33
|
-
// wire POSITION: we parse message.reasoning_details, so it is message-attached.
|
|
34
|
-
// plurnk is tools-in-body (SPEC §2), so reasoning is never tool-call-attached and
|
|
35
|
-
// subtype is constant here; the field is structural, future-proofing the seam.
|
|
36
|
-
// An ARRAY of items (not a single object) so N distinct reasoning ids never
|
|
37
|
-
// re-collide the identity this fixes. Blobs verbatim, never decoded.
|
|
38
|
-
export type EncryptedReasoningItem = { id: string | null; subtype: string; encrypted: Array<{ data: string; format: string | null }> };
|
|
39
|
-
|
|
40
|
-
type RawEncrypted = { id: string | null; data: string; format: string | null };
|
|
41
|
-
|
|
42
|
-
// Group accumulated encrypted entries into items by wire `id` (id-less entries
|
|
43
|
-
// stand alone, never merged). Order preserved.
|
|
44
|
-
const groupEncrypted = (entries: Iterable<RawEncrypted>): EncryptedReasoningItem[] => {
|
|
45
|
-
const items: EncryptedReasoningItem[] = [];
|
|
46
|
-
const byId = new Map<string, EncryptedReasoningItem>();
|
|
47
|
-
for (const e of entries) {
|
|
48
|
-
const blob = { data: e.data, format: e.format };
|
|
49
|
-
if (e.id === null) { items.push({ id: null, subtype: "message", encrypted: [blob] }); continue; }
|
|
50
|
-
let item = byId.get(e.id);
|
|
51
|
-
if (item === undefined) { item = { id: e.id, subtype: "message", encrypted: [] }; byId.set(e.id, item); items.push(item); }
|
|
52
|
-
item.encrypted.push(blob);
|
|
53
|
-
}
|
|
54
|
-
return items;
|
|
55
|
-
};
|
|
56
|
-
|
|
57
|
-
const encryptedFromDetails = (details: unknown): EncryptedReasoningItem[] => {
|
|
58
|
-
if (!Array.isArray(details)) return [];
|
|
59
|
-
const raw: RawEncrypted[] = [];
|
|
60
|
-
for (const e of details) {
|
|
61
|
-
const entry = e as { type?: unknown; id?: unknown; data?: unknown; format?: unknown };
|
|
62
|
-
if (entry?.type !== "reasoning.encrypted" || typeof entry.data !== "string") continue;
|
|
63
|
-
raw.push({ id: typeof entry.id === "string" ? entry.id : null, data: entry.data, format: typeof entry.format === "string" ? entry.format : null });
|
|
64
|
-
}
|
|
65
|
-
return groupEncrypted(raw);
|
|
66
|
-
};
|
|
67
|
-
|
|
68
|
-
export type StreamResponse = {
|
|
69
|
-
model: string | null;
|
|
70
|
-
content: string;
|
|
71
|
-
reasoning_content: string;
|
|
72
|
-
// Sealed relay reasoning (#482) — empty for the open-reasoning backends.
|
|
73
|
-
reasoning_encrypted: EncryptedReasoningItem[];
|
|
74
|
-
finish_reason: string | null;
|
|
75
|
-
usage: RawUsage | null;
|
|
76
|
-
chunkMetadata: Record<string, unknown>;
|
|
77
|
-
// #36: per-token logprobs parsed from choices[0].logprobs.content[], present
|
|
78
|
-
// only when the request asked for them (else the field is absent → null).
|
|
79
|
-
logprobs: TokenLogprob[] | null;
|
|
80
|
-
// #36: the verbatim response body, populated only when captureRawBody is set.
|
|
81
|
-
rawBody: unknown;
|
|
82
|
-
};
|
|
83
|
-
|
|
84
|
-
// Map an OpenAI-style `logprobs.content[]` array to the canonical structured view
|
|
85
|
-
// (#36). Reads the RAW `logprob` (not `sampling_logprob`); `top_logprobs` → `top`.
|
|
86
|
-
// Returns null when the shape is absent — never synthesizes.
|
|
87
|
-
const parseLogprobs = (raw: unknown): TokenLogprob[] | null => {
|
|
88
|
-
const content = (raw as { content?: unknown } | null | undefined)?.content;
|
|
89
|
-
if (!Array.isArray(content)) return null;
|
|
90
|
-
return content.map((entry) => {
|
|
91
|
-
const { token, logprob, top_logprobs } = entry as { token: string; logprob: number; top_logprobs?: unknown };
|
|
92
|
-
const top = Array.isArray(top_logprobs)
|
|
93
|
-
? top_logprobs.map((a) => ({ token: (a as TokenLogprob).token, logprob: (a as TokenLogprob).logprob }))
|
|
94
|
-
: undefined;
|
|
95
|
-
return top !== undefined ? { token, logprob, top } : { token, logprob };
|
|
96
|
-
});
|
|
97
|
-
};
|
|
98
|
-
|
|
99
|
-
// Cloudflare/CDN EDGE status codes (520-527): infrastructure failures the proxy
|
|
100
|
-
// returns (as HTML error pages), NOT OpenAI/API statuses. A retry re-incurs the
|
|
101
|
-
// same origin wait, so they fail-fast (#543).
|
|
102
|
-
const EDGE_LABELS: ReadonlyMap<number, string> = new Map([
|
|
103
|
-
[520, "web server returned an unknown error"], [521, "web server is down"],
|
|
104
|
-
[522, "connection timed out"], [523, "origin is unreachable"], [524, "origin timeout"],
|
|
105
|
-
[525, "SSL handshake failed"], [526, "invalid SSL certificate"], [527, "railgun error"],
|
|
106
|
-
]);
|
|
107
|
-
export const isEdgeStatus = (status: number): boolean => status >= 520 && status <= 527;
|
|
108
|
-
|
|
109
|
-
export class OpenAiHttpError extends Error {
|
|
110
|
-
readonly status: number;
|
|
111
|
-
readonly body: string;
|
|
112
|
-
readonly retryAfter: number | null;
|
|
113
|
-
constructor(status: number, body: string, retryAfter: number | null) {
|
|
114
|
-
super(OpenAiHttpError.#describe(status, body));
|
|
115
|
-
this.status = status;
|
|
116
|
-
this.body = body;
|
|
117
|
-
this.retryAfter = retryAfter;
|
|
118
|
-
}
|
|
119
|
-
// A non-JSON error body (a proxy/CDN HTML page) collapses to one line and drops
|
|
120
|
-
// the misleading "OpenAI" prefix - an edge code is not an API status (#543).
|
|
121
|
-
// JSON API errors pass through verbatim.
|
|
122
|
-
static #describe(status: number, body: string): string {
|
|
123
|
-
if (body.trimStart().startsWith("<")) return `${status} ${EDGE_LABELS.get(status) ?? "edge/proxy error"}`;
|
|
124
|
-
return `OpenAI ${status} - ${body}`;
|
|
125
|
-
}
|
|
126
|
-
}
|
|
127
|
-
|
|
128
|
-
export class StreamIdleError extends Error {
|
|
129
|
-
readonly timeoutMs: number;
|
|
130
|
-
constructor(timeoutMs: number) {
|
|
131
|
-
super(`stream received no body bytes for ${timeoutMs}ms`);
|
|
132
|
-
this.name = "StreamIdleError";
|
|
133
|
-
this.timeoutMs = timeoutMs;
|
|
134
|
-
}
|
|
135
|
-
}
|
|
136
|
-
|
|
137
|
-
const parseRetryAfter = (header: string | null): number | null => {
|
|
138
|
-
if (header === null) return null;
|
|
139
|
-
const asInt = Number.parseInt(header, 10);
|
|
140
|
-
if (Number.isFinite(asInt)) return asInt * 1000;
|
|
141
|
-
const asDate = Date.parse(header);
|
|
142
|
-
if (Number.isFinite(asDate)) return Math.max(0, asDate - Date.now());
|
|
143
|
-
return null;
|
|
144
|
-
};
|
|
145
|
-
|
|
146
|
-
// Non-streaming sibling. Same request/error handling, but POSTs without
|
|
147
|
-
// `stream` and parses the single JSON body into the SAME StreamResponse shape.
|
|
148
|
-
// For backends whose STREAMING response misbehaves (e.g. Fireworks labels
|
|
149
|
-
// grammar-constrained output as `reasoning_content` instead of `content`) —
|
|
150
|
-
// the Provider contract is atomic either way, so the transport is free to
|
|
151
|
-
// choose. The fetch timeout (AbortSignal) bounds the wait; there is no proxy
|
|
152
|
-
// between us and the backend that would idle out a non-streamed request.
|
|
153
|
-
export const chatCompletion = async ({ url, headers, body, signal, fetch, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
|
|
154
|
-
const response = await fetch(url, {
|
|
155
|
-
method: "POST",
|
|
156
|
-
headers: { "Content-Type": "application/json", ...headers },
|
|
157
|
-
body: JSON.stringify(body),
|
|
158
|
-
signal,
|
|
159
|
-
});
|
|
160
|
-
if (!response.ok) {
|
|
161
|
-
const errorBody = await response.text();
|
|
162
|
-
throw new OpenAiHttpError(response.status, errorBody, parseRetryAfter(response.headers.get("retry-after")));
|
|
163
|
-
}
|
|
164
|
-
const j = (await response.json()) as Record<string, unknown>;
|
|
165
|
-
const choices = j.choices as Array<Record<string, unknown>> | undefined;
|
|
166
|
-
const choice = (choices?.[0] ?? {}) as Record<string, unknown>;
|
|
167
|
-
const msg = (choice.message ?? {}) as Record<string, unknown>;
|
|
168
|
-
const reasoning = msg.reasoning_content ?? msg.reasoning ?? msg.thinking ?? "";
|
|
169
|
-
const chunkMetadata: Record<string, unknown> = {};
|
|
170
|
-
for (const [k, v] of Object.entries(j)) if (k !== "choices" && k !== "usage") chunkMetadata[k] = v;
|
|
171
|
-
return {
|
|
172
|
-
model: typeof j.model === "string" ? j.model : null,
|
|
173
|
-
content: typeof msg.content === "string" ? msg.content : "",
|
|
174
|
-
reasoning_content: typeof reasoning === "string" ? reasoning : "",
|
|
175
|
-
reasoning_encrypted: encryptedFromDetails(msg.reasoning_details),
|
|
176
|
-
finish_reason: typeof choice.finish_reason === "string" ? choice.finish_reason : null,
|
|
177
|
-
usage: (j.usage ?? null) as StreamResponse["usage"],
|
|
178
|
-
chunkMetadata,
|
|
179
|
-
logprobs: parseLogprobs(choice.logprobs),
|
|
180
|
-
// Non-streamed: the parsed JSON IS the verbatim wire body, exact.
|
|
181
|
-
rawBody: captureRawBody === true ? j : undefined,
|
|
182
|
-
};
|
|
183
|
-
};
|
|
184
|
-
|
|
185
|
-
export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody, streamIdleTimeoutMs }: StreamRequest): Promise<StreamResponse> => {
|
|
186
|
-
const requestBody = { ...body, stream: true, stream_options: { include_usage: true } };
|
|
187
|
-
|
|
188
|
-
const response = await fetch(url, {
|
|
189
|
-
method: "POST",
|
|
190
|
-
headers: { "Content-Type": "application/json", ...headers },
|
|
191
|
-
body: JSON.stringify(requestBody),
|
|
192
|
-
signal,
|
|
193
|
-
});
|
|
194
|
-
|
|
195
|
-
if (!response.ok) {
|
|
196
|
-
const errorBody = await response.text();
|
|
197
|
-
throw new OpenAiHttpError(response.status, errorBody, parseRetryAfter(response.headers.get("retry-after")));
|
|
198
|
-
}
|
|
199
|
-
|
|
200
|
-
if (response.body === null) throw new Error("OpenAI response body is null");
|
|
201
|
-
const reader = response.body.getReader();
|
|
202
|
-
const decoder = new TextDecoder();
|
|
203
|
-
|
|
204
|
-
let buffer = "";
|
|
205
|
-
let content = "";
|
|
206
|
-
let reasoning_content = "";
|
|
207
|
-
let usage: StreamResponse["usage"] = null;
|
|
208
|
-
let model: string | null = null;
|
|
209
|
-
let finish_reason: string | null = null;
|
|
210
|
-
const chunkMetadata: Record<string, unknown> = {};
|
|
211
|
-
// #36: logprobs stream as per-chunk choices[0].logprobs.content[] deltas —
|
|
212
|
-
// accumulate the raw entries across chunks, map once at the end.
|
|
213
|
-
const logprobEntries: unknown[] = [];
|
|
214
|
-
// #482: encrypted reasoning_details stream chunked — concatenate `data` per
|
|
215
|
-
// reassembly key (index when present, else id, else a counter); the id/format
|
|
216
|
-
// ride along and items group by id at the end.
|
|
217
|
-
const encryptedByKey = new Map<string, RawEncrypted>();
|
|
218
|
-
let encryptedNoKey = 0;
|
|
219
|
-
|
|
220
|
-
while (true) {
|
|
221
|
-
const read = reader.read();
|
|
222
|
-
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
223
|
-
const idle = streamIdleTimeoutMs !== undefined && streamIdleTimeoutMs > 0
|
|
224
|
-
? new Promise<never>((_resolve, reject) => {
|
|
225
|
-
timer = setTimeout(() => reject(new StreamIdleError(streamIdleTimeoutMs)), streamIdleTimeoutMs);
|
|
226
|
-
})
|
|
227
|
-
: null;
|
|
228
|
-
let result: Awaited<ReturnType<typeof reader.read>>;
|
|
229
|
-
try {
|
|
230
|
-
result = idle === null ? await read : await Promise.race([read, idle]);
|
|
231
|
-
} catch (err) {
|
|
232
|
-
if (err instanceof StreamIdleError) void reader.cancel(err).catch(() => undefined);
|
|
233
|
-
throw err;
|
|
234
|
-
} finally {
|
|
235
|
-
if (timer !== undefined) clearTimeout(timer);
|
|
236
|
-
}
|
|
237
|
-
const { done, value } = result;
|
|
238
|
-
if (done) break;
|
|
239
|
-
buffer += decoder.decode(value, { stream: true });
|
|
240
|
-
const lines = buffer.split("\n");
|
|
241
|
-
buffer = lines.pop() ?? "";
|
|
242
|
-
|
|
243
|
-
for (const rawLine of lines) {
|
|
244
|
-
const line = rawLine.trim();
|
|
245
|
-
if (!line.startsWith("data:")) continue;
|
|
246
|
-
const payload = line.slice(5).trimStart();
|
|
247
|
-
if (payload === "[DONE]" || payload === "") continue;
|
|
248
|
-
|
|
249
|
-
let chunk: Record<string, unknown>;
|
|
250
|
-
try { chunk = JSON.parse(payload) as Record<string, unknown>; } catch { continue; }
|
|
251
|
-
|
|
252
|
-
// A streaming server may flush HTTP 200 headers before inference
|
|
253
|
-
// completes, then report a terminal failure as an SSE error frame.
|
|
254
|
-
// That frame is a failed exchange, never an empty completion.
|
|
255
|
-
if (chunk.error !== null && typeof chunk.error === "object") {
|
|
256
|
-
const status = typeof chunk.status === "number"
|
|
257
|
-
&& Number.isInteger(chunk.status)
|
|
258
|
-
&& chunk.status >= 400
|
|
259
|
-
&& chunk.status <= 599
|
|
260
|
-
? chunk.status
|
|
261
|
-
: 500;
|
|
262
|
-
throw new OpenAiHttpError(status, JSON.stringify({ error: chunk.error }), null);
|
|
263
|
-
}
|
|
264
|
-
|
|
265
|
-
if (typeof chunk.model === "string") model = chunk.model;
|
|
266
|
-
if (chunk.usage !== undefined && chunk.usage !== null) usage = chunk.usage as StreamResponse["usage"];
|
|
267
|
-
|
|
268
|
-
for (const [k, v] of Object.entries(chunk)) {
|
|
269
|
-
if (k === "choices" || k === "usage") continue;
|
|
270
|
-
chunkMetadata[k] = v;
|
|
271
|
-
}
|
|
272
|
-
|
|
273
|
-
const choices = chunk.choices as Array<Record<string, unknown>> | undefined;
|
|
274
|
-
const choice = choices?.[0];
|
|
275
|
-
if (choice === undefined) continue;
|
|
276
|
-
if (typeof choice.finish_reason === "string") finish_reason = choice.finish_reason;
|
|
277
|
-
|
|
278
|
-
const chunkLogprobs = (choice.logprobs as { content?: unknown } | undefined)?.content;
|
|
279
|
-
if (Array.isArray(chunkLogprobs)) logprobEntries.push(...chunkLogprobs);
|
|
280
|
-
|
|
281
|
-
const delta = choice.delta as Record<string, unknown> | undefined;
|
|
282
|
-
if (delta === undefined) continue;
|
|
283
|
-
if (typeof delta.content === "string") content += delta.content;
|
|
284
|
-
// Reasoning surfaces under different field names per provider.
|
|
285
|
-
if (typeof delta.reasoning_content === "string") reasoning_content += delta.reasoning_content;
|
|
286
|
-
if (typeof delta.reasoning === "string") reasoning_content += delta.reasoning;
|
|
287
|
-
if (typeof delta.thinking === "string") reasoning_content += delta.thinking;
|
|
288
|
-
if (Array.isArray(delta.reasoning_details)) {
|
|
289
|
-
for (const e of delta.reasoning_details) {
|
|
290
|
-
const entry = e as { type?: unknown; id?: unknown; data?: unknown; format?: unknown; index?: unknown };
|
|
291
|
-
if (entry?.type !== "reasoning.encrypted" || typeof entry.data !== "string") continue;
|
|
292
|
-
const id = typeof entry.id === "string" ? entry.id : null;
|
|
293
|
-
const key = typeof entry.index === "number" ? `i${entry.index}` : id ?? `n${encryptedNoKey++}`;
|
|
294
|
-
const prev = encryptedByKey.get(key);
|
|
295
|
-
if (prev !== undefined) prev.data += entry.data;
|
|
296
|
-
else encryptedByKey.set(key, { id, data: entry.data, format: typeof entry.format === "string" ? entry.format : null });
|
|
297
|
-
}
|
|
298
|
-
}
|
|
299
|
-
}
|
|
300
|
-
}
|
|
301
|
-
|
|
302
|
-
const logprobs = logprobEntries.length > 0 ? parseLogprobs({ content: logprobEntries }) : null;
|
|
303
|
-
// Streamed turns have no single verbatim wire body; reassemble the equivalent
|
|
304
|
-
// (#36) — chunk-level fields (chunkMetadata) + the collected choice — only when
|
|
305
|
-
// asked, so serving turns pay nothing.
|
|
306
|
-
const rawBody = captureRawBody === true
|
|
307
|
-
? { ...chunkMetadata, model, usage, choices: [{ index: 0, message: { content, reasoning_content }, finish_reason, logprobs: logprobs !== null ? { content: logprobEntries } : null }] }
|
|
308
|
-
: undefined;
|
|
309
|
-
return { model, content, reasoning_content, reasoning_encrypted: groupEncrypted(encryptedByKey.values()), finish_reason, usage, chunkMetadata, logprobs, rawBody };
|
|
310
|
-
};
|