@plurnk/plurnk-providers 1.3.5 → 1.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/.env.defaults +35 -45
  2. package/README.md +44 -53
  3. package/SPEC.md +215 -371
  4. package/dist/AiSdkProvider.d.ts +78 -0
  5. package/dist/AiSdkProvider.d.ts.map +1 -0
  6. package/dist/AiSdkProvider.js +591 -0
  7. package/dist/AiSdkProvider.js.map +1 -0
  8. package/dist/OpenAICompat.d.ts +1 -2
  9. package/dist/OpenAICompat.d.ts.map +1 -1
  10. package/dist/OpenAICompat.js +39 -117
  11. package/dist/OpenAICompat.js.map +1 -1
  12. package/dist/ProviderRegistry.d.ts.map +1 -1
  13. package/dist/ProviderRegistry.js +37 -24
  14. package/dist/ProviderRegistry.js.map +1 -1
  15. package/dist/aiSdkTransport.d.ts +52 -0
  16. package/dist/aiSdkTransport.d.ts.map +1 -0
  17. package/dist/aiSdkTransport.js +294 -0
  18. package/dist/aiSdkTransport.js.map +1 -0
  19. package/dist/catalogProvider.d.ts +15 -0
  20. package/dist/catalogProvider.d.ts.map +1 -0
  21. package/dist/catalogProvider.js +103 -0
  22. package/dist/catalogProvider.js.map +1 -0
  23. package/dist/compatibleProvider.d.ts +3 -0
  24. package/dist/compatibleProvider.d.ts.map +1 -0
  25. package/dist/compatibleProvider.js +146 -0
  26. package/dist/compatibleProvider.js.map +1 -0
  27. package/dist/discover.d.ts.map +1 -1
  28. package/dist/discover.js.map +1 -1
  29. package/dist/env.d.ts +1 -0
  30. package/dist/env.d.ts.map +1 -1
  31. package/dist/env.js +13 -6
  32. package/dist/env.js.map +1 -1
  33. package/dist/index.d.ts +4 -6
  34. package/dist/index.d.ts.map +1 -1
  35. package/dist/index.js +3 -7
  36. package/dist/index.js.map +1 -1
  37. package/dist/ollama.d.ts +3 -0
  38. package/dist/ollama.d.ts.map +1 -0
  39. package/dist/ollama.js +39 -0
  40. package/dist/ollama.js.map +1 -0
  41. package/dist/openai.d.ts +2 -4
  42. package/dist/openai.d.ts.map +1 -1
  43. package/dist/openai.js +1 -2
  44. package/dist/openai.js.map +1 -1
  45. package/dist/sdkModels.d.ts +13 -0
  46. package/dist/sdkModels.d.ts.map +1 -0
  47. package/dist/sdkModels.js +153 -0
  48. package/dist/sdkModels.js.map +1 -0
  49. package/dist/standardProviders.d.ts.map +1 -1
  50. package/dist/standardProviders.js +0 -1
  51. package/dist/standardProviders.js.map +1 -1
  52. package/dist/telemetry.d.ts.map +1 -1
  53. package/dist/telemetry.js +20 -9
  54. package/dist/telemetry.js.map +1 -1
  55. package/dist/types.d.ts +3 -2
  56. package/dist/types.d.ts.map +1 -1
  57. package/package.json +18 -10
  58. package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +193 -152
  59. package/src/{OpenAICompat.ts → AiSdkProvider.ts} +83 -127
  60. package/src/Mock.test.ts +1 -1
  61. package/src/ProviderRegistry.test.ts +40 -27
  62. package/src/ProviderRegistry.ts +35 -24
  63. package/src/aiSdkTransport.test.ts +253 -0
  64. package/src/aiSdkTransport.ts +369 -0
  65. package/src/boundaries.test.ts +2 -2
  66. package/src/catalogProvider.test.ts +100 -0
  67. package/src/catalogProvider.ts +151 -0
  68. package/src/compatibleProvider.test.ts +44 -0
  69. package/src/compatibleProvider.ts +205 -0
  70. package/src/discover.test.ts +12 -12
  71. package/src/discover.ts +3 -6
  72. package/src/env.ts +14 -6
  73. package/src/index.ts +6 -10
  74. package/src/ollama.ts +63 -0
  75. package/src/openai.ts +2 -8
  76. package/src/sdkModels.test.ts +47 -0
  77. package/src/sdkModels.ts +194 -0
  78. package/src/telemetry.test.ts +17 -10
  79. package/src/telemetry.ts +22 -14
  80. package/src/types.ts +5 -8
  81. package/src/aiSdkAdapter.spike.test.ts +0 -242
  82. package/src/openaiStream.ts +0 -310
  83. package/src/standardProviders.test.ts +0 -939
  84. package/src/standardProviders.ts +0 -631
@@ -1,242 +0,0 @@
1
- import assert from "node:assert/strict";
2
- import { describe, it } from "node:test";
3
- import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
4
- import { APICallError, streamText } from "ai";
5
-
6
- const encoder = new TextEncoder();
7
-
8
- const sseResponse = (...chunks: object[]): Response => new Response(
9
- new ReadableStream({
10
- start(controller) {
11
- for (const chunk of chunks) {
12
- controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`));
13
- }
14
- controller.enqueue(encoder.encode("data: [DONE]\n\n"));
15
- controller.close();
16
- },
17
- }),
18
- { headers: { "content-type": "text/event-stream" } },
19
- );
20
-
21
- describe("AI SDK adapter spike", () => {
22
- it("preserves PLURNK request extensions and complete stream evidence", async () => {
23
- let requestBody: Record<string, unknown> | undefined;
24
- const rawChunks = [
25
- {
26
- id: "response-1",
27
- object: "chat.completion.chunk",
28
- created: 1,
29
- model: "test-model",
30
- choices: [{
31
- index: 0,
32
- delta: { role: "assistant", reasoning_content: "because " },
33
- finish_reason: null,
34
- }],
35
- },
36
- {
37
- id: "response-1",
38
- object: "chat.completion.chunk",
39
- created: 1,
40
- model: "test-model",
41
- choices: [{
42
- index: 0,
43
- delta: { content: "answer" },
44
- finish_reason: null,
45
- }],
46
- },
47
- {
48
- id: "response-1",
49
- object: "chat.completion.chunk",
50
- created: 1,
51
- model: "test-model",
52
- choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
53
- usage: {
54
- prompt_tokens: 3,
55
- completion_tokens: 5,
56
- total_tokens: 8,
57
- completion_tokens_details: { reasoning_tokens: 2 },
58
- },
59
- },
60
- ];
61
- const provider = createOpenAICompatible({
62
- name: "spike",
63
- baseURL: "https://example.test/v1",
64
- apiKey: "test-key",
65
- includeUsage: true,
66
- fetch: async (_input, init) => {
67
- requestBody = JSON.parse(String(init?.body)) as Record<string, unknown>;
68
- return sseResponse(...rawChunks);
69
- },
70
- });
71
-
72
- const result = streamText({
73
- model: provider("test-model"),
74
- messages: [{ role: "user", content: "question" }],
75
- maxOutputTokens: 64,
76
- temperature: 0.25,
77
- maxRetries: 0,
78
- timeout: { totalMs: 1_000, chunkMs: 500 },
79
- includeRawChunks: true,
80
- providerOptions: {
81
- spike: {
82
- grammar: "root ::= \"answer\"",
83
- id_slot: 2,
84
- enable_thinking: true,
85
- service_tier: "flex",
86
- },
87
- },
88
- });
89
- const parts = [];
90
- for await (const part of result.fullStream) parts.push(part);
91
-
92
- assert.equal(requestBody?.model, "test-model");
93
- assert.equal(requestBody?.max_tokens, 64);
94
- assert.equal(requestBody?.temperature, 0.25);
95
- assert.equal(requestBody?.grammar, "root ::= \"answer\"");
96
- assert.equal(requestBody?.id_slot, 2);
97
- assert.equal(requestBody?.enable_thinking, true);
98
- assert.equal(requestBody?.service_tier, "flex");
99
- assert.equal(requestBody?.stream, true);
100
- assert.deepEqual(requestBody?.stream_options, { include_usage: true });
101
-
102
- assert.equal(await result.text, "answer");
103
- assert.equal(await result.reasoningText, "because ");
104
- assert.equal(await result.finishReason, "stop");
105
- assert.equal(await result.rawFinishReason, "stop");
106
- const usage = await result.usage;
107
- assert.equal(usage.inputTokens, 3);
108
- assert.equal(usage.outputTokens, 5);
109
- assert.equal(usage.outputTokenDetails.reasoningTokens, 2);
110
- assert.equal(usage.outputTokenDetails.textTokens, 3);
111
- assert.equal(usage.totalTokens, 8);
112
- assert.deepEqual(
113
- parts.filter((part) => part.type === "raw").map((part) => part.rawValue),
114
- rawChunks,
115
- );
116
- });
117
-
118
- it("surfaces typed HTTP failures without retrying", async () => {
119
- let requests = 0;
120
- const provider = createOpenAICompatible({
121
- name: "spike",
122
- baseURL: "https://example.test/v1",
123
- apiKey: "test-key",
124
- fetch: async () => {
125
- requests += 1;
126
- return new Response(
127
- JSON.stringify({ error: { message: "rate limited", type: "rate_limit" } }),
128
- {
129
- status: 429,
130
- headers: {
131
- "content-type": "application/json",
132
- "retry-after": "3",
133
- "x-request-id": "request-1",
134
- },
135
- },
136
- );
137
- },
138
- });
139
- const result = streamText({
140
- model: provider("test-model"),
141
- prompt: "question",
142
- maxRetries: 0,
143
- onError: () => {},
144
- });
145
-
146
- const parts = [];
147
- for await (const part of result.fullStream) parts.push(part);
148
- const errorPart = parts.find((part) => part.type === "error");
149
- assert.ok(errorPart?.type === "error");
150
- assert.ok(APICallError.isInstance(errorPart.error));
151
- assert.equal(errorPart.error.statusCode, 429);
152
- assert.equal(errorPart.error.isRetryable, true);
153
- assert.equal(errorPart.error.responseHeaders?.["retry-after"], "3");
154
- assert.equal(errorPart.error.responseHeaders?.["x-request-id"], "request-1");
155
- assert.match(errorPart.error.responseBody ?? "", /rate limited/);
156
- assert.equal(requests, 1);
157
- });
158
-
159
- it("enforces stream-idle timeout through the transport contract", async () => {
160
- const provider = createOpenAICompatible({
161
- name: "spike",
162
- baseURL: "https://example.test/v1",
163
- apiKey: "test-key",
164
- fetch: async () => new Response(
165
- new ReadableStream({
166
- start(controller) {
167
- controller.enqueue(encoder.encode(`data: ${JSON.stringify({
168
- id: "response-1",
169
- object: "chat.completion.chunk",
170
- created: 1,
171
- model: "test-model",
172
- choices: [{
173
- index: 0,
174
- delta: { role: "assistant", content: "started" },
175
- finish_reason: null,
176
- }],
177
- })}\n\n`));
178
- setTimeout(() => controller.close(), 100);
179
- },
180
- }),
181
- { headers: { "content-type": "text/event-stream" } },
182
- ),
183
- });
184
- const result = streamText({
185
- model: provider("test-model"),
186
- prompt: "question",
187
- maxRetries: 0,
188
- timeout: { totalMs: 500, chunkMs: 25 },
189
- onError: () => {},
190
- });
191
-
192
- await assert.rejects(
193
- () => Promise.resolve(result.text),
194
- (error: unknown) => {
195
- assert.match(String(error), /timed out|timeout/i);
196
- return true;
197
- },
198
- );
199
- });
200
-
201
- it("propagates caller cancellation into the injected transport", async () => {
202
- const caller = new AbortController();
203
- let transportAborted = false;
204
- const provider = createOpenAICompatible({
205
- name: "spike",
206
- baseURL: "https://example.test/v1",
207
- apiKey: "test-key",
208
- fetch: async (_input, init) => {
209
- const transportSignal = init?.signal as AbortSignal;
210
- return await new Promise<Response>((_resolve, reject) => {
211
- transportSignal.addEventListener(
212
- "abort",
213
- () => {
214
- transportAborted = true;
215
- reject(transportSignal.reason);
216
- },
217
- { once: true },
218
- );
219
- });
220
- },
221
- });
222
- const result = streamText({
223
- model: provider("test-model"),
224
- prompt: "question",
225
- maxRetries: 0,
226
- abortSignal: caller.signal,
227
- onError: () => {},
228
- });
229
- const partsPromise = (async () => {
230
- const parts = [];
231
- for await (const part of result.fullStream) parts.push(part);
232
- return parts;
233
- })();
234
-
235
- await new Promise((resolve) => setTimeout(resolve, 0));
236
- caller.abort(new Error("caller stopped"));
237
- const parts = await partsPromise;
238
-
239
- assert.equal(transportAborted, true);
240
- assert.ok(parts.some((part) => part.type === "abort"));
241
- });
242
- });
@@ -1,310 +0,0 @@
1
- // SSE client for OpenAI-compatible /chat/completions. Streaming keeps long
2
- // completions alive through CDN proxies; the aggregated result is returned as
3
- // one StreamResponse (the Provider contract is atomic — no partial resolves).
4
- // Adapted from rummy's proven implementation; previously copy-pasted byte-for-
5
- // byte into every @plurnk/plurnk-providers-* sibling, now shared from here.
6
-
7
- type StreamRequest = {
8
- url: string;
9
- headers: Record<string, string>;
10
- body: Record<string, unknown>;
11
- signal: AbortSignal;
12
- fetch: ProviderFetch;
13
- // #36: assemble the verbatim wire body onto StreamResponse.rawBody. Off by
14
- // default so a serving turn never pays the reassembly/retention cost.
15
- captureRawBody?: boolean;
16
- // Maximum silence between streamed response-body chunks. Undefined/zero
17
- // disables this clock; the caller's signal still owns the total deadline.
18
- streamIdleTimeoutMs?: number;
19
- };
20
-
21
- import type { RawUsage } from "./usage.ts";
22
- import type { TokenLogprob } from "./types.ts";
23
-
24
- export type ProviderFetch = typeof globalThis.fetch;
25
-
26
- // Sealed reasoning (#482, widened per client). A relay backend (OpenRouter
27
- // fronting OpenAI o-series) returns the chain-of-thought ENCRYPTED as
28
- // reasoning_details entries ({ type: "reasoning.encrypted", id, data, format,
29
- // index }) while readable text still rides reasoning/reasoning_content (verified
30
- // live: o4-mini via OpenRouter — reasoning null, one encrypted entry, format
31
- // "openai-responses-v1"). The ITEM shape preserves the wire's `id` (a flat blob
32
- // list dropped item identity — the widening's whole point) and a `subtype` from
33
- // wire POSITION: we parse message.reasoning_details, so it is message-attached.
34
- // plurnk is tools-in-body (SPEC §2), so reasoning is never tool-call-attached and
35
- // subtype is constant here; the field is structural, future-proofing the seam.
36
- // An ARRAY of items (not a single object) so N distinct reasoning ids never
37
- // re-collide the identity this fixes. Blobs verbatim, never decoded.
38
- export type EncryptedReasoningItem = { id: string | null; subtype: string; encrypted: Array<{ data: string; format: string | null }> };
39
-
40
- type RawEncrypted = { id: string | null; data: string; format: string | null };
41
-
42
- // Group accumulated encrypted entries into items by wire `id` (id-less entries
43
- // stand alone, never merged). Order preserved.
44
- const groupEncrypted = (entries: Iterable<RawEncrypted>): EncryptedReasoningItem[] => {
45
- const items: EncryptedReasoningItem[] = [];
46
- const byId = new Map<string, EncryptedReasoningItem>();
47
- for (const e of entries) {
48
- const blob = { data: e.data, format: e.format };
49
- if (e.id === null) { items.push({ id: null, subtype: "message", encrypted: [blob] }); continue; }
50
- let item = byId.get(e.id);
51
- if (item === undefined) { item = { id: e.id, subtype: "message", encrypted: [] }; byId.set(e.id, item); items.push(item); }
52
- item.encrypted.push(blob);
53
- }
54
- return items;
55
- };
56
-
57
- const encryptedFromDetails = (details: unknown): EncryptedReasoningItem[] => {
58
- if (!Array.isArray(details)) return [];
59
- const raw: RawEncrypted[] = [];
60
- for (const e of details) {
61
- const entry = e as { type?: unknown; id?: unknown; data?: unknown; format?: unknown };
62
- if (entry?.type !== "reasoning.encrypted" || typeof entry.data !== "string") continue;
63
- raw.push({ id: typeof entry.id === "string" ? entry.id : null, data: entry.data, format: typeof entry.format === "string" ? entry.format : null });
64
- }
65
- return groupEncrypted(raw);
66
- };
67
-
68
- export type StreamResponse = {
69
- model: string | null;
70
- content: string;
71
- reasoning_content: string;
72
- // Sealed relay reasoning (#482) — empty for the open-reasoning backends.
73
- reasoning_encrypted: EncryptedReasoningItem[];
74
- finish_reason: string | null;
75
- usage: RawUsage | null;
76
- chunkMetadata: Record<string, unknown>;
77
- // #36: per-token logprobs parsed from choices[0].logprobs.content[], present
78
- // only when the request asked for them (else the field is absent → null).
79
- logprobs: TokenLogprob[] | null;
80
- // #36: the verbatim response body, populated only when captureRawBody is set.
81
- rawBody: unknown;
82
- };
83
-
84
- // Map an OpenAI-style `logprobs.content[]` array to the canonical structured view
85
- // (#36). Reads the RAW `logprob` (not `sampling_logprob`); `top_logprobs` → `top`.
86
- // Returns null when the shape is absent — never synthesizes.
87
- const parseLogprobs = (raw: unknown): TokenLogprob[] | null => {
88
- const content = (raw as { content?: unknown } | null | undefined)?.content;
89
- if (!Array.isArray(content)) return null;
90
- return content.map((entry) => {
91
- const { token, logprob, top_logprobs } = entry as { token: string; logprob: number; top_logprobs?: unknown };
92
- const top = Array.isArray(top_logprobs)
93
- ? top_logprobs.map((a) => ({ token: (a as TokenLogprob).token, logprob: (a as TokenLogprob).logprob }))
94
- : undefined;
95
- return top !== undefined ? { token, logprob, top } : { token, logprob };
96
- });
97
- };
98
-
99
- // Cloudflare/CDN EDGE status codes (520-527): infrastructure failures the proxy
100
- // returns (as HTML error pages), NOT OpenAI/API statuses. A retry re-incurs the
101
- // same origin wait, so they fail-fast (#543).
102
- const EDGE_LABELS: ReadonlyMap<number, string> = new Map([
103
- [520, "web server returned an unknown error"], [521, "web server is down"],
104
- [522, "connection timed out"], [523, "origin is unreachable"], [524, "origin timeout"],
105
- [525, "SSL handshake failed"], [526, "invalid SSL certificate"], [527, "railgun error"],
106
- ]);
107
- export const isEdgeStatus = (status: number): boolean => status >= 520 && status <= 527;
108
-
109
- export class OpenAiHttpError extends Error {
110
- readonly status: number;
111
- readonly body: string;
112
- readonly retryAfter: number | null;
113
- constructor(status: number, body: string, retryAfter: number | null) {
114
- super(OpenAiHttpError.#describe(status, body));
115
- this.status = status;
116
- this.body = body;
117
- this.retryAfter = retryAfter;
118
- }
119
- // A non-JSON error body (a proxy/CDN HTML page) collapses to one line and drops
120
- // the misleading "OpenAI" prefix - an edge code is not an API status (#543).
121
- // JSON API errors pass through verbatim.
122
- static #describe(status: number, body: string): string {
123
- if (body.trimStart().startsWith("<")) return `${status} ${EDGE_LABELS.get(status) ?? "edge/proxy error"}`;
124
- return `OpenAI ${status} - ${body}`;
125
- }
126
- }
127
-
128
- export class StreamIdleError extends Error {
129
- readonly timeoutMs: number;
130
- constructor(timeoutMs: number) {
131
- super(`stream received no body bytes for ${timeoutMs}ms`);
132
- this.name = "StreamIdleError";
133
- this.timeoutMs = timeoutMs;
134
- }
135
- }
136
-
137
- const parseRetryAfter = (header: string | null): number | null => {
138
- if (header === null) return null;
139
- const asInt = Number.parseInt(header, 10);
140
- if (Number.isFinite(asInt)) return asInt * 1000;
141
- const asDate = Date.parse(header);
142
- if (Number.isFinite(asDate)) return Math.max(0, asDate - Date.now());
143
- return null;
144
- };
145
-
146
- // Non-streaming sibling. Same request/error handling, but POSTs without
147
- // `stream` and parses the single JSON body into the SAME StreamResponse shape.
148
- // For backends whose STREAMING response misbehaves (e.g. Fireworks labels
149
- // grammar-constrained output as `reasoning_content` instead of `content`) —
150
- // the Provider contract is atomic either way, so the transport is free to
151
- // choose. The fetch timeout (AbortSignal) bounds the wait; there is no proxy
152
- // between us and the backend that would idle out a non-streamed request.
153
- export const chatCompletion = async ({ url, headers, body, signal, fetch, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
154
- const response = await fetch(url, {
155
- method: "POST",
156
- headers: { "Content-Type": "application/json", ...headers },
157
- body: JSON.stringify(body),
158
- signal,
159
- });
160
- if (!response.ok) {
161
- const errorBody = await response.text();
162
- throw new OpenAiHttpError(response.status, errorBody, parseRetryAfter(response.headers.get("retry-after")));
163
- }
164
- const j = (await response.json()) as Record<string, unknown>;
165
- const choices = j.choices as Array<Record<string, unknown>> | undefined;
166
- const choice = (choices?.[0] ?? {}) as Record<string, unknown>;
167
- const msg = (choice.message ?? {}) as Record<string, unknown>;
168
- const reasoning = msg.reasoning_content ?? msg.reasoning ?? msg.thinking ?? "";
169
- const chunkMetadata: Record<string, unknown> = {};
170
- for (const [k, v] of Object.entries(j)) if (k !== "choices" && k !== "usage") chunkMetadata[k] = v;
171
- return {
172
- model: typeof j.model === "string" ? j.model : null,
173
- content: typeof msg.content === "string" ? msg.content : "",
174
- reasoning_content: typeof reasoning === "string" ? reasoning : "",
175
- reasoning_encrypted: encryptedFromDetails(msg.reasoning_details),
176
- finish_reason: typeof choice.finish_reason === "string" ? choice.finish_reason : null,
177
- usage: (j.usage ?? null) as StreamResponse["usage"],
178
- chunkMetadata,
179
- logprobs: parseLogprobs(choice.logprobs),
180
- // Non-streamed: the parsed JSON IS the verbatim wire body, exact.
181
- rawBody: captureRawBody === true ? j : undefined,
182
- };
183
- };
184
-
185
- export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody, streamIdleTimeoutMs }: StreamRequest): Promise<StreamResponse> => {
186
- const requestBody = { ...body, stream: true, stream_options: { include_usage: true } };
187
-
188
- const response = await fetch(url, {
189
- method: "POST",
190
- headers: { "Content-Type": "application/json", ...headers },
191
- body: JSON.stringify(requestBody),
192
- signal,
193
- });
194
-
195
- if (!response.ok) {
196
- const errorBody = await response.text();
197
- throw new OpenAiHttpError(response.status, errorBody, parseRetryAfter(response.headers.get("retry-after")));
198
- }
199
-
200
- if (response.body === null) throw new Error("OpenAI response body is null");
201
- const reader = response.body.getReader();
202
- const decoder = new TextDecoder();
203
-
204
- let buffer = "";
205
- let content = "";
206
- let reasoning_content = "";
207
- let usage: StreamResponse["usage"] = null;
208
- let model: string | null = null;
209
- let finish_reason: string | null = null;
210
- const chunkMetadata: Record<string, unknown> = {};
211
- // #36: logprobs stream as per-chunk choices[0].logprobs.content[] deltas —
212
- // accumulate the raw entries across chunks, map once at the end.
213
- const logprobEntries: unknown[] = [];
214
- // #482: encrypted reasoning_details stream chunked — concatenate `data` per
215
- // reassembly key (index when present, else id, else a counter); the id/format
216
- // ride along and items group by id at the end.
217
- const encryptedByKey = new Map<string, RawEncrypted>();
218
- let encryptedNoKey = 0;
219
-
220
- while (true) {
221
- const read = reader.read();
222
- let timer: ReturnType<typeof setTimeout> | undefined;
223
- const idle = streamIdleTimeoutMs !== undefined && streamIdleTimeoutMs > 0
224
- ? new Promise<never>((_resolve, reject) => {
225
- timer = setTimeout(() => reject(new StreamIdleError(streamIdleTimeoutMs)), streamIdleTimeoutMs);
226
- })
227
- : null;
228
- let result: Awaited<ReturnType<typeof reader.read>>;
229
- try {
230
- result = idle === null ? await read : await Promise.race([read, idle]);
231
- } catch (err) {
232
- if (err instanceof StreamIdleError) void reader.cancel(err).catch(() => undefined);
233
- throw err;
234
- } finally {
235
- if (timer !== undefined) clearTimeout(timer);
236
- }
237
- const { done, value } = result;
238
- if (done) break;
239
- buffer += decoder.decode(value, { stream: true });
240
- const lines = buffer.split("\n");
241
- buffer = lines.pop() ?? "";
242
-
243
- for (const rawLine of lines) {
244
- const line = rawLine.trim();
245
- if (!line.startsWith("data:")) continue;
246
- const payload = line.slice(5).trimStart();
247
- if (payload === "[DONE]" || payload === "") continue;
248
-
249
- let chunk: Record<string, unknown>;
250
- try { chunk = JSON.parse(payload) as Record<string, unknown>; } catch { continue; }
251
-
252
- // A streaming server may flush HTTP 200 headers before inference
253
- // completes, then report a terminal failure as an SSE error frame.
254
- // That frame is a failed exchange, never an empty completion.
255
- if (chunk.error !== null && typeof chunk.error === "object") {
256
- const status = typeof chunk.status === "number"
257
- && Number.isInteger(chunk.status)
258
- && chunk.status >= 400
259
- && chunk.status <= 599
260
- ? chunk.status
261
- : 500;
262
- throw new OpenAiHttpError(status, JSON.stringify({ error: chunk.error }), null);
263
- }
264
-
265
- if (typeof chunk.model === "string") model = chunk.model;
266
- if (chunk.usage !== undefined && chunk.usage !== null) usage = chunk.usage as StreamResponse["usage"];
267
-
268
- for (const [k, v] of Object.entries(chunk)) {
269
- if (k === "choices" || k === "usage") continue;
270
- chunkMetadata[k] = v;
271
- }
272
-
273
- const choices = chunk.choices as Array<Record<string, unknown>> | undefined;
274
- const choice = choices?.[0];
275
- if (choice === undefined) continue;
276
- if (typeof choice.finish_reason === "string") finish_reason = choice.finish_reason;
277
-
278
- const chunkLogprobs = (choice.logprobs as { content?: unknown } | undefined)?.content;
279
- if (Array.isArray(chunkLogprobs)) logprobEntries.push(...chunkLogprobs);
280
-
281
- const delta = choice.delta as Record<string, unknown> | undefined;
282
- if (delta === undefined) continue;
283
- if (typeof delta.content === "string") content += delta.content;
284
- // Reasoning surfaces under different field names per provider.
285
- if (typeof delta.reasoning_content === "string") reasoning_content += delta.reasoning_content;
286
- if (typeof delta.reasoning === "string") reasoning_content += delta.reasoning;
287
- if (typeof delta.thinking === "string") reasoning_content += delta.thinking;
288
- if (Array.isArray(delta.reasoning_details)) {
289
- for (const e of delta.reasoning_details) {
290
- const entry = e as { type?: unknown; id?: unknown; data?: unknown; format?: unknown; index?: unknown };
291
- if (entry?.type !== "reasoning.encrypted" || typeof entry.data !== "string") continue;
292
- const id = typeof entry.id === "string" ? entry.id : null;
293
- const key = typeof entry.index === "number" ? `i${entry.index}` : id ?? `n${encryptedNoKey++}`;
294
- const prev = encryptedByKey.get(key);
295
- if (prev !== undefined) prev.data += entry.data;
296
- else encryptedByKey.set(key, { id, data: entry.data, format: typeof entry.format === "string" ? entry.format : null });
297
- }
298
- }
299
- }
300
- }
301
-
302
- const logprobs = logprobEntries.length > 0 ? parseLogprobs({ content: logprobEntries }) : null;
303
- // Streamed turns have no single verbatim wire body; reassemble the equivalent
304
- // (#36) — chunk-level fields (chunkMetadata) + the collected choice — only when
305
- // asked, so serving turns pay nothing.
306
- const rawBody = captureRawBody === true
307
- ? { ...chunkMetadata, model, usage, choices: [{ index: 0, message: { content, reasoning_content }, finish_reason, logprobs: logprobs !== null ? { content: logprobEntries } : null }] }
308
- : undefined;
309
- return { model, content, reasoning_content, reasoning_encrypted: groupEncrypted(encryptedByKey.values()), finish_reason, usage, chunkMetadata, logprobs, rawBody };
310
- };