alchemy 2.0.0-beta.45 → 2.0.0-beta.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/exec.js +18 -10
- package/bin/exec.js.map +1 -1
- package/lib/AlchemyContext.d.ts +0 -1
- package/lib/AlchemyContext.d.ts.map +1 -1
- package/lib/AlchemyContext.js.map +1 -1
- package/lib/Apply.js +2 -2
- package/lib/Apply.js.map +1 -1
- package/lib/Cli/commands/_shared.d.ts +7 -1
- package/lib/Cli/commands/_shared.d.ts.map +1 -1
- package/lib/Cli/commands/_shared.js +1 -0
- package/lib/Cli/commands/_shared.js.map +1 -1
- package/lib/Cli/commands/cloudflare.d.ts.map +1 -1
- package/lib/Cli/commands/cloudflare.js +5 -1
- package/lib/Cli/commands/cloudflare.js.map +1 -1
- package/lib/Cli/commands/login.d.ts +1 -1
- package/lib/Cli/commands/login.d.ts.map +1 -1
- package/lib/Cli/commands/login.js +45 -42
- package/lib/Cli/commands/login.js.map +1 -1
- package/lib/Cli/commands/logs.js +1 -1
- package/lib/Cli/commands/logs.js.map +1 -1
- package/lib/Cli/commands/state.js +1 -1
- package/lib/Cli/commands/state.js.map +1 -1
- package/lib/Cli/commands/tail.js +1 -1
- package/lib/Cli/commands/tail.js.map +1 -1
- package/lib/Cloudflare/AiGateway/AiGateway.d.ts +112 -1
- package/lib/Cloudflare/AiGateway/AiGateway.d.ts.map +1 -1
- package/lib/Cloudflare/AiGateway/AiGateway.js +112 -1
- package/lib/Cloudflare/AiGateway/AiGateway.js.map +1 -1
- package/lib/Cloudflare/AiGateway/AiGatewayBinding.d.ts +43 -10
- package/lib/Cloudflare/AiGateway/AiGatewayBinding.d.ts.map +1 -1
- package/lib/Cloudflare/AiGateway/AiGatewayBinding.js +38 -10
- package/lib/Cloudflare/AiGateway/AiGatewayBinding.js.map +1 -1
- package/lib/Cloudflare/AiGateway/LanguageModel.d.ts +35 -0
- package/lib/Cloudflare/AiGateway/LanguageModel.d.ts.map +1 -0
- package/lib/Cloudflare/AiGateway/LanguageModel.js +576 -0
- package/lib/Cloudflare/AiGateway/LanguageModel.js.map +1 -0
- package/lib/Cloudflare/AiGateway/index.d.ts +1 -0
- package/lib/Cloudflare/AiGateway/index.d.ts.map +1 -1
- package/lib/Cloudflare/AiGateway/index.js +1 -0
- package/lib/Cloudflare/AiGateway/index.js.map +1 -1
- package/lib/Cloudflare/Providers.d.ts.map +1 -1
- package/lib/Cloudflare/Providers.js +5 -1
- package/lib/Cloudflare/Providers.js.map +1 -1
- package/lib/Cloudflare/Queue/QueueConsumer.d.ts.map +1 -1
- package/lib/Cloudflare/Queue/QueueConsumer.js +1 -2
- package/lib/Cloudflare/Queue/QueueConsumer.js.map +1 -1
- package/lib/Cloudflare/StateStore/Api.d.ts +1 -1
- package/lib/Cloudflare/StateStore/Api.js +1 -1
- package/lib/Cloudflare/StateStore/State.d.ts +11 -65
- package/lib/Cloudflare/StateStore/State.d.ts.map +1 -1
- package/lib/Cloudflare/StateStore/State.js +313 -368
- package/lib/Cloudflare/StateStore/State.js.map +1 -1
- package/lib/Cloudflare/StateStore/Store.d.ts.map +1 -1
- package/lib/Cloudflare/StateStore/Store.js +16 -29
- package/lib/Cloudflare/StateStore/Store.js.map +1 -1
- package/lib/Cloudflare/Vectorize/VectorizeIndex.d.ts +114 -0
- package/lib/Cloudflare/Vectorize/VectorizeIndex.d.ts.map +1 -0
- package/lib/Cloudflare/Vectorize/VectorizeIndex.js +157 -0
- package/lib/Cloudflare/Vectorize/VectorizeIndex.js.map +1 -0
- package/lib/Cloudflare/Vectorize/VectorizeIndexBinding.d.ts +38 -0
- package/lib/Cloudflare/Vectorize/VectorizeIndexBinding.d.ts.map +1 -0
- package/lib/Cloudflare/Vectorize/VectorizeIndexBinding.js +44 -0
- package/lib/Cloudflare/Vectorize/VectorizeIndexBinding.js.map +1 -0
- package/lib/Cloudflare/Vectorize/VectorizeMetadataIndex.d.ts +72 -0
- package/lib/Cloudflare/Vectorize/VectorizeMetadataIndex.d.ts.map +1 -0
- package/lib/Cloudflare/Vectorize/VectorizeMetadataIndex.js +131 -0
- package/lib/Cloudflare/Vectorize/VectorizeMetadataIndex.js.map +1 -0
- package/lib/Cloudflare/Vectorize/index.d.ts +4 -0
- package/lib/Cloudflare/Vectorize/index.d.ts.map +1 -0
- package/lib/Cloudflare/Vectorize/index.js +4 -0
- package/lib/Cloudflare/Vectorize/index.js.map +1 -0
- package/lib/Cloudflare/Workers/DurableObjectBridge.d.ts.map +1 -1
- package/lib/Cloudflare/Workers/DurableObjectBridge.js +10 -1
- package/lib/Cloudflare/Workers/DurableObjectBridge.js.map +1 -1
- package/lib/Cloudflare/Workers/LocalWorkerProvider.d.ts.map +1 -1
- package/lib/Cloudflare/Workers/LocalWorkerProvider.js +13 -1
- package/lib/Cloudflare/Workers/LocalWorkerProvider.js.map +1 -1
- package/lib/Cloudflare/Workers/Rpc.d.ts.map +1 -1
- package/lib/Cloudflare/Workers/Rpc.js +20 -2
- package/lib/Cloudflare/Workers/Rpc.js.map +1 -1
- package/lib/Cloudflare/Workers/WorkerAsyncBindings.d.ts.map +1 -1
- package/lib/Cloudflare/Workers/WorkerAsyncBindings.js +8 -0
- package/lib/Cloudflare/Workers/WorkerAsyncBindings.js.map +1 -1
- package/lib/Cloudflare/Workers/WorkerBinding.d.ts +2 -1
- package/lib/Cloudflare/Workers/WorkerBinding.d.ts.map +1 -1
- package/lib/Cloudflare/Workers/WorkerBinding.js.map +1 -1
- package/lib/Cloudflare/Workers/WorkerBridge.d.ts.map +1 -1
- package/lib/Cloudflare/Workers/WorkerBridge.js +20 -2
- package/lib/Cloudflare/Workers/WorkerBridge.js.map +1 -1
- package/lib/Cloudflare/index.d.ts +1 -0
- package/lib/Cloudflare/index.d.ts.map +1 -1
- package/lib/Cloudflare/index.js +1 -0
- package/lib/Cloudflare/index.js.map +1 -1
- package/lib/Output.js +2 -2
- package/lib/Output.js.map +1 -1
- package/lib/Plan.d.ts +2 -2
- package/lib/Plan.d.ts.map +1 -1
- package/lib/Plan.js +3 -3
- package/lib/Plan.js.map +1 -1
- package/lib/Stack.d.ts.map +1 -1
- package/lib/Stack.js +6 -0
- package/lib/Stack.js.map +1 -1
- package/lib/State/HttpStateStore.d.ts +4 -0
- package/lib/State/HttpStateStore.d.ts.map +1 -1
- package/lib/State/HttpStateStore.js +13 -0
- package/lib/State/HttpStateStore.js.map +1 -1
- package/lib/State/InMemoryState.d.ts +2 -1
- package/lib/State/InMemoryState.d.ts.map +1 -1
- package/lib/State/InMemoryState.js +3 -3
- package/lib/State/InMemoryState.js.map +1 -1
- package/lib/State/LocalState.d.ts.map +1 -1
- package/lib/State/LocalState.js +5 -1
- package/lib/State/LocalState.js.map +1 -1
- package/lib/State/State.d.ts +2 -2
- package/lib/State/State.d.ts.map +1 -1
- package/lib/State/State.js.map +1 -1
- package/lib/Util/poll.d.ts +30 -0
- package/lib/Util/poll.d.ts.map +1 -0
- package/lib/Util/poll.js +26 -0
- package/lib/Util/poll.js.map +1 -0
- package/lib/tsconfig.test.tsbuildinfo +1 -1
- package/package.json +7 -7
- package/src/AlchemyContext.ts +0 -1
- package/src/Apply.ts +2 -2
- package/src/Cli/commands/_shared.ts +7 -1
- package/src/Cli/commands/cloudflare.ts +5 -3
- package/src/Cli/commands/login.ts +69 -60
- package/src/Cli/commands/logs.ts +1 -1
- package/src/Cli/commands/state.ts +1 -1
- package/src/Cli/commands/tail.ts +1 -1
- package/src/Cloudflare/AiGateway/AiGateway.ts +112 -1
- package/src/Cloudflare/AiGateway/AiGatewayBinding.ts +52 -10
- package/src/Cloudflare/AiGateway/LanguageModel.ts +907 -0
- package/src/Cloudflare/AiGateway/index.ts +1 -0
- package/src/Cloudflare/Providers.ts +7 -0
- package/src/Cloudflare/Queue/QueueConsumer.ts +0 -2
- package/src/Cloudflare/StateStore/Api.ts +1 -1
- package/src/Cloudflare/StateStore/State.ts +452 -508
- package/src/Cloudflare/StateStore/Store.ts +22 -35
- package/src/Cloudflare/Vectorize/VectorizeIndex.ts +261 -0
- package/src/Cloudflare/Vectorize/VectorizeIndexBinding.ts +103 -0
- package/src/Cloudflare/Vectorize/VectorizeMetadataIndex.ts +208 -0
- package/src/Cloudflare/Vectorize/index.ts +3 -0
- package/src/Cloudflare/Workers/DurableObjectBridge.ts +10 -1
- package/src/Cloudflare/Workers/LocalWorkerProvider.ts +13 -1
- package/src/Cloudflare/Workers/Rpc.ts +38 -9
- package/src/Cloudflare/Workers/WorkerAsyncBindings.ts +7 -0
- package/src/Cloudflare/Workers/WorkerBinding.ts +2 -0
- package/src/Cloudflare/Workers/WorkerBridge.ts +20 -2
- package/src/Cloudflare/index.ts +1 -0
- package/src/Output.ts +2 -2
- package/src/Plan.ts +4 -4
- package/src/Stack.ts +6 -0
- package/src/State/HttpStateStore.ts +28 -0
- package/src/State/InMemoryState.ts +87 -85
- package/src/State/LocalState.ts +15 -1
- package/src/State/State.ts +5 -4
- package/src/Util/poll.ts +47 -0
|
@@ -0,0 +1,907 @@
|
|
|
1
|
+
import * as Effect from "effect/Effect";
|
|
2
|
+
import * as Layer from "effect/Layer";
|
|
3
|
+
import * as Stream from "effect/Stream";
|
|
4
|
+
import {
|
|
5
|
+
AiError,
|
|
6
|
+
LanguageModel as AiLanguageModel,
|
|
7
|
+
IdGenerator,
|
|
8
|
+
Prompt,
|
|
9
|
+
Response,
|
|
10
|
+
Tool,
|
|
11
|
+
} from "effect/unstable/ai";
|
|
12
|
+
import * as Sse from "effect/unstable/encoding/Sse";
|
|
13
|
+
import type { RuntimeContext } from "../../RuntimeContext.ts";
|
|
14
|
+
import type { AiGatewayClient } from "./AiGatewayBinding.ts";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Options for constructing an AI Gateway-backed Workers AI LanguageModel.
|
|
18
|
+
*/
|
|
19
|
+
export interface LanguageModelOptions {
|
|
20
|
+
/** Already-bound AI Gateway client from `AiGatewayBinding.bind(gateway)`. */
|
|
21
|
+
readonly client: AiGatewayClient;
|
|
22
|
+
/** Workers AI model id, e.g. `@cf/meta/llama-3.3-70b-instruct-fp8-fast`. */
|
|
23
|
+
readonly model: string;
|
|
24
|
+
/** Optional per-call defaults; overridable per request via `providerOptions`. */
|
|
25
|
+
readonly parameters?: {
|
|
26
|
+
readonly temperature?: number;
|
|
27
|
+
readonly maxTokens?: number;
|
|
28
|
+
readonly topP?: number;
|
|
29
|
+
readonly topK?: number;
|
|
30
|
+
readonly seed?: number;
|
|
31
|
+
readonly frequencyPenalty?: number;
|
|
32
|
+
readonly presencePenalty?: number;
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Provide a {@link AiLanguageModel.LanguageModel} layer backed by the supplied
|
|
38
|
+
* AI Gateway client and Workers AI model.
|
|
39
|
+
*/
|
|
40
|
+
export const makeLanguageModelLayer = (
|
|
41
|
+
options: LanguageModelOptions,
|
|
42
|
+
): Layer.Layer<AiLanguageModel.LanguageModel, never, RuntimeContext> =>
|
|
43
|
+
Layer.effect(AiLanguageModel.LanguageModel, makeLanguageModel(options));
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Build a {@link AiLanguageModel.Service} that proxies generateText/streamText
|
|
47
|
+
* through the supplied AI Gateway client to a Workers AI model.
|
|
48
|
+
*/
|
|
49
|
+
export const makeLanguageModel = ({
|
|
50
|
+
client,
|
|
51
|
+
model,
|
|
52
|
+
parameters,
|
|
53
|
+
}: LanguageModelOptions): Effect.Effect<
|
|
54
|
+
AiLanguageModel.Service,
|
|
55
|
+
never,
|
|
56
|
+
RuntimeContext
|
|
57
|
+
> =>
|
|
58
|
+
Effect.gen(function* () {
|
|
59
|
+
const ai = yield* client.raw;
|
|
60
|
+
const gatewayId = yield* client.id;
|
|
61
|
+
|
|
62
|
+
const callRaw = (
|
|
63
|
+
body: WorkersAiInputs,
|
|
64
|
+
method: "generateText" | "streamText",
|
|
65
|
+
): Effect.Effect<Response, AiError.AiError> =>
|
|
66
|
+
Effect.tryPromise({
|
|
67
|
+
try: () =>
|
|
68
|
+
ai.run(
|
|
69
|
+
model as keyof AiModels,
|
|
70
|
+
body as unknown as AiModels[keyof AiModels]["inputs"],
|
|
71
|
+
{
|
|
72
|
+
gateway: { id: gatewayId },
|
|
73
|
+
returnRawResponse: true,
|
|
74
|
+
},
|
|
75
|
+
),
|
|
76
|
+
catch: (cause) => toAiError(cause, method),
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
return yield* AiLanguageModel.make({
|
|
80
|
+
generateText: (options) =>
|
|
81
|
+
Effect.gen(function* () {
|
|
82
|
+
const body = toRequestBody({ options, parameters, stream: false });
|
|
83
|
+
const resp = yield* callRaw(body, "generateText");
|
|
84
|
+
const json = yield* Effect.tryPromise({
|
|
85
|
+
try: () => resp.json() as Promise<Record<string, unknown>>,
|
|
86
|
+
catch: (cause) => toAiError(cause, "generateText"),
|
|
87
|
+
});
|
|
88
|
+
return yield* parseGenerateText(json);
|
|
89
|
+
}),
|
|
90
|
+
streamText: (options) =>
|
|
91
|
+
Stream.unwrap(
|
|
92
|
+
Effect.gen(function* () {
|
|
93
|
+
const idGen = yield* IdGenerator.IdGenerator;
|
|
94
|
+
const body = toRequestBody({ options, parameters, stream: true });
|
|
95
|
+
const resp = yield* callRaw(body, "streamText");
|
|
96
|
+
return parseStreamText(resp, idGen);
|
|
97
|
+
}),
|
|
98
|
+
),
|
|
99
|
+
});
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
// ---------------------------------------------------------------------------
|
|
103
|
+
// Wire format types (Workers AI request)
|
|
104
|
+
//
|
|
105
|
+
// Workers AI returns two response shapes depending on the model:
|
|
106
|
+
// - Native: { response: "...", tool_calls: [...] , usage }
|
|
107
|
+
// - OpenAI: { choices: [{ message: { content, tool_calls, reasoning_content } }], usage }
|
|
108
|
+
//
|
|
109
|
+
// We accept both defensively — schemas would over-constrain.
|
|
110
|
+
// ---------------------------------------------------------------------------
|
|
111
|
+
|
|
112
|
+
interface WorkersAiMessage {
|
|
113
|
+
readonly role: "system" | "user" | "assistant" | "tool";
|
|
114
|
+
readonly content?: unknown;
|
|
115
|
+
readonly name?: string;
|
|
116
|
+
readonly tool_call_id?: string;
|
|
117
|
+
readonly tool_calls?: ReadonlyArray<{
|
|
118
|
+
readonly id: string;
|
|
119
|
+
readonly type: "function";
|
|
120
|
+
readonly function: { readonly name: string; readonly arguments: string };
|
|
121
|
+
}>;
|
|
122
|
+
readonly reasoning?: string;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
interface WorkersAiToolDef {
|
|
126
|
+
readonly type: "function";
|
|
127
|
+
readonly function: {
|
|
128
|
+
readonly name: string;
|
|
129
|
+
readonly description?: string;
|
|
130
|
+
readonly parameters: unknown;
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
type WorkersAiToolChoice =
|
|
135
|
+
| "auto"
|
|
136
|
+
| "required"
|
|
137
|
+
| "none"
|
|
138
|
+
| { readonly type: "function"; readonly function: { readonly name: string } };
|
|
139
|
+
|
|
140
|
+
interface WorkersAiInputs {
|
|
141
|
+
readonly messages: ReadonlyArray<WorkersAiMessage>;
|
|
142
|
+
readonly tools?: ReadonlyArray<WorkersAiToolDef>;
|
|
143
|
+
readonly tool_choice?: WorkersAiToolChoice;
|
|
144
|
+
readonly stream?: boolean;
|
|
145
|
+
readonly stream_options?: { readonly include_usage: boolean };
|
|
146
|
+
readonly max_tokens?: number;
|
|
147
|
+
readonly temperature?: number;
|
|
148
|
+
readonly top_p?: number;
|
|
149
|
+
readonly top_k?: number;
|
|
150
|
+
readonly random_seed?: number;
|
|
151
|
+
readonly frequency_penalty?: number;
|
|
152
|
+
readonly presence_penalty?: number;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
// ---------------------------------------------------------------------------
|
|
156
|
+
// Prompt → Workers AI messages (pure, no .push mutation)
|
|
157
|
+
// ---------------------------------------------------------------------------
|
|
158
|
+
|
|
159
|
+
const uint8ArrayToBase64 = (bytes: Uint8Array): string => {
|
|
160
|
+
let binary = "";
|
|
161
|
+
const chunkSize = 8192;
|
|
162
|
+
for (let i = 0; i < bytes.length; i += chunkSize) {
|
|
163
|
+
const chunk = bytes.subarray(i, Math.min(i + chunkSize, bytes.length));
|
|
164
|
+
binary += String.fromCharCode(...chunk);
|
|
165
|
+
}
|
|
166
|
+
return btoa(binary);
|
|
167
|
+
};
|
|
168
|
+
|
|
169
|
+
const fileToImageUrl = (
|
|
170
|
+
data: string | Uint8Array | URL,
|
|
171
|
+
mediaType: string,
|
|
172
|
+
): string => {
|
|
173
|
+
if (data instanceof URL) return data.toString();
|
|
174
|
+
if (data instanceof Uint8Array) {
|
|
175
|
+
return `data:${mediaType};base64,${uint8ArrayToBase64(data)}`;
|
|
176
|
+
}
|
|
177
|
+
if (data.startsWith("data:") || data.startsWith("http")) return data;
|
|
178
|
+
return `data:${mediaType};base64,${data}`;
|
|
179
|
+
};
|
|
180
|
+
|
|
181
|
+
const convertPromptToMessages = (
|
|
182
|
+
prompt: Prompt.Prompt,
|
|
183
|
+
): ReadonlyArray<WorkersAiMessage> =>
|
|
184
|
+
prompt.content.flatMap((m): ReadonlyArray<WorkersAiMessage> => {
|
|
185
|
+
switch (m.role) {
|
|
186
|
+
case "system":
|
|
187
|
+
return [{ role: "system", content: m.content }];
|
|
188
|
+
case "user":
|
|
189
|
+
return [toUserMessage(m.content)];
|
|
190
|
+
case "assistant":
|
|
191
|
+
return [toAssistantMessage(m.content)];
|
|
192
|
+
case "tool":
|
|
193
|
+
return m.content.flatMap(toToolMessage);
|
|
194
|
+
}
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
const toUserMessage = (
|
|
198
|
+
parts: Prompt.UserMessage["content"],
|
|
199
|
+
): WorkersAiMessage => {
|
|
200
|
+
const text = parts
|
|
201
|
+
.flatMap((p) => (p.type === "text" ? [p.text] : []))
|
|
202
|
+
.join("\n");
|
|
203
|
+
const images = parts.flatMap((p) =>
|
|
204
|
+
p.type === "file"
|
|
205
|
+
? [
|
|
206
|
+
{
|
|
207
|
+
type: "image_url" as const,
|
|
208
|
+
image_url: { url: fileToImageUrl(p.data, p.mediaType) },
|
|
209
|
+
},
|
|
210
|
+
]
|
|
211
|
+
: [],
|
|
212
|
+
);
|
|
213
|
+
if (images.length === 0) return { role: "user", content: text };
|
|
214
|
+
return {
|
|
215
|
+
role: "user",
|
|
216
|
+
content: [
|
|
217
|
+
...(text.length > 0 ? [{ type: "text" as const, text }] : []),
|
|
218
|
+
...images,
|
|
219
|
+
],
|
|
220
|
+
};
|
|
221
|
+
};
|
|
222
|
+
|
|
223
|
+
const toAssistantMessage = (
|
|
224
|
+
parts: Prompt.AssistantMessage["content"],
|
|
225
|
+
): WorkersAiMessage => {
|
|
226
|
+
const text = parts
|
|
227
|
+
.flatMap((p) => (p.type === "text" ? [p.text] : []))
|
|
228
|
+
.join("");
|
|
229
|
+
const reasoning = parts
|
|
230
|
+
.flatMap((p) => (p.type === "reasoning" ? [p.text] : []))
|
|
231
|
+
.join("");
|
|
232
|
+
const toolCalls = parts.flatMap((p) =>
|
|
233
|
+
p.type === "tool-call"
|
|
234
|
+
? [
|
|
235
|
+
{
|
|
236
|
+
id: p.id,
|
|
237
|
+
type: "function" as const,
|
|
238
|
+
function: {
|
|
239
|
+
name: p.name,
|
|
240
|
+
arguments:
|
|
241
|
+
typeof p.params === "string"
|
|
242
|
+
? p.params
|
|
243
|
+
: JSON.stringify(p.params ?? {}),
|
|
244
|
+
},
|
|
245
|
+
},
|
|
246
|
+
]
|
|
247
|
+
: [],
|
|
248
|
+
);
|
|
249
|
+
return {
|
|
250
|
+
role: "assistant",
|
|
251
|
+
content: text,
|
|
252
|
+
...(reasoning ? { reasoning } : {}),
|
|
253
|
+
...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),
|
|
254
|
+
};
|
|
255
|
+
};
|
|
256
|
+
|
|
257
|
+
const toToolMessage = (
|
|
258
|
+
part: Prompt.ToolMessage["content"][number],
|
|
259
|
+
): ReadonlyArray<WorkersAiMessage> =>
|
|
260
|
+
part.type === "tool-result"
|
|
261
|
+
? [
|
|
262
|
+
{
|
|
263
|
+
role: "tool",
|
|
264
|
+
name: part.name,
|
|
265
|
+
tool_call_id: part.id,
|
|
266
|
+
content:
|
|
267
|
+
typeof part.result === "string"
|
|
268
|
+
? part.result
|
|
269
|
+
: JSON.stringify(part.result),
|
|
270
|
+
},
|
|
271
|
+
]
|
|
272
|
+
: [];
|
|
273
|
+
|
|
274
|
+
// ---------------------------------------------------------------------------
|
|
275
|
+
// Tools / tool_choice
|
|
276
|
+
// ---------------------------------------------------------------------------
|
|
277
|
+
|
|
278
|
+
const prepareTools = (
|
|
279
|
+
tools: ReadonlyArray<Tool.Any>,
|
|
280
|
+
toolChoice: AiLanguageModel.ProviderOptions["toolChoice"],
|
|
281
|
+
): {
|
|
282
|
+
tools?: ReadonlyArray<WorkersAiToolDef>;
|
|
283
|
+
tool_choice?: WorkersAiToolChoice;
|
|
284
|
+
} => {
|
|
285
|
+
if (tools.length === 0) return {};
|
|
286
|
+
const mapped: ReadonlyArray<WorkersAiToolDef> = tools.map((tool) => ({
|
|
287
|
+
type: "function",
|
|
288
|
+
function: {
|
|
289
|
+
name: tool.name,
|
|
290
|
+
description: Tool.getDescription(tool),
|
|
291
|
+
parameters: Tool.getJsonSchema(tool),
|
|
292
|
+
},
|
|
293
|
+
}));
|
|
294
|
+
|
|
295
|
+
if (toolChoice === "auto" || toolChoice == null) {
|
|
296
|
+
return { tools: mapped, tool_choice: "auto" };
|
|
297
|
+
}
|
|
298
|
+
if (toolChoice === "none") return { tools: mapped, tool_choice: "none" };
|
|
299
|
+
if (toolChoice === "required") {
|
|
300
|
+
return { tools: mapped, tool_choice: "required" };
|
|
301
|
+
}
|
|
302
|
+
if (typeof toolChoice === "object" && "tool" in toolChoice) {
|
|
303
|
+
return {
|
|
304
|
+
tools: mapped.filter((t) => t.function.name === toolChoice.tool),
|
|
305
|
+
tool_choice: "required",
|
|
306
|
+
};
|
|
307
|
+
}
|
|
308
|
+
if (typeof toolChoice === "object" && "oneOf" in toolChoice) {
|
|
309
|
+
const allowed = new Set(toolChoice.oneOf);
|
|
310
|
+
return {
|
|
311
|
+
tools: mapped.filter((t) => allowed.has(t.function.name)),
|
|
312
|
+
tool_choice: toolChoice.mode === "required" ? "required" : "auto",
|
|
313
|
+
};
|
|
314
|
+
}
|
|
315
|
+
return { tools: mapped, tool_choice: "auto" };
|
|
316
|
+
};
|
|
317
|
+
|
|
318
|
+
// ---------------------------------------------------------------------------
|
|
319
|
+
// Request body
|
|
320
|
+
// ---------------------------------------------------------------------------
|
|
321
|
+
|
|
322
|
+
const toRequestBody = ({
|
|
323
|
+
options,
|
|
324
|
+
parameters,
|
|
325
|
+
stream,
|
|
326
|
+
}: {
|
|
327
|
+
readonly options: AiLanguageModel.ProviderOptions;
|
|
328
|
+
readonly parameters: LanguageModelOptions["parameters"];
|
|
329
|
+
readonly stream: boolean;
|
|
330
|
+
}): WorkersAiInputs => {
|
|
331
|
+
const messages = convertPromptToMessages(options.prompt);
|
|
332
|
+
const { tools, tool_choice } = prepareTools(
|
|
333
|
+
options.tools,
|
|
334
|
+
options.toolChoice,
|
|
335
|
+
);
|
|
336
|
+
return {
|
|
337
|
+
messages,
|
|
338
|
+
...(tools !== undefined ? { tools } : {}),
|
|
339
|
+
...(tool_choice !== undefined ? { tool_choice } : {}),
|
|
340
|
+
// `stream_options.include_usage` is the OpenAI-compatible opt-in for
|
|
341
|
+
// usage tokens to appear in the final streamed chunk. Without it most
|
|
342
|
+
// Workers AI models omit `usage` from the stream entirely, leaving the
|
|
343
|
+
// `finish` part with zeroed counts.
|
|
344
|
+
...(stream
|
|
345
|
+
? { stream: true, stream_options: { include_usage: true } }
|
|
346
|
+
: {}),
|
|
347
|
+
...(parameters?.maxTokens !== undefined
|
|
348
|
+
? { max_tokens: parameters.maxTokens }
|
|
349
|
+
: {}),
|
|
350
|
+
...(parameters?.temperature !== undefined
|
|
351
|
+
? { temperature: parameters.temperature }
|
|
352
|
+
: {}),
|
|
353
|
+
...(parameters?.topP !== undefined ? { top_p: parameters.topP } : {}),
|
|
354
|
+
...(parameters?.topK !== undefined ? { top_k: parameters.topK } : {}),
|
|
355
|
+
...(parameters?.seed !== undefined ? { random_seed: parameters.seed } : {}),
|
|
356
|
+
...(parameters?.frequencyPenalty !== undefined
|
|
357
|
+
? { frequency_penalty: parameters.frequencyPenalty }
|
|
358
|
+
: {}),
|
|
359
|
+
...(parameters?.presencePenalty !== undefined
|
|
360
|
+
? { presence_penalty: parameters.presencePenalty }
|
|
361
|
+
: {}),
|
|
362
|
+
};
|
|
363
|
+
};
|
|
364
|
+
|
|
365
|
+
// ---------------------------------------------------------------------------
|
|
366
|
+
// Finish reason / usage mapping
|
|
367
|
+
// ---------------------------------------------------------------------------
|
|
368
|
+
|
|
369
|
+
const mapFinishReason = (raw: unknown): Response.FinishReason => {
|
|
370
|
+
switch (raw) {
|
|
371
|
+
case "stop":
|
|
372
|
+
return "stop";
|
|
373
|
+
case "length":
|
|
374
|
+
case "model_length":
|
|
375
|
+
return "length";
|
|
376
|
+
case "tool_calls":
|
|
377
|
+
return "tool-calls";
|
|
378
|
+
case "content_filter":
|
|
379
|
+
case "content-filter":
|
|
380
|
+
return "content-filter";
|
|
381
|
+
case "error":
|
|
382
|
+
return "error";
|
|
383
|
+
case undefined:
|
|
384
|
+
case null:
|
|
385
|
+
return "unknown";
|
|
386
|
+
default:
|
|
387
|
+
return "other";
|
|
388
|
+
}
|
|
389
|
+
};
|
|
390
|
+
|
|
391
|
+
const mapUsage = (raw: Record<string, unknown> | undefined): Response.Usage => {
|
|
392
|
+
const usage = (raw?.usage as Record<string, unknown> | undefined) ?? {};
|
|
393
|
+
const promptTokens = (usage.prompt_tokens as number | undefined) ?? 0;
|
|
394
|
+
const completionTokens = (usage.completion_tokens as number | undefined) ?? 0;
|
|
395
|
+
const cached = (
|
|
396
|
+
usage.prompt_tokens_details as { cached_tokens?: number } | undefined
|
|
397
|
+
)?.cached_tokens;
|
|
398
|
+
// Construct an actual `Response.Usage` instance — `Schema.Class<Usage>`
|
|
399
|
+
// encodes by going through the class constructor / `isInstance` check, so a
|
|
400
|
+
// plain struct that "matches" the encoded shape isn't enough.
|
|
401
|
+
return new Response.Usage({
|
|
402
|
+
inputTokens: {
|
|
403
|
+
uncached:
|
|
404
|
+
cached !== undefined
|
|
405
|
+
? Math.max(0, promptTokens - cached)
|
|
406
|
+
: promptTokens,
|
|
407
|
+
total: promptTokens,
|
|
408
|
+
cacheRead: cached ?? 0,
|
|
409
|
+
cacheWrite: 0,
|
|
410
|
+
},
|
|
411
|
+
outputTokens: {
|
|
412
|
+
total: completionTokens,
|
|
413
|
+
text: 0,
|
|
414
|
+
reasoning: 0,
|
|
415
|
+
},
|
|
416
|
+
});
|
|
417
|
+
};
|
|
418
|
+
|
|
419
|
+
// ---------------------------------------------------------------------------
|
|
420
|
+
// generateText: JSON → Response.PartEncoded[]
|
|
421
|
+
//
|
|
422
|
+
// Normalize the dual-shape (native + OpenAI) response into a single
|
|
423
|
+
// `DecodedResponse` once, then build the part list with pure spreads.
|
|
424
|
+
// ---------------------------------------------------------------------------
|
|
425
|
+
|
|
426
|
+
interface DecodedToolCall {
|
|
427
|
+
readonly rawId: string;
|
|
428
|
+
readonly name: string;
|
|
429
|
+
readonly arguments: unknown;
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
interface DecodedResponse {
|
|
433
|
+
readonly text: string | undefined;
|
|
434
|
+
readonly reasoning: string | undefined;
|
|
435
|
+
readonly toolCalls: ReadonlyArray<DecodedToolCall>;
|
|
436
|
+
readonly finishReason: string | undefined;
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
const decodeResponse = (raw: Record<string, unknown>): DecodedResponse => {
|
|
440
|
+
const choice = (
|
|
441
|
+
raw.choices as
|
|
442
|
+
| Array<{
|
|
443
|
+
message?: {
|
|
444
|
+
content?: string | null;
|
|
445
|
+
reasoning_content?: string;
|
|
446
|
+
reasoning?: string;
|
|
447
|
+
tool_calls?: ReadonlyArray<Record<string, unknown>>;
|
|
448
|
+
};
|
|
449
|
+
finish_reason?: string;
|
|
450
|
+
}>
|
|
451
|
+
| undefined
|
|
452
|
+
)?.[0];
|
|
453
|
+
const message = choice?.message;
|
|
454
|
+
|
|
455
|
+
const openAiText = message?.content;
|
|
456
|
+
const text =
|
|
457
|
+
typeof openAiText === "string" && openAiText.length > 0
|
|
458
|
+
? openAiText
|
|
459
|
+
: nativeTextOf(raw.response);
|
|
460
|
+
const reasoning = message?.reasoning_content ?? message?.reasoning;
|
|
461
|
+
const rawToolCalls =
|
|
462
|
+
message?.tool_calls ??
|
|
463
|
+
(Array.isArray(raw.tool_calls)
|
|
464
|
+
? (raw.tool_calls as ReadonlyArray<Record<string, unknown>>)
|
|
465
|
+
: []);
|
|
466
|
+
|
|
467
|
+
return {
|
|
468
|
+
text,
|
|
469
|
+
reasoning: reasoning && reasoning.length > 0 ? reasoning : undefined,
|
|
470
|
+
toolCalls: rawToolCalls.flatMap(decodeToolCall),
|
|
471
|
+
finishReason:
|
|
472
|
+
choice?.finish_reason ?? (raw.finish_reason as string | undefined),
|
|
473
|
+
};
|
|
474
|
+
};
|
|
475
|
+
|
|
476
|
+
const nativeTextOf = (raw: unknown): string | undefined => {
|
|
477
|
+
if (raw == null) return undefined;
|
|
478
|
+
if (typeof raw === "object") return JSON.stringify(raw);
|
|
479
|
+
const text = String(raw);
|
|
480
|
+
return text.length > 0 ? text : undefined;
|
|
481
|
+
};
|
|
482
|
+
|
|
483
|
+
const decodeToolCall = (
|
|
484
|
+
tc: Record<string, unknown>,
|
|
485
|
+
): ReadonlyArray<DecodedToolCall> => {
|
|
486
|
+
const fn = tc.function as { name?: string; arguments?: unknown } | undefined;
|
|
487
|
+
const rawId = (tc.id as string | undefined) ?? "";
|
|
488
|
+
if (fn?.name) {
|
|
489
|
+
return [{ rawId, name: fn.name, arguments: fn.arguments ?? "" }];
|
|
490
|
+
}
|
|
491
|
+
const flatName = tc.name as string | undefined;
|
|
492
|
+
if (flatName) {
|
|
493
|
+
return [{ rawId, name: flatName, arguments: tc.arguments ?? "" }];
|
|
494
|
+
}
|
|
495
|
+
return [];
|
|
496
|
+
};
|
|
497
|
+
|
|
498
|
+
const tryParseJsonArgs = (raw: unknown): unknown => {
|
|
499
|
+
if (typeof raw !== "string") return raw;
|
|
500
|
+
try {
|
|
501
|
+
return JSON.parse(raw);
|
|
502
|
+
} catch {
|
|
503
|
+
// Leave as raw string; the framework's tool-result decoder will fail loudly.
|
|
504
|
+
return raw;
|
|
505
|
+
}
|
|
506
|
+
};
|
|
507
|
+
|
|
508
|
+
const parseGenerateText = Effect.fnUntraced(function* (
|
|
509
|
+
raw: Record<string, unknown>,
|
|
510
|
+
) {
|
|
511
|
+
const idGen = yield* IdGenerator.IdGenerator;
|
|
512
|
+
const decoded = decodeResponse(raw);
|
|
513
|
+
|
|
514
|
+
const toolCallParts = yield* Effect.forEach(decoded.toolCalls, (tc) =>
|
|
515
|
+
Effect.gen(function* () {
|
|
516
|
+
const id = tc.rawId || (yield* idGen.generateId());
|
|
517
|
+
return {
|
|
518
|
+
type: "tool-call" as const,
|
|
519
|
+
id,
|
|
520
|
+
name: tc.name,
|
|
521
|
+
params: tryParseJsonArgs(tc.arguments),
|
|
522
|
+
};
|
|
523
|
+
}),
|
|
524
|
+
);
|
|
525
|
+
|
|
526
|
+
const finish = mapFinishReason(
|
|
527
|
+
decoded.finishReason ??
|
|
528
|
+
(decoded.toolCalls.length > 0 ? "tool_calls" : "stop"),
|
|
529
|
+
);
|
|
530
|
+
|
|
531
|
+
return [
|
|
532
|
+
...(decoded.reasoning !== undefined
|
|
533
|
+
? [{ type: "reasoning" as const, text: decoded.reasoning }]
|
|
534
|
+
: []),
|
|
535
|
+
...(decoded.text !== undefined && decoded.text.length > 0
|
|
536
|
+
? [{ type: "text" as const, text: decoded.text }]
|
|
537
|
+
: []),
|
|
538
|
+
...toolCallParts,
|
|
539
|
+
{
|
|
540
|
+
type: "finish" as const,
|
|
541
|
+
reason: finish,
|
|
542
|
+
usage: mapUsage(raw),
|
|
543
|
+
response: undefined,
|
|
544
|
+
},
|
|
545
|
+
] satisfies ReadonlyArray<Response.PartEncoded>;
|
|
546
|
+
});
|
|
547
|
+
|
|
548
|
+
// ---------------------------------------------------------------------------
|
|
549
|
+
// streamText: SSE byte stream → Stream<Response.StreamPartEncoded>
|
|
550
|
+
//
|
|
551
|
+
// Immutable `StreamState` is threaded through `Stream.mapAccumEffect`. The
|
|
552
|
+
// per-chunk output buffer (`parts: Array<StreamPartEncoded>`) is mutable for
|
|
553
|
+
// performance — it's scoped to one chunk, never escapes the handler, and lets
|
|
554
|
+
// us avoid the O(n²) array-spread that pure threading would force in the hot
|
|
555
|
+
// path. This matches the pattern Effect's own `@effect/ai-*` adapters use.
|
|
556
|
+
// ---------------------------------------------------------------------------
|
|
557
|
+
|
|
558
|
+
interface StreamState {
|
|
559
|
+
readonly textId: string | undefined;
|
|
560
|
+
readonly reasoningId: string | undefined;
|
|
561
|
+
readonly toolCalls: ReadonlyMap<
|
|
562
|
+
number,
|
|
563
|
+
{ readonly id: string; readonly name: string }
|
|
564
|
+
>;
|
|
565
|
+
readonly lastToolIndex: number | undefined;
|
|
566
|
+
readonly closedToolIndices: ReadonlySet<number>;
|
|
567
|
+
readonly usage: Record<string, unknown> | undefined;
|
|
568
|
+
readonly finishReason: string | undefined;
|
|
569
|
+
readonly receivedAnyData: boolean;
|
|
570
|
+
readonly receivedDone: boolean;
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
const initialStreamState = (): StreamState => ({
|
|
574
|
+
textId: undefined,
|
|
575
|
+
reasoningId: undefined,
|
|
576
|
+
toolCalls: new Map(),
|
|
577
|
+
lastToolIndex: undefined,
|
|
578
|
+
closedToolIndices: new Set(),
|
|
579
|
+
usage: undefined,
|
|
580
|
+
finishReason: undefined,
|
|
581
|
+
receivedAnyData: false,
|
|
582
|
+
receivedDone: false,
|
|
583
|
+
});
|
|
584
|
+
|
|
585
|
+
type StreamParts = Array<Response.StreamPartEncoded>;
|
|
586
|
+
|
|
587
|
+
const tryParseJson = (data: string): Record<string, unknown> | undefined => {
|
|
588
|
+
try {
|
|
589
|
+
const v = JSON.parse(data);
|
|
590
|
+
return v && typeof v === "object"
|
|
591
|
+
? (v as Record<string, unknown>)
|
|
592
|
+
: undefined;
|
|
593
|
+
} catch {
|
|
594
|
+
return undefined;
|
|
595
|
+
}
|
|
596
|
+
};
|
|
597
|
+
|
|
598
|
+
const isNullFinalizationToolCall = (tc: Record<string, unknown>): boolean => {
|
|
599
|
+
const fn = tc.function as Record<string, unknown> | undefined;
|
|
600
|
+
const name = fn?.name ?? tc.name ?? null;
|
|
601
|
+
const args = fn?.arguments ?? tc.arguments ?? null;
|
|
602
|
+
const id = tc.id ?? null;
|
|
603
|
+
return !id && !name && (!args || args === "");
|
|
604
|
+
};
|
|
605
|
+
|
|
606
|
+
const closeReasoning = (
|
|
607
|
+
state: StreamState,
|
|
608
|
+
parts: StreamParts,
|
|
609
|
+
): StreamState => {
|
|
610
|
+
if (state.reasoningId === undefined) return state;
|
|
611
|
+
parts.push({ type: "reasoning-end", id: state.reasoningId });
|
|
612
|
+
return { ...state, reasoningId: undefined };
|
|
613
|
+
};
|
|
614
|
+
|
|
615
|
+
const closeToolCall = (
|
|
616
|
+
state: StreamState,
|
|
617
|
+
index: number,
|
|
618
|
+
parts: StreamParts,
|
|
619
|
+
): StreamState => {
|
|
620
|
+
if (state.closedToolIndices.has(index)) return state;
|
|
621
|
+
const tc = state.toolCalls.get(index);
|
|
622
|
+
if (!tc) return state;
|
|
623
|
+
parts.push({ type: "tool-params-end", id: tc.id });
|
|
624
|
+
const closed = new Set(state.closedToolIndices);
|
|
625
|
+
closed.add(index);
|
|
626
|
+
return { ...state, closedToolIndices: closed };
|
|
627
|
+
};
|
|
628
|
+
|
|
629
|
+
const emitTextDelta = (
|
|
630
|
+
state: StreamState,
|
|
631
|
+
delta: string,
|
|
632
|
+
parts: StreamParts,
|
|
633
|
+
idGen: IdGenerator.Service,
|
|
634
|
+
): Effect.Effect<StreamState> =>
|
|
635
|
+
Effect.gen(function* () {
|
|
636
|
+
let s = closeReasoning(state, parts);
|
|
637
|
+
if (s.textId === undefined) {
|
|
638
|
+
const id = yield* idGen.generateId();
|
|
639
|
+
parts.push({ type: "text-start", id });
|
|
640
|
+
s = { ...s, textId: id };
|
|
641
|
+
}
|
|
642
|
+
parts.push({ type: "text-delta", id: s.textId!, delta });
|
|
643
|
+
return s;
|
|
644
|
+
});
|
|
645
|
+
|
|
646
|
+
const emitReasoningDelta = (
|
|
647
|
+
state: StreamState,
|
|
648
|
+
delta: string,
|
|
649
|
+
parts: StreamParts,
|
|
650
|
+
idGen: IdGenerator.Service,
|
|
651
|
+
): Effect.Effect<StreamState> =>
|
|
652
|
+
Effect.gen(function* () {
|
|
653
|
+
let s = state;
|
|
654
|
+
if (s.reasoningId === undefined) {
|
|
655
|
+
const id = yield* idGen.generateId();
|
|
656
|
+
parts.push({ type: "reasoning-start", id });
|
|
657
|
+
s = { ...s, reasoningId: id };
|
|
658
|
+
}
|
|
659
|
+
parts.push({ type: "reasoning-delta", id: s.reasoningId!, delta });
|
|
660
|
+
return s;
|
|
661
|
+
});
|
|
662
|
+
|
|
663
|
+
const handleToolDeltas = (
|
|
664
|
+
state: StreamState,
|
|
665
|
+
deltas: ReadonlyArray<Record<string, unknown>>,
|
|
666
|
+
parts: StreamParts,
|
|
667
|
+
idGen: IdGenerator.Service,
|
|
668
|
+
): Effect.Effect<StreamState> =>
|
|
669
|
+
Effect.gen(function* () {
|
|
670
|
+
let s = state;
|
|
671
|
+
for (const d of deltas) {
|
|
672
|
+
if (isNullFinalizationToolCall(d)) {
|
|
673
|
+
if (s.lastToolIndex !== undefined) {
|
|
674
|
+
s = closeToolCall(s, s.lastToolIndex, parts);
|
|
675
|
+
}
|
|
676
|
+
continue;
|
|
677
|
+
}
|
|
678
|
+
const idx = (d.index as number | undefined) ?? 0;
|
|
679
|
+
const fn = d.function as
|
|
680
|
+
| { name?: string; arguments?: string }
|
|
681
|
+
| undefined;
|
|
682
|
+
const name = fn?.name ?? (d.name as string | undefined) ?? "";
|
|
683
|
+
const args = fn?.arguments ?? (d.arguments as string | undefined) ?? "";
|
|
684
|
+
const rawId = (d.id as string | undefined) ?? "";
|
|
685
|
+
|
|
686
|
+
const existing = s.toolCalls.get(idx);
|
|
687
|
+
if (existing === undefined) {
|
|
688
|
+
if (s.lastToolIndex !== undefined && s.lastToolIndex !== idx) {
|
|
689
|
+
s = closeToolCall(s, s.lastToolIndex, parts);
|
|
690
|
+
}
|
|
691
|
+
const id = rawId || (yield* idGen.generateId());
|
|
692
|
+
const entry = { id, name };
|
|
693
|
+
const next = new Map(s.toolCalls);
|
|
694
|
+
next.set(idx, entry);
|
|
695
|
+
s = { ...s, toolCalls: next, lastToolIndex: idx };
|
|
696
|
+
parts.push({ type: "tool-params-start", id, name });
|
|
697
|
+
if (args.length > 0) {
|
|
698
|
+
parts.push({ type: "tool-params-delta", id, delta: args });
|
|
699
|
+
}
|
|
700
|
+
} else {
|
|
701
|
+
s = { ...s, lastToolIndex: idx };
|
|
702
|
+
if (args.length > 0) {
|
|
703
|
+
parts.push({
|
|
704
|
+
type: "tool-params-delta",
|
|
705
|
+
id: existing.id,
|
|
706
|
+
delta: args,
|
|
707
|
+
});
|
|
708
|
+
}
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
return s;
|
|
712
|
+
});
|
|
713
|
+
|
|
714
|
+
const hasNonZeroUsage = (raw: unknown): boolean => {
|
|
715
|
+
if (raw == null || typeof raw !== "object") return false;
|
|
716
|
+
const u = raw as Record<string, unknown>;
|
|
717
|
+
const prompt = (u.prompt_tokens as number | undefined) ?? 0;
|
|
718
|
+
const completion = (u.completion_tokens as number | undefined) ?? 0;
|
|
719
|
+
const total = (u.total_tokens as number | undefined) ?? 0;
|
|
720
|
+
return prompt > 0 || completion > 0 || total > 0;
|
|
721
|
+
};
|
|
722
|
+
|
|
723
|
+
const updateChunkMeta = (
|
|
724
|
+
state: StreamState,
|
|
725
|
+
chunk: Record<string, unknown>,
|
|
726
|
+
): StreamState => {
|
|
727
|
+
let s = state;
|
|
728
|
+
// Workers AI's native stream emits the real usage chunk, then a
|
|
729
|
+
// "zero-valued terminator" chunk where every count is 0 (it also re-emits
|
|
730
|
+
// `usage` with all zeros). Treat the zero chunk as a no-op so we keep the
|
|
731
|
+
// meaningful counts.
|
|
732
|
+
if (chunk.usage !== undefined && hasNonZeroUsage(chunk.usage)) {
|
|
733
|
+
s = { ...s, usage: chunk };
|
|
734
|
+
}
|
|
735
|
+
const choices = chunk.choices as
|
|
736
|
+
| Array<{ finish_reason?: string }>
|
|
737
|
+
| undefined;
|
|
738
|
+
const finish =
|
|
739
|
+
choices?.[0]?.finish_reason ?? (chunk.finish_reason as string | undefined);
|
|
740
|
+
if (finish != null) s = { ...s, finishReason: finish };
|
|
741
|
+
return s;
|
|
742
|
+
};
|
|
743
|
+
|
|
744
|
+
const handleNativeText = (
|
|
745
|
+
state: StreamState,
|
|
746
|
+
chunk: Record<string, unknown>,
|
|
747
|
+
parts: StreamParts,
|
|
748
|
+
idGen: IdGenerator.Service,
|
|
749
|
+
): Effect.Effect<StreamState> => {
|
|
750
|
+
const native = chunk.response;
|
|
751
|
+
if (native == null || native === "") return Effect.succeed(state);
|
|
752
|
+
const text =
|
|
753
|
+
typeof native === "object" ? JSON.stringify(native) : String(native);
|
|
754
|
+
if (text.length === 0) return Effect.succeed(state);
|
|
755
|
+
return emitTextDelta(state, text, parts, idGen);
|
|
756
|
+
};
|
|
757
|
+
|
|
758
|
+
const handleNativeToolCalls = (
|
|
759
|
+
state: StreamState,
|
|
760
|
+
chunk: Record<string, unknown>,
|
|
761
|
+
parts: StreamParts,
|
|
762
|
+
idGen: IdGenerator.Service,
|
|
763
|
+
): Effect.Effect<StreamState> => {
|
|
764
|
+
if (!Array.isArray(chunk.tool_calls)) return Effect.succeed(state);
|
|
765
|
+
return Effect.gen(function* () {
|
|
766
|
+
const s = closeReasoning(state, parts);
|
|
767
|
+
return yield* handleToolDeltas(
|
|
768
|
+
s,
|
|
769
|
+
chunk.tool_calls as ReadonlyArray<Record<string, unknown>>,
|
|
770
|
+
parts,
|
|
771
|
+
idGen,
|
|
772
|
+
);
|
|
773
|
+
});
|
|
774
|
+
};
|
|
775
|
+
|
|
776
|
+
const handleOpenAiDelta = (
|
|
777
|
+
state: StreamState,
|
|
778
|
+
chunk: Record<string, unknown>,
|
|
779
|
+
parts: StreamParts,
|
|
780
|
+
idGen: IdGenerator.Service,
|
|
781
|
+
): Effect.Effect<StreamState> => {
|
|
782
|
+
const delta = (
|
|
783
|
+
chunk.choices as Array<{ delta?: Record<string, unknown> }> | undefined
|
|
784
|
+
)?.[0]?.delta;
|
|
785
|
+
if (!delta) return Effect.succeed(state);
|
|
786
|
+
return Effect.gen(function* () {
|
|
787
|
+
let s = state;
|
|
788
|
+
const reasoning = (delta.reasoning_content ?? delta.reasoning) as
|
|
789
|
+
| string
|
|
790
|
+
| undefined;
|
|
791
|
+
if (reasoning && reasoning.length > 0) {
|
|
792
|
+
s = yield* emitReasoningDelta(s, reasoning, parts, idGen);
|
|
793
|
+
}
|
|
794
|
+
const text = delta.content as string | undefined;
|
|
795
|
+
if (text && text.length > 0) {
|
|
796
|
+
s = yield* emitTextDelta(s, text, parts, idGen);
|
|
797
|
+
}
|
|
798
|
+
const toolDeltas = delta.tool_calls as
|
|
799
|
+
| ReadonlyArray<Record<string, unknown>>
|
|
800
|
+
| undefined;
|
|
801
|
+
if (Array.isArray(toolDeltas)) {
|
|
802
|
+
s = closeReasoning(s, parts);
|
|
803
|
+
s = yield* handleToolDeltas(s, toolDeltas, parts, idGen);
|
|
804
|
+
}
|
|
805
|
+
return s;
|
|
806
|
+
});
|
|
807
|
+
};
|
|
808
|
+
|
|
809
|
+
const handleStreamChunk = (
|
|
810
|
+
state: StreamState,
|
|
811
|
+
data: string,
|
|
812
|
+
idGen: IdGenerator.Service,
|
|
813
|
+
): Effect.Effect<
|
|
814
|
+
readonly [StreamState, ReadonlyArray<Response.StreamPartEncoded>]
|
|
815
|
+
> =>
|
|
816
|
+
Effect.gen(function* () {
|
|
817
|
+
if (data === "") return [state, []] as const;
|
|
818
|
+
if (data === "[DONE]") {
|
|
819
|
+
return [{ ...state, receivedDone: true }, []] as const;
|
|
820
|
+
}
|
|
821
|
+
const chunk = tryParseJson(data);
|
|
822
|
+
if (chunk === undefined) return [state, []] as const;
|
|
823
|
+
|
|
824
|
+
const parts: StreamParts = [];
|
|
825
|
+
let s: StreamState = { ...state, receivedAnyData: true };
|
|
826
|
+
s = updateChunkMeta(s, chunk);
|
|
827
|
+
s = yield* handleNativeText(s, chunk, parts, idGen);
|
|
828
|
+
s = yield* handleNativeToolCalls(s, chunk, parts, idGen);
|
|
829
|
+
s = yield* handleOpenAiDelta(s, chunk, parts, idGen);
|
|
830
|
+
return [s, parts] as const;
|
|
831
|
+
});
|
|
832
|
+
|
|
833
|
+
const finalizeStream = (
|
|
834
|
+
state: StreamState,
|
|
835
|
+
): ReadonlyArray<Response.StreamPartEncoded> => {
|
|
836
|
+
const parts: StreamParts = [];
|
|
837
|
+
let s = state;
|
|
838
|
+
for (const [idx] of s.toolCalls) s = closeToolCall(s, idx, parts);
|
|
839
|
+
s = closeReasoning(s, parts);
|
|
840
|
+
if (s.textId !== undefined) parts.push({ type: "text-end", id: s.textId });
|
|
841
|
+
|
|
842
|
+
// Three cases for the final reason:
|
|
843
|
+
// 1. The model emitted an explicit `finish_reason` → map it.
|
|
844
|
+
// 2. The stream ended cleanly (`[DONE]` seen) but no reason → "stop".
|
|
845
|
+
// Workers AI's native shape never includes `finish_reason`,
|
|
846
|
+
// so without this rule every native-mode stream would report
|
|
847
|
+
// `unknown` despite completing successfully.
|
|
848
|
+
// 3. The stream ended abnormally (no `[DONE]`, no reason) → "error".
|
|
849
|
+
const reason: Response.FinishReason =
|
|
850
|
+
s.finishReason !== undefined
|
|
851
|
+
? mapFinishReason(s.finishReason)
|
|
852
|
+
: s.receivedDone
|
|
853
|
+
? "stop"
|
|
854
|
+
: s.receivedAnyData
|
|
855
|
+
? "error"
|
|
856
|
+
: "unknown";
|
|
857
|
+
|
|
858
|
+
parts.push({
|
|
859
|
+
type: "finish",
|
|
860
|
+
reason,
|
|
861
|
+
usage: mapUsage(s.usage),
|
|
862
|
+
response: undefined,
|
|
863
|
+
});
|
|
864
|
+
return parts;
|
|
865
|
+
};
|
|
866
|
+
|
|
867
|
+
const parseStreamText = (
|
|
868
|
+
resp: Response,
|
|
869
|
+
idGen: IdGenerator.Service,
|
|
870
|
+
): Stream.Stream<Response.StreamPartEncoded, AiError.AiError> => {
|
|
871
|
+
const body = resp.body;
|
|
872
|
+
if (body === null) {
|
|
873
|
+
return Stream.fromIterable<Response.StreamPartEncoded>(
|
|
874
|
+
finalizeStream(initialStreamState()),
|
|
875
|
+
);
|
|
876
|
+
}
|
|
877
|
+
return Stream.fromReadableStream<Uint8Array, AiError.AiError>({
|
|
878
|
+
evaluate: () => body,
|
|
879
|
+
onError: (cause) => toAiError(cause, "streamText"),
|
|
880
|
+
}).pipe(
|
|
881
|
+
Stream.decodeText(),
|
|
882
|
+
Stream.pipeThroughChannel(Sse.decode<AiError.AiError, unknown>()),
|
|
883
|
+
Stream.catchTag("Retry", (retry) => Stream.die(retry)),
|
|
884
|
+
Stream.mapAccumEffect(
|
|
885
|
+
initialStreamState,
|
|
886
|
+
(state, event) => handleStreamChunk(state, event.data, idGen),
|
|
887
|
+
{ onHalt: (state) => finalizeStream(state) },
|
|
888
|
+
),
|
|
889
|
+
);
|
|
890
|
+
};
|
|
891
|
+
|
|
892
|
+
// ---------------------------------------------------------------------------
|
|
893
|
+
// Error mapping
|
|
894
|
+
// ---------------------------------------------------------------------------
|
|
895
|
+
|
|
896
|
+
const toAiError = (
|
|
897
|
+
cause: unknown,
|
|
898
|
+
method: "generateText" | "streamText",
|
|
899
|
+
): AiError.AiError =>
|
|
900
|
+
AiError.AiError.make({
|
|
901
|
+
module: "Cloudflare.AiGateway.LanguageModel",
|
|
902
|
+
method,
|
|
903
|
+
reason: new AiError.UnknownError({
|
|
904
|
+
description:
|
|
905
|
+
cause instanceof Error ? cause.message : "AI Gateway request failed",
|
|
906
|
+
}),
|
|
907
|
+
});
|