alchemy 2.0.0-beta.46 → 2.0.0-beta.47

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/bin/exec.js +18 -10
  2. package/bin/exec.js.map +1 -1
  3. package/lib/Apply.js +2 -2
  4. package/lib/Apply.js.map +1 -1
  5. package/lib/Cli/commands/_shared.d.ts +7 -1
  6. package/lib/Cli/commands/_shared.d.ts.map +1 -1
  7. package/lib/Cli/commands/_shared.js +1 -0
  8. package/lib/Cli/commands/_shared.js.map +1 -1
  9. package/lib/Cli/commands/login.d.ts +1 -1
  10. package/lib/Cli/commands/login.d.ts.map +1 -1
  11. package/lib/Cli/commands/login.js +45 -42
  12. package/lib/Cli/commands/login.js.map +1 -1
  13. package/lib/Cli/commands/logs.js +1 -1
  14. package/lib/Cli/commands/logs.js.map +1 -1
  15. package/lib/Cli/commands/state.js +1 -1
  16. package/lib/Cli/commands/state.js.map +1 -1
  17. package/lib/Cli/commands/tail.js +1 -1
  18. package/lib/Cli/commands/tail.js.map +1 -1
  19. package/lib/Cloudflare/AiGateway/AiGateway.d.ts +112 -1
  20. package/lib/Cloudflare/AiGateway/AiGateway.d.ts.map +1 -1
  21. package/lib/Cloudflare/AiGateway/AiGateway.js +112 -1
  22. package/lib/Cloudflare/AiGateway/AiGateway.js.map +1 -1
  23. package/lib/Cloudflare/AiGateway/AiGatewayBinding.d.ts +43 -10
  24. package/lib/Cloudflare/AiGateway/AiGatewayBinding.d.ts.map +1 -1
  25. package/lib/Cloudflare/AiGateway/AiGatewayBinding.js +38 -10
  26. package/lib/Cloudflare/AiGateway/AiGatewayBinding.js.map +1 -1
  27. package/lib/Cloudflare/AiGateway/LanguageModel.d.ts +35 -0
  28. package/lib/Cloudflare/AiGateway/LanguageModel.d.ts.map +1 -0
  29. package/lib/Cloudflare/AiGateway/LanguageModel.js +576 -0
  30. package/lib/Cloudflare/AiGateway/LanguageModel.js.map +1 -0
  31. package/lib/Cloudflare/AiGateway/index.d.ts +1 -0
  32. package/lib/Cloudflare/AiGateway/index.d.ts.map +1 -1
  33. package/lib/Cloudflare/AiGateway/index.js +1 -0
  34. package/lib/Cloudflare/AiGateway/index.js.map +1 -1
  35. package/lib/Cloudflare/StateStore/State.d.ts.map +1 -1
  36. package/lib/Cloudflare/StateStore/State.js +82 -78
  37. package/lib/Cloudflare/StateStore/State.js.map +1 -1
  38. package/lib/Cloudflare/Workers/DurableObjectBridge.d.ts.map +1 -1
  39. package/lib/Cloudflare/Workers/DurableObjectBridge.js +10 -1
  40. package/lib/Cloudflare/Workers/DurableObjectBridge.js.map +1 -1
  41. package/lib/Cloudflare/Workers/LocalWorkerProvider.d.ts.map +1 -1
  42. package/lib/Cloudflare/Workers/LocalWorkerProvider.js +13 -1
  43. package/lib/Cloudflare/Workers/LocalWorkerProvider.js.map +1 -1
  44. package/lib/Cloudflare/Workers/Rpc.d.ts.map +1 -1
  45. package/lib/Cloudflare/Workers/Rpc.js +20 -2
  46. package/lib/Cloudflare/Workers/Rpc.js.map +1 -1
  47. package/lib/Cloudflare/Workers/WorkerBridge.d.ts.map +1 -1
  48. package/lib/Cloudflare/Workers/WorkerBridge.js +20 -2
  49. package/lib/Cloudflare/Workers/WorkerBridge.js.map +1 -1
  50. package/lib/Output.js +2 -2
  51. package/lib/Output.js.map +1 -1
  52. package/lib/Plan.d.ts +2 -2
  53. package/lib/Plan.d.ts.map +1 -1
  54. package/lib/Plan.js +3 -3
  55. package/lib/Plan.js.map +1 -1
  56. package/lib/Stack.d.ts.map +1 -1
  57. package/lib/Stack.js +6 -0
  58. package/lib/Stack.js.map +1 -1
  59. package/lib/State/InMemoryState.d.ts +2 -1
  60. package/lib/State/InMemoryState.d.ts.map +1 -1
  61. package/lib/State/InMemoryState.js +3 -3
  62. package/lib/State/InMemoryState.js.map +1 -1
  63. package/lib/State/LocalState.d.ts.map +1 -1
  64. package/lib/State/LocalState.js +5 -1
  65. package/lib/State/LocalState.js.map +1 -1
  66. package/lib/State/State.d.ts +2 -2
  67. package/lib/State/State.d.ts.map +1 -1
  68. package/lib/State/State.js.map +1 -1
  69. package/lib/tsconfig.test.tsbuildinfo +1 -1
  70. package/package.json +1 -1
  71. package/src/Apply.ts +2 -2
  72. package/src/Cli/commands/_shared.ts +7 -1
  73. package/src/Cli/commands/login.ts +69 -60
  74. package/src/Cli/commands/logs.ts +1 -1
  75. package/src/Cli/commands/state.ts +1 -1
  76. package/src/Cli/commands/tail.ts +1 -1
  77. package/src/Cloudflare/AiGateway/AiGateway.ts +112 -1
  78. package/src/Cloudflare/AiGateway/AiGatewayBinding.ts +52 -10
  79. package/src/Cloudflare/AiGateway/LanguageModel.ts +907 -0
  80. package/src/Cloudflare/AiGateway/index.ts +1 -0
  81. package/src/Cloudflare/StateStore/State.ts +113 -107
  82. package/src/Cloudflare/Workers/DurableObjectBridge.ts +10 -1
  83. package/src/Cloudflare/Workers/LocalWorkerProvider.ts +13 -1
  84. package/src/Cloudflare/Workers/Rpc.ts +38 -9
  85. package/src/Cloudflare/Workers/WorkerBridge.ts +20 -2
  86. package/src/Output.ts +2 -2
  87. package/src/Plan.ts +4 -4
  88. package/src/Stack.ts +6 -0
  89. package/src/State/InMemoryState.ts +87 -85
  90. package/src/State/LocalState.ts +15 -1
  91. package/src/State/State.ts +5 -4
@@ -0,0 +1,907 @@
1
+ import * as Effect from "effect/Effect";
2
+ import * as Layer from "effect/Layer";
3
+ import * as Stream from "effect/Stream";
4
+ import {
5
+ AiError,
6
+ LanguageModel as AiLanguageModel,
7
+ IdGenerator,
8
+ Prompt,
9
+ Response,
10
+ Tool,
11
+ } from "effect/unstable/ai";
12
+ import * as Sse from "effect/unstable/encoding/Sse";
13
+ import type { RuntimeContext } from "../../RuntimeContext.ts";
14
+ import type { AiGatewayClient } from "./AiGatewayBinding.ts";
15
+
16
+ /**
17
+ * Options for constructing an AI Gateway-backed Workers AI LanguageModel.
18
+ */
19
+ export interface LanguageModelOptions {
20
+ /** Already-bound AI Gateway client from `AiGatewayBinding.bind(gateway)`. */
21
+ readonly client: AiGatewayClient;
22
+ /** Workers AI model id, e.g. `@cf/meta/llama-3.3-70b-instruct-fp8-fast`. */
23
+ readonly model: string;
24
+ /** Optional per-call defaults; overridable per request via `providerOptions`. */
25
+ readonly parameters?: {
26
+ readonly temperature?: number;
27
+ readonly maxTokens?: number;
28
+ readonly topP?: number;
29
+ readonly topK?: number;
30
+ readonly seed?: number;
31
+ readonly frequencyPenalty?: number;
32
+ readonly presencePenalty?: number;
33
+ };
34
+ }
35
+
36
+ /**
37
+ * Provide a {@link AiLanguageModel.LanguageModel} layer backed by the supplied
38
+ * AI Gateway client and Workers AI model.
39
+ */
40
+ export const makeLanguageModelLayer = (
41
+ options: LanguageModelOptions,
42
+ ): Layer.Layer<AiLanguageModel.LanguageModel, never, RuntimeContext> =>
43
+ Layer.effect(AiLanguageModel.LanguageModel, makeLanguageModel(options));
44
+
45
+ /**
46
+ * Build a {@link AiLanguageModel.Service} that proxies generateText/streamText
47
+ * through the supplied AI Gateway client to a Workers AI model.
48
+ */
49
+ export const makeLanguageModel = ({
50
+ client,
51
+ model,
52
+ parameters,
53
+ }: LanguageModelOptions): Effect.Effect<
54
+ AiLanguageModel.Service,
55
+ never,
56
+ RuntimeContext
57
+ > =>
58
+ Effect.gen(function* () {
59
+ const ai = yield* client.raw;
60
+ const gatewayId = yield* client.id;
61
+
62
+ const callRaw = (
63
+ body: WorkersAiInputs,
64
+ method: "generateText" | "streamText",
65
+ ): Effect.Effect<Response, AiError.AiError> =>
66
+ Effect.tryPromise({
67
+ try: () =>
68
+ ai.run(
69
+ model as keyof AiModels,
70
+ body as unknown as AiModels[keyof AiModels]["inputs"],
71
+ {
72
+ gateway: { id: gatewayId },
73
+ returnRawResponse: true,
74
+ },
75
+ ),
76
+ catch: (cause) => toAiError(cause, method),
77
+ });
78
+
79
+ return yield* AiLanguageModel.make({
80
+ generateText: (options) =>
81
+ Effect.gen(function* () {
82
+ const body = toRequestBody({ options, parameters, stream: false });
83
+ const resp = yield* callRaw(body, "generateText");
84
+ const json = yield* Effect.tryPromise({
85
+ try: () => resp.json() as Promise<Record<string, unknown>>,
86
+ catch: (cause) => toAiError(cause, "generateText"),
87
+ });
88
+ return yield* parseGenerateText(json);
89
+ }),
90
+ streamText: (options) =>
91
+ Stream.unwrap(
92
+ Effect.gen(function* () {
93
+ const idGen = yield* IdGenerator.IdGenerator;
94
+ const body = toRequestBody({ options, parameters, stream: true });
95
+ const resp = yield* callRaw(body, "streamText");
96
+ return parseStreamText(resp, idGen);
97
+ }),
98
+ ),
99
+ });
100
+ });
101
+
102
+ // ---------------------------------------------------------------------------
103
+ // Wire format types (Workers AI request)
104
+ //
105
+ // Workers AI returns two response shapes depending on the model:
106
+ // - Native: { response: "...", tool_calls: [...] , usage }
107
+ // - OpenAI: { choices: [{ message: { content, tool_calls, reasoning_content } }], usage }
108
+ //
109
+ // We accept both defensively — schemas would over-constrain.
110
+ // ---------------------------------------------------------------------------
111
+
112
+ interface WorkersAiMessage {
113
+ readonly role: "system" | "user" | "assistant" | "tool";
114
+ readonly content?: unknown;
115
+ readonly name?: string;
116
+ readonly tool_call_id?: string;
117
+ readonly tool_calls?: ReadonlyArray<{
118
+ readonly id: string;
119
+ readonly type: "function";
120
+ readonly function: { readonly name: string; readonly arguments: string };
121
+ }>;
122
+ readonly reasoning?: string;
123
+ }
124
+
125
+ interface WorkersAiToolDef {
126
+ readonly type: "function";
127
+ readonly function: {
128
+ readonly name: string;
129
+ readonly description?: string;
130
+ readonly parameters: unknown;
131
+ };
132
+ }
133
+
134
+ type WorkersAiToolChoice =
135
+ | "auto"
136
+ | "required"
137
+ | "none"
138
+ | { readonly type: "function"; readonly function: { readonly name: string } };
139
+
140
+ interface WorkersAiInputs {
141
+ readonly messages: ReadonlyArray<WorkersAiMessage>;
142
+ readonly tools?: ReadonlyArray<WorkersAiToolDef>;
143
+ readonly tool_choice?: WorkersAiToolChoice;
144
+ readonly stream?: boolean;
145
+ readonly stream_options?: { readonly include_usage: boolean };
146
+ readonly max_tokens?: number;
147
+ readonly temperature?: number;
148
+ readonly top_p?: number;
149
+ readonly top_k?: number;
150
+ readonly random_seed?: number;
151
+ readonly frequency_penalty?: number;
152
+ readonly presence_penalty?: number;
153
+ }
154
+
155
+ // ---------------------------------------------------------------------------
156
+ // Prompt → Workers AI messages (pure, no .push mutation)
157
+ // ---------------------------------------------------------------------------
158
+
159
+ const uint8ArrayToBase64 = (bytes: Uint8Array): string => {
160
+ let binary = "";
161
+ const chunkSize = 8192;
162
+ for (let i = 0; i < bytes.length; i += chunkSize) {
163
+ const chunk = bytes.subarray(i, Math.min(i + chunkSize, bytes.length));
164
+ binary += String.fromCharCode(...chunk);
165
+ }
166
+ return btoa(binary);
167
+ };
168
+
169
+ const fileToImageUrl = (
170
+ data: string | Uint8Array | URL,
171
+ mediaType: string,
172
+ ): string => {
173
+ if (data instanceof URL) return data.toString();
174
+ if (data instanceof Uint8Array) {
175
+ return `data:${mediaType};base64,${uint8ArrayToBase64(data)}`;
176
+ }
177
+ if (data.startsWith("data:") || data.startsWith("http")) return data;
178
+ return `data:${mediaType};base64,${data}`;
179
+ };
180
+
181
+ const convertPromptToMessages = (
182
+ prompt: Prompt.Prompt,
183
+ ): ReadonlyArray<WorkersAiMessage> =>
184
+ prompt.content.flatMap((m): ReadonlyArray<WorkersAiMessage> => {
185
+ switch (m.role) {
186
+ case "system":
187
+ return [{ role: "system", content: m.content }];
188
+ case "user":
189
+ return [toUserMessage(m.content)];
190
+ case "assistant":
191
+ return [toAssistantMessage(m.content)];
192
+ case "tool":
193
+ return m.content.flatMap(toToolMessage);
194
+ }
195
+ });
196
+
197
+ const toUserMessage = (
198
+ parts: Prompt.UserMessage["content"],
199
+ ): WorkersAiMessage => {
200
+ const text = parts
201
+ .flatMap((p) => (p.type === "text" ? [p.text] : []))
202
+ .join("\n");
203
+ const images = parts.flatMap((p) =>
204
+ p.type === "file"
205
+ ? [
206
+ {
207
+ type: "image_url" as const,
208
+ image_url: { url: fileToImageUrl(p.data, p.mediaType) },
209
+ },
210
+ ]
211
+ : [],
212
+ );
213
+ if (images.length === 0) return { role: "user", content: text };
214
+ return {
215
+ role: "user",
216
+ content: [
217
+ ...(text.length > 0 ? [{ type: "text" as const, text }] : []),
218
+ ...images,
219
+ ],
220
+ };
221
+ };
222
+
223
+ const toAssistantMessage = (
224
+ parts: Prompt.AssistantMessage["content"],
225
+ ): WorkersAiMessage => {
226
+ const text = parts
227
+ .flatMap((p) => (p.type === "text" ? [p.text] : []))
228
+ .join("");
229
+ const reasoning = parts
230
+ .flatMap((p) => (p.type === "reasoning" ? [p.text] : []))
231
+ .join("");
232
+ const toolCalls = parts.flatMap((p) =>
233
+ p.type === "tool-call"
234
+ ? [
235
+ {
236
+ id: p.id,
237
+ type: "function" as const,
238
+ function: {
239
+ name: p.name,
240
+ arguments:
241
+ typeof p.params === "string"
242
+ ? p.params
243
+ : JSON.stringify(p.params ?? {}),
244
+ },
245
+ },
246
+ ]
247
+ : [],
248
+ );
249
+ return {
250
+ role: "assistant",
251
+ content: text,
252
+ ...(reasoning ? { reasoning } : {}),
253
+ ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),
254
+ };
255
+ };
256
+
257
+ const toToolMessage = (
258
+ part: Prompt.ToolMessage["content"][number],
259
+ ): ReadonlyArray<WorkersAiMessage> =>
260
+ part.type === "tool-result"
261
+ ? [
262
+ {
263
+ role: "tool",
264
+ name: part.name,
265
+ tool_call_id: part.id,
266
+ content:
267
+ typeof part.result === "string"
268
+ ? part.result
269
+ : JSON.stringify(part.result),
270
+ },
271
+ ]
272
+ : [];
273
+
274
+ // ---------------------------------------------------------------------------
275
+ // Tools / tool_choice
276
+ // ---------------------------------------------------------------------------
277
+
278
+ const prepareTools = (
279
+ tools: ReadonlyArray<Tool.Any>,
280
+ toolChoice: AiLanguageModel.ProviderOptions["toolChoice"],
281
+ ): {
282
+ tools?: ReadonlyArray<WorkersAiToolDef>;
283
+ tool_choice?: WorkersAiToolChoice;
284
+ } => {
285
+ if (tools.length === 0) return {};
286
+ const mapped: ReadonlyArray<WorkersAiToolDef> = tools.map((tool) => ({
287
+ type: "function",
288
+ function: {
289
+ name: tool.name,
290
+ description: Tool.getDescription(tool),
291
+ parameters: Tool.getJsonSchema(tool),
292
+ },
293
+ }));
294
+
295
+ if (toolChoice === "auto" || toolChoice == null) {
296
+ return { tools: mapped, tool_choice: "auto" };
297
+ }
298
+ if (toolChoice === "none") return { tools: mapped, tool_choice: "none" };
299
+ if (toolChoice === "required") {
300
+ return { tools: mapped, tool_choice: "required" };
301
+ }
302
+ if (typeof toolChoice === "object" && "tool" in toolChoice) {
303
+ return {
304
+ tools: mapped.filter((t) => t.function.name === toolChoice.tool),
305
+ tool_choice: "required",
306
+ };
307
+ }
308
+ if (typeof toolChoice === "object" && "oneOf" in toolChoice) {
309
+ const allowed = new Set(toolChoice.oneOf);
310
+ return {
311
+ tools: mapped.filter((t) => allowed.has(t.function.name)),
312
+ tool_choice: toolChoice.mode === "required" ? "required" : "auto",
313
+ };
314
+ }
315
+ return { tools: mapped, tool_choice: "auto" };
316
+ };
317
+
318
+ // ---------------------------------------------------------------------------
319
+ // Request body
320
+ // ---------------------------------------------------------------------------
321
+
322
+ const toRequestBody = ({
323
+ options,
324
+ parameters,
325
+ stream,
326
+ }: {
327
+ readonly options: AiLanguageModel.ProviderOptions;
328
+ readonly parameters: LanguageModelOptions["parameters"];
329
+ readonly stream: boolean;
330
+ }): WorkersAiInputs => {
331
+ const messages = convertPromptToMessages(options.prompt);
332
+ const { tools, tool_choice } = prepareTools(
333
+ options.tools,
334
+ options.toolChoice,
335
+ );
336
+ return {
337
+ messages,
338
+ ...(tools !== undefined ? { tools } : {}),
339
+ ...(tool_choice !== undefined ? { tool_choice } : {}),
340
+ // `stream_options.include_usage` is the OpenAI-compatible opt-in for
341
+ // usage tokens to appear in the final streamed chunk. Without it most
342
+ // Workers AI models omit `usage` from the stream entirely, leaving the
343
+ // `finish` part with zeroed counts.
344
+ ...(stream
345
+ ? { stream: true, stream_options: { include_usage: true } }
346
+ : {}),
347
+ ...(parameters?.maxTokens !== undefined
348
+ ? { max_tokens: parameters.maxTokens }
349
+ : {}),
350
+ ...(parameters?.temperature !== undefined
351
+ ? { temperature: parameters.temperature }
352
+ : {}),
353
+ ...(parameters?.topP !== undefined ? { top_p: parameters.topP } : {}),
354
+ ...(parameters?.topK !== undefined ? { top_k: parameters.topK } : {}),
355
+ ...(parameters?.seed !== undefined ? { random_seed: parameters.seed } : {}),
356
+ ...(parameters?.frequencyPenalty !== undefined
357
+ ? { frequency_penalty: parameters.frequencyPenalty }
358
+ : {}),
359
+ ...(parameters?.presencePenalty !== undefined
360
+ ? { presence_penalty: parameters.presencePenalty }
361
+ : {}),
362
+ };
363
+ };
364
+
365
+ // ---------------------------------------------------------------------------
366
+ // Finish reason / usage mapping
367
+ // ---------------------------------------------------------------------------
368
+
369
+ const mapFinishReason = (raw: unknown): Response.FinishReason => {
370
+ switch (raw) {
371
+ case "stop":
372
+ return "stop";
373
+ case "length":
374
+ case "model_length":
375
+ return "length";
376
+ case "tool_calls":
377
+ return "tool-calls";
378
+ case "content_filter":
379
+ case "content-filter":
380
+ return "content-filter";
381
+ case "error":
382
+ return "error";
383
+ case undefined:
384
+ case null:
385
+ return "unknown";
386
+ default:
387
+ return "other";
388
+ }
389
+ };
390
+
391
+ const mapUsage = (raw: Record<string, unknown> | undefined): Response.Usage => {
392
+ const usage = (raw?.usage as Record<string, unknown> | undefined) ?? {};
393
+ const promptTokens = (usage.prompt_tokens as number | undefined) ?? 0;
394
+ const completionTokens = (usage.completion_tokens as number | undefined) ?? 0;
395
+ const cached = (
396
+ usage.prompt_tokens_details as { cached_tokens?: number } | undefined
397
+ )?.cached_tokens;
398
+ // Construct an actual `Response.Usage` instance — `Schema.Class<Usage>`
399
+ // encodes by going through the class constructor / `isInstance` check, so a
400
+ // plain struct that "matches" the encoded shape isn't enough.
401
+ return new Response.Usage({
402
+ inputTokens: {
403
+ uncached:
404
+ cached !== undefined
405
+ ? Math.max(0, promptTokens - cached)
406
+ : promptTokens,
407
+ total: promptTokens,
408
+ cacheRead: cached ?? 0,
409
+ cacheWrite: 0,
410
+ },
411
+ outputTokens: {
412
+ total: completionTokens,
413
+ text: 0,
414
+ reasoning: 0,
415
+ },
416
+ });
417
+ };
418
+
419
+ // ---------------------------------------------------------------------------
420
+ // generateText: JSON → Response.PartEncoded[]
421
+ //
422
+ // Normalize the dual-shape (native + OpenAI) response into a single
423
+ // `DecodedResponse` once, then build the part list with pure spreads.
424
+ // ---------------------------------------------------------------------------
425
+
426
+ interface DecodedToolCall {
427
+ readonly rawId: string;
428
+ readonly name: string;
429
+ readonly arguments: unknown;
430
+ }
431
+
432
+ interface DecodedResponse {
433
+ readonly text: string | undefined;
434
+ readonly reasoning: string | undefined;
435
+ readonly toolCalls: ReadonlyArray<DecodedToolCall>;
436
+ readonly finishReason: string | undefined;
437
+ }
438
+
439
+ const decodeResponse = (raw: Record<string, unknown>): DecodedResponse => {
440
+ const choice = (
441
+ raw.choices as
442
+ | Array<{
443
+ message?: {
444
+ content?: string | null;
445
+ reasoning_content?: string;
446
+ reasoning?: string;
447
+ tool_calls?: ReadonlyArray<Record<string, unknown>>;
448
+ };
449
+ finish_reason?: string;
450
+ }>
451
+ | undefined
452
+ )?.[0];
453
+ const message = choice?.message;
454
+
455
+ const openAiText = message?.content;
456
+ const text =
457
+ typeof openAiText === "string" && openAiText.length > 0
458
+ ? openAiText
459
+ : nativeTextOf(raw.response);
460
+ const reasoning = message?.reasoning_content ?? message?.reasoning;
461
+ const rawToolCalls =
462
+ message?.tool_calls ??
463
+ (Array.isArray(raw.tool_calls)
464
+ ? (raw.tool_calls as ReadonlyArray<Record<string, unknown>>)
465
+ : []);
466
+
467
+ return {
468
+ text,
469
+ reasoning: reasoning && reasoning.length > 0 ? reasoning : undefined,
470
+ toolCalls: rawToolCalls.flatMap(decodeToolCall),
471
+ finishReason:
472
+ choice?.finish_reason ?? (raw.finish_reason as string | undefined),
473
+ };
474
+ };
475
+
476
+ const nativeTextOf = (raw: unknown): string | undefined => {
477
+ if (raw == null) return undefined;
478
+ if (typeof raw === "object") return JSON.stringify(raw);
479
+ const text = String(raw);
480
+ return text.length > 0 ? text : undefined;
481
+ };
482
+
483
+ const decodeToolCall = (
484
+ tc: Record<string, unknown>,
485
+ ): ReadonlyArray<DecodedToolCall> => {
486
+ const fn = tc.function as { name?: string; arguments?: unknown } | undefined;
487
+ const rawId = (tc.id as string | undefined) ?? "";
488
+ if (fn?.name) {
489
+ return [{ rawId, name: fn.name, arguments: fn.arguments ?? "" }];
490
+ }
491
+ const flatName = tc.name as string | undefined;
492
+ if (flatName) {
493
+ return [{ rawId, name: flatName, arguments: tc.arguments ?? "" }];
494
+ }
495
+ return [];
496
+ };
497
+
498
+ const tryParseJsonArgs = (raw: unknown): unknown => {
499
+ if (typeof raw !== "string") return raw;
500
+ try {
501
+ return JSON.parse(raw);
502
+ } catch {
503
+ // Leave as raw string; the framework's tool-result decoder will fail loudly.
504
+ return raw;
505
+ }
506
+ };
507
+
508
+ const parseGenerateText = Effect.fnUntraced(function* (
509
+ raw: Record<string, unknown>,
510
+ ) {
511
+ const idGen = yield* IdGenerator.IdGenerator;
512
+ const decoded = decodeResponse(raw);
513
+
514
+ const toolCallParts = yield* Effect.forEach(decoded.toolCalls, (tc) =>
515
+ Effect.gen(function* () {
516
+ const id = tc.rawId || (yield* idGen.generateId());
517
+ return {
518
+ type: "tool-call" as const,
519
+ id,
520
+ name: tc.name,
521
+ params: tryParseJsonArgs(tc.arguments),
522
+ };
523
+ }),
524
+ );
525
+
526
+ const finish = mapFinishReason(
527
+ decoded.finishReason ??
528
+ (decoded.toolCalls.length > 0 ? "tool_calls" : "stop"),
529
+ );
530
+
531
+ return [
532
+ ...(decoded.reasoning !== undefined
533
+ ? [{ type: "reasoning" as const, text: decoded.reasoning }]
534
+ : []),
535
+ ...(decoded.text !== undefined && decoded.text.length > 0
536
+ ? [{ type: "text" as const, text: decoded.text }]
537
+ : []),
538
+ ...toolCallParts,
539
+ {
540
+ type: "finish" as const,
541
+ reason: finish,
542
+ usage: mapUsage(raw),
543
+ response: undefined,
544
+ },
545
+ ] satisfies ReadonlyArray<Response.PartEncoded>;
546
+ });
547
+
548
+ // ---------------------------------------------------------------------------
549
+ // streamText: SSE byte stream → Stream<Response.StreamPartEncoded>
550
+ //
551
+ // Immutable `StreamState` is threaded through `Stream.mapAccumEffect`. The
552
+ // per-chunk output buffer (`parts: Array<StreamPartEncoded>`) is mutable for
553
+ // performance — it's scoped to one chunk, never escapes the handler, and lets
554
+ // us avoid the O(n²) array-spread that pure threading would force in the hot
555
+ // path. This matches the pattern Effect's own `@effect/ai-*` adapters use.
556
+ // ---------------------------------------------------------------------------
557
+
558
+ interface StreamState {
559
+ readonly textId: string | undefined;
560
+ readonly reasoningId: string | undefined;
561
+ readonly toolCalls: ReadonlyMap<
562
+ number,
563
+ { readonly id: string; readonly name: string }
564
+ >;
565
+ readonly lastToolIndex: number | undefined;
566
+ readonly closedToolIndices: ReadonlySet<number>;
567
+ readonly usage: Record<string, unknown> | undefined;
568
+ readonly finishReason: string | undefined;
569
+ readonly receivedAnyData: boolean;
570
+ readonly receivedDone: boolean;
571
+ }
572
+
573
+ const initialStreamState = (): StreamState => ({
574
+ textId: undefined,
575
+ reasoningId: undefined,
576
+ toolCalls: new Map(),
577
+ lastToolIndex: undefined,
578
+ closedToolIndices: new Set(),
579
+ usage: undefined,
580
+ finishReason: undefined,
581
+ receivedAnyData: false,
582
+ receivedDone: false,
583
+ });
584
+
585
+ type StreamParts = Array<Response.StreamPartEncoded>;
586
+
587
+ const tryParseJson = (data: string): Record<string, unknown> | undefined => {
588
+ try {
589
+ const v = JSON.parse(data);
590
+ return v && typeof v === "object"
591
+ ? (v as Record<string, unknown>)
592
+ : undefined;
593
+ } catch {
594
+ return undefined;
595
+ }
596
+ };
597
+
598
+ const isNullFinalizationToolCall = (tc: Record<string, unknown>): boolean => {
599
+ const fn = tc.function as Record<string, unknown> | undefined;
600
+ const name = fn?.name ?? tc.name ?? null;
601
+ const args = fn?.arguments ?? tc.arguments ?? null;
602
+ const id = tc.id ?? null;
603
+ return !id && !name && (!args || args === "");
604
+ };
605
+
606
+ const closeReasoning = (
607
+ state: StreamState,
608
+ parts: StreamParts,
609
+ ): StreamState => {
610
+ if (state.reasoningId === undefined) return state;
611
+ parts.push({ type: "reasoning-end", id: state.reasoningId });
612
+ return { ...state, reasoningId: undefined };
613
+ };
614
+
615
+ const closeToolCall = (
616
+ state: StreamState,
617
+ index: number,
618
+ parts: StreamParts,
619
+ ): StreamState => {
620
+ if (state.closedToolIndices.has(index)) return state;
621
+ const tc = state.toolCalls.get(index);
622
+ if (!tc) return state;
623
+ parts.push({ type: "tool-params-end", id: tc.id });
624
+ const closed = new Set(state.closedToolIndices);
625
+ closed.add(index);
626
+ return { ...state, closedToolIndices: closed };
627
+ };
628
+
629
+ const emitTextDelta = (
630
+ state: StreamState,
631
+ delta: string,
632
+ parts: StreamParts,
633
+ idGen: IdGenerator.Service,
634
+ ): Effect.Effect<StreamState> =>
635
+ Effect.gen(function* () {
636
+ let s = closeReasoning(state, parts);
637
+ if (s.textId === undefined) {
638
+ const id = yield* idGen.generateId();
639
+ parts.push({ type: "text-start", id });
640
+ s = { ...s, textId: id };
641
+ }
642
+ parts.push({ type: "text-delta", id: s.textId!, delta });
643
+ return s;
644
+ });
645
+
646
+ const emitReasoningDelta = (
647
+ state: StreamState,
648
+ delta: string,
649
+ parts: StreamParts,
650
+ idGen: IdGenerator.Service,
651
+ ): Effect.Effect<StreamState> =>
652
+ Effect.gen(function* () {
653
+ let s = state;
654
+ if (s.reasoningId === undefined) {
655
+ const id = yield* idGen.generateId();
656
+ parts.push({ type: "reasoning-start", id });
657
+ s = { ...s, reasoningId: id };
658
+ }
659
+ parts.push({ type: "reasoning-delta", id: s.reasoningId!, delta });
660
+ return s;
661
+ });
662
+
663
+ const handleToolDeltas = (
664
+ state: StreamState,
665
+ deltas: ReadonlyArray<Record<string, unknown>>,
666
+ parts: StreamParts,
667
+ idGen: IdGenerator.Service,
668
+ ): Effect.Effect<StreamState> =>
669
+ Effect.gen(function* () {
670
+ let s = state;
671
+ for (const d of deltas) {
672
+ if (isNullFinalizationToolCall(d)) {
673
+ if (s.lastToolIndex !== undefined) {
674
+ s = closeToolCall(s, s.lastToolIndex, parts);
675
+ }
676
+ continue;
677
+ }
678
+ const idx = (d.index as number | undefined) ?? 0;
679
+ const fn = d.function as
680
+ | { name?: string; arguments?: string }
681
+ | undefined;
682
+ const name = fn?.name ?? (d.name as string | undefined) ?? "";
683
+ const args = fn?.arguments ?? (d.arguments as string | undefined) ?? "";
684
+ const rawId = (d.id as string | undefined) ?? "";
685
+
686
+ const existing = s.toolCalls.get(idx);
687
+ if (existing === undefined) {
688
+ if (s.lastToolIndex !== undefined && s.lastToolIndex !== idx) {
689
+ s = closeToolCall(s, s.lastToolIndex, parts);
690
+ }
691
+ const id = rawId || (yield* idGen.generateId());
692
+ const entry = { id, name };
693
+ const next = new Map(s.toolCalls);
694
+ next.set(idx, entry);
695
+ s = { ...s, toolCalls: next, lastToolIndex: idx };
696
+ parts.push({ type: "tool-params-start", id, name });
697
+ if (args.length > 0) {
698
+ parts.push({ type: "tool-params-delta", id, delta: args });
699
+ }
700
+ } else {
701
+ s = { ...s, lastToolIndex: idx };
702
+ if (args.length > 0) {
703
+ parts.push({
704
+ type: "tool-params-delta",
705
+ id: existing.id,
706
+ delta: args,
707
+ });
708
+ }
709
+ }
710
+ }
711
+ return s;
712
+ });
713
+
714
+ const hasNonZeroUsage = (raw: unknown): boolean => {
715
+ if (raw == null || typeof raw !== "object") return false;
716
+ const u = raw as Record<string, unknown>;
717
+ const prompt = (u.prompt_tokens as number | undefined) ?? 0;
718
+ const completion = (u.completion_tokens as number | undefined) ?? 0;
719
+ const total = (u.total_tokens as number | undefined) ?? 0;
720
+ return prompt > 0 || completion > 0 || total > 0;
721
+ };
722
+
723
+ const updateChunkMeta = (
724
+ state: StreamState,
725
+ chunk: Record<string, unknown>,
726
+ ): StreamState => {
727
+ let s = state;
728
+ // Workers AI's native stream emits the real usage chunk, then a
729
+ // "zero-valued terminator" chunk where every count is 0 (it also re-emits
730
+ // `usage` with all zeros). Treat the zero chunk as a no-op so we keep the
731
+ // meaningful counts.
732
+ if (chunk.usage !== undefined && hasNonZeroUsage(chunk.usage)) {
733
+ s = { ...s, usage: chunk };
734
+ }
735
+ const choices = chunk.choices as
736
+ | Array<{ finish_reason?: string }>
737
+ | undefined;
738
+ const finish =
739
+ choices?.[0]?.finish_reason ?? (chunk.finish_reason as string | undefined);
740
+ if (finish != null) s = { ...s, finishReason: finish };
741
+ return s;
742
+ };
743
+
744
+ const handleNativeText = (
745
+ state: StreamState,
746
+ chunk: Record<string, unknown>,
747
+ parts: StreamParts,
748
+ idGen: IdGenerator.Service,
749
+ ): Effect.Effect<StreamState> => {
750
+ const native = chunk.response;
751
+ if (native == null || native === "") return Effect.succeed(state);
752
+ const text =
753
+ typeof native === "object" ? JSON.stringify(native) : String(native);
754
+ if (text.length === 0) return Effect.succeed(state);
755
+ return emitTextDelta(state, text, parts, idGen);
756
+ };
757
+
758
+ const handleNativeToolCalls = (
759
+ state: StreamState,
760
+ chunk: Record<string, unknown>,
761
+ parts: StreamParts,
762
+ idGen: IdGenerator.Service,
763
+ ): Effect.Effect<StreamState> => {
764
+ if (!Array.isArray(chunk.tool_calls)) return Effect.succeed(state);
765
+ return Effect.gen(function* () {
766
+ const s = closeReasoning(state, parts);
767
+ return yield* handleToolDeltas(
768
+ s,
769
+ chunk.tool_calls as ReadonlyArray<Record<string, unknown>>,
770
+ parts,
771
+ idGen,
772
+ );
773
+ });
774
+ };
775
+
776
+ const handleOpenAiDelta = (
777
+ state: StreamState,
778
+ chunk: Record<string, unknown>,
779
+ parts: StreamParts,
780
+ idGen: IdGenerator.Service,
781
+ ): Effect.Effect<StreamState> => {
782
+ const delta = (
783
+ chunk.choices as Array<{ delta?: Record<string, unknown> }> | undefined
784
+ )?.[0]?.delta;
785
+ if (!delta) return Effect.succeed(state);
786
+ return Effect.gen(function* () {
787
+ let s = state;
788
+ const reasoning = (delta.reasoning_content ?? delta.reasoning) as
789
+ | string
790
+ | undefined;
791
+ if (reasoning && reasoning.length > 0) {
792
+ s = yield* emitReasoningDelta(s, reasoning, parts, idGen);
793
+ }
794
+ const text = delta.content as string | undefined;
795
+ if (text && text.length > 0) {
796
+ s = yield* emitTextDelta(s, text, parts, idGen);
797
+ }
798
+ const toolDeltas = delta.tool_calls as
799
+ | ReadonlyArray<Record<string, unknown>>
800
+ | undefined;
801
+ if (Array.isArray(toolDeltas)) {
802
+ s = closeReasoning(s, parts);
803
+ s = yield* handleToolDeltas(s, toolDeltas, parts, idGen);
804
+ }
805
+ return s;
806
+ });
807
+ };
808
+
809
+ const handleStreamChunk = (
810
+ state: StreamState,
811
+ data: string,
812
+ idGen: IdGenerator.Service,
813
+ ): Effect.Effect<
814
+ readonly [StreamState, ReadonlyArray<Response.StreamPartEncoded>]
815
+ > =>
816
+ Effect.gen(function* () {
817
+ if (data === "") return [state, []] as const;
818
+ if (data === "[DONE]") {
819
+ return [{ ...state, receivedDone: true }, []] as const;
820
+ }
821
+ const chunk = tryParseJson(data);
822
+ if (chunk === undefined) return [state, []] as const;
823
+
824
+ const parts: StreamParts = [];
825
+ let s: StreamState = { ...state, receivedAnyData: true };
826
+ s = updateChunkMeta(s, chunk);
827
+ s = yield* handleNativeText(s, chunk, parts, idGen);
828
+ s = yield* handleNativeToolCalls(s, chunk, parts, idGen);
829
+ s = yield* handleOpenAiDelta(s, chunk, parts, idGen);
830
+ return [s, parts] as const;
831
+ });
832
+
833
+ const finalizeStream = (
834
+ state: StreamState,
835
+ ): ReadonlyArray<Response.StreamPartEncoded> => {
836
+ const parts: StreamParts = [];
837
+ let s = state;
838
+ for (const [idx] of s.toolCalls) s = closeToolCall(s, idx, parts);
839
+ s = closeReasoning(s, parts);
840
+ if (s.textId !== undefined) parts.push({ type: "text-end", id: s.textId });
841
+
842
+ // Three cases for the final reason:
843
+ // 1. The model emitted an explicit `finish_reason` → map it.
844
+ // 2. The stream ended cleanly (`[DONE]` seen) but no reason → "stop".
845
+ // Workers AI's native shape never includes `finish_reason`,
846
+ // so without this rule every native-mode stream would report
847
+ // `unknown` despite completing successfully.
848
+ // 3. The stream ended abnormally (no `[DONE]`, no reason) → "error".
849
+ const reason: Response.FinishReason =
850
+ s.finishReason !== undefined
851
+ ? mapFinishReason(s.finishReason)
852
+ : s.receivedDone
853
+ ? "stop"
854
+ : s.receivedAnyData
855
+ ? "error"
856
+ : "unknown";
857
+
858
+ parts.push({
859
+ type: "finish",
860
+ reason,
861
+ usage: mapUsage(s.usage),
862
+ response: undefined,
863
+ });
864
+ return parts;
865
+ };
866
+
867
+ const parseStreamText = (
868
+ resp: Response,
869
+ idGen: IdGenerator.Service,
870
+ ): Stream.Stream<Response.StreamPartEncoded, AiError.AiError> => {
871
+ const body = resp.body;
872
+ if (body === null) {
873
+ return Stream.fromIterable<Response.StreamPartEncoded>(
874
+ finalizeStream(initialStreamState()),
875
+ );
876
+ }
877
+ return Stream.fromReadableStream<Uint8Array, AiError.AiError>({
878
+ evaluate: () => body,
879
+ onError: (cause) => toAiError(cause, "streamText"),
880
+ }).pipe(
881
+ Stream.decodeText(),
882
+ Stream.pipeThroughChannel(Sse.decode<AiError.AiError, unknown>()),
883
+ Stream.catchTag("Retry", (retry) => Stream.die(retry)),
884
+ Stream.mapAccumEffect(
885
+ initialStreamState,
886
+ (state, event) => handleStreamChunk(state, event.data, idGen),
887
+ { onHalt: (state) => finalizeStream(state) },
888
+ ),
889
+ );
890
+ };
891
+
892
+ // ---------------------------------------------------------------------------
893
+ // Error mapping
894
+ // ---------------------------------------------------------------------------
895
+
896
+ const toAiError = (
897
+ cause: unknown,
898
+ method: "generateText" | "streamText",
899
+ ): AiError.AiError =>
900
+ AiError.AiError.make({
901
+ module: "Cloudflare.AiGateway.LanguageModel",
902
+ method,
903
+ reason: new AiError.UnknownError({
904
+ description:
905
+ cause instanceof Error ? cause.message : "AI Gateway request failed",
906
+ }),
907
+ });