@aphrody/m3-ai 3.3.4 → 3.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,400 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // Streaming client of an OpenAI-compatible `/v1/chat/completions` endpoint (llama-server,
3
+ // vLLM, the Aphrody gateway...): SSE framing, chunk decoding, tool-call accumulation, a pure
4
+ // reducer for the conversation state, and typed errors that tell "model unavailable" apart.
5
+ // Pure module: no React, no DOM; `fetch` is injectable.
6
+
7
+ import type { Citation } from "./citations";
8
+
9
+ export type ChatRole = "system" | "user" | "assistant" | "tool";
10
+
11
+ /** One complete tool call, as OpenAI returns it once every delta is merged. */
12
+ export interface OpenAIToolCall {
13
+ id: string;
14
+ type: "function";
15
+ function: { name: string; arguments: string };
16
+ }
17
+
18
+ /** One message of the conversation, in the shape the chat components render. */
19
+ export interface ChatMessage {
20
+ id: string;
21
+ role: ChatRole;
22
+ content: string;
23
+ /** `reasoning_content` deltas (reasoning models), kept apart from the answer. */
24
+ reasoning?: string;
25
+ toolCalls?: OpenAIToolCall[];
26
+ /** For a `tool` message: the call it answers. */
27
+ toolCallId?: string;
28
+ finishReason?: string | null;
29
+ /** Grounded answer: the numbered passages `[n]` the text refers to. */
30
+ citations?: Citation[];
31
+ /** The backend refused to answer (out of the sources' scope). */
32
+ refusal?: boolean;
33
+ }
34
+
35
+ /** What one SSE chunk contributes. */
36
+ export type ChatStreamEvent =
37
+ | { type: "content"; text: string }
38
+ | { type: "reasoning"; text: string }
39
+ | {
40
+ type: "tool_call_delta";
41
+ index: number;
42
+ id?: string;
43
+ name?: string;
44
+ arguments?: string;
45
+ }
46
+ | { type: "citations"; citations: Citation[] }
47
+ | { type: "refusal"; text?: string }
48
+ | { type: "finish"; reason: string }
49
+ | { type: "done" };
50
+
51
+ export type ChatStreamErrorKind = "network" | "http" | "aborted" | "parse" | "stream";
52
+
53
+ /** Typed failure of a chat request. `unavailable` marks a model that cannot answer right now. */
54
+ export class ChatStreamError extends Error {
55
+ readonly kind: ChatStreamErrorKind;
56
+ readonly status?: number;
57
+ readonly unavailable: boolean;
58
+
59
+ constructor(
60
+ kind: ChatStreamErrorKind,
61
+ message: string,
62
+ options: { status?: number; cause?: unknown } = {},
63
+ ) {
64
+ super(message, options.cause === undefined ? undefined : { cause: options.cause });
65
+ this.name = "ChatStreamError";
66
+ this.kind = kind;
67
+ this.status = options.status;
68
+ this.unavailable = kind === "network" || isUnavailableStatus(options.status);
69
+ }
70
+ }
71
+
72
+ /** HTTP statuses that mean "no model to answer" rather than "bad request". */
73
+ export function isUnavailableStatus(status: number | undefined): boolean {
74
+ return status === 404 || status === 502 || status === 503 || status === 504;
75
+ }
76
+
77
+ /**
78
+ * Incremental SSE parser: feed it decoded text in arbitrary pieces, it calls `onData` with the
79
+ * payload of every complete event (multi-line `data:` fields joined by `\n`). Comments and
80
+ * other fields are ignored.
81
+ */
82
+ export function createSseParser(onData: (data: string) => void): {
83
+ push(chunk: string): void;
84
+ flush(): void;
85
+ } {
86
+ let buffer = "";
87
+ let data: string[] = [];
88
+ const dispatch = () => {
89
+ if (data.length > 0) onData(data.join("\n"));
90
+ data = [];
91
+ };
92
+ const line = (raw: string) => {
93
+ const text = raw.endsWith("\r") ? raw.slice(0, -1) : raw;
94
+ if (text === "") return dispatch();
95
+ if (text.startsWith(":")) return;
96
+ const colon = text.indexOf(":");
97
+ const field = colon === -1 ? text : text.slice(0, colon);
98
+ let value = colon === -1 ? "" : text.slice(colon + 1);
99
+ if (value.startsWith(" ")) value = value.slice(1);
100
+ if (field === "data") data.push(value);
101
+ };
102
+ return {
103
+ push(chunk) {
104
+ buffer += chunk;
105
+ let newline = buffer.indexOf("\n");
106
+ while (newline !== -1) {
107
+ line(buffer.slice(0, newline));
108
+ buffer = buffer.slice(newline + 1);
109
+ newline = buffer.indexOf("\n");
110
+ }
111
+ },
112
+ flush() {
113
+ if (buffer.length > 0) line(buffer);
114
+ buffer = "";
115
+ dispatch();
116
+ },
117
+ };
118
+ }
119
+
120
+ interface RawChunk {
121
+ choices?: {
122
+ delta?: {
123
+ content?: string | null;
124
+ reasoning_content?: string | null;
125
+ refusal?: string | null;
126
+ tool_calls?: {
127
+ index?: number;
128
+ id?: string;
129
+ function?: { name?: string; arguments?: string };
130
+ }[];
131
+ };
132
+ finish_reason?: string | null;
133
+ }[];
134
+ citations?: Citation[];
135
+ refusal?: boolean;
136
+ error?: { message?: string } | string;
137
+ }
138
+
139
+ /**
140
+ * Decode one `data:` payload of a chat-completions stream into events. `[DONE]` yields `done`.
141
+ * A payload carrying `error` throws a `stream` error; malformed JSON throws a `parse` error.
142
+ * Top-level `citations` / `refusal: true` (grounded backends such as Shenron) are passed on.
143
+ */
144
+ export function parseChatCompletionChunk(payload: string): ChatStreamEvent[] {
145
+ const trimmed = payload.trim();
146
+ if (trimmed === "") return [];
147
+ if (trimmed === "[DONE]") return [{ type: "done" }];
148
+ let chunk: RawChunk;
149
+ try {
150
+ chunk = JSON.parse(trimmed) as RawChunk;
151
+ } catch (cause) {
152
+ throw new ChatStreamError("parse", "invalid chat-completions chunk", { cause });
153
+ }
154
+ if (chunk.error) {
155
+ const message = typeof chunk.error === "string" ? chunk.error : chunk.error.message;
156
+ throw new ChatStreamError("stream", message || "stream error");
157
+ }
158
+ const events: ChatStreamEvent[] = [];
159
+ for (const choice of chunk.choices ?? []) {
160
+ const delta = choice.delta;
161
+ if (delta?.reasoning_content) events.push({ type: "reasoning", text: delta.reasoning_content });
162
+ if (delta?.content) events.push({ type: "content", text: delta.content });
163
+ if (delta?.refusal) events.push({ type: "refusal", text: delta.refusal });
164
+ for (const call of delta?.tool_calls ?? []) {
165
+ events.push({
166
+ type: "tool_call_delta",
167
+ index: call.index ?? 0,
168
+ id: call.id,
169
+ name: call.function?.name,
170
+ arguments: call.function?.arguments,
171
+ });
172
+ }
173
+ if (choice.finish_reason) events.push({ type: "finish", reason: choice.finish_reason });
174
+ }
175
+ if (Array.isArray(chunk.citations))
176
+ events.push({ type: "citations", citations: chunk.citations });
177
+ if (chunk.refusal === true) events.push({ type: "refusal" });
178
+ return events;
179
+ }
180
+
181
+ /** Merge a tool-call delta into the calls accumulated so far (returns a new array). */
182
+ export function accumulateToolCall(
183
+ calls: readonly OpenAIToolCall[],
184
+ delta: Extract<ChatStreamEvent, { type: "tool_call_delta" }>,
185
+ ): OpenAIToolCall[] {
186
+ const next = calls.slice();
187
+ const current = next[delta.index] ?? {
188
+ id: "",
189
+ type: "function" as const,
190
+ function: { name: "", arguments: "" },
191
+ };
192
+ next[delta.index] = {
193
+ id: delta.id ?? current.id,
194
+ type: "function",
195
+ function: {
196
+ name: current.function.name + (delta.name ?? ""),
197
+ arguments: current.function.arguments + (delta.arguments ?? ""),
198
+ },
199
+ };
200
+ return next;
201
+ }
202
+
203
+ /** The request body messages, in the OpenAI wire shape. */
204
+ export function toOpenAIMessages(messages: readonly ChatMessage[]): Record<string, unknown>[] {
205
+ return messages.map((message) => {
206
+ const wire: Record<string, unknown> = { role: message.role, content: message.content };
207
+ if (message.toolCalls?.length) wire.tool_calls = message.toolCalls;
208
+ if (message.toolCallId) wire.tool_call_id = message.toolCallId;
209
+ return wire;
210
+ });
211
+ }
212
+
213
+ export interface StreamChatOptions {
214
+ /** API root including `/v1` (e.g. `http://127.0.0.1:8080/v1`). */
215
+ baseUrl: string;
216
+ model?: string;
217
+ messages: readonly ChatMessage[];
218
+ /** Extra body fields (`temperature`, `tools`, `max_tokens`...). */
219
+ body?: Record<string, unknown>;
220
+ headers?: Record<string, string>;
221
+ signal?: AbortSignal;
222
+ fetch?: typeof fetch;
223
+ }
224
+
225
+ /** `<baseUrl>/chat/completions`, tolerant of a trailing slash. */
226
+ export function chatCompletionsUrl(baseUrl: string): string {
227
+ return `${baseUrl.replace(/\/+$/, "")}/chat/completions`;
228
+ }
229
+
230
+ /**
231
+ * POST a streaming chat completion and yield its events until `[DONE]` or the end of the body.
232
+ *
233
+ * Throws `ChatStreamError`: `network` (fetch rejected: server down), `http` (non-2xx status;
234
+ * `unavailable` for 404/502/503/504), `aborted` (the signal fired), `parse` / `stream`.
235
+ */
236
+ export async function* streamChatCompletion(
237
+ options: StreamChatOptions,
238
+ ): AsyncGenerator<ChatStreamEvent, void, undefined> {
239
+ const doFetch = options.fetch ?? globalThis.fetch;
240
+ let response: Response;
241
+ try {
242
+ response = await doFetch(chatCompletionsUrl(options.baseUrl), {
243
+ method: "POST",
244
+ headers: {
245
+ "content-type": "application/json",
246
+ accept: "text/event-stream",
247
+ ...options.headers,
248
+ },
249
+ body: JSON.stringify({
250
+ ...options.body,
251
+ ...(options.model ? { model: options.model } : {}),
252
+ messages: toOpenAIMessages(options.messages),
253
+ stream: true,
254
+ }),
255
+ signal: options.signal,
256
+ });
257
+ } catch (cause) {
258
+ if (options.signal?.aborted) throw new ChatStreamError("aborted", "request aborted", { cause });
259
+ throw new ChatStreamError("network", "chat endpoint unreachable", { cause });
260
+ }
261
+ if (!response.ok) {
262
+ let detail = "";
263
+ try {
264
+ detail = (await response.text()).slice(0, 500);
265
+ } catch {
266
+ // body unreadable: the status alone is reported
267
+ }
268
+ throw new ChatStreamError("http", detail || `HTTP ${response.status}`, {
269
+ status: response.status,
270
+ });
271
+ }
272
+ if (!response.body) throw new ChatStreamError("stream", "response has no body");
273
+
274
+ const queue: string[] = [];
275
+ const parser = createSseParser((data) => queue.push(data));
276
+ const reader = response.body.getReader();
277
+ const decoder = new TextDecoder();
278
+ try {
279
+ for (;;) {
280
+ let result: Awaited<ReturnType<typeof reader.read>>;
281
+ try {
282
+ // oxlint-disable-next-line no-await-in-loop -- SSE chunks are read in order, one at a time.
283
+ result = await reader.read();
284
+ } catch (cause) {
285
+ if (options.signal?.aborted)
286
+ throw new ChatStreamError("aborted", "request aborted", { cause });
287
+ throw new ChatStreamError("stream", "stream interrupted", { cause });
288
+ }
289
+ if (result.done) {
290
+ parser.push(decoder.decode());
291
+ parser.flush();
292
+ } else {
293
+ parser.push(decoder.decode(result.value, { stream: true }));
294
+ }
295
+ while (queue.length > 0) {
296
+ for (const event of parseChatCompletionChunk(queue.shift()!)) {
297
+ yield event;
298
+ if (event.type === "done") return;
299
+ }
300
+ }
301
+ if (result.done) return;
302
+ }
303
+ } finally {
304
+ reader.releaseLock();
305
+ }
306
+ }
307
+
308
+ // ── Conversation state ─────────────────────────────────────────────────────────────────────
309
+
310
+ export type ChatStatus = "idle" | "streaming" | "error" | "unavailable";
311
+
312
+ export interface ChatState {
313
+ messages: ChatMessage[];
314
+ status: ChatStatus;
315
+ error: ChatStreamError | null;
316
+ }
317
+
318
+ export type ChatAction =
319
+ | { type: "start"; user: ChatMessage; assistantId: string }
320
+ | { type: "event"; assistantId: string; event: ChatStreamEvent }
321
+ | { type: "finish" }
322
+ | { type: "fail"; error: ChatStreamError }
323
+ | { type: "stop" }
324
+ | { type: "reset"; messages?: ChatMessage[] }
325
+ | { type: "set"; messages: ChatMessage[] };
326
+
327
+ export function initialChatState(messages: ChatMessage[] = []): ChatState {
328
+ return { messages, status: "idle", error: null };
329
+ }
330
+
331
+ function applyEvent(message: ChatMessage, event: ChatStreamEvent): ChatMessage {
332
+ switch (event.type) {
333
+ case "content":
334
+ return { ...message, content: message.content + event.text };
335
+ case "reasoning":
336
+ return { ...message, reasoning: (message.reasoning ?? "") + event.text };
337
+ case "tool_call_delta":
338
+ return { ...message, toolCalls: accumulateToolCall(message.toolCalls ?? [], event) };
339
+ case "citations":
340
+ return { ...message, citations: event.citations };
341
+ case "refusal":
342
+ return { ...message, refusal: true };
343
+ case "finish":
344
+ return { ...message, finishReason: event.reason };
345
+ case "done":
346
+ return message;
347
+ }
348
+ }
349
+
350
+ /** Pure reducer of a streamed conversation (the state behind `useOpenAIChat`). */
351
+ export function chatReducer(state: ChatState, action: ChatAction): ChatState {
352
+ switch (action.type) {
353
+ case "start":
354
+ return {
355
+ status: "streaming",
356
+ error: null,
357
+ messages: [
358
+ ...state.messages,
359
+ action.user,
360
+ { id: action.assistantId, role: "assistant", content: "" },
361
+ ],
362
+ };
363
+ case "event":
364
+ return {
365
+ ...state,
366
+ messages: state.messages.map((message) =>
367
+ message.id === action.assistantId ? applyEvent(message, action.event) : message,
368
+ ),
369
+ };
370
+ case "finish":
371
+ case "stop":
372
+ return { ...state, status: "idle" };
373
+ case "fail": {
374
+ // An empty assistant turn left by a failed request is dropped.
375
+ const last = state.messages.at(-1);
376
+ const messages =
377
+ last && last.role === "assistant" && !last.content && !last.toolCalls?.length
378
+ ? state.messages.slice(0, -1)
379
+ : state.messages;
380
+ return {
381
+ messages,
382
+ status: action.error.unavailable ? "unavailable" : "error",
383
+ error: action.error,
384
+ };
385
+ }
386
+ case "reset":
387
+ return initialChatState(action.messages ?? []);
388
+ case "set":
389
+ return { ...state, messages: action.messages };
390
+ }
391
+ }
392
+
393
+ let counter = 0;
394
+ /** Message id: `crypto.randomUUID` when present, else a process-local counter. */
395
+ export function newMessageId(): string {
396
+ const uuid = globalThis.crypto?.randomUUID?.();
397
+ if (uuid) return uuid;
398
+ counter += 1;
399
+ return `msg-${Date.now().toString(36)}-${counter}`;
400
+ }
@@ -0,0 +1,126 @@
1
+ // SPDX-License-Identifier: Apache-2.0
2
+ // React hook over `streamChatCompletion`: one conversation streamed from an OpenAI-compatible
3
+ // endpoint, with stop, reset and the `unavailable` status of a model that cannot answer.
4
+
5
+ "use client";
6
+
7
+ import { useCallback, useEffect, useReducer, useRef } from "react";
8
+
9
+ import {
10
+ type ChatMessage,
11
+ type ChatStatus,
12
+ ChatStreamError,
13
+ chatReducer,
14
+ initialChatState,
15
+ newMessageId,
16
+ streamChatCompletion,
17
+ } from "./openaiChatStream";
18
+
19
+ export interface UseOpenAIChatOptions {
20
+ /** API root including `/v1` (e.g. `http://127.0.0.1:8080/v1`). */
21
+ baseUrl: string;
22
+ model?: string;
23
+ fetch?: typeof fetch;
24
+ headers?: Record<string, string>;
25
+ initialMessages?: ChatMessage[];
26
+ /** Messages sent before the conversation on every request (system prompt...). */
27
+ preamble?: ChatMessage[];
28
+ /** Extra body fields (`temperature`, `tools`...). */
29
+ body?: Record<string, unknown>;
30
+ /** Called with the finished assistant message. */
31
+ onFinish?: (message: ChatMessage) => void;
32
+ onError?: (error: ChatStreamError) => void;
33
+ }
34
+
35
+ export interface UseOpenAIChatResult {
36
+ messages: ChatMessage[];
37
+ status: ChatStatus;
38
+ error: ChatStreamError | null;
39
+ /** Send a user message and stream the answer. Ignored while streaming. */
40
+ send: (content: string) => Promise<void>;
41
+ /** Abort the running request; the partial answer stays. */
42
+ stop: () => void;
43
+ /** Abort and clear (or replace) the conversation. */
44
+ reset: (messages?: ChatMessage[]) => void;
45
+ setMessages: (messages: ChatMessage[]) => void;
46
+ }
47
+
48
+ export function useOpenAIChat(options: UseOpenAIChatOptions): UseOpenAIChatResult {
49
+ const [state, dispatch] = useReducer(chatReducer, options.initialMessages, initialChatState);
50
+ const controller = useRef<AbortController | null>(null);
51
+ const latest = useRef({ options, state });
52
+
53
+ useEffect(() => {
54
+ latest.current = { options, state };
55
+ });
56
+ useEffect(() => () => controller.current?.abort(), []);
57
+
58
+ const send = useCallback(async (content: string) => {
59
+ const { options: o, state: s } = latest.current;
60
+ if (s.status === "streaming" || controller.current || !content.trim()) return;
61
+ const user: ChatMessage = { id: newMessageId(), role: "user", content };
62
+ const assistantId = newMessageId();
63
+ const abort = new AbortController();
64
+ controller.current = abort;
65
+ dispatch({ type: "start", user, assistantId });
66
+ let assistant: ChatMessage = { id: assistantId, role: "assistant", content: "" };
67
+ try {
68
+ for await (const event of streamChatCompletion({
69
+ baseUrl: o.baseUrl,
70
+ model: o.model,
71
+ messages: [...(o.preamble ?? []), ...s.messages, user],
72
+ body: o.body,
73
+ headers: o.headers,
74
+ fetch: o.fetch,
75
+ signal: abort.signal,
76
+ })) {
77
+ dispatch({ type: "event", assistantId, event });
78
+ assistant = chatReducer(
79
+ { messages: [assistant], status: "streaming", error: null },
80
+ { type: "event", assistantId, event },
81
+ ).messages[0]!;
82
+ }
83
+ dispatch({ type: "finish" });
84
+ o.onFinish?.(assistant);
85
+ } catch (cause) {
86
+ const error =
87
+ cause instanceof ChatStreamError
88
+ ? cause
89
+ : new ChatStreamError("stream", String(cause), { cause });
90
+ if (error.kind === "aborted" || abort.signal.aborted) {
91
+ dispatch({ type: "stop" });
92
+ } else {
93
+ dispatch({ type: "fail", error });
94
+ o.onError?.(error);
95
+ }
96
+ } finally {
97
+ if (controller.current === abort) controller.current = null;
98
+ }
99
+ }, []);
100
+
101
+ const stop = useCallback(() => {
102
+ controller.current?.abort();
103
+ controller.current = null;
104
+ dispatch({ type: "stop" });
105
+ }, []);
106
+
107
+ const reset = useCallback((messages?: ChatMessage[]) => {
108
+ controller.current?.abort();
109
+ controller.current = null;
110
+ dispatch({ type: "reset", messages });
111
+ }, []);
112
+
113
+ const setMessages = useCallback((messages: ChatMessage[]) => {
114
+ dispatch({ type: "set", messages });
115
+ }, []);
116
+
117
+ return {
118
+ messages: state.messages,
119
+ status: state.status,
120
+ error: state.error,
121
+ send,
122
+ stop,
123
+ reset,
124
+ setMessages,
125
+ };
126
+ }