@volter/twin-openai 0.1.2 → 2.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -30
- package/defaults/handlers.json +10 -0
- package/dist/defaults/handlers.json +10 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +29 -0
- package/dist/src/generated/surface.gen.json +1 -0
- package/dist/src/generated/ui.gen.json +1 -0
- package/dist/src/index.d.ts +19 -0
- package/dist/src/index.js +72 -0
- package/dist/src/manifest.d.ts +6 -0
- package/dist/src/manifest.js +323 -0
- package/dist/src/openai-budget.d.ts +53 -0
- package/dist/src/openai-budget.js +147 -0
- package/dist/src/openai-capabilities.d.ts +4 -0
- package/dist/src/openai-capabilities.js +1569 -0
- package/dist/src/openai-conformance.d.ts +13 -0
- package/dist/src/openai-conformance.js +116 -0
- package/dist/src/openai-connector.d.ts +86 -0
- package/dist/src/openai-connector.js +291 -0
- package/dist/src/openai-media.d.ts +43 -0
- package/dist/src/openai-media.js +257 -0
- package/dist/src/openai-models.d.ts +74 -0
- package/dist/src/openai-models.js +148 -0
- package/dist/src/openai-scenario.d.ts +51 -0
- package/dist/src/openai-scenario.js +166 -0
- package/dist/src/openai-server.d.ts +40 -0
- package/dist/src/openai-server.js +126 -0
- package/dist/src/openai-stub.d.ts +82 -0
- package/dist/src/openai-stub.js +256 -0
- package/dist/src/openai-twin.d.ts +182 -0
- package/dist/src/openai-twin.js +1117 -0
- package/dist/src/openai-types.d.ts +194 -0
- package/dist/src/openai-types.js +4 -0
- package/dist/src/openai-webhooks.d.ts +47 -0
- package/dist/src/openai-webhooks.js +99 -0
- package/dist/src/screens/api-keys.d.ts +16 -0
- package/dist/src/screens/api-keys.js +131 -0
- package/dist/src/screens/session.d.ts +22 -0
- package/dist/src/screens/session.js +115 -0
- package/dist/src/semantics/assistants.d.ts +2 -0
- package/dist/src/semantics/assistants.js +331 -0
- package/dist/src/semantics/audio.d.ts +2 -0
- package/dist/src/semantics/audio.js +27 -0
- package/dist/src/semantics/batches.d.ts +4 -0
- package/dist/src/semantics/batches.js +86 -0
- package/dist/src/semantics/chat-completions.d.ts +3 -0
- package/dist/src/semantics/chat-completions.js +58 -0
- package/dist/src/semantics/containers.d.ts +2 -0
- package/dist/src/semantics/containers.js +147 -0
- package/dist/src/semantics/embeddings.d.ts +2 -0
- package/dist/src/semantics/embeddings.js +13 -0
- package/dist/src/semantics/evals.d.ts +2 -0
- package/dist/src/semantics/evals.js +173 -0
- package/dist/src/semantics/files.d.ts +13 -0
- package/dist/src/semantics/files.js +59 -0
- package/dist/src/semantics/fine-tuning.d.ts +4 -0
- package/dist/src/semantics/fine-tuning.js +178 -0
- package/dist/src/semantics/images.d.ts +2 -0
- package/dist/src/semantics/images.js +18 -0
- package/dist/src/semantics/index.d.ts +8 -0
- package/dist/src/semantics/index.js +46 -0
- package/dist/src/semantics/models.d.ts +2 -0
- package/dist/src/semantics/models.js +34 -0
- package/dist/src/semantics/moderations.d.ts +2 -0
- package/dist/src/semantics/moderations.js +12 -0
- package/dist/src/semantics/organization.d.ts +2 -0
- package/dist/src/semantics/organization.js +67 -0
- package/dist/src/semantics/progress.d.ts +22 -0
- package/dist/src/semantics/progress.js +63 -0
- package/dist/src/semantics/responses.d.ts +3 -0
- package/dist/src/semantics/responses.js +153 -0
- package/dist/src/semantics/shared.d.ts +32 -0
- package/dist/src/semantics/shared.js +69 -0
- package/dist/src/semantics/uploads.d.ts +2 -0
- package/dist/src/semantics/uploads.js +84 -0
- package/dist/src/semantics/vector-stores.d.ts +2 -0
- package/dist/src/semantics/vector-stores.js +281 -0
- package/dist/test-fixtures/openai-openapi-operations.SOURCE.md +18 -0
- package/dist/test-fixtures/openai-openapi-operations.json +1849 -0
- package/package.json +21 -10
- package/src/cli.ts +9 -7
- package/src/generated/surface.gen.json +1 -0
- package/src/generated/ui.gen.json +1 -0
- package/src/index.ts +20 -10
- package/src/manifest.ts +343 -0
- package/src/openai-budget.ts +4 -4
- package/src/openai-capabilities.ts +177 -195
- package/src/openai-conformance.ts +1 -1
- package/src/openai-connector.ts +40 -43
- package/src/openai-media.ts +225 -0
- package/src/openai-models.ts +145 -15
- package/src/openai-scenario.ts +46 -10
- package/src/openai-server.ts +65 -108
- package/src/openai-stub.ts +54 -30
- package/src/openai-twin.ts +760 -1665
- package/src/openai-types.ts +24 -6
- package/src/openai-webhooks.ts +3 -2
- package/src/screens/api-keys.tsx +138 -0
- package/src/screens/session.tsx +131 -0
- package/src/semantics/assistants.ts +336 -0
- package/src/semantics/audio.ts +31 -0
- package/src/semantics/batches.ts +88 -0
- package/src/semantics/chat-completions.ts +66 -0
- package/src/semantics/containers.ts +151 -0
- package/src/semantics/embeddings.ts +19 -0
- package/src/semantics/evals.ts +182 -0
- package/src/semantics/files.ts +67 -0
- package/src/semantics/fine-tuning.ts +185 -0
- package/src/semantics/images.ts +23 -0
- package/src/semantics/index.ts +52 -0
- package/src/semantics/models.ts +41 -0
- package/src/semantics/moderations.ts +14 -0
- package/src/semantics/organization.ts +76 -0
- package/src/semantics/progress.ts +72 -0
- package/src/semantics/responses.ts +151 -0
- package/src/semantics/shared.ts +82 -0
- package/src/semantics/uploads.ts +92 -0
- package/src/semantics/vector-stores.ts +279 -0
- package/test-fixtures/openai-openapi-operations.SOURCE.md +4 -5
- package/test-fixtures/openai-openapi-operations.json +224 -1334
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
// Assistants API (beta) semantics: assistants, threads, their messages, and runs with their steps.
|
|
2
|
+
// Reads and deletes are the derived core's: an assistant, a thread, and a thread's messages (each
|
|
3
|
+
// under its thread), and cancelling a run (the machine in ../manifest.ts moves its `status`). A run
|
|
4
|
+
// is created `queued`, as OpenAI creates it; it cannot invoke a model, so when a read of it first
|
|
5
|
+
// looks (./progress.ts) a run with function tools stops at `requires_action` for them (the placeholder
|
|
6
|
+
// model calls them) and, once their outputs are submitted, completes, appending a labeled stub reply to
|
|
7
|
+
// its thread; its steps (a tool-call step, a message-creation step) are read off the run, never stored;
|
|
8
|
+
// a cancelled run ends `cancelled`.
|
|
9
|
+
import type { Semantics, SemanticsContext } from '@volter/world-core';
|
|
10
|
+
import { estimateTokens, stubToolCall } from '../openai-stub.ts';
|
|
11
|
+
import { coreAnswer, progress } from './progress.ts';
|
|
12
|
+
import { at, epoch, invalid, missing, objectOr, pick, type Row } from './shared.ts';
|
|
13
|
+
|
|
14
|
+
const ASSISTANT = 'AssistantObject';
|
|
15
|
+
const THREAD = 'ThreadObject';
|
|
16
|
+
const MESSAGE = 'MessageObject';
|
|
17
|
+
const RUN = 'RunObject';
|
|
18
|
+
|
|
19
|
+
// ── assistants ──────────────────────────────────────────────────────────────────────────────
|
|
20
|
+
|
|
21
|
+
/** An assistant's tool resources: as sent, with the empty store list a file_search tool searches when none is named
|
|
22
|
+
* (the reference's modify example answers `"tool_resources": {"file_search": {"vector_store_ids": []}}` for an
|
|
23
|
+
* assistant given the file_search tool and no stores: https://platform.openai.com/docs/api-reference/assistants/modifyAssistant). */
|
|
24
|
+
function toolResources(tools: unknown, sent: unknown): Row {
|
|
25
|
+
const out = { ...((sent && typeof sent === 'object' ? sent : {}) as Row) };
|
|
26
|
+
if (Array.isArray(tools) && tools.some((t) => (t as { type?: unknown })?.type === 'file_search') && !out.file_search) out.file_search = { vector_store_ids: [] };
|
|
27
|
+
// and the empty file list a code interpreter works on (the reference's listRuns and getThread examples answer
|
|
28
|
+
// `"code_interpreter": {"file_ids": []}`)
|
|
29
|
+
if (Array.isArray(tools) && tools.some((t) => (t as { type?: unknown })?.type === 'code_interpreter') && !out.code_interpreter) out.code_interpreter = { file_ids: [] };
|
|
30
|
+
return out;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
const createAssistant: Semantics = async (ctx) => {
|
|
34
|
+
const p = ctx.params;
|
|
35
|
+
if (p.model === undefined || p.model === '') return invalid(ctx, 'you must provide a model parameter', 'model');
|
|
36
|
+
const tools = Array.isArray(p.tools) ? p.tools : [];
|
|
37
|
+
const fields = {
|
|
38
|
+
object: 'assistant', created_at: epoch(ctx),
|
|
39
|
+
name: p.name ?? null, description: p.description ?? null,
|
|
40
|
+
model: String(p.model), instructions: p.instructions ?? null,
|
|
41
|
+
tools,
|
|
42
|
+
tool_resources: toolResources(tools, p.tool_resources),
|
|
43
|
+
metadata: objectOr(p.metadata, {}),
|
|
44
|
+
// sampling "Defaults to 1" (https://platform.openai.com/docs/api-reference/assistants/object, `temperature`, `top_p`)
|
|
45
|
+
temperature: p.temperature ?? 1, top_p: p.top_p ?? 1,
|
|
46
|
+
response_format: p.response_format ?? 'auto',
|
|
47
|
+
};
|
|
48
|
+
return ctx.reply(await ctx.write(ASSISTANT, ctx.mint(ASSISTANT), fields, 'assistant.create'));
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
// modify replaces only the fields the request names
|
|
52
|
+
const modifyAssistant: Semantics = async (ctx) => {
|
|
53
|
+
const id = String(ctx.id);
|
|
54
|
+
const current = ctx.get(ASSISTANT, id);
|
|
55
|
+
if (!current) return ctx.notFound(ASSISTANT, id);
|
|
56
|
+
const fields = pick(ctx.params, ['name', 'description', 'model', 'instructions', 'tools', 'tool_resources', 'metadata', 'temperature', 'top_p', 'response_format']);
|
|
57
|
+
if (fields.tools !== undefined || fields.tool_resources !== undefined) fields.tool_resources = toolResources(fields.tools ?? current.tools, fields.tool_resources ?? current.tool_resources);
|
|
58
|
+
return ctx.reply(await ctx.write(ASSISTANT, id, fields, 'assistant.update'));
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
// ── threads and messages ────────────────────────────────────────────────────────────────────
|
|
62
|
+
|
|
63
|
+
/** The id of the next message appended to a thread: the thread's next sequence number. */
|
|
64
|
+
const nextMessage = (ctx: SemanticsContext, threadId: string): { seq: number; id: string } => {
|
|
65
|
+
const seq = ctx.rowsRaw(MESSAGE, { withDeleted: true }).filter((r) => r.thread_id === threadId).length + 1;
|
|
66
|
+
return { seq, id: `msg-twin-${threadId}-${seq}` };
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
/** A message appended to a thread, named by the thread's next sequence number (or the id a run
|
|
70
|
+
* reserved for its reply). */
|
|
71
|
+
async function append(ctx: SemanticsContext, threadId: string, p: Row, reserved?: string): Promise<Row> {
|
|
72
|
+
const next = nextMessage(ctx, threadId);
|
|
73
|
+
const id = reserved ?? next.id;
|
|
74
|
+
const seq = reserved ? Number(reserved.split('-').at(-1)) : next.seq;
|
|
75
|
+
const text = typeof p.content === 'string'
|
|
76
|
+
? p.content
|
|
77
|
+
: Array.isArray(p.content) ? (p.content as Row[]).map((part) => String((part as { text?: { value?: unknown } }).text?.value ?? part.text ?? '')).join('') : '';
|
|
78
|
+
const fields = {
|
|
79
|
+
object: 'thread.message', created_at: epoch(ctx), thread_id: threadId, role: typeof p.role === 'string' ? p.role : 'user',
|
|
80
|
+
// a message added to a thread is whole when it is added
|
|
81
|
+
status: 'completed', completed_at: epoch(ctx), incomplete_at: null, incomplete_details: null,
|
|
82
|
+
content: [{ type: 'text', text: { value: text, annotations: [] } }],
|
|
83
|
+
assistant_id: p.assistant_id ?? null, run_id: p.run_id ?? null,
|
|
84
|
+
attachments: Array.isArray(p.attachments) ? p.attachments : [],
|
|
85
|
+
metadata: objectOr(p.metadata, {}),
|
|
86
|
+
_seq: seq,
|
|
87
|
+
};
|
|
88
|
+
return ctx.write(MESSAGE, id, fields, 'message.create');
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** The thread the path names (a deleted one still holds its messages and runs), or OpenAI's answer. */
|
|
92
|
+
function thread(ctx: SemanticsContext): { id: string } | { answer: Response } {
|
|
93
|
+
const id = at(ctx, 'thread_id');
|
|
94
|
+
return ctx.row(THREAD, id, { withDeleted: true }) ? { id } : { answer: ctx.notFound(THREAD, id) };
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
const createThread: Semantics = async (ctx) => {
|
|
98
|
+
const p = ctx.params;
|
|
99
|
+
const id = ctx.mint(THREAD);
|
|
100
|
+
// a thread made with no tool resources has none: `{}` (the reference's createThread examples)
|
|
101
|
+
const body = await ctx.write(THREAD, id, { object: 'thread', created_at: epoch(ctx), metadata: objectOr(p.metadata, {}), tool_resources: objectOr(p.tool_resources, {}) }, 'thread.create');
|
|
102
|
+
// messages given at creation are the thread's first
|
|
103
|
+
if (Array.isArray(p.messages)) for (const m of p.messages as Row[]) await append(ctx, id, m);
|
|
104
|
+
return ctx.reply(body);
|
|
105
|
+
};
|
|
106
|
+
|
|
107
|
+
const modifyThread: Semantics = async (ctx) => {
|
|
108
|
+
const id = String(ctx.id);
|
|
109
|
+
if (!ctx.get(THREAD, id)) return ctx.notFound(THREAD, id);
|
|
110
|
+
return ctx.reply(await ctx.write(THREAD, id, pick(ctx.params, ['metadata', 'tool_resources']), 'thread.update'));
|
|
111
|
+
};
|
|
112
|
+
|
|
113
|
+
const createMessage: Semantics = async (ctx) => {
|
|
114
|
+
const found = thread(ctx);
|
|
115
|
+
if ('answer' in found) return found.answer;
|
|
116
|
+
if (ctx.params.content === undefined) return invalid(ctx, 'you must provide a content parameter', 'content');
|
|
117
|
+
return ctx.reply(await append(ctx, found.id, ctx.params));
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
// ── runs and steps ──────────────────────────────────────────────────────────────────────────
|
|
121
|
+
|
|
122
|
+
/** The run the path names under its thread, raw (its reply message is bookkeeping), or OpenAI's answer. */
|
|
123
|
+
function run(ctx: SemanticsContext): { run: Row } | { answer: Response } {
|
|
124
|
+
const id = at(ctx, 'run_id');
|
|
125
|
+
const found = ctx.row(RUN, id);
|
|
126
|
+
return found && found.thread_id === at(ctx, 'thread_id') ? { run: found } : { answer: ctx.notFound(RUN, id) };
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
const replyText = (model: string): string => `[twin-stub:${model}] deterministic assistant run output (no model weights are run)`;
|
|
130
|
+
|
|
131
|
+
/** What a completed run carries: its reply message and usage; a run that waited on its tools keeps when it started. */
|
|
132
|
+
function finishRun(ctx: SemanticsContext): (row: Row, end: string) => Row {
|
|
133
|
+
return (row, end) => {
|
|
134
|
+
if (end !== 'completed') return {};
|
|
135
|
+
const text = replyText(String(row.model));
|
|
136
|
+
return { _reply_msg: nextMessage(ctx, String(row.thread_id)).id, expires_at: null, ...(row.started_at ? { started_at: row.started_at } : {}), usage: { prompt_tokens: 0, completion_tokens: estimateTokens(text), total_tokens: estimateTokens(text) } };
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** The function tools a run's model calls: its placeholder decision is to call them (each one, or the one
|
|
141
|
+
* `tool_choice` names, or the first when calls are not parallel), unless `tool_choice` is `none` or their outputs
|
|
142
|
+
* were already submitted. */
|
|
143
|
+
function toolCalls(r: Row, userText = ''): Row[] {
|
|
144
|
+
const functions = (Array.isArray(r.tools) ? r.tools : []).filter((t) => (t as { type?: unknown }).type === 'function') as Row[];
|
|
145
|
+
if (!functions.length || r.tool_choice === 'none' || r._tools_answered) return [];
|
|
146
|
+
const named = r.tool_choice && typeof r.tool_choice === 'object' ? String(((r.tool_choice as Row).function as Row | undefined)?.name ?? '') : '';
|
|
147
|
+
const chosen = named ? functions.filter((t) => (t.function as Row | undefined)?.name === named) : r.parallel_tool_calls === false ? functions.slice(0, 1) : functions;
|
|
148
|
+
const run = String(r.id).replace(/[^A-Za-z0-9]+/g, '_');
|
|
149
|
+
return chosen.map((tool, i) => {
|
|
150
|
+
const call = stubToolCall([tool], i + 1, undefined, userText)!;
|
|
151
|
+
return { id: `call_${run}_${i + 1}`, type: 'function', function: call.function };
|
|
152
|
+
});
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** A run whose model calls its function tools stops at `requires_action`, naming the calls it waits on
|
|
156
|
+
* (https://platform.openai.com/docs/assistants/tools/function-calling: "the Run will enter a requires_action
|
|
157
|
+
* status"); the vendor's moves, decided and written at once. */
|
|
158
|
+
async function waitOnTools(ctx: SemanticsContext, id: string): Promise<boolean> {
|
|
159
|
+
return ctx.atomically<boolean>((rows) => {
|
|
160
|
+
const current = rows(RUN).find((r) => r.id === id);
|
|
161
|
+
if (!current || current.status !== 'queued') return { value: false };
|
|
162
|
+
// the thread's last user message is what the model answers, and what its calls take their values from
|
|
163
|
+
const asked = rows(MESSAGE).filter((m) => m.thread_id === current.thread_id && m.role === 'user').sort((a, b) => Number(a._seq ?? 0) - Number(b._seq ?? 0)).at(-1);
|
|
164
|
+
const calls = toolCalls(current, String(((asked?.content as Row[] | undefined)?.[0]?.text as { value?: unknown } | undefined)?.value ?? ''));
|
|
165
|
+
if (!calls.length) return { value: false };
|
|
166
|
+
if (ctx.legal(RUN, 'status', ctx.call.operation.id, 'queued', 'in_progress', id, 'vendor') || ctx.legal(RUN, 'status', ctx.call.operation.id, 'in_progress', 'requires_action', id, 'vendor')) return { value: false };
|
|
167
|
+
const fields = { status: 'requires_action', started_at: epoch(ctx), required_action: { type: 'submit_tool_outputs', submit_tool_outputs: { tool_calls: calls } }, _tool_calls: calls };
|
|
168
|
+
return { value: true, write: { resource: RUN, id, fields, operation: 'run.update' } };
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/** Observe the vendor's work on a run: it stops for its function tools, else the read that completes it appends
|
|
173
|
+
* the reply to its thread. */
|
|
174
|
+
async function work(ctx: SemanticsContext, id: string): Promise<void> {
|
|
175
|
+
if (await waitOnTools(ctx, id)) return;
|
|
176
|
+
const moved = await progress(ctx, RUN, id, finishRun(ctx));
|
|
177
|
+
if (moved?.to !== 'completed') return;
|
|
178
|
+
const r = moved.row;
|
|
179
|
+
await append(ctx, String(r.thread_id), { role: 'assistant', content: replyText(String(r.model)), assistant_id: r.assistant_id, run_id: r.id }, String(r._reply_msg));
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
/** A run's steps, newest first: the call of its function tools (done once their outputs are in, each call with its
|
|
183
|
+
* output), and, once it completed, the creation of its reply message. */
|
|
184
|
+
function steps(r: Row): Row[] {
|
|
185
|
+
const base = { object: 'thread.run.step', run_id: String(r.id), assistant_id: String(r.assistant_id), thread_id: String(r.thread_id), cancelled_at: null, expired_at: null, failed_at: null, last_error: null, metadata: {} };
|
|
186
|
+
const out: Row[] = [];
|
|
187
|
+
const calls = (r._tool_calls ?? []) as Row[];
|
|
188
|
+
if (calls.length) {
|
|
189
|
+
const outputs = (r._tool_outputs ?? {}) as Record<string, string>;
|
|
190
|
+
const done = r._tools_answered === true;
|
|
191
|
+
const cancelled = !done && (r.status === 'cancelled' || r.status === 'cancelling');
|
|
192
|
+
out.push({
|
|
193
|
+
...base, id: `step-${r.id}-1`, created_at: Number(r.started_at ?? r.created_at ?? 0), type: 'tool_calls',
|
|
194
|
+
status: done ? 'completed' : cancelled ? 'cancelled' : 'in_progress', completed_at: done ? Number(r._tools_answered_at ?? r.started_at ?? 0) : null,
|
|
195
|
+
...(cancelled ? { cancelled_at: Number(r.cancelled_at ?? r.started_at ?? 0) } : {}),
|
|
196
|
+
step_details: { type: 'tool_calls', tool_calls: calls.map((c) => ({ ...c, function: { ...(c.function as Row), output: done ? (outputs[String(c.id)] ?? null) : null } })) },
|
|
197
|
+
usage: null,
|
|
198
|
+
});
|
|
199
|
+
}
|
|
200
|
+
if (r.status === 'completed') {
|
|
201
|
+
const created = Number(r.completed_at ?? r.created_at ?? 0);
|
|
202
|
+
out.unshift({
|
|
203
|
+
...base, id: `step-${r.id}-${out.length + 1}`, created_at: created,
|
|
204
|
+
type: 'message_creation', status: 'completed', completed_at: created,
|
|
205
|
+
step_details: { type: 'message_creation', message_creation: { message_id: String(r._reply_msg ?? '') } },
|
|
206
|
+
usage: r.usage ?? null,
|
|
207
|
+
});
|
|
208
|
+
}
|
|
209
|
+
return out;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
const createRun: Semantics = async (ctx) => {
|
|
213
|
+
const found = thread(ctx);
|
|
214
|
+
if ('answer' in found) return found.answer;
|
|
215
|
+
const p = ctx.params;
|
|
216
|
+
if (p.assistant_id === undefined || p.assistant_id === '') return invalid(ctx, 'you must provide an assistant_id parameter', 'assistant_id');
|
|
217
|
+
const assistantId = String(p.assistant_id);
|
|
218
|
+
const assistant = ctx.get(ASSISTANT, assistantId);
|
|
219
|
+
if (!assistant) return ctx.notFound(ASSISTANT, assistantId);
|
|
220
|
+
const id = ctx.mint(RUN);
|
|
221
|
+
const created = epoch(ctx);
|
|
222
|
+
const model = typeof p.model === 'string' && p.model ? p.model : String(assistant.model);
|
|
223
|
+
const fields = {
|
|
224
|
+
object: 'thread.run', created_at: created, thread_id: found.id, assistant_id: assistantId,
|
|
225
|
+
status: 'queued', model, instructions: p.instructions ?? assistant.instructions ?? null,
|
|
226
|
+
tools: Array.isArray(p.tools) ? p.tools : (assistant.tools ?? []),
|
|
227
|
+
// the tool resources it runs with, its assistant's (the reference's cancelRun example; spec/patches.json)
|
|
228
|
+
tool_resources: assistant.tool_resources ?? {},
|
|
229
|
+
// a run in flight expires ten minutes after it was created; a completed one no longer does (the reference's
|
|
230
|
+
// streamed run: `expires_at` 600 seconds past `created_at` until `thread.run.completed` answers it null)
|
|
231
|
+
started_at: null, completed_at: null, expires_at: created + 600, cancelled_at: null, failed_at: null,
|
|
232
|
+
required_action: null, last_error: null, usage: null, incomplete_details: null,
|
|
233
|
+
// the request's sampling, else the assistant's (https://platform.openai.com/docs/api-reference/runs/object)
|
|
234
|
+
temperature: p.temperature ?? assistant.temperature ?? 1, top_p: p.top_p ?? assistant.top_p ?? 1,
|
|
235
|
+
// the request's settings as sent, or OpenAI's defaults for a run
|
|
236
|
+
max_prompt_tokens: p.max_prompt_tokens ?? null, max_completion_tokens: p.max_completion_tokens ?? null,
|
|
237
|
+
parallel_tool_calls: p.parallel_tool_calls ?? true, tool_choice: p.tool_choice ?? 'auto',
|
|
238
|
+
response_format: p.response_format ?? assistant.response_format ?? 'auto', truncation_strategy: p.truncation_strategy ?? { type: 'auto', last_messages: null },
|
|
239
|
+
metadata: objectOr(p.metadata, {}),
|
|
240
|
+
};
|
|
241
|
+
const created_ = await ctx.write(RUN, id, fields, 'run.create');
|
|
242
|
+
return p.stream === true ? streamRun(ctx, created_) : ctx.reply(created_);
|
|
243
|
+
};
|
|
244
|
+
|
|
245
|
+
/** A run created with `stream: true` answers its events as it runs (https://platform.openai.com/docs/api-reference/assistants-streaming/events):
|
|
246
|
+
* created, queued and in progress, then either its tool-call step and `thread.run.requires_action` (a run that waits on
|
|
247
|
+
* its function tools), or its message-creation step, the reply message with its text as deltas, and the run completed;
|
|
248
|
+
* then `done`. Submitted tool outputs stream the same way from the finished tool-call step. The twin works the run in
|
|
249
|
+
* the stream as a read would; each event is a server-sent `event:` with its object as `data`. */
|
|
250
|
+
async function streamRun(ctx: SemanticsContext, queued: Row, submitted = false): Promise<Response> {
|
|
251
|
+
const id = String(queued.id);
|
|
252
|
+
await work(ctx, id);
|
|
253
|
+
const done = ctx.get(RUN, id)!;
|
|
254
|
+
const all = steps(ctx.row(RUN, id)!);
|
|
255
|
+
const toolStep = all.find((x) => x.type === 'tool_calls');
|
|
256
|
+
const step = all.find((x) => x.type === 'message_creation');
|
|
257
|
+
const message = step ? ctx.get(MESSAGE, String((step.step_details as { message_creation: { message_id: string } }).message_creation.message_id)) : undefined;
|
|
258
|
+
const started = { ...done, status: 'in_progress', completed_at: null, required_action: null, usage: null };
|
|
259
|
+
const events: Array<[string, unknown]> = submitted
|
|
260
|
+
? [['thread.run.step.completed', toolStep], ['thread.run.queued', queued], ['thread.run.in_progress', started]]
|
|
261
|
+
: [['thread.run.created', queued], ['thread.run.queued', queued], ['thread.run.in_progress', started]];
|
|
262
|
+
if (done.status === 'requires_action' && toolStep) {
|
|
263
|
+
events.push(['thread.run.step.created', toolStep], ['thread.run.step.in_progress', toolStep], ['thread.run.requires_action', done]);
|
|
264
|
+
} else {
|
|
265
|
+
if (step && message) {
|
|
266
|
+
const text = String(((message.content as Row[])[0]?.text as { value?: unknown } | undefined)?.value ?? '');
|
|
267
|
+
const stepRunning = { ...step, status: 'in_progress', completed_at: null, usage: null };
|
|
268
|
+
const messageRunning = { ...message, status: 'in_progress', completed_at: null, content: [] };
|
|
269
|
+
events.push(['thread.run.step.created', stepRunning], ['thread.run.step.in_progress', stepRunning], ['thread.message.created', messageRunning], ['thread.message.in_progress', messageRunning]);
|
|
270
|
+
for (let i = 0; i < text.length; i += 20) events.push(['thread.message.delta', { id: message.id, object: 'thread.message.delta', delta: { content: [{ index: 0, type: 'text', text: { value: text.slice(i, i + 20), annotations: [] } }] } }]);
|
|
271
|
+
events.push(['thread.message.completed', message], ['thread.run.step.completed', step]);
|
|
272
|
+
}
|
|
273
|
+
events.push(['thread.run.completed', done]);
|
|
274
|
+
}
|
|
275
|
+
const body = events.map(([event, data]) => `event: ${event}\ndata: ${JSON.stringify(data)}\n\n`).join('') + 'event: done\ndata: [DONE]\n\n';
|
|
276
|
+
return ctx.raw(body, { headers: { 'content-type': 'text/event-stream; charset=utf-8', 'cache-control': 'no-cache' } });
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/** Submitting a waiting run's tool outputs queues it again to finish (https://platform.openai.com/docs/api-reference/runs/submitToolOutputs):
|
|
280
|
+
* only a run at `requires_action` takes them, and they must answer every call it waits on, and only those. */
|
|
281
|
+
const submitToolOutputs: Semantics = async (ctx) => {
|
|
282
|
+
const found = run(ctx);
|
|
283
|
+
if ('answer' in found) return found.answer;
|
|
284
|
+
const r = found.run;
|
|
285
|
+
const refusal = ctx.legal(RUN, 'status', 'submitToolOuputsToRun', r.status, 'queued');
|
|
286
|
+
if (refusal) return ctx.refuse(refusal);
|
|
287
|
+
const outputs = Array.isArray(ctx.params.tool_outputs) ? (ctx.params.tool_outputs as Row[]) : undefined;
|
|
288
|
+
if (!outputs) return invalid(ctx, 'you must provide a tool_outputs array', 'tool_outputs');
|
|
289
|
+
const waiting = ((r._tool_calls ?? []) as Row[]).map((c) => String(c.id));
|
|
290
|
+
const given = outputs.map((o) => String(o.tool_call_id ?? ''));
|
|
291
|
+
const unknown = given.find((g) => !waiting.includes(g));
|
|
292
|
+
if (unknown !== undefined) return invalid(ctx, `Invalid tool_call_id: ${unknown}. No tool call with that id is waiting on this run.`, 'tool_outputs');
|
|
293
|
+
const missingIds = waiting.filter((w) => !given.includes(w));
|
|
294
|
+
if (missingIds.length) return invalid(ctx, `Expected tool outputs for call_ids ${JSON.stringify(waiting)}, got ${JSON.stringify(given)}`, 'tool_outputs');
|
|
295
|
+
const fields = { status: 'queued', required_action: null, _tools_answered: true, _tools_answered_at: epoch(ctx), _tool_outputs: Object.fromEntries(outputs.map((o) => [String(o.tool_call_id), String(o.output ?? '')])) };
|
|
296
|
+
const queued = await ctx.write(RUN, String(r.id), fields, 'run.update');
|
|
297
|
+
return ctx.params.stream === true ? streamRun(ctx, queued, true) : ctx.reply(queued);
|
|
298
|
+
};
|
|
299
|
+
|
|
300
|
+
const listRunSteps: Semantics = async (ctx) => {
|
|
301
|
+
await work(ctx, at(ctx, 'run_id'));
|
|
302
|
+
const found = run(ctx);
|
|
303
|
+
if ('answer' in found) return found.answer;
|
|
304
|
+
const data = steps(found.run);
|
|
305
|
+
return ctx.reply({ object: 'list', data, has_more: false, first_id: data[0]?.id ?? null, last_id: data.at(-1)?.id ?? null });
|
|
306
|
+
};
|
|
307
|
+
|
|
308
|
+
const getRunStep: Semantics = async (ctx) => {
|
|
309
|
+
await work(ctx, at(ctx, 'run_id'));
|
|
310
|
+
const found = run(ctx);
|
|
311
|
+
if ('answer' in found) return found.answer;
|
|
312
|
+
const id = at(ctx, 'step_id');
|
|
313
|
+
const step = steps(found.run).find((s) => s.id === id);
|
|
314
|
+
return step ? ctx.reply(step) : missing(ctx, `No run step found with id '${id}'.`);
|
|
315
|
+
};
|
|
316
|
+
|
|
317
|
+
export const assistants: Record<string, Semantics> = {
|
|
318
|
+
createAssistant,
|
|
319
|
+
modifyAssistant,
|
|
320
|
+
createThread,
|
|
321
|
+
modifyThread,
|
|
322
|
+
createMessage,
|
|
323
|
+
createRun,
|
|
324
|
+
submitToolOuputsToRun: submitToolOutputs,
|
|
325
|
+
listRunSteps,
|
|
326
|
+
getRunStep,
|
|
327
|
+
getRun: async (ctx) => {
|
|
328
|
+
await work(ctx, at(ctx, 'run_id'));
|
|
329
|
+
return coreAnswer(ctx);
|
|
330
|
+
},
|
|
331
|
+
listRuns: async (ctx) => {
|
|
332
|
+
const threadId = at(ctx, 'thread_id');
|
|
333
|
+
for (const r of ctx.rowsRaw(RUN).filter((x) => x.thread_id === threadId)) await work(ctx, String(r.id));
|
|
334
|
+
return coreAnswer(ctx);
|
|
335
|
+
},
|
|
336
|
+
};
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
// Audio: labeled deterministic transcripts, and speech as a real audio file of silence; the twin runs no
|
|
2
|
+
// speech model. Transcription and translation take the audio as a multipart file (the SDK's shape),
|
|
3
|
+
// refuse anything that is not an audio file OpenAI transcribes, and name the file in the transcript.
|
|
4
|
+
import type { Semantics } from '@volter/world-core';
|
|
5
|
+
import { handleSpeech, handleTranscription, type OpenAIResponseEnvelope } from '../openai-twin.ts';
|
|
6
|
+
import { recordUsage, send } from './shared.ts';
|
|
7
|
+
|
|
8
|
+
/** A transcription or translation is billed by the seconds of audio heard (the usage report's `seconds`). */
|
|
9
|
+
async function recordSeconds(ctx: Parameters<Semantics>[0], r: ReturnType<typeof handleTranscription>): Promise<void> {
|
|
10
|
+
if (r.status === 200 && 'seconds' in r && typeof r.seconds === 'number') await recordUsage(ctx, 'audio_transcriptions', String(ctx.params.model), 0, 0, { seconds: Math.ceil(r.seconds) });
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export const audio: Record<string, Semantics> = {
|
|
14
|
+
createTranscription: async (ctx) => {
|
|
15
|
+
const r = handleTranscription(ctx.params, false);
|
|
16
|
+
await recordSeconds(ctx, r);
|
|
17
|
+
return 'events' in r && r.events ? ctx.sse(r.events) : send(ctx, r as OpenAIResponseEnvelope);
|
|
18
|
+
},
|
|
19
|
+
createTranslation: async (ctx) => {
|
|
20
|
+
const r = handleTranscription(ctx.params, true);
|
|
21
|
+
await recordSeconds(ctx, r);
|
|
22
|
+
return send(ctx, r as OpenAIResponseEnvelope);
|
|
23
|
+
},
|
|
24
|
+
// the audio file itself, not JSON
|
|
25
|
+
createSpeech: async (ctx) => {
|
|
26
|
+
const r = handleSpeech(ctx.params);
|
|
27
|
+
// speech is billed by the characters it speaks (the usage report's `characters`)
|
|
28
|
+
if ('audio' in r) await recordUsage(ctx, 'audio_speeches', String(ctx.params.model), 0, 0, { characters: String(ctx.params.input ?? '').length });
|
|
29
|
+
return 'audio' in r ? ctx.raw(r.audio as Uint8Array<ArrayBuffer>, { headers: { 'content-type': r.type } }) : send(ctx, r);
|
|
30
|
+
},
|
|
31
|
+
};
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
// Batch semantics. A batch is created `validating`, as OpenAI creates it, and cancel is the derived
|
|
2
|
+
// core's (the machine in ../manifest.ts moves `status`: an in-flight batch goes to `cancelling`).
|
|
3
|
+
// OpenAI works a batch on its own; the twin works it when a read first looks (./progress.ts): it
|
|
4
|
+
// ends `completed` with its output file written (one result line per request line of its input file, answered
|
|
5
|
+
// by the pack's stub), or `cancelled` if it was being cancelled.
|
|
6
|
+
import type { Semantics, SemanticsContext } from '@volter/world-core';
|
|
7
|
+
import { coreAnswer, inFlight, progress } from './progress.ts';
|
|
8
|
+
import { epoch, invalid, objectOr, type Row } from './shared.ts';
|
|
9
|
+
|
|
10
|
+
/** The request lines of a batch's input file. */
|
|
11
|
+
const requestLines = (ctx: SemanticsContext, row: Row): string[] =>
|
|
12
|
+
String(ctx.row('OpenAIFile', String(row.input_file_id), { withDeleted: true })?._content ?? '').split('\n').filter((l) => l.trim());
|
|
13
|
+
|
|
14
|
+
/** What a finished batch carries: its output file and its request counts, every request answered
|
|
15
|
+
* (https://platform.openai.com/docs/api-reference/batch/object, `request_counts`). */
|
|
16
|
+
const finish = (ctx: SemanticsContext) => (row: Row, end: string): Row => {
|
|
17
|
+
if (end !== 'completed') return {};
|
|
18
|
+
const total = requestLines(ctx, row).length;
|
|
19
|
+
return { output_file_id: `file-twin-batchout-${String(row.id)}`, request_counts: { total, completed: total, failed: 0 } };
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
const create: Semantics = async (ctx) => {
|
|
23
|
+
const p = ctx.params;
|
|
24
|
+
if (p.input_file_id === undefined || p.input_file_id === '') return invalid(ctx, 'you must provide an input_file_id parameter', 'input_file_id');
|
|
25
|
+
if (p.endpoint === undefined || p.endpoint === '') return invalid(ctx, 'you must provide an endpoint parameter', 'endpoint');
|
|
26
|
+
if (p.completion_window === undefined) return invalid(ctx, 'you must provide a completion_window parameter', 'completion_window');
|
|
27
|
+
const id = ctx.mint('Batch');
|
|
28
|
+
const created = epoch(ctx);
|
|
29
|
+
const fields = {
|
|
30
|
+
object: 'batch',
|
|
31
|
+
endpoint: String(p.endpoint),
|
|
32
|
+
input_file_id: String(p.input_file_id),
|
|
33
|
+
completion_window: String(p.completion_window),
|
|
34
|
+
status: 'validating',
|
|
35
|
+
output_file_id: null,
|
|
36
|
+
error_file_id: null,
|
|
37
|
+
created_at: created,
|
|
38
|
+
// the batch object's timestamps, each null until the batch gets there; it expires at the end of its completion
|
|
39
|
+
// window (https://platform.openai.com/docs/api-reference/batch/object)
|
|
40
|
+
in_progress_at: null,
|
|
41
|
+
expires_at: created + 24 * 3600,
|
|
42
|
+
finalizing_at: null,
|
|
43
|
+
completed_at: null,
|
|
44
|
+
failed_at: null,
|
|
45
|
+
expired_at: null,
|
|
46
|
+
cancelling_at: null,
|
|
47
|
+
cancelled_at: null,
|
|
48
|
+
errors: null,
|
|
49
|
+
request_counts: { total: 0, completed: 0, failed: 0 },
|
|
50
|
+
metadata: objectOr(p.metadata, null),
|
|
51
|
+
};
|
|
52
|
+
return ctx.reply(await ctx.write('Batch', id, fields, 'batch.create'));
|
|
53
|
+
};
|
|
54
|
+
|
|
55
|
+
/** Work a batch on; the read that completes it writes its output file, a result for each request line of its input. */
|
|
56
|
+
async function work(ctx: SemanticsContext, id: string): Promise<void> {
|
|
57
|
+
const moved = await progress(ctx, 'Batch', id, finish(ctx));
|
|
58
|
+
if (moved?.to !== 'completed') return;
|
|
59
|
+
const lines = requestLines(ctx, moved.row);
|
|
60
|
+
const results = lines.map((line, i) => {
|
|
61
|
+
let req: Row = {};
|
|
62
|
+
try { req = JSON.parse(line) as Row; } catch { /* a line that is not JSON answers an error for it */ }
|
|
63
|
+
return JSON.stringify({ id: `batch_req_twin_${i + 1}`, custom_id: req.custom_id ?? null, response: { status_code: 200, request_id: `req_twin_${i + 1}`, body: { object: 'chat.completion', model: (req.body as Row | undefined)?.model ?? null, choices: [{ index: 0, message: { role: 'assistant', content: '[twin-stub] deterministic batch result' }, finish_reason: 'stop' }] } }, error: null });
|
|
64
|
+
});
|
|
65
|
+
const content = results.length ? `${results.join('\n')}\n` : '';
|
|
66
|
+
// the output is deleted thirty days after the batch completes (developers.openai.com/api/docs/guides/batch)
|
|
67
|
+
await ctx.write('OpenAIFile', String(moved.row.output_file_id), { object: 'file', bytes: new TextEncoder().encode(content).length, created_at: epoch(ctx), expires_at: Number(moved.row.completed_at ?? epoch(ctx)) + 30 * 24 * 3600, filename: 'batch_output.jsonl', purpose: 'batch_output', status: 'processed', _content: content }, 'file.create');
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// the organization's batches, each one's work observed first
|
|
71
|
+
const listBatches: Semantics = async (ctx) => {
|
|
72
|
+
for (const row of ctx.rowsRaw('Batch')) if (inFlight('Batch', row)) await work(ctx, String(row.id));
|
|
73
|
+
return coreAnswer(ctx);
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
export const batches: Record<string, Semantics> = {
|
|
77
|
+
createBatch: create,
|
|
78
|
+
retrieveBatch: async (ctx) => {
|
|
79
|
+
await work(ctx, String(ctx.id));
|
|
80
|
+
return coreAnswer(ctx);
|
|
81
|
+
},
|
|
82
|
+
listBatches,
|
|
83
|
+
};
|
|
84
|
+
|
|
85
|
+
/** Every batch OpenAI is still working on, observed (a read observes all the vendor's work before it answers). */
|
|
86
|
+
export async function observeBatches(ctx: SemanticsContext): Promise<void> {
|
|
87
|
+
for (const row of ctx.rowsRaw('Batch')) if (inFlight('Batch', row)) await work(ctx, String(row.id));
|
|
88
|
+
}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// Chat completion semantics. A completion is the pack's behaviour, never a model: a scripted
|
|
2
|
+
// scenario's turn when one is loaded and a handler matches, else the labeled deterministic stub.
|
|
3
|
+
// Every completion is recorded (the call is the change); one made with `store: true` is also
|
|
4
|
+
// readable, with its request messages, until deleted. List, retrieve and delete of stored
|
|
5
|
+
// completions are the derived core's (the manifest makes only `_stored` ones readable).
|
|
6
|
+
import type { Semantics } from '@volter/world-core';
|
|
7
|
+
import { type OpenAIScenarioEngine, serveScenario } from '../openai-scenario.ts';
|
|
8
|
+
import { buildChatCompletion, streamChat, validateChat } from '../openai-twin.ts';
|
|
9
|
+
import type { ChatCompletion } from '../openai-types.ts';
|
|
10
|
+
import { page, recordUsage, send, type Row } from './shared.ts';
|
|
11
|
+
|
|
12
|
+
const COMPLETION = 'CreateChatCompletionResponse';
|
|
13
|
+
|
|
14
|
+
const create = (engine: OpenAIScenarioEngine | undefined): Semantics => async (ctx) => {
|
|
15
|
+
const validated = validateChat(ctx.params);
|
|
16
|
+
if ('error' in validated) return send(ctx, validated.error);
|
|
17
|
+
const args = validated.args;
|
|
18
|
+
// the scenario decides the turn and honors a fault before any completion exists: a status fault
|
|
19
|
+
// is this vendor's own refusal envelope, decided before a stream would start
|
|
20
|
+
const decision = engine ? await serveScenario(engine, { model: args.model, messages: args.messages, tools: args.tools, maxTokens: args.maxTokens }, new URL(ctx.call.request.url).pathname) : undefined;
|
|
21
|
+
if (decision?.kind === 'fault') return send(ctx, decision.result);
|
|
22
|
+
const events: Array<{ data: unknown }> = [];
|
|
23
|
+
const result = args.stream ? streamChat(args, (e) => { if (e.data) events.push({ data: e.data }); }, ctx.occurredAt, decision) : buildChatCompletion(args, ctx.occurredAt, decision);
|
|
24
|
+
// the request's messages are kept for the stored completion's /messages
|
|
25
|
+
// as the messages list shows them (https://platform.openai.com/docs/api-reference/chat/getMessages): the text, the
|
|
26
|
+
// author's name or null, and the parts when the content was sent as parts, else null
|
|
27
|
+
const inputMessages = args.messages.map((m, i) => {
|
|
28
|
+
const parts = Array.isArray(m.content) ? (m.content as Array<{ type?: unknown; text?: unknown }>) : null;
|
|
29
|
+
const content = parts ? parts.filter((c) => c.type === 'text').map((c) => String(c.text ?? '')).join('') : (m.content ?? null);
|
|
30
|
+
return { id: `${result.id}-msg-${i}`, role: m.role, content, name: (m as { name?: unknown }).name ?? null, content_parts: parts };
|
|
31
|
+
});
|
|
32
|
+
// a stored completion keeps its metadata, `{}` when none was sent (https://platform.openai.com/docs/api-reference/chat/object)
|
|
33
|
+
// and reads back each reply message with its tool calls and legacy function call, null when it made none (the
|
|
34
|
+
// reference's getChatCompletion and listChatCompletions examples)
|
|
35
|
+
// A stored completion also keeps the request that made it: its id, and the settings it was sampled with, as sent or
|
|
36
|
+
// their defaults (the reference's getChatCompletion example: request_id, seed, temperature, top_p, presence_penalty,
|
|
37
|
+
// frequency_penalty, input_user, tools, tool_choice, response_format; spec/patches.json adds them to the object)
|
|
38
|
+
const p = ctx.params;
|
|
39
|
+
const settings = {
|
|
40
|
+
request_id: `req_twin_${result.id.replace(/^chatcmpl-twin-/, '')}`,
|
|
41
|
+
seed: typeof p.seed === 'number' ? p.seed : parseInt(result.id.replace(/^chatcmpl-twin-/, ''), 36),
|
|
42
|
+
temperature: p.temperature ?? 1, top_p: p.top_p ?? 1, presence_penalty: p.presence_penalty ?? 0, frequency_penalty: p.frequency_penalty ?? 0,
|
|
43
|
+
input_user: p.user ?? null, tools: p.tools ?? null, tool_choice: p.tool_choice ?? null, response_format: p.response_format ?? null,
|
|
44
|
+
};
|
|
45
|
+
const stored = args.store === true ? { metadata: result.metadata ?? {}, ...settings, choices: result.choices.map((c) => ({ ...c, message: { ...c.message, tool_calls: c.message.tool_calls ?? null, function_call: null } })) } : {};
|
|
46
|
+
const { vendorData } = await ctx.writeDetailed(COMPLETION, result.id, { ...result, ...stored, _stored: args.store === true, _input_messages: inputMessages }, 'chat.completions.create');
|
|
47
|
+
// live use: the head performed the call, and the model's own words are the answer
|
|
48
|
+
const live = vendorData as Partial<ChatCompletion> | undefined;
|
|
49
|
+
const answer = live && typeof live === 'object' && Array.isArray(live.choices) ? (live as ChatCompletion) : result;
|
|
50
|
+
await recordUsage(ctx, 'completions', answer.model, answer.usage?.prompt_tokens ?? result.usage.prompt_tokens, answer.usage?.completion_tokens ?? result.usage.completion_tokens);
|
|
51
|
+
return args.stream ? ctx.sse(events) : ctx.reply(answer);
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
// the request's messages, kept when the completion was stored
|
|
55
|
+
const messages: Semantics = async (ctx) => {
|
|
56
|
+
const id = String(ctx.id);
|
|
57
|
+
const row = ctx.row(COMPLETION, id);
|
|
58
|
+
return row ? page(ctx, (row._input_messages as Row[] | undefined) ?? []) : ctx.notFound(COMPLETION, id);
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
export function chatCompletions(engine: OpenAIScenarioEngine | undefined): Record<string, Semantics> {
|
|
62
|
+
return {
|
|
63
|
+
createChatCompletion: create(engine),
|
|
64
|
+
getChatCompletionMessages: messages,
|
|
65
|
+
};
|
|
66
|
+
}
|