@volter/twin-openai 0.1.2 → 2.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -30
- package/defaults/handlers.json +10 -0
- package/dist/defaults/handlers.json +10 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +29 -0
- package/dist/src/generated/surface.gen.json +1 -0
- package/dist/src/generated/ui.gen.json +1 -0
- package/dist/src/index.d.ts +19 -0
- package/dist/src/index.js +72 -0
- package/dist/src/manifest.d.ts +6 -0
- package/dist/src/manifest.js +323 -0
- package/dist/src/openai-budget.d.ts +53 -0
- package/dist/src/openai-budget.js +147 -0
- package/dist/src/openai-capabilities.d.ts +4 -0
- package/dist/src/openai-capabilities.js +1569 -0
- package/dist/src/openai-conformance.d.ts +13 -0
- package/dist/src/openai-conformance.js +116 -0
- package/dist/src/openai-connector.d.ts +86 -0
- package/dist/src/openai-connector.js +291 -0
- package/dist/src/openai-media.d.ts +43 -0
- package/dist/src/openai-media.js +257 -0
- package/dist/src/openai-models.d.ts +74 -0
- package/dist/src/openai-models.js +148 -0
- package/dist/src/openai-scenario.d.ts +51 -0
- package/dist/src/openai-scenario.js +166 -0
- package/dist/src/openai-server.d.ts +40 -0
- package/dist/src/openai-server.js +126 -0
- package/dist/src/openai-stub.d.ts +82 -0
- package/dist/src/openai-stub.js +256 -0
- package/dist/src/openai-twin.d.ts +182 -0
- package/dist/src/openai-twin.js +1117 -0
- package/dist/src/openai-types.d.ts +194 -0
- package/dist/src/openai-types.js +4 -0
- package/dist/src/openai-webhooks.d.ts +47 -0
- package/dist/src/openai-webhooks.js +99 -0
- package/dist/src/screens/api-keys.d.ts +16 -0
- package/dist/src/screens/api-keys.js +131 -0
- package/dist/src/screens/session.d.ts +22 -0
- package/dist/src/screens/session.js +115 -0
- package/dist/src/semantics/assistants.d.ts +2 -0
- package/dist/src/semantics/assistants.js +331 -0
- package/dist/src/semantics/audio.d.ts +2 -0
- package/dist/src/semantics/audio.js +27 -0
- package/dist/src/semantics/batches.d.ts +4 -0
- package/dist/src/semantics/batches.js +86 -0
- package/dist/src/semantics/chat-completions.d.ts +3 -0
- package/dist/src/semantics/chat-completions.js +58 -0
- package/dist/src/semantics/containers.d.ts +2 -0
- package/dist/src/semantics/containers.js +147 -0
- package/dist/src/semantics/embeddings.d.ts +2 -0
- package/dist/src/semantics/embeddings.js +13 -0
- package/dist/src/semantics/evals.d.ts +2 -0
- package/dist/src/semantics/evals.js +173 -0
- package/dist/src/semantics/files.d.ts +13 -0
- package/dist/src/semantics/files.js +59 -0
- package/dist/src/semantics/fine-tuning.d.ts +4 -0
- package/dist/src/semantics/fine-tuning.js +178 -0
- package/dist/src/semantics/images.d.ts +2 -0
- package/dist/src/semantics/images.js +18 -0
- package/dist/src/semantics/index.d.ts +8 -0
- package/dist/src/semantics/index.js +46 -0
- package/dist/src/semantics/models.d.ts +2 -0
- package/dist/src/semantics/models.js +34 -0
- package/dist/src/semantics/moderations.d.ts +2 -0
- package/dist/src/semantics/moderations.js +12 -0
- package/dist/src/semantics/organization.d.ts +2 -0
- package/dist/src/semantics/organization.js +67 -0
- package/dist/src/semantics/progress.d.ts +22 -0
- package/dist/src/semantics/progress.js +63 -0
- package/dist/src/semantics/responses.d.ts +3 -0
- package/dist/src/semantics/responses.js +153 -0
- package/dist/src/semantics/shared.d.ts +32 -0
- package/dist/src/semantics/shared.js +69 -0
- package/dist/src/semantics/uploads.d.ts +2 -0
- package/dist/src/semantics/uploads.js +84 -0
- package/dist/src/semantics/vector-stores.d.ts +2 -0
- package/dist/src/semantics/vector-stores.js +281 -0
- package/dist/test-fixtures/openai-openapi-operations.SOURCE.md +18 -0
- package/dist/test-fixtures/openai-openapi-operations.json +1849 -0
- package/package.json +21 -10
- package/src/cli.ts +9 -7
- package/src/generated/surface.gen.json +1 -0
- package/src/generated/ui.gen.json +1 -0
- package/src/index.ts +20 -10
- package/src/manifest.ts +343 -0
- package/src/openai-budget.ts +4 -4
- package/src/openai-capabilities.ts +177 -195
- package/src/openai-conformance.ts +1 -1
- package/src/openai-connector.ts +40 -43
- package/src/openai-media.ts +225 -0
- package/src/openai-models.ts +145 -15
- package/src/openai-scenario.ts +46 -10
- package/src/openai-server.ts +65 -108
- package/src/openai-stub.ts +54 -30
- package/src/openai-twin.ts +760 -1665
- package/src/openai-types.ts +24 -6
- package/src/openai-webhooks.ts +3 -2
- package/src/screens/api-keys.tsx +138 -0
- package/src/screens/session.tsx +131 -0
- package/src/semantics/assistants.ts +336 -0
- package/src/semantics/audio.ts +31 -0
- package/src/semantics/batches.ts +88 -0
- package/src/semantics/chat-completions.ts +66 -0
- package/src/semantics/containers.ts +151 -0
- package/src/semantics/embeddings.ts +19 -0
- package/src/semantics/evals.ts +182 -0
- package/src/semantics/files.ts +67 -0
- package/src/semantics/fine-tuning.ts +185 -0
- package/src/semantics/images.ts +23 -0
- package/src/semantics/index.ts +52 -0
- package/src/semantics/models.ts +41 -0
- package/src/semantics/moderations.ts +14 -0
- package/src/semantics/organization.ts +76 -0
- package/src/semantics/progress.ts +72 -0
- package/src/semantics/responses.ts +151 -0
- package/src/semantics/shared.ts +82 -0
- package/src/semantics/uploads.ts +92 -0
- package/src/semantics/vector-stores.ts +279 -0
- package/test-fixtures/openai-openapi-operations.SOURCE.md +4 -5
- package/test-fixtures/openai-openapi-operations.json +224 -1334
|
@@ -0,0 +1,1117 @@
|
|
|
1
|
+
// OpenAI twin: the pack's behaviour and its in-process door. `handleOpenAITwinRequest` answers one
|
|
2
|
+
// request through the pack's fetch (openai-server.ts), which the derived dispatch owns: an operation
|
|
3
|
+
// OpenAI's spec publishes is a semantics handler (semantics/), the derived core (manifest.ts), or
|
|
4
|
+
// OpenAI's unknown-URL 404. What lives here is what those share:
|
|
5
|
+
// • the generative stubs: chat completions and responses (validation, the deterministic stub or a
|
|
6
|
+
// scripted scenario's turn, the streaming event sequences), embeddings (pseudo-vectors),
|
|
7
|
+
// moderations, images and audio (labeled placeholders); never a model's output;
|
|
8
|
+
// • the Responses API's atomic id allocation and background-poll completion;
|
|
9
|
+
// • the usage and costs reports over the recorded usage.
|
|
10
|
+
// No real OpenAI is called from this path.
|
|
11
|
+
import { applyTwinWriteAtomic } from '@volter/world-core';
|
|
12
|
+
import { buildLogprobs, countPromptTokens, estimateTokens, lastUserText, moderateText, pseudoEmbedding, stubAssistantText, stubJsonObject, stubToolCall, } from "./openai-stub.js";
|
|
13
|
+
import { DEFAULT_EFFORT, modelOf, reasons, unsupported } from "./openai-models.js";
|
|
14
|
+
import { scenarioTurn, scriptedChoice } from "./openai-scenario.js";
|
|
15
|
+
import { audioFormat, base64, imageFormat, partBytes, placeholderPng, pngSize, requestedSize, silentAac, silentFlac, silentMp3, silentOpus, silentPcm, silentWav, wavSeconds } from "./openai-media.js";
|
|
16
|
+
import { createOpenAITwinFetch } from "./openai-server.js";
|
|
17
|
+
const SERVICE = 'openai';
|
|
18
|
+
// ── vendor-shaped errors ──────────────────────────────────────────────────────────────
|
|
19
|
+
function errBody(type, message, code = null, param = null) {
|
|
20
|
+
return { error: { message, type, param, code } };
|
|
21
|
+
}
|
|
22
|
+
function invalidRequest(message, param = null, code = null) {
|
|
23
|
+
return { status: 400, body: errBody('invalid_request_error', message, code, param) };
|
|
24
|
+
}
|
|
25
|
+
function nowEpoch(occurredAt) {
|
|
26
|
+
return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
|
|
27
|
+
}
|
|
28
|
+
// ── usage ledger (real recorded usage → the usage/costs reporting endpoints) ──────────────
|
|
29
|
+
// Every billable inference call appends a usage record (model + token counts + a synthetic cost
|
|
30
|
+
// computed from a per-model price table). The /v1/organization/usage|costs endpoints aggregate
|
|
31
|
+
// these REAL recorded rows — nothing is hardcoded; with no traffic the report is genuinely empty.
|
|
32
|
+
// Per-model USD price per 1M tokens (a faithful slice of the published price sheet; deterministic).
|
|
33
|
+
const MODEL_PRICES = {
|
|
34
|
+
'gpt-4o': { input: 2.5, output: 10 },
|
|
35
|
+
'gpt-4o-mini': { input: 0.15, output: 0.6 },
|
|
36
|
+
'gpt-4.1': { input: 2, output: 8 },
|
|
37
|
+
'gpt-4.1-mini': { input: 0.4, output: 1.6 },
|
|
38
|
+
'gpt-4-turbo': { input: 10, output: 30 },
|
|
39
|
+
'o3': { input: 2, output: 8 },
|
|
40
|
+
'o4-mini': { input: 1.1, output: 4.4 },
|
|
41
|
+
'gpt-3.5-turbo': { input: 0.5, output: 1.5 },
|
|
42
|
+
'text-embedding-3-small': { input: 0.02, output: 0 },
|
|
43
|
+
'text-embedding-3-large': { input: 0.13, output: 0 },
|
|
44
|
+
'text-embedding-ada-002': { input: 0.1, output: 0 },
|
|
45
|
+
};
|
|
46
|
+
function modelPrice(model) {
|
|
47
|
+
return MODEL_PRICES[model] ?? { input: 0, output: 0 };
|
|
48
|
+
}
|
|
49
|
+
export function usageCost(model, inputTokens, outputTokens) {
|
|
50
|
+
const p = modelPrice(model);
|
|
51
|
+
return (inputTokens / 1e6) * p.input + (outputTokens / 1e6) * p.output;
|
|
52
|
+
}
|
|
53
|
+
const WIDTHS = {
|
|
54
|
+
'1m': { seconds: 60, limit: 60, max: 1440 },
|
|
55
|
+
'1h': { seconds: 3600, limit: 24, max: 168 },
|
|
56
|
+
'1d': { seconds: 86400, limit: 7, max: 31 },
|
|
57
|
+
};
|
|
58
|
+
/** A report's query from its parameters; Costs buckets by day only and pages up to 180 of them. */
|
|
59
|
+
export function reportQuery(params, now, costs) {
|
|
60
|
+
const width = WIDTHS[String(params.bucket_width ?? '1d')] ?? WIDTHS['1d'];
|
|
61
|
+
const max = costs ? 180 : width.max;
|
|
62
|
+
const asked = Number(params.limit);
|
|
63
|
+
const limit = Number.isInteger(asked) && asked >= 1 ? Math.min(asked, max) : width.limit;
|
|
64
|
+
const start = Number(params.start_time);
|
|
65
|
+
const end = params.end_time !== undefined ? Number(params.end_time) : now;
|
|
66
|
+
const page = /^page_(\d+)$/.exec(String(params.page ?? ''));
|
|
67
|
+
const groups = params.group_by;
|
|
68
|
+
return { start, end, width: width.seconds, limit, from: page ? Number(page[1]) : start, groupBy: (Array.isArray(groups) ? groups : groups === undefined ? [] : [groups]).map(String) };
|
|
69
|
+
}
|
|
70
|
+
/** Buckets of one page of the query, each holding `results(rows in its window)`. */
|
|
71
|
+
export function bucketed(records, q, results) {
|
|
72
|
+
const data = [];
|
|
73
|
+
let at = q.from;
|
|
74
|
+
while (at < q.end && data.length < q.limit) {
|
|
75
|
+
const until = Math.min(at + q.width, q.end);
|
|
76
|
+
data.push({ object: 'bucket', start_time: at, end_time: until, results: results(records.filter((r) => Number(r.created_at) >= at && Number(r.created_at) < until), at, until) });
|
|
77
|
+
at = until;
|
|
78
|
+
}
|
|
79
|
+
return { object: 'page', data, has_more: at < q.end, next_page: at < q.end ? `page_${at}` : null };
|
|
80
|
+
}
|
|
81
|
+
/** Rows grouped by `key` when grouping by it, else all in one group keyed null; none when there are no rows. */
|
|
82
|
+
function grouped(rows, by) {
|
|
83
|
+
const out = new Map();
|
|
84
|
+
for (const r of rows) {
|
|
85
|
+
const k = by === null ? null : String(r[by]);
|
|
86
|
+
out.set(k, [...(out.get(k) ?? []), r]);
|
|
87
|
+
}
|
|
88
|
+
return [...out.entries()].sort((a, b) => (String(a[0]) < String(b[0]) ? -1 : 1));
|
|
89
|
+
}
|
|
90
|
+
/** What each usage kind's result counts, and the object it answers as (the spec's Usage*Result schemas;
|
|
91
|
+
* https://platform.openai.com/docs/api-reference/usage). A result names the grouping it answers (model, and for images
|
|
92
|
+
* their size and source) and null for the groupings not asked for. */
|
|
93
|
+
const USAGE_RESULTS = {
|
|
94
|
+
completions: {
|
|
95
|
+
object: 'organization.usage.completions.result',
|
|
96
|
+
// the twin caches nothing and hears, sees and speaks nothing: every token is uncached text
|
|
97
|
+
count: (rs) => {
|
|
98
|
+
const input = sum(rs, 'input_tokens');
|
|
99
|
+
const output = sum(rs, 'output_tokens');
|
|
100
|
+
return { input_tokens: input, input_cached_tokens: 0, input_cache_write_tokens: 0, input_uncached_tokens: input, output_tokens: output, input_text_tokens: input, output_text_tokens: output, input_cached_text_tokens: 0, input_audio_tokens: 0, input_cached_audio_tokens: 0, output_audio_tokens: 0, input_image_tokens: 0, input_cached_image_tokens: 0, output_image_tokens: 0, num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null, batch: null, service_tier: null };
|
|
101
|
+
},
|
|
102
|
+
},
|
|
103
|
+
embeddings: { object: 'organization.usage.embeddings.result', count: (rs) => ({ input_tokens: sum(rs, 'input_tokens'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
|
|
104
|
+
moderations: { object: 'organization.usage.moderations.result', count: (rs) => ({ input_tokens: sum(rs, 'input_tokens'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
|
|
105
|
+
audio_speeches: { object: 'organization.usage.audio_speeches.result', count: (rs) => ({ characters: sum(rs, 'characters'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
|
|
106
|
+
audio_transcriptions: { object: 'organization.usage.audio_transcriptions.result', count: (rs) => ({ seconds: sum(rs, 'seconds'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
|
|
107
|
+
images: { object: 'organization.usage.images.result', by: ['size', 'source'], count: (rs) => ({ images: sum(rs, 'images'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
|
|
108
|
+
code_interpreter_sessions: { object: 'organization.usage.code_interpreter_sessions.result', count: (rs) => ({ num_sessions: sum(rs, 'num_sessions') }) },
|
|
109
|
+
file_search_calls: { object: 'organization.usage.file_searches.result', count: (rs) => ({ num_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null, vector_store_id: null }) },
|
|
110
|
+
web_search_calls: { object: 'organization.usage.web_searches.result', count: (rs) => ({ num_model_requests: sum(rs, 'num_model_requests'), num_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null, context_level: null }) },
|
|
111
|
+
};
|
|
112
|
+
const sum = (rows, k) => rows.reduce((n, r) => n + Number(r[k] ?? 0), 0);
|
|
113
|
+
/** The kinds whose results name a model. */
|
|
114
|
+
const MODELLESS = new Set(['code_interpreter_sessions', 'file_search_calls']);
|
|
115
|
+
/** The usage report over the recorded usage rows, for one kind ('completions', 'embeddings', …). */
|
|
116
|
+
export function usageReport(records, kind, q) {
|
|
117
|
+
const spec = USAGE_RESULTS[kind ?? 'completions'] ?? USAGE_RESULTS.completions;
|
|
118
|
+
const byModel = q.groupBy.includes('model') && !MODELLESS.has(kind ?? '');
|
|
119
|
+
const extra = (spec.by ?? []).filter((k) => q.groupBy.includes(k));
|
|
120
|
+
return bucketed(records.filter((r) => kind === undefined || r.kind === kind), q, (rows) => groupedBy(rows, [...(byModel ? ['model'] : []), ...extra]).map(([key, rs]) => ({
|
|
121
|
+
object: spec.object,
|
|
122
|
+
...spec.count(rs),
|
|
123
|
+
project_id: null,
|
|
124
|
+
...(MODELLESS.has(kind ?? '') ? {} : { model: byModel ? key.model : null }),
|
|
125
|
+
...(spec.by ? Object.fromEntries(spec.by.map((k) => [k, extra.includes(k) ? key[k] : null])) : {}),
|
|
126
|
+
})));
|
|
127
|
+
}
|
|
128
|
+
/** Rows grouped by the values of `keys`, in key order; one group of all rows when no key; none when there are no rows. */
|
|
129
|
+
function groupedBy(rows, keys) {
|
|
130
|
+
const out = new Map();
|
|
131
|
+
for (const r of rows) {
|
|
132
|
+
const key = Object.fromEntries(keys.map((k) => [k, r[k] === undefined ? null : String(r[k])]));
|
|
133
|
+
const id = JSON.stringify(key);
|
|
134
|
+
out.set(id, [key, [...(out.get(id)?.[1] ?? []), r]]);
|
|
135
|
+
}
|
|
136
|
+
return [...out.entries()].sort((a, b) => (a[0] < b[0] ? -1 : 1)).map(([, v]) => v);
|
|
137
|
+
}
|
|
138
|
+
/** The costs report: the recorded usage priced per model, per bucket, by line item (usage kind) when grouped so. */
|
|
139
|
+
export function costsReport(records, q) {
|
|
140
|
+
const byLine = q.groupBy.includes('line_item');
|
|
141
|
+
return bucketed(records, q, (rows) => grouped(rows, byLine ? 'kind' : null).map(([line, rs]) => ({
|
|
142
|
+
object: 'organization.costs.result',
|
|
143
|
+
amount: { value: rs.reduce((n, r) => n + Number(r.cost_usd ?? 0), 0), currency: 'usd' },
|
|
144
|
+
line_item: line, project_id: null, api_key_id: null, quantity: null, quantity_unit: null,
|
|
145
|
+
})));
|
|
146
|
+
}
|
|
147
|
+
const SYSTEM_FINGERPRINT = 'fp_twin_stub';
|
|
148
|
+
/** The roles a chat message may have (the spec's ChatCompletionRequestMessage variants). */
|
|
149
|
+
const ROLES = new Set(['developer', 'system', 'user', 'assistant', 'tool', 'function']);
|
|
150
|
+
/** OpenAI's refusal of a chat message whose role is not one of the request message types
|
|
151
|
+
* (https://platform.openai.com/docs/api-reference/chat/create, `messages`). */
|
|
152
|
+
function invalidRole() {
|
|
153
|
+
return { error: invalidRequest("each message must have a valid 'role'", 'messages') };
|
|
154
|
+
}
|
|
155
|
+
// The chat options below are OpenAI's (https://platform.openai.com/docs/api-reference/chat/create); each
|
|
156
|
+
// parser answers the option's value or OpenAI's refusal of it.
|
|
157
|
+
/** `n`: how many choices to generate, at least one. */
|
|
158
|
+
function chatN(raw) {
|
|
159
|
+
const n = Number(raw);
|
|
160
|
+
return Number.isInteger(n) && n >= 1 ? n : { error: invalidRequest("'n' must be an integer >= 1", 'n') };
|
|
161
|
+
}
|
|
162
|
+
/** `max_completion_tokens` (or the legacy `max_tokens`): a cap of at least one token. */
|
|
163
|
+
function chatMaxTokens(raw) {
|
|
164
|
+
const max = Number(raw);
|
|
165
|
+
return Number.isInteger(max) && max >= 1 ? max : { error: invalidRequest("'max_tokens' must be an integer >= 1", 'max_tokens') };
|
|
166
|
+
}
|
|
167
|
+
/** `stop`: one sequence or a list of them. */
|
|
168
|
+
function chatStop(raw) {
|
|
169
|
+
if (typeof raw === 'string')
|
|
170
|
+
return [raw];
|
|
171
|
+
return Array.isArray(raw) ? raw : { error: invalidRequest("'stop' must be a string or array of strings", 'stop') };
|
|
172
|
+
}
|
|
173
|
+
/** `tool_choice`: 'auto' | 'none' | 'required' | { type:'function', function:{ name } }. */
|
|
174
|
+
function chatToolChoice(raw) {
|
|
175
|
+
if (raw === 'auto' || raw === 'none' || raw === 'required')
|
|
176
|
+
return raw;
|
|
177
|
+
if (!raw || typeof raw !== 'object')
|
|
178
|
+
return { error: invalidRequest("'tool_choice' must be 'auto'/'none'/'required' or a named function", 'tool_choice') };
|
|
179
|
+
const fn = raw.function;
|
|
180
|
+
return typeof fn?.name === 'string' ? { name: fn.name } : { error: invalidRequest("invalid 'tool_choice' — named choice requires function.name", 'tool_choice') };
|
|
181
|
+
}
|
|
182
|
+
/** OpenAI's refusal of a `response_format` whose type is none of text, json_object and json_schema. */
|
|
183
|
+
function badResponseFormat() {
|
|
184
|
+
return { error: invalidRequest("'response_format.type' must be 'text', 'json_object', or 'json_schema'", 'response_format') };
|
|
185
|
+
}
|
|
186
|
+
/** `top_logprobs`: 0 to 20 alternatives per token, only with `logprobs: true`. */
|
|
187
|
+
function chatTopLogprobs(raw, logprobs) {
|
|
188
|
+
const top = Number(raw);
|
|
189
|
+
if (!Number.isInteger(top) || top < 0 || top > 20)
|
|
190
|
+
return { error: invalidRequest("'top_logprobs' must be an integer between 0 and 20", 'top_logprobs') };
|
|
191
|
+
return logprobs ? top : { error: invalidRequest("'top_logprobs' requires 'logprobs' to be true", 'top_logprobs') };
|
|
192
|
+
}
|
|
193
|
+
/** `seed`: an integer for best-effort reproducible sampling. */
|
|
194
|
+
function chatSeed(raw) {
|
|
195
|
+
const seed = Number(raw);
|
|
196
|
+
return Number.isInteger(seed) ? seed : { error: invalidRequest("'seed' must be an integer", 'seed') };
|
|
197
|
+
}
|
|
198
|
+
/** `logit_bias`: token ids mapped to a bias from -100 to 100. */
|
|
199
|
+
function chatLogitBias(raw) {
|
|
200
|
+
if (!raw || typeof raw !== 'object' || Array.isArray(raw))
|
|
201
|
+
return { error: invalidRequest("'logit_bias' must be an object mapping token ids to bias values", 'logit_bias') };
|
|
202
|
+
const bias = {};
|
|
203
|
+
for (const [k, v] of Object.entries(raw)) {
|
|
204
|
+
const num = Number(v);
|
|
205
|
+
if (!Number.isFinite(num) || num < -100 || num > 100)
|
|
206
|
+
return { error: invalidRequest("each 'logit_bias' value must be a number between -100 and 100", 'logit_bias') };
|
|
207
|
+
bias[k] = num;
|
|
208
|
+
}
|
|
209
|
+
return bias;
|
|
210
|
+
}
|
|
211
|
+
/** `prediction`: predicted output, `{ type: 'content', content }`, its content a string or text parts. */
|
|
212
|
+
function chatPrediction(raw) {
|
|
213
|
+
const p = raw;
|
|
214
|
+
if (!p || typeof p !== 'object' || p.type !== 'content' || p.content === undefined)
|
|
215
|
+
return { error: invalidRequest("'prediction' must be an object with type 'content' and a content field", 'prediction') };
|
|
216
|
+
if (typeof p.content === 'string')
|
|
217
|
+
return p.content;
|
|
218
|
+
return Array.isArray(p.content) ? p.content.map((c) => (c && typeof c === 'object' ? String(c.text ?? '') : String(c))).join('') : '';
|
|
219
|
+
}
|
|
220
|
+
/** `modalities` with `audio`: a spoken answer needs `audio: { voice, format }` (the vendor's rule). The twin
|
|
221
|
+
* can't synthesize speech, so the returned bytes are a labeled stub; the envelope is faithful. */
|
|
222
|
+
function chatAudio(modalities, audio) {
|
|
223
|
+
if (!Array.isArray(modalities))
|
|
224
|
+
return { error: invalidRequest("'modalities' must be an array", 'modalities') };
|
|
225
|
+
if (!modalities.includes('audio'))
|
|
226
|
+
return undefined;
|
|
227
|
+
const a = audio;
|
|
228
|
+
if (!a || typeof a !== 'object' || typeof a.voice !== 'string' || typeof a.format !== 'string')
|
|
229
|
+
return { error: invalidRequest("'audio' with a 'voice' and 'format' is required when 'modalities' includes 'audio'", 'audio') };
|
|
230
|
+
return { voice: a.voice, format: a.format };
|
|
231
|
+
}
|
|
232
|
+
export function validateChat(params) {
|
|
233
|
+
if (params.model === undefined || params.model === '')
|
|
234
|
+
return { error: invalidRequest("you must provide a model parameter", 'model') };
|
|
235
|
+
if (typeof params.model !== 'string')
|
|
236
|
+
return { error: invalidRequest("'model' must be a string", 'model') };
|
|
237
|
+
if (!Array.isArray(params.messages))
|
|
238
|
+
return { error: invalidRequest("you must provide a messages parameter", 'messages') };
|
|
239
|
+
if (params.messages.length === 0)
|
|
240
|
+
return { error: invalidRequest("[] is too short - 'messages'", 'messages') };
|
|
241
|
+
const messages = params.messages;
|
|
242
|
+
if (messages.some((m) => !m || typeof m !== 'object' || !ROLES.has(String(m.role))))
|
|
243
|
+
return invalidRole();
|
|
244
|
+
// Each option the request may carry is parsed by its own function (below): a function returns the
|
|
245
|
+
// option's value, or OpenAI's refusal of it as `{ error }`.
|
|
246
|
+
const n = params.n === undefined ? 1 : chatN(params.n);
|
|
247
|
+
if (typeof n !== 'number')
|
|
248
|
+
return n;
|
|
249
|
+
// max_completion_tokens is the current name; max_tokens is the legacy alias.
|
|
250
|
+
const maxRaw = params.max_completion_tokens ?? params.max_tokens;
|
|
251
|
+
const maxTokens = maxRaw === undefined ? undefined : chatMaxTokens(maxRaw);
|
|
252
|
+
if (typeof maxTokens === 'object')
|
|
253
|
+
return maxTokens;
|
|
254
|
+
const stop = params.stop === undefined ? undefined : chatStop(params.stop);
|
|
255
|
+
if (stop && !Array.isArray(stop))
|
|
256
|
+
return stop;
|
|
257
|
+
const toolChoice = params.tool_choice === undefined ? undefined : chatToolChoice(params.tool_choice);
|
|
258
|
+
if (toolChoice && typeof toolChoice === 'object' && 'error' in toolChoice)
|
|
259
|
+
return toolChoice;
|
|
260
|
+
// response_format: { type:'text' | 'json_object' | 'json_schema', json_schema? }.
|
|
261
|
+
let responseFormat = { kind: 'text' };
|
|
262
|
+
const rf = params.response_format;
|
|
263
|
+
if (rf !== undefined) {
|
|
264
|
+
if (!rf || typeof rf !== 'object')
|
|
265
|
+
return { error: invalidRequest("'response_format' must be an object", 'response_format') };
|
|
266
|
+
const t = rf.type ?? 'text';
|
|
267
|
+
const format = t === 'json_schema' ? { kind: 'json_schema', schema: rf.json_schema } : t === 'json_object' || t === 'text' ? { kind: t } : undefined;
|
|
268
|
+
if (!format)
|
|
269
|
+
return badResponseFormat();
|
|
270
|
+
responseFormat = format;
|
|
271
|
+
}
|
|
272
|
+
// stream_options.include_usage → emit a final usage-only chunk in the stream.
|
|
273
|
+
const so = params.stream_options;
|
|
274
|
+
const includeUsage = !!(so && typeof so === 'object' && so.include_usage === true);
|
|
275
|
+
// logprobs (boolean) + top_logprobs (0..20, requires logprobs:true) → per-token logprob detail.
|
|
276
|
+
const logprobs = params.logprobs === true;
|
|
277
|
+
const topLogprobs = params.top_logprobs === undefined ? undefined : chatTopLogprobs(params.top_logprobs, logprobs);
|
|
278
|
+
if (typeof topLogprobs === 'object')
|
|
279
|
+
return topLogprobs;
|
|
280
|
+
// seed → reproducible sampling (the twin is already deterministic; we echo it via fingerprint).
|
|
281
|
+
const seed = params.seed === undefined ? undefined : chatSeed(params.seed);
|
|
282
|
+
if (typeof seed === 'object')
|
|
283
|
+
return seed;
|
|
284
|
+
// logit_bias → a map of token-id → bias in [-100, 100].
|
|
285
|
+
const logitBias = params.logit_bias === undefined ? undefined : chatLogitBias(params.logit_bias);
|
|
286
|
+
if (logitBias && 'error' in logitBias)
|
|
287
|
+
return logitBias;
|
|
288
|
+
// prediction → predicted outputs ({ type:'content', content }); content may be a string or parts.
|
|
289
|
+
const prediction = params.prediction === undefined ? undefined : chatPrediction(params.prediction);
|
|
290
|
+
if (typeof prediction === 'object')
|
|
291
|
+
return prediction;
|
|
292
|
+
// modalities + audio → audio output.
|
|
293
|
+
const audioOutput = params.modalities === undefined ? undefined : chatAudio(params.modalities, params.audio);
|
|
294
|
+
if (audioOutput && 'error' in audioOutput)
|
|
295
|
+
return audioOutput;
|
|
296
|
+
// store + metadata → stored completions (retrievable later); metadata must be a flat object.
|
|
297
|
+
// a model refuses what its page says it does not take (openai-models.ts); a reasoning model reasons with the effort
|
|
298
|
+
// asked for, else the spec's default
|
|
299
|
+
const refused = unsupported(params.model, { effort: params.reasoning_effort, effortParam: 'reasoning_effort', temperature: params.temperature, top_p: params.top_p, logprobs: params.logprobs, chatTools: params.tools ?? params.functions });
|
|
300
|
+
if (refused)
|
|
301
|
+
return { error: invalidRequest(refused.message, refused.param, 'unsupported_value') };
|
|
302
|
+
const reasoningEffort = typeof params.reasoning_effort === 'string' ? params.reasoning_effort : reasons(params.model) ? DEFAULT_EFFORT : undefined;
|
|
303
|
+
const store = params.store === true;
|
|
304
|
+
let metadata;
|
|
305
|
+
if (params.metadata !== undefined) {
|
|
306
|
+
if (!params.metadata || typeof params.metadata !== 'object' || Array.isArray(params.metadata))
|
|
307
|
+
return { error: invalidRequest("'metadata' must be an object", 'metadata') };
|
|
308
|
+
metadata = params.metadata;
|
|
309
|
+
}
|
|
310
|
+
return {
|
|
311
|
+
args: {
|
|
312
|
+
model: params.model,
|
|
313
|
+
messages,
|
|
314
|
+
tools: params.tools,
|
|
315
|
+
functions: params.functions,
|
|
316
|
+
n,
|
|
317
|
+
...(maxTokens !== undefined ? { maxTokens } : {}),
|
|
318
|
+
...(stop !== undefined ? { stop } : {}),
|
|
319
|
+
stream: params.stream === true,
|
|
320
|
+
...(toolChoice !== undefined ? { toolChoice } : {}),
|
|
321
|
+
parallelToolCalls: params.parallel_tool_calls !== false,
|
|
322
|
+
responseFormat,
|
|
323
|
+
includeUsage,
|
|
324
|
+
logprobs,
|
|
325
|
+
...(topLogprobs !== undefined ? { topLogprobs } : {}),
|
|
326
|
+
...(seed !== undefined ? { seed } : {}),
|
|
327
|
+
...(logitBias !== undefined ? { logitBias: logitBias } : {}),
|
|
328
|
+
...(prediction !== undefined ? { prediction } : {}),
|
|
329
|
+
...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
|
|
330
|
+
store,
|
|
331
|
+
...(metadata !== undefined ? { metadata } : {}),
|
|
332
|
+
...(audioOutput !== undefined ? { audioOutput: audioOutput } : {}),
|
|
333
|
+
},
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
/** The text cut at the EARLIEST-occurring stop sequence across the whole `stop` list, not whichever is
|
|
337
|
+
* listed first: the real vendor's "stop generation at the first hit", regardless of array order. */
|
|
338
|
+
function stoppedAt(text, stops) {
|
|
339
|
+
let stopAt = -1;
|
|
340
|
+
for (const s of stops) {
|
|
341
|
+
if (!s)
|
|
342
|
+
continue;
|
|
343
|
+
const i = text.indexOf(s);
|
|
344
|
+
if (i >= 0 && (stopAt < 0 || i < stopAt))
|
|
345
|
+
stopAt = i;
|
|
346
|
+
}
|
|
347
|
+
return stopAt >= 0 ? text.slice(0, stopAt) : text;
|
|
348
|
+
}
|
|
349
|
+
/** The text cut to a `max_completion_tokens` cap (about four characters a token); the choice finishes `length`. */
|
|
350
|
+
function cutAt(text, maxTokens) {
|
|
351
|
+
return text.slice(0, maxTokens * 4);
|
|
352
|
+
}
|
|
353
|
+
/** modalities:['audio'] → the assistant replies with an audio object (content is null, the text lives in
|
|
354
|
+
* audio.transcript). The twin can't synthesize speech, so `data` is a labeled-stub base64 string; the shape
|
|
355
|
+
* (id/data/transcript/expires_at) is vendor-faithful. */
|
|
356
|
+
function audioChoice(args, idx, transcript, logprobs, finish) {
|
|
357
|
+
const voice = args.audioOutput;
|
|
358
|
+
const stubBytes = `[twin-stub:${args.model}] no real audio synthesis — voice=${voice.voice} format=${voice.format}; transcript: ${transcript}`;
|
|
359
|
+
const audio = {
|
|
360
|
+
id: `audio-twin-${stableSuffix(args)}-${idx}`,
|
|
361
|
+
data: Buffer.from(stubBytes, 'utf8').toString('base64'),
|
|
362
|
+
transcript,
|
|
363
|
+
expires_at: nowEpoch() + 3600,
|
|
364
|
+
};
|
|
365
|
+
return {
|
|
366
|
+
choice: { index: idx, message: { role: 'assistant', content: null, refusal: null, audio }, logprobs, finish_reason: finish },
|
|
367
|
+
completionTokens: estimateTokens(transcript),
|
|
368
|
+
};
|
|
369
|
+
}
|
|
370
|
+
// Build ONE deterministic stub choice (index `idx`). Honors tools/functions (tool_calls +
|
|
371
|
+
// finish_reason tool_calls), tool_choice (none/required/named), parallel_tool_calls,
|
|
372
|
+
// response_format (json_object/json_schema), max_tokens, and stop sequences.
|
|
373
|
+
function buildChoice(args, idx) {
|
|
374
|
+
const toolsSource = args.tools ?? args.functions;
|
|
375
|
+
const hasTools = Array.isArray(toolsSource) && toolsSource.length > 0;
|
|
376
|
+
// tool_choice gates whether the stub calls a tool: 'none' forbids it; a named/required choice
|
|
377
|
+
// forces it (even when the heuristic otherwise would not); 'auto'/default calls when tools exist.
|
|
378
|
+
const forbidTools = args.toolChoice === 'none';
|
|
379
|
+
const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
|
|
380
|
+
// a model answers a tool's result in words unless the caller insists on another call; one that always
|
|
381
|
+
// called a tool would never let an app's tool loop end
|
|
382
|
+
const answeringTool = ['tool', 'function'].includes(String(args.messages.at(-1)?.role));
|
|
383
|
+
const wantTool = hasTools && !forbidTools && (!answeringTool || forcedName !== undefined || args.toolChoice === 'required');
|
|
384
|
+
if (wantTool) {
|
|
385
|
+
// parallel_tool_calls (default true) → the stub may emit one call per provided tool; a named
|
|
386
|
+
// choice or parallel:false collapses to a single call.
|
|
387
|
+
// (the tools are there, so each call is made)
|
|
388
|
+
const calls = forcedName || args.parallelToolCalls === false
|
|
389
|
+
? [stubToolCall(toolsSource, idx + 1, forcedName, lastUserText(args.messages))]
|
|
390
|
+
: toolsSource.map((tool, t) => stubToolCall([tool], idx * 100 + t + 1, undefined, lastUserText(args.messages)));
|
|
391
|
+
return {
|
|
392
|
+
choice: { index: idx, message: { role: 'assistant', content: null, tool_calls: calls, refusal: null }, logprobs: null, finish_reason: 'tool_calls' },
|
|
393
|
+
completionTokens: estimateTokens(JSON.stringify(calls)),
|
|
394
|
+
};
|
|
395
|
+
}
|
|
396
|
+
// json_mode: when response_format requests json_object/json_schema, the content is valid JSON.
|
|
397
|
+
// prediction (predicted outputs): a real model uses the prediction to speed decoding but still
|
|
398
|
+
// returns its own generation — the twin echoes the predicted content (clearly still a stub) so
|
|
399
|
+
// the prediction round-trips, then reports accepted/rejected prediction tokens in usage.
|
|
400
|
+
let text = args.prediction !== undefined
|
|
401
|
+
? `[twin-stub:${args.model}] predicted-output echo: ${args.prediction}`
|
|
402
|
+
: args.responseFormat.kind === 'json_object'
|
|
403
|
+
? stubJsonObject(args.messages, args.model)
|
|
404
|
+
: args.responseFormat.kind === 'json_schema'
|
|
405
|
+
? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
|
|
406
|
+
: stubAssistantText(args.messages, args.model);
|
|
407
|
+
if (args.stop)
|
|
408
|
+
text = stoppedAt(text, args.stop);
|
|
409
|
+
const capped = args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens;
|
|
410
|
+
if (capped)
|
|
411
|
+
text = cutAt(text, args.maxTokens);
|
|
412
|
+
const finish = capped ? 'length' : 'stop';
|
|
413
|
+
const logprobs = args.logprobs ? buildLogprobs(text, args.topLogprobs ?? 0) : null;
|
|
414
|
+
if (args.audioOutput)
|
|
415
|
+
return audioChoice(args, idx, text, logprobs, finish);
|
|
416
|
+
return {
|
|
417
|
+
// a text answer carries its annotations (none: the twin cites no web source), as OpenAI's does
|
|
418
|
+
// (https://platform.openai.com/docs/api-reference/chat/object, `choices[].message.annotations`)
|
|
419
|
+
choice: { index: idx, message: { role: 'assistant', content: text, refusal: null, annotations: [] }, logprobs, finish_reason: finish },
|
|
420
|
+
completionTokens: estimateTokens(text),
|
|
421
|
+
};
|
|
422
|
+
}
|
|
423
|
+
/** Predicted-output accounting: the twin echoes the prediction, so every predicted token is "accepted". */
|
|
424
|
+
function predictionUsage(prediction) {
|
|
425
|
+
return { accepted_prediction_tokens: estimateTokens(prediction), rejected_prediction_tokens: 0 };
|
|
426
|
+
}
|
|
427
|
+
export function buildChatCompletion(args, occurredAt, decision) {
|
|
428
|
+
const promptTokens = countPromptTokens(args.messages);
|
|
429
|
+
// Scenario handlers: a fired handler scripts the assistant turn; a miss teaches in the stub. The
|
|
430
|
+
// decision was made (and any fault honored) by the request handler through the engine's serve(); this
|
|
431
|
+
// builder only realizes it — a status fault never reaches here.
|
|
432
|
+
const turn = decision && scenarioTurn(decision);
|
|
433
|
+
const choices = [];
|
|
434
|
+
let completionTokens = 0;
|
|
435
|
+
for (let i = 0; i < args.n; i++) {
|
|
436
|
+
const { choice, completionTokens: ct } = turn?.scripted ? scriptedChoice(turn.scripted, i) : buildChoice(args, i);
|
|
437
|
+
if (turn?.missTeach && typeof choice.message.content === 'string')
|
|
438
|
+
choice.message.content += turn.missTeach;
|
|
439
|
+
choices.push(choice);
|
|
440
|
+
completionTokens += ct;
|
|
441
|
+
}
|
|
442
|
+
// usage breaks its counts down as OpenAI's does (https://platform.openai.com/docs/api-reference/chat/object,
|
|
443
|
+
// `usage.prompt_tokens_details`, `usage.completion_tokens_details`): the twin caches nothing and reasons and
|
|
444
|
+
// speaks nothing, so those are zero
|
|
445
|
+
const usage = {
|
|
446
|
+
prompt_tokens: promptTokens, completion_tokens: completionTokens, total_tokens: promptTokens + completionTokens,
|
|
447
|
+
prompt_tokens_details: { cached_tokens: 0, audio_tokens: 0 },
|
|
448
|
+
completion_tokens_details: { reasoning_tokens: 0, audio_tokens: 0, accepted_prediction_tokens: 0, rejected_prediction_tokens: 0 },
|
|
449
|
+
};
|
|
450
|
+
// prediction (predicted outputs): real responses report how many predicted tokens were accepted
|
|
451
|
+
// vs rejected. The twin echoes the prediction, so all predicted tokens are "accepted".
|
|
452
|
+
if (args.prediction !== undefined)
|
|
453
|
+
usage.completion_tokens_details = { ...usage.completion_tokens_details, ...predictionUsage(args.prediction) };
|
|
454
|
+
// a reasoning model's reasoning tokens are billed as output tokens (https://developers.openai.com/api/docs/guides/reasoning:
|
|
455
|
+
// "they still occupy space in the model's context window and are billed as output tokens")
|
|
456
|
+
if (args.reasoningEffort !== undefined && reasons(args.model)) {
|
|
457
|
+
const spent = REASONING_BUDGET[args.reasoningEffort] ?? 0;
|
|
458
|
+
usage.completion_tokens += spent;
|
|
459
|
+
usage.total_tokens += spent;
|
|
460
|
+
usage.completion_tokens_details = { ...usage.completion_tokens_details, reasoning_tokens: spent };
|
|
461
|
+
}
|
|
462
|
+
return {
|
|
463
|
+
id: `chatcmpl-twin-${stableSuffix(args)}`,
|
|
464
|
+
object: 'chat.completion',
|
|
465
|
+
created: nowEpoch(occurredAt),
|
|
466
|
+
model: args.model,
|
|
467
|
+
choices,
|
|
468
|
+
usage,
|
|
469
|
+
system_fingerprint: SYSTEM_FINGERPRINT,
|
|
470
|
+
// the tier that served the request: the standard one, which is what `auto` and `default` pick for a project
|
|
471
|
+
// not on Scale Tier (https://platform.openai.com/docs/api-reference/chat/object, `service_tier`)
|
|
472
|
+
service_tier: 'default',
|
|
473
|
+
...(args.metadata !== undefined ? { metadata: args.metadata } : {}),
|
|
474
|
+
};
|
|
475
|
+
}
|
|
476
|
+
// A deterministic id suffix from the request (so ids are stable + assertable, like the twin's
|
|
477
|
+
// other deterministic outputs). Hash of the prompt text + model + seed (seed changes sampling,
|
|
478
|
+
// so it changes the response id the way a real seed-distinct request does).
|
|
479
|
+
function stableSuffix(args) {
|
|
480
|
+
let h = 0x811c9dc5;
|
|
481
|
+
const s = JSON.stringify(args.messages) + args.model + (args.seed !== undefined ? `|seed=${args.seed}` : '');
|
|
482
|
+
for (let i = 0; i < s.length; i++) {
|
|
483
|
+
h ^= s.charCodeAt(i);
|
|
484
|
+
h = Math.imul(h, 0x01000193);
|
|
485
|
+
}
|
|
486
|
+
return (h >>> 0).toString(36);
|
|
487
|
+
}
|
|
488
|
+
// Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order.
|
|
489
|
+
function chunkText(text) {
|
|
490
|
+
if (!text)
|
|
491
|
+
return [];
|
|
492
|
+
const out = [];
|
|
493
|
+
for (let i = 0; i < text.length; i += 20)
|
|
494
|
+
out.push(text.slice(i, i + 20));
|
|
495
|
+
return out;
|
|
496
|
+
}
|
|
497
|
+
/** A streamed choice's tool calls: one tool_calls delta-pair per call, each carrying its own `index`
|
|
498
|
+
* (parallel tool calls). */
|
|
499
|
+
function streamToolCalls(calls, idx, base, sink) {
|
|
500
|
+
calls.forEach((tc, tIdx) => {
|
|
501
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, logprobs: null, finish_reason: null }] } });
|
|
502
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, logprobs: null, finish_reason: null }] } });
|
|
503
|
+
});
|
|
504
|
+
}
|
|
505
|
+
/**
|
|
506
|
+
* Emit the vendor-faithful Chat Completions streaming sequence into the injected sink (NO
|
|
507
|
+
* sockets, NO setTimeout). Real order: a first chunk with `delta:{role:'assistant'}`, then
|
|
508
|
+
* `delta:{content}` chunks (or tool_calls deltas), then a final chunk with `finish_reason`,
|
|
509
|
+
* then `[DONE]`. Deterministic + synchronous so a collector can assert the full sequence.
|
|
510
|
+
*/
|
|
511
|
+
export function streamChat(args, sink, occurredAt, decision) {
|
|
512
|
+
const full = buildChatCompletion(args, occurredAt, decision);
|
|
513
|
+
const base = { id: full.id, object: 'chat.completion.chunk', created: full.created, model: full.model, system_fingerprint: SYSTEM_FINGERPRINT };
|
|
514
|
+
for (const choice of full.choices) {
|
|
515
|
+
const idx = choice.index;
|
|
516
|
+
// role chunk
|
|
517
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, logprobs: null, finish_reason: null }] } });
|
|
518
|
+
if (choice.message.tool_calls && choice.message.tool_calls.length)
|
|
519
|
+
streamToolCalls(choice.message.tool_calls, idx, base, sink);
|
|
520
|
+
else {
|
|
521
|
+
for (const piece of chunkText(choice.message.content ?? '')) {
|
|
522
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, logprobs: null, finish_reason: null }] } });
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: {}, logprobs: null, finish_reason: choice.finish_reason }] } });
|
|
526
|
+
}
|
|
527
|
+
// stream_options.include_usage → a final chunk with an empty choices array carrying `usage`.
|
|
528
|
+
if (args.includeUsage) {
|
|
529
|
+
sink({ data: { ...base, choices: [], usage: full.usage } });
|
|
530
|
+
}
|
|
531
|
+
sink({ done: true });
|
|
532
|
+
return full;
|
|
533
|
+
}
|
|
534
|
+
export function responseMessages(items, instructions) {
|
|
535
|
+
const messages = instructions ? [{ role: 'system', content: instructions }] : [];
|
|
536
|
+
const partsText = (content) => typeof content === 'string' ? content : Array.isArray(content) ? content.map((c) => c && typeof c === 'object' ? String(c.text ?? JSON.stringify(c)) : String(c)).join('\n') : '';
|
|
537
|
+
for (const item of items) {
|
|
538
|
+
if (item.type === 'function_call') {
|
|
539
|
+
messages.push({ role: 'assistant', content: '', tool_calls: [{ id: String(item.call_id ?? item.id ?? ''), type: 'function', function: { name: String(item.name ?? ''), arguments: typeof item.arguments === 'string' ? item.arguments : JSON.stringify(item.arguments ?? {}) } }] });
|
|
540
|
+
}
|
|
541
|
+
else if (item.type === 'function_call_output') {
|
|
542
|
+
messages.push({ role: 'tool', content: partsText(item.output), tool_call_id: String(item.call_id ?? '') });
|
|
543
|
+
}
|
|
544
|
+
else if (item.type === 'message' || item.type === undefined) {
|
|
545
|
+
messages.push({ role: (item.role ?? 'user'), content: partsText(item.content) });
|
|
546
|
+
}
|
|
547
|
+
// Reasoning and other typed items remain in state, without inventing user turns.
|
|
548
|
+
}
|
|
549
|
+
return messages;
|
|
550
|
+
}
|
|
551
|
+
/** OpenAI's refusal of an `input` list holding something other than input items (objects)
|
|
552
|
+
* (https://platform.openai.com/docs/api-reference/responses/create, `input`). */
|
|
553
|
+
function itemsNotObjects() {
|
|
554
|
+
return { error: invalidRequest("'input' items must be objects", 'input') };
|
|
555
|
+
}
|
|
556
|
+
export function validateResponses(params) {
|
|
557
|
+
if (params.model === undefined || params.model === '')
|
|
558
|
+
return { error: invalidRequest("you must provide a model parameter", 'model') };
|
|
559
|
+
if (typeof params.model !== 'string')
|
|
560
|
+
return { error: invalidRequest("'model' must be a string", 'model') };
|
|
561
|
+
if (params.input === undefined)
|
|
562
|
+
return { error: invalidRequest("you must provide an input parameter", 'input') };
|
|
563
|
+
// Keep vendor items as state; the messages view is only a projection for the scenario.
|
|
564
|
+
let inputItems;
|
|
565
|
+
if (typeof params.input === 'string') {
|
|
566
|
+
inputItems = [{ type: 'message', role: 'user', content: [{ type: 'input_text', text: params.input }] }];
|
|
567
|
+
}
|
|
568
|
+
else if (Array.isArray(params.input)) {
|
|
569
|
+
if (params.input.some((item) => !item || typeof item !== 'object' || Array.isArray(item)))
|
|
570
|
+
return itemsNotObjects();
|
|
571
|
+
inputItems = params.input.map((item) => {
|
|
572
|
+
if (item.type !== undefined && item.type !== 'message')
|
|
573
|
+
return { ...item };
|
|
574
|
+
return { ...item, type: 'message', role: item.role ?? 'user', content: typeof item.content === 'string' ? [{ type: item.role === 'assistant' ? 'output_text' : 'input_text', text: item.content }] : item.content };
|
|
575
|
+
});
|
|
576
|
+
}
|
|
577
|
+
else {
|
|
578
|
+
return { error: invalidRequest("'input' must be a string or an array of input items", 'input') };
|
|
579
|
+
}
|
|
580
|
+
const instructions = typeof params.instructions === 'string' ? params.instructions : undefined;
|
|
581
|
+
const messages = responseMessages(inputItems, instructions);
|
|
582
|
+
const inputText = responseMessages(inputItems).map((m) => m.content).filter(Boolean).join('\n');
|
|
583
|
+
if (params.tools !== undefined && !Array.isArray(params.tools))
|
|
584
|
+
return { error: invalidRequest("'tools' must be an array", 'tools') };
|
|
585
|
+
const maxOut = params.max_output_tokens;
|
|
586
|
+
if (maxOut !== undefined && (typeof maxOut !== 'number' || !Number.isInteger(maxOut) || maxOut < 1))
|
|
587
|
+
return { error: invalidRequest("'max_output_tokens' must be a positive integer", 'max_output_tokens') };
|
|
588
|
+
const prev = params.previous_response_id;
|
|
589
|
+
if (prev !== undefined && (typeof prev !== 'string' || !prev))
|
|
590
|
+
return { error: invalidRequest("'previous_response_id' must be a string", 'previous_response_id') };
|
|
591
|
+
// reasoning.effort → the model spends a (stubbed) reasoning budget; the item shape is faithful.
|
|
592
|
+
let reasoningEffort;
|
|
593
|
+
if (params.reasoning !== undefined) {
|
|
594
|
+
const r = params.reasoning;
|
|
595
|
+
if (!r || typeof r !== 'object' || Array.isArray(r))
|
|
596
|
+
return { error: invalidRequest("'reasoning' must be an object", 'reasoning') };
|
|
597
|
+
const effort = r.effort;
|
|
598
|
+
if (effort !== undefined && effort !== null)
|
|
599
|
+
reasoningEffort = effort;
|
|
600
|
+
}
|
|
601
|
+
// a model refuses what its page says it does not take (openai-models.ts); a reasoning model reasons with the effort
|
|
602
|
+
// asked for, else the spec's default
|
|
603
|
+
const refused = unsupported(params.model, { effort: reasoningEffort, effortParam: 'reasoning.effort', temperature: params.temperature, top_p: params.top_p });
|
|
604
|
+
if (refused)
|
|
605
|
+
return { error: invalidRequest(refused.message, refused.param, 'unsupported_value') };
|
|
606
|
+
if (reasoningEffort === undefined && reasons(params.model))
|
|
607
|
+
reasoningEffort = DEFAULT_EFFORT;
|
|
608
|
+
const reasoningSummary = params.reasoning && typeof params.reasoning.summary === 'string' ? String(params.reasoning.summary) : undefined;
|
|
609
|
+
return {
|
|
610
|
+
args: {
|
|
611
|
+
model: params.model,
|
|
612
|
+
inputItems,
|
|
613
|
+
instructions,
|
|
614
|
+
inputText,
|
|
615
|
+
messages,
|
|
616
|
+
stream: params.stream === true,
|
|
617
|
+
store: params.store !== false, // OpenAI defaults store=true (stored & retrievable)
|
|
618
|
+
background: params.background === true,
|
|
619
|
+
...(typeof prev === 'string' ? { previousResponseId: prev } : {}),
|
|
620
|
+
...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
|
|
621
|
+
...(Array.isArray(params.tools) ? { tools: params.tools } : {}),
|
|
622
|
+
...(typeof maxOut === 'number' ? { maxTokens: maxOut } : {}),
|
|
623
|
+
...(reasoningSummary !== undefined ? { reasoningSummary } : {}),
|
|
624
|
+
// OpenAI's defaults for what was not sent (the Response schema's): temperature and top_p 1, tools in parallel, auto;
|
|
625
|
+
// plain text out, no truncation, no end user, no tool-call cap, no alternative tokens, the default tier
|
|
626
|
+
// (https://platform.openai.com/docs/api-reference/responses/object)
|
|
627
|
+
echo: {
|
|
628
|
+
metadata: params.metadata ?? {}, temperature: params.temperature ?? 1, top_p: params.top_p ?? 1, parallel_tool_calls: params.parallel_tool_calls ?? true, tool_choice: params.tool_choice ?? 'auto',
|
|
629
|
+
text: params.text ?? { format: { type: 'text' } }, truncation: params.truncation ?? 'disabled', user: params.user ?? null, max_tool_calls: params.max_tool_calls ?? null,
|
|
630
|
+
top_logprobs: params.top_logprobs ?? 0, service_tier: ['flex', 'priority', 'scale'].includes(String(params.service_tier)) ? params.service_tier : 'default',
|
|
631
|
+
},
|
|
632
|
+
},
|
|
633
|
+
};
|
|
634
|
+
}
|
|
635
|
+
/** A tool as a response answers it: as sent, with what OpenAI fills in for what was not. A web search preview searches
|
|
636
|
+
* with `medium` context from the United States unless told otherwise (the spec's WebSearchPreviewTool: "`medium` is
|
|
637
|
+
* the default"; `user_location` "If omitted or null, defaults to the United States"); a function tool sent without
|
|
638
|
+
* `strict` answers strict, as the reference's Functions example answers the tool it sends; a file search tool, its
|
|
639
|
+
* default filter and ranking
|
|
640
|
+
* (https://platform.openai.com/docs/api-reference/responses/create). */
|
|
641
|
+
const TOOL_DEFAULTS = {
|
|
642
|
+
function: (t) => ({ ...t, strict: t.strict ?? true }),
|
|
643
|
+
// a file search sent without filters or ranking answers no filter and the automatic ranker with no threshold, as the
|
|
644
|
+
// reference's File search example answers the tool it sends
|
|
645
|
+
file_search: (t) => ({ ...t, filters: t.filters ?? null, ranking_options: t.ranking_options ?? { ranker: 'auto', score_threshold: 0 } }),
|
|
646
|
+
// and the domains it is limited to, none unless sent, as the reference's Web search example answers `"domains": []`
|
|
647
|
+
web_search_preview: (t) => ({ ...t, domains: t.domains ?? [], search_context_size: t.search_context_size ?? 'medium', user_location: t.user_location ?? { type: 'approximate', city: null, country: 'US', region: null, timezone: null } }),
|
|
648
|
+
};
|
|
649
|
+
TOOL_DEFAULTS.web_search_preview_2025_03_11 = TOOL_DEFAULTS.web_search_preview;
|
|
650
|
+
function answeredTool(tool) {
|
|
651
|
+
const t = (tool && typeof tool === 'object' ? tool : {});
|
|
652
|
+
return TOOL_DEFAULTS[String(t.type)]?.(t) ?? tool;
|
|
653
|
+
}
|
|
654
|
+
/** The built-in tool calls a response makes: a web search for a web search tool, a file search over the named stores for a
|
|
655
|
+
* file search tool, each querying the input's text (its results are not included unless asked for: `results` null). */
|
|
656
|
+
function builtInCalls(args, suffix) {
|
|
657
|
+
const query = args.inputText.slice(0, 200);
|
|
658
|
+
const out = [];
|
|
659
|
+
for (const [i, tool] of (Array.isArray(args.tools) ? args.tools : []).entries()) {
|
|
660
|
+
const type = String(tool?.type ?? '');
|
|
661
|
+
if (/^web_search/.test(type))
|
|
662
|
+
out.push({ type: 'web_search_call', id: `ws-twin-${suffix}-${i}`, status: 'completed', action: { type: 'search', query } });
|
|
663
|
+
if (type === 'file_search')
|
|
664
|
+
out.push({ type: 'file_search_call', id: `fs-twin-${suffix}-${i}`, status: 'completed', queries: [query], results: null });
|
|
665
|
+
}
|
|
666
|
+
return out;
|
|
667
|
+
}
|
|
668
|
+
/** What every response answers besides its output: the request's settings as sent (or OpenAI's defaults), no error,
|
|
669
|
+
* no program access (https://platform.openai.com/docs/api-reference/responses/object). */
|
|
670
|
+
const responseSettings = (args) => ({
|
|
671
|
+
instructions: args.instructions ?? null, tools: Array.isArray(args.tools) ? args.tools.map(answeredTool) : [], ...args.echo,
|
|
672
|
+
access_programs: null, error: null, incomplete_details: null,
|
|
673
|
+
// whether it is kept: the spec's own Response example answers `"store": true` (spec/patches.json)
|
|
674
|
+
background: args.background, store: args.store, max_output_tokens: args.maxTokens ?? null,
|
|
675
|
+
previous_response_id: args.previousResponseId ?? null,
|
|
676
|
+
reasoning: { effort: args.reasoningEffort ?? null, summary: args.reasoningSummary ?? null },
|
|
677
|
+
});
|
|
678
|
+
// A deterministic reasoning-token budget per effort level (more effort → more reasoning tokens).
|
|
679
|
+
const REASONING_BUDGET = { none: 0, minimal: 8, low: 16, medium: 48, high: 128, xhigh: 256, max: 512 };
|
|
680
|
+
/** The function call an unscripted response makes: when the caller requires one or names one, or on
|
|
681
|
+
* `auto` unless the input ends with a tool's result, which a model answers in words. */
|
|
682
|
+
function responseToolCall(args) {
|
|
683
|
+
const choice = args.echo.tool_choice;
|
|
684
|
+
const named = choice && typeof choice === 'object' ? String(choice.name ?? '') : undefined;
|
|
685
|
+
const answering = args.inputItems.at(-1)?.type === 'function_call_output';
|
|
686
|
+
const functions = (Array.isArray(args.tools) ? args.tools : []).filter((t) => t.type === 'function');
|
|
687
|
+
if (!functions.length || choice === 'none' || (answering && !named && choice !== 'required'))
|
|
688
|
+
return null;
|
|
689
|
+
return stubToolCall(functions, 1, named, lastUserText(args.messages));
|
|
690
|
+
}
|
|
691
|
+
export function buildResponse(args, occurredAt, idSuffix, decision) {
|
|
692
|
+
// The scenario decides the turn exactly as it does for chat completions: a fired handler scripts the
|
|
693
|
+
// text and the tool calls; a miss answers the labeled stub and teaches. The request handler has
|
|
694
|
+
// already honored a fault; only a content decision reaches here.
|
|
695
|
+
const suffix = idSuffix ?? String(nowEpoch(occurredAt));
|
|
696
|
+
const turn = decision && scenarioTurn(decision, `call-twin-${suffix}`);
|
|
697
|
+
const scripted = turn?.scripted;
|
|
698
|
+
let text = scripted ? (scripted.text ?? '') : stubAssistantText(args.messages, args.model) + (turn?.missTeach ?? '');
|
|
699
|
+
const inputTokens = countPromptTokens(args.messages);
|
|
700
|
+
const messageItem = {
|
|
701
|
+
type: 'message',
|
|
702
|
+
id: `msg-twin-${suffix}`,
|
|
703
|
+
status: 'completed',
|
|
704
|
+
role: 'assistant',
|
|
705
|
+
// no log probabilities unless asked for (`include: ["message.output_text.logprobs"]`): an empty list
|
|
706
|
+
content: [{ type: 'output_text', text, annotations: [], logprobs: [] }],
|
|
707
|
+
};
|
|
708
|
+
// unscripted, the stub calls a function tool as the chat stub does (buildChoice)
|
|
709
|
+
const stubbed = scripted ? null : responseToolCall(args);
|
|
710
|
+
if (stubbed)
|
|
711
|
+
text = '';
|
|
712
|
+
const calls = (scripted?.toolCalls ?? (stubbed ? [stubbed] : [])).map((tc, i) => ({ type: 'function_call', id: `fc-twin-${suffix}-${i}`, status: 'completed', call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments }));
|
|
713
|
+
const messageTokens = estimateTokens(text) + estimateTokens(calls.map((c) => c.arguments).join(''));
|
|
714
|
+
const output = [];
|
|
715
|
+
// usage breaks its counts down as OpenAI's does (https://platform.openai.com/docs/api-reference/responses/object, `usage`):
|
|
716
|
+
// the twin caches nothing, so no input token was read from or written to a cache
|
|
717
|
+
const usage = { input_tokens: inputTokens, input_tokens_details: { cached_tokens: 0, cache_write_tokens: 0 }, output_tokens: messageTokens, output_tokens_details: { reasoning_tokens: 0 }, total_tokens: inputTokens + messageTokens };
|
|
718
|
+
// reasoning.effort → a faithful reasoning item (labeled-stub summary) BEFORE the message item,
|
|
719
|
+
// plus output_tokens_details.reasoning_tokens in usage (counted into output_tokens, like the vendor).
|
|
720
|
+
if (args.reasoningEffort) {
|
|
721
|
+
const reasoningTokens = REASONING_BUDGET[args.reasoningEffort] ?? 16;
|
|
722
|
+
// a summary only when one was asked for (`reasoning.summary`); without, OpenAI answers the item's summary empty
|
|
723
|
+
// (https://platform.openai.com/docs/guides/reasoning#reasoning-summaries)
|
|
724
|
+
const reasoningItem = {
|
|
725
|
+
type: 'reasoning',
|
|
726
|
+
id: `rs-twin-${suffix}`,
|
|
727
|
+
summary: args.reasoningSummary ? [{ type: 'summary_text', text: `[twin-stub] reasoning summary (effort=${args.reasoningEffort}); the twin cannot run the model, so the chain-of-thought is not real.` }] : [],
|
|
728
|
+
};
|
|
729
|
+
// only a model that reasons lists its reasoning item (openai-models.ts)
|
|
730
|
+
if (reasons(args.model))
|
|
731
|
+
output.push(reasoningItem);
|
|
732
|
+
usage.output_tokens += reasoningTokens;
|
|
733
|
+
usage.total_tokens += reasoningTokens;
|
|
734
|
+
usage.output_tokens_details = { reasoning_tokens: reasoningTokens };
|
|
735
|
+
}
|
|
736
|
+
// the built-in tools the request offers, which OpenAI runs itself before answering: the placeholder model searches
|
|
737
|
+
// with each one offered, on the input's text, unless `tool_choice` is `none` (the reference's Web search and File search
|
|
738
|
+
// examples answer a `web_search_call` / `file_search_call` item before the message:
|
|
739
|
+
// https://platform.openai.com/docs/guides/tools-web-search, https://platform.openai.com/docs/guides/tools-file-search)
|
|
740
|
+
if (!scripted && args.echo.tool_choice !== 'none')
|
|
741
|
+
output.push(...builtInCalls(args, suffix));
|
|
742
|
+
// A scripted turn with tool calls and no words carries the calls alone, as the vendor does.
|
|
743
|
+
if (text || !calls.length)
|
|
744
|
+
output.push(messageItem);
|
|
745
|
+
output.push(...calls);
|
|
746
|
+
return {
|
|
747
|
+
id: `resp-twin-${suffix}`,
|
|
748
|
+
object: 'response',
|
|
749
|
+
created_at: nowEpoch(occurredAt),
|
|
750
|
+
status: 'completed',
|
|
751
|
+
// answered whole at once: completed the instant it was created
|
|
752
|
+
completed_at: nowEpoch(occurredAt),
|
|
753
|
+
model: args.model,
|
|
754
|
+
// no `output_text`: the spec's is an "SDK-only convenience property" the SDKs compute from `output`
|
|
755
|
+
output,
|
|
756
|
+
usage,
|
|
757
|
+
...responseSettings(args),
|
|
758
|
+
};
|
|
759
|
+
}
|
|
760
|
+
// Unstored responses have no state to overwrite. Stored identities are allocated atomically below.
|
|
761
|
+
export function responseSuffix(args) {
|
|
762
|
+
let h = 0x811c9dc5;
|
|
763
|
+
const s = JSON.stringify([args.model, args.inputItems ?? args.messages, args.previousResponseId, args.instructions]);
|
|
764
|
+
for (let i = 0; i < s.length; i++) {
|
|
765
|
+
h ^= s.charCodeAt(i);
|
|
766
|
+
h = Math.imul(h, 0x01000193);
|
|
767
|
+
}
|
|
768
|
+
return (h >>> 0).toString(36);
|
|
769
|
+
}
|
|
770
|
+
// Allocate and create under the kernel's cross-process lock. Repeated requests are occurrences;
|
|
771
|
+
// a content hash alone would overwrite an earlier response (and its queued scenario decision).
|
|
772
|
+
export async function createStoredResponse(args, req, decision) {
|
|
773
|
+
const { value } = await applyTwinWriteAtomic(SERVICE, (resources) => {
|
|
774
|
+
let ordinal = 0;
|
|
775
|
+
for (const row of resources) {
|
|
776
|
+
if (row.type !== 'response')
|
|
777
|
+
continue;
|
|
778
|
+
const match = /^resp-twin-(\d+)$/.exec(String(row.id));
|
|
779
|
+
if (match)
|
|
780
|
+
ordinal = Math.max(ordinal, Number(match[1]));
|
|
781
|
+
}
|
|
782
|
+
const suffix = String(ordinal + 1);
|
|
783
|
+
const inputItems = args.inputItems.map((item, i) => ({ ...item, id: item.id ?? `item-in-${suffix}-${i}` }));
|
|
784
|
+
const resp = args.background ? {
|
|
785
|
+
id: `resp-twin-${suffix}`, object: 'response', created_at: nowEpoch(req.occurredAt),
|
|
786
|
+
status: 'queued', completed_at: null, model: args.model, output: [],
|
|
787
|
+
usage: null, ...responseSettings(args),
|
|
788
|
+
} : buildResponse(args, req.occurredAt, suffix, decision);
|
|
789
|
+
return { kind: 'write', value: resp, write: {
|
|
790
|
+
operation: 'response.create', subjectType: 'response', subjectId: resp.id,
|
|
791
|
+
fields: { ...resp, _stored: true, _input_items: inputItems,
|
|
792
|
+
...(args.background ? { _bg_args: JSON.stringify(args), _bg_suffix: suffix, _bg_decision: decision ? JSON.stringify(decision) : null } : {}),
|
|
793
|
+
},
|
|
794
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
795
|
+
} };
|
|
796
|
+
}, req.root);
|
|
797
|
+
return value;
|
|
798
|
+
}
|
|
799
|
+
export function emitResponse(resp, sink) {
|
|
800
|
+
sink({ data: { type: 'response.created', response: { ...resp, output: [] } } });
|
|
801
|
+
// Emit each output item in order, as the vendor streams them: a reasoning item whole, the message
|
|
802
|
+
// item's text as deltas at its own index, a function call's arguments as one delta then done.
|
|
803
|
+
resp.output.forEach((item, i) => {
|
|
804
|
+
sink({ data: { type: 'response.output_item.added', output_index: i, item: item.type === 'function_call' ? { ...item, arguments: '' } : item } });
|
|
805
|
+
if (item.type === 'message') {
|
|
806
|
+
const text = item.content[0]?.text ?? '';
|
|
807
|
+
for (const piece of chunkText(text))
|
|
808
|
+
sink({ data: { type: 'response.output_text.delta', output_index: i, content_index: 0, delta: piece } });
|
|
809
|
+
sink({ data: { type: 'response.output_text.done', output_index: i, content_index: 0, text } });
|
|
810
|
+
}
|
|
811
|
+
if (item.type === 'function_call') {
|
|
812
|
+
sink({ data: { type: 'response.function_call_arguments.delta', output_index: i, item_id: item.id, delta: item.arguments } });
|
|
813
|
+
sink({ data: { type: 'response.function_call_arguments.done', output_index: i, item_id: item.id, arguments: item.arguments } });
|
|
814
|
+
}
|
|
815
|
+
sink({ data: { type: 'response.output_item.done', output_index: i, item } });
|
|
816
|
+
});
|
|
817
|
+
sink({ data: { type: 'response.completed', response: resp } });
|
|
818
|
+
sink({ done: true });
|
|
819
|
+
return resp;
|
|
820
|
+
}
|
|
821
|
+
// ── Embeddings (deterministic pseudo-vectors) ───────────────────────────────────────────
|
|
822
|
+
export function handleEmbeddings(params) {
|
|
823
|
+
if (params.model === undefined || params.model === '')
|
|
824
|
+
return invalidRequest("you must provide a model parameter", 'model');
|
|
825
|
+
if (params.input === undefined)
|
|
826
|
+
return invalidRequest("you must provide an input parameter", 'input');
|
|
827
|
+
const model = String(params.model);
|
|
828
|
+
const inputs = typeof params.input === 'string'
|
|
829
|
+
? [params.input]
|
|
830
|
+
: Array.isArray(params.input)
|
|
831
|
+
? params.input.map((x) => (typeof x === 'string' ? x : JSON.stringify(x)))
|
|
832
|
+
: [];
|
|
833
|
+
if (inputs.length === 0)
|
|
834
|
+
return invalidRequest("'input' must be a non-empty string or array", 'input');
|
|
835
|
+
const dimensions = params.dimensions !== undefined ? Number(params.dimensions) : defaultDims(model);
|
|
836
|
+
if (!Number.isInteger(dimensions) || dimensions < 1)
|
|
837
|
+
return invalidRequest("'dimensions' must be a positive integer", 'dimensions');
|
|
838
|
+
// The `openai` SDK defaults to encoding_format:'base64' and decodes a base64 Float32 buffer
|
|
839
|
+
// back into numbers client-side. Honor both formats faithfully.
|
|
840
|
+
const asBase64 = params.encoding_format === 'base64';
|
|
841
|
+
let promptTokens = 0;
|
|
842
|
+
const data = inputs.map((text, index) => {
|
|
843
|
+
promptTokens += estimateTokens(text);
|
|
844
|
+
const vec = pseudoEmbedding(text, dimensions);
|
|
845
|
+
return { object: 'embedding', index, embedding: (asBase64 ? floatsToBase64(vec) : vec) };
|
|
846
|
+
});
|
|
847
|
+
const out = { object: 'list', data, model, usage: { prompt_tokens: promptTokens, total_tokens: promptTokens } };
|
|
848
|
+
return { status: 200, body: out };
|
|
849
|
+
}
|
|
850
|
+
function defaultDims(model) {
|
|
851
|
+
if (model.includes('large'))
|
|
852
|
+
return 3072;
|
|
853
|
+
return 1536; // small + ada-002
|
|
854
|
+
}
|
|
855
|
+
/** Encode a float vector as a base64 little-endian Float32 buffer (the SDK's base64 format). */
|
|
856
|
+
function floatsToBase64(vec) {
|
|
857
|
+
const buf = new ArrayBuffer(vec.length * 4);
|
|
858
|
+
const view = new DataView(buf);
|
|
859
|
+
for (let i = 0; i < vec.length; i++)
|
|
860
|
+
view.setFloat32(i * 4, vec[i], true);
|
|
861
|
+
// btoa over the raw bytes (available in Bun/browser).
|
|
862
|
+
let bin = '';
|
|
863
|
+
const bytes = new Uint8Array(buf);
|
|
864
|
+
for (let i = 0; i < bytes.length; i++)
|
|
865
|
+
bin += String.fromCharCode(bytes[i]);
|
|
866
|
+
return btoa(bin);
|
|
867
|
+
}
|
|
868
|
+
// ── Moderations (deterministic) ─────────────────────────────────────────────────────────
|
|
869
|
+
export function handleModerations(params, occurredAt) {
|
|
870
|
+
if (params.input === undefined)
|
|
871
|
+
return invalidRequest("you must provide an input parameter", 'input');
|
|
872
|
+
// an array of strings is several inputs, one result each; an array of text and image parts is ONE multimodal input,
|
|
873
|
+
// one result (https://platform.openai.com/docs/api-reference/moderations/create, `input`)
|
|
874
|
+
const list = Array.isArray(params.input) ? params.input : [];
|
|
875
|
+
const parts = list.length > 0 && list.every((x) => x && typeof x === 'object' && !Array.isArray(x));
|
|
876
|
+
const inputs = typeof params.input === 'string'
|
|
877
|
+
? [{ text: params.input, image: false }]
|
|
878
|
+
: parts
|
|
879
|
+
? [{ text: list.filter((x) => x.type === 'text').map((x) => String(x.text ?? '')).join('\n'), image: list.some((x) => x.type === 'image_url') }]
|
|
880
|
+
: list.map((x) => ({ text: typeof x === 'string' ? x : JSON.stringify(x), image: false }));
|
|
881
|
+
if (inputs.length === 0)
|
|
882
|
+
return invalidRequest("'input' must be a non-empty string or array", 'input');
|
|
883
|
+
const model = modelOf('createModeration', params);
|
|
884
|
+
const results = inputs.map((t) => moderateText(t.text, t.image));
|
|
885
|
+
return { status: 200, body: { id: `modr-twin-${nowEpoch(occurredAt)}`, model, results } };
|
|
886
|
+
}
|
|
887
|
+
// ── Images (a real placeholder image, no model: ./openai-media.ts) ─────────────────────────────────
|
|
888
|
+
const gptImage = (model) => /^(gpt-image|chatgpt-image)/.test(model);
|
|
889
|
+
/** One answered image: the GPT image models give the image itself as base64, always; `dall-e-2` and `dall-e-3` a URL
|
|
890
|
+
* unless `response_format` is `b64_json`, and `dall-e-3` its revised prompt (developers.openai.com/api/reference/
|
|
891
|
+
* resources/images/methods/generate, `response_format`; the Image object's `b64_json`, `url`, `revised_prompt`). */
|
|
892
|
+
function imageData(params, model, size, seed, kind, i, occurredAt) {
|
|
893
|
+
return gptImage(model) || params.response_format === 'b64_json' ? { b64_json: base64(placeholderPng(size, `${seed} #${i + 1}`)) } : dalleUrl(model, seed, kind, i, occurredAt);
|
|
894
|
+
}
|
|
895
|
+
/** A DALL·E image as its URL, the default for `dall-e-2` and `dall-e-3`. */
|
|
896
|
+
function dalleUrl(model, seed, kind, i, occurredAt) {
|
|
897
|
+
return { url: `https://twin.invalid/openai-${kind}-stub/${nowEpoch(occurredAt)}-${i}.png`, ...(model === 'dall-e-3' ? { revised_prompt: `[twin-stub] ${seed}` } : {}) };
|
|
898
|
+
}
|
|
899
|
+
function imageCount(params) {
|
|
900
|
+
const n = Number(params.n ?? 1);
|
|
901
|
+
return Number.isInteger(n) && n > 0 ? n : 1;
|
|
902
|
+
}
|
|
903
|
+
/** A GPT image model's answer: the images with the tokens they cost (the image generation guide's 1024x1024
|
|
904
|
+
* medium-quality image is 1056 output tokens, scaled here by area; the twin reads no pixels, so it counts no input image
|
|
905
|
+
* tokens: https://platform.openai.com/docs/guides/image-generation#cost-and-latency), or, with `stream`, the
|
|
906
|
+
* `partial_images` asked (none by default: "When set to 0, the response will be a single image sent in one streaming
|
|
907
|
+
* event") then the completed image (https://platform.openai.com/docs/api-reference/images-streaming). */
|
|
908
|
+
function gptImageAnswer(params, size, data, kind, occurredAt) {
|
|
909
|
+
const text = estimateTokens(String(params.prompt ?? ''));
|
|
910
|
+
const output = Math.ceil((1056 * size.width * size.height) / (1024 * 1024)) * data.length;
|
|
911
|
+
const usage = { total_tokens: text + output, input_tokens: text, output_tokens: output, input_tokens_details: { text_tokens: text, image_tokens: 0 } };
|
|
912
|
+
const created = nowEpoch(occurredAt);
|
|
913
|
+
if (params.stream !== true && params.stream !== 'true')
|
|
914
|
+
return { status: 200, body: { created, data, usage } };
|
|
915
|
+
const settings = {
|
|
916
|
+
created_at: created, size: `${size.width}x${size.height}`,
|
|
917
|
+
// what `auto` settles on for a placeholder: an opaque PNG of medium quality
|
|
918
|
+
quality: params.quality && params.quality !== 'auto' ? params.quality : 'medium',
|
|
919
|
+
background: params.background && params.background !== 'auto' ? params.background : 'opaque',
|
|
920
|
+
output_format: params.output_format ?? 'png',
|
|
921
|
+
};
|
|
922
|
+
const partials = Math.min(3, Math.max(0, Number(params.partial_images ?? 0) || 0));
|
|
923
|
+
const b64 = String(data[0]?.b64_json ?? '');
|
|
924
|
+
const events = [
|
|
925
|
+
...Array.from({ length: partials }, (_, i) => ({ event: `${kind}.partial_image`, data: { type: `${kind}.partial_image`, b64_json: b64, ...settings, partial_image_index: i } })),
|
|
926
|
+
{ event: `${kind}.completed`, data: { type: `${kind}.completed`, b64_json: b64, ...settings, usage } },
|
|
927
|
+
];
|
|
928
|
+
return { status: 200, body: { created, data, usage }, events };
|
|
929
|
+
}
|
|
930
|
+
export function handleImages(params, occurredAt) {
|
|
931
|
+
if (params.prompt === undefined || params.prompt === '')
|
|
932
|
+
return invalidRequest("you must provide a prompt parameter", 'prompt');
|
|
933
|
+
const model = modelOf('createImage', params);
|
|
934
|
+
const size = requestedSize(params.size);
|
|
935
|
+
const data = Array.from({ length: imageCount(params) }, (_, i) => imageData(params, model, size, String(params.prompt), 'image', i, occurredAt));
|
|
936
|
+
if (gptImage(model))
|
|
937
|
+
return gptImageAnswer(params, size, data, 'image_generation', occurredAt);
|
|
938
|
+
return { status: 200, body: { created: nowEpoch(occurredAt), data } };
|
|
939
|
+
}
|
|
940
|
+
/** OpenAI's refusal of an upload that is not an image the model takes. */
|
|
941
|
+
const badImage = (formats) => invalidRequest(`Invalid image file: 'image' must be ${formats}.`, 'image', 'invalid_image_format');
|
|
942
|
+
// An EDIT takes an image file and a prompt; a VARIATION (dall-e-2 only) a square PNG. Each is told by its bytes.
|
|
943
|
+
export function handleImageEdit(params, occurredAt) {
|
|
944
|
+
// a GPT image model edits several images sent as `image[]` (the form keeps the last one named so)
|
|
945
|
+
const image = params.image ?? params['image[]'];
|
|
946
|
+
if (image === undefined || image === '')
|
|
947
|
+
return invalidRequest("you must provide an image to edit", 'image');
|
|
948
|
+
if (params.prompt === undefined || params.prompt === '')
|
|
949
|
+
return invalidRequest("you must provide a prompt parameter", 'prompt');
|
|
950
|
+
const model = modelOf('createImageEdit', params);
|
|
951
|
+
const bytes = partBytes(image);
|
|
952
|
+
const format = bytes && imageFormat(bytes);
|
|
953
|
+
if (gptImage(model) ? !format : format !== 'png' || !square(bytes))
|
|
954
|
+
return badImage(gptImage(model) ? 'a png, webp, or jpg file' : 'a square png file');
|
|
955
|
+
const size = requestedSize(params.size, format === 'png' ? pngSize(bytes) : undefined);
|
|
956
|
+
const data = Array.from({ length: imageCount(params) }, (_, i) => imageData(params, model, size, String(params.prompt), 'image-edit', i, occurredAt));
|
|
957
|
+
if (gptImage(model))
|
|
958
|
+
return gptImageAnswer(params, size, data, 'image_edit', occurredAt);
|
|
959
|
+
return { status: 200, body: { created: nowEpoch(occurredAt), data } };
|
|
960
|
+
}
|
|
961
|
+
const square = (png) => { const s = pngSize(png); return s.width === s.height; };
|
|
962
|
+
export function handleImageVariation(params, occurredAt) {
|
|
963
|
+
if (params.image === undefined || params.image === '')
|
|
964
|
+
return invalidRequest("you must provide an image", 'image');
|
|
965
|
+
const bytes = partBytes(params.image);
|
|
966
|
+
if (!bytes || imageFormat(bytes) !== 'png' || !square(bytes))
|
|
967
|
+
return badImage('a valid square png file');
|
|
968
|
+
const size = requestedSize(params.size);
|
|
969
|
+
const data = Array.from({ length: imageCount(params) }, (_, i) => imageData(params, modelOf('createImageVariation', params), size, 'variation', 'image-variation', i, occurredAt));
|
|
970
|
+
return { status: 200, body: { created: nowEpoch(occurredAt), data } };
|
|
971
|
+
}
|
|
972
|
+
// ── Audio (deterministic stubs; the twin runs no model) ──────────────────────
|
|
973
|
+
// Transcription/translation cannot run a real speech model, so the twin returns a clearly
|
|
974
|
+
// labeled deterministic transcript naming the uploaded file. It takes the audio file itself, in a
|
|
975
|
+
// format OpenAI transcribes (./openai-media.ts), never a file's name. The response SHAPE
|
|
976
|
+
// (json / verbose_json / text) is vendor-faithful.
|
|
977
|
+
/** A multipart list field (`include[]`, `timestamp_granularities[]`), however the form named it. */
|
|
978
|
+
function formList(params, name) {
|
|
979
|
+
const v = params[name] ?? params[`${name}[]`];
|
|
980
|
+
return (Array.isArray(v) ? v : v === undefined ? [] : [v]).map(String);
|
|
981
|
+
}
|
|
982
|
+
/** What a transcription is billed on: the GPT-4o transcribe models by tokens, whisper-1 and the diarizing model by the
|
|
983
|
+
* audio's seconds (https://platform.openai.com/docs/api-reference/audio/json-object, `usage`). The twin counts ten
|
|
984
|
+
* audio tokens a second and its transcript's text tokens; neither is a model's count. */
|
|
985
|
+
function transcriptionUsage(model, seconds, text) {
|
|
986
|
+
if (model.startsWith('whisper') || model.includes('diarize'))
|
|
987
|
+
return { type: 'duration', seconds: Math.ceil(seconds) };
|
|
988
|
+
const audio = Math.ceil(seconds * 10);
|
|
989
|
+
const out = estimateTokens(text);
|
|
990
|
+
return { type: 'tokens', input_tokens: audio, input_token_details: { text_tokens: 0, audio_tokens: audio }, output_tokens: out, total_tokens: audio + out };
|
|
991
|
+
}
|
|
992
|
+
/** The transcript's words spread evenly over the audio: the twin hears nothing, so its timings are the file's length
|
|
993
|
+
* shared out, not a model's alignment. */
|
|
994
|
+
function words(text, seconds) {
|
|
995
|
+
const all = text.split(/\s+/).filter(Boolean);
|
|
996
|
+
const each = seconds / Math.max(1, all.length);
|
|
997
|
+
return all.map((word, i) => ({ word, start: i * each, end: (i + 1) * each }));
|
|
998
|
+
}
|
|
999
|
+
/** A transcription (or translation) of the uploaded audio: a labeled transcript naming the file, in the response
|
|
1000
|
+
* format asked (json, text, verbose_json with the word or segment timestamps asked, or diarized_json), with the
|
|
1001
|
+
* token log probabilities when `include[]=logprobs`, and as `transcript.text.delta` events then
|
|
1002
|
+
* `transcript.text.done` when `stream=true` on a model that streams (https://platform.openai.com/docs/api-reference/audio/createTranscription). */
|
|
1003
|
+
export function handleTranscription(params, translate) {
|
|
1004
|
+
if (params.file === undefined || params.file === '')
|
|
1005
|
+
return invalidRequest("you must provide a file parameter", 'file');
|
|
1006
|
+
if (params.model === undefined || params.model === '')
|
|
1007
|
+
return invalidRequest("you must provide a model parameter", 'model');
|
|
1008
|
+
const bytes = partBytes(params.file);
|
|
1009
|
+
if (!bytes || !audioFormat(bytes))
|
|
1010
|
+
return invalidRequest("Invalid file format. Supported formats: ['flac', 'm4a', 'mp3', 'mp4', 'mpeg', 'mpga', 'oga', 'ogg', 'wav', 'webm']", 'file', 'invalid_value');
|
|
1011
|
+
const filename = String(params.file.name || 'upload');
|
|
1012
|
+
const model = String(params.model);
|
|
1013
|
+
const verb = translate ? 'translation' : 'transcription';
|
|
1014
|
+
const text = `[twin-stub] deterministic ${verb} of ${filename} (no speech model is run)`;
|
|
1015
|
+
const format = typeof params.response_format === 'string' ? params.response_format : 'json';
|
|
1016
|
+
// the file's own length where the twin can read it (a WAV's header), else a labeled second
|
|
1017
|
+
const duration = wavSeconds(bytes) ?? 1.0;
|
|
1018
|
+
if (format === 'text')
|
|
1019
|
+
return { status: 200, body: text, seconds: duration };
|
|
1020
|
+
if (translate)
|
|
1021
|
+
return { status: 200, seconds: duration, body: format === 'verbose_json' ? { task: 'translation', language: 'english', duration, text, segments: [{ id: 0, start: 0, end: duration, text }] } : { text } };
|
|
1022
|
+
const usage = transcriptionUsage(model, duration, text);
|
|
1023
|
+
if (format === 'diarized_json') {
|
|
1024
|
+
// one speaker the twin cannot tell apart: the first name the caller knows, else OpenAI's first label
|
|
1025
|
+
const speaker = formList(params, 'known_speaker_names')[0] ?? 'A';
|
|
1026
|
+
return { status: 200, seconds: duration, body: { task: 'transcribe', duration, text, segments: [{ type: 'transcript.text.segment', id: 'seg_001', start: 0, end: duration, text, speaker }], usage } };
|
|
1027
|
+
}
|
|
1028
|
+
if (format === 'verbose_json') {
|
|
1029
|
+
const granularities = formList(params, 'timestamp_granularities');
|
|
1030
|
+
const segment = { id: 0, seek: 0, start: 0, end: duration, text, tokens: [], temperature: 0, avg_logprob: 0, compression_ratio: 1, no_speech_prob: 0 };
|
|
1031
|
+
return {
|
|
1032
|
+
status: 200,
|
|
1033
|
+
seconds: duration,
|
|
1034
|
+
body: {
|
|
1035
|
+
task: 'transcribe', language: typeof params.language === 'string' ? params.language : 'english', duration, text,
|
|
1036
|
+
...(granularities.includes('word') ? { words: words(text, duration) } : {}),
|
|
1037
|
+
...(!granularities.length || granularities.includes('segment') ? { segments: [segment] } : {}),
|
|
1038
|
+
usage,
|
|
1039
|
+
},
|
|
1040
|
+
};
|
|
1041
|
+
}
|
|
1042
|
+
const logprobs = formList(params, 'include').includes('logprobs') ? buildLogprobs(text, 0).content.map(({ token, logprob, bytes: b }) => ({ token, logprob, bytes: b })) : undefined;
|
|
1043
|
+
const body = { text, ...(logprobs ? { logprobs } : {}), usage };
|
|
1044
|
+
// whisper-1 does not stream ("Streaming is not supported for the whisper-1 model and will be ignored")
|
|
1045
|
+
if (String(params.stream) === 'true' && !model.startsWith('whisper')) {
|
|
1046
|
+
const pieces = text.match(/\s*\S+/g) ?? [];
|
|
1047
|
+
const events = pieces.map((delta) => ({ data: { type: 'transcript.text.delta', delta, ...(logprobs ? { logprobs: buildLogprobs(delta, 0).content.map(({ token, logprob, bytes: b }) => ({ token, logprob, bytes: b })) } : {}) } }));
|
|
1048
|
+
events.push({ data: { type: 'transcript.text.done', text, ...(logprobs ? { logprobs } : {}), usage } });
|
|
1049
|
+
return { status: 200, body, seconds: duration, events };
|
|
1050
|
+
}
|
|
1051
|
+
return { status: 200, body, seconds: duration };
|
|
1052
|
+
}
|
|
1053
|
+
// Text-to-speech: runs no speech model, but answers the audio file itself in the requested format, as OpenAI does
|
|
1054
|
+
// (a second of silence: ./openai-media.ts), with that format's Content-Type.
|
|
1055
|
+
const SPEECH = {
|
|
1056
|
+
mp3: { type: 'audio/mpeg', file: silentMp3 },
|
|
1057
|
+
opus: { type: 'audio/opus', file: silentOpus },
|
|
1058
|
+
aac: { type: 'audio/aac', file: silentAac },
|
|
1059
|
+
flac: { type: 'audio/flac', file: silentFlac },
|
|
1060
|
+
wav: { type: 'audio/wav', file: silentWav },
|
|
1061
|
+
pcm: { type: 'audio/pcm', file: silentPcm },
|
|
1062
|
+
};
|
|
1063
|
+
export function handleSpeech(params) {
|
|
1064
|
+
if (params.model === undefined || params.model === '')
|
|
1065
|
+
return invalidRequest("you must provide a model parameter", 'model');
|
|
1066
|
+
if (params.input === undefined || params.input === '')
|
|
1067
|
+
return invalidRequest("you must provide an input parameter", 'input');
|
|
1068
|
+
if (params.voice === undefined || params.voice === '')
|
|
1069
|
+
return invalidRequest("you must provide a voice parameter", 'voice');
|
|
1070
|
+
const format = SPEECH[String(params.response_format ?? 'mp3')];
|
|
1071
|
+
if (!format)
|
|
1072
|
+
return invalidRequest("'response_format' must be one of 'mp3', 'opus', 'aac', 'flac', 'wav', 'pcm'", 'response_format');
|
|
1073
|
+
return { status: 200, audio: format.file(), type: format.type };
|
|
1074
|
+
}
|
|
1075
|
+
// ── public entry: cross-cutting protocol (auth / rate-limit) then route ─────────────────────
|
|
1076
|
+
// The HTTP server passes request headers, so on the live wire every call is auth/rate-limit-
|
|
1077
|
+
// checked; in-process trusted calls (capability verify, connector) omit `headers`/`apiKey` and are
|
|
1078
|
+
// NOT gated.
|
|
1079
|
+
/** OpenAI's API as one call: the request goes through the pack's own dispatch (the derived dispatch over its
|
|
1080
|
+
* semantics, its derived core and the hand-written routes below), so a caller that holds no HTTP
|
|
1081
|
+
* server reaches exactly what an SDK does. A caller that presents no credential is a trusted
|
|
1082
|
+
* in-process caller and is not held to the auth rule; `sseSink` receives a stream's events. */
|
|
1083
|
+
export async function handleOpenAITwinRequest(req) {
|
|
1084
|
+
const twin = createOpenAITwinFetch({
|
|
1085
|
+
...(req.root !== undefined ? { root: req.root } : {}),
|
|
1086
|
+
readOnly: req.readOnly ?? false,
|
|
1087
|
+
...(req.scenarioEngine ? { scenarioEngine: req.scenarioEngine } : {}),
|
|
1088
|
+
...(req.occurredAt ? { clock: () => req.occurredAt } : {}),
|
|
1089
|
+
});
|
|
1090
|
+
const headers = { ...(req.headers ?? {}) };
|
|
1091
|
+
if (req.apiKey !== undefined)
|
|
1092
|
+
headers.authorization = `Bearer ${req.apiKey}`;
|
|
1093
|
+
else if (req.headers === undefined)
|
|
1094
|
+
headers.authorization = 'Bearer sk-twin-in-process';
|
|
1095
|
+
const method = req.method.toUpperCase();
|
|
1096
|
+
if (typeof req.body === 'string' && method !== 'GET' && method !== 'HEAD' && !headers['content-type'])
|
|
1097
|
+
headers['content-type'] = 'application/json';
|
|
1098
|
+
const path = req.path.startsWith('/') ? req.path : `/${req.path}`;
|
|
1099
|
+
const response = await twin(new Request(`https://api.openai.com${path}`, { method, headers, ...(req.body !== undefined && method !== 'GET' && method !== 'HEAD' ? { body: req.body } : {}) }));
|
|
1100
|
+
const answerHeaders = {};
|
|
1101
|
+
response.headers.forEach((v, k) => { if (k !== 'content-type')
|
|
1102
|
+
answerHeaders[k] = v; });
|
|
1103
|
+
const type = response.headers.get('content-type') ?? '';
|
|
1104
|
+
const text = await response.text();
|
|
1105
|
+
if (type.includes('text/event-stream')) {
|
|
1106
|
+
// the frames back into events, for a caller that collects them
|
|
1107
|
+
for (const frame of text.split('\n\n')) {
|
|
1108
|
+
const data = frame.split('\n').filter((l) => l.startsWith('data: ')).map((l) => l.slice(6)).join('\n');
|
|
1109
|
+
if (!data)
|
|
1110
|
+
continue;
|
|
1111
|
+
req.sseSink?.(data === '[DONE]' ? { done: true } : { data: JSON.parse(data) });
|
|
1112
|
+
}
|
|
1113
|
+
return { status: response.status, body: null, headers: answerHeaders };
|
|
1114
|
+
}
|
|
1115
|
+
const body = type.includes('json') ? (text ? JSON.parse(text) : null) : text;
|
|
1116
|
+
return { status: response.status, body, headers: answerHeaders };
|
|
1117
|
+
}
|