@volter/twin-openai 0.1.1 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. package/README.md +33 -30
  2. package/defaults/handlers.json +10 -0
  3. package/dist/defaults/handlers.json +10 -0
  4. package/dist/src/cli.d.ts +2 -0
  5. package/dist/src/cli.js +29 -0
  6. package/dist/src/generated/surface.gen.json +1 -0
  7. package/dist/src/generated/ui.gen.json +1 -0
  8. package/dist/src/index.d.ts +19 -0
  9. package/dist/src/index.js +72 -0
  10. package/dist/src/manifest.d.ts +6 -0
  11. package/dist/src/manifest.js +323 -0
  12. package/dist/src/openai-budget.d.ts +53 -0
  13. package/dist/src/openai-budget.js +147 -0
  14. package/dist/src/openai-capabilities.d.ts +4 -0
  15. package/dist/src/openai-capabilities.js +1569 -0
  16. package/dist/src/openai-conformance.d.ts +13 -0
  17. package/dist/src/openai-conformance.js +116 -0
  18. package/dist/src/openai-connector.d.ts +86 -0
  19. package/dist/src/openai-connector.js +291 -0
  20. package/dist/src/openai-media.d.ts +43 -0
  21. package/dist/src/openai-media.js +257 -0
  22. package/dist/src/openai-models.d.ts +74 -0
  23. package/dist/src/openai-models.js +148 -0
  24. package/dist/src/openai-scenario.d.ts +51 -0
  25. package/dist/src/openai-scenario.js +166 -0
  26. package/dist/src/openai-server.d.ts +40 -0
  27. package/dist/src/openai-server.js +126 -0
  28. package/dist/src/openai-stub.d.ts +82 -0
  29. package/dist/src/openai-stub.js +256 -0
  30. package/dist/src/openai-twin.d.ts +182 -0
  31. package/dist/src/openai-twin.js +1117 -0
  32. package/dist/src/openai-types.d.ts +194 -0
  33. package/dist/src/openai-types.js +4 -0
  34. package/dist/src/openai-webhooks.d.ts +47 -0
  35. package/dist/src/openai-webhooks.js +99 -0
  36. package/dist/src/screens/api-keys.d.ts +16 -0
  37. package/dist/src/screens/api-keys.js +131 -0
  38. package/dist/src/screens/session.d.ts +22 -0
  39. package/dist/src/screens/session.js +115 -0
  40. package/dist/src/semantics/assistants.d.ts +2 -0
  41. package/dist/src/semantics/assistants.js +331 -0
  42. package/dist/src/semantics/audio.d.ts +2 -0
  43. package/dist/src/semantics/audio.js +27 -0
  44. package/dist/src/semantics/batches.d.ts +4 -0
  45. package/dist/src/semantics/batches.js +86 -0
  46. package/dist/src/semantics/chat-completions.d.ts +3 -0
  47. package/dist/src/semantics/chat-completions.js +58 -0
  48. package/dist/src/semantics/containers.d.ts +2 -0
  49. package/dist/src/semantics/containers.js +147 -0
  50. package/dist/src/semantics/embeddings.d.ts +2 -0
  51. package/dist/src/semantics/embeddings.js +13 -0
  52. package/dist/src/semantics/evals.d.ts +2 -0
  53. package/dist/src/semantics/evals.js +173 -0
  54. package/dist/src/semantics/files.d.ts +13 -0
  55. package/dist/src/semantics/files.js +59 -0
  56. package/dist/src/semantics/fine-tuning.d.ts +4 -0
  57. package/dist/src/semantics/fine-tuning.js +178 -0
  58. package/dist/src/semantics/images.d.ts +2 -0
  59. package/dist/src/semantics/images.js +18 -0
  60. package/dist/src/semantics/index.d.ts +8 -0
  61. package/dist/src/semantics/index.js +46 -0
  62. package/dist/src/semantics/models.d.ts +2 -0
  63. package/dist/src/semantics/models.js +34 -0
  64. package/dist/src/semantics/moderations.d.ts +2 -0
  65. package/dist/src/semantics/moderations.js +12 -0
  66. package/dist/src/semantics/organization.d.ts +2 -0
  67. package/dist/src/semantics/organization.js +67 -0
  68. package/dist/src/semantics/progress.d.ts +22 -0
  69. package/dist/src/semantics/progress.js +63 -0
  70. package/dist/src/semantics/responses.d.ts +3 -0
  71. package/dist/src/semantics/responses.js +153 -0
  72. package/dist/src/semantics/shared.d.ts +32 -0
  73. package/dist/src/semantics/shared.js +69 -0
  74. package/dist/src/semantics/uploads.d.ts +2 -0
  75. package/dist/src/semantics/uploads.js +84 -0
  76. package/dist/src/semantics/vector-stores.d.ts +2 -0
  77. package/dist/src/semantics/vector-stores.js +281 -0
  78. package/dist/test-fixtures/openai-openapi-operations.SOURCE.md +18 -0
  79. package/dist/test-fixtures/openai-openapi-operations.json +1849 -0
  80. package/package.json +21 -10
  81. package/src/cli.ts +9 -7
  82. package/src/generated/surface.gen.json +1 -0
  83. package/src/generated/ui.gen.json +1 -0
  84. package/src/index.ts +20 -10
  85. package/src/manifest.ts +343 -0
  86. package/src/openai-budget.ts +4 -4
  87. package/src/openai-capabilities.ts +177 -195
  88. package/src/openai-conformance.ts +1 -1
  89. package/src/openai-connector.ts +40 -43
  90. package/src/openai-media.ts +225 -0
  91. package/src/openai-models.ts +145 -15
  92. package/src/openai-scenario.ts +46 -10
  93. package/src/openai-server.ts +65 -108
  94. package/src/openai-stub.ts +54 -30
  95. package/src/openai-twin.ts +760 -1665
  96. package/src/openai-types.ts +24 -6
  97. package/src/openai-webhooks.ts +2 -1
  98. package/src/screens/api-keys.tsx +138 -0
  99. package/src/screens/session.tsx +131 -0
  100. package/src/semantics/assistants.ts +336 -0
  101. package/src/semantics/audio.ts +31 -0
  102. package/src/semantics/batches.ts +88 -0
  103. package/src/semantics/chat-completions.ts +66 -0
  104. package/src/semantics/containers.ts +151 -0
  105. package/src/semantics/embeddings.ts +19 -0
  106. package/src/semantics/evals.ts +182 -0
  107. package/src/semantics/files.ts +67 -0
  108. package/src/semantics/fine-tuning.ts +185 -0
  109. package/src/semantics/images.ts +23 -0
  110. package/src/semantics/index.ts +52 -0
  111. package/src/semantics/models.ts +41 -0
  112. package/src/semantics/moderations.ts +14 -0
  113. package/src/semantics/organization.ts +76 -0
  114. package/src/semantics/progress.ts +72 -0
  115. package/src/semantics/responses.ts +151 -0
  116. package/src/semantics/shared.ts +82 -0
  117. package/src/semantics/uploads.ts +92 -0
  118. package/src/semantics/vector-stores.ts +279 -0
  119. package/test-fixtures/openai-openapi-operations.SOURCE.md +4 -5
  120. package/test-fixtures/openai-openapi-operations.json +224 -1334
@@ -0,0 +1,1117 @@
1
+ // OpenAI twin: the pack's behaviour and its in-process door. `handleOpenAITwinRequest` answers one
2
+ // request through the pack's fetch (openai-server.ts), which the derived dispatch owns: an operation
3
+ // OpenAI's spec publishes is a semantics handler (semantics/), the derived core (manifest.ts), or
4
+ // OpenAI's unknown-URL 404. What lives here is what those share:
5
+ // • the generative stubs: chat completions and responses (validation, the deterministic stub or a
6
+ // scripted scenario's turn, the streaming event sequences), embeddings (pseudo-vectors),
7
+ // moderations, images and audio (labeled placeholders); never a model's output;
8
+ // • the Responses API's atomic id allocation and background-poll completion;
9
+ // • the usage and costs reports over the recorded usage.
10
+ // No real OpenAI is called from this path.
11
+ import { applyTwinWriteAtomic } from '@volter/world-core';
12
+ import { buildLogprobs, countPromptTokens, estimateTokens, lastUserText, moderateText, pseudoEmbedding, stubAssistantText, stubJsonObject, stubToolCall, } from "./openai-stub.js";
13
+ import { DEFAULT_EFFORT, modelOf, reasons, unsupported } from "./openai-models.js";
14
+ import { scenarioTurn, scriptedChoice } from "./openai-scenario.js";
15
+ import { audioFormat, base64, imageFormat, partBytes, placeholderPng, pngSize, requestedSize, silentAac, silentFlac, silentMp3, silentOpus, silentPcm, silentWav, wavSeconds } from "./openai-media.js";
16
+ import { createOpenAITwinFetch } from "./openai-server.js";
17
+ const SERVICE = 'openai';
18
+ // ── vendor-shaped errors ──────────────────────────────────────────────────────────────
19
+ function errBody(type, message, code = null, param = null) {
20
+ return { error: { message, type, param, code } };
21
+ }
22
+ function invalidRequest(message, param = null, code = null) {
23
+ return { status: 400, body: errBody('invalid_request_error', message, code, param) };
24
+ }
25
+ function nowEpoch(occurredAt) {
26
+ return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
27
+ }
28
+ // ── usage ledger (real recorded usage → the usage/costs reporting endpoints) ──────────────
29
+ // Every billable inference call appends a usage record (model + token counts + a synthetic cost
30
+ // computed from a per-model price table). The /v1/organization/usage|costs endpoints aggregate
31
+ // these REAL recorded rows — nothing is hardcoded; with no traffic the report is genuinely empty.
32
+ // Per-model USD price per 1M tokens (a faithful slice of the published price sheet; deterministic).
33
+ const MODEL_PRICES = {
34
+ 'gpt-4o': { input: 2.5, output: 10 },
35
+ 'gpt-4o-mini': { input: 0.15, output: 0.6 },
36
+ 'gpt-4.1': { input: 2, output: 8 },
37
+ 'gpt-4.1-mini': { input: 0.4, output: 1.6 },
38
+ 'gpt-4-turbo': { input: 10, output: 30 },
39
+ 'o3': { input: 2, output: 8 },
40
+ 'o4-mini': { input: 1.1, output: 4.4 },
41
+ 'gpt-3.5-turbo': { input: 0.5, output: 1.5 },
42
+ 'text-embedding-3-small': { input: 0.02, output: 0 },
43
+ 'text-embedding-3-large': { input: 0.13, output: 0 },
44
+ 'text-embedding-ada-002': { input: 0.1, output: 0 },
45
+ };
46
+ function modelPrice(model) {
47
+ return MODEL_PRICES[model] ?? { input: 0, output: 0 };
48
+ }
49
+ export function usageCost(model, inputTokens, outputTokens) {
50
+ const p = modelPrice(model);
51
+ return (inputTokens / 1e6) * p.input + (outputTokens / 1e6) * p.output;
52
+ }
53
+ const WIDTHS = {
54
+ '1m': { seconds: 60, limit: 60, max: 1440 },
55
+ '1h': { seconds: 3600, limit: 24, max: 168 },
56
+ '1d': { seconds: 86400, limit: 7, max: 31 },
57
+ };
58
+ /** A report's query from its parameters; Costs buckets by day only and pages up to 180 of them. */
59
+ export function reportQuery(params, now, costs) {
60
+ const width = WIDTHS[String(params.bucket_width ?? '1d')] ?? WIDTHS['1d'];
61
+ const max = costs ? 180 : width.max;
62
+ const asked = Number(params.limit);
63
+ const limit = Number.isInteger(asked) && asked >= 1 ? Math.min(asked, max) : width.limit;
64
+ const start = Number(params.start_time);
65
+ const end = params.end_time !== undefined ? Number(params.end_time) : now;
66
+ const page = /^page_(\d+)$/.exec(String(params.page ?? ''));
67
+ const groups = params.group_by;
68
+ return { start, end, width: width.seconds, limit, from: page ? Number(page[1]) : start, groupBy: (Array.isArray(groups) ? groups : groups === undefined ? [] : [groups]).map(String) };
69
+ }
70
+ /** Buckets of one page of the query, each holding `results(rows in its window)`. */
71
+ export function bucketed(records, q, results) {
72
+ const data = [];
73
+ let at = q.from;
74
+ while (at < q.end && data.length < q.limit) {
75
+ const until = Math.min(at + q.width, q.end);
76
+ data.push({ object: 'bucket', start_time: at, end_time: until, results: results(records.filter((r) => Number(r.created_at) >= at && Number(r.created_at) < until), at, until) });
77
+ at = until;
78
+ }
79
+ return { object: 'page', data, has_more: at < q.end, next_page: at < q.end ? `page_${at}` : null };
80
+ }
81
+ /** Rows grouped by `key` when grouping by it, else all in one group keyed null; none when there are no rows. */
82
+ function grouped(rows, by) {
83
+ const out = new Map();
84
+ for (const r of rows) {
85
+ const k = by === null ? null : String(r[by]);
86
+ out.set(k, [...(out.get(k) ?? []), r]);
87
+ }
88
+ return [...out.entries()].sort((a, b) => (String(a[0]) < String(b[0]) ? -1 : 1));
89
+ }
90
+ /** What each usage kind's result counts, and the object it answers as (the spec's Usage*Result schemas;
91
+ * https://platform.openai.com/docs/api-reference/usage). A result names the grouping it answers (model, and for images
92
+ * their size and source) and null for the groupings not asked for. */
93
+ const USAGE_RESULTS = {
94
+ completions: {
95
+ object: 'organization.usage.completions.result',
96
+ // the twin caches nothing and hears, sees and speaks nothing: every token is uncached text
97
+ count: (rs) => {
98
+ const input = sum(rs, 'input_tokens');
99
+ const output = sum(rs, 'output_tokens');
100
+ return { input_tokens: input, input_cached_tokens: 0, input_cache_write_tokens: 0, input_uncached_tokens: input, output_tokens: output, input_text_tokens: input, output_text_tokens: output, input_cached_text_tokens: 0, input_audio_tokens: 0, input_cached_audio_tokens: 0, output_audio_tokens: 0, input_image_tokens: 0, input_cached_image_tokens: 0, output_image_tokens: 0, num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null, batch: null, service_tier: null };
101
+ },
102
+ },
103
+ embeddings: { object: 'organization.usage.embeddings.result', count: (rs) => ({ input_tokens: sum(rs, 'input_tokens'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
104
+ moderations: { object: 'organization.usage.moderations.result', count: (rs) => ({ input_tokens: sum(rs, 'input_tokens'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
105
+ audio_speeches: { object: 'organization.usage.audio_speeches.result', count: (rs) => ({ characters: sum(rs, 'characters'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
106
+ audio_transcriptions: { object: 'organization.usage.audio_transcriptions.result', count: (rs) => ({ seconds: sum(rs, 'seconds'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
107
+ images: { object: 'organization.usage.images.result', by: ['size', 'source'], count: (rs) => ({ images: sum(rs, 'images'), num_model_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null }) },
108
+ code_interpreter_sessions: { object: 'organization.usage.code_interpreter_sessions.result', count: (rs) => ({ num_sessions: sum(rs, 'num_sessions') }) },
109
+ file_search_calls: { object: 'organization.usage.file_searches.result', count: (rs) => ({ num_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null, vector_store_id: null }) },
110
+ web_search_calls: { object: 'organization.usage.web_searches.result', count: (rs) => ({ num_model_requests: sum(rs, 'num_model_requests'), num_requests: sum(rs, 'num_model_requests'), user_id: null, api_key_id: null, context_level: null }) },
111
+ };
112
+ const sum = (rows, k) => rows.reduce((n, r) => n + Number(r[k] ?? 0), 0);
113
+ /** The kinds whose results name a model. */
114
+ const MODELLESS = new Set(['code_interpreter_sessions', 'file_search_calls']);
115
+ /** The usage report over the recorded usage rows, for one kind ('completions', 'embeddings', …). */
116
+ export function usageReport(records, kind, q) {
117
+ const spec = USAGE_RESULTS[kind ?? 'completions'] ?? USAGE_RESULTS.completions;
118
+ const byModel = q.groupBy.includes('model') && !MODELLESS.has(kind ?? '');
119
+ const extra = (spec.by ?? []).filter((k) => q.groupBy.includes(k));
120
+ return bucketed(records.filter((r) => kind === undefined || r.kind === kind), q, (rows) => groupedBy(rows, [...(byModel ? ['model'] : []), ...extra]).map(([key, rs]) => ({
121
+ object: spec.object,
122
+ ...spec.count(rs),
123
+ project_id: null,
124
+ ...(MODELLESS.has(kind ?? '') ? {} : { model: byModel ? key.model : null }),
125
+ ...(spec.by ? Object.fromEntries(spec.by.map((k) => [k, extra.includes(k) ? key[k] : null])) : {}),
126
+ })));
127
+ }
128
+ /** Rows grouped by the values of `keys`, in key order; one group of all rows when no key; none when there are no rows. */
129
+ function groupedBy(rows, keys) {
130
+ const out = new Map();
131
+ for (const r of rows) {
132
+ const key = Object.fromEntries(keys.map((k) => [k, r[k] === undefined ? null : String(r[k])]));
133
+ const id = JSON.stringify(key);
134
+ out.set(id, [key, [...(out.get(id)?.[1] ?? []), r]]);
135
+ }
136
+ return [...out.entries()].sort((a, b) => (a[0] < b[0] ? -1 : 1)).map(([, v]) => v);
137
+ }
138
+ /** The costs report: the recorded usage priced per model, per bucket, by line item (usage kind) when grouped so. */
139
+ export function costsReport(records, q) {
140
+ const byLine = q.groupBy.includes('line_item');
141
+ return bucketed(records, q, (rows) => grouped(rows, byLine ? 'kind' : null).map(([line, rs]) => ({
142
+ object: 'organization.costs.result',
143
+ amount: { value: rs.reduce((n, r) => n + Number(r.cost_usd ?? 0), 0), currency: 'usd' },
144
+ line_item: line, project_id: null, api_key_id: null, quantity: null, quantity_unit: null,
145
+ })));
146
+ }
147
+ const SYSTEM_FINGERPRINT = 'fp_twin_stub';
148
+ /** The roles a chat message may have (the spec's ChatCompletionRequestMessage variants). */
149
+ const ROLES = new Set(['developer', 'system', 'user', 'assistant', 'tool', 'function']);
150
+ /** OpenAI's refusal of a chat message whose role is not one of the request message types
151
+ * (https://platform.openai.com/docs/api-reference/chat/create, `messages`). */
152
+ function invalidRole() {
153
+ return { error: invalidRequest("each message must have a valid 'role'", 'messages') };
154
+ }
155
+ // The chat options below are OpenAI's (https://platform.openai.com/docs/api-reference/chat/create); each
156
+ // parser answers the option's value or OpenAI's refusal of it.
157
+ /** `n`: how many choices to generate, at least one. */
158
+ function chatN(raw) {
159
+ const n = Number(raw);
160
+ return Number.isInteger(n) && n >= 1 ? n : { error: invalidRequest("'n' must be an integer >= 1", 'n') };
161
+ }
162
+ /** `max_completion_tokens` (or the legacy `max_tokens`): a cap of at least one token. */
163
+ function chatMaxTokens(raw) {
164
+ const max = Number(raw);
165
+ return Number.isInteger(max) && max >= 1 ? max : { error: invalidRequest("'max_tokens' must be an integer >= 1", 'max_tokens') };
166
+ }
167
+ /** `stop`: one sequence or a list of them. */
168
+ function chatStop(raw) {
169
+ if (typeof raw === 'string')
170
+ return [raw];
171
+ return Array.isArray(raw) ? raw : { error: invalidRequest("'stop' must be a string or array of strings", 'stop') };
172
+ }
173
+ /** `tool_choice`: 'auto' | 'none' | 'required' | { type:'function', function:{ name } }. */
174
+ function chatToolChoice(raw) {
175
+ if (raw === 'auto' || raw === 'none' || raw === 'required')
176
+ return raw;
177
+ if (!raw || typeof raw !== 'object')
178
+ return { error: invalidRequest("'tool_choice' must be 'auto'/'none'/'required' or a named function", 'tool_choice') };
179
+ const fn = raw.function;
180
+ return typeof fn?.name === 'string' ? { name: fn.name } : { error: invalidRequest("invalid 'tool_choice' — named choice requires function.name", 'tool_choice') };
181
+ }
182
+ /** OpenAI's refusal of a `response_format` whose type is none of text, json_object and json_schema. */
183
+ function badResponseFormat() {
184
+ return { error: invalidRequest("'response_format.type' must be 'text', 'json_object', or 'json_schema'", 'response_format') };
185
+ }
186
+ /** `top_logprobs`: 0 to 20 alternatives per token, only with `logprobs: true`. */
187
+ function chatTopLogprobs(raw, logprobs) {
188
+ const top = Number(raw);
189
+ if (!Number.isInteger(top) || top < 0 || top > 20)
190
+ return { error: invalidRequest("'top_logprobs' must be an integer between 0 and 20", 'top_logprobs') };
191
+ return logprobs ? top : { error: invalidRequest("'top_logprobs' requires 'logprobs' to be true", 'top_logprobs') };
192
+ }
193
+ /** `seed`: an integer for best-effort reproducible sampling. */
194
+ function chatSeed(raw) {
195
+ const seed = Number(raw);
196
+ return Number.isInteger(seed) ? seed : { error: invalidRequest("'seed' must be an integer", 'seed') };
197
+ }
198
+ /** `logit_bias`: token ids mapped to a bias from -100 to 100. */
199
+ function chatLogitBias(raw) {
200
+ if (!raw || typeof raw !== 'object' || Array.isArray(raw))
201
+ return { error: invalidRequest("'logit_bias' must be an object mapping token ids to bias values", 'logit_bias') };
202
+ const bias = {};
203
+ for (const [k, v] of Object.entries(raw)) {
204
+ const num = Number(v);
205
+ if (!Number.isFinite(num) || num < -100 || num > 100)
206
+ return { error: invalidRequest("each 'logit_bias' value must be a number between -100 and 100", 'logit_bias') };
207
+ bias[k] = num;
208
+ }
209
+ return bias;
210
+ }
211
+ /** `prediction`: predicted output, `{ type: 'content', content }`, its content a string or text parts. */
212
+ function chatPrediction(raw) {
213
+ const p = raw;
214
+ if (!p || typeof p !== 'object' || p.type !== 'content' || p.content === undefined)
215
+ return { error: invalidRequest("'prediction' must be an object with type 'content' and a content field", 'prediction') };
216
+ if (typeof p.content === 'string')
217
+ return p.content;
218
+ return Array.isArray(p.content) ? p.content.map((c) => (c && typeof c === 'object' ? String(c.text ?? '') : String(c))).join('') : '';
219
+ }
220
+ /** `modalities` with `audio`: a spoken answer needs `audio: { voice, format }` (the vendor's rule). The twin
221
+ * can't synthesize speech, so the returned bytes are a labeled stub; the envelope is faithful. */
222
+ function chatAudio(modalities, audio) {
223
+ if (!Array.isArray(modalities))
224
+ return { error: invalidRequest("'modalities' must be an array", 'modalities') };
225
+ if (!modalities.includes('audio'))
226
+ return undefined;
227
+ const a = audio;
228
+ if (!a || typeof a !== 'object' || typeof a.voice !== 'string' || typeof a.format !== 'string')
229
+ return { error: invalidRequest("'audio' with a 'voice' and 'format' is required when 'modalities' includes 'audio'", 'audio') };
230
+ return { voice: a.voice, format: a.format };
231
+ }
232
+ export function validateChat(params) {
233
+ if (params.model === undefined || params.model === '')
234
+ return { error: invalidRequest("you must provide a model parameter", 'model') };
235
+ if (typeof params.model !== 'string')
236
+ return { error: invalidRequest("'model' must be a string", 'model') };
237
+ if (!Array.isArray(params.messages))
238
+ return { error: invalidRequest("you must provide a messages parameter", 'messages') };
239
+ if (params.messages.length === 0)
240
+ return { error: invalidRequest("[] is too short - 'messages'", 'messages') };
241
+ const messages = params.messages;
242
+ if (messages.some((m) => !m || typeof m !== 'object' || !ROLES.has(String(m.role))))
243
+ return invalidRole();
244
+ // Each option the request may carry is parsed by its own function (below): a function returns the
245
+ // option's value, or OpenAI's refusal of it as `{ error }`.
246
+ const n = params.n === undefined ? 1 : chatN(params.n);
247
+ if (typeof n !== 'number')
248
+ return n;
249
+ // max_completion_tokens is the current name; max_tokens is the legacy alias.
250
+ const maxRaw = params.max_completion_tokens ?? params.max_tokens;
251
+ const maxTokens = maxRaw === undefined ? undefined : chatMaxTokens(maxRaw);
252
+ if (typeof maxTokens === 'object')
253
+ return maxTokens;
254
+ const stop = params.stop === undefined ? undefined : chatStop(params.stop);
255
+ if (stop && !Array.isArray(stop))
256
+ return stop;
257
+ const toolChoice = params.tool_choice === undefined ? undefined : chatToolChoice(params.tool_choice);
258
+ if (toolChoice && typeof toolChoice === 'object' && 'error' in toolChoice)
259
+ return toolChoice;
260
+ // response_format: { type:'text' | 'json_object' | 'json_schema', json_schema? }.
261
+ let responseFormat = { kind: 'text' };
262
+ const rf = params.response_format;
263
+ if (rf !== undefined) {
264
+ if (!rf || typeof rf !== 'object')
265
+ return { error: invalidRequest("'response_format' must be an object", 'response_format') };
266
+ const t = rf.type ?? 'text';
267
+ const format = t === 'json_schema' ? { kind: 'json_schema', schema: rf.json_schema } : t === 'json_object' || t === 'text' ? { kind: t } : undefined;
268
+ if (!format)
269
+ return badResponseFormat();
270
+ responseFormat = format;
271
+ }
272
+ // stream_options.include_usage → emit a final usage-only chunk in the stream.
273
+ const so = params.stream_options;
274
+ const includeUsage = !!(so && typeof so === 'object' && so.include_usage === true);
275
+ // logprobs (boolean) + top_logprobs (0..20, requires logprobs:true) → per-token logprob detail.
276
+ const logprobs = params.logprobs === true;
277
+ const topLogprobs = params.top_logprobs === undefined ? undefined : chatTopLogprobs(params.top_logprobs, logprobs);
278
+ if (typeof topLogprobs === 'object')
279
+ return topLogprobs;
280
+ // seed → reproducible sampling (the twin is already deterministic; we echo it via fingerprint).
281
+ const seed = params.seed === undefined ? undefined : chatSeed(params.seed);
282
+ if (typeof seed === 'object')
283
+ return seed;
284
+ // logit_bias → a map of token-id → bias in [-100, 100].
285
+ const logitBias = params.logit_bias === undefined ? undefined : chatLogitBias(params.logit_bias);
286
+ if (logitBias && 'error' in logitBias)
287
+ return logitBias;
288
+ // prediction → predicted outputs ({ type:'content', content }); content may be a string or parts.
289
+ const prediction = params.prediction === undefined ? undefined : chatPrediction(params.prediction);
290
+ if (typeof prediction === 'object')
291
+ return prediction;
292
+ // modalities + audio → audio output.
293
+ const audioOutput = params.modalities === undefined ? undefined : chatAudio(params.modalities, params.audio);
294
+ if (audioOutput && 'error' in audioOutput)
295
+ return audioOutput;
296
+ // store + metadata → stored completions (retrievable later); metadata must be a flat object.
297
+ // a model refuses what its page says it does not take (openai-models.ts); a reasoning model reasons with the effort
298
+ // asked for, else the spec's default
299
+ const refused = unsupported(params.model, { effort: params.reasoning_effort, effortParam: 'reasoning_effort', temperature: params.temperature, top_p: params.top_p, logprobs: params.logprobs, chatTools: params.tools ?? params.functions });
300
+ if (refused)
301
+ return { error: invalidRequest(refused.message, refused.param, 'unsupported_value') };
302
+ const reasoningEffort = typeof params.reasoning_effort === 'string' ? params.reasoning_effort : reasons(params.model) ? DEFAULT_EFFORT : undefined;
303
+ const store = params.store === true;
304
+ let metadata;
305
+ if (params.metadata !== undefined) {
306
+ if (!params.metadata || typeof params.metadata !== 'object' || Array.isArray(params.metadata))
307
+ return { error: invalidRequest("'metadata' must be an object", 'metadata') };
308
+ metadata = params.metadata;
309
+ }
310
+ return {
311
+ args: {
312
+ model: params.model,
313
+ messages,
314
+ tools: params.tools,
315
+ functions: params.functions,
316
+ n,
317
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
318
+ ...(stop !== undefined ? { stop } : {}),
319
+ stream: params.stream === true,
320
+ ...(toolChoice !== undefined ? { toolChoice } : {}),
321
+ parallelToolCalls: params.parallel_tool_calls !== false,
322
+ responseFormat,
323
+ includeUsage,
324
+ logprobs,
325
+ ...(topLogprobs !== undefined ? { topLogprobs } : {}),
326
+ ...(seed !== undefined ? { seed } : {}),
327
+ ...(logitBias !== undefined ? { logitBias: logitBias } : {}),
328
+ ...(prediction !== undefined ? { prediction } : {}),
329
+ ...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
330
+ store,
331
+ ...(metadata !== undefined ? { metadata } : {}),
332
+ ...(audioOutput !== undefined ? { audioOutput: audioOutput } : {}),
333
+ },
334
+ };
335
+ }
336
+ /** The text cut at the EARLIEST-occurring stop sequence across the whole `stop` list, not whichever is
337
+ * listed first: the real vendor's "stop generation at the first hit", regardless of array order. */
338
+ function stoppedAt(text, stops) {
339
+ let stopAt = -1;
340
+ for (const s of stops) {
341
+ if (!s)
342
+ continue;
343
+ const i = text.indexOf(s);
344
+ if (i >= 0 && (stopAt < 0 || i < stopAt))
345
+ stopAt = i;
346
+ }
347
+ return stopAt >= 0 ? text.slice(0, stopAt) : text;
348
+ }
349
+ /** The text cut to a `max_completion_tokens` cap (about four characters a token); the choice finishes `length`. */
350
+ function cutAt(text, maxTokens) {
351
+ return text.slice(0, maxTokens * 4);
352
+ }
353
+ /** modalities:['audio'] → the assistant replies with an audio object (content is null, the text lives in
354
+ * audio.transcript). The twin can't synthesize speech, so `data` is a labeled-stub base64 string; the shape
355
+ * (id/data/transcript/expires_at) is vendor-faithful. */
356
+ function audioChoice(args, idx, transcript, logprobs, finish) {
357
+ const voice = args.audioOutput;
358
+ const stubBytes = `[twin-stub:${args.model}] no real audio synthesis — voice=${voice.voice} format=${voice.format}; transcript: ${transcript}`;
359
+ const audio = {
360
+ id: `audio-twin-${stableSuffix(args)}-${idx}`,
361
+ data: Buffer.from(stubBytes, 'utf8').toString('base64'),
362
+ transcript,
363
+ expires_at: nowEpoch() + 3600,
364
+ };
365
+ return {
366
+ choice: { index: idx, message: { role: 'assistant', content: null, refusal: null, audio }, logprobs, finish_reason: finish },
367
+ completionTokens: estimateTokens(transcript),
368
+ };
369
+ }
370
+ // Build ONE deterministic stub choice (index `idx`). Honors tools/functions (tool_calls +
371
+ // finish_reason tool_calls), tool_choice (none/required/named), parallel_tool_calls,
372
+ // response_format (json_object/json_schema), max_tokens, and stop sequences.
373
+ function buildChoice(args, idx) {
374
+ const toolsSource = args.tools ?? args.functions;
375
+ const hasTools = Array.isArray(toolsSource) && toolsSource.length > 0;
376
+ // tool_choice gates whether the stub calls a tool: 'none' forbids it; a named/required choice
377
+ // forces it (even when the heuristic otherwise would not); 'auto'/default calls when tools exist.
378
+ const forbidTools = args.toolChoice === 'none';
379
+ const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
380
+ // a model answers a tool's result in words unless the caller insists on another call; one that always
381
+ // called a tool would never let an app's tool loop end
382
+ const answeringTool = ['tool', 'function'].includes(String(args.messages.at(-1)?.role));
383
+ const wantTool = hasTools && !forbidTools && (!answeringTool || forcedName !== undefined || args.toolChoice === 'required');
384
+ if (wantTool) {
385
+ // parallel_tool_calls (default true) → the stub may emit one call per provided tool; a named
386
+ // choice or parallel:false collapses to a single call.
387
+ // (the tools are there, so each call is made)
388
+ const calls = forcedName || args.parallelToolCalls === false
389
+ ? [stubToolCall(toolsSource, idx + 1, forcedName, lastUserText(args.messages))]
390
+ : toolsSource.map((tool, t) => stubToolCall([tool], idx * 100 + t + 1, undefined, lastUserText(args.messages)));
391
+ return {
392
+ choice: { index: idx, message: { role: 'assistant', content: null, tool_calls: calls, refusal: null }, logprobs: null, finish_reason: 'tool_calls' },
393
+ completionTokens: estimateTokens(JSON.stringify(calls)),
394
+ };
395
+ }
396
+ // json_mode: when response_format requests json_object/json_schema, the content is valid JSON.
397
+ // prediction (predicted outputs): a real model uses the prediction to speed decoding but still
398
+ // returns its own generation — the twin echoes the predicted content (clearly still a stub) so
399
+ // the prediction round-trips, then reports accepted/rejected prediction tokens in usage.
400
+ let text = args.prediction !== undefined
401
+ ? `[twin-stub:${args.model}] predicted-output echo: ${args.prediction}`
402
+ : args.responseFormat.kind === 'json_object'
403
+ ? stubJsonObject(args.messages, args.model)
404
+ : args.responseFormat.kind === 'json_schema'
405
+ ? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
406
+ : stubAssistantText(args.messages, args.model);
407
+ if (args.stop)
408
+ text = stoppedAt(text, args.stop);
409
+ const capped = args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens;
410
+ if (capped)
411
+ text = cutAt(text, args.maxTokens);
412
+ const finish = capped ? 'length' : 'stop';
413
+ const logprobs = args.logprobs ? buildLogprobs(text, args.topLogprobs ?? 0) : null;
414
+ if (args.audioOutput)
415
+ return audioChoice(args, idx, text, logprobs, finish);
416
+ return {
417
+ // a text answer carries its annotations (none: the twin cites no web source), as OpenAI's does
418
+ // (https://platform.openai.com/docs/api-reference/chat/object, `choices[].message.annotations`)
419
+ choice: { index: idx, message: { role: 'assistant', content: text, refusal: null, annotations: [] }, logprobs, finish_reason: finish },
420
+ completionTokens: estimateTokens(text),
421
+ };
422
+ }
423
+ /** Predicted-output accounting: the twin echoes the prediction, so every predicted token is "accepted". */
424
+ function predictionUsage(prediction) {
425
+ return { accepted_prediction_tokens: estimateTokens(prediction), rejected_prediction_tokens: 0 };
426
+ }
427
+ export function buildChatCompletion(args, occurredAt, decision) {
428
+ const promptTokens = countPromptTokens(args.messages);
429
+ // Scenario handlers: a fired handler scripts the assistant turn; a miss teaches in the stub. The
430
+ // decision was made (and any fault honored) by the request handler through the engine's serve(); this
431
+ // builder only realizes it — a status fault never reaches here.
432
+ const turn = decision && scenarioTurn(decision);
433
+ const choices = [];
434
+ let completionTokens = 0;
435
+ for (let i = 0; i < args.n; i++) {
436
+ const { choice, completionTokens: ct } = turn?.scripted ? scriptedChoice(turn.scripted, i) : buildChoice(args, i);
437
+ if (turn?.missTeach && typeof choice.message.content === 'string')
438
+ choice.message.content += turn.missTeach;
439
+ choices.push(choice);
440
+ completionTokens += ct;
441
+ }
442
+ // usage breaks its counts down as OpenAI's does (https://platform.openai.com/docs/api-reference/chat/object,
443
+ // `usage.prompt_tokens_details`, `usage.completion_tokens_details`): the twin caches nothing and reasons and
444
+ // speaks nothing, so those are zero
445
+ const usage = {
446
+ prompt_tokens: promptTokens, completion_tokens: completionTokens, total_tokens: promptTokens + completionTokens,
447
+ prompt_tokens_details: { cached_tokens: 0, audio_tokens: 0 },
448
+ completion_tokens_details: { reasoning_tokens: 0, audio_tokens: 0, accepted_prediction_tokens: 0, rejected_prediction_tokens: 0 },
449
+ };
450
+ // prediction (predicted outputs): real responses report how many predicted tokens were accepted
451
+ // vs rejected. The twin echoes the prediction, so all predicted tokens are "accepted".
452
+ if (args.prediction !== undefined)
453
+ usage.completion_tokens_details = { ...usage.completion_tokens_details, ...predictionUsage(args.prediction) };
454
+ // a reasoning model's reasoning tokens are billed as output tokens (https://developers.openai.com/api/docs/guides/reasoning:
455
+ // "they still occupy space in the model's context window and are billed as output tokens")
456
+ if (args.reasoningEffort !== undefined && reasons(args.model)) {
457
+ const spent = REASONING_BUDGET[args.reasoningEffort] ?? 0;
458
+ usage.completion_tokens += spent;
459
+ usage.total_tokens += spent;
460
+ usage.completion_tokens_details = { ...usage.completion_tokens_details, reasoning_tokens: spent };
461
+ }
462
+ return {
463
+ id: `chatcmpl-twin-${stableSuffix(args)}`,
464
+ object: 'chat.completion',
465
+ created: nowEpoch(occurredAt),
466
+ model: args.model,
467
+ choices,
468
+ usage,
469
+ system_fingerprint: SYSTEM_FINGERPRINT,
470
+ // the tier that served the request: the standard one, which is what `auto` and `default` pick for a project
471
+ // not on Scale Tier (https://platform.openai.com/docs/api-reference/chat/object, `service_tier`)
472
+ service_tier: 'default',
473
+ ...(args.metadata !== undefined ? { metadata: args.metadata } : {}),
474
+ };
475
+ }
476
+ // A deterministic id suffix from the request (so ids are stable + assertable, like the twin's
477
+ // other deterministic outputs). Hash of the prompt text + model + seed (seed changes sampling,
478
+ // so it changes the response id the way a real seed-distinct request does).
479
+ function stableSuffix(args) {
480
+ let h = 0x811c9dc5;
481
+ const s = JSON.stringify(args.messages) + args.model + (args.seed !== undefined ? `|seed=${args.seed}` : '');
482
+ for (let i = 0; i < s.length; i++) {
483
+ h ^= s.charCodeAt(i);
484
+ h = Math.imul(h, 0x01000193);
485
+ }
486
+ return (h >>> 0).toString(36);
487
+ }
488
+ // Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order.
489
+ function chunkText(text) {
490
+ if (!text)
491
+ return [];
492
+ const out = [];
493
+ for (let i = 0; i < text.length; i += 20)
494
+ out.push(text.slice(i, i + 20));
495
+ return out;
496
+ }
497
+ /** A streamed choice's tool calls: one tool_calls delta-pair per call, each carrying its own `index`
498
+ * (parallel tool calls). */
499
+ function streamToolCalls(calls, idx, base, sink) {
500
+ calls.forEach((tc, tIdx) => {
501
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: '' } }] }, logprobs: null, finish_reason: null }] } });
502
+ sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, function: { arguments: tc.function.arguments } }] }, logprobs: null, finish_reason: null }] } });
503
+ });
504
+ }
505
+ /**
506
+ * Emit the vendor-faithful Chat Completions streaming sequence into the injected sink (NO
507
+ * sockets, NO setTimeout). Real order: a first chunk with `delta:{role:'assistant'}`, then
508
+ * `delta:{content}` chunks (or tool_calls deltas), then a final chunk with `finish_reason`,
509
+ * then `[DONE]`. Deterministic + synchronous so a collector can assert the full sequence.
510
+ */
511
+ export function streamChat(args, sink, occurredAt, decision) {
512
+ const full = buildChatCompletion(args, occurredAt, decision);
513
+ const base = { id: full.id, object: 'chat.completion.chunk', created: full.created, model: full.model, system_fingerprint: SYSTEM_FINGERPRINT };
514
+ for (const choice of full.choices) {
515
+ const idx = choice.index;
516
+ // role chunk
517
+ sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, logprobs: null, finish_reason: null }] } });
518
+ if (choice.message.tool_calls && choice.message.tool_calls.length)
519
+ streamToolCalls(choice.message.tool_calls, idx, base, sink);
520
+ else {
521
+ for (const piece of chunkText(choice.message.content ?? '')) {
522
+ sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, logprobs: null, finish_reason: null }] } });
523
+ }
524
+ }
525
+ sink({ data: { ...base, choices: [{ index: idx, delta: {}, logprobs: null, finish_reason: choice.finish_reason }] } });
526
+ }
527
+ // stream_options.include_usage → a final chunk with an empty choices array carrying `usage`.
528
+ if (args.includeUsage) {
529
+ sink({ data: { ...base, choices: [], usage: full.usage } });
530
+ }
531
+ sink({ done: true });
532
+ return full;
533
+ }
534
+ export function responseMessages(items, instructions) {
535
+ const messages = instructions ? [{ role: 'system', content: instructions }] : [];
536
+ const partsText = (content) => typeof content === 'string' ? content : Array.isArray(content) ? content.map((c) => c && typeof c === 'object' ? String(c.text ?? JSON.stringify(c)) : String(c)).join('\n') : '';
537
+ for (const item of items) {
538
+ if (item.type === 'function_call') {
539
+ messages.push({ role: 'assistant', content: '', tool_calls: [{ id: String(item.call_id ?? item.id ?? ''), type: 'function', function: { name: String(item.name ?? ''), arguments: typeof item.arguments === 'string' ? item.arguments : JSON.stringify(item.arguments ?? {}) } }] });
540
+ }
541
+ else if (item.type === 'function_call_output') {
542
+ messages.push({ role: 'tool', content: partsText(item.output), tool_call_id: String(item.call_id ?? '') });
543
+ }
544
+ else if (item.type === 'message' || item.type === undefined) {
545
+ messages.push({ role: (item.role ?? 'user'), content: partsText(item.content) });
546
+ }
547
+ // Reasoning and other typed items remain in state, without inventing user turns.
548
+ }
549
+ return messages;
550
+ }
551
+ /** OpenAI's refusal of an `input` list holding something other than input items (objects)
552
+ * (https://platform.openai.com/docs/api-reference/responses/create, `input`). */
553
+ function itemsNotObjects() {
554
+ return { error: invalidRequest("'input' items must be objects", 'input') };
555
+ }
556
+ export function validateResponses(params) {
557
+ if (params.model === undefined || params.model === '')
558
+ return { error: invalidRequest("you must provide a model parameter", 'model') };
559
+ if (typeof params.model !== 'string')
560
+ return { error: invalidRequest("'model' must be a string", 'model') };
561
+ if (params.input === undefined)
562
+ return { error: invalidRequest("you must provide an input parameter", 'input') };
563
+ // Keep vendor items as state; the messages view is only a projection for the scenario.
564
+ let inputItems;
565
+ if (typeof params.input === 'string') {
566
+ inputItems = [{ type: 'message', role: 'user', content: [{ type: 'input_text', text: params.input }] }];
567
+ }
568
+ else if (Array.isArray(params.input)) {
569
+ if (params.input.some((item) => !item || typeof item !== 'object' || Array.isArray(item)))
570
+ return itemsNotObjects();
571
+ inputItems = params.input.map((item) => {
572
+ if (item.type !== undefined && item.type !== 'message')
573
+ return { ...item };
574
+ return { ...item, type: 'message', role: item.role ?? 'user', content: typeof item.content === 'string' ? [{ type: item.role === 'assistant' ? 'output_text' : 'input_text', text: item.content }] : item.content };
575
+ });
576
+ }
577
+ else {
578
+ return { error: invalidRequest("'input' must be a string or an array of input items", 'input') };
579
+ }
580
+ const instructions = typeof params.instructions === 'string' ? params.instructions : undefined;
581
+ const messages = responseMessages(inputItems, instructions);
582
+ const inputText = responseMessages(inputItems).map((m) => m.content).filter(Boolean).join('\n');
583
+ if (params.tools !== undefined && !Array.isArray(params.tools))
584
+ return { error: invalidRequest("'tools' must be an array", 'tools') };
585
+ const maxOut = params.max_output_tokens;
586
+ if (maxOut !== undefined && (typeof maxOut !== 'number' || !Number.isInteger(maxOut) || maxOut < 1))
587
+ return { error: invalidRequest("'max_output_tokens' must be a positive integer", 'max_output_tokens') };
588
+ const prev = params.previous_response_id;
589
+ if (prev !== undefined && (typeof prev !== 'string' || !prev))
590
+ return { error: invalidRequest("'previous_response_id' must be a string", 'previous_response_id') };
591
+ // reasoning.effort → the model spends a (stubbed) reasoning budget; the item shape is faithful.
592
+ let reasoningEffort;
593
+ if (params.reasoning !== undefined) {
594
+ const r = params.reasoning;
595
+ if (!r || typeof r !== 'object' || Array.isArray(r))
596
+ return { error: invalidRequest("'reasoning' must be an object", 'reasoning') };
597
+ const effort = r.effort;
598
+ if (effort !== undefined && effort !== null)
599
+ reasoningEffort = effort;
600
+ }
601
+ // a model refuses what its page says it does not take (openai-models.ts); a reasoning model reasons with the effort
602
+ // asked for, else the spec's default
603
+ const refused = unsupported(params.model, { effort: reasoningEffort, effortParam: 'reasoning.effort', temperature: params.temperature, top_p: params.top_p });
604
+ if (refused)
605
+ return { error: invalidRequest(refused.message, refused.param, 'unsupported_value') };
606
+ if (reasoningEffort === undefined && reasons(params.model))
607
+ reasoningEffort = DEFAULT_EFFORT;
608
+ const reasoningSummary = params.reasoning && typeof params.reasoning.summary === 'string' ? String(params.reasoning.summary) : undefined;
609
+ return {
610
+ args: {
611
+ model: params.model,
612
+ inputItems,
613
+ instructions,
614
+ inputText,
615
+ messages,
616
+ stream: params.stream === true,
617
+ store: params.store !== false, // OpenAI defaults store=true (stored & retrievable)
618
+ background: params.background === true,
619
+ ...(typeof prev === 'string' ? { previousResponseId: prev } : {}),
620
+ ...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
621
+ ...(Array.isArray(params.tools) ? { tools: params.tools } : {}),
622
+ ...(typeof maxOut === 'number' ? { maxTokens: maxOut } : {}),
623
+ ...(reasoningSummary !== undefined ? { reasoningSummary } : {}),
624
+ // OpenAI's defaults for what was not sent (the Response schema's): temperature and top_p 1, tools in parallel, auto;
625
+ // plain text out, no truncation, no end user, no tool-call cap, no alternative tokens, the default tier
626
+ // (https://platform.openai.com/docs/api-reference/responses/object)
627
+ echo: {
628
+ metadata: params.metadata ?? {}, temperature: params.temperature ?? 1, top_p: params.top_p ?? 1, parallel_tool_calls: params.parallel_tool_calls ?? true, tool_choice: params.tool_choice ?? 'auto',
629
+ text: params.text ?? { format: { type: 'text' } }, truncation: params.truncation ?? 'disabled', user: params.user ?? null, max_tool_calls: params.max_tool_calls ?? null,
630
+ top_logprobs: params.top_logprobs ?? 0, service_tier: ['flex', 'priority', 'scale'].includes(String(params.service_tier)) ? params.service_tier : 'default',
631
+ },
632
+ },
633
+ };
634
+ }
635
+ /** A tool as a response answers it: as sent, with what OpenAI fills in for what was not. A web search preview searches
636
+ * with `medium` context from the United States unless told otherwise (the spec's WebSearchPreviewTool: "`medium` is
637
+ * the default"; `user_location` "If omitted or null, defaults to the United States"); a function tool sent without
638
+ * `strict` answers strict, as the reference's Functions example answers the tool it sends; a file search tool, its
639
+ * default filter and ranking
640
+ * (https://platform.openai.com/docs/api-reference/responses/create). */
641
+ const TOOL_DEFAULTS = {
642
+ function: (t) => ({ ...t, strict: t.strict ?? true }),
643
+ // a file search sent without filters or ranking answers no filter and the automatic ranker with no threshold, as the
644
+ // reference's File search example answers the tool it sends
645
+ file_search: (t) => ({ ...t, filters: t.filters ?? null, ranking_options: t.ranking_options ?? { ranker: 'auto', score_threshold: 0 } }),
646
+ // and the domains it is limited to, none unless sent, as the reference's Web search example answers `"domains": []`
647
+ web_search_preview: (t) => ({ ...t, domains: t.domains ?? [], search_context_size: t.search_context_size ?? 'medium', user_location: t.user_location ?? { type: 'approximate', city: null, country: 'US', region: null, timezone: null } }),
648
+ };
649
+ TOOL_DEFAULTS.web_search_preview_2025_03_11 = TOOL_DEFAULTS.web_search_preview;
650
+ function answeredTool(tool) {
651
+ const t = (tool && typeof tool === 'object' ? tool : {});
652
+ return TOOL_DEFAULTS[String(t.type)]?.(t) ?? tool;
653
+ }
654
+ /** The built-in tool calls a response makes: a web search for a web search tool, a file search over the named stores for a
655
+ * file search tool, each querying the input's text (its results are not included unless asked for: `results` null). */
656
+ function builtInCalls(args, suffix) {
657
+ const query = args.inputText.slice(0, 200);
658
+ const out = [];
659
+ for (const [i, tool] of (Array.isArray(args.tools) ? args.tools : []).entries()) {
660
+ const type = String(tool?.type ?? '');
661
+ if (/^web_search/.test(type))
662
+ out.push({ type: 'web_search_call', id: `ws-twin-${suffix}-${i}`, status: 'completed', action: { type: 'search', query } });
663
+ if (type === 'file_search')
664
+ out.push({ type: 'file_search_call', id: `fs-twin-${suffix}-${i}`, status: 'completed', queries: [query], results: null });
665
+ }
666
+ return out;
667
+ }
668
+ /** What every response answers besides its output: the request's settings as sent (or OpenAI's defaults), no error,
669
+ * no program access (https://platform.openai.com/docs/api-reference/responses/object). */
670
+ const responseSettings = (args) => ({
671
+ instructions: args.instructions ?? null, tools: Array.isArray(args.tools) ? args.tools.map(answeredTool) : [], ...args.echo,
672
+ access_programs: null, error: null, incomplete_details: null,
673
+ // whether it is kept: the spec's own Response example answers `"store": true` (spec/patches.json)
674
+ background: args.background, store: args.store, max_output_tokens: args.maxTokens ?? null,
675
+ previous_response_id: args.previousResponseId ?? null,
676
+ reasoning: { effort: args.reasoningEffort ?? null, summary: args.reasoningSummary ?? null },
677
+ });
678
+ // A deterministic reasoning-token budget per effort level (more effort → more reasoning tokens).
679
+ const REASONING_BUDGET = { none: 0, minimal: 8, low: 16, medium: 48, high: 128, xhigh: 256, max: 512 };
680
+ /** The function call an unscripted response makes: when the caller requires one or names one, or on
681
+ * `auto` unless the input ends with a tool's result, which a model answers in words. */
682
+ function responseToolCall(args) {
683
+ const choice = args.echo.tool_choice;
684
+ const named = choice && typeof choice === 'object' ? String(choice.name ?? '') : undefined;
685
+ const answering = args.inputItems.at(-1)?.type === 'function_call_output';
686
+ const functions = (Array.isArray(args.tools) ? args.tools : []).filter((t) => t.type === 'function');
687
+ if (!functions.length || choice === 'none' || (answering && !named && choice !== 'required'))
688
+ return null;
689
+ return stubToolCall(functions, 1, named, lastUserText(args.messages));
690
+ }
691
+ export function buildResponse(args, occurredAt, idSuffix, decision) {
692
+ // The scenario decides the turn exactly as it does for chat completions: a fired handler scripts the
693
+ // text and the tool calls; a miss answers the labeled stub and teaches. The request handler has
694
+ // already honored a fault; only a content decision reaches here.
695
+ const suffix = idSuffix ?? String(nowEpoch(occurredAt));
696
+ const turn = decision && scenarioTurn(decision, `call-twin-${suffix}`);
697
+ const scripted = turn?.scripted;
698
+ let text = scripted ? (scripted.text ?? '') : stubAssistantText(args.messages, args.model) + (turn?.missTeach ?? '');
699
+ const inputTokens = countPromptTokens(args.messages);
700
+ const messageItem = {
701
+ type: 'message',
702
+ id: `msg-twin-${suffix}`,
703
+ status: 'completed',
704
+ role: 'assistant',
705
+ // no log probabilities unless asked for (`include: ["message.output_text.logprobs"]`): an empty list
706
+ content: [{ type: 'output_text', text, annotations: [], logprobs: [] }],
707
+ };
708
+ // unscripted, the stub calls a function tool as the chat stub does (buildChoice)
709
+ const stubbed = scripted ? null : responseToolCall(args);
710
+ if (stubbed)
711
+ text = '';
712
+ const calls = (scripted?.toolCalls ?? (stubbed ? [stubbed] : [])).map((tc, i) => ({ type: 'function_call', id: `fc-twin-${suffix}-${i}`, status: 'completed', call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments }));
713
+ const messageTokens = estimateTokens(text) + estimateTokens(calls.map((c) => c.arguments).join(''));
714
+ const output = [];
715
+ // usage breaks its counts down as OpenAI's does (https://platform.openai.com/docs/api-reference/responses/object, `usage`):
716
+ // the twin caches nothing, so no input token was read from or written to a cache
717
+ const usage = { input_tokens: inputTokens, input_tokens_details: { cached_tokens: 0, cache_write_tokens: 0 }, output_tokens: messageTokens, output_tokens_details: { reasoning_tokens: 0 }, total_tokens: inputTokens + messageTokens };
718
+ // reasoning.effort → a faithful reasoning item (labeled-stub summary) BEFORE the message item,
719
+ // plus output_tokens_details.reasoning_tokens in usage (counted into output_tokens, like the vendor).
720
+ if (args.reasoningEffort) {
721
+ const reasoningTokens = REASONING_BUDGET[args.reasoningEffort] ?? 16;
722
+ // a summary only when one was asked for (`reasoning.summary`); without, OpenAI answers the item's summary empty
723
+ // (https://platform.openai.com/docs/guides/reasoning#reasoning-summaries)
724
+ const reasoningItem = {
725
+ type: 'reasoning',
726
+ id: `rs-twin-${suffix}`,
727
+ summary: args.reasoningSummary ? [{ type: 'summary_text', text: `[twin-stub] reasoning summary (effort=${args.reasoningEffort}); the twin cannot run the model, so the chain-of-thought is not real.` }] : [],
728
+ };
729
+ // only a model that reasons lists its reasoning item (openai-models.ts)
730
+ if (reasons(args.model))
731
+ output.push(reasoningItem);
732
+ usage.output_tokens += reasoningTokens;
733
+ usage.total_tokens += reasoningTokens;
734
+ usage.output_tokens_details = { reasoning_tokens: reasoningTokens };
735
+ }
736
+ // the built-in tools the request offers, which OpenAI runs itself before answering: the placeholder model searches
737
+ // with each one offered, on the input's text, unless `tool_choice` is `none` (the reference's Web search and File search
738
+ // examples answer a `web_search_call` / `file_search_call` item before the message:
739
+ // https://platform.openai.com/docs/guides/tools-web-search, https://platform.openai.com/docs/guides/tools-file-search)
740
+ if (!scripted && args.echo.tool_choice !== 'none')
741
+ output.push(...builtInCalls(args, suffix));
742
+ // A scripted turn with tool calls and no words carries the calls alone, as the vendor does.
743
+ if (text || !calls.length)
744
+ output.push(messageItem);
745
+ output.push(...calls);
746
+ return {
747
+ id: `resp-twin-${suffix}`,
748
+ object: 'response',
749
+ created_at: nowEpoch(occurredAt),
750
+ status: 'completed',
751
+ // answered whole at once: completed the instant it was created
752
+ completed_at: nowEpoch(occurredAt),
753
+ model: args.model,
754
+ // no `output_text`: the spec's is an "SDK-only convenience property" the SDKs compute from `output`
755
+ output,
756
+ usage,
757
+ ...responseSettings(args),
758
+ };
759
+ }
760
+ // Unstored responses have no state to overwrite. Stored identities are allocated atomically below.
761
+ export function responseSuffix(args) {
762
+ let h = 0x811c9dc5;
763
+ const s = JSON.stringify([args.model, args.inputItems ?? args.messages, args.previousResponseId, args.instructions]);
764
+ for (let i = 0; i < s.length; i++) {
765
+ h ^= s.charCodeAt(i);
766
+ h = Math.imul(h, 0x01000193);
767
+ }
768
+ return (h >>> 0).toString(36);
769
+ }
770
+ // Allocate and create under the kernel's cross-process lock. Repeated requests are occurrences;
771
+ // a content hash alone would overwrite an earlier response (and its queued scenario decision).
772
+ export async function createStoredResponse(args, req, decision) {
773
+ const { value } = await applyTwinWriteAtomic(SERVICE, (resources) => {
774
+ let ordinal = 0;
775
+ for (const row of resources) {
776
+ if (row.type !== 'response')
777
+ continue;
778
+ const match = /^resp-twin-(\d+)$/.exec(String(row.id));
779
+ if (match)
780
+ ordinal = Math.max(ordinal, Number(match[1]));
781
+ }
782
+ const suffix = String(ordinal + 1);
783
+ const inputItems = args.inputItems.map((item, i) => ({ ...item, id: item.id ?? `item-in-${suffix}-${i}` }));
784
+ const resp = args.background ? {
785
+ id: `resp-twin-${suffix}`, object: 'response', created_at: nowEpoch(req.occurredAt),
786
+ status: 'queued', completed_at: null, model: args.model, output: [],
787
+ usage: null, ...responseSettings(args),
788
+ } : buildResponse(args, req.occurredAt, suffix, decision);
789
+ return { kind: 'write', value: resp, write: {
790
+ operation: 'response.create', subjectType: 'response', subjectId: resp.id,
791
+ fields: { ...resp, _stored: true, _input_items: inputItems,
792
+ ...(args.background ? { _bg_args: JSON.stringify(args), _bg_suffix: suffix, _bg_decision: decision ? JSON.stringify(decision) : null } : {}),
793
+ },
794
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
795
+ } };
796
+ }, req.root);
797
+ return value;
798
+ }
799
+ export function emitResponse(resp, sink) {
800
+ sink({ data: { type: 'response.created', response: { ...resp, output: [] } } });
801
+ // Emit each output item in order, as the vendor streams them: a reasoning item whole, the message
802
+ // item's text as deltas at its own index, a function call's arguments as one delta then done.
803
+ resp.output.forEach((item, i) => {
804
+ sink({ data: { type: 'response.output_item.added', output_index: i, item: item.type === 'function_call' ? { ...item, arguments: '' } : item } });
805
+ if (item.type === 'message') {
806
+ const text = item.content[0]?.text ?? '';
807
+ for (const piece of chunkText(text))
808
+ sink({ data: { type: 'response.output_text.delta', output_index: i, content_index: 0, delta: piece } });
809
+ sink({ data: { type: 'response.output_text.done', output_index: i, content_index: 0, text } });
810
+ }
811
+ if (item.type === 'function_call') {
812
+ sink({ data: { type: 'response.function_call_arguments.delta', output_index: i, item_id: item.id, delta: item.arguments } });
813
+ sink({ data: { type: 'response.function_call_arguments.done', output_index: i, item_id: item.id, arguments: item.arguments } });
814
+ }
815
+ sink({ data: { type: 'response.output_item.done', output_index: i, item } });
816
+ });
817
+ sink({ data: { type: 'response.completed', response: resp } });
818
+ sink({ done: true });
819
+ return resp;
820
+ }
821
+ // ── Embeddings (deterministic pseudo-vectors) ───────────────────────────────────────────
822
+ export function handleEmbeddings(params) {
823
+ if (params.model === undefined || params.model === '')
824
+ return invalidRequest("you must provide a model parameter", 'model');
825
+ if (params.input === undefined)
826
+ return invalidRequest("you must provide an input parameter", 'input');
827
+ const model = String(params.model);
828
+ const inputs = typeof params.input === 'string'
829
+ ? [params.input]
830
+ : Array.isArray(params.input)
831
+ ? params.input.map((x) => (typeof x === 'string' ? x : JSON.stringify(x)))
832
+ : [];
833
+ if (inputs.length === 0)
834
+ return invalidRequest("'input' must be a non-empty string or array", 'input');
835
+ const dimensions = params.dimensions !== undefined ? Number(params.dimensions) : defaultDims(model);
836
+ if (!Number.isInteger(dimensions) || dimensions < 1)
837
+ return invalidRequest("'dimensions' must be a positive integer", 'dimensions');
838
+ // The `openai` SDK defaults to encoding_format:'base64' and decodes a base64 Float32 buffer
839
+ // back into numbers client-side. Honor both formats faithfully.
840
+ const asBase64 = params.encoding_format === 'base64';
841
+ let promptTokens = 0;
842
+ const data = inputs.map((text, index) => {
843
+ promptTokens += estimateTokens(text);
844
+ const vec = pseudoEmbedding(text, dimensions);
845
+ return { object: 'embedding', index, embedding: (asBase64 ? floatsToBase64(vec) : vec) };
846
+ });
847
+ const out = { object: 'list', data, model, usage: { prompt_tokens: promptTokens, total_tokens: promptTokens } };
848
+ return { status: 200, body: out };
849
+ }
850
+ function defaultDims(model) {
851
+ if (model.includes('large'))
852
+ return 3072;
853
+ return 1536; // small + ada-002
854
+ }
855
+ /** Encode a float vector as a base64 little-endian Float32 buffer (the SDK's base64 format). */
856
+ function floatsToBase64(vec) {
857
+ const buf = new ArrayBuffer(vec.length * 4);
858
+ const view = new DataView(buf);
859
+ for (let i = 0; i < vec.length; i++)
860
+ view.setFloat32(i * 4, vec[i], true);
861
+ // btoa over the raw bytes (available in Bun/browser).
862
+ let bin = '';
863
+ const bytes = new Uint8Array(buf);
864
+ for (let i = 0; i < bytes.length; i++)
865
+ bin += String.fromCharCode(bytes[i]);
866
+ return btoa(bin);
867
+ }
868
+ // ── Moderations (deterministic) ─────────────────────────────────────────────────────────
869
+ export function handleModerations(params, occurredAt) {
870
+ if (params.input === undefined)
871
+ return invalidRequest("you must provide an input parameter", 'input');
872
+ // an array of strings is several inputs, one result each; an array of text and image parts is ONE multimodal input,
873
+ // one result (https://platform.openai.com/docs/api-reference/moderations/create, `input`)
874
+ const list = Array.isArray(params.input) ? params.input : [];
875
+ const parts = list.length > 0 && list.every((x) => x && typeof x === 'object' && !Array.isArray(x));
876
+ const inputs = typeof params.input === 'string'
877
+ ? [{ text: params.input, image: false }]
878
+ : parts
879
+ ? [{ text: list.filter((x) => x.type === 'text').map((x) => String(x.text ?? '')).join('\n'), image: list.some((x) => x.type === 'image_url') }]
880
+ : list.map((x) => ({ text: typeof x === 'string' ? x : JSON.stringify(x), image: false }));
881
+ if (inputs.length === 0)
882
+ return invalidRequest("'input' must be a non-empty string or array", 'input');
883
+ const model = modelOf('createModeration', params);
884
+ const results = inputs.map((t) => moderateText(t.text, t.image));
885
+ return { status: 200, body: { id: `modr-twin-${nowEpoch(occurredAt)}`, model, results } };
886
+ }
887
+ // ── Images (a real placeholder image, no model: ./openai-media.ts) ─────────────────────────────────
888
+ const gptImage = (model) => /^(gpt-image|chatgpt-image)/.test(model);
889
+ /** One answered image: the GPT image models give the image itself as base64, always; `dall-e-2` and `dall-e-3` a URL
890
+ * unless `response_format` is `b64_json`, and `dall-e-3` its revised prompt (developers.openai.com/api/reference/
891
+ * resources/images/methods/generate, `response_format`; the Image object's `b64_json`, `url`, `revised_prompt`). */
892
+ function imageData(params, model, size, seed, kind, i, occurredAt) {
893
+ return gptImage(model) || params.response_format === 'b64_json' ? { b64_json: base64(placeholderPng(size, `${seed} #${i + 1}`)) } : dalleUrl(model, seed, kind, i, occurredAt);
894
+ }
895
+ /** A DALL·E image as its URL, the default for `dall-e-2` and `dall-e-3`. */
896
+ function dalleUrl(model, seed, kind, i, occurredAt) {
897
+ return { url: `https://twin.invalid/openai-${kind}-stub/${nowEpoch(occurredAt)}-${i}.png`, ...(model === 'dall-e-3' ? { revised_prompt: `[twin-stub] ${seed}` } : {}) };
898
+ }
899
+ function imageCount(params) {
900
+ const n = Number(params.n ?? 1);
901
+ return Number.isInteger(n) && n > 0 ? n : 1;
902
+ }
903
+ /** A GPT image model's answer: the images with the tokens they cost (the image generation guide's 1024x1024
904
+ * medium-quality image is 1056 output tokens, scaled here by area; the twin reads no pixels, so it counts no input image
905
+ * tokens: https://platform.openai.com/docs/guides/image-generation#cost-and-latency), or, with `stream`, the
906
+ * `partial_images` asked (none by default: "When set to 0, the response will be a single image sent in one streaming
907
+ * event") then the completed image (https://platform.openai.com/docs/api-reference/images-streaming). */
908
+ function gptImageAnswer(params, size, data, kind, occurredAt) {
909
+ const text = estimateTokens(String(params.prompt ?? ''));
910
+ const output = Math.ceil((1056 * size.width * size.height) / (1024 * 1024)) * data.length;
911
+ const usage = { total_tokens: text + output, input_tokens: text, output_tokens: output, input_tokens_details: { text_tokens: text, image_tokens: 0 } };
912
+ const created = nowEpoch(occurredAt);
913
+ if (params.stream !== true && params.stream !== 'true')
914
+ return { status: 200, body: { created, data, usage } };
915
+ const settings = {
916
+ created_at: created, size: `${size.width}x${size.height}`,
917
+ // what `auto` settles on for a placeholder: an opaque PNG of medium quality
918
+ quality: params.quality && params.quality !== 'auto' ? params.quality : 'medium',
919
+ background: params.background && params.background !== 'auto' ? params.background : 'opaque',
920
+ output_format: params.output_format ?? 'png',
921
+ };
922
+ const partials = Math.min(3, Math.max(0, Number(params.partial_images ?? 0) || 0));
923
+ const b64 = String(data[0]?.b64_json ?? '');
924
+ const events = [
925
+ ...Array.from({ length: partials }, (_, i) => ({ event: `${kind}.partial_image`, data: { type: `${kind}.partial_image`, b64_json: b64, ...settings, partial_image_index: i } })),
926
+ { event: `${kind}.completed`, data: { type: `${kind}.completed`, b64_json: b64, ...settings, usage } },
927
+ ];
928
+ return { status: 200, body: { created, data, usage }, events };
929
+ }
930
+ export function handleImages(params, occurredAt) {
931
+ if (params.prompt === undefined || params.prompt === '')
932
+ return invalidRequest("you must provide a prompt parameter", 'prompt');
933
+ const model = modelOf('createImage', params);
934
+ const size = requestedSize(params.size);
935
+ const data = Array.from({ length: imageCount(params) }, (_, i) => imageData(params, model, size, String(params.prompt), 'image', i, occurredAt));
936
+ if (gptImage(model))
937
+ return gptImageAnswer(params, size, data, 'image_generation', occurredAt);
938
+ return { status: 200, body: { created: nowEpoch(occurredAt), data } };
939
+ }
940
+ /** OpenAI's refusal of an upload that is not an image the model takes. */
941
+ const badImage = (formats) => invalidRequest(`Invalid image file: 'image' must be ${formats}.`, 'image', 'invalid_image_format');
942
+ // An EDIT takes an image file and a prompt; a VARIATION (dall-e-2 only) a square PNG. Each is told by its bytes.
943
+ export function handleImageEdit(params, occurredAt) {
944
+ // a GPT image model edits several images sent as `image[]` (the form keeps the last one named so)
945
+ const image = params.image ?? params['image[]'];
946
+ if (image === undefined || image === '')
947
+ return invalidRequest("you must provide an image to edit", 'image');
948
+ if (params.prompt === undefined || params.prompt === '')
949
+ return invalidRequest("you must provide a prompt parameter", 'prompt');
950
+ const model = modelOf('createImageEdit', params);
951
+ const bytes = partBytes(image);
952
+ const format = bytes && imageFormat(bytes);
953
+ if (gptImage(model) ? !format : format !== 'png' || !square(bytes))
954
+ return badImage(gptImage(model) ? 'a png, webp, or jpg file' : 'a square png file');
955
+ const size = requestedSize(params.size, format === 'png' ? pngSize(bytes) : undefined);
956
+ const data = Array.from({ length: imageCount(params) }, (_, i) => imageData(params, model, size, String(params.prompt), 'image-edit', i, occurredAt));
957
+ if (gptImage(model))
958
+ return gptImageAnswer(params, size, data, 'image_edit', occurredAt);
959
+ return { status: 200, body: { created: nowEpoch(occurredAt), data } };
960
+ }
961
+ const square = (png) => { const s = pngSize(png); return s.width === s.height; };
962
+ export function handleImageVariation(params, occurredAt) {
963
+ if (params.image === undefined || params.image === '')
964
+ return invalidRequest("you must provide an image", 'image');
965
+ const bytes = partBytes(params.image);
966
+ if (!bytes || imageFormat(bytes) !== 'png' || !square(bytes))
967
+ return badImage('a valid square png file');
968
+ const size = requestedSize(params.size);
969
+ const data = Array.from({ length: imageCount(params) }, (_, i) => imageData(params, modelOf('createImageVariation', params), size, 'variation', 'image-variation', i, occurredAt));
970
+ return { status: 200, body: { created: nowEpoch(occurredAt), data } };
971
+ }
972
+ // ── Audio (deterministic stubs; the twin runs no model) ──────────────────────
973
+ // Transcription/translation cannot run a real speech model, so the twin returns a clearly
974
+ // labeled deterministic transcript naming the uploaded file. It takes the audio file itself, in a
975
+ // format OpenAI transcribes (./openai-media.ts), never a file's name. The response SHAPE
976
+ // (json / verbose_json / text) is vendor-faithful.
977
+ /** A multipart list field (`include[]`, `timestamp_granularities[]`), however the form named it. */
978
+ function formList(params, name) {
979
+ const v = params[name] ?? params[`${name}[]`];
980
+ return (Array.isArray(v) ? v : v === undefined ? [] : [v]).map(String);
981
+ }
982
+ /** What a transcription is billed on: the GPT-4o transcribe models by tokens, whisper-1 and the diarizing model by the
983
+ * audio's seconds (https://platform.openai.com/docs/api-reference/audio/json-object, `usage`). The twin counts ten
984
+ * audio tokens a second and its transcript's text tokens; neither is a model's count. */
985
+ function transcriptionUsage(model, seconds, text) {
986
+ if (model.startsWith('whisper') || model.includes('diarize'))
987
+ return { type: 'duration', seconds: Math.ceil(seconds) };
988
+ const audio = Math.ceil(seconds * 10);
989
+ const out = estimateTokens(text);
990
+ return { type: 'tokens', input_tokens: audio, input_token_details: { text_tokens: 0, audio_tokens: audio }, output_tokens: out, total_tokens: audio + out };
991
+ }
992
+ /** The transcript's words spread evenly over the audio: the twin hears nothing, so its timings are the file's length
993
+ * shared out, not a model's alignment. */
994
+ function words(text, seconds) {
995
+ const all = text.split(/\s+/).filter(Boolean);
996
+ const each = seconds / Math.max(1, all.length);
997
+ return all.map((word, i) => ({ word, start: i * each, end: (i + 1) * each }));
998
+ }
999
+ /** A transcription (or translation) of the uploaded audio: a labeled transcript naming the file, in the response
1000
+ * format asked (json, text, verbose_json with the word or segment timestamps asked, or diarized_json), with the
1001
+ * token log probabilities when `include[]=logprobs`, and as `transcript.text.delta` events then
1002
+ * `transcript.text.done` when `stream=true` on a model that streams (https://platform.openai.com/docs/api-reference/audio/createTranscription). */
1003
+ export function handleTranscription(params, translate) {
1004
+ if (params.file === undefined || params.file === '')
1005
+ return invalidRequest("you must provide a file parameter", 'file');
1006
+ if (params.model === undefined || params.model === '')
1007
+ return invalidRequest("you must provide a model parameter", 'model');
1008
+ const bytes = partBytes(params.file);
1009
+ if (!bytes || !audioFormat(bytes))
1010
+ return invalidRequest("Invalid file format. Supported formats: ['flac', 'm4a', 'mp3', 'mp4', 'mpeg', 'mpga', 'oga', 'ogg', 'wav', 'webm']", 'file', 'invalid_value');
1011
+ const filename = String(params.file.name || 'upload');
1012
+ const model = String(params.model);
1013
+ const verb = translate ? 'translation' : 'transcription';
1014
+ const text = `[twin-stub] deterministic ${verb} of ${filename} (no speech model is run)`;
1015
+ const format = typeof params.response_format === 'string' ? params.response_format : 'json';
1016
+ // the file's own length where the twin can read it (a WAV's header), else a labeled second
1017
+ const duration = wavSeconds(bytes) ?? 1.0;
1018
+ if (format === 'text')
1019
+ return { status: 200, body: text, seconds: duration };
1020
+ if (translate)
1021
+ return { status: 200, seconds: duration, body: format === 'verbose_json' ? { task: 'translation', language: 'english', duration, text, segments: [{ id: 0, start: 0, end: duration, text }] } : { text } };
1022
+ const usage = transcriptionUsage(model, duration, text);
1023
+ if (format === 'diarized_json') {
1024
+ // one speaker the twin cannot tell apart: the first name the caller knows, else OpenAI's first label
1025
+ const speaker = formList(params, 'known_speaker_names')[0] ?? 'A';
1026
+ return { status: 200, seconds: duration, body: { task: 'transcribe', duration, text, segments: [{ type: 'transcript.text.segment', id: 'seg_001', start: 0, end: duration, text, speaker }], usage } };
1027
+ }
1028
+ if (format === 'verbose_json') {
1029
+ const granularities = formList(params, 'timestamp_granularities');
1030
+ const segment = { id: 0, seek: 0, start: 0, end: duration, text, tokens: [], temperature: 0, avg_logprob: 0, compression_ratio: 1, no_speech_prob: 0 };
1031
+ return {
1032
+ status: 200,
1033
+ seconds: duration,
1034
+ body: {
1035
+ task: 'transcribe', language: typeof params.language === 'string' ? params.language : 'english', duration, text,
1036
+ ...(granularities.includes('word') ? { words: words(text, duration) } : {}),
1037
+ ...(!granularities.length || granularities.includes('segment') ? { segments: [segment] } : {}),
1038
+ usage,
1039
+ },
1040
+ };
1041
+ }
1042
+ const logprobs = formList(params, 'include').includes('logprobs') ? buildLogprobs(text, 0).content.map(({ token, logprob, bytes: b }) => ({ token, logprob, bytes: b })) : undefined;
1043
+ const body = { text, ...(logprobs ? { logprobs } : {}), usage };
1044
+ // whisper-1 does not stream ("Streaming is not supported for the whisper-1 model and will be ignored")
1045
+ if (String(params.stream) === 'true' && !model.startsWith('whisper')) {
1046
+ const pieces = text.match(/\s*\S+/g) ?? [];
1047
+ const events = pieces.map((delta) => ({ data: { type: 'transcript.text.delta', delta, ...(logprobs ? { logprobs: buildLogprobs(delta, 0).content.map(({ token, logprob, bytes: b }) => ({ token, logprob, bytes: b })) } : {}) } }));
1048
+ events.push({ data: { type: 'transcript.text.done', text, ...(logprobs ? { logprobs } : {}), usage } });
1049
+ return { status: 200, body, seconds: duration, events };
1050
+ }
1051
+ return { status: 200, body, seconds: duration };
1052
+ }
1053
+ // Text-to-speech: runs no speech model, but answers the audio file itself in the requested format, as OpenAI does
1054
+ // (a second of silence: ./openai-media.ts), with that format's Content-Type.
1055
+ const SPEECH = {
1056
+ mp3: { type: 'audio/mpeg', file: silentMp3 },
1057
+ opus: { type: 'audio/opus', file: silentOpus },
1058
+ aac: { type: 'audio/aac', file: silentAac },
1059
+ flac: { type: 'audio/flac', file: silentFlac },
1060
+ wav: { type: 'audio/wav', file: silentWav },
1061
+ pcm: { type: 'audio/pcm', file: silentPcm },
1062
+ };
1063
+ export function handleSpeech(params) {
1064
+ if (params.model === undefined || params.model === '')
1065
+ return invalidRequest("you must provide a model parameter", 'model');
1066
+ if (params.input === undefined || params.input === '')
1067
+ return invalidRequest("you must provide an input parameter", 'input');
1068
+ if (params.voice === undefined || params.voice === '')
1069
+ return invalidRequest("you must provide a voice parameter", 'voice');
1070
+ const format = SPEECH[String(params.response_format ?? 'mp3')];
1071
+ if (!format)
1072
+ return invalidRequest("'response_format' must be one of 'mp3', 'opus', 'aac', 'flac', 'wav', 'pcm'", 'response_format');
1073
+ return { status: 200, audio: format.file(), type: format.type };
1074
+ }
1075
+ // ── public entry: cross-cutting protocol (auth / rate-limit) then route ─────────────────────
1076
+ // The HTTP server passes request headers, so on the live wire every call is auth/rate-limit-
1077
+ // checked; in-process trusted calls (capability verify, connector) omit `headers`/`apiKey` and are
1078
+ // NOT gated.
1079
+ /** OpenAI's API as one call: the request goes through the pack's own dispatch (the derived dispatch over its
1080
+ * semantics, its derived core and the hand-written routes below), so a caller that holds no HTTP
1081
+ * server reaches exactly what an SDK does. A caller that presents no credential is a trusted
1082
+ * in-process caller and is not held to the auth rule; `sseSink` receives a stream's events. */
1083
+ export async function handleOpenAITwinRequest(req) {
1084
+ const twin = createOpenAITwinFetch({
1085
+ ...(req.root !== undefined ? { root: req.root } : {}),
1086
+ readOnly: req.readOnly ?? false,
1087
+ ...(req.scenarioEngine ? { scenarioEngine: req.scenarioEngine } : {}),
1088
+ ...(req.occurredAt ? { clock: () => req.occurredAt } : {}),
1089
+ });
1090
+ const headers = { ...(req.headers ?? {}) };
1091
+ if (req.apiKey !== undefined)
1092
+ headers.authorization = `Bearer ${req.apiKey}`;
1093
+ else if (req.headers === undefined)
1094
+ headers.authorization = 'Bearer sk-twin-in-process';
1095
+ const method = req.method.toUpperCase();
1096
+ if (typeof req.body === 'string' && method !== 'GET' && method !== 'HEAD' && !headers['content-type'])
1097
+ headers['content-type'] = 'application/json';
1098
+ const path = req.path.startsWith('/') ? req.path : `/${req.path}`;
1099
+ const response = await twin(new Request(`https://api.openai.com${path}`, { method, headers, ...(req.body !== undefined && method !== 'GET' && method !== 'HEAD' ? { body: req.body } : {}) }));
1100
+ const answerHeaders = {};
1101
+ response.headers.forEach((v, k) => { if (k !== 'content-type')
1102
+ answerHeaders[k] = v; });
1103
+ const type = response.headers.get('content-type') ?? '';
1104
+ const text = await response.text();
1105
+ if (type.includes('text/event-stream')) {
1106
+ // the frames back into events, for a caller that collects them
1107
+ for (const frame of text.split('\n\n')) {
1108
+ const data = frame.split('\n').filter((l) => l.startsWith('data: ')).map((l) => l.slice(6)).join('\n');
1109
+ if (!data)
1110
+ continue;
1111
+ req.sseSink?.(data === '[DONE]' ? { done: true } : { data: JSON.parse(data) });
1112
+ }
1113
+ return { status: response.status, body: null, headers: answerHeaders };
1114
+ }
1115
+ const body = type.includes('json') ? (text ? JSON.parse(text) : null) : text;
1116
+ return { status: response.status, body, headers: answerHeaders };
1117
+ }