@volter/twin-openai 0.1.2 → 2.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. package/README.md +33 -30
  2. package/defaults/handlers.json +10 -0
  3. package/dist/defaults/handlers.json +10 -0
  4. package/dist/src/cli.d.ts +2 -0
  5. package/dist/src/cli.js +29 -0
  6. package/dist/src/generated/surface.gen.json +1 -0
  7. package/dist/src/generated/ui.gen.json +1 -0
  8. package/dist/src/index.d.ts +19 -0
  9. package/dist/src/index.js +72 -0
  10. package/dist/src/manifest.d.ts +6 -0
  11. package/dist/src/manifest.js +323 -0
  12. package/dist/src/openai-budget.d.ts +53 -0
  13. package/dist/src/openai-budget.js +147 -0
  14. package/dist/src/openai-capabilities.d.ts +4 -0
  15. package/dist/src/openai-capabilities.js +1569 -0
  16. package/dist/src/openai-conformance.d.ts +13 -0
  17. package/dist/src/openai-conformance.js +116 -0
  18. package/dist/src/openai-connector.d.ts +86 -0
  19. package/dist/src/openai-connector.js +291 -0
  20. package/dist/src/openai-media.d.ts +43 -0
  21. package/dist/src/openai-media.js +257 -0
  22. package/dist/src/openai-models.d.ts +74 -0
  23. package/dist/src/openai-models.js +148 -0
  24. package/dist/src/openai-scenario.d.ts +51 -0
  25. package/dist/src/openai-scenario.js +166 -0
  26. package/dist/src/openai-server.d.ts +40 -0
  27. package/dist/src/openai-server.js +126 -0
  28. package/dist/src/openai-stub.d.ts +82 -0
  29. package/dist/src/openai-stub.js +256 -0
  30. package/dist/src/openai-twin.d.ts +182 -0
  31. package/dist/src/openai-twin.js +1117 -0
  32. package/dist/src/openai-types.d.ts +194 -0
  33. package/dist/src/openai-types.js +4 -0
  34. package/dist/src/openai-webhooks.d.ts +47 -0
  35. package/dist/src/openai-webhooks.js +99 -0
  36. package/dist/src/screens/api-keys.d.ts +16 -0
  37. package/dist/src/screens/api-keys.js +131 -0
  38. package/dist/src/screens/session.d.ts +22 -0
  39. package/dist/src/screens/session.js +115 -0
  40. package/dist/src/semantics/assistants.d.ts +2 -0
  41. package/dist/src/semantics/assistants.js +331 -0
  42. package/dist/src/semantics/audio.d.ts +2 -0
  43. package/dist/src/semantics/audio.js +27 -0
  44. package/dist/src/semantics/batches.d.ts +4 -0
  45. package/dist/src/semantics/batches.js +86 -0
  46. package/dist/src/semantics/chat-completions.d.ts +3 -0
  47. package/dist/src/semantics/chat-completions.js +58 -0
  48. package/dist/src/semantics/containers.d.ts +2 -0
  49. package/dist/src/semantics/containers.js +147 -0
  50. package/dist/src/semantics/embeddings.d.ts +2 -0
  51. package/dist/src/semantics/embeddings.js +13 -0
  52. package/dist/src/semantics/evals.d.ts +2 -0
  53. package/dist/src/semantics/evals.js +173 -0
  54. package/dist/src/semantics/files.d.ts +13 -0
  55. package/dist/src/semantics/files.js +59 -0
  56. package/dist/src/semantics/fine-tuning.d.ts +4 -0
  57. package/dist/src/semantics/fine-tuning.js +178 -0
  58. package/dist/src/semantics/images.d.ts +2 -0
  59. package/dist/src/semantics/images.js +18 -0
  60. package/dist/src/semantics/index.d.ts +8 -0
  61. package/dist/src/semantics/index.js +46 -0
  62. package/dist/src/semantics/models.d.ts +2 -0
  63. package/dist/src/semantics/models.js +34 -0
  64. package/dist/src/semantics/moderations.d.ts +2 -0
  65. package/dist/src/semantics/moderations.js +12 -0
  66. package/dist/src/semantics/organization.d.ts +2 -0
  67. package/dist/src/semantics/organization.js +67 -0
  68. package/dist/src/semantics/progress.d.ts +22 -0
  69. package/dist/src/semantics/progress.js +63 -0
  70. package/dist/src/semantics/responses.d.ts +3 -0
  71. package/dist/src/semantics/responses.js +153 -0
  72. package/dist/src/semantics/shared.d.ts +32 -0
  73. package/dist/src/semantics/shared.js +69 -0
  74. package/dist/src/semantics/uploads.d.ts +2 -0
  75. package/dist/src/semantics/uploads.js +84 -0
  76. package/dist/src/semantics/vector-stores.d.ts +2 -0
  77. package/dist/src/semantics/vector-stores.js +281 -0
  78. package/dist/test-fixtures/openai-openapi-operations.SOURCE.md +18 -0
  79. package/dist/test-fixtures/openai-openapi-operations.json +1849 -0
  80. package/package.json +21 -10
  81. package/src/cli.ts +9 -7
  82. package/src/generated/surface.gen.json +1 -0
  83. package/src/generated/ui.gen.json +1 -0
  84. package/src/index.ts +20 -10
  85. package/src/manifest.ts +343 -0
  86. package/src/openai-budget.ts +4 -4
  87. package/src/openai-capabilities.ts +177 -195
  88. package/src/openai-conformance.ts +1 -1
  89. package/src/openai-connector.ts +40 -43
  90. package/src/openai-media.ts +225 -0
  91. package/src/openai-models.ts +145 -15
  92. package/src/openai-scenario.ts +46 -10
  93. package/src/openai-server.ts +65 -108
  94. package/src/openai-stub.ts +54 -30
  95. package/src/openai-twin.ts +760 -1665
  96. package/src/openai-types.ts +24 -6
  97. package/src/openai-webhooks.ts +3 -2
  98. package/src/screens/api-keys.tsx +138 -0
  99. package/src/screens/session.tsx +131 -0
  100. package/src/semantics/assistants.ts +336 -0
  101. package/src/semantics/audio.ts +31 -0
  102. package/src/semantics/batches.ts +88 -0
  103. package/src/semantics/chat-completions.ts +66 -0
  104. package/src/semantics/containers.ts +151 -0
  105. package/src/semantics/embeddings.ts +19 -0
  106. package/src/semantics/evals.ts +182 -0
  107. package/src/semantics/files.ts +67 -0
  108. package/src/semantics/fine-tuning.ts +185 -0
  109. package/src/semantics/images.ts +23 -0
  110. package/src/semantics/index.ts +52 -0
  111. package/src/semantics/models.ts +41 -0
  112. package/src/semantics/moderations.ts +14 -0
  113. package/src/semantics/organization.ts +76 -0
  114. package/src/semantics/progress.ts +72 -0
  115. package/src/semantics/responses.ts +151 -0
  116. package/src/semantics/shared.ts +82 -0
  117. package/src/semantics/uploads.ts +92 -0
  118. package/src/semantics/vector-stores.ts +279 -0
  119. package/test-fixtures/openai-openapi-operations.SOURCE.md +4 -5
  120. package/test-fixtures/openai-openapi-operations.json +224 -1334
@@ -5,30 +5,35 @@
5
5
  // `expected:'done'` only on capabilities we genuinely claim, so a broken one shows as a
6
6
  // regression. Grow this toward the API's *full* surface every cycle.
7
7
  //
8
- // THE HONEST CARVE-OUTS: `openai.chat.inference` and `openai.embeddings.real_vectors` are the
9
- // out-of-scope entries — the twin returns a DETERMINISTIC STUB for model generation (no
10
- // weights) and DETERMINISTIC pseudo-vectors for embeddings, while the protocol envelope (shape/
11
- // streaming/tool_calls/usage) is faithful. `openai.images.real_pixels` and
12
- // `openai.files.real_bytes` carve out the binary content. Surfaced here + in README ## Coverage.
8
+ // Every capability is done or todo. A twin is a deterministic, offline model of the vendor's API
9
+ // contract; the labeled stub completion and the deterministic pseudo-vector ARE the twin's
10
+ // answer (see `openai.chat.stub_labeled`, `openai.embeddings.deterministic`), not a shortfall
11
+ // from a "real" one. Every entry here is either done or todo.
13
12
  import { mkdtempSync, rmSync } from 'node:fs';
14
13
  import { tmpdir } from 'node:os';
15
14
  import { join } from 'node:path';
16
- import { checkCapabilities, type CapabilityReport, type CapabilitySpec, verifyBoundary, isInfrastructureError, harnessError } from '@volter/twin-tooling';
15
+ import { checkCapabilities, type CapabilityReport, type CapabilitySpec, verifyBoundary, isInfrastructureError, harnessError } from '@volter/world-tooling';
17
16
  import { handleOpenAITwinRequest, type OpenAIResponseEnvelope } from './openai-twin.ts';
17
+ import { audioFormat, imageFormat, placeholderPng, wavSeconds } from './openai-media.ts';
18
+ import { createOpenAITwinFetch } from './openai-server.ts';
18
19
  import { buildSignedOpenAIWebhook, verifyOpenAIWebhook, OpenAIWebhookVerificationError, computeOpenAIWebhookSignature } from './openai-webhooks.ts';
19
20
  import type { SseEvent } from './openai-types.ts';
20
21
 
21
- // (OpenAI is an API-first vendor with no meaningful product UI worth mirroring — ARCHITECTURE.md
22
+ // (OpenAI is an API-first vendor with no meaningful product UI worth mirroring — docs/contributing/architecture.md
22
23
  // C1b — so this pack ships no mirror, and there are no UI capabilities to verify.)
23
24
 
24
25
  // ── API verify: drive REAL requests against a fresh temp root, then assert status/shape ──
25
- type Step = { m: string; p: string; b?: unknown };
26
+ type Step = { m: string; p: string; b?: unknown; at?: string };
27
+ /** The World instant the checks run at unless a step names one: 2026-01-01, when every model and API family the checks
28
+ * use is live on OpenAI's timeline (./openai-models.ts: dall-e until 2026-05-12, the Assistants API until 2026-08-26,
29
+ * fine-tuning by a new organization until 2026-05-07). */
30
+ const CHECKS_AT = '2026-01-01T00:00:00.000Z';
26
31
  type Body = Record<string, any>;
27
32
 
28
33
  /** Run a sequence of real OpenAI requests against an isolated root; return all responses. */
29
34
  async function withRoot(steps: (h: (s: Step) => Promise<OpenAIResponseEnvelope>) => Promise<boolean>): Promise<boolean> {
30
35
  const root = mkdtempSync(join(tmpdir(), 'openai-cap-'));
31
- const h = (s: Step) => handleOpenAITwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : JSON.stringify(s.b), root });
36
+ const h = (s: Step) => handleOpenAITwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : s.b instanceof FormData ? s.b : JSON.stringify(s.b), root, occurredAt: s.at ?? CHECKS_AT });
32
37
  try {
33
38
  return await verifyBoundary('openai.withRoot', () => steps(h));
34
39
  } finally {
@@ -41,7 +46,7 @@ function withStream(body: unknown, fn: (events: SseEvent[], final: OpenAIRespons
41
46
  return new Promise<boolean>((resolve, reject) => {
42
47
  const root = mkdtempSync(join(tmpdir(), 'openai-cap-'));
43
48
  const events: SseEvent[] = [];
44
- handleOpenAITwinRequest({ method: 'POST', path: bodyPath(body), body: JSON.stringify(body), root, sseSink: (e) => events.push(e) })
49
+ handleOpenAITwinRequest({ method: 'POST', path: bodyPath(body), body: JSON.stringify(body), root, occurredAt: CHECKS_AT, sseSink: (e) => events.push(e) })
45
50
  .then((final) => resolve(fn(events, final)))
46
51
  .catch((err) => { if (isInfrastructureError(err)) reject(harnessError('openai.withStream', err)); else resolve(false); })
47
52
  .finally(() => rmSync(root, { recursive: true, force: true }));
@@ -51,12 +56,12 @@ function bodyPath(body: unknown): string {
51
56
  return body && typeof body === 'object' && 'input' in (body as Record<string, unknown>) ? '/v1/responses' : '/v1/chat/completions';
52
57
  }
53
58
 
54
- /** Like withRoot, but the request helper passes request HEADERS through (for auth/rate-limit/
55
- * idempotency, which the trusted no-headers withRoot helper deliberately never triggers). */
59
+ /** Like withRoot, but the request helper passes request HEADERS through (for auth and rate-limit,
60
+ * which the trusted no-headers withRoot helper deliberately never triggers). */
56
61
  type StepH = Step & { headers?: Record<string, string> };
57
62
  async function withRootH(steps: (h: (s: StepH) => Promise<OpenAIResponseEnvelope>) => Promise<boolean>): Promise<boolean> {
58
63
  const root = mkdtempSync(join(tmpdir(), 'openai-cap-'));
59
- const h = (s: StepH) => handleOpenAITwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : JSON.stringify(s.b), root, ...(s.headers ? { headers: s.headers } : {}) });
64
+ const h = (s: StepH) => handleOpenAITwinRequest({ method: s.m, path: s.p, body: s.b === undefined ? undefined : JSON.stringify(s.b), root, occurredAt: CHECKS_AT, ...(s.headers ? { headers: s.headers } : {}) });
60
65
  try {
61
66
  return await verifyBoundary('openai.withRootH', () => steps(h));
62
67
  } finally {
@@ -67,24 +72,40 @@ async function withRootH(steps: (h: (s: StepH) => Promise<OpenAIResponseEnvelope
67
72
  const ok = (r: OpenAIResponseEnvelope) => r.status >= 200 && r.status < 300;
68
73
  const id = (r: OpenAIResponseEnvelope) => (r.body as Body)?.id as string;
69
74
  const field = (r: OpenAIResponseEnvelope, k: string) => (r.body as Body)?.[k];
75
+ /** A response's text, as the SDKs' `output_text` aggregates it from its message items (the wire carries no `output_text`). */
76
+ const outputText = (r: OpenAIResponseEnvelope): string => (((r.body as Body)?.output ?? []) as Body[]).filter((i) => i.type === 'message').flatMap((i) => (i.content ?? []) as Body[]).filter((c) => c.type === 'output_text').map((c) => String(c.text)).join('');
77
+
78
+ /** A multipart upload: text fields and one file part of the given bytes. */
79
+ function form(fields: Record<string, string>, name: string, bytes: Uint8Array, filename: string, type: string): FormData {
80
+ const f = new FormData();
81
+ for (const [k, v] of Object.entries(fields)) f.append(k, v);
82
+ f.append(name, new File([bytes as Uint8Array<ArrayBuffer>], filename, { type }));
83
+ return f;
84
+ }
85
+ /** A tenth of a second of 8 kHz 8-bit mono silence as a WAV file. */
86
+ const SILENT_WAV = (() => {
87
+ const data = 800;
88
+ const b = new Uint8Array(44 + data).fill(0x80);
89
+ const v = new DataView(b.buffer);
90
+ b.set([...'RIFF'].map((c) => c.charCodeAt(0)), 0);
91
+ v.setUint32(4, 36 + data, true);
92
+ b.set([...'WAVEfmt '].map((c) => c.charCodeAt(0)), 8);
93
+ v.setUint32(16, 16, true); v.setUint16(20, 1, true); v.setUint16(22, 1, true); v.setUint32(24, 8000, true); v.setUint32(28, 8000, true); v.setUint16(32, 1, true); v.setUint16(34, 8, true);
94
+ b.set([...'data'].map((c) => c.charCodeAt(0)), 36);
95
+ v.setUint32(40, data, true);
96
+ return b;
97
+ })();
70
98
 
71
99
  // ── shorthands (mirror the stripe/anthropic manifests) ──
72
100
  const done = (id: string, area: string, title: string, dimension: CapabilitySpec['dimension'], tier: CapabilitySpec['tier'], verify: CapabilitySpec['verify']): CapabilitySpec => ({ id, area, title, dimension, tier, expected: 'done', verify });
73
101
  const todo = (id: string, area: string, title: string, dimension: CapabilitySpec['dimension'], tier: CapabilitySpec['tier']): CapabilitySpec => ({ id, area, title, dimension, tier, expected: 'todo' });
74
- const outOfScope = (id: string, area: string, title: string, dimension: CapabilitySpec['dimension'], tier: CapabilitySpec['tier'], reason: string): CapabilitySpec => ({ id, area, title, dimension, tier, expected: 'todo', outOfScope: reason });
75
102
 
76
103
  const CHAT = (extra: Record<string, unknown> = {}) => ({ model: 'gpt-4o', messages: [{ role: 'user', content: 'hello twin' }], ...extra });
77
104
 
78
105
  export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
79
106
  // ── THE HONEST CARVE-OUTS ─────────────────────────────────────────────────────────────
80
- outOfScope('openai.chat.inference', 'chat', 'Real model inference (generation from model weights)', 'api', 'core',
81
- 'Out of scope: the twin cannot run the model — chat/responses return a DETERMINISTIC STUB completion (clearly labeled), never real model output. The protocol envelope (shape/streaming/tool_calls/usage) is faithful; only the generated text is a stub.'),
82
- outOfScope('openai.embeddings.real_vectors', 'embeddings', 'Real embedding vectors (semantic values)', 'api', 'core',
83
- 'Out of scope: the twin cannot run the embedding model — /v1/embeddings returns DETERMINISTIC pseudo-vectors seeded from the input hash. The shape, dimensions, and determinism are faithful; the values carry no semantic meaning.'),
84
- outOfScope('openai.images.real_pixels', 'images', 'Real image pixels (generated image bytes)', 'api', 'common',
85
- 'Out of scope: the twin cannot run the image model — /v1/images/generations returns the faithful response shape with a placeholder URL/revised_prompt; no real pixels are produced.'),
86
- outOfScope('openai.files.real_bytes', 'files', 'Real opaque file storage of arbitrary bytes', 'api', 'niche',
87
- 'Out of scope: the twin stores file metadata + supplied text content faithfully, but does not persist arbitrary binary blobs as a real object store would.'),
107
+ todo('openai.images.served_bytes', 'images', 'Image URLs resolve: the twin serves deterministic placeholder image bytes at the URLs it returns (today they point at the dead twin.invalid host)', 'api', 'common'),
108
+ todo('openai.files.binary_content', 'files', 'Files: persist and return arbitrary binary content, not only supplied text (a blob store, as the S3 twin already does)', 'api', 'niche'),
88
109
 
89
110
  // ── Chat Completions (the protocol envelope — faithful) ────────────────────────────────
90
111
  done('openai.chat.create', 'chat', 'Chat: create → faithful envelope (id/object/created/model/choices/usage)', 'api', 'core', () =>
@@ -423,7 +444,7 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
423
444
  done('openai.responses.stub_labeled', 'responses', 'Responses stub output is clearly labeled (not real output)', 'api', 'core', () =>
424
445
  withRoot(async (h) => {
425
446
  const r = await h({ m: 'POST', p: '/v1/responses', b: { model: 'gpt-4o', input: 'echo me responses' } });
426
- const text = (r.body as Body).output_text as string;
447
+ const text = outputText(r);
427
448
  return ok(r) && text.includes('[twin-stub') && text.includes('echo me responses');
428
449
  }),
429
450
  ),
@@ -440,11 +461,11 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
440
461
  return noModel.status === 400 && noInput.status === 400;
441
462
  }),
442
463
  ),
443
- done('openai.responses.streaming', 'responses', 'Responses streaming events (created → output_text.delta → completed → [DONE])', 'api', 'common', () =>
464
+ done('openai.responses.streaming', 'responses', 'Responses streaming events (created → output_text.delta → completed, no sentinel)', 'api', 'common', () =>
444
465
  withStream({ model: 'gpt-4o', input: 'stream responses', stream: true }, (events) => {
445
466
  const types = events.filter((e) => !e.done).map((e) => e.data!.type);
446
- const doneLast = events[events.length - 1]?.done === true;
447
- return types.includes('response.created') && types.includes('response.output_text.delta') && types.includes('response.completed') && doneLast;
467
+ const completedLast = types[types.length - 1] === 'response.completed' && !events.some((e) => e.done);
468
+ return types.includes('response.created') && types.includes('response.output_text.delta') && completedLast;
448
469
  }),
449
470
  ),
450
471
  done('openai.responses.retrieve', 'responses', 'Responses: stored by default, retrieve + delete by id (+ 404)', 'api', 'common', () =>
@@ -452,7 +473,7 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
452
473
  const c = await h({ m: 'POST', p: '/v1/responses', b: { model: 'gpt-4o', input: 'store me' } });
453
474
  if (!ok(c)) return false;
454
475
  const g = await h({ m: 'GET', p: `/v1/responses/${id(c)}` });
455
- if (!ok(g) || id(g) !== id(c) || (g.body as Body).object !== 'response' || (g.body as Body).output_text !== (c.body as Body).output_text) return false;
476
+ if (!ok(g) || id(g) !== id(c) || (g.body as Body).object !== 'response' || outputText(g) !== outputText(c)) return false;
456
477
  const del = await h({ m: 'DELETE', p: `/v1/responses/${id(c)}` });
457
478
  if (!ok(del) || (del.body as Body).deleted !== true) return false;
458
479
  const after = await h({ m: 'GET', p: `/v1/responses/${id(c)}` });
@@ -463,17 +484,16 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
463
484
  return after.status === 404 && missing.status === 404 && nsGet.status === 404;
464
485
  }),
465
486
  ),
466
- outOfScope('openai.responses.tools', 'responses', 'Responses built-in tools (web_search/file_search/computer_use)', 'api', 'niche',
467
- 'Out of scope: the hosted built-in tools (web_search, computer_use, code_interpreter) require live model+infrastructure side effects the twin cannot reproduce offline — faking a web_search result or a computer_use action would fabricate output and violate the honest-stub contract. (User-defined function tools ARE modeled via openai.tools.*.)'),
487
+ todo('openai.responses.tools', 'responses', 'Responses built-in tools (web_search/file_search/computer_use/code_interpreter): the tool-call envelope with a deterministic, labeled twin-stub result, exactly as chat stubs generation (user-defined function tools are already modeled via openai.tools.*)', 'api', 'niche'),
468
488
  done('openai.responses.reasoning', 'responses', 'Responses reasoning items + reasoning.effort (faithful reasoning item shape; labeled stub summary)', 'api', 'niche', () =>
469
489
  withRoot(async (h) => {
470
- // no reasoning param → no reasoning item, no reasoning_tokens
471
- const plain = await h({ m: 'POST', p: '/v1/responses', b: { model: 'o4-mini', input: 'think about this' } });
490
+ // a model that does not reason → no reasoning item, no reasoning_tokens
491
+ const plain = await h({ m: 'POST', p: '/v1/responses', b: { model: 'gpt-4o', input: 'think about this' } });
472
492
  if (!ok(plain)) return false;
473
493
  if ((plain.body as Body).output.some((o: Body) => o.type === 'reasoning')) return false;
474
- if ((plain.body as Body).usage.output_tokens_details !== undefined) return false;
475
- // reasoning.effort:'high' → a reasoning item (labeled stub) BEFORE the message item + reasoning_tokens
476
- const r = await h({ m: 'POST', p: '/v1/responses', b: { model: 'o4-mini', input: 'think about this', reasoning: { effort: 'high' } } });
494
+ if ((plain.body as Body).usage.output_tokens_details?.reasoning_tokens !== 0) return false;
495
+ // reasoning.effort:'high' with a summary asked for → a reasoning item (labeled stub summary) BEFORE the message item + reasoning_tokens
496
+ const r = await h({ m: 'POST', p: '/v1/responses', b: { model: 'o4-mini', input: 'think about this', reasoning: { effort: 'high', summary: 'auto' } } });
477
497
  if (!ok(r)) return false;
478
498
  const b = r.body as Body;
479
499
  const items = b.output as Body[];
@@ -607,7 +627,9 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
607
627
  ),
608
628
  done('openai.models.delete', 'models', 'Delete a fine-tuned model (job output); base models refuse; 404 unknown', 'api', 'niche', () =>
609
629
  withRoot(async (h) => {
610
- const job = await h({ m: 'POST', p: '/v1/fine_tuning/jobs', b: { model: 'gpt-4o-mini', training_file: 'file-train' } });
630
+ const created = await h({ m: 'POST', p: '/v1/fine_tuning/jobs', b: { model: 'gpt-4o-mini', training_file: 'file-train' } });
631
+ // the model exists once the job has trained: a read observes that
632
+ const job = await h({ m: 'GET', p: `/v1/fine_tuning/jobs/${id(created)}` });
611
633
  const ftModel = (job.body as Body).fine_tuned_model as string;
612
634
  // the minted fine-tuned model is retrievable + listed
613
635
  const ret = await h({ m: 'GET', p: `/v1/models/${encodeURIComponent(ftModel)}` });
@@ -704,7 +726,10 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
704
726
  const r = await h({ m: 'POST', p: '/v1/batches', b: { input_file_id: 'file-x', endpoint: '/v1/chat/completions', completion_window: '24h' } });
705
727
  if (!ok(r) || (r.body as Body).object !== 'batch' || !String(id(r)).startsWith('batch-')) return false;
706
728
  const b = r.body as Body;
707
- return b.endpoint === '/v1/chat/completions' && b.status === 'completed' && typeof b.output_file_id === 'string';
729
+ if (b.endpoint !== '/v1/chat/completions' || b.status !== 'validating' || b.output_file_id !== null) return false;
730
+ // OpenAI works the batch on its own: a read finds it completed with its output file
731
+ const g = await h({ m: 'GET', p: `/v1/batches/${id(r)}` });
732
+ return field(g, 'status') === 'completed' && typeof field(g, 'output_file_id') === 'string';
708
733
  }),
709
734
  ),
710
735
  done('openai.batches.retrieve', 'batches', 'Batches: retrieve by id (+ 404 unknown)', 'api', 'core', () =>
@@ -722,12 +747,16 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
722
747
  return ok(l) && (l.body as Body).object === 'list' && (l.body as Body).data.length === 1;
723
748
  }),
724
749
  ),
725
- done('openai.batches.cancel', 'batches', 'Batches: cancel sets status cancelled (404 unknown)', 'api', 'common', () =>
750
+ done('openai.batches.cancel', 'batches', 'Batches: cancel an in-flight batch (cancelling, then cancelled; 409 once done; 404 unknown)', 'api', 'common', () =>
726
751
  withRoot(async (h) => {
727
752
  const c = await h({ m: 'POST', p: '/v1/batches', b: { input_file_id: 'file-x', endpoint: '/v1/chat/completions', completion_window: '24h' } });
728
753
  const cancel = await h({ m: 'POST', p: `/v1/batches/${id(c)}/cancel` });
754
+ if (!ok(cancel) || field(cancel, 'status') !== 'cancelling') return false;
755
+ const read = await h({ m: 'GET', p: `/v1/batches/${id(c)}` });
756
+ if (field(read, 'status') !== 'cancelled') return false;
757
+ const again = await h({ m: 'POST', p: `/v1/batches/${id(c)}/cancel` });
729
758
  const missing = await h({ m: 'POST', p: '/v1/batches/batch-nope/cancel' });
730
- return ok(cancel) && field(cancel, 'status') === 'cancelled' && missing.status === 404;
759
+ return again.status === 409 && missing.status === 404;
731
760
  }),
732
761
  ),
733
762
  done('openai.batches.validation', 'batches', 'Batches: input_file_id/endpoint/completion_window required', 'api', 'common', () =>
@@ -744,17 +773,18 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
744
773
  // a real, well-formed batch id + object/fields must exist — not merely "both sides undefined".
745
774
  if (typeof cid !== 'string' || !cid.startsWith('batch-')) return false;
746
775
  const cb = c.body as Body;
747
- if (cb.object !== 'batch' || cb.input_file_id !== 'file-x' || cb.status !== 'completed') return false;
776
+ if (cb.object !== 'batch' || cb.input_file_id !== 'file-x' || cb.status !== 'validating') return false;
748
777
  const g1 = await h({ m: 'GET', p: `/v1/batches/${cid}` });
749
778
  const g2 = await h({ m: 'GET', p: `/v1/batches/${cid}` });
750
779
  // the GET must actually round-trip the persisted resource's real fields (kernel-backed) —
751
780
  // not merely a matching-because-undefined id.
752
781
  const same = (r: OpenAIResponseEnvelope) => {
753
782
  const rb = r.body as Body;
754
- return rb.object === 'batch' && rb.input_file_id === cb.input_file_id && rb.endpoint === cb.endpoint &&
755
- rb.status === cb.status && rb.output_file_id === cb.output_file_id;
783
+ return rb.object === 'batch' && rb.input_file_id === cb.input_file_id && rb.endpoint === cb.endpoint;
756
784
  };
757
- return ok(g1) && ok(g2) && id(g1) === cid && id(g2) === cid && same(g1) && same(g2);
785
+ // the first read observes the vendor's work; the second reads what it stored
786
+ return ok(g1) && ok(g2) && id(g1) === cid && id(g2) === cid && same(g1) && same(g2) &&
787
+ field(g1, 'status') === 'completed' && field(g2, 'status') === 'completed' && field(g2, 'output_file_id') === field(g1, 'output_file_id');
758
788
  }),
759
789
  ),
760
790
 
@@ -764,7 +794,10 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
764
794
  const r = await h({ m: 'POST', p: '/v1/fine_tuning/jobs', b: { model: 'gpt-4o-mini', training_file: 'file-train' } });
765
795
  if (!ok(r) || (r.body as Body).object !== 'fine_tuning.job' || !String(id(r)).startsWith('ftjob-')) return false;
766
796
  const b = r.body as Body;
767
- return b.model === 'gpt-4o-mini' && b.status === 'succeeded' && String(b.fine_tuned_model).startsWith('ft:');
797
+ if (b.model !== 'gpt-4o-mini' || b.status !== 'validating_files' || b.fine_tuned_model !== null) return false;
798
+ // OpenAI trains on its own: a read finds the job succeeded with its model
799
+ const g = await h({ m: 'GET', p: `/v1/fine_tuning/jobs/${id(r)}` });
800
+ return field(g, 'status') === 'succeeded' && String(field(g, 'fine_tuned_model')).startsWith('ft:');
768
801
  }),
769
802
  ),
770
803
  done('openai.fine_tuning.retrieve', 'fine_tuning', 'Fine-tuning: retrieve + list jobs', 'api', 'common', () =>
@@ -867,10 +900,10 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
867
900
  const batch = await h({ m: 'POST', p: `/v1/vector_stores/${id(vs)}/file_batches`, b: { file_ids: ['file-a', 'file-b', 'file-c'] } });
868
901
  if (!ok(batch)) return false;
869
902
  const bb = batch.body as Body;
870
- if (bb.object !== 'vector_store.file_batch' || bb.status !== 'completed' || bb.file_counts?.total !== 3) return false;
871
- // retrieve the batch
903
+ if (bb.object !== 'vector_store.files_batch' || bb.status !== 'in_progress' || bb.file_counts?.total !== 3) return false;
904
+ // retrieve the batch: the read observes OpenAI's processing done
872
905
  const get = await h({ m: 'GET', p: `/v1/vector_stores/${id(vs)}/file_batches/${bb.id}` });
873
- if (!ok(get) || (get.body as Body).id !== bb.id) return false;
906
+ if (!ok(get) || (get.body as Body).id !== bb.id || (get.body as Body).status !== 'completed') return false;
874
907
  // list this batch's files
875
908
  const files = await h({ m: 'GET', p: `/v1/vector_stores/${id(vs)}/file_batches/${bb.id}/files` });
876
909
  if (!ok(files) || ((files.body as Body).data as Body[]).length !== 3) return false;
@@ -900,56 +933,65 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
900
933
  return r.status === 400 && (r.body as Body).error?.param === 'prompt';
901
934
  }),
902
935
  ),
903
- done('openai.images.edits', 'images', 'Image edits (image+prompt) + variations (image) → faithful response shape', 'api', 'niche', () =>
936
+ done('openai.images.edits', 'images', 'Image edits (an image file + prompt; base64 PNG from a GPT image model) + variations (a square PNG) → faithful response shape; a text upload refused', 'api', 'niche', () =>
904
937
  withRoot(async (h) => {
905
- const edit = await h({ m: 'POST', p: '/v1/images/edits', b: { image: 'base.png', prompt: 'add a hat' } });
906
- if (!ok(edit) || typeof (edit.body as Body).created !== 'number' || typeof (edit.body as Body).data[0].url !== 'string') return false;
907
- const varn = await h({ m: 'POST', p: '/v1/images/variations', b: { image: 'base.png', n: 2 } });
938
+ const png = placeholderPng({ width: 64, height: 64 }, 'base');
939
+ const edit = await h({ m: 'POST', p: '/v1/images/edits', b: form({ model: 'gpt-image-1', prompt: 'add a hat' }, 'image', png, 'base.png', 'image/png') });
940
+ const b64 = (edit.body as Body).data?.[0]?.b64_json;
941
+ if (!ok(edit) || typeof (edit.body as Body).created !== 'number' || typeof b64 !== 'string' || imageFormat(Uint8Array.from(atob(b64), (c) => c.charCodeAt(0))) !== 'png') return false;
942
+ const varn = await h({ m: 'POST', p: '/v1/images/variations', b: form({ n: '2' }, 'image', png, 'base.png', 'image/png') });
908
943
  if (!ok(varn) || ((varn.body as Body).data as Body[]).length !== 2) return false;
909
- // validation: edits require image AND prompt; variations require image
944
+ // validation: edits require image AND prompt, an image file; variations require a square png
910
945
  const noImg = await h({ m: 'POST', p: '/v1/images/edits', b: { prompt: 'x' } });
911
- const noPrompt = await h({ m: 'POST', p: '/v1/images/edits', b: { image: 'base.png' } });
912
- const varNoImg = await h({ m: 'POST', p: '/v1/images/variations', b: {} });
913
- return noImg.status === 400 && noPrompt.status === 400 && varNoImg.status === 400;
946
+ const noPrompt = await h({ m: 'POST', p: '/v1/images/edits', b: form({ model: 'gpt-image-1' }, 'image', png, 'base.png', 'image/png') });
947
+ const text = await h({ m: 'POST', p: '/v1/images/edits', b: form({ model: 'gpt-image-1', prompt: 'x' }, 'image', new TextEncoder().encode('PNG-base'), 'base.png', 'image/png') });
948
+ const wide = await h({ m: 'POST', p: '/v1/images/variations', b: form({}, 'image', placeholderPng({ width: 64, height: 32 }, 'wide'), 'wide.png', 'image/png') });
949
+ return noImg.status === 400 && noPrompt.status === 400 && text.status === 400 && (text.body as Body).error?.param === 'image' && wide.status === 400;
914
950
  }),
915
951
  ),
916
952
 
917
953
  // ── Audio ─────────────────────────────────────────────────────────────────────────────
918
- done('openai.audio.transcriptions', 'audio', 'Audio transcriptions: labeled stub transcript, json + verbose_json + text shapes', 'api', 'common', () =>
954
+ done('openai.audio.transcriptions', 'audio', 'Audio transcriptions: labeled stub transcript of an audio file, json + verbose_json + text shapes; a text upload refused', 'api', 'common', () =>
919
955
  withRoot(async (h) => {
920
- const j = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: { file: 'speech.mp3', model: 'whisper-1' } });
956
+ const speech = (extra: Record<string, string>) => form({ model: 'whisper-1', ...extra }, 'file', SILENT_WAV, 'speech.wav', 'audio/wav');
957
+ const j = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: speech({}) });
921
958
  if (!ok(j) || typeof (j.body as Body).text !== 'string' || !String((j.body as Body).text).includes('[twin-stub')) return false;
922
- const v = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: { file: 'speech.mp3', model: 'whisper-1', response_format: 'verbose_json' } });
923
- if (!ok(v) || (v.body as Body).task !== 'transcription' || !Array.isArray((v.body as Body).segments)) return false;
924
- const t = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: { file: 'speech.mp3', model: 'whisper-1', response_format: 'text' } });
959
+ const v = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: speech({ response_format: 'verbose_json' }) });
960
+ if (!ok(v) || (v.body as Body).task !== 'transcribe' || !Array.isArray((v.body as Body).segments)) return false;
961
+ const t = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: speech({ response_format: 'text' }) });
925
962
  if (!ok(t) || typeof t.body !== 'string') return false;
926
963
  const noFile = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: { model: 'whisper-1' } });
927
- const noModel = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: { file: 'speech.mp3' } });
928
- return noFile.status === 400 && (noFile.body as Body).error?.param === 'file' && noModel.status === 400;
964
+ const noModel = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: form({}, 'file', SILENT_WAV, 'speech.wav', 'audio/wav') });
965
+ const text = await h({ m: 'POST', p: '/v1/audio/transcriptions', b: form({ model: 'whisper-1' }, 'file', new TextEncoder().encode('M4A-speech'), 'speech.m4a', 'audio/mp4') });
966
+ return noFile.status === 400 && (noFile.body as Body).error?.param === 'file' && noModel.status === 400 && text.status === 400 && (text.body as Body).error?.param === 'file';
929
967
  }),
930
968
  ),
931
969
  done('openai.audio.translations', 'audio', 'Audio translations: labeled stub, task translation, english language', 'api', 'niche', () =>
932
970
  withRoot(async (h) => {
933
- const v = await h({ m: 'POST', p: '/v1/audio/translations', b: { file: 'foreign.mp3', model: 'whisper-1', response_format: 'verbose_json' } });
971
+ const v = await h({ m: 'POST', p: '/v1/audio/translations', b: form({ model: 'whisper-1', response_format: 'verbose_json' }, 'file', SILENT_WAV, 'foreign.wav', 'audio/wav') });
934
972
  const noFile = await h({ m: 'POST', p: '/v1/audio/translations', b: { model: 'whisper-1' } });
935
973
  return ok(v) && (v.body as Body).task === 'translation' && (v.body as Body).language === 'english' && noFile.status === 400;
936
974
  }),
937
975
  ),
938
- done('openai.audio.speech', 'audio', 'Text-to-speech: deterministic labeled audio payload (+ model/input/voice required)', 'api', 'common', () =>
939
- withRoot(async (h) => {
940
- const r = await h({ m: 'POST', p: '/v1/audio/speech', b: { model: 'tts-1', input: 'hello world', voice: 'alloy' } });
941
- if (!ok(r) || (r.body as Body).object !== 'audio.speech' || typeof (r.body as Body).audio_base64 !== 'string') return false;
942
- // deterministic: same input → same payload; decoded marker labels it a stub
943
- const r2 = await h({ m: 'POST', p: '/v1/audio/speech', b: { model: 'tts-1', input: 'hello world', voice: 'alloy' } });
944
- if ((r.body as Body).audio_base64 !== (r2.body as Body).audio_base64) return false;
945
- if (!atob((r.body as Body).audio_base64 as string).includes('[twin-stub-audio]')) return false;
946
- const noVoice = await h({ m: 'POST', p: '/v1/audio/speech', b: { model: 'tts-1', input: 'x' } });
947
- const noInput = await h({ m: 'POST', p: '/v1/audio/speech', b: { model: 'tts-1', voice: 'alloy' } });
948
- return noVoice.status === 400 && (noVoice.body as Body).error?.param === 'voice' && noInput.status === 400;
949
- }),
950
- ),
951
- outOfScope('openai.audio.realtime', 'audio', 'Realtime API (websocket sessions)', 'api', 'niche',
952
- 'Out of scope: the Realtime API is a stateful bidirectional WebSocket carrying live audio to/from a running speech model — both the transport and the real-time model inference are irreproducible offline. The twin runs no model and the verify() harness is HTTP/offline, so a faithful realtime session cannot be modeled without fabrication.'),
976
+ done('openai.audio.speech', 'audio', 'Text-to-speech: the audio file itself in the requested format with its Content-Type (mp3 by default, wav), a real file of silence (+ model/input/voice required)', 'api', 'common', async () => {
977
+ const root = mkdtempSync(join(tmpdir(), 'openai-cap-'));
978
+ try {
979
+ return await verifyBoundary('openai.audio.speech', async () => {
980
+ const fetch = createOpenAITwinFetch({ root });
981
+ const speak = (b: unknown) => fetch(new Request('https://api.openai.com/v1/audio/speech', { method: 'POST', headers: { authorization: 'Bearer sk-twin', 'content-type': 'application/json' }, body: JSON.stringify(b) }));
982
+ const mp3 = await speak({ model: 'tts-1', input: 'hello world', voice: 'alloy' });
983
+ if (mp3.status !== 200 || mp3.headers.get('content-type') !== 'audio/mpeg' || audioFormat(new Uint8Array(await mp3.arrayBuffer())) !== 'mp3') return false;
984
+ const wav = await speak({ model: 'tts-1', input: 'hello world', voice: 'alloy', response_format: 'wav' });
985
+ if (wav.headers.get('content-type') !== 'audio/wav' || wavSeconds(new Uint8Array(await wav.arrayBuffer())) !== 1) return false;
986
+ const noVoice = await speak({ model: 'tts-1', input: 'x' });
987
+ const noInput = await speak({ model: 'tts-1', voice: 'alloy' });
988
+ return noVoice.status === 400 && ((await noVoice.json()) as Body).error?.param === 'voice' && noInput.status === 400;
989
+ });
990
+ } finally {
991
+ rmSync(root, { recursive: true, force: true });
992
+ }
993
+ }),
994
+ todo('openai.audio.realtime', 'audio', 'Realtime API: the stateful bidirectional websocket session protocol with deterministic labeled stub audio/text events (the repo serves other non-HTTP wire protocols over the kernel)', 'api', 'niche'),
953
995
 
954
996
  // ── Assistants (beta) ─────────────────────────────────────────────────────────────────
955
997
  done('openai.assistants.crud', 'assistants', 'Assistants: create/retrieve/update/list/delete (+ model required, 404)', 'api', 'niche', () =>
@@ -1010,17 +1052,17 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1010
1052
  const noAsst = await h({ m: 'POST', p: `/v1/threads/${id(t)}/runs`, b: {} });
1011
1053
  const badAsst = await h({ m: 'POST', p: `/v1/threads/${id(t)}/runs`, b: { assistant_id: 'asst-nope' } });
1012
1054
  if (noAsst.status !== 400 || badAsst.status !== 404) return false;
1013
- // create a run → completed, stub assistant reply appended to the thread
1055
+ // create a run → queued, as OpenAI queues it
1014
1056
  const run = await h({ m: 'POST', p: `/v1/threads/${id(t)}/runs`, b: { assistant_id: id(a) } });
1015
- if (!ok(run) || !String(id(run)).startsWith('run-') || (run.body as Body).object !== 'thread.run' || (run.body as Body).status !== 'completed') return false;
1057
+ if (!ok(run) || !String(id(run)).startsWith('run-') || (run.body as Body).object !== 'thread.run' || (run.body as Body).status !== 'queued') return false;
1016
1058
  if ((run.body as Body).assistant_id !== id(a) || (run.body as Body).thread_id !== id(t)) return false;
1059
+ // retrieving it observes the run completing
1060
+ const gr = await h({ m: 'GET', p: `/v1/threads/${id(t)}/runs/${id(run)}` });
1061
+ if (!ok(gr) || id(gr) !== id(run) || field(gr, 'status') !== 'completed') return false;
1017
1062
  // the thread now has the user turn + the stub assistant reply
1018
1063
  const msgs = ((await h({ m: 'GET', p: `/v1/threads/${id(t)}/messages` })).body as Body).data as Body[];
1019
1064
  const assistantMsg = msgs.find((mm) => mm.role === 'assistant');
1020
1065
  if (!assistantMsg || !String(assistantMsg.content[0].text.value).includes('[twin-stub')) return false;
1021
- // retrieve the run + list runs
1022
- const gr = await h({ m: 'GET', p: `/v1/threads/${id(t)}/runs/${id(run)}` });
1023
- if (!ok(gr) || id(gr) !== id(run)) return false;
1024
1066
  const lr = await h({ m: 'GET', p: `/v1/threads/${id(t)}/runs` });
1025
1067
  if (((lr.body as Body).data as Body[]).length !== 1) return false;
1026
1068
  // run steps: a message_creation step pointing at the reply message
@@ -1030,9 +1072,12 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1030
1072
  if (sd[0]!.step_details.message_creation.message_id !== assistantMsg.id) return false;
1031
1073
  const gs = await h({ m: 'GET', p: `/v1/threads/${id(t)}/runs/${id(run)}/steps/${sd[0]!.id}` });
1032
1074
  if (!ok(gs) || (gs.body as Body).id !== sd[0]!.id) return false;
1033
- // cancel
1034
- const cancel = await h({ m: 'POST', p: `/v1/threads/${id(t)}/runs/${id(run)}/cancel` });
1035
- if (!ok(cancel) || (cancel.body as Body).status !== 'cancelled') return false;
1075
+ // cancel a fresh run while it is queued → cancelling, then cancelled; a completed run refuses
1076
+ const run2 = await h({ m: 'POST', p: `/v1/threads/${id(t)}/runs`, b: { assistant_id: id(a) } });
1077
+ const cancel = await h({ m: 'POST', p: `/v1/threads/${id(t)}/runs/${id(run2)}/cancel` });
1078
+ if (!ok(cancel) || field(cancel, 'status') !== 'cancelling') return false;
1079
+ if (field(await h({ m: 'GET', p: `/v1/threads/${id(t)}/runs/${id(run2)}` }), 'status') !== 'cancelled') return false;
1080
+ if ((await h({ m: 'POST', p: `/v1/threads/${id(t)}/runs/${id(run)}/cancel` })).status !== 400) return false;
1036
1081
  const badRun = await h({ m: 'GET', p: `/v1/threads/${id(t)}/runs/run-nope` });
1037
1082
  return badRun.status === 404;
1038
1083
  }),
@@ -1074,11 +1119,11 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1074
1119
  if ((upd.body as Body).name !== 'Alpha2') return false;
1075
1120
  const arch = await h({ m: 'POST', p: `/v1/organization/projects/${id(c)}/archive` });
1076
1121
  if (!ok(arch) || (arch.body as Body).status !== 'archived') return false;
1077
- // archived project hidden by default, visible with include_archived
1122
+ // archived project hidden by default, visible with include_archived; the organization's Default project stays
1078
1123
  const def = ((await h({ m: 'GET', p: '/v1/organization/projects' })).body as Body).data as Body[];
1079
1124
  const withArch = ((await h({ m: 'GET', p: '/v1/organization/projects?include_archived=true' })).body as Body).data as Body[];
1080
1125
  const missing = await h({ m: 'GET', p: '/v1/organization/projects/proj-nope' });
1081
- return def.length === 0 && withArch.length === 1 && missing.status === 404;
1126
+ return def.length === 1 && def[0]!.name === 'Default project' && withArch.length === 2 && missing.status === 404;
1082
1127
  }),
1083
1128
  ),
1084
1129
  done('openai.admin.api_keys', 'admin', 'Admin: project API keys create (one-time secret) / list / retrieve (redacted) / delete', 'connector', 'niche', () =>
@@ -1103,17 +1148,19 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1103
1148
  ),
1104
1149
  done('openai.usage.costs', 'admin', 'Usage + costs reporting endpoints computed over REAL recorded usage', 'api', 'niche', () =>
1105
1150
  withRoot(async (h) => {
1106
- // with no traffic the report is genuinely EMPTY (nothing hardcoded)
1107
- const empty = await h({ m: 'GET', p: '/v1/organization/usage/completions' });
1108
- if (!ok(empty) || (empty.body as Body).object !== 'page' || ((empty.body as Body).data as Body[]).length !== 0) return false;
1109
- const emptyCosts = await h({ m: 'GET', p: '/v1/organization/costs' });
1110
- if (((emptyCosts.body as Body).data as Body[]).length !== 0) return false;
1151
+ // today, one daily bucket; with no traffic its results are genuinely EMPTY (nothing hardcoded)
1152
+ const day = Date.parse(CHECKS_AT) / 1000;
1153
+ const TODAY = `start_time=${day}&end_time=${day + 86_400}`;
1154
+ const empty = await h({ m: 'GET', p: `/v1/organization/usage/completions?${TODAY}&group_by=model` });
1155
+ if (!ok(empty) || (empty.body as Body).object !== 'page' || ((empty.body as Body).data as Body[]).length !== 1 || ((empty.body as Body).data[0].results as Body[]).length !== 0) return false;
1156
+ const emptyCosts = await h({ m: 'GET', p: `/v1/organization/costs?${TODAY}&group_by=line_item` });
1157
+ if (((emptyCosts.body as Body).data[0].results as Body[]).length !== 0) return false;
1111
1158
  // generate real usage: two chat calls (gpt-4o) + one embeddings call
1112
1159
  const c1 = await h({ m: 'POST', p: '/v1/chat/completions', b: CHAT() });
1113
1160
  const c2 = await h({ m: 'POST', p: '/v1/chat/completions', b: CHAT({ messages: [{ role: 'user', content: 'another prompt entirely' }] }) });
1114
1161
  await h({ m: 'POST', p: '/v1/embeddings', b: { model: 'text-embedding-3-small', input: 'embed me' } });
1115
1162
  // completions usage aggregates the two chat calls under gpt-4o
1116
- const usage = await h({ m: 'GET', p: '/v1/organization/usage/completions' });
1163
+ const usage = await h({ m: 'GET', p: `/v1/organization/usage/completions?${TODAY}&group_by=model` });
1117
1164
  const buckets = (usage.body as Body).data as Body[];
1118
1165
  if (buckets.length !== 1) return false;
1119
1166
  const results = buckets[0]!.results as Body[];
@@ -1125,11 +1172,11 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1125
1172
  if (gpt4o.input_tokens !== expectedIn || gpt4o.output_tokens !== expectedOut) return false;
1126
1173
  // embeddings usage is filtered out of the completions view, present under embeddings
1127
1174
  if (results.some((r) => r.model.startsWith('text-embedding'))) return false;
1128
- const emb = await h({ m: 'GET', p: '/v1/organization/usage/embeddings' });
1175
+ const emb = await h({ m: 'GET', p: `/v1/organization/usage/embeddings?${TODAY}&group_by=model` });
1129
1176
  const embResults = ((emb.body as Body).data as Body[])[0]!.results as Body[];
1130
1177
  if (!embResults.some((r) => r.model === 'text-embedding-3-small' && r.object === 'organization.usage.embeddings.result')) return false;
1131
1178
  // costs: a non-zero amount computed from the recorded usage (not hardcoded)
1132
- const costs = await h({ m: 'GET', p: '/v1/organization/costs' });
1179
+ const costs = await h({ m: 'GET', p: `/v1/organization/costs?${TODAY}&group_by=line_item` });
1133
1180
  const costResults = ((costs.body as Body).data as Body[])[0]!.results as Body[];
1134
1181
  const completionsCost = costResults.find((r) => r.line_item === 'completions');
1135
1182
  if (!completionsCost || completionsCost.object !== 'organization.costs.result' || completionsCost.amount.currency !== 'usd' || !(completionsCost.amount.value > 0)) return false;
@@ -1154,9 +1201,12 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1154
1201
  if (dsc?.type !== 'custom') return false;
1155
1202
  const list = await h({ m: 'GET', p: '/v1/evals' });
1156
1203
  if (!ok(list) || !(list.body as Body).data.some((e: Body) => e.id === evalId)) return false;
1157
- // kick off a run; assert completed + result_counts + per_model_usage reflects the run model
1158
- const run = await h({ m: 'POST', p: `/v1/evals/${evalId}/runs`, b: { name: 'run-1', data_source: { type: 'completions', model: 'gpt-4.1', source: { type: 'file_content', content: [] } } } });
1159
- if (!ok(run) || field(run, 'object') !== 'eval.run' || field(run, 'status') !== 'completed') return false;
1204
+ // kick off a run (queued, as OpenAI queues it); a read observes it graded: completed +
1205
+ // result_counts + per_model_usage reflects the run model
1206
+ const created = await h({ m: 'POST', p: `/v1/evals/${evalId}/runs`, b: { name: 'run-1', data_source: { type: 'completions', model: 'gpt-4.1', source: { type: 'file_content', content: [] } } } });
1207
+ if (!ok(created) || field(created, 'object') !== 'eval.run' || field(created, 'status') !== 'queued') return false;
1208
+ const run = await h({ m: 'GET', p: `/v1/evals/${evalId}/runs/${id(created)}` });
1209
+ if (!ok(run) || field(run, 'status') !== 'completed') return false;
1160
1210
  const rc = field(run, 'result_counts') as Body;
1161
1211
  if (rc.passed !== 1 || rc.total !== 1 || field(run, 'eval_id') !== evalId) return false;
1162
1212
  // the run echoes the model from the request data_source (not a hardcoded default)
@@ -1192,9 +1242,10 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1192
1242
  if (!ok(got) || id(got) !== cid) return false;
1193
1243
  const list = await h({ m: 'GET', p: '/v1/containers' });
1194
1244
  if (!ok(list) || !(list.body as Body).data.some((x: Body) => x.id === cid)) return false;
1195
- // container file with inline text content → read back metadata + content
1196
- const cf = await h({ m: 'POST', p: `/v1/containers/${cid}/files`, b: { path: '/mnt/data/a.txt', content: 'hello sandbox' } });
1197
- if (!ok(cf) || field(cf, 'object') !== 'container.file' || field(cf, 'container_id') !== cid || field(cf, 'bytes') !== 'hello sandbox'.length || field(cf, 'path') !== '/mnt/data/a.txt') return false;
1245
+ // container file from a stored File (the JSON form OpenAI takes) → read back metadata + content
1246
+ const src = await h({ m: 'POST', p: '/v1/files', b: { purpose: 'user_data', filename: 'a.txt', content: 'hello sandbox' } });
1247
+ const cf = await h({ m: 'POST', p: `/v1/containers/${cid}/files`, b: { file_id: id(src) } });
1248
+ if (!ok(cf) || field(cf, 'object') !== 'container.file' || field(cf, 'container_id') !== cid || field(cf, 'bytes') !== 'hello sandbox'.length || field(cf, 'source') !== 'file_id') return false;
1198
1249
  const fid = id(cf);
1199
1250
  const content = await h({ m: 'GET', p: `/v1/containers/${cid}/files/${fid}/content` });
1200
1251
  if (!ok(content) || content.body !== 'hello sandbox') return false;
@@ -1215,36 +1266,42 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1215
1266
  ),
1216
1267
 
1217
1268
  // ── Background responses (queued → poll → completed | cancel) ──────────────────────────
1218
- done('openai.responses.background', 'responses', 'Background responses (background:true → queued status + poll + cancel)', 'api', 'niche', () =>
1269
+ done('openai.responses.background', 'responses', 'Background responses (background:true → queued, then completed over time on the World clock + poll + cancel)', 'api', 'niche', () =>
1219
1270
  withRoot(async (h) => {
1271
+ // the World instants of the calls: created, then polled an hour later, when OpenAI's work is done
1272
+ const T = '2026-03-01T00:00:00.000Z';
1273
+ const LATER = '2026-03-01T01:00:00.000Z';
1220
1274
  // background:true → immediate queued response with no output yet
1221
- const bg = await h({ m: 'POST', p: '/v1/responses', b: { model: 'gpt-4o', input: 'background task A', background: true } });
1275
+ const bg = await h({ m: 'POST', p: '/v1/responses', b: { model: 'gpt-4o', input: 'background task A', background: true }, at: T });
1222
1276
  if (!ok(bg) || field(bg, 'status') !== 'queued' || field(bg, 'background') !== true) return false;
1223
- if (field(bg, 'output_text') !== null || ((field(bg, 'output') as unknown[]) ?? []).length !== 0) return false;
1277
+ if (((field(bg, 'output') as unknown[]) ?? []).length !== 0) return false;
1224
1278
  const rid = id(bg);
1225
- // first poll → transitions to completed with the real stub output + usage
1226
- const poll = await h({ m: 'GET', p: `/v1/responses/${rid}` });
1279
+ // polled at the instant it was made, it is still queued; an hour later it has completed with the stub output + usage
1280
+ const early = await h({ m: 'GET', p: `/v1/responses/${rid}`, at: T });
1281
+ if (!ok(early) || field(early, 'status') !== 'queued') return false;
1282
+ const poll = await h({ m: 'GET', p: `/v1/responses/${rid}`, at: LATER });
1227
1283
  if (!ok(poll) || field(poll, 'status') !== 'completed') return false;
1228
- const text = field(poll, 'output_text');
1229
- if (typeof text !== 'string' || !text.includes('[twin-stub')) return false;
1284
+ const text = outputText(poll);
1285
+ if (!text.includes('[twin-stub')) return false;
1230
1286
  // usage must be coherent: total = input + output, and output > 0 (the stub produced text)
1231
1287
  const usage = field(poll, 'usage') as Body;
1232
1288
  if (!usage || typeof usage.output_tokens !== 'number' || usage.output_tokens <= 0) return false;
1233
1289
  if (usage.total_tokens !== usage.input_tokens + usage.output_tokens) return false;
1234
1290
  // the completion is billable: a usage_record was recorded → the usage report is non-empty
1235
- const report = await h({ m: 'GET', p: '/v1/organization/usage/responses' });
1291
+ const day = Date.parse(T) / 1000;
1292
+ const report = await h({ m: 'GET', p: `/v1/organization/usage/responses?start_time=${day}&end_time=${day + 86_400}`, at: LATER });
1236
1293
  const buckets = (report.body as Body).data as Body[];
1237
- if (!ok(report) || buckets.length === 0) return false;
1294
+ if (!ok(report) || buckets.length === 0 || (buckets[0]!.results as Body[]).length === 0) return false;
1238
1295
  // a second poll stays completed (idempotent) with the same output text (no re-billing visible)
1239
- const poll2 = await h({ m: 'GET', p: `/v1/responses/${rid}` });
1240
- if (field(poll2, 'status') !== 'completed' || field(poll2, 'output_text') !== text) return false;
1296
+ const poll2 = await h({ m: 'GET', p: `/v1/responses/${rid}`, at: LATER });
1297
+ if (field(poll2, 'status') !== 'completed' || outputText(poll2) !== text) return false;
1241
1298
  // cancel a still-queued background response → cancelled
1242
- const bg2 = await h({ m: 'POST', p: '/v1/responses', b: { model: 'gpt-4o', input: 'background task B', background: true } });
1299
+ const bg2 = await h({ m: 'POST', p: '/v1/responses', b: { model: 'gpt-4o', input: 'background task B', background: true }, at: LATER });
1243
1300
  const rid2 = id(bg2);
1244
- const cancel = await h({ m: 'POST', p: `/v1/responses/${rid2}/cancel` });
1301
+ const cancel = await h({ m: 'POST', p: `/v1/responses/${rid2}/cancel`, at: LATER });
1245
1302
  if (!ok(cancel) || field(cancel, 'status') !== 'cancelled') return false;
1246
1303
  // negative: cancel an already-completed response → 400; cancel unknown id → 404
1247
- const reCancel = await h({ m: 'POST', p: `/v1/responses/${rid}/cancel` });
1304
+ const reCancel = await h({ m: 'POST', p: `/v1/responses/${rid}/cancel`, at: LATER });
1248
1305
  const missing = await h({ m: 'POST', p: '/v1/responses/resp-nope/cancel' });
1249
1306
  return reCancel.status === 400 && missing.status === 404;
1250
1307
  }),
@@ -1254,16 +1311,16 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1254
1311
  done('openai.fine_tuning.pause_resume', 'fine_tuning', 'Fine-tuning job pause / resume — status transitions', 'api', 'niche', () =>
1255
1312
  withRoot(async (h) => {
1256
1313
  const job = await h({ m: 'POST', p: '/v1/fine_tuning/jobs', b: { model: 'gpt-4o-mini', training_file: 'file-twin-1' } });
1257
- if (!ok(job) || field(job, 'status') !== 'succeeded') return false;
1314
+ if (!ok(job) || field(job, 'status') !== 'validating_files') return false;
1258
1315
  const jid = id(job);
1259
- // pause → status 'paused'
1316
+ // pause while in flight → status 'paused' (a paused job does not progress on reads)
1260
1317
  const paused = await h({ m: 'POST', p: `/v1/fine_tuning/jobs/${jid}/pause` });
1261
1318
  if (!ok(paused) || field(paused, 'status') !== 'paused') return false;
1262
1319
  const readPaused = await h({ m: 'GET', p: `/v1/fine_tuning/jobs/${jid}` });
1263
1320
  if (field(readPaused, 'status') !== 'paused') return false;
1264
- // resume → back to 'succeeded', and the persisted state round-trips
1321
+ // resume → queued again; the next read observes it finishing
1265
1322
  const resumed = await h({ m: 'POST', p: `/v1/fine_tuning/jobs/${jid}/resume` });
1266
- if (!ok(resumed) || field(resumed, 'status') !== 'succeeded') return false;
1323
+ if (!ok(resumed) || field(resumed, 'status') !== 'queued') return false;
1267
1324
  const readResumed = await h({ m: 'GET', p: `/v1/fine_tuning/jobs/${jid}` });
1268
1325
  if (field(readResumed, 'status') !== 'succeeded') return false;
1269
1326
  // negative: resuming the now-succeeded (non-paused) job → 400; pause unknown job → 404
@@ -1375,26 +1432,6 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1375
1432
  return hdrs['x-ratelimit-remaining-requests'] === '0' && typeof hdrs['x-ratelimit-limit-tokens'] === 'string';
1376
1433
  }),
1377
1434
  ),
1378
- done('openai.protocol.idempotency', 'errors', 'Idempotency-Key header dedup — re-issue replays the same response; a new key → a new result', 'api', 'niche', () =>
1379
- withRootH(async (h) => {
1380
- const auth = { authorization: 'Bearer sk-twin-good' };
1381
- // a mutation (file create) with an Idempotency-Key → re-issue replays the SAME id (no double create)
1382
- const first = await h({ m: 'POST', p: '/v1/files', b: { purpose: 'batch', filename: 'a.jsonl', content: '{}' }, headers: { ...auth, 'idempotency-key': 'key-A' } });
1383
- if (!ok(first)) return false;
1384
- const id1 = (first.body as Body).id as string;
1385
- const replay = await h({ m: 'POST', p: '/v1/files', b: { purpose: 'batch', filename: 'a.jsonl', content: '{}' }, headers: { ...auth, 'idempotency-key': 'key-A' } });
1386
- if (!ok(replay) || (replay.body as Body).id !== id1) return false;
1387
- // the side effect happened ONCE: exactly one file exists
1388
- const listed = (await h({ m: 'GET', p: '/v1/files', headers: auth })).body as Body;
1389
- if ((listed.data as Body[]).length !== 1) return false;
1390
- // a DIFFERENT key → a new, distinct create
1391
- const second = await h({ m: 'POST', p: '/v1/files', b: { purpose: 'batch', filename: 'a.jsonl', content: '{}' }, headers: { ...auth, 'idempotency-key': 'key-B' } });
1392
- if (!ok(second) || (second.body as Body).id === id1) return false;
1393
- const listed2 = (await h({ m: 'GET', p: '/v1/files', headers: auth })).body as Body;
1394
- // even the replayed body is byte-identical to the first response (full envelope replay)
1395
- return (listed2.data as Body[]).length === 2 && JSON.stringify(replay.body) === JSON.stringify(first.body);
1396
- }),
1397
- ),
1398
1435
  done('openai.protocol.pagination', 'errors', 'Cursor pagination: after/limit + has_more/first_id/last_id across list endpoints', 'api', 'common', () =>
1399
1436
  withRoot(async (h) => {
1400
1437
  // create 3 files; page with limit=2 → has_more + last_id; then after=last_id → final page
@@ -1424,7 +1461,7 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1424
1461
  ),
1425
1462
  done('openai.connector.pull_map', 'connector', 'Connector pulls + maps real state into the twin (offline, injected client)', 'connector', 'core', async () => {
1426
1463
  const { mapBatch, mapFile, syncOpenAIFromReal } = await import('./openai-connector.ts');
1427
- const { projectResources } = await import('@volter/twin');
1464
+ const { projectResources } = await import('@volter/world-core');
1428
1465
  const root = mkdtempSync(join(tmpdir(), 'openai-cap-'));
1429
1466
  try {
1430
1467
  const mappedFile = mapFile({ id: 'file-real', purpose: 'batch', bytes: 12, status: 'processed', created_at: 1 });
@@ -1447,31 +1484,6 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1447
1484
  rmSync(root, { recursive: true, force: true });
1448
1485
  }
1449
1486
  }),
1450
- done('openai.connector.push_confirm', 'connector', 'Connector pushes pending local writes to real + confirms them (offline, injected client)', 'connector', 'core', async () => {
1451
- const { pushPendingOpenAIActions, openaiRequestForAction } = await import('./openai-connector.ts');
1452
- const { pendingActions } = await import('@volter/twin');
1453
- const root = mkdtempSync(join(tmpdir(), 'openai-cap-'));
1454
- try {
1455
- const created = await handleOpenAITwinRequest({ method: 'POST', path: '/v1/batches', body: JSON.stringify({ input_file_id: 'file-x', endpoint: '/v1/chat/completions', completion_window: '24h' }), root, occurredAt: '2026-06-15T00:00:00Z' });
1456
- if (created.status !== 200) return false;
1457
- if (pendingActions('openai', root).length === 0) return false;
1458
- const createReq = openaiRequestForAction({ operation: 'batch.create', subject: { type: 'batch', id: 'batch-twin-1' } });
1459
- const cancelReq = openaiRequestForAction({ operation: 'batch.cancel', subject: { type: 'batch', id: 'batch-twin-1' } });
1460
- const delReq = openaiRequestForAction({ operation: 'file.delete', subject: { type: 'file', id: 'file-twin-1' } });
1461
- if (createReq.path !== '/v1/batches' || cancelReq.path !== '/v1/batches/batch-twin-1/cancel' || delReq.method !== 'DELETE') return false;
1462
- const calls: string[] = [];
1463
- const fake = async (m: 'GET' | 'POST' | 'DELETE', p: string) => { calls.push(`${m} ${p}`); return { id: 'batch-real-1' }; };
1464
- const res = await pushPendingOpenAIActions(fake, { root, occurredAt: '2026-06-15T00:00:01Z' });
1465
- if (res.pushed < 1 || calls.length < 1) return false;
1466
- const after = await pushPendingOpenAIActions(fake, { root, occurredAt: '2026-06-15T00:00:02Z' });
1467
- return pendingActions('openai', root).length === 0 && after.pushed === 0;
1468
- } catch (err) {
1469
- if (isInfrastructureError(err)) throw harnessError('openai.connector.push_confirm', err);
1470
- return false;
1471
- } finally {
1472
- rmSync(root, { recursive: true, force: true });
1473
- }
1474
- }),
1475
1487
  done('openai.connector.unsupported_op_fails', 'connector', 'Connector refuses to silently drop an unsupported push op', 'connector', 'common', async () => {
1476
1488
  const { pushOpenAIAction } = await import('./openai-connector.ts');
1477
1489
  try {
@@ -1484,36 +1496,6 @@ export const OPENAI_CAPABILITIES: CapabilitySpec[] = [
1484
1496
  return false;
1485
1497
  }
1486
1498
  }),
1487
- done('openai.connector.full_sync', 'connector', 'Connector full bi-directional sync: push pending then pull all collections (idempotent)', 'connector', 'common', async () => {
1488
- const { fullSyncOpenAI } = await import('./openai-connector.ts');
1489
- const { pendingActions } = await import('@volter/twin');
1490
- const root = mkdtempSync(join(tmpdir(), 'openai-cap-'));
1491
- try {
1492
- // a local write becomes a pending action
1493
- const created = await handleOpenAITwinRequest({ method: 'POST', path: '/v1/batches', body: JSON.stringify({ input_file_id: 'file-x', endpoint: '/v1/chat/completions', completion_window: '24h' }), root, occurredAt: '2026-06-15T00:00:00Z' });
1494
- if (created.status !== 200 || pendingActions('openai', root).length === 0) return false;
1495
- const calls: string[] = [];
1496
- const fake = async (m: 'GET' | 'POST' | 'DELETE', p: string) => {
1497
- calls.push(`${m} ${p}`);
1498
- if (m === 'GET' && p.startsWith('/v1/files')) return { data: [{ id: 'file-real', purpose: 'batch', bytes: 5, status: 'processed', created_at: 1 }] };
1499
- if (m === 'GET' && p.startsWith('/v1/batches')) return { data: [{ id: 'batch-real', status: 'completed', created_at: 1, request_counts: { total: 1 } }] };
1500
- if (m === 'GET') return { data: [] };
1501
- return { id: 'batch-real-pushed' };
1502
- };
1503
- const res = await fullSyncOpenAI(fake, { root, occurredAt: '2026-06-15T00:00:01Z' });
1504
- // pushed the pending write, observed real state, appended deltas, swept all 4 collections
1505
- if (res.pushed < 1 || res.deltasAppended < 1 || res.collections < 4) return false;
1506
- if (pendingActions('openai', root).length !== 0) return false;
1507
- // idempotent: a second run with no pending + identical real state changes nothing
1508
- const again = await fullSyncOpenAI(fake, { root, occurredAt: '2026-06-15T00:00:02Z' });
1509
- return again.pushed === 0 && again.deltasAppended === 0;
1510
- } catch (err) {
1511
- if (isInfrastructureError(err)) throw harnessError('openai.connector.full_sync', err);
1512
- return false;
1513
- } finally {
1514
- rmSync(root, { recursive: true, force: true });
1515
- }
1516
- }),
1517
1499
 
1518
1500
  // ── Conformance harness ───────────────────────────────────────────────────────────────
1519
1501
  done('openai.conformance.envelopes', 'conformance', 'Offline conformance harness passes (envelope shapes across the surface)', 'connector', 'core', async () => {