@urun-sh/openai 0.6.16 → 0.6.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  import { Server, IncomingMessage } from 'node:http';
2
- import { f as ProxyHandlerOptions, P as ProxyClients } from '../responses-turn-vMefRdgJ.cjs';
3
- export { D as DialObserver, a as DialReport, g as ProxyIdentity, h as ProxyVideoOutLane, S as SessionGoneError, U as UnknownModelError } from '../responses-turn-vMefRdgJ.cjs';
2
+ import { f as ProxyHandlerOptions, P as ProxyClients } from '../responses-turn-B91w2YKn.cjs';
3
+ export { D as DialObserver, a as DialReport, g as ProxyIdentity, h as ProxyVideoOutLane, S as SessionGoneError, U as UnknownModelError } from '../responses-turn-B91w2YKn.cjs';
4
4
  export { h as GEMINI_LIVE_PATH, V as VIDEO_OUT_SETUP_KEY } from '../translator-CO8W_hmJ.cjs';
5
5
  import { WebSocket } from 'ws';
6
6
  import '../models-DUdx_Y6X.cjs';
@@ -339,6 +339,17 @@ interface RealtimeBinding<S = unknown> {
339
339
  };
340
340
  /** `model` the binding names for this conversation (overrides session config default). */
341
341
  model?: string;
342
+ /**
343
+ * The GA transcription-session intent (`?intent=transcription`): this
344
+ * connection is a TRANSCRIPTION session — transcription events flow
345
+ * without the client opting in, and `response.create` answers the loud
346
+ * `unsupported` error (the connection stays open) because there are no
347
+ * model turns to create. Set by the shared binding composition (binding.ts)
348
+ * only when the resolver's URL carried the intent AND the model's catalog
349
+ * row vouches stt (the binding carries `turns`). Absent = the realtime
350
+ * (voice) session, today's behaviour.
351
+ */
352
+ intent?: 'transcription';
342
353
  /** True when `session` is a resumed authorized session (vs a fresh private one). */
343
354
  resumed?: boolean;
344
355
  /**
@@ -350,8 +361,14 @@ interface RealtimeBinding<S = unknown> {
350
361
  * `turn_start` IS the native barge-in — the adapter drives server-VAD
351
362
  * emulation, real transcription events, and model-visible committed audio
352
363
  * from it. Absent = the backend wired no turn lane: manual mode only.
364
+ *
365
+ * The CONSUMER passes its own lifecycle `signal` (the realtime connection's
366
+ * turn-lane controller): the lane must END its stream subscription when it
367
+ * aborts — the transport races its pulls against it — so the subscription
368
+ * never outlives the consumer. A consumer that passes none gets an
369
+ * un-raced lane (its own teardown is then its own problem).
353
370
  */
354
- turns?: () => AsyncIterable<{
371
+ turns?: (signal?: AbortSignal) => AsyncIterable<{
355
372
  kind: 'delta' | 'final' | 'turn_start' | 'turn_end' | 'turn_eager_end' | 'turn_resumed';
356
373
  text?: string;
357
374
  confidence?: number;
@@ -446,10 +463,39 @@ declare class OpenAIRealtimeConnection<S = unknown> {
446
463
  private closed;
447
464
  /** The backend's turn/transcript lane consumer (binding.turns), if wired. */
448
465
  private turnLaneAbort;
449
- /** Accumulated transcript of the OPEN user turn (delta/final events). */
466
+ /**
467
+ * The CONSUMER-SIDE turn-lane iterator — held so teardown can end the lane
468
+ * subscription through the async-iterator return() protocol when it is
469
+ * parked at a YIELD (no pending pull for the abort to unwind; a plain
470
+ * for-await has no handle to close). Null once the consumer loop exits.
471
+ */
472
+ private turnLaneIterator;
473
+ /** Accumulated transcript of the OPEN user turn (delta/final events). The
474
+ * `final` flag says the engine finalized the accumulated text (its last
475
+ * transcript event was a `final`) — a manual commit may complete on it. */
450
476
  private turnTranscript;
451
477
  /** Transcript of the LAST committed turn (the model-visible commit). */
452
478
  private lastCommittedTranscript;
479
+ /**
480
+ * A manual transcription-session commit whose item is still awaiting the
481
+ * lane's final text (the item id of the committed item): interim deltas
482
+ * meanwhile carry THIS id, and the lane's next `final`/`turn_end`
483
+ * discharges exactly one `…transcription.completed` for it. No timers.
484
+ */
485
+ private pendingTranscriptItemId;
486
+ /**
487
+ * The engine turn a manual transcription-session commit ALREADY discharged
488
+ * (the discharged item's id): the lane's `final` completed the committed
489
+ * item (or the commit completed it on the lane's already-final text) while
490
+ * the engine's `turn_end` for THAT utterance is still to come. That
491
+ * `turn_end` is the boundary signal only — the server-side commit it
492
+ * would otherwise open mints a SECOND item for the same utterance and
493
+ * clears audio the client already appended for the next one. Consumed by
494
+ * that `turn_end` itself (one-shot); invalidated by the next `turn_start`
495
+ * (a new engine utterance opened — its boundary was not the discharged
496
+ * one).
497
+ */
498
+ private manuallyDischargedTurnItem;
453
499
  constructor(opts: {
454
500
  ws: WebSocket;
455
501
  binding: RealtimeBinding<S>;
@@ -464,8 +510,33 @@ declare class OpenAIRealtimeConnection<S = unknown> {
464
510
  * is the native barge-in, turn_end is the native commit, deltas/finals are
465
511
  * the REAL input transcription. This is the server-VAD primitive, not an
466
512
  * emulation gap: the voice engine itself decides turn boundaries.
513
+ *
514
+ * TEARDOWN ENDS THE SUBSCRIPTION FROM EVERY PARKING STATE, which a plain
515
+ * `for await` cannot do (its `break` only runs after the next event — a
516
+ * quiet engine parks the pull forever, and a `return()` enqueued behind a
517
+ * pending `next()` waits for it to settle):
518
+ * - the `abort` threaded into `turns()` unwinds a PENDING pull (the
519
+ * transport races its `messages()` pulls against it);
520
+ * - the HELD consumer iterator's `return()` (teardown) closes a lane
521
+ * parked at a YIELD with no pending pull.
467
522
  */
468
523
  private startTurnLane;
524
+ /** True when this connection IS a transcription session (the GA verdict:
525
+ * the binding's `?intent=transcription`, or the client's own
526
+ * `session.type: "transcription"` — the latter only accepted behind a
527
+ * turns lane, so a transcription session always has a transcript source). */
528
+ private isTranscriptionSession;
529
+ /** True when input-transcription events flow: the client opted in on a
530
+ * voice session, or ALWAYS on a transcription session (no opt-in needed). */
531
+ private transcriptionEventsOn;
532
+ /**
533
+ * Discharge a pending manual transcription commit: the ONE
534
+ * `…transcription.completed` for the committed item, carrying the text the
535
+ * lane delivered for its audio, and a reset shared accumulator (the
536
+ * utterance is client-delivered; the next utterance starts clean). The
537
+ * lane's own boundary events drive this — no timers anywhere.
538
+ */
539
+ private completePendingTranscription;
469
540
  /** One backend turn/transcript event → REAL GA protocol events. */
470
541
  private onTurnEvent;
471
542
  /** The committed turn's transcript, consumed once (null = none). */
@@ -647,8 +718,21 @@ declare function realtimeModelOf(req: IncomingMessage): string | undefined;
647
718
  * (`sessionHandle` once → `openAudioOn(handle, signal)` +
648
719
  * `createResponseOn(handle, …)`), so no turn on this connection can ever
649
720
  * resolve a different pooled session than the bridge.
721
+ * - An stt verdict (the catalog `task` gate `supportsTranscription`, the
722
+ * SAME router-backed oracle the /v1/audio/transcriptions lane rides) with
723
+ * no `openTurnsOn` seam ⇒ 501 — a transcription-capable binding without
724
+ * its transcript source is never wired silently. A row that OVERRIDES
725
+ * the platform's default transcript stream (`engine_args.transcript_stream`,
726
+ * resolved through `transcriptStreamFor` when the seam is wired) is the
727
+ * same-shape 501 — the lane never listens on a stream the engine does
728
+ * not write.
729
+ * - `?intent=transcription` on anything but an stt row ⇒ 400 naming the
730
+ * model; any other intent value ⇒ 400 naming the accepted values. Every
731
+ * refusal lands BEFORE the pooled session is resolved (`sessionHandle`)
732
+ * and the bridge opens — a refused upgrade admits nothing. On an stt row
733
+ * the intent is recorded on the binding (a transcription session).
650
734
  */
651
- declare function realtimeBindingFor(clients: Pick<ProxyClients, 'sessionHandle' | 'createResponseOn'> & Partial<Pick<ProxyClients, 'supportsAudio' | 'openAudioOn'>>, model: string | undefined, signal?: AbortSignal): Promise<RealtimeBinding<ProxyClients>>;
735
+ declare function realtimeBindingFor(clients: Pick<ProxyClients, 'sessionHandle' | 'createResponseOn'> & Partial<Pick<ProxyClients, 'supportsAudio' | 'openAudioOn' | 'supportsTranscription' | 'openTurnsOn' | 'transcriptStreamFor'>>, model: string | undefined, signal?: AbortSignal, intent?: string): Promise<RealtimeBinding<ProxyClients>>;
652
736
  /**
653
737
  * The LOCAL lane resolver: fixed pooled clients, optional fixed apiKey (the
654
738
  * same shared-secret rule the local Gemini Live lane applies — a mismatched
@@ -1,6 +1,6 @@
1
1
  import { Server, IncomingMessage } from 'node:http';
2
- import { f as ProxyHandlerOptions, P as ProxyClients } from '../responses-turn-EwxfRJJg.js';
3
- export { D as DialObserver, a as DialReport, g as ProxyIdentity, h as ProxyVideoOutLane, S as SessionGoneError, U as UnknownModelError } from '../responses-turn-EwxfRJJg.js';
2
+ import { f as ProxyHandlerOptions, P as ProxyClients } from '../responses-turn-FcEDaymF.js';
3
+ export { D as DialObserver, a as DialReport, g as ProxyIdentity, h as ProxyVideoOutLane, S as SessionGoneError, U as UnknownModelError } from '../responses-turn-FcEDaymF.js';
4
4
  export { h as GEMINI_LIVE_PATH, V as VIDEO_OUT_SETUP_KEY } from '../translator-Ddd65sXR.js';
5
5
  import { WebSocket } from 'ws';
6
6
  import '../models-DUdx_Y6X.js';
@@ -81,9 +81,11 @@ interface RealtimeBinding<S = unknown> {
81
81
 
82
82
  model?: string;
83
83
 
84
+ intent?: 'transcription';
85
+
84
86
  resumed?: boolean;
85
87
 
86
- turns?: () => AsyncIterable<{
88
+ turns?: (signal?: AbortSignal) => AsyncIterable<{
87
89
  kind: 'delta' | 'final' | 'turn_start' | 'turn_end' | 'turn_eager_end' | 'turn_resumed';
88
90
  text?: string;
89
91
  confidence?: number;
@@ -128,9 +130,15 @@ declare class OpenAIRealtimeConnection<S = unknown> {
128
130
 
129
131
  private turnLaneAbort;
130
132
 
133
+ private turnLaneIterator;
134
+
131
135
  private turnTranscript;
132
136
 
133
137
  private lastCommittedTranscript;
138
+
139
+ private pendingTranscriptItemId;
140
+
141
+ private manuallyDischargedTurnItem;
134
142
  constructor(opts: {
135
143
  ws: WebSocket;
136
144
  binding: RealtimeBinding<S>;
@@ -142,6 +150,12 @@ declare class OpenAIRealtimeConnection<S = unknown> {
142
150
 
143
151
  private startTurnLane;
144
152
 
153
+ private isTranscriptionSession;
154
+
155
+ private transcriptionEventsOn;
156
+
157
+ private completePendingTranscription;
158
+
145
159
  private onTurnEvent;
146
160
 
147
161
  private takeTurnTranscript;
@@ -188,7 +202,7 @@ declare class OpenAIRealtimeConnection<S = unknown> {
188
202
 
189
203
  declare function realtimeModelOf(req: IncomingMessage): string | undefined;
190
204
 
191
- declare function realtimeBindingFor(clients: Pick<ProxyClients, 'sessionHandle' | 'createResponseOn'> & Partial<Pick<ProxyClients, 'supportsAudio' | 'openAudioOn'>>, model: string | undefined, signal?: AbortSignal): Promise<RealtimeBinding<ProxyClients>>;
205
+ declare function realtimeBindingFor(clients: Pick<ProxyClients, 'sessionHandle' | 'createResponseOn'> & Partial<Pick<ProxyClients, 'supportsAudio' | 'openAudioOn' | 'supportsTranscription' | 'openTurnsOn' | 'transcriptStreamFor'>>, model: string | undefined, signal?: AbortSignal, intent?: string): Promise<RealtimeBinding<ProxyClients>>;
192
206
 
193
207
  declare function fixedClientsResolver(clients: ProxyClients, apiKey?: string): {
194
208
  resolve(req: IncomingMessage, signal?: AbortSignal): Promise<RealtimeBinding<ProxyClients>>;
@@ -1 +1 @@
1
- import{B as l,C as a,D as x,E as s,F as O,G as f,S as I,a as n,b as p,c as m}from"../chunk-2PIYH7O7.js";import{d as r,e as i}from"../chunk-TMQJY56F.js";import{a as o,h as t}from"../chunk-3JWYIKHM.js";import"../chunk-6M5JK4YC.js";import{c as e}from"../chunk-YSFSRI3D.js";e();export{o as GEMINI_LIVE_PATH,a as OPENAI_REALTIME_PATH,l as OpenAIRealtimeConnection,n as ResumptionRegistry,i as SessionGoneError,r as UnknownModelError,t as VIDEO_OUT_SETUP_KEY,p as attachGeminiLive,x as attachOpenAIRealtime,I as createOpenAIProxy,f as fixedClientsResolver,O as realtimeBindingFor,s as realtimeModelOf,m as serveGeminiLiveSocket};
1
+ import{B as l,C as a,D as x,E as s,G as O,H as f,T as I,a as n,b as p,c as m}from"../chunk-3U7YQ7LE.js";import{f as r,g as i}from"../chunk-NYSZ3RGB.js";import{a as o,h as t}from"../chunk-3JWYIKHM.js";import"../chunk-6M5JK4YC.js";import{c as e}from"../chunk-YSFSRI3D.js";e();export{o as GEMINI_LIVE_PATH,a as OPENAI_REALTIME_PATH,l as OpenAIRealtimeConnection,n as ResumptionRegistry,i as SessionGoneError,r as UnknownModelError,t as VIDEO_OUT_SETUP_KEY,p as attachGeminiLive,x as attachOpenAIRealtime,I as createOpenAIProxy,f as fixedClientsResolver,O as realtimeBindingFor,s as realtimeModelOf,m as serveGeminiLiveSocket};
@@ -151,6 +151,8 @@ interface InferenceRecord {
151
151
  readonly prompt_tokens: number | null;
152
152
  readonly completion_tokens: number | null;
153
153
  readonly output_units: number | null;
154
+ /** THE OUTPUT-MINUTE QUANTITY — see {@link InferenceTurn.outputMinutes}. */
155
+ readonly output_minutes: number | null;
154
156
  /**
155
157
  * THE INSTANCE that served this request — see
156
158
  * {@link InferenceTurn.sessionId}. On the log line as well as the ledger,
@@ -379,7 +381,8 @@ interface ModelPriceRow {
379
381
  }
380
382
 
381
383
  /**
382
- * THE OPENROUTER PROVIDER DOCUMENT — `GET /v1/models?format=openrouter`.
384
+ * THE OPENROUTER PROVIDER DOCUMENT — `GET /partners/openrouter/models`
385
+ * (`GET /v1/models?format=openrouter` answers 308 here for one release).
383
386
  *
384
387
  * OpenRouter's provider monitor polls this document (schema 2.4, per
385
388
  * openrouter.ai/docs/guides/community/for-providers §1) to list a provider's
@@ -608,28 +611,30 @@ declare class UnknownModelError extends Error {
608
611
  declare class SessionGoneError extends Error {
609
612
  }
610
613
  /**
611
- * OpenAI-shaped model list (same shape as models.ts listModels). When the
612
- * catalog oracle is configured, each entry ALSO carries the catalog
613
- * enrichment fields (additive JSON — OpenAI clients ignore unknown fields;
614
- * the OpenRouter + Vercel AI Gateway provider listings need them). Entries
615
- * whose slug matches no catalog row stay at the bare four fields.
614
+ * One `/v1/models` row — a CLOSED allow-list of externally contracted keys
615
+ * (owner ruling, ENG-472: publish what an external contract names, never the
616
+ * implementation). The four OpenAI `Model` fields, plus — only when known —
617
+ * the Hugging Face pair (`context_length`, `pricing`), the TTS `voices` and
618
+ * the `capabilities` object. `name`, `description`, `task`, `engine`,
619
+ * `gpu_spec` and `warm` are GONE from the type so they cannot be set; every
620
+ * row builder goes through `publicModelEntry` (openrouter-doc.ts), which
621
+ * copies exactly these keys.
616
622
  */
617
623
  interface RouterModelEntry {
624
+ /** OpenAI `Model`: the model identifier callers reference in the API. */
618
625
  id: string;
626
+ /** OpenAI `Model`: always "model". */
619
627
  object: 'model';
628
+ /** OpenAI `Model`: the Unix time the model id first appeared in the uRun catalog (0 until the catalog carries it). */
620
629
  created: number;
630
+ /** OpenAI `Model`: the organization that owns the model. */
621
631
  owned_by: string;
622
- /** Display name — the canonical `<model_id>:<variant>` catalog ref. */
623
- name?: string;
624
- /** User-facing description from the catalog (console Endpoints copy). */
625
- description?: string;
626
- /** Catalog modality (chat | code | agent | vl | audio | image | ...). */
627
- task?: string;
628
- /** Serving engine (vllm | sglang | llamacpp | ...). */
629
- engine?: string;
630
- /** The catalog lane this app deploys, e.g. 'rtx6000:1'. */
631
- gpu_spec?: string;
632
- /** Context window in tokens (chat rows carry it in engine_args). */
632
+ /**
633
+ * Hugging Face provider comparison table — its top-level `context_length`
634
+ * read (HF also gates `:fastest`/`:cheapest` on it). Only when the catalog
635
+ * states it and the placements agree (openrouter-doc.ts
636
+ * `placementsContextLength`).
637
+ */
633
638
  context_length?: number;
634
639
  /**
635
640
  * Per-token price, USD per MILLION tokens — the exact shape Hugging Face
@@ -639,8 +644,6 @@ interface RouterModelEntry {
639
644
  * a zero or an estimate here would be a silent lie on a public table.
640
645
  */
641
646
  pricing?: TokenPricing;
642
- /** Idle-floor replica count for a shared-lane model (U6 reconciler). */
643
- warm?: number;
644
647
  /**
645
648
  * The voices a `tts`-task model synthesizes under (ENG-331 — OpenAI's
646
649
  * speech API names a voice per request, so a caller needs the list).
@@ -662,7 +665,7 @@ interface RouterModelEntry {
662
665
  * whose placements disagree, a malformed value, or a slug with no catalog
663
666
  * row omit the object entirely — unknown, never guessed
664
667
  * (openrouter-doc.ts `placementsReasoning`, applied by both listing
665
- * lanes).
668
+ * lanes). The conformance kit's voice_profile check reads it.
666
669
  */
667
670
  capabilities?: {
668
671
  reasoning: boolean;
@@ -672,10 +675,11 @@ interface RouterModelList {
672
675
  object: 'list';
673
676
  data: RouterModelEntry[];
674
677
  }
675
- /** One `/v1/images/models` entry: the OpenAI model object plus the image
678
+ /** One `/v1/images/models` entry: the allow-listed model row plus the image
676
679
  * modes the catalog rows vouch for (the SAME per-placement consensus
677
680
  * {@link ModelRouter.supportsImages} applies — a row set that vouches
678
- * nothing is NOT listed). */
681
+ * nothing is NOT listed). `image_modes` is allow-listed for THIS listing
682
+ * only (publicModelEntry's image overload). */
679
683
  interface RouterImageModelEntry extends RouterModelEntry {
680
684
  image_modes: ImageCapabilityMode[];
681
685
  }
@@ -1100,7 +1104,8 @@ declare class ModelRouter<S> {
1100
1104
  */
1101
1105
  modelList(): Promise<RouterModelList>;
1102
1106
  /**
1103
- * The OpenRouter PROVIDER document (`GET /v1/models?format=openrouter`):
1107
+ * The OpenRouter PROVIDER document (`GET /partners/openrouter/models`;
1108
+ * `GET /v1/models?format=openrouter` 308-redirects there for one release):
1104
1109
  * schema 2.4 per openrouter.ai/docs/guides/community/for-providers §1 —
1105
1110
  * "an endpoint that returns all models that should be served by
1106
1111
  * OpenRouter". That is the SHARED block's sellable endpoints (the models
@@ -1216,6 +1221,25 @@ declare class ModelRouter<S> {
1216
1221
  * an infrastructure outage.
1217
1222
  */
1218
1223
  supportsTranscription(model: string | undefined): Promise<boolean>;
1224
+ /**
1225
+ * THE REALTIME TRANSCRIPTION LANE'S STREAM-NAME RESOLUTION (openai-realtime
1226
+ * binding.ts, ENG-461): the §5 stream the model's engine emits its own
1227
+ * transcript events on — `engine_args.transcript_stream`, defaulting to
1228
+ * the platform's {@link STT_TRANSCRIPT_STREAM} when no row carries the
1229
+ * field, over the SAME catalog join {@link supportsTranscription} uses
1230
+ * (exact `(model_id, variant)` for a shared-lane hit,
1231
+ * {@link catalogRowsForSlug} for a caller-org app). Deliberately NOT
1232
+ * task-gated: the binding consults it only behind a `transcribes` verdict,
1233
+ * and the resolution itself is a pure field consensus. The lane's consume
1234
+ * side reads the PLATFORM DEFAULT only — a row that OVERRIDES the default
1235
+ * is the binding's loud 501 naming the field, never a silent listen on a
1236
+ * stream the engine does not write; a catalog that cannot vow ONE name
1237
+ * (disagreeing placements, a non-string or empty value) is the loud
1238
+ * defect, never a guess. Undeployed uRun-namespace models throw
1239
+ * {@link UnknownModelError}, and a CONFIGURED catalog oracle that FAILS
1240
+ * propagates — never a fabricated default.
1241
+ */
1242
+ transcriptStreamFor(model: string | undefined): Promise<string>;
1219
1243
  /**
1220
1244
  * THE SPEECH LANE'S VOICE-SET RESOLUTION (proxy/speech.ts): the voice set
1221
1245
  * `/v1/models` publishes for `model` — resolved by the SAME resolver the
@@ -1266,6 +1290,43 @@ declare class ModelRouter<S> {
1266
1290
  closeAll(): Promise<void>;
1267
1291
  }
1268
1292
 
1293
+ /**
1294
+ * THE TURN/TRANSCRIPT LANE over the named §5 `stt` data stream — the
1295
+ * engine's OWN transcript source for the OpenAI Realtime transcription
1296
+ * intent (ENG-461).
1297
+ *
1298
+ * The delivery path is the platform's existing one (urun-python
1299
+ * `urun.serve.transcribe_bridge`): the streaming STT engine emits its own
1300
+ * transcript events on the session-scoped named stream {@link
1301
+ * STT_TRANSCRIPT_STREAM} — the SAME §5 named-DATA mechanism the text-lane
1302
+ * token deltas ride, in the text lane's own wire shape (`{"t": kind,
1303
+ * "delta": text, "start_ms": int}`). Nothing here invents a framing: the
1304
+ * lane is `session.stream('stt').messages()` (core `SessionStream.messages`,
1305
+ * contract §5 — the downstream twin of the `emit` produce seam), and this
1306
+ * module only DECODES each payload onto the `RealtimeBinding.turns` event
1307
+ * shape the realtime adapter already consumes.
1308
+ *
1309
+ * `confidence` in the wire payload is SEMANTIC END-OF-TURN confidence (the
1310
+ * engine's own VAD verdict that the user finished speaking), NOT
1311
+ * transcription confidence — carrying it onto the GA surface would
1312
+ * misrepresent it, so the decoder DROPS it by design (the adapter reads the
1313
+ * boundary from `turn_end`, which is the same verdict as an event).
1314
+ */
1315
+
1316
+ /** The turn kinds the transcribe bridge emits (TranscriptEvent kinds). */
1317
+ type SessionTurnKind = 'delta' | 'final' | 'turn_start' | 'turn_end' | 'turn_eager_end' | 'turn_resumed';
1318
+ /**
1319
+ * One decoded transcript-lane event — the `RealtimeBinding.turns` event
1320
+ * shape. `text` is the wire `delta` field (a transcript PIECE — the adapter
1321
+ * accumulates), `startMs` the additive `start_ms` clock. The wire
1322
+ * `confidence` (semantic end-of-turn confidence) is deliberately absent.
1323
+ */
1324
+ interface SessionTurnLaneEvent {
1325
+ kind: SessionTurnKind;
1326
+ text?: string;
1327
+ startMs?: number;
1328
+ }
1329
+
1269
1330
  /** One ready replica row from the presence read. */
1270
1331
  interface RuntimePresenceRow {
1271
1332
  runtime_id: string;
@@ -1485,6 +1546,13 @@ interface InferenceUsageRecord {
1485
1546
  prompt_tokens: number | null;
1486
1547
  completion_tokens: number | null;
1487
1548
  output_units: number | null;
1549
+ /**
1550
+ * THE OUTPUT-MINUTE QUANTITY — a text-to-speech row's delivered audio
1551
+ * duration in minutes, the quantity its `per_output_minute` price reads.
1552
+ * null = this modality has no such quantity (a token lane, an unmeasured
1553
+ * TTS answer) — never "zero of it".
1554
+ */
1555
+ output_minutes: number | null;
1488
1556
  /** NULL = NOT YET PRICED (ENG-317). Never "free". */
1489
1557
  cost_nano_usd: number | null;
1490
1558
  price_version: string | null;
@@ -1661,6 +1729,17 @@ interface LedgerWrite {
1661
1729
  prompt_tokens: number | null;
1662
1730
  completion_tokens: number | null;
1663
1731
  output_units: number | null;
1732
+ /**
1733
+ * THE OUTPUT-MINUTE QUANTITY — the delivered audio duration of a
1734
+ * text-to-speech answer, in minutes, carried from the record
1735
+ * ({@link InferenceRecord.output_minutes}). The control plane prices a
1736
+ * `per_output_minute` task from THIS field; `output_units` stays the
1737
+ * per-image/per-clip quantity. NULL = the request has no such quantity —
1738
+ * a discarded answer, or a lane that never measured one. NEVER zero of it,
1739
+ * and never coerced: an unmeasured TTS row is written unpriced
1740
+ * (`output_minutes_absent`), not at a billed zero.
1741
+ */
1742
+ output_minutes: number | null;
1664
1743
  /**
1665
1744
  * THE WEIGHT (ENG-417) — the serve runtime's own residency in whole
1666
1745
  * milliseconds. An allocation input, never GPU time and never a cost. NULL
@@ -1736,7 +1815,8 @@ interface LedgerRecorded {
1736
1815
  * Why the row carries no price, or null when it is priced. One of
1737
1816
  * `not_a_catalog_model` | `outcome_not_billable` | `http_status_not_success`
1738
1817
  * | `task_unit_undeclared` | `task_not_token_billed` | `token_counts_absent`
1739
- * | `no_active_price` | `no_active_input_price` | `no_active_output_price`.
1818
+ * | `output_minutes_absent` | `no_active_price` | `no_active_input_price` |
1819
+ * `no_active_output_price`.
1740
1820
  * It is a NORMAL answer, not an error: no PER-TOKEN rate has been seeded
1741
1821
  * yet, so this lane has nothing to price with. (Not "the price book is
1742
1822
  * empty" — `model_prices` has carried live `per_minute` rows since
@@ -2085,9 +2165,10 @@ interface ProxyClients {
2085
2165
  /** `listModels(...)` result (an OpenAI model list object). */
2086
2166
  listModels(): Promise<unknown>;
2087
2167
  /**
2088
- * The OpenRouter provider document (`GET /v1/models?format=openrouter`).
2168
+ * The OpenRouter provider document (`GET /partners/openrouter/models`; the
2169
+ * old `GET /v1/models?format=openrouter` spelling 308-redirects there).
2089
2170
  * Optional at the seam: hand-built test clients may omit it, in which case
2090
- * the format=openrouter branch answers 501 — loud, never a silent empty.
2171
+ * that route answers 501 — loud, never a silent empty.
2091
2172
  */
2092
2173
  openRouterModels?(): Promise<unknown>;
2093
2174
  /**
@@ -2265,6 +2346,46 @@ interface ProxyClients {
2265
2346
  * queued admission going back only at the LAST waiter's abort.
2266
2347
  */
2267
2348
  openAudioOn?(handle: string, signal?: AbortSignal): Promise<PinnedAudioLane>;
2349
+ /**
2350
+ * THE TURN/TRANSCRIPT LANE OPEN (the realtime transcription intent,
2351
+ * ENG-461): the STT engine's OWN transcript events — the §5 `stt` named
2352
+ * data stream (urun.serve.transcribe_bridge's wire shape: `t` =
2353
+ * delta/final/turn_start/turn_end/turn_eager_end/turn_resumed, `delta`
2354
+ * text pieces, additive `start_ms`), consumed off the EXACT pooled session
2355
+ * the handle names (ModelRouter.sessionForHandle — a gone or replaced
2356
+ * session throws {@link SessionGoneError} LOUDLY, never a fresh session
2357
+ * opened behind the handle's back) and decoded onto the
2358
+ * `RealtimeBinding.turns` event shape (the wire's semantic-VAD
2359
+ * `confidence` is end-of-turn confidence, NOT transcription confidence —
2360
+ * the lane deliberately does not carry it). One lane per call; the
2361
+ * realtime binding calls it once per connection. The returned iterable is
2362
+ * lazy. `signal` is the CONSUMER's own lifecycle (the realtime
2363
+ * connection's turn-lane controller, aborted at its teardown): the
2364
+ * transport races every `messages()` pull against it and ends the stream
2365
+ * subscription on every exit path, so the lane never outlives its
2366
+ * consumer — a quiet engine cannot pin it open behind a never-settling
2367
+ * pull. Optional at the seam exactly like
2368
+ * {@link openAudioOn}: the binding's stt verdict answers its absence with
2369
+ * the loud 501 — never a transcription-capable binding with no transcript
2370
+ * source. An already-aborted signal rejects on the first pull — never
2371
+ * opens a lane for a vanished consumer.
2372
+ */
2373
+ openTurnsOn?(handle: string, signal?: AbortSignal): AsyncIterable<SessionTurnLaneEvent>;
2374
+ /**
2375
+ * THE REALTIME TRANSCRIPTION LANE'S STREAM-NAME RESOLUTION (openai-realtime
2376
+ * binding.ts): the §5 transcript stream the model's catalog rows vow
2377
+ * (ModelRouter.transcriptStreamFor — `engine_args.transcript_stream`,
2378
+ * defaulting to the platform's `stt`). The lane's consume side reads the
2379
+ * platform DEFAULT stream name only, so the binding answers a row that
2380
+ * OVERRIDES the default with the loud 501 naming the field — never a
2381
+ * silent listen on a stream the engine does not write, and never a
2382
+ * transcript-less transcription session. Optional at the seam exactly
2383
+ * like {@link supportsTranscription}: hand-built clients own their own
2384
+ * lane (their `openTurnsOn` decides what to read — no catalog row exists
2385
+ * to check), so its absence skips the catalog check; every router-backed
2386
+ * backhaul wires it.
2387
+ */
2388
+ transcriptStreamFor?(model: string | undefined): Promise<string>;
2268
2389
  /**
2269
2390
  * Open (or reuse) the NATIVE video FRAME lane on the pooled session `model`
2270
2391
  * routes to (transport/media.ts `enableSessionVideo`: discrete JPEG frames →
@@ -51,6 +51,8 @@ interface InferenceRecord {
51
51
  readonly completion_tokens: number | null;
52
52
  readonly output_units: number | null;
53
53
 
54
+ readonly output_minutes: number | null;
55
+
54
56
  readonly session_id: string | null;
55
57
 
56
58
  readonly queue_ms: number | null;
@@ -176,27 +178,19 @@ declare class SessionGoneError extends Error {
176
178
  }
177
179
 
178
180
  interface RouterModelEntry {
179
- id: string;
180
- object: 'model';
181
- created: number;
182
- owned_by: string;
183
181
 
184
- name?: string;
185
-
186
- description?: string;
182
+ id: string;
187
183
 
188
- task?: string;
184
+ object: 'model';
189
185
 
190
- engine?: string;
186
+ created: number;
191
187
 
192
- gpu_spec?: string;
188
+ owned_by: string;
193
189
 
194
190
  context_length?: number;
195
191
 
196
192
  pricing?: TokenPricing;
197
193
 
198
- warm?: number;
199
-
200
194
  voices?: string[];
201
195
 
202
196
  capabilities?: {
@@ -344,6 +338,8 @@ declare class ModelRouter<S> {
344
338
 
345
339
  supportsTranscription(model: string | undefined): Promise<boolean>;
346
340
 
341
+ transcriptStreamFor(model: string | undefined): Promise<string>;
342
+
347
343
  speechVoices(model: string | undefined): Promise<SpeechVoicesResolution>;
348
344
 
349
345
  imageModelList(): Promise<RouterImageModelList>;
@@ -353,6 +349,14 @@ declare class ModelRouter<S> {
353
349
  closeAll(): Promise<void>;
354
350
  }
355
351
 
352
+ type SessionTurnKind = 'delta' | 'final' | 'turn_start' | 'turn_end' | 'turn_eager_end' | 'turn_resumed';
353
+
354
+ interface SessionTurnLaneEvent {
355
+ kind: SessionTurnKind;
356
+ text?: string;
357
+ startMs?: number;
358
+ }
359
+
356
360
  interface RuntimePresenceRow {
357
361
  runtime_id: string;
358
362
 
@@ -438,6 +442,8 @@ interface InferenceUsageRecord {
438
442
  completion_tokens: number | null;
439
443
  output_units: number | null;
440
444
 
445
+ output_minutes: number | null;
446
+
441
447
  cost_nano_usd: number | null;
442
448
  price_version: string | null;
443
449
  priced_at: string | null;
@@ -480,6 +486,8 @@ interface LedgerWrite {
480
486
  completion_tokens: number | null;
481
487
  output_units: number | null;
482
488
 
489
+ output_minutes: number | null;
490
+
483
491
  queue_ms: number | null;
484
492
  prefill_ms: number | null;
485
493
  decode_ms: number | null;
@@ -609,6 +617,10 @@ interface ProxyClients {
609
617
 
610
618
  openAudioOn?(handle: string, signal?: AbortSignal): Promise<PinnedAudioLane>;
611
619
 
620
+ openTurnsOn?(handle: string, signal?: AbortSignal): AsyncIterable<SessionTurnLaneEvent>;
621
+
622
+ transcriptStreamFor?(model: string | undefined): Promise<string>;
623
+
612
624
  openVideo?(model: string | undefined): Promise<ProxyVideoLane>;
613
625
 
614
626
  openVideoOut?(model: string | undefined): Promise<ProxyVideoOutLane>;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@urun-sh/openai",
3
- "version": "0.6.16",
3
+ "version": "0.6.18",
4
4
  "description": "OpenAI-compatible Realtime + Responses SDK over uRun session primitives (not websockets).",
5
5
  "type": "module",
6
6
  "main": "./dist/index.cjs",
@@ -70,7 +70,7 @@
70
70
  "build:bin": "bun build --compile --external @roamhq/wrtc src/proxy/bin.ts --outfile dist/bin/urun-openai && node scripts/assert-bin-transport.mjs"
71
71
  },
72
72
  "peerDependencies": {
73
- "@urun-sh/core": "^0.6.16"
73
+ "@urun-sh/core": "^0.6.18"
74
74
  },
75
75
  "dependencies": {
76
76
  "@prometheus-io/client": "^0.16.1",