@urun-sh/openai 0.5.5 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/dist/{ResponsesClient-BYx3YLGo.d.ts → ResponsesClient-CSSrOYD8.d.ts} +7 -1
  2. package/dist/{ResponsesClient-Dft3bg3b.d.cts → ResponsesClient-_OZERUjH.d.cts} +11 -1
  3. package/dist/chunk-3JWYIKHM.js +2 -0
  4. package/dist/chunk-45FJ5GTP.js +7 -0
  5. package/dist/chunk-6M5JK4YC.js +1 -0
  6. package/dist/chunk-QTVTCMJU.js +1 -0
  7. package/dist/chunk-TDRZZGUK.js +1 -0
  8. package/dist/chunk-TPH77ZOW.js +58 -0
  9. package/dist/chunk-WCMX5VOF.js +4 -0
  10. package/dist/chunk-YXK42SKC.js +1 -0
  11. package/dist/gemini-live.cjs +2 -2
  12. package/dist/gemini-live.d.cts +79 -7
  13. package/dist/gemini-live.d.ts +32 -7
  14. package/dist/gemini-live.js +1 -1
  15. package/dist/hosted/bin.cjs +43 -27
  16. package/dist/hosted/bin.js +4 -4
  17. package/dist/hosted/index.cjs +41 -26
  18. package/dist/hosted/index.d.cts +710 -17
  19. package/dist/hosted/index.d.ts +171 -6
  20. package/dist/hosted/index.js +1 -1
  21. package/dist/index.cjs +1 -1
  22. package/dist/index.d.cts +5 -5
  23. package/dist/index.d.ts +5 -5
  24. package/dist/index.js +1 -1
  25. package/dist/{models-NYMZrklp.d.cts → models-DUdx_Y6X.d.cts} +1 -1
  26. package/dist/pi-extension/index.cjs +7 -7
  27. package/dist/pi-extension/index.d.cts +2 -2
  28. package/dist/pi-extension/index.d.ts +2 -2
  29. package/dist/pi-extension/index.js +1 -1
  30. package/dist/pi-extension/standalone.cjs +53 -53
  31. package/dist/proxy/cli.cjs +55 -43
  32. package/dist/proxy/cli.js +14 -13
  33. package/dist/proxy/index.cjs +31 -28
  34. package/dist/proxy/index.d.cts +132 -18
  35. package/dist/proxy/index.d.ts +28 -11
  36. package/dist/proxy/index.js +1 -6
  37. package/dist/responses-turn-BfdBbey8.d.cts +2394 -0
  38. package/dist/responses-turn-CdhIre_a.d.ts +652 -0
  39. package/dist/{translator-CcDBEfvm.d.cts → translator-CO8W_hmJ.d.cts} +3 -3
  40. package/dist/{translator-C9uPKypK.d.ts → translator-Ddd65sXR.d.ts} +1 -1
  41. package/dist/{types-lsVTbNcH.d.cts → types-CHPtJx6d.d.cts} +13 -0
  42. package/dist/{types-lsVTbNcH.d.ts → types-CHPtJx6d.d.ts} +4 -0
  43. package/dist/{video-out-D20UuJ8G.d.cts → video-out-BsuLqlID.d.cts} +6 -6
  44. package/dist/{video-out-CWesbk12.d.ts → video-out-Cawtm7SF.d.ts} +1 -1
  45. package/package.json +14 -11
  46. package/dist/chunk-5BZCM3RS.js +0 -4
  47. package/dist/chunk-5NXM4IO3.js +0 -2
  48. package/dist/chunk-CWJRDBDC.js +0 -1
  49. package/dist/chunk-DMD5UENK.js +0 -1
  50. package/dist/chunk-GBBY3PCZ.js +0 -1
  51. package/dist/chunk-I2Q3B3OG.js +0 -6
  52. package/dist/chunk-OI2OY32M.js +0 -1
  53. package/dist/chunk-QVF7NF7G.js +0 -44
  54. package/dist/responses-turn-KOAoIqZ-.d.ts +0 -200
  55. package/dist/responses-turn-OrO4euEN.d.cts +0 -513
  56. /package/dist/{models-NYMZrklp.d.ts → models-DUdx_Y6X.d.ts} +0 -0
@@ -0,0 +1,2394 @@
1
+ import { C as CatalogRow } from './models-DUdx_Y6X.cjs';
2
+ import { IncomingMessage } from 'node:http';
3
+ import { a as AudioBridge, h as VideoFrameLane, j as VideoOutLane } from './video-out-BsuLqlID.cjs';
4
+
5
+ /**
6
+ * WHAT a proxy serves — the `identity` block on `GET /stats`. Reuse of a
7
+ * running proxy is allowed iff every field matches the launching invocation
8
+ * EXACTLY: a partial match (same app, different fn; same everything, older
9
+ * proxy_version) silently routes the agent at the wrong backend, which is
10
+ * worse than any error.
11
+ */
12
+ interface ProxyIdentity {
13
+ app: string;
14
+ org: string;
15
+ fn: string;
16
+ /**
17
+ * The control-plane URL the proxy's backhaul session opens against
18
+ * (URUN_BASE_URL verbatim — same org/app/fn on staging vs prod are
19
+ * DIFFERENT backends; review finding, #285). Compared exactly: a cosmetic
20
+ * difference (trailing slash) merely refuses reuse and spawns an ephemeral
21
+ * proxy — the safe direction.
22
+ */
23
+ base_url: string;
24
+ proxy_version: string;
25
+ }
26
+
27
+ /**
28
+ * THE BILLING CORRELATION SEAM — one uuid per `/v1` request, returned as the
29
+ * `Inference-Id` response header, carried as the suffix of the body id the
30
+ * request answers with, and recorded in one structured log line.
31
+ *
32
+ * WHY IT EXISTS: Hugging Face Inference Providers bill routed requests by a
33
+ * unique id the provider returns as a RESPONSE HEADER. Their spec is explicit:
34
+ * "Make sure this header is present on every response you return, including
35
+ * streaming responses. If it's missing, we have no way to match the request
36
+ * and it can't be billed." A response that leaves this proxy without the
37
+ * header is therefore an UNBILLED response — a revenue bug, not a cosmetic
38
+ * one — which is why the header is set on the response object in the request
39
+ * handler's synchronous prefix, before any lane can flush a head.
40
+ *
41
+ * WHAT THIS MINT COVERS: the three HTTP lanes' body ids (`chatcmpl-…`,
42
+ * `resp_…`, `msg_…`) and the `/v1/responses` WebSocket lane's response ids —
43
+ * all of which previously minted their own `Math.random().toString(36)`
44
+ * string, four unrelated id spaces nothing downstream could join on. Only
45
+ * `executeResponsesTurn`'s required `responseId` argument is COMPILE-enforced;
46
+ * the rest is enforced by review, so keep new lanes on this mint.
47
+ *
48
+ * WHAT IT DOES NOT COVER (still separate id spaces, recorded on ENG-309):
49
+ * - `transport/decode.ts` builds `resp_${requestId}` from the SERVE-lane
50
+ * request id that `responses/ResponsesClient.ts` mints (`req_<n>`); the
51
+ * HTTP lanes overwrite it with the id from here, but joining a billing row
52
+ * to the runtime's own usage receipt still needs that id plumbed through
53
+ * `ProxyClients.createResponse`.
54
+ * - the OpenAI Realtime lane (`proxy/openai-realtime/protocol.ts`) mints its
55
+ * own `resp_` ids for WS realtime turns — which, being built from a
56
+ * `req_`-prefixed seed, currently read `resp_req_<hex>` (a double prefix).
57
+ */
58
+
59
+ type TurnCatalogIdentity = {
60
+ kind: 'catalog';
61
+ model_id: string;
62
+ variant: string;
63
+ } | {
64
+ kind: 'not-a-catalog-model';
65
+ model: string | null;
66
+ } | {
67
+ kind: 'unresolved';
68
+ reason: string;
69
+ };
70
+ /**
71
+ * WHAT ONE DIAL REPORTS ABOUT ITSELF — everything the ledger reads off the
72
+ * dial that actually served, reported together, from that dial.
73
+ *
74
+ * ONE CALLBACK, BOTH FACTS, and that is the point rather than a convenience.
75
+ * `session_id` (ENG-417) and the billing identity (ENG-414) are the two halves
76
+ * of the same question — which instance ran this, and what it ran — and a
77
+ * re-home makes them disagree the moment they are sourced separately: the row
78
+ * named the REPLACEMENT instance while naming the identity the ABANDONED dial
79
+ * resolved. Two callbacks firing from the same loop about the same dial would
80
+ * be two things free to drift apart again, which is the whole defect this seam
81
+ * exists to remove. Widen this payload; never add a second callback beside it.
82
+ *
83
+ * FIRED ONCE PER DIAL, LAST DIAL WINS. `rehome.ts` performs exactly one
84
+ * re-dial when a backhaul dies before producing client-visible content, and
85
+ * reports again before the replacement is used — so what the ledger keeps is
86
+ * the dial that served.
87
+ */
88
+ interface DialReport {
89
+ /**
90
+ * The uRun session id of the backhaul about to be used, or null when the
91
+ * pool entry carries no session identity. Never a fabricated id: a wrong
92
+ * link moves cost attribution between machines.
93
+ */
94
+ sessionId: string | null;
95
+ /**
96
+ * The billing identity THIS dial resolved — the ledger's own vocabulary, not
97
+ * the router's. It comes off the SAME `resolveTarget` answer the dial used
98
+ * (`ModelRouter.sessionFor`), never a second resolution taken alongside it:
99
+ * a resolution taken at a different moment can answer differently, and then
100
+ * the row names a variant nobody served. Both flavours travel this way — the
101
+ * catalog `(model_id, variant)` and the caller-org app slug, which ENG-412
102
+ * made billing-bearing.
103
+ */
104
+ identity: TurnCatalogIdentity;
105
+ }
106
+ /** The observer one dial reports itself to. See {@link DialReport}. */
107
+ type DialObserver = (dial: DialReport) => void;
108
+ /**
109
+ * ONE completed `/v1` request, as data. This is the SINGLE per-request fact
110
+ * object: {@link inferenceRequestLine} serializes it into the billing
111
+ * correlation log line, and `proxy/metrics.ts` `InferenceMetrics.observe`
112
+ * counts the same object into Prometheus. Two sinks, one record — a metric
113
+ * that could disagree with the billing log would be worse than no metric.
114
+ *
115
+ * Nothing here is a secret or a body: no API key, no prompt, no completion.
116
+ * The caller is identified by `org` (a control-plane identifier), never by the
117
+ * Bearer key that resolved it.
118
+ */
119
+ interface InferenceRecord {
120
+ readonly event: 'inference_request';
121
+ readonly ts: string;
122
+ readonly inference_id: string;
123
+ readonly method: string;
124
+ readonly path: string;
125
+ readonly status: number | null;
126
+ readonly delivered: boolean;
127
+ readonly stream_error: boolean;
128
+ readonly model: string | null;
129
+ readonly stream: boolean | null;
130
+ readonly org: string | null;
131
+ readonly duration_ms: number;
132
+ readonly ttft_ms: number | null;
133
+ /**
134
+ * The usage quantities the per-request ledger is written from. NULL means
135
+ * the request HAS no such quantity — never zero of it. See
136
+ * {@link InferenceTurn.promptTokens}.
137
+ *
138
+ * They are on the RECORD rather than read separately by the ledger writer
139
+ * for the reason the record exists at all: one object, every sink. A writer
140
+ * that took a second walk over the request could disagree with the log line
141
+ * and the metric, and the disagreement would be invisible.
142
+ *
143
+ * NOTE WHAT IS NOT HERE: `gpu_seconds`. There is no such quantity, and as
144
+ * of ENG-417 there is no such ledger column either. Exclusive per-request
145
+ * GPU occupancy is not obtainable under continuous batching, and wall-clock
146
+ * is SHARED occupancy — a per-request stamp would double-count against
147
+ * `usage_events.metered_gpu_seconds`, which meters the instance once. What
148
+ * is here instead is the LINK ({@link session_id}) and the WEIGHT (the
149
+ * residency milliseconds below), from which the share is DERIVED.
150
+ */
151
+ readonly prompt_tokens: number | null;
152
+ readonly completion_tokens: number | null;
153
+ readonly output_units: number | null;
154
+ /**
155
+ * THE INSTANCE that served this request — see
156
+ * {@link InferenceTurn.sessionId}. On the log line as well as the ledger,
157
+ * so a row that was never written can still be traced to the session it
158
+ * would have been attributed to.
159
+ */
160
+ readonly session_id: string | null;
161
+ /**
162
+ * THE WEIGHT — see {@link InferenceTurn.timing}. Flattened onto the record
163
+ * rather than nested, because this record is also a log line and a flat
164
+ * field is what a log query can select. NULL means the runtime reported
165
+ * nothing; it never means zero.
166
+ */
167
+ readonly queue_ms: number | null;
168
+ readonly prefill_ms: number | null;
169
+ readonly decode_ms: number | null;
170
+ readonly total_ms: number | null;
171
+ /**
172
+ * The identity that served this request, reported by the dial that served it
173
+ * — see {@link TurnCatalogIdentity} and {@link DialReport}. The ledger writer
174
+ * reads it from here rather than asking the router again, and the log line
175
+ * carries it so a row that was never written can be traced to the reason
176
+ * from the same record.
177
+ */
178
+ readonly catalog: TurnCatalogIdentity;
179
+ }
180
+
181
+ /**
182
+ * THE SHARED-MODEL LANE (shared-endpoints U8, urun-infra#1490 §7) — the pure
183
+ * per-request resolver over the catalog edge function's `shared` block:
184
+ *
185
+ * { shared_org_id, endpoints: [{ model_id, variant, gpu_spec, app_slug,
186
+ * function, warm, ready, ready_runtimes,
187
+ * ready_since }],
188
+ * regions: [...], compliance: { zdr, hipaa } }
189
+ *
190
+ * Every function here is PURE and synchronous over the parsed block: the
191
+ * catalog fetch (U5's edge-function surface) is a separate seam and is NOT
192
+ * built yet — a caller hands the block in, or doesn't. An absent/empty
193
+ * `shared` block means "no shared models" and every lookup says so (null),
194
+ * never a stub.
195
+ *
196
+ * POSTURE (the two error classes, deliberately distinct):
197
+ * - `null` → this model is not in the shared block. The caller
198
+ * keeps its own resolution order; on the hosted lane that ends in the
199
+ * loud UnknownModelError 404. NOT an error — "not mine" is an answer.
200
+ * - `SharedCatalogError` → the block itself is broken (malformed shape, or
201
+ * a bare model id whose compat_default flag is not unique). This is a
202
+ * platform catalog defect, not a caller mistake: it surfaces as a LOUD
203
+ * 500 naming the row, never a silent resolution to "some" variant.
204
+ */
205
+ /** One shared endpoint row — the edge function's `endpoints[]` element. */
206
+ interface SharedEndpointRow {
207
+ model_id: string;
208
+ variant: string;
209
+ /** The shared org's deployed app slug the session dial targets. */
210
+ app_slug: string;
211
+ /** The serve function name on that app. */
212
+ function: string;
213
+ /** PLACEMENT-level GPU spec, e.g. 'rtx6000:1' (mirrors CatalogRow). */
214
+ gpu_spec?: string | null;
215
+ /** Idle-floor replica count (the reconciler's `warm`, U6). */
216
+ warm?: number | null;
217
+ /**
218
+ * The compat_default flag (U4): a BARE model id resolves to THE flagged
219
+ * row. Exactly one per model_id — zero or two is a loud catalog defect.
220
+ */
221
+ compat_default?: boolean | null;
222
+ /**
223
+ * The catalog row's output-free opt-in (owner decision 2026-09-21, the
224
+ * `reflex-latest:bf16` decision lane): a ZERO per-token OUTPUT rate is a
225
+ * deliberate published price for this endpoint — the model is a
226
+ * non-generative decision/action surface whose tokens are billed on
227
+ * input alone, so the OpenRouter document emits `cost_usd: '0'`
228
+ * completion pricing instead of refusing it. Absent (pre-flag edge
229
+ * function) and null both mean NOT opted in — the zero-output refusal
230
+ * stands.
231
+ */
232
+ output_free?: boolean | null;
233
+ /**
234
+ * Runtime-plane readiness (up-debounce, down-immediate upstream). ABSENT
235
+ * on blocks from an edge function that predates the readiness seam — the
236
+ * OpenRouter document builder treats that as NOT ready, loudly (see
237
+ * openrouter-doc.ts); `null` is likewise unknown, not ready.
238
+ */
239
+ ready?: boolean | null;
240
+ /** How many ready runtimes stand behind `ready` (null when unknown). */
241
+ ready_runtimes?: number | null;
242
+ /** ISO 8601 instant the endpoint became (and stayed) ready; null when not. */
243
+ ready_since?: string | null;
244
+ }
245
+ /** The catalog edge function's `shared` block. */
246
+ interface SharedCatalogBlock {
247
+ shared_org_id: string;
248
+ endpoints: SharedEndpointRow[];
249
+ /**
250
+ * Deployment-topology region codes (AWS-style, e.g. 'us-west-2') — the
251
+ * physical surface the shared endpoints serve from. Absent/empty means the
252
+ * topology is undeclared.
253
+ */
254
+ regions?: string[] | null;
255
+ /**
256
+ * Operator-declared data-handling posture (zero data retention, HIPAA).
257
+ * Present ONLY when an operator has declared BOTH flags — and nothing
258
+ * else: the wire contract is exactly {zdr, hipaa}, and the parser treats
259
+ * a further key as a catalog defect. ABSENT means UNDECLARED — never a
260
+ * default, because these are claims about where customer data goes.
261
+ */
262
+ compliance?: {
263
+ zdr: boolean;
264
+ hipaa: boolean;
265
+ } | null;
266
+ }
267
+
268
+ /**
269
+ * THE PER-TOKEN PRICE SURFACE — what `GET /v1/models` may say about money.
270
+ *
271
+ * Hugging Face reads our OpenAI-shaped model list to populate its public
272
+ * Inference-Provider comparison table and to power its `:fastest` and
273
+ * `:cheapest` routing. The required per-model shape is
274
+ *
275
+ * "pricing": { "input": <USD per MILLION input tokens>,
276
+ * "output": <USD per MILLION output tokens> }
277
+ *
278
+ * so this module resolves exactly that pair — per catalog model — out of
279
+ * urun-infra's `public.model_prices` surface (shared-endpoints U3,
280
+ * migration 20271003000003), and NOTHING else.
281
+ *
282
+ * OMISSION IS THE CONTRACT. A model with no per-token price row carries NO
283
+ * `pricing` key at all — never `0`, never a GPU-minute rate reinterpreted as
284
+ * a token rate, never an estimate. `model_prices` prices GPU TIME today
285
+ * (`unit in ('per_request','per_minute','per_output_unit')`, and the seeded
286
+ * rows are per_minute GPU-minutes); turning GPU-minutes into per-token rates
287
+ * is a PRICING DECISION a human makes, tracked as ENG-317. A fabricated
288
+ * number here would be a silent lie on a public comparison table that other
289
+ * people's routing decisions depend on, which is strictly worse than an
290
+ * absent field: HF renders an unpriced model as unpriced, and that is true.
291
+ *
292
+ * TWO GAPS THAT MUST CLOSE UPSTREAM BEFORE THIS SURFACE CAN EVER BE
293
+ * NON-EMPTY. Both are recorded on **ENG-317**, which is the durable record
294
+ * — this comment only points at it, because a code comment and a PR body
295
+ * are exactly the channels CLAUDE.md rule 2 names as insufficient:
296
+ *
297
+ * 1. NO PER-TOKEN UNIT EXISTS YET (ENG-317). `model_prices_unit_check`
298
+ * allows only per_request | per_minute | per_output_unit. The two unit
299
+ * strings this module recognizes — {@link PER_MILLION_INPUT_UNIT} and
300
+ * {@link PER_MILLION_OUTPUT_UNIT}, with `price_usd` read VERBATIM as
301
+ * USD per million tokens (no conversion, and `numeric(12,8)` has ample
302
+ * resolution at per-million scale — a per-single-token unit would not)
303
+ * — are a PROPOSED contract ENG-317 must ratify and a migration must
304
+ * add. Until then every live read yields zero per-token rows and every
305
+ * entry is unpriced. (If ENG-317 ratifies different names, a row
306
+ * carrying them is LOUD here, not skipped — see
307
+ * {@link KNOWN_NON_TOKEN_UNITS}.)
308
+ * 2. THE ANON READ IS RLS-BLOCKED (ENG-317). `model_prices` grants
309
+ * `select` to `anon` (migration 20271003000003 line 107), but its only
310
+ * policy (`model_prices_select_all`, line 112) is
311
+ * `for select to authenticated` — so a read with the proxy's catalog
312
+ * anon key returns `[]` whatever the table holds, indistinguishably
313
+ * from "no rows". (Contrast `model_catalog`, whose policy is
314
+ * `to anon, authenticated`; migration 20270226000000 calls that "the
315
+ * one intentional public read".) urun-infra must add the anon policy
316
+ * or publish prices through the catalog edge function.
317
+ *
318
+ * CONSEQUENCE, STATED PLAINLY: no `pricing` field can appear in production
319
+ * today. This module is correct and inert until gap 2 is fixed.
320
+ *
321
+ * WHAT IS A DEFECT (loud) vs WHAT IS SIMPLY UNPRICED (skipped):
322
+ * - a row whose `unit` is not one of the two per-token units → SKIPPED.
323
+ * A GPU-minute or per-request price is not evidence of a token rate; it
324
+ * is not this surface's business.
325
+ * - a row outside its effective window → SKIPPED. That is what
326
+ * effective-dating means.
327
+ * - a model with only ONE of the two rates → that model stays UNPRICED.
328
+ * Half a price is a lie; HF needs both numbers or neither.
329
+ * - a row that DOES claim a per-token unit but cannot be read (no
330
+ * model_id, unparseable price/date, a missing column, a non-object row,
331
+ * two rows of the same side with the same winning `effective_at` and
332
+ * DIFFERENT prices) → {@link ModelPriceError}. A broken price on a
333
+ * public table is worse than no price, and an order-dependent pick
334
+ * between two same-instant rows would be exactly the silent guess this
335
+ * repo forbids.
336
+ *
337
+ * THE ERROR TAXONOMY MATTERS TO CALLERS, so it is deliberate and narrow:
338
+ * {@link ModelPriceError} means THE DATA IS A DEFECT and no honest price can
339
+ * be derived — the /v1/models lane lets it PROPAGATE (a self-contradictory
340
+ * price table must be seen and fixed, not rounded down to "unpriced"). A
341
+ * plain `Error` from {@link fetchModelPriceRows} means the surface could
342
+ * not be READ (transport/status) — that one the listing may degrade past,
343
+ * because model DISCOVERY must not die with the price oracle. Those are the
344
+ * only two failure modes; nothing here returns an empty list to paper over
345
+ * either.
346
+ *
347
+ * THE SURFACE IS READ IN TWO STEPS, and the split is load-bearing:
348
+ * {@link parseModelPriceRows} validates the wire rows (timeless facts — a
349
+ * caller may CACHE these), and {@link resolveTokenPrices} answers which of
350
+ * them is in force AT AN INSTANT (a per-REQUEST question — caching THAT
351
+ * would serve a rate past its own `expires_at`). See each function's note.
352
+ */
353
+
354
+ /** USD per 1,000,000 tokens — verbatim the HF Inference-Provider shape. */
355
+ interface TokenPricing {
356
+ input: number;
357
+ output: number;
358
+ }
359
+ /** Which half of the pair a recognized unit names. */
360
+ type Side = 'input' | 'output';
361
+ /**
362
+ * ONE VALIDATED per-token price row — the cacheable, TIME-INDEPENDENT half
363
+ * of this surface. Everything here is a fact about the row itself; nothing
364
+ * about WHICH row is in force, because that depends on when you ask (see
365
+ * {@link resolveTokenPrices}). The read/validate step yields these; the
366
+ * time-dependent step consumes them.
367
+ */
368
+ interface ModelPriceRow {
369
+ model_id: string;
370
+ /** null = the model-level rate for every variant without its own row. */
371
+ variant: string | null;
372
+ side: Side;
373
+ /** USD per 1,000,000 tokens, verbatim from `price_usd`. */
374
+ usd: number;
375
+ /** Epoch ms of `effective_at`. */
376
+ effective_at: number;
377
+ /** Epoch ms of `expires_at`, or null for a row that never expires. */
378
+ expires_at: number | null;
379
+ }
380
+
381
+ /**
382
+ * THE OPENROUTER PROVIDER DOCUMENT — `GET /v1/models?format=openrouter`.
383
+ *
384
+ * OpenRouter's provider monitor polls this document (schema 2.4, per
385
+ * openrouter.ai/docs/guides/community/for-providers §1) to list a provider's
386
+ * models on the marketplace. The document answers ONE question: "all models
387
+ * that should be served by OpenRouter" — the platform's SELLABLE surface.
388
+ * That surface is the catalog edge function's SHARED block (the shared
389
+ * endpoints OpenRouter's callers resolve by `model_id:variant` on the chat
390
+ * surface); caller-org apps are deliberately NOT in it. A caller-org app is
391
+ * a tenancy-private chat surface behind one org's key — declaring it here
392
+ * would sell another tenant's private deployment as a public marketplace
393
+ * SKU. The document still rides the tenancy seam (`openRouterModels` on the
394
+ * tenant client), so each caller sees the same sellable listing.
395
+ *
396
+ * THE SHARED JOIN: each endpoint joins the catalog EXACTLY on
397
+ * (model_id, variant) — the SAME join the shared-lane consumers (supportsAudio,
398
+ * modelList's shared arm, imageModelList) already apply. No slug heuristics: the
399
+ * shared org's app slug need not exist in
400
+ * the caller's catalog, and a slug guess would be a second, bespoke join.
401
+ *
402
+ * MODALITY GATING: only rows whose catalog `task` names a chat-completions
403
+ * shape (chat | code | agent | vl) are declared. A voice/video/image app
404
+ * behind a chat surface would stream garbage into OpenRouter's baseline
405
+ * tests; models the catalog cannot vouch for are OMITTED, never guessed.
406
+ * With no shared block (or no catalog oracle) configured the document is
407
+ * `{"data": []}` — honest emptiness, not invented capability.
408
+ *
409
+ * PRICING — schema 2.4 nests `pricing` arrays on the modality that owns
410
+ * them ({type 'prompt'} on the text INPUT modality, {type 'completion'} on
411
+ * the text OUTPUT modality, `cost_usd` a USD string PER TOKEN). The price
412
+ * source is model-prices.ts — the SAME per-token join the OpenAI listing
413
+ * uses ({@link pricingFor}); the per-million rate is shifted to per-token
414
+ * on its decimal string, never by float division. An entry with no
415
+ * per-token row — today MOST of them, because `model_prices` prices GPU
416
+ * minutes (`per_minute` et al.), which is not evidence about tokens —
417
+ * carries NO pricing array at all: the schema's rule is "a modality with
418
+ * no pricing array is simply unpriced". Never a zero, never a GPU-minute
419
+ * rate relabelled as a token rate (ENG-317 owns the human pricing
420
+ * decision that would make such rows exist) — with ONE owner-ratified
421
+ * exception: `reflex-latest:bf16` is a non-generative decision lane whose
422
+ * catalog row declares `output_free` (owner decision 2026-09-21), so its
423
+ * ZERO completion rate is the published price (`cost_usd: '0'`), billed
424
+ * on input alone. The zero reaches this builder only through
425
+ * model-prices.ts's allowlist (routing resolves prices against the
426
+ * shared block's `output_free` set); the builder itself just renders
427
+ * whatever a resolved price says.
428
+ *
429
+ * The full closed-value-domain schema ships as OpenAPI 3.1 at
430
+ * openrouter.ai/docs/assets/provider-monitor-schema-v2.openapi.json.
431
+ *
432
+ * THE OPERATIONAL FIELDS (schema 2.4, the "launch control" surface):
433
+ * - `is_ready` is the shared endpoint's SELLABLE readiness:
434
+ * `runtime ready AND priced` — `endpoint.ready === true` AND a per-token
435
+ * price row for this exact (model_id, variant). Priced means the price
436
+ * book RESOLVES a pair for the SKU: an output-free row's `{input: >0,
437
+ * output: 0}` is a resolved pair (model-prices.ts's allowlist decides,
438
+ * from the shared block's `output_free` flag), so an input-priced,
439
+ * output-free, runtime-ready endpoint IS ready. An unpriced SKU is
440
+ * never sellable: OpenRouter's `is_ready: false` keeps an endpoint
441
+ * hidden, and we must never expose a model we cannot bill, so a ready
442
+ * runtime without a price is emitted NOT ready. Runtime readiness is
443
+ * THE ONE DELIBERATELY CONSERVATIVE DEFAULT
444
+ * in this module: a block whose endpoints carry NO `ready` field at all
445
+ * predates the readiness seam (the catalog edge function is not yet
446
+ * redeployed), and UNKNOWN READINESS IS NOT READINESS — every such
447
+ * endpoint is declared `is_ready: false` AND one loud warning per
448
+ * document build names the missing seam, so the default is visible,
449
+ * never silent. A `null` ready is likewise unknown, not ready. Endpoints
450
+ * of the SAME identity follow the existing dedup rule (first occurrence
451
+ * in the deterministic shared-block order wins).
452
+ * - `is_free` is `false` for every declared model: we bill. An unpriced row
453
+ * is UNPRICED, not free — `is_free: true` would mint a `:free` SKU whose
454
+ * every price OpenRouter ignores. The same holds for an output-free row:
455
+ * its input side bills, so it is not a `:free` SKU either — the zero
456
+ * lives in the completion `cost_usd`, nowhere else.
457
+ * - `deprecation_date` is OMITTED: the schema makes it optional and no
458
+ * shared endpoint has a deprecation schedule to declare.
459
+ * - `datacenters` maps the block's deployment-topology region codes
460
+ * (AWS-style) to the schema's `{ country_code, region }` elements through
461
+ * an EXPLICIT region→country table (a country cannot be derived from a
462
+ * prefix — sa-east-1 is BR). Unmappable codes are omitted and WARNED,
463
+ * never guessed; absent/empty topology omits the field.
464
+ * - `compliance` is emitted ONLY when the block's `compliance` is present:
465
+ * zdr/hipaa are claims about where customer data goes, DECLARED by an
466
+ * operator — never inferred, never defaulted, and never half-emitted.
467
+ * - `tokenizer` names the model family the catalog row's `hf_repo`
468
+ * carries (e.g. `Inferact/Qwen3.8-27B-NVFP4` → `Qwen`) through an
469
+ * explicit, BOUNDED table over the repo's model-NAME tokens — whole
470
+ * tokens only, version-aware for the families whose generations differ
471
+ * (llama-2 is not llama-3); a repo that names no known family OMITS the
472
+ * field.
473
+ */
474
+
475
+ type OpenRouterProviderDoc = {
476
+ data: OpenRouterProviderModel[];
477
+ };
478
+ /** One schema-2.4 `datacenters` element: ISO 3166-1 alpha-2 country + region. */
479
+ interface OpenRouterDatacenter {
480
+ country_code: string;
481
+ region?: string;
482
+ }
483
+ /** The operator-declared data-handling posture (schema-2.4 `compliance`). */
484
+ interface OpenRouterCompliance {
485
+ zdr: boolean;
486
+ hipaa: boolean;
487
+ }
488
+ interface OpenRouterProviderModel {
489
+ schema_version: '2.4';
490
+ /**
491
+ * The EXACT id OpenRouter sends back as `model`: the verbatim
492
+ * `<model_id>:<variant>` external identity (see MODEL-IDENTITY.md) — never
493
+ * the deployment slug.
494
+ */
495
+ id: string;
496
+ name: string;
497
+ created: number;
498
+ /** Valid enum: int4|int8|fp4|mxfp4|nvfp4|fp6|fp8|mxfp8|fp16|bf16|fp32|null. */
499
+ quantization: string | null;
500
+ /** Tokenizer family name (e.g. 'Qwen'); omitted when the repo names none. */
501
+ tokenizer?: string;
502
+ description: string;
503
+ hugging_face_id: string;
504
+ /** SELLABLE readiness — `endpoint.ready === true` AND a per-token price. */
505
+ is_ready: boolean;
506
+ /** `false` always: we bill; an unpriced row is unpriced, not free. */
507
+ is_free: boolean;
508
+ input_modalities: Array<Record<string, unknown>>;
509
+ output_modalities: Array<Record<string, unknown>>;
510
+ /** Physical serving topology; omitted when the block declares none. */
511
+ datacenters?: OpenRouterDatacenter[];
512
+ /** Operator-declared posture; omitted entirely when undeclared. */
513
+ compliance?: OpenRouterCompliance;
514
+ }
515
+ /**
516
+ * The resolution of ONE tts model's voice set — ONE answer BOTH consumers
517
+ * read, so the lane can never drift from the listing:
518
+ * - `published` is the set the model's /v1/models entry carries ([] =
519
+ * publish none).
520
+ * - `defect` is non-null ONLY when the catalog PRESENTED
521
+ * `engine_args.voices` that could not be vouched (empty, malformed, or
522
+ * not an identical non-empty string array on every GPU placement). The
523
+ * publish side keeps its long-standing posture through a defect (the
524
+ * card list for the qwen3 class, nothing otherwise — the listing is a
525
+ * best-effort enrichment), but the speech lane must never silently
526
+ * validate against the card's nine over a catalog that SPOKE — it fails
527
+ * loud (500) on `defect` instead.
528
+ */
529
+ interface SpeechVoicesResolution {
530
+ /** The voices the /v1/models entry publishes for this model ([] = none). */
531
+ readonly published: readonly string[];
532
+ /** Why the catalog's own voices value could not be vouched, when it presented one. */
533
+ readonly defect: string | null;
534
+ }
535
+
536
+ /**
537
+ * Per-model → per-app routing for the compat proxy (owner directive
538
+ * 2026-08-07): swapping the model in a coding harness routes the request to
539
+ * the org's DEPLOYED app for that model. v1 is deployed-only — a uRun model
540
+ * that is not deployed gets a loud 404 naming `urun serve <id>`; a later
541
+ * phase (explicitly out of scope here; urun-infra#1490 shared-endpoints)
542
+ * auto-creates from the model catalog on first request.
543
+ *
544
+ * MODEL-ID SURFACE (the documented mapping): a model may be named by
545
+ * - the app slug itself ("qwen3-6-27b-bf16"), or
546
+ * - the catalog id ("qwen3.6-27b"), or
547
+ * - the catalog id:variant ("qwen3.6-27b:bf16"),
548
+ * where slugification mirrors urun-cli `serve.py _default_app_name` exactly:
549
+ * lowercase, every non-alphanumeric-non-dash character becomes "-", leading/
550
+ * trailing dashes stripped (catalog id "qwen3.6-27b" + variant "bf16" → app
551
+ * "qwen3-6-27b-bf16"). COLLISION RULE: an exact slug match always wins over
552
+ * the catalog-id (prefix) interpretation.
553
+ *
554
+ * RESOLUTION ORDER (one canonical path, documented end to end):
555
+ * 1. model absent / "urun" / an alias of the startup app → DEFAULT app.
556
+ * 2. exact slug match on a deployed app that declares serves="openai" →
557
+ * that app.
558
+ * 3. catalog-id form matching exactly one deployed app → that app
559
+ * (two or more candidates → loud ambiguity error naming them).
560
+ * 4. the name maps to an org app that has NO function declaring the
561
+ * serve protocol (`serves="openai"`) → loud 404
562
+ * naming `urun serve <model>` and the available models.
563
+ * 5. the name matches a catalog row but no deployed app → loud 404
564
+ * naming `urun serve <id>` (deployed-only v1).
565
+ * 6. anything else — a model name outside the uRun namespace entirely
566
+ * (e.g. the harness's own upstream default, "claude-*"/"gpt-*") →
567
+ * DEFAULT app. This IS today's single-app contract, kept deliberately
568
+ * so `urun compat <agent>` with the agent's stock model keeps working
569
+ * with zero new env; the per-model /stats table records every such
570
+ * mapping so it is visible, never silent. Models the proxy ADVERTISES
571
+ * on /v1/models can never land here — they resolve (2/3) or fail loud
572
+ * (4/5) above.
573
+ *
574
+ * NO-DEFAULT MODE (`defaultApp: null`) — the HOSTED multi-tenant lane
575
+ * (`src/hosted/`): a shared endpoint serving every org has no "the app this
576
+ * proxy was started for", so rules 1 and 6 have nothing to fall back TO.
577
+ * Rather than inventing one (picking "some" app for a caller would be the
578
+ * worst kind of silent divergence), both rules become the SAME loud
579
+ * {@link UnknownModelError} that rules 4/5 already raise: name a deployed
580
+ * model, here is the list. Rules 2–5 are byte-for-byte the local behavior —
581
+ * one router, one resolution order, two configurations.
582
+ */
583
+
584
+ /** One org app row from `GET {orgApi}/apps` (urun-cli `ApiClient.list_apps`). */
585
+ interface DeployedApp {
586
+ app_slug: string;
587
+ /** The serve function the BACKHAUL dials on the app (URUN_FUNCTION); no
588
+ * longer a routing membership test — routing reads `serves`. */
589
+ function_name?: string | null;
590
+ /** Declared serve protocol (`"openai"`), or null/absent = undeclared = the
591
+ * proxy must not route this app. THE routing membership test. */
592
+ serves?: string | null;
593
+ deployment_status?: string | null;
594
+ }
595
+ /**
596
+ * A model that does not resolve to a deployed app the proxy may serve.
597
+ * Rendered by the server in each lane's NATIVE error format as a 404 — never
598
+ * silently served by the default app.
599
+ */
600
+ declare class UnknownModelError extends Error {
601
+ }
602
+ /**
603
+ * A session handle names a pooled session that no longer exists (closed,
604
+ * evicted after its pod died, replaced by a re-home, or the proxy restarted).
605
+ * The caller asked to REATTACH that exact session — opening a fresh one and
606
+ * calling it "resumed" would be a silent lie, so this is always loud.
607
+ */
608
+ declare class SessionGoneError extends Error {
609
+ }
610
+ /**
611
+ * OpenAI-shaped model list (same shape as models.ts listModels). When the
612
+ * catalog oracle is configured, each entry ALSO carries the catalog
613
+ * enrichment fields (additive JSON — OpenAI clients ignore unknown fields;
614
+ * the OpenRouter + Vercel AI Gateway provider listings need them). Entries
615
+ * whose slug matches no catalog row stay at the bare four fields.
616
+ */
617
+ interface RouterModelEntry {
618
+ id: string;
619
+ object: 'model';
620
+ created: number;
621
+ owned_by: string;
622
+ /** Display name — the canonical `<model_id>:<variant>` catalog ref. */
623
+ name?: string;
624
+ /** User-facing description from the catalog (console Endpoints copy). */
625
+ description?: string;
626
+ /** Catalog modality (chat | code | agent | vl | audio | image | ...). */
627
+ task?: string;
628
+ /** Serving engine (vllm | sglang | llamacpp | ...). */
629
+ engine?: string;
630
+ /** The catalog lane this app deploys, e.g. 'rtx6000:1'. */
631
+ gpu_spec?: string;
632
+ /** Context window in tokens (chat rows carry it in engine_args). */
633
+ context_length?: number;
634
+ /**
635
+ * Per-token price, USD per MILLION tokens — the exact shape Hugging Face
636
+ * reads off `GET /v1/models` for its provider comparison table and its
637
+ * `:fastest`/`:cheapest` routing. ABSENT means UNPRICED and is the honest
638
+ * answer for every model with no per-token price row (model-prices.ts);
639
+ * a zero or an estimate here would be a silent lie on a public table.
640
+ */
641
+ pricing?: TokenPricing;
642
+ /** Idle-floor replica count for a shared-lane model (U6 reconciler). */
643
+ warm?: number;
644
+ /**
645
+ * The voices a `tts`-task model synthesizes under (ENG-331 — OpenAI's
646
+ * speech API names a voice per request, so a caller needs the list).
647
+ * Catalog-sourced from `engine_args.voices` when the row carries it;
648
+ * otherwise the qwen3 CustomVoice timbres for that one model class (see
649
+ * openrouter-doc.ts `QWEN3_CUSTOMVOICE_VOICES`). Absent on every non-tts
650
+ * entry, and on tts entries whose voices the catalog does not vouch.
651
+ */
652
+ voices?: string[];
653
+ }
654
+ interface RouterModelList {
655
+ object: 'list';
656
+ data: RouterModelEntry[];
657
+ }
658
+ /** One `/v1/images/models` entry: the OpenAI model object plus the image
659
+ * modes the catalog rows vouch for (the SAME per-placement consensus
660
+ * {@link ModelRouter.supportsImages} applies — a row set that vouches
661
+ * nothing is NOT listed). */
662
+ interface RouterImageModelEntry extends RouterModelEntry {
663
+ image_modes: ImageCapabilityMode[];
664
+ }
665
+ interface RouterImageModelList {
666
+ object: 'list';
667
+ data: RouterImageModelEntry[];
668
+ }
669
+ /**
670
+ * The dial target one resolved model names: the app slug plus — for a SHARED
671
+ * lane (the catalog `shared` block, the cross-org session dial) — the exact
672
+ * serve function on the shared org's app and the flag that switches the
673
+ * control plane's shared-admission carve-out on. A plain caller-org app
674
+ * carries only `appSlug`; the router's `fnName` (the function the backhaul
675
+ * dials) applies there.
676
+ */
677
+ interface RouteTarget {
678
+ appSlug: string;
679
+ /** The serve function on the TARGET app (a shared lane pins it per row). */
680
+ fnName?: string;
681
+ /**
682
+ * CROSS-ORG SHARED DIAL: the backhaul mints the client token with
683
+ * `sharedApp: true` and starts the session with `shared_app: true`, so the
684
+ * control plane resolves the app from the platform's shared org while the
685
+ * session and its usage stay attributed to the CALLER org.
686
+ */
687
+ sharedApp?: boolean;
688
+ /** Shared lanes only: the catalog id the request pinned (capability joins). */
689
+ modelId?: string;
690
+ /** Shared lanes only: the catalog variant the request pinned. */
691
+ variant?: string;
692
+ /**
693
+ * Shared lanes only: the pod org (`shared.shared_org_id` of the block that
694
+ * resolved the lane). THE STATELESS DECISION TOKEN'S PROVIDER CLAIM — the
695
+ * decision intake's provider-aware org gate accepts tenant OR provider
696
+ * matching the pod org, and on a shared pod the caller-org tenant differs
697
+ * from the shared-org pod. Sourced from the SAME block that resolved the
698
+ * lane, never a second env spelling, so key and pod-org identity cannot
699
+ * drift.
700
+ */
701
+ sharedOrgId?: string;
702
+ }
703
+ interface ModelRouterOptions<S> {
704
+ /**
705
+ * The startup app slug (URUN_APP) — the DEFAULT model — or `null` for the
706
+ * hosted multi-tenant lane, which has no per-proxy default app: there,
707
+ * every request must NAME a deployed model and an unnamed/unknown one
708
+ * fails loud instead of silently landing somewhere (see the module header,
709
+ * "NO-DEFAULT MODE").
710
+ */
711
+ defaultApp: string | null;
712
+ /**
713
+ * The serve function the BACKHAUL dials on the resolved app (URUN_FUNCTION)
714
+ * — shared lanes may pin their own per-row function instead (RouteTarget
715
+ * fnName). NOT a routing membership test any more: routing reads the app's
716
+ * declared `serves` protocol (see {@link ModelRouter.servable}).
717
+ */
718
+ fnName: string;
719
+ /**
720
+ * Open a backhaul session for a resolved dial target (called at most once
721
+ * per pool key). The SECOND argument is the POOL KEY the entry will
722
+ * occupy — the same string `sessionFor`/`speechSessionFor` store it under
723
+ * — so an implementer can wire per-slot machinery (the backhaul's
724
+ * gone-watch eviction and the speech lane's subject scoping) against the
725
+ * slot the entry ACTUALLY holds: a lane-scoped open (`speech:<slug>`)
726
+ * that wired its eviction on the unscoped key would leave a dead entry
727
+ * stranded in the pool forever. Implementers that ignore it behave
728
+ * exactly as before.
729
+ */
730
+ openSession: (target: RouteTarget, poolKey: string) => S | Promise<S>;
731
+ /** Terminal release for one pool entry (Session.end() underneath). */
732
+ closeSession: (entry: S) => Promise<void>;
733
+ /**
734
+ * List the org's deployed apps, or null when the credentials cannot
735
+ * (URUN_JWT lane: the pre-vended token is scoped to the default app, so
736
+ * there is no org listing AND no cross-app session — routing degrades to
737
+ * the default-app-only contract, which is exactly today's behavior).
738
+ */
739
+ listApps: (() => Promise<DeployedApp[]>) | null;
740
+ /**
741
+ * Catalog rows (models.ts fetchCatalogRows) as the uRun-namespace oracle
742
+ * for rule 5 AND the /v1/models enrichment + OpenRouter provider-doc
743
+ * source, or null when catalog access is not configured. Optional fields
744
+ * beyond model_id/variant are tolerated (thin rows still typecheck).
745
+ */
746
+ listCatalog: (() => Promise<CatalogRow[]>) | null;
747
+ /**
748
+ * The VALIDATED per-token price rows (model-prices.ts
749
+ * `fetchModelPriceRows`) behind the `pricing` field of the /v1/models
750
+ * listing, or null when no price source is configured. A REQUIRED key
751
+ * exactly like {@link listCatalog}: an optional one would let a lane ship
752
+ * a silently unpriced public listing without ever saying so.
753
+ *
754
+ * These are ROWS, not resolved prices, precisely so the router may cache
755
+ * them: which row is in force is a per-REQUEST question answered by
756
+ * `resolveTokenPrices` against the exact `effective_at` / `expires_at`
757
+ * boundaries. Handing back already-resolved prices would let a cached
758
+ * answer outlive the window it was resolved in.
759
+ */
760
+ listPrices: (() => Promise<ModelPriceRow[]>) | null;
761
+ /**
762
+ * Where this router announces a DEGRADATION — an unreadable price
763
+ * surface, a model refused a price, a catalog blip that suppresses
764
+ * pricing. Optional only in WHERE it goes: left unset the router writes
765
+ * to `console.warn`, never to nothing (see {@link ModelRouter.warn}).
766
+ * Mirrors the `warn` sink `catalogFromEnv` (cli.ts) already takes.
767
+ */
768
+ warn?: (line: string) => void;
769
+ /** Deployed-apps cache TTL (the list changes on deploys, not per request). */
770
+ appsTtlMs?: number;
771
+ /**
772
+ * The stable NATIVE identity of one pooled entry (the uRun session id in
773
+ * the proxy wiring — the same identity the serve-side session-affinity tag
774
+ * rides, urun-python#1556). Powers the session-identity seam
775
+ * ({@link ModelRouter.handleFor} / {@link ModelRouter.sessionForHandle});
776
+ * a router without it fails LOUD on those calls, never approximates.
777
+ */
778
+ sessionKey?: (entry: S) => string;
779
+ /**
780
+ * The catalog edge function's `shared` block (shared-endpoints U8), or
781
+ * null when shared-model routing is not configured — every phase-1 caller
782
+ * leaves it absent and behaves byte-for-byte as before. Same TTL cache
783
+ * discipline as the apps/catalog oracles.
784
+ */
785
+ shared?: (() => Promise<SharedCatalogBlock | null>) | null;
786
+ /**
787
+ * OWNER RULING (2026-09-21, shared-endpoints inference-proxy): the model
788
+ * listing is the platform's SHARED surface and nothing else. When true,
789
+ * {@link ModelRouter.modelList} advertises ONLY the catalog `shared`
790
+ * block's endpoints and never the caller-org's deployed apps — the same
791
+ * id space {@link ModelRouter.openRouterModels} publishes, so every
792
+ * listing format agrees. The hosted lane (hosted/tenants.ts) sets this;
793
+ * the local `urun compat` lane leaves it absent and behaves
794
+ * byte-for-byte as before (a caller's own apps stay listed there).
795
+ * An absent shared block (or none configured) is an EMPTY listing —
796
+ * "no shared models" is an answer, never a fallback to the org's apps.
797
+ */
798
+ sharedOnlyListing?: boolean;
799
+ }
800
+ /**
801
+ * The image-capability modes the Images lane gates on (images.ts's
802
+ * ImageCapabilityMode — declared here because the router is the LOWER layer:
803
+ * the lane's gate type stays structurally satisfied by
804
+ * {@link ModelRouter.supportsImages} with no router→lane import).
805
+ */
806
+ type ImageCapabilityMode = 'generation' | 'edit';
807
+ /**
808
+ * The session pool: one backhaul session per deployed app, keyed by app slug,
809
+ * opened lazily on the first request that routes to it and reused for every
810
+ * subsequent one. The startup app is seeded eagerly by the CLI. Sessions
811
+ * close on proxy shutdown via {@link closeAll}; there is NO idle-close policy
812
+ * (deliberate v1 simplification — noted as a follow-up in the PR).
813
+ */
814
+ declare class ModelRouter<S> {
815
+ private readonly opts;
816
+ private readonly pool;
817
+ private appsCache;
818
+ private rowsCache;
819
+ constructor(opts: ModelRouterOptions<S>);
820
+ /** Seed an already-open session (the CLI's eagerly-opened startup app). */
821
+ seed(appSlug: string, entry: S): void;
822
+ private deployedApps;
823
+ /**
824
+ * The shared block with the same TTL discipline as {@link deployedApps}.
825
+ * Null when not configured. Malformed blocks throw loudly
826
+ * (SharedCatalogError) on first use — never a silently half-empty table.
827
+ */
828
+ private sharedBlock;
829
+ private sharedCache;
830
+ /**
831
+ * Catalog rows with the same TTL discipline as {@link deployedApps} (the
832
+ * catalog changes on reseed migrations, not per request). Null when no
833
+ * oracle is configured — callers degrade to the bare four-field entries.
834
+ */
835
+ private catalogRows;
836
+ private pricesCache;
837
+ /**
838
+ * The per-token price ROWS with the same TTL discipline as
839
+ * {@link catalogRows} (the price book changes on ops writes, not per
840
+ * request). No source configured ⇒ [] — every model unpriced, which is
841
+ * the honest listing until per-token rows exist (ENG-317).
842
+ *
843
+ * ROWS, never resolved prices: the rows are timeless facts, so caching
844
+ * them is safe, whereas caching the prices IN FORCE would keep serving a
845
+ * rate past its own `expires_at` — or withhold one past its
846
+ * `effective_at` — for the rest of the TTL window. The in-force question
847
+ * is answered per request in {@link pricesForListing}.
848
+ */
849
+ private priceRows;
850
+ /**
851
+ * The price surface for the /v1/models listing. Three distinct outcomes,
852
+ * and NONE of them is silent:
853
+ *
854
+ * - the surface cannot be READ (transport, status, or the read's own
855
+ * deadline — a plain Error) ⇒ every model lists UNPRICED and the
856
+ * reason is WARNED. Model discovery must not die with the price
857
+ * oracle, but an outage rendering as an ordinary unpriced listing —
858
+ * indistinguishable from today's inert state — is exactly the silent
859
+ * degradation CLAUDE.md forbids, so it says so out loud.
860
+ * - one model's rows are a DEFECT (a duplicate rate for one instant, a
861
+ * zero rate) ⇒ that MODEL is unpriced and the refusal is WARNED. The
862
+ * blast radius is the model the bad row belongs to; it used to be the
863
+ * whole listing for every tenant.
864
+ * - the READ itself is not the surface we contracted for (a malformed
865
+ * row, a missing column, an unknown `unit` — {@link ModelPriceError})
866
+ * ⇒ PROPAGATES. Nothing in a payload that violates its own column
867
+ * contract can be trusted, so half a price table is not served as if
868
+ * complete — the same posture `parseSharedBlock` takes on a malformed
869
+ * routing block.
870
+ *
871
+ * `outputFree` is the shared block's EXACT `(model_id, variant)` opt-in
872
+ * set ({@link outputFreeKey} keys, built from `SharedEndpointRow.output_free`):
873
+ * the only rows whose zero OUTPUT rate the resolver may publish — see
874
+ * model-prices.ts for the one-flag-one-lane rule. Absent/empty set: every
875
+ * zero keeps the refusal.
876
+ *
877
+ * The resolve step sits OUTSIDE the catch deliberately: it is our own
878
+ * code, so a programming error in it must surface as a crash, not become
879
+ * an empty price list.
880
+ */
881
+ private pricesForListing;
882
+ /**
883
+ * The EXACT `(model_id, variant)` pairs the shared block declares
884
+ * output-free ({@link outputFreeKey} keys) — the only allowlist
885
+ * resolveTokenPrices accepts a zero OUTPUT rate against. Only a
886
+ * `true` flag contributes: absent and null are the un-flagged default.
887
+ */
888
+ private outputFreeKeys;
889
+ /**
890
+ * Where a degradation says so. Defaults to `console.warn` rather than to
891
+ * nothing: a caller may ROUTE the signal (the CLI sends it to stderr, as
892
+ * `catalogFromEnv` already does), but no caller can switch it off, because
893
+ * an unannounced degradation is the failure mode this repo treats as a
894
+ * time bomb.
895
+ */
896
+ private warn;
897
+ /**
898
+ * Apps this proxy may serve: active AND declaring the OpenAI-compatible
899
+ * serve protocol (`@app.function(serves="openai")`). The OLD rule — a
900
+ * function literally named `serve` — is GONE: the name carried no
901
+ * capability meaning (an app with a function named `serve_runtime` could
902
+ * never be routed), and the declaration is what the deploy actually knows.
903
+ * `fnName` survives only as the function the BACKHAUL dials.
904
+ */
905
+ private servable;
906
+ private availableIds;
907
+ /**
908
+ * NO-DEFAULT MODE's terminal for rules 1 and 6: there is no app to fall
909
+ * back to, so say so loudly and list what the CALLER'S org actually has.
910
+ * Never returns.
911
+ */
912
+ private noDefaultApp;
913
+ /**
914
+ * Resolve a request's `model` to its dial target — the documented
915
+ * resolution order from the module header. Throws
916
+ * {@link UnknownModelError} for a uRun model that is not deployed (rules
917
+ * 4/5). A SHARED-lane hit (the catalog `shared` block) resolves to the
918
+ * SHARED org's app with `sharedApp: true` — the cross-org session dial.
919
+ */
920
+ private resolveTarget;
921
+ /**
922
+ * THE BILLING IDENTITY of a request's model: the CATALOG `(model_id,
923
+ * variant)` the request actually resolved to, or `null` when the route
924
+ * names no catalog model.
925
+ *
926
+ * WHY THE LEDGER CANNOT USE THE CALLER'S STRING INSTEAD. The per-request
927
+ * ledger prices on `(model_id, variant)` (urun-infra
928
+ * `urun_record_inference_request`), and the caller's `model` is not that:
929
+ * a BARE catalog id resolves through {@link resolveSharedModel} to the
930
+ * model's `compat_default` VARIANT, so recording the raw string would file
931
+ * the request under a variant nobody served and price it off the
932
+ * model-level wildcard — a rate that exists precisely to price the variants
933
+ * nobody named explicitly. Splitting the string on its last `:` would be a
934
+ * heuristic on a money path. This is the resolution the dial itself used.
935
+ *
936
+ * NULL FOR A CALLER-ORG APP, and that is a REFUSAL rather than a gap. An
937
+ * org's own deployed app has no catalog identity at all — its identity is
938
+ * an app slug, which is a different namespace from `model_prices.model_id`
939
+ * — so there is no honest `(model_id, variant)` to record. The ledger
940
+ * writer declines to write such a request rather than inventing one; see
941
+ * `proxy/ledger.ts`.
942
+ *
943
+ * CHEAP TO CALL: it rides {@link resolveTarget}, whose app listing, catalog
944
+ * and shared block are all cached on this router, so the ledger writer can
945
+ * ask AFTER the response has closed without re-hitting the control plane.
946
+ */
947
+ catalogIdentityFor(model: string | undefined): Promise<{
948
+ model_id: string;
949
+ variant: string;
950
+ } | null>;
951
+ /**
952
+ * THE CALLER-ORG APP LANE's OWN IDENTITY (ENG-412) — the app slug this
953
+ * request actually resolves to.
954
+ *
955
+ * The ledger records it for the lane {@link catalogIdentityFor} answers null
956
+ * for, and the caller's RAW `model` string is not it: routing trims, caps and
957
+ * matches that string before selecting an app, so two spellings of one app
958
+ * would otherwise be filed as two identities and no app-level reconciliation
959
+ * could undo it afterwards.
960
+ *
961
+ * Rides {@link resolveTarget}'s cached listings and OPENS NO SESSION.
962
+ */
963
+ servingAppSlugFor(model: string | undefined): Promise<string>;
964
+ /**
965
+ * THE STATELESS DIRECT PATH's resolution seam: the dial target for a
966
+ * model — the pure resolution (app slug, fn, shared flag) with NO session
967
+ * opened. The stateless decision dial (stateless-dispatch.ts) uses it to
968
+ * discover ready replicas for the resolved (org, app, fn) instead of
969
+ * minting a session; every public no-dial accessor above rides this same
970
+ * private resolution.
971
+ */
972
+ targetFor(model: string | undefined): Promise<RouteTarget>;
973
+ /**
974
+ * Resolve a request's `model` to its POOL KEY — the app slug for a
975
+ * caller-org app, `shared:<slug>` for a shared-lane dial (the two never
976
+ * share a pool slot, {@link poolKeyOf}).
977
+ */
978
+ resolveApp(model: string | undefined): Promise<string>;
979
+ /**
980
+ * The pooled session for a model — opened lazily, reused afterwards.
981
+ *
982
+ * IT ALSO REPORTS THE DIAL'S BILLING IDENTITY (ENG-414), off the SAME
983
+ * `resolveTarget` answer it just selected the session with. That is the
984
+ * whole point of returning it from here: the ledger must be able to name
985
+ * what served a request WITHOUT taking a second resolution. This router
986
+ * reads its app list, catalog and shared block through caches with a TTL, so
987
+ * a second resolution taken microseconds later can answer differently — and
988
+ * then the row names a variant nobody served. `rehome.ts` reports this
989
+ * identity once per dial, so a request that re-homes is billed on the
990
+ * REPLACEMENT's identity rather than the abandoned dial's.
991
+ *
992
+ * THE POOL ENTRY COULD NOT CARRY IT. `S` is the generic pooled session and
993
+ * knows nothing about models; the identity is a property of the ROUTE, which
994
+ * is what this method resolved and the entry never saw.
995
+ */
996
+ sessionFor(model: string | undefined): Promise<{
997
+ app: string;
998
+ entry: S;
999
+ identity: TurnCatalogIdentity;
1000
+ }>;
1001
+ /**
1002
+ * The SPEECH lane's pooled session for a model — the SAME resolution and
1003
+ * acquisition as {@link sessionFor}, stored under a LANE-SCOPED pool key
1004
+ * (`speech:` + the ordinary key) so the session a speech collector
1005
+ * subscribes to is one NO interactive lane (text turns via
1006
+ * `rehomingCreateResponse`/`sessionFor`, Realtime and Gemini Live via
1007
+ * `openAudio`/`handleFor`) can ever resolve: the per-session AudioBridge
1008
+ * broadcasts every untagged output frame to every subscriber, so a shared
1009
+ * session is exactly the mixed-audio defect (greptile urun-ts#500). The
1010
+ * entry's OWN lifecycle (gone-watch eviction, wedge re-salt, billing
1011
+ * identity) is identical — only the slot differs, which is the whole point.
1012
+ */
1013
+ speechSessionFor(model: string | undefined): Promise<{
1014
+ app: string;
1015
+ entry: S;
1016
+ identity: TurnCatalogIdentity;
1017
+ }>;
1018
+ /**
1019
+ * The ONE acquisition path both lanes ride — opened lazily under exactly
1020
+ * `key`, reused afterwards. A failed open never poisons the slot.
1021
+ */
1022
+ private acquire;
1023
+ private keyOf;
1024
+ /**
1025
+ * SESSION-IDENTITY SEAM (a): the opaque stable handle for the pooled
1026
+ * session currently serving `model`'s turns. Rides the SAME acquisition
1027
+ * path as every request ({@link sessionFor}) — the session opens lazily if
1028
+ * this model has none yet — and derives the handle from native identity
1029
+ * (app slug + uRun session id), zero bespoke bookkeeping.
1030
+ */
1031
+ handleFor(model: string | undefined): Promise<{
1032
+ app: string;
1033
+ handle: string;
1034
+ }>;
1035
+ /**
1036
+ * SESSION-IDENTITY SEAM (a), SPEECH lane: the handle for the SPEECH-scoped
1037
+ * pooled session ({@link speechSessionFor} — `speech:<key>` as the handle's
1038
+ * pool key, so {@link sessionForHandle} resolves exactly the speech slot).
1039
+ * The handle codec needs no change: it already carries whatever pool key
1040
+ * minted it, and an interactive {@link handleFor} can never produce one.
1041
+ */
1042
+ speechHandleFor(model: string | undefined): Promise<{
1043
+ app: string;
1044
+ handle: string;
1045
+ }>;
1046
+ /**
1047
+ * SESSION-IDENTITY SEAM (b): the exact pooled session a handle names.
1048
+ * NEVER opens a fresh session — a handle whose session is gone (closed,
1049
+ * evicted, re-homed to a replacement, proxy restarted) or malformed throws
1050
+ * {@link SessionGoneError} loudly. Resume is reattach-or-fail, not
1051
+ * reattach-or-quietly-restart.
1052
+ */
1053
+ sessionForHandle(handle: string): Promise<{
1054
+ app: string;
1055
+ entry: S;
1056
+ }>;
1057
+ /**
1058
+ * Drop ONE pooled session whose backhaul died (its pod was restarted /
1059
+ * drained / deleted) and release it — the next {@link sessionFor} opens a
1060
+ * fresh one, i.e. asks the control plane for a new assignment. Used by the
1061
+ * one-shot re-home (rehome.ts, urun-sh/urun-python#1592).
1062
+ *
1063
+ * IDENTITY-GUARDED (the same rule the pi lane's SessionPool follows): a
1064
+ * concurrent request that already re-homed this app has put a NEWER entry
1065
+ * under the key, and evicting that would close a healthy session out from
1066
+ * under it.
1067
+ */
1068
+ evict(app: string, entry: S): Promise<void>;
1069
+ /**
1070
+ * `GET /v1/models`: the org's deployed serve apps as model entries, the
1071
+ * default app FIRST. On the JWT lane (no org listing) this is the default
1072
+ * app plus any app already in the pool — the gap is called out loudly in
1073
+ * the PR, not papered over here.
1074
+ *
1075
+ * THE HUGGING FACE FIELDS: every entry carries `context_length` (from the
1076
+ * catalog's engine_args, omitted when unknown or when the placements
1077
+ * disagree) and `pricing` (USD per MILLION input/output tokens, omitted
1078
+ * ENTIRELY when the model has no per-token price row — see
1079
+ * model-prices.ts; ENG-317 owns the pricing decision that makes such rows
1080
+ * exist). HF reads both to build its public provider comparison table.
1081
+ * A price surface that cannot be READ lists everything unpriced; a price
1082
+ * surface that is a DEFECT fails this call loudly (see below).
1083
+ */
1084
+ modelList(): Promise<RouterModelList>;
1085
+ /**
1086
+ * The OpenRouter PROVIDER document (`GET /v1/models?format=openrouter`):
1087
+ * schema 2.4 per openrouter.ai/docs/guides/community/for-providers §1 —
1088
+ * "an endpoint that returns all models that should be served by
1089
+ * OpenRouter". That is the SHARED block's sellable endpoints (the models
1090
+ * callers resolve by `model_id:variant` on the chat surface), joined to
1091
+ * the catalog EXACTLY on (model_id, variant) — the SAME join
1092
+ * {@link supportsAudio} applies — and gated to chat-completions shape
1093
+ * (task chat | code | agent | vl; a voice or video app behind a chat
1094
+ * surface would stream garbage). Caller-org apps are deliberately NOT
1095
+ * listed: they are tenancy-private chat surfaces, and a provider
1096
+ * document that published them would sell another tenant's private
1097
+ * deployment as a public marketplace SKU. Pricing follows
1098
+ * {@link pricesForListing}'s posture (an unreadable surface prices
1099
+ * nothing and WARNS — discovery outlives the price oracle); a model with
1100
+ * no per-token row carries NO pricing array (the schema rule: "a
1101
+ * modality with no pricing array is simply unpriced"), and the block's
1102
+ * `output_free` endpoints are the ONLY rows whose zero OUTPUT rate
1103
+ * prices as `cost_usd: '0'` — the allowlist resolves prices IN the
1104
+ * pricing step below, so the document cannot emit a zero the catalog
1105
+ * did not opt into.
1106
+ *
1107
+ * A CONFIGURED catalog read that FAILS propagates — the provider
1108
+ * document cannot vouch a sellable model from nothing, and a half-empty
1109
+ * one must never be served as if complete.
1110
+ *
1111
+ * The builder's own loud announcements (the absent readiness seam that
1112
+ * conservatively declares every endpoint not ready, unmappable region
1113
+ * codes) go through this router's {@link warn} sink, so a lane that
1114
+ * routes degradations somewhere sees them too — never a default print.
1115
+ */
1116
+ openRouterModels(): Promise<OpenRouterProviderDoc>;
1117
+ /**
1118
+ * IMAGE CAPABILITY GATE (the Images lane's `imageModels` oracle, images.ts):
1119
+ * is `model`'s deployed app a native destination for `mode`? Authorized
1120
+ * resolution ({@link resolveApp} — undeployed uRun models throw
1121
+ * {@link UnknownModelError}, which the lane renders as a 404) then the
1122
+ * SAME catalog join as the /v1/models enrichment (PR421's rowForSlug):
1123
+ * exact `<model_id>-<variant>` slug across ALL its GPU-placement rows, a
1124
+ * bare model_id only when one variant owns it.
1125
+ *
1126
+ * Capability is read ONLY from catalog modality metadata, never from the
1127
+ * serve function name: every row must carry task 'image' INVARIANTLY across
1128
+ * the placements, and the mode comes from engine_args.expects_image —
1129
+ * false ⇒ pure text-to-image ('generation'; diffusers v0.38.0
1130
+ * pipeline_qwenimage_edit_plus raises torch.cat on an EMPTY image list, so
1131
+ * an edit destination is never generation-capable); true or ABSENT (the
1132
+ * native default) ⇒ 'edit'. The consensus rule is per-placement:
1133
+ * capability must agree across every placement — a CONFLICTING flag (mixed
1134
+ * true/false/absent) or a non-boolean one is a refused capability, never
1135
+ * silently defaulted to edit.
1136
+ *
1137
+ * A CONFIGURED catalog oracle that FAILS propagates its rejection — the
1138
+ * gate never fabricates a `false` (which would render as a misleading 404)
1139
+ * out of an infrastructure outage. On the CALLER-ORG lane, no oracle
1140
+ * configured, no matching row, or a slug the catalog cannot vouch for ⇒
1141
+ * false (loud 404 upstream). On the SHARED lane the same questions are
1142
+ * LOUDER: a shared endpoint is published capability, so an unresolvable
1143
+ * row or an unvouchable mode throws {@link SharedCatalogError} (the 500
1144
+ * path) instead of a 404 that would misattribute a catalog defect to the
1145
+ * caller's request.
1146
+ */
1147
+ supportsImages(model: string | undefined, mode: ImageCapabilityMode): Promise<boolean>;
1148
+ /**
1149
+ * AUDIO MODALITY GATE (the OpenAI Realtime lane's modality oracle): does
1150
+ * `model`'s deployed app carry catalog `task` stt or tts — INVARIANTLY
1151
+ * across every GPU-placement row ({@link audioModalityForRows})? The SAME
1152
+ * shape as {@link supportsImages}: authorized resolution
1153
+ * ({@link resolveTarget} — undeployed uRun models throw
1154
+ * {@link UnknownModelError}), then the SAME catalog join
1155
+ * ({@link catalogRowsForSlug}): exact `<model_id>-<variant>` slug across
1156
+ * ALL its placements, a bare model_id only when one variant owns it.
1157
+ *
1158
+ * A SHARED-lane hit joins on the lane's catalog id + variant (the SAME
1159
+ * join {@link imageModelList} uses for the shared block) — the shared org's
1160
+ * app slug need not exist in the caller's own catalog by slug.
1161
+ *
1162
+ * A CONFIGURED catalog oracle that FAILS propagates its rejection — the
1163
+ * gate never fabricates a `false` out of an infrastructure outage. No
1164
+ * oracle configured, no matching row, or a task the catalog cannot vouch
1165
+ * for ⇒ false — the realtime resolver renders that as the loud 404.
1166
+ */
1167
+ supportsAudio(model: string | undefined): Promise<boolean>;
1168
+ /**
1169
+ * THE SPEECH LANE'S GATE (proxy/speech.ts, `POST /v1/audio/speech`): does
1170
+ * `model`'s deployed app carry catalog `task` **tts** — INVARIANTLY across
1171
+ * every GPU-placement row ({@link audioModalityForRows} answering exactly
1172
+ * `'tts'`)? The SAME shape as {@link supportsImages}/{@link supportsAudio}:
1173
+ * authorized resolution ({@link resolveTarget} — undeployed uRun models
1174
+ * throw {@link UnknownModelError}), then the SAME catalog join
1175
+ * ({@link catalogRowsForSlug}); a SHARED-lane hit joins on the lane's exact
1176
+ * catalog id + variant, exactly like {@link supportsAudio}.
1177
+ *
1178
+ * An `stt` verdict is deliberately NOT speech-capable: a transcription
1179
+ * model must never be asked to synthesize, and a chat model on this route
1180
+ * is the loud 404 the images lane renders for non-image models. A
1181
+ * CONFIGURED catalog oracle that FAILS propagates its rejection — the gate
1182
+ * never fabricates a `false` out of an infrastructure outage.
1183
+ */
1184
+ supportsSpeech(model: string | undefined): Promise<boolean>;
1185
+ /**
1186
+ * THE SPEECH LANE'S VOICE-SET RESOLUTION (proxy/speech.ts): the voice set
1187
+ * `/v1/models` publishes for `model` — resolved by the SAME resolver the
1188
+ * listing's enrichment applies (openrouter-doc.ts `speechVoicesFor`, via
1189
+ * {@link speechVoicesForMatches}) over the SAME catalog join
1190
+ * {@link supportsSpeech} uses (exact `(model_id, variant)` for a
1191
+ * shared-lane hit, {@link catalogRowsForSlug} for a caller-org app), so
1192
+ * the lane enforces exactly ONE list: the one the listing publishes — the
1193
+ * catalog-vouched `engine_args.voices`, else the qwen3 CustomVoice card's
1194
+ * nine for that one model class. A catalog that PRESENTS a voices value
1195
+ * it cannot vouch answers WITH the defect set on the resolution — the
1196
+ * lane fails loud (500) rather than silently validating against the card
1197
+ * fallback. Undeployed uRun-namespace models throw
1198
+ * {@link UnknownModelError} (the lane's 404), and a CONFIGURED catalog
1199
+ * oracle that FAILS propagates — never a fabricated empty set.
1200
+ */
1201
+ speechVoices(model: string | undefined): Promise<SpeechVoicesResolution>;
1202
+ /**
1203
+ * `GET /v1/images/models`: the IMAGE-CAPABLE slice of the model surface —
1204
+ * the caller-org deployed apps whose catalog rows vouch an image mode
1205
+ * ({@link imageModesForRows}, the same consensus {@link supportsImages}
1206
+ * applies), plus the shared block's endpoints joined against each
1207
+ * endpoint's OWN catalog row (exact model_id + variant, EXPLICIT boolean
1208
+ * expects_image — {@link sharedImageModes}). A configured catalog oracle
1209
+ * that FAILS propagates, and so does a shared endpoint the catalog cannot
1210
+ * vouch (unresolvable row / unvouchable mode → {@link SharedCatalogError}):
1211
+ * an image listing cannot vouch capability from nothing or advertise a
1212
+ * silently defaulted mode, so — unlike {@link modelList} — there is NO
1213
+ * bare-entries degradation here.
1214
+ */
1215
+ imageModelList(): Promise<RouterImageModelList>;
1216
+ /** Close every pooled session (Session.end() underneath) — proxy shutdown. */
1217
+ closeAll(): Promise<void>;
1218
+ }
1219
+
1220
+ /** One ready replica row from the presence read. */
1221
+ interface RuntimePresenceRow {
1222
+ runtime_id: string;
1223
+ /** Base URL of the runtime's :8099 endpoint, e.g. `http://10.0.1.5:8099`. */
1224
+ address: string;
1225
+ ready: boolean;
1226
+ /** ISO 8601 instant of the runtime's last presence heartbeat. */
1227
+ last_seen: string;
1228
+ status?: string | null;
1229
+ }
1230
+ /** One resolved dial target — the router's pure resolution (no session).
1231
+ * The shared arm's fields ride through so the dial can mint the
1232
+ * provider-aware token and the handlers can repin the CATALOG billing pair:
1233
+ * `sharedOrgId` is the shared block's `shared_org_id` (the pod org). */
1234
+ interface DialTarget {
1235
+ appSlug: string;
1236
+ fnName?: string;
1237
+ sharedApp?: boolean;
1238
+ modelId?: string;
1239
+ variant?: string;
1240
+ sharedOrgId?: string;
1241
+ }
1242
+ interface StatelessDecisionDialOptions {
1243
+ /** Resolve a request's model to its dial target WITHOUT opening a session. */
1244
+ targetFor: (model: string | undefined) => Promise<DialTarget>;
1245
+ apiUrl: string;
1246
+ apiKey: string;
1247
+ /** The caller org's id — the minted token's `tenant` claim. */
1248
+ orgId: string;
1249
+ /** The serve function name the resolved app must expose (the backhaul's
1250
+ * URUN_FUNCTION default, 'serve'). */
1251
+ fnName: string;
1252
+ /** The shared secret render injects (URUN_SESSION_TOKEN_SECRET). */
1253
+ secret: string;
1254
+ /** THE SHARED LANE'S credential (URUN_SHARED_ORG_API_KEY): the shared
1255
+ * org's API key, so discovery and decisions can reach the SHARED org's
1256
+ * pods too. The provider org id comes from the resolved target's
1257
+ * `sharedOrgId` (the block's `shared_org_id`), never a second env spelling
1258
+ * of it — key and pod-org identity cannot drift apart. Absent ⇒ shared
1259
+ * models keep the session lane (the v1 boundary narrows to "no shared
1260
+ * credential mounted" instead of "shared never"). */
1261
+ shared?: {
1262
+ apiKey: string;
1263
+ };
1264
+ fetchImpl?: typeof fetch;
1265
+ nowMs?: () => number;
1266
+ /** Per-attempt POST budget override (test seam; production rides
1267
+ * DECISION_TIMEOUT_MS). */
1268
+ timeoutMs?: number;
1269
+ }
1270
+ /**
1271
+ * The stateless decision dial. `decide` resolves the target, discovers ready
1272
+ * replicas, picks the least-outstanding one, mints the scoped token and POSTs
1273
+ * the flat body to the runtime's `/v1/decision` intake.
1274
+ */
1275
+ declare class StatelessDecisionDial {
1276
+ private readonly opts;
1277
+ private readonly replicas;
1278
+ private readonly nowMs;
1279
+ /** Per-(presence key, app, fn) fresh listing, TTL-bounded to
1280
+ * PRESENCE_FRESH_S. A per-DECISION control-plane read would be the same
1281
+ * design error the session dial was — a network round-trip for state the
1282
+ * runtime plane re-proves every ~5s heartbeat. The cache serves the
1283
+ * caller from memory, refreshes in the BACKGROUND past half the TTL, and
1284
+ * evicts a replica (or the whole entry) the moment a dial to its cached
1285
+ * address fails. */
1286
+ private readonly presenceCache;
1287
+ constructor(opts: StatelessDecisionDialOptions);
1288
+ /** Enabled = the pod secret is configured. Absent ⇒ the session lane serves
1289
+ * (byte-for-byte behavior). */
1290
+ get enabled(): boolean;
1291
+ /**
1292
+ * Resolve a request's lane target WITHOUT dialing: `null` for a SHARED-lane
1293
+ * model when the dial holds NO shared credential — the caller keeps the
1294
+ * session lane. WITH the shared key, a shared target is dialable and rides
1295
+ * through with its full identity (modelId/variant for the billing repin,
1296
+ * `sharedOrgId` for the provider-aware token). The HTTP handlers call this
1297
+ * FIRST and route non-null targets into decide; decide re-resolves its own
1298
+ * target so the one it reports is the one it actually dialed.
1299
+ */
1300
+ targetFor(model: string | undefined): Promise<DialTarget | null>;
1301
+ /**
1302
+ * One decision. Returns the intake's `(status, body)` plus the SYNTHESIZED
1303
+ * completed body the lane stamps its billing receipt from — the runtime's
1304
+ * chat body carries `usage` (OpenAI shape) and `urun_timing`; both are the
1305
+ * exact fields the ledger's receipt stamper reads, mapped once here so the
1306
+ * stateless lane prices identically to the session lane. The RESOLVED
1307
+ * target rides along: the handlers repin the billing identity from it
1308
+ * (the dial's own answer, handed out — never asked for again) before
1309
+ * stamping.
1310
+ */
1311
+ decide(body: Record<string, unknown>, model: string | undefined): Promise<{
1312
+ status: number;
1313
+ body: unknown;
1314
+ completed: unknown;
1315
+ target: DialTarget & {
1316
+ fnName: string;
1317
+ };
1318
+ }>;
1319
+ /** Serve the caller from the cache; a MISS (or a TTL-expired entry) reads
1320
+ * synchronously — the cold path is the one decision that pays the hop —
1321
+ * and an entry past half the TTL fires a single-flight BACKGROUND refresh
1322
+ * that never blocks this caller. */
1323
+ private ensureRows;
1324
+ private readPresence;
1325
+ /** LEAST-OUTSTANDING: pick the ready replica with the fewest in-flight
1326
+ * decisions (ties → the first), skipping replicas in failure cooldown. */
1327
+ pick(rows: RuntimePresenceRow[]): string | null;
1328
+ private post;
1329
+ }
1330
+
1331
+ /**
1332
+ * THE CANONICAL USAGE-QUERY LANE — `POST /v1/usage/requests`.
1333
+ *
1334
+ * WHAT IT IS. A batch lookup over uRun's per-request inference ledger
1335
+ * (`public.inference_requests`, urun-infra ENG-316): "given these inference
1336
+ * ids, what do we hold for each?" — the cost and the usage quantities, scoped
1337
+ * to the calling key's org. OpenAI defines no shape for this, so the shape is
1338
+ * uRun's own: the ledger's own column vocabulary, and the same id space as the
1339
+ * `Inference-Id` response header every `/v1` answer already carries
1340
+ * (inference-id.ts).
1341
+ *
1342
+ * WHY IT IS CANONICAL AND NOT HUGGING-FACE-SHAPED. The owner has ruled that
1343
+ * distributor-specific functionality must not land on the canonical surface.
1344
+ * HF's billing poll — `{"requestIds":[...]}` answered with
1345
+ * `{"requests":[{"requestId","costNanoUsd"}]}` — is HF-proprietary. It gets a
1346
+ * THIN TRANSLATOR over this lane (partners/huggingface.ts, the ENG-376 adapter
1347
+ * pattern): field renames and nothing else. Every other distributor's billing
1348
+ * adapter reads this same lane, and so can the console and support.
1349
+ *
1350
+ * WHERE THE DATA COMES FROM. This process holds no database credential — by
1351
+ * design (hosted/auth.ts: "NO STANDING CREDENTIAL"). The lookup rides the
1352
+ * CALLER'S OWN key to the control plane's `inference-usage` edge function,
1353
+ * which derives the org from that key server-side and runs the org-scoped
1354
+ * `urun_inference_usage_lookup` RPC. So cross-org isolation here is the same
1355
+ * structural property every other lane has: this process never holds a
1356
+ * credential that spans orgs, and the org is never a request field.
1357
+ *
1358
+ * THREE CONTRACTS THAT ARE NOT NEGOTIABLE, because each one is money:
1359
+ *
1360
+ * 1. `cost_nano_usd: null` MEANS NOT YET PRICED — NEVER FREE. Every row in
1361
+ * production carries NULL today: ENG-317's price machinery has landed but
1362
+ * seeds no price, and nothing stamps the ledger yet. This lane reports
1363
+ * the NULL verbatim. A caller that renders it as 0 is inventing a price
1364
+ * nobody set; a caller billing a third party from it must simply not
1365
+ * answer for that id yet.
1366
+ * 2. AN ID WE HOLD NOTHING FOR IS ABSENT FROM THE ANSWER, and an id
1367
+ * belonging to another org is indistinguishable from one that never
1368
+ * existed. That is deliberate: anything else makes this surface a
1369
+ * cross-org existence oracle. A row this lane could not VALIDATE is
1370
+ * absent for the same reason and is therefore indistinguishable from
1371
+ * those two — which is precisely why it is reported (ENG-416), so that
1372
+ * the one absence we caused ourselves is visible on our side.
1373
+ * 3. IDS THE CALLER DID NOT ASK ABOUT ARE NEVER RETURNED. The id list is
1374
+ * the query; the control plane bounds the answer by it, and
1375
+ * {@link assertRequestedOnly} re-checks it here rather than trusting the
1376
+ * upstream to have done so.
1377
+ *
1378
+ * WHAT THIS LANE DOES NOT DO: it does not filter on `http_status` (only
1379
+ * 2xx/3xx being billable is an HF rule, and support must be able to ask about
1380
+ * a failed request), and it never returns `gpu_seconds` — a column ENG-417
1381
+ * DROPPED from the ledger entirely, because a per-request GPU-seconds figure
1382
+ * double-counts against `usage_events.metered_gpu_seconds` (which meters the
1383
+ * instance once) and, under continuous batching, can never be reconciled to
1384
+ * it. The guard below is kept and is STRONGER for the removal: it now asserts
1385
+ * that a column which does not exist has not come back.
1386
+ *
1387
+ * WHAT IT DOES RETURN THAT IS EASY TO MISS: `identity_kind`. `model_id` speaks
1388
+ * one of two unrelated namespaces — a catalog identity, or the caller's own
1389
+ * deployed app slug — and the two can COLLIDE as strings, so the identity is
1390
+ * never reported without the namespace that makes it readable.
1391
+ *
1392
+ * AND IT RESOLVES NO PRICE — stated because the surrounding system is about to
1393
+ * grow several, and someone will come looking here. Nothing in this lane or in
1394
+ * its Hugging Face translator reads `model_prices`, picks a rate, or knows what
1395
+ * a model costs. It reports the `cost_nano_usd` a writer already stamped on the
1396
+ * row, at REQUEST grain, so two requests for the same model may carry different
1397
+ * costs and different `price_version`s and this code is indifferent to why.
1398
+ *
1399
+ * That means the one-price-per-model assumption is NOT baked in here. It lives
1400
+ * one layer down, in `urun_active_model_price`, which resolves on
1401
+ * `(model_id, variant)` and separates the session lane from the HF lane BY UNIT
1402
+ * — fine while unit and audience correlate, and filed as ENG-404 for when they
1403
+ * stop (per-minute credit tiers put two audiences on one unit). When that is
1404
+ * fixed, the audience rides `price_version`, which this lane already carries
1405
+ * through per row and never interprets. No change is required here.
1406
+ */
1407
+
1408
+ /**
1409
+ * ONE ledger row as this lane answers it — the control-plane RPC's column
1410
+ * list, in the ledger's own vocabulary. `gpu_seconds` and `org_id` are absent
1411
+ * by construction, not by omission here (see the module header).
1412
+ */
1413
+ interface InferenceUsageRecord {
1414
+ /** The uuid this request's `Inference-Id` response header carried. */
1415
+ inference_id: string;
1416
+ /** The identity that served it, IN THE NAMESPACE {@link identity_kind} names. */
1417
+ model_id: string;
1418
+ /**
1419
+ * WHICH NAMESPACE {@link model_id} SPEAKS: `catalog_model` (the
1420
+ * `model_prices` vocabulary — everything a distributor routes) or
1421
+ * `caller_org_app` (the caller's own deployed app, which has no catalog
1422
+ * identity and is therefore PERMANENTLY unpriceable — never merely "not
1423
+ * priced yet"). Reported as a plain string rather than a union, because a
1424
+ * value this lane does not recognise must reach the caller as data rather
1425
+ * than fail the batch: an unknown lane is a forward-compatible control plane,
1426
+ * not a corrupt row.
1427
+ */
1428
+ identity_kind: string;
1429
+ /** null = the model-level row (the `model_prices` key semantics). */
1430
+ variant: string | null;
1431
+ task: string;
1432
+ http_status: number;
1433
+ started_at: string;
1434
+ completed_at: string;
1435
+ /** null = this modality has no such quantity (not "zero of it"). */
1436
+ prompt_tokens: number | null;
1437
+ completion_tokens: number | null;
1438
+ output_units: number | null;
1439
+ /** NULL = NOT YET PRICED (ENG-317). Never "free". */
1440
+ cost_nano_usd: number | null;
1441
+ price_version: string | null;
1442
+ priced_at: string | null;
1443
+ }
1444
+ /**
1445
+ * ONE row this lane REFUSED TO ANSWER ABOUT, handed to the reporter so that a
1446
+ * dropped row is a visible event rather than a silent absence.
1447
+ *
1448
+ * It carries no row content — `where` is a position and `reason` is an
1449
+ * already-redacted {@link UsageSurfaceError} message (names and types, never
1450
+ * values). See {@link describeKeys}.
1451
+ */
1452
+ interface UsageRowRejection {
1453
+ /** `usage.requests[N]` — WHICH row, never the row. */
1454
+ where: string;
1455
+ /** Why it could not be validated. */
1456
+ reason: string;
1457
+ }
1458
+ /**
1459
+ * A sink for {@link UsageRowRejection}s.
1460
+ *
1461
+ * `Promise<void>` IS PART OF THE TYPE BECAUSE IT WAS PART OF THE TYPE ANYWAY.
1462
+ * An `async` function is assignable to a `=> void` signature, so declaring this
1463
+ * synchronous never prevented an async sink — it only hid one, by putting its
1464
+ * failure on a later tick where the containment could not see it. Saying so in
1465
+ * the type is what lets {@link reportRejectedRow} actually handle it.
1466
+ */
1467
+ type UsageRowReporter = (rejection: UsageRowRejection) => void | Promise<void>;
1468
+ /**
1469
+ * The batch-lookup seam. The hosted endpoint supplies one backed by the
1470
+ * control plane's `inference-usage` function; an embedder that supplies none
1471
+ * gets a LOUD 501 on the lane rather than a silently empty answer — the same
1472
+ * posture the image lanes take for `supportsImages`.
1473
+ *
1474
+ * IT IS TWO-PHASE, and the split is the admission boundary rather than
1475
+ * decoration: {@link authorize} runs BEFORE the request body is read, so a
1476
+ * request with no credential is refused without this process allocating or
1477
+ * parsing anything the caller sent.
1478
+ */
1479
+ interface UsageLane {
1480
+ /**
1481
+ * Admission. Returns the caller's credential, or throws (a 401 through the
1482
+ * handler's `statusOf`) when it is missing or malformed. Runs before the
1483
+ * body is touched, and must NOT make a network call — the surface that owns
1484
+ * the ledger authenticates the credential itself when {@link lookup} uses
1485
+ * it, and a second verifier here would be a second source of truth about
1486
+ * the same key.
1487
+ */
1488
+ authorize(req: IncomingMessage): string;
1489
+ /** The batch read, as the holder of the credential `authorize` returned. */
1490
+ lookup(credential: string, inferenceIds: readonly string[]): Promise<InferenceUsageRecord[]>;
1491
+ }
1492
+
1493
+ /**
1494
+ * THE PER-REQUEST LEDGER WRITE — one served `/v1` inference becomes one
1495
+ * durable, already-priced row in `public.inference_requests`.
1496
+ *
1497
+ * WHY THIS EXISTS. uRun answers Hugging Face's billing poll out of a ledger
1498
+ * that, until this module, NOTHING HAD EVER WRITTEN. ENG-316 landed the table
1499
+ * with no writer, ENG-317 landed the price machinery with no stamping, and
1500
+ * ENG-318 landed the reader over both and stated the consequence: with every
1501
+ * cost NULL, a poll answers `{"requests": null}` for every batch and every
1502
+ * request is written off unbilled ~30 minutes later. This module is the hop
1503
+ * from the correlation record the proxy already emits to a row.
1504
+ *
1505
+ * THE THREE RULES THIS MODULE EXISTS TO KEEP (owner ruling, ENG-409):
1506
+ *
1507
+ * 1. THE WRITE IS OFF THE RESPONSE PATH, and the price is stamped in that
1508
+ * same write. It runs from the response's `'close'` hook — after the
1509
+ * customer has their answer — so a ledger round trip can never be in the
1510
+ * latency of an inference. It is not a later sweep either: a re-price
1511
+ * landing after HF has been answered disagrees with a bill already
1512
+ * issued (ENG-406), so the row is born priced or born unpriced.
1513
+ *
1514
+ * 2. A LEDGER WRITE MUST NEVER FAIL AN INFERENCE REQUEST. Everything here
1515
+ * is fire-and-forget and cannot throw into the request path: the answer
1516
+ * is already delivered, the socket is already closing, and a failure is
1517
+ * LOUD IN THE LOG and nowhere else. {@link recordInference} returns
1518
+ * `void` for exactly that reason — there is no promise a caller could
1519
+ * accidentally await, and nothing to reject. The queue and the retries
1520
+ * added by ENG-415 change nothing about this: they all happen behind that
1521
+ * same `void`, after the customer has been answered.
1522
+ *
1523
+ * 3. A REQUEST THAT CANNOT BE PRICED IS WRITTEN UNPRICED, NEVER AT ZERO AND
1524
+ * NEVER DROPPED. Pricing itself happens in the database
1525
+ * (`urun_record_inference_request`), which is where `model_prices` is
1526
+ * readable and where a `price_version` can name the rate rows that
1527
+ * produced a number. This module supplies the ONE fact the database
1528
+ * cannot see — {@link billableOutcome} — and nothing else about money.
1529
+ *
1530
+ * WHAT IS DELIBERATELY NOT HERE:
1531
+ * * NO PRICES, NO RATES, NO MODEL NAMES, NO ARITHMETIC. Grep it: there is
1532
+ * no rate and no model id below. The price book is data (ENG-317) and a
1533
+ * rate change must take effect without a deploy.
1534
+ * * NO PARTNER VOCABULARY. Hugging Face's wire shape lives in
1535
+ * `proxy/partners/huggingface.ts` and nowhere else. What crosses this
1536
+ * module is uRun's own ledger vocabulary, which every distributor's
1537
+ * adapter and the console read alike.
1538
+ * * NO `gpu_seconds`, AND NO LEDGER COLUMN FOR ONE (ENG-417). Exclusive
1539
+ * per-request GPU occupancy is not obtainable under continuous batching,
1540
+ * and wall-clock there is SHARED occupancy — a per-request figure would
1541
+ * double-count against `usage_events.metered_gpu_seconds`, which meters
1542
+ * the instance once and is reduced by the hourly `usage_accrual` sweep so
1543
+ * that ledger sums stay exactly the metered wall-clock cost. What this
1544
+ * module sends instead is a LINK (`session_id`) and a WEIGHT (the
1545
+ * residency milliseconds), from which the share is DERIVED at
1546
+ * reconciliation so the shares sum to the instance total by construction.
1547
+ * The residency is NOT GPU time: under continuous batching the sum of
1548
+ * concurrent `decode_ms` EXCEEDS the session's wall-clock, which is
1549
+ * precisely why it is only ever meaningful as a normalized fraction.
1550
+ * * NO DURABLE BUFFER. Rows are queued in memory and nowhere else. The
1551
+ * inference-proxy Deployment declares NO VOLUMES — not a PVC, not even an
1552
+ * emptyDir — so there is nowhere on this pod that survives a SIGKILL, and
1553
+ * making one would mean a StatefulSet with a per-replica volume on an
1554
+ * internet-facing HA front door where a scaled-down replica's volume holds
1555
+ * unflushed rows forever. What survives instead is the `inference_request`
1556
+ * line this proxy writes to stdout for EVERY `/v1` request, which
1557
+ * fluent-bit ships off the pod into Loki; that is the reconciliation
1558
+ * record, and it is a way to KNOW what was lost rather than a way to bill
1559
+ * it. See {@link drainLedgerWrites}.
1560
+ *
1561
+ * RETRY, AND THE RULE IT LIVES UNDER (ENG-415). This module used to do NO
1562
+ * retry, for a good reason: a retry after a timeout cannot know whether the
1563
+ * first attempt landed, and the row it would re-send came back as an
1564
+ * undifferentiated duplicate — so retrying turned an ambiguous outcome into a
1565
+ * loud error that looked exactly like a forged row. That reason is now gone:
1566
+ * the control plane's 409 carries the row already stored, so a retry whose
1567
+ * first attempt landed resolves as `already_written` instead. Retry is
1568
+ * therefore allowed for EXACTLY ONE failure reason and no other — see
1569
+ * {@link isRetryable}, which is where the rule lives rather than in a
1570
+ * judgement at the call site.
1571
+ */
1572
+
1573
+ /**
1574
+ * ONE ledger row, in the ledger's own vocabulary — exactly the body the
1575
+ * control plane's `inference-ledger` function takes.
1576
+ *
1577
+ * `org_id` and `api_key_id` are ABSENT BY CONSTRUCTION rather than omitted
1578
+ * here: the control plane derives both from the credential the write rides,
1579
+ * so there is no field on this surface that could name another org.
1580
+ */
1581
+ interface LedgerWrite {
1582
+ /** The uuid this request's `Inference-Id` response header carried. */
1583
+ inference_id: string;
1584
+ /**
1585
+ * The identity that served it — IN THE NAMESPACE {@link identity_kind}
1586
+ * NAMES. Never the caller's raw `model` string.
1587
+ */
1588
+ model_id: string;
1589
+ /**
1590
+ * WHICH NAMESPACE {@link model_id} SPEAKS (ENG-412): a catalog identity
1591
+ * (`public.model_prices` vocabulary), or the caller's own deployed app slug.
1592
+ * The two are unrelated namespaces that can nonetheless COLLIDE as strings,
1593
+ * so the row states which one it is rather than leaving it to be inferred —
1594
+ * and the control plane refuses to price anything but a catalog row, in the
1595
+ * writer's predicate and again in a table CHECK.
1596
+ */
1597
+ identity_kind: 'catalog_model' | 'caller_org_app';
1598
+ variant: string | null;
1599
+ task: string;
1600
+ http_status: number;
1601
+ started_at: string;
1602
+ completed_at: string;
1603
+ /** See {@link billableOutcome}. The one fact the database cannot see. */
1604
+ billable_outcome: boolean;
1605
+ /**
1606
+ * THE INSTANCE that served it (ENG-417) — the link the control plane
1607
+ * verifies against the org before writing. NULL when this request's dial
1608
+ * exposed no session identity; never a fabricated id.
1609
+ */
1610
+ session_id: string | null;
1611
+ /** NULL = this request has no such quantity. NEVER zero of it. */
1612
+ prompt_tokens: number | null;
1613
+ completion_tokens: number | null;
1614
+ output_units: number | null;
1615
+ /**
1616
+ * THE WEIGHT (ENG-417) — the serve runtime's own residency in whole
1617
+ * milliseconds. An allocation input, never GPU time and never a cost. NULL
1618
+ * means the runtime reported nothing, which is not zero: a weightless
1619
+ * request drops out of its instance's allocation and inflates every other
1620
+ * request's share.
1621
+ */
1622
+ queue_ms: number | null;
1623
+ prefill_ms: number | null;
1624
+ decode_ms: number | null;
1625
+ total_ms: number | null;
1626
+ /**
1627
+ * PROXY-OBSERVED TIME TO FIRST STREAMED CONTENT BYTE, in whole
1628
+ * milliseconds — what the CALLER experienced, and therefore the number
1629
+ * Hugging Face's 5s provider-listing gate is about (ENG-337).
1630
+ *
1631
+ * THE RECORD IS THE ONE SOURCE. This is {@link InferenceRecord.ttft_ms}
1632
+ * carried through, measured at the one place TTFT was already measured
1633
+ * (`ProxyStats.track` stamps `turn.ttftMs` on the first text delta) — the
1634
+ * same number the billing log line and the `urun_inference_ttft_seconds`
1635
+ * histogram already publish. The writer adds no second measurement, so the
1636
+ * row cannot disagree with either sink.
1637
+ *
1638
+ * NULL FOR NON-STREAM, NULL FOR UNOBSERVED, NEVER ZERO FOR EITHER. The
1639
+ * record stamps `ttft_ms` only for a streamed turn: a caller who receives
1640
+ * one JSON body at the end was never shown a "first byte", and publishing
1641
+ * the internal timing in its place would put flattering numbers into
1642
+ * exactly the series the HF gate reads. A stream that produced no text
1643
+ * delta before it ended was never observed at all. Null is the only honest
1644
+ * spelling of both; a 0 here would assert a sub-millisecond first byte
1645
+ * nobody measured. (A MEASURED sub-millisecond first byte does round to 0 —
1646
+ * that is a measurement, not an unknown; which side of the `??` a value
1647
+ * arrived on is the whole difference.)
1648
+ *
1649
+ * NOT THE RUNTIME'S `urun_timing.ttft_ms`, which is `queue_ms +
1650
+ * prefill_ms` and still deliberately does not ride the row (see
1651
+ * `TurnResidency`): that is the engine's view of its own internals, this
1652
+ * is the wall-clock the caller saw — network, proxy, cold-wake and all.
1653
+ * Neither can derive the other, so neither is a second spelling of the
1654
+ * other.
1655
+ */
1656
+ ttft_ms: number | null;
1657
+ /**
1658
+ * PROXY-OBSERVED REQUEST RECEIPT → LAST BYTE, in whole milliseconds: the
1659
+ * end-to-end latency the caller experienced for the whole answer
1660
+ * (ENG-337).
1661
+ *
1662
+ * `Math.round` of the record's `duration_ms` — THE SAME MEASUREMENT
1663
+ * {@link startedAtOf} derives the row's `started_at` from, so the row's
1664
+ * window columns cannot disagree by construction: `started_at + e2e_ms`
1665
+ * and `completed_at` describe one clock. Rounded HERE, at the one place
1666
+ * the row is built, because the ledger's column
1667
+ * (`urun_record_inference_request`'s `p_e2e_ms`) is an `integer` — never
1668
+ * rounded a second time somewhere a new spelling could drift from this
1669
+ * one.
1670
+ *
1671
+ * ALWAYS PRESENT on rows this writer produces, unlike {@link ttft_ms}: the
1672
+ * duration is measured for every completed request, streamed or not,
1673
+ * failed or not. The TYPE still admits null because the wire column is
1674
+ * nullable — the contract is stated rather than implied by what this
1675
+ * writer happens to emit today.
1676
+ */
1677
+ e2e_ms: number | null;
1678
+ }
1679
+ /** What the control plane answers with — the row as it was actually written. */
1680
+ interface LedgerRecorded {
1681
+ inference_id: string;
1682
+ /** NULL = NOT YET PRICED (ENG-317). Never "free". */
1683
+ cost_nano_usd: number | null;
1684
+ /** `in=<model_prices.id>;out=<model_prices.id>` when priced. */
1685
+ price_version: string | null;
1686
+ /**
1687
+ * Why the row carries no price, or null when it is priced. One of
1688
+ * `not_a_catalog_model` | `outcome_not_billable` | `http_status_not_success`
1689
+ * | `task_unit_undeclared` | `task_not_token_billed` | `token_counts_absent`
1690
+ * | `no_active_price` | `no_active_input_price` | `no_active_output_price`.
1691
+ * It is a NORMAL answer, not an error: no PER-TOKEN rate has been seeded
1692
+ * yet, so this lane has nothing to price with. (Not "the price book is
1693
+ * empty" — `model_prices` has carried live `per_minute` rows since
1694
+ * 2026-09-10; what is absent is `per_million_input_tokens` /
1695
+ * `per_million_output_tokens`.)
1696
+ *
1697
+ * Only `not_a_catalog_model` is PERMANENT. A caller-org app has no catalog
1698
+ * identity, so no price book will ever cover it; every other reason means
1699
+ * "not priced YET".
1700
+ *
1701
+ * THE TWO `task_*` REASONS ARE ENG-427, and they are the reason
1702
+ * {@link LedgerWrite.task} is worth carrying honestly. The control plane prices a row IN THE UNIT ITS
1703
+ * TASK DECLARES (`public.task_billing_units`), never in the unit the data
1704
+ * happens to look like:
1705
+ *
1706
+ * * `task_not_token_billed` — this lane's billable quantity is not tokens
1707
+ * (an image lane's is {@link LedgerWrite.output_units}), so its token counts are not
1708
+ * a price for it. A genuine `{0, 0}` receipt on such a lane would
1709
+ * otherwise settle at `cost_nano_usd = 0` PERMANENTLY — the control
1710
+ * plane's append-only rule allows a deliberate re-price and never a
1711
+ * return to NULL — which is an invoice asserting the work was free when
1712
+ * its billable quantity was never measured.
1713
+ * * `task_unit_undeclared` — nobody has declared what this task bills in,
1714
+ * so nothing is priced. Adding a lane to {@link ledgerTaskOf} without a
1715
+ * matching declaration fails CLOSED and says which fix it wants.
1716
+ */
1717
+ unpriced_reason: string | null;
1718
+ }
1719
+ /**
1720
+ * WHY A LEDGER WRITE DID NOT LAND — a FIXED, SMALL vocabulary, because this is
1721
+ * what an alert matches on.
1722
+ *
1723
+ * ENG-409 made the fabricated-row hazard (ENG-410) detectable and stopped
1724
+ * there: a duplicate `inference_id` raises 23505, surfaces as a 409, and
1725
+ * arrives here as a rejected write. But it arrived carrying only an English
1726
+ * SENTENCE, which meant the only way to alert on "somebody wrote this row
1727
+ * ahead of us" was a regex over prose. An alert that is a substring match on a
1728
+ * message nobody promised to keep stable is an alert that dies silently the
1729
+ * first time the wording is improved — and this particular alert is the ONLY
1730
+ * signal that a customer is under-billing itself.
1731
+ *
1732
+ * So the reason is a FIELD, drawn from this closed set, and it is also the
1733
+ * metric label ({@link LedgerOutcome}):
1734
+ *
1735
+ * * `duplicate` — the id was already in the ledger and the stored row is
1736
+ * PROVABLY NOT THE ONE WE WROTE. Since the 409 carries that row (ENG-415),
1737
+ * a matching one is our own landed attempt and is not a failure at all
1738
+ * ({@link LedgerAlreadyWritten}); what reaches here is somebody else's row
1739
+ * — or a row we could not read back, which fails to this loud side on
1740
+ * purpose. This is the ENG-410 signal.
1741
+ * * `duplicate_foreign` — the id is taken by a row THIS ORG CANNOT READ.
1742
+ * `inference_id` is a global primary key, so it is reachable, and it can
1743
+ * never be our own write. The loudest state in the lane.
1744
+ * * `auth` — the control plane refused the caller's org key.
1745
+ * * `surface` — the control plane could not be reached, answered a status
1746
+ * we do not accept, or answered a body we could not validate. THE ONLY
1747
+ * RETRYABLE ONE (see {@link isRetryable}).
1748
+ * * `unknown` — the lane rejected with something that carries NO reason at
1749
+ * all. It is its OWN value rather than being folded into `surface`
1750
+ * precisely so it cannot hide: a reason-less rejection is a defect in the
1751
+ * lane, and counting it as a surface error would file a bug as weather —
1752
+ * and would retry it, which is how a defect becomes a loop.
1753
+ * * `dwell_expired` — the row waited longer than {@link LEDGER_MAX_DWELL_MS}
1754
+ * and was given up on. Revenue lost, named as revenue lost.
1755
+ * * `dropped_on_exit` — the row was still queued when the shutdown drain ran
1756
+ * out. One per row, per pod, per rolling deploy that went badly.
1757
+ */
1758
+ declare const LEDGER_FAILURE_REASONS: readonly ["duplicate", "duplicate_foreign", "auth", "surface", "unknown", "dwell_expired", "dropped_on_exit"];
1759
+ type LedgerFailureReason = (typeof LEDGER_FAILURE_REASONS)[number];
1760
+ /**
1761
+ * Implemented by every error a {@link LedgerLane.write} may reject with, so
1762
+ * the reason travels ON the error instead of being re-derived from its text.
1763
+ *
1764
+ * It is a FIELD rather than a class hierarchy on purpose: the auth rejection
1765
+ * must stay an `instanceof ProxyAuthError` for every existing caller, and a
1766
+ * class can only have one base.
1767
+ */
1768
+ interface LedgerFailure {
1769
+ readonly ledgerFailureReason: LedgerFailureReason;
1770
+ }
1771
+ /**
1772
+ * WHY A ROW WAS NOT EVEN ATTEMPTED. Also a closed vocabulary, and for the same
1773
+ * reason: a served request that produced no ledger row is revenue that can
1774
+ * never be recovered, so each of these is a counted event and not merely a log
1775
+ * line somebody might read.
1776
+ *
1777
+ * `not_a_catalog_model` WAS IN THIS LIST AND IS DELIBERATELY GONE (ENG-412). A
1778
+ * request served by a caller's own deployed app used to be skipped here; it
1779
+ * now WRITES a row in its own declared namespace (`identity_kind:
1780
+ * 'caller_org_app'`), so nothing can emit that skip any more. A member of a
1781
+ * closed, counted vocabulary that is structurally unreachable is a counter
1782
+ * that can only ever read zero — it invites the reader to conclude the case
1783
+ * never happens, when in truth the case stopped being a skip at all.
1784
+ *
1785
+ * DO NOT CONFUSE IT WITH THE UNPRICED REASON OF THE SAME NAME, which is alive
1786
+ * and is the whole point of ENG-412: the control plane still answers
1787
+ * `unpriced_reason: 'not_a_catalog_model'` for those rows (see
1788
+ * {@link LedgerRecorded}). The row exists and is counted; it simply can never
1789
+ * carry a price. Two vocabularies, one word, opposite states — a written row
1790
+ * versus no row at all.
1791
+ *
1792
+ * `no_model_named` replaces it for the one case on that lane that still cannot
1793
+ * be written: `model_id` is NOT NULL, a caller-org app's identity IS the
1794
+ * caller's model string, and a request that named no model leaves nothing
1795
+ * honest to record.
1796
+ */
1797
+ declare const LEDGER_SKIP_REASONS: readonly ["no_response_head", "no_org", "model_unresolved", "no_model_named", "queue_full"];
1798
+ type LedgerSkipReason = (typeof LEDGER_SKIP_REASONS)[number];
1799
+ /**
1800
+ * THE TWO PRICING STATES a landed row can be in — ENG-317's
1801
+ * NULL-is-not-zero distinction, as a metric label value.
1802
+ *
1803
+ * A CONST ARRAY rather than an inline union because the vocabulary is now
1804
+ * ENUMERATED, not merely constrained: {@link LEDGER_OUTCOMES} materialises
1805
+ * every series this counter can ever produce, and a pricing state that existed
1806
+ * only in a type could not be enumerated. Adding a third state here adds its
1807
+ * series everywhere without a second edit.
1808
+ */
1809
+ declare const LEDGER_PRICING_STATES: readonly ["priced", "unpriced"];
1810
+ type LedgerPricingState = (typeof LEDGER_PRICING_STATES)[number];
1811
+ /**
1812
+ * The TERMINAL fate of one request's ledger row, as a bounded pair of metric
1813
+ * labels. Every `/v1` request that reaches a lane produces exactly one of
1814
+ * these, so `sum(urun_inference_ledger_writes_total)` is the number of
1815
+ * billable requests the proxy has accounted for — and the `skipped` and
1816
+ * `failed` arms are the leak.
1817
+ *
1818
+ * EVERY VALUE IS OWNED BY THIS MODULE. `reason` is drawn from
1819
+ * {@link LEDGER_SKIP_REASONS}, {@link LEDGER_FAILURE_REASONS} or the two
1820
+ * pricing states below — never from the control plane's answer and never from
1821
+ * anything a caller can influence. That is what keeps the series set bounded
1822
+ * without a `BoundedLabel`: an `unpriced_reason` echoed verbatim into a label
1823
+ * would let a surface that changed shape blow up the TSDB.
1824
+ */
1825
+ type LedgerOutcome =
1826
+ /** The row is in the ledger. `priced` / `unpriced` is ENG-317's NULL-is-not-zero distinction. */
1827
+ {
1828
+ outcome: 'written';
1829
+ reason: LedgerPricingState;
1830
+ }
1831
+ /**
1832
+ * The row was ALREADY in the ledger and it is OURS — an earlier attempt
1833
+ * landed and we never learned it (ENG-415). A terminal SUCCESS: the request
1834
+ * is accounted for exactly once, and it is counted separately from `written`
1835
+ * only so "how often does the write path go ambiguous" is answerable.
1836
+ */
1837
+ | {
1838
+ outcome: 'already_written';
1839
+ reason: LedgerPricingState;
1840
+ }
1841
+ /** No row was attempted. */
1842
+ | {
1843
+ outcome: 'skipped';
1844
+ reason: LedgerSkipReason;
1845
+ }
1846
+ /** A row was attempted and did not land. */
1847
+ | {
1848
+ outcome: 'failed';
1849
+ reason: LedgerFailureReason;
1850
+ };
1851
+ /**
1852
+ * The ledger write seam. Only the HOSTED endpoint supplies one (the write
1853
+ * rides the caller's own org API key to the control plane; this process holds
1854
+ * no database credential). A proxy WITHOUT one does not participate in the
1855
+ * ledger at all — the `usage` / `supportsImages` posture: configured absence,
1856
+ * declared at the seam, not a degradation discovered at runtime. The local
1857
+ * CLI proxy is single-tenant and bills nothing, so it has none.
1858
+ */
1859
+ interface LedgerLane {
1860
+ /**
1861
+ * Write one row, as the holder of the credential on `req`. The request is
1862
+ * over by the time this is called; `req` is carried only for its
1863
+ * `Authorization` header, never re-read.
1864
+ */
1865
+ write(req: IncomingMessage, entry: LedgerWrite): Promise<LedgerRecorded>;
1866
+ /**
1867
+ * Count ONE terminal outcome. REQUIRED, unlike {@link report} — which has a
1868
+ * real default (this process's stdout) — because there is no default for
1869
+ * "count it nowhere" that is not a silent no-op. A lane that genuinely
1870
+ * counts nothing has to say so at the seam, in code a reviewer can see.
1871
+ *
1872
+ * This is the alerting surface for ENG-410: `outcome="failed"` with
1873
+ * `reason="duplicate"` is the fabricated-row signal, and the log line is
1874
+ * only the per-request detail behind it.
1875
+ */
1876
+ observe(outcome: LedgerOutcome): void;
1877
+ /**
1878
+ * Where this lane's own structured diagnostics go — one newline-terminated
1879
+ * JSON object per event. Absent, they go to this process's stdout (the
1880
+ * hosted endpoint's collected container log). Injectable so tests assert on
1881
+ * them without racing the process streams.
1882
+ *
1883
+ * These are DELIBERATELY NOT the `requestLog` sink: that sink carries
1884
+ * exactly one `inference_request` record per `/v1` request and readers count
1885
+ * on that.
1886
+ */
1887
+ report?(line: string): void;
1888
+ }
1889
+
1890
+ /**
1891
+ * The shared Responses execution/store seam — the ONE execution + event-
1892
+ * envelope path and the authorized tenant store, imported by BOTH the HTTP
1893
+ * SSE lane (server.ts) and the `/v1/responses` WebSocket lane
1894
+ * (responses-ws.ts). This module deliberately imports NO transport acceptor
1895
+ * and NO HTTP lane module: its only imports are type-only, so the module
1896
+ * graph server.ts → responses-ws.ts → responses-turn.ts is acyclic and no
1897
+ * Vitest/Bun/Node module-evaluation order can observe an uninitialized
1898
+ * binding (the constructor-failure class of bug this extraction removes).
1899
+ */
1900
+
1901
+ /** The request envelope every Responses-shaped upstream call carries. */
1902
+ interface ResponsesCreateParams {
1903
+ model?: string;
1904
+ input: unknown;
1905
+ stream?: boolean;
1906
+ tools?: unknown;
1907
+ tool_choice?: unknown;
1908
+ temperature?: number;
1909
+ max_output_tokens?: number;
1910
+ /**
1911
+ * System/developer instructions, forwarded to the serve envelope — where
1912
+ * they become ONE prepended `system`-role message (the protocol's native
1913
+ * per-request system input; there is NO envelope-level instructions field).
1914
+ * Typed `string` end to end (transport/encode.ts) so no cast bridges the seam.
1915
+ */
1916
+ instructions?: string;
1917
+ top_p?: number;
1918
+ stop?: string[];
1919
+ /**
1920
+ * Reasoning controls (urun-python #1667): forwarded VERBATIM to the serve
1921
+ * envelope — `chat_template_kwargs.enable_thinking:false` is the think-off
1922
+ * switch. Absent -> absent (no default injection; server-side validation).
1923
+ */
1924
+ reasoning_effort?: string;
1925
+ chat_template_kwargs?: Record<string, unknown>;
1926
+ /**
1927
+ * OpenAI `response_format` — structured output (ENG-310). On the Responses
1928
+ * lane the caller spells this `text.format`; `textFormatOf` lifts it onto
1929
+ * this ONE field so both lanes hand the serve envelope the same key.
1930
+ * Forwarded verbatim — the serve runtime is the one validator.
1931
+ */
1932
+ response_format?: unknown;
1933
+ /**
1934
+ * The SPEECH lane's per-request TTS voice (ENG-331, additive like `tools`):
1935
+ * the speaker id a `tts`-task engine synthesizes under (qwen3 CustomVoice
1936
+ * timbre, e.g. "Ryan"). Forwarded on the serve envelope so the runtime's
1937
+ * speak bridge can apply it through the engine's own per-session override
1938
+ * seam (`SpeakSessionConfig.voice` → `set_voice`, urun-python
1939
+ * tts_engine.py). Absent -> key absent: the engine's catalog-configured
1940
+ * default voice applies, today's behavior.
1941
+ */
1942
+ voice?: string;
1943
+ }
1944
+ /** How one pinned session ended (the native phase machinery's terminal step). */
1945
+ interface SessionEndInfo {
1946
+ /**
1947
+ * Milliseconds until the session's native deadline (`endsAt`), or null when
1948
+ * the app declared no maximum session length. NOTE (missing primitive,
1949
+ * called out in the PR): core exposes no PRE-expiry notice event — this
1950
+ * callback fires AT terminal loss, so timeLeftMs is ~0 on expiry.
1951
+ */
1952
+ timeLeftMs: number | null;
1953
+ /** The terminal reason (expired / ended / error), for the loud close. */
1954
+ reason: string;
1955
+ }
1956
+ /** The upstream calls the proxy makes — injectable (tests; alt transports). */
1957
+ interface ProxyClients {
1958
+ /**
1959
+ * The NON-SECRET tenancy label of the org this backhaul belongs to — the
1960
+ * control plane's org id, set by the hosted registry from the verified
1961
+ * `CallerIdentity` (`hosted/tenants.ts`). The request handler copies it onto
1962
+ * the request's `InferenceTurn`, which is what puts an `org` label on the
1963
+ * Prometheus series and an `org` field on the billing record.
1964
+ *
1965
+ * THE ORG, NEVER THE KEY: the Bearer key is the credential on this surface,
1966
+ * so it must never reach a metric, a log, or a dashboard. Absent on the
1967
+ * single-tenant local proxy and on hand-built test clients.
1968
+ */
1969
+ readonly tenant?: string;
1970
+ /**
1971
+ * `UrunResponses(session).responses.create` — an async iterable of Responses
1972
+ * stream events.
1973
+ *
1974
+ * `onDial` is WHAT THE DIAL REPORTS ABOUT ITSELF (ENG-417 + ENG-414): the
1975
+ * implementation calls it once PER DIAL with a {@link DialReport} — the uRun
1976
+ * session id of the backhaul it is ABOUT TO USE, and the billing identity
1977
+ * that dial's acquisition resolved. So a request that re-homes reports the
1978
+ * REPLACEMENT on both counts: the instance that actually served it, and the
1979
+ * identity it was actually served under.
1980
+ *
1981
+ * IT IS A SECOND ARGUMENT, NOT A FIELD ON THE ENVELOPE, and that is
1982
+ * load-bearing. `params` is forwarded VERBATIM to the serve runtime by every
1983
+ * implementation of this seam, including hand-built embedder clients; a
1984
+ * function smuggled onto it would reach a JSON serializer, be dropped
1985
+ * silently, and leave one implementation (the one that remembered to strip
1986
+ * it) behaving differently from the rest. Out of band, the envelope stays
1987
+ * exactly what it was and an implementation that ignores `onDial` simply
1988
+ * records no link.
1989
+ *
1990
+ * IT IS A CALLBACK RATHER THAN A RETURN VALUE because the answer is not
1991
+ * known when `createResponse` returns: `rehome.ts` dials inside the async
1992
+ * generator and may dial a SECOND time before the first event is yielded.
1993
+ * Asking beforehand — an earlier draft of ENG-417 did — has two defects at
1994
+ * once: it records a session the turn may abandon, and it ALLOCATES a pooled
1995
+ * backhaul for requests still going to be refused by local validation, which
1996
+ * is GPU capacity spent on a 400.
1997
+ *
1998
+ * THE IDENTITY ARRIVES THROUGH `ModelRouter.sessionFor` AND NOT FROM THE
1999
+ * POOL ENTRY, and a reader who does not know why will "simplify" it back
2000
+ * into the bug. The obvious-looking move is to widen
2001
+ * `RehomeOptions.sessionIdOf` the way the session id arrived. IT CANNOT
2002
+ * WORK: `entry` is the GENERIC pooled session (`rehome.ts` is parameterised
2003
+ * over it) and knows nothing about models, so the identity is not derivable
2004
+ * from it. The other obvious move — resolving the identity separately inside
2005
+ * the dial loop — is worse: it would be a SECOND RESOLUTION of the same
2006
+ * question, which is precisely the defect that got ENG-417's pre-dial pin
2007
+ * rejected in review (a resolution taken at a different moment from the dial
2008
+ * can answer differently, and then the row names a variant nobody served).
2009
+ * So the acquisition returns the identity alongside the entry, and the loop
2010
+ * reports what it already resolved.
2011
+ *
2012
+ * BOTH IDENTITY FLAVOURS RIDE IT. The catalog `(model_id, variant)` a row is
2013
+ * priced on, and the caller-org APP SLUG that ENG-412 made billing-bearing —
2014
+ * it lands in `model_id` with `identity_kind: 'caller_org_app'`, on rows
2015
+ * that previously were not written at all. They had the identical re-home
2016
+ * window and they close by the same line: the dial reports whichever flavour
2017
+ * it resolved.
2018
+ *
2019
+ * WIDEN {@link DialReport}; do not add a second callback beside it. Two
2020
+ * callbacks firing from the same loop about the same dial would be two
2021
+ * things that can disagree, which is the whole defect this seam exists to
2022
+ * remove.
2023
+ */
2024
+ createResponse(params: ResponsesCreateParams, onDial?: DialObserver): Promise<AsyncIterable<unknown>> | AsyncIterable<unknown>;
2025
+ /** `listModels(...)` result (an OpenAI model list object). */
2026
+ listModels(): Promise<unknown>;
2027
+ /**
2028
+ * The OpenRouter provider document (`GET /v1/models?format=openrouter`).
2029
+ * Optional at the seam: hand-built test clients may omit it, in which case
2030
+ * the format=openrouter branch answers 501 — loud, never a silent empty.
2031
+ */
2032
+ openRouterModels?(): Promise<unknown>;
2033
+ /**
2034
+ * The Images lane's capability gate (images.ts `ImageModelGate`), backed by
2035
+ * ModelRouter.supportsImages — the caller-org catalog's modality metadata,
2036
+ * per-caller by construction; never function-name inference. Optional at
2037
+ * the seam: hand-built clients may omit it, in which case the mounted
2038
+ * image lanes answer 501 — loud, never a silent allow-all.
2039
+ */
2040
+ supportsImages?(model: string | undefined, mode: 'generation' | 'edit'): Promise<boolean>;
2041
+ /**
2042
+ * The realtime/audio lane's modality gate (openai-realtime/binding.ts),
2043
+ * backed by ModelRouter.supportsAudio — the caller-org catalog's `task`
2044
+ * column (stt | tts, invariant across placements), per-caller by
2045
+ * construction; never function-name inference. Optional at the seam like
2046
+ * supportsImages: hand-built clients may omit it, in which case the
2047
+ * realtime lane refuses the upgrade LOUDLY (501) — never a silent
2048
+ * allow-all, and never a chat model binding an audio session.
2049
+ */
2050
+ supportsAudio?(model: string | undefined): Promise<boolean>;
2051
+ /**
2052
+ * The speech lane's capability gate (proxy/speech.ts), backed by
2053
+ * ModelRouter.supportsSpeech — the caller-org catalog's `task` column,
2054
+ * which must vouch `tts` (NOT `stt`: a transcription model must never be
2055
+ * asked to synthesize) INVARIANTLY across placements. Optional at the seam
2056
+ * exactly like {@link supportsImages}: hand-built clients may omit it, in
2057
+ * which case the mounted speech lane answers the loud 501 — never a
2058
+ * silent allow-all, and never a chat model asked to speak.
2059
+ */
2060
+ supportsSpeech?(model: string | undefined): Promise<boolean>;
2061
+ /**
2062
+ * THE SPEECH LANE'S VOICE-SET RESOLUTION (proxy/speech.ts): the voices
2063
+ * `/v1/models` publishes for `model` (ModelRouter.speechVoices — the SAME
2064
+ * openrouter-doc resolver the listing's enrichment applies, over the SAME
2065
+ * catalog join the tts gate uses), so the lane validates the requested
2066
+ * `voice` against exactly the set the listing publishes — ONE list
2067
+ * everywhere, never a second copy. The resolution also carries the catalog
2068
+ * defect (`engine_args.voices` present but empty/malformed/disagreeing
2069
+ * across placements), which the lane answers with the loud 500 — never a
2070
+ * silent fallback to the qwen3 card's nine. Optional at the seam exactly
2071
+ * like {@link supportsSpeech}: hand-built clients may omit it, in which
2072
+ * case the mounted speech lane answers the loud 501 — skipping the check
2073
+ * silently would downgrade the lane to unvalidated voices.
2074
+ */
2075
+ speechVoices?(model: string | undefined): Promise<SpeechVoicesResolution>;
2076
+ /**
2077
+ * `GET /v1/images/models` — the image-capable slice of the model surface
2078
+ * (ModelRouter.imageModelList: deployed task-image apps + the shared block
2079
+ * joined against the catalog). Optional at the seam for the same reason;
2080
+ * its absence is the loud 501, never a silently empty list.
2081
+ */
2082
+ listImageModels?(): Promise<unknown>;
2083
+ /**
2084
+ * THE PRE-DIAL DEFAULT FOR THE LEDGER'S BILLING IDENTITY
2085
+ * (ModelRouter.catalogIdentityFor): the CATALOG `(model_id, variant)` a
2086
+ * request's `model` actually resolved to, or `null` when it names no catalog
2087
+ * model (a caller-org app, whose identity is an app slug — a different
2088
+ * namespace from `model_prices.model_id`).
2089
+ *
2090
+ * The ledger prices on `(model_id, variant)`, and the caller's raw string is
2091
+ * not that: a bare catalog id resolves to the model's `compat_default`
2092
+ * variant, so recording the string would file the request under a variant
2093
+ * nobody served.
2094
+ *
2095
+ * THE DIAL HAS THE LAST WORD (ENG-414). What a served row records is the
2096
+ * identity `onDial` reported, off the acquisition that dial used — so a
2097
+ * re-homed request is billed on the replacement. This seam is what a request
2098
+ * that reaches NO dial records instead, which is why it is still resolved
2099
+ * on the request path.
2100
+ *
2101
+ * REQUIRED AT THE SEAM (ENG-429), unlike `supportsImages`. ENG-414 changed
2102
+ * what an omission MEANS: a backhaul with no `catalogIdentity` but a
2103
+ * dialling `createResponse` used to write no ledger row at all, and now
2104
+ * writes — and prices — one on the dial's own report. Honest, but it means
2105
+ * a lane could acquire BILLING behaviour by OMITTING a member, which is the
2106
+ * wrong direction for a default on a revenue path. So "no seam" is a
2107
+ * COMPILE ERROR now, the same move as the exhaustive `Record<...>`
2108
+ * registration checks used elsewhere in this codebase: a contract the
2109
+ * typechecker holds, not a convention a reviewer must notice. An implementer
2110
+ * that genuinely has no catalog oracle says so EXPLICITLY — a function of
2111
+ * its own, in code a reviewer can see — never by leaving the member out, and
2112
+ * never an invented identity. (The runtime guard in server.ts
2113
+ * `pinCatalogIdentity` stays for UNTYPED embedders: the seam is a public
2114
+ * surface callable from plain JS, where a member can still be absent at
2115
+ * runtime, and the guard's answer there remains the loud `unresolved` pin.)
2116
+ */
2117
+ catalogIdentity(model: string | undefined): Promise<{
2118
+ model_id: string;
2119
+ variant: string;
2120
+ } | null>;
2121
+ /**
2122
+ * THE CALLER-ORG APP's OWN IDENTITY (ENG-412): the app slug `model` actually
2123
+ * RESOLVES to, for the lane where {@link catalogIdentity} answers null. Like
2124
+ * {@link catalogIdentity} it is the PRE-DIAL DEFAULT — the dial reports this
2125
+ * flavour too, and has the last word (ENG-414).
2126
+ *
2127
+ * IT IS NOT THE CALLER'S RAW STRING. Routing trims, length-caps and matches
2128
+ * that string before selecting an app, so a differently cased, punctuated or
2129
+ * whitespace-padded name is served by one canonical slug while the raw value
2130
+ * says something else — and the ledger would then file two spellings of the
2131
+ * same app as two different identities, which no app-level reconciliation
2132
+ * can undo afterwards.
2133
+ *
2134
+ * CHEAP AND ALLOCATION-FREE: it rides `ModelRouter.resolveTarget`, whose app
2135
+ * listing, catalog and shared block are cached promises. It opens no session
2136
+ * — asking which app would serve must never BE the thing that spends GPU
2137
+ * capacity.
2138
+ */
2139
+ servingAppSlug?(model: string | undefined): Promise<string | null>;
2140
+ /**
2141
+ * Open (or reuse) the NATIVE audio lanes on the pooled session `model` routes
2142
+ * to — the same first-party machinery as RealtimeClient.enableAudio
2143
+ * (transport/media.ts `enableSessionAudio`: AudioBridge over the session's
2144
+ * `audio` stream lane at 24 kHz). One lane per pooled
2145
+ * session; repeat calls return the same lane. Optional at the seam because
2146
+ * text-only embeddings exist — but a surface that RECEIVES audio while the
2147
+ * embedder wired no `openAudio` must fail LOUD, never drop chunks.
2148
+ *
2149
+ * `signal` aborts when the CLIENT that asked for the lane went away while
2150
+ * the lane's admission was still pending (the realtime lane's raw socket
2151
+ * close) — the backhaul aborts its `whenLive` wait on it and hands the
2152
+ * queued admission ticket back through core `Session.cancel()`, so a
2153
+ * vanished client never holds the sense pod's one-session slot.
2154
+ */
2155
+ openAudio?(model: string | undefined, signal?: AbortSignal): Promise<ProxyAudioLane>;
2156
+ /**
2157
+ * THE PINNED AUDIO OPEN (speech lane): the NATIVE audio lane of the EXACT
2158
+ * pooled session `handle` names (ModelRouter.sessionForHandle — a gone or
2159
+ * replaced session throws {@link SessionGoneError} LOUDLY, never a fresh
2160
+ * session opened behind the handle's back). The speech lane opens its
2161
+ * audio collector here and dials its response turn on the SAME handle
2162
+ * ({@link createResponseOn}), so BOTH legs are one session identity by
2163
+ * construction: a re-home can never move the turn to a second session
2164
+ * while the collector stays subscribed to the first (which missed the
2165
+ * replacement's audio and answered a voiceless 502, or worse, served a
2166
+ * concurrent collector another session's frames). Optional at the seam
2167
+ * exactly like {@link openAudio}; the speech lane answers its absence
2168
+ * with the loud 501 — never a voiceless downgrade.
2169
+ */
2170
+ openAudioOn?(handle: string): Promise<PinnedAudioLane>;
2171
+ /**
2172
+ * Open (or reuse) the NATIVE video FRAME lane on the pooled session `model`
2173
+ * routes to (transport/media.ts `enableSessionVideo`: discrete JPEG frames →
2174
+ * `stream('rt-video-in').emit`, the §5 named-DATA image-bytes path — NOT the
2175
+ * RTP media plane, which requires an already-H.264-encoded track). One lane
2176
+ * per pooled session; repeat calls return the same lane. Optional at the seam
2177
+ * because text-only embeddings exist — but a surface that RECEIVES video
2178
+ * frames while the embedder wired no `openVideo` must fail LOUD, never drop
2179
+ * frames.
2180
+ */
2181
+ openVideo?(model: string | undefined): Promise<ProxyVideoLane>;
2182
+ /**
2183
+ * Open (or reuse) the NATIVE video OUTPUT lane on the pooled session
2184
+ * `model` routes to (transport/video-out.ts `openSessionVideoOutLane`):
2185
+ * decoded spec-v1 records consumed from the session's named §5
2186
+ * `rt-video-out` downstream. The Live adapter fans each record onto the
2187
+ * SAME Bidi WS as an extension server message (spec v2, WS-inline) — no
2188
+ * side-channel transport. Optional at the seam; an ABSENT seam means the
2189
+ * `urun.videoOut` capability is refused by omission in setupComplete
2190
+ * (vanilla parity), while a wired seam that cannot open the lane must
2191
+ * throw LOUD — never a quiet downgrade.
2192
+ */
2193
+ openVideoOut?(model: string | undefined): Promise<ProxyVideoOutLane>;
2194
+ /**
2195
+ * SESSION-IDENTITY SEAM (ModelRouter.handleFor): the opaque stable handle
2196
+ * for the pooled session currently serving `model`'s turns. Derived from
2197
+ * native identity (app slug + uRun session id) — the same identity the
2198
+ * serve-side session-affinity tag rides (urun-python#1556/#1582).
2199
+ */
2200
+ sessionHandle(model: string | undefined): Promise<string>;
2201
+ /**
2202
+ * THE SPEECH LANE'S OWN HANDLE (greptile urun-ts#500): the opaque handle
2203
+ * for the SPEECH-SCOPED pooled session (`speech:<pool key>` — routing.ts
2204
+ * {@link ModelRouter.speechSessionFor}), a slot no interactive lane can
2205
+ * resolve. The speech lane resolves its collector AND its turn through
2206
+ * THIS handle ({@link openAudioOn} + {@link createResponseOn} unchanged),
2207
+ * because the per-session AudioBridge broadcasts every untagged output
2208
+ * frame to every subscriber: a speech collection on the INTERACTIVE
2209
+ * pooled session would stitch a concurrent Realtime/Gemini Live turn's
2210
+ * frames into the returned WAV. Optional at the seam exactly like
2211
+ * {@link openAudioOn}; the speech lane answers its absence with the loud
2212
+ * 501 — NEVER a silent fall-back to the shared interactive session
2213
+ * (that "fallback" is precisely the mixed-audio defect).
2214
+ */
2215
+ speechSessionHandle?(model: string | undefined): Promise<string>;
2216
+ /**
2217
+ * EVICT the SPEECH-scoped pooled session a handle names (CodeRabbit
2218
+ * urun-ts#500, round 8) — through the pool's OWN identity-guarded
2219
+ * eviction path (ModelRouter.evict under the handle's `speech:<key>`
2220
+ * slot, never a bespoke close), so the NEXT speechSessionHandle acquire
2221
+ * opens a FRESH session. The lane calls this after a collector failure
2222
+ * that leaves the session's audio state unknown — above all a turn-cap
2223
+ * timeout, where the serialized lane is released while the pinned
2224
+ * session may still be emitting the timed-out turn's audio: a later
2225
+ * request on the SAME entry would collect those stale frames into its
2226
+ * own answer. Throws {@link SessionGoneError} when the handle names a
2227
+ * session the pool no longer holds (the eviction's goal is then already
2228
+ * true — the lane treats it as success). Optional at the seam exactly
2229
+ * like {@link speechSessionHandle}; the speech lane answers its absence
2230
+ * with the loud 501 — never a silent reuse of the possibly-stale entry.
2231
+ */
2232
+ evictSpeechSession?(handle: string): Promise<void>;
2233
+ /**
2234
+ * createResponse PINNED to the exact session a handle names
2235
+ * (ModelRouter.sessionForHandle). Throws SessionGoneError LOUDLY when that
2236
+ * session is gone or was replaced — never silently opens a fresh session
2237
+ * while claiming resume. Deliberately NO re-home on this path: re-homing
2238
+ * would swap the pinned session out from under the caller.
2239
+ */
2240
+ createResponseOn(handle: string, params: ResponsesCreateParams): Promise<AsyncIterable<unknown>> | AsyncIterable<unknown>;
2241
+ /**
2242
+ * Subscribe to the pinned session's terminal end via core's NATIVE phase
2243
+ * machinery (Session.onPhase → terminal 'expired'/'ended'/'error'). Fires
2244
+ * `cb` once. Throws SessionGoneError if the handle's session is already
2245
+ * gone — which doubles as the loud reattach check at resume time. Returns
2246
+ * the unsubscribe.
2247
+ */
2248
+ onSessionEnd(handle: string, cb: (end: SessionEndInfo) => void): Promise<() => void>;
2249
+ /**
2250
+ * THE STATELESS DIRECT PATH (stateless-dispatch.ts): the decision dial to
2251
+ * the serving pods' `POST /v1/decision` intake, present only when the
2252
+ * deployment carries the runtime scoped-token secret. Absent ⇒ every
2253
+ * decision rides the session lane exactly as before (the staged rollout).
2254
+ * Per-org: the HOSTED registry builds one per tenant so the minted token's
2255
+ * tenant claim is always the CALLER's org.
2256
+ */
2257
+ readonly stateless?: StatelessDecisionDial;
2258
+ }
2259
+ /**
2260
+ * The audio lane handle `openAudio` returns — structurally the transport
2261
+ * AudioBridge (media.ts): base64 PCM16 @24 kHz mono in both directions.
2262
+ * `sessionId` is the pooled uRun session the lane rides (core `Session.id`),
2263
+ * when the backhaul reports it — the identity the realtime lane's structured
2264
+ * lifecycle log carries so an orphaned session is traceable from the proxy
2265
+ * log alone. Null is the HONEST ABSENCE (the same line {@link
2266
+ * PinnedAudioLane.sessionId} and the re-homing dial's `sessionIdOf` take —
2267
+ * an id-less session reports no id, never a fabricated one).
2268
+ */
2269
+ type ProxyAudioLane = Pick<AudioBridge, 'appendInputAudio' | 'onOutputAudio'> & {
2270
+ sessionId?: string | null;
2271
+ };
2272
+ /**
2273
+ * The PINNED audio lane handle `openAudioOn` returns ({@link ProxyClients.openAudioOn}):
2274
+ * the session's native audio lane PLUS the identity of the exact pooled session
2275
+ * it is bound to — the two facts the speech lane needs to keep its collector and
2276
+ * its response turn on ONE session identity.
2277
+ */
2278
+ interface PinnedAudioLane extends ProxyAudioLane {
2279
+ /**
2280
+ * The uRun session id of the pooled session this lane is bound to — the
2281
+ * honest instance link for a request whose collector and turn ride one
2282
+ * handle, the same derivation the dial report's `sessionId` uses. Null
2283
+ * when the entry carries no native session id; never a fabricated one.
2284
+ */
2285
+ readonly sessionId: string | null;
2286
+ }
2287
+ /**
2288
+ * The video frame-lane handle `openVideo` returns — structurally the
2289
+ * transport VideoFrameLane (media.ts): one raw encoded JPEG frame per call,
2290
+ * input-only (Live-style protocols have no video OUT modality).
2291
+ */
2292
+ type ProxyVideoLane = Pick<VideoFrameLane, 'sendInputFrame'>;
2293
+ /**
2294
+ * The video-OUT lane handle `openVideoOut` returns — structurally the
2295
+ * transport VideoOutLane (video-out.ts): the codec plus frame/error fan-out
2296
+ * the WS-inline delivery subscribes.
2297
+ */
2298
+ type ProxyVideoOutLane = Pick<VideoOutLane, 'codec' | 'onOutputFrame' | 'onLaneError'>;
2299
+ /**
2300
+ * Per-request `ProxyClients` resolution — the seam the HOSTED multi-tenant
2301
+ * server (`src/hosted/`) plugs into so ONE handler implementation serves
2302
+ * every org: the hosted server resolves the caller's org from its Bearer API
2303
+ * key and returns THAT org's backhaul. The local CLI passes a fixed
2304
+ * `ProxyClients` instead; both go through the identical handler body.
2305
+ *
2306
+ * Throwing from here is the loud path: the thrown error surfaces in the
2307
+ * lane's native error envelope (see {@link ProxyHandlerOptions.statusOf}).
2308
+ */
2309
+ type ProxyClientsFor = (req: IncomingMessage) => ProxyClients | Promise<ProxyClients>;
2310
+ interface ProxyHandlerOptions {
2311
+ /** A fixed backhaul (local CLI) or a per-request resolver (hosted server). */
2312
+ clients: ProxyClients | ProxyClientsFor;
2313
+ /** Optional bearer the local agent must present (never forwarded upstream). */
2314
+ apiKey?: string;
2315
+ /**
2316
+ * WHAT this proxy serves (app/org/fn/base_url + proxy_version), surfaced as
2317
+ * the `identity` block on `GET /stats` so a second `urun compat` invocation
2318
+ * can reuse this proxy iff the identity matches its own exactly. Only the
2319
+ * standalone `urun compat proxy` command sets it — a launch-mode proxy is
2320
+ * child-owned (it dies with its child) and an embedder that omits it is
2321
+ * simply never reused.
2322
+ */
2323
+ identity?: ProxyIdentity;
2324
+ /**
2325
+ * Map a thrown error onto an HTTP status + error `type` before the generic
2326
+ * 500. The hosted server uses it to turn its auth/tenancy failures into a
2327
+ * 401 in the OpenAI envelope. Returning null means "not mine" — the error
2328
+ * takes the ordinary loud 500 path.
2329
+ */
2330
+ statusOf?: (err: unknown) => {
2331
+ status: number;
2332
+ openaiType: string;
2333
+ anthropicType: string;
2334
+ } | null;
2335
+ /**
2336
+ * Where the per-request structured log line goes — ONE newline-terminated
2337
+ * JSON object per `/v1` request, carrying the request's `Inference-Id`
2338
+ * (proxy/inference-id.ts `inferenceRequestLine`) plus method, path, status
2339
+ * and whether the answer was delivered / failed mid-stream, written when the
2340
+ * response closes (so streams and client aborts are recorded too). This is
2341
+ * the billing correlation record a later usage ledger is keyed on, so it is
2342
+ * emitted unconditionally — never behind a verbosity flag.
2343
+ *
2344
+ * Absent, the record goes to this process's stdout through the repo's
2345
+ * canonical guarded write (`server.ts` `writeInferenceLogToStdout`), which is
2346
+ * the hosted endpoint's collected container log. The local CLI proxy passes
2347
+ * its own sink instead, because the operator owns that terminal — see
2348
+ * `proxy/cli.ts` `startProxy`. Injectable so tests assert on the lines
2349
+ * without racing the process streams.
2350
+ */
2351
+ requestLog?: (line: string) => void;
2352
+ /**
2353
+ * THE CANONICAL USAGE-QUERY SEAM (`POST /v1/usage/requests`, and the
2354
+ * Hugging Face translator over it) — a batch lookup of what the per-request
2355
+ * inference ledger holds for a list of `Inference-Id`s, scoped to the
2356
+ * caller's own org. See proxy/usage.ts.
2357
+ *
2358
+ * Only the HOSTED endpoint supplies one (the read rides the caller's own
2359
+ * org API key to the control plane; this process holds no database
2360
+ * credential). A proxy without it answers those routes with a LOUD 501 —
2361
+ * the `supportsImages` posture — and never with an empty result, which a
2362
+ * billing caller cannot tell from "nothing is priced yet".
2363
+ */
2364
+ usage?: UsageLane;
2365
+ /**
2366
+ * THE PER-REQUEST LEDGER WRITE SEAM (proxy/ledger.ts) — one durable,
2367
+ * already-priced row per billable `/v1` request, written from the response's
2368
+ * `'close'` hook and therefore OFF the response path.
2369
+ *
2370
+ * Only the HOSTED endpoint supplies one (the write rides the caller's own
2371
+ * org API key to the control plane; this process holds no database
2372
+ * credential). A proxy WITHOUT one does not participate in the ledger at
2373
+ * all — the `usage` / `supportsImages` posture of declared absence. The
2374
+ * local CLI proxy is single-tenant and bills nothing, so it has none.
2375
+ *
2376
+ * It is deliberately NOT a place to do work on the hot path, and it cannot
2377
+ * be: `recordInference` is fire-and-forget, returns void, and never throws.
2378
+ */
2379
+ ledger?: LedgerLane;
2380
+ /**
2381
+ * The SECOND sink for the very same per-request record `requestLog`
2382
+ * serializes — the hosted endpoint passes its Prometheus collector
2383
+ * (`proxy/metrics.ts` `InferenceMetrics.observe`). Called once per `/v1`
2384
+ * request, from the same `'close'` hook, with the same object.
2385
+ *
2386
+ * Optional because the local CLI proxy publishes no metrics port; absent, a
2387
+ * request is logged and not counted. It is deliberately NOT a place to do
2388
+ * work: it runs on the response's close hook, so anything slow here delays
2389
+ * the socket teardown.
2390
+ */
2391
+ observe?: (record: InferenceRecord) => void;
2392
+ }
2393
+
2394
+ export { type DialObserver as D, type InferenceRecord as I, type LedgerFailure as L, ModelRouter as M, type ProxyClients as P, SessionGoneError as S, UnknownModelError as U, type DialReport as a, type InferenceUsageRecord as b, type LedgerOutcome as c, type LedgerRecorded as d, type LedgerWrite as e, type ProxyHandlerOptions as f, type ProxyIdentity as g, type ProxyVideoOutLane as h, type UsageRowReporter as i };