@urun-sh/openai 0.5.5 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/dist/{ResponsesClient-BYx3YLGo.d.ts → ResponsesClient-BE3-hx3m.d.ts} +2 -0
  2. package/dist/{ResponsesClient-Dft3bg3b.d.cts → ResponsesClient-DjZWjlFH.d.cts} +2 -0
  3. package/dist/chunk-2T2YYBVX.js +1 -0
  4. package/dist/chunk-3WAJD62J.js +4 -0
  5. package/dist/{chunk-K23AZPI4.js → chunk-5HKWNK3O.js} +1 -1
  6. package/dist/chunk-7M6LY6DR.js +2 -0
  7. package/dist/chunk-BSHT6RZZ.js +1 -0
  8. package/dist/{chunk-GBBY3PCZ.js → chunk-FP4RSAIE.js} +1 -1
  9. package/dist/chunk-HR5H6S7L.js +1 -0
  10. package/dist/chunk-NY23USZF.js +57 -0
  11. package/dist/chunk-VLRMJRLS.js +6 -0
  12. package/dist/gemini-live.cjs +2 -2
  13. package/dist/gemini-live.d.cts +17 -2
  14. package/dist/gemini-live.d.ts +7 -2
  15. package/dist/gemini-live.js +1 -1
  16. package/dist/hosted/bin.cjs +40 -26
  17. package/dist/hosted/bin.js +4 -4
  18. package/dist/hosted/index.cjs +39 -26
  19. package/dist/hosted/index.d.cts +643 -16
  20. package/dist/hosted/index.d.ts +153 -5
  21. package/dist/hosted/index.js +1 -1
  22. package/dist/index.cjs +1 -1
  23. package/dist/index.d.cts +4 -4
  24. package/dist/index.d.ts +4 -4
  25. package/dist/index.js +1 -1
  26. package/dist/{models-NYMZrklp.d.cts → models-DUdx_Y6X.d.cts} +1 -1
  27. package/dist/pi-extension/index.cjs +7 -7
  28. package/dist/pi-extension/index.d.cts +1 -1
  29. package/dist/pi-extension/index.d.ts +1 -1
  30. package/dist/pi-extension/index.js +1 -1
  31. package/dist/pi-extension/standalone.cjs +48 -48
  32. package/dist/proxy/cli.cjs +53 -43
  33. package/dist/proxy/cli.js +14 -14
  34. package/dist/proxy/index.cjs +32 -30
  35. package/dist/proxy/index.d.cts +84 -8
  36. package/dist/proxy/index.d.ts +16 -6
  37. package/dist/proxy/index.js +1 -6
  38. package/dist/responses-turn-Dr37N3kH.d.ts +486 -0
  39. package/dist/responses-turn-tNUHq5b0.d.cts +1807 -0
  40. package/dist/{translator-C9uPKypK.d.ts → translator-BCoRFaTs.d.ts} +1 -1
  41. package/dist/{translator-CcDBEfvm.d.cts → translator-Bh0Bp_Ie.d.cts} +3 -3
  42. package/dist/{video-out-D20UuJ8G.d.cts → video-out-CCksIcj8.d.cts} +5 -5
  43. package/package.json +12 -11
  44. package/dist/chunk-5BZCM3RS.js +0 -4
  45. package/dist/chunk-5NXM4IO3.js +0 -2
  46. package/dist/chunk-CWJRDBDC.js +0 -1
  47. package/dist/chunk-DMD5UENK.js +0 -1
  48. package/dist/chunk-I2Q3B3OG.js +0 -6
  49. package/dist/chunk-OI2OY32M.js +0 -1
  50. package/dist/chunk-QVF7NF7G.js +0 -44
  51. package/dist/responses-turn-KOAoIqZ-.d.ts +0 -200
  52. package/dist/responses-turn-OrO4euEN.d.cts +0 -513
  53. /package/dist/{chunk-YSFSRI3D.js → chunk-5CCJUNH7.js} +0 -0
  54. /package/dist/{models-NYMZrklp.d.ts → models-DUdx_Y6X.d.ts} +0 -0
  55. /package/dist/{video-out-CWesbk12.d.ts → video-out-IGQ4YQ-c.d.ts} +0 -0
@@ -0,0 +1,1807 @@
1
+ import { C as CatalogRow } from './models-DUdx_Y6X.cjs';
2
+ import { IncomingMessage } from 'node:http';
3
+ import { a as AudioBridge, h as VideoFrameLane, j as VideoOutLane } from './video-out-CCksIcj8.cjs';
4
+
5
+ /**
6
+ * WHAT a proxy serves — the `identity` block on `GET /stats`. Reuse of a
7
+ * running proxy is allowed iff every field matches the launching invocation
8
+ * EXACTLY: a partial match (same app, different fn; same everything, older
9
+ * proxy_version) silently routes the agent at the wrong backend, which is
10
+ * worse than any error.
11
+ */
12
+ interface ProxyIdentity {
13
+ app: string;
14
+ org: string;
15
+ fn: string;
16
+ /**
17
+ * The control-plane URL the proxy's backhaul session opens against
18
+ * (URUN_BASE_URL verbatim — same org/app/fn on staging vs prod are
19
+ * DIFFERENT backends; review finding, #285). Compared exactly: a cosmetic
20
+ * difference (trailing slash) merely refuses reuse and spawns an ephemeral
21
+ * proxy — the safe direction.
22
+ */
23
+ base_url: string;
24
+ proxy_version: string;
25
+ }
26
+
27
+ /**
28
+ * THE BILLING CORRELATION SEAM — one uuid per `/v1` request, returned as the
29
+ * `Inference-Id` response header, carried as the suffix of the body id the
30
+ * request answers with, and recorded in one structured log line.
31
+ *
32
+ * WHY IT EXISTS: Hugging Face Inference Providers bill routed requests by a
33
+ * unique id the provider returns as a RESPONSE HEADER. Their spec is explicit:
34
+ * "Make sure this header is present on every response you return, including
35
+ * streaming responses. If it's missing, we have no way to match the request
36
+ * and it can't be billed." A response that leaves this proxy without the
37
+ * header is therefore an UNBILLED response — a revenue bug, not a cosmetic
38
+ * one — which is why the header is set on the response object in the request
39
+ * handler's synchronous prefix, before any lane can flush a head.
40
+ *
41
+ * WHAT THIS MINT COVERS: the three HTTP lanes' body ids (`chatcmpl-…`,
42
+ * `resp_…`, `msg_…`) and the `/v1/responses` WebSocket lane's response ids —
43
+ * all of which previously minted their own `Math.random().toString(36)`
44
+ * string, four unrelated id spaces nothing downstream could join on. Only
45
+ * `executeResponsesTurn`'s required `responseId` argument is COMPILE-enforced;
46
+ * the rest is enforced by review, so keep new lanes on this mint.
47
+ *
48
+ * WHAT IT DOES NOT COVER (still separate id spaces, recorded on ENG-309):
49
+ * - `transport/decode.ts` builds `resp_${requestId}` from the SERVE-lane
50
+ * request id that `responses/ResponsesClient.ts` mints (`req_<n>`); the
51
+ * HTTP lanes overwrite it with the id from here, but joining a billing row
52
+ * to the runtime's own usage receipt still needs that id plumbed through
53
+ * `ProxyClients.createResponse`.
54
+ * - the OpenAI Realtime lane (`proxy/openai-realtime/protocol.ts`) mints its
55
+ * own `resp_` ids for WS realtime turns — which, being built from a
56
+ * `req_`-prefixed seed, currently read `resp_req_<hex>` (a double prefix).
57
+ */
58
+
59
+ type TurnCatalogIdentity = {
60
+ kind: 'catalog';
61
+ model_id: string;
62
+ variant: string;
63
+ } | {
64
+ kind: 'not-a-catalog-model';
65
+ model: string | null;
66
+ } | {
67
+ kind: 'unresolved';
68
+ reason: string;
69
+ };
70
+ /**
71
+ * WHAT ONE DIAL REPORTS ABOUT ITSELF — everything the ledger reads off the
72
+ * dial that actually served, reported together, from that dial.
73
+ *
74
+ * ONE CALLBACK, BOTH FACTS, and that is the point rather than a convenience.
75
+ * `session_id` (ENG-417) and the billing identity (ENG-414) are the two halves
76
+ * of the same question — which instance ran this, and what it ran — and a
77
+ * re-home makes them disagree the moment they are sourced separately: the row
78
+ * named the REPLACEMENT instance while naming the identity the ABANDONED dial
79
+ * resolved. Two callbacks firing from the same loop about the same dial would
80
+ * be two things free to drift apart again, which is the whole defect this seam
81
+ * exists to remove. Widen this payload; never add a second callback beside it.
82
+ *
83
+ * FIRED ONCE PER DIAL, LAST DIAL WINS. `rehome.ts` performs exactly one
84
+ * re-dial when a backhaul dies before producing client-visible content, and
85
+ * reports again before the replacement is used — so what the ledger keeps is
86
+ * the dial that served.
87
+ */
88
+ interface DialReport {
89
+ /**
90
+ * The uRun session id of the backhaul about to be used, or null when the
91
+ * pool entry carries no session identity. Never a fabricated id: a wrong
92
+ * link moves cost attribution between machines.
93
+ */
94
+ sessionId: string | null;
95
+ /**
96
+ * The billing identity THIS dial resolved — the ledger's own vocabulary, not
97
+ * the router's. It comes off the SAME `resolveTarget` answer the dial used
98
+ * (`ModelRouter.sessionFor`), never a second resolution taken alongside it:
99
+ * a resolution taken at a different moment can answer differently, and then
100
+ * the row names a variant nobody served. Both flavours travel this way — the
101
+ * catalog `(model_id, variant)` and the caller-org app slug, which ENG-412
102
+ * made billing-bearing.
103
+ */
104
+ identity: TurnCatalogIdentity;
105
+ }
106
+ /** The observer one dial reports itself to. See {@link DialReport}. */
107
+ type DialObserver = (dial: DialReport) => void;
108
+ /**
109
+ * ONE completed `/v1` request, as data. This is the SINGLE per-request fact
110
+ * object: {@link inferenceRequestLine} serializes it into the billing
111
+ * correlation log line, and `proxy/metrics.ts` `InferenceMetrics.observe`
112
+ * counts the same object into Prometheus. Two sinks, one record — a metric
113
+ * that could disagree with the billing log would be worse than no metric.
114
+ *
115
+ * Nothing here is a secret or a body: no API key, no prompt, no completion.
116
+ * The caller is identified by `org` (a control-plane identifier), never by the
117
+ * Bearer key that resolved it.
118
+ */
119
+ interface InferenceRecord {
120
+ readonly event: 'inference_request';
121
+ readonly ts: string;
122
+ readonly inference_id: string;
123
+ readonly method: string;
124
+ readonly path: string;
125
+ readonly status: number | null;
126
+ readonly delivered: boolean;
127
+ readonly stream_error: boolean;
128
+ readonly model: string | null;
129
+ readonly stream: boolean | null;
130
+ readonly org: string | null;
131
+ readonly duration_ms: number;
132
+ readonly ttft_ms: number | null;
133
+ /**
134
+ * The usage quantities the per-request ledger is written from. NULL means
135
+ * the request HAS no such quantity — never zero of it. See
136
+ * {@link InferenceTurn.promptTokens}.
137
+ *
138
+ * They are on the RECORD rather than read separately by the ledger writer
139
+ * for the reason the record exists at all: one object, every sink. A writer
140
+ * that took a second walk over the request could disagree with the log line
141
+ * and the metric, and the disagreement would be invisible.
142
+ *
143
+ * NOTE WHAT IS NOT HERE: `gpu_seconds`. There is no such quantity, and as
144
+ * of ENG-417 there is no such ledger column either. Exclusive per-request
145
+ * GPU occupancy is not obtainable under continuous batching, and wall-clock
146
+ * is SHARED occupancy — a per-request stamp would double-count against
147
+ * `usage_events.metered_gpu_seconds`, which meters the instance once. What
148
+ * is here instead is the LINK ({@link session_id}) and the WEIGHT (the
149
+ * residency milliseconds below), from which the share is DERIVED.
150
+ */
151
+ readonly prompt_tokens: number | null;
152
+ readonly completion_tokens: number | null;
153
+ readonly output_units: number | null;
154
+ /**
155
+ * THE INSTANCE that served this request — see
156
+ * {@link InferenceTurn.sessionId}. On the log line as well as the ledger,
157
+ * so a row that was never written can still be traced to the session it
158
+ * would have been attributed to.
159
+ */
160
+ readonly session_id: string | null;
161
+ /**
162
+ * THE WEIGHT — see {@link InferenceTurn.timing}. Flattened onto the record
163
+ * rather than nested, because this record is also a log line and a flat
164
+ * field is what a log query can select. NULL means the runtime reported
165
+ * nothing; it never means zero.
166
+ */
167
+ readonly queue_ms: number | null;
168
+ readonly prefill_ms: number | null;
169
+ readonly decode_ms: number | null;
170
+ readonly total_ms: number | null;
171
+ /**
172
+ * The identity that served this request, reported by the dial that served it
173
+ * — see {@link TurnCatalogIdentity} and {@link DialReport}. The ledger writer
174
+ * reads it from here rather than asking the router again, and the log line
175
+ * carries it so a row that was never written can be traced to the reason
176
+ * from the same record.
177
+ */
178
+ readonly catalog: TurnCatalogIdentity;
179
+ }
180
+
181
+ /**
182
+ * THE SHARED-MODEL LANE (shared-endpoints U8, urun-infra#1490 §7) — the pure
183
+ * per-request resolver over the catalog edge function's `shared` block:
184
+ *
185
+ * { shared_org_id, endpoints: [{ model_id, variant, gpu_spec, app_slug,
186
+ * function, warm }] }
187
+ *
188
+ * Every function here is PURE and synchronous over the parsed block: the
189
+ * catalog fetch (U5's edge-function surface) is a separate seam and is NOT
190
+ * built yet — a caller hands the block in, or doesn't. An absent/empty
191
+ * `shared` block means "no shared models" and every lookup says so (null),
192
+ * never a stub.
193
+ *
194
+ * POSTURE (the two error classes, deliberately distinct):
195
+ * - `null` → this model is not in the shared block. The caller
196
+ * keeps its own resolution order; on the hosted lane that ends in the
197
+ * loud UnknownModelError 404. NOT an error — "not mine" is an answer.
198
+ * - `SharedCatalogError` → the block itself is broken (malformed shape, or
199
+ * a bare model id whose compat_default flag is not unique). This is a
200
+ * platform catalog defect, not a caller mistake: it surfaces as a LOUD
201
+ * 500 naming the row, never a silent resolution to "some" variant.
202
+ */
203
+ /** One shared endpoint row — the edge function's `endpoints[]` element. */
204
+ interface SharedEndpointRow {
205
+ model_id: string;
206
+ variant: string;
207
+ /** The shared org's deployed app slug the session dial targets. */
208
+ app_slug: string;
209
+ /** The serve function name on that app. */
210
+ function: string;
211
+ /** PLACEMENT-level GPU spec, e.g. 'rtx6000:1' (mirrors CatalogRow). */
212
+ gpu_spec?: string | null;
213
+ /** Idle-floor replica count (the reconciler's `warm`, U6). */
214
+ warm?: number | null;
215
+ /**
216
+ * The compat_default flag (U4): a BARE model id resolves to THE flagged
217
+ * row. Exactly one per model_id — zero or two is a loud catalog defect.
218
+ */
219
+ compat_default?: boolean | null;
220
+ }
221
+ /** The catalog edge function's `shared` block. */
222
+ interface SharedCatalogBlock {
223
+ shared_org_id: string;
224
+ endpoints: SharedEndpointRow[];
225
+ }
226
+
227
+ /**
228
+ * THE PER-TOKEN PRICE SURFACE — what `GET /v1/models` may say about money.
229
+ *
230
+ * Hugging Face reads our OpenAI-shaped model list to populate its public
231
+ * Inference-Provider comparison table and to power its `:fastest` and
232
+ * `:cheapest` routing. The required per-model shape is
233
+ *
234
+ * "pricing": { "input": <USD per MILLION input tokens>,
235
+ * "output": <USD per MILLION output tokens> }
236
+ *
237
+ * so this module resolves exactly that pair — per catalog model — out of
238
+ * urun-infra's `public.model_prices` surface (shared-endpoints U3,
239
+ * migration 20271003000003), and NOTHING else.
240
+ *
241
+ * OMISSION IS THE CONTRACT. A model with no per-token price row carries NO
242
+ * `pricing` key at all — never `0`, never a GPU-minute rate reinterpreted as
243
+ * a token rate, never an estimate. `model_prices` prices GPU TIME today
244
+ * (`unit in ('per_request','per_minute','per_output_unit')`, and the seeded
245
+ * rows are per_minute GPU-minutes); turning GPU-minutes into per-token rates
246
+ * is a PRICING DECISION a human makes, tracked as ENG-317. A fabricated
247
+ * number here would be a silent lie on a public comparison table that other
248
+ * people's routing decisions depend on, which is strictly worse than an
249
+ * absent field: HF renders an unpriced model as unpriced, and that is true.
250
+ *
251
+ * TWO GAPS THAT MUST CLOSE UPSTREAM BEFORE THIS SURFACE CAN EVER BE
252
+ * NON-EMPTY. Both are recorded on **ENG-317**, which is the durable record
253
+ * — this comment only points at it, because a code comment and a PR body
254
+ * are exactly the channels CLAUDE.md rule 2 names as insufficient:
255
+ *
256
+ * 1. NO PER-TOKEN UNIT EXISTS YET (ENG-317). `model_prices_unit_check`
257
+ * allows only per_request | per_minute | per_output_unit. The two unit
258
+ * strings this module recognizes — {@link PER_MILLION_INPUT_UNIT} and
259
+ * {@link PER_MILLION_OUTPUT_UNIT}, with `price_usd` read VERBATIM as
260
+ * USD per million tokens (no conversion, and `numeric(12,8)` has ample
261
+ * resolution at per-million scale — a per-single-token unit would not)
262
+ * — are a PROPOSED contract ENG-317 must ratify and a migration must
263
+ * add. Until then every live read yields zero per-token rows and every
264
+ * entry is unpriced. (If ENG-317 ratifies different names, a row
265
+ * carrying them is LOUD here, not skipped — see
266
+ * {@link KNOWN_NON_TOKEN_UNITS}.)
267
+ * 2. THE ANON READ IS RLS-BLOCKED (ENG-317). `model_prices` grants
268
+ * `select` to `anon` (migration 20271003000003 line 107), but its only
269
+ * policy (`model_prices_select_all`, line 112) is
270
+ * `for select to authenticated` — so a read with the proxy's catalog
271
+ * anon key returns `[]` whatever the table holds, indistinguishably
272
+ * from "no rows". (Contrast `model_catalog`, whose policy is
273
+ * `to anon, authenticated`; migration 20270226000000 calls that "the
274
+ * one intentional public read".) urun-infra must add the anon policy
275
+ * or publish prices through the catalog edge function.
276
+ *
277
+ * CONSEQUENCE, STATED PLAINLY: no `pricing` field can appear in production
278
+ * today. This module is correct and inert until gap 2 is fixed.
279
+ *
280
+ * WHAT IS A DEFECT (loud) vs WHAT IS SIMPLY UNPRICED (skipped):
281
+ * - a row whose `unit` is not one of the two per-token units → SKIPPED.
282
+ * A GPU-minute or per-request price is not evidence of a token rate; it
283
+ * is not this surface's business.
284
+ * - a row outside its effective window → SKIPPED. That is what
285
+ * effective-dating means.
286
+ * - a model with only ONE of the two rates → that model stays UNPRICED.
287
+ * Half a price is a lie; HF needs both numbers or neither.
288
+ * - a row that DOES claim a per-token unit but cannot be read (no
289
+ * model_id, unparseable price/date, a missing column, a non-object row,
290
+ * two rows of the same side with the same winning `effective_at` and
291
+ * DIFFERENT prices) → {@link ModelPriceError}. A broken price on a
292
+ * public table is worse than no price, and an order-dependent pick
293
+ * between two same-instant rows would be exactly the silent guess this
294
+ * repo forbids.
295
+ *
296
+ * THE ERROR TAXONOMY MATTERS TO CALLERS, so it is deliberate and narrow:
297
+ * {@link ModelPriceError} means THE DATA IS A DEFECT and no honest price can
298
+ * be derived — the /v1/models lane lets it PROPAGATE (a self-contradictory
299
+ * price table must be seen and fixed, not rounded down to "unpriced"). A
300
+ * plain `Error` from {@link fetchModelPriceRows} means the surface could
301
+ * not be READ (transport/status) — that one the listing may degrade past,
302
+ * because model DISCOVERY must not die with the price oracle. Those are the
303
+ * only two failure modes; nothing here returns an empty list to paper over
304
+ * either.
305
+ *
306
+ * THE SURFACE IS READ IN TWO STEPS, and the split is load-bearing:
307
+ * {@link parseModelPriceRows} validates the wire rows (timeless facts — a
308
+ * caller may CACHE these), and {@link resolveTokenPrices} answers which of
309
+ * them is in force AT AN INSTANT (a per-REQUEST question — caching THAT
310
+ * would serve a rate past its own `expires_at`). See each function's note.
311
+ */
312
+
313
+ /** USD per 1,000,000 tokens — verbatim the HF Inference-Provider shape. */
314
+ interface TokenPricing {
315
+ input: number;
316
+ output: number;
317
+ }
318
+ /** Which half of the pair a recognized unit names. */
319
+ type Side = 'input' | 'output';
320
+ /**
321
+ * ONE VALIDATED per-token price row — the cacheable, TIME-INDEPENDENT half
322
+ * of this surface. Everything here is a fact about the row itself; nothing
323
+ * about WHICH row is in force, because that depends on when you ask (see
324
+ * {@link resolveTokenPrices}). The read/validate step yields these; the
325
+ * time-dependent step consumes them.
326
+ */
327
+ interface ModelPriceRow {
328
+ model_id: string;
329
+ /** null = the model-level rate for every variant without its own row. */
330
+ variant: string | null;
331
+ side: Side;
332
+ /** USD per 1,000,000 tokens, verbatim from `price_usd`. */
333
+ usd: number;
334
+ /** Epoch ms of `effective_at`. */
335
+ effective_at: number;
336
+ /** Epoch ms of `expires_at`, or null for a row that never expires. */
337
+ expires_at: number | null;
338
+ }
339
+
340
+ /**
341
+ * THE OPENROUTER PROVIDER DOCUMENT — `GET /v1/models?format=openrouter`.
342
+ *
343
+ * OpenRouter's provider monitor polls this document (schema 2.4, per
344
+ * openrouter.ai/docs/guides/community/for-providers §1) to list a provider's
345
+ * models on the marketplace. The document answers ONE question: "all models
346
+ * that should be served by OpenRouter" — the platform's SELLABLE surface.
347
+ * That surface is the catalog edge function's SHARED block (the shared
348
+ * endpoints OpenRouter's callers resolve by `model_id:variant` on the chat
349
+ * surface); caller-org apps are deliberately NOT in it. A caller-org app is
350
+ * a tenancy-private chat surface behind one org's key — declaring it here
351
+ * would sell another tenant's private deployment as a public marketplace
352
+ * SKU. The document still rides the tenancy seam (`openRouterModels` on the
353
+ * tenant client), so each caller sees the same sellable listing.
354
+ *
355
+ * THE SHARED JOIN: each endpoint joins the catalog EXACTLY on
356
+ * (model_id, variant) — the SAME join the shared-lane consumers (supportsAudio,
357
+ * modelList's shared arm, imageModelList) already apply. No slug heuristics: the
358
+ * shared org's app slug need not exist in
359
+ * the caller's catalog, and a slug guess would be a second, bespoke join.
360
+ *
361
+ * MODALITY GATING: only rows whose catalog `task` names a chat-completions
362
+ * shape (chat | code | agent | vl) are declared. A voice/video/image app
363
+ * behind a chat surface would stream garbage into OpenRouter's baseline
364
+ * tests; models the catalog cannot vouch for are OMITTED, never guessed.
365
+ * With no shared block (or no catalog oracle) configured the document is
366
+ * `{"data": []}` — honest emptiness, not invented capability.
367
+ *
368
+ * PRICING — schema 2.4 nests `pricing` arrays on the modality that owns
369
+ * them ({type 'prompt'} on the text INPUT modality, {type 'completion'} on
370
+ * the text OUTPUT modality, `cost_usd` a USD string PER TOKEN). The price
371
+ * source is model-prices.ts — the SAME per-token join the OpenAI listing
372
+ * uses ({@link pricingFor}); the per-million rate is shifted to per-token
373
+ * on its decimal string, never by float division. An entry with no
374
+ * per-token row — today ALL of them, because `model_prices` prices GPU
375
+ * minutes (`per_minute` et al.), which is not evidence about tokens —
376
+ * carries NO pricing array at all: the schema's rule is "a modality with
377
+ * no pricing array is simply unpriced". Never a zero, never a GPU-minute
378
+ * rate relabelled as a token rate (ENG-317 owns the human pricing
379
+ * decision that would make such rows exist).
380
+ *
381
+ * The full closed-value-domain schema ships as OpenAPI 3.1 at
382
+ * openrouter.ai/docs/assets/provider-monitor-schema-v2.openapi.json.
383
+ */
384
+
385
+ type OpenRouterProviderDoc = {
386
+ data: OpenRouterProviderModel[];
387
+ };
388
+ interface OpenRouterProviderModel {
389
+ schema_version: '2.4';
390
+ /** The EXACT id OpenRouter sends back as `model` — the app slug. */
391
+ id: string;
392
+ name: string;
393
+ created: number;
394
+ /** Valid enum: int4|int8|fp4|mxfp4|nvfp4|fp6|fp8|mxfp8|fp16|bf16|fp32|null. */
395
+ quantization: string | null;
396
+ description: string;
397
+ hugging_face_id: string;
398
+ input_modalities: Array<Record<string, unknown>>;
399
+ output_modalities: Array<Record<string, unknown>>;
400
+ }
401
+
402
+ /**
403
+ * Per-model → per-app routing for the compat proxy (owner directive
404
+ * 2026-08-07): swapping the model in a coding harness routes the request to
405
+ * the org's DEPLOYED app for that model. v1 is deployed-only — a uRun model
406
+ * that is not deployed gets a loud 404 naming `urun serve <id>`; a later
407
+ * phase (explicitly out of scope here; urun-infra#1490 shared-endpoints)
408
+ * auto-creates from the model catalog on first request.
409
+ *
410
+ * MODEL-ID SURFACE (the documented mapping): a model may be named by
411
+ * - the app slug itself ("qwen3-6-27b-bf16"), or
412
+ * - the catalog id ("qwen3.6-27b"), or
413
+ * - the catalog id:variant ("qwen3.6-27b:bf16"),
414
+ * where slugification mirrors urun-cli `serve.py _default_app_name` exactly:
415
+ * lowercase, every non-alphanumeric-non-dash character becomes "-", leading/
416
+ * trailing dashes stripped (catalog id "qwen3.6-27b" + variant "bf16" → app
417
+ * "qwen3-6-27b-bf16"). COLLISION RULE: an exact slug match always wins over
418
+ * the catalog-id (prefix) interpretation.
419
+ *
420
+ * RESOLUTION ORDER (one canonical path, documented end to end):
421
+ * 1. model absent / "urun" / an alias of the startup app → DEFAULT app.
422
+ * 2. exact slug match on a deployed serve app → that app.
423
+ * 3. catalog-id form matching exactly one deployed app → that app
424
+ * (two or more candidates → loud ambiguity error naming them).
425
+ * 4. the name maps to an org app that is NOT an active app exposing the
426
+ * proxy's serve function → loud 404
427
+ * naming `urun serve <model>` and the available models.
428
+ * 5. the name matches a catalog row but no deployed app → loud 404
429
+ * naming `urun serve <id>` (deployed-only v1).
430
+ * 6. anything else — a model name outside the uRun namespace entirely
431
+ * (e.g. the harness's own upstream default, "claude-*"/"gpt-*") →
432
+ * DEFAULT app. This IS today's single-app contract, kept deliberately
433
+ * so `urun compat <agent>` with the agent's stock model keeps working
434
+ * with zero new env; the per-model /stats table records every such
435
+ * mapping so it is visible, never silent. Models the proxy ADVERTISES
436
+ * on /v1/models can never land here — they resolve (2/3) or fail loud
437
+ * (4/5) above.
438
+ *
439
+ * NO-DEFAULT MODE (`defaultApp: null`) — the HOSTED multi-tenant lane
440
+ * (`src/hosted/`): a shared endpoint serving every org has no "the app this
441
+ * proxy was started for", so rules 1 and 6 have nothing to fall back TO.
442
+ * Rather than inventing one (picking "some" app for a caller would be the
443
+ * worst kind of silent divergence), both rules become the SAME loud
444
+ * {@link UnknownModelError} that rules 4/5 already raise: name a deployed
445
+ * model, here is the list. Rules 2–5 are byte-for-byte the local behavior —
446
+ * one router, one resolution order, two configurations.
447
+ */
448
+
449
+ /** One org app row from `GET {orgApi}/apps` (urun-cli `ApiClient.list_apps`). */
450
+ interface DeployedApp {
451
+ app_slug: string;
452
+ function_name?: string | null;
453
+ deployment_status?: string | null;
454
+ [k: string]: unknown;
455
+ }
456
+ /**
457
+ * A model that does not resolve to a deployed app the proxy may serve.
458
+ * Rendered by the server in each lane's NATIVE error format as a 404 — never
459
+ * silently served by the default app.
460
+ */
461
+ declare class UnknownModelError extends Error {
462
+ }
463
+ /**
464
+ * A session handle names a pooled session that no longer exists (closed,
465
+ * evicted after its pod died, replaced by a re-home, or the proxy restarted).
466
+ * The caller asked to REATTACH that exact session — opening a fresh one and
467
+ * calling it "resumed" would be a silent lie, so this is always loud.
468
+ */
469
+ declare class SessionGoneError extends Error {
470
+ }
471
+ /**
472
+ * OpenAI-shaped model list (same shape as models.ts listModels). When the
473
+ * catalog oracle is configured, each entry ALSO carries the catalog
474
+ * enrichment fields (additive JSON — OpenAI clients ignore unknown fields;
475
+ * the OpenRouter + Vercel AI Gateway provider listings need them). Entries
476
+ * whose slug matches no catalog row stay at the bare four fields.
477
+ */
478
+ interface RouterModelEntry {
479
+ id: string;
480
+ object: 'model';
481
+ created: number;
482
+ owned_by: string;
483
+ /** Display name — the canonical `<model_id>:<variant>` catalog ref. */
484
+ name?: string;
485
+ /** User-facing description from the catalog (console Endpoints copy). */
486
+ description?: string;
487
+ /** Catalog modality (chat | code | agent | vl | audio | image | ...). */
488
+ task?: string;
489
+ /** Serving engine (vllm | sglang | llamacpp | ...). */
490
+ engine?: string;
491
+ /** The catalog lane this app deploys, e.g. 'rtx6000:1'. */
492
+ gpu_spec?: string;
493
+ /** Context window in tokens (chat rows carry it in engine_args). */
494
+ context_length?: number;
495
+ /**
496
+ * Per-token price, USD per MILLION tokens — the exact shape Hugging Face
497
+ * reads off `GET /v1/models` for its provider comparison table and its
498
+ * `:fastest`/`:cheapest` routing. ABSENT means UNPRICED and is the honest
499
+ * answer for every model with no per-token price row (model-prices.ts);
500
+ * a zero or an estimate here would be a silent lie on a public table.
501
+ */
502
+ pricing?: TokenPricing;
503
+ /** Idle-floor replica count for a shared-lane model (U6 reconciler). */
504
+ warm?: number;
505
+ }
506
+ interface RouterModelList {
507
+ object: 'list';
508
+ data: RouterModelEntry[];
509
+ }
510
+ /** One `/v1/images/models` entry: the OpenAI model object plus the image
511
+ * modes the catalog rows vouch for (the SAME per-placement consensus
512
+ * {@link ModelRouter.supportsImages} applies — a row set that vouches
513
+ * nothing is NOT listed). */
514
+ interface RouterImageModelEntry extends RouterModelEntry {
515
+ image_modes: ImageCapabilityMode[];
516
+ }
517
+ interface RouterImageModelList {
518
+ object: 'list';
519
+ data: RouterImageModelEntry[];
520
+ }
521
+ /**
522
+ * The dial target one resolved model names: the app slug plus — for a SHARED
523
+ * lane (the catalog `shared` block, the cross-org session dial) — the exact
524
+ * serve function on the shared org's app and the flag that switches the
525
+ * control plane's shared-admission carve-out on. A plain caller-org app
526
+ * carries only `appSlug`; the router's own `fnName` applies there.
527
+ */
528
+ interface RouteTarget {
529
+ appSlug: string;
530
+ /** The serve function on the TARGET app (a shared lane pins it per row). */
531
+ fnName?: string;
532
+ /**
533
+ * CROSS-ORG SHARED DIAL: the backhaul mints the client token with
534
+ * `sharedApp: true` and starts the session with `shared_app: true`, so the
535
+ * control plane resolves the app from the platform's shared org while the
536
+ * session and its usage stay attributed to the CALLER org.
537
+ */
538
+ sharedApp?: boolean;
539
+ /** Shared lanes only: the catalog id the request pinned (capability joins). */
540
+ modelId?: string;
541
+ /** Shared lanes only: the catalog variant the request pinned. */
542
+ variant?: string;
543
+ }
544
+ interface ModelRouterOptions<S> {
545
+ /**
546
+ * The startup app slug (URUN_APP) — the DEFAULT model — or `null` for the
547
+ * hosted multi-tenant lane, which has no per-proxy default app: there,
548
+ * every request must NAME a deployed model and an unnamed/unknown one
549
+ * fails loud instead of silently landing somewhere (see the module header,
550
+ * "NO-DEFAULT MODE").
551
+ */
552
+ defaultApp: string | null;
553
+ /** The serve function name every routed app must expose (URUN_FUNCTION). */
554
+ fnName: string;
555
+ /** Open a backhaul session for a resolved dial target (called at most once per pool key). */
556
+ openSession: (target: RouteTarget) => S | Promise<S>;
557
+ /** Terminal release for one pool entry (Session.end() underneath). */
558
+ closeSession: (entry: S) => Promise<void>;
559
+ /**
560
+ * List the org's deployed apps, or null when the credentials cannot
561
+ * (URUN_JWT lane: the pre-vended token is scoped to the default app, so
562
+ * there is no org listing AND no cross-app session — routing degrades to
563
+ * the default-app-only contract, which is exactly today's behavior).
564
+ */
565
+ listApps: (() => Promise<DeployedApp[]>) | null;
566
+ /**
567
+ * Catalog rows (models.ts fetchCatalogRows) as the uRun-namespace oracle
568
+ * for rule 5 AND the /v1/models enrichment + OpenRouter provider-doc
569
+ * source, or null when catalog access is not configured. Optional fields
570
+ * beyond model_id/variant are tolerated (thin rows still typecheck).
571
+ */
572
+ listCatalog: (() => Promise<CatalogRow[]>) | null;
573
+ /**
574
+ * The VALIDATED per-token price rows (model-prices.ts
575
+ * `fetchModelPriceRows`) behind the `pricing` field of the /v1/models
576
+ * listing, or null when no price source is configured. A REQUIRED key
577
+ * exactly like {@link listCatalog}: an optional one would let a lane ship
578
+ * a silently unpriced public listing without ever saying so.
579
+ *
580
+ * These are ROWS, not resolved prices, precisely so the router may cache
581
+ * them: which row is in force is a per-REQUEST question answered by
582
+ * `resolveTokenPrices` against the exact `effective_at` / `expires_at`
583
+ * boundaries. Handing back already-resolved prices would let a cached
584
+ * answer outlive the window it was resolved in.
585
+ */
586
+ listPrices: (() => Promise<ModelPriceRow[]>) | null;
587
+ /**
588
+ * Where this router announces a DEGRADATION — an unreadable price
589
+ * surface, a model refused a price, a catalog blip that suppresses
590
+ * pricing. Optional only in WHERE it goes: left unset the router writes
591
+ * to `console.warn`, never to nothing (see {@link ModelRouter.warn}).
592
+ * Mirrors the `warn` sink `catalogFromEnv` (cli.ts) already takes.
593
+ */
594
+ warn?: (line: string) => void;
595
+ /** Deployed-apps cache TTL (the list changes on deploys, not per request). */
596
+ appsTtlMs?: number;
597
+ /**
598
+ * The stable NATIVE identity of one pooled entry (the uRun session id in
599
+ * the proxy wiring — the same identity the serve-side session-affinity tag
600
+ * rides, urun-python#1556). Powers the session-identity seam
601
+ * ({@link ModelRouter.handleFor} / {@link ModelRouter.sessionForHandle});
602
+ * a router without it fails LOUD on those calls, never approximates.
603
+ */
604
+ sessionKey?: (entry: S) => string;
605
+ /**
606
+ * The catalog edge function's `shared` block (shared-endpoints U8), or
607
+ * null when shared-model routing is not configured — every phase-1 caller
608
+ * leaves it absent and behaves byte-for-byte as before. Same TTL cache
609
+ * discipline as the apps/catalog oracles.
610
+ */
611
+ shared?: (() => Promise<SharedCatalogBlock | null>) | null;
612
+ }
613
+ /**
614
+ * The image-capability modes the Images lane gates on (images.ts's
615
+ * ImageCapabilityMode — declared here because the router is the LOWER layer:
616
+ * the lane's gate type stays structurally satisfied by
617
+ * {@link ModelRouter.supportsImages} with no router→lane import).
618
+ */
619
+ type ImageCapabilityMode = 'generation' | 'edit';
620
+ /**
621
+ * The session pool: one backhaul session per deployed app, keyed by app slug,
622
+ * opened lazily on the first request that routes to it and reused for every
623
+ * subsequent one. The startup app is seeded eagerly by the CLI. Sessions
624
+ * close on proxy shutdown via {@link closeAll}; there is NO idle-close policy
625
+ * (deliberate v1 simplification — noted as a follow-up in the PR).
626
+ */
627
+ declare class ModelRouter<S> {
628
+ private readonly opts;
629
+ private readonly pool;
630
+ private appsCache;
631
+ private rowsCache;
632
+ constructor(opts: ModelRouterOptions<S>);
633
+ /** Seed an already-open session (the CLI's eagerly-opened startup app). */
634
+ seed(appSlug: string, entry: S): void;
635
+ private deployedApps;
636
+ /**
637
+ * The shared block with the same TTL discipline as {@link deployedApps}.
638
+ * Null when not configured. Malformed blocks throw loudly
639
+ * (SharedCatalogError) on first use — never a silently half-empty table.
640
+ */
641
+ private sharedBlock;
642
+ private sharedCache;
643
+ /**
644
+ * Catalog rows with the same TTL discipline as {@link deployedApps} (the
645
+ * catalog changes on reseed migrations, not per request). Null when no
646
+ * oracle is configured — callers degrade to the bare four-field entries.
647
+ */
648
+ private catalogRows;
649
+ private pricesCache;
650
+ /**
651
+ * The per-token price ROWS with the same TTL discipline as
652
+ * {@link catalogRows} (the price book changes on ops writes, not per
653
+ * request). No source configured ⇒ [] — every model unpriced, which is
654
+ * the honest listing until per-token rows exist (ENG-317).
655
+ *
656
+ * ROWS, never resolved prices: the rows are timeless facts, so caching
657
+ * them is safe, whereas caching the prices IN FORCE would keep serving a
658
+ * rate past its own `expires_at` — or withhold one past its
659
+ * `effective_at` — for the rest of the TTL window. The in-force question
660
+ * is answered per request in {@link pricesForListing}.
661
+ */
662
+ private priceRows;
663
+ /**
664
+ * The price surface for the /v1/models listing. Three distinct outcomes,
665
+ * and NONE of them is silent:
666
+ *
667
+ * - the surface cannot be READ (transport, status, or the read's own
668
+ * deadline — a plain Error) ⇒ every model lists UNPRICED and the
669
+ * reason is WARNED. Model discovery must not die with the price
670
+ * oracle, but an outage rendering as an ordinary unpriced listing —
671
+ * indistinguishable from today's inert state — is exactly the silent
672
+ * degradation CLAUDE.md forbids, so it says so out loud.
673
+ * - one model's rows are a DEFECT (a duplicate rate for one instant, a
674
+ * zero rate) ⇒ that MODEL is unpriced and the refusal is WARNED. The
675
+ * blast radius is the model the bad row belongs to; it used to be the
676
+ * whole listing for every tenant.
677
+ * - the READ itself is not the surface we contracted for (a malformed
678
+ * row, a missing column, an unknown `unit` — {@link ModelPriceError})
679
+ * ⇒ PROPAGATES. Nothing in a payload that violates its own column
680
+ * contract can be trusted, so half a price table is not served as if
681
+ * complete — the same posture `parseSharedBlock` takes on a malformed
682
+ * routing block.
683
+ *
684
+ * The resolve step sits OUTSIDE the catch deliberately: it is our own
685
+ * code, so a programming error in it must surface as a crash, not become
686
+ * an empty price list.
687
+ */
688
+ private pricesForListing;
689
+ /**
690
+ * Where a degradation says so. Defaults to `console.warn` rather than to
691
+ * nothing: a caller may ROUTE the signal (the CLI sends it to stderr, as
692
+ * `catalogFromEnv` already does), but no caller can switch it off, because
693
+ * an unannounced degradation is the failure mode this repo treats as a
694
+ * time bomb.
695
+ */
696
+ private warn;
697
+ /** Apps this proxy may serve: active AND exposing the serve function. */
698
+ private servable;
699
+ private availableIds;
700
+ /**
701
+ * NO-DEFAULT MODE's terminal for rules 1 and 6: there is no app to fall
702
+ * back to, so say so loudly and list what the CALLER'S org actually has.
703
+ * Never returns.
704
+ */
705
+ private noDefaultApp;
706
+ /**
707
+ * Resolve a request's `model` to its dial target — the documented
708
+ * resolution order from the module header. Throws
709
+ * {@link UnknownModelError} for a uRun model that is not deployed (rules
710
+ * 4/5). A SHARED-lane hit (the catalog `shared` block) resolves to the
711
+ * SHARED org's app with `sharedApp: true` — the cross-org session dial.
712
+ */
713
+ private resolveTarget;
714
+ /**
715
+ * THE BILLING IDENTITY of a request's model: the CATALOG `(model_id,
716
+ * variant)` the request actually resolved to, or `null` when the route
717
+ * names no catalog model.
718
+ *
719
+ * WHY THE LEDGER CANNOT USE THE CALLER'S STRING INSTEAD. The per-request
720
+ * ledger prices on `(model_id, variant)` (urun-infra
721
+ * `urun_record_inference_request`), and the caller's `model` is not that:
722
+ * a BARE catalog id resolves through {@link resolveSharedModel} to the
723
+ * model's `compat_default` VARIANT, so recording the raw string would file
724
+ * the request under a variant nobody served and price it off the
725
+ * model-level wildcard — a rate that exists precisely to price the variants
726
+ * nobody named explicitly. Splitting the string on its last `:` would be a
727
+ * heuristic on a money path. This is the resolution the dial itself used.
728
+ *
729
+ * NULL FOR A CALLER-ORG APP, and that is a REFUSAL rather than a gap. An
730
+ * org's own deployed app has no catalog identity at all — its identity is
731
+ * an app slug, which is a different namespace from `model_prices.model_id`
732
+ * — so there is no honest `(model_id, variant)` to record. The ledger
733
+ * writer declines to write such a request rather than inventing one; see
734
+ * `proxy/ledger.ts`.
735
+ *
736
+ * CHEAP TO CALL: it rides {@link resolveTarget}, whose app listing, catalog
737
+ * and shared block are all cached on this router, so the ledger writer can
738
+ * ask AFTER the response has closed without re-hitting the control plane.
739
+ */
740
+ catalogIdentityFor(model: string | undefined): Promise<{
741
+ model_id: string;
742
+ variant: string;
743
+ } | null>;
744
+ /**
745
+ * THE CALLER-ORG APP LANE's OWN IDENTITY (ENG-412) — the app slug this
746
+ * request actually resolves to.
747
+ *
748
+ * The ledger records it for the lane {@link catalogIdentityFor} answers null
749
+ * for, and the caller's RAW `model` string is not it: routing trims, caps and
750
+ * matches that string before selecting an app, so two spellings of one app
751
+ * would otherwise be filed as two identities and no app-level reconciliation
752
+ * could undo it afterwards.
753
+ *
754
+ * Rides {@link resolveTarget}'s cached listings and OPENS NO SESSION.
755
+ */
756
+ servingAppSlugFor(model: string | undefined): Promise<string>;
757
+ /**
758
+ * Resolve a request's `model` to its POOL KEY — the app slug for a
759
+ * caller-org app, `shared:<slug>` for a shared-lane dial (the two never
760
+ * share a pool slot, {@link poolKeyOf}).
761
+ */
762
+ resolveApp(model: string | undefined): Promise<string>;
763
+ /**
764
+ * The pooled session for a model — opened lazily, reused afterwards.
765
+ *
766
+ * IT ALSO REPORTS THE DIAL'S BILLING IDENTITY (ENG-414), off the SAME
767
+ * `resolveTarget` answer it just selected the session with. That is the
768
+ * whole point of returning it from here: the ledger must be able to name
769
+ * what served a request WITHOUT taking a second resolution. This router
770
+ * reads its app list, catalog and shared block through caches with a TTL, so
771
+ * a second resolution taken microseconds later can answer differently — and
772
+ * then the row names a variant nobody served. `rehome.ts` reports this
773
+ * identity once per dial, so a request that re-homes is billed on the
774
+ * REPLACEMENT's identity rather than the abandoned dial's.
775
+ *
776
+ * THE POOL ENTRY COULD NOT CARRY IT. `S` is the generic pooled session and
777
+ * knows nothing about models; the identity is a property of the ROUTE, which
778
+ * is what this method resolved and the entry never saw.
779
+ */
780
+ sessionFor(model: string | undefined): Promise<{
781
+ app: string;
782
+ entry: S;
783
+ identity: TurnCatalogIdentity;
784
+ }>;
785
+ private keyOf;
786
+ /**
787
+ * SESSION-IDENTITY SEAM (a): the opaque stable handle for the pooled
788
+ * session currently serving `model`'s turns. Rides the SAME acquisition
789
+ * path as every request ({@link sessionFor}) — the session opens lazily if
790
+ * this model has none yet — and derives the handle from native identity
791
+ * (app slug + uRun session id), zero bespoke bookkeeping.
792
+ */
793
+ handleFor(model: string | undefined): Promise<{
794
+ app: string;
795
+ handle: string;
796
+ }>;
797
+ /**
798
+ * SESSION-IDENTITY SEAM (b): the exact pooled session a handle names.
799
+ * NEVER opens a fresh session — a handle whose session is gone (closed,
800
+ * evicted, re-homed to a replacement, proxy restarted) or malformed throws
801
+ * {@link SessionGoneError} loudly. Resume is reattach-or-fail, not
802
+ * reattach-or-quietly-restart.
803
+ */
804
+ sessionForHandle(handle: string): Promise<{
805
+ app: string;
806
+ entry: S;
807
+ }>;
808
+ /**
809
+ * Drop ONE pooled session whose backhaul died (its pod was restarted /
810
+ * drained / deleted) and release it — the next {@link sessionFor} opens a
811
+ * fresh one, i.e. asks the control plane for a new assignment. Used by the
812
+ * one-shot re-home (rehome.ts, urun-sh/urun-python#1592).
813
+ *
814
+ * IDENTITY-GUARDED (the same rule the pi lane's SessionPool follows): a
815
+ * concurrent request that already re-homed this app has put a NEWER entry
816
+ * under the key, and evicting that would close a healthy session out from
817
+ * under it.
818
+ */
819
+ evict(app: string, entry: S): Promise<void>;
820
+ /**
821
+ * `GET /v1/models`: the org's deployed serve apps as model entries, the
822
+ * default app FIRST. On the JWT lane (no org listing) this is the default
823
+ * app plus any app already in the pool — the gap is called out loudly in
824
+ * the PR, not papered over here.
825
+ *
826
+ * THE HUGGING FACE FIELDS: every entry carries `context_length` (from the
827
+ * catalog's engine_args, omitted when unknown or when the placements
828
+ * disagree) and `pricing` (USD per MILLION input/output tokens, omitted
829
+ * ENTIRELY when the model has no per-token price row — see
830
+ * model-prices.ts; ENG-317 owns the pricing decision that makes such rows
831
+ * exist). HF reads both to build its public provider comparison table.
832
+ * A price surface that cannot be READ lists everything unpriced; a price
833
+ * surface that is a DEFECT fails this call loudly (see below).
834
+ */
835
+ modelList(): Promise<RouterModelList>;
836
+ /**
837
+ * The OpenRouter PROVIDER document (`GET /v1/models?format=openrouter`):
838
+ * schema 2.4 per openrouter.ai/docs/guides/community/for-providers §1 —
839
+ * "an endpoint that returns all models that should be served by
840
+ * OpenRouter". That is the SHARED block's sellable endpoints (the models
841
+ * callers resolve by `model_id:variant` on the chat surface), joined to
842
+ * the catalog EXACTLY on (model_id, variant) — the SAME join
843
+ * {@link supportsAudio} applies — and gated to chat-completions shape
844
+ * (task chat | code | agent | vl; a voice or video app behind a chat
845
+ * surface would stream garbage). Caller-org apps are deliberately NOT
846
+ * listed: they are tenancy-private chat surfaces, and a provider
847
+ * document that published them would sell another tenant's private
848
+ * deployment as a public marketplace SKU. Pricing follows
849
+ * {@link pricesForListing}'s posture (an unreadable surface prices
850
+ * nothing and WARNS — discovery outlives the price oracle); a model with
851
+ * no per-token row carries NO pricing array (the schema rule: "a
852
+ * modality with no pricing array is simply unpriced").
853
+ *
854
+ * A CONFIGURED catalog read that FAILS propagates — the provider
855
+ * document cannot vouch a sellable model from nothing, and a half-empty
856
+ * one must never be served as if complete.
857
+ */
858
+ openRouterModels(): Promise<OpenRouterProviderDoc>;
859
+ /**
860
+ * IMAGE CAPABILITY GATE (the Images lane's `imageModels` oracle, images.ts):
861
+ * is `model`'s deployed app a native destination for `mode`? Authorized
862
+ * resolution ({@link resolveApp} — undeployed uRun models throw
863
+ * {@link UnknownModelError}, which the lane renders as a 404) then the
864
+ * SAME catalog join as the /v1/models enrichment (PR421's rowForSlug):
865
+ * exact `<model_id>-<variant>` slug across ALL its GPU-placement rows, a
866
+ * bare model_id only when one variant owns it.
867
+ *
868
+ * Capability is read ONLY from catalog modality metadata, never from the
869
+ * serve function name: every row must carry task 'image' INVARIANTLY across
870
+ * the placements, and the mode comes from engine_args.expects_image —
871
+ * false ⇒ pure text-to-image ('generation'; diffusers v0.38.0
872
+ * pipeline_qwenimage_edit_plus raises torch.cat on an EMPTY image list, so
873
+ * an edit destination is never generation-capable); true or ABSENT (the
874
+ * native default) ⇒ 'edit'. The consensus rule is per-placement:
875
+ * capability must agree across every placement — a CONFLICTING flag (mixed
876
+ * true/false/absent) or a non-boolean one is a refused capability, never
877
+ * silently defaulted to edit.
878
+ *
879
+ * A CONFIGURED catalog oracle that FAILS propagates its rejection — the
880
+ * gate never fabricates a `false` (which would render as a misleading 404)
881
+ * out of an infrastructure outage. On the CALLER-ORG lane, no oracle
882
+ * configured, no matching row, or a slug the catalog cannot vouch for ⇒
883
+ * false (loud 404 upstream). On the SHARED lane the same questions are
884
+ * LOUDER: a shared endpoint is published capability, so an unresolvable
885
+ * row or an unvouchable mode throws {@link SharedCatalogError} (the 500
886
+ * path) instead of a 404 that would misattribute a catalog defect to the
887
+ * caller's request.
888
+ */
889
+ supportsImages(model: string | undefined, mode: ImageCapabilityMode): Promise<boolean>;
890
+ /**
891
+ * AUDIO MODALITY GATE (the OpenAI Realtime lane's modality oracle): does
892
+ * `model`'s deployed app carry catalog `task` stt or tts — INVARIANTLY
893
+ * across every GPU-placement row ({@link audioModalityForRows})? The SAME
894
+ * shape as {@link supportsImages}: authorized resolution
895
+ * ({@link resolveTarget} — undeployed uRun models throw
896
+ * {@link UnknownModelError}), then the SAME catalog join
897
+ * ({@link catalogRowsForSlug}): exact `<model_id>-<variant>` slug across
898
+ * ALL its placements, a bare model_id only when one variant owns it.
899
+ *
900
+ * A SHARED-lane hit joins on the lane's catalog id + variant (the SAME
901
+ * join {@link imageModelList} uses for the shared block) — the shared org's
902
+ * app slug need not exist in the caller's own catalog by slug.
903
+ *
904
+ * A CONFIGURED catalog oracle that FAILS propagates its rejection — the
905
+ * gate never fabricates a `false` out of an infrastructure outage. No
906
+ * oracle configured, no matching row, or a task the catalog cannot vouch
907
+ * for ⇒ false — the realtime resolver renders that as the loud 404.
908
+ */
909
+ supportsAudio(model: string | undefined): Promise<boolean>;
910
+ /**
911
+ * `GET /v1/images/models`: the IMAGE-CAPABLE slice of the model surface —
912
+ * the caller-org deployed apps whose catalog rows vouch an image mode
913
+ * ({@link imageModesForRows}, the same consensus {@link supportsImages}
914
+ * applies), plus the shared block's endpoints joined against each
915
+ * endpoint's OWN catalog row (exact model_id + variant, EXPLICIT boolean
916
+ * expects_image — {@link sharedImageModes}). A configured catalog oracle
917
+ * that FAILS propagates, and so does a shared endpoint the catalog cannot
918
+ * vouch (unresolvable row / unvouchable mode → {@link SharedCatalogError}):
919
+ * an image listing cannot vouch capability from nothing or advertise a
920
+ * silently defaulted mode, so — unlike {@link modelList} — there is NO
921
+ * bare-entries degradation here.
922
+ */
923
+ imageModelList(): Promise<RouterImageModelList>;
924
+ /** Close every pooled session (Session.end() underneath) — proxy shutdown. */
925
+ closeAll(): Promise<void>;
926
+ }
927
+
928
+ /**
929
+ * THE CANONICAL USAGE-QUERY LANE — `POST /v1/usage/requests`.
930
+ *
931
+ * WHAT IT IS. A batch lookup over uRun's per-request inference ledger
932
+ * (`public.inference_requests`, urun-infra ENG-316): "given these inference
933
+ * ids, what do we hold for each?" — the cost and the usage quantities, scoped
934
+ * to the calling key's org. OpenAI defines no shape for this, so the shape is
935
+ * uRun's own: the ledger's own column vocabulary, and the same id space as the
936
+ * `Inference-Id` response header every `/v1` answer already carries
937
+ * (inference-id.ts).
938
+ *
939
+ * WHY IT IS CANONICAL AND NOT HUGGING-FACE-SHAPED. The owner has ruled that
940
+ * distributor-specific functionality must not land on the canonical surface.
941
+ * HF's billing poll — `{"requestIds":[...]}` answered with
942
+ * `{"requests":[{"requestId","costNanoUsd"}]}` — is HF-proprietary. It gets a
943
+ * THIN TRANSLATOR over this lane (partners/huggingface.ts, the ENG-376 adapter
944
+ * pattern): field renames and nothing else. Every other distributor's billing
945
+ * adapter reads this same lane, and so can the console and support.
946
+ *
947
+ * WHERE THE DATA COMES FROM. This process holds no database credential — by
948
+ * design (hosted/auth.ts: "NO STANDING CREDENTIAL"). The lookup rides the
949
+ * CALLER'S OWN key to the control plane's `inference-usage` edge function,
950
+ * which derives the org from that key server-side and runs the org-scoped
951
+ * `urun_inference_usage_lookup` RPC. So cross-org isolation here is the same
952
+ * structural property every other lane has: this process never holds a
953
+ * credential that spans orgs, and the org is never a request field.
954
+ *
955
+ * THREE CONTRACTS THAT ARE NOT NEGOTIABLE, because each one is money:
956
+ *
957
+ * 1. `cost_nano_usd: null` MEANS NOT YET PRICED — NEVER FREE. Every row in
958
+ * production carries NULL today: ENG-317's price machinery has landed but
959
+ * seeds no price, and nothing stamps the ledger yet. This lane reports
960
+ * the NULL verbatim. A caller that renders it as 0 is inventing a price
961
+ * nobody set; a caller billing a third party from it must simply not
962
+ * answer for that id yet.
963
+ * 2. AN ID WE HOLD NOTHING FOR IS ABSENT FROM THE ANSWER, and an id
964
+ * belonging to another org is indistinguishable from one that never
965
+ * existed. That is deliberate: anything else makes this surface a
966
+ * cross-org existence oracle. A row this lane could not VALIDATE is
967
+ * absent for the same reason and is therefore indistinguishable from
968
+ * those two — which is precisely why it is reported (ENG-416), so that
969
+ * the one absence we caused ourselves is visible on our side.
970
+ * 3. IDS THE CALLER DID NOT ASK ABOUT ARE NEVER RETURNED. The id list is
971
+ * the query; the control plane bounds the answer by it, and
972
+ * {@link assertRequestedOnly} re-checks it here rather than trusting the
973
+ * upstream to have done so.
974
+ *
975
+ * WHAT THIS LANE DOES NOT DO: it does not filter on `http_status` (only
976
+ * 2xx/3xx being billable is an HF rule, and support must be able to ask about
977
+ * a failed request), and it never returns `gpu_seconds` — a column ENG-417
978
+ * DROPPED from the ledger entirely, because a per-request GPU-seconds figure
979
+ * double-counts against `usage_events.metered_gpu_seconds` (which meters the
980
+ * instance once) and, under continuous batching, can never be reconciled to
981
+ * it. The guard below is kept and is STRONGER for the removal: it now asserts
982
+ * that a column which does not exist has not come back.
983
+ *
984
+ * WHAT IT DOES RETURN THAT IS EASY TO MISS: `identity_kind`. `model_id` speaks
985
+ * one of two unrelated namespaces — a catalog identity, or the caller's own
986
+ * deployed app slug — and the two can COLLIDE as strings, so the identity is
987
+ * never reported without the namespace that makes it readable.
988
+ *
989
+ * AND IT RESOLVES NO PRICE — stated because the surrounding system is about to
990
+ * grow several, and someone will come looking here. Nothing in this lane or in
991
+ * its Hugging Face translator reads `model_prices`, picks a rate, or knows what
992
+ * a model costs. It reports the `cost_nano_usd` a writer already stamped on the
993
+ * row, at REQUEST grain, so two requests for the same model may carry different
994
+ * costs and different `price_version`s and this code is indifferent to why.
995
+ *
996
+ * That means the one-price-per-model assumption is NOT baked in here. It lives
997
+ * one layer down, in `urun_active_model_price`, which resolves on
998
+ * `(model_id, variant)` and separates the session lane from the HF lane BY UNIT
999
+ * — fine while unit and audience correlate, and filed as ENG-404 for when they
1000
+ * stop (per-minute credit tiers put two audiences on one unit). When that is
1001
+ * fixed, the audience rides `price_version`, which this lane already carries
1002
+ * through per row and never interprets. No change is required here.
1003
+ */
1004
+
1005
+ /**
1006
+ * ONE ledger row as this lane answers it — the control-plane RPC's column
1007
+ * list, in the ledger's own vocabulary. `gpu_seconds` and `org_id` are absent
1008
+ * by construction, not by omission here (see the module header).
1009
+ */
1010
+ interface InferenceUsageRecord {
1011
+ /** The uuid this request's `Inference-Id` response header carried. */
1012
+ inference_id: string;
1013
+ /** The identity that served it, IN THE NAMESPACE {@link identity_kind} names. */
1014
+ model_id: string;
1015
+ /**
1016
+ * WHICH NAMESPACE {@link model_id} SPEAKS: `catalog_model` (the
1017
+ * `model_prices` vocabulary — everything a distributor routes) or
1018
+ * `caller_org_app` (the caller's own deployed app, which has no catalog
1019
+ * identity and is therefore PERMANENTLY unpriceable — never merely "not
1020
+ * priced yet"). Reported as a plain string rather than a union, because a
1021
+ * value this lane does not recognise must reach the caller as data rather
1022
+ * than fail the batch: an unknown lane is a forward-compatible control plane,
1023
+ * not a corrupt row.
1024
+ */
1025
+ identity_kind: string;
1026
+ /** null = the model-level row (the `model_prices` key semantics). */
1027
+ variant: string | null;
1028
+ task: string;
1029
+ http_status: number;
1030
+ started_at: string;
1031
+ completed_at: string;
1032
+ /** null = this modality has no such quantity (not "zero of it"). */
1033
+ prompt_tokens: number | null;
1034
+ completion_tokens: number | null;
1035
+ output_units: number | null;
1036
+ /** NULL = NOT YET PRICED (ENG-317). Never "free". */
1037
+ cost_nano_usd: number | null;
1038
+ price_version: string | null;
1039
+ priced_at: string | null;
1040
+ }
1041
+ /**
1042
+ * ONE row this lane REFUSED TO ANSWER ABOUT, handed to the reporter so that a
1043
+ * dropped row is a visible event rather than a silent absence.
1044
+ *
1045
+ * It carries no row content — `where` is a position and `reason` is an
1046
+ * already-redacted {@link UsageSurfaceError} message (names and types, never
1047
+ * values). See {@link describeKeys}.
1048
+ */
1049
+ interface UsageRowRejection {
1050
+ /** `usage.requests[N]` — WHICH row, never the row. */
1051
+ where: string;
1052
+ /** Why it could not be validated. */
1053
+ reason: string;
1054
+ }
1055
+ /**
1056
+ * A sink for {@link UsageRowRejection}s.
1057
+ *
1058
+ * `Promise<void>` IS PART OF THE TYPE BECAUSE IT WAS PART OF THE TYPE ANYWAY.
1059
+ * An `async` function is assignable to a `=> void` signature, so declaring this
1060
+ * synchronous never prevented an async sink — it only hid one, by putting its
1061
+ * failure on a later tick where the containment could not see it. Saying so in
1062
+ * the type is what lets {@link reportRejectedRow} actually handle it.
1063
+ */
1064
+ type UsageRowReporter = (rejection: UsageRowRejection) => void | Promise<void>;
1065
+ /**
1066
+ * The batch-lookup seam. The hosted endpoint supplies one backed by the
1067
+ * control plane's `inference-usage` function; an embedder that supplies none
1068
+ * gets a LOUD 501 on the lane rather than a silently empty answer — the same
1069
+ * posture the image lanes take for `supportsImages`.
1070
+ *
1071
+ * IT IS TWO-PHASE, and the split is the admission boundary rather than
1072
+ * decoration: {@link authorize} runs BEFORE the request body is read, so a
1073
+ * request with no credential is refused without this process allocating or
1074
+ * parsing anything the caller sent.
1075
+ */
1076
+ interface UsageLane {
1077
+ /**
1078
+ * Admission. Returns the caller's credential, or throws (a 401 through the
1079
+ * handler's `statusOf`) when it is missing or malformed. Runs before the
1080
+ * body is touched, and must NOT make a network call — the surface that owns
1081
+ * the ledger authenticates the credential itself when {@link lookup} uses
1082
+ * it, and a second verifier here would be a second source of truth about
1083
+ * the same key.
1084
+ */
1085
+ authorize(req: IncomingMessage): string;
1086
+ /** The batch read, as the holder of the credential `authorize` returned. */
1087
+ lookup(credential: string, inferenceIds: readonly string[]): Promise<InferenceUsageRecord[]>;
1088
+ }
1089
+
1090
+ /**
1091
+ * THE PER-REQUEST LEDGER WRITE — one served `/v1` inference becomes one
1092
+ * durable, already-priced row in `public.inference_requests`.
1093
+ *
1094
+ * WHY THIS EXISTS. uRun answers Hugging Face's billing poll out of a ledger
1095
+ * that, until this module, NOTHING HAD EVER WRITTEN. ENG-316 landed the table
1096
+ * with no writer, ENG-317 landed the price machinery with no stamping, and
1097
+ * ENG-318 landed the reader over both and stated the consequence: with every
1098
+ * cost NULL, a poll answers `{"requests": null}` for every batch and every
1099
+ * request is written off unbilled ~30 minutes later. This module is the hop
1100
+ * from the correlation record the proxy already emits to a row.
1101
+ *
1102
+ * THE THREE RULES THIS MODULE EXISTS TO KEEP (owner ruling, ENG-409):
1103
+ *
1104
+ * 1. THE WRITE IS OFF THE RESPONSE PATH, and the price is stamped in that
1105
+ * same write. It runs from the response's `'close'` hook — after the
1106
+ * customer has their answer — so a ledger round trip can never be in the
1107
+ * latency of an inference. It is not a later sweep either: a re-price
1108
+ * landing after HF has been answered disagrees with a bill already
1109
+ * issued (ENG-406), so the row is born priced or born unpriced.
1110
+ *
1111
+ * 2. A LEDGER WRITE MUST NEVER FAIL AN INFERENCE REQUEST. Everything here
1112
+ * is fire-and-forget and cannot throw into the request path: the answer
1113
+ * is already delivered, the socket is already closing, and a failure is
1114
+ * LOUD IN THE LOG and nowhere else. {@link recordInference} returns
1115
+ * `void` for exactly that reason — there is no promise a caller could
1116
+ * accidentally await, and nothing to reject. The queue and the retries
1117
+ * added by ENG-415 change nothing about this: they all happen behind that
1118
+ * same `void`, after the customer has been answered.
1119
+ *
1120
+ * 3. A REQUEST THAT CANNOT BE PRICED IS WRITTEN UNPRICED, NEVER AT ZERO AND
1121
+ * NEVER DROPPED. Pricing itself happens in the database
1122
+ * (`urun_record_inference_request`), which is where `model_prices` is
1123
+ * readable and where a `price_version` can name the rate rows that
1124
+ * produced a number. This module supplies the ONE fact the database
1125
+ * cannot see — {@link billableOutcome} — and nothing else about money.
1126
+ *
1127
+ * WHAT IS DELIBERATELY NOT HERE:
1128
+ * * NO PRICES, NO RATES, NO MODEL NAMES, NO ARITHMETIC. Grep it: there is
1129
+ * no rate and no model id below. The price book is data (ENG-317) and a
1130
+ * rate change must take effect without a deploy.
1131
+ * * NO PARTNER VOCABULARY. Hugging Face's wire shape lives in
1132
+ * `proxy/partners/huggingface.ts` and nowhere else. What crosses this
1133
+ * module is uRun's own ledger vocabulary, which every distributor's
1134
+ * adapter and the console read alike.
1135
+ * * NO `gpu_seconds`, AND NO LEDGER COLUMN FOR ONE (ENG-417). Exclusive
1136
+ * per-request GPU occupancy is not obtainable under continuous batching,
1137
+ * and wall-clock there is SHARED occupancy — a per-request figure would
1138
+ * double-count against `usage_events.metered_gpu_seconds`, which meters
1139
+ * the instance once and is reduced by the hourly `usage_accrual` sweep so
1140
+ * that ledger sums stay exactly the metered wall-clock cost. What this
1141
+ * module sends instead is a LINK (`session_id`) and a WEIGHT (the
1142
+ * residency milliseconds), from which the share is DERIVED at
1143
+ * reconciliation so the shares sum to the instance total by construction.
1144
+ * The residency is NOT GPU time: under continuous batching the sum of
1145
+ * concurrent `decode_ms` EXCEEDS the session's wall-clock, which is
1146
+ * precisely why it is only ever meaningful as a normalized fraction.
1147
+ * * NO DURABLE BUFFER. Rows are queued in memory and nowhere else. The
1148
+ * inference-proxy Deployment declares NO VOLUMES — not a PVC, not even an
1149
+ * emptyDir — so there is nowhere on this pod that survives a SIGKILL, and
1150
+ * making one would mean a StatefulSet with a per-replica volume on an
1151
+ * internet-facing HA front door where a scaled-down replica's volume holds
1152
+ * unflushed rows forever. What survives instead is the `inference_request`
1153
+ * line this proxy writes to stdout for EVERY `/v1` request, which
1154
+ * fluent-bit ships off the pod into Loki; that is the reconciliation
1155
+ * record, and it is a way to KNOW what was lost rather than a way to bill
1156
+ * it. See {@link drainLedgerWrites}.
1157
+ *
1158
+ * RETRY, AND THE RULE IT LIVES UNDER (ENG-415). This module used to do NO
1159
+ * retry, for a good reason: a retry after a timeout cannot know whether the
1160
+ * first attempt landed, and the row it would re-send came back as an
1161
+ * undifferentiated duplicate — so retrying turned an ambiguous outcome into a
1162
+ * loud error that looked exactly like a forged row. That reason is now gone:
1163
+ * the control plane's 409 carries the row already stored, so a retry whose
1164
+ * first attempt landed resolves as `already_written` instead. Retry is
1165
+ * therefore allowed for EXACTLY ONE failure reason and no other — see
1166
+ * {@link isRetryable}, which is where the rule lives rather than in a
1167
+ * judgement at the call site.
1168
+ */
1169
+
1170
+ /**
1171
+ * ONE ledger row, in the ledger's own vocabulary — exactly the body the
1172
+ * control plane's `inference-ledger` function takes.
1173
+ *
1174
+ * `org_id` and `api_key_id` are ABSENT BY CONSTRUCTION rather than omitted
1175
+ * here: the control plane derives both from the credential the write rides,
1176
+ * so there is no field on this surface that could name another org.
1177
+ */
1178
+ interface LedgerWrite {
1179
+ /** The uuid this request's `Inference-Id` response header carried. */
1180
+ inference_id: string;
1181
+ /**
1182
+ * The identity that served it — IN THE NAMESPACE {@link identity_kind}
1183
+ * NAMES. Never the caller's raw `model` string.
1184
+ */
1185
+ model_id: string;
1186
+ /**
1187
+ * WHICH NAMESPACE {@link model_id} SPEAKS (ENG-412): a catalog identity
1188
+ * (`public.model_prices` vocabulary), or the caller's own deployed app slug.
1189
+ * The two are unrelated namespaces that can nonetheless COLLIDE as strings,
1190
+ * so the row states which one it is rather than leaving it to be inferred —
1191
+ * and the control plane refuses to price anything but a catalog row, in the
1192
+ * writer's predicate and again in a table CHECK.
1193
+ */
1194
+ identity_kind: 'catalog_model' | 'caller_org_app';
1195
+ variant: string | null;
1196
+ task: string;
1197
+ http_status: number;
1198
+ started_at: string;
1199
+ completed_at: string;
1200
+ /** See {@link billableOutcome}. The one fact the database cannot see. */
1201
+ billable_outcome: boolean;
1202
+ /**
1203
+ * THE INSTANCE that served it (ENG-417) — the link the control plane
1204
+ * verifies against the org before writing. NULL when this request's dial
1205
+ * exposed no session identity; never a fabricated id.
1206
+ */
1207
+ session_id: string | null;
1208
+ /** NULL = this request has no such quantity. NEVER zero of it. */
1209
+ prompt_tokens: number | null;
1210
+ completion_tokens: number | null;
1211
+ output_units: number | null;
1212
+ /**
1213
+ * THE WEIGHT (ENG-417) — the serve runtime's own residency in whole
1214
+ * milliseconds. An allocation input, never GPU time and never a cost. NULL
1215
+ * means the runtime reported nothing, which is not zero: a weightless
1216
+ * request drops out of its instance's allocation and inflates every other
1217
+ * request's share.
1218
+ */
1219
+ queue_ms: number | null;
1220
+ prefill_ms: number | null;
1221
+ decode_ms: number | null;
1222
+ total_ms: number | null;
1223
+ }
1224
+ /** What the control plane answers with — the row as it was actually written. */
1225
+ interface LedgerRecorded {
1226
+ inference_id: string;
1227
+ /** NULL = NOT YET PRICED (ENG-317). Never "free". */
1228
+ cost_nano_usd: number | null;
1229
+ /** `in=<model_prices.id>;out=<model_prices.id>` when priced. */
1230
+ price_version: string | null;
1231
+ /**
1232
+ * Why the row carries no price, or null when it is priced. One of
1233
+ * `not_a_catalog_model` | `outcome_not_billable` | `http_status_not_success`
1234
+ * | `task_unit_undeclared` | `task_not_token_billed` | `token_counts_absent`
1235
+ * | `no_active_price` | `no_active_input_price` | `no_active_output_price`.
1236
+ * It is a NORMAL answer, not an error: no PER-TOKEN rate has been seeded
1237
+ * yet, so this lane has nothing to price with. (Not "the price book is
1238
+ * empty" — `model_prices` has carried live `per_minute` rows since
1239
+ * 2026-09-10; what is absent is `per_million_input_tokens` /
1240
+ * `per_million_output_tokens`.)
1241
+ *
1242
+ * Only `not_a_catalog_model` is PERMANENT. A caller-org app has no catalog
1243
+ * identity, so no price book will ever cover it; every other reason means
1244
+ * "not priced YET".
1245
+ *
1246
+ * THE TWO `task_*` REASONS ARE ENG-427, and they are the reason
1247
+ * {@link LedgerWrite.task} is worth carrying honestly. The control plane prices a row IN THE UNIT ITS
1248
+ * TASK DECLARES (`public.task_billing_units`), never in the unit the data
1249
+ * happens to look like:
1250
+ *
1251
+ * * `task_not_token_billed` — this lane's billable quantity is not tokens
1252
+ * (an image lane's is {@link LedgerWrite.output_units}), so its token counts are not
1253
+ * a price for it. A genuine `{0, 0}` receipt on such a lane would
1254
+ * otherwise settle at `cost_nano_usd = 0` PERMANENTLY — the control
1255
+ * plane's append-only rule allows a deliberate re-price and never a
1256
+ * return to NULL — which is an invoice asserting the work was free when
1257
+ * its billable quantity was never measured.
1258
+ * * `task_unit_undeclared` — nobody has declared what this task bills in,
1259
+ * so nothing is priced. Adding a lane to {@link ledgerTaskOf} without a
1260
+ * matching declaration fails CLOSED and says which fix it wants.
1261
+ */
1262
+ unpriced_reason: string | null;
1263
+ }
1264
+ /**
1265
+ * WHY A LEDGER WRITE DID NOT LAND — a FIXED, SMALL vocabulary, because this is
1266
+ * what an alert matches on.
1267
+ *
1268
+ * ENG-409 made the fabricated-row hazard (ENG-410) detectable and stopped
1269
+ * there: a duplicate `inference_id` raises 23505, surfaces as a 409, and
1270
+ * arrives here as a rejected write. But it arrived carrying only an English
1271
+ * SENTENCE, which meant the only way to alert on "somebody wrote this row
1272
+ * ahead of us" was a regex over prose. An alert that is a substring match on a
1273
+ * message nobody promised to keep stable is an alert that dies silently the
1274
+ * first time the wording is improved — and this particular alert is the ONLY
1275
+ * signal that a customer is under-billing itself.
1276
+ *
1277
+ * So the reason is a FIELD, drawn from this closed set, and it is also the
1278
+ * metric label ({@link LedgerOutcome}):
1279
+ *
1280
+ * * `duplicate` — the id was already in the ledger and the stored row is
1281
+ * PROVABLY NOT THE ONE WE WROTE. Since the 409 carries that row (ENG-415),
1282
+ * a matching one is our own landed attempt and is not a failure at all
1283
+ * ({@link LedgerAlreadyWritten}); what reaches here is somebody else's row
1284
+ * — or a row we could not read back, which fails to this loud side on
1285
+ * purpose. This is the ENG-410 signal.
1286
+ * * `duplicate_foreign` — the id is taken by a row THIS ORG CANNOT READ.
1287
+ * `inference_id` is a global primary key, so it is reachable, and it can
1288
+ * never be our own write. The loudest state in the lane.
1289
+ * * `auth` — the control plane refused the caller's org key.
1290
+ * * `surface` — the control plane could not be reached, answered a status
1291
+ * we do not accept, or answered a body we could not validate. THE ONLY
1292
+ * RETRYABLE ONE (see {@link isRetryable}).
1293
+ * * `unknown` — the lane rejected with something that carries NO reason at
1294
+ * all. It is its OWN value rather than being folded into `surface`
1295
+ * precisely so it cannot hide: a reason-less rejection is a defect in the
1296
+ * lane, and counting it as a surface error would file a bug as weather —
1297
+ * and would retry it, which is how a defect becomes a loop.
1298
+ * * `dwell_expired` — the row waited longer than {@link LEDGER_MAX_DWELL_MS}
1299
+ * and was given up on. Revenue lost, named as revenue lost.
1300
+ * * `dropped_on_exit` — the row was still queued when the shutdown drain ran
1301
+ * out. One per row, per pod, per rolling deploy that went badly.
1302
+ */
1303
+ declare const LEDGER_FAILURE_REASONS: readonly ["duplicate", "duplicate_foreign", "auth", "surface", "unknown", "dwell_expired", "dropped_on_exit"];
1304
+ type LedgerFailureReason = (typeof LEDGER_FAILURE_REASONS)[number];
1305
+ /**
1306
+ * Implemented by every error a {@link LedgerLane.write} may reject with, so
1307
+ * the reason travels ON the error instead of being re-derived from its text.
1308
+ *
1309
+ * It is a FIELD rather than a class hierarchy on purpose: the auth rejection
1310
+ * must stay an `instanceof ProxyAuthError` for every existing caller, and a
1311
+ * class can only have one base.
1312
+ */
1313
+ interface LedgerFailure {
1314
+ readonly ledgerFailureReason: LedgerFailureReason;
1315
+ }
1316
+ /**
1317
+ * WHY A ROW WAS NOT EVEN ATTEMPTED. Also a closed vocabulary, and for the same
1318
+ * reason: a served request that produced no ledger row is revenue that can
1319
+ * never be recovered, so each of these is a counted event and not merely a log
1320
+ * line somebody might read.
1321
+ *
1322
+ * `not_a_catalog_model` WAS IN THIS LIST AND IS DELIBERATELY GONE (ENG-412). A
1323
+ * request served by a caller's own deployed app used to be skipped here; it
1324
+ * now WRITES a row in its own declared namespace (`identity_kind:
1325
+ * 'caller_org_app'`), so nothing can emit that skip any more. A member of a
1326
+ * closed, counted vocabulary that is structurally unreachable is a counter
1327
+ * that can only ever read zero — it invites the reader to conclude the case
1328
+ * never happens, when in truth the case stopped being a skip at all.
1329
+ *
1330
+ * DO NOT CONFUSE IT WITH THE UNPRICED REASON OF THE SAME NAME, which is alive
1331
+ * and is the whole point of ENG-412: the control plane still answers
1332
+ * `unpriced_reason: 'not_a_catalog_model'` for those rows (see
1333
+ * {@link LedgerRecorded}). The row exists and is counted; it simply can never
1334
+ * carry a price. Two vocabularies, one word, opposite states — a written row
1335
+ * versus no row at all.
1336
+ *
1337
+ * `no_model_named` replaces it for the one case on that lane that still cannot
1338
+ * be written: `model_id` is NOT NULL, a caller-org app's identity IS the
1339
+ * caller's model string, and a request that named no model leaves nothing
1340
+ * honest to record.
1341
+ */
1342
+ declare const LEDGER_SKIP_REASONS: readonly ["no_response_head", "no_org", "model_unresolved", "no_model_named", "queue_full"];
1343
+ type LedgerSkipReason = (typeof LEDGER_SKIP_REASONS)[number];
1344
+ /**
1345
+ * The TERMINAL fate of one request's ledger row, as a bounded pair of metric
1346
+ * labels. Every `/v1` request that reaches a lane produces exactly one of
1347
+ * these, so `sum(urun_inference_ledger_writes_total)` is the number of
1348
+ * billable requests the proxy has accounted for — and the `skipped` and
1349
+ * `failed` arms are the leak.
1350
+ *
1351
+ * EVERY VALUE IS OWNED BY THIS MODULE. `reason` is drawn from
1352
+ * {@link LEDGER_SKIP_REASONS}, {@link LEDGER_FAILURE_REASONS} or the two
1353
+ * pricing states below — never from the control plane's answer and never from
1354
+ * anything a caller can influence. That is what keeps the series set bounded
1355
+ * without a `BoundedLabel`: an `unpriced_reason` echoed verbatim into a label
1356
+ * would let a surface that changed shape blow up the TSDB.
1357
+ */
1358
+ type LedgerOutcome =
1359
+ /** The row is in the ledger. `priced` / `unpriced` is ENG-317's NULL-is-not-zero distinction. */
1360
+ {
1361
+ outcome: 'written';
1362
+ reason: 'priced' | 'unpriced';
1363
+ }
1364
+ /**
1365
+ * The row was ALREADY in the ledger and it is OURS — an earlier attempt
1366
+ * landed and we never learned it (ENG-415). A terminal SUCCESS: the request
1367
+ * is accounted for exactly once, and it is counted separately from `written`
1368
+ * only so "how often does the write path go ambiguous" is answerable.
1369
+ */
1370
+ | {
1371
+ outcome: 'already_written';
1372
+ reason: 'priced' | 'unpriced';
1373
+ }
1374
+ /** No row was attempted. */
1375
+ | {
1376
+ outcome: 'skipped';
1377
+ reason: LedgerSkipReason;
1378
+ }
1379
+ /** A row was attempted and did not land. */
1380
+ | {
1381
+ outcome: 'failed';
1382
+ reason: LedgerFailureReason;
1383
+ };
1384
+ /**
1385
+ * The ledger write seam. Only the HOSTED endpoint supplies one (the write
1386
+ * rides the caller's own org API key to the control plane; this process holds
1387
+ * no database credential). A proxy WITHOUT one does not participate in the
1388
+ * ledger at all — the `usage` / `supportsImages` posture: configured absence,
1389
+ * declared at the seam, not a degradation discovered at runtime. The local
1390
+ * CLI proxy is single-tenant and bills nothing, so it has none.
1391
+ */
1392
+ interface LedgerLane {
1393
+ /**
1394
+ * Write one row, as the holder of the credential on `req`. The request is
1395
+ * over by the time this is called; `req` is carried only for its
1396
+ * `Authorization` header, never re-read.
1397
+ */
1398
+ write(req: IncomingMessage, entry: LedgerWrite): Promise<LedgerRecorded>;
1399
+ /**
1400
+ * Count ONE terminal outcome. REQUIRED, unlike {@link report} — which has a
1401
+ * real default (this process's stdout) — because there is no default for
1402
+ * "count it nowhere" that is not a silent no-op. A lane that genuinely
1403
+ * counts nothing has to say so at the seam, in code a reviewer can see.
1404
+ *
1405
+ * This is the alerting surface for ENG-410: `outcome="failed"` with
1406
+ * `reason="duplicate"` is the fabricated-row signal, and the log line is
1407
+ * only the per-request detail behind it.
1408
+ */
1409
+ observe(outcome: LedgerOutcome): void;
1410
+ /**
1411
+ * Where this lane's own structured diagnostics go — one newline-terminated
1412
+ * JSON object per event. Absent, they go to this process's stdout (the
1413
+ * hosted endpoint's collected container log). Injectable so tests assert on
1414
+ * them without racing the process streams.
1415
+ *
1416
+ * These are DELIBERATELY NOT the `requestLog` sink: that sink carries
1417
+ * exactly one `inference_request` record per `/v1` request and readers count
1418
+ * on that.
1419
+ */
1420
+ report?(line: string): void;
1421
+ }
1422
+
1423
+ /**
1424
+ * The shared Responses execution/store seam — the ONE execution + event-
1425
+ * envelope path and the authorized tenant store, imported by BOTH the HTTP
1426
+ * SSE lane (server.ts) and the `/v1/responses` WebSocket lane
1427
+ * (responses-ws.ts). This module deliberately imports NO transport acceptor
1428
+ * and NO HTTP lane module: its only imports are type-only, so the module
1429
+ * graph server.ts → responses-ws.ts → responses-turn.ts is acyclic and no
1430
+ * Vitest/Bun/Node module-evaluation order can observe an uninitialized
1431
+ * binding (the constructor-failure class of bug this extraction removes).
1432
+ */
1433
+
1434
+ /** The request envelope every Responses-shaped upstream call carries. */
1435
+ interface ResponsesCreateParams {
1436
+ model?: string;
1437
+ input: unknown;
1438
+ stream?: boolean;
1439
+ tools?: unknown;
1440
+ tool_choice?: unknown;
1441
+ temperature?: number;
1442
+ max_output_tokens?: number;
1443
+ /**
1444
+ * System/developer instructions, forwarded to the serve envelope — where
1445
+ * they become ONE prepended `system`-role message (the protocol's native
1446
+ * per-request system input; there is NO envelope-level instructions field).
1447
+ * Typed `string` end to end (transport/encode.ts) so no cast bridges the seam.
1448
+ */
1449
+ instructions?: string;
1450
+ top_p?: number;
1451
+ stop?: string[];
1452
+ /**
1453
+ * Reasoning controls (urun-python #1667): forwarded VERBATIM to the serve
1454
+ * envelope — `chat_template_kwargs.enable_thinking:false` is the think-off
1455
+ * switch. Absent -> absent (no default injection; server-side validation).
1456
+ */
1457
+ reasoning_effort?: string;
1458
+ chat_template_kwargs?: Record<string, unknown>;
1459
+ /**
1460
+ * OpenAI `response_format` — structured output (ENG-310). On the Responses
1461
+ * lane the caller spells this `text.format`; `textFormatOf` lifts it onto
1462
+ * this ONE field so both lanes hand the serve envelope the same key.
1463
+ * Forwarded verbatim — the serve runtime is the one validator.
1464
+ */
1465
+ response_format?: unknown;
1466
+ }
1467
+ /** How one pinned session ended (the native phase machinery's terminal step). */
1468
+ interface SessionEndInfo {
1469
+ /**
1470
+ * Milliseconds until the session's native deadline (`endsAt`), or null when
1471
+ * the app declared no maximum session length. NOTE (missing primitive,
1472
+ * called out in the PR): core exposes no PRE-expiry notice event — this
1473
+ * callback fires AT terminal loss, so timeLeftMs is ~0 on expiry.
1474
+ */
1475
+ timeLeftMs: number | null;
1476
+ /** The terminal reason (expired / ended / error), for the loud close. */
1477
+ reason: string;
1478
+ }
1479
+ /** The upstream calls the proxy makes — injectable (tests; alt transports). */
1480
+ interface ProxyClients {
1481
+ /**
1482
+ * The NON-SECRET tenancy label of the org this backhaul belongs to — the
1483
+ * control plane's org id, set by the hosted registry from the verified
1484
+ * `CallerIdentity` (`hosted/tenants.ts`). The request handler copies it onto
1485
+ * the request's `InferenceTurn`, which is what puts an `org` label on the
1486
+ * Prometheus series and an `org` field on the billing record.
1487
+ *
1488
+ * THE ORG, NEVER THE KEY: the Bearer key is the credential on this surface,
1489
+ * so it must never reach a metric, a log, or a dashboard. Absent on the
1490
+ * single-tenant local proxy and on hand-built test clients.
1491
+ */
1492
+ readonly tenant?: string;
1493
+ /**
1494
+ * `UrunResponses(session).responses.create` — an async iterable of Responses
1495
+ * stream events.
1496
+ *
1497
+ * `onDial` is WHAT THE DIAL REPORTS ABOUT ITSELF (ENG-417 + ENG-414): the
1498
+ * implementation calls it once PER DIAL with a {@link DialReport} — the uRun
1499
+ * session id of the backhaul it is ABOUT TO USE, and the billing identity
1500
+ * that dial's acquisition resolved. So a request that re-homes reports the
1501
+ * REPLACEMENT on both counts: the instance that actually served it, and the
1502
+ * identity it was actually served under.
1503
+ *
1504
+ * IT IS A SECOND ARGUMENT, NOT A FIELD ON THE ENVELOPE, and that is
1505
+ * load-bearing. `params` is forwarded VERBATIM to the serve runtime by every
1506
+ * implementation of this seam, including hand-built embedder clients; a
1507
+ * function smuggled onto it would reach a JSON serializer, be dropped
1508
+ * silently, and leave one implementation (the one that remembered to strip
1509
+ * it) behaving differently from the rest. Out of band, the envelope stays
1510
+ * exactly what it was and an implementation that ignores `onDial` simply
1511
+ * records no link.
1512
+ *
1513
+ * IT IS A CALLBACK RATHER THAN A RETURN VALUE because the answer is not
1514
+ * known when `createResponse` returns: `rehome.ts` dials inside the async
1515
+ * generator and may dial a SECOND time before the first event is yielded.
1516
+ * Asking beforehand — an earlier draft of ENG-417 did — has two defects at
1517
+ * once: it records a session the turn may abandon, and it ALLOCATES a pooled
1518
+ * backhaul for requests still going to be refused by local validation, which
1519
+ * is GPU capacity spent on a 400.
1520
+ *
1521
+ * THE IDENTITY ARRIVES THROUGH `ModelRouter.sessionFor` AND NOT FROM THE
1522
+ * POOL ENTRY, and a reader who does not know why will "simplify" it back
1523
+ * into the bug. The obvious-looking move is to widen
1524
+ * `RehomeOptions.sessionIdOf` the way the session id arrived. IT CANNOT
1525
+ * WORK: `entry` is the GENERIC pooled session (`rehome.ts` is parameterised
1526
+ * over it) and knows nothing about models, so the identity is not derivable
1527
+ * from it. The other obvious move — resolving the identity separately inside
1528
+ * the dial loop — is worse: it would be a SECOND RESOLUTION of the same
1529
+ * question, which is precisely the defect that got ENG-417's pre-dial pin
1530
+ * rejected in review (a resolution taken at a different moment from the dial
1531
+ * can answer differently, and then the row names a variant nobody served).
1532
+ * So the acquisition returns the identity alongside the entry, and the loop
1533
+ * reports what it already resolved.
1534
+ *
1535
+ * BOTH IDENTITY FLAVOURS RIDE IT. The catalog `(model_id, variant)` a row is
1536
+ * priced on, and the caller-org APP SLUG that ENG-412 made billing-bearing —
1537
+ * it lands in `model_id` with `identity_kind: 'caller_org_app'`, on rows
1538
+ * that previously were not written at all. They had the identical re-home
1539
+ * window and they close by the same line: the dial reports whichever flavour
1540
+ * it resolved.
1541
+ *
1542
+ * WIDEN {@link DialReport}; do not add a second callback beside it. Two
1543
+ * callbacks firing from the same loop about the same dial would be two
1544
+ * things that can disagree, which is the whole defect this seam exists to
1545
+ * remove.
1546
+ */
1547
+ createResponse(params: ResponsesCreateParams, onDial?: DialObserver): Promise<AsyncIterable<unknown>> | AsyncIterable<unknown>;
1548
+ /** `listModels(...)` result (an OpenAI model list object). */
1549
+ listModels(): Promise<unknown>;
1550
+ /**
1551
+ * The OpenRouter provider document (`GET /v1/models?format=openrouter`).
1552
+ * Optional at the seam: hand-built test clients may omit it, in which case
1553
+ * the format=openrouter branch answers 501 — loud, never a silent empty.
1554
+ */
1555
+ openRouterModels?(): Promise<unknown>;
1556
+ /**
1557
+ * The Images lane's capability gate (images.ts `ImageModelGate`), backed by
1558
+ * ModelRouter.supportsImages — the caller-org catalog's modality metadata,
1559
+ * per-caller by construction; never function-name inference. Optional at
1560
+ * the seam: hand-built clients may omit it, in which case the mounted
1561
+ * image lanes answer 501 — loud, never a silent allow-all.
1562
+ */
1563
+ supportsImages?(model: string | undefined, mode: 'generation' | 'edit'): Promise<boolean>;
1564
+ /**
1565
+ * The realtime/audio lane's modality gate (openai-realtime/binding.ts),
1566
+ * backed by ModelRouter.supportsAudio — the caller-org catalog's `task`
1567
+ * column (stt | tts, invariant across placements), per-caller by
1568
+ * construction; never function-name inference. Optional at the seam like
1569
+ * supportsImages: hand-built clients may omit it, in which case the
1570
+ * realtime lane refuses the upgrade LOUDLY (501) — never a silent
1571
+ * allow-all, and never a chat model binding an audio session.
1572
+ */
1573
+ supportsAudio?(model: string | undefined): Promise<boolean>;
1574
+ /**
1575
+ * `GET /v1/images/models` — the image-capable slice of the model surface
1576
+ * (ModelRouter.imageModelList: deployed task-image apps + the shared block
1577
+ * joined against the catalog). Optional at the seam for the same reason;
1578
+ * its absence is the loud 501, never a silently empty list.
1579
+ */
1580
+ listImageModels?(): Promise<unknown>;
1581
+ /**
1582
+ * THE PRE-DIAL DEFAULT FOR THE LEDGER'S BILLING IDENTITY
1583
+ * (ModelRouter.catalogIdentityFor): the CATALOG `(model_id, variant)` a
1584
+ * request's `model` actually resolved to, or `null` when it names no catalog
1585
+ * model (a caller-org app, whose identity is an app slug — a different
1586
+ * namespace from `model_prices.model_id`).
1587
+ *
1588
+ * The ledger prices on `(model_id, variant)`, and the caller's raw string is
1589
+ * not that: a bare catalog id resolves to the model's `compat_default`
1590
+ * variant, so recording the string would file the request under a variant
1591
+ * nobody served.
1592
+ *
1593
+ * THE DIAL HAS THE LAST WORD (ENG-414). What a served row records is the
1594
+ * identity `onDial` reported, off the acquisition that dial used — so a
1595
+ * re-homed request is billed on the replacement. This seam is what a request
1596
+ * that reaches NO dial records instead, which is why it is still resolved
1597
+ * on the request path.
1598
+ *
1599
+ * REQUIRED AT THE SEAM (ENG-429), unlike `supportsImages`. ENG-414 changed
1600
+ * what an omission MEANS: a backhaul with no `catalogIdentity` but a
1601
+ * dialling `createResponse` used to write no ledger row at all, and now
1602
+ * writes — and prices — one on the dial's own report. Honest, but it means
1603
+ * a lane could acquire BILLING behaviour by OMITTING a member, which is the
1604
+ * wrong direction for a default on a revenue path. So "no seam" is a
1605
+ * COMPILE ERROR now, the same move as the exhaustive `Record<...>`
1606
+ * registration checks used elsewhere in this codebase: a contract the
1607
+ * typechecker holds, not a convention a reviewer must notice. An implementer
1608
+ * that genuinely has no catalog oracle says so EXPLICITLY — a function of
1609
+ * its own, in code a reviewer can see — never by leaving the member out, and
1610
+ * never an invented identity. (The runtime guard in server.ts
1611
+ * `pinCatalogIdentity` stays for UNTYPED embedders: the seam is a public
1612
+ * surface callable from plain JS, where a member can still be absent at
1613
+ * runtime, and the guard's answer there remains the loud `unresolved` pin.)
1614
+ */
1615
+ catalogIdentity(model: string | undefined): Promise<{
1616
+ model_id: string;
1617
+ variant: string;
1618
+ } | null>;
1619
+ /**
1620
+ * THE CALLER-ORG APP's OWN IDENTITY (ENG-412): the app slug `model` actually
1621
+ * RESOLVES to, for the lane where {@link catalogIdentity} answers null. Like
1622
+ * {@link catalogIdentity} it is the PRE-DIAL DEFAULT — the dial reports this
1623
+ * flavour too, and has the last word (ENG-414).
1624
+ *
1625
+ * IT IS NOT THE CALLER'S RAW STRING. Routing trims, length-caps and matches
1626
+ * that string before selecting an app, so a differently cased, punctuated or
1627
+ * whitespace-padded name is served by one canonical slug while the raw value
1628
+ * says something else — and the ledger would then file two spellings of the
1629
+ * same app as two different identities, which no app-level reconciliation
1630
+ * can undo afterwards.
1631
+ *
1632
+ * CHEAP AND ALLOCATION-FREE: it rides `ModelRouter.resolveTarget`, whose app
1633
+ * listing, catalog and shared block are cached promises. It opens no session
1634
+ * — asking which app would serve must never BE the thing that spends GPU
1635
+ * capacity.
1636
+ */
1637
+ servingAppSlug?(model: string | undefined): Promise<string | null>;
1638
+ /**
1639
+ * Open (or reuse) the NATIVE audio lanes on the pooled session `model` routes
1640
+ * to — the same first-party machinery as RealtimeClient.enableAudio
1641
+ * (transport/media.ts `enableSessionAudio`: AudioBridge over the session's
1642
+ * `audio` stream lane at 24 kHz). One lane per pooled
1643
+ * session; repeat calls return the same lane. Optional at the seam because
1644
+ * text-only embeddings exist — but a surface that RECEIVES audio while the
1645
+ * embedder wired no `openAudio` must fail LOUD, never drop chunks.
1646
+ */
1647
+ openAudio?(model: string | undefined): Promise<ProxyAudioLane>;
1648
+ /**
1649
+ * Open (or reuse) the NATIVE video FRAME lane on the pooled session `model`
1650
+ * routes to (transport/media.ts `enableSessionVideo`: discrete JPEG frames →
1651
+ * `stream('rt-video-in').emit`, the §5 named-DATA image-bytes path — NOT the
1652
+ * RTP media plane, which requires an already-H.264-encoded track). One lane
1653
+ * per pooled session; repeat calls return the same lane. Optional at the seam
1654
+ * because text-only embeddings exist — but a surface that RECEIVES video
1655
+ * frames while the embedder wired no `openVideo` must fail LOUD, never drop
1656
+ * frames.
1657
+ */
1658
+ openVideo?(model: string | undefined): Promise<ProxyVideoLane>;
1659
+ /**
1660
+ * Open (or reuse) the NATIVE video OUTPUT lane on the pooled session
1661
+ * `model` routes to (transport/video-out.ts `openSessionVideoOutLane`):
1662
+ * decoded spec-v1 records consumed from the session's named §5
1663
+ * `rt-video-out` downstream. The Live adapter fans each record onto the
1664
+ * SAME Bidi WS as an extension server message (spec v2, WS-inline) — no
1665
+ * side-channel transport. Optional at the seam; an ABSENT seam means the
1666
+ * `urun.videoOut` capability is refused by omission in setupComplete
1667
+ * (vanilla parity), while a wired seam that cannot open the lane must
1668
+ * throw LOUD — never a quiet downgrade.
1669
+ */
1670
+ openVideoOut?(model: string | undefined): Promise<ProxyVideoOutLane>;
1671
+ /**
1672
+ * SESSION-IDENTITY SEAM (ModelRouter.handleFor): the opaque stable handle
1673
+ * for the pooled session currently serving `model`'s turns. Derived from
1674
+ * native identity (app slug + uRun session id) — the same identity the
1675
+ * serve-side session-affinity tag rides (urun-python#1556/#1582).
1676
+ */
1677
+ sessionHandle(model: string | undefined): Promise<string>;
1678
+ /**
1679
+ * createResponse PINNED to the exact session a handle names
1680
+ * (ModelRouter.sessionForHandle). Throws SessionGoneError LOUDLY when that
1681
+ * session is gone or was replaced — never silently opens a fresh session
1682
+ * while claiming resume. Deliberately NO re-home on this path: re-homing
1683
+ * would swap the pinned session out from under the caller.
1684
+ */
1685
+ createResponseOn(handle: string, params: ResponsesCreateParams): Promise<AsyncIterable<unknown>> | AsyncIterable<unknown>;
1686
+ /**
1687
+ * Subscribe to the pinned session's terminal end via core's NATIVE phase
1688
+ * machinery (Session.onPhase → terminal 'expired'/'ended'/'error'). Fires
1689
+ * `cb` once. Throws SessionGoneError if the handle's session is already
1690
+ * gone — which doubles as the loud reattach check at resume time. Returns
1691
+ * the unsubscribe.
1692
+ */
1693
+ onSessionEnd(handle: string, cb: (end: SessionEndInfo) => void): Promise<() => void>;
1694
+ }
1695
+ /**
1696
+ * The audio lane handle `openAudio` returns — structurally the transport
1697
+ * AudioBridge (media.ts): base64 PCM16 @24 kHz mono in both directions.
1698
+ */
1699
+ type ProxyAudioLane = Pick<AudioBridge, 'appendInputAudio' | 'onOutputAudio'>;
1700
+ /**
1701
+ * The video frame-lane handle `openVideo` returns — structurally the
1702
+ * transport VideoFrameLane (media.ts): one raw encoded JPEG frame per call,
1703
+ * input-only (Live-style protocols have no video OUT modality).
1704
+ */
1705
+ type ProxyVideoLane = Pick<VideoFrameLane, 'sendInputFrame'>;
1706
+ /**
1707
+ * The video-OUT lane handle `openVideoOut` returns — structurally the
1708
+ * transport VideoOutLane (video-out.ts): the codec plus frame/error fan-out
1709
+ * the WS-inline delivery subscribes.
1710
+ */
1711
+ type ProxyVideoOutLane = Pick<VideoOutLane, 'codec' | 'onOutputFrame' | 'onLaneError'>;
1712
+ /**
1713
+ * Per-request `ProxyClients` resolution — the seam the HOSTED multi-tenant
1714
+ * server (`src/hosted/`) plugs into so ONE handler implementation serves
1715
+ * every org: the hosted server resolves the caller's org from its Bearer API
1716
+ * key and returns THAT org's backhaul. The local CLI passes a fixed
1717
+ * `ProxyClients` instead; both go through the identical handler body.
1718
+ *
1719
+ * Throwing from here is the loud path: the thrown error surfaces in the
1720
+ * lane's native error envelope (see {@link ProxyHandlerOptions.statusOf}).
1721
+ */
1722
+ type ProxyClientsFor = (req: IncomingMessage) => ProxyClients | Promise<ProxyClients>;
1723
+ interface ProxyHandlerOptions {
1724
+ /** A fixed backhaul (local CLI) or a per-request resolver (hosted server). */
1725
+ clients: ProxyClients | ProxyClientsFor;
1726
+ /** Optional bearer the local agent must present (never forwarded upstream). */
1727
+ apiKey?: string;
1728
+ /**
1729
+ * WHAT this proxy serves (app/org/fn/base_url + proxy_version), surfaced as
1730
+ * the `identity` block on `GET /stats` so a second `urun compat` invocation
1731
+ * can reuse this proxy iff the identity matches its own exactly. Only the
1732
+ * standalone `urun compat proxy` command sets it — a launch-mode proxy is
1733
+ * child-owned (it dies with its child) and an embedder that omits it is
1734
+ * simply never reused.
1735
+ */
1736
+ identity?: ProxyIdentity;
1737
+ /**
1738
+ * Map a thrown error onto an HTTP status + error `type` before the generic
1739
+ * 500. The hosted server uses it to turn its auth/tenancy failures into a
1740
+ * 401 in the OpenAI envelope. Returning null means "not mine" — the error
1741
+ * takes the ordinary loud 500 path.
1742
+ */
1743
+ statusOf?: (err: unknown) => {
1744
+ status: number;
1745
+ openaiType: string;
1746
+ anthropicType: string;
1747
+ } | null;
1748
+ /**
1749
+ * Where the per-request structured log line goes — ONE newline-terminated
1750
+ * JSON object per `/v1` request, carrying the request's `Inference-Id`
1751
+ * (proxy/inference-id.ts `inferenceRequestLine`) plus method, path, status
1752
+ * and whether the answer was delivered / failed mid-stream, written when the
1753
+ * response closes (so streams and client aborts are recorded too). This is
1754
+ * the billing correlation record a later usage ledger is keyed on, so it is
1755
+ * emitted unconditionally — never behind a verbosity flag.
1756
+ *
1757
+ * Absent, the record goes to this process's stdout through the repo's
1758
+ * canonical guarded write (`server.ts` `writeInferenceLogToStdout`), which is
1759
+ * the hosted endpoint's collected container log. The local CLI proxy passes
1760
+ * its own sink instead, because the operator owns that terminal — see
1761
+ * `proxy/cli.ts` `startProxy`. Injectable so tests assert on the lines
1762
+ * without racing the process streams.
1763
+ */
1764
+ requestLog?: (line: string) => void;
1765
+ /**
1766
+ * THE CANONICAL USAGE-QUERY SEAM (`POST /v1/usage/requests`, and the
1767
+ * Hugging Face translator over it) — a batch lookup of what the per-request
1768
+ * inference ledger holds for a list of `Inference-Id`s, scoped to the
1769
+ * caller's own org. See proxy/usage.ts.
1770
+ *
1771
+ * Only the HOSTED endpoint supplies one (the read rides the caller's own
1772
+ * org API key to the control plane; this process holds no database
1773
+ * credential). A proxy without it answers those routes with a LOUD 501 —
1774
+ * the `supportsImages` posture — and never with an empty result, which a
1775
+ * billing caller cannot tell from "nothing is priced yet".
1776
+ */
1777
+ usage?: UsageLane;
1778
+ /**
1779
+ * THE PER-REQUEST LEDGER WRITE SEAM (proxy/ledger.ts) — one durable,
1780
+ * already-priced row per billable `/v1` request, written from the response's
1781
+ * `'close'` hook and therefore OFF the response path.
1782
+ *
1783
+ * Only the HOSTED endpoint supplies one (the write rides the caller's own
1784
+ * org API key to the control plane; this process holds no database
1785
+ * credential). A proxy WITHOUT one does not participate in the ledger at
1786
+ * all — the `usage` / `supportsImages` posture of declared absence. The
1787
+ * local CLI proxy is single-tenant and bills nothing, so it has none.
1788
+ *
1789
+ * It is deliberately NOT a place to do work on the hot path, and it cannot
1790
+ * be: `recordInference` is fire-and-forget, returns void, and never throws.
1791
+ */
1792
+ ledger?: LedgerLane;
1793
+ /**
1794
+ * The SECOND sink for the very same per-request record `requestLog`
1795
+ * serializes — the hosted endpoint passes its Prometheus collector
1796
+ * (`proxy/metrics.ts` `InferenceMetrics.observe`). Called once per `/v1`
1797
+ * request, from the same `'close'` hook, with the same object.
1798
+ *
1799
+ * Optional because the local CLI proxy publishes no metrics port; absent, a
1800
+ * request is logged and not counted. It is deliberately NOT a place to do
1801
+ * work: it runs on the response's close hook, so anything slow here delays
1802
+ * the socket teardown.
1803
+ */
1804
+ observe?: (record: InferenceRecord) => void;
1805
+ }
1806
+
1807
+ export { type DialObserver as D, type InferenceRecord as I, type LedgerFailure as L, ModelRouter as M, type ProxyClients as P, SessionGoneError as S, UnknownModelError as U, type DialReport as a, type InferenceUsageRecord as b, type LedgerOutcome as c, type LedgerRecorded as d, type LedgerWrite as e, type ProxyHandlerOptions as f, type ProxyIdentity as g, type ProxyVideoOutLane as h, type UsageRowReporter as i };