@urun-sh/openai 0.5.5 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{ResponsesClient-BYx3YLGo.d.ts → ResponsesClient-BE3-hx3m.d.ts} +2 -0
- package/dist/{ResponsesClient-Dft3bg3b.d.cts → ResponsesClient-DjZWjlFH.d.cts} +2 -0
- package/dist/chunk-2T2YYBVX.js +1 -0
- package/dist/chunk-3WAJD62J.js +4 -0
- package/dist/{chunk-K23AZPI4.js → chunk-5HKWNK3O.js} +1 -1
- package/dist/chunk-7M6LY6DR.js +2 -0
- package/dist/chunk-BSHT6RZZ.js +1 -0
- package/dist/{chunk-GBBY3PCZ.js → chunk-FP4RSAIE.js} +1 -1
- package/dist/chunk-HR5H6S7L.js +1 -0
- package/dist/chunk-NY23USZF.js +57 -0
- package/dist/chunk-VLRMJRLS.js +6 -0
- package/dist/gemini-live.cjs +2 -2
- package/dist/gemini-live.d.cts +17 -2
- package/dist/gemini-live.d.ts +7 -2
- package/dist/gemini-live.js +1 -1
- package/dist/hosted/bin.cjs +40 -26
- package/dist/hosted/bin.js +4 -4
- package/dist/hosted/index.cjs +39 -26
- package/dist/hosted/index.d.cts +643 -16
- package/dist/hosted/index.d.ts +153 -5
- package/dist/hosted/index.js +1 -1
- package/dist/index.cjs +1 -1
- package/dist/index.d.cts +4 -4
- package/dist/index.d.ts +4 -4
- package/dist/index.js +1 -1
- package/dist/{models-NYMZrklp.d.cts → models-DUdx_Y6X.d.cts} +1 -1
- package/dist/pi-extension/index.cjs +7 -7
- package/dist/pi-extension/index.d.cts +1 -1
- package/dist/pi-extension/index.d.ts +1 -1
- package/dist/pi-extension/index.js +1 -1
- package/dist/pi-extension/standalone.cjs +48 -48
- package/dist/proxy/cli.cjs +53 -43
- package/dist/proxy/cli.js +14 -14
- package/dist/proxy/index.cjs +32 -30
- package/dist/proxy/index.d.cts +84 -8
- package/dist/proxy/index.d.ts +16 -6
- package/dist/proxy/index.js +1 -6
- package/dist/responses-turn-Dr37N3kH.d.ts +486 -0
- package/dist/responses-turn-tNUHq5b0.d.cts +1807 -0
- package/dist/{translator-C9uPKypK.d.ts → translator-BCoRFaTs.d.ts} +1 -1
- package/dist/{translator-CcDBEfvm.d.cts → translator-Bh0Bp_Ie.d.cts} +3 -3
- package/dist/{video-out-D20UuJ8G.d.cts → video-out-CCksIcj8.d.cts} +5 -5
- package/package.json +12 -11
- package/dist/chunk-5BZCM3RS.js +0 -4
- package/dist/chunk-5NXM4IO3.js +0 -2
- package/dist/chunk-CWJRDBDC.js +0 -1
- package/dist/chunk-DMD5UENK.js +0 -1
- package/dist/chunk-I2Q3B3OG.js +0 -6
- package/dist/chunk-OI2OY32M.js +0 -1
- package/dist/chunk-QVF7NF7G.js +0 -44
- package/dist/responses-turn-KOAoIqZ-.d.ts +0 -200
- package/dist/responses-turn-OrO4euEN.d.cts +0 -513
- /package/dist/{chunk-YSFSRI3D.js → chunk-5CCJUNH7.js} +0 -0
- /package/dist/{models-NYMZrklp.d.ts → models-DUdx_Y6X.d.ts} +0 -0
- /package/dist/{video-out-CWesbk12.d.ts → video-out-IGQ4YQ-c.d.ts} +0 -0
|
@@ -0,0 +1,1807 @@
|
|
|
1
|
+
import { C as CatalogRow } from './models-DUdx_Y6X.cjs';
|
|
2
|
+
import { IncomingMessage } from 'node:http';
|
|
3
|
+
import { a as AudioBridge, h as VideoFrameLane, j as VideoOutLane } from './video-out-CCksIcj8.cjs';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* WHAT a proxy serves — the `identity` block on `GET /stats`. Reuse of a
|
|
7
|
+
* running proxy is allowed iff every field matches the launching invocation
|
|
8
|
+
* EXACTLY: a partial match (same app, different fn; same everything, older
|
|
9
|
+
* proxy_version) silently routes the agent at the wrong backend, which is
|
|
10
|
+
* worse than any error.
|
|
11
|
+
*/
|
|
12
|
+
interface ProxyIdentity {
|
|
13
|
+
app: string;
|
|
14
|
+
org: string;
|
|
15
|
+
fn: string;
|
|
16
|
+
/**
|
|
17
|
+
* The control-plane URL the proxy's backhaul session opens against
|
|
18
|
+
* (URUN_BASE_URL verbatim — same org/app/fn on staging vs prod are
|
|
19
|
+
* DIFFERENT backends; review finding, #285). Compared exactly: a cosmetic
|
|
20
|
+
* difference (trailing slash) merely refuses reuse and spawns an ephemeral
|
|
21
|
+
* proxy — the safe direction.
|
|
22
|
+
*/
|
|
23
|
+
base_url: string;
|
|
24
|
+
proxy_version: string;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* THE BILLING CORRELATION SEAM — one uuid per `/v1` request, returned as the
|
|
29
|
+
* `Inference-Id` response header, carried as the suffix of the body id the
|
|
30
|
+
* request answers with, and recorded in one structured log line.
|
|
31
|
+
*
|
|
32
|
+
* WHY IT EXISTS: Hugging Face Inference Providers bill routed requests by a
|
|
33
|
+
* unique id the provider returns as a RESPONSE HEADER. Their spec is explicit:
|
|
34
|
+
* "Make sure this header is present on every response you return, including
|
|
35
|
+
* streaming responses. If it's missing, we have no way to match the request
|
|
36
|
+
* and it can't be billed." A response that leaves this proxy without the
|
|
37
|
+
* header is therefore an UNBILLED response — a revenue bug, not a cosmetic
|
|
38
|
+
* one — which is why the header is set on the response object in the request
|
|
39
|
+
* handler's synchronous prefix, before any lane can flush a head.
|
|
40
|
+
*
|
|
41
|
+
* WHAT THIS MINT COVERS: the three HTTP lanes' body ids (`chatcmpl-…`,
|
|
42
|
+
* `resp_…`, `msg_…`) and the `/v1/responses` WebSocket lane's response ids —
|
|
43
|
+
* all of which previously minted their own `Math.random().toString(36)`
|
|
44
|
+
* string, four unrelated id spaces nothing downstream could join on. Only
|
|
45
|
+
* `executeResponsesTurn`'s required `responseId` argument is COMPILE-enforced;
|
|
46
|
+
* the rest is enforced by review, so keep new lanes on this mint.
|
|
47
|
+
*
|
|
48
|
+
* WHAT IT DOES NOT COVER (still separate id spaces, recorded on ENG-309):
|
|
49
|
+
* - `transport/decode.ts` builds `resp_${requestId}` from the SERVE-lane
|
|
50
|
+
* request id that `responses/ResponsesClient.ts` mints (`req_<n>`); the
|
|
51
|
+
* HTTP lanes overwrite it with the id from here, but joining a billing row
|
|
52
|
+
* to the runtime's own usage receipt still needs that id plumbed through
|
|
53
|
+
* `ProxyClients.createResponse`.
|
|
54
|
+
* - the OpenAI Realtime lane (`proxy/openai-realtime/protocol.ts`) mints its
|
|
55
|
+
* own `resp_` ids for WS realtime turns — which, being built from a
|
|
56
|
+
* `req_`-prefixed seed, currently read `resp_req_<hex>` (a double prefix).
|
|
57
|
+
*/
|
|
58
|
+
|
|
59
|
+
type TurnCatalogIdentity = {
|
|
60
|
+
kind: 'catalog';
|
|
61
|
+
model_id: string;
|
|
62
|
+
variant: string;
|
|
63
|
+
} | {
|
|
64
|
+
kind: 'not-a-catalog-model';
|
|
65
|
+
model: string | null;
|
|
66
|
+
} | {
|
|
67
|
+
kind: 'unresolved';
|
|
68
|
+
reason: string;
|
|
69
|
+
};
|
|
70
|
+
/**
|
|
71
|
+
* WHAT ONE DIAL REPORTS ABOUT ITSELF — everything the ledger reads off the
|
|
72
|
+
* dial that actually served, reported together, from that dial.
|
|
73
|
+
*
|
|
74
|
+
* ONE CALLBACK, BOTH FACTS, and that is the point rather than a convenience.
|
|
75
|
+
* `session_id` (ENG-417) and the billing identity (ENG-414) are the two halves
|
|
76
|
+
* of the same question — which instance ran this, and what it ran — and a
|
|
77
|
+
* re-home makes them disagree the moment they are sourced separately: the row
|
|
78
|
+
* named the REPLACEMENT instance while naming the identity the ABANDONED dial
|
|
79
|
+
* resolved. Two callbacks firing from the same loop about the same dial would
|
|
80
|
+
* be two things free to drift apart again, which is the whole defect this seam
|
|
81
|
+
* exists to remove. Widen this payload; never add a second callback beside it.
|
|
82
|
+
*
|
|
83
|
+
* FIRED ONCE PER DIAL, LAST DIAL WINS. `rehome.ts` performs exactly one
|
|
84
|
+
* re-dial when a backhaul dies before producing client-visible content, and
|
|
85
|
+
* reports again before the replacement is used — so what the ledger keeps is
|
|
86
|
+
* the dial that served.
|
|
87
|
+
*/
|
|
88
|
+
interface DialReport {
|
|
89
|
+
/**
|
|
90
|
+
* The uRun session id of the backhaul about to be used, or null when the
|
|
91
|
+
* pool entry carries no session identity. Never a fabricated id: a wrong
|
|
92
|
+
* link moves cost attribution between machines.
|
|
93
|
+
*/
|
|
94
|
+
sessionId: string | null;
|
|
95
|
+
/**
|
|
96
|
+
* The billing identity THIS dial resolved — the ledger's own vocabulary, not
|
|
97
|
+
* the router's. It comes off the SAME `resolveTarget` answer the dial used
|
|
98
|
+
* (`ModelRouter.sessionFor`), never a second resolution taken alongside it:
|
|
99
|
+
* a resolution taken at a different moment can answer differently, and then
|
|
100
|
+
* the row names a variant nobody served. Both flavours travel this way — the
|
|
101
|
+
* catalog `(model_id, variant)` and the caller-org app slug, which ENG-412
|
|
102
|
+
* made billing-bearing.
|
|
103
|
+
*/
|
|
104
|
+
identity: TurnCatalogIdentity;
|
|
105
|
+
}
|
|
106
|
+
/** The observer one dial reports itself to. See {@link DialReport}. */
|
|
107
|
+
type DialObserver = (dial: DialReport) => void;
|
|
108
|
+
/**
|
|
109
|
+
* ONE completed `/v1` request, as data. This is the SINGLE per-request fact
|
|
110
|
+
* object: {@link inferenceRequestLine} serializes it into the billing
|
|
111
|
+
* correlation log line, and `proxy/metrics.ts` `InferenceMetrics.observe`
|
|
112
|
+
* counts the same object into Prometheus. Two sinks, one record — a metric
|
|
113
|
+
* that could disagree with the billing log would be worse than no metric.
|
|
114
|
+
*
|
|
115
|
+
* Nothing here is a secret or a body: no API key, no prompt, no completion.
|
|
116
|
+
* The caller is identified by `org` (a control-plane identifier), never by the
|
|
117
|
+
* Bearer key that resolved it.
|
|
118
|
+
*/
|
|
119
|
+
interface InferenceRecord {
|
|
120
|
+
readonly event: 'inference_request';
|
|
121
|
+
readonly ts: string;
|
|
122
|
+
readonly inference_id: string;
|
|
123
|
+
readonly method: string;
|
|
124
|
+
readonly path: string;
|
|
125
|
+
readonly status: number | null;
|
|
126
|
+
readonly delivered: boolean;
|
|
127
|
+
readonly stream_error: boolean;
|
|
128
|
+
readonly model: string | null;
|
|
129
|
+
readonly stream: boolean | null;
|
|
130
|
+
readonly org: string | null;
|
|
131
|
+
readonly duration_ms: number;
|
|
132
|
+
readonly ttft_ms: number | null;
|
|
133
|
+
/**
|
|
134
|
+
* The usage quantities the per-request ledger is written from. NULL means
|
|
135
|
+
* the request HAS no such quantity — never zero of it. See
|
|
136
|
+
* {@link InferenceTurn.promptTokens}.
|
|
137
|
+
*
|
|
138
|
+
* They are on the RECORD rather than read separately by the ledger writer
|
|
139
|
+
* for the reason the record exists at all: one object, every sink. A writer
|
|
140
|
+
* that took a second walk over the request could disagree with the log line
|
|
141
|
+
* and the metric, and the disagreement would be invisible.
|
|
142
|
+
*
|
|
143
|
+
* NOTE WHAT IS NOT HERE: `gpu_seconds`. There is no such quantity, and as
|
|
144
|
+
* of ENG-417 there is no such ledger column either. Exclusive per-request
|
|
145
|
+
* GPU occupancy is not obtainable under continuous batching, and wall-clock
|
|
146
|
+
* is SHARED occupancy — a per-request stamp would double-count against
|
|
147
|
+
* `usage_events.metered_gpu_seconds`, which meters the instance once. What
|
|
148
|
+
* is here instead is the LINK ({@link session_id}) and the WEIGHT (the
|
|
149
|
+
* residency milliseconds below), from which the share is DERIVED.
|
|
150
|
+
*/
|
|
151
|
+
readonly prompt_tokens: number | null;
|
|
152
|
+
readonly completion_tokens: number | null;
|
|
153
|
+
readonly output_units: number | null;
|
|
154
|
+
/**
|
|
155
|
+
* THE INSTANCE that served this request — see
|
|
156
|
+
* {@link InferenceTurn.sessionId}. On the log line as well as the ledger,
|
|
157
|
+
* so a row that was never written can still be traced to the session it
|
|
158
|
+
* would have been attributed to.
|
|
159
|
+
*/
|
|
160
|
+
readonly session_id: string | null;
|
|
161
|
+
/**
|
|
162
|
+
* THE WEIGHT — see {@link InferenceTurn.timing}. Flattened onto the record
|
|
163
|
+
* rather than nested, because this record is also a log line and a flat
|
|
164
|
+
* field is what a log query can select. NULL means the runtime reported
|
|
165
|
+
* nothing; it never means zero.
|
|
166
|
+
*/
|
|
167
|
+
readonly queue_ms: number | null;
|
|
168
|
+
readonly prefill_ms: number | null;
|
|
169
|
+
readonly decode_ms: number | null;
|
|
170
|
+
readonly total_ms: number | null;
|
|
171
|
+
/**
|
|
172
|
+
* The identity that served this request, reported by the dial that served it
|
|
173
|
+
* — see {@link TurnCatalogIdentity} and {@link DialReport}. The ledger writer
|
|
174
|
+
* reads it from here rather than asking the router again, and the log line
|
|
175
|
+
* carries it so a row that was never written can be traced to the reason
|
|
176
|
+
* from the same record.
|
|
177
|
+
*/
|
|
178
|
+
readonly catalog: TurnCatalogIdentity;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* THE SHARED-MODEL LANE (shared-endpoints U8, urun-infra#1490 §7) — the pure
|
|
183
|
+
* per-request resolver over the catalog edge function's `shared` block:
|
|
184
|
+
*
|
|
185
|
+
* { shared_org_id, endpoints: [{ model_id, variant, gpu_spec, app_slug,
|
|
186
|
+
* function, warm }] }
|
|
187
|
+
*
|
|
188
|
+
* Every function here is PURE and synchronous over the parsed block: the
|
|
189
|
+
* catalog fetch (U5's edge-function surface) is a separate seam and is NOT
|
|
190
|
+
* built yet — a caller hands the block in, or doesn't. An absent/empty
|
|
191
|
+
* `shared` block means "no shared models" and every lookup says so (null),
|
|
192
|
+
* never a stub.
|
|
193
|
+
*
|
|
194
|
+
* POSTURE (the two error classes, deliberately distinct):
|
|
195
|
+
* - `null` → this model is not in the shared block. The caller
|
|
196
|
+
* keeps its own resolution order; on the hosted lane that ends in the
|
|
197
|
+
* loud UnknownModelError 404. NOT an error — "not mine" is an answer.
|
|
198
|
+
* - `SharedCatalogError` → the block itself is broken (malformed shape, or
|
|
199
|
+
* a bare model id whose compat_default flag is not unique). This is a
|
|
200
|
+
* platform catalog defect, not a caller mistake: it surfaces as a LOUD
|
|
201
|
+
* 500 naming the row, never a silent resolution to "some" variant.
|
|
202
|
+
*/
|
|
203
|
+
/** One shared endpoint row — the edge function's `endpoints[]` element. */
|
|
204
|
+
interface SharedEndpointRow {
|
|
205
|
+
model_id: string;
|
|
206
|
+
variant: string;
|
|
207
|
+
/** The shared org's deployed app slug the session dial targets. */
|
|
208
|
+
app_slug: string;
|
|
209
|
+
/** The serve function name on that app. */
|
|
210
|
+
function: string;
|
|
211
|
+
/** PLACEMENT-level GPU spec, e.g. 'rtx6000:1' (mirrors CatalogRow). */
|
|
212
|
+
gpu_spec?: string | null;
|
|
213
|
+
/** Idle-floor replica count (the reconciler's `warm`, U6). */
|
|
214
|
+
warm?: number | null;
|
|
215
|
+
/**
|
|
216
|
+
* The compat_default flag (U4): a BARE model id resolves to THE flagged
|
|
217
|
+
* row. Exactly one per model_id — zero or two is a loud catalog defect.
|
|
218
|
+
*/
|
|
219
|
+
compat_default?: boolean | null;
|
|
220
|
+
}
|
|
221
|
+
/** The catalog edge function's `shared` block. */
|
|
222
|
+
interface SharedCatalogBlock {
|
|
223
|
+
shared_org_id: string;
|
|
224
|
+
endpoints: SharedEndpointRow[];
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* THE PER-TOKEN PRICE SURFACE — what `GET /v1/models` may say about money.
|
|
229
|
+
*
|
|
230
|
+
* Hugging Face reads our OpenAI-shaped model list to populate its public
|
|
231
|
+
* Inference-Provider comparison table and to power its `:fastest` and
|
|
232
|
+
* `:cheapest` routing. The required per-model shape is
|
|
233
|
+
*
|
|
234
|
+
* "pricing": { "input": <USD per MILLION input tokens>,
|
|
235
|
+
* "output": <USD per MILLION output tokens> }
|
|
236
|
+
*
|
|
237
|
+
* so this module resolves exactly that pair — per catalog model — out of
|
|
238
|
+
* urun-infra's `public.model_prices` surface (shared-endpoints U3,
|
|
239
|
+
* migration 20271003000003), and NOTHING else.
|
|
240
|
+
*
|
|
241
|
+
* OMISSION IS THE CONTRACT. A model with no per-token price row carries NO
|
|
242
|
+
* `pricing` key at all — never `0`, never a GPU-minute rate reinterpreted as
|
|
243
|
+
* a token rate, never an estimate. `model_prices` prices GPU TIME today
|
|
244
|
+
* (`unit in ('per_request','per_minute','per_output_unit')`, and the seeded
|
|
245
|
+
* rows are per_minute GPU-minutes); turning GPU-minutes into per-token rates
|
|
246
|
+
* is a PRICING DECISION a human makes, tracked as ENG-317. A fabricated
|
|
247
|
+
* number here would be a silent lie on a public comparison table that other
|
|
248
|
+
* people's routing decisions depend on, which is strictly worse than an
|
|
249
|
+
* absent field: HF renders an unpriced model as unpriced, and that is true.
|
|
250
|
+
*
|
|
251
|
+
* TWO GAPS THAT MUST CLOSE UPSTREAM BEFORE THIS SURFACE CAN EVER BE
|
|
252
|
+
* NON-EMPTY. Both are recorded on **ENG-317**, which is the durable record
|
|
253
|
+
* — this comment only points at it, because a code comment and a PR body
|
|
254
|
+
* are exactly the channels CLAUDE.md rule 2 names as insufficient:
|
|
255
|
+
*
|
|
256
|
+
* 1. NO PER-TOKEN UNIT EXISTS YET (ENG-317). `model_prices_unit_check`
|
|
257
|
+
* allows only per_request | per_minute | per_output_unit. The two unit
|
|
258
|
+
* strings this module recognizes — {@link PER_MILLION_INPUT_UNIT} and
|
|
259
|
+
* {@link PER_MILLION_OUTPUT_UNIT}, with `price_usd` read VERBATIM as
|
|
260
|
+
* USD per million tokens (no conversion, and `numeric(12,8)` has ample
|
|
261
|
+
* resolution at per-million scale — a per-single-token unit would not)
|
|
262
|
+
* — are a PROPOSED contract ENG-317 must ratify and a migration must
|
|
263
|
+
* add. Until then every live read yields zero per-token rows and every
|
|
264
|
+
* entry is unpriced. (If ENG-317 ratifies different names, a row
|
|
265
|
+
* carrying them is LOUD here, not skipped — see
|
|
266
|
+
* {@link KNOWN_NON_TOKEN_UNITS}.)
|
|
267
|
+
* 2. THE ANON READ IS RLS-BLOCKED (ENG-317). `model_prices` grants
|
|
268
|
+
* `select` to `anon` (migration 20271003000003 line 107), but its only
|
|
269
|
+
* policy (`model_prices_select_all`, line 112) is
|
|
270
|
+
* `for select to authenticated` — so a read with the proxy's catalog
|
|
271
|
+
* anon key returns `[]` whatever the table holds, indistinguishably
|
|
272
|
+
* from "no rows". (Contrast `model_catalog`, whose policy is
|
|
273
|
+
* `to anon, authenticated`; migration 20270226000000 calls that "the
|
|
274
|
+
* one intentional public read".) urun-infra must add the anon policy
|
|
275
|
+
* or publish prices through the catalog edge function.
|
|
276
|
+
*
|
|
277
|
+
* CONSEQUENCE, STATED PLAINLY: no `pricing` field can appear in production
|
|
278
|
+
* today. This module is correct and inert until gap 2 is fixed.
|
|
279
|
+
*
|
|
280
|
+
* WHAT IS A DEFECT (loud) vs WHAT IS SIMPLY UNPRICED (skipped):
|
|
281
|
+
* - a row whose `unit` is not one of the two per-token units → SKIPPED.
|
|
282
|
+
* A GPU-minute or per-request price is not evidence of a token rate; it
|
|
283
|
+
* is not this surface's business.
|
|
284
|
+
* - a row outside its effective window → SKIPPED. That is what
|
|
285
|
+
* effective-dating means.
|
|
286
|
+
* - a model with only ONE of the two rates → that model stays UNPRICED.
|
|
287
|
+
* Half a price is a lie; HF needs both numbers or neither.
|
|
288
|
+
* - a row that DOES claim a per-token unit but cannot be read (no
|
|
289
|
+
* model_id, unparseable price/date, a missing column, a non-object row,
|
|
290
|
+
* two rows of the same side with the same winning `effective_at` and
|
|
291
|
+
* DIFFERENT prices) → {@link ModelPriceError}. A broken price on a
|
|
292
|
+
* public table is worse than no price, and an order-dependent pick
|
|
293
|
+
* between two same-instant rows would be exactly the silent guess this
|
|
294
|
+
* repo forbids.
|
|
295
|
+
*
|
|
296
|
+
* THE ERROR TAXONOMY MATTERS TO CALLERS, so it is deliberate and narrow:
|
|
297
|
+
* {@link ModelPriceError} means THE DATA IS A DEFECT and no honest price can
|
|
298
|
+
* be derived — the /v1/models lane lets it PROPAGATE (a self-contradictory
|
|
299
|
+
* price table must be seen and fixed, not rounded down to "unpriced"). A
|
|
300
|
+
* plain `Error` from {@link fetchModelPriceRows} means the surface could
|
|
301
|
+
* not be READ (transport/status) — that one the listing may degrade past,
|
|
302
|
+
* because model DISCOVERY must not die with the price oracle. Those are the
|
|
303
|
+
* only two failure modes; nothing here returns an empty list to paper over
|
|
304
|
+
* either.
|
|
305
|
+
*
|
|
306
|
+
* THE SURFACE IS READ IN TWO STEPS, and the split is load-bearing:
|
|
307
|
+
* {@link parseModelPriceRows} validates the wire rows (timeless facts — a
|
|
308
|
+
* caller may CACHE these), and {@link resolveTokenPrices} answers which of
|
|
309
|
+
* them is in force AT AN INSTANT (a per-REQUEST question — caching THAT
|
|
310
|
+
* would serve a rate past its own `expires_at`). See each function's note.
|
|
311
|
+
*/
|
|
312
|
+
|
|
313
|
+
/** USD per 1,000,000 tokens — verbatim the HF Inference-Provider shape. */
|
|
314
|
+
interface TokenPricing {
|
|
315
|
+
input: number;
|
|
316
|
+
output: number;
|
|
317
|
+
}
|
|
318
|
+
/** Which half of the pair a recognized unit names. */
|
|
319
|
+
type Side = 'input' | 'output';
|
|
320
|
+
/**
|
|
321
|
+
* ONE VALIDATED per-token price row — the cacheable, TIME-INDEPENDENT half
|
|
322
|
+
* of this surface. Everything here is a fact about the row itself; nothing
|
|
323
|
+
* about WHICH row is in force, because that depends on when you ask (see
|
|
324
|
+
* {@link resolveTokenPrices}). The read/validate step yields these; the
|
|
325
|
+
* time-dependent step consumes them.
|
|
326
|
+
*/
|
|
327
|
+
interface ModelPriceRow {
|
|
328
|
+
model_id: string;
|
|
329
|
+
/** null = the model-level rate for every variant without its own row. */
|
|
330
|
+
variant: string | null;
|
|
331
|
+
side: Side;
|
|
332
|
+
/** USD per 1,000,000 tokens, verbatim from `price_usd`. */
|
|
333
|
+
usd: number;
|
|
334
|
+
/** Epoch ms of `effective_at`. */
|
|
335
|
+
effective_at: number;
|
|
336
|
+
/** Epoch ms of `expires_at`, or null for a row that never expires. */
|
|
337
|
+
expires_at: number | null;
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* THE OPENROUTER PROVIDER DOCUMENT — `GET /v1/models?format=openrouter`.
|
|
342
|
+
*
|
|
343
|
+
* OpenRouter's provider monitor polls this document (schema 2.4, per
|
|
344
|
+
* openrouter.ai/docs/guides/community/for-providers §1) to list a provider's
|
|
345
|
+
* models on the marketplace. The document answers ONE question: "all models
|
|
346
|
+
* that should be served by OpenRouter" — the platform's SELLABLE surface.
|
|
347
|
+
* That surface is the catalog edge function's SHARED block (the shared
|
|
348
|
+
* endpoints OpenRouter's callers resolve by `model_id:variant` on the chat
|
|
349
|
+
* surface); caller-org apps are deliberately NOT in it. A caller-org app is
|
|
350
|
+
* a tenancy-private chat surface behind one org's key — declaring it here
|
|
351
|
+
* would sell another tenant's private deployment as a public marketplace
|
|
352
|
+
* SKU. The document still rides the tenancy seam (`openRouterModels` on the
|
|
353
|
+
* tenant client), so each caller sees the same sellable listing.
|
|
354
|
+
*
|
|
355
|
+
* THE SHARED JOIN: each endpoint joins the catalog EXACTLY on
|
|
356
|
+
* (model_id, variant) — the SAME join the shared-lane consumers (supportsAudio,
|
|
357
|
+
* modelList's shared arm, imageModelList) already apply. No slug heuristics: the
|
|
358
|
+
* shared org's app slug need not exist in
|
|
359
|
+
* the caller's catalog, and a slug guess would be a second, bespoke join.
|
|
360
|
+
*
|
|
361
|
+
* MODALITY GATING: only rows whose catalog `task` names a chat-completions
|
|
362
|
+
* shape (chat | code | agent | vl) are declared. A voice/video/image app
|
|
363
|
+
* behind a chat surface would stream garbage into OpenRouter's baseline
|
|
364
|
+
* tests; models the catalog cannot vouch for are OMITTED, never guessed.
|
|
365
|
+
* With no shared block (or no catalog oracle) configured the document is
|
|
366
|
+
* `{"data": []}` — honest emptiness, not invented capability.
|
|
367
|
+
*
|
|
368
|
+
* PRICING — schema 2.4 nests `pricing` arrays on the modality that owns
|
|
369
|
+
* them ({type 'prompt'} on the text INPUT modality, {type 'completion'} on
|
|
370
|
+
* the text OUTPUT modality, `cost_usd` a USD string PER TOKEN). The price
|
|
371
|
+
* source is model-prices.ts — the SAME per-token join the OpenAI listing
|
|
372
|
+
* uses ({@link pricingFor}); the per-million rate is shifted to per-token
|
|
373
|
+
* on its decimal string, never by float division. An entry with no
|
|
374
|
+
* per-token row — today ALL of them, because `model_prices` prices GPU
|
|
375
|
+
* minutes (`per_minute` et al.), which is not evidence about tokens —
|
|
376
|
+
* carries NO pricing array at all: the schema's rule is "a modality with
|
|
377
|
+
* no pricing array is simply unpriced". Never a zero, never a GPU-minute
|
|
378
|
+
* rate relabelled as a token rate (ENG-317 owns the human pricing
|
|
379
|
+
* decision that would make such rows exist).
|
|
380
|
+
*
|
|
381
|
+
* The full closed-value-domain schema ships as OpenAPI 3.1 at
|
|
382
|
+
* openrouter.ai/docs/assets/provider-monitor-schema-v2.openapi.json.
|
|
383
|
+
*/
|
|
384
|
+
|
|
385
|
+
type OpenRouterProviderDoc = {
|
|
386
|
+
data: OpenRouterProviderModel[];
|
|
387
|
+
};
|
|
388
|
+
interface OpenRouterProviderModel {
|
|
389
|
+
schema_version: '2.4';
|
|
390
|
+
/** The EXACT id OpenRouter sends back as `model` — the app slug. */
|
|
391
|
+
id: string;
|
|
392
|
+
name: string;
|
|
393
|
+
created: number;
|
|
394
|
+
/** Valid enum: int4|int8|fp4|mxfp4|nvfp4|fp6|fp8|mxfp8|fp16|bf16|fp32|null. */
|
|
395
|
+
quantization: string | null;
|
|
396
|
+
description: string;
|
|
397
|
+
hugging_face_id: string;
|
|
398
|
+
input_modalities: Array<Record<string, unknown>>;
|
|
399
|
+
output_modalities: Array<Record<string, unknown>>;
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
/**
|
|
403
|
+
* Per-model → per-app routing for the compat proxy (owner directive
|
|
404
|
+
* 2026-08-07): swapping the model in a coding harness routes the request to
|
|
405
|
+
* the org's DEPLOYED app for that model. v1 is deployed-only — a uRun model
|
|
406
|
+
* that is not deployed gets a loud 404 naming `urun serve <id>`; a later
|
|
407
|
+
* phase (explicitly out of scope here; urun-infra#1490 shared-endpoints)
|
|
408
|
+
* auto-creates from the model catalog on first request.
|
|
409
|
+
*
|
|
410
|
+
* MODEL-ID SURFACE (the documented mapping): a model may be named by
|
|
411
|
+
* - the app slug itself ("qwen3-6-27b-bf16"), or
|
|
412
|
+
* - the catalog id ("qwen3.6-27b"), or
|
|
413
|
+
* - the catalog id:variant ("qwen3.6-27b:bf16"),
|
|
414
|
+
* where slugification mirrors urun-cli `serve.py _default_app_name` exactly:
|
|
415
|
+
* lowercase, every non-alphanumeric-non-dash character becomes "-", leading/
|
|
416
|
+
* trailing dashes stripped (catalog id "qwen3.6-27b" + variant "bf16" → app
|
|
417
|
+
* "qwen3-6-27b-bf16"). COLLISION RULE: an exact slug match always wins over
|
|
418
|
+
* the catalog-id (prefix) interpretation.
|
|
419
|
+
*
|
|
420
|
+
* RESOLUTION ORDER (one canonical path, documented end to end):
|
|
421
|
+
* 1. model absent / "urun" / an alias of the startup app → DEFAULT app.
|
|
422
|
+
* 2. exact slug match on a deployed serve app → that app.
|
|
423
|
+
* 3. catalog-id form matching exactly one deployed app → that app
|
|
424
|
+
* (two or more candidates → loud ambiguity error naming them).
|
|
425
|
+
* 4. the name maps to an org app that is NOT an active app exposing the
|
|
426
|
+
* proxy's serve function → loud 404
|
|
427
|
+
* naming `urun serve <model>` and the available models.
|
|
428
|
+
* 5. the name matches a catalog row but no deployed app → loud 404
|
|
429
|
+
* naming `urun serve <id>` (deployed-only v1).
|
|
430
|
+
* 6. anything else — a model name outside the uRun namespace entirely
|
|
431
|
+
* (e.g. the harness's own upstream default, "claude-*"/"gpt-*") →
|
|
432
|
+
* DEFAULT app. This IS today's single-app contract, kept deliberately
|
|
433
|
+
* so `urun compat <agent>` with the agent's stock model keeps working
|
|
434
|
+
* with zero new env; the per-model /stats table records every such
|
|
435
|
+
* mapping so it is visible, never silent. Models the proxy ADVERTISES
|
|
436
|
+
* on /v1/models can never land here — they resolve (2/3) or fail loud
|
|
437
|
+
* (4/5) above.
|
|
438
|
+
*
|
|
439
|
+
* NO-DEFAULT MODE (`defaultApp: null`) — the HOSTED multi-tenant lane
|
|
440
|
+
* (`src/hosted/`): a shared endpoint serving every org has no "the app this
|
|
441
|
+
* proxy was started for", so rules 1 and 6 have nothing to fall back TO.
|
|
442
|
+
* Rather than inventing one (picking "some" app for a caller would be the
|
|
443
|
+
* worst kind of silent divergence), both rules become the SAME loud
|
|
444
|
+
* {@link UnknownModelError} that rules 4/5 already raise: name a deployed
|
|
445
|
+
* model, here is the list. Rules 2–5 are byte-for-byte the local behavior —
|
|
446
|
+
* one router, one resolution order, two configurations.
|
|
447
|
+
*/
|
|
448
|
+
|
|
449
|
+
/** One org app row from `GET {orgApi}/apps` (urun-cli `ApiClient.list_apps`). */
|
|
450
|
+
interface DeployedApp {
|
|
451
|
+
app_slug: string;
|
|
452
|
+
function_name?: string | null;
|
|
453
|
+
deployment_status?: string | null;
|
|
454
|
+
[k: string]: unknown;
|
|
455
|
+
}
|
|
456
|
+
/**
|
|
457
|
+
* A model that does not resolve to a deployed app the proxy may serve.
|
|
458
|
+
* Rendered by the server in each lane's NATIVE error format as a 404 — never
|
|
459
|
+
* silently served by the default app.
|
|
460
|
+
*/
|
|
461
|
+
declare class UnknownModelError extends Error {
|
|
462
|
+
}
|
|
463
|
+
/**
|
|
464
|
+
* A session handle names a pooled session that no longer exists (closed,
|
|
465
|
+
* evicted after its pod died, replaced by a re-home, or the proxy restarted).
|
|
466
|
+
* The caller asked to REATTACH that exact session — opening a fresh one and
|
|
467
|
+
* calling it "resumed" would be a silent lie, so this is always loud.
|
|
468
|
+
*/
|
|
469
|
+
declare class SessionGoneError extends Error {
|
|
470
|
+
}
|
|
471
|
+
/**
|
|
472
|
+
* OpenAI-shaped model list (same shape as models.ts listModels). When the
|
|
473
|
+
* catalog oracle is configured, each entry ALSO carries the catalog
|
|
474
|
+
* enrichment fields (additive JSON — OpenAI clients ignore unknown fields;
|
|
475
|
+
* the OpenRouter + Vercel AI Gateway provider listings need them). Entries
|
|
476
|
+
* whose slug matches no catalog row stay at the bare four fields.
|
|
477
|
+
*/
|
|
478
|
+
interface RouterModelEntry {
|
|
479
|
+
id: string;
|
|
480
|
+
object: 'model';
|
|
481
|
+
created: number;
|
|
482
|
+
owned_by: string;
|
|
483
|
+
/** Display name — the canonical `<model_id>:<variant>` catalog ref. */
|
|
484
|
+
name?: string;
|
|
485
|
+
/** User-facing description from the catalog (console Endpoints copy). */
|
|
486
|
+
description?: string;
|
|
487
|
+
/** Catalog modality (chat | code | agent | vl | audio | image | ...). */
|
|
488
|
+
task?: string;
|
|
489
|
+
/** Serving engine (vllm | sglang | llamacpp | ...). */
|
|
490
|
+
engine?: string;
|
|
491
|
+
/** The catalog lane this app deploys, e.g. 'rtx6000:1'. */
|
|
492
|
+
gpu_spec?: string;
|
|
493
|
+
/** Context window in tokens (chat rows carry it in engine_args). */
|
|
494
|
+
context_length?: number;
|
|
495
|
+
/**
|
|
496
|
+
* Per-token price, USD per MILLION tokens — the exact shape Hugging Face
|
|
497
|
+
* reads off `GET /v1/models` for its provider comparison table and its
|
|
498
|
+
* `:fastest`/`:cheapest` routing. ABSENT means UNPRICED and is the honest
|
|
499
|
+
* answer for every model with no per-token price row (model-prices.ts);
|
|
500
|
+
* a zero or an estimate here would be a silent lie on a public table.
|
|
501
|
+
*/
|
|
502
|
+
pricing?: TokenPricing;
|
|
503
|
+
/** Idle-floor replica count for a shared-lane model (U6 reconciler). */
|
|
504
|
+
warm?: number;
|
|
505
|
+
}
|
|
506
|
+
interface RouterModelList {
|
|
507
|
+
object: 'list';
|
|
508
|
+
data: RouterModelEntry[];
|
|
509
|
+
}
|
|
510
|
+
/** One `/v1/images/models` entry: the OpenAI model object plus the image
|
|
511
|
+
* modes the catalog rows vouch for (the SAME per-placement consensus
|
|
512
|
+
* {@link ModelRouter.supportsImages} applies — a row set that vouches
|
|
513
|
+
* nothing is NOT listed). */
|
|
514
|
+
interface RouterImageModelEntry extends RouterModelEntry {
|
|
515
|
+
image_modes: ImageCapabilityMode[];
|
|
516
|
+
}
|
|
517
|
+
interface RouterImageModelList {
|
|
518
|
+
object: 'list';
|
|
519
|
+
data: RouterImageModelEntry[];
|
|
520
|
+
}
|
|
521
|
+
/**
|
|
522
|
+
* The dial target one resolved model names: the app slug plus — for a SHARED
|
|
523
|
+
* lane (the catalog `shared` block, the cross-org session dial) — the exact
|
|
524
|
+
* serve function on the shared org's app and the flag that switches the
|
|
525
|
+
* control plane's shared-admission carve-out on. A plain caller-org app
|
|
526
|
+
* carries only `appSlug`; the router's own `fnName` applies there.
|
|
527
|
+
*/
|
|
528
|
+
interface RouteTarget {
|
|
529
|
+
appSlug: string;
|
|
530
|
+
/** The serve function on the TARGET app (a shared lane pins it per row). */
|
|
531
|
+
fnName?: string;
|
|
532
|
+
/**
|
|
533
|
+
* CROSS-ORG SHARED DIAL: the backhaul mints the client token with
|
|
534
|
+
* `sharedApp: true` and starts the session with `shared_app: true`, so the
|
|
535
|
+
* control plane resolves the app from the platform's shared org while the
|
|
536
|
+
* session and its usage stay attributed to the CALLER org.
|
|
537
|
+
*/
|
|
538
|
+
sharedApp?: boolean;
|
|
539
|
+
/** Shared lanes only: the catalog id the request pinned (capability joins). */
|
|
540
|
+
modelId?: string;
|
|
541
|
+
/** Shared lanes only: the catalog variant the request pinned. */
|
|
542
|
+
variant?: string;
|
|
543
|
+
}
|
|
544
|
+
interface ModelRouterOptions<S> {
|
|
545
|
+
/**
|
|
546
|
+
* The startup app slug (URUN_APP) — the DEFAULT model — or `null` for the
|
|
547
|
+
* hosted multi-tenant lane, which has no per-proxy default app: there,
|
|
548
|
+
* every request must NAME a deployed model and an unnamed/unknown one
|
|
549
|
+
* fails loud instead of silently landing somewhere (see the module header,
|
|
550
|
+
* "NO-DEFAULT MODE").
|
|
551
|
+
*/
|
|
552
|
+
defaultApp: string | null;
|
|
553
|
+
/** The serve function name every routed app must expose (URUN_FUNCTION). */
|
|
554
|
+
fnName: string;
|
|
555
|
+
/** Open a backhaul session for a resolved dial target (called at most once per pool key). */
|
|
556
|
+
openSession: (target: RouteTarget) => S | Promise<S>;
|
|
557
|
+
/** Terminal release for one pool entry (Session.end() underneath). */
|
|
558
|
+
closeSession: (entry: S) => Promise<void>;
|
|
559
|
+
/**
|
|
560
|
+
* List the org's deployed apps, or null when the credentials cannot
|
|
561
|
+
* (URUN_JWT lane: the pre-vended token is scoped to the default app, so
|
|
562
|
+
* there is no org listing AND no cross-app session — routing degrades to
|
|
563
|
+
* the default-app-only contract, which is exactly today's behavior).
|
|
564
|
+
*/
|
|
565
|
+
listApps: (() => Promise<DeployedApp[]>) | null;
|
|
566
|
+
/**
|
|
567
|
+
* Catalog rows (models.ts fetchCatalogRows) as the uRun-namespace oracle
|
|
568
|
+
* for rule 5 AND the /v1/models enrichment + OpenRouter provider-doc
|
|
569
|
+
* source, or null when catalog access is not configured. Optional fields
|
|
570
|
+
* beyond model_id/variant are tolerated (thin rows still typecheck).
|
|
571
|
+
*/
|
|
572
|
+
listCatalog: (() => Promise<CatalogRow[]>) | null;
|
|
573
|
+
/**
|
|
574
|
+
* The VALIDATED per-token price rows (model-prices.ts
|
|
575
|
+
* `fetchModelPriceRows`) behind the `pricing` field of the /v1/models
|
|
576
|
+
* listing, or null when no price source is configured. A REQUIRED key
|
|
577
|
+
* exactly like {@link listCatalog}: an optional one would let a lane ship
|
|
578
|
+
* a silently unpriced public listing without ever saying so.
|
|
579
|
+
*
|
|
580
|
+
* These are ROWS, not resolved prices, precisely so the router may cache
|
|
581
|
+
* them: which row is in force is a per-REQUEST question answered by
|
|
582
|
+
* `resolveTokenPrices` against the exact `effective_at` / `expires_at`
|
|
583
|
+
* boundaries. Handing back already-resolved prices would let a cached
|
|
584
|
+
* answer outlive the window it was resolved in.
|
|
585
|
+
*/
|
|
586
|
+
listPrices: (() => Promise<ModelPriceRow[]>) | null;
|
|
587
|
+
/**
|
|
588
|
+
* Where this router announces a DEGRADATION — an unreadable price
|
|
589
|
+
* surface, a model refused a price, a catalog blip that suppresses
|
|
590
|
+
* pricing. Optional only in WHERE it goes: left unset the router writes
|
|
591
|
+
* to `console.warn`, never to nothing (see {@link ModelRouter.warn}).
|
|
592
|
+
* Mirrors the `warn` sink `catalogFromEnv` (cli.ts) already takes.
|
|
593
|
+
*/
|
|
594
|
+
warn?: (line: string) => void;
|
|
595
|
+
/** Deployed-apps cache TTL (the list changes on deploys, not per request). */
|
|
596
|
+
appsTtlMs?: number;
|
|
597
|
+
/**
|
|
598
|
+
* The stable NATIVE identity of one pooled entry (the uRun session id in
|
|
599
|
+
* the proxy wiring — the same identity the serve-side session-affinity tag
|
|
600
|
+
* rides, urun-python#1556). Powers the session-identity seam
|
|
601
|
+
* ({@link ModelRouter.handleFor} / {@link ModelRouter.sessionForHandle});
|
|
602
|
+
* a router without it fails LOUD on those calls, never approximates.
|
|
603
|
+
*/
|
|
604
|
+
sessionKey?: (entry: S) => string;
|
|
605
|
+
/**
|
|
606
|
+
* The catalog edge function's `shared` block (shared-endpoints U8), or
|
|
607
|
+
* null when shared-model routing is not configured — every phase-1 caller
|
|
608
|
+
* leaves it absent and behaves byte-for-byte as before. Same TTL cache
|
|
609
|
+
* discipline as the apps/catalog oracles.
|
|
610
|
+
*/
|
|
611
|
+
shared?: (() => Promise<SharedCatalogBlock | null>) | null;
|
|
612
|
+
}
|
|
613
|
+
/**
|
|
614
|
+
* The image-capability modes the Images lane gates on (images.ts's
|
|
615
|
+
* ImageCapabilityMode — declared here because the router is the LOWER layer:
|
|
616
|
+
* the lane's gate type stays structurally satisfied by
|
|
617
|
+
* {@link ModelRouter.supportsImages} with no router→lane import).
|
|
618
|
+
*/
|
|
619
|
+
type ImageCapabilityMode = 'generation' | 'edit';
|
|
620
|
+
/**
|
|
621
|
+
* The session pool: one backhaul session per deployed app, keyed by app slug,
|
|
622
|
+
* opened lazily on the first request that routes to it and reused for every
|
|
623
|
+
* subsequent one. The startup app is seeded eagerly by the CLI. Sessions
|
|
624
|
+
* close on proxy shutdown via {@link closeAll}; there is NO idle-close policy
|
|
625
|
+
* (deliberate v1 simplification — noted as a follow-up in the PR).
|
|
626
|
+
*/
|
|
627
|
+
declare class ModelRouter<S> {
|
|
628
|
+
private readonly opts;
|
|
629
|
+
private readonly pool;
|
|
630
|
+
private appsCache;
|
|
631
|
+
private rowsCache;
|
|
632
|
+
constructor(opts: ModelRouterOptions<S>);
|
|
633
|
+
/** Seed an already-open session (the CLI's eagerly-opened startup app). */
|
|
634
|
+
seed(appSlug: string, entry: S): void;
|
|
635
|
+
private deployedApps;
|
|
636
|
+
/**
|
|
637
|
+
* The shared block with the same TTL discipline as {@link deployedApps}.
|
|
638
|
+
* Null when not configured. Malformed blocks throw loudly
|
|
639
|
+
* (SharedCatalogError) on first use — never a silently half-empty table.
|
|
640
|
+
*/
|
|
641
|
+
private sharedBlock;
|
|
642
|
+
private sharedCache;
|
|
643
|
+
/**
|
|
644
|
+
* Catalog rows with the same TTL discipline as {@link deployedApps} (the
|
|
645
|
+
* catalog changes on reseed migrations, not per request). Null when no
|
|
646
|
+
* oracle is configured — callers degrade to the bare four-field entries.
|
|
647
|
+
*/
|
|
648
|
+
private catalogRows;
|
|
649
|
+
private pricesCache;
|
|
650
|
+
/**
|
|
651
|
+
* The per-token price ROWS with the same TTL discipline as
|
|
652
|
+
* {@link catalogRows} (the price book changes on ops writes, not per
|
|
653
|
+
* request). No source configured ⇒ [] — every model unpriced, which is
|
|
654
|
+
* the honest listing until per-token rows exist (ENG-317).
|
|
655
|
+
*
|
|
656
|
+
* ROWS, never resolved prices: the rows are timeless facts, so caching
|
|
657
|
+
* them is safe, whereas caching the prices IN FORCE would keep serving a
|
|
658
|
+
* rate past its own `expires_at` — or withhold one past its
|
|
659
|
+
* `effective_at` — for the rest of the TTL window. The in-force question
|
|
660
|
+
* is answered per request in {@link pricesForListing}.
|
|
661
|
+
*/
|
|
662
|
+
private priceRows;
|
|
663
|
+
/**
|
|
664
|
+
* The price surface for the /v1/models listing. Three distinct outcomes,
|
|
665
|
+
* and NONE of them is silent:
|
|
666
|
+
*
|
|
667
|
+
* - the surface cannot be READ (transport, status, or the read's own
|
|
668
|
+
* deadline — a plain Error) ⇒ every model lists UNPRICED and the
|
|
669
|
+
* reason is WARNED. Model discovery must not die with the price
|
|
670
|
+
* oracle, but an outage rendering as an ordinary unpriced listing —
|
|
671
|
+
* indistinguishable from today's inert state — is exactly the silent
|
|
672
|
+
* degradation CLAUDE.md forbids, so it says so out loud.
|
|
673
|
+
* - one model's rows are a DEFECT (a duplicate rate for one instant, a
|
|
674
|
+
* zero rate) ⇒ that MODEL is unpriced and the refusal is WARNED. The
|
|
675
|
+
* blast radius is the model the bad row belongs to; it used to be the
|
|
676
|
+
* whole listing for every tenant.
|
|
677
|
+
* - the READ itself is not the surface we contracted for (a malformed
|
|
678
|
+
* row, a missing column, an unknown `unit` — {@link ModelPriceError})
|
|
679
|
+
* ⇒ PROPAGATES. Nothing in a payload that violates its own column
|
|
680
|
+
* contract can be trusted, so half a price table is not served as if
|
|
681
|
+
* complete — the same posture `parseSharedBlock` takes on a malformed
|
|
682
|
+
* routing block.
|
|
683
|
+
*
|
|
684
|
+
* The resolve step sits OUTSIDE the catch deliberately: it is our own
|
|
685
|
+
* code, so a programming error in it must surface as a crash, not become
|
|
686
|
+
* an empty price list.
|
|
687
|
+
*/
|
|
688
|
+
private pricesForListing;
|
|
689
|
+
/**
|
|
690
|
+
* Where a degradation says so. Defaults to `console.warn` rather than to
|
|
691
|
+
* nothing: a caller may ROUTE the signal (the CLI sends it to stderr, as
|
|
692
|
+
* `catalogFromEnv` already does), but no caller can switch it off, because
|
|
693
|
+
* an unannounced degradation is the failure mode this repo treats as a
|
|
694
|
+
* time bomb.
|
|
695
|
+
*/
|
|
696
|
+
private warn;
|
|
697
|
+
/** Apps this proxy may serve: active AND exposing the serve function. */
|
|
698
|
+
private servable;
|
|
699
|
+
private availableIds;
|
|
700
|
+
/**
|
|
701
|
+
* NO-DEFAULT MODE's terminal for rules 1 and 6: there is no app to fall
|
|
702
|
+
* back to, so say so loudly and list what the CALLER'S org actually has.
|
|
703
|
+
* Never returns.
|
|
704
|
+
*/
|
|
705
|
+
private noDefaultApp;
|
|
706
|
+
/**
|
|
707
|
+
* Resolve a request's `model` to its dial target — the documented
|
|
708
|
+
* resolution order from the module header. Throws
|
|
709
|
+
* {@link UnknownModelError} for a uRun model that is not deployed (rules
|
|
710
|
+
* 4/5). A SHARED-lane hit (the catalog `shared` block) resolves to the
|
|
711
|
+
* SHARED org's app with `sharedApp: true` — the cross-org session dial.
|
|
712
|
+
*/
|
|
713
|
+
private resolveTarget;
|
|
714
|
+
/**
|
|
715
|
+
* THE BILLING IDENTITY of a request's model: the CATALOG `(model_id,
|
|
716
|
+
* variant)` the request actually resolved to, or `null` when the route
|
|
717
|
+
* names no catalog model.
|
|
718
|
+
*
|
|
719
|
+
* WHY THE LEDGER CANNOT USE THE CALLER'S STRING INSTEAD. The per-request
|
|
720
|
+
* ledger prices on `(model_id, variant)` (urun-infra
|
|
721
|
+
* `urun_record_inference_request`), and the caller's `model` is not that:
|
|
722
|
+
* a BARE catalog id resolves through {@link resolveSharedModel} to the
|
|
723
|
+
* model's `compat_default` VARIANT, so recording the raw string would file
|
|
724
|
+
* the request under a variant nobody served and price it off the
|
|
725
|
+
* model-level wildcard — a rate that exists precisely to price the variants
|
|
726
|
+
* nobody named explicitly. Splitting the string on its last `:` would be a
|
|
727
|
+
* heuristic on a money path. This is the resolution the dial itself used.
|
|
728
|
+
*
|
|
729
|
+
* NULL FOR A CALLER-ORG APP, and that is a REFUSAL rather than a gap. An
|
|
730
|
+
* org's own deployed app has no catalog identity at all — its identity is
|
|
731
|
+
* an app slug, which is a different namespace from `model_prices.model_id`
|
|
732
|
+
* — so there is no honest `(model_id, variant)` to record. The ledger
|
|
733
|
+
* writer declines to write such a request rather than inventing one; see
|
|
734
|
+
* `proxy/ledger.ts`.
|
|
735
|
+
*
|
|
736
|
+
* CHEAP TO CALL: it rides {@link resolveTarget}, whose app listing, catalog
|
|
737
|
+
* and shared block are all cached on this router, so the ledger writer can
|
|
738
|
+
* ask AFTER the response has closed without re-hitting the control plane.
|
|
739
|
+
*/
|
|
740
|
+
catalogIdentityFor(model: string | undefined): Promise<{
|
|
741
|
+
model_id: string;
|
|
742
|
+
variant: string;
|
|
743
|
+
} | null>;
|
|
744
|
+
/**
|
|
745
|
+
* THE CALLER-ORG APP LANE's OWN IDENTITY (ENG-412) — the app slug this
|
|
746
|
+
* request actually resolves to.
|
|
747
|
+
*
|
|
748
|
+
* The ledger records it for the lane {@link catalogIdentityFor} answers null
|
|
749
|
+
* for, and the caller's RAW `model` string is not it: routing trims, caps and
|
|
750
|
+
* matches that string before selecting an app, so two spellings of one app
|
|
751
|
+
* would otherwise be filed as two identities and no app-level reconciliation
|
|
752
|
+
* could undo it afterwards.
|
|
753
|
+
*
|
|
754
|
+
* Rides {@link resolveTarget}'s cached listings and OPENS NO SESSION.
|
|
755
|
+
*/
|
|
756
|
+
servingAppSlugFor(model: string | undefined): Promise<string>;
|
|
757
|
+
/**
|
|
758
|
+
* Resolve a request's `model` to its POOL KEY — the app slug for a
|
|
759
|
+
* caller-org app, `shared:<slug>` for a shared-lane dial (the two never
|
|
760
|
+
* share a pool slot, {@link poolKeyOf}).
|
|
761
|
+
*/
|
|
762
|
+
resolveApp(model: string | undefined): Promise<string>;
|
|
763
|
+
/**
|
|
764
|
+
* The pooled session for a model — opened lazily, reused afterwards.
|
|
765
|
+
*
|
|
766
|
+
* IT ALSO REPORTS THE DIAL'S BILLING IDENTITY (ENG-414), off the SAME
|
|
767
|
+
* `resolveTarget` answer it just selected the session with. That is the
|
|
768
|
+
* whole point of returning it from here: the ledger must be able to name
|
|
769
|
+
* what served a request WITHOUT taking a second resolution. This router
|
|
770
|
+
* reads its app list, catalog and shared block through caches with a TTL, so
|
|
771
|
+
* a second resolution taken microseconds later can answer differently — and
|
|
772
|
+
* then the row names a variant nobody served. `rehome.ts` reports this
|
|
773
|
+
* identity once per dial, so a request that re-homes is billed on the
|
|
774
|
+
* REPLACEMENT's identity rather than the abandoned dial's.
|
|
775
|
+
*
|
|
776
|
+
* THE POOL ENTRY COULD NOT CARRY IT. `S` is the generic pooled session and
|
|
777
|
+
* knows nothing about models; the identity is a property of the ROUTE, which
|
|
778
|
+
* is what this method resolved and the entry never saw.
|
|
779
|
+
*/
|
|
780
|
+
sessionFor(model: string | undefined): Promise<{
|
|
781
|
+
app: string;
|
|
782
|
+
entry: S;
|
|
783
|
+
identity: TurnCatalogIdentity;
|
|
784
|
+
}>;
|
|
785
|
+
private keyOf;
|
|
786
|
+
/**
|
|
787
|
+
* SESSION-IDENTITY SEAM (a): the opaque stable handle for the pooled
|
|
788
|
+
* session currently serving `model`'s turns. Rides the SAME acquisition
|
|
789
|
+
* path as every request ({@link sessionFor}) — the session opens lazily if
|
|
790
|
+
* this model has none yet — and derives the handle from native identity
|
|
791
|
+
* (app slug + uRun session id), zero bespoke bookkeeping.
|
|
792
|
+
*/
|
|
793
|
+
handleFor(model: string | undefined): Promise<{
|
|
794
|
+
app: string;
|
|
795
|
+
handle: string;
|
|
796
|
+
}>;
|
|
797
|
+
/**
|
|
798
|
+
* SESSION-IDENTITY SEAM (b): the exact pooled session a handle names.
|
|
799
|
+
* NEVER opens a fresh session — a handle whose session is gone (closed,
|
|
800
|
+
* evicted, re-homed to a replacement, proxy restarted) or malformed throws
|
|
801
|
+
* {@link SessionGoneError} loudly. Resume is reattach-or-fail, not
|
|
802
|
+
* reattach-or-quietly-restart.
|
|
803
|
+
*/
|
|
804
|
+
sessionForHandle(handle: string): Promise<{
|
|
805
|
+
app: string;
|
|
806
|
+
entry: S;
|
|
807
|
+
}>;
|
|
808
|
+
/**
|
|
809
|
+
* Drop ONE pooled session whose backhaul died (its pod was restarted /
|
|
810
|
+
* drained / deleted) and release it — the next {@link sessionFor} opens a
|
|
811
|
+
* fresh one, i.e. asks the control plane for a new assignment. Used by the
|
|
812
|
+
* one-shot re-home (rehome.ts, urun-sh/urun-python#1592).
|
|
813
|
+
*
|
|
814
|
+
* IDENTITY-GUARDED (the same rule the pi lane's SessionPool follows): a
|
|
815
|
+
* concurrent request that already re-homed this app has put a NEWER entry
|
|
816
|
+
* under the key, and evicting that would close a healthy session out from
|
|
817
|
+
* under it.
|
|
818
|
+
*/
|
|
819
|
+
evict(app: string, entry: S): Promise<void>;
|
|
820
|
+
/**
|
|
821
|
+
* `GET /v1/models`: the org's deployed serve apps as model entries, the
|
|
822
|
+
* default app FIRST. On the JWT lane (no org listing) this is the default
|
|
823
|
+
* app plus any app already in the pool — the gap is called out loudly in
|
|
824
|
+
* the PR, not papered over here.
|
|
825
|
+
*
|
|
826
|
+
* THE HUGGING FACE FIELDS: every entry carries `context_length` (from the
|
|
827
|
+
* catalog's engine_args, omitted when unknown or when the placements
|
|
828
|
+
* disagree) and `pricing` (USD per MILLION input/output tokens, omitted
|
|
829
|
+
* ENTIRELY when the model has no per-token price row — see
|
|
830
|
+
* model-prices.ts; ENG-317 owns the pricing decision that makes such rows
|
|
831
|
+
* exist). HF reads both to build its public provider comparison table.
|
|
832
|
+
* A price surface that cannot be READ lists everything unpriced; a price
|
|
833
|
+
* surface that is a DEFECT fails this call loudly (see below).
|
|
834
|
+
*/
|
|
835
|
+
modelList(): Promise<RouterModelList>;
|
|
836
|
+
/**
|
|
837
|
+
* The OpenRouter PROVIDER document (`GET /v1/models?format=openrouter`):
|
|
838
|
+
* schema 2.4 per openrouter.ai/docs/guides/community/for-providers §1 —
|
|
839
|
+
* "an endpoint that returns all models that should be served by
|
|
840
|
+
* OpenRouter". That is the SHARED block's sellable endpoints (the models
|
|
841
|
+
* callers resolve by `model_id:variant` on the chat surface), joined to
|
|
842
|
+
* the catalog EXACTLY on (model_id, variant) — the SAME join
|
|
843
|
+
* {@link supportsAudio} applies — and gated to chat-completions shape
|
|
844
|
+
* (task chat | code | agent | vl; a voice or video app behind a chat
|
|
845
|
+
* surface would stream garbage). Caller-org apps are deliberately NOT
|
|
846
|
+
* listed: they are tenancy-private chat surfaces, and a provider
|
|
847
|
+
* document that published them would sell another tenant's private
|
|
848
|
+
* deployment as a public marketplace SKU. Pricing follows
|
|
849
|
+
* {@link pricesForListing}'s posture (an unreadable surface prices
|
|
850
|
+
* nothing and WARNS — discovery outlives the price oracle); a model with
|
|
851
|
+
* no per-token row carries NO pricing array (the schema rule: "a
|
|
852
|
+
* modality with no pricing array is simply unpriced").
|
|
853
|
+
*
|
|
854
|
+
* A CONFIGURED catalog read that FAILS propagates — the provider
|
|
855
|
+
* document cannot vouch a sellable model from nothing, and a half-empty
|
|
856
|
+
* one must never be served as if complete.
|
|
857
|
+
*/
|
|
858
|
+
openRouterModels(): Promise<OpenRouterProviderDoc>;
|
|
859
|
+
/**
|
|
860
|
+
* IMAGE CAPABILITY GATE (the Images lane's `imageModels` oracle, images.ts):
|
|
861
|
+
* is `model`'s deployed app a native destination for `mode`? Authorized
|
|
862
|
+
* resolution ({@link resolveApp} — undeployed uRun models throw
|
|
863
|
+
* {@link UnknownModelError}, which the lane renders as a 404) then the
|
|
864
|
+
* SAME catalog join as the /v1/models enrichment (PR421's rowForSlug):
|
|
865
|
+
* exact `<model_id>-<variant>` slug across ALL its GPU-placement rows, a
|
|
866
|
+
* bare model_id only when one variant owns it.
|
|
867
|
+
*
|
|
868
|
+
* Capability is read ONLY from catalog modality metadata, never from the
|
|
869
|
+
* serve function name: every row must carry task 'image' INVARIANTLY across
|
|
870
|
+
* the placements, and the mode comes from engine_args.expects_image —
|
|
871
|
+
* false ⇒ pure text-to-image ('generation'; diffusers v0.38.0
|
|
872
|
+
* pipeline_qwenimage_edit_plus raises torch.cat on an EMPTY image list, so
|
|
873
|
+
* an edit destination is never generation-capable); true or ABSENT (the
|
|
874
|
+
* native default) ⇒ 'edit'. The consensus rule is per-placement:
|
|
875
|
+
* capability must agree across every placement — a CONFLICTING flag (mixed
|
|
876
|
+
* true/false/absent) or a non-boolean one is a refused capability, never
|
|
877
|
+
* silently defaulted to edit.
|
|
878
|
+
*
|
|
879
|
+
* A CONFIGURED catalog oracle that FAILS propagates its rejection — the
|
|
880
|
+
* gate never fabricates a `false` (which would render as a misleading 404)
|
|
881
|
+
* out of an infrastructure outage. On the CALLER-ORG lane, no oracle
|
|
882
|
+
* configured, no matching row, or a slug the catalog cannot vouch for ⇒
|
|
883
|
+
* false (loud 404 upstream). On the SHARED lane the same questions are
|
|
884
|
+
* LOUDER: a shared endpoint is published capability, so an unresolvable
|
|
885
|
+
* row or an unvouchable mode throws {@link SharedCatalogError} (the 500
|
|
886
|
+
* path) instead of a 404 that would misattribute a catalog defect to the
|
|
887
|
+
* caller's request.
|
|
888
|
+
*/
|
|
889
|
+
supportsImages(model: string | undefined, mode: ImageCapabilityMode): Promise<boolean>;
|
|
890
|
+
/**
|
|
891
|
+
* AUDIO MODALITY GATE (the OpenAI Realtime lane's modality oracle): does
|
|
892
|
+
* `model`'s deployed app carry catalog `task` stt or tts — INVARIANTLY
|
|
893
|
+
* across every GPU-placement row ({@link audioModalityForRows})? The SAME
|
|
894
|
+
* shape as {@link supportsImages}: authorized resolution
|
|
895
|
+
* ({@link resolveTarget} — undeployed uRun models throw
|
|
896
|
+
* {@link UnknownModelError}), then the SAME catalog join
|
|
897
|
+
* ({@link catalogRowsForSlug}): exact `<model_id>-<variant>` slug across
|
|
898
|
+
* ALL its placements, a bare model_id only when one variant owns it.
|
|
899
|
+
*
|
|
900
|
+
* A SHARED-lane hit joins on the lane's catalog id + variant (the SAME
|
|
901
|
+
* join {@link imageModelList} uses for the shared block) — the shared org's
|
|
902
|
+
* app slug need not exist in the caller's own catalog by slug.
|
|
903
|
+
*
|
|
904
|
+
* A CONFIGURED catalog oracle that FAILS propagates its rejection — the
|
|
905
|
+
* gate never fabricates a `false` out of an infrastructure outage. No
|
|
906
|
+
* oracle configured, no matching row, or a task the catalog cannot vouch
|
|
907
|
+
* for ⇒ false — the realtime resolver renders that as the loud 404.
|
|
908
|
+
*/
|
|
909
|
+
supportsAudio(model: string | undefined): Promise<boolean>;
|
|
910
|
+
/**
|
|
911
|
+
* `GET /v1/images/models`: the IMAGE-CAPABLE slice of the model surface —
|
|
912
|
+
* the caller-org deployed apps whose catalog rows vouch an image mode
|
|
913
|
+
* ({@link imageModesForRows}, the same consensus {@link supportsImages}
|
|
914
|
+
* applies), plus the shared block's endpoints joined against each
|
|
915
|
+
* endpoint's OWN catalog row (exact model_id + variant, EXPLICIT boolean
|
|
916
|
+
* expects_image — {@link sharedImageModes}). A configured catalog oracle
|
|
917
|
+
* that FAILS propagates, and so does a shared endpoint the catalog cannot
|
|
918
|
+
* vouch (unresolvable row / unvouchable mode → {@link SharedCatalogError}):
|
|
919
|
+
* an image listing cannot vouch capability from nothing or advertise a
|
|
920
|
+
* silently defaulted mode, so — unlike {@link modelList} — there is NO
|
|
921
|
+
* bare-entries degradation here.
|
|
922
|
+
*/
|
|
923
|
+
imageModelList(): Promise<RouterImageModelList>;
|
|
924
|
+
/** Close every pooled session (Session.end() underneath) — proxy shutdown. */
|
|
925
|
+
closeAll(): Promise<void>;
|
|
926
|
+
}
|
|
927
|
+
|
|
928
|
+
/**
|
|
929
|
+
* THE CANONICAL USAGE-QUERY LANE — `POST /v1/usage/requests`.
|
|
930
|
+
*
|
|
931
|
+
* WHAT IT IS. A batch lookup over uRun's per-request inference ledger
|
|
932
|
+
* (`public.inference_requests`, urun-infra ENG-316): "given these inference
|
|
933
|
+
* ids, what do we hold for each?" — the cost and the usage quantities, scoped
|
|
934
|
+
* to the calling key's org. OpenAI defines no shape for this, so the shape is
|
|
935
|
+
* uRun's own: the ledger's own column vocabulary, and the same id space as the
|
|
936
|
+
* `Inference-Id` response header every `/v1` answer already carries
|
|
937
|
+
* (inference-id.ts).
|
|
938
|
+
*
|
|
939
|
+
* WHY IT IS CANONICAL AND NOT HUGGING-FACE-SHAPED. The owner has ruled that
|
|
940
|
+
* distributor-specific functionality must not land on the canonical surface.
|
|
941
|
+
* HF's billing poll — `{"requestIds":[...]}` answered with
|
|
942
|
+
* `{"requests":[{"requestId","costNanoUsd"}]}` — is HF-proprietary. It gets a
|
|
943
|
+
* THIN TRANSLATOR over this lane (partners/huggingface.ts, the ENG-376 adapter
|
|
944
|
+
* pattern): field renames and nothing else. Every other distributor's billing
|
|
945
|
+
* adapter reads this same lane, and so can the console and support.
|
|
946
|
+
*
|
|
947
|
+
* WHERE THE DATA COMES FROM. This process holds no database credential — by
|
|
948
|
+
* design (hosted/auth.ts: "NO STANDING CREDENTIAL"). The lookup rides the
|
|
949
|
+
* CALLER'S OWN key to the control plane's `inference-usage` edge function,
|
|
950
|
+
* which derives the org from that key server-side and runs the org-scoped
|
|
951
|
+
* `urun_inference_usage_lookup` RPC. So cross-org isolation here is the same
|
|
952
|
+
* structural property every other lane has: this process never holds a
|
|
953
|
+
* credential that spans orgs, and the org is never a request field.
|
|
954
|
+
*
|
|
955
|
+
* THREE CONTRACTS THAT ARE NOT NEGOTIABLE, because each one is money:
|
|
956
|
+
*
|
|
957
|
+
* 1. `cost_nano_usd: null` MEANS NOT YET PRICED — NEVER FREE. Every row in
|
|
958
|
+
* production carries NULL today: ENG-317's price machinery has landed but
|
|
959
|
+
* seeds no price, and nothing stamps the ledger yet. This lane reports
|
|
960
|
+
* the NULL verbatim. A caller that renders it as 0 is inventing a price
|
|
961
|
+
* nobody set; a caller billing a third party from it must simply not
|
|
962
|
+
* answer for that id yet.
|
|
963
|
+
* 2. AN ID WE HOLD NOTHING FOR IS ABSENT FROM THE ANSWER, and an id
|
|
964
|
+
* belonging to another org is indistinguishable from one that never
|
|
965
|
+
* existed. That is deliberate: anything else makes this surface a
|
|
966
|
+
* cross-org existence oracle. A row this lane could not VALIDATE is
|
|
967
|
+
* absent for the same reason and is therefore indistinguishable from
|
|
968
|
+
* those two — which is precisely why it is reported (ENG-416), so that
|
|
969
|
+
* the one absence we caused ourselves is visible on our side.
|
|
970
|
+
* 3. IDS THE CALLER DID NOT ASK ABOUT ARE NEVER RETURNED. The id list is
|
|
971
|
+
* the query; the control plane bounds the answer by it, and
|
|
972
|
+
* {@link assertRequestedOnly} re-checks it here rather than trusting the
|
|
973
|
+
* upstream to have done so.
|
|
974
|
+
*
|
|
975
|
+
* WHAT THIS LANE DOES NOT DO: it does not filter on `http_status` (only
|
|
976
|
+
* 2xx/3xx being billable is an HF rule, and support must be able to ask about
|
|
977
|
+
* a failed request), and it never returns `gpu_seconds` — a column ENG-417
|
|
978
|
+
* DROPPED from the ledger entirely, because a per-request GPU-seconds figure
|
|
979
|
+
* double-counts against `usage_events.metered_gpu_seconds` (which meters the
|
|
980
|
+
* instance once) and, under continuous batching, can never be reconciled to
|
|
981
|
+
* it. The guard below is kept and is STRONGER for the removal: it now asserts
|
|
982
|
+
* that a column which does not exist has not come back.
|
|
983
|
+
*
|
|
984
|
+
* WHAT IT DOES RETURN THAT IS EASY TO MISS: `identity_kind`. `model_id` speaks
|
|
985
|
+
* one of two unrelated namespaces — a catalog identity, or the caller's own
|
|
986
|
+
* deployed app slug — and the two can COLLIDE as strings, so the identity is
|
|
987
|
+
* never reported without the namespace that makes it readable.
|
|
988
|
+
*
|
|
989
|
+
* AND IT RESOLVES NO PRICE — stated because the surrounding system is about to
|
|
990
|
+
* grow several, and someone will come looking here. Nothing in this lane or in
|
|
991
|
+
* its Hugging Face translator reads `model_prices`, picks a rate, or knows what
|
|
992
|
+
* a model costs. It reports the `cost_nano_usd` a writer already stamped on the
|
|
993
|
+
* row, at REQUEST grain, so two requests for the same model may carry different
|
|
994
|
+
* costs and different `price_version`s and this code is indifferent to why.
|
|
995
|
+
*
|
|
996
|
+
* That means the one-price-per-model assumption is NOT baked in here. It lives
|
|
997
|
+
* one layer down, in `urun_active_model_price`, which resolves on
|
|
998
|
+
* `(model_id, variant)` and separates the session lane from the HF lane BY UNIT
|
|
999
|
+
* — fine while unit and audience correlate, and filed as ENG-404 for when they
|
|
1000
|
+
* stop (per-minute credit tiers put two audiences on one unit). When that is
|
|
1001
|
+
* fixed, the audience rides `price_version`, which this lane already carries
|
|
1002
|
+
* through per row and never interprets. No change is required here.
|
|
1003
|
+
*/
|
|
1004
|
+
|
|
1005
|
+
/**
|
|
1006
|
+
* ONE ledger row as this lane answers it — the control-plane RPC's column
|
|
1007
|
+
* list, in the ledger's own vocabulary. `gpu_seconds` and `org_id` are absent
|
|
1008
|
+
* by construction, not by omission here (see the module header).
|
|
1009
|
+
*/
|
|
1010
|
+
interface InferenceUsageRecord {
|
|
1011
|
+
/** The uuid this request's `Inference-Id` response header carried. */
|
|
1012
|
+
inference_id: string;
|
|
1013
|
+
/** The identity that served it, IN THE NAMESPACE {@link identity_kind} names. */
|
|
1014
|
+
model_id: string;
|
|
1015
|
+
/**
|
|
1016
|
+
* WHICH NAMESPACE {@link model_id} SPEAKS: `catalog_model` (the
|
|
1017
|
+
* `model_prices` vocabulary — everything a distributor routes) or
|
|
1018
|
+
* `caller_org_app` (the caller's own deployed app, which has no catalog
|
|
1019
|
+
* identity and is therefore PERMANENTLY unpriceable — never merely "not
|
|
1020
|
+
* priced yet"). Reported as a plain string rather than a union, because a
|
|
1021
|
+
* value this lane does not recognise must reach the caller as data rather
|
|
1022
|
+
* than fail the batch: an unknown lane is a forward-compatible control plane,
|
|
1023
|
+
* not a corrupt row.
|
|
1024
|
+
*/
|
|
1025
|
+
identity_kind: string;
|
|
1026
|
+
/** null = the model-level row (the `model_prices` key semantics). */
|
|
1027
|
+
variant: string | null;
|
|
1028
|
+
task: string;
|
|
1029
|
+
http_status: number;
|
|
1030
|
+
started_at: string;
|
|
1031
|
+
completed_at: string;
|
|
1032
|
+
/** null = this modality has no such quantity (not "zero of it"). */
|
|
1033
|
+
prompt_tokens: number | null;
|
|
1034
|
+
completion_tokens: number | null;
|
|
1035
|
+
output_units: number | null;
|
|
1036
|
+
/** NULL = NOT YET PRICED (ENG-317). Never "free". */
|
|
1037
|
+
cost_nano_usd: number | null;
|
|
1038
|
+
price_version: string | null;
|
|
1039
|
+
priced_at: string | null;
|
|
1040
|
+
}
|
|
1041
|
+
/**
|
|
1042
|
+
* ONE row this lane REFUSED TO ANSWER ABOUT, handed to the reporter so that a
|
|
1043
|
+
* dropped row is a visible event rather than a silent absence.
|
|
1044
|
+
*
|
|
1045
|
+
* It carries no row content — `where` is a position and `reason` is an
|
|
1046
|
+
* already-redacted {@link UsageSurfaceError} message (names and types, never
|
|
1047
|
+
* values). See {@link describeKeys}.
|
|
1048
|
+
*/
|
|
1049
|
+
interface UsageRowRejection {
|
|
1050
|
+
/** `usage.requests[N]` — WHICH row, never the row. */
|
|
1051
|
+
where: string;
|
|
1052
|
+
/** Why it could not be validated. */
|
|
1053
|
+
reason: string;
|
|
1054
|
+
}
|
|
1055
|
+
/**
|
|
1056
|
+
* A sink for {@link UsageRowRejection}s.
|
|
1057
|
+
*
|
|
1058
|
+
* `Promise<void>` IS PART OF THE TYPE BECAUSE IT WAS PART OF THE TYPE ANYWAY.
|
|
1059
|
+
* An `async` function is assignable to a `=> void` signature, so declaring this
|
|
1060
|
+
* synchronous never prevented an async sink — it only hid one, by putting its
|
|
1061
|
+
* failure on a later tick where the containment could not see it. Saying so in
|
|
1062
|
+
* the type is what lets {@link reportRejectedRow} actually handle it.
|
|
1063
|
+
*/
|
|
1064
|
+
type UsageRowReporter = (rejection: UsageRowRejection) => void | Promise<void>;
|
|
1065
|
+
/**
|
|
1066
|
+
* The batch-lookup seam. The hosted endpoint supplies one backed by the
|
|
1067
|
+
* control plane's `inference-usage` function; an embedder that supplies none
|
|
1068
|
+
* gets a LOUD 501 on the lane rather than a silently empty answer — the same
|
|
1069
|
+
* posture the image lanes take for `supportsImages`.
|
|
1070
|
+
*
|
|
1071
|
+
* IT IS TWO-PHASE, and the split is the admission boundary rather than
|
|
1072
|
+
* decoration: {@link authorize} runs BEFORE the request body is read, so a
|
|
1073
|
+
* request with no credential is refused without this process allocating or
|
|
1074
|
+
* parsing anything the caller sent.
|
|
1075
|
+
*/
|
|
1076
|
+
interface UsageLane {
|
|
1077
|
+
/**
|
|
1078
|
+
* Admission. Returns the caller's credential, or throws (a 401 through the
|
|
1079
|
+
* handler's `statusOf`) when it is missing or malformed. Runs before the
|
|
1080
|
+
* body is touched, and must NOT make a network call — the surface that owns
|
|
1081
|
+
* the ledger authenticates the credential itself when {@link lookup} uses
|
|
1082
|
+
* it, and a second verifier here would be a second source of truth about
|
|
1083
|
+
* the same key.
|
|
1084
|
+
*/
|
|
1085
|
+
authorize(req: IncomingMessage): string;
|
|
1086
|
+
/** The batch read, as the holder of the credential `authorize` returned. */
|
|
1087
|
+
lookup(credential: string, inferenceIds: readonly string[]): Promise<InferenceUsageRecord[]>;
|
|
1088
|
+
}
|
|
1089
|
+
|
|
1090
|
+
/**
|
|
1091
|
+
* THE PER-REQUEST LEDGER WRITE — one served `/v1` inference becomes one
|
|
1092
|
+
* durable, already-priced row in `public.inference_requests`.
|
|
1093
|
+
*
|
|
1094
|
+
* WHY THIS EXISTS. uRun answers Hugging Face's billing poll out of a ledger
|
|
1095
|
+
* that, until this module, NOTHING HAD EVER WRITTEN. ENG-316 landed the table
|
|
1096
|
+
* with no writer, ENG-317 landed the price machinery with no stamping, and
|
|
1097
|
+
* ENG-318 landed the reader over both and stated the consequence: with every
|
|
1098
|
+
* cost NULL, a poll answers `{"requests": null}` for every batch and every
|
|
1099
|
+
* request is written off unbilled ~30 minutes later. This module is the hop
|
|
1100
|
+
* from the correlation record the proxy already emits to a row.
|
|
1101
|
+
*
|
|
1102
|
+
* THE THREE RULES THIS MODULE EXISTS TO KEEP (owner ruling, ENG-409):
|
|
1103
|
+
*
|
|
1104
|
+
* 1. THE WRITE IS OFF THE RESPONSE PATH, and the price is stamped in that
|
|
1105
|
+
* same write. It runs from the response's `'close'` hook — after the
|
|
1106
|
+
* customer has their answer — so a ledger round trip can never be in the
|
|
1107
|
+
* latency of an inference. It is not a later sweep either: a re-price
|
|
1108
|
+
* landing after HF has been answered disagrees with a bill already
|
|
1109
|
+
* issued (ENG-406), so the row is born priced or born unpriced.
|
|
1110
|
+
*
|
|
1111
|
+
* 2. A LEDGER WRITE MUST NEVER FAIL AN INFERENCE REQUEST. Everything here
|
|
1112
|
+
* is fire-and-forget and cannot throw into the request path: the answer
|
|
1113
|
+
* is already delivered, the socket is already closing, and a failure is
|
|
1114
|
+
* LOUD IN THE LOG and nowhere else. {@link recordInference} returns
|
|
1115
|
+
* `void` for exactly that reason — there is no promise a caller could
|
|
1116
|
+
* accidentally await, and nothing to reject. The queue and the retries
|
|
1117
|
+
* added by ENG-415 change nothing about this: they all happen behind that
|
|
1118
|
+
* same `void`, after the customer has been answered.
|
|
1119
|
+
*
|
|
1120
|
+
* 3. A REQUEST THAT CANNOT BE PRICED IS WRITTEN UNPRICED, NEVER AT ZERO AND
|
|
1121
|
+
* NEVER DROPPED. Pricing itself happens in the database
|
|
1122
|
+
* (`urun_record_inference_request`), which is where `model_prices` is
|
|
1123
|
+
* readable and where a `price_version` can name the rate rows that
|
|
1124
|
+
* produced a number. This module supplies the ONE fact the database
|
|
1125
|
+
* cannot see — {@link billableOutcome} — and nothing else about money.
|
|
1126
|
+
*
|
|
1127
|
+
* WHAT IS DELIBERATELY NOT HERE:
|
|
1128
|
+
* * NO PRICES, NO RATES, NO MODEL NAMES, NO ARITHMETIC. Grep it: there is
|
|
1129
|
+
* no rate and no model id below. The price book is data (ENG-317) and a
|
|
1130
|
+
* rate change must take effect without a deploy.
|
|
1131
|
+
* * NO PARTNER VOCABULARY. Hugging Face's wire shape lives in
|
|
1132
|
+
* `proxy/partners/huggingface.ts` and nowhere else. What crosses this
|
|
1133
|
+
* module is uRun's own ledger vocabulary, which every distributor's
|
|
1134
|
+
* adapter and the console read alike.
|
|
1135
|
+
* * NO `gpu_seconds`, AND NO LEDGER COLUMN FOR ONE (ENG-417). Exclusive
|
|
1136
|
+
* per-request GPU occupancy is not obtainable under continuous batching,
|
|
1137
|
+
* and wall-clock there is SHARED occupancy — a per-request figure would
|
|
1138
|
+
* double-count against `usage_events.metered_gpu_seconds`, which meters
|
|
1139
|
+
* the instance once and is reduced by the hourly `usage_accrual` sweep so
|
|
1140
|
+
* that ledger sums stay exactly the metered wall-clock cost. What this
|
|
1141
|
+
* module sends instead is a LINK (`session_id`) and a WEIGHT (the
|
|
1142
|
+
* residency milliseconds), from which the share is DERIVED at
|
|
1143
|
+
* reconciliation so the shares sum to the instance total by construction.
|
|
1144
|
+
* The residency is NOT GPU time: under continuous batching the sum of
|
|
1145
|
+
* concurrent `decode_ms` EXCEEDS the session's wall-clock, which is
|
|
1146
|
+
* precisely why it is only ever meaningful as a normalized fraction.
|
|
1147
|
+
* * NO DURABLE BUFFER. Rows are queued in memory and nowhere else. The
|
|
1148
|
+
* inference-proxy Deployment declares NO VOLUMES — not a PVC, not even an
|
|
1149
|
+
* emptyDir — so there is nowhere on this pod that survives a SIGKILL, and
|
|
1150
|
+
* making one would mean a StatefulSet with a per-replica volume on an
|
|
1151
|
+
* internet-facing HA front door where a scaled-down replica's volume holds
|
|
1152
|
+
* unflushed rows forever. What survives instead is the `inference_request`
|
|
1153
|
+
* line this proxy writes to stdout for EVERY `/v1` request, which
|
|
1154
|
+
* fluent-bit ships off the pod into Loki; that is the reconciliation
|
|
1155
|
+
* record, and it is a way to KNOW what was lost rather than a way to bill
|
|
1156
|
+
* it. See {@link drainLedgerWrites}.
|
|
1157
|
+
*
|
|
1158
|
+
* RETRY, AND THE RULE IT LIVES UNDER (ENG-415). This module used to do NO
|
|
1159
|
+
* retry, for a good reason: a retry after a timeout cannot know whether the
|
|
1160
|
+
* first attempt landed, and the row it would re-send came back as an
|
|
1161
|
+
* undifferentiated duplicate — so retrying turned an ambiguous outcome into a
|
|
1162
|
+
* loud error that looked exactly like a forged row. That reason is now gone:
|
|
1163
|
+
* the control plane's 409 carries the row already stored, so a retry whose
|
|
1164
|
+
* first attempt landed resolves as `already_written` instead. Retry is
|
|
1165
|
+
* therefore allowed for EXACTLY ONE failure reason and no other — see
|
|
1166
|
+
* {@link isRetryable}, which is where the rule lives rather than in a
|
|
1167
|
+
* judgement at the call site.
|
|
1168
|
+
*/
|
|
1169
|
+
|
|
1170
|
+
/**
|
|
1171
|
+
* ONE ledger row, in the ledger's own vocabulary — exactly the body the
|
|
1172
|
+
* control plane's `inference-ledger` function takes.
|
|
1173
|
+
*
|
|
1174
|
+
* `org_id` and `api_key_id` are ABSENT BY CONSTRUCTION rather than omitted
|
|
1175
|
+
* here: the control plane derives both from the credential the write rides,
|
|
1176
|
+
* so there is no field on this surface that could name another org.
|
|
1177
|
+
*/
|
|
1178
|
+
interface LedgerWrite {
|
|
1179
|
+
/** The uuid this request's `Inference-Id` response header carried. */
|
|
1180
|
+
inference_id: string;
|
|
1181
|
+
/**
|
|
1182
|
+
* The identity that served it — IN THE NAMESPACE {@link identity_kind}
|
|
1183
|
+
* NAMES. Never the caller's raw `model` string.
|
|
1184
|
+
*/
|
|
1185
|
+
model_id: string;
|
|
1186
|
+
/**
|
|
1187
|
+
* WHICH NAMESPACE {@link model_id} SPEAKS (ENG-412): a catalog identity
|
|
1188
|
+
* (`public.model_prices` vocabulary), or the caller's own deployed app slug.
|
|
1189
|
+
* The two are unrelated namespaces that can nonetheless COLLIDE as strings,
|
|
1190
|
+
* so the row states which one it is rather than leaving it to be inferred —
|
|
1191
|
+
* and the control plane refuses to price anything but a catalog row, in the
|
|
1192
|
+
* writer's predicate and again in a table CHECK.
|
|
1193
|
+
*/
|
|
1194
|
+
identity_kind: 'catalog_model' | 'caller_org_app';
|
|
1195
|
+
variant: string | null;
|
|
1196
|
+
task: string;
|
|
1197
|
+
http_status: number;
|
|
1198
|
+
started_at: string;
|
|
1199
|
+
completed_at: string;
|
|
1200
|
+
/** See {@link billableOutcome}. The one fact the database cannot see. */
|
|
1201
|
+
billable_outcome: boolean;
|
|
1202
|
+
/**
|
|
1203
|
+
* THE INSTANCE that served it (ENG-417) — the link the control plane
|
|
1204
|
+
* verifies against the org before writing. NULL when this request's dial
|
|
1205
|
+
* exposed no session identity; never a fabricated id.
|
|
1206
|
+
*/
|
|
1207
|
+
session_id: string | null;
|
|
1208
|
+
/** NULL = this request has no such quantity. NEVER zero of it. */
|
|
1209
|
+
prompt_tokens: number | null;
|
|
1210
|
+
completion_tokens: number | null;
|
|
1211
|
+
output_units: number | null;
|
|
1212
|
+
/**
|
|
1213
|
+
* THE WEIGHT (ENG-417) — the serve runtime's own residency in whole
|
|
1214
|
+
* milliseconds. An allocation input, never GPU time and never a cost. NULL
|
|
1215
|
+
* means the runtime reported nothing, which is not zero: a weightless
|
|
1216
|
+
* request drops out of its instance's allocation and inflates every other
|
|
1217
|
+
* request's share.
|
|
1218
|
+
*/
|
|
1219
|
+
queue_ms: number | null;
|
|
1220
|
+
prefill_ms: number | null;
|
|
1221
|
+
decode_ms: number | null;
|
|
1222
|
+
total_ms: number | null;
|
|
1223
|
+
}
|
|
1224
|
+
/** What the control plane answers with — the row as it was actually written. */
|
|
1225
|
+
interface LedgerRecorded {
|
|
1226
|
+
inference_id: string;
|
|
1227
|
+
/** NULL = NOT YET PRICED (ENG-317). Never "free". */
|
|
1228
|
+
cost_nano_usd: number | null;
|
|
1229
|
+
/** `in=<model_prices.id>;out=<model_prices.id>` when priced. */
|
|
1230
|
+
price_version: string | null;
|
|
1231
|
+
/**
|
|
1232
|
+
* Why the row carries no price, or null when it is priced. One of
|
|
1233
|
+
* `not_a_catalog_model` | `outcome_not_billable` | `http_status_not_success`
|
|
1234
|
+
* | `task_unit_undeclared` | `task_not_token_billed` | `token_counts_absent`
|
|
1235
|
+
* | `no_active_price` | `no_active_input_price` | `no_active_output_price`.
|
|
1236
|
+
* It is a NORMAL answer, not an error: no PER-TOKEN rate has been seeded
|
|
1237
|
+
* yet, so this lane has nothing to price with. (Not "the price book is
|
|
1238
|
+
* empty" — `model_prices` has carried live `per_minute` rows since
|
|
1239
|
+
* 2026-09-10; what is absent is `per_million_input_tokens` /
|
|
1240
|
+
* `per_million_output_tokens`.)
|
|
1241
|
+
*
|
|
1242
|
+
* Only `not_a_catalog_model` is PERMANENT. A caller-org app has no catalog
|
|
1243
|
+
* identity, so no price book will ever cover it; every other reason means
|
|
1244
|
+
* "not priced YET".
|
|
1245
|
+
*
|
|
1246
|
+
* THE TWO `task_*` REASONS ARE ENG-427, and they are the reason
|
|
1247
|
+
* {@link LedgerWrite.task} is worth carrying honestly. The control plane prices a row IN THE UNIT ITS
|
|
1248
|
+
* TASK DECLARES (`public.task_billing_units`), never in the unit the data
|
|
1249
|
+
* happens to look like:
|
|
1250
|
+
*
|
|
1251
|
+
* * `task_not_token_billed` — this lane's billable quantity is not tokens
|
|
1252
|
+
* (an image lane's is {@link LedgerWrite.output_units}), so its token counts are not
|
|
1253
|
+
* a price for it. A genuine `{0, 0}` receipt on such a lane would
|
|
1254
|
+
* otherwise settle at `cost_nano_usd = 0` PERMANENTLY — the control
|
|
1255
|
+
* plane's append-only rule allows a deliberate re-price and never a
|
|
1256
|
+
* return to NULL — which is an invoice asserting the work was free when
|
|
1257
|
+
* its billable quantity was never measured.
|
|
1258
|
+
* * `task_unit_undeclared` — nobody has declared what this task bills in,
|
|
1259
|
+
* so nothing is priced. Adding a lane to {@link ledgerTaskOf} without a
|
|
1260
|
+
* matching declaration fails CLOSED and says which fix it wants.
|
|
1261
|
+
*/
|
|
1262
|
+
unpriced_reason: string | null;
|
|
1263
|
+
}
|
|
1264
|
+
/**
|
|
1265
|
+
* WHY A LEDGER WRITE DID NOT LAND — a FIXED, SMALL vocabulary, because this is
|
|
1266
|
+
* what an alert matches on.
|
|
1267
|
+
*
|
|
1268
|
+
* ENG-409 made the fabricated-row hazard (ENG-410) detectable and stopped
|
|
1269
|
+
* there: a duplicate `inference_id` raises 23505, surfaces as a 409, and
|
|
1270
|
+
* arrives here as a rejected write. But it arrived carrying only an English
|
|
1271
|
+
* SENTENCE, which meant the only way to alert on "somebody wrote this row
|
|
1272
|
+
* ahead of us" was a regex over prose. An alert that is a substring match on a
|
|
1273
|
+
* message nobody promised to keep stable is an alert that dies silently the
|
|
1274
|
+
* first time the wording is improved — and this particular alert is the ONLY
|
|
1275
|
+
* signal that a customer is under-billing itself.
|
|
1276
|
+
*
|
|
1277
|
+
* So the reason is a FIELD, drawn from this closed set, and it is also the
|
|
1278
|
+
* metric label ({@link LedgerOutcome}):
|
|
1279
|
+
*
|
|
1280
|
+
* * `duplicate` — the id was already in the ledger and the stored row is
|
|
1281
|
+
* PROVABLY NOT THE ONE WE WROTE. Since the 409 carries that row (ENG-415),
|
|
1282
|
+
* a matching one is our own landed attempt and is not a failure at all
|
|
1283
|
+
* ({@link LedgerAlreadyWritten}); what reaches here is somebody else's row
|
|
1284
|
+
* — or a row we could not read back, which fails to this loud side on
|
|
1285
|
+
* purpose. This is the ENG-410 signal.
|
|
1286
|
+
* * `duplicate_foreign` — the id is taken by a row THIS ORG CANNOT READ.
|
|
1287
|
+
* `inference_id` is a global primary key, so it is reachable, and it can
|
|
1288
|
+
* never be our own write. The loudest state in the lane.
|
|
1289
|
+
* * `auth` — the control plane refused the caller's org key.
|
|
1290
|
+
* * `surface` — the control plane could not be reached, answered a status
|
|
1291
|
+
* we do not accept, or answered a body we could not validate. THE ONLY
|
|
1292
|
+
* RETRYABLE ONE (see {@link isRetryable}).
|
|
1293
|
+
* * `unknown` — the lane rejected with something that carries NO reason at
|
|
1294
|
+
* all. It is its OWN value rather than being folded into `surface`
|
|
1295
|
+
* precisely so it cannot hide: a reason-less rejection is a defect in the
|
|
1296
|
+
* lane, and counting it as a surface error would file a bug as weather —
|
|
1297
|
+
* and would retry it, which is how a defect becomes a loop.
|
|
1298
|
+
* * `dwell_expired` — the row waited longer than {@link LEDGER_MAX_DWELL_MS}
|
|
1299
|
+
* and was given up on. Revenue lost, named as revenue lost.
|
|
1300
|
+
* * `dropped_on_exit` — the row was still queued when the shutdown drain ran
|
|
1301
|
+
* out. One per row, per pod, per rolling deploy that went badly.
|
|
1302
|
+
*/
|
|
1303
|
+
declare const LEDGER_FAILURE_REASONS: readonly ["duplicate", "duplicate_foreign", "auth", "surface", "unknown", "dwell_expired", "dropped_on_exit"];
|
|
1304
|
+
type LedgerFailureReason = (typeof LEDGER_FAILURE_REASONS)[number];
|
|
1305
|
+
/**
|
|
1306
|
+
* Implemented by every error a {@link LedgerLane.write} may reject with, so
|
|
1307
|
+
* the reason travels ON the error instead of being re-derived from its text.
|
|
1308
|
+
*
|
|
1309
|
+
* It is a FIELD rather than a class hierarchy on purpose: the auth rejection
|
|
1310
|
+
* must stay an `instanceof ProxyAuthError` for every existing caller, and a
|
|
1311
|
+
* class can only have one base.
|
|
1312
|
+
*/
|
|
1313
|
+
interface LedgerFailure {
|
|
1314
|
+
readonly ledgerFailureReason: LedgerFailureReason;
|
|
1315
|
+
}
|
|
1316
|
+
/**
|
|
1317
|
+
* WHY A ROW WAS NOT EVEN ATTEMPTED. Also a closed vocabulary, and for the same
|
|
1318
|
+
* reason: a served request that produced no ledger row is revenue that can
|
|
1319
|
+
* never be recovered, so each of these is a counted event and not merely a log
|
|
1320
|
+
* line somebody might read.
|
|
1321
|
+
*
|
|
1322
|
+
* `not_a_catalog_model` WAS IN THIS LIST AND IS DELIBERATELY GONE (ENG-412). A
|
|
1323
|
+
* request served by a caller's own deployed app used to be skipped here; it
|
|
1324
|
+
* now WRITES a row in its own declared namespace (`identity_kind:
|
|
1325
|
+
* 'caller_org_app'`), so nothing can emit that skip any more. A member of a
|
|
1326
|
+
* closed, counted vocabulary that is structurally unreachable is a counter
|
|
1327
|
+
* that can only ever read zero — it invites the reader to conclude the case
|
|
1328
|
+
* never happens, when in truth the case stopped being a skip at all.
|
|
1329
|
+
*
|
|
1330
|
+
* DO NOT CONFUSE IT WITH THE UNPRICED REASON OF THE SAME NAME, which is alive
|
|
1331
|
+
* and is the whole point of ENG-412: the control plane still answers
|
|
1332
|
+
* `unpriced_reason: 'not_a_catalog_model'` for those rows (see
|
|
1333
|
+
* {@link LedgerRecorded}). The row exists and is counted; it simply can never
|
|
1334
|
+
* carry a price. Two vocabularies, one word, opposite states — a written row
|
|
1335
|
+
* versus no row at all.
|
|
1336
|
+
*
|
|
1337
|
+
* `no_model_named` replaces it for the one case on that lane that still cannot
|
|
1338
|
+
* be written: `model_id` is NOT NULL, a caller-org app's identity IS the
|
|
1339
|
+
* caller's model string, and a request that named no model leaves nothing
|
|
1340
|
+
* honest to record.
|
|
1341
|
+
*/
|
|
1342
|
+
declare const LEDGER_SKIP_REASONS: readonly ["no_response_head", "no_org", "model_unresolved", "no_model_named", "queue_full"];
|
|
1343
|
+
type LedgerSkipReason = (typeof LEDGER_SKIP_REASONS)[number];
|
|
1344
|
+
/**
|
|
1345
|
+
* The TERMINAL fate of one request's ledger row, as a bounded pair of metric
|
|
1346
|
+
* labels. Every `/v1` request that reaches a lane produces exactly one of
|
|
1347
|
+
* these, so `sum(urun_inference_ledger_writes_total)` is the number of
|
|
1348
|
+
* billable requests the proxy has accounted for — and the `skipped` and
|
|
1349
|
+
* `failed` arms are the leak.
|
|
1350
|
+
*
|
|
1351
|
+
* EVERY VALUE IS OWNED BY THIS MODULE. `reason` is drawn from
|
|
1352
|
+
* {@link LEDGER_SKIP_REASONS}, {@link LEDGER_FAILURE_REASONS} or the two
|
|
1353
|
+
* pricing states below — never from the control plane's answer and never from
|
|
1354
|
+
* anything a caller can influence. That is what keeps the series set bounded
|
|
1355
|
+
* without a `BoundedLabel`: an `unpriced_reason` echoed verbatim into a label
|
|
1356
|
+
* would let a surface that changed shape blow up the TSDB.
|
|
1357
|
+
*/
|
|
1358
|
+
type LedgerOutcome =
|
|
1359
|
+
/** The row is in the ledger. `priced` / `unpriced` is ENG-317's NULL-is-not-zero distinction. */
|
|
1360
|
+
{
|
|
1361
|
+
outcome: 'written';
|
|
1362
|
+
reason: 'priced' | 'unpriced';
|
|
1363
|
+
}
|
|
1364
|
+
/**
|
|
1365
|
+
* The row was ALREADY in the ledger and it is OURS — an earlier attempt
|
|
1366
|
+
* landed and we never learned it (ENG-415). A terminal SUCCESS: the request
|
|
1367
|
+
* is accounted for exactly once, and it is counted separately from `written`
|
|
1368
|
+
* only so "how often does the write path go ambiguous" is answerable.
|
|
1369
|
+
*/
|
|
1370
|
+
| {
|
|
1371
|
+
outcome: 'already_written';
|
|
1372
|
+
reason: 'priced' | 'unpriced';
|
|
1373
|
+
}
|
|
1374
|
+
/** No row was attempted. */
|
|
1375
|
+
| {
|
|
1376
|
+
outcome: 'skipped';
|
|
1377
|
+
reason: LedgerSkipReason;
|
|
1378
|
+
}
|
|
1379
|
+
/** A row was attempted and did not land. */
|
|
1380
|
+
| {
|
|
1381
|
+
outcome: 'failed';
|
|
1382
|
+
reason: LedgerFailureReason;
|
|
1383
|
+
};
|
|
1384
|
+
/**
|
|
1385
|
+
* The ledger write seam. Only the HOSTED endpoint supplies one (the write
|
|
1386
|
+
* rides the caller's own org API key to the control plane; this process holds
|
|
1387
|
+
* no database credential). A proxy WITHOUT one does not participate in the
|
|
1388
|
+
* ledger at all — the `usage` / `supportsImages` posture: configured absence,
|
|
1389
|
+
* declared at the seam, not a degradation discovered at runtime. The local
|
|
1390
|
+
* CLI proxy is single-tenant and bills nothing, so it has none.
|
|
1391
|
+
*/
|
|
1392
|
+
interface LedgerLane {
|
|
1393
|
+
/**
|
|
1394
|
+
* Write one row, as the holder of the credential on `req`. The request is
|
|
1395
|
+
* over by the time this is called; `req` is carried only for its
|
|
1396
|
+
* `Authorization` header, never re-read.
|
|
1397
|
+
*/
|
|
1398
|
+
write(req: IncomingMessage, entry: LedgerWrite): Promise<LedgerRecorded>;
|
|
1399
|
+
/**
|
|
1400
|
+
* Count ONE terminal outcome. REQUIRED, unlike {@link report} — which has a
|
|
1401
|
+
* real default (this process's stdout) — because there is no default for
|
|
1402
|
+
* "count it nowhere" that is not a silent no-op. A lane that genuinely
|
|
1403
|
+
* counts nothing has to say so at the seam, in code a reviewer can see.
|
|
1404
|
+
*
|
|
1405
|
+
* This is the alerting surface for ENG-410: `outcome="failed"` with
|
|
1406
|
+
* `reason="duplicate"` is the fabricated-row signal, and the log line is
|
|
1407
|
+
* only the per-request detail behind it.
|
|
1408
|
+
*/
|
|
1409
|
+
observe(outcome: LedgerOutcome): void;
|
|
1410
|
+
/**
|
|
1411
|
+
* Where this lane's own structured diagnostics go — one newline-terminated
|
|
1412
|
+
* JSON object per event. Absent, they go to this process's stdout (the
|
|
1413
|
+
* hosted endpoint's collected container log). Injectable so tests assert on
|
|
1414
|
+
* them without racing the process streams.
|
|
1415
|
+
*
|
|
1416
|
+
* These are DELIBERATELY NOT the `requestLog` sink: that sink carries
|
|
1417
|
+
* exactly one `inference_request` record per `/v1` request and readers count
|
|
1418
|
+
* on that.
|
|
1419
|
+
*/
|
|
1420
|
+
report?(line: string): void;
|
|
1421
|
+
}
|
|
1422
|
+
|
|
1423
|
+
/**
|
|
1424
|
+
* The shared Responses execution/store seam — the ONE execution + event-
|
|
1425
|
+
* envelope path and the authorized tenant store, imported by BOTH the HTTP
|
|
1426
|
+
* SSE lane (server.ts) and the `/v1/responses` WebSocket lane
|
|
1427
|
+
* (responses-ws.ts). This module deliberately imports NO transport acceptor
|
|
1428
|
+
* and NO HTTP lane module: its only imports are type-only, so the module
|
|
1429
|
+
* graph server.ts → responses-ws.ts → responses-turn.ts is acyclic and no
|
|
1430
|
+
* Vitest/Bun/Node module-evaluation order can observe an uninitialized
|
|
1431
|
+
* binding (the constructor-failure class of bug this extraction removes).
|
|
1432
|
+
*/
|
|
1433
|
+
|
|
1434
|
+
/** The request envelope every Responses-shaped upstream call carries. */
|
|
1435
|
+
interface ResponsesCreateParams {
|
|
1436
|
+
model?: string;
|
|
1437
|
+
input: unknown;
|
|
1438
|
+
stream?: boolean;
|
|
1439
|
+
tools?: unknown;
|
|
1440
|
+
tool_choice?: unknown;
|
|
1441
|
+
temperature?: number;
|
|
1442
|
+
max_output_tokens?: number;
|
|
1443
|
+
/**
|
|
1444
|
+
* System/developer instructions, forwarded to the serve envelope — where
|
|
1445
|
+
* they become ONE prepended `system`-role message (the protocol's native
|
|
1446
|
+
* per-request system input; there is NO envelope-level instructions field).
|
|
1447
|
+
* Typed `string` end to end (transport/encode.ts) so no cast bridges the seam.
|
|
1448
|
+
*/
|
|
1449
|
+
instructions?: string;
|
|
1450
|
+
top_p?: number;
|
|
1451
|
+
stop?: string[];
|
|
1452
|
+
/**
|
|
1453
|
+
* Reasoning controls (urun-python #1667): forwarded VERBATIM to the serve
|
|
1454
|
+
* envelope — `chat_template_kwargs.enable_thinking:false` is the think-off
|
|
1455
|
+
* switch. Absent -> absent (no default injection; server-side validation).
|
|
1456
|
+
*/
|
|
1457
|
+
reasoning_effort?: string;
|
|
1458
|
+
chat_template_kwargs?: Record<string, unknown>;
|
|
1459
|
+
/**
|
|
1460
|
+
* OpenAI `response_format` — structured output (ENG-310). On the Responses
|
|
1461
|
+
* lane the caller spells this `text.format`; `textFormatOf` lifts it onto
|
|
1462
|
+
* this ONE field so both lanes hand the serve envelope the same key.
|
|
1463
|
+
* Forwarded verbatim — the serve runtime is the one validator.
|
|
1464
|
+
*/
|
|
1465
|
+
response_format?: unknown;
|
|
1466
|
+
}
|
|
1467
|
+
/** How one pinned session ended (the native phase machinery's terminal step). */
|
|
1468
|
+
interface SessionEndInfo {
|
|
1469
|
+
/**
|
|
1470
|
+
* Milliseconds until the session's native deadline (`endsAt`), or null when
|
|
1471
|
+
* the app declared no maximum session length. NOTE (missing primitive,
|
|
1472
|
+
* called out in the PR): core exposes no PRE-expiry notice event — this
|
|
1473
|
+
* callback fires AT terminal loss, so timeLeftMs is ~0 on expiry.
|
|
1474
|
+
*/
|
|
1475
|
+
timeLeftMs: number | null;
|
|
1476
|
+
/** The terminal reason (expired / ended / error), for the loud close. */
|
|
1477
|
+
reason: string;
|
|
1478
|
+
}
|
|
1479
|
+
/** The upstream calls the proxy makes — injectable (tests; alt transports). */
|
|
1480
|
+
interface ProxyClients {
|
|
1481
|
+
/**
|
|
1482
|
+
* The NON-SECRET tenancy label of the org this backhaul belongs to — the
|
|
1483
|
+
* control plane's org id, set by the hosted registry from the verified
|
|
1484
|
+
* `CallerIdentity` (`hosted/tenants.ts`). The request handler copies it onto
|
|
1485
|
+
* the request's `InferenceTurn`, which is what puts an `org` label on the
|
|
1486
|
+
* Prometheus series and an `org` field on the billing record.
|
|
1487
|
+
*
|
|
1488
|
+
* THE ORG, NEVER THE KEY: the Bearer key is the credential on this surface,
|
|
1489
|
+
* so it must never reach a metric, a log, or a dashboard. Absent on the
|
|
1490
|
+
* single-tenant local proxy and on hand-built test clients.
|
|
1491
|
+
*/
|
|
1492
|
+
readonly tenant?: string;
|
|
1493
|
+
/**
|
|
1494
|
+
* `UrunResponses(session).responses.create` — an async iterable of Responses
|
|
1495
|
+
* stream events.
|
|
1496
|
+
*
|
|
1497
|
+
* `onDial` is WHAT THE DIAL REPORTS ABOUT ITSELF (ENG-417 + ENG-414): the
|
|
1498
|
+
* implementation calls it once PER DIAL with a {@link DialReport} — the uRun
|
|
1499
|
+
* session id of the backhaul it is ABOUT TO USE, and the billing identity
|
|
1500
|
+
* that dial's acquisition resolved. So a request that re-homes reports the
|
|
1501
|
+
* REPLACEMENT on both counts: the instance that actually served it, and the
|
|
1502
|
+
* identity it was actually served under.
|
|
1503
|
+
*
|
|
1504
|
+
* IT IS A SECOND ARGUMENT, NOT A FIELD ON THE ENVELOPE, and that is
|
|
1505
|
+
* load-bearing. `params` is forwarded VERBATIM to the serve runtime by every
|
|
1506
|
+
* implementation of this seam, including hand-built embedder clients; a
|
|
1507
|
+
* function smuggled onto it would reach a JSON serializer, be dropped
|
|
1508
|
+
* silently, and leave one implementation (the one that remembered to strip
|
|
1509
|
+
* it) behaving differently from the rest. Out of band, the envelope stays
|
|
1510
|
+
* exactly what it was and an implementation that ignores `onDial` simply
|
|
1511
|
+
* records no link.
|
|
1512
|
+
*
|
|
1513
|
+
* IT IS A CALLBACK RATHER THAN A RETURN VALUE because the answer is not
|
|
1514
|
+
* known when `createResponse` returns: `rehome.ts` dials inside the async
|
|
1515
|
+
* generator and may dial a SECOND time before the first event is yielded.
|
|
1516
|
+
* Asking beforehand — an earlier draft of ENG-417 did — has two defects at
|
|
1517
|
+
* once: it records a session the turn may abandon, and it ALLOCATES a pooled
|
|
1518
|
+
* backhaul for requests still going to be refused by local validation, which
|
|
1519
|
+
* is GPU capacity spent on a 400.
|
|
1520
|
+
*
|
|
1521
|
+
* THE IDENTITY ARRIVES THROUGH `ModelRouter.sessionFor` AND NOT FROM THE
|
|
1522
|
+
* POOL ENTRY, and a reader who does not know why will "simplify" it back
|
|
1523
|
+
* into the bug. The obvious-looking move is to widen
|
|
1524
|
+
* `RehomeOptions.sessionIdOf` the way the session id arrived. IT CANNOT
|
|
1525
|
+
* WORK: `entry` is the GENERIC pooled session (`rehome.ts` is parameterised
|
|
1526
|
+
* over it) and knows nothing about models, so the identity is not derivable
|
|
1527
|
+
* from it. The other obvious move — resolving the identity separately inside
|
|
1528
|
+
* the dial loop — is worse: it would be a SECOND RESOLUTION of the same
|
|
1529
|
+
* question, which is precisely the defect that got ENG-417's pre-dial pin
|
|
1530
|
+
* rejected in review (a resolution taken at a different moment from the dial
|
|
1531
|
+
* can answer differently, and then the row names a variant nobody served).
|
|
1532
|
+
* So the acquisition returns the identity alongside the entry, and the loop
|
|
1533
|
+
* reports what it already resolved.
|
|
1534
|
+
*
|
|
1535
|
+
* BOTH IDENTITY FLAVOURS RIDE IT. The catalog `(model_id, variant)` a row is
|
|
1536
|
+
* priced on, and the caller-org APP SLUG that ENG-412 made billing-bearing —
|
|
1537
|
+
* it lands in `model_id` with `identity_kind: 'caller_org_app'`, on rows
|
|
1538
|
+
* that previously were not written at all. They had the identical re-home
|
|
1539
|
+
* window and they close by the same line: the dial reports whichever flavour
|
|
1540
|
+
* it resolved.
|
|
1541
|
+
*
|
|
1542
|
+
* WIDEN {@link DialReport}; do not add a second callback beside it. Two
|
|
1543
|
+
* callbacks firing from the same loop about the same dial would be two
|
|
1544
|
+
* things that can disagree, which is the whole defect this seam exists to
|
|
1545
|
+
* remove.
|
|
1546
|
+
*/
|
|
1547
|
+
createResponse(params: ResponsesCreateParams, onDial?: DialObserver): Promise<AsyncIterable<unknown>> | AsyncIterable<unknown>;
|
|
1548
|
+
/** `listModels(...)` result (an OpenAI model list object). */
|
|
1549
|
+
listModels(): Promise<unknown>;
|
|
1550
|
+
/**
|
|
1551
|
+
* The OpenRouter provider document (`GET /v1/models?format=openrouter`).
|
|
1552
|
+
* Optional at the seam: hand-built test clients may omit it, in which case
|
|
1553
|
+
* the format=openrouter branch answers 501 — loud, never a silent empty.
|
|
1554
|
+
*/
|
|
1555
|
+
openRouterModels?(): Promise<unknown>;
|
|
1556
|
+
/**
|
|
1557
|
+
* The Images lane's capability gate (images.ts `ImageModelGate`), backed by
|
|
1558
|
+
* ModelRouter.supportsImages — the caller-org catalog's modality metadata,
|
|
1559
|
+
* per-caller by construction; never function-name inference. Optional at
|
|
1560
|
+
* the seam: hand-built clients may omit it, in which case the mounted
|
|
1561
|
+
* image lanes answer 501 — loud, never a silent allow-all.
|
|
1562
|
+
*/
|
|
1563
|
+
supportsImages?(model: string | undefined, mode: 'generation' | 'edit'): Promise<boolean>;
|
|
1564
|
+
/**
|
|
1565
|
+
* The realtime/audio lane's modality gate (openai-realtime/binding.ts),
|
|
1566
|
+
* backed by ModelRouter.supportsAudio — the caller-org catalog's `task`
|
|
1567
|
+
* column (stt | tts, invariant across placements), per-caller by
|
|
1568
|
+
* construction; never function-name inference. Optional at the seam like
|
|
1569
|
+
* supportsImages: hand-built clients may omit it, in which case the
|
|
1570
|
+
* realtime lane refuses the upgrade LOUDLY (501) — never a silent
|
|
1571
|
+
* allow-all, and never a chat model binding an audio session.
|
|
1572
|
+
*/
|
|
1573
|
+
supportsAudio?(model: string | undefined): Promise<boolean>;
|
|
1574
|
+
/**
|
|
1575
|
+
* `GET /v1/images/models` — the image-capable slice of the model surface
|
|
1576
|
+
* (ModelRouter.imageModelList: deployed task-image apps + the shared block
|
|
1577
|
+
* joined against the catalog). Optional at the seam for the same reason;
|
|
1578
|
+
* its absence is the loud 501, never a silently empty list.
|
|
1579
|
+
*/
|
|
1580
|
+
listImageModels?(): Promise<unknown>;
|
|
1581
|
+
/**
|
|
1582
|
+
* THE PRE-DIAL DEFAULT FOR THE LEDGER'S BILLING IDENTITY
|
|
1583
|
+
* (ModelRouter.catalogIdentityFor): the CATALOG `(model_id, variant)` a
|
|
1584
|
+
* request's `model` actually resolved to, or `null` when it names no catalog
|
|
1585
|
+
* model (a caller-org app, whose identity is an app slug — a different
|
|
1586
|
+
* namespace from `model_prices.model_id`).
|
|
1587
|
+
*
|
|
1588
|
+
* The ledger prices on `(model_id, variant)`, and the caller's raw string is
|
|
1589
|
+
* not that: a bare catalog id resolves to the model's `compat_default`
|
|
1590
|
+
* variant, so recording the string would file the request under a variant
|
|
1591
|
+
* nobody served.
|
|
1592
|
+
*
|
|
1593
|
+
* THE DIAL HAS THE LAST WORD (ENG-414). What a served row records is the
|
|
1594
|
+
* identity `onDial` reported, off the acquisition that dial used — so a
|
|
1595
|
+
* re-homed request is billed on the replacement. This seam is what a request
|
|
1596
|
+
* that reaches NO dial records instead, which is why it is still resolved
|
|
1597
|
+
* on the request path.
|
|
1598
|
+
*
|
|
1599
|
+
* REQUIRED AT THE SEAM (ENG-429), unlike `supportsImages`. ENG-414 changed
|
|
1600
|
+
* what an omission MEANS: a backhaul with no `catalogIdentity` but a
|
|
1601
|
+
* dialling `createResponse` used to write no ledger row at all, and now
|
|
1602
|
+
* writes — and prices — one on the dial's own report. Honest, but it means
|
|
1603
|
+
* a lane could acquire BILLING behaviour by OMITTING a member, which is the
|
|
1604
|
+
* wrong direction for a default on a revenue path. So "no seam" is a
|
|
1605
|
+
* COMPILE ERROR now, the same move as the exhaustive `Record<...>`
|
|
1606
|
+
* registration checks used elsewhere in this codebase: a contract the
|
|
1607
|
+
* typechecker holds, not a convention a reviewer must notice. An implementer
|
|
1608
|
+
* that genuinely has no catalog oracle says so EXPLICITLY — a function of
|
|
1609
|
+
* its own, in code a reviewer can see — never by leaving the member out, and
|
|
1610
|
+
* never an invented identity. (The runtime guard in server.ts
|
|
1611
|
+
* `pinCatalogIdentity` stays for UNTYPED embedders: the seam is a public
|
|
1612
|
+
* surface callable from plain JS, where a member can still be absent at
|
|
1613
|
+
* runtime, and the guard's answer there remains the loud `unresolved` pin.)
|
|
1614
|
+
*/
|
|
1615
|
+
catalogIdentity(model: string | undefined): Promise<{
|
|
1616
|
+
model_id: string;
|
|
1617
|
+
variant: string;
|
|
1618
|
+
} | null>;
|
|
1619
|
+
/**
|
|
1620
|
+
* THE CALLER-ORG APP's OWN IDENTITY (ENG-412): the app slug `model` actually
|
|
1621
|
+
* RESOLVES to, for the lane where {@link catalogIdentity} answers null. Like
|
|
1622
|
+
* {@link catalogIdentity} it is the PRE-DIAL DEFAULT — the dial reports this
|
|
1623
|
+
* flavour too, and has the last word (ENG-414).
|
|
1624
|
+
*
|
|
1625
|
+
* IT IS NOT THE CALLER'S RAW STRING. Routing trims, length-caps and matches
|
|
1626
|
+
* that string before selecting an app, so a differently cased, punctuated or
|
|
1627
|
+
* whitespace-padded name is served by one canonical slug while the raw value
|
|
1628
|
+
* says something else — and the ledger would then file two spellings of the
|
|
1629
|
+
* same app as two different identities, which no app-level reconciliation
|
|
1630
|
+
* can undo afterwards.
|
|
1631
|
+
*
|
|
1632
|
+
* CHEAP AND ALLOCATION-FREE: it rides `ModelRouter.resolveTarget`, whose app
|
|
1633
|
+
* listing, catalog and shared block are cached promises. It opens no session
|
|
1634
|
+
* — asking which app would serve must never BE the thing that spends GPU
|
|
1635
|
+
* capacity.
|
|
1636
|
+
*/
|
|
1637
|
+
servingAppSlug?(model: string | undefined): Promise<string | null>;
|
|
1638
|
+
/**
|
|
1639
|
+
* Open (or reuse) the NATIVE audio lanes on the pooled session `model` routes
|
|
1640
|
+
* to — the same first-party machinery as RealtimeClient.enableAudio
|
|
1641
|
+
* (transport/media.ts `enableSessionAudio`: AudioBridge over the session's
|
|
1642
|
+
* `audio` stream lane at 24 kHz). One lane per pooled
|
|
1643
|
+
* session; repeat calls return the same lane. Optional at the seam because
|
|
1644
|
+
* text-only embeddings exist — but a surface that RECEIVES audio while the
|
|
1645
|
+
* embedder wired no `openAudio` must fail LOUD, never drop chunks.
|
|
1646
|
+
*/
|
|
1647
|
+
openAudio?(model: string | undefined): Promise<ProxyAudioLane>;
|
|
1648
|
+
/**
|
|
1649
|
+
* Open (or reuse) the NATIVE video FRAME lane on the pooled session `model`
|
|
1650
|
+
* routes to (transport/media.ts `enableSessionVideo`: discrete JPEG frames →
|
|
1651
|
+
* `stream('rt-video-in').emit`, the §5 named-DATA image-bytes path — NOT the
|
|
1652
|
+
* RTP media plane, which requires an already-H.264-encoded track). One lane
|
|
1653
|
+
* per pooled session; repeat calls return the same lane. Optional at the seam
|
|
1654
|
+
* because text-only embeddings exist — but a surface that RECEIVES video
|
|
1655
|
+
* frames while the embedder wired no `openVideo` must fail LOUD, never drop
|
|
1656
|
+
* frames.
|
|
1657
|
+
*/
|
|
1658
|
+
openVideo?(model: string | undefined): Promise<ProxyVideoLane>;
|
|
1659
|
+
/**
|
|
1660
|
+
* Open (or reuse) the NATIVE video OUTPUT lane on the pooled session
|
|
1661
|
+
* `model` routes to (transport/video-out.ts `openSessionVideoOutLane`):
|
|
1662
|
+
* decoded spec-v1 records consumed from the session's named §5
|
|
1663
|
+
* `rt-video-out` downstream. The Live adapter fans each record onto the
|
|
1664
|
+
* SAME Bidi WS as an extension server message (spec v2, WS-inline) — no
|
|
1665
|
+
* side-channel transport. Optional at the seam; an ABSENT seam means the
|
|
1666
|
+
* `urun.videoOut` capability is refused by omission in setupComplete
|
|
1667
|
+
* (vanilla parity), while a wired seam that cannot open the lane must
|
|
1668
|
+
* throw LOUD — never a quiet downgrade.
|
|
1669
|
+
*/
|
|
1670
|
+
openVideoOut?(model: string | undefined): Promise<ProxyVideoOutLane>;
|
|
1671
|
+
/**
|
|
1672
|
+
* SESSION-IDENTITY SEAM (ModelRouter.handleFor): the opaque stable handle
|
|
1673
|
+
* for the pooled session currently serving `model`'s turns. Derived from
|
|
1674
|
+
* native identity (app slug + uRun session id) — the same identity the
|
|
1675
|
+
* serve-side session-affinity tag rides (urun-python#1556/#1582).
|
|
1676
|
+
*/
|
|
1677
|
+
sessionHandle(model: string | undefined): Promise<string>;
|
|
1678
|
+
/**
|
|
1679
|
+
* createResponse PINNED to the exact session a handle names
|
|
1680
|
+
* (ModelRouter.sessionForHandle). Throws SessionGoneError LOUDLY when that
|
|
1681
|
+
* session is gone or was replaced — never silently opens a fresh session
|
|
1682
|
+
* while claiming resume. Deliberately NO re-home on this path: re-homing
|
|
1683
|
+
* would swap the pinned session out from under the caller.
|
|
1684
|
+
*/
|
|
1685
|
+
createResponseOn(handle: string, params: ResponsesCreateParams): Promise<AsyncIterable<unknown>> | AsyncIterable<unknown>;
|
|
1686
|
+
/**
|
|
1687
|
+
* Subscribe to the pinned session's terminal end via core's NATIVE phase
|
|
1688
|
+
* machinery (Session.onPhase → terminal 'expired'/'ended'/'error'). Fires
|
|
1689
|
+
* `cb` once. Throws SessionGoneError if the handle's session is already
|
|
1690
|
+
* gone — which doubles as the loud reattach check at resume time. Returns
|
|
1691
|
+
* the unsubscribe.
|
|
1692
|
+
*/
|
|
1693
|
+
onSessionEnd(handle: string, cb: (end: SessionEndInfo) => void): Promise<() => void>;
|
|
1694
|
+
}
|
|
1695
|
+
/**
|
|
1696
|
+
* The audio lane handle `openAudio` returns — structurally the transport
|
|
1697
|
+
* AudioBridge (media.ts): base64 PCM16 @24 kHz mono in both directions.
|
|
1698
|
+
*/
|
|
1699
|
+
type ProxyAudioLane = Pick<AudioBridge, 'appendInputAudio' | 'onOutputAudio'>;
|
|
1700
|
+
/**
|
|
1701
|
+
* The video frame-lane handle `openVideo` returns — structurally the
|
|
1702
|
+
* transport VideoFrameLane (media.ts): one raw encoded JPEG frame per call,
|
|
1703
|
+
* input-only (Live-style protocols have no video OUT modality).
|
|
1704
|
+
*/
|
|
1705
|
+
type ProxyVideoLane = Pick<VideoFrameLane, 'sendInputFrame'>;
|
|
1706
|
+
/**
|
|
1707
|
+
* The video-OUT lane handle `openVideoOut` returns — structurally the
|
|
1708
|
+
* transport VideoOutLane (video-out.ts): the codec plus frame/error fan-out
|
|
1709
|
+
* the WS-inline delivery subscribes.
|
|
1710
|
+
*/
|
|
1711
|
+
type ProxyVideoOutLane = Pick<VideoOutLane, 'codec' | 'onOutputFrame' | 'onLaneError'>;
|
|
1712
|
+
/**
|
|
1713
|
+
* Per-request `ProxyClients` resolution — the seam the HOSTED multi-tenant
|
|
1714
|
+
* server (`src/hosted/`) plugs into so ONE handler implementation serves
|
|
1715
|
+
* every org: the hosted server resolves the caller's org from its Bearer API
|
|
1716
|
+
* key and returns THAT org's backhaul. The local CLI passes a fixed
|
|
1717
|
+
* `ProxyClients` instead; both go through the identical handler body.
|
|
1718
|
+
*
|
|
1719
|
+
* Throwing from here is the loud path: the thrown error surfaces in the
|
|
1720
|
+
* lane's native error envelope (see {@link ProxyHandlerOptions.statusOf}).
|
|
1721
|
+
*/
|
|
1722
|
+
type ProxyClientsFor = (req: IncomingMessage) => ProxyClients | Promise<ProxyClients>;
|
|
1723
|
+
interface ProxyHandlerOptions {
|
|
1724
|
+
/** A fixed backhaul (local CLI) or a per-request resolver (hosted server). */
|
|
1725
|
+
clients: ProxyClients | ProxyClientsFor;
|
|
1726
|
+
/** Optional bearer the local agent must present (never forwarded upstream). */
|
|
1727
|
+
apiKey?: string;
|
|
1728
|
+
/**
|
|
1729
|
+
* WHAT this proxy serves (app/org/fn/base_url + proxy_version), surfaced as
|
|
1730
|
+
* the `identity` block on `GET /stats` so a second `urun compat` invocation
|
|
1731
|
+
* can reuse this proxy iff the identity matches its own exactly. Only the
|
|
1732
|
+
* standalone `urun compat proxy` command sets it — a launch-mode proxy is
|
|
1733
|
+
* child-owned (it dies with its child) and an embedder that omits it is
|
|
1734
|
+
* simply never reused.
|
|
1735
|
+
*/
|
|
1736
|
+
identity?: ProxyIdentity;
|
|
1737
|
+
/**
|
|
1738
|
+
* Map a thrown error onto an HTTP status + error `type` before the generic
|
|
1739
|
+
* 500. The hosted server uses it to turn its auth/tenancy failures into a
|
|
1740
|
+
* 401 in the OpenAI envelope. Returning null means "not mine" — the error
|
|
1741
|
+
* takes the ordinary loud 500 path.
|
|
1742
|
+
*/
|
|
1743
|
+
statusOf?: (err: unknown) => {
|
|
1744
|
+
status: number;
|
|
1745
|
+
openaiType: string;
|
|
1746
|
+
anthropicType: string;
|
|
1747
|
+
} | null;
|
|
1748
|
+
/**
|
|
1749
|
+
* Where the per-request structured log line goes — ONE newline-terminated
|
|
1750
|
+
* JSON object per `/v1` request, carrying the request's `Inference-Id`
|
|
1751
|
+
* (proxy/inference-id.ts `inferenceRequestLine`) plus method, path, status
|
|
1752
|
+
* and whether the answer was delivered / failed mid-stream, written when the
|
|
1753
|
+
* response closes (so streams and client aborts are recorded too). This is
|
|
1754
|
+
* the billing correlation record a later usage ledger is keyed on, so it is
|
|
1755
|
+
* emitted unconditionally — never behind a verbosity flag.
|
|
1756
|
+
*
|
|
1757
|
+
* Absent, the record goes to this process's stdout through the repo's
|
|
1758
|
+
* canonical guarded write (`server.ts` `writeInferenceLogToStdout`), which is
|
|
1759
|
+
* the hosted endpoint's collected container log. The local CLI proxy passes
|
|
1760
|
+
* its own sink instead, because the operator owns that terminal — see
|
|
1761
|
+
* `proxy/cli.ts` `startProxy`. Injectable so tests assert on the lines
|
|
1762
|
+
* without racing the process streams.
|
|
1763
|
+
*/
|
|
1764
|
+
requestLog?: (line: string) => void;
|
|
1765
|
+
/**
|
|
1766
|
+
* THE CANONICAL USAGE-QUERY SEAM (`POST /v1/usage/requests`, and the
|
|
1767
|
+
* Hugging Face translator over it) — a batch lookup of what the per-request
|
|
1768
|
+
* inference ledger holds for a list of `Inference-Id`s, scoped to the
|
|
1769
|
+
* caller's own org. See proxy/usage.ts.
|
|
1770
|
+
*
|
|
1771
|
+
* Only the HOSTED endpoint supplies one (the read rides the caller's own
|
|
1772
|
+
* org API key to the control plane; this process holds no database
|
|
1773
|
+
* credential). A proxy without it answers those routes with a LOUD 501 —
|
|
1774
|
+
* the `supportsImages` posture — and never with an empty result, which a
|
|
1775
|
+
* billing caller cannot tell from "nothing is priced yet".
|
|
1776
|
+
*/
|
|
1777
|
+
usage?: UsageLane;
|
|
1778
|
+
/**
|
|
1779
|
+
* THE PER-REQUEST LEDGER WRITE SEAM (proxy/ledger.ts) — one durable,
|
|
1780
|
+
* already-priced row per billable `/v1` request, written from the response's
|
|
1781
|
+
* `'close'` hook and therefore OFF the response path.
|
|
1782
|
+
*
|
|
1783
|
+
* Only the HOSTED endpoint supplies one (the write rides the caller's own
|
|
1784
|
+
* org API key to the control plane; this process holds no database
|
|
1785
|
+
* credential). A proxy WITHOUT one does not participate in the ledger at
|
|
1786
|
+
* all — the `usage` / `supportsImages` posture of declared absence. The
|
|
1787
|
+
* local CLI proxy is single-tenant and bills nothing, so it has none.
|
|
1788
|
+
*
|
|
1789
|
+
* It is deliberately NOT a place to do work on the hot path, and it cannot
|
|
1790
|
+
* be: `recordInference` is fire-and-forget, returns void, and never throws.
|
|
1791
|
+
*/
|
|
1792
|
+
ledger?: LedgerLane;
|
|
1793
|
+
/**
|
|
1794
|
+
* The SECOND sink for the very same per-request record `requestLog`
|
|
1795
|
+
* serializes — the hosted endpoint passes its Prometheus collector
|
|
1796
|
+
* (`proxy/metrics.ts` `InferenceMetrics.observe`). Called once per `/v1`
|
|
1797
|
+
* request, from the same `'close'` hook, with the same object.
|
|
1798
|
+
*
|
|
1799
|
+
* Optional because the local CLI proxy publishes no metrics port; absent, a
|
|
1800
|
+
* request is logged and not counted. It is deliberately NOT a place to do
|
|
1801
|
+
* work: it runs on the response's close hook, so anything slow here delays
|
|
1802
|
+
* the socket teardown.
|
|
1803
|
+
*/
|
|
1804
|
+
observe?: (record: InferenceRecord) => void;
|
|
1805
|
+
}
|
|
1806
|
+
|
|
1807
|
+
export { type DialObserver as D, type InferenceRecord as I, type LedgerFailure as L, ModelRouter as M, type ProxyClients as P, SessionGoneError as S, UnknownModelError as U, type DialReport as a, type InferenceUsageRecord as b, type LedgerOutcome as c, type LedgerRecorded as d, type LedgerWrite as e, type ProxyHandlerOptions as f, type ProxyIdentity as g, type ProxyVideoOutLane as h, type UsageRowReporter as i };
|