@urun-sh/openai 0.5.4 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/dist/{ResponsesClient-BYx3YLGo.d.ts → ResponsesClient-BE3-hx3m.d.ts} +2 -0
  2. package/dist/{ResponsesClient-Dft3bg3b.d.cts → ResponsesClient-DjZWjlFH.d.cts} +2 -0
  3. package/dist/chunk-2T2YYBVX.js +1 -0
  4. package/dist/chunk-3WAJD62J.js +4 -0
  5. package/dist/{chunk-K23AZPI4.js → chunk-5HKWNK3O.js} +1 -1
  6. package/dist/chunk-7M6LY6DR.js +2 -0
  7. package/dist/chunk-BSHT6RZZ.js +1 -0
  8. package/dist/{chunk-GBBY3PCZ.js → chunk-FP4RSAIE.js} +1 -1
  9. package/dist/chunk-HR5H6S7L.js +1 -0
  10. package/dist/chunk-NY23USZF.js +57 -0
  11. package/dist/chunk-VLRMJRLS.js +6 -0
  12. package/dist/gemini-live.cjs +2 -2
  13. package/dist/gemini-live.d.cts +17 -2
  14. package/dist/gemini-live.d.ts +7 -2
  15. package/dist/gemini-live.js +1 -1
  16. package/dist/hosted/bin.cjs +65 -37
  17. package/dist/hosted/bin.js +3 -3
  18. package/dist/hosted/index.cjs +46 -19
  19. package/dist/hosted/index.d.cts +760 -272
  20. package/dist/hosted/index.d.ts +203 -107
  21. package/dist/hosted/index.js +1 -1
  22. package/dist/index.cjs +1 -1
  23. package/dist/index.d.cts +4 -4
  24. package/dist/index.d.ts +4 -4
  25. package/dist/index.js +1 -1
  26. package/dist/{models-NYMZrklp.d.cts → models-DUdx_Y6X.d.cts} +1 -1
  27. package/dist/pi-extension/index.cjs +7 -7
  28. package/dist/pi-extension/index.d.cts +1 -1
  29. package/dist/pi-extension/index.d.ts +1 -1
  30. package/dist/pi-extension/index.js +1 -1
  31. package/dist/pi-extension/standalone.cjs +48 -48
  32. package/dist/proxy/cli.cjs +54 -40
  33. package/dist/proxy/cli.js +14 -14
  34. package/dist/proxy/index.cjs +33 -22
  35. package/dist/proxy/index.d.cts +546 -4
  36. package/dist/proxy/index.d.ts +185 -4
  37. package/dist/proxy/index.js +1 -1
  38. package/dist/responses-turn-Dr37N3kH.d.ts +486 -0
  39. package/dist/responses-turn-tNUHq5b0.d.cts +1807 -0
  40. package/dist/{translator-C9uPKypK.d.ts → translator-BCoRFaTs.d.ts} +1 -1
  41. package/dist/{translator-CcDBEfvm.d.cts → translator-Bh0Bp_Ie.d.cts} +3 -3
  42. package/dist/{video-out-D20UuJ8G.d.cts → video-out-CCksIcj8.d.cts} +5 -5
  43. package/package.json +12 -11
  44. package/dist/chunk-4MSBS7M6.js +0 -1
  45. package/dist/chunk-5NXM4IO3.js +0 -2
  46. package/dist/chunk-CWJRDBDC.js +0 -1
  47. package/dist/chunk-OI2OY32M.js +0 -1
  48. package/dist/chunk-RQBD4OAA.js +0 -6
  49. package/dist/chunk-STFRT5OY.js +0 -1
  50. package/dist/chunk-VNM4M4P3.js +0 -40
  51. package/dist/server-DuH_5OD9.d.ts +0 -84
  52. package/dist/server-wmUnMewW.d.cts +0 -206
  53. /package/dist/{chunk-YSFSRI3D.js → chunk-5CCJUNH7.js} +0 -0
  54. /package/dist/{models-NYMZrklp.d.ts → models-DUdx_Y6X.d.ts} +0 -0
  55. /package/dist/{video-out-CWesbk12.d.ts → video-out-IGQ4YQ-c.d.ts} +0 -0
@@ -1,283 +1,147 @@
1
1
  import { Server } from 'node:http';
2
+ import * as _prometheus_io_client from '@prometheus-io/client';
3
+ import { Registry } from '@prometheus-io/client';
4
+ import { c as LedgerOutcome, I as InferenceRecord, S as SessionGoneError, P as ProxyClients, M as ModelRouter, i as UsageRowReporter, b as InferenceUsageRecord, e as LedgerWrite, d as LedgerRecorded, L as LedgerFailure } from '../responses-turn-tNUHq5b0.cjs';
2
5
  import { createClientToken } from '@urun-sh/core';
3
- import { U as UrunResponses } from '../ResponsesClient-Dft3bg3b.cjs';
6
+ import { U as UrunResponses } from '../ResponsesClient-DjZWjlFH.cjs';
4
7
  import { b as UrunSessionLike } from '../types-lsVTbNcH.cjs';
5
- import { C as CatalogRow } from '../models-NYMZrklp.cjs';
6
- import { P as ProxyClients } from '../server-wmUnMewW.cjs';
7
- import '../video-out-D20UuJ8G.cjs';
8
+ import '../models-DUdx_Y6X.cjs';
9
+ import '../video-out-CCksIcj8.cjs';
8
10
 
9
11
  /**
10
- * THE OPENROUTER PROVIDER DOCUMENT — `GET /v1/models?format=openrouter`.
12
+ * The container's metrics port — a MODULE CONSTANT, not an env var, per the
13
+ * hosted config mandate (`hosted/config.ts`: "env vars are NOT a config
14
+ * mechanism"). 9464 is the Prometheus exporter port the OpenTelemetry
15
+ * ecosystem registered, and it is the interface contract the deployment chart
16
+ * and its ServiceMonitor are built against.
11
17
  *
12
- * OpenRouter's provider monitor polls this document (schema 2.4, per
13
- * openrouter.ai/docs/guides/community/for-providers) to list a provider's
14
- * models on the marketplace. The `id` field is the EXACT model identifier
15
- * OpenRouter sends back as `model` on chat completions — which here is the
16
- * app slug, the same id the OpenAI surface lists. One id space, two views.
17
- *
18
- * MODALITY GATING: only rows whose catalog `task` names a chat-completions
19
- * shape (chat | code | agent | vl) are declared. A voice/video/image app
20
- * behind a chat surface would stream garbage into OpenRouter's baseline
21
- * tests; models the catalog cannot vouch for are OMITTED, never guessed.
22
- * With no catalog oracle configured the document is `{"data": []}` — honest
23
- * emptiness, not invented capability (the chart turns the oracle on).
24
- *
25
- * PRICING is deliberately omitted entirely: the schema's rule is "a modality
26
- * with no pricing array is simply unpriced", OpenRouter configures pricing
27
- * during provider onboarding, and the per-model price surface is
28
- * urun-infra U3 (model_prices) — when that lands, this module grows a
29
- * `prices` input and emits the input/output pricing arrays.
30
- *
31
- * The full closed-value-domain schema ships as OpenAPI 3.1 at
32
- * openrouter.ai/docs/assets/provider-monitor-schema-v2.openapi.json.
18
+ * PRIVATE, and private STRUCTURALLY rather than by convention: the public
19
+ * HTTPRoute allowlist (`k8s-manifests` `_public-matches.tpl`) routes only
20
+ * `/v1` to the Service's 8080 port, so nothing on this port is reachable from
21
+ * the internet. It is served by its OWN `http.Server`
22
+ * ({@link createMetricsServer}) rather than as a route on the main server, so
23
+ * a future public path can never accidentally expose it.
33
24
  */
34
-
35
- type OpenRouterProviderDoc = {
36
- data: OpenRouterProviderModel[];
37
- };
38
- interface OpenRouterProviderModel {
39
- schema_version: '2.4';
40
- /** The EXACT id OpenRouter sends back as `model` — the app slug. */
41
- id: string;
42
- name: string;
43
- created: number;
44
- /** Valid enum: int4|int8|fp4|mxfp4|nvfp4|fp6|fp8|mxfp8|fp16|bf16|fp32|null. */
45
- quantization: string | null;
46
- description: string;
47
- hugging_face_id: string;
48
- input_modalities: Array<Record<string, unknown>>;
49
- output_modalities: Array<Record<string, unknown>>;
50
- }
51
-
25
+ declare const METRICS_PORT = 9464;
26
+ /** The path the ServiceMonitor scrapes. */
27
+ declare const METRICS_PATH = "/metrics";
28
+ /** The label value every bounded label collapses to once its budget is spent. */
29
+ declare const OTHER_LABEL = "(other)";
30
+ /** The label value for a request that named no model. */
31
+ declare const NO_MODEL_LABEL = "(none)";
52
32
  /**
53
- * Per-model → per-app routing for the compat proxy (owner directive
54
- * 2026-08-07): swapping the model in a coding harness routes the request to
55
- * the org's DEPLOYED app for that model. v1 is deployed-only — a uRun model
56
- * that is not deployed gets a loud 404 naming `urun serve <id>`; a later
57
- * phase (explicitly out of scope here; urun-infra#1490 shared-endpoints)
58
- * auto-creates from the model catalog on first request.
59
- *
60
- * MODEL-ID SURFACE (the documented mapping): a model may be named by
61
- * - the app slug itself ("qwen3-6-27b-bf16"), or
62
- * - the catalog id ("qwen3.6-27b"), or
63
- * - the catalog id:variant ("qwen3.6-27b:bf16"),
64
- * where slugification mirrors urun-cli `serve.py _default_app_name` exactly:
65
- * lowercase, every non-alphanumeric-non-dash character becomes "-", leading/
66
- * trailing dashes stripped (catalog id "qwen3.6-27b" + variant "bf16" → app
67
- * "qwen3-6-27b-bf16"). COLLISION RULE: an exact slug match always wins over
68
- * the catalog-id (prefix) interpretation.
69
- *
70
- * RESOLUTION ORDER (one canonical path, documented end to end):
71
- * 1. model absent / "urun" / an alias of the startup app → DEFAULT app.
72
- * 2. exact slug match on a deployed serve app → that app.
73
- * 3. catalog-id form matching exactly one deployed app → that app
74
- * (two or more candidates → loud ambiguity error naming them).
75
- * 4. the name maps to an org app that is NOT an active app exposing the
76
- * proxy's serve function → loud 404
77
- * naming `urun serve <model>` and the available models.
78
- * 5. the name matches a catalog row but no deployed app → loud 404
79
- * naming `urun serve <id>` (deployed-only v1).
80
- * 6. anything else — a model name outside the uRun namespace entirely
81
- * (e.g. the harness's own upstream default, "claude-*"/"gpt-*") →
82
- * DEFAULT app. This IS today's single-app contract, kept deliberately
83
- * so `urun compat <agent>` with the agent's stock model keeps working
84
- * with zero new env; the per-model /stats table records every such
85
- * mapping so it is visible, never silent. Models the proxy ADVERTISES
86
- * on /v1/models can never land here — they resolve (2/3) or fail loud
87
- * (4/5) above.
88
- *
89
- * NO-DEFAULT MODE (`defaultApp: null`) — the HOSTED multi-tenant lane
90
- * (`src/hosted/`): a shared endpoint serving every org has no "the app this
91
- * proxy was started for", so rules 1 and 6 have nothing to fall back TO.
92
- * Rather than inventing one (picking "some" app for a caller would be the
93
- * worst kind of silent divergence), both rules become the SAME loud
94
- * {@link UnknownModelError} that rules 4/5 already raise: name a deployed
95
- * model, here is the list. Rules 2–5 are byte-for-byte the local behavior —
96
- * one router, one resolution order, two configurations.
33
+ * Distinct `model` values that get their own series. Above this the label
34
+ * collapses to {@link OTHER_LABEL} — a caller sending a fresh random model
35
+ * string per request must not be able to grow this process's series set (and
36
+ * therefore Prometheus's) without bound.
97
37
  */
98
-
99
- /** One org app row from `GET {orgApi}/apps` (urun-cli `ApiClient.list_apps`). */
100
- interface DeployedApp {
101
- app_slug: string;
102
- function_name?: string | null;
103
- deployment_status?: string | null;
104
- [k: string]: unknown;
105
- }
38
+ declare const MAX_MODEL_LABELS = 64;
106
39
  /**
107
- * A session handle names a pooled session that no longer exists (closed,
108
- * evicted after its pod died, replaced by a re-home, or the proxy restarted).
109
- * The caller asked to REATTACH that exact session — opening a fresh one and
110
- * calling it "resumed" would be a silent lie, so this is always loud.
40
+ * Distinct `org` values that get their own series. Bounded for the same reason
41
+ * even though org ids come from the control plane rather than the caller: the
42
+ * number of orgs is unbounded over time, and a metric's cardinality must be a
43
+ * property of the code, not of how successful sales were.
111
44
  */
112
- declare class SessionGoneError extends Error {
113
- }
45
+ declare const MAX_ORG_LABELS = 128;
46
+ /** Collapse a client-supplied path onto the bounded lane label. */
47
+ declare function laneOf(path: string): string;
114
48
  /**
115
- * OpenAI-shaped model list (same shape as models.ts listModels). When the
116
- * catalog oracle is configured, each entry ALSO carries the catalog
117
- * enrichment fields (additive JSON — OpenAI clients ignore unknown fields;
118
- * the OpenRouter + Vercel AI Gateway provider listings need them). Entries
119
- * whose slug matches no catalog row stay at the bare four fields.
120
- */
121
- interface RouterModelEntry {
122
- id: string;
123
- object: 'model';
124
- created: number;
125
- owned_by: string;
126
- /** Display name — the canonical `<model_id>:<variant>` catalog ref. */
127
- name?: string;
128
- /** User-facing description from the catalog (console Endpoints copy). */
129
- description?: string;
130
- /** Catalog modality (chat | code | agent | vl | audio | image | ...). */
131
- task?: string;
132
- /** Serving engine (vllm | sglang | llamacpp | ...). */
133
- engine?: string;
134
- /** The catalog lane this app deploys, e.g. 'rtx6000:1'. */
135
- gpu_spec?: string;
136
- /** Context window in tokens (chat rows carry it in engine_args). */
137
- context_length?: number;
138
- }
139
- interface RouterModelList {
140
- object: 'list';
141
- data: RouterModelEntry[];
142
- }
143
- interface ModelRouterOptions<S> {
144
- /**
145
- * The startup app slug (URUN_APP) — the DEFAULT model — or `null` for the
146
- * hosted multi-tenant lane, which has no per-proxy default app: there,
147
- * every request must NAME a deployed model and an unnamed/unknown one
148
- * fails loud instead of silently landing somewhere (see the module header,
149
- * "NO-DEFAULT MODE").
150
- */
151
- defaultApp: string | null;
152
- /** The serve function name every routed app must expose (URUN_FUNCTION). */
153
- fnName: string;
154
- /** Open a backhaul session for an app slug (called at most once per app). */
155
- openSession: (appSlug: string) => S | Promise<S>;
156
- /** Terminal release for one pool entry (Session.end() underneath). */
157
- closeSession: (entry: S) => Promise<void>;
158
- /**
159
- * List the org's deployed apps, or null when the credentials cannot
160
- * (URUN_JWT lane: the pre-vended token is scoped to the default app, so
161
- * there is no org listing AND no cross-app session — routing degrades to
162
- * the default-app-only contract, which is exactly today's behavior).
163
- */
164
- listApps: (() => Promise<DeployedApp[]>) | null;
165
- /**
166
- * Catalog rows (models.ts fetchCatalogRows) as the uRun-namespace oracle
167
- * for rule 5 AND the /v1/models enrichment + OpenRouter provider-doc
168
- * source, or null when catalog access is not configured. Optional fields
169
- * beyond model_id/variant are tolerated (thin rows still typecheck).
170
- */
171
- listCatalog: (() => Promise<CatalogRow[]>) | null;
172
- /** Deployed-apps cache TTL (the list changes on deploys, not per request). */
173
- appsTtlMs?: number;
49
+ * The status class label. `none` is its own class rather than being folded
50
+ * into 5xx: it means NO head ever flushed (an aborted upload, a client that
51
+ * vanished during the body read), which `inferenceRequestRecord` records as a
52
+ * null status precisely so nothing downstream invents a 200 the client was
53
+ * never sent.
54
+ */
55
+ declare function statusClassOf(status: number | null): string;
56
+ /**
57
+ * A label whose distinct-value budget is FIXED at construction. The first
58
+ * `max` distinct values are ADMITTED; everything after is {@link OTHER_LABEL},
59
+ * permanently. Deliberately NOT an LRU: an LRU would let a flooding caller
60
+ * evict the real models and quietly rewrite history in the TSDB, where each
61
+ * admitted-then-evicted-then-readmitted value leaves a broken series behind.
62
+ *
63
+ * ADMISSION IS EARNED, AND THAT IS THE WHOLE DEFENCE. A budget alone does not
64
+ * protect anything: one authenticated caller sending `MAX_MODEL_LABELS`
65
+ * requests naming random models — every one of them answered 404, because an
66
+ * unknown model is refused — would spend the entire budget and collapse every
67
+ * REAL model onto `(other)` until the pod restarts. That kills the per-model
68
+ * TTFT and rate panels, which are exactly the two the Hugging Face latency
69
+ * gate is read from. So a value is admitted only when `admit` is true, and the
70
+ * caller passes `admit` for answers that were actually served; a refusal can
71
+ * still be COUNTED (under `(other)`) but can never consume budget.
72
+ */
73
+ declare class BoundedLabel {
74
+ private readonly max;
75
+ private readonly admitted;
76
+ private overflows;
77
+ constructor(max: number);
174
78
  /**
175
- * The stable NATIVE identity of one pooled entry (the uRun session id in
176
- * the proxy wiring — the same identity the serve-side session-affinity tag
177
- * rides, urun-python#1556). Powers the session-identity seam
178
- * ({@link ModelRouter.handleFor} / {@link ModelRouter.sessionForHandle});
179
- * a router without it fails LOUD on those calls, never approximates.
79
+ * @param admit whether this observation may spend budget on a new value.
80
+ * False for refused requests, so a caller cannot blind the label by naming
81
+ * models that do not exist.
180
82
  */
181
- sessionKey?: (entry: S) => string;
83
+ of(value: string | null, admit?: boolean): string;
84
+ /** Distinct values admitted so far — published as a gauge (see below). */
85
+ get size(): number;
86
+ /** Admissible observations refused because the budget was spent. */
87
+ get overflowCount(): number;
182
88
  }
183
89
  /**
184
- * The session pool: one backhaul session per deployed app, keyed by app slug,
185
- * opened lazily on the first request that routes to it and reused for every
186
- * subsequent one. The startup app is seeded eagerly by the CLI. Sessions
187
- * close on proxy shutdown via {@link closeAll}; there is NO idle-close policy
188
- * (deliberate v1 simplification — noted as a follow-up in the PR).
90
+ * The hosted proxy's metric set, over its OWN {@link Registry} rather than the
91
+ * library's global `register`. Per-instance because the alternative is
92
+ * process-global mutable state that two tests (or two embedders) silently
93
+ * share — `Counter` construction on an already-registered name throws, so a
94
+ * global registry would make the second `createHostedProxy` in a process fail.
189
95
  */
190
- declare class ModelRouter<S> {
191
- private readonly opts;
192
- private readonly pool;
193
- private appsCache;
194
- private rowsCache;
195
- constructor(opts: ModelRouterOptions<S>);
196
- /** Seed an already-open session (the CLI's eagerly-opened startup app). */
197
- seed(appSlug: string, entry: S): void;
198
- private deployedApps;
96
+ declare class InferenceMetrics {
97
+ readonly registry: Registry<_prometheus_io_client.RegistryContentType>;
98
+ private readonly modelLabel;
99
+ private readonly orgLabel;
100
+ private readonly requests;
101
+ private readonly duration;
102
+ private readonly ttft;
103
+ private readonly streamErrors;
104
+ private readonly undelivered;
105
+ private readonly ledgerWrites;
106
+ private readonly ledgerQueue;
107
+ private readonly labelCardinality;
108
+ private readonly labelOverflows;
109
+ constructor(options?: {
110
+ defaultMetrics?: boolean;
111
+ });
199
112
  /**
200
- * Catalog rows with the same TTL discipline as {@link deployedApps} (the
201
- * catalog changes on reseed migrations, not per request). Null when no
202
- * oracle is configured — callers degrade to the bare four-field entries.
203
- */
204
- private catalogRows;
205
- /** Apps this proxy may serve: active AND exposing the serve function. */
206
- private servable;
207
- private availableIds;
208
- /**
209
- * NO-DEFAULT MODE's terminal for rules 1 and 6: there is no app to fall
210
- * back to, so say so loudly and list what the CALLER'S org actually has.
211
- * Never returns.
212
- */
213
- private noDefaultApp;
214
- /**
215
- * Resolve a request's `model` to an app slug — the documented resolution
216
- * order from the module header. Throws {@link UnknownModelError} for a uRun
217
- * model that is not deployed (rules 4/5).
218
- */
219
- resolveApp(model: string | undefined): Promise<string>;
220
- /** The pooled session for a model — opened lazily, reused afterwards. */
221
- sessionFor(model: string | undefined): Promise<{
222
- app: string;
223
- entry: S;
224
- }>;
225
- private keyOf;
226
- /**
227
- * SESSION-IDENTITY SEAM (a): the opaque stable handle for the pooled
228
- * session currently serving `model`'s turns. Rides the SAME acquisition
229
- * path as every request ({@link sessionFor}) — the session opens lazily if
230
- * this model has none yet — and derives the handle from native identity
231
- * (app slug + uRun session id), zero bespoke bookkeeping.
232
- */
233
- handleFor(model: string | undefined): Promise<{
234
- app: string;
235
- handle: string;
236
- }>;
237
- /**
238
- * SESSION-IDENTITY SEAM (b): the exact pooled session a handle names.
239
- * NEVER opens a fresh session — a handle whose session is gone (closed,
240
- * evicted, re-homed to a replacement, proxy restarted) or malformed throws
241
- * {@link SessionGoneError} loudly. Resume is reattach-or-fail, not
242
- * reattach-or-quietly-restart.
243
- */
244
- sessionForHandle(handle: string): Promise<{
245
- app: string;
246
- entry: S;
247
- }>;
248
- /**
249
- * Drop ONE pooled session whose backhaul died (its pod was restarted /
250
- * drained / deleted) and release it — the next {@link sessionFor} opens a
251
- * fresh one, i.e. asks the control plane for a new assignment. Used by the
252
- * one-shot re-home (rehome.ts, urun-sh/urun-python#1592).
113
+ * Count ONE terminal ledger outcome — the third sink for a request that
114
+ * `proxy/ledger.ts` already logs, counted so it can be alerted on.
253
115
  *
254
- * IDENTITY-GUARDED (the same rule the pi lane's SessionPool follows): a
255
- * concurrent request that already re-homed this app has put a NEWER entry
256
- * under the key, and evicting that would close a healthy session out from
257
- * under it.
258
- */
259
- evict(app: string, entry: S): Promise<void>;
260
- /**
261
- * `GET /v1/models`: the org's deployed serve apps as model entries, the
262
- * default app FIRST. On the JWT lane (no org listing) this is the default
263
- * app plus any app already in the pool — the gap is called out loudly in
264
- * the PR, not papered over here.
116
+ * Takes the very {@link LedgerOutcome} the log line carries, for the same
117
+ * reason {@link observe} takes the record verbatim: a counter derived from a
118
+ * second walk would eventually disagree with the billing log, and the
119
+ * disagreement would be invisible.
265
120
  */
266
- modelList(): Promise<RouterModelList>;
121
+ observeLedger(outcome: LedgerOutcome): void;
267
122
  /**
268
- * The OpenRouter PROVIDER document (`GET /v1/models?format=openrouter`):
269
- * schema 2.4 per openrouter.ai/docs/guides/community/for-providers. Only
270
- * chat-completions-shaped models are declared (task chat | code | agent |
271
- * vl — a voice or video app behind a chat surface would stream garbage);
272
- * models the catalog cannot vouch for are OMITTED, never guessed. Pricing
273
- * is deliberately omitted entirely (schema rule: "A modality with no
274
- * pricing array is simply unpriced" — OpenRouter configures pricing during
275
- * provider onboarding; the per-model price surface is urun-infra U3).
123
+ * Count ONE completed `/v1` request. Takes the very record
124
+ * `inferenceRequestLine` serializes, so the counters and the billing log can
125
+ * never drift apart.
276
126
  */
277
- openRouterModels(): Promise<OpenRouterProviderDoc>;
278
- /** Close every pooled session (Session.end() underneath) — proxy shutdown. */
279
- closeAll(): Promise<void>;
127
+ observe(record: InferenceRecord): void;
128
+ /** The exposition text a scrape answers with. */
129
+ scrape(): Promise<string>;
130
+ /** The `Content-Type` that exposition must be served under. */
131
+ get contentType(): string;
280
132
  }
133
+ /**
134
+ * The PRIVATE metrics listener — its own `http.Server` on
135
+ * {@link METRICS_PORT}, serving exactly `GET /metrics` and 404 for everything
136
+ * else.
137
+ *
138
+ * A SEPARATE SERVER, not a route on the main one, is the whole point: the
139
+ * public gateway forwards only to the main port, so "not public" is a property
140
+ * of the socket rather than a rule the `/v1` router has to keep remembering.
141
+ * It also means a scrape can still be answered while the main server is
142
+ * draining, which is exactly when the numbers matter most.
143
+ */
144
+ declare function createMetricsServer(metrics: InferenceMetrics): Server;
281
145
 
282
146
  /**
283
147
  * HOSTED AUTH — the Bearer key IS the identity AND the tenancy.
@@ -401,17 +265,24 @@ declare class OrgResolver {
401
265
  * `consecutiveFailures` resets on every healthy sync, so a recovered
402
266
  * blip never trips this.
403
267
  *
404
- * 2. Core Session's own phase surface (`Session.onPhase` → 'paused' /
405
- * 'error') — the compute binding the pool already tracks (the same signal
406
- * cli.ts `onSessionEnd` rides).
268
+ * Core Session's own phase surface (`Session.onPhase` → 'live' /
269
+ * 'paused' / 'error'). The verdict arms ONLY on a `live` sighting: a
270
+ * binding must EXIST before it can be lost. A session born into its
271
+ * pre-live wait (the cold wake) or re-attached in `error` (resume runs
272
+ * the function from the top, session-semantics.md) is ACQUIRING
273
+ * compute, never losing it — core's own whenLive gate treats both as
274
+ * WAITABLE, and the unguarded latch 502-looped every shared chat/agent
275
+ * model on exactly those sightings (prod 2026-09-14: the entry was
276
+ * evicted before it ever served, and "retry to get a fresh session"
277
+ * re-dialed into the same verdict forever).
407
278
  *
408
- * NOTE what "gone" means HERE, because it is narrower than it reads: this
409
- * pool holds a session with COMPUTE BOUND to it, and a `paused` phase says
410
- * that binding is gone. The session ITSELF is not gone — it is parked,
411
- * listed, and attachable by name (see
412
- * urun-python/docs/design/session-semantics.md). The eviction is correct
413
- * because the POOL ENTRY is what became unusable; nothing here is entitled
414
- * to conclude the session is over.
279
+ * NOTE what "gone" means HERE, because it is narrower than it reads:
280
+ * this pool holds a session with COMPUTE BOUND to it, and a post-live
281
+ * `paused` phase says that binding is gone. The session ITSELF is not
282
+ * gone — it is parked, listed, and attachable by name (see
283
+ * urun-python/docs/design/session-semantics.md). The eviction is
284
+ * correct because the POOL ENTRY is what became unusable; nothing here
285
+ * is entitled to conclude the session is over.
415
286
  *
416
287
  * Consumers (cli.ts):
417
288
  * - {@link watchSessionGone} per pooled entry; `onGone` → `router.evict`
@@ -479,12 +350,24 @@ interface CatalogConfig {
479
350
  */
480
351
  type OwnedSession = UrunSessionLike & {
481
352
  end: () => Promise<unknown>;
353
+ /**
354
+ * The SDK's NATIVE local release (core `Session.detach`): drop the media
355
+ * transport and decrement the reference count while the named session
356
+ * stays live and attachable. Optional here because only the
357
+ * `release: 'detach'` lane requires it — that lane fails LOUD when a
358
+ * session object lacks it (no approximation, no fake detach).
359
+ */
360
+ detach?: () => void;
482
361
  id?: string;
483
362
  endsAt?: Date | null;
484
363
  onPhase?: (handler: (phase: {
485
364
  name: string;
486
365
  reason?: string;
487
366
  }) => void) => () => void;
367
+ whenLive?: (options?: {
368
+ timeout?: number;
369
+ signal?: AbortSignal;
370
+ }) => Promise<void>;
488
371
  };
489
372
  /**
490
373
  * One pooled backhaul: the session, its (stateful) Responses client, and the
@@ -584,6 +467,24 @@ declare class TenantRegistry {
584
467
  /** Insertion-ordered = LRU order, because a hit re-inserts at the end. */
585
468
  private readonly tenants;
586
469
  constructor(opts: TenantRegistryOptions);
470
+ /**
471
+ * THE NON-SECRET TENANCY LABEL, attached by the REGISTRY rather than by
472
+ * whichever builder produced the backhaul.
473
+ *
474
+ * The caller's org id is what the per-request record and the Prometheus
475
+ * series are broken down by (`proxy/responses-turn.ts` `ProxyClients.tenant`)
476
+ * — never `caller.apiKey`, which is the credential on this surface and would
477
+ * become a credential on every dashboard. Attaching it here means a backhaul
478
+ * can never come back unlabelled because a builder forgot.
479
+ *
480
+ * `Object.create`, not a mutation and not a spread: the source object keeps
481
+ * its identity and its prototype (so a class-based `ProxyClients` keeps its
482
+ * methods), and no other caller's backhaul is touched. Done ONCE per tenant,
483
+ * at build time — the returned object is then cached and reused for every
484
+ * later request from that key, which matters because the `/v1/responses`
485
+ * thread store is keyed on this very object identity.
486
+ */
487
+ private label;
587
488
  private build;
588
489
  /** The backhaul for THIS caller, opened on first use and reused after. */
589
490
  clientsFor(caller: CallerIdentity): ProxyClients;
@@ -592,6 +493,358 @@ declare class TenantRegistry {
592
493
  /** Shutdown: end every pooled session through `Session.end()`. */
593
494
  closeAll(): Promise<void>;
594
495
  }
496
+ /** The most distinct LIVE conversations one replica keeps private backhauls for. */
497
+ declare const MAX_CONVERSATIONS = 512;
498
+ /** The most resume handles the conversation index remembers (bounded memory). */
499
+ declare const MAX_CONVERSATION_HANDLES = 1024;
500
+ /**
501
+ * Every conversation slot holds a conversation (active or resumable) — a NEW
502
+ * bootstrap/setup is refused. Explicit and honest: never an LRU-detach of
503
+ * someone else's conversation to admit a new one.
504
+ */
505
+ declare class ConversationCapacityError extends Error {
506
+ constructor();
507
+ }
508
+ /**
509
+ * One conversation's private backhaul. Its pooled sessions are reachable by
510
+ * NO other conversation of any caller: the token subject is conversation-
511
+ * unique, so the platform's own session dedupe never coalesces them.
512
+ */
513
+ interface PrivateConversationBackhaul {
514
+ clients: ProxyClients;
515
+ /**
516
+ * LOCAL release of every pooled session — `release: 'detach'` underneath,
517
+ * i.e. core's NATIVE `Session.detach()`: the media transport drops, the
518
+ * refcount decrements, the NAMED session stays live and attachable. Never
519
+ * `Session.end()`: no caller has explicitly ended anything.
520
+ */
521
+ release(): Promise<void>;
522
+ }
523
+ interface ConversationBackhaulOptions {
524
+ /** The session-gateway base sessions are opened against. */
525
+ baseUrl: string;
526
+ /** The org control-plane API base (`GET {apiUrl}/apps`). */
527
+ apiUrl: string;
528
+ /** The model-catalog oracle for routing rule 5, or null. */
529
+ catalog: CatalogConfig | null;
530
+ /** Injected in tests; production builds real uRun backhauls. */
531
+ build?: (caller: CallerIdentity, conversation: {
532
+ id: string;
533
+ subject: string;
534
+ }) => PrivateConversationBackhaul;
535
+ }
536
+ /**
537
+ * The per-CONNECTION Live/realtime seam — transport-neutral (Gemini Live,
538
+ * OpenAI Realtime, ...). `clients` lazily opens the caller's PRIVATE
539
+ * conversation backhaul on first use; `bindResumeHandle` indexes each handle
540
+ * the protocol lane mints; `clientsForResume` re-attaches the conversation a
541
+ * handle belongs to — SAME native identity, owner binding enforced, loud
542
+ * SessionGoneError when the conversation is gone; `release` drops THIS
543
+ * attachment when the socket closes.
544
+ */
545
+ interface PrivateConversationConnection {
546
+ clients: ProxyClients;
547
+ bindResumeHandle(handle: string): void;
548
+ clientsForResume(handle: string): Promise<ProxyClients>;
549
+ /**
550
+ * Release THIS attachment: called by the surface when its socket closes.
551
+ * When the LAST attachment of a conversation goes, its native transport is
552
+ * detached immediately — while the conversation's native session
553
+ * REFERENCES (and its resume handles) are RETAINED, so a later resume
554
+ * re-attaches the SAME named session through canonical attach-by-name.
555
+ * Idempotent.
556
+ */
557
+ release(): Promise<void>;
558
+ }
559
+ /**
560
+ * Process-local registry of per-conversation private backhauls, bounded by
561
+ * {@link MAX_CONVERSATIONS}. A conversation whose last attachment closes
562
+ * KEEPS its entry: its native session references + resume handles survive so
563
+ * a later resume re-attaches the SAME named session (canonical
564
+ * attach-by-name). At capacity, only such DETACHED entries are evicted
565
+ * (oldest first — their resume capability is the only thing lost, the named
566
+ * sessions stay attachable server-side); ACTIVE conversations are never
567
+ * touched, and when every slot is active a new bootstrap/setup is refused
568
+ * with an explicit capacity error. NO HA CLAIM: this registry is
569
+ * per-process; a replica restart loses every conversation (loud gone), and
570
+ * no other replica can serve one.
571
+ */
572
+ declare class ConversationBackhauls {
573
+ private readonly opts;
574
+ /** apiKey → conversationId → entry (insertion-ordered = LRU). */
575
+ private readonly byCaller;
576
+ /** resume handle → owning conversation (the owner binding's index). */
577
+ private readonly byHandle;
578
+ constructor(opts: ConversationBackhaulOptions);
579
+ private build;
580
+ /** Find (LRU-refresh) or create the conversation entry, then ATTACH to it. */
581
+ private attachEntry;
582
+ private get conversationCount();
583
+ /**
584
+ * Free a slot for a NEW conversation: evict the OLDEST DETACHED
585
+ * conversation (one whose last attachment already closed — only its resume
586
+ * metadata is lost, no live transport is touched, no session is ended).
587
+ * ACTIVE conversations are never evicted. When every slot holds an active
588
+ * conversation, refuse LOUD with an explicit capacity error — the new
589
+ * bootstrap/setup fails, existing callers are untouched.
590
+ */
591
+ private evictOverflow;
592
+ private dropHandleIndexFor;
593
+ /**
594
+ * The per-upgrade seam for one authenticated caller — ONE attachment. The
595
+ * conversation backhaul opens on the FIRST seam call of a fresh connection;
596
+ * a resumed connection replaces `clients` wholesale via `clientsForResume`
597
+ * before any seam call, so the lazy path is never taken on resume. The
598
+ * attachment is released by `release()` (the socket's close path): when the
599
+ * LAST attachment of a conversation goes, its native transport is detached
600
+ * immediately and its resume metadata is retained — no resource keepalive,
601
+ * no resume loss.
602
+ */
603
+ connectionFor(caller: CallerIdentity): PrivateConversationConnection;
604
+ /**
605
+ * Drain-time LOCAL release: detach every conversation's sessions (native
606
+ * detach — the named sessions pause server-side, nothing is ended). Handles
607
+ * and conversations are forgotten; process exit is the conversation's end.
608
+ */
609
+ detachAll(): Promise<void>;
610
+ }
611
+
612
+ /**
613
+ * THE HOSTED USAGE READ — the control-plane half of the canonical usage lane.
614
+ *
615
+ * The proxy holds no database credential and no standing platform secret
616
+ * (hosted/auth.ts). So the ledger read rides the CALLER'S OWN org API key to
617
+ * the control plane's `inference-usage` edge function — the same mechanism and
618
+ * the same credential `TenantRegistry` uses for `GET {apiUrl}/apps` — and the
619
+ * control plane derives the org from that key server-side and runs the
620
+ * org-scoped `urun_inference_usage_lookup` RPC.
621
+ *
622
+ * THAT IS WHY THIS LANE DOES NOT PRE-VERIFY THE KEY WITH `OrgResolver`. The
623
+ * surface that owns the data authenticates the key itself, so a second
624
+ * verification here would be a second source of truth about the same
625
+ * credential — and the org binding it produced would not be the one the read
626
+ * was actually scoped by. `bearerFrom` still runs at the call site, so a
627
+ * missing or malformed Authorization header is a loud 401 without a round trip.
628
+ *
629
+ * EVERY FAILURE IS LOUD. There is no shape of failure that answers with an
630
+ * empty `requests` list, because "we hold nothing for these ids" is precisely
631
+ * the answer that makes a billing caller stop asking — Hugging Face writes a
632
+ * request off ~30 minutes after it was served — so an outage that could wear
633
+ * that costume would be an invisible revenue hole.
634
+ */
635
+
636
+ interface UsageClientOptions {
637
+ /** The org control-plane API base (`{apiUrl}`) the function is mounted under. */
638
+ apiUrl: string;
639
+ /** Injected in tests; production uses the platform fetch. */
640
+ fetchImpl?: typeof fetch;
641
+ /**
642
+ * Where a row this lane REFUSED TO ANSWER ABOUT is reported (ENG-416).
643
+ *
644
+ * A malformed ledger row is isolated rather than allowed to 502 the whole
645
+ * 10,000-id batch, and the row it dropped is then MISSING from the answer —
646
+ * which a billing caller cannot tell apart from an id we do not hold, and
647
+ * reads as "not priced yet" until the request is written off unbilled. So
648
+ * the drop must be an event somebody can alert on. Absent, it goes to this
649
+ * process's stdout (`parseUsageRecords`'s own default), never nowhere.
650
+ *
651
+ * It may be async — a sink that ships the line somewhere usually is — and a
652
+ * rejected promise from it is contained exactly as a synchronous throw is.
653
+ */
654
+ onRejectedRow?: UsageRowReporter;
655
+ }
656
+ /** The batch ledger read, as one call per request. */
657
+ declare class UsageClient {
658
+ private readonly opts;
659
+ private readonly fetchImpl;
660
+ constructor(opts: UsageClientOptions);
661
+ /**
662
+ * Look up what we hold for these ids, as the holder of `apiKey`.
663
+ *
664
+ * Three outcomes, all explicit:
665
+ * - the control plane answers → validated {@link InferenceUsageRecord}s;
666
+ * - it rejects the key (401/403) → {@link ProxyAuthError} (401);
667
+ * - anything else → {@link UsageSurfaceError} (502).
668
+ */
669
+ lookup(apiKey: string, inferenceIds: readonly string[]): Promise<InferenceUsageRecord[]>;
670
+ }
671
+
672
+ /**
673
+ * THE HOSTED LEDGER WRITE — the control-plane half of the per-request ledger.
674
+ *
675
+ * The proxy holds no database credential and no standing platform secret
676
+ * (hosted/auth.ts: "NO STANDING CREDENTIAL... every upstream call it makes
677
+ * rides the CALLER'S OWN key"). So the ledger write rides the caller's own org
678
+ * API key to the control plane's `inference-ledger` function — the same
679
+ * mechanism and the same credential the ENG-318 read already uses — and the
680
+ * control plane derives the org AND the api_key_id from that key server-side
681
+ * before calling `urun_record_inference_request`.
682
+ *
683
+ * THAT PROPERTY IS SOUND ON THE READ AND INVERTED ON THE WRITE, and it is
684
+ * stated here because it is invisible in the code. On a read, riding the
685
+ * caller's key is precisely what enforces org scoping. On a WRITE TO A BILLING
686
+ * LEDGER it means the party being billed is the party authenticated to write
687
+ * the bill: a caller can read its `Inference-Id` off a streaming response head
688
+ * and race a fabricated row in ahead of this one. The control plane refuses
689
+ * the second write rather than upserting, so the attempt is LOUD instead of
690
+ * silent — see {@link LedgerDuplicateError}.
691
+ *
692
+ * ============================ THE ENG-410 RULING ============================
693
+ *
694
+ * That is a DETECTION, and it is the DELIBERATE, OWNER-APPROVED choice — not a
695
+ * gap somebody failed to close. The decision and the condition that reopens it
696
+ * are recorded here because the next person to read this file will otherwise
697
+ * re-derive the wrong answer from first principles.
698
+ *
699
+ * WHAT THE ATTACK ACTUALLY IS, bounded honestly. `org_id` and `api_key_id` are
700
+ * derived from the credential and the control-plane surface REFUSES both as
701
+ * body fields, so cross-tenant billing INJECTION is structurally impossible: a
702
+ * forged row can only ever land in the forger's OWN org. The only rational
703
+ * attack is therefore SELF-under-billing, it is not really a race (the header
704
+ * flushes at the head of a stream and this write happens on close, so the
705
+ * attacker has the whole generation), and it cannot be made quiet — every
706
+ * stolen request costs the attacker one refused write and one error line.
707
+ *
708
+ * REJECTED — GIVE THE PROXY A STANDING PLATFORM CREDENTIAL FOR THE WRITE. This
709
+ * is the obvious-sounding move and it is STRICTLY WORSE THAN THE FLAW. For a
710
+ * platform credential to REPLACE the caller's key on this write, `org_id` has
711
+ * to come back into the request body — there is nothing else left to carry
712
+ * tenancy. That deletes the one structural property that makes this surface
713
+ * safe at all, and it converts a proxy compromise from "whatever caller keys
714
+ * happen to be in flight" into "arbitrary billing rows for EVERY org, written
715
+ * from an internet-facing pod". The generalised rule, which outlives this
716
+ * module: A CREDENTIAL MAY BE ADDITIVE TO THE CALLER'S KEY, NEVER A
717
+ * REPLACEMENT FOR IT, because the caller's key is what carries tenancy.
718
+ *
719
+ * REJECTED — HAVE THE CONTROL PLANE DERIVE MORE AND TRUST THE BODY LESS. It
720
+ * already derives everything it can (`org_id`, `api_key_id`) and refuses what
721
+ * it must (`gpu_seconds`). Of what is left, the fields that DECIDE MONEY are
722
+ * exactly the fields it has no independent source for: nothing in the platform
723
+ * reports per-request token counts anywhere except this wire, and the
724
+ * `inference_id` is minted here. Authenticating the id would not help either —
725
+ * the attacker holds a LEGITIMATE id, read from its own response head.
726
+ *
727
+ * DEFERRED, WITH A TRIGGER — MAKE THE ROW UNFORGEABLE (a proxy-held key that
728
+ * signs the row's CONTENT, verified by the control plane, ADDITIVE to the
729
+ * caller's key so tenancy is untouched). This is the real fix and its blast
730
+ * radius is small: a stolen signing key only restores today's position,
731
+ * because the row still lands in whatever org the thief's own key names. It is
732
+ * NOT built yet because its own failure mode is worse than the flaw's while
733
+ * nothing bills off this table: a misconfigured or badly-rotated key takes
734
+ * 100% OF LEDGER ROWS TO ZERO ACROSS EVERY ORG, where the flaw takes some rows
735
+ * to zero for one customer attacking itself, loudly. Fitting a cryptographic
736
+ * gate to a writer that has never once run in production, in the week it first
737
+ * writes a row, is how a remedy costs more than the bug.
738
+ *
739
+ * THE TRIGGER, EXPLICITLY: BUILD IT BEFORE ANY PRODUCT BILLS OFF
740
+ * `public.inference_requests`. Today HF traffic authenticates as HF's own
741
+ * org (the end user never holds a uRun key) and self-serve is billed off
742
+ * `usage_events`, so no party both can run the attack and benefits from it.
743
+ * The credits / concurrency-tier / retention-tier product is the moment that
744
+ * stops being true, and it must not be the moment this is discovered.
745
+ *
746
+ * UNTIL THEN THE DETECTION IS THE CONTROL, so it is built to be OPERATED
747
+ * rather than merely to exist: the refusal carries a machine-readable
748
+ * `ledgerFailureReason` (`proxy/ledger.ts` {@link LedgerFailure}) that reaches
749
+ * both the structured log line and a Prometheus counter, so the alert is a
750
+ * field match and not a regex over this paragraph's prose.
751
+ *
752
+ * ===========================================================================
753
+ *
754
+ * EVERY FAILURE IS LOUD AND NONE OF THEM REACHES THE CUSTOMER. This runs from
755
+ * the response's `'close'` hook, after the answer is delivered, so there is no
756
+ * request left to fail: `proxy/ledger.ts` catches everything here and writes
757
+ * one structured log line. A write that did not land is revenue that was never
758
+ * recorded, which is exactly the thing that must never be silent.
759
+ */
760
+
761
+ /**
762
+ * The control-plane route this writes. Relative to the org control-plane API
763
+ * base (`{apiUrl}`, i.e. `https://api.urun.sh/v1`), which the deployment's
764
+ * gateway rewrites onto the Supabase Edge Functions host — the same base and
765
+ * the same rewrite `GET {apiUrl}/apps` and the usage read already ride.
766
+ */
767
+ declare const LEDGER_FUNCTION_PATH = "/inference-ledger";
768
+ /**
769
+ * How long one write may take before it is abandoned. A single-row insert plus
770
+ * two indexed price lookups is milliseconds; this bound exists so a stalled
771
+ * socket becomes a log line rather than a promise retained for the life of the
772
+ * pod. Deliberately TIGHTER than the usage read's 15s: that read answers a
773
+ * live poller that is waiting, while this write has nobody waiting on it and
774
+ * every second it holds is a slot in the in-flight bound.
775
+ */
776
+ declare const LEDGER_WRITE_TIMEOUT_MS = 10000;
777
+ /**
778
+ * The row was already in the ledger. Its own type because it is not a fault
779
+ * and must not be logged as one: it means somebody wrote this `inference_id`
780
+ * before us, which is either a double-write defect in this proxy or a row
781
+ * fabricated by the org being billed. Nothing was written and nothing was
782
+ * overwritten.
783
+ */
784
+ declare class LedgerDuplicateError extends Error implements LedgerFailure {
785
+ /**
786
+ * THE ENG-410 SIGNAL, as a field an alert can match.
787
+ *
788
+ * `duplicate` now means PROVABLY NOT OURS. Since the 409 carries the stored
789
+ * row (ENG-415), a conflict whose row matches what we are re-sending is our
790
+ * OWN earlier attempt and resolves as {@link LedgerAlreadyWrittenError}
791
+ * instead — so what is left here is a row somebody else wrote, or a row we
792
+ * could not read back to check. Both fail to the LOUD side on purpose.
793
+ *
794
+ * `duplicate_foreign` is the loudest state in the lane: the id is taken by a
795
+ * row THIS ORG CANNOT READ. `inference_requests.inference_id` is a global
796
+ * primary key, so it is reachable, and it can never be our own write.
797
+ */
798
+ readonly ledgerFailureReason: 'duplicate' | 'duplicate_foreign';
799
+ constructor(message: string, reason?: 'duplicate' | 'duplicate_foreign');
800
+ }
801
+ /** The control-plane write surface could not be reached or refused the row. */
802
+ declare class LedgerSurfaceError extends Error implements LedgerFailure {
803
+ readonly ledgerFailureReason: "surface";
804
+ constructor(message: string);
805
+ }
806
+ interface LedgerClientOptions {
807
+ /** The org control-plane API base (`{apiUrl}`) the function is mounted under. */
808
+ apiUrl: string;
809
+ /** Injected in tests; production uses the platform fetch. */
810
+ fetchImpl?: typeof fetch;
811
+ }
812
+ /**
813
+ * Validate the control plane's answer into the record the caller logs.
814
+ *
815
+ * IT IS VALIDATED RATHER THAN TRUSTED because what it carries is a COST. A
816
+ * surface that changed shape under us must not be able to produce a log line
817
+ * claiming a request was priced at a number nobody returned — and
818
+ * `unpriced_reason` is what tells an operator that traffic is being served
819
+ * that nobody can bill, so a missing one is not a detail.
820
+ */
821
+ declare function parseLedgerRecorded(raw: unknown, where: string): LedgerRecorded;
822
+ /** One ledger row, written as the holder of the caller's own org API key. */
823
+ declare class LedgerClient {
824
+ private readonly opts;
825
+ private readonly fetchImpl;
826
+ constructor(opts: LedgerClientOptions);
827
+ /**
828
+ * Write one row. Five outcomes, all explicit:
829
+ * - the control plane records it → the validated {@link LedgerRecorded},
830
+ * priced or with an `unpriced_reason`;
831
+ * - the id is already there AND the stored row is OURS (409) →
832
+ * {@link LedgerAlreadyWrittenError}, which is a SUCCESS: an earlier
833
+ * attempt landed;
834
+ * - the id is already there and the row is NOT ours, or cannot be read
835
+ * back to check (409) → {@link LedgerDuplicateError};
836
+ * - it rejects the key (401/403) → {@link LedgerAuthError};
837
+ * - anything else → {@link LedgerSurfaceError}.
838
+ *
839
+ * THE CALLER RETRIES, NOT THIS METHOD (ENG-415). One call is one attempt;
840
+ * `proxy/ledger.ts` owns the queue, the backoff and the dwell bound, and it
841
+ * retries only the one reason that is safe to retry. What makes any of it
842
+ * safe is the fifth outcome above: a retry whose first attempt secretly
843
+ * landed comes back as {@link LedgerAlreadyWrittenError}, not as something
844
+ * indistinguishable from a forged row.
845
+ */
846
+ write(apiKey: string, entry: LedgerWrite): Promise<LedgerRecorded>;
847
+ }
595
848
 
596
849
  /**
597
850
  * The hosted endpoint's configuration.
@@ -670,8 +923,25 @@ declare function resolveHostedConfig(env?: NodeJS.ProcessEnv): HostedConfig;
670
923
  * reachable, because a replica that cannot verify API keys
671
924
  * or list apps can serve nothing and must be pulled out of
672
925
  * the Service rather than answering 502s.
673
- * * /v1/... the compat surface, org-scoped by the Bearer key.
674
- * * anything else → 404 in the OpenAI error envelope.
926
+ * * /v1/... the compat surface, org-scoped by the Bearer key, including
927
+ * the canonical usage-query lane POST /v1/usage/requests
928
+ * (proxy/usage.ts — a batch read of the per-request
929
+ * inference ledger).
930
+ * POST /partners/<name>/... the ENG-376 partner ADAPTER surface: a
931
+ * partner's own wire shape translated onto a canonical lane,
932
+ * sharing its auth and error mapping. Today: the Hugging
933
+ * Face billing poll. Ingress/charts must allowlist it.
934
+ * WS /ws/google.ai.generativelanguage.v1beta.GenerativeService.
935
+ * BidiGenerateContent — the Gemini Live surface, org-scoped by the
936
+ * Gemini credential (x-goog-api-key / Authorization: Bearer; `?key=`
937
+ * refused). Ingress/charts must allowlist that path.
938
+ * WS /v1/realtime — the OpenAI Realtime surface (GA protocol subset),
939
+ * org-scoped by the SAME Bearer api key as the /v1 lane; the
940
+ * caller-org catalog's `task` column (stt | tts) gates which models
941
+ * may bind an audio session. Ingress/charts must allowlist that
942
+ * path too.
943
+ * * anything else → 404 in the OpenAI error envelope (unclaimed WS
944
+ * upgrades get their own final refusal).
675
945
  *
676
946
  * WHAT IS DELIBERATELY NOT HERE: `/stats`. The local proxy exposes it for the
677
947
  * launcher's reuse probe; on a shared endpoint it would publish one tenant's
@@ -698,14 +968,232 @@ interface HostedProxyOptions extends HostedConfig {
698
968
  resolver?: OrgResolver;
699
969
  /** Injected in tests so backhauls are stubbed at the seam. */
700
970
  registry?: TenantRegistry;
971
+ /** Injected in tests so conversation backhauls are stubbed at the seam. */
972
+ conversations?: ConversationBackhauls;
973
+ /**
974
+ * Injected in tests so the canonical usage lane is exercised without a
975
+ * control plane. Production builds one against `apiUrl` — the same base and
976
+ * the same caller-key credential `GET {apiUrl}/apps` rides.
977
+ */
978
+ usageClient?: UsageClient;
979
+ /**
980
+ * Injected in tests so the per-request LEDGER WRITE is exercised without a
981
+ * control plane. Production builds one against `apiUrl` — the same base and
982
+ * the same caller-key credential the usage read and `GET {apiUrl}/apps`
983
+ * ride.
984
+ */
985
+ ledgerClient?: LedgerClient;
986
+ /**
987
+ * Sink for the ledger lane's own structured diagnostics (proxy/ledger.ts:
988
+ * one JSON object per write, skip or failure). DELIBERATELY SEPARATE from
989
+ * {@link requestLog}, which carries exactly one `inference_request` record
990
+ * per `/v1` request and whose readers count on that. Production leaves it
991
+ * unset, which is this container's stdout.
992
+ */
993
+ ledgerLog?: (line: string) => void;
994
+ /**
995
+ * Sink for the shared handler's per-request billing log line (the
996
+ * `Inference-Id` record — proxy/responses-turn.ts `ProxyHandlerOptions`).
997
+ * Injected in tests; production leaves it unset, which is this container's
998
+ * stdout — the log stream the platform collects.
999
+ */
1000
+ requestLog?: (line: string) => void;
1001
+ /**
1002
+ * Collect Node process metrics (event-loop lag, heap, GC) alongside the
1003
+ * request metrics. Production leaves it on; tests turn it off so an
1004
+ * assertion on the exposition text is not swamped by process noise.
1005
+ */
1006
+ defaultMetrics?: boolean;
701
1007
  }
702
1008
  /**
703
- * Build (not listen) the hosted endpoint. The caller owns listen/close, and
704
- * `closeAll` releases every pooled backhaul through `Session.end()`.
1009
+ * Build (not listen) the hosted endpoint. The caller owns listen/close,
1010
+ * `closeAll` releases every pooled backhaul through `Session.end()`, and
1011
+ * `closeWebSockets` gracefully closes every attached WS surface — a drain
1012
+ * must run it BEFORE `server.close()`/`closeAllConnections()` (see
1013
+ * {@link drainServer}).
705
1014
  */
706
1015
  declare function createHostedProxy(options: HostedProxyOptions): {
707
1016
  server: Server;
1017
+ /**
1018
+ * The PRIVATE Prometheus listener (`proxy/metrics.ts` — GET /metrics on
1019
+ * {@link METRICS_PORT}). Its own server, not a route on the one above, so
1020
+ * "the metrics port is not public" is a property of the socket rather than
1021
+ * a rule the `/v1` router must keep remembering. The caller owns
1022
+ * listen/close, exactly as it does for `server`.
1023
+ */
1024
+ metricsServer: Server;
1025
+ metrics: InferenceMetrics;
1026
+ /**
1027
+ * The PRIVATE ext-auth listener the edge calls to turn the caller's Bearer
1028
+ * key into a non-secret bucket id (`hosted/edge-identity.ts`). Its own
1029
+ * socket for the same reason the metrics port is: the public HTTPRoutes
1030
+ * forward only to the `/v1` port, so it cannot be reached from outside.
1031
+ * The caller owns listen/close.
1032
+ */
1033
+ edgeIdentityServer: Server;
708
1034
  closeAll: () => Promise<void>;
1035
+ closeWebSockets: () => Promise<void>;
709
1036
  };
710
1037
 
711
- export { BIND_HOST, type CallerIdentity, ControlPlaneUnavailableError, DEFAULT_PORT, type HostedConfig, type HostedProxyOptions, MAX_CACHED_KEYS, MAX_TENANTS, ORG_BINDING_TTL_MS, OrgResolver, type OrgResolverOptions, ProxyAuthError, READINESS_TIMEOUT_MS, READINESS_TTL_MS, SERVE_FUNCTION, TenantRegistry, type TenantRegistryOptions, bearerFrom, createHostedProxy, resolveHostedConfig, tenantSubject };
1038
+ /**
1039
+ * THE EDGE IDENTITY SURFACE — the one thing Envoy cannot compute for itself.
1040
+ *
1041
+ * WHY IT EXISTS. Envoy Gateway's per-caller rate limiting needs a DISTINCT
1042
+ * bucket per caller, and the only caller identity on this endpoint is the
1043
+ * Bearer org API key (`hosted/auth.ts`: "the Bearer key IS the identity AND
1044
+ * the tenancy"). Envoy Gateway can key a global rate limit on a header's
1045
+ * distinct values — but the descriptor VALUE is what the rate-limit service
1046
+ * sends to Redis and what Redis stores as part of its key. Keying directly on
1047
+ * `Authorization` would therefore park live customer API keys, in plaintext,
1048
+ * in an in-cluster Redis whose NetworkPolicy is inert on both prod clusters
1049
+ * today (the VPC CNI node agent runs without `--enable-network-policy`, as the
1050
+ * valkey chart already documents). That is a credential store nobody designed,
1051
+ * audited, or rotates.
1052
+ *
1053
+ * Envoy's Lua filter has no hashing primitive and its rate-limit actions have
1054
+ * no transform, so SOMETHING has to turn the key into a non-secret id before
1055
+ * the rate-limit filter runs. Envoy Gateway's own mechanism for that is
1056
+ * `SecurityPolicy.extAuth`, whose response headers are merged into the request
1057
+ * ("coexisting headers will be overridden") before the later rate-limit filter
1058
+ * reads them. This module is that service.
1059
+ *
1060
+ * IT IS NOT AN AUTHORIZATION GATE, AND IT MUST NEVER BECOME ONE. It answers
1061
+ * 200 to everything. Authentication stays where it already is — in the proxy,
1062
+ * which alone can answer 401 in the lane's native error envelope with the
1063
+ * `Inference-Id` header HF bills on. A deny here would instead produce Envoy's
1064
+ * bodiless refusal, which an OpenAI SDK surfaces as an unparseable error.
1065
+ *
1066
+ * IT HAS NO DEPENDENCIES, ON PURPOSE. Read a header, hash it, answer. No
1067
+ * control-plane call, no cache, no I/O, nothing that can be slow or down. It
1068
+ * sits in the request path of `inference.urun.sh`, so the only acceptable
1069
+ * failure budget is none — and the `SecurityPolicy` that calls it is
1070
+ * additionally configured `failOpen`, so even losing it entirely degrades to
1071
+ * "no rate limiting", never to "no inference".
1072
+ */
1073
+
1074
+ /**
1075
+ * The container's edge-identity port — a MODULE CONSTANT, not an env var, per
1076
+ * the hosted config mandate (`hosted/config.ts`). Its own socket, like the
1077
+ * metrics port: private by construction, since the public HTTPRoutes forward
1078
+ * only to the `/v1` port.
1079
+ */
1080
+ declare const EDGE_IDENTITY_PORT = 9465;
1081
+ /**
1082
+ * THE BUCKET ID. Envoy Gateway's `BackendTrafficPolicy` keys the caller's
1083
+ * rate-limit counter on this header's DISTINCT values, which means the value
1084
+ * here IS the input the rate-limit service builds its Redis counter key from.
1085
+ *
1086
+ * THAT IS WHY IT MUST NOT BE THE VALUE COMMITTED IN `values.yaml`. The tier
1087
+ * fingerprint below is written into a git-tracked manifest so an operator can
1088
+ * give one caller its own limit. If the bucket id were the same digest, then
1089
+ * ANYONE WHO CAN READ THE REPO could reconstruct a tiered caller's counter key
1090
+ * — and the counter store is reachable by any pod in the cluster with
1091
+ * `+incrby` and `+expire` granted, so they could spend Hugging Face's
1092
+ * allowance or stretch the window and hold HF's own probe at 429. That is the
1093
+ * delisting event this whole pair exists to prevent, and non-invertibility of
1094
+ * SHA-256 does nothing about it: the attacker never needs the key, only the
1095
+ * digest, and the digest was published on purpose.
1096
+ *
1097
+ * So the two digests are DIFFERENT one-way functions of the same token, split
1098
+ * by domain separator ({@link BUCKET_DOMAIN} vs {@link TIER_DOMAIN}). Knowing
1099
+ * the tier fingerprint gives no path to the bucket id: getting there would
1100
+ * require recovering the token from its digest, which is a preimage attack on
1101
+ * SHA-256 over 256 bits of `randomBytes` entropy.
1102
+ */
1103
+ declare const EDGE_KEY_ID_HEADER = "x-urun-key-id";
1104
+ /**
1105
+ * THE TIER SELECTOR — matched with `Exact` / `RegularExpression` in the
1106
+ * BackendTrafficPolicy to decide WHICH limit applies, never to bucket.
1107
+ *
1108
+ * This is the digest an operator commits to `rateLimit.tiers[].keyIdSha256` in
1109
+ * k8s-manifests, so treat it as PUBLIC: everything about the design has to
1110
+ * hold when an attacker knows it.
1111
+ *
1112
+ * WHY A SECOND HEADER AT ALL. An Envoy Gateway rate-limit rule needs both
1113
+ * kinds of match on the caller at once: `Distinct` (give this caller their own
1114
+ * counter) AND an exact/inverted match (is this the partner key, or everyone
1115
+ * else). Whether a single selector may list the SAME header name under two
1116
+ * different match types is not something this repo can verify without a
1117
+ * cluster, and getting it wrong means Envoy Gateway rejects the policy —
1118
+ * which, on a fail-open limiter, is indistinguishable from "rate limiting
1119
+ * works" until somebody floods us. Two headers remove the question; carrying
1120
+ * two DIFFERENT digests is what makes publishing one of them safe.
1121
+ */
1122
+ declare const EDGE_KEY_FINGERPRINT_HEADER = "x-urun-key-fingerprint";
1123
+ /**
1124
+ * Domain separator for the BUCKET digest. Never appears in a manifest, a log,
1125
+ * or a metric — the bucket id is derived here and read only by Envoy.
1126
+ */
1127
+ declare const BUCKET_DOMAIN = "urun-ratelimit-bucket:";
1128
+ /**
1129
+ * Domain separator for the TIER digest — the one an operator reproduces from a
1130
+ * key with `printf 'urun-ratelimit-tier:%s' "$KEY" | sha256sum | cut -d' ' -f1`.
1131
+ *
1132
+ * A plain `sha256sum "$KEY"` would be the OLD, unsafe value: identical to the
1133
+ * bucket id, and therefore a published counter key. The separator is what
1134
+ * keeps the two apart, so it is part of the operator contract and cannot drift.
1135
+ */
1136
+ declare const TIER_DOMAIN = "urun-ratelimit-tier:";
1137
+ /**
1138
+ * The bucket every request WITHOUT a Bearer token shares. A single shared
1139
+ * bucket rather than no bucket at all: unauthenticated floods must be limited
1140
+ * too, and they have no identity to spread across. They cannot consume a real
1141
+ * caller's bucket because no real key hashes to this value.
1142
+ */
1143
+ declare const ANONYMOUS_KEY_ID = "anonymous";
1144
+ /**
1145
+ * The caller's BUCKET id — `sha256(BUCKET_DOMAIN || token)`, hex.
1146
+ *
1147
+ * SECRET BY CONSTRUCTION, not because the digest is one-way but because the
1148
+ * only published digest is a DIFFERENT one. This value is the input Envoy's
1149
+ * rate-limit descriptor carries, so it is what the counter key in the
1150
+ * rate-limit cache is built from; publishing it would publish a writable
1151
+ * counter key (see {@link EDGE_KEY_ID_HEADER}).
1152
+ */
1153
+ declare function edgeBucketId(bearerToken: string): string;
1154
+ /**
1155
+ * The caller's TIER fingerprint — `sha256(TIER_DOMAIN || token)`, hex.
1156
+ *
1157
+ * PUBLIC BY DESIGN: this is the value an operator commits to
1158
+ * `rateLimit.tiers[].keyIdSha256`. The runbook is
1159
+ * `printf 'urun-ratelimit-tier:%s' "$KEY" | sha256sum | cut -d' ' -f1`.
1160
+ *
1161
+ * The domain prefix is not decoration — a bare `sha256sum "$KEY"` reproduces
1162
+ * the BUCKET id, which is exactly the value that must never be committed. The
1163
+ * `cut` is load-bearing too: `sha256sum` prints the digest, two spaces, then
1164
+ * the input name, and a tier carrying that trailing ` -` matches nothing and
1165
+ * silently leaves the caller on the default limit.
1166
+ */
1167
+ declare function edgeTierFingerprint(bearerToken: string): string;
1168
+ /**
1169
+ * The bearer token of an `Authorization` header, or null.
1170
+ *
1171
+ * Deliberately NOT `hosted/auth.ts` `bearerFrom`: that one THROWS a 401-shaped
1172
+ * error for a missing or malformed header, which is the correct behaviour for
1173
+ * the lane that authenticates. Here a malformed header must produce the
1174
+ * anonymous bucket and a 200, because refusing is the proxy's job and this
1175
+ * service must never refuse.
1176
+ */
1177
+ declare function bearerTokenOf(headerValue: string | undefined): string | null;
1178
+ /**
1179
+ * The PRIVATE ext-auth listener. Answers 200 to every method and every path —
1180
+ * Envoy appends the ORIGINAL request path to the configured ext-auth path, so
1181
+ * this service sees `/ratelimit-identity/v1/chat/completions` and friends and
1182
+ * must not care.
1183
+ *
1184
+ * The response carries exactly the two identity headers below, and the
1185
+ * `SecurityPolicy` forwards exactly those two (`headersToBackend`). Nothing
1186
+ * about the caller's request, and certainly not their key, is echoed.
1187
+ *
1188
+ * A CALLER CANNOT PRESENT THEIR OWN. `headersToBackend` is documented as
1189
+ * overriding coexisting headers, so whatever a client sent under these names is
1190
+ * replaced on every request this service answers. There is deliberately NO
1191
+ * `ClientTrafficPolicy` stripping them at the listener: that policy would
1192
+ * attach to the shared `public-gw` (api.urun.sh rides it too) to close a gap
1193
+ * that grants nothing — in the only case where a client value survives, the
1194
+ * ext-auth hop has already failed open, and fail-open grants MORE than any
1195
+ * spoofed bucket could. The k8s half records the same decision.
1196
+ */
1197
+ declare function createEdgeIdentityServer(): Server;
1198
+
1199
+ export { ANONYMOUS_KEY_ID, BIND_HOST, BUCKET_DOMAIN, BoundedLabel, type CallerIdentity, ControlPlaneUnavailableError, type ConversationBackhaulOptions, ConversationBackhauls, ConversationCapacityError, DEFAULT_PORT, EDGE_IDENTITY_PORT, EDGE_KEY_FINGERPRINT_HEADER, EDGE_KEY_ID_HEADER, type HostedConfig, type HostedProxyOptions, InferenceMetrics, LEDGER_FUNCTION_PATH, LEDGER_WRITE_TIMEOUT_MS, LedgerClient, type LedgerClientOptions, LedgerDuplicateError, LedgerSurfaceError, MAX_CACHED_KEYS, MAX_CONVERSATIONS, MAX_CONVERSATION_HANDLES, MAX_MODEL_LABELS, MAX_ORG_LABELS, MAX_TENANTS, METRICS_PATH, METRICS_PORT, NO_MODEL_LABEL, ORG_BINDING_TTL_MS, OTHER_LABEL, OrgResolver, type OrgResolverOptions, type PrivateConversationBackhaul, type PrivateConversationConnection, ProxyAuthError, READINESS_TIMEOUT_MS, READINESS_TTL_MS, SERVE_FUNCTION, TIER_DOMAIN, TenantRegistry, type TenantRegistryOptions, bearerFrom, bearerTokenOf, createEdgeIdentityServer, createHostedProxy, createMetricsServer, edgeBucketId, edgeTierFingerprint, laneOf, parseLedgerRecorded, resolveHostedConfig, statusClassOf, tenantSubject };