@urun-sh/openai 0.5.4 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{ResponsesClient-BYx3YLGo.d.ts → ResponsesClient-BE3-hx3m.d.ts} +2 -0
- package/dist/{ResponsesClient-Dft3bg3b.d.cts → ResponsesClient-DjZWjlFH.d.cts} +2 -0
- package/dist/chunk-2T2YYBVX.js +1 -0
- package/dist/chunk-3WAJD62J.js +4 -0
- package/dist/{chunk-K23AZPI4.js → chunk-5HKWNK3O.js} +1 -1
- package/dist/chunk-7M6LY6DR.js +2 -0
- package/dist/chunk-BSHT6RZZ.js +1 -0
- package/dist/{chunk-GBBY3PCZ.js → chunk-FP4RSAIE.js} +1 -1
- package/dist/chunk-HR5H6S7L.js +1 -0
- package/dist/chunk-NY23USZF.js +57 -0
- package/dist/chunk-VLRMJRLS.js +6 -0
- package/dist/gemini-live.cjs +2 -2
- package/dist/gemini-live.d.cts +17 -2
- package/dist/gemini-live.d.ts +7 -2
- package/dist/gemini-live.js +1 -1
- package/dist/hosted/bin.cjs +65 -37
- package/dist/hosted/bin.js +3 -3
- package/dist/hosted/index.cjs +46 -19
- package/dist/hosted/index.d.cts +760 -272
- package/dist/hosted/index.d.ts +203 -107
- package/dist/hosted/index.js +1 -1
- package/dist/index.cjs +1 -1
- package/dist/index.d.cts +4 -4
- package/dist/index.d.ts +4 -4
- package/dist/index.js +1 -1
- package/dist/{models-NYMZrklp.d.cts → models-DUdx_Y6X.d.cts} +1 -1
- package/dist/pi-extension/index.cjs +7 -7
- package/dist/pi-extension/index.d.cts +1 -1
- package/dist/pi-extension/index.d.ts +1 -1
- package/dist/pi-extension/index.js +1 -1
- package/dist/pi-extension/standalone.cjs +48 -48
- package/dist/proxy/cli.cjs +54 -40
- package/dist/proxy/cli.js +14 -14
- package/dist/proxy/index.cjs +33 -22
- package/dist/proxy/index.d.cts +546 -4
- package/dist/proxy/index.d.ts +185 -4
- package/dist/proxy/index.js +1 -1
- package/dist/responses-turn-Dr37N3kH.d.ts +486 -0
- package/dist/responses-turn-tNUHq5b0.d.cts +1807 -0
- package/dist/{translator-C9uPKypK.d.ts → translator-BCoRFaTs.d.ts} +1 -1
- package/dist/{translator-CcDBEfvm.d.cts → translator-Bh0Bp_Ie.d.cts} +3 -3
- package/dist/{video-out-D20UuJ8G.d.cts → video-out-CCksIcj8.d.cts} +5 -5
- package/package.json +12 -11
- package/dist/chunk-4MSBS7M6.js +0 -1
- package/dist/chunk-5NXM4IO3.js +0 -2
- package/dist/chunk-CWJRDBDC.js +0 -1
- package/dist/chunk-OI2OY32M.js +0 -1
- package/dist/chunk-RQBD4OAA.js +0 -6
- package/dist/chunk-STFRT5OY.js +0 -1
- package/dist/chunk-VNM4M4P3.js +0 -40
- package/dist/server-DuH_5OD9.d.ts +0 -84
- package/dist/server-wmUnMewW.d.cts +0 -206
- /package/dist/{chunk-YSFSRI3D.js → chunk-5CCJUNH7.js} +0 -0
- /package/dist/{models-NYMZrklp.d.ts → models-DUdx_Y6X.d.ts} +0 -0
- /package/dist/{video-out-CWesbk12.d.ts → video-out-IGQ4YQ-c.d.ts} +0 -0
package/dist/hosted/index.d.cts
CHANGED
|
@@ -1,283 +1,147 @@
|
|
|
1
1
|
import { Server } from 'node:http';
|
|
2
|
+
import * as _prometheus_io_client from '@prometheus-io/client';
|
|
3
|
+
import { Registry } from '@prometheus-io/client';
|
|
4
|
+
import { c as LedgerOutcome, I as InferenceRecord, S as SessionGoneError, P as ProxyClients, M as ModelRouter, i as UsageRowReporter, b as InferenceUsageRecord, e as LedgerWrite, d as LedgerRecorded, L as LedgerFailure } from '../responses-turn-tNUHq5b0.cjs';
|
|
2
5
|
import { createClientToken } from '@urun-sh/core';
|
|
3
|
-
import { U as UrunResponses } from '../ResponsesClient-
|
|
6
|
+
import { U as UrunResponses } from '../ResponsesClient-DjZWjlFH.cjs';
|
|
4
7
|
import { b as UrunSessionLike } from '../types-lsVTbNcH.cjs';
|
|
5
|
-
import
|
|
6
|
-
import
|
|
7
|
-
import '../video-out-D20UuJ8G.cjs';
|
|
8
|
+
import '../models-DUdx_Y6X.cjs';
|
|
9
|
+
import '../video-out-CCksIcj8.cjs';
|
|
8
10
|
|
|
9
11
|
/**
|
|
10
|
-
*
|
|
12
|
+
* The container's metrics port — a MODULE CONSTANT, not an env var, per the
|
|
13
|
+
* hosted config mandate (`hosted/config.ts`: "env vars are NOT a config
|
|
14
|
+
* mechanism"). 9464 is the Prometheus exporter port the OpenTelemetry
|
|
15
|
+
* ecosystem registered, and it is the interface contract the deployment chart
|
|
16
|
+
* and its ServiceMonitor are built against.
|
|
11
17
|
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
* MODALITY GATING: only rows whose catalog `task` names a chat-completions
|
|
19
|
-
* shape (chat | code | agent | vl) are declared. A voice/video/image app
|
|
20
|
-
* behind a chat surface would stream garbage into OpenRouter's baseline
|
|
21
|
-
* tests; models the catalog cannot vouch for are OMITTED, never guessed.
|
|
22
|
-
* With no catalog oracle configured the document is `{"data": []}` — honest
|
|
23
|
-
* emptiness, not invented capability (the chart turns the oracle on).
|
|
24
|
-
*
|
|
25
|
-
* PRICING is deliberately omitted entirely: the schema's rule is "a modality
|
|
26
|
-
* with no pricing array is simply unpriced", OpenRouter configures pricing
|
|
27
|
-
* during provider onboarding, and the per-model price surface is
|
|
28
|
-
* urun-infra U3 (model_prices) — when that lands, this module grows a
|
|
29
|
-
* `prices` input and emits the input/output pricing arrays.
|
|
30
|
-
*
|
|
31
|
-
* The full closed-value-domain schema ships as OpenAPI 3.1 at
|
|
32
|
-
* openrouter.ai/docs/assets/provider-monitor-schema-v2.openapi.json.
|
|
18
|
+
* PRIVATE, and private STRUCTURALLY rather than by convention: the public
|
|
19
|
+
* HTTPRoute allowlist (`k8s-manifests` `_public-matches.tpl`) routes only
|
|
20
|
+
* `/v1` to the Service's 8080 port, so nothing on this port is reachable from
|
|
21
|
+
* the internet. It is served by its OWN `http.Server`
|
|
22
|
+
* ({@link createMetricsServer}) rather than as a route on the main server, so
|
|
23
|
+
* a future public path can never accidentally expose it.
|
|
33
24
|
*/
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
id: string;
|
|
42
|
-
name: string;
|
|
43
|
-
created: number;
|
|
44
|
-
/** Valid enum: int4|int8|fp4|mxfp4|nvfp4|fp6|fp8|mxfp8|fp16|bf16|fp32|null. */
|
|
45
|
-
quantization: string | null;
|
|
46
|
-
description: string;
|
|
47
|
-
hugging_face_id: string;
|
|
48
|
-
input_modalities: Array<Record<string, unknown>>;
|
|
49
|
-
output_modalities: Array<Record<string, unknown>>;
|
|
50
|
-
}
|
|
51
|
-
|
|
25
|
+
declare const METRICS_PORT = 9464;
|
|
26
|
+
/** The path the ServiceMonitor scrapes. */
|
|
27
|
+
declare const METRICS_PATH = "/metrics";
|
|
28
|
+
/** The label value every bounded label collapses to once its budget is spent. */
|
|
29
|
+
declare const OTHER_LABEL = "(other)";
|
|
30
|
+
/** The label value for a request that named no model. */
|
|
31
|
+
declare const NO_MODEL_LABEL = "(none)";
|
|
52
32
|
/**
|
|
53
|
-
*
|
|
54
|
-
*
|
|
55
|
-
*
|
|
56
|
-
*
|
|
57
|
-
* phase (explicitly out of scope here; urun-infra#1490 shared-endpoints)
|
|
58
|
-
* auto-creates from the model catalog on first request.
|
|
59
|
-
*
|
|
60
|
-
* MODEL-ID SURFACE (the documented mapping): a model may be named by
|
|
61
|
-
* - the app slug itself ("qwen3-6-27b-bf16"), or
|
|
62
|
-
* - the catalog id ("qwen3.6-27b"), or
|
|
63
|
-
* - the catalog id:variant ("qwen3.6-27b:bf16"),
|
|
64
|
-
* where slugification mirrors urun-cli `serve.py _default_app_name` exactly:
|
|
65
|
-
* lowercase, every non-alphanumeric-non-dash character becomes "-", leading/
|
|
66
|
-
* trailing dashes stripped (catalog id "qwen3.6-27b" + variant "bf16" → app
|
|
67
|
-
* "qwen3-6-27b-bf16"). COLLISION RULE: an exact slug match always wins over
|
|
68
|
-
* the catalog-id (prefix) interpretation.
|
|
69
|
-
*
|
|
70
|
-
* RESOLUTION ORDER (one canonical path, documented end to end):
|
|
71
|
-
* 1. model absent / "urun" / an alias of the startup app → DEFAULT app.
|
|
72
|
-
* 2. exact slug match on a deployed serve app → that app.
|
|
73
|
-
* 3. catalog-id form matching exactly one deployed app → that app
|
|
74
|
-
* (two or more candidates → loud ambiguity error naming them).
|
|
75
|
-
* 4. the name maps to an org app that is NOT an active app exposing the
|
|
76
|
-
* proxy's serve function → loud 404
|
|
77
|
-
* naming `urun serve <model>` and the available models.
|
|
78
|
-
* 5. the name matches a catalog row but no deployed app → loud 404
|
|
79
|
-
* naming `urun serve <id>` (deployed-only v1).
|
|
80
|
-
* 6. anything else — a model name outside the uRun namespace entirely
|
|
81
|
-
* (e.g. the harness's own upstream default, "claude-*"/"gpt-*") →
|
|
82
|
-
* DEFAULT app. This IS today's single-app contract, kept deliberately
|
|
83
|
-
* so `urun compat <agent>` with the agent's stock model keeps working
|
|
84
|
-
* with zero new env; the per-model /stats table records every such
|
|
85
|
-
* mapping so it is visible, never silent. Models the proxy ADVERTISES
|
|
86
|
-
* on /v1/models can never land here — they resolve (2/3) or fail loud
|
|
87
|
-
* (4/5) above.
|
|
88
|
-
*
|
|
89
|
-
* NO-DEFAULT MODE (`defaultApp: null`) — the HOSTED multi-tenant lane
|
|
90
|
-
* (`src/hosted/`): a shared endpoint serving every org has no "the app this
|
|
91
|
-
* proxy was started for", so rules 1 and 6 have nothing to fall back TO.
|
|
92
|
-
* Rather than inventing one (picking "some" app for a caller would be the
|
|
93
|
-
* worst kind of silent divergence), both rules become the SAME loud
|
|
94
|
-
* {@link UnknownModelError} that rules 4/5 already raise: name a deployed
|
|
95
|
-
* model, here is the list. Rules 2–5 are byte-for-byte the local behavior —
|
|
96
|
-
* one router, one resolution order, two configurations.
|
|
33
|
+
* Distinct `model` values that get their own series. Above this the label
|
|
34
|
+
* collapses to {@link OTHER_LABEL} — a caller sending a fresh random model
|
|
35
|
+
* string per request must not be able to grow this process's series set (and
|
|
36
|
+
* therefore Prometheus's) without bound.
|
|
97
37
|
*/
|
|
98
|
-
|
|
99
|
-
/** One org app row from `GET {orgApi}/apps` (urun-cli `ApiClient.list_apps`). */
|
|
100
|
-
interface DeployedApp {
|
|
101
|
-
app_slug: string;
|
|
102
|
-
function_name?: string | null;
|
|
103
|
-
deployment_status?: string | null;
|
|
104
|
-
[k: string]: unknown;
|
|
105
|
-
}
|
|
38
|
+
declare const MAX_MODEL_LABELS = 64;
|
|
106
39
|
/**
|
|
107
|
-
*
|
|
108
|
-
*
|
|
109
|
-
*
|
|
110
|
-
*
|
|
40
|
+
* Distinct `org` values that get their own series. Bounded for the same reason
|
|
41
|
+
* even though org ids come from the control plane rather than the caller: the
|
|
42
|
+
* number of orgs is unbounded over time, and a metric's cardinality must be a
|
|
43
|
+
* property of the code, not of how successful sales were.
|
|
111
44
|
*/
|
|
112
|
-
declare
|
|
113
|
-
|
|
45
|
+
declare const MAX_ORG_LABELS = 128;
|
|
46
|
+
/** Collapse a client-supplied path onto the bounded lane label. */
|
|
47
|
+
declare function laneOf(path: string): string;
|
|
114
48
|
/**
|
|
115
|
-
*
|
|
116
|
-
*
|
|
117
|
-
*
|
|
118
|
-
*
|
|
119
|
-
*
|
|
120
|
-
*/
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
/**
|
|
145
|
-
* The startup app slug (URUN_APP) — the DEFAULT model — or `null` for the
|
|
146
|
-
* hosted multi-tenant lane, which has no per-proxy default app: there,
|
|
147
|
-
* every request must NAME a deployed model and an unnamed/unknown one
|
|
148
|
-
* fails loud instead of silently landing somewhere (see the module header,
|
|
149
|
-
* "NO-DEFAULT MODE").
|
|
150
|
-
*/
|
|
151
|
-
defaultApp: string | null;
|
|
152
|
-
/** The serve function name every routed app must expose (URUN_FUNCTION). */
|
|
153
|
-
fnName: string;
|
|
154
|
-
/** Open a backhaul session for an app slug (called at most once per app). */
|
|
155
|
-
openSession: (appSlug: string) => S | Promise<S>;
|
|
156
|
-
/** Terminal release for one pool entry (Session.end() underneath). */
|
|
157
|
-
closeSession: (entry: S) => Promise<void>;
|
|
158
|
-
/**
|
|
159
|
-
* List the org's deployed apps, or null when the credentials cannot
|
|
160
|
-
* (URUN_JWT lane: the pre-vended token is scoped to the default app, so
|
|
161
|
-
* there is no org listing AND no cross-app session — routing degrades to
|
|
162
|
-
* the default-app-only contract, which is exactly today's behavior).
|
|
163
|
-
*/
|
|
164
|
-
listApps: (() => Promise<DeployedApp[]>) | null;
|
|
165
|
-
/**
|
|
166
|
-
* Catalog rows (models.ts fetchCatalogRows) as the uRun-namespace oracle
|
|
167
|
-
* for rule 5 AND the /v1/models enrichment + OpenRouter provider-doc
|
|
168
|
-
* source, or null when catalog access is not configured. Optional fields
|
|
169
|
-
* beyond model_id/variant are tolerated (thin rows still typecheck).
|
|
170
|
-
*/
|
|
171
|
-
listCatalog: (() => Promise<CatalogRow[]>) | null;
|
|
172
|
-
/** Deployed-apps cache TTL (the list changes on deploys, not per request). */
|
|
173
|
-
appsTtlMs?: number;
|
|
49
|
+
* The status class label. `none` is its own class rather than being folded
|
|
50
|
+
* into 5xx: it means NO head ever flushed (an aborted upload, a client that
|
|
51
|
+
* vanished during the body read), which `inferenceRequestRecord` records as a
|
|
52
|
+
* null status precisely so nothing downstream invents a 200 the client was
|
|
53
|
+
* never sent.
|
|
54
|
+
*/
|
|
55
|
+
declare function statusClassOf(status: number | null): string;
|
|
56
|
+
/**
|
|
57
|
+
* A label whose distinct-value budget is FIXED at construction. The first
|
|
58
|
+
* `max` distinct values are ADMITTED; everything after is {@link OTHER_LABEL},
|
|
59
|
+
* permanently. Deliberately NOT an LRU: an LRU would let a flooding caller
|
|
60
|
+
* evict the real models and quietly rewrite history in the TSDB, where each
|
|
61
|
+
* admitted-then-evicted-then-readmitted value leaves a broken series behind.
|
|
62
|
+
*
|
|
63
|
+
* ADMISSION IS EARNED, AND THAT IS THE WHOLE DEFENCE. A budget alone does not
|
|
64
|
+
* protect anything: one authenticated caller sending `MAX_MODEL_LABELS`
|
|
65
|
+
* requests naming random models — every one of them answered 404, because an
|
|
66
|
+
* unknown model is refused — would spend the entire budget and collapse every
|
|
67
|
+
* REAL model onto `(other)` until the pod restarts. That kills the per-model
|
|
68
|
+
* TTFT and rate panels, which are exactly the two the Hugging Face latency
|
|
69
|
+
* gate is read from. So a value is admitted only when `admit` is true, and the
|
|
70
|
+
* caller passes `admit` for answers that were actually served; a refusal can
|
|
71
|
+
* still be COUNTED (under `(other)`) but can never consume budget.
|
|
72
|
+
*/
|
|
73
|
+
declare class BoundedLabel {
|
|
74
|
+
private readonly max;
|
|
75
|
+
private readonly admitted;
|
|
76
|
+
private overflows;
|
|
77
|
+
constructor(max: number);
|
|
174
78
|
/**
|
|
175
|
-
*
|
|
176
|
-
*
|
|
177
|
-
*
|
|
178
|
-
* ({@link ModelRouter.handleFor} / {@link ModelRouter.sessionForHandle});
|
|
179
|
-
* a router without it fails LOUD on those calls, never approximates.
|
|
79
|
+
* @param admit whether this observation may spend budget on a new value.
|
|
80
|
+
* False for refused requests, so a caller cannot blind the label by naming
|
|
81
|
+
* models that do not exist.
|
|
180
82
|
*/
|
|
181
|
-
|
|
83
|
+
of(value: string | null, admit?: boolean): string;
|
|
84
|
+
/** Distinct values admitted so far — published as a gauge (see below). */
|
|
85
|
+
get size(): number;
|
|
86
|
+
/** Admissible observations refused because the budget was spent. */
|
|
87
|
+
get overflowCount(): number;
|
|
182
88
|
}
|
|
183
89
|
/**
|
|
184
|
-
* The
|
|
185
|
-
*
|
|
186
|
-
*
|
|
187
|
-
*
|
|
188
|
-
*
|
|
90
|
+
* The hosted proxy's metric set, over its OWN {@link Registry} rather than the
|
|
91
|
+
* library's global `register`. Per-instance because the alternative is
|
|
92
|
+
* process-global mutable state that two tests (or two embedders) silently
|
|
93
|
+
* share — `Counter` construction on an already-registered name throws, so a
|
|
94
|
+
* global registry would make the second `createHostedProxy` in a process fail.
|
|
189
95
|
*/
|
|
190
|
-
declare class
|
|
191
|
-
|
|
192
|
-
private readonly
|
|
193
|
-
private
|
|
194
|
-
private
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
private
|
|
96
|
+
declare class InferenceMetrics {
|
|
97
|
+
readonly registry: Registry<_prometheus_io_client.RegistryContentType>;
|
|
98
|
+
private readonly modelLabel;
|
|
99
|
+
private readonly orgLabel;
|
|
100
|
+
private readonly requests;
|
|
101
|
+
private readonly duration;
|
|
102
|
+
private readonly ttft;
|
|
103
|
+
private readonly streamErrors;
|
|
104
|
+
private readonly undelivered;
|
|
105
|
+
private readonly ledgerWrites;
|
|
106
|
+
private readonly ledgerQueue;
|
|
107
|
+
private readonly labelCardinality;
|
|
108
|
+
private readonly labelOverflows;
|
|
109
|
+
constructor(options?: {
|
|
110
|
+
defaultMetrics?: boolean;
|
|
111
|
+
});
|
|
199
112
|
/**
|
|
200
|
-
*
|
|
201
|
-
*
|
|
202
|
-
* oracle is configured — callers degrade to the bare four-field entries.
|
|
203
|
-
*/
|
|
204
|
-
private catalogRows;
|
|
205
|
-
/** Apps this proxy may serve: active AND exposing the serve function. */
|
|
206
|
-
private servable;
|
|
207
|
-
private availableIds;
|
|
208
|
-
/**
|
|
209
|
-
* NO-DEFAULT MODE's terminal for rules 1 and 6: there is no app to fall
|
|
210
|
-
* back to, so say so loudly and list what the CALLER'S org actually has.
|
|
211
|
-
* Never returns.
|
|
212
|
-
*/
|
|
213
|
-
private noDefaultApp;
|
|
214
|
-
/**
|
|
215
|
-
* Resolve a request's `model` to an app slug — the documented resolution
|
|
216
|
-
* order from the module header. Throws {@link UnknownModelError} for a uRun
|
|
217
|
-
* model that is not deployed (rules 4/5).
|
|
218
|
-
*/
|
|
219
|
-
resolveApp(model: string | undefined): Promise<string>;
|
|
220
|
-
/** The pooled session for a model — opened lazily, reused afterwards. */
|
|
221
|
-
sessionFor(model: string | undefined): Promise<{
|
|
222
|
-
app: string;
|
|
223
|
-
entry: S;
|
|
224
|
-
}>;
|
|
225
|
-
private keyOf;
|
|
226
|
-
/**
|
|
227
|
-
* SESSION-IDENTITY SEAM (a): the opaque stable handle for the pooled
|
|
228
|
-
* session currently serving `model`'s turns. Rides the SAME acquisition
|
|
229
|
-
* path as every request ({@link sessionFor}) — the session opens lazily if
|
|
230
|
-
* this model has none yet — and derives the handle from native identity
|
|
231
|
-
* (app slug + uRun session id), zero bespoke bookkeeping.
|
|
232
|
-
*/
|
|
233
|
-
handleFor(model: string | undefined): Promise<{
|
|
234
|
-
app: string;
|
|
235
|
-
handle: string;
|
|
236
|
-
}>;
|
|
237
|
-
/**
|
|
238
|
-
* SESSION-IDENTITY SEAM (b): the exact pooled session a handle names.
|
|
239
|
-
* NEVER opens a fresh session — a handle whose session is gone (closed,
|
|
240
|
-
* evicted, re-homed to a replacement, proxy restarted) or malformed throws
|
|
241
|
-
* {@link SessionGoneError} loudly. Resume is reattach-or-fail, not
|
|
242
|
-
* reattach-or-quietly-restart.
|
|
243
|
-
*/
|
|
244
|
-
sessionForHandle(handle: string): Promise<{
|
|
245
|
-
app: string;
|
|
246
|
-
entry: S;
|
|
247
|
-
}>;
|
|
248
|
-
/**
|
|
249
|
-
* Drop ONE pooled session whose backhaul died (its pod was restarted /
|
|
250
|
-
* drained / deleted) and release it — the next {@link sessionFor} opens a
|
|
251
|
-
* fresh one, i.e. asks the control plane for a new assignment. Used by the
|
|
252
|
-
* one-shot re-home (rehome.ts, urun-sh/urun-python#1592).
|
|
113
|
+
* Count ONE terminal ledger outcome — the third sink for a request that
|
|
114
|
+
* `proxy/ledger.ts` already logs, counted so it can be alerted on.
|
|
253
115
|
*
|
|
254
|
-
*
|
|
255
|
-
*
|
|
256
|
-
*
|
|
257
|
-
*
|
|
258
|
-
*/
|
|
259
|
-
evict(app: string, entry: S): Promise<void>;
|
|
260
|
-
/**
|
|
261
|
-
* `GET /v1/models`: the org's deployed serve apps as model entries, the
|
|
262
|
-
* default app FIRST. On the JWT lane (no org listing) this is the default
|
|
263
|
-
* app plus any app already in the pool — the gap is called out loudly in
|
|
264
|
-
* the PR, not papered over here.
|
|
116
|
+
* Takes the very {@link LedgerOutcome} the log line carries, for the same
|
|
117
|
+
* reason {@link observe} takes the record verbatim: a counter derived from a
|
|
118
|
+
* second walk would eventually disagree with the billing log, and the
|
|
119
|
+
* disagreement would be invisible.
|
|
265
120
|
*/
|
|
266
|
-
|
|
121
|
+
observeLedger(outcome: LedgerOutcome): void;
|
|
267
122
|
/**
|
|
268
|
-
*
|
|
269
|
-
*
|
|
270
|
-
*
|
|
271
|
-
* vl — a voice or video app behind a chat surface would stream garbage);
|
|
272
|
-
* models the catalog cannot vouch for are OMITTED, never guessed. Pricing
|
|
273
|
-
* is deliberately omitted entirely (schema rule: "A modality with no
|
|
274
|
-
* pricing array is simply unpriced" — OpenRouter configures pricing during
|
|
275
|
-
* provider onboarding; the per-model price surface is urun-infra U3).
|
|
123
|
+
* Count ONE completed `/v1` request. Takes the very record
|
|
124
|
+
* `inferenceRequestLine` serializes, so the counters and the billing log can
|
|
125
|
+
* never drift apart.
|
|
276
126
|
*/
|
|
277
|
-
|
|
278
|
-
/**
|
|
279
|
-
|
|
127
|
+
observe(record: InferenceRecord): void;
|
|
128
|
+
/** The exposition text a scrape answers with. */
|
|
129
|
+
scrape(): Promise<string>;
|
|
130
|
+
/** The `Content-Type` that exposition must be served under. */
|
|
131
|
+
get contentType(): string;
|
|
280
132
|
}
|
|
133
|
+
/**
|
|
134
|
+
* The PRIVATE metrics listener — its own `http.Server` on
|
|
135
|
+
* {@link METRICS_PORT}, serving exactly `GET /metrics` and 404 for everything
|
|
136
|
+
* else.
|
|
137
|
+
*
|
|
138
|
+
* A SEPARATE SERVER, not a route on the main one, is the whole point: the
|
|
139
|
+
* public gateway forwards only to the main port, so "not public" is a property
|
|
140
|
+
* of the socket rather than a rule the `/v1` router has to keep remembering.
|
|
141
|
+
* It also means a scrape can still be answered while the main server is
|
|
142
|
+
* draining, which is exactly when the numbers matter most.
|
|
143
|
+
*/
|
|
144
|
+
declare function createMetricsServer(metrics: InferenceMetrics): Server;
|
|
281
145
|
|
|
282
146
|
/**
|
|
283
147
|
* HOSTED AUTH — the Bearer key IS the identity AND the tenancy.
|
|
@@ -401,17 +265,24 @@ declare class OrgResolver {
|
|
|
401
265
|
* `consecutiveFailures` resets on every healthy sync, so a recovered
|
|
402
266
|
* blip never trips this.
|
|
403
267
|
*
|
|
404
|
-
*
|
|
405
|
-
* 'error')
|
|
406
|
-
*
|
|
268
|
+
* Core Session's own phase surface (`Session.onPhase` → 'live' /
|
|
269
|
+
* 'paused' / 'error'). The verdict arms ONLY on a `live` sighting: a
|
|
270
|
+
* binding must EXIST before it can be lost. A session born into its
|
|
271
|
+
* pre-live wait (the cold wake) or re-attached in `error` (resume runs
|
|
272
|
+
* the function from the top, session-semantics.md) is ACQUIRING
|
|
273
|
+
* compute, never losing it — core's own whenLive gate treats both as
|
|
274
|
+
* WAITABLE, and the unguarded latch 502-looped every shared chat/agent
|
|
275
|
+
* model on exactly those sightings (prod 2026-09-14: the entry was
|
|
276
|
+
* evicted before it ever served, and "retry to get a fresh session"
|
|
277
|
+
* re-dialed into the same verdict forever).
|
|
407
278
|
*
|
|
408
|
-
* NOTE what "gone" means HERE, because it is narrower than it reads:
|
|
409
|
-
* pool holds a session with COMPUTE BOUND to it, and a
|
|
410
|
-
* that binding is gone. The session ITSELF is not
|
|
411
|
-
* listed, and attachable by name (see
|
|
412
|
-
* urun-python/docs/design/session-semantics.md). The eviction is
|
|
413
|
-
* because the POOL ENTRY is what became unusable; nothing here
|
|
414
|
-
* to conclude the session is over.
|
|
279
|
+
* NOTE what "gone" means HERE, because it is narrower than it reads:
|
|
280
|
+
* this pool holds a session with COMPUTE BOUND to it, and a post-live
|
|
281
|
+
* `paused` phase says that binding is gone. The session ITSELF is not
|
|
282
|
+
* gone — it is parked, listed, and attachable by name (see
|
|
283
|
+
* urun-python/docs/design/session-semantics.md). The eviction is
|
|
284
|
+
* correct because the POOL ENTRY is what became unusable; nothing here
|
|
285
|
+
* is entitled to conclude the session is over.
|
|
415
286
|
*
|
|
416
287
|
* Consumers (cli.ts):
|
|
417
288
|
* - {@link watchSessionGone} per pooled entry; `onGone` → `router.evict`
|
|
@@ -479,12 +350,24 @@ interface CatalogConfig {
|
|
|
479
350
|
*/
|
|
480
351
|
type OwnedSession = UrunSessionLike & {
|
|
481
352
|
end: () => Promise<unknown>;
|
|
353
|
+
/**
|
|
354
|
+
* The SDK's NATIVE local release (core `Session.detach`): drop the media
|
|
355
|
+
* transport and decrement the reference count while the named session
|
|
356
|
+
* stays live and attachable. Optional here because only the
|
|
357
|
+
* `release: 'detach'` lane requires it — that lane fails LOUD when a
|
|
358
|
+
* session object lacks it (no approximation, no fake detach).
|
|
359
|
+
*/
|
|
360
|
+
detach?: () => void;
|
|
482
361
|
id?: string;
|
|
483
362
|
endsAt?: Date | null;
|
|
484
363
|
onPhase?: (handler: (phase: {
|
|
485
364
|
name: string;
|
|
486
365
|
reason?: string;
|
|
487
366
|
}) => void) => () => void;
|
|
367
|
+
whenLive?: (options?: {
|
|
368
|
+
timeout?: number;
|
|
369
|
+
signal?: AbortSignal;
|
|
370
|
+
}) => Promise<void>;
|
|
488
371
|
};
|
|
489
372
|
/**
|
|
490
373
|
* One pooled backhaul: the session, its (stateful) Responses client, and the
|
|
@@ -584,6 +467,24 @@ declare class TenantRegistry {
|
|
|
584
467
|
/** Insertion-ordered = LRU order, because a hit re-inserts at the end. */
|
|
585
468
|
private readonly tenants;
|
|
586
469
|
constructor(opts: TenantRegistryOptions);
|
|
470
|
+
/**
|
|
471
|
+
* THE NON-SECRET TENANCY LABEL, attached by the REGISTRY rather than by
|
|
472
|
+
* whichever builder produced the backhaul.
|
|
473
|
+
*
|
|
474
|
+
* The caller's org id is what the per-request record and the Prometheus
|
|
475
|
+
* series are broken down by (`proxy/responses-turn.ts` `ProxyClients.tenant`)
|
|
476
|
+
* — never `caller.apiKey`, which is the credential on this surface and would
|
|
477
|
+
* become a credential on every dashboard. Attaching it here means a backhaul
|
|
478
|
+
* can never come back unlabelled because a builder forgot.
|
|
479
|
+
*
|
|
480
|
+
* `Object.create`, not a mutation and not a spread: the source object keeps
|
|
481
|
+
* its identity and its prototype (so a class-based `ProxyClients` keeps its
|
|
482
|
+
* methods), and no other caller's backhaul is touched. Done ONCE per tenant,
|
|
483
|
+
* at build time — the returned object is then cached and reused for every
|
|
484
|
+
* later request from that key, which matters because the `/v1/responses`
|
|
485
|
+
* thread store is keyed on this very object identity.
|
|
486
|
+
*/
|
|
487
|
+
private label;
|
|
587
488
|
private build;
|
|
588
489
|
/** The backhaul for THIS caller, opened on first use and reused after. */
|
|
589
490
|
clientsFor(caller: CallerIdentity): ProxyClients;
|
|
@@ -592,6 +493,358 @@ declare class TenantRegistry {
|
|
|
592
493
|
/** Shutdown: end every pooled session through `Session.end()`. */
|
|
593
494
|
closeAll(): Promise<void>;
|
|
594
495
|
}
|
|
496
|
+
/** The most distinct LIVE conversations one replica keeps private backhauls for. */
|
|
497
|
+
declare const MAX_CONVERSATIONS = 512;
|
|
498
|
+
/** The most resume handles the conversation index remembers (bounded memory). */
|
|
499
|
+
declare const MAX_CONVERSATION_HANDLES = 1024;
|
|
500
|
+
/**
|
|
501
|
+
* Every conversation slot holds a conversation (active or resumable) — a NEW
|
|
502
|
+
* bootstrap/setup is refused. Explicit and honest: never an LRU-detach of
|
|
503
|
+
* someone else's conversation to admit a new one.
|
|
504
|
+
*/
|
|
505
|
+
declare class ConversationCapacityError extends Error {
|
|
506
|
+
constructor();
|
|
507
|
+
}
|
|
508
|
+
/**
|
|
509
|
+
* One conversation's private backhaul. Its pooled sessions are reachable by
|
|
510
|
+
* NO other conversation of any caller: the token subject is conversation-
|
|
511
|
+
* unique, so the platform's own session dedupe never coalesces them.
|
|
512
|
+
*/
|
|
513
|
+
interface PrivateConversationBackhaul {
|
|
514
|
+
clients: ProxyClients;
|
|
515
|
+
/**
|
|
516
|
+
* LOCAL release of every pooled session — `release: 'detach'` underneath,
|
|
517
|
+
* i.e. core's NATIVE `Session.detach()`: the media transport drops, the
|
|
518
|
+
* refcount decrements, the NAMED session stays live and attachable. Never
|
|
519
|
+
* `Session.end()`: no caller has explicitly ended anything.
|
|
520
|
+
*/
|
|
521
|
+
release(): Promise<void>;
|
|
522
|
+
}
|
|
523
|
+
interface ConversationBackhaulOptions {
|
|
524
|
+
/** The session-gateway base sessions are opened against. */
|
|
525
|
+
baseUrl: string;
|
|
526
|
+
/** The org control-plane API base (`GET {apiUrl}/apps`). */
|
|
527
|
+
apiUrl: string;
|
|
528
|
+
/** The model-catalog oracle for routing rule 5, or null. */
|
|
529
|
+
catalog: CatalogConfig | null;
|
|
530
|
+
/** Injected in tests; production builds real uRun backhauls. */
|
|
531
|
+
build?: (caller: CallerIdentity, conversation: {
|
|
532
|
+
id: string;
|
|
533
|
+
subject: string;
|
|
534
|
+
}) => PrivateConversationBackhaul;
|
|
535
|
+
}
|
|
536
|
+
/**
|
|
537
|
+
* The per-CONNECTION Live/realtime seam — transport-neutral (Gemini Live,
|
|
538
|
+
* OpenAI Realtime, ...). `clients` lazily opens the caller's PRIVATE
|
|
539
|
+
* conversation backhaul on first use; `bindResumeHandle` indexes each handle
|
|
540
|
+
* the protocol lane mints; `clientsForResume` re-attaches the conversation a
|
|
541
|
+
* handle belongs to — SAME native identity, owner binding enforced, loud
|
|
542
|
+
* SessionGoneError when the conversation is gone; `release` drops THIS
|
|
543
|
+
* attachment when the socket closes.
|
|
544
|
+
*/
|
|
545
|
+
interface PrivateConversationConnection {
|
|
546
|
+
clients: ProxyClients;
|
|
547
|
+
bindResumeHandle(handle: string): void;
|
|
548
|
+
clientsForResume(handle: string): Promise<ProxyClients>;
|
|
549
|
+
/**
|
|
550
|
+
* Release THIS attachment: called by the surface when its socket closes.
|
|
551
|
+
* When the LAST attachment of a conversation goes, its native transport is
|
|
552
|
+
* detached immediately — while the conversation's native session
|
|
553
|
+
* REFERENCES (and its resume handles) are RETAINED, so a later resume
|
|
554
|
+
* re-attaches the SAME named session through canonical attach-by-name.
|
|
555
|
+
* Idempotent.
|
|
556
|
+
*/
|
|
557
|
+
release(): Promise<void>;
|
|
558
|
+
}
|
|
559
|
+
/**
|
|
560
|
+
* Process-local registry of per-conversation private backhauls, bounded by
|
|
561
|
+
* {@link MAX_CONVERSATIONS}. A conversation whose last attachment closes
|
|
562
|
+
* KEEPS its entry: its native session references + resume handles survive so
|
|
563
|
+
* a later resume re-attaches the SAME named session (canonical
|
|
564
|
+
* attach-by-name). At capacity, only such DETACHED entries are evicted
|
|
565
|
+
* (oldest first — their resume capability is the only thing lost, the named
|
|
566
|
+
* sessions stay attachable server-side); ACTIVE conversations are never
|
|
567
|
+
* touched, and when every slot is active a new bootstrap/setup is refused
|
|
568
|
+
* with an explicit capacity error. NO HA CLAIM: this registry is
|
|
569
|
+
* per-process; a replica restart loses every conversation (loud gone), and
|
|
570
|
+
* no other replica can serve one.
|
|
571
|
+
*/
|
|
572
|
+
declare class ConversationBackhauls {
|
|
573
|
+
private readonly opts;
|
|
574
|
+
/** apiKey → conversationId → entry (insertion-ordered = LRU). */
|
|
575
|
+
private readonly byCaller;
|
|
576
|
+
/** resume handle → owning conversation (the owner binding's index). */
|
|
577
|
+
private readonly byHandle;
|
|
578
|
+
constructor(opts: ConversationBackhaulOptions);
|
|
579
|
+
private build;
|
|
580
|
+
/** Find (LRU-refresh) or create the conversation entry, then ATTACH to it. */
|
|
581
|
+
private attachEntry;
|
|
582
|
+
private get conversationCount();
|
|
583
|
+
/**
|
|
584
|
+
* Free a slot for a NEW conversation: evict the OLDEST DETACHED
|
|
585
|
+
* conversation (one whose last attachment already closed — only its resume
|
|
586
|
+
* metadata is lost, no live transport is touched, no session is ended).
|
|
587
|
+
* ACTIVE conversations are never evicted. When every slot holds an active
|
|
588
|
+
* conversation, refuse LOUD with an explicit capacity error — the new
|
|
589
|
+
* bootstrap/setup fails, existing callers are untouched.
|
|
590
|
+
*/
|
|
591
|
+
private evictOverflow;
|
|
592
|
+
private dropHandleIndexFor;
|
|
593
|
+
/**
|
|
594
|
+
* The per-upgrade seam for one authenticated caller — ONE attachment. The
|
|
595
|
+
* conversation backhaul opens on the FIRST seam call of a fresh connection;
|
|
596
|
+
* a resumed connection replaces `clients` wholesale via `clientsForResume`
|
|
597
|
+
* before any seam call, so the lazy path is never taken on resume. The
|
|
598
|
+
* attachment is released by `release()` (the socket's close path): when the
|
|
599
|
+
* LAST attachment of a conversation goes, its native transport is detached
|
|
600
|
+
* immediately and its resume metadata is retained — no resource keepalive,
|
|
601
|
+
* no resume loss.
|
|
602
|
+
*/
|
|
603
|
+
connectionFor(caller: CallerIdentity): PrivateConversationConnection;
|
|
604
|
+
/**
|
|
605
|
+
* Drain-time LOCAL release: detach every conversation's sessions (native
|
|
606
|
+
* detach — the named sessions pause server-side, nothing is ended). Handles
|
|
607
|
+
* and conversations are forgotten; process exit is the conversation's end.
|
|
608
|
+
*/
|
|
609
|
+
detachAll(): Promise<void>;
|
|
610
|
+
}
|
|
611
|
+
|
|
612
|
+
/**
|
|
613
|
+
* THE HOSTED USAGE READ — the control-plane half of the canonical usage lane.
|
|
614
|
+
*
|
|
615
|
+
* The proxy holds no database credential and no standing platform secret
|
|
616
|
+
* (hosted/auth.ts). So the ledger read rides the CALLER'S OWN org API key to
|
|
617
|
+
* the control plane's `inference-usage` edge function — the same mechanism and
|
|
618
|
+
* the same credential `TenantRegistry` uses for `GET {apiUrl}/apps` — and the
|
|
619
|
+
* control plane derives the org from that key server-side and runs the
|
|
620
|
+
* org-scoped `urun_inference_usage_lookup` RPC.
|
|
621
|
+
*
|
|
622
|
+
* THAT IS WHY THIS LANE DOES NOT PRE-VERIFY THE KEY WITH `OrgResolver`. The
|
|
623
|
+
* surface that owns the data authenticates the key itself, so a second
|
|
624
|
+
* verification here would be a second source of truth about the same
|
|
625
|
+
* credential — and the org binding it produced would not be the one the read
|
|
626
|
+
* was actually scoped by. `bearerFrom` still runs at the call site, so a
|
|
627
|
+
* missing or malformed Authorization header is a loud 401 without a round trip.
|
|
628
|
+
*
|
|
629
|
+
* EVERY FAILURE IS LOUD. There is no shape of failure that answers with an
|
|
630
|
+
* empty `requests` list, because "we hold nothing for these ids" is precisely
|
|
631
|
+
* the answer that makes a billing caller stop asking — Hugging Face writes a
|
|
632
|
+
* request off ~30 minutes after it was served — so an outage that could wear
|
|
633
|
+
* that costume would be an invisible revenue hole.
|
|
634
|
+
*/
|
|
635
|
+
|
|
636
|
+
interface UsageClientOptions {
|
|
637
|
+
/** The org control-plane API base (`{apiUrl}`) the function is mounted under. */
|
|
638
|
+
apiUrl: string;
|
|
639
|
+
/** Injected in tests; production uses the platform fetch. */
|
|
640
|
+
fetchImpl?: typeof fetch;
|
|
641
|
+
/**
|
|
642
|
+
* Where a row this lane REFUSED TO ANSWER ABOUT is reported (ENG-416).
|
|
643
|
+
*
|
|
644
|
+
* A malformed ledger row is isolated rather than allowed to 502 the whole
|
|
645
|
+
* 10,000-id batch, and the row it dropped is then MISSING from the answer —
|
|
646
|
+
* which a billing caller cannot tell apart from an id we do not hold, and
|
|
647
|
+
* reads as "not priced yet" until the request is written off unbilled. So
|
|
648
|
+
* the drop must be an event somebody can alert on. Absent, it goes to this
|
|
649
|
+
* process's stdout (`parseUsageRecords`'s own default), never nowhere.
|
|
650
|
+
*
|
|
651
|
+
* It may be async — a sink that ships the line somewhere usually is — and a
|
|
652
|
+
* rejected promise from it is contained exactly as a synchronous throw is.
|
|
653
|
+
*/
|
|
654
|
+
onRejectedRow?: UsageRowReporter;
|
|
655
|
+
}
|
|
656
|
+
/** The batch ledger read, as one call per request. */
|
|
657
|
+
declare class UsageClient {
|
|
658
|
+
private readonly opts;
|
|
659
|
+
private readonly fetchImpl;
|
|
660
|
+
constructor(opts: UsageClientOptions);
|
|
661
|
+
/**
|
|
662
|
+
* Look up what we hold for these ids, as the holder of `apiKey`.
|
|
663
|
+
*
|
|
664
|
+
* Three outcomes, all explicit:
|
|
665
|
+
* - the control plane answers → validated {@link InferenceUsageRecord}s;
|
|
666
|
+
* - it rejects the key (401/403) → {@link ProxyAuthError} (401);
|
|
667
|
+
* - anything else → {@link UsageSurfaceError} (502).
|
|
668
|
+
*/
|
|
669
|
+
lookup(apiKey: string, inferenceIds: readonly string[]): Promise<InferenceUsageRecord[]>;
|
|
670
|
+
}
|
|
671
|
+
|
|
672
|
+
/**
|
|
673
|
+
* THE HOSTED LEDGER WRITE — the control-plane half of the per-request ledger.
|
|
674
|
+
*
|
|
675
|
+
* The proxy holds no database credential and no standing platform secret
|
|
676
|
+
* (hosted/auth.ts: "NO STANDING CREDENTIAL... every upstream call it makes
|
|
677
|
+
* rides the CALLER'S OWN key"). So the ledger write rides the caller's own org
|
|
678
|
+
* API key to the control plane's `inference-ledger` function — the same
|
|
679
|
+
* mechanism and the same credential the ENG-318 read already uses — and the
|
|
680
|
+
* control plane derives the org AND the api_key_id from that key server-side
|
|
681
|
+
* before calling `urun_record_inference_request`.
|
|
682
|
+
*
|
|
683
|
+
* THAT PROPERTY IS SOUND ON THE READ AND INVERTED ON THE WRITE, and it is
|
|
684
|
+
* stated here because it is invisible in the code. On a read, riding the
|
|
685
|
+
* caller's key is precisely what enforces org scoping. On a WRITE TO A BILLING
|
|
686
|
+
* LEDGER it means the party being billed is the party authenticated to write
|
|
687
|
+
* the bill: a caller can read its `Inference-Id` off a streaming response head
|
|
688
|
+
* and race a fabricated row in ahead of this one. The control plane refuses
|
|
689
|
+
* the second write rather than upserting, so the attempt is LOUD instead of
|
|
690
|
+
* silent — see {@link LedgerDuplicateError}.
|
|
691
|
+
*
|
|
692
|
+
* ============================ THE ENG-410 RULING ============================
|
|
693
|
+
*
|
|
694
|
+
* That is a DETECTION, and it is the DELIBERATE, OWNER-APPROVED choice — not a
|
|
695
|
+
* gap somebody failed to close. The decision and the condition that reopens it
|
|
696
|
+
* are recorded here because the next person to read this file will otherwise
|
|
697
|
+
* re-derive the wrong answer from first principles.
|
|
698
|
+
*
|
|
699
|
+
* WHAT THE ATTACK ACTUALLY IS, bounded honestly. `org_id` and `api_key_id` are
|
|
700
|
+
* derived from the credential and the control-plane surface REFUSES both as
|
|
701
|
+
* body fields, so cross-tenant billing INJECTION is structurally impossible: a
|
|
702
|
+
* forged row can only ever land in the forger's OWN org. The only rational
|
|
703
|
+
* attack is therefore SELF-under-billing, it is not really a race (the header
|
|
704
|
+
* flushes at the head of a stream and this write happens on close, so the
|
|
705
|
+
* attacker has the whole generation), and it cannot be made quiet — every
|
|
706
|
+
* stolen request costs the attacker one refused write and one error line.
|
|
707
|
+
*
|
|
708
|
+
* REJECTED — GIVE THE PROXY A STANDING PLATFORM CREDENTIAL FOR THE WRITE. This
|
|
709
|
+
* is the obvious-sounding move and it is STRICTLY WORSE THAN THE FLAW. For a
|
|
710
|
+
* platform credential to REPLACE the caller's key on this write, `org_id` has
|
|
711
|
+
* to come back into the request body — there is nothing else left to carry
|
|
712
|
+
* tenancy. That deletes the one structural property that makes this surface
|
|
713
|
+
* safe at all, and it converts a proxy compromise from "whatever caller keys
|
|
714
|
+
* happen to be in flight" into "arbitrary billing rows for EVERY org, written
|
|
715
|
+
* from an internet-facing pod". The generalised rule, which outlives this
|
|
716
|
+
* module: A CREDENTIAL MAY BE ADDITIVE TO THE CALLER'S KEY, NEVER A
|
|
717
|
+
* REPLACEMENT FOR IT, because the caller's key is what carries tenancy.
|
|
718
|
+
*
|
|
719
|
+
* REJECTED — HAVE THE CONTROL PLANE DERIVE MORE AND TRUST THE BODY LESS. It
|
|
720
|
+
* already derives everything it can (`org_id`, `api_key_id`) and refuses what
|
|
721
|
+
* it must (`gpu_seconds`). Of what is left, the fields that DECIDE MONEY are
|
|
722
|
+
* exactly the fields it has no independent source for: nothing in the platform
|
|
723
|
+
* reports per-request token counts anywhere except this wire, and the
|
|
724
|
+
* `inference_id` is minted here. Authenticating the id would not help either —
|
|
725
|
+
* the attacker holds a LEGITIMATE id, read from its own response head.
|
|
726
|
+
*
|
|
727
|
+
* DEFERRED, WITH A TRIGGER — MAKE THE ROW UNFORGEABLE (a proxy-held key that
|
|
728
|
+
* signs the row's CONTENT, verified by the control plane, ADDITIVE to the
|
|
729
|
+
* caller's key so tenancy is untouched). This is the real fix and its blast
|
|
730
|
+
* radius is small: a stolen signing key only restores today's position,
|
|
731
|
+
* because the row still lands in whatever org the thief's own key names. It is
|
|
732
|
+
* NOT built yet because its own failure mode is worse than the flaw's while
|
|
733
|
+
* nothing bills off this table: a misconfigured or badly-rotated key takes
|
|
734
|
+
* 100% OF LEDGER ROWS TO ZERO ACROSS EVERY ORG, where the flaw takes some rows
|
|
735
|
+
* to zero for one customer attacking itself, loudly. Fitting a cryptographic
|
|
736
|
+
* gate to a writer that has never once run in production, in the week it first
|
|
737
|
+
* writes a row, is how a remedy costs more than the bug.
|
|
738
|
+
*
|
|
739
|
+
* THE TRIGGER, EXPLICITLY: BUILD IT BEFORE ANY PRODUCT BILLS OFF
|
|
740
|
+
* `public.inference_requests`. Today HF traffic authenticates as HF's own
|
|
741
|
+
* org (the end user never holds a uRun key) and self-serve is billed off
|
|
742
|
+
* `usage_events`, so no party both can run the attack and benefits from it.
|
|
743
|
+
* The credits / concurrency-tier / retention-tier product is the moment that
|
|
744
|
+
* stops being true, and it must not be the moment this is discovered.
|
|
745
|
+
*
|
|
746
|
+
* UNTIL THEN THE DETECTION IS THE CONTROL, so it is built to be OPERATED
|
|
747
|
+
* rather than merely to exist: the refusal carries a machine-readable
|
|
748
|
+
* `ledgerFailureReason` (`proxy/ledger.ts` {@link LedgerFailure}) that reaches
|
|
749
|
+
* both the structured log line and a Prometheus counter, so the alert is a
|
|
750
|
+
* field match and not a regex over this paragraph's prose.
|
|
751
|
+
*
|
|
752
|
+
* ===========================================================================
|
|
753
|
+
*
|
|
754
|
+
* EVERY FAILURE IS LOUD AND NONE OF THEM REACHES THE CUSTOMER. This runs from
|
|
755
|
+
* the response's `'close'` hook, after the answer is delivered, so there is no
|
|
756
|
+
* request left to fail: `proxy/ledger.ts` catches everything here and writes
|
|
757
|
+
* one structured log line. A write that did not land is revenue that was never
|
|
758
|
+
* recorded, which is exactly the thing that must never be silent.
|
|
759
|
+
*/
|
|
760
|
+
|
|
761
|
+
/**
|
|
762
|
+
* The control-plane route this writes. Relative to the org control-plane API
|
|
763
|
+
* base (`{apiUrl}`, i.e. `https://api.urun.sh/v1`), which the deployment's
|
|
764
|
+
* gateway rewrites onto the Supabase Edge Functions host — the same base and
|
|
765
|
+
* the same rewrite `GET {apiUrl}/apps` and the usage read already ride.
|
|
766
|
+
*/
|
|
767
|
+
declare const LEDGER_FUNCTION_PATH = "/inference-ledger";
|
|
768
|
+
/**
|
|
769
|
+
* How long one write may take before it is abandoned. A single-row insert plus
|
|
770
|
+
* two indexed price lookups is milliseconds; this bound exists so a stalled
|
|
771
|
+
* socket becomes a log line rather than a promise retained for the life of the
|
|
772
|
+
* pod. Deliberately TIGHTER than the usage read's 15s: that read answers a
|
|
773
|
+
* live poller that is waiting, while this write has nobody waiting on it and
|
|
774
|
+
* every second it holds is a slot in the in-flight bound.
|
|
775
|
+
*/
|
|
776
|
+
declare const LEDGER_WRITE_TIMEOUT_MS = 10000;
|
|
777
|
+
/**
|
|
778
|
+
* The row was already in the ledger. Its own type because it is not a fault
|
|
779
|
+
* and must not be logged as one: it means somebody wrote this `inference_id`
|
|
780
|
+
* before us, which is either a double-write defect in this proxy or a row
|
|
781
|
+
* fabricated by the org being billed. Nothing was written and nothing was
|
|
782
|
+
* overwritten.
|
|
783
|
+
*/
|
|
784
|
+
declare class LedgerDuplicateError extends Error implements LedgerFailure {
|
|
785
|
+
/**
|
|
786
|
+
* THE ENG-410 SIGNAL, as a field an alert can match.
|
|
787
|
+
*
|
|
788
|
+
* `duplicate` now means PROVABLY NOT OURS. Since the 409 carries the stored
|
|
789
|
+
* row (ENG-415), a conflict whose row matches what we are re-sending is our
|
|
790
|
+
* OWN earlier attempt and resolves as {@link LedgerAlreadyWrittenError}
|
|
791
|
+
* instead — so what is left here is a row somebody else wrote, or a row we
|
|
792
|
+
* could not read back to check. Both fail to the LOUD side on purpose.
|
|
793
|
+
*
|
|
794
|
+
* `duplicate_foreign` is the loudest state in the lane: the id is taken by a
|
|
795
|
+
* row THIS ORG CANNOT READ. `inference_requests.inference_id` is a global
|
|
796
|
+
* primary key, so it is reachable, and it can never be our own write.
|
|
797
|
+
*/
|
|
798
|
+
readonly ledgerFailureReason: 'duplicate' | 'duplicate_foreign';
|
|
799
|
+
constructor(message: string, reason?: 'duplicate' | 'duplicate_foreign');
|
|
800
|
+
}
|
|
801
|
+
/** The control-plane write surface could not be reached or refused the row. */
|
|
802
|
+
declare class LedgerSurfaceError extends Error implements LedgerFailure {
|
|
803
|
+
readonly ledgerFailureReason: "surface";
|
|
804
|
+
constructor(message: string);
|
|
805
|
+
}
|
|
806
|
+
interface LedgerClientOptions {
|
|
807
|
+
/** The org control-plane API base (`{apiUrl}`) the function is mounted under. */
|
|
808
|
+
apiUrl: string;
|
|
809
|
+
/** Injected in tests; production uses the platform fetch. */
|
|
810
|
+
fetchImpl?: typeof fetch;
|
|
811
|
+
}
|
|
812
|
+
/**
|
|
813
|
+
* Validate the control plane's answer into the record the caller logs.
|
|
814
|
+
*
|
|
815
|
+
* IT IS VALIDATED RATHER THAN TRUSTED because what it carries is a COST. A
|
|
816
|
+
* surface that changed shape under us must not be able to produce a log line
|
|
817
|
+
* claiming a request was priced at a number nobody returned — and
|
|
818
|
+
* `unpriced_reason` is what tells an operator that traffic is being served
|
|
819
|
+
* that nobody can bill, so a missing one is not a detail.
|
|
820
|
+
*/
|
|
821
|
+
declare function parseLedgerRecorded(raw: unknown, where: string): LedgerRecorded;
|
|
822
|
+
/** One ledger row, written as the holder of the caller's own org API key. */
|
|
823
|
+
declare class LedgerClient {
|
|
824
|
+
private readonly opts;
|
|
825
|
+
private readonly fetchImpl;
|
|
826
|
+
constructor(opts: LedgerClientOptions);
|
|
827
|
+
/**
|
|
828
|
+
* Write one row. Five outcomes, all explicit:
|
|
829
|
+
* - the control plane records it → the validated {@link LedgerRecorded},
|
|
830
|
+
* priced or with an `unpriced_reason`;
|
|
831
|
+
* - the id is already there AND the stored row is OURS (409) →
|
|
832
|
+
* {@link LedgerAlreadyWrittenError}, which is a SUCCESS: an earlier
|
|
833
|
+
* attempt landed;
|
|
834
|
+
* - the id is already there and the row is NOT ours, or cannot be read
|
|
835
|
+
* back to check (409) → {@link LedgerDuplicateError};
|
|
836
|
+
* - it rejects the key (401/403) → {@link LedgerAuthError};
|
|
837
|
+
* - anything else → {@link LedgerSurfaceError}.
|
|
838
|
+
*
|
|
839
|
+
* THE CALLER RETRIES, NOT THIS METHOD (ENG-415). One call is one attempt;
|
|
840
|
+
* `proxy/ledger.ts` owns the queue, the backoff and the dwell bound, and it
|
|
841
|
+
* retries only the one reason that is safe to retry. What makes any of it
|
|
842
|
+
* safe is the fifth outcome above: a retry whose first attempt secretly
|
|
843
|
+
* landed comes back as {@link LedgerAlreadyWrittenError}, not as something
|
|
844
|
+
* indistinguishable from a forged row.
|
|
845
|
+
*/
|
|
846
|
+
write(apiKey: string, entry: LedgerWrite): Promise<LedgerRecorded>;
|
|
847
|
+
}
|
|
595
848
|
|
|
596
849
|
/**
|
|
597
850
|
* The hosted endpoint's configuration.
|
|
@@ -670,8 +923,25 @@ declare function resolveHostedConfig(env?: NodeJS.ProcessEnv): HostedConfig;
|
|
|
670
923
|
* reachable, because a replica that cannot verify API keys
|
|
671
924
|
* or list apps can serve nothing and must be pulled out of
|
|
672
925
|
* the Service rather than answering 502s.
|
|
673
|
-
* * /v1/... the compat surface, org-scoped by the Bearer key
|
|
674
|
-
*
|
|
926
|
+
* * /v1/... the compat surface, org-scoped by the Bearer key, including
|
|
927
|
+
* the canonical usage-query lane POST /v1/usage/requests
|
|
928
|
+
* (proxy/usage.ts — a batch read of the per-request
|
|
929
|
+
* inference ledger).
|
|
930
|
+
* POST /partners/<name>/... the ENG-376 partner ADAPTER surface: a
|
|
931
|
+
* partner's own wire shape translated onto a canonical lane,
|
|
932
|
+
* sharing its auth and error mapping. Today: the Hugging
|
|
933
|
+
* Face billing poll. Ingress/charts must allowlist it.
|
|
934
|
+
* WS /ws/google.ai.generativelanguage.v1beta.GenerativeService.
|
|
935
|
+
* BidiGenerateContent — the Gemini Live surface, org-scoped by the
|
|
936
|
+
* Gemini credential (x-goog-api-key / Authorization: Bearer; `?key=`
|
|
937
|
+
* refused). Ingress/charts must allowlist that path.
|
|
938
|
+
* WS /v1/realtime — the OpenAI Realtime surface (GA protocol subset),
|
|
939
|
+
* org-scoped by the SAME Bearer api key as the /v1 lane; the
|
|
940
|
+
* caller-org catalog's `task` column (stt | tts) gates which models
|
|
941
|
+
* may bind an audio session. Ingress/charts must allowlist that
|
|
942
|
+
* path too.
|
|
943
|
+
* * anything else → 404 in the OpenAI error envelope (unclaimed WS
|
|
944
|
+
* upgrades get their own final refusal).
|
|
675
945
|
*
|
|
676
946
|
* WHAT IS DELIBERATELY NOT HERE: `/stats`. The local proxy exposes it for the
|
|
677
947
|
* launcher's reuse probe; on a shared endpoint it would publish one tenant's
|
|
@@ -698,14 +968,232 @@ interface HostedProxyOptions extends HostedConfig {
|
|
|
698
968
|
resolver?: OrgResolver;
|
|
699
969
|
/** Injected in tests so backhauls are stubbed at the seam. */
|
|
700
970
|
registry?: TenantRegistry;
|
|
971
|
+
/** Injected in tests so conversation backhauls are stubbed at the seam. */
|
|
972
|
+
conversations?: ConversationBackhauls;
|
|
973
|
+
/**
|
|
974
|
+
* Injected in tests so the canonical usage lane is exercised without a
|
|
975
|
+
* control plane. Production builds one against `apiUrl` — the same base and
|
|
976
|
+
* the same caller-key credential `GET {apiUrl}/apps` rides.
|
|
977
|
+
*/
|
|
978
|
+
usageClient?: UsageClient;
|
|
979
|
+
/**
|
|
980
|
+
* Injected in tests so the per-request LEDGER WRITE is exercised without a
|
|
981
|
+
* control plane. Production builds one against `apiUrl` — the same base and
|
|
982
|
+
* the same caller-key credential the usage read and `GET {apiUrl}/apps`
|
|
983
|
+
* ride.
|
|
984
|
+
*/
|
|
985
|
+
ledgerClient?: LedgerClient;
|
|
986
|
+
/**
|
|
987
|
+
* Sink for the ledger lane's own structured diagnostics (proxy/ledger.ts:
|
|
988
|
+
* one JSON object per write, skip or failure). DELIBERATELY SEPARATE from
|
|
989
|
+
* {@link requestLog}, which carries exactly one `inference_request` record
|
|
990
|
+
* per `/v1` request and whose readers count on that. Production leaves it
|
|
991
|
+
* unset, which is this container's stdout.
|
|
992
|
+
*/
|
|
993
|
+
ledgerLog?: (line: string) => void;
|
|
994
|
+
/**
|
|
995
|
+
* Sink for the shared handler's per-request billing log line (the
|
|
996
|
+
* `Inference-Id` record — proxy/responses-turn.ts `ProxyHandlerOptions`).
|
|
997
|
+
* Injected in tests; production leaves it unset, which is this container's
|
|
998
|
+
* stdout — the log stream the platform collects.
|
|
999
|
+
*/
|
|
1000
|
+
requestLog?: (line: string) => void;
|
|
1001
|
+
/**
|
|
1002
|
+
* Collect Node process metrics (event-loop lag, heap, GC) alongside the
|
|
1003
|
+
* request metrics. Production leaves it on; tests turn it off so an
|
|
1004
|
+
* assertion on the exposition text is not swamped by process noise.
|
|
1005
|
+
*/
|
|
1006
|
+
defaultMetrics?: boolean;
|
|
701
1007
|
}
|
|
702
1008
|
/**
|
|
703
|
-
* Build (not listen) the hosted endpoint. The caller owns listen/close,
|
|
704
|
-
* `closeAll` releases every pooled backhaul through `Session.end()
|
|
1009
|
+
* Build (not listen) the hosted endpoint. The caller owns listen/close,
|
|
1010
|
+
* `closeAll` releases every pooled backhaul through `Session.end()`, and
|
|
1011
|
+
* `closeWebSockets` gracefully closes every attached WS surface — a drain
|
|
1012
|
+
* must run it BEFORE `server.close()`/`closeAllConnections()` (see
|
|
1013
|
+
* {@link drainServer}).
|
|
705
1014
|
*/
|
|
706
1015
|
declare function createHostedProxy(options: HostedProxyOptions): {
|
|
707
1016
|
server: Server;
|
|
1017
|
+
/**
|
|
1018
|
+
* The PRIVATE Prometheus listener (`proxy/metrics.ts` — GET /metrics on
|
|
1019
|
+
* {@link METRICS_PORT}). Its own server, not a route on the one above, so
|
|
1020
|
+
* "the metrics port is not public" is a property of the socket rather than
|
|
1021
|
+
* a rule the `/v1` router must keep remembering. The caller owns
|
|
1022
|
+
* listen/close, exactly as it does for `server`.
|
|
1023
|
+
*/
|
|
1024
|
+
metricsServer: Server;
|
|
1025
|
+
metrics: InferenceMetrics;
|
|
1026
|
+
/**
|
|
1027
|
+
* The PRIVATE ext-auth listener the edge calls to turn the caller's Bearer
|
|
1028
|
+
* key into a non-secret bucket id (`hosted/edge-identity.ts`). Its own
|
|
1029
|
+
* socket for the same reason the metrics port is: the public HTTPRoutes
|
|
1030
|
+
* forward only to the `/v1` port, so it cannot be reached from outside.
|
|
1031
|
+
* The caller owns listen/close.
|
|
1032
|
+
*/
|
|
1033
|
+
edgeIdentityServer: Server;
|
|
708
1034
|
closeAll: () => Promise<void>;
|
|
1035
|
+
closeWebSockets: () => Promise<void>;
|
|
709
1036
|
};
|
|
710
1037
|
|
|
711
|
-
|
|
1038
|
+
/**
|
|
1039
|
+
* THE EDGE IDENTITY SURFACE — the one thing Envoy cannot compute for itself.
|
|
1040
|
+
*
|
|
1041
|
+
* WHY IT EXISTS. Envoy Gateway's per-caller rate limiting needs a DISTINCT
|
|
1042
|
+
* bucket per caller, and the only caller identity on this endpoint is the
|
|
1043
|
+
* Bearer org API key (`hosted/auth.ts`: "the Bearer key IS the identity AND
|
|
1044
|
+
* the tenancy"). Envoy Gateway can key a global rate limit on a header's
|
|
1045
|
+
* distinct values — but the descriptor VALUE is what the rate-limit service
|
|
1046
|
+
* sends to Redis and what Redis stores as part of its key. Keying directly on
|
|
1047
|
+
* `Authorization` would therefore park live customer API keys, in plaintext,
|
|
1048
|
+
* in an in-cluster Redis whose NetworkPolicy is inert on both prod clusters
|
|
1049
|
+
* today (the VPC CNI node agent runs without `--enable-network-policy`, as the
|
|
1050
|
+
* valkey chart already documents). That is a credential store nobody designed,
|
|
1051
|
+
* audited, or rotates.
|
|
1052
|
+
*
|
|
1053
|
+
* Envoy's Lua filter has no hashing primitive and its rate-limit actions have
|
|
1054
|
+
* no transform, so SOMETHING has to turn the key into a non-secret id before
|
|
1055
|
+
* the rate-limit filter runs. Envoy Gateway's own mechanism for that is
|
|
1056
|
+
* `SecurityPolicy.extAuth`, whose response headers are merged into the request
|
|
1057
|
+
* ("coexisting headers will be overridden") before the later rate-limit filter
|
|
1058
|
+
* reads them. This module is that service.
|
|
1059
|
+
*
|
|
1060
|
+
* IT IS NOT AN AUTHORIZATION GATE, AND IT MUST NEVER BECOME ONE. It answers
|
|
1061
|
+
* 200 to everything. Authentication stays where it already is — in the proxy,
|
|
1062
|
+
* which alone can answer 401 in the lane's native error envelope with the
|
|
1063
|
+
* `Inference-Id` header HF bills on. A deny here would instead produce Envoy's
|
|
1064
|
+
* bodiless refusal, which an OpenAI SDK surfaces as an unparseable error.
|
|
1065
|
+
*
|
|
1066
|
+
* IT HAS NO DEPENDENCIES, ON PURPOSE. Read a header, hash it, answer. No
|
|
1067
|
+
* control-plane call, no cache, no I/O, nothing that can be slow or down. It
|
|
1068
|
+
* sits in the request path of `inference.urun.sh`, so the only acceptable
|
|
1069
|
+
* failure budget is none — and the `SecurityPolicy` that calls it is
|
|
1070
|
+
* additionally configured `failOpen`, so even losing it entirely degrades to
|
|
1071
|
+
* "no rate limiting", never to "no inference".
|
|
1072
|
+
*/
|
|
1073
|
+
|
|
1074
|
+
/**
|
|
1075
|
+
* The container's edge-identity port — a MODULE CONSTANT, not an env var, per
|
|
1076
|
+
* the hosted config mandate (`hosted/config.ts`). Its own socket, like the
|
|
1077
|
+
* metrics port: private by construction, since the public HTTPRoutes forward
|
|
1078
|
+
* only to the `/v1` port.
|
|
1079
|
+
*/
|
|
1080
|
+
declare const EDGE_IDENTITY_PORT = 9465;
|
|
1081
|
+
/**
|
|
1082
|
+
* THE BUCKET ID. Envoy Gateway's `BackendTrafficPolicy` keys the caller's
|
|
1083
|
+
* rate-limit counter on this header's DISTINCT values, which means the value
|
|
1084
|
+
* here IS the input the rate-limit service builds its Redis counter key from.
|
|
1085
|
+
*
|
|
1086
|
+
* THAT IS WHY IT MUST NOT BE THE VALUE COMMITTED IN `values.yaml`. The tier
|
|
1087
|
+
* fingerprint below is written into a git-tracked manifest so an operator can
|
|
1088
|
+
* give one caller its own limit. If the bucket id were the same digest, then
|
|
1089
|
+
* ANYONE WHO CAN READ THE REPO could reconstruct a tiered caller's counter key
|
|
1090
|
+
* — and the counter store is reachable by any pod in the cluster with
|
|
1091
|
+
* `+incrby` and `+expire` granted, so they could spend Hugging Face's
|
|
1092
|
+
* allowance or stretch the window and hold HF's own probe at 429. That is the
|
|
1093
|
+
* delisting event this whole pair exists to prevent, and non-invertibility of
|
|
1094
|
+
* SHA-256 does nothing about it: the attacker never needs the key, only the
|
|
1095
|
+
* digest, and the digest was published on purpose.
|
|
1096
|
+
*
|
|
1097
|
+
* So the two digests are DIFFERENT one-way functions of the same token, split
|
|
1098
|
+
* by domain separator ({@link BUCKET_DOMAIN} vs {@link TIER_DOMAIN}). Knowing
|
|
1099
|
+
* the tier fingerprint gives no path to the bucket id: getting there would
|
|
1100
|
+
* require recovering the token from its digest, which is a preimage attack on
|
|
1101
|
+
* SHA-256 over 256 bits of `randomBytes` entropy.
|
|
1102
|
+
*/
|
|
1103
|
+
declare const EDGE_KEY_ID_HEADER = "x-urun-key-id";
|
|
1104
|
+
/**
|
|
1105
|
+
* THE TIER SELECTOR — matched with `Exact` / `RegularExpression` in the
|
|
1106
|
+
* BackendTrafficPolicy to decide WHICH limit applies, never to bucket.
|
|
1107
|
+
*
|
|
1108
|
+
* This is the digest an operator commits to `rateLimit.tiers[].keyIdSha256` in
|
|
1109
|
+
* k8s-manifests, so treat it as PUBLIC: everything about the design has to
|
|
1110
|
+
* hold when an attacker knows it.
|
|
1111
|
+
*
|
|
1112
|
+
* WHY A SECOND HEADER AT ALL. An Envoy Gateway rate-limit rule needs both
|
|
1113
|
+
* kinds of match on the caller at once: `Distinct` (give this caller their own
|
|
1114
|
+
* counter) AND an exact/inverted match (is this the partner key, or everyone
|
|
1115
|
+
* else). Whether a single selector may list the SAME header name under two
|
|
1116
|
+
* different match types is not something this repo can verify without a
|
|
1117
|
+
* cluster, and getting it wrong means Envoy Gateway rejects the policy —
|
|
1118
|
+
* which, on a fail-open limiter, is indistinguishable from "rate limiting
|
|
1119
|
+
* works" until somebody floods us. Two headers remove the question; carrying
|
|
1120
|
+
* two DIFFERENT digests is what makes publishing one of them safe.
|
|
1121
|
+
*/
|
|
1122
|
+
declare const EDGE_KEY_FINGERPRINT_HEADER = "x-urun-key-fingerprint";
|
|
1123
|
+
/**
|
|
1124
|
+
* Domain separator for the BUCKET digest. Never appears in a manifest, a log,
|
|
1125
|
+
* or a metric — the bucket id is derived here and read only by Envoy.
|
|
1126
|
+
*/
|
|
1127
|
+
declare const BUCKET_DOMAIN = "urun-ratelimit-bucket:";
|
|
1128
|
+
/**
|
|
1129
|
+
* Domain separator for the TIER digest — the one an operator reproduces from a
|
|
1130
|
+
* key with `printf 'urun-ratelimit-tier:%s' "$KEY" | sha256sum | cut -d' ' -f1`.
|
|
1131
|
+
*
|
|
1132
|
+
* A plain `sha256sum "$KEY"` would be the OLD, unsafe value: identical to the
|
|
1133
|
+
* bucket id, and therefore a published counter key. The separator is what
|
|
1134
|
+
* keeps the two apart, so it is part of the operator contract and cannot drift.
|
|
1135
|
+
*/
|
|
1136
|
+
declare const TIER_DOMAIN = "urun-ratelimit-tier:";
|
|
1137
|
+
/**
|
|
1138
|
+
* The bucket every request WITHOUT a Bearer token shares. A single shared
|
|
1139
|
+
* bucket rather than no bucket at all: unauthenticated floods must be limited
|
|
1140
|
+
* too, and they have no identity to spread across. They cannot consume a real
|
|
1141
|
+
* caller's bucket because no real key hashes to this value.
|
|
1142
|
+
*/
|
|
1143
|
+
declare const ANONYMOUS_KEY_ID = "anonymous";
|
|
1144
|
+
/**
|
|
1145
|
+
* The caller's BUCKET id — `sha256(BUCKET_DOMAIN || token)`, hex.
|
|
1146
|
+
*
|
|
1147
|
+
* SECRET BY CONSTRUCTION, not because the digest is one-way but because the
|
|
1148
|
+
* only published digest is a DIFFERENT one. This value is the input Envoy's
|
|
1149
|
+
* rate-limit descriptor carries, so it is what the counter key in the
|
|
1150
|
+
* rate-limit cache is built from; publishing it would publish a writable
|
|
1151
|
+
* counter key (see {@link EDGE_KEY_ID_HEADER}).
|
|
1152
|
+
*/
|
|
1153
|
+
declare function edgeBucketId(bearerToken: string): string;
|
|
1154
|
+
/**
|
|
1155
|
+
* The caller's TIER fingerprint — `sha256(TIER_DOMAIN || token)`, hex.
|
|
1156
|
+
*
|
|
1157
|
+
* PUBLIC BY DESIGN: this is the value an operator commits to
|
|
1158
|
+
* `rateLimit.tiers[].keyIdSha256`. The runbook is
|
|
1159
|
+
* `printf 'urun-ratelimit-tier:%s' "$KEY" | sha256sum | cut -d' ' -f1`.
|
|
1160
|
+
*
|
|
1161
|
+
* The domain prefix is not decoration — a bare `sha256sum "$KEY"` reproduces
|
|
1162
|
+
* the BUCKET id, which is exactly the value that must never be committed. The
|
|
1163
|
+
* `cut` is load-bearing too: `sha256sum` prints the digest, two spaces, then
|
|
1164
|
+
* the input name, and a tier carrying that trailing ` -` matches nothing and
|
|
1165
|
+
* silently leaves the caller on the default limit.
|
|
1166
|
+
*/
|
|
1167
|
+
declare function edgeTierFingerprint(bearerToken: string): string;
|
|
1168
|
+
/**
|
|
1169
|
+
* The bearer token of an `Authorization` header, or null.
|
|
1170
|
+
*
|
|
1171
|
+
* Deliberately NOT `hosted/auth.ts` `bearerFrom`: that one THROWS a 401-shaped
|
|
1172
|
+
* error for a missing or malformed header, which is the correct behaviour for
|
|
1173
|
+
* the lane that authenticates. Here a malformed header must produce the
|
|
1174
|
+
* anonymous bucket and a 200, because refusing is the proxy's job and this
|
|
1175
|
+
* service must never refuse.
|
|
1176
|
+
*/
|
|
1177
|
+
declare function bearerTokenOf(headerValue: string | undefined): string | null;
|
|
1178
|
+
/**
|
|
1179
|
+
* The PRIVATE ext-auth listener. Answers 200 to every method and every path —
|
|
1180
|
+
* Envoy appends the ORIGINAL request path to the configured ext-auth path, so
|
|
1181
|
+
* this service sees `/ratelimit-identity/v1/chat/completions` and friends and
|
|
1182
|
+
* must not care.
|
|
1183
|
+
*
|
|
1184
|
+
* The response carries exactly the two identity headers below, and the
|
|
1185
|
+
* `SecurityPolicy` forwards exactly those two (`headersToBackend`). Nothing
|
|
1186
|
+
* about the caller's request, and certainly not their key, is echoed.
|
|
1187
|
+
*
|
|
1188
|
+
* A CALLER CANNOT PRESENT THEIR OWN. `headersToBackend` is documented as
|
|
1189
|
+
* overriding coexisting headers, so whatever a client sent under these names is
|
|
1190
|
+
* replaced on every request this service answers. There is deliberately NO
|
|
1191
|
+
* `ClientTrafficPolicy` stripping them at the listener: that policy would
|
|
1192
|
+
* attach to the shared `public-gw` (api.urun.sh rides it too) to close a gap
|
|
1193
|
+
* that grants nothing — in the only case where a client value survives, the
|
|
1194
|
+
* ext-auth hop has already failed open, and fail-open grants MORE than any
|
|
1195
|
+
* spoofed bucket could. The k8s half records the same decision.
|
|
1196
|
+
*/
|
|
1197
|
+
declare function createEdgeIdentityServer(): Server;
|
|
1198
|
+
|
|
1199
|
+
export { ANONYMOUS_KEY_ID, BIND_HOST, BUCKET_DOMAIN, BoundedLabel, type CallerIdentity, ControlPlaneUnavailableError, type ConversationBackhaulOptions, ConversationBackhauls, ConversationCapacityError, DEFAULT_PORT, EDGE_IDENTITY_PORT, EDGE_KEY_FINGERPRINT_HEADER, EDGE_KEY_ID_HEADER, type HostedConfig, type HostedProxyOptions, InferenceMetrics, LEDGER_FUNCTION_PATH, LEDGER_WRITE_TIMEOUT_MS, LedgerClient, type LedgerClientOptions, LedgerDuplicateError, LedgerSurfaceError, MAX_CACHED_KEYS, MAX_CONVERSATIONS, MAX_CONVERSATION_HANDLES, MAX_MODEL_LABELS, MAX_ORG_LABELS, MAX_TENANTS, METRICS_PATH, METRICS_PORT, NO_MODEL_LABEL, ORG_BINDING_TTL_MS, OTHER_LABEL, OrgResolver, type OrgResolverOptions, type PrivateConversationBackhaul, type PrivateConversationConnection, ProxyAuthError, READINESS_TIMEOUT_MS, READINESS_TTL_MS, SERVE_FUNCTION, TIER_DOMAIN, TenantRegistry, type TenantRegistryOptions, bearerFrom, bearerTokenOf, createEdgeIdentityServer, createHostedProxy, createMetricsServer, edgeBucketId, edgeTierFingerprint, laneOf, parseLedgerRecorded, resolveHostedConfig, statusClassOf, tenantSubject };
|