@leaf233/dsh-llm-rate-limiter 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +211 -1
- package/README.md +80 -17
- package/README.zh.md +284 -0
- package/lib/client.js +182 -140
- package/lib/index.js +160 -54
- package/lib/status-rpc.js +227 -9
- package/lib/types/config.js +48 -16
- package/lib/types/index.js +1 -1
- package/package.json +13 -11
- package/COMPATIBILITY.md +0 -150
package/lib/index.js
CHANGED
|
@@ -2,28 +2,36 @@
|
|
|
2
2
|
* dsh-llm-rate-limiter — Host (server-side) entry point.
|
|
3
3
|
*
|
|
4
4
|
* Intercepts the `llm/stream` waterfall and enforces per-model rate limits.
|
|
5
|
-
*
|
|
6
|
-
*
|
|
5
|
+
*
|
|
6
|
+
* Configuration is the profile ENTRY's own `config` (this module's exported
|
|
7
|
+
* {@link Config} schema), read through the volatile references Cordis builds
|
|
8
|
+
* for it. `dsh-settings` projects the volatile fields of every active entry
|
|
9
|
+
* into a form the browser can edit, so the plugin ships no settings service
|
|
10
|
+
* of its own. (DSH 0.1.5 had a `ctx.settings.register()` namespace API; 0.1.7
|
|
11
|
+
* removed it, and its browser half dropped `settingsScope` for `configForms`.
|
|
12
|
+
* The migration record lives in CHANGELOG.md under `[0.3.0]`.)
|
|
7
13
|
*
|
|
8
14
|
* The `llm/stream` waterfall signature is:
|
|
9
15
|
* (options: GenerateOptions, next: () => AsyncIterable<StreamChunk>) => AsyncIterable<StreamChunk>
|
|
10
16
|
*
|
|
11
17
|
* where `options` always has `.provider` (string) and `.model` (string).
|
|
12
18
|
*
|
|
13
|
-
* The settings service is injected dynamically (like dshmarket's
|
|
14
|
-
* installMarketSettings): `ctx.inject(['settings'], (scopedCtx) => ...)`, so a
|
|
15
|
-
* host without a settings provider simply runs without the GUI wiring — the
|
|
16
|
-
* composed entry stays as configured. Inherited from the DSH house pattern.
|
|
17
|
-
*
|
|
18
19
|
* @module dsh-llm-rate-limiter
|
|
19
20
|
*/
|
|
20
21
|
|
|
21
22
|
import { TokenBucketStrategy } from "./strategies/token-bucket.js";
|
|
22
23
|
import { SlidingWindowStrategy } from "./strategies/sliding-window.js";
|
|
23
|
-
import {
|
|
24
|
-
import { CHANNEL, createStatusChannel } from "./status-rpc.js";
|
|
24
|
+
import { Config } from "./types/config.js";
|
|
25
|
+
import { CHANNEL, createChannelRoute, createStatusChannel } from "./status-rpc.js";
|
|
25
26
|
|
|
26
27
|
const name = "llm-rate-limiter";
|
|
28
|
+
/**
|
|
29
|
+
* Settings namespace / profile entry id.
|
|
30
|
+
*
|
|
31
|
+
* `dsh-settings` keys every form by `entry.options.id` (`lib/index.js:432`),
|
|
32
|
+
* which is the `id` this plugin's `cordis.patch.yml` row declares. Renaming
|
|
33
|
+
* that row id therefore renames this namespace too — keep the two in step.
|
|
34
|
+
*/
|
|
27
35
|
const SETTINGS_NS = "llm-rate-limiter";
|
|
28
36
|
|
|
29
37
|
/** Maximum retained events in the status ring (bounded memory). */
|
|
@@ -33,9 +41,45 @@ const SNAPSHOT_EVENT_COUNT = 8;
|
|
|
33
41
|
|
|
34
42
|
/* ── Helpers ─────────────────────────────────────────────────────────── */
|
|
35
43
|
|
|
44
|
+
/** Built-in defaults, used when a reference yields nothing (or no config at all). */
|
|
45
|
+
const FALLBACK = {
|
|
46
|
+
enabled: true,
|
|
47
|
+
strategy: "token-bucket",
|
|
48
|
+
defaults: { maxConcurrent: 5, maxRpm: 60, burstSize: 10, refillRate: 1 },
|
|
49
|
+
models: {},
|
|
50
|
+
onThrottled: "queue",
|
|
51
|
+
maxQueueWaitMs: 60_000,
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Read one volatile reference from the parsed Config.
|
|
56
|
+
*
|
|
57
|
+
* The loader mutates volatile references IN PLACE when only raw config values
|
|
58
|
+
* changed (`cordis-plugin-loader/lib/index.js:393` `_commitVolatile()`), so
|
|
59
|
+
* the reference object stays valid for the plugin's whole lifetime and its
|
|
60
|
+
* `get()` always yields the current value. Call this at each use rather than
|
|
61
|
+
* caching the unwrapped value.
|
|
62
|
+
*
|
|
63
|
+
* @param ref - a `.volatile()` field of the parsed Config, or undefined when
|
|
64
|
+
* the entry was mounted with no config at all.
|
|
65
|
+
* @param fallback - value used when the reference is absent.
|
|
66
|
+
* @returns the current plain value.
|
|
67
|
+
*/
|
|
68
|
+
function read(ref, fallback) {
|
|
69
|
+
if (ref === undefined || ref === null) return fallback;
|
|
70
|
+
if (typeof ref.get === "function") {
|
|
71
|
+
const value = ref.get();
|
|
72
|
+
return value === undefined || value === null ? fallback : value;
|
|
73
|
+
}
|
|
74
|
+
return ref;
|
|
75
|
+
}
|
|
76
|
+
|
|
36
77
|
/**
|
|
37
78
|
* Merge defaults + per-model overrides into a flat config object that both
|
|
38
79
|
* strategy constructors understand.
|
|
80
|
+
*
|
|
81
|
+
* @param cfg - plain config object (`{ enabled, strategy, defaults, models, ... }`).
|
|
82
|
+
* @param key - `"provider/model"` of the call being limited.
|
|
39
83
|
*/
|
|
40
84
|
function resolveModelConfig(cfg, key) {
|
|
41
85
|
const o = cfg.models?.[key] ?? {};
|
|
@@ -92,22 +136,26 @@ function terminalChunk(modelKey, signal) {
|
|
|
92
136
|
|
|
93
137
|
/**
|
|
94
138
|
* @param {import("@deepseek-ai/cordis").Context} ctx
|
|
139
|
+
* @param {object} [config] - parsed Config; volatile fields are live references.
|
|
95
140
|
*/
|
|
96
|
-
function apply(ctx) {
|
|
141
|
+
function apply(ctx, config) {
|
|
97
142
|
const logger = ctx.logger?.("rate-limiter");
|
|
98
143
|
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
144
|
+
/**
|
|
145
|
+
* Snapshot the whole config as one plain object, reading each reference at
|
|
146
|
+
* call time. Derived settings are not cached: a volatile-only change updates
|
|
147
|
+
* the references in place without re-running `apply`.
|
|
148
|
+
*/
|
|
149
|
+
function readConfig() {
|
|
150
|
+
return {
|
|
151
|
+
enabled: read(config?.enabled, FALLBACK.enabled),
|
|
152
|
+
strategy: read(config?.strategy, FALLBACK.strategy),
|
|
153
|
+
defaults: read(config?.defaults, FALLBACK.defaults),
|
|
154
|
+
models: read(config?.models, FALLBACK.models),
|
|
155
|
+
onThrottled: read(config?.onThrottled, FALLBACK.onThrottled),
|
|
156
|
+
maxQueueWaitMs: read(config?.maxQueueWaitMs, FALLBACK.maxQueueWaitMs),
|
|
157
|
+
};
|
|
158
|
+
}
|
|
111
159
|
|
|
112
160
|
/** @type {Map<string, TokenBucketStrategy | SlidingWindowStrategy>} */
|
|
113
161
|
const limiters = new Map();
|
|
@@ -135,6 +183,7 @@ function apply(ctx) {
|
|
|
135
183
|
|
|
136
184
|
/** Build the JSON-safe snapshot served on the "snapshot" endpoint. */
|
|
137
185
|
function buildSnapshot() {
|
|
186
|
+
const cfg = readConfig();
|
|
138
187
|
const models = {};
|
|
139
188
|
for (const [key, limiter] of limiters) {
|
|
140
189
|
try {
|
|
@@ -162,57 +211,114 @@ function apply(ctx) {
|
|
|
162
211
|
rev += 1;
|
|
163
212
|
}
|
|
164
213
|
|
|
165
|
-
// ── Settings
|
|
166
|
-
//
|
|
167
|
-
//
|
|
214
|
+
// ── Settings presentation (dynamic inject) ─────────────────
|
|
215
|
+
// This plugin ships its own browser page, so it opts out of any
|
|
216
|
+
// schema-generated page. Wrapped in ctx.inject so a host WITHOUT a settings
|
|
217
|
+
// service still runs the limiter — the callback simply never runs. The
|
|
218
|
+
// optional child ctx names the plugin fiber the policy belongs to, which is
|
|
219
|
+
// why `ctx.fiber` is passed explicitly (`dsh-settings/lib/index.js:370`).
|
|
168
220
|
ctx.inject(["settings"], (scopedCtx) => {
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
// Replace cfg with live settings value on every change.
|
|
176
|
-
cfg = scope.get();
|
|
177
|
-
scope.watch((c) => { cfg = c; });
|
|
178
|
-
|
|
179
|
-
// Dispose limiters on settings teardown.
|
|
180
|
-
scopedCtx.effect(() => () => {
|
|
181
|
-
logger?.info("rate limiter settings scope tearing down (%s)", SETTINGS_NS);
|
|
182
|
-
for (const limiter of limiters.values()) limiter.dispose();
|
|
183
|
-
limiters.clear();
|
|
184
|
-
}, "rate-limiter: settings scope teardown");
|
|
221
|
+
if (typeof scopedCtx.settings?.configure !== "function") return; // older/foreign settings seam: keep running
|
|
222
|
+
scopedCtx.effect(
|
|
223
|
+
() => scopedCtx.settings.configure({ auto: false }, ctx.fiber),
|
|
224
|
+
"rate-limiter: settings presentation",
|
|
225
|
+
);
|
|
185
226
|
});
|
|
186
227
|
|
|
187
|
-
// ── Status
|
|
228
|
+
// ── Status channel (dynamic inject) ────────────────────────
|
|
188
229
|
// Same defensive shape as the settings wiring: on a host without a
|
|
189
230
|
// connection service the callback never runs and the plugin stays silent.
|
|
190
|
-
//
|
|
231
|
+
// Either path below registers inside an effect owned by this fiber, so
|
|
232
|
+
// unloading the plugin withdraws the channel automatically.
|
|
233
|
+
//
|
|
234
|
+
// Two mounting paths exist, and both speak the same wire protocol, so the
|
|
235
|
+
// browser half never needs to know which one is live:
|
|
236
|
+
//
|
|
237
|
+
// 1. connection.rpc.handle(channel, handler) — the framework's generic
|
|
238
|
+
// channel registry (preferred: the framework owns the route, the
|
|
239
|
+
// request validation, and the withdrawal).
|
|
240
|
+
//
|
|
241
|
+
// 2. a self-registered prefix route reusing connection.requestRejection()
|
|
242
|
+
// — the fallback for hosts where path 1 is broken. DSH 0.1.5-rc.3
|
|
243
|
+
// changed dsh-client-connection's own `inject` from
|
|
244
|
+
// ["webServer", "credentials"] to ["credentials"] while its
|
|
245
|
+
// HostConnectionService.register() still dereferences
|
|
246
|
+
// `owner.webServer`, and Cordis rebinds a cross-fiber service's `ctx`
|
|
247
|
+
// to the *reader's* fiber — so rpc.handle() throws
|
|
248
|
+
// 'cannot get property "webServer" without inject'. Registering the
|
|
249
|
+
// route on our own fiber (which injects webServer below) avoids that
|
|
250
|
+
// while still using the framework's own 403/401 fence.
|
|
251
|
+
// Still unfixed in 0.1.7-rc.1 (dsh-client-connection/lib/index.js:798
|
|
252
|
+
// keeps `inject = ["credentials"]`), so the fallback still fires.
|
|
191
253
|
ctx.inject(["connection"], (scopedCtx) => {
|
|
192
254
|
const connection = scopedCtx.get("connection");
|
|
255
|
+
const channel = createStatusChannel({ buildSnapshot, resetTotals });
|
|
256
|
+
|
|
257
|
+
// ── Path 1: the framework's generic channel registry ──
|
|
193
258
|
const handle = typeof connection?.rpc?.handle === "function"
|
|
194
259
|
? connection.rpc.handle.bind(connection.rpc)
|
|
195
260
|
: undefined;
|
|
196
|
-
if (handle === undefined) return; // older host: degrade silently
|
|
197
261
|
|
|
262
|
+
if (handle !== undefined) {
|
|
263
|
+
try {
|
|
264
|
+
scopedCtx.effect(() => {
|
|
265
|
+
const unregister = handle(CHANNEL, channel);
|
|
266
|
+
return () => { unregister(); };
|
|
267
|
+
}, "rate-limiter: status rpc channel");
|
|
268
|
+
logger?.info("rate limiter status channel mounted via connection.rpc (%s)", CHANNEL);
|
|
269
|
+
return;
|
|
270
|
+
} catch (err) {
|
|
271
|
+
logger?.warn(
|
|
272
|
+
"connection.rpc channel registration failed (%s); falling back to a self-registered route",
|
|
273
|
+
err instanceof Error ? err.message : String(err),
|
|
274
|
+
);
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// ── Path 2: self-registered route (needs webServer on THIS fiber) ──
|
|
279
|
+
// The fence is mandatory: without connection.requestRejection() we would
|
|
280
|
+
// be publishing an unauthenticated route, so we refuse to mount at all.
|
|
281
|
+
// Probing once here turns a broken fence into a mount-time refusal instead
|
|
282
|
+
// of a per-request surprise.
|
|
198
283
|
try {
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
return
|
|
202
|
-
}
|
|
284
|
+
if (typeof connection?.requestRejection !== "function") {
|
|
285
|
+
logger?.warn("connection.requestRejection unavailable; status channel not mounted");
|
|
286
|
+
return;
|
|
287
|
+
}
|
|
288
|
+
connection.requestRejection({ headers: {} });
|
|
203
289
|
} catch (err) {
|
|
204
|
-
logger?.warn(
|
|
290
|
+
logger?.warn(
|
|
291
|
+
"connection.requestRejection unusable (%s); refusing to mount an unfenced status route",
|
|
292
|
+
err instanceof Error ? err.message : String(err),
|
|
293
|
+
);
|
|
205
294
|
return;
|
|
206
295
|
}
|
|
207
296
|
|
|
208
|
-
|
|
297
|
+
scopedCtx.inject(["webServer"], (webCtx) => {
|
|
298
|
+
const webServer = webCtx.get("webServer");
|
|
299
|
+
if (typeof webServer?.register !== "function") return; // no web carrier: degrade silently
|
|
300
|
+
|
|
301
|
+
try {
|
|
302
|
+
webCtx.effect(() => webServer.register(createChannelRoute({
|
|
303
|
+
channel: CHANNEL,
|
|
304
|
+
handler: channel,
|
|
305
|
+
reject: (req) => connection.requestRejection(req),
|
|
306
|
+
})), "rate-limiter: status route (self-registered)");
|
|
307
|
+
} catch (err) {
|
|
308
|
+
logger?.warn("status channel registration failed: %s", err instanceof Error ? err.message : String(err));
|
|
309
|
+
return;
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
logger?.info("rate limiter status channel mounted via self-registered route (%s)", CHANNEL);
|
|
313
|
+
});
|
|
209
314
|
});
|
|
210
315
|
|
|
211
316
|
// ── LLM stream interceptor ─────────────────────────────────
|
|
212
|
-
//
|
|
213
|
-
//
|
|
214
|
-
// with defaults; replaced by the settings watcher when available).
|
|
317
|
+
// Reads the live Config on every call, so a settings edit that reaches the
|
|
318
|
+
// loader takes effect on the next request without a restart.
|
|
215
319
|
const disposeLlmListener = ctx.on("llm/stream", async function* rateLimitInterceptor(options, next) {
|
|
320
|
+
const cfg = readConfig();
|
|
321
|
+
|
|
216
322
|
if (!cfg.enabled) {
|
|
217
323
|
yield* next();
|
|
218
324
|
return;
|
|
@@ -297,4 +403,4 @@ function apply(ctx) {
|
|
|
297
403
|
logger?.info("rate limiter active — llm/stream interceptor registered");
|
|
298
404
|
}
|
|
299
405
|
|
|
300
|
-
export { apply, name, SETTINGS_NS };
|
|
406
|
+
export { apply, name, SETTINGS_NS, Config };
|
package/lib/status-rpc.js
CHANGED
|
@@ -2,15 +2,27 @@
|
|
|
2
2
|
* dsh-llm-rate-limiter — status RPC channel (Host half).
|
|
3
3
|
*
|
|
4
4
|
* Exposes live rate-limiter statistics to the browser over the framework's
|
|
5
|
-
* generic Connection RPC channel
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* endpoints and shapes envelopes.
|
|
5
|
+
* generic Connection RPC channel. The framework supplies POST+JSON transport,
|
|
6
|
+
* the Host/Origin fence (403) and browser authentication (401); this module
|
|
7
|
+
* only answers endpoints and shapes envelopes.
|
|
9
8
|
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
9
|
+
* Two mounting paths exist, and both speak the *same* wire protocol so the
|
|
10
|
+
* browser half never needs to know which one is active:
|
|
11
|
+
*
|
|
12
|
+
* 1. `connection.rpc.handle(channel, handler)` — the framework's own generic
|
|
13
|
+
* channel registry. Preferred: the framework owns the route, the request
|
|
14
|
+
* validation, and the fiber-scoped withdrawal.
|
|
15
|
+
*
|
|
16
|
+
* 2. `createChannelRoute({ channel, handler, reject })` — a self-registered
|
|
17
|
+
* prefix route for hosts where path 1 is broken. DSH 0.1.5-rc.3 changed
|
|
18
|
+
* `@deepseek-ai/dsh-client-connection`'s own `inject` from
|
|
19
|
+
* `["webServer", "credentials"]` to `["credentials"]` while
|
|
20
|
+
* `HostConnectionService.register()` still dereferences
|
|
21
|
+
* `owner.webServer`, so `rpc.handle` throws
|
|
22
|
+
* `cannot get property "webServer" without inject`. This route is
|
|
23
|
+
* registered on our own fiber (which *can* see `webServer`) and reuses the
|
|
24
|
+
* connection service's public `requestRejection()` fence, so the 403/401
|
|
25
|
+
* policy is still the framework's.
|
|
14
26
|
*
|
|
15
27
|
* Envelope contract (mirrors the Connection wire schema):
|
|
16
28
|
* success: { ok: true, value: <JSON-safe> }
|
|
@@ -25,6 +37,27 @@
|
|
|
25
37
|
*/
|
|
26
38
|
const CHANNEL = "/llm-rate-limiter";
|
|
27
39
|
|
|
40
|
+
/**
|
|
41
|
+
* Endpoint grammar for one path segment, copied verbatim from the framework's
|
|
42
|
+
* `ENDPOINT_SEGMENT_PATTERN` in dsh-client-connection. Keeping it identical is
|
|
43
|
+
* what makes the self-registered route accept and reject exactly the same URLs
|
|
44
|
+
* as `rpc.handle` would.
|
|
45
|
+
*/
|
|
46
|
+
const ENDPOINT_SEGMENT_PATTERN = /^[A-Za-z0-9_$.-]+$/;
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Buffered-body cap for the self-registered route. Both endpoints are called
|
|
50
|
+
* with an empty payload, so this is a defensive memory bound rather than a
|
|
51
|
+
* real limit (the framework uses its own, much larger, carrier cap).
|
|
52
|
+
*/
|
|
53
|
+
const MAX_REQUEST_BODY_BYTES = 1024 * 1024;
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Correlation id echoed when the request envelope cannot be parsed at all —
|
|
57
|
+
* the same literal the framework uses (`INVALID_REQUEST_RPC_ID`).
|
|
58
|
+
*/
|
|
59
|
+
const INVALID_REQUEST_RPC_ID = "invalid-request";
|
|
60
|
+
|
|
28
61
|
/**
|
|
29
62
|
* Build a failure envelope. Field-for-field the same shape the Connection
|
|
30
63
|
* host accepts, so callers can rely on `ok` discrimination alone.
|
|
@@ -64,4 +97,189 @@ function createStatusChannel({ buildSnapshot, resetTotals }) {
|
|
|
64
97
|
};
|
|
65
98
|
}
|
|
66
99
|
|
|
67
|
-
|
|
100
|
+
/* ── self-registered route (the DSH 0.1.5-rc.3 fallback) ───────────────── */
|
|
101
|
+
|
|
102
|
+
/** Narrow an unknown JSON value to a plain object. */
|
|
103
|
+
function isRecord(value) {
|
|
104
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Extract the endpoint from a channel-relative pathname, mirroring the
|
|
109
|
+
* framework's `endpointFromPath`: a single `channel/<endpoint>` segment whose
|
|
110
|
+
* segments all satisfy {@link ENDPOINT_SEGMENT_PATTERN}.
|
|
111
|
+
*
|
|
112
|
+
* @param {string} channel - the channel prefix (e.g. `/llm-rate-limiter`).
|
|
113
|
+
* @param {string} pathname - request pathname.
|
|
114
|
+
* @returns {string | undefined} the endpoint, or undefined when the URL is not
|
|
115
|
+
* a well-formed member of this channel.
|
|
116
|
+
*/
|
|
117
|
+
function endpointFromPath(channel, pathname) {
|
|
118
|
+
if (!pathname.startsWith(`${channel}/`)) return undefined;
|
|
119
|
+
const endpoint = pathname.slice(channel.length + 1);
|
|
120
|
+
const segments = endpoint.split("/");
|
|
121
|
+
if (segments.some((segment) => segment === "" || segment === "." || segment === ".." || !ENDPOINT_SEGMENT_PATTERN.test(segment))) return undefined;
|
|
122
|
+
return endpoint;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Read and decode the request body, refusing anything above `limit` bytes so a
|
|
127
|
+
* hostile caller cannot make the host buffer without bound.
|
|
128
|
+
*
|
|
129
|
+
* @param {import("node:http").IncomingMessage} req - node request stream.
|
|
130
|
+
* @param {number} limit - maximum accepted byte count.
|
|
131
|
+
* @returns {Promise<string>} the decoded UTF-8 body.
|
|
132
|
+
* @throws {Error & { code: "TOO_LARGE" }} when the body exceeds `limit`.
|
|
133
|
+
*/
|
|
134
|
+
async function readBody(req, limit) {
|
|
135
|
+
const chunks = [];
|
|
136
|
+
let received = 0;
|
|
137
|
+
for await (const chunk of req) {
|
|
138
|
+
const buffer = typeof chunk === "string" ? Buffer.from(chunk) : chunk;
|
|
139
|
+
received += buffer.byteLength;
|
|
140
|
+
if (received > limit) {
|
|
141
|
+
const error = new Error("request body too large");
|
|
142
|
+
error.code = "TOO_LARGE";
|
|
143
|
+
throw error;
|
|
144
|
+
}
|
|
145
|
+
chunks.push(buffer);
|
|
146
|
+
}
|
|
147
|
+
return Buffer.concat(chunks).toString("utf8");
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Write one `server-response` envelope as JSON — byte-for-byte the shape
|
|
152
|
+
* `Response.json(...)` produces on the framework's own route.
|
|
153
|
+
*
|
|
154
|
+
* @param {import("node:http").ServerResponse} res - response to own.
|
|
155
|
+
* @param {string} rpcId - correlation id to echo.
|
|
156
|
+
* @param {object} result - the `{ ok, value }` / `{ ok, error }` envelope.
|
|
157
|
+
*/
|
|
158
|
+
function sendEnvelope(res, rpcId, result) {
|
|
159
|
+
res.writeHead(200, { "content-type": "application/json" });
|
|
160
|
+
res.end(JSON.stringify({ type: "server-response", rpcId, result }));
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Build a webserver prefix route that speaks the Connection RPC wire protocol.
|
|
165
|
+
*
|
|
166
|
+
* This is the fallback carrier for {@link createStatusChannel} on hosts where
|
|
167
|
+
* `connection.rpc.handle` throws. Every request-shape decision mirrors the
|
|
168
|
+
* framework's `rpcFetchHandler` so the browser half is byte-compatible with the
|
|
169
|
+
* preferred path:
|
|
170
|
+
*
|
|
171
|
+
* | condition | response |
|
|
172
|
+
* |---|---|
|
|
173
|
+
* | fence/auth rejected by `reject` | `403` (or `401`) + `forbidden`/`unauthorized` |
|
|
174
|
+
* | non-POST, or URL outside the channel | `404 not found` |
|
|
175
|
+
* | `content-type` not `application/json` | `415` |
|
|
176
|
+
* | body over {@link MAX_REQUEST_BODY_BYTES} | `413` |
|
|
177
|
+
* | body not JSON | `400 body is not JSON` |
|
|
178
|
+
* | envelope not a `client-request` | `gateway/bad-request` envelope |
|
|
179
|
+
* | `method` ≠ endpoint | `gateway/bad-request` envelope |
|
|
180
|
+
* | handler threw | `500 handler failure: ...` |
|
|
181
|
+
* | otherwise | `200` + `server-response` envelope |
|
|
182
|
+
*
|
|
183
|
+
* @param {object} deps
|
|
184
|
+
* @param {string} deps.channel - channel prefix to own (e.g. `/llm-rate-limiter`).
|
|
185
|
+
* @param {(endpoint: string, payload: unknown, signal?: AbortSignal) => Promise<object>} deps.handler - endpoint dispatcher.
|
|
186
|
+
* @param {(req: import("node:http").IncomingMessage) => number | undefined} deps.reject - the framework's `connection.requestRejection` fence; returns an HTTP status to refuse with, or undefined to allow.
|
|
187
|
+
* @returns {{ kind: "prefix", path: string, handler: (req: import("node:http").IncomingMessage, res: import("node:http").ServerResponse) => Promise<void> }}
|
|
188
|
+
*/
|
|
189
|
+
function createChannelRoute({ channel, handler, reject }) {
|
|
190
|
+
return {
|
|
191
|
+
kind: "prefix",
|
|
192
|
+
path: channel,
|
|
193
|
+
handler: async (req, res) => {
|
|
194
|
+
// 1. The framework's own fence: Host/Origin trust (403) then browser
|
|
195
|
+
// authentication (401). Reusing it keeps the security policy identical.
|
|
196
|
+
const rejection = reject(req);
|
|
197
|
+
if (rejection !== undefined) {
|
|
198
|
+
res.writeHead(rejection);
|
|
199
|
+
res.end(rejection === 401 ? "unauthorized" : "forbidden");
|
|
200
|
+
return;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
// 2. POST-only, and only inside this channel (framework: 404 otherwise).
|
|
204
|
+
const url = new URL(req.url ?? "/", "http://dsh.internal");
|
|
205
|
+
const endpoint = endpointFromPath(channel, url.pathname);
|
|
206
|
+
if (req.method !== "POST" || endpoint === undefined) {
|
|
207
|
+
res.writeHead(404);
|
|
208
|
+
res.end("not found");
|
|
209
|
+
return;
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
// 3. JSON only (framework: 415).
|
|
213
|
+
const mediaType = String(req.headers["content-type"] ?? "").split(";", 1)[0].trim().toLowerCase();
|
|
214
|
+
if (mediaType !== "application/json") {
|
|
215
|
+
res.writeHead(415);
|
|
216
|
+
res.end("content type must be application/json");
|
|
217
|
+
return;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// 4. Buffer the body under a defensive cap, then parse it.
|
|
221
|
+
let raw;
|
|
222
|
+
try {
|
|
223
|
+
raw = await readBody(req, MAX_REQUEST_BODY_BYTES);
|
|
224
|
+
} catch (error) {
|
|
225
|
+
if (error?.code === "TOO_LARGE") {
|
|
226
|
+
res.writeHead(413, { connection: "close" });
|
|
227
|
+
res.end();
|
|
228
|
+
req.destroy?.();
|
|
229
|
+
return;
|
|
230
|
+
}
|
|
231
|
+
throw error;
|
|
232
|
+
}
|
|
233
|
+
let body;
|
|
234
|
+
try {
|
|
235
|
+
body = JSON.parse(raw);
|
|
236
|
+
} catch {
|
|
237
|
+
res.writeHead(400);
|
|
238
|
+
res.end("body is not JSON");
|
|
239
|
+
return;
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
// 5. Validate the client-request envelope; echo the correlation id
|
|
243
|
+
// whenever it is a string, exactly like the framework does.
|
|
244
|
+
const rpcId = typeof body?.rpcId === "string" ? body.rpcId : INVALID_REQUEST_RPC_ID;
|
|
245
|
+
if (!isRecord(body) || body.type !== "client-request" || typeof body.rpcId !== "string" || typeof body.method !== "string") {
|
|
246
|
+
sendEnvelope(res, rpcId, failure("gateway/bad-request", "invalid client-request message", { issues: [] }));
|
|
247
|
+
return;
|
|
248
|
+
}
|
|
249
|
+
if (body.method !== endpoint) {
|
|
250
|
+
sendEnvelope(res, rpcId, failure(
|
|
251
|
+
"gateway/bad-request",
|
|
252
|
+
`method ${JSON.stringify(body.method)} does not match endpoint ${JSON.stringify(endpoint)}`,
|
|
253
|
+
{ issues: [] },
|
|
254
|
+
));
|
|
255
|
+
return;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
// 6. Dispatch. Tie the signal to the response lifetime, as the framework's
|
|
259
|
+
// http bridge does, so a dropped client can cancel the work.
|
|
260
|
+
const abort = new AbortController();
|
|
261
|
+
res.on?.("close", () => {
|
|
262
|
+
if (!res.writableEnded) abort.abort();
|
|
263
|
+
});
|
|
264
|
+
let result;
|
|
265
|
+
try {
|
|
266
|
+
result = await handler(endpoint, body.payload, abort.signal);
|
|
267
|
+
} catch (error) {
|
|
268
|
+
res.writeHead(500);
|
|
269
|
+
res.end(`handler failure: ${String(error)}`);
|
|
270
|
+
return;
|
|
271
|
+
}
|
|
272
|
+
sendEnvelope(res, rpcId, result);
|
|
273
|
+
},
|
|
274
|
+
};
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
export {
|
|
278
|
+
CHANNEL,
|
|
279
|
+
MAX_REQUEST_BODY_BYTES,
|
|
280
|
+
createChannelRoute,
|
|
281
|
+
createStatusChannel,
|
|
282
|
+
endpointFromPath,
|
|
283
|
+
failure,
|
|
284
|
+
success,
|
|
285
|
+
};
|
package/lib/types/config.js
CHANGED
|
@@ -1,8 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Rate limiter configuration schema.
|
|
3
3
|
*
|
|
4
|
-
* The
|
|
5
|
-
*
|
|
4
|
+
* The plugin's Config is the profile entry's own schema: Cordis validates the
|
|
5
|
+
* entry's `config` against it, so the host reads values from the references
|
|
6
|
+
* this schema produces rather than from a settings namespace.
|
|
7
|
+
*
|
|
8
|
+
* Fields marked `.volatile()` become separately addressable live references
|
|
9
|
+
* (`config.enabled.get()`). Only volatile fields appear in the settings form
|
|
10
|
+
* (`dsh-settings`'s `volatileForm()`), and an entry whose Config has NO
|
|
11
|
+
* volatile field is dropped from the form list entirely — so the set of
|
|
12
|
+
* `.volatile()` marks below IS the settings UI's field list.
|
|
6
13
|
*
|
|
7
14
|
* @module dsh-llm-rate-limiter/types/config
|
|
8
15
|
*/
|
|
@@ -12,6 +19,12 @@ import Schema from "@deepseek-ai/schemastery";
|
|
|
12
19
|
/**
|
|
13
20
|
* Per-model overrides — key is `"provider/model"` (e.g. `"deepseek/deepseek-chat"`).
|
|
14
21
|
* Every field is optional; omitted fields inherit from `defaults`.
|
|
22
|
+
*
|
|
23
|
+
* This sub-schema deliberately carries NO `.volatile()`: it is the `inner`
|
|
24
|
+
* of a `dict`, and two volatile layers on one path are rejected
|
|
25
|
+
* (`volatile fields require a fixed object path without an enclosing volatile
|
|
26
|
+
* field`, `schemastery/lib/index.mjs:246`). The whole `models` dict is made
|
|
27
|
+
* volatile instead — see below.
|
|
15
28
|
*/
|
|
16
29
|
const ModelOverride = Schema.object({
|
|
17
30
|
maxConcurrent: Schema.number().step(1).min(1).description("Max concurrent requests for this model"),
|
|
@@ -21,33 +34,52 @@ const ModelOverride = Schema.object({
|
|
|
21
34
|
enabled: Schema.boolean().description("Enable/disable rate limiting for this model"),
|
|
22
35
|
});
|
|
23
36
|
|
|
24
|
-
|
|
37
|
+
/**
|
|
38
|
+
* Per-model defaults — applied to every model unless overridden.
|
|
39
|
+
* The object is volatile AS A WHOLE so `config.defaults.get()` yields one
|
|
40
|
+
* plain object and the settings form edits `["defaults", field]` paths.
|
|
41
|
+
*/
|
|
42
|
+
const Defaults = Schema.object({
|
|
43
|
+
maxConcurrent: Schema.number().step(1).min(1).default(5).description("Max concurrent requests"),
|
|
44
|
+
maxRpm: Schema.number().step(1).min(1).default(60).description("Max requests per minute"),
|
|
45
|
+
burstSize: Schema.number().step(1).min(1).default(10).description("Burst capacity (token-bucket)"),
|
|
46
|
+
refillRate: Schema.number().min(0.1).default(1).description("Tokens per second (token-bucket)"),
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
export const Config = Schema.object({
|
|
25
50
|
/** Master switch */
|
|
26
|
-
enabled: Schema.boolean().default(true).description("Enable the rate limiter globally"),
|
|
51
|
+
enabled: Schema.boolean().default(true).volatile().description("Enable the rate limiter globally"),
|
|
27
52
|
|
|
28
53
|
/** Default strategy */
|
|
29
54
|
strategy: Schema.union([
|
|
30
55
|
Schema.const("token-bucket").description("Token Bucket — allows bursts up to burstSize"),
|
|
31
56
|
Schema.const("sliding-window").description("Sliding Window — smooth, fixed requests-per-minute"),
|
|
32
|
-
]).default("token-bucket").description("Rate limiting algorithm"),
|
|
57
|
+
]).default("token-bucket").volatile().description("Rate limiting algorithm"),
|
|
33
58
|
|
|
34
59
|
/** Global defaults — applied to every model unless overridden */
|
|
35
|
-
defaults:
|
|
36
|
-
maxConcurrent: Schema.number().step(1).min(1).default(5).description("Max concurrent requests"),
|
|
37
|
-
maxRpm: Schema.number().step(1).min(1).default(60).description("Max requests per minute"),
|
|
38
|
-
burstSize: Schema.number().step(1).min(1).default(10).description("Burst capacity (token-bucket)"),
|
|
39
|
-
refillRate: Schema.number().min(0.1).default(1).description("Tokens per second (token-bucket)"),
|
|
40
|
-
}).default({}).description("Default limits applied to all models"),
|
|
60
|
+
defaults: Defaults.default({}).volatile().description("Default limits applied to all models"),
|
|
41
61
|
|
|
42
|
-
/**
|
|
43
|
-
|
|
62
|
+
/**
|
|
63
|
+
* Per-model overrides — key is "provider/model".
|
|
64
|
+
*
|
|
65
|
+
* WHOLE-BLOCK volatile: `dict(inner).volatile()`, never
|
|
66
|
+
* `dict(inner.volatile())`. The latter marks the dict's `inner` while the
|
|
67
|
+
* `sKey`/`inner` nodes are "blocked" (`schemastery/lib/index.mjs:249-250`),
|
|
68
|
+
* which throws at validation time.
|
|
69
|
+
*
|
|
70
|
+
* Because `isVolatilePath` (`dsh-settings/lib/index.js:153-158`) returns
|
|
71
|
+
* true on reaching ANY volatile node without inspecting its descendants,
|
|
72
|
+
* paths such as `["models", "provider/model", "maxRpm"]` remain writable
|
|
73
|
+
* through the settings form under this whole-block mark.
|
|
74
|
+
*/
|
|
75
|
+
models: Schema.dict(ModelOverride).default({}).volatile().description("Per-model rate limit overrides"),
|
|
44
76
|
|
|
45
77
|
/** Queue behaviour when a request is throttled */
|
|
46
78
|
onThrottled: Schema.union([
|
|
47
79
|
Schema.const("queue").description("Wait in queue until a slot opens"),
|
|
48
80
|
Schema.const("reject").description("Immediately return an error"),
|
|
49
|
-
]).default("queue").description("What happens when a request exceeds the limit"),
|
|
81
|
+
]).default("queue").volatile().description("What happens when a request exceeds the limit"),
|
|
50
82
|
|
|
51
83
|
/** Maximum time a request waits in queue before being rejected */
|
|
52
|
-
maxQueueWaitMs: Schema.number().step(1).min(1000).default(60_000).description("Max queue wait (ms)"),
|
|
53
|
-
}).description("LLM call rate limiter settings");
|
|
84
|
+
maxQueueWaitMs: Schema.number().step(1).min(1000).default(60_000).volatile().description("Max queue wait (ms)"),
|
|
85
|
+
}).description("LLM call rate limiter settings");
|
package/lib/types/index.js
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export {
|
|
1
|
+
export { Config } from "./config.js";
|