@leaf233/dsh-llm-rate-limiter 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -2,28 +2,36 @@
2
2
  * dsh-llm-rate-limiter — Host (server-side) entry point.
3
3
  *
4
4
  * Intercepts the `llm/stream` waterfall and enforces per-model rate limits.
5
- * Configuration lives in the "llm-rate-limiter" settings namespace, editable
6
- * from the browser GUI.
5
+ *
6
+ * Configuration is the profile ENTRY's own `config` (this module's exported
7
+ * {@link Config} schema), read through the volatile references Cordis builds
8
+ * for it. `dsh-settings` projects the volatile fields of every active entry
9
+ * into a form the browser can edit, so the plugin ships no settings service
10
+ * of its own. (DSH 0.1.5 had a `ctx.settings.register()` namespace API; 0.1.7
11
+ * removed it, and its browser half dropped `settingsScope` for `configForms`.
12
+ * The migration record lives in CHANGELOG.md under `[0.3.0]`.)
7
13
  *
8
14
  * The `llm/stream` waterfall signature is:
9
15
  * (options: GenerateOptions, next: () => AsyncIterable<StreamChunk>) => AsyncIterable<StreamChunk>
10
16
  *
11
17
  * where `options` always has `.provider` (string) and `.model` (string).
12
18
  *
13
- * The settings service is injected dynamically (like dshmarket's
14
- * installMarketSettings): `ctx.inject(['settings'], (scopedCtx) => ...)`, so a
15
- * host without a settings provider simply runs without the GUI wiring — the
16
- * composed entry stays as configured. Inherited from the DSH house pattern.
17
- *
18
19
  * @module dsh-llm-rate-limiter
19
20
  */
20
21
 
21
22
  import { TokenBucketStrategy } from "./strategies/token-bucket.js";
22
23
  import { SlidingWindowStrategy } from "./strategies/sliding-window.js";
23
- import { RateLimiterConfig } from "./types/config.js";
24
- import { CHANNEL, createStatusChannel } from "./status-rpc.js";
24
+ import { Config } from "./types/config.js";
25
+ import { CHANNEL, createChannelRoute, createStatusChannel } from "./status-rpc.js";
25
26
 
26
27
  const name = "llm-rate-limiter";
28
+ /**
29
+ * Settings namespace / profile entry id.
30
+ *
31
+ * `dsh-settings` keys every form by `entry.options.id` (`lib/index.js:432`),
32
+ * which is the `id` this plugin's `cordis.patch.yml` row declares. Renaming
33
+ * that row id therefore renames this namespace too — keep the two in step.
34
+ */
27
35
  const SETTINGS_NS = "llm-rate-limiter";
28
36
 
29
37
  /** Maximum retained events in the status ring (bounded memory). */
@@ -33,9 +41,45 @@ const SNAPSHOT_EVENT_COUNT = 8;
33
41
 
34
42
  /* ── Helpers ─────────────────────────────────────────────────────────── */
35
43
 
44
+ /** Built-in defaults, used when a reference yields nothing (or no config at all). */
45
+ const FALLBACK = {
46
+ enabled: true,
47
+ strategy: "token-bucket",
48
+ defaults: { maxConcurrent: 5, maxRpm: 60, burstSize: 10, refillRate: 1 },
49
+ models: {},
50
+ onThrottled: "queue",
51
+ maxQueueWaitMs: 60_000,
52
+ };
53
+
54
+ /**
55
+ * Read one volatile reference from the parsed Config.
56
+ *
57
+ * The loader mutates volatile references IN PLACE when only raw config values
58
+ * changed (`cordis-plugin-loader/lib/index.js:393` `_commitVolatile()`), so
59
+ * the reference object stays valid for the plugin's whole lifetime and its
60
+ * `get()` always yields the current value. Call this at each use rather than
61
+ * caching the unwrapped value.
62
+ *
63
+ * @param ref - a `.volatile()` field of the parsed Config, or undefined when
64
+ * the entry was mounted with no config at all.
65
+ * @param fallback - value used when the reference is absent.
66
+ * @returns the current plain value.
67
+ */
68
+ function read(ref, fallback) {
69
+ if (ref === undefined || ref === null) return fallback;
70
+ if (typeof ref.get === "function") {
71
+ const value = ref.get();
72
+ return value === undefined || value === null ? fallback : value;
73
+ }
74
+ return ref;
75
+ }
76
+
36
77
  /**
37
78
  * Merge defaults + per-model overrides into a flat config object that both
38
79
  * strategy constructors understand.
80
+ *
81
+ * @param cfg - plain config object (`{ enabled, strategy, defaults, models, ... }`).
82
+ * @param key - `"provider/model"` of the call being limited.
39
83
  */
40
84
  function resolveModelConfig(cfg, key) {
41
85
  const o = cfg.models?.[key] ?? {};
@@ -92,22 +136,26 @@ function terminalChunk(modelKey, signal) {
92
136
 
93
137
  /**
94
138
  * @param {import("@deepseek-ai/cordis").Context} ctx
139
+ * @param {object} [config] - parsed Config; volatile fields are live references.
95
140
  */
96
- function apply(ctx) {
141
+ function apply(ctx, config) {
97
142
  const logger = ctx.logger?.("rate-limiter");
98
143
 
99
- // Shared mutable config state. Starts with built-in defaults; once the
100
- // settings service is available the watcher replaces it with the live
101
- // user-configured value on every change.
102
- const defaultBase = {
103
- enabled: true,
104
- strategy: "token-bucket",
105
- defaults: { maxConcurrent: 5, maxRpm: 60, burstSize: 10, refillRate: 1 },
106
- models: {},
107
- onThrottled: "queue",
108
- maxQueueWaitMs: 60_000,
109
- };
110
- let cfg = { ...defaultBase };
144
+ /**
145
+ * Snapshot the whole config as one plain object, reading each reference at
146
+ * call time. Derived settings are not cached: a volatile-only change updates
147
+ * the references in place without re-running `apply`.
148
+ */
149
+ function readConfig() {
150
+ return {
151
+ enabled: read(config?.enabled, FALLBACK.enabled),
152
+ strategy: read(config?.strategy, FALLBACK.strategy),
153
+ defaults: read(config?.defaults, FALLBACK.defaults),
154
+ models: read(config?.models, FALLBACK.models),
155
+ onThrottled: read(config?.onThrottled, FALLBACK.onThrottled),
156
+ maxQueueWaitMs: read(config?.maxQueueWaitMs, FALLBACK.maxQueueWaitMs),
157
+ };
158
+ }
111
159
 
112
160
  /** @type {Map<string, TokenBucketStrategy | SlidingWindowStrategy>} */
113
161
  const limiters = new Map();
@@ -135,6 +183,7 @@ function apply(ctx) {
135
183
 
136
184
  /** Build the JSON-safe snapshot served on the "snapshot" endpoint. */
137
185
  function buildSnapshot() {
186
+ const cfg = readConfig();
138
187
  const models = {};
139
188
  for (const [key, limiter] of limiters) {
140
189
  try {
@@ -162,57 +211,114 @@ function apply(ctx) {
162
211
  rev += 1;
163
212
  }
164
213
 
165
- // ── Settings wiring (dynamic inject) ───────────────────────
166
- // Wrapped in ctx.inject so the plugin works on older DSH versions that
167
- // don't have a settings service at all — the callback simply never runs.
214
+ // ── Settings presentation (dynamic inject) ─────────────────
215
+ // This plugin ships its own browser page, so it opts out of any
216
+ // schema-generated page. Wrapped in ctx.inject so a host WITHOUT a settings
217
+ // service still runs the limiter — the callback simply never runs. The
218
+ // optional child ctx names the plugin fiber the policy belongs to, which is
219
+ // why `ctx.fiber` is passed explicitly (`dsh-settings/lib/index.js:370`).
168
220
  ctx.inject(["settings"], (scopedCtx) => {
169
- const scope = scopedCtx.settings.register(SETTINGS_NS, RateLimiterConfig, {
170
- base: defaultBase,
171
- });
172
-
173
- logger?.info("rate limiter settings namespace registered (%s)", SETTINGS_NS);
174
-
175
- // Replace cfg with live settings value on every change.
176
- cfg = scope.get();
177
- scope.watch((c) => { cfg = c; });
178
-
179
- // Dispose limiters on settings teardown.
180
- scopedCtx.effect(() => () => {
181
- logger?.info("rate limiter settings scope tearing down (%s)", SETTINGS_NS);
182
- for (const limiter of limiters.values()) limiter.dispose();
183
- limiters.clear();
184
- }, "rate-limiter: settings scope teardown");
221
+ if (typeof scopedCtx.settings?.configure !== "function") return; // older/foreign settings seam: keep running
222
+ scopedCtx.effect(
223
+ () => scopedCtx.settings.configure({ auto: false }, ctx.fiber),
224
+ "rate-limiter: settings presentation",
225
+ );
185
226
  });
186
227
 
187
- // ── Status RPC channel (dynamic inject) ────────────────────
228
+ // ── Status channel (dynamic inject) ────────────────────────
188
229
  // Same defensive shape as the settings wiring: on a host without a
189
230
  // connection service the callback never runs and the plugin stays silent.
190
- // The registration is owned by this fiber, so unloading withdraws it.
231
+ // Either path below registers inside an effect owned by this fiber, so
232
+ // unloading the plugin withdraws the channel automatically.
233
+ //
234
+ // Two mounting paths exist, and both speak the same wire protocol, so the
235
+ // browser half never needs to know which one is live:
236
+ //
237
+ // 1. connection.rpc.handle(channel, handler) — the framework's generic
238
+ // channel registry (preferred: the framework owns the route, the
239
+ // request validation, and the withdrawal).
240
+ //
241
+ // 2. a self-registered prefix route reusing connection.requestRejection()
242
+ // — the fallback for hosts where path 1 is broken. DSH 0.1.5-rc.3
243
+ // changed dsh-client-connection's own `inject` from
244
+ // ["webServer", "credentials"] to ["credentials"] while its
245
+ // HostConnectionService.register() still dereferences
246
+ // `owner.webServer`, and Cordis rebinds a cross-fiber service's `ctx`
247
+ // to the *reader's* fiber — so rpc.handle() throws
248
+ // 'cannot get property "webServer" without inject'. Registering the
249
+ // route on our own fiber (which injects webServer below) avoids that
250
+ // while still using the framework's own 403/401 fence.
251
+ // Still unfixed in 0.1.7-rc.1 (dsh-client-connection/lib/index.js:798
252
+ // keeps `inject = ["credentials"]`), so the fallback still fires.
191
253
  ctx.inject(["connection"], (scopedCtx) => {
192
254
  const connection = scopedCtx.get("connection");
255
+ const channel = createStatusChannel({ buildSnapshot, resetTotals });
256
+
257
+ // ── Path 1: the framework's generic channel registry ──
193
258
  const handle = typeof connection?.rpc?.handle === "function"
194
259
  ? connection.rpc.handle.bind(connection.rpc)
195
260
  : undefined;
196
- if (handle === undefined) return; // older host: degrade silently
197
261
 
262
+ if (handle !== undefined) {
263
+ try {
264
+ scopedCtx.effect(() => {
265
+ const unregister = handle(CHANNEL, channel);
266
+ return () => { unregister(); };
267
+ }, "rate-limiter: status rpc channel");
268
+ logger?.info("rate limiter status channel mounted via connection.rpc (%s)", CHANNEL);
269
+ return;
270
+ } catch (err) {
271
+ logger?.warn(
272
+ "connection.rpc channel registration failed (%s); falling back to a self-registered route",
273
+ err instanceof Error ? err.message : String(err),
274
+ );
275
+ }
276
+ }
277
+
278
+ // ── Path 2: self-registered route (needs webServer on THIS fiber) ──
279
+ // The fence is mandatory: without connection.requestRejection() we would
280
+ // be publishing an unauthenticated route, so we refuse to mount at all.
281
+ // Probing once here turns a broken fence into a mount-time refusal instead
282
+ // of a per-request surprise.
198
283
  try {
199
- scopedCtx.effect(() => {
200
- const unregister = handle(CHANNEL, createStatusChannel({ buildSnapshot, resetTotals }));
201
- return () => { unregister(); };
202
- }, "rate-limiter: status rpc channel");
284
+ if (typeof connection?.requestRejection !== "function") {
285
+ logger?.warn("connection.requestRejection unavailable; status channel not mounted");
286
+ return;
287
+ }
288
+ connection.requestRejection({ headers: {} });
203
289
  } catch (err) {
204
- logger?.warn("status channel registration failed: %s", err instanceof Error ? err.message : String(err));
290
+ logger?.warn(
291
+ "connection.requestRejection unusable (%s); refusing to mount an unfenced status route",
292
+ err instanceof Error ? err.message : String(err),
293
+ );
205
294
  return;
206
295
  }
207
296
 
208
- logger?.info("rate limiter status channel mounted (%s)", CHANNEL);
297
+ scopedCtx.inject(["webServer"], (webCtx) => {
298
+ const webServer = webCtx.get("webServer");
299
+ if (typeof webServer?.register !== "function") return; // no web carrier: degrade silently
300
+
301
+ try {
302
+ webCtx.effect(() => webServer.register(createChannelRoute({
303
+ channel: CHANNEL,
304
+ handler: channel,
305
+ reject: (req) => connection.requestRejection(req),
306
+ })), "rate-limiter: status route (self-registered)");
307
+ } catch (err) {
308
+ logger?.warn("status channel registration failed: %s", err instanceof Error ? err.message : String(err));
309
+ return;
310
+ }
311
+
312
+ logger?.info("rate limiter status channel mounted via self-registered route (%s)", CHANNEL);
313
+ });
209
314
  });
210
315
 
211
316
  // ── LLM stream interceptor ─────────────────────────────────
212
- // ctx.on('llm/stream') doesn't need the settings service — it's a plain
213
- // waterfall listener that reads the shared `cfg` variable (initialised
214
- // with defaults; replaced by the settings watcher when available).
317
+ // Reads the live Config on every call, so a settings edit that reaches the
318
+ // loader takes effect on the next request without a restart.
215
319
  const disposeLlmListener = ctx.on("llm/stream", async function* rateLimitInterceptor(options, next) {
320
+ const cfg = readConfig();
321
+
216
322
  if (!cfg.enabled) {
217
323
  yield* next();
218
324
  return;
@@ -297,4 +403,4 @@ function apply(ctx) {
297
403
  logger?.info("rate limiter active — llm/stream interceptor registered");
298
404
  }
299
405
 
300
- export { apply, name, SETTINGS_NS };
406
+ export { apply, name, SETTINGS_NS, Config };
package/lib/status-rpc.js CHANGED
@@ -2,15 +2,27 @@
2
2
  * dsh-llm-rate-limiter — status RPC channel (Host half).
3
3
  *
4
4
  * Exposes live rate-limiter statistics to the browser over the framework's
5
- * generic Connection RPC channel — the same idiom dsh-context uses for its
6
- * detail channel. The framework supplies POST+JSON transport, the Host/Origin
7
- * fence (403) and browser authentication (401); this module only answers
8
- * endpoints and shapes envelopes.
5
+ * generic Connection RPC channel. The framework supplies POST+JSON transport,
6
+ * the Host/Origin fence (403) and browser authentication (401); this module
7
+ * only answers endpoints and shapes envelopes.
9
8
  *
10
- * Registration is owned by the calling fiber (`register()` wraps
11
- * `owner.webServer.register` in `owner.effect`), so unloading the plugin
12
- * withdraws the channel automatically — there is no connection bookkeeping
13
- * here.
9
+ * Two mounting paths exist, and both speak the *same* wire protocol so the
10
+ * browser half never needs to know which one is active:
11
+ *
12
+ * 1. `connection.rpc.handle(channel, handler)` — the framework's own generic
13
+ * channel registry. Preferred: the framework owns the route, the request
14
+ * validation, and the fiber-scoped withdrawal.
15
+ *
16
+ * 2. `createChannelRoute({ channel, handler, reject })` — a self-registered
17
+ * prefix route for hosts where path 1 is broken. DSH 0.1.5-rc.3 changed
18
+ * `@deepseek-ai/dsh-client-connection`'s own `inject` from
19
+ * `["webServer", "credentials"]` to `["credentials"]` while
20
+ * `HostConnectionService.register()` still dereferences
21
+ * `owner.webServer`, so `rpc.handle` throws
22
+ * `cannot get property "webServer" without inject`. This route is
23
+ * registered on our own fiber (which *can* see `webServer`) and reuses the
24
+ * connection service's public `requestRejection()` fence, so the 403/401
25
+ * policy is still the framework's.
14
26
  *
15
27
  * Envelope contract (mirrors the Connection wire schema):
16
28
  * success: { ok: true, value: <JSON-safe> }
@@ -25,6 +37,27 @@
25
37
  */
26
38
  const CHANNEL = "/llm-rate-limiter";
27
39
 
40
+ /**
41
+ * Endpoint grammar for one path segment, copied verbatim from the framework's
42
+ * `ENDPOINT_SEGMENT_PATTERN` in dsh-client-connection. Keeping it identical is
43
+ * what makes the self-registered route accept and reject exactly the same URLs
44
+ * as `rpc.handle` would.
45
+ */
46
+ const ENDPOINT_SEGMENT_PATTERN = /^[A-Za-z0-9_$.-]+$/;
47
+
48
+ /**
49
+ * Buffered-body cap for the self-registered route. Both endpoints are called
50
+ * with an empty payload, so this is a defensive memory bound rather than a
51
+ * real limit (the framework uses its own, much larger, carrier cap).
52
+ */
53
+ const MAX_REQUEST_BODY_BYTES = 1024 * 1024;
54
+
55
+ /**
56
+ * Correlation id echoed when the request envelope cannot be parsed at all —
57
+ * the same literal the framework uses (`INVALID_REQUEST_RPC_ID`).
58
+ */
59
+ const INVALID_REQUEST_RPC_ID = "invalid-request";
60
+
28
61
  /**
29
62
  * Build a failure envelope. Field-for-field the same shape the Connection
30
63
  * host accepts, so callers can rely on `ok` discrimination alone.
@@ -64,4 +97,189 @@ function createStatusChannel({ buildSnapshot, resetTotals }) {
64
97
  };
65
98
  }
66
99
 
67
- export { CHANNEL, createStatusChannel, failure, success };
100
+ /* ── self-registered route (the DSH 0.1.5-rc.3 fallback) ───────────────── */
101
+
102
+ /** Narrow an unknown JSON value to a plain object. */
103
+ function isRecord(value) {
104
+ return typeof value === "object" && value !== null && !Array.isArray(value);
105
+ }
106
+
107
+ /**
108
+ * Extract the endpoint from a channel-relative pathname, mirroring the
109
+ * framework's `endpointFromPath`: a single `channel/<endpoint>` segment whose
110
+ * segments all satisfy {@link ENDPOINT_SEGMENT_PATTERN}.
111
+ *
112
+ * @param {string} channel - the channel prefix (e.g. `/llm-rate-limiter`).
113
+ * @param {string} pathname - request pathname.
114
+ * @returns {string | undefined} the endpoint, or undefined when the URL is not
115
+ * a well-formed member of this channel.
116
+ */
117
+ function endpointFromPath(channel, pathname) {
118
+ if (!pathname.startsWith(`${channel}/`)) return undefined;
119
+ const endpoint = pathname.slice(channel.length + 1);
120
+ const segments = endpoint.split("/");
121
+ if (segments.some((segment) => segment === "" || segment === "." || segment === ".." || !ENDPOINT_SEGMENT_PATTERN.test(segment))) return undefined;
122
+ return endpoint;
123
+ }
124
+
125
+ /**
126
+ * Read and decode the request body, refusing anything above `limit` bytes so a
127
+ * hostile caller cannot make the host buffer without bound.
128
+ *
129
+ * @param {import("node:http").IncomingMessage} req - node request stream.
130
+ * @param {number} limit - maximum accepted byte count.
131
+ * @returns {Promise<string>} the decoded UTF-8 body.
132
+ * @throws {Error & { code: "TOO_LARGE" }} when the body exceeds `limit`.
133
+ */
134
+ async function readBody(req, limit) {
135
+ const chunks = [];
136
+ let received = 0;
137
+ for await (const chunk of req) {
138
+ const buffer = typeof chunk === "string" ? Buffer.from(chunk) : chunk;
139
+ received += buffer.byteLength;
140
+ if (received > limit) {
141
+ const error = new Error("request body too large");
142
+ error.code = "TOO_LARGE";
143
+ throw error;
144
+ }
145
+ chunks.push(buffer);
146
+ }
147
+ return Buffer.concat(chunks).toString("utf8");
148
+ }
149
+
150
+ /**
151
+ * Write one `server-response` envelope as JSON — byte-for-byte the shape
152
+ * `Response.json(...)` produces on the framework's own route.
153
+ *
154
+ * @param {import("node:http").ServerResponse} res - response to own.
155
+ * @param {string} rpcId - correlation id to echo.
156
+ * @param {object} result - the `{ ok, value }` / `{ ok, error }` envelope.
157
+ */
158
+ function sendEnvelope(res, rpcId, result) {
159
+ res.writeHead(200, { "content-type": "application/json" });
160
+ res.end(JSON.stringify({ type: "server-response", rpcId, result }));
161
+ }
162
+
163
+ /**
164
+ * Build a webserver prefix route that speaks the Connection RPC wire protocol.
165
+ *
166
+ * This is the fallback carrier for {@link createStatusChannel} on hosts where
167
+ * `connection.rpc.handle` throws. Every request-shape decision mirrors the
168
+ * framework's `rpcFetchHandler` so the browser half is byte-compatible with the
169
+ * preferred path:
170
+ *
171
+ * | condition | response |
172
+ * |---|---|
173
+ * | fence/auth rejected by `reject` | `403` (or `401`) + `forbidden`/`unauthorized` |
174
+ * | non-POST, or URL outside the channel | `404 not found` |
175
+ * | `content-type` not `application/json` | `415` |
176
+ * | body over {@link MAX_REQUEST_BODY_BYTES} | `413` |
177
+ * | body not JSON | `400 body is not JSON` |
178
+ * | envelope not a `client-request` | `gateway/bad-request` envelope |
179
+ * | `method` ≠ endpoint | `gateway/bad-request` envelope |
180
+ * | handler threw | `500 handler failure: ...` |
181
+ * | otherwise | `200` + `server-response` envelope |
182
+ *
183
+ * @param {object} deps
184
+ * @param {string} deps.channel - channel prefix to own (e.g. `/llm-rate-limiter`).
185
+ * @param {(endpoint: string, payload: unknown, signal?: AbortSignal) => Promise<object>} deps.handler - endpoint dispatcher.
186
+ * @param {(req: import("node:http").IncomingMessage) => number | undefined} deps.reject - the framework's `connection.requestRejection` fence; returns an HTTP status to refuse with, or undefined to allow.
187
+ * @returns {{ kind: "prefix", path: string, handler: (req: import("node:http").IncomingMessage, res: import("node:http").ServerResponse) => Promise<void> }}
188
+ */
189
+ function createChannelRoute({ channel, handler, reject }) {
190
+ return {
191
+ kind: "prefix",
192
+ path: channel,
193
+ handler: async (req, res) => {
194
+ // 1. The framework's own fence: Host/Origin trust (403) then browser
195
+ // authentication (401). Reusing it keeps the security policy identical.
196
+ const rejection = reject(req);
197
+ if (rejection !== undefined) {
198
+ res.writeHead(rejection);
199
+ res.end(rejection === 401 ? "unauthorized" : "forbidden");
200
+ return;
201
+ }
202
+
203
+ // 2. POST-only, and only inside this channel (framework: 404 otherwise).
204
+ const url = new URL(req.url ?? "/", "http://dsh.internal");
205
+ const endpoint = endpointFromPath(channel, url.pathname);
206
+ if (req.method !== "POST" || endpoint === undefined) {
207
+ res.writeHead(404);
208
+ res.end("not found");
209
+ return;
210
+ }
211
+
212
+ // 3. JSON only (framework: 415).
213
+ const mediaType = String(req.headers["content-type"] ?? "").split(";", 1)[0].trim().toLowerCase();
214
+ if (mediaType !== "application/json") {
215
+ res.writeHead(415);
216
+ res.end("content type must be application/json");
217
+ return;
218
+ }
219
+
220
+ // 4. Buffer the body under a defensive cap, then parse it.
221
+ let raw;
222
+ try {
223
+ raw = await readBody(req, MAX_REQUEST_BODY_BYTES);
224
+ } catch (error) {
225
+ if (error?.code === "TOO_LARGE") {
226
+ res.writeHead(413, { connection: "close" });
227
+ res.end();
228
+ req.destroy?.();
229
+ return;
230
+ }
231
+ throw error;
232
+ }
233
+ let body;
234
+ try {
235
+ body = JSON.parse(raw);
236
+ } catch {
237
+ res.writeHead(400);
238
+ res.end("body is not JSON");
239
+ return;
240
+ }
241
+
242
+ // 5. Validate the client-request envelope; echo the correlation id
243
+ // whenever it is a string, exactly like the framework does.
244
+ const rpcId = typeof body?.rpcId === "string" ? body.rpcId : INVALID_REQUEST_RPC_ID;
245
+ if (!isRecord(body) || body.type !== "client-request" || typeof body.rpcId !== "string" || typeof body.method !== "string") {
246
+ sendEnvelope(res, rpcId, failure("gateway/bad-request", "invalid client-request message", { issues: [] }));
247
+ return;
248
+ }
249
+ if (body.method !== endpoint) {
250
+ sendEnvelope(res, rpcId, failure(
251
+ "gateway/bad-request",
252
+ `method ${JSON.stringify(body.method)} does not match endpoint ${JSON.stringify(endpoint)}`,
253
+ { issues: [] },
254
+ ));
255
+ return;
256
+ }
257
+
258
+ // 6. Dispatch. Tie the signal to the response lifetime, as the framework's
259
+ // http bridge does, so a dropped client can cancel the work.
260
+ const abort = new AbortController();
261
+ res.on?.("close", () => {
262
+ if (!res.writableEnded) abort.abort();
263
+ });
264
+ let result;
265
+ try {
266
+ result = await handler(endpoint, body.payload, abort.signal);
267
+ } catch (error) {
268
+ res.writeHead(500);
269
+ res.end(`handler failure: ${String(error)}`);
270
+ return;
271
+ }
272
+ sendEnvelope(res, rpcId, result);
273
+ },
274
+ };
275
+ }
276
+
277
+ export {
278
+ CHANNEL,
279
+ MAX_REQUEST_BODY_BYTES,
280
+ createChannelRoute,
281
+ createStatusChannel,
282
+ endpointFromPath,
283
+ failure,
284
+ success,
285
+ };
@@ -1,8 +1,15 @@
1
1
  /**
2
2
  * Rate limiter configuration schema.
3
3
  *
4
- * The settings namespace is "llm-rate-limiter". Users override values through
5
- * the DSH settings document; the host merges them on top of `base`.
4
+ * The plugin's Config is the profile entry's own schema: Cordis validates the
5
+ * entry's `config` against it, so the host reads values from the references
6
+ * this schema produces rather than from a settings namespace.
7
+ *
8
+ * Fields marked `.volatile()` become separately addressable live references
9
+ * (`config.enabled.get()`). Only volatile fields appear in the settings form
10
+ * (`dsh-settings`'s `volatileForm()`), and an entry whose Config has NO
11
+ * volatile field is dropped from the form list entirely — so the set of
12
+ * `.volatile()` marks below IS the settings UI's field list.
6
13
  *
7
14
  * @module dsh-llm-rate-limiter/types/config
8
15
  */
@@ -12,6 +19,12 @@ import Schema from "@deepseek-ai/schemastery";
12
19
  /**
13
20
  * Per-model overrides — key is `"provider/model"` (e.g. `"deepseek/deepseek-chat"`).
14
21
  * Every field is optional; omitted fields inherit from `defaults`.
22
+ *
23
+ * This sub-schema deliberately carries NO `.volatile()`: it is the `inner`
24
+ * of a `dict`, and two volatile layers on one path are rejected
25
+ * (`volatile fields require a fixed object path without an enclosing volatile
26
+ * field`, `schemastery/lib/index.mjs:246`). The whole `models` dict is made
27
+ * volatile instead — see below.
15
28
  */
16
29
  const ModelOverride = Schema.object({
17
30
  maxConcurrent: Schema.number().step(1).min(1).description("Max concurrent requests for this model"),
@@ -21,33 +34,52 @@ const ModelOverride = Schema.object({
21
34
  enabled: Schema.boolean().description("Enable/disable rate limiting for this model"),
22
35
  });
23
36
 
24
- export const RateLimiterConfig = Schema.object({
37
+ /**
38
+ * Per-model defaults — applied to every model unless overridden.
39
+ * The object is volatile AS A WHOLE so `config.defaults.get()` yields one
40
+ * plain object and the settings form edits `["defaults", field]` paths.
41
+ */
42
+ const Defaults = Schema.object({
43
+ maxConcurrent: Schema.number().step(1).min(1).default(5).description("Max concurrent requests"),
44
+ maxRpm: Schema.number().step(1).min(1).default(60).description("Max requests per minute"),
45
+ burstSize: Schema.number().step(1).min(1).default(10).description("Burst capacity (token-bucket)"),
46
+ refillRate: Schema.number().min(0.1).default(1).description("Tokens per second (token-bucket)"),
47
+ });
48
+
49
+ export const Config = Schema.object({
25
50
  /** Master switch */
26
- enabled: Schema.boolean().default(true).description("Enable the rate limiter globally"),
51
+ enabled: Schema.boolean().default(true).volatile().description("Enable the rate limiter globally"),
27
52
 
28
53
  /** Default strategy */
29
54
  strategy: Schema.union([
30
55
  Schema.const("token-bucket").description("Token Bucket — allows bursts up to burstSize"),
31
56
  Schema.const("sliding-window").description("Sliding Window — smooth, fixed requests-per-minute"),
32
- ]).default("token-bucket").description("Rate limiting algorithm"),
57
+ ]).default("token-bucket").volatile().description("Rate limiting algorithm"),
33
58
 
34
59
  /** Global defaults — applied to every model unless overridden */
35
- defaults: Schema.object({
36
- maxConcurrent: Schema.number().step(1).min(1).default(5).description("Max concurrent requests"),
37
- maxRpm: Schema.number().step(1).min(1).default(60).description("Max requests per minute"),
38
- burstSize: Schema.number().step(1).min(1).default(10).description("Burst capacity (token-bucket)"),
39
- refillRate: Schema.number().min(0.1).default(1).description("Tokens per second (token-bucket)"),
40
- }).default({}).description("Default limits applied to all models"),
60
+ defaults: Defaults.default({}).volatile().description("Default limits applied to all models"),
41
61
 
42
- /** Per-model overrides — key is "provider/model" */
43
- models: Schema.dict(ModelOverride).description("Per-model rate limit overrides"),
62
+ /**
63
+ * Per-model overrides — key is "provider/model".
64
+ *
65
+ * WHOLE-BLOCK volatile: `dict(inner).volatile()`, never
66
+ * `dict(inner.volatile())`. The latter marks the dict's `inner` while the
67
+ * `sKey`/`inner` nodes are "blocked" (`schemastery/lib/index.mjs:249-250`),
68
+ * which throws at validation time.
69
+ *
70
+ * Because `isVolatilePath` (`dsh-settings/lib/index.js:153-158`) returns
71
+ * true on reaching ANY volatile node without inspecting its descendants,
72
+ * paths such as `["models", "provider/model", "maxRpm"]` remain writable
73
+ * through the settings form under this whole-block mark.
74
+ */
75
+ models: Schema.dict(ModelOverride).default({}).volatile().description("Per-model rate limit overrides"),
44
76
 
45
77
  /** Queue behaviour when a request is throttled */
46
78
  onThrottled: Schema.union([
47
79
  Schema.const("queue").description("Wait in queue until a slot opens"),
48
80
  Schema.const("reject").description("Immediately return an error"),
49
- ]).default("queue").description("What happens when a request exceeds the limit"),
81
+ ]).default("queue").volatile().description("What happens when a request exceeds the limit"),
50
82
 
51
83
  /** Maximum time a request waits in queue before being rejected */
52
- maxQueueWaitMs: Schema.number().step(1).min(1000).default(60_000).description("Max queue wait (ms)"),
53
- }).description("LLM call rate limiter settings");
84
+ maxQueueWaitMs: Schema.number().step(1).min(1000).default(60_000).volatile().description("Max queue wait (ms)"),
85
+ }).description("LLM call rate limiter settings");
@@ -1 +1 @@
1
- export { RateLimiterConfig } from "./config.js";
1
+ export { Config } from "./config.js";