llm-runtime-dock 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +58 -3
- package/dist/index.js +244 -48
- package/package.json +26 -26
package/README.md
CHANGED
|
@@ -222,6 +222,7 @@ field.
|
|
|
222
222
|
server:
|
|
223
223
|
host: 127.0.0.1
|
|
224
224
|
port: 8787
|
|
225
|
+
idle_unload: 60m # unload after an hour of quiet; 0 to never
|
|
225
226
|
|
|
226
227
|
runtimes:
|
|
227
228
|
mtplx:
|
|
@@ -291,6 +292,13 @@ Clients send the logical id and never see the backend model:
|
|
|
291
292
|
|
|
292
293
|
### Fields you will actually set
|
|
293
294
|
|
|
295
|
+
On the gateway itself:
|
|
296
|
+
|
|
297
|
+
| field | meaning |
|
|
298
|
+
| --------------- | ------------------------------------------------------------------ |
|
|
299
|
+
| `host` / `port` | where `lrd` listens. Defaults to `127.0.0.1:8787` |
|
|
300
|
+
| `idle_unload` | unload the loaded model after this long with nothing to do (below) |
|
|
301
|
+
|
|
294
302
|
On a runtime:
|
|
295
303
|
|
|
296
304
|
| field | meaning |
|
|
@@ -335,7 +343,9 @@ Some arguments belong to the gateway and are rejected rather than silently
|
|
|
335
343
|
merged: `--host`, `--port`, `--model`, served-id flags like `--identifier`,
|
|
336
344
|
`--api-key`, and idle-unload flags such as LM Studio's `--ttl` — an idle
|
|
337
345
|
auto-unload would drop the model behind the gateway's back and leave its view of
|
|
338
|
-
the world wrong.
|
|
346
|
+
the world wrong. That is about _who owns the timer_, not about the idea: the
|
|
347
|
+
gateway runs one itself, in front of its own bookkeeping, and you set it with
|
|
348
|
+
`idle_unload` (below). The check covers `extra_args` too, and the error names the
|
|
339
349
|
field to use instead. The full table, with the reasoning per flag, is in
|
|
340
350
|
[§12](docs/05-configuration.md#configuration).
|
|
341
351
|
|
|
@@ -356,6 +366,51 @@ releasing foreign mtplx server on :8001 to free the resident slot
|
|
|
356
366
|
|
|
357
367
|
The reasoning is in [§8 of the specification](docs/03-lifecycle.md#lifecycle).
|
|
358
368
|
|
|
369
|
+
### Unloading when you stop using it
|
|
370
|
+
|
|
371
|
+
`lrd serve` is a long-lived process, and until something asks for a different
|
|
372
|
+
model nothing frees the one it is holding. Leave the gateway running after a
|
|
373
|
+
morning's work and that 27B is still in memory at midnight.
|
|
374
|
+
|
|
375
|
+
So it unloads by itself after an hour of quiet:
|
|
376
|
+
|
|
377
|
+
```yaml
|
|
378
|
+
server:
|
|
379
|
+
idle_unload: 60m # 0 never unloads
|
|
380
|
+
```
|
|
381
|
+
|
|
382
|
+
Write it as `60m`, `1h`, `90s`, `3600000ms`, or a bare number of milliseconds.
|
|
383
|
+
`lrd serve --idle-unload 90s` overrides it for one run, and `lrd serve` prints
|
|
384
|
+
the window it is using on startup:
|
|
385
|
+
|
|
386
|
+
```text
|
|
387
|
+
idle: 1h then unload
|
|
388
|
+
```
|
|
389
|
+
|
|
390
|
+
The next request loads the model again exactly as the first one did — the cost
|
|
391
|
+
is one reload after an hour of not working, which is the trade the default is
|
|
392
|
+
picked for. Turn it off with `0` if you would rather keep the model warm
|
|
393
|
+
indefinitely.
|
|
394
|
+
|
|
395
|
+
Two things it will not do:
|
|
396
|
+
|
|
397
|
+
- **it never unloads a `keep_resident` model.** That flag means "for as long as
|
|
398
|
+
the gateway runs", and an idle spell is not the gateway stopping;
|
|
399
|
+
- **it will not stop a server it did not start.** If `lrd` attached to an MTPLX
|
|
400
|
+
or custom server you launched yourself, stopping it is the only way to free
|
|
401
|
+
that memory — and doing that because `lrd` went quiet is not its call. It says
|
|
402
|
+
so in the log and leaves it alone. On LM Studio, oMLX and Ollama the question
|
|
403
|
+
never arises: unloading one model leaves the server, and everyone else on it,
|
|
404
|
+
untouched.
|
|
405
|
+
|
|
406
|
+
When it has unloaded something, `lrd status` says so rather than just showing an
|
|
407
|
+
empty slot:
|
|
408
|
+
|
|
409
|
+
```text
|
|
410
|
+
resident: none
|
|
411
|
+
released: coding-quality (idle, stop_server)
|
|
412
|
+
```
|
|
413
|
+
|
|
359
414
|
### Keeping one model always loaded
|
|
360
415
|
|
|
361
416
|
Sometimes one model should never leave memory — a small one a coding agent
|
|
@@ -419,8 +474,8 @@ kept: summariser (mtplx, ready, 0 active)
|
|
|
419
474
|
serving: summariser
|
|
420
475
|
```
|
|
421
476
|
|
|
422
|
-
Its lifetime is the gateway's
|
|
423
|
-
kept entries included — nothing would be left to free the memory otherwise, and
|
|
477
|
+
Its lifetime is the gateway's — the idle unload above does not touch it either.
|
|
478
|
+
Stopping `lrd serve` releases everything loaded, kept entries included — nothing would be left to free the memory otherwise, and
|
|
424
479
|
a model outliving the process that loaded it is exactly the leak this gateway
|
|
425
480
|
exists to prevent.
|
|
426
481
|
|
package/dist/index.js
CHANGED
|
@@ -11308,6 +11308,50 @@ var applyAgent = async (options) => {
|
|
|
11308
11308
|
};
|
|
11309
11309
|
};
|
|
11310
11310
|
|
|
11311
|
+
// packages/core/dist/config/duration.js
|
|
11312
|
+
var UNIT_MS = {
|
|
11313
|
+
ms: 1,
|
|
11314
|
+
s: 1e3,
|
|
11315
|
+
m: 6e4,
|
|
11316
|
+
h: 36e5
|
|
11317
|
+
};
|
|
11318
|
+
var SPELLING = "3600000, 3600000ms, 3600s, 60m or 1h; 0 disables";
|
|
11319
|
+
var parseDuration = (value, field) => {
|
|
11320
|
+
if (typeof value === "number") {
|
|
11321
|
+
if (!Number.isInteger(value) || value < 0) {
|
|
11322
|
+
throw cliError("CONFIG_INVALID", `${field} must be a whole number of milliseconds or more`, {
|
|
11323
|
+
details: { field, value: String(value) },
|
|
11324
|
+
hint: `accepted: ${SPELLING}`
|
|
11325
|
+
});
|
|
11326
|
+
}
|
|
11327
|
+
return value;
|
|
11328
|
+
}
|
|
11329
|
+
if (typeof value !== "string") {
|
|
11330
|
+
throw cliError("CONFIG_INVALID", `${field} must be a duration`, {
|
|
11331
|
+
details: { field, type: typeof value },
|
|
11332
|
+
hint: `accepted: ${SPELLING}`
|
|
11333
|
+
});
|
|
11334
|
+
}
|
|
11335
|
+
const match = value.trim().match(/^(\d+)\s*(ms|s|m|h)?$/i);
|
|
11336
|
+
if (!match) {
|
|
11337
|
+
throw cliError("CONFIG_INVALID", `${field} is not a duration: "${value}"`, {
|
|
11338
|
+
details: { field, value },
|
|
11339
|
+
hint: `accepted: ${SPELLING}`
|
|
11340
|
+
});
|
|
11341
|
+
}
|
|
11342
|
+
return Number(match[1]) * UNIT_MS[(match[2] ?? "ms").toLowerCase()];
|
|
11343
|
+
};
|
|
11344
|
+
var formatDuration = (ms) => {
|
|
11345
|
+
if (ms <= 0)
|
|
11346
|
+
return "off";
|
|
11347
|
+
for (const unit of ["h", "m", "s"]) {
|
|
11348
|
+
const size = UNIT_MS[unit];
|
|
11349
|
+
if (ms % size === 0)
|
|
11350
|
+
return `${ms / size}${unit}`;
|
|
11351
|
+
}
|
|
11352
|
+
return `${ms}ms`;
|
|
11353
|
+
};
|
|
11354
|
+
|
|
11311
11355
|
// packages/core/dist/config/load.js
|
|
11312
11356
|
var import_yaml = __toESM(require_dist(), 1);
|
|
11313
11357
|
import { readFileSync as readFileSync2 } from "node:fs";
|
|
@@ -16474,7 +16518,25 @@ var modelEntrySchema = object({
|
|
|
16474
16518
|
}).strict();
|
|
16475
16519
|
var serverSchema = object({
|
|
16476
16520
|
host: string2().min(1).default("127.0.0.1"),
|
|
16477
|
-
port: number2().int().min(1).max(65535).default(8787)
|
|
16521
|
+
port: number2().int().min(1).max(65535).default(8787),
|
|
16522
|
+
/**
|
|
16523
|
+
* Release the rotating occupant after this long with nothing to do (§29).
|
|
16524
|
+
* Defaults to an hour; `0` switches it off.
|
|
16525
|
+
*
|
|
16526
|
+
* Shape only here: a duration may be written `60m`, `1h`, `90s`, `3600000ms`
|
|
16527
|
+
* or as a bare number of milliseconds, and `parseDuration` in ./duration.ts
|
|
16528
|
+
* is what resolves it — in `buildConfig`, so the error names the YAML key.
|
|
16529
|
+
*
|
|
16530
|
+
* The first key in this block that is not a bind setting. It sits here
|
|
16531
|
+
* rather than on a model because it is a property of this gateway process,
|
|
16532
|
+
* not of any one entry.
|
|
16533
|
+
*/
|
|
16534
|
+
idle_unload: union([number2(), string2()], {
|
|
16535
|
+
// zod's own text for a failed union is a bare "Invalid input", which
|
|
16536
|
+
// says nothing at all for a field whose whole point is that it accepts
|
|
16537
|
+
// several spellings.
|
|
16538
|
+
error: "expected a duration such as 60m, 1h, 90s or a number of milliseconds"
|
|
16539
|
+
}).optional()
|
|
16478
16540
|
}).strict();
|
|
16479
16541
|
var agentsSchema = object({
|
|
16480
16542
|
claude: object({
|
|
@@ -16497,6 +16559,7 @@ var rawConfigSchema = object({
|
|
|
16497
16559
|
|
|
16498
16560
|
// packages/core/dist/config/load.js
|
|
16499
16561
|
var DEFAULT_HOST = "127.0.0.1";
|
|
16562
|
+
var DEFAULT_IDLE_UNLOAD_MS = 36e5;
|
|
16500
16563
|
var loadConfig = (registry2, options = {}) => {
|
|
16501
16564
|
const location = resolveConfigLocation(options);
|
|
16502
16565
|
const contents = options.contents ?? readConfigFile(location);
|
|
@@ -16632,7 +16695,12 @@ var buildConfig = (raw, location, registry2) => {
|
|
|
16632
16695
|
}
|
|
16633
16696
|
checkKeptResidency(models, runtimes, runtimeAdapters);
|
|
16634
16697
|
checkAgentRoles(raw.agents, models);
|
|
16635
|
-
|
|
16698
|
+
const server = {
|
|
16699
|
+
host: raw.server.host,
|
|
16700
|
+
port: raw.server.port,
|
|
16701
|
+
idleUnloadMs: raw.server.idle_unload === void 0 ? DEFAULT_IDLE_UNLOAD_MS : parseDuration(raw.server.idle_unload, "server.idle_unload")
|
|
16702
|
+
};
|
|
16703
|
+
return { location, server, runtimes, models, agents: raw.agents, raw };
|
|
16636
16704
|
};
|
|
16637
16705
|
var checkKeptResidency = (models, runtimes, adapters) => {
|
|
16638
16706
|
for (const instance of models.values()) {
|
|
@@ -17965,11 +18033,13 @@ var listLogicalModels = (config2) => {
|
|
|
17965
18033
|
// packages/core/dist/scheduler.js
|
|
17966
18034
|
var DEFAULT_READY_TIMEOUT_MS = 3e5;
|
|
17967
18035
|
var DEFAULT_DRAIN_TIMEOUT_MS = 6e5;
|
|
18036
|
+
var DEFAULT_IDLE_UNLOAD_MS2 = 36e5;
|
|
17968
18037
|
var createScheduler = (options) => {
|
|
17969
18038
|
const registry2 = options.registry;
|
|
17970
18039
|
const logger = options.logger ?? nullLogger;
|
|
17971
18040
|
const readyTimeoutMs = options.readyTimeoutMs ?? DEFAULT_READY_TIMEOUT_MS;
|
|
17972
18041
|
const drainTimeoutMs = options.drainTimeoutMs ?? DEFAULT_DRAIN_TIMEOUT_MS;
|
|
18042
|
+
const idleUnloadMs = options.idleUnloadMs ?? DEFAULT_IDLE_UNLOAD_MS2;
|
|
17973
18043
|
let resident = null;
|
|
17974
18044
|
const kept = /* @__PURE__ */ new Map();
|
|
17975
18045
|
let serving = null;
|
|
@@ -17978,6 +18048,9 @@ var createScheduler = (options) => {
|
|
|
17978
18048
|
let pumping = false;
|
|
17979
18049
|
let switching = null;
|
|
17980
18050
|
let lastRelease = null;
|
|
18051
|
+
let idleTimer;
|
|
18052
|
+
let idleSweep;
|
|
18053
|
+
let shuttingDown = false;
|
|
17981
18054
|
let drainWaiters = [];
|
|
17982
18055
|
const describe3 = (slot) => ({
|
|
17983
18056
|
modelId: slot.instance.id,
|
|
@@ -18006,6 +18079,7 @@ var createScheduler = (options) => {
|
|
|
18006
18079
|
};
|
|
18007
18080
|
const keepLoadedFor = (instance) => [...kept.values()].filter((slot) => slot.instance.id !== instance.id && slot.instance.runtimeId === instance.runtimeId).map((slot) => slot.adapter.servedModelId(slot.instance));
|
|
18008
18081
|
const acquire = (instance, options2 = {}) => {
|
|
18082
|
+
clearIdle();
|
|
18009
18083
|
return new Promise((resolve2, reject) => {
|
|
18010
18084
|
const waiter = {
|
|
18011
18085
|
instance,
|
|
@@ -18061,6 +18135,9 @@ var createScheduler = (options) => {
|
|
|
18061
18135
|
});
|
|
18062
18136
|
};
|
|
18063
18137
|
const shutdown = async () => {
|
|
18138
|
+
shuttingDown = true;
|
|
18139
|
+
clearIdle();
|
|
18140
|
+
await idleSweep;
|
|
18064
18141
|
for (const waiter of waiters.splice(0)) {
|
|
18065
18142
|
waiter.settled = true;
|
|
18066
18143
|
pending -= 1;
|
|
@@ -18070,7 +18147,7 @@ var createScheduler = (options) => {
|
|
|
18070
18147
|
if (!slot)
|
|
18071
18148
|
continue;
|
|
18072
18149
|
try {
|
|
18073
|
-
await releaseSlot(slot);
|
|
18150
|
+
await releaseSlot(slot, "shutdown");
|
|
18074
18151
|
} catch (error2) {
|
|
18075
18152
|
logger.error("failed to release the resident slot during shutdown", {
|
|
18076
18153
|
event: "slot.release_failed",
|
|
@@ -18082,6 +18159,7 @@ var createScheduler = (options) => {
|
|
|
18082
18159
|
kept.clear();
|
|
18083
18160
|
resident = null;
|
|
18084
18161
|
serving = null;
|
|
18162
|
+
shuttingDown = false;
|
|
18085
18163
|
};
|
|
18086
18164
|
const pump = async () => {
|
|
18087
18165
|
if (pumping)
|
|
@@ -18114,6 +18192,7 @@ var createScheduler = (options) => {
|
|
|
18114
18192
|
pumping = false;
|
|
18115
18193
|
if (waiters.length > 0)
|
|
18116
18194
|
void pump();
|
|
18195
|
+
armIdle();
|
|
18117
18196
|
}
|
|
18118
18197
|
};
|
|
18119
18198
|
const settle = (waiter, action) => {
|
|
@@ -18168,7 +18247,7 @@ var createScheduler = (options) => {
|
|
|
18168
18247
|
if (existing)
|
|
18169
18248
|
await discard(existing);
|
|
18170
18249
|
if (!instance.keepResident && resident && resident.instance.id !== instance.id) {
|
|
18171
|
-
await releaseRotating(resident);
|
|
18250
|
+
await releaseRotating(resident, "switch");
|
|
18172
18251
|
resident = null;
|
|
18173
18252
|
} else if (instance.keepResident && resident) {
|
|
18174
18253
|
logger.info("leaving the current occupant loaded for a kept entry", {
|
|
@@ -18185,7 +18264,8 @@ var createScheduler = (options) => {
|
|
|
18185
18264
|
state: "starting",
|
|
18186
18265
|
ownership: "unknown",
|
|
18187
18266
|
leases: 0,
|
|
18188
|
-
since: Date.now()
|
|
18267
|
+
since: Date.now(),
|
|
18268
|
+
idleSkipped: false
|
|
18189
18269
|
};
|
|
18190
18270
|
const keepLoaded = keepLoadedFor(instance);
|
|
18191
18271
|
if (instance.keepResident)
|
|
@@ -18286,14 +18366,14 @@ var createScheduler = (options) => {
|
|
|
18286
18366
|
slot.state = wasFailed ? "failed" : "ready";
|
|
18287
18367
|
}
|
|
18288
18368
|
};
|
|
18289
|
-
const releaseRotating = async (slot) => {
|
|
18369
|
+
const releaseRotating = async (slot, reason) => {
|
|
18290
18370
|
const wasFailed = slot.state === "failed";
|
|
18291
18371
|
slot.state = "stopping";
|
|
18292
18372
|
if (wasFailed) {
|
|
18293
18373
|
const health = await slot.adapter.health(slot.instance).catch(() => ({ state: "unreachable" }));
|
|
18294
18374
|
if (health.state === "unreachable") {
|
|
18295
18375
|
try {
|
|
18296
|
-
await releaseSlot(slot);
|
|
18376
|
+
await releaseSlot(slot, reason);
|
|
18297
18377
|
} catch (error2) {
|
|
18298
18378
|
logger.warn("ignoring release failure for a runtime that is already gone", {
|
|
18299
18379
|
event: "slot.release_after_crash",
|
|
@@ -18301,14 +18381,14 @@ var createScheduler = (options) => {
|
|
|
18301
18381
|
adapter: slot.adapter.id,
|
|
18302
18382
|
error: error2.message
|
|
18303
18383
|
});
|
|
18304
|
-
lastRelease = { modelId: slot.instance.id, via: slot.adapter.modelRelease };
|
|
18384
|
+
lastRelease = { modelId: slot.instance.id, via: slot.adapter.modelRelease, reason };
|
|
18305
18385
|
}
|
|
18306
18386
|
if (serving === slot)
|
|
18307
18387
|
serving = null;
|
|
18308
18388
|
return;
|
|
18309
18389
|
}
|
|
18310
18390
|
}
|
|
18311
|
-
await releaseSlot(slot);
|
|
18391
|
+
await releaseSlot(slot, reason);
|
|
18312
18392
|
if (serving === slot)
|
|
18313
18393
|
serving = null;
|
|
18314
18394
|
};
|
|
@@ -18318,7 +18398,7 @@ var createScheduler = (options) => {
|
|
|
18318
18398
|
resident = null;
|
|
18319
18399
|
if (serving === slot)
|
|
18320
18400
|
serving = null;
|
|
18321
|
-
await releaseRotating(slot).catch((error2) => {
|
|
18401
|
+
await releaseRotating(slot, "switch").catch((error2) => {
|
|
18322
18402
|
logger.warn("could not release a runtime before rebuilding it", {
|
|
18323
18403
|
event: "slot.release_failed",
|
|
18324
18404
|
runtime: slot.instance.id,
|
|
@@ -18327,7 +18407,7 @@ var createScheduler = (options) => {
|
|
|
18327
18407
|
});
|
|
18328
18408
|
});
|
|
18329
18409
|
};
|
|
18330
|
-
const releaseSlot = async (slot) => {
|
|
18410
|
+
const releaseSlot = async (slot, reason) => {
|
|
18331
18411
|
const log = logger.child({
|
|
18332
18412
|
runtime: slot.instance.id,
|
|
18333
18413
|
adapter: slot.adapter.id,
|
|
@@ -18344,13 +18424,84 @@ var createScheduler = (options) => {
|
|
|
18344
18424
|
throw wrapReleaseError(error2, slot);
|
|
18345
18425
|
}
|
|
18346
18426
|
slot.state = "stopped";
|
|
18347
|
-
lastRelease = { modelId: slot.instance.id, via: slot.adapter.modelRelease };
|
|
18427
|
+
lastRelease = { modelId: slot.instance.id, via: slot.adapter.modelRelease, reason };
|
|
18348
18428
|
log.info("resident slot freed", {
|
|
18349
18429
|
event: "slot.released",
|
|
18350
18430
|
via: slot.adapter.modelRelease,
|
|
18431
|
+
reason,
|
|
18351
18432
|
durationMs: Date.now() - startedAt
|
|
18352
18433
|
});
|
|
18353
18434
|
};
|
|
18435
|
+
const idleReleasable = (slot) => slot.adapter.modelRelease === "unload_model" || slot.ownership === "spawned";
|
|
18436
|
+
const clearIdle = () => {
|
|
18437
|
+
if (idleTimer === void 0)
|
|
18438
|
+
return;
|
|
18439
|
+
clearTimeout(idleTimer);
|
|
18440
|
+
idleTimer = void 0;
|
|
18441
|
+
};
|
|
18442
|
+
const armIdle = () => {
|
|
18443
|
+
clearIdle();
|
|
18444
|
+
if (idleUnloadMs <= 0 || shuttingDown)
|
|
18445
|
+
return;
|
|
18446
|
+
if (pumping || pending > 0 || waiters.length > 0)
|
|
18447
|
+
return;
|
|
18448
|
+
if (!resident || resident.idleSkipped || resident.leases > 0)
|
|
18449
|
+
return;
|
|
18450
|
+
for (const slot of kept.values())
|
|
18451
|
+
if (slot.leases > 0)
|
|
18452
|
+
return;
|
|
18453
|
+
idleTimer = setTimeout(() => {
|
|
18454
|
+
idleSweep = sweepIdle().finally(() => {
|
|
18455
|
+
idleSweep = void 0;
|
|
18456
|
+
});
|
|
18457
|
+
}, idleUnloadMs);
|
|
18458
|
+
idleTimer.unref();
|
|
18459
|
+
};
|
|
18460
|
+
const sweepIdle = async () => {
|
|
18461
|
+
idleTimer = void 0;
|
|
18462
|
+
if (pumping || waiters.length > 0 || shuttingDown)
|
|
18463
|
+
return;
|
|
18464
|
+
const slot = resident;
|
|
18465
|
+
if (!slot)
|
|
18466
|
+
return;
|
|
18467
|
+
pumping = true;
|
|
18468
|
+
try {
|
|
18469
|
+
if (slot.leases > 0)
|
|
18470
|
+
return;
|
|
18471
|
+
if (!idleReleasable(slot)) {
|
|
18472
|
+
slot.idleSkipped = true;
|
|
18473
|
+
logger.info(`leaving the foreign ${slot.adapter.id} server on :${slot.instance.port ?? "?"} loaded: only a switch may stop a server the gateway did not start`, {
|
|
18474
|
+
event: "slot.idle_skipped",
|
|
18475
|
+
runtime: slot.instance.id,
|
|
18476
|
+
adapter: slot.adapter.id,
|
|
18477
|
+
ownership: slot.ownership
|
|
18478
|
+
});
|
|
18479
|
+
return;
|
|
18480
|
+
}
|
|
18481
|
+
logger.info("releasing the resident slot after an idle period", {
|
|
18482
|
+
event: "slot.idle_release",
|
|
18483
|
+
runtime: slot.instance.id,
|
|
18484
|
+
adapter: slot.adapter.id,
|
|
18485
|
+
idleMs: idleUnloadMs
|
|
18486
|
+
});
|
|
18487
|
+
await releaseRotating(slot, "idle");
|
|
18488
|
+
if (resident === slot)
|
|
18489
|
+
resident = null;
|
|
18490
|
+
} catch (error2) {
|
|
18491
|
+
logger.warn("idle release failed; leaving the runtime loaded", {
|
|
18492
|
+
event: "slot.idle_release_failed",
|
|
18493
|
+
runtime: slot.instance.id,
|
|
18494
|
+
adapter: slot.adapter.id,
|
|
18495
|
+
error: error2.message
|
|
18496
|
+
});
|
|
18497
|
+
} finally {
|
|
18498
|
+
pumping = false;
|
|
18499
|
+
if (waiters.length > 0)
|
|
18500
|
+
void pump();
|
|
18501
|
+
else
|
|
18502
|
+
armIdle();
|
|
18503
|
+
}
|
|
18504
|
+
};
|
|
18354
18505
|
const waitForDrain = async (slot) => {
|
|
18355
18506
|
if (slot.leases <= 0)
|
|
18356
18507
|
return;
|
|
@@ -18404,7 +18555,8 @@ var createDockService = (options) => {
|
|
|
18404
18555
|
registry: registry2,
|
|
18405
18556
|
logger,
|
|
18406
18557
|
readyTimeoutMs: options.readyTimeoutMs,
|
|
18407
|
-
drainTimeoutMs: options.drainTimeoutMs
|
|
18558
|
+
drainTimeoutMs: options.drainTimeoutMs,
|
|
18559
|
+
...options.idleUnloadMs !== void 0 ? { idleUnloadMs: options.idleUnloadMs } : {}
|
|
18408
18560
|
});
|
|
18409
18561
|
const resolve2 = (clientModelId) => resolveModel(config2, clientModelId);
|
|
18410
18562
|
const status = () => ({
|
|
@@ -21865,10 +22017,12 @@ var runServe = async (context, options = {}) => {
|
|
|
21865
22017
|
logger: context.logger,
|
|
21866
22018
|
...debugDir ? { dir: debugDir } : {}
|
|
21867
22019
|
}) : null;
|
|
22020
|
+
const idleUnloadMs = options.idleUnload === void 0 ? config2.server.idleUnloadMs : parseDuration(options.idleUnload, "--idle-unload");
|
|
21868
22021
|
const service = createDockService({
|
|
21869
22022
|
config: config2,
|
|
21870
22023
|
registry: context.adapters,
|
|
21871
22024
|
logger: context.logger,
|
|
22025
|
+
idleUnloadMs,
|
|
21872
22026
|
...tap ? { tap } : {}
|
|
21873
22027
|
});
|
|
21874
22028
|
const gateway = await startGateway({
|
|
@@ -21880,13 +22034,14 @@ var runServe = async (context, options = {}) => {
|
|
|
21880
22034
|
});
|
|
21881
22035
|
if (!context.options.json) {
|
|
21882
22036
|
const theme = context.theme;
|
|
21883
|
-
const labels = ["config", "endpoint", "models"];
|
|
22037
|
+
const labels = ["config", "endpoint", "models", "idle"];
|
|
21884
22038
|
const width = labelWidth(labels);
|
|
21885
22039
|
const line = (label, value) => context.out(keyValue(label, value, width, { label: theme.label }));
|
|
21886
22040
|
const models = servedModelIds(config2);
|
|
21887
22041
|
line("config", theme.path(config2.location.path));
|
|
21888
22042
|
line("endpoint", theme.url(gateway.url));
|
|
21889
22043
|
line("models", models.length > 0 ? models.map((id) => theme.id(id)).join(", ") : theme.muted("(none)"));
|
|
22044
|
+
line("idle", idleUnloadMs > 0 ? `${theme.id(formatDuration(idleUnloadMs))} ${theme.muted("then unload")}` : theme.muted("off"));
|
|
21890
22045
|
}
|
|
21891
22046
|
if (tap) {
|
|
21892
22047
|
context.err(`${context.theme.label("debug capture:")} ${context.theme.path(tap.path)}`);
|
|
@@ -22129,6 +22284,7 @@ var printStatus = (context, endpoint, status) => {
|
|
|
22129
22284
|
"state",
|
|
22130
22285
|
"kept",
|
|
22131
22286
|
"serving",
|
|
22287
|
+
"released",
|
|
22132
22288
|
"queue"
|
|
22133
22289
|
]);
|
|
22134
22290
|
const line = (label, value, style) => context.out(keyValue(label, value, width, { label: theme.label, value: style }));
|
|
@@ -22142,6 +22298,9 @@ var printStatus = (context, endpoint, status) => {
|
|
|
22142
22298
|
line("state", status.resident.state, theme.state);
|
|
22143
22299
|
} else {
|
|
22144
22300
|
line("resident", "none", theme.muted);
|
|
22301
|
+
if (status.lastRelease) {
|
|
22302
|
+
line("released", `${theme.id(status.lastRelease.modelId)} (${status.lastRelease.reason}, ${status.lastRelease.via})`);
|
|
22303
|
+
}
|
|
22145
22304
|
}
|
|
22146
22305
|
if (status.kept.length > 0) {
|
|
22147
22306
|
for (const entry of status.kept) {
|
|
@@ -22208,7 +22367,7 @@ var runCli = async (options) => {
|
|
|
22208
22367
|
};
|
|
22209
22368
|
return createCliContext({ ...options, version: version2 }, merged);
|
|
22210
22369
|
};
|
|
22211
|
-
shared(program2.command("serve")).description("start the gateway").option("--host <host>", "bind address (default: from configuration)").option("--port <port>", "bind port (default: from configuration)", (value) => Number(value)).option("--debug", "capture every proxied request and response to an NDJSON file").option("--debug-dir <path>", "where --debug writes (default: a temp subdirectory)").action(async (local, command) => {
|
|
22370
|
+
shared(program2.command("serve")).description("start the gateway").option("--host <host>", "bind address (default: from configuration)").option("--port <port>", "bind port (default: from configuration)", (value) => Number(value)).option("--debug", "capture every proxied request and response to an NDJSON file").option("--debug-dir <path>", "where --debug writes (default: a temp subdirectory)").option("--idle-unload <duration>", "release the loaded model after this long with nothing to do, e.g. 60m or 0 for never").action(async (local, command) => {
|
|
22212
22371
|
await runServe(contextFor(command), local);
|
|
22213
22372
|
});
|
|
22214
22373
|
shared(program2.command("status"), true).description("ask a running gateway what it is doing").action(async (_local, command) => {
|
|
@@ -23157,10 +23316,22 @@ var createMtplxAdapter = (options = {}) => {
|
|
|
23157
23316
|
return (parsed.models ?? []).filter((entry) => typeof entry.repo_id === "string").map((entry) => ({
|
|
23158
23317
|
id: entry.repo_id,
|
|
23159
23318
|
suggestedId: suggestLogicalId(entry.repo_id),
|
|
23160
|
-
//
|
|
23161
|
-
//
|
|
23162
|
-
//
|
|
23163
|
-
|
|
23319
|
+
// `unusable` means the runtime says it will not load this, and the
|
|
23320
|
+
// only field here that says so is `has_config` — MTPLX reads it as
|
|
23321
|
+
// `(dir / 'config.json').exists()`, i.e. the directory is not a model.
|
|
23322
|
+
//
|
|
23323
|
+
// Deliberately *not* `validation.ok`: that field is the MTP runtime
|
|
23324
|
+
// contract, and this adapter used to read it as servability. It is
|
|
23325
|
+
// not. MTPLX's own launch gate says so — "the gate asks one question,
|
|
23326
|
+
// can this artifact execute; verification tier stays a label, it must
|
|
23327
|
+
// never block loading" — and a pack with no contract degrades to
|
|
23328
|
+
// target-only AR on its own (`native-ar-only-missing-mtp` flips
|
|
23329
|
+
// `--generation-mode` to `ar` and loads). Skipping those hid two
|
|
23330
|
+
// working models, so a missing contract is slower, never unusable.
|
|
23331
|
+
//
|
|
23332
|
+
// `=== false`, not `!entry.has_config`: an older `mtplx` that omits
|
|
23333
|
+
// the field must not have its whole catalogue declared unusable.
|
|
23334
|
+
...entry.has_config === false ? { unusable: "MTPLX reports no config.json in this model pack" } : {}
|
|
23164
23335
|
}));
|
|
23165
23336
|
} catch {
|
|
23166
23337
|
return [];
|
|
@@ -25092,13 +25263,17 @@ var createCodexIntegration = (options = {}) => {
|
|
|
25092
25263
|
}
|
|
25093
25264
|
const model = plan.roles.model;
|
|
25094
25265
|
const providers = isRecord2(existing.model_providers) ? { ...existing.model_providers } : {};
|
|
25266
|
+
const previous = isRecord2(providers[PROVIDER_ID]) ? providers[PROVIDER_ID] : {};
|
|
25095
25267
|
const provider = {
|
|
25268
|
+
...previous,
|
|
25096
25269
|
name: PROVIDER_ID,
|
|
25097
25270
|
base_url: `${plan.gatewayBaseUrl}/v1`
|
|
25098
25271
|
};
|
|
25099
25272
|
const mapped = plan.models.find((entry) => entry.id === model);
|
|
25100
25273
|
if (mapped?.apiKeyEnv) {
|
|
25101
25274
|
provider.env_key = mapped.apiKeyEnv;
|
|
25275
|
+
} else {
|
|
25276
|
+
delete provider.env_key;
|
|
25102
25277
|
}
|
|
25103
25278
|
providers[PROVIDER_ID] = provider;
|
|
25104
25279
|
const merged = { ...existing };
|
|
@@ -26544,6 +26719,7 @@ function applyEdits(text, edits) {
|
|
|
26544
26719
|
// packages/agents/opencode/dist/index.js
|
|
26545
26720
|
var PROVIDER_ID2 = "llm-runtime-dock";
|
|
26546
26721
|
var PROVIDER_NAME = "LLM Runtime Dock";
|
|
26722
|
+
var ENV_REFERENCE = /^\{env:[^}]+\}$/;
|
|
26547
26723
|
var createOpenCodeIntegration = (options = {}) => {
|
|
26548
26724
|
const id = "opencode";
|
|
26549
26725
|
const displayName = "OpenCode";
|
|
@@ -26559,49 +26735,54 @@ var createOpenCodeIntegration = (options = {}) => {
|
|
|
26559
26735
|
const isInstalled = async () => {
|
|
26560
26736
|
return existsSync6(configDir);
|
|
26561
26737
|
};
|
|
26738
|
+
const setPath = (content, path2, value) => {
|
|
26739
|
+
let next = content;
|
|
26740
|
+
const formatting = { formattingOptions: { insertSpaces: true, tabSize: 2 } };
|
|
26741
|
+
for (const edit of modify(next, path2, value, formatting)) {
|
|
26742
|
+
next = applyEdits(next, [edit]);
|
|
26743
|
+
}
|
|
26744
|
+
return next;
|
|
26745
|
+
};
|
|
26562
26746
|
const render = async (plan) => {
|
|
26563
26747
|
const path2 = configPath();
|
|
26564
26748
|
const source = plan.existing ?? "{}";
|
|
26565
26749
|
const warnings = [];
|
|
26566
26750
|
const errors = [];
|
|
26567
|
-
parse5(source, errors, { allowTrailingComma: true });
|
|
26751
|
+
const parsed = parse5(source, errors, { allowTrailingComma: true });
|
|
26568
26752
|
if (errors.length > 0) {
|
|
26569
26753
|
const first = errors[0];
|
|
26570
26754
|
throw cliError("AGENT_CONFIG_UNREADABLE", `${path2} could not be parsed (${first ? printParseErrorCode(first.error) : "unknown error"} at offset ${first?.offset ?? 0})`, { details: { agent: id, path: path2 }, hint: "fix the file by hand; nothing was written" });
|
|
26571
26755
|
}
|
|
26756
|
+
const previous = providerBlock(parsed);
|
|
26572
26757
|
const defaultModel = plan.roles.default;
|
|
26573
|
-
const
|
|
26574
|
-
|
|
26575
|
-
|
|
26576
|
-
|
|
26577
|
-
|
|
26578
|
-
if (model.outputLimit !== void 0)
|
|
26579
|
-
limit.output = model.outputLimit;
|
|
26580
|
-
models[model.id] = {
|
|
26581
|
-
name: model.name,
|
|
26582
|
-
...Object.keys(limit).length > 0 ? { limit } : {}
|
|
26583
|
-
};
|
|
26584
|
-
}
|
|
26758
|
+
const base = ["provider", PROVIDER_ID2];
|
|
26759
|
+
let content = source.trim() === "" ? "{}" : source;
|
|
26760
|
+
content = setPath(content, [...base, "npm"], "@ai-sdk/openai-compatible");
|
|
26761
|
+
content = setPath(content, [...base, "name"], PROVIDER_NAME);
|
|
26762
|
+
content = setPath(content, [...base, "options", "baseURL"], `${plan.gatewayBaseUrl}/v1`);
|
|
26585
26763
|
const secretModel = plan.models.find((model) => model.apiKeyEnv !== void 0);
|
|
26586
|
-
const
|
|
26764
|
+
const existingKey = previousApiKey(previous);
|
|
26587
26765
|
if (secretModel?.apiKeyEnv) {
|
|
26588
|
-
|
|
26766
|
+
content = setPath(content, [...base, "options", "apiKey"], `{env:${secretModel.apiKeyEnv}}`);
|
|
26767
|
+
} else if (typeof existingKey === "string" && ENV_REFERENCE.test(existingKey)) {
|
|
26768
|
+
content = setPath(content, [...base, "options", "apiKey"], void 0);
|
|
26589
26769
|
}
|
|
26590
|
-
const
|
|
26591
|
-
|
|
26592
|
-
|
|
26593
|
-
|
|
26594
|
-
|
|
26595
|
-
|
|
26596
|
-
|
|
26597
|
-
|
|
26598
|
-
|
|
26599
|
-
|
|
26770
|
+
for (const model of plan.models) {
|
|
26771
|
+
content = setPath(content, [...base, "models", model.id, "name"], model.name);
|
|
26772
|
+
if (model.contextLimit !== void 0) {
|
|
26773
|
+
content = setPath(content, [...base, "models", model.id, "limit", "context"], model.contextLimit);
|
|
26774
|
+
}
|
|
26775
|
+
if (model.outputLimit !== void 0) {
|
|
26776
|
+
content = setPath(content, [...base, "models", model.id, "limit", "output"], model.outputLimit);
|
|
26777
|
+
}
|
|
26778
|
+
}
|
|
26779
|
+
const served = new Set(plan.models.map((model) => model.id));
|
|
26780
|
+
for (const stale of Object.keys(previousModels(previous))) {
|
|
26781
|
+
if (!served.has(stale))
|
|
26782
|
+
content = setPath(content, [...base, "models", stale], void 0);
|
|
26600
26783
|
}
|
|
26601
26784
|
if (defaultModel) {
|
|
26602
|
-
|
|
26603
|
-
content = applyEdits(content, [edit]);
|
|
26604
|
-
}
|
|
26785
|
+
content = setPath(content, ["model"], `${PROVIDER_ID2}/${defaultModel}`);
|
|
26605
26786
|
}
|
|
26606
26787
|
if (!content.endsWith("\n"))
|
|
26607
26788
|
content += "\n";
|
|
@@ -26619,6 +26800,21 @@ var createOpenCodeIntegration = (options = {}) => {
|
|
|
26619
26800
|
render
|
|
26620
26801
|
};
|
|
26621
26802
|
};
|
|
26803
|
+
var isRecord3 = (value) => {
|
|
26804
|
+
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
26805
|
+
};
|
|
26806
|
+
var providerBlock = (parsed) => {
|
|
26807
|
+
if (!isRecord3(parsed) || !isRecord3(parsed.provider))
|
|
26808
|
+
return {};
|
|
26809
|
+
const block = parsed.provider[PROVIDER_ID2];
|
|
26810
|
+
return isRecord3(block) ? block : {};
|
|
26811
|
+
};
|
|
26812
|
+
var previousModels = (block) => {
|
|
26813
|
+
return isRecord3(block.models) ? block.models : {};
|
|
26814
|
+
};
|
|
26815
|
+
var previousApiKey = (block) => {
|
|
26816
|
+
return isRecord3(block.options) ? block.options.apiKey : void 0;
|
|
26817
|
+
};
|
|
26622
26818
|
|
|
26623
26819
|
// src/index.ts
|
|
26624
26820
|
var defaultAdapters = (logger) => {
|
|
@@ -26638,7 +26834,7 @@ var main = async (argv = process.argv) => {
|
|
|
26638
26834
|
argv: stripArgSeparator(argv.slice(2)),
|
|
26639
26835
|
adapters: defaultAdapters(),
|
|
26640
26836
|
agents: defaultAgents(),
|
|
26641
|
-
version: "0.1.
|
|
26837
|
+
version: "0.1.1"
|
|
26642
26838
|
});
|
|
26643
26839
|
if (code !== 0) process.exit(code);
|
|
26644
26840
|
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "llm-runtime-dock",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.1",
|
|
4
4
|
"description": "Local OpenAI/Anthropic-compatible LLM gateway and runtime lifecycle manager",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "BartRuSec",
|
|
@@ -16,7 +16,6 @@
|
|
|
16
16
|
"engines": {
|
|
17
17
|
"node": ">=24.0.0"
|
|
18
18
|
},
|
|
19
|
-
"packageManager": "pnpm@11.15.1",
|
|
20
19
|
"bin": {
|
|
21
20
|
"lrd": "./bin/lrd.mjs"
|
|
22
21
|
},
|
|
@@ -44,11 +43,34 @@
|
|
|
44
43
|
"ollama",
|
|
45
44
|
"coding-agents"
|
|
46
45
|
],
|
|
46
|
+
"devDependencies": {
|
|
47
|
+
"@eslint/js": "^10.0.1",
|
|
48
|
+
"@types/node": "^24",
|
|
49
|
+
"esbuild": "^0.28.2",
|
|
50
|
+
"eslint": "^10.9.1",
|
|
51
|
+
"prettier": "^3.4.2",
|
|
52
|
+
"rimraf": "^6.0.1",
|
|
53
|
+
"typescript": "^6.0.3",
|
|
54
|
+
"typescript-eslint": "^8.18.1",
|
|
55
|
+
"vitest": "^4.1.11",
|
|
56
|
+
"@llm-runtime-dock/adapter-custom": "0.1.1",
|
|
57
|
+
"@llm-runtime-dock/adapter-lm-studio": "0.1.1",
|
|
58
|
+
"@llm-runtime-dock/adapter-omlx": "0.1.1",
|
|
59
|
+
"@llm-runtime-dock/adapter-ollama": "0.1.1",
|
|
60
|
+
"@llm-runtime-dock/agent-codex": "0.1.1",
|
|
61
|
+
"@llm-runtime-dock/agent-opencode": "0.1.1",
|
|
62
|
+
"@llm-runtime-dock/core": "0.1.1",
|
|
63
|
+
"@llm-runtime-dock/adapter-mtplx": "0.1.1",
|
|
64
|
+
"@llm-runtime-dock/cli": "0.1.1",
|
|
65
|
+
"@llm-runtime-dock/agent-claude": "0.1.1",
|
|
66
|
+
"@llm-runtime-dock/gateway": "0.1.1"
|
|
67
|
+
},
|
|
47
68
|
"scripts": {
|
|
48
69
|
"build": "node scripts/sync-version.mjs && pnpm -r run build && node scripts/bundle.mjs",
|
|
49
70
|
"bundle": "node scripts/bundle.mjs",
|
|
50
71
|
"version:sync": "node scripts/sync-version.mjs",
|
|
51
|
-
"
|
|
72
|
+
"version:bump": "node scripts/bump-version.mjs",
|
|
73
|
+
"release:github": "node scripts/release-github.mjs",
|
|
52
74
|
"dev": "node scripts/dev.mjs",
|
|
53
75
|
"clean": "pnpm -r run clean && rimraf dist",
|
|
54
76
|
"typecheck": "node scripts/sync-version.mjs --check && pnpm -r run typecheck && tsc -p tsconfig.json",
|
|
@@ -60,27 +82,5 @@
|
|
|
60
82
|
"format:check": "prettier --check .",
|
|
61
83
|
"test:unit": "pnpm -r run test",
|
|
62
84
|
"test:e2e": "vitest run"
|
|
63
|
-
},
|
|
64
|
-
"devDependencies": {
|
|
65
|
-
"@eslint/js": "^10.0.1",
|
|
66
|
-
"@llm-runtime-dock/adapter-custom": "workspace:*",
|
|
67
|
-
"@llm-runtime-dock/adapter-lm-studio": "workspace:*",
|
|
68
|
-
"@llm-runtime-dock/adapter-mtplx": "workspace:*",
|
|
69
|
-
"@llm-runtime-dock/adapter-omlx": "workspace:*",
|
|
70
|
-
"@llm-runtime-dock/adapter-ollama": "workspace:*",
|
|
71
|
-
"@llm-runtime-dock/agent-claude": "workspace:*",
|
|
72
|
-
"@llm-runtime-dock/agent-codex": "workspace:*",
|
|
73
|
-
"@llm-runtime-dock/agent-opencode": "workspace:*",
|
|
74
|
-
"@llm-runtime-dock/cli": "workspace:*",
|
|
75
|
-
"@llm-runtime-dock/core": "workspace:*",
|
|
76
|
-
"@llm-runtime-dock/gateway": "workspace:*",
|
|
77
|
-
"@types/node": "^24",
|
|
78
|
-
"esbuild": "^0.28.2",
|
|
79
|
-
"eslint": "^10.9.1",
|
|
80
|
-
"prettier": "^3.4.2",
|
|
81
|
-
"rimraf": "^6.0.1",
|
|
82
|
-
"typescript": "^6.0.3",
|
|
83
|
-
"typescript-eslint": "^8.18.1",
|
|
84
|
-
"vitest": "^4.1.11"
|
|
85
85
|
}
|
|
86
|
-
}
|
|
86
|
+
}
|