@infersec/conduit 1.112.0 → 1.113.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/apiClient/index.d.ts +2 -1
- package/dist/cli.js +512 -59
- package/dist/cli.sea.cjs +512 -59
- package/dist/modelManagement/ModelManager.d.ts +6 -0
- package/dist/modelManagement/classifyEngineError.d.ts +12 -0
- package/dist/reporting/engineExecutionReporter.d.ts +54 -0
- package/dist/reporting/index.d.ts +1 -0
- package/dist/requestHandlers/createConduitAnthropicAPIReferenceHandlers.d.ts +1 -1
- package/dist/utils/machineInfo.d.ts +18 -0
- package/package.json +1 -1
package/dist/cli.sea.cjs
CHANGED
|
@@ -19986,6 +19986,32 @@ object$5({
|
|
|
19986
19986
|
sizeBytes: number$1().int().nonnegative().nullable()
|
|
19987
19987
|
});
|
|
19988
19988
|
|
|
19989
|
+
const EngineExecutionErrorTypeSchema = _enum$1([
|
|
19990
|
+
"config",
|
|
19991
|
+
"crash",
|
|
19992
|
+
"oom",
|
|
19993
|
+
"prompt",
|
|
19994
|
+
"unknown"
|
|
19995
|
+
]);
|
|
19996
|
+
const EngineExecutionReportPayloadSchema = object$5({
|
|
19997
|
+
avgTps: number$1().nonnegative().finite().default(0),
|
|
19998
|
+
completionTokens: number$1().int().nonnegative().default(0),
|
|
19999
|
+
durationMs: number$1().int().nonnegative().default(0),
|
|
20000
|
+
engineType: LLMEngineSchema.nullable(),
|
|
20001
|
+
engineVersion: string$2().max(64).nullable().default(null),
|
|
20002
|
+
errorDetail: string$2().max(2048).nullable().default(null),
|
|
20003
|
+
errorType: EngineExecutionErrorTypeSchema.nullable(),
|
|
20004
|
+
extraArgs: array$1(tuple([string$2().min(1).max(128), string$2().max(512)]))
|
|
20005
|
+
.max(256)
|
|
20006
|
+
.default([]),
|
|
20007
|
+
finishedAtISO: string$2().datetime({ offset: true }),
|
|
20008
|
+
peakTps: number$1().nonnegative().finite().nullable().default(null),
|
|
20009
|
+
promptTokens: number$1().int().nonnegative().default(0),
|
|
20010
|
+
runAtISO: string$2().datetime({ offset: true }),
|
|
20011
|
+
success: boolean$1(),
|
|
20012
|
+
ttftMs: number$1().int().nonnegative().default(0),
|
|
20013
|
+
totalTokens: number$1().int().nonnegative().default(0)
|
|
20014
|
+
});
|
|
19989
20015
|
const InferenceAgentLLMMetricsPayloadSchema = object$5({
|
|
19990
20016
|
bytes: number$1().int().nonnegative(),
|
|
19991
20017
|
completionTokens: number$1().int().nonnegative(),
|
|
@@ -20324,6 +20350,23 @@ const API_SERVICE_CONDUIT_API_REFERENCE = {
|
|
|
20324
20350
|
}
|
|
20325
20351
|
}
|
|
20326
20352
|
},
|
|
20353
|
+
"/conduit/api/v1/source/:sourceID/engine/execution": {
|
|
20354
|
+
POST: {
|
|
20355
|
+
auth: {
|
|
20356
|
+
type: "api-key"
|
|
20357
|
+
},
|
|
20358
|
+
body: EngineExecutionReportPayloadSchema,
|
|
20359
|
+
parameters: {
|
|
20360
|
+
sourceID: ULIDSchema
|
|
20361
|
+
},
|
|
20362
|
+
response: {
|
|
20363
|
+
schema: object$5({
|
|
20364
|
+
acknowledged: literal(true)
|
|
20365
|
+
}),
|
|
20366
|
+
type: "rest"
|
|
20367
|
+
}
|
|
20368
|
+
}
|
|
20369
|
+
},
|
|
20327
20370
|
"/conduit/api/v1/source/:sourceID/requests/:requestID/chunk": {
|
|
20328
20371
|
POST: {
|
|
20329
20372
|
auth: {
|
|
@@ -113781,6 +113824,19 @@ function createAPIClient({ apiKey, apiURL, inferenceSourceID, logger }) {
|
|
|
113781
113824
|
route: "/conduit/api/v1/source/:sourceID/state"
|
|
113782
113825
|
});
|
|
113783
113826
|
},
|
|
113827
|
+
reportEngineExecution: async (payload) => {
|
|
113828
|
+
await fetchByReference({
|
|
113829
|
+
baseURL: apiURL,
|
|
113830
|
+
body: payload,
|
|
113831
|
+
fetch: fetchWithAPIKey,
|
|
113832
|
+
method: "POST",
|
|
113833
|
+
parameters: {
|
|
113834
|
+
sourceID: inferenceSourceID
|
|
113835
|
+
},
|
|
113836
|
+
reference: API_SERVICE_CONDUIT_API_REFERENCE,
|
|
113837
|
+
route: "/conduit/api/v1/source/:sourceID/engine/execution"
|
|
113838
|
+
});
|
|
113839
|
+
},
|
|
113784
113840
|
reportPromptMetrics: async (payload) => {
|
|
113785
113841
|
await fetchByReference({
|
|
113786
113842
|
baseURL: apiURL,
|
|
@@ -125641,6 +125697,9 @@ class ModelManager extends EventEmitter {
|
|
|
125641
125697
|
lifecycleState = "stopped";
|
|
125642
125698
|
downloadLockHandle = null;
|
|
125643
125699
|
stopRequested = false;
|
|
125700
|
+
lastEngineExitCode = null;
|
|
125701
|
+
lastEngineExitSignal = null;
|
|
125702
|
+
reachedRunningState = false;
|
|
125644
125703
|
modelsDirectory;
|
|
125645
125704
|
constructor({ contextLength, engineConfig, enginePort, engineType, logger, model, root }) {
|
|
125646
125705
|
super();
|
|
@@ -125763,6 +125822,9 @@ class ModelManager extends EventEmitter {
|
|
|
125763
125822
|
this.lifecycleState = "starting";
|
|
125764
125823
|
this.lastEngineError = null;
|
|
125765
125824
|
this.stopRequested = false;
|
|
125825
|
+
this.lastEngineExitCode = null;
|
|
125826
|
+
this.lastEngineExitSignal = null;
|
|
125827
|
+
this.reachedRunningState = false;
|
|
125766
125828
|
this.logger.info("Starting LLM engine", {
|
|
125767
125829
|
agentEngineType: this.engine
|
|
125768
125830
|
});
|
|
@@ -125793,6 +125855,7 @@ class ModelManager extends EventEmitter {
|
|
|
125793
125855
|
throw err;
|
|
125794
125856
|
}
|
|
125795
125857
|
this.lifecycleState = "running";
|
|
125858
|
+
this.reachedRunningState = true;
|
|
125796
125859
|
this.emit("engineReady");
|
|
125797
125860
|
}
|
|
125798
125861
|
async stop() {
|
|
@@ -125817,6 +125880,7 @@ class ModelManager extends EventEmitter {
|
|
|
125817
125880
|
this.lifecycleState = "stopped";
|
|
125818
125881
|
return;
|
|
125819
125882
|
}
|
|
125883
|
+
this.reachedRunningState = false;
|
|
125820
125884
|
this.lifecycleState = "stopping";
|
|
125821
125885
|
this.stopRequested = true;
|
|
125822
125886
|
await processManager.stop();
|
|
@@ -125829,9 +125893,18 @@ class ModelManager extends EventEmitter {
|
|
|
125829
125893
|
this.lifecycleState === "starting" ||
|
|
125830
125894
|
this.lifecycleState === "errored");
|
|
125831
125895
|
}
|
|
125896
|
+
get lastExitCode() {
|
|
125897
|
+
return this.lastEngineExitCode;
|
|
125898
|
+
}
|
|
125899
|
+
get lastExitSignal() {
|
|
125900
|
+
return this.lastEngineExitSignal;
|
|
125901
|
+
}
|
|
125832
125902
|
get state() {
|
|
125833
125903
|
return this.lifecycleState;
|
|
125834
125904
|
}
|
|
125905
|
+
get wasRunning() {
|
|
125906
|
+
return this.reachedRunningState;
|
|
125907
|
+
}
|
|
125835
125908
|
async checkEngineReadiness() {
|
|
125836
125909
|
switch (this.engine) {
|
|
125837
125910
|
case "llama.cpp": {
|
|
@@ -125945,6 +126018,7 @@ class ModelManager extends EventEmitter {
|
|
|
125945
126018
|
if (readiness === "ready") {
|
|
125946
126019
|
this.clearHealthPoll();
|
|
125947
126020
|
this.lifecycleState = "running";
|
|
126021
|
+
this.reachedRunningState = true;
|
|
125948
126022
|
this.emit("engineReady");
|
|
125949
126023
|
}
|
|
125950
126024
|
})
|
|
@@ -126007,6 +126081,8 @@ class ModelManager extends EventEmitter {
|
|
|
126007
126081
|
}));
|
|
126008
126082
|
});
|
|
126009
126083
|
processManager.on("stopped", (code, signal) => {
|
|
126084
|
+
this.lastEngineExitCode = code;
|
|
126085
|
+
this.lastEngineExitSignal = signal;
|
|
126010
126086
|
if (hasTerminated) {
|
|
126011
126087
|
return;
|
|
126012
126088
|
}
|
|
@@ -126074,6 +126150,69 @@ class ModelManager extends EventEmitter {
|
|
|
126074
126150
|
}
|
|
126075
126151
|
}
|
|
126076
126152
|
|
|
126153
|
+
// Ordered most-specific first: a message mentioning a prompt-size failure should classify as
|
|
126154
|
+
// "prompt" even if it also touches memory text; memory outranks config because OOM kills frequently
|
|
126155
|
+
// emit sparse stderr. These are heuristics, not exhaustively enumerated engine vocabularies.
|
|
126156
|
+
const OOM_PATTERNS = [
|
|
126157
|
+
/CUDA out of memory/i,
|
|
126158
|
+
/No available memory for the cache blocks/i,
|
|
126159
|
+
/\bOOMKilled\b/,
|
|
126160
|
+
/out of memory/i,
|
|
126161
|
+
/MemoryError/i,
|
|
126162
|
+
/Cannot allocate memory/i
|
|
126163
|
+
];
|
|
126164
|
+
const PROMPT_PATTERNS = [
|
|
126165
|
+
/prompt is too long/i,
|
|
126166
|
+
/maximum context length exceeded/i,
|
|
126167
|
+
/too many tokens/i
|
|
126168
|
+
];
|
|
126169
|
+
const CONFIG_PATTERNS = [
|
|
126170
|
+
/unrecognized argument/i,
|
|
126171
|
+
/invalid argument/i,
|
|
126172
|
+
/error while loading state_dict/i,
|
|
126173
|
+
/Architecture not understood/i,
|
|
126174
|
+
/No such file or directory/i
|
|
126175
|
+
];
|
|
126176
|
+
/**
|
|
126177
|
+
* Coarse engine-failure classification used only to fill `engine_execution.error_type`. Exit code and
|
|
126178
|
+
* terminating signal are consulted FIRST (SIGKILL/SIGSEGV are reliable OOM signals regardless of
|
|
126179
|
+
* how much stderr the engine produced); the message text then refines the category. Anything
|
|
126180
|
+
* unrecognized is "unknown".
|
|
126181
|
+
*/
|
|
126182
|
+
function classifyEngineFailure({ error, exitCode, signal }) {
|
|
126183
|
+
// 137 = SIGKILL (kernel OOM-killer), 139 = SIGSEGV (illegal memory access). Treat both as OOM
|
|
126184
|
+
// so that memory-pressure deaths do not degrade to "unknown" when stderr is truncated.
|
|
126185
|
+
if (exitCode === 137 || exitCode === 139) {
|
|
126186
|
+
return "oom";
|
|
126187
|
+
}
|
|
126188
|
+
// Direct OS kills (SIGKILL/SIGSEGV) leave no meaningful exit code; honor them like 137/139.
|
|
126189
|
+
if (signal === "SIGKILL" || signal === "SIGSEGV") {
|
|
126190
|
+
return "oom";
|
|
126191
|
+
}
|
|
126192
|
+
const text = `${error.message}`.slice(0, 4000);
|
|
126193
|
+
for (const pattern of PROMPT_PATTERNS) {
|
|
126194
|
+
if (pattern.test(text)) {
|
|
126195
|
+
return "prompt";
|
|
126196
|
+
}
|
|
126197
|
+
}
|
|
126198
|
+
for (const pattern of OOM_PATTERNS) {
|
|
126199
|
+
if (pattern.test(text)) {
|
|
126200
|
+
return "oom";
|
|
126201
|
+
}
|
|
126202
|
+
}
|
|
126203
|
+
for (const pattern of CONFIG_PATTERNS) {
|
|
126204
|
+
if (pattern.test(text)) {
|
|
126205
|
+
return "config";
|
|
126206
|
+
}
|
|
126207
|
+
}
|
|
126208
|
+
// A process that exited abnormally with no recognizable diagnostic: treat as a crash rather
|
|
126209
|
+
// than "unknown" so the two buckets distinguish "we saw nothing" from "it died badly".
|
|
126210
|
+
if (exitCode !== null && exitCode !== 0) {
|
|
126211
|
+
return "crash";
|
|
126212
|
+
}
|
|
126213
|
+
return "unknown";
|
|
126214
|
+
}
|
|
126215
|
+
|
|
126077
126216
|
const EXCEPTION_LINE_PATTERN = /([A-Za-z_][A-Za-z0-9_]*(?:Error|Exception)):\s*(.+)/;
|
|
126078
126217
|
const FALLBACK_DETAIL_MAX_LENGTH = 300;
|
|
126079
126218
|
const FALLBACK_RAW_MAX_LENGTH = 500;
|
|
@@ -158165,12 +158304,20 @@ async function detectVRAMViaSysfs(pciBusSuffix) {
|
|
|
158165
158304
|
const deviceDir = path$1.join(DRM_PATH, entry, "device");
|
|
158166
158305
|
const totalStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_vram_total"));
|
|
158167
158306
|
const usedStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_vram_used"));
|
|
158307
|
+
const gttTotalStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_gtt_total"));
|
|
158308
|
+
const gttUsedStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_gtt_used"));
|
|
158168
158309
|
const totalBytes = totalStr !== null ? parseInt(totalStr, 10) : null;
|
|
158169
158310
|
const usedBytes = usedStr !== null ? parseInt(usedStr, 10) : null;
|
|
158311
|
+
const gttTotalBytes = gttTotalStr !== null ? parseInt(gttTotalStr, 10) : null;
|
|
158312
|
+
const gttUsedBytes = gttUsedStr !== null ? parseInt(gttUsedStr, 10) : null;
|
|
158170
158313
|
const validTotal = totalBytes !== null && Number.isFinite(totalBytes) && totalBytes >= 0;
|
|
158171
158314
|
const validUsed = usedBytes !== null && Number.isFinite(usedBytes) && usedBytes >= 0;
|
|
158315
|
+
const validGttTotal = gttTotalBytes !== null && Number.isFinite(gttTotalBytes) && gttTotalBytes >= 0;
|
|
158316
|
+
const validGttUsed = gttUsedBytes !== null && Number.isFinite(gttUsedBytes) && gttUsedBytes >= 0;
|
|
158172
158317
|
if (validTotal) {
|
|
158173
158318
|
return {
|
|
158319
|
+
gttTotalBytes: validGttTotal ? gttTotalBytes : null,
|
|
158320
|
+
gttUsedBytes: validGttUsed ? gttUsedBytes : null,
|
|
158174
158321
|
memoryTotalBytes: totalBytes,
|
|
158175
158322
|
memoryUsedBytes: validUsed ? usedBytes : null
|
|
158176
158323
|
};
|
|
@@ -158180,36 +158327,74 @@ async function detectVRAMViaSysfs(pciBusSuffix) {
|
|
|
158180
158327
|
catch {
|
|
158181
158328
|
// sysfs not available
|
|
158182
158329
|
}
|
|
158183
|
-
return {
|
|
158330
|
+
return {
|
|
158331
|
+
gttTotalBytes: null,
|
|
158332
|
+
gttUsedBytes: null,
|
|
158333
|
+
memoryTotalBytes: null,
|
|
158334
|
+
memoryUsedBytes: null
|
|
158335
|
+
};
|
|
158336
|
+
}
|
|
158337
|
+
// rocm-smi accepts a single --showmeminfo type per invocation, so VRAM and GTT
|
|
158338
|
+
// arrive from separate calls; both share this per-card parser.
|
|
158339
|
+
function parseRocmSmiMemory({ keyPrefix, stdout }) {
|
|
158340
|
+
const parsed = JSON.parse(stdout);
|
|
158341
|
+
const results = [];
|
|
158342
|
+
const cards = Object.entries(parsed).filter(([key]) => key.startsWith("card"));
|
|
158343
|
+
for (const [, data] of cards) {
|
|
158344
|
+
const bus = data["PCI Bus"] ?? null;
|
|
158345
|
+
const totalStr = data[`${keyPrefix} Total Memory (B)`] ?? null;
|
|
158346
|
+
const usedStr = data[`${keyPrefix} Total Used Memory (B)`] ?? null;
|
|
158347
|
+
const totalBytes = totalStr !== null ? parseInt(totalStr, 10) : null;
|
|
158348
|
+
const usedBytes = usedStr !== null ? parseInt(usedStr, 10) : null;
|
|
158349
|
+
const validTotal = totalBytes !== null && Number.isFinite(totalBytes) && totalBytes >= 0;
|
|
158350
|
+
const validUsed = usedBytes !== null && Number.isFinite(usedBytes) && usedBytes >= 0;
|
|
158351
|
+
if (bus) {
|
|
158352
|
+
results.push({
|
|
158353
|
+
bus,
|
|
158354
|
+
memoryTotalBytes: validTotal ? totalBytes : null,
|
|
158355
|
+
memoryUsedBytes: validUsed ? usedBytes : null
|
|
158356
|
+
});
|
|
158357
|
+
}
|
|
158358
|
+
}
|
|
158359
|
+
return results;
|
|
158184
158360
|
}
|
|
158185
|
-
|
|
158361
|
+
const ROCM_SMI_TIMEOUT_MS = 10_000;
|
|
158362
|
+
async function detectVRAMViaRocmSmi({ logger }) {
|
|
158186
158363
|
try {
|
|
158187
|
-
const
|
|
158188
|
-
"--showbus",
|
|
158189
|
-
|
|
158190
|
-
|
|
158191
|
-
"--json"
|
|
158364
|
+
const [vramResult, gttResult] = await Promise.allSettled([
|
|
158365
|
+
execa("rocm-smi", ["--showbus", "--showmeminfo", "vram", "--json"], {
|
|
158366
|
+
timeout: ROCM_SMI_TIMEOUT_MS
|
|
158367
|
+
}),
|
|
158368
|
+
execa("rocm-smi", ["--showbus", "--showmeminfo", "gtt", "--json"], {
|
|
158369
|
+
timeout: ROCM_SMI_TIMEOUT_MS
|
|
158370
|
+
})
|
|
158192
158371
|
]);
|
|
158193
|
-
|
|
158194
|
-
|
|
158195
|
-
|
|
158196
|
-
|
|
158197
|
-
|
|
158198
|
-
|
|
158199
|
-
|
|
158200
|
-
|
|
158201
|
-
const usedBytes = usedStr !== null ? parseInt(usedStr, 10) : null;
|
|
158202
|
-
const validTotal = totalBytes !== null && Number.isFinite(totalBytes) && totalBytes >= 0;
|
|
158203
|
-
const validUsed = usedBytes !== null && Number.isFinite(usedBytes) && usedBytes >= 0;
|
|
158204
|
-
if (bus) {
|
|
158205
|
-
results.push({
|
|
158206
|
-
bus,
|
|
158207
|
-
memoryTotalBytes: validTotal ? totalBytes : null,
|
|
158208
|
-
memoryUsedBytes: validUsed ? usedBytes : null
|
|
158372
|
+
if (vramResult.status !== "fulfilled")
|
|
158373
|
+
return [];
|
|
158374
|
+
let gttEntries = [];
|
|
158375
|
+
if (gttResult.status === "fulfilled") {
|
|
158376
|
+
try {
|
|
158377
|
+
gttEntries = parseRocmSmiMemory({
|
|
158378
|
+
keyPrefix: "GTT",
|
|
158379
|
+
stdout: gttResult.value.stdout
|
|
158209
158380
|
});
|
|
158210
158381
|
}
|
|
158382
|
+
catch (error) {
|
|
158383
|
+
// Unusable GTT output (e.g. older rocm-smi) degrades to VRAM-only
|
|
158384
|
+
logger.warn("rocm-smi GTT output parse failed", { error: asError(error) });
|
|
158385
|
+
}
|
|
158211
158386
|
}
|
|
158212
|
-
|
|
158387
|
+
const gttByBus = new Map(gttEntries.map(entry => [entry.bus, entry]));
|
|
158388
|
+
return parseRocmSmiMemory({ keyPrefix: "VRAM", stdout: vramResult.value.stdout }).map(vram => {
|
|
158389
|
+
const gtt = gttByBus.get(vram.bus);
|
|
158390
|
+
return {
|
|
158391
|
+
bus: vram.bus,
|
|
158392
|
+
gttTotalBytes: gtt?.memoryTotalBytes ?? null,
|
|
158393
|
+
gttUsedBytes: gtt?.memoryUsedBytes ?? null,
|
|
158394
|
+
memoryTotalBytes: vram.memoryTotalBytes,
|
|
158395
|
+
memoryUsedBytes: vram.memoryUsedBytes
|
|
158396
|
+
};
|
|
158397
|
+
});
|
|
158213
158398
|
}
|
|
158214
158399
|
catch {
|
|
158215
158400
|
return [];
|
|
@@ -158274,8 +158459,9 @@ async function detectGPUsViaNvidiaSmi({ logger }) {
|
|
|
158274
158459
|
vendor: "NVIDIA"
|
|
158275
158460
|
});
|
|
158276
158461
|
}
|
|
158277
|
-
//
|
|
158278
|
-
// memory. Their compute pool is system RAM, so fall back to
|
|
158462
|
+
// Shared/unified-memory devices (e.g. NVIDIA GB10 on DGX Spark) report
|
|
158463
|
+
// "[N/A]" for memory. Their compute pool is system RAM, so fall back to
|
|
158464
|
+
// /proc/meminfo. (AMD iGPUs get the same treatment via GTT merging.)
|
|
158279
158465
|
if (gpus.some(gpu => gpu.memoryTotalBytes === null)) {
|
|
158280
158466
|
const systemMemory = await readSystemMemoryBytes({ logger });
|
|
158281
158467
|
const total = systemMemory.totalBytes;
|
|
@@ -158299,7 +158485,7 @@ async function detectGPUsViaNvidiaSmi({ logger }) {
|
|
|
158299
158485
|
}
|
|
158300
158486
|
}
|
|
158301
158487
|
function buildMergedGPUs(options) {
|
|
158302
|
-
const { lspciGPUs, nvidiaGPUs, rocmVRAM, siGPUs, sysfsVRAMMap } = options;
|
|
158488
|
+
const { lspciGPUs, nvidiaGPUs, rocmVRAM, siGPUs, systemTotalBytes, sysfsVRAMMap } = options;
|
|
158303
158489
|
const rocmByBus = new Map();
|
|
158304
158490
|
for (const entry of rocmVRAM) {
|
|
158305
158491
|
const key = normalizeBusAddress(entry.bus);
|
|
@@ -158340,6 +158526,7 @@ function buildMergedGPUs(options) {
|
|
|
158340
158526
|
gpu: existing,
|
|
158341
158527
|
key,
|
|
158342
158528
|
rocmByBus,
|
|
158529
|
+
systemTotalBytes,
|
|
158343
158530
|
sysfsVRAMMap
|
|
158344
158531
|
});
|
|
158345
158532
|
}
|
|
@@ -158359,30 +158546,73 @@ function buildMergedGPUs(options) {
|
|
|
158359
158546
|
gpu,
|
|
158360
158547
|
key,
|
|
158361
158548
|
rocmByBus,
|
|
158549
|
+
systemTotalBytes,
|
|
158362
158550
|
sysfsVRAMMap
|
|
158363
158551
|
});
|
|
158364
158552
|
byBus.set(key, gpu);
|
|
158365
158553
|
}
|
|
158366
158554
|
return [...byBus.values()];
|
|
158367
158555
|
}
|
|
158368
|
-
|
|
158556
|
+
// Shared-memory GPUs (AMD iGPUs) expose a small dedicated VRAM carve-out via
|
|
158557
|
+
// mem_info_vram_total while the real usable pool - GTT - is carved dynamically
|
|
158558
|
+
// from system RAM. When GTT exceeds VRAM the device is treated as integrated
|
|
158559
|
+
// and the two pools combine (capped by installed RAM). Note that GTT defaults
|
|
158560
|
+
// to half of system RAM even on discrete GPUs, so a dGPU with less VRAM than
|
|
158561
|
+
// that also merges - acceptable, since GTT remains real addressable memory
|
|
158562
|
+
// for amdgpu compute (albeit slower over PCIe).
|
|
158563
|
+
function mergeGTTMemory({ gttTotalBytes, gttUsedBytes, systemTotalBytes, vramTotalBytes, vramUsedBytes }) {
|
|
158564
|
+
if (vramTotalBytes === null || !Number.isFinite(vramTotalBytes)) {
|
|
158565
|
+
return { memoryTotalBytes: null, memoryUsedBytes: null };
|
|
158566
|
+
}
|
|
158567
|
+
const isIntegrated = gttTotalBytes !== null && gttTotalBytes > vramTotalBytes;
|
|
158568
|
+
if (!isIntegrated) {
|
|
158569
|
+
return { memoryTotalBytes: vramTotalBytes, memoryUsedBytes: vramUsedBytes };
|
|
158570
|
+
}
|
|
158571
|
+
const combined = vramTotalBytes + gttTotalBytes;
|
|
158572
|
+
const capped = systemTotalBytes !== null && systemTotalBytes > 0
|
|
158573
|
+
? Math.min(combined, systemTotalBytes)
|
|
158574
|
+
: combined;
|
|
158575
|
+
const hasCompleteUsage = vramUsedBytes !== null &&
|
|
158576
|
+
Number.isFinite(vramUsedBytes) &&
|
|
158577
|
+
gttUsedBytes !== null &&
|
|
158578
|
+
Number.isFinite(gttUsedBytes);
|
|
158579
|
+
const memoryUsedBytes = hasCompleteUsage
|
|
158580
|
+
? Math.min(vramUsedBytes + gttUsedBytes, capped)
|
|
158581
|
+
: null;
|
|
158582
|
+
return { memoryTotalBytes: capped, memoryUsedBytes };
|
|
158583
|
+
}
|
|
158584
|
+
function applySysfsOrRocmVRAM({ gpu, key, rocmByBus, systemTotalBytes, sysfsVRAMMap }) {
|
|
158369
158585
|
const sysfs = sysfsVRAMMap.get(key);
|
|
158370
|
-
let
|
|
158371
|
-
let
|
|
158372
|
-
|
|
158586
|
+
let vramTotalBytes = sysfs?.memoryTotalBytes ?? null;
|
|
158587
|
+
let vramUsedBytes = sysfs?.memoryUsedBytes ?? null;
|
|
158588
|
+
let gttTotalBytes = sysfs?.gttTotalBytes ?? null;
|
|
158589
|
+
let gttUsedBytes = sysfs?.gttUsedBytes ?? null;
|
|
158590
|
+
if (vramTotalBytes === null) {
|
|
158373
158591
|
const rocm = rocmByBus.get(key);
|
|
158374
158592
|
if (rocm) {
|
|
158375
|
-
|
|
158376
|
-
|
|
158377
|
-
|
|
158378
|
-
|
|
158379
|
-
|
|
158593
|
+
vramTotalBytes = rocm.memoryTotalBytes;
|
|
158594
|
+
vramUsedBytes = rocm.memoryUsedBytes;
|
|
158595
|
+
gttTotalBytes = rocm.gttTotalBytes;
|
|
158596
|
+
gttUsedBytes = rocm.gttUsedBytes;
|
|
158597
|
+
}
|
|
158598
|
+
}
|
|
158599
|
+
const merged = mergeGTTMemory({
|
|
158600
|
+
gttTotalBytes,
|
|
158601
|
+
gttUsedBytes,
|
|
158602
|
+
systemTotalBytes,
|
|
158603
|
+
vramTotalBytes,
|
|
158604
|
+
vramUsedBytes
|
|
158605
|
+
});
|
|
158606
|
+
if (merged.memoryTotalBytes === null || !Number.isFinite(merged.memoryTotalBytes))
|
|
158380
158607
|
return;
|
|
158381
|
-
gpu.memoryTotalBytes =
|
|
158382
|
-
gpu.memoryUsedBytes =
|
|
158608
|
+
gpu.memoryTotalBytes = merged.memoryTotalBytes;
|
|
158609
|
+
gpu.memoryUsedBytes =
|
|
158610
|
+
merged.memoryUsedBytes !== null && Number.isFinite(merged.memoryUsedBytes)
|
|
158611
|
+
? merged.memoryUsedBytes
|
|
158612
|
+
: null;
|
|
158383
158613
|
gpu.memoryFreeBytes =
|
|
158384
|
-
|
|
158385
|
-
? Math.max(
|
|
158614
|
+
gpu.memoryUsedBytes !== null
|
|
158615
|
+
? Math.max(gpu.memoryTotalBytes - gpu.memoryUsedBytes, 0)
|
|
158386
158616
|
: null;
|
|
158387
158617
|
}
|
|
158388
158618
|
async function collectMachineMetadata({ logger }) {
|
|
@@ -158393,7 +158623,7 @@ async function collectMachineMetadata({ logger }) {
|
|
|
158393
158623
|
si.graphics(),
|
|
158394
158624
|
detectGPUsViaLspci(),
|
|
158395
158625
|
detectGPUsViaNvidiaSmi({ logger }),
|
|
158396
|
-
detectVRAMViaRocmSmi()
|
|
158626
|
+
detectVRAMViaRocmSmi({ logger })
|
|
158397
158627
|
]);
|
|
158398
158628
|
const cpuInfo = cpuResult.status === "fulfilled" ? cpuResult.value : null;
|
|
158399
158629
|
const memInfo = memResult.status === "fulfilled" ? memResult.value : null;
|
|
@@ -158448,6 +158678,7 @@ async function collectMachineMetadata({ logger }) {
|
|
|
158448
158678
|
nvidiaGPUs: resolvedNvidiaGPUs,
|
|
158449
158679
|
rocmVRAM: resolvedRocmVRAM,
|
|
158450
158680
|
siGPUs,
|
|
158681
|
+
systemTotalBytes: memInfo?.total ?? null,
|
|
158451
158682
|
sysfsVRAMMap
|
|
158452
158683
|
});
|
|
158453
158684
|
const machineMetadata = {
|
|
@@ -158508,6 +158739,119 @@ async function detectDockerVersion() {
|
|
|
158508
158739
|
}
|
|
158509
158740
|
}
|
|
158510
158741
|
|
|
158742
|
+
/**
|
|
158743
|
+
* Flattens flat CLI extra-arg tokens into [arg, value] pairs, sorted by ARG NAME (ascending, ties by
|
|
158744
|
+
* value). `--flag=value` pairs split on the first `=`; a bare `--flag` consumes the following token
|
|
158745
|
+
* as its value when that token does not start with "-" (classic CLI convention); anything else
|
|
158746
|
+
* (flags, non-strings) is dropped.
|
|
158747
|
+
*/
|
|
158748
|
+
function pairExtraArgs(tokens) {
|
|
158749
|
+
if (!Array.isArray(tokens)) {
|
|
158750
|
+
return [];
|
|
158751
|
+
}
|
|
158752
|
+
const list = tokens;
|
|
158753
|
+
const pairs = [];
|
|
158754
|
+
let index = 0;
|
|
158755
|
+
while (index < list.length) {
|
|
158756
|
+
const token = list[index];
|
|
158757
|
+
if (typeof token !== "string" || token.length === 0 || !token.startsWith("-")) {
|
|
158758
|
+
index++;
|
|
158759
|
+
continue;
|
|
158760
|
+
}
|
|
158761
|
+
const separator = token.indexOf("=");
|
|
158762
|
+
if (separator > -1) {
|
|
158763
|
+
const arg = token.slice(0, separator);
|
|
158764
|
+
if (arg.length > 0) {
|
|
158765
|
+
pairs.push([arg, token.slice(separator + 1)]);
|
|
158766
|
+
}
|
|
158767
|
+
index++;
|
|
158768
|
+
continue;
|
|
158769
|
+
}
|
|
158770
|
+
const next = list[index + 1];
|
|
158771
|
+
const consumesNext = typeof next === "string" && next.length > 0 && !next.startsWith("-");
|
|
158772
|
+
if (consumesNext) {
|
|
158773
|
+
pairs.push([token, next]);
|
|
158774
|
+
index += 2;
|
|
158775
|
+
}
|
|
158776
|
+
else {
|
|
158777
|
+
pairs.push([token, ""]);
|
|
158778
|
+
index++;
|
|
158779
|
+
}
|
|
158780
|
+
}
|
|
158781
|
+
return pairs.sort((a, b) => a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0);
|
|
158782
|
+
}
|
|
158783
|
+
/**
|
|
158784
|
+
* Files AT MOST ONE engine_execution report per engine startup. `beginStartup(runAt)` re-arms the
|
|
158785
|
+
* latch on every fresh model start (initial boot or cycle) and stamps the startup epoch used as the
|
|
158786
|
+
* row's `run_at`. Whichever of the three report* paths fires first wins: startup failure, in-flight
|
|
158787
|
+
* crash, or first successful full prompt completion.
|
|
158788
|
+
*/
|
|
158789
|
+
class EngineExecutionReporter {
|
|
158790
|
+
options;
|
|
158791
|
+
currentStartupAt = null;
|
|
158792
|
+
reportedForCurrentStartup = false;
|
|
158793
|
+
constructor(options) {
|
|
158794
|
+
this.options = options;
|
|
158795
|
+
}
|
|
158796
|
+
/** Re-arms the latch; the stamp becomes the row's `run_at` (moment startup began). */
|
|
158797
|
+
beginStartup(runAt) {
|
|
158798
|
+
this.currentStartupAt = runAt;
|
|
158799
|
+
this.reportedForCurrentStartup = false;
|
|
158800
|
+
}
|
|
158801
|
+
/** Reports a startup failure (rejected prepare/start, readiness timeout, pre-ready death). */
|
|
158802
|
+
async reportStartupFailure(report) {
|
|
158803
|
+
await this.file(report, false);
|
|
158804
|
+
}
|
|
158805
|
+
/** Reports a spontaneous crash of an engine that had reached the running state. */
|
|
158806
|
+
async reportRuntimeCrash(report) {
|
|
158807
|
+
await this.file(report, false);
|
|
158808
|
+
}
|
|
158809
|
+
/** Reports the first fully-responded, token-bearing prompt completion since startup. */
|
|
158810
|
+
async reportSuccess(report) {
|
|
158811
|
+
await this.file(report, true);
|
|
158812
|
+
}
|
|
158813
|
+
async file(report, success) {
|
|
158814
|
+
if (this.reportedForCurrentStartup || this.currentStartupAt === null) {
|
|
158815
|
+
return;
|
|
158816
|
+
}
|
|
158817
|
+
// Latch BEFORE the network call: a thrown POST cannot double-file for this startup.
|
|
158818
|
+
this.reportedForCurrentStartup = true;
|
|
158819
|
+
const context = this.options.buildContext();
|
|
158820
|
+
const payload = {
|
|
158821
|
+
avgTps: report.throughput.avgTps,
|
|
158822
|
+
completionTokens: report.usage.completionTokens,
|
|
158823
|
+
durationMs: report.durationMs,
|
|
158824
|
+
engineType: context.engineType,
|
|
158825
|
+
engineVersion: context.engineVersion,
|
|
158826
|
+
errorDetail: success ? null : report.errorDetail,
|
|
158827
|
+
errorType: success ? null : report.errorType,
|
|
158828
|
+
extraArgs: context.extraArgsPairs,
|
|
158829
|
+
finishedAtISO: new Date().toISOString(),
|
|
158830
|
+
peakTps: report.throughput.peakTps,
|
|
158831
|
+
promptTokens: report.usage.promptTokens,
|
|
158832
|
+
runAtISO: this.currentStartupAt.toISOString(),
|
|
158833
|
+
success,
|
|
158834
|
+
ttftMs: report.ttftMs,
|
|
158835
|
+
totalTokens: report.usage.totalTokens
|
|
158836
|
+
};
|
|
158837
|
+
try {
|
|
158838
|
+
await this.options.report(payload);
|
|
158839
|
+
this.options.logger.info("Engine execution outcome reported", {
|
|
158840
|
+
inferenceSourceID: this.options.sourceLabel,
|
|
158841
|
+
success
|
|
158842
|
+
});
|
|
158843
|
+
}
|
|
158844
|
+
catch (error) {
|
|
158845
|
+
// Losing one report is preferable to filing two; the latch stays latched.
|
|
158846
|
+
this.options.logger.warn("Failed to report engine execution outcome", {
|
|
158847
|
+
error: asError(error),
|
|
158848
|
+
inferenceSourceID: this.options.sourceLabel,
|
|
158849
|
+
success
|
|
158850
|
+
});
|
|
158851
|
+
}
|
|
158852
|
+
}
|
|
158853
|
+
}
|
|
158854
|
+
|
|
158511
158855
|
async function createApplication({ abortController, apiClient, configuration, logger }) {
|
|
158512
158856
|
ensureDockerValidEnv();
|
|
158513
158857
|
logger.info("Fetching conduit configuration");
|
|
@@ -158539,6 +158883,87 @@ async function createApplication({ abortController, apiClient, configuration, lo
|
|
|
158539
158883
|
error: asError(error)
|
|
158540
158884
|
});
|
|
158541
158885
|
}
|
|
158886
|
+
const reporter = new EngineExecutionReporter({
|
|
158887
|
+
buildContext: () => {
|
|
158888
|
+
const engineType = (conduitConfiguration.engineConfig?.type ??
|
|
158889
|
+
"llama.cpp");
|
|
158890
|
+
const versions = {
|
|
158891
|
+
exllamav3: machine?.exllamav3Version ?? null,
|
|
158892
|
+
"llama.cpp": machine?.llamaCppVersion ?? null,
|
|
158893
|
+
"mlx-lm": machine?.mlxlmVersion ?? null,
|
|
158894
|
+
sglang: machine?.sglangVersion ?? null,
|
|
158895
|
+
"tensorrt-llm": machine?.tensorrtLlmVersion ?? null,
|
|
158896
|
+
vllm: machine?.vllmVersion ?? null
|
|
158897
|
+
};
|
|
158898
|
+
return {
|
|
158899
|
+
engineType,
|
|
158900
|
+
engineVersion: versions[engineType] ?? null,
|
|
158901
|
+
extraArgsPairs: pairExtraArgs(conduitConfiguration.engineConfig?.extraArgs)
|
|
158902
|
+
};
|
|
158903
|
+
},
|
|
158904
|
+
logger,
|
|
158905
|
+
report: payload => apiClient.reportEngineExecution(payload),
|
|
158906
|
+
sourceLabel: configuration.inferenceSourceID
|
|
158907
|
+
});
|
|
158908
|
+
// Intercept the prompt-metrics chokepoint so the first fully-responded, token-bearing prompt of
|
|
158909
|
+
// each fresh startup files the one-shot engine_execution success report. Handlers close over the
|
|
158910
|
+
// SAME `apiClient` object and read `reportPromptMetrics` at request-dispatch time (which always
|
|
158911
|
+
// follows this point), so the wrapped method is what they invoke.
|
|
158912
|
+
const rawReportPromptMetrics = apiClient.reportPromptMetrics;
|
|
158913
|
+
apiClient.reportPromptMetrics = async (payload) => {
|
|
158914
|
+
if (payload.successful && payload.completionTokens > 0 && payload.latencyMs > 0) {
|
|
158915
|
+
// The one-shot report is kicked off and its settlement attached HERE (before any await): if the
|
|
158916
|
+
// metrics path throws below, the report promise must still be able to log its own rejection.
|
|
158917
|
+
const successReport = reporter
|
|
158918
|
+
.reportSuccess({
|
|
158919
|
+
durationMs: payload.latencyMs,
|
|
158920
|
+
errorDetail: null,
|
|
158921
|
+
errorType: null,
|
|
158922
|
+
throughput: {
|
|
158923
|
+
avgTps: payload.tokensPerSecond,
|
|
158924
|
+
peakTps: null
|
|
158925
|
+
},
|
|
158926
|
+
ttftMs: payload.timeToFirstTokenMs ?? 0,
|
|
158927
|
+
usage: {
|
|
158928
|
+
completionTokens: payload.completionTokens,
|
|
158929
|
+
promptTokens: payload.promptTokens,
|
|
158930
|
+
totalTokens: payload.totalTokens
|
|
158931
|
+
}
|
|
158932
|
+
})
|
|
158933
|
+
.catch(error => {
|
|
158934
|
+
logger.warn("Engine execution success report failed", {
|
|
158935
|
+
error: asError(error)
|
|
158936
|
+
});
|
|
158937
|
+
});
|
|
158938
|
+
await rawReportPromptMetrics(payload);
|
|
158939
|
+
await successReport;
|
|
158940
|
+
return;
|
|
158941
|
+
}
|
|
158942
|
+
await rawReportPromptMetrics(payload);
|
|
158943
|
+
};
|
|
158944
|
+
const SECRET_ARG_MASK_PATTERN = /(-{1,2}[A-Za-z0-9_.]*(?:api[-_]?key|hf[-_]?token|token)(?:\s+|[=:]))\S+/gi;
|
|
158945
|
+
// Assembles the payload shared by the startup-failure and runtime-crash report paths. Stderr
|
|
158946
|
+
// may echo secrets, so mask `--api-key`/token-looking args before they reach the DB.
|
|
158947
|
+
function buildCrashReport(error, exitCode, signal) {
|
|
158948
|
+
const classification = classifyEngineFailure({ error, exitCode, signal });
|
|
158949
|
+
const raw = normalizeEngineError(error.message);
|
|
158950
|
+
const masked = raw.replace(SECRET_ARG_MASK_PATTERN, "$1***");
|
|
158951
|
+
return {
|
|
158952
|
+
durationMs: 0,
|
|
158953
|
+
errorDetail: masked.slice(0, 2048),
|
|
158954
|
+
errorType: classification,
|
|
158955
|
+
ttftMs: 0,
|
|
158956
|
+
throughput: {
|
|
158957
|
+
avgTps: 0,
|
|
158958
|
+
peakTps: null
|
|
158959
|
+
},
|
|
158960
|
+
usage: {
|
|
158961
|
+
completionTokens: 0,
|
|
158962
|
+
promptTokens: 0,
|
|
158963
|
+
totalTokens: 0
|
|
158964
|
+
}
|
|
158965
|
+
};
|
|
158966
|
+
}
|
|
158542
158967
|
const conduitStateManager = new ConduitStateManager({
|
|
158543
158968
|
initialState: {
|
|
158544
158969
|
state: "initialising"
|
|
@@ -158590,6 +159015,17 @@ async function createApplication({ abortController, apiClient, configuration, lo
|
|
|
158590
159015
|
});
|
|
158591
159016
|
stopRequestedByControl = false;
|
|
158592
159017
|
setErrorState({ error: normalizeEngineError(err.message) });
|
|
159018
|
+
// Spontaneous death of a SERVING engine → crash report, suppressed by the latch if the
|
|
159019
|
+
// startup's one-shot outcome was already filed. Startup-path failures report from
|
|
159020
|
+
// `startEngine`'s catch; this listener is the RUNTIME-crash path only.
|
|
159021
|
+
if (modelManager.wasRunning && !err.message.includes("interrupted by stop request")) {
|
|
159022
|
+
const crashReport = buildCrashReport(err, modelManager.lastExitCode, modelManager.lastExitSignal);
|
|
159023
|
+
reporter.reportRuntimeCrash(crashReport).catch(crashReportError => {
|
|
159024
|
+
logger.warn("Engine execution crash report failed", {
|
|
159025
|
+
error: asError(crashReportError)
|
|
159026
|
+
});
|
|
159027
|
+
});
|
|
159028
|
+
}
|
|
158593
159029
|
});
|
|
158594
159030
|
modelManager.on("engineReady", () => {
|
|
158595
159031
|
setOnlineState();
|
|
@@ -158649,24 +159085,40 @@ async function createApplication({ abortController, apiClient, configuration, lo
|
|
|
158649
159085
|
};
|
|
158650
159086
|
async function startEngine() {
|
|
158651
159087
|
logger.info("Engine start requested");
|
|
158652
|
-
|
|
158653
|
-
|
|
158654
|
-
|
|
158655
|
-
|
|
158656
|
-
|
|
158657
|
-
|
|
158658
|
-
|
|
159088
|
+
reporter.beginStartup(new Date());
|
|
159089
|
+
try {
|
|
159090
|
+
conduitStateManager.setState({
|
|
159091
|
+
modelFileName,
|
|
159092
|
+
modelName,
|
|
159093
|
+
state: "downloadingModelFiles",
|
|
159094
|
+
totalProgress: {
|
|
159095
|
+
file: 0,
|
|
159096
|
+
total: 0
|
|
159097
|
+
}
|
|
159098
|
+
});
|
|
159099
|
+
await conduitStateReportManager.reportNow();
|
|
159100
|
+
await modelManager.prepare({
|
|
159101
|
+
onDownloadProgress: reportDownloadProgress
|
|
159102
|
+
});
|
|
159103
|
+
conduitStateManager.setState({
|
|
159104
|
+
state: "bootingEngine"
|
|
159105
|
+
});
|
|
159106
|
+
await conduitStateReportManager.reportNow();
|
|
159107
|
+
await modelManager.start();
|
|
159108
|
+
}
|
|
159109
|
+
catch (error) {
|
|
159110
|
+
const parsedError = asError(error);
|
|
159111
|
+
// Operator-initiated aborts are not startup failures worth reporting.
|
|
159112
|
+
if (!parsedError.message.includes("interrupted by stop request")) {
|
|
159113
|
+
const startupReport = buildCrashReport(parsedError, modelManager.lastExitCode, modelManager.lastExitSignal);
|
|
159114
|
+
reporter.reportStartupFailure(startupReport).catch(startupReportError => {
|
|
159115
|
+
logger.warn("Engine execution startup report failed", {
|
|
159116
|
+
error: asError(startupReportError)
|
|
159117
|
+
});
|
|
159118
|
+
});
|
|
158659
159119
|
}
|
|
158660
|
-
|
|
158661
|
-
|
|
158662
|
-
await modelManager.prepare({
|
|
158663
|
-
onDownloadProgress: reportDownloadProgress
|
|
158664
|
-
});
|
|
158665
|
-
conduitStateManager.setState({
|
|
158666
|
-
state: "bootingEngine"
|
|
158667
|
-
});
|
|
158668
|
-
await conduitStateReportManager.reportNow();
|
|
158669
|
-
await modelManager.start();
|
|
159120
|
+
throw error;
|
|
159121
|
+
}
|
|
158670
159122
|
}
|
|
158671
159123
|
async function stopEngine({ reason }) {
|
|
158672
159124
|
if (!modelManager.canStop) {
|
|
@@ -363121,12 +363573,13 @@ async function runModelFit(options) {
|
|
|
363121
363573
|
console.log();
|
|
363122
363574
|
const feasible = filterFeasibleModels({ detection, models: recommendedModels });
|
|
363123
363575
|
if (feasible.length === 0) {
|
|
363576
|
+
const smallestTierGB = Math.min(...recommendedModels.map(model => model.vramTierGB));
|
|
363124
363577
|
console.error("No recommended models fit this hardware. " +
|
|
363125
363578
|
`Budget: ${formatBytes$1(detection.gpus.some(gpu => gpu.memoryTotalBytes)
|
|
363126
363579
|
? Math.max(...detection.gpus
|
|
363127
363580
|
.map(gpu => gpu.memoryTotalBytes ?? 0)
|
|
363128
363581
|
.filter(bytes => bytes > 0))
|
|
363129
|
-
: detection.memory.totalBytes)}. Smallest tier starts at
|
|
363582
|
+
: detection.memory.totalBytes)}. Smallest tier starts at ${smallestTierGB} GB.`);
|
|
363130
363583
|
process.exitCode = 1;
|
|
363131
363584
|
return;
|
|
363132
363585
|
}
|