@infersec/conduit 1.112.0 → 1.113.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.sea.cjs CHANGED
@@ -19986,6 +19986,32 @@ object$5({
19986
19986
  sizeBytes: number$1().int().nonnegative().nullable()
19987
19987
  });
19988
19988
 
19989
+ const EngineExecutionErrorTypeSchema = _enum$1([
19990
+ "config",
19991
+ "crash",
19992
+ "oom",
19993
+ "prompt",
19994
+ "unknown"
19995
+ ]);
19996
+ const EngineExecutionReportPayloadSchema = object$5({
19997
+ avgTps: number$1().nonnegative().finite().default(0),
19998
+ completionTokens: number$1().int().nonnegative().default(0),
19999
+ durationMs: number$1().int().nonnegative().default(0),
20000
+ engineType: LLMEngineSchema.nullable(),
20001
+ engineVersion: string$2().max(64).nullable().default(null),
20002
+ errorDetail: string$2().max(2048).nullable().default(null),
20003
+ errorType: EngineExecutionErrorTypeSchema.nullable(),
20004
+ extraArgs: array$1(tuple([string$2().min(1).max(128), string$2().max(512)]))
20005
+ .max(256)
20006
+ .default([]),
20007
+ finishedAtISO: string$2().datetime({ offset: true }),
20008
+ peakTps: number$1().nonnegative().finite().nullable().default(null),
20009
+ promptTokens: number$1().int().nonnegative().default(0),
20010
+ runAtISO: string$2().datetime({ offset: true }),
20011
+ success: boolean$1(),
20012
+ ttftMs: number$1().int().nonnegative().default(0),
20013
+ totalTokens: number$1().int().nonnegative().default(0)
20014
+ });
19989
20015
  const InferenceAgentLLMMetricsPayloadSchema = object$5({
19990
20016
  bytes: number$1().int().nonnegative(),
19991
20017
  completionTokens: number$1().int().nonnegative(),
@@ -20324,6 +20350,23 @@ const API_SERVICE_CONDUIT_API_REFERENCE = {
20324
20350
  }
20325
20351
  }
20326
20352
  },
20353
+ "/conduit/api/v1/source/:sourceID/engine/execution": {
20354
+ POST: {
20355
+ auth: {
20356
+ type: "api-key"
20357
+ },
20358
+ body: EngineExecutionReportPayloadSchema,
20359
+ parameters: {
20360
+ sourceID: ULIDSchema
20361
+ },
20362
+ response: {
20363
+ schema: object$5({
20364
+ acknowledged: literal(true)
20365
+ }),
20366
+ type: "rest"
20367
+ }
20368
+ }
20369
+ },
20327
20370
  "/conduit/api/v1/source/:sourceID/requests/:requestID/chunk": {
20328
20371
  POST: {
20329
20372
  auth: {
@@ -113781,6 +113824,19 @@ function createAPIClient({ apiKey, apiURL, inferenceSourceID, logger }) {
113781
113824
  route: "/conduit/api/v1/source/:sourceID/state"
113782
113825
  });
113783
113826
  },
113827
+ reportEngineExecution: async (payload) => {
113828
+ await fetchByReference({
113829
+ baseURL: apiURL,
113830
+ body: payload,
113831
+ fetch: fetchWithAPIKey,
113832
+ method: "POST",
113833
+ parameters: {
113834
+ sourceID: inferenceSourceID
113835
+ },
113836
+ reference: API_SERVICE_CONDUIT_API_REFERENCE,
113837
+ route: "/conduit/api/v1/source/:sourceID/engine/execution"
113838
+ });
113839
+ },
113784
113840
  reportPromptMetrics: async (payload) => {
113785
113841
  await fetchByReference({
113786
113842
  baseURL: apiURL,
@@ -125641,6 +125697,9 @@ class ModelManager extends EventEmitter {
125641
125697
  lifecycleState = "stopped";
125642
125698
  downloadLockHandle = null;
125643
125699
  stopRequested = false;
125700
+ lastEngineExitCode = null;
125701
+ lastEngineExitSignal = null;
125702
+ reachedRunningState = false;
125644
125703
  modelsDirectory;
125645
125704
  constructor({ contextLength, engineConfig, enginePort, engineType, logger, model, root }) {
125646
125705
  super();
@@ -125763,6 +125822,9 @@ class ModelManager extends EventEmitter {
125763
125822
  this.lifecycleState = "starting";
125764
125823
  this.lastEngineError = null;
125765
125824
  this.stopRequested = false;
125825
+ this.lastEngineExitCode = null;
125826
+ this.lastEngineExitSignal = null;
125827
+ this.reachedRunningState = false;
125766
125828
  this.logger.info("Starting LLM engine", {
125767
125829
  agentEngineType: this.engine
125768
125830
  });
@@ -125793,6 +125855,7 @@ class ModelManager extends EventEmitter {
125793
125855
  throw err;
125794
125856
  }
125795
125857
  this.lifecycleState = "running";
125858
+ this.reachedRunningState = true;
125796
125859
  this.emit("engineReady");
125797
125860
  }
125798
125861
  async stop() {
@@ -125817,6 +125880,7 @@ class ModelManager extends EventEmitter {
125817
125880
  this.lifecycleState = "stopped";
125818
125881
  return;
125819
125882
  }
125883
+ this.reachedRunningState = false;
125820
125884
  this.lifecycleState = "stopping";
125821
125885
  this.stopRequested = true;
125822
125886
  await processManager.stop();
@@ -125829,9 +125893,18 @@ class ModelManager extends EventEmitter {
125829
125893
  this.lifecycleState === "starting" ||
125830
125894
  this.lifecycleState === "errored");
125831
125895
  }
125896
+ get lastExitCode() {
125897
+ return this.lastEngineExitCode;
125898
+ }
125899
+ get lastExitSignal() {
125900
+ return this.lastEngineExitSignal;
125901
+ }
125832
125902
  get state() {
125833
125903
  return this.lifecycleState;
125834
125904
  }
125905
+ get wasRunning() {
125906
+ return this.reachedRunningState;
125907
+ }
125835
125908
  async checkEngineReadiness() {
125836
125909
  switch (this.engine) {
125837
125910
  case "llama.cpp": {
@@ -125945,6 +126018,7 @@ class ModelManager extends EventEmitter {
125945
126018
  if (readiness === "ready") {
125946
126019
  this.clearHealthPoll();
125947
126020
  this.lifecycleState = "running";
126021
+ this.reachedRunningState = true;
125948
126022
  this.emit("engineReady");
125949
126023
  }
125950
126024
  })
@@ -126007,6 +126081,8 @@ class ModelManager extends EventEmitter {
126007
126081
  }));
126008
126082
  });
126009
126083
  processManager.on("stopped", (code, signal) => {
126084
+ this.lastEngineExitCode = code;
126085
+ this.lastEngineExitSignal = signal;
126010
126086
  if (hasTerminated) {
126011
126087
  return;
126012
126088
  }
@@ -126074,6 +126150,69 @@ class ModelManager extends EventEmitter {
126074
126150
  }
126075
126151
  }
126076
126152
 
126153
+ // Ordered most-specific first: a message mentioning a prompt-size failure should classify as
126154
+ // "prompt" even if it also touches memory text; memory outranks config because OOM kills frequently
126155
+ // emit sparse stderr. These are heuristics, not exhaustively enumerated engine vocabularies.
126156
+ const OOM_PATTERNS = [
126157
+ /CUDA out of memory/i,
126158
+ /No available memory for the cache blocks/i,
126159
+ /\bOOMKilled\b/,
126160
+ /out of memory/i,
126161
+ /MemoryError/i,
126162
+ /Cannot allocate memory/i
126163
+ ];
126164
+ const PROMPT_PATTERNS = [
126165
+ /prompt is too long/i,
126166
+ /maximum context length exceeded/i,
126167
+ /too many tokens/i
126168
+ ];
126169
+ const CONFIG_PATTERNS = [
126170
+ /unrecognized argument/i,
126171
+ /invalid argument/i,
126172
+ /error while loading state_dict/i,
126173
+ /Architecture not understood/i,
126174
+ /No such file or directory/i
126175
+ ];
126176
+ /**
126177
+ * Coarse engine-failure classification used only to fill `engine_execution.error_type`. Exit code and
126178
+ * terminating signal are consulted FIRST (SIGKILL/SIGSEGV are reliable OOM signals regardless of
126179
+ * how much stderr the engine produced); the message text then refines the category. Anything
126180
+ * unrecognized is "unknown".
126181
+ */
126182
+ function classifyEngineFailure({ error, exitCode, signal }) {
126183
+ // 137 = SIGKILL (kernel OOM-killer), 139 = SIGSEGV (illegal memory access). Treat both as OOM
126184
+ // so that memory-pressure deaths do not degrade to "unknown" when stderr is truncated.
126185
+ if (exitCode === 137 || exitCode === 139) {
126186
+ return "oom";
126187
+ }
126188
+ // Direct OS kills (SIGKILL/SIGSEGV) leave no meaningful exit code; honor them like 137/139.
126189
+ if (signal === "SIGKILL" || signal === "SIGSEGV") {
126190
+ return "oom";
126191
+ }
126192
+ const text = `${error.message}`.slice(0, 4000);
126193
+ for (const pattern of PROMPT_PATTERNS) {
126194
+ if (pattern.test(text)) {
126195
+ return "prompt";
126196
+ }
126197
+ }
126198
+ for (const pattern of OOM_PATTERNS) {
126199
+ if (pattern.test(text)) {
126200
+ return "oom";
126201
+ }
126202
+ }
126203
+ for (const pattern of CONFIG_PATTERNS) {
126204
+ if (pattern.test(text)) {
126205
+ return "config";
126206
+ }
126207
+ }
126208
+ // A process that exited abnormally with no recognizable diagnostic: treat as a crash rather
126209
+ // than "unknown" so the two buckets distinguish "we saw nothing" from "it died badly".
126210
+ if (exitCode !== null && exitCode !== 0) {
126211
+ return "crash";
126212
+ }
126213
+ return "unknown";
126214
+ }
126215
+
126077
126216
  const EXCEPTION_LINE_PATTERN = /([A-Za-z_][A-Za-z0-9_]*(?:Error|Exception)):\s*(.+)/;
126078
126217
  const FALLBACK_DETAIL_MAX_LENGTH = 300;
126079
126218
  const FALLBACK_RAW_MAX_LENGTH = 500;
@@ -158165,12 +158304,20 @@ async function detectVRAMViaSysfs(pciBusSuffix) {
158165
158304
  const deviceDir = path$1.join(DRM_PATH, entry, "device");
158166
158305
  const totalStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_vram_total"));
158167
158306
  const usedStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_vram_used"));
158307
+ const gttTotalStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_gtt_total"));
158308
+ const gttUsedStr = await readSysfsFile(path$1.join(deviceDir, "mem_info_gtt_used"));
158168
158309
  const totalBytes = totalStr !== null ? parseInt(totalStr, 10) : null;
158169
158310
  const usedBytes = usedStr !== null ? parseInt(usedStr, 10) : null;
158311
+ const gttTotalBytes = gttTotalStr !== null ? parseInt(gttTotalStr, 10) : null;
158312
+ const gttUsedBytes = gttUsedStr !== null ? parseInt(gttUsedStr, 10) : null;
158170
158313
  const validTotal = totalBytes !== null && Number.isFinite(totalBytes) && totalBytes >= 0;
158171
158314
  const validUsed = usedBytes !== null && Number.isFinite(usedBytes) && usedBytes >= 0;
158315
+ const validGttTotal = gttTotalBytes !== null && Number.isFinite(gttTotalBytes) && gttTotalBytes >= 0;
158316
+ const validGttUsed = gttUsedBytes !== null && Number.isFinite(gttUsedBytes) && gttUsedBytes >= 0;
158172
158317
  if (validTotal) {
158173
158318
  return {
158319
+ gttTotalBytes: validGttTotal ? gttTotalBytes : null,
158320
+ gttUsedBytes: validGttUsed ? gttUsedBytes : null,
158174
158321
  memoryTotalBytes: totalBytes,
158175
158322
  memoryUsedBytes: validUsed ? usedBytes : null
158176
158323
  };
@@ -158180,36 +158327,74 @@ async function detectVRAMViaSysfs(pciBusSuffix) {
158180
158327
  catch {
158181
158328
  // sysfs not available
158182
158329
  }
158183
- return { memoryTotalBytes: null, memoryUsedBytes: null };
158330
+ return {
158331
+ gttTotalBytes: null,
158332
+ gttUsedBytes: null,
158333
+ memoryTotalBytes: null,
158334
+ memoryUsedBytes: null
158335
+ };
158336
+ }
158337
+ // rocm-smi accepts a single --showmeminfo type per invocation, so VRAM and GTT
158338
+ // arrive from separate calls; both share this per-card parser.
158339
+ function parseRocmSmiMemory({ keyPrefix, stdout }) {
158340
+ const parsed = JSON.parse(stdout);
158341
+ const results = [];
158342
+ const cards = Object.entries(parsed).filter(([key]) => key.startsWith("card"));
158343
+ for (const [, data] of cards) {
158344
+ const bus = data["PCI Bus"] ?? null;
158345
+ const totalStr = data[`${keyPrefix} Total Memory (B)`] ?? null;
158346
+ const usedStr = data[`${keyPrefix} Total Used Memory (B)`] ?? null;
158347
+ const totalBytes = totalStr !== null ? parseInt(totalStr, 10) : null;
158348
+ const usedBytes = usedStr !== null ? parseInt(usedStr, 10) : null;
158349
+ const validTotal = totalBytes !== null && Number.isFinite(totalBytes) && totalBytes >= 0;
158350
+ const validUsed = usedBytes !== null && Number.isFinite(usedBytes) && usedBytes >= 0;
158351
+ if (bus) {
158352
+ results.push({
158353
+ bus,
158354
+ memoryTotalBytes: validTotal ? totalBytes : null,
158355
+ memoryUsedBytes: validUsed ? usedBytes : null
158356
+ });
158357
+ }
158358
+ }
158359
+ return results;
158184
158360
  }
158185
- async function detectVRAMViaRocmSmi() {
158361
+ const ROCM_SMI_TIMEOUT_MS = 10_000;
158362
+ async function detectVRAMViaRocmSmi({ logger }) {
158186
158363
  try {
158187
- const { stdout } = await execa("rocm-smi", [
158188
- "--showbus",
158189
- "--showmeminfo",
158190
- "vram",
158191
- "--json"
158364
+ const [vramResult, gttResult] = await Promise.allSettled([
158365
+ execa("rocm-smi", ["--showbus", "--showmeminfo", "vram", "--json"], {
158366
+ timeout: ROCM_SMI_TIMEOUT_MS
158367
+ }),
158368
+ execa("rocm-smi", ["--showbus", "--showmeminfo", "gtt", "--json"], {
158369
+ timeout: ROCM_SMI_TIMEOUT_MS
158370
+ })
158192
158371
  ]);
158193
- const parsed = JSON.parse(stdout);
158194
- const results = [];
158195
- const cards = Object.entries(parsed).filter(([key]) => key.startsWith("card"));
158196
- for (const [, data] of cards) {
158197
- const bus = data["PCI Bus"] ?? null;
158198
- const totalStr = data["VRAM Total Memory (B)"] ?? null;
158199
- const usedStr = data["VRAM Total Used Memory (B)"] ?? null;
158200
- const totalBytes = totalStr !== null ? parseInt(totalStr, 10) : null;
158201
- const usedBytes = usedStr !== null ? parseInt(usedStr, 10) : null;
158202
- const validTotal = totalBytes !== null && Number.isFinite(totalBytes) && totalBytes >= 0;
158203
- const validUsed = usedBytes !== null && Number.isFinite(usedBytes) && usedBytes >= 0;
158204
- if (bus) {
158205
- results.push({
158206
- bus,
158207
- memoryTotalBytes: validTotal ? totalBytes : null,
158208
- memoryUsedBytes: validUsed ? usedBytes : null
158372
+ if (vramResult.status !== "fulfilled")
158373
+ return [];
158374
+ let gttEntries = [];
158375
+ if (gttResult.status === "fulfilled") {
158376
+ try {
158377
+ gttEntries = parseRocmSmiMemory({
158378
+ keyPrefix: "GTT",
158379
+ stdout: gttResult.value.stdout
158209
158380
  });
158210
158381
  }
158382
+ catch (error) {
158383
+ // Unusable GTT output (e.g. older rocm-smi) degrades to VRAM-only
158384
+ logger.warn("rocm-smi GTT output parse failed", { error: asError(error) });
158385
+ }
158211
158386
  }
158212
- return results;
158387
+ const gttByBus = new Map(gttEntries.map(entry => [entry.bus, entry]));
158388
+ return parseRocmSmiMemory({ keyPrefix: "VRAM", stdout: vramResult.value.stdout }).map(vram => {
158389
+ const gtt = gttByBus.get(vram.bus);
158390
+ return {
158391
+ bus: vram.bus,
158392
+ gttTotalBytes: gtt?.memoryTotalBytes ?? null,
158393
+ gttUsedBytes: gtt?.memoryUsedBytes ?? null,
158394
+ memoryTotalBytes: vram.memoryTotalBytes,
158395
+ memoryUsedBytes: vram.memoryUsedBytes
158396
+ };
158397
+ });
158213
158398
  }
158214
158399
  catch {
158215
158400
  return [];
@@ -158274,8 +158459,9 @@ async function detectGPUsViaNvidiaSmi({ logger }) {
158274
158459
  vendor: "NVIDIA"
158275
158460
  });
158276
158461
  }
158277
- // Unified-memory devices (e.g. NVIDIA GB10 on DGX Spark) report "[N/A]" for
158278
- // memory. Their compute pool is system RAM, so fall back to /proc/meminfo.
158462
+ // Shared/unified-memory devices (e.g. NVIDIA GB10 on DGX Spark) report
158463
+ // "[N/A]" for memory. Their compute pool is system RAM, so fall back to
158464
+ // /proc/meminfo. (AMD iGPUs get the same treatment via GTT merging.)
158279
158465
  if (gpus.some(gpu => gpu.memoryTotalBytes === null)) {
158280
158466
  const systemMemory = await readSystemMemoryBytes({ logger });
158281
158467
  const total = systemMemory.totalBytes;
@@ -158299,7 +158485,7 @@ async function detectGPUsViaNvidiaSmi({ logger }) {
158299
158485
  }
158300
158486
  }
158301
158487
  function buildMergedGPUs(options) {
158302
- const { lspciGPUs, nvidiaGPUs, rocmVRAM, siGPUs, sysfsVRAMMap } = options;
158488
+ const { lspciGPUs, nvidiaGPUs, rocmVRAM, siGPUs, systemTotalBytes, sysfsVRAMMap } = options;
158303
158489
  const rocmByBus = new Map();
158304
158490
  for (const entry of rocmVRAM) {
158305
158491
  const key = normalizeBusAddress(entry.bus);
@@ -158340,6 +158526,7 @@ function buildMergedGPUs(options) {
158340
158526
  gpu: existing,
158341
158527
  key,
158342
158528
  rocmByBus,
158529
+ systemTotalBytes,
158343
158530
  sysfsVRAMMap
158344
158531
  });
158345
158532
  }
@@ -158359,30 +158546,73 @@ function buildMergedGPUs(options) {
158359
158546
  gpu,
158360
158547
  key,
158361
158548
  rocmByBus,
158549
+ systemTotalBytes,
158362
158550
  sysfsVRAMMap
158363
158551
  });
158364
158552
  byBus.set(key, gpu);
158365
158553
  }
158366
158554
  return [...byBus.values()];
158367
158555
  }
158368
- function applySysfsOrRocmVRAM({ gpu, key, rocmByBus, sysfsVRAMMap }) {
158556
+ // Shared-memory GPUs (AMD iGPUs) expose a small dedicated VRAM carve-out via
158557
+ // mem_info_vram_total while the real usable pool - GTT - is carved dynamically
158558
+ // from system RAM. When GTT exceeds VRAM the device is treated as integrated
158559
+ // and the two pools combine (capped by installed RAM). Note that GTT defaults
158560
+ // to half of system RAM even on discrete GPUs, so a dGPU with less VRAM than
158561
+ // that also merges - acceptable, since GTT remains real addressable memory
158562
+ // for amdgpu compute (albeit slower over PCIe).
158563
+ function mergeGTTMemory({ gttTotalBytes, gttUsedBytes, systemTotalBytes, vramTotalBytes, vramUsedBytes }) {
158564
+ if (vramTotalBytes === null || !Number.isFinite(vramTotalBytes)) {
158565
+ return { memoryTotalBytes: null, memoryUsedBytes: null };
158566
+ }
158567
+ const isIntegrated = gttTotalBytes !== null && gttTotalBytes > vramTotalBytes;
158568
+ if (!isIntegrated) {
158569
+ return { memoryTotalBytes: vramTotalBytes, memoryUsedBytes: vramUsedBytes };
158570
+ }
158571
+ const combined = vramTotalBytes + gttTotalBytes;
158572
+ const capped = systemTotalBytes !== null && systemTotalBytes > 0
158573
+ ? Math.min(combined, systemTotalBytes)
158574
+ : combined;
158575
+ const hasCompleteUsage = vramUsedBytes !== null &&
158576
+ Number.isFinite(vramUsedBytes) &&
158577
+ gttUsedBytes !== null &&
158578
+ Number.isFinite(gttUsedBytes);
158579
+ const memoryUsedBytes = hasCompleteUsage
158580
+ ? Math.min(vramUsedBytes + gttUsedBytes, capped)
158581
+ : null;
158582
+ return { memoryTotalBytes: capped, memoryUsedBytes };
158583
+ }
158584
+ function applySysfsOrRocmVRAM({ gpu, key, rocmByBus, systemTotalBytes, sysfsVRAMMap }) {
158369
158585
  const sysfs = sysfsVRAMMap.get(key);
158370
- let totalBytes = sysfs?.memoryTotalBytes ?? null;
158371
- let usedBytes = sysfs?.memoryUsedBytes ?? null;
158372
- if (totalBytes === null) {
158586
+ let vramTotalBytes = sysfs?.memoryTotalBytes ?? null;
158587
+ let vramUsedBytes = sysfs?.memoryUsedBytes ?? null;
158588
+ let gttTotalBytes = sysfs?.gttTotalBytes ?? null;
158589
+ let gttUsedBytes = sysfs?.gttUsedBytes ?? null;
158590
+ if (vramTotalBytes === null) {
158373
158591
  const rocm = rocmByBus.get(key);
158374
158592
  if (rocm) {
158375
- totalBytes = rocm.memoryTotalBytes;
158376
- usedBytes = rocm.memoryUsedBytes;
158377
- }
158378
- }
158379
- if (totalBytes === null || !Number.isFinite(totalBytes))
158593
+ vramTotalBytes = rocm.memoryTotalBytes;
158594
+ vramUsedBytes = rocm.memoryUsedBytes;
158595
+ gttTotalBytes = rocm.gttTotalBytes;
158596
+ gttUsedBytes = rocm.gttUsedBytes;
158597
+ }
158598
+ }
158599
+ const merged = mergeGTTMemory({
158600
+ gttTotalBytes,
158601
+ gttUsedBytes,
158602
+ systemTotalBytes,
158603
+ vramTotalBytes,
158604
+ vramUsedBytes
158605
+ });
158606
+ if (merged.memoryTotalBytes === null || !Number.isFinite(merged.memoryTotalBytes))
158380
158607
  return;
158381
- gpu.memoryTotalBytes = totalBytes;
158382
- gpu.memoryUsedBytes = usedBytes !== null && Number.isFinite(usedBytes) ? usedBytes : null;
158608
+ gpu.memoryTotalBytes = merged.memoryTotalBytes;
158609
+ gpu.memoryUsedBytes =
158610
+ merged.memoryUsedBytes !== null && Number.isFinite(merged.memoryUsedBytes)
158611
+ ? merged.memoryUsedBytes
158612
+ : null;
158383
158613
  gpu.memoryFreeBytes =
158384
- usedBytes !== null && Number.isFinite(usedBytes)
158385
- ? Math.max(totalBytes - usedBytes, 0)
158614
+ gpu.memoryUsedBytes !== null
158615
+ ? Math.max(gpu.memoryTotalBytes - gpu.memoryUsedBytes, 0)
158386
158616
  : null;
158387
158617
  }
158388
158618
  async function collectMachineMetadata({ logger }) {
@@ -158393,7 +158623,7 @@ async function collectMachineMetadata({ logger }) {
158393
158623
  si.graphics(),
158394
158624
  detectGPUsViaLspci(),
158395
158625
  detectGPUsViaNvidiaSmi({ logger }),
158396
- detectVRAMViaRocmSmi()
158626
+ detectVRAMViaRocmSmi({ logger })
158397
158627
  ]);
158398
158628
  const cpuInfo = cpuResult.status === "fulfilled" ? cpuResult.value : null;
158399
158629
  const memInfo = memResult.status === "fulfilled" ? memResult.value : null;
@@ -158448,6 +158678,7 @@ async function collectMachineMetadata({ logger }) {
158448
158678
  nvidiaGPUs: resolvedNvidiaGPUs,
158449
158679
  rocmVRAM: resolvedRocmVRAM,
158450
158680
  siGPUs,
158681
+ systemTotalBytes: memInfo?.total ?? null,
158451
158682
  sysfsVRAMMap
158452
158683
  });
158453
158684
  const machineMetadata = {
@@ -158508,6 +158739,119 @@ async function detectDockerVersion() {
158508
158739
  }
158509
158740
  }
158510
158741
 
158742
+ /**
158743
+ * Flattens flat CLI extra-arg tokens into [arg, value] pairs, sorted by ARG NAME (ascending, ties by
158744
+ * value). `--flag=value` pairs split on the first `=`; a bare `--flag` consumes the following token
158745
+ * as its value when that token does not start with "-" (classic CLI convention); anything else
158746
+ * (flags, non-strings) is dropped.
158747
+ */
158748
+ function pairExtraArgs(tokens) {
158749
+ if (!Array.isArray(tokens)) {
158750
+ return [];
158751
+ }
158752
+ const list = tokens;
158753
+ const pairs = [];
158754
+ let index = 0;
158755
+ while (index < list.length) {
158756
+ const token = list[index];
158757
+ if (typeof token !== "string" || token.length === 0 || !token.startsWith("-")) {
158758
+ index++;
158759
+ continue;
158760
+ }
158761
+ const separator = token.indexOf("=");
158762
+ if (separator > -1) {
158763
+ const arg = token.slice(0, separator);
158764
+ if (arg.length > 0) {
158765
+ pairs.push([arg, token.slice(separator + 1)]);
158766
+ }
158767
+ index++;
158768
+ continue;
158769
+ }
158770
+ const next = list[index + 1];
158771
+ const consumesNext = typeof next === "string" && next.length > 0 && !next.startsWith("-");
158772
+ if (consumesNext) {
158773
+ pairs.push([token, next]);
158774
+ index += 2;
158775
+ }
158776
+ else {
158777
+ pairs.push([token, ""]);
158778
+ index++;
158779
+ }
158780
+ }
158781
+ return pairs.sort((a, b) => a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0);
158782
+ }
158783
+ /**
158784
+ * Files AT MOST ONE engine_execution report per engine startup. `beginStartup(runAt)` re-arms the
158785
+ * latch on every fresh model start (initial boot or cycle) and stamps the startup epoch used as the
158786
+ * row's `run_at`. Whichever of the three report* paths fires first wins: startup failure, in-flight
158787
+ * crash, or first successful full prompt completion.
158788
+ */
158789
+ class EngineExecutionReporter {
158790
+ options;
158791
+ currentStartupAt = null;
158792
+ reportedForCurrentStartup = false;
158793
+ constructor(options) {
158794
+ this.options = options;
158795
+ }
158796
+ /** Re-arms the latch; the stamp becomes the row's `run_at` (moment startup began). */
158797
+ beginStartup(runAt) {
158798
+ this.currentStartupAt = runAt;
158799
+ this.reportedForCurrentStartup = false;
158800
+ }
158801
+ /** Reports a startup failure (rejected prepare/start, readiness timeout, pre-ready death). */
158802
+ async reportStartupFailure(report) {
158803
+ await this.file(report, false);
158804
+ }
158805
+ /** Reports a spontaneous crash of an engine that had reached the running state. */
158806
+ async reportRuntimeCrash(report) {
158807
+ await this.file(report, false);
158808
+ }
158809
+ /** Reports the first fully-responded, token-bearing prompt completion since startup. */
158810
+ async reportSuccess(report) {
158811
+ await this.file(report, true);
158812
+ }
158813
+ async file(report, success) {
158814
+ if (this.reportedForCurrentStartup || this.currentStartupAt === null) {
158815
+ return;
158816
+ }
158817
+ // Latch BEFORE the network call: a thrown POST cannot double-file for this startup.
158818
+ this.reportedForCurrentStartup = true;
158819
+ const context = this.options.buildContext();
158820
+ const payload = {
158821
+ avgTps: report.throughput.avgTps,
158822
+ completionTokens: report.usage.completionTokens,
158823
+ durationMs: report.durationMs,
158824
+ engineType: context.engineType,
158825
+ engineVersion: context.engineVersion,
158826
+ errorDetail: success ? null : report.errorDetail,
158827
+ errorType: success ? null : report.errorType,
158828
+ extraArgs: context.extraArgsPairs,
158829
+ finishedAtISO: new Date().toISOString(),
158830
+ peakTps: report.throughput.peakTps,
158831
+ promptTokens: report.usage.promptTokens,
158832
+ runAtISO: this.currentStartupAt.toISOString(),
158833
+ success,
158834
+ ttftMs: report.ttftMs,
158835
+ totalTokens: report.usage.totalTokens
158836
+ };
158837
+ try {
158838
+ await this.options.report(payload);
158839
+ this.options.logger.info("Engine execution outcome reported", {
158840
+ inferenceSourceID: this.options.sourceLabel,
158841
+ success
158842
+ });
158843
+ }
158844
+ catch (error) {
158845
+ // Losing one report is preferable to filing two; the latch stays latched.
158846
+ this.options.logger.warn("Failed to report engine execution outcome", {
158847
+ error: asError(error),
158848
+ inferenceSourceID: this.options.sourceLabel,
158849
+ success
158850
+ });
158851
+ }
158852
+ }
158853
+ }
158854
+
158511
158855
  async function createApplication({ abortController, apiClient, configuration, logger }) {
158512
158856
  ensureDockerValidEnv();
158513
158857
  logger.info("Fetching conduit configuration");
@@ -158539,6 +158883,87 @@ async function createApplication({ abortController, apiClient, configuration, lo
158539
158883
  error: asError(error)
158540
158884
  });
158541
158885
  }
158886
+ const reporter = new EngineExecutionReporter({
158887
+ buildContext: () => {
158888
+ const engineType = (conduitConfiguration.engineConfig?.type ??
158889
+ "llama.cpp");
158890
+ const versions = {
158891
+ exllamav3: machine?.exllamav3Version ?? null,
158892
+ "llama.cpp": machine?.llamaCppVersion ?? null,
158893
+ "mlx-lm": machine?.mlxlmVersion ?? null,
158894
+ sglang: machine?.sglangVersion ?? null,
158895
+ "tensorrt-llm": machine?.tensorrtLlmVersion ?? null,
158896
+ vllm: machine?.vllmVersion ?? null
158897
+ };
158898
+ return {
158899
+ engineType,
158900
+ engineVersion: versions[engineType] ?? null,
158901
+ extraArgsPairs: pairExtraArgs(conduitConfiguration.engineConfig?.extraArgs)
158902
+ };
158903
+ },
158904
+ logger,
158905
+ report: payload => apiClient.reportEngineExecution(payload),
158906
+ sourceLabel: configuration.inferenceSourceID
158907
+ });
158908
+ // Intercept the prompt-metrics chokepoint so the first fully-responded, token-bearing prompt of
158909
+ // each fresh startup files the one-shot engine_execution success report. Handlers close over the
158910
+ // SAME `apiClient` object and read `reportPromptMetrics` at request-dispatch time (which always
158911
+ // follows this point), so the wrapped method is what they invoke.
158912
+ const rawReportPromptMetrics = apiClient.reportPromptMetrics;
158913
+ apiClient.reportPromptMetrics = async (payload) => {
158914
+ if (payload.successful && payload.completionTokens > 0 && payload.latencyMs > 0) {
158915
+ // The one-shot report is kicked off and its settlement attached HERE (before any await): if the
158916
+ // metrics path throws below, the report promise must still be able to log its own rejection.
158917
+ const successReport = reporter
158918
+ .reportSuccess({
158919
+ durationMs: payload.latencyMs,
158920
+ errorDetail: null,
158921
+ errorType: null,
158922
+ throughput: {
158923
+ avgTps: payload.tokensPerSecond,
158924
+ peakTps: null
158925
+ },
158926
+ ttftMs: payload.timeToFirstTokenMs ?? 0,
158927
+ usage: {
158928
+ completionTokens: payload.completionTokens,
158929
+ promptTokens: payload.promptTokens,
158930
+ totalTokens: payload.totalTokens
158931
+ }
158932
+ })
158933
+ .catch(error => {
158934
+ logger.warn("Engine execution success report failed", {
158935
+ error: asError(error)
158936
+ });
158937
+ });
158938
+ await rawReportPromptMetrics(payload);
158939
+ await successReport;
158940
+ return;
158941
+ }
158942
+ await rawReportPromptMetrics(payload);
158943
+ };
158944
+ const SECRET_ARG_MASK_PATTERN = /(-{1,2}[A-Za-z0-9_.]*(?:api[-_]?key|hf[-_]?token|token)(?:\s+|[=:]))\S+/gi;
158945
+ // Assembles the payload shared by the startup-failure and runtime-crash report paths. Stderr
158946
+ // may echo secrets, so mask `--api-key`/token-looking args before they reach the DB.
158947
+ function buildCrashReport(error, exitCode, signal) {
158948
+ const classification = classifyEngineFailure({ error, exitCode, signal });
158949
+ const raw = normalizeEngineError(error.message);
158950
+ const masked = raw.replace(SECRET_ARG_MASK_PATTERN, "$1***");
158951
+ return {
158952
+ durationMs: 0,
158953
+ errorDetail: masked.slice(0, 2048),
158954
+ errorType: classification,
158955
+ ttftMs: 0,
158956
+ throughput: {
158957
+ avgTps: 0,
158958
+ peakTps: null
158959
+ },
158960
+ usage: {
158961
+ completionTokens: 0,
158962
+ promptTokens: 0,
158963
+ totalTokens: 0
158964
+ }
158965
+ };
158966
+ }
158542
158967
  const conduitStateManager = new ConduitStateManager({
158543
158968
  initialState: {
158544
158969
  state: "initialising"
@@ -158590,6 +159015,17 @@ async function createApplication({ abortController, apiClient, configuration, lo
158590
159015
  });
158591
159016
  stopRequestedByControl = false;
158592
159017
  setErrorState({ error: normalizeEngineError(err.message) });
159018
+ // Spontaneous death of a SERVING engine → crash report, suppressed by the latch if the
159019
+ // startup's one-shot outcome was already filed. Startup-path failures report from
159020
+ // `startEngine`'s catch; this listener is the RUNTIME-crash path only.
159021
+ if (modelManager.wasRunning && !err.message.includes("interrupted by stop request")) {
159022
+ const crashReport = buildCrashReport(err, modelManager.lastExitCode, modelManager.lastExitSignal);
159023
+ reporter.reportRuntimeCrash(crashReport).catch(crashReportError => {
159024
+ logger.warn("Engine execution crash report failed", {
159025
+ error: asError(crashReportError)
159026
+ });
159027
+ });
159028
+ }
158593
159029
  });
158594
159030
  modelManager.on("engineReady", () => {
158595
159031
  setOnlineState();
@@ -158649,24 +159085,40 @@ async function createApplication({ abortController, apiClient, configuration, lo
158649
159085
  };
158650
159086
  async function startEngine() {
158651
159087
  logger.info("Engine start requested");
158652
- conduitStateManager.setState({
158653
- modelFileName,
158654
- modelName,
158655
- state: "downloadingModelFiles",
158656
- totalProgress: {
158657
- file: 0,
158658
- total: 0
159088
+ reporter.beginStartup(new Date());
159089
+ try {
159090
+ conduitStateManager.setState({
159091
+ modelFileName,
159092
+ modelName,
159093
+ state: "downloadingModelFiles",
159094
+ totalProgress: {
159095
+ file: 0,
159096
+ total: 0
159097
+ }
159098
+ });
159099
+ await conduitStateReportManager.reportNow();
159100
+ await modelManager.prepare({
159101
+ onDownloadProgress: reportDownloadProgress
159102
+ });
159103
+ conduitStateManager.setState({
159104
+ state: "bootingEngine"
159105
+ });
159106
+ await conduitStateReportManager.reportNow();
159107
+ await modelManager.start();
159108
+ }
159109
+ catch (error) {
159110
+ const parsedError = asError(error);
159111
+ // Operator-initiated aborts are not startup failures worth reporting.
159112
+ if (!parsedError.message.includes("interrupted by stop request")) {
159113
+ const startupReport = buildCrashReport(parsedError, modelManager.lastExitCode, modelManager.lastExitSignal);
159114
+ reporter.reportStartupFailure(startupReport).catch(startupReportError => {
159115
+ logger.warn("Engine execution startup report failed", {
159116
+ error: asError(startupReportError)
159117
+ });
159118
+ });
158659
159119
  }
158660
- });
158661
- await conduitStateReportManager.reportNow();
158662
- await modelManager.prepare({
158663
- onDownloadProgress: reportDownloadProgress
158664
- });
158665
- conduitStateManager.setState({
158666
- state: "bootingEngine"
158667
- });
158668
- await conduitStateReportManager.reportNow();
158669
- await modelManager.start();
159120
+ throw error;
159121
+ }
158670
159122
  }
158671
159123
  async function stopEngine({ reason }) {
158672
159124
  if (!modelManager.canStop) {
@@ -363121,12 +363573,13 @@ async function runModelFit(options) {
363121
363573
  console.log();
363122
363574
  const feasible = filterFeasibleModels({ detection, models: recommendedModels });
363123
363575
  if (feasible.length === 0) {
363576
+ const smallestTierGB = Math.min(...recommendedModels.map(model => model.vramTierGB));
363124
363577
  console.error("No recommended models fit this hardware. " +
363125
363578
  `Budget: ${formatBytes$1(detection.gpus.some(gpu => gpu.memoryTotalBytes)
363126
363579
  ? Math.max(...detection.gpus
363127
363580
  .map(gpu => gpu.memoryTotalBytes ?? 0)
363128
363581
  .filter(bytes => bytes > 0))
363129
- : detection.memory.totalBytes)}. Smallest tier starts at 2 GB.`);
363582
+ : detection.memory.totalBytes)}. Smallest tier starts at ${smallestTierGB} GB.`);
363130
363583
  process.exitCode = 1;
363131
363584
  return;
363132
363585
  }