badgr-cli 1.0.29 → 1.0.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "badgr-cli",
3
- "version": "1.0.29",
3
+ "version": "1.0.30",
4
4
  "description": "Badgr, run or serve GPU workloads from one command",
5
5
  "type": "module",
6
6
  "bin": {
@@ -1,7 +1,7 @@
1
- import readline from 'readline';
2
1
  import { requireApiKey } from '../config.js';
3
2
  import { callApi, terminateDeployment } from '../api.js';
4
3
  import { addReceipt, updateReceipt, generateReceiptId } from '../store.js';
4
+ import { normalizeTier, callWithFallback, HIGH_RATE_THRESHOLD } from '../fallback.js';
5
5
 
6
6
  /**
7
7
  * badgr run python train.py # gpu=auto, attached
@@ -21,8 +21,10 @@ export function parseRunArgs(args) {
21
21
  if (args[i] === '--max-price') { flags.maxPrice = parseFloat(args[++i]); i++; continue; }
22
22
  if (args[i] === '--name') { flags.name = args[++i]; i++; continue; }
23
23
  if (args[i] === '--detach') { flags.detach = true; i++; continue; }
24
- if (args[i] === '--fallback') { flags.fallback = args[++i]; i++; continue; }
25
- if (args[i] === '--no-fallback') { flags.noFallback = true; i++; continue; }
24
+ if (args[i] === '--fallback') { flags.fallback = args[++i]; i++; continue; }
25
+ if (args[i] === '--no-fallback') { flags.noFallback = true; i++; continue; }
26
+ if (args[i] === '--strict-capacity') { flags.noFallback = true; i++; continue; }
27
+ if (args[i] === '--no-expanded-search') { flags.noFallback = true; i++; continue; }
26
28
  if (args[i] === '--max-runtime') { flags.maxRuntime = parseFloat(args[++i]); i++; continue; }
27
29
  if (args[i] === '--max-cost') { flags.maxCost = parseFloat(args[++i]); i++; continue; }
28
30
  positional.push(args[i++]);
@@ -30,6 +32,25 @@ export function parseRunArgs(args) {
30
32
  return { flags, positional };
31
33
  }
32
34
 
35
+ // Mirror of backend workload_profile.py — kept in sync for pre-flight display.
36
+ const _PROFILES = {
37
+ smoke_test: { label: 'smoke test', vram: '4 GB', gpus: ['RTX 3080', 'RTX 3090', 'RTX 4090'] },
38
+ lora_finetune: { label: 'fine-tuning (LoRA)', vram: '40+ GB', gpus: ['A6000', 'L40S', 'A100'] },
39
+ image_gen: { label: 'image generation', vram: '16+ GB', gpus: ['RTX 4090', 'A6000', 'L40S'] },
40
+ inference_small: { label: 'inference (7B–8B model)', vram: '24+ GB', gpus: ['RTX 4090', 'A6000', 'L40S'] },
41
+ general: { label: 'GPU job', vram: '16+ GB', gpus: ['RTX 4090', 'RTX 3090', 'A6000', 'L40S'] },
42
+ };
43
+
44
+ function inferProfileFromCommand(cmdStr) {
45
+ const s = cmdStr.toLowerCase();
46
+ if (/print\s*\(|['"]hello/.test(s) && s.length < 100) return 'smoke_test';
47
+ if (/lora|qlora|finetune|fine[_-]tun|peft/.test(s)) return 'lora_finetune';
48
+ if (/diffusion|stable.?diff|sdxl|sd.?xl|comfyui|a1111|invoke|kohya/.test(s)) return 'image_gen';
49
+ if (/vllm|tgi|text.generation.inference/.test(s)) return 'inference_small';
50
+ if (/\btrain\.py\b/.test(s)) return 'lora_finetune';
51
+ return 'general';
52
+ }
53
+
33
54
  function fmtRuntime(ms) {
34
55
  const s = Math.round(ms / 1000);
35
56
  if (s < 60) return `${s}s`;
@@ -82,7 +103,7 @@ function renderStatusBar(chalk, { elapsedMs, ratePerHour, gpuUtil, cpuUtil, maxR
82
103
  }
83
104
 
84
105
  // Wait for status to leave 'starting'/'queued'/'provisioning'.
85
- // This covers image pull time — runtime limit does NOT start until this returns.
106
+ // The runtime limit does NOT start until this returns.
86
107
  // Returns the final dep object with status 'running' or 'failed'.
87
108
  async function waitForRunning(config, depId, chalk) {
88
109
  const POLL_MS = 3000;
@@ -285,11 +306,6 @@ async function attachToJob(config, depId, { chalk, maxRuntimeMs = null, maxCost
285
306
  }
286
307
  }
287
308
 
288
- function askConfirm(prompt) {
289
- const rl = readline.createInterface({ input: process.stdin, output: process.stdout });
290
- return new Promise(resolve => rl.question(prompt, ans => { rl.close(); resolve(ans.trim()); }));
291
- }
292
-
293
309
  export async function runCommand(config, args, chalk) {
294
310
  const { flags, positional } = parseRunArgs(args);
295
311
 
@@ -312,10 +328,7 @@ export async function runCommand(config, args, chalk) {
312
328
  const maxRuntimeMs = flags.maxRuntime ? flags.maxRuntime * 60 * 1000 : null;
313
329
  const maxCost = flags.maxCost ?? null;
314
330
 
315
- // Tier 1 = managed routing (default). Tier 2 = marketplace routing, opt-in via --tier 2.
316
- const effectiveTier = (flags.tier === '2' || flags.tier === 'tier2' || flags.tier === 'tier-2')
317
- ? '2'
318
- : (flags.tier || '1');
331
+ const effectiveTier = normalizeTier(flags.tier);
319
332
 
320
333
  // ── Auto GPU selection (no --gpu specified) ────────────────────────────────
321
334
  let gpu = flags.gpu ? flags.gpu.toUpperCase().replace('-', '_') : 'auto';
@@ -324,7 +337,16 @@ export async function runCommand(config, args, chalk) {
324
337
  console.log(chalk.bold('\n⚡ Running GPU job\n'));
325
338
  if (command) console.log(` ${chalk.bold('Command:')} ${command.join(' ')}`);
326
339
  if (image) console.log(` ${chalk.bold('Image:')} ${image}`);
327
- if (effectiveTier === '2') console.log(` ${chalk.dim('(tier 2 marketplace routing)')}`)
340
+ // Show workload estimate so the user knows what Badgr inferred.
341
+ if (command) {
342
+ const profKey = inferProfileFromCommand(cmdStr);
343
+ const prof = _PROFILES[profKey];
344
+ console.log();
345
+ console.log(` ${chalk.bold('Estimated workload:')} ${prof.label}`);
346
+ console.log(` ${chalk.bold('Estimated minimum VRAM:')} ${prof.vram}`);
347
+ if (maxCost) console.log(` ${chalk.bold('Max cost:')} $${maxCost.toFixed(2)}`);
348
+ if (flags.maxRuntime) console.log(` ${chalk.bold('Max runtime:')} ${flags.maxRuntime}min`);
349
+ }
328
350
  console.log();
329
351
  } else {
330
352
  // Specific GPU requested — show header
@@ -343,12 +365,12 @@ export async function runCommand(config, args, chalk) {
343
365
  }
344
366
  }
345
367
 
346
- console.log(chalk.dim(' Finding reliable capacity...'));
368
+ console.log(chalk.dim(' Finding suitable capacity...'));
347
369
  if (process.env.BADGR_DEBUG === '1' || process.env.BADGR_DEBUG === 'true') {
348
370
  console.log(chalk.dim(` API: ${config.baseUrl}`));
349
371
  }
350
372
 
351
- function buildBody(gpuOverride) {
373
+ function buildBody(gpuOverride, tierOverride) {
352
374
  const effectiveRegion = flags.region ? flags.region.toUpperCase() : undefined;
353
375
  return {
354
376
  command,
@@ -358,94 +380,28 @@ export async function runCommand(config, args, chalk) {
358
380
  ...(effectiveRegion ? { region: effectiveRegion } : {}),
359
381
  max_price_per_hour: flags.maxPrice,
360
382
  name: flags.name,
361
- tier: effectiveTier,
383
+ tier: tierOverride || effectiveTier,
362
384
  };
363
385
  }
364
386
 
365
387
  let dep;
366
388
  try {
367
- dep = await callApi('/run', {
368
- method: 'POST',
369
- apiKey: config.apiKey,
370
- baseUrl: config.baseUrl,
371
- body: buildBody(),
372
- });
389
+ dep = await callWithFallback(
390
+ '/run',
391
+ { apiKey: config.apiKey, baseUrl: config.baseUrl },
392
+ (tierOverride) => buildBody(undefined, tierOverride),
393
+ effectiveTier,
394
+ chalk,
395
+ { thing: 'job', cmd: 'badgr run' },
396
+ );
373
397
  } catch (err) {
374
- const d = err.errorData;
375
- if (d?.code === 'NO_CAPACITY_MATCH') {
376
- if (effectiveTier === '2') {
377
- console.error(chalk.red('\n ✗ No GPU capacity available right now on any provider.\n'));
378
- console.error(chalk.dim(' Try `badgr capacity` to see what\'s available, or try again shortly.'));
379
- process.exit(1);
380
- }
381
-
382
- // Tier 1 out of capacity — offer tier 2 marketplace routing.
383
- if (process.stdin.isTTY) {
384
- const answer = await askConfirm(
385
- `\n No tier 1 capacity available. Try tier 2 marketplace routing? [${chalk.bold('Enter')}/${chalk.bold('q')}]: `
386
- );
387
- if (answer.toLowerCase() === 'q') {
388
- console.log(chalk.dim('\n Cancelled.\n'));
389
- process.exit(0);
390
- }
391
- } else {
392
- console.log(chalk.dim('\n No tier 1 capacity — trying tier 2 marketplace routing...\n'));
393
- }
394
-
395
- console.log(chalk.dim(' Searching tier 2 capacity...'));
396
- try {
397
- dep = await callApi('/run', {
398
- method: 'POST',
399
- apiKey: config.apiKey,
400
- baseUrl: config.baseUrl,
401
- body: { ...buildBody(), tier: '2' },
402
- });
403
- } catch (err2) {
404
- const d2 = err2.errorData;
405
- if (d2?.code === 'NO_CAPACITY_MATCH') {
406
- console.error(chalk.red('\n ✗ No GPU capacity available right now on any provider.\n'));
407
- console.error(chalk.dim(' Try `badgr capacity` to see what\'s available, or try again shortly.'));
408
- } else if (d2?.code === 'PROVISIONING_FAILED' || d2?.code === 'PROVIDER_ADAPTER_ERROR') {
409
- console.error(chalk.red(`\n ✗ Budget provider found capacity but could not start the machine. Please try again.\n`));
410
- // (error detail kept below)
411
- if (process.env.BADGR_DEBUG === '1' || process.env.BADGR_DEBUG === 'true') {
412
- if (d2?.debug_error) console.error(chalk.dim(` Provider detail: ${d2.debug_error}`));
413
- } else {
414
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr run … for a full trace.\n`));
415
- }
416
- } else {
417
- console.error(chalk.red(`\n ✗ Could not start job on tier 2: ${err2.message}\n`));
418
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr run … for a full trace.`));
419
- }
420
- process.exit(1);
421
- }
422
- } else if (d?.code === 'PROVISIONING_FAILED' || d?.code === 'PROVIDER_ADAPTER_ERROR') {
423
- if (d?.low_cost_provider_failed) {
424
- console.error(chalk.red(`\n ✗ Tier 2 unavailable. Tier 1 also unavailable. Try again shortly.\n`));
425
- } else {
426
- console.error(chalk.red(`\n ✗ Badgr found ${gpu} capacity but could not start the machine. Please try again.\n`));
427
- }
428
- if (process.env.BADGR_DEBUG === '1' || process.env.BADGR_DEBUG === 'true') {
429
- if (d?.debug_error) console.error(chalk.dim(` Provider detail: ${d.debug_error}`));
430
- } else {
431
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr run … for a full trace.\n`));
432
- }
433
- process.exit(1);
434
- } else if (err.isPaymentRequired) {
398
+ if (err.isPaymentRequired) {
435
399
  console.error(chalk.yellow(err.message));
436
400
  const rerun = ['badgr run', ...args].join(' ');
437
401
  console.error(chalk.dim(`After payment, rerun:\n ${rerun}\n`));
438
402
  process.exit(1);
439
- } else {
440
- console.error(chalk.red(`\n ✗ Could not start job: ${err.message}\n`));
441
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr run … for a full trace.`));
442
- console.error(chalk.dim(` Check config: badgr config\n`));
443
- process.exit(1);
444
403
  }
445
- }
446
-
447
- if (dep.provider_fallback_note === 'tier2_failed_using_tier1') {
448
- console.log(chalk.yellow(' ℹ Tier 2 unavailable. Running on Tier 1 instead.\n'));
404
+ throw err;
449
405
  }
450
406
 
451
407
  const rcptId = dep.receipt_id || generateReceiptId();
@@ -455,20 +411,27 @@ export async function runCommand(config, args, chalk) {
455
411
  deploymentId: dep.deployment_id,
456
412
  gpu: dep.gpu_type,
457
413
  providerRoute: dep.provider ?? null,
414
+ tier: dep.tier ?? null,
458
415
  maxCost: maxCost ?? null,
459
416
  maxRuntime: flags.maxRuntime ?? null,
460
417
  status: dep.status,
461
418
  createdAt: new Date().toISOString(),
462
419
  });
463
420
 
464
- console.log();
421
+ const rate = dep.cost_per_hour || 0;
422
+
423
+ console.log(chalk.dim(' Capacity found.\n'));
465
424
  console.log(chalk.bold(` Job ID: ${chalk.cyan(dep.deployment_id)}`));
466
425
  console.log(` ${chalk.bold('GPU:')} ${dep.gpu_type} × ${dep.gpu_count}`);
467
- if (dep.workload_desc) console.log(` ${chalk.bold('Workload:')} ${dep.workload_desc}`);
468
- if (dep.tier) console.log(` ${chalk.bold('Tier:')} ${dep.tier}`);
469
- if (dep.cost_per_hour > 0) console.log(` ${chalk.bold('Rate:')} $${dep.cost_per_hour.toFixed(2)}/hr`);
426
+ if (rate > 0) console.log(` ${chalk.bold('Rate:')} $${rate.toFixed(2)}/hr`);
427
+ if (maxCost) console.log(` ${chalk.bold('Max cost:')} $${maxCost.toFixed(2)}`);
470
428
  console.log(` ${chalk.bold('Receipt:')} ${chalk.dim(rcptId)}`);
471
429
 
430
+ if (rate > HIGH_RATE_THRESHOLD && !maxCost) {
431
+ console.log(chalk.yellow(`\n Selected capacity rate: $${rate.toFixed(2)}/hr`));
432
+ console.log(chalk.dim(' Tip: use --max-cost to enforce a hard ceiling.'));
433
+ }
434
+
472
435
  if (detach) {
473
436
  console.log(`\n ${chalk.bold('Logs:')} ${dep.logs_url || `badgr logs ${dep.deployment_id}`}`);
474
437
  console.log(chalk.dim(`\n Detached. Track progress: badgr logs ${dep.deployment_id}\n`));
@@ -476,7 +439,7 @@ export async function runCommand(config, args, chalk) {
476
439
  }
477
440
 
478
441
  // Wait through startup phases (queued → provisioning → starting → running).
479
- // The max-runtime clock does NOT start until this returns — image pull time is free.
442
+ // The max-runtime clock does NOT start until this returns.
480
443
  const STARTUP_STATES = new Set(['queued', 'provisioning', 'starting']);
481
444
  if (STARTUP_STATES.has(dep.status)) {
482
445
  console.log();
@@ -1,12 +1,6 @@
1
- import readline from 'readline';
2
1
  import { requireApiKey } from '../config.js';
3
- import { callApi } from '../api.js';
4
2
  import { addDeployment, addReceipt, updateReceipt, generateReceiptId } from '../store.js';
5
-
6
- function askConfirm(prompt) {
7
- const rl = readline.createInterface({ input: process.stdin, output: process.stdout });
8
- return new Promise(resolve => rl.question(prompt, ans => { rl.close(); resolve(ans.trim()); }));
9
- }
3
+ import { normalizeTier, callWithFallback, HIGH_RATE_THRESHOLD } from '../fallback.js';
10
4
 
11
5
  /**
12
6
  * badgr serve meta-llama/Llama-3.1-8B-Instruct
@@ -25,13 +19,33 @@ export function parseServeArgs(args) {
25
19
  if (args[i] === '--tier') { flags.tier = args[++i]; i++; continue; }
26
20
  if (args[i] === '--max-price') { flags.maxPrice = parseFloat(args[++i]); i++; continue; }
27
21
  if (args[i] === '--name') { flags.name = args[++i]; i++; continue; }
28
- if (args[i] === '--no-wait') { flags.noWait = true; i++; continue; }
22
+ if (args[i] === '--no-wait') { flags.noWait = true; i++; continue; }
23
+ if (args[i] === '--max-cost') { flags.maxCost = parseFloat(args[++i]); i++; continue; }
24
+ if (args[i] === '--strict-capacity') { flags.noMarketplaceFallback = true; i++; continue; }
25
+ if (args[i] === '--no-expanded-search') { flags.noMarketplaceFallback = true; i++; continue; }
29
26
  positional.push(args[i++]);
30
27
  }
31
28
  const model = positional[0] || null;
32
29
  return { model, flags };
33
30
  }
34
31
 
32
+ // Mirror of backend workload_profile.py infer_profile_from_model — for pre-flight display.
33
+ function _inferServeProfile(modelName) {
34
+ const s = modelName.toLowerCase();
35
+ const moe = s.match(/(\d+)x(\d+)b/);
36
+ let paramsB;
37
+ if (moe) {
38
+ paramsB = parseInt(moe[1]) * parseInt(moe[2]);
39
+ } else {
40
+ const m = s.match(/(\d+)b/);
41
+ paramsB = m ? parseInt(m[1]) : null;
42
+ }
43
+ if (paramsB === null) return { label: 'inference (7B–8B model)', vram: '24+ GB', gpus: ['RTX 4090', 'A6000', 'L40S'] };
44
+ if (paramsB <= 9) return { label: 'inference (7B–8B model)', vram: '24+ GB', gpus: ['RTX 4090', 'A6000', 'L40S'] };
45
+ if (paramsB <= 35) return { label: 'inference (30B–34B model)', vram: '40+ GB', gpus: ['A6000', 'L40S', 'A100'] };
46
+ return { label: 'inference (70B+ model)', vram: '80+ GB', gpus: ['H100', 'A100'] };
47
+ }
48
+
35
49
  function _serveStageLabel(elapsedSec) {
36
50
  if (elapsedSec < 45) return 'Starting vLLM…';
37
51
  if (elapsedSec < 150) return 'Downloading model…';
@@ -74,118 +88,68 @@ export async function serveCommand(config, args, chalk) {
74
88
  const gpu = flags.gpu ? flags.gpu.toUpperCase().replace('-', '_') : 'AUTO';
75
89
  const gpuLabel = gpu === 'AUTO' ? 'auto' : gpu;
76
90
 
77
- // Tier 1 = managed routing (default). Tier 2 = marketplace routing, opt-in via --tier 2.
78
- const effectiveTier = (flags.tier === '2' || flags.tier === 'tier2' || flags.tier === 'tier-2')
79
- ? '2'
80
- : (flags.tier || '1');
91
+ const effectiveTier = normalizeTier(flags.tier);
81
92
 
82
- console.log(chalk.bold('\nServing model\n'));
93
+ console.log(chalk.bold('\n⚡ Serving model\n'));
83
94
  console.log(` ${chalk.bold('Model:')} ${model}`);
84
95
  console.log(` ${chalk.bold('GPU:')} ${gpuLabel}`);
85
- if (effectiveTier === '2') console.log(` ${chalk.dim('(tier 2 — marketplace routing)')}`);
96
+
97
+ // Show inferred workload so the user knows what Badgr detected.
98
+ if (gpu === 'AUTO') {
99
+ const prof = _inferServeProfile(model);
100
+ console.log();
101
+ console.log(` ${chalk.bold('Estimated workload:')} ${prof.label}`);
102
+ console.log(` ${chalk.bold('Estimated minimum VRAM:')} ${prof.vram}`);
103
+ if (flags.maxCost) console.log(` ${chalk.bold('Max cost:')} $${flags.maxCost.toFixed(2)}`);
104
+ }
86
105
  console.log();
87
- process.stdout.write(chalk.dim(' Finding reliable capacity...\n'));
106
+ process.stdout.write(chalk.dim(' Finding suitable capacity...\n'));
88
107
 
89
108
  const effectiveRegion = flags.region ? flags.region.toUpperCase() : undefined;
90
109
 
91
- function buildBody(gpuOverride, regionOverride) {
110
+ function buildBody(gpuOverride, tierOverride) {
92
111
  return {
93
112
  model,
94
113
  gpu: gpuOverride || gpu,
95
114
  gpu_count: flags.count || 1,
96
- ...(regionOverride || effectiveRegion ? { region: regionOverride || effectiveRegion } : {}),
115
+ ...(effectiveRegion ? { region: effectiveRegion } : {}),
97
116
  max_price_per_hour: flags.maxPrice,
98
117
  name: flags.name,
99
- tier: effectiveTier,
118
+ tier: tierOverride || effectiveTier,
100
119
  };
101
120
  }
102
121
 
103
122
  let dep;
104
123
  try {
105
- dep = await callApi('/serve', {
106
- method: 'POST',
107
- apiKey: config.apiKey,
108
- baseUrl: config.baseUrl,
109
- body: buildBody(),
110
- });
124
+ dep = await callWithFallback(
125
+ '/serve',
126
+ { apiKey: config.apiKey, baseUrl: config.baseUrl },
127
+ (tierOverride) => buildBody(undefined, tierOverride),
128
+ effectiveTier,
129
+ chalk,
130
+ { thing: 'endpoint', cmd: 'badgr serve' },
131
+ );
111
132
  } catch (err) {
112
- const d = err.errorData;
113
- if (d?.code === 'NO_CAPACITY_MATCH') {
114
- if (effectiveTier === '2') {
115
- console.error(chalk.red('\n ✗ No GPU capacity available right now on any provider.\n'));
116
- console.error(chalk.dim(' Run `badgr capacity` to see alternatives, or try again shortly.'));
117
- process.exit(1);
118
- }
119
-
120
- // Tier 1 out of capacity — offer tier 2 marketplace routing.
121
- if (process.stdin.isTTY) {
122
- const answer = await askConfirm(
123
- `\n No tier 1 capacity available. Try tier 2 marketplace routing? [${chalk.bold('Enter')}/${chalk.bold('q')}]: `
124
- );
125
- if (answer.toLowerCase() === 'q') {
126
- console.log(chalk.dim('\n Cancelled.\n'));
127
- process.exit(0);
128
- }
129
- } else {
130
- console.log(chalk.dim('\n No tier 1 capacity — trying tier 2 marketplace routing...\n'));
131
- }
132
-
133
- console.log(chalk.dim(' Searching tier 2 capacity...'));
134
- try {
135
- dep = await callApi('/serve', {
136
- method: 'POST',
137
- apiKey: config.apiKey,
138
- baseUrl: config.baseUrl,
139
- body: { ...buildBody(), tier: '2' },
140
- });
141
- } catch (err2) {
142
- const d2 = err2.errorData;
143
- if (d2?.code === 'NO_CAPACITY_MATCH') {
144
- console.error(chalk.red('\n ✗ No GPU capacity available right now on any provider.\n'));
145
- console.error(chalk.dim(' Run `badgr capacity` to see alternatives, or try again shortly.'));
146
- } else if (d2?.code === 'PROVISIONING_FAILED' || d2?.code === 'PROVIDER_ADAPTER_ERROR') {
147
- console.error(chalk.red(`\n ✗ Budget provider found capacity but could not start the endpoint. Please try again.\n`));
148
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr serve … for a full trace.\n`));
149
- } else {
150
- console.error(chalk.red(`\n ✗ Could not start endpoint on tier 2: ${err2.message}\n`));
151
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr serve … for a full trace.\n`));
152
- }
153
- process.exit(1);
154
- }
155
- } else if (d?.code === 'PROVISIONING_FAILED' || d?.code === 'PROVIDER_ADAPTER_ERROR') {
156
- if (d?.low_cost_provider_failed) {
157
- console.error(chalk.red(`\n ✗ Tier 2 unavailable. Tier 1 also unavailable. Try again shortly.\n`));
158
- } else {
159
- console.error(chalk.red(`\n ✗ Badgr found capacity but could not start the endpoint. Please try again.\n`));
160
- }
161
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr serve … for a full trace.\n`));
162
- process.exit(1);
163
- } else {
164
- const failRcptId = generateReceiptId();
165
- addReceipt({
166
- receiptId: failRcptId,
167
- action: 'badgr serve',
168
- model,
169
- gpu: gpuLabel,
170
- status: 'failed',
171
- failureType: 'infrastructure',
172
- createdAt: new Date().toISOString(),
173
- });
174
- if (err.isPaymentRequired) {
175
- console.error(chalk.yellow(err.message));
176
- const rerun = `badgr serve ${args.join(' ')}`;
177
- console.error(chalk.dim(`After payment, rerun:\n ${rerun}\n`));
178
- return;
179
- }
180
- console.error(chalk.red(`\n ✗ Could not start endpoint: ${err.message}`));
181
- console.error(chalk.dim(`\n Receipt: ${failRcptId}`));
182
- console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr serve … for a full trace.\n`));
133
+ const failRcptId = generateReceiptId();
134
+ addReceipt({
135
+ receiptId: failRcptId,
136
+ action: 'badgr serve',
137
+ model,
138
+ gpu: gpuLabel,
139
+ status: 'failed',
140
+ failureType: 'infrastructure',
141
+ createdAt: new Date().toISOString(),
142
+ });
143
+ if (err.isPaymentRequired) {
144
+ console.error(chalk.yellow(err.message));
145
+ const rerun = `badgr serve ${args.join(' ')}`;
146
+ console.error(chalk.dim(`After payment, rerun:\n ${rerun}\n`));
183
147
  return;
184
148
  }
185
- }
186
-
187
- if (dep.provider_fallback_note === 'tier2_failed_using_tier1') {
188
- console.log(chalk.yellow(' ℹ Tier 2 unavailable. Running on Tier 1 instead.\n'));
149
+ console.error(chalk.red(`\n ✗ Could not start endpoint: ${err.message}`));
150
+ console.error(chalk.dim(`\n Receipt: ${failRcptId}`));
151
+ console.error(chalk.dim(` Run BADGR_DEBUG=1 badgr serve … for a full trace.\n`));
152
+ return;
189
153
  }
190
154
 
191
155
  addDeployment({
@@ -200,6 +164,8 @@ export async function serveCommand(config, args, chalk) {
200
164
  receiptId: dep.receipt_id,
201
165
  createdAt: new Date().toISOString(),
202
166
  costPerHour: dep.cost_per_hour || 0,
167
+ providerRoute: dep.provider ?? null,
168
+ tier: dep.tier ?? null,
203
169
  });
204
170
 
205
171
  const rcptId = dep.receipt_id || generateReceiptId();
@@ -208,6 +174,8 @@ export async function serveCommand(config, args, chalk) {
208
174
  action: 'badgr serve',
209
175
  deploymentId: dep.deployment_id,
210
176
  gpu: dep.gpu_type,
177
+ providerRoute: dep.provider ?? null,
178
+ tier: dep.tier ?? null,
211
179
  status: dep.status,
212
180
  createdAt: new Date().toISOString(),
213
181
  });
@@ -234,15 +202,24 @@ export async function serveCommand(config, args, chalk) {
234
202
  console.log();
235
203
  }
236
204
 
237
- console.log(` ${chalk.bold('Base URL:')} ${chalk.cyan(endpointUrl)}`);
238
- console.log(` ${chalk.bold('Model:')} ${dep.model || model}`);
239
- console.log(` ${chalk.bold('GPU:')} ${dep.gpu_type} × ${dep.gpu_count}`);
240
- if (dep.tier) console.log(` ${chalk.bold('Tier:')} ${dep.tier}`);
241
- if (dep.cost_per_hour > 0) console.log(` ${chalk.bold('Rate:')} $${dep.cost_per_hour.toFixed(2)}/hr`);
242
- console.log(` ${chalk.bold('Logs:')} ${chalk.dim(`badgr logs ${dep.deployment_id}`)}`);
243
- console.log(` ${chalk.bold('Stop billing:')} ${chalk.dim(`badgr down ${dep.deployment_id}`)}`);
205
+ const serveRate = dep.cost_per_hour || 0;
206
+
207
+ console.log(` ${chalk.bold('Base URL:')} ${chalk.cyan(endpointUrl)}`);
208
+ console.log(` ${chalk.bold('Model:')} ${dep.model || model}`);
209
+ console.log(` ${chalk.bold('GPU:')} ${dep.gpu_type} × ${dep.gpu_count}`);
210
+ if (serveRate > 0) console.log(` ${chalk.bold('Rate:')} $${serveRate.toFixed(2)}/hr`);
211
+ if (flags.maxCost) console.log(` ${chalk.bold('Max cost:')} $${flags.maxCost.toFixed(2)}`);
212
+ console.log(` ${chalk.bold('Receipt:')} ${chalk.dim(rcptId)}`);
213
+ console.log(` ${chalk.bold('Logs:')} ${chalk.dim(`badgr logs ${dep.deployment_id}`)}`);
214
+ console.log(` ${chalk.bold('Stop billing:')} ${chalk.cyan(`badgr down ${dep.deployment_id}`)}`);
244
215
  console.log();
245
216
 
217
+ if (serveRate > HIGH_RATE_THRESHOLD && !flags.maxCost) {
218
+ console.log(chalk.yellow(` Selected capacity rate: $${serveRate.toFixed(2)}/hr`));
219
+ console.log(chalk.dim(' Tip: use --max-cost to enforce a hard ceiling.\n'));
220
+ }
221
+ console.log(chalk.dim(' Billing continues until you run: ') + chalk.cyan(`badgr down ${dep.deployment_id}`));
222
+
246
223
  if (endpointReady) {
247
224
  const keySnip = config.apiKey?.slice(0, 8) || 'sk-...';
248
225
  console.log(` ${chalk.bold('Use with OpenAI SDK:')}`);
package/src/fallback.js CHANGED
@@ -1,5 +1,80 @@
1
1
  import readline from 'readline';
2
2
 
3
+ // ── Shared routing helpers ────────────────────────────────────────────────────
4
+
5
+ /** Rates above this threshold trigger a visible warning when no --max-cost is set. */
6
+ export const HIGH_RATE_THRESHOLD = 3.00;
7
+
8
+ /** Normalise --tier flag variants to '1' or '2'. */
9
+ export function normalizeTier(tier) {
10
+ return (tier === '2' || tier === 'tier2' || tier === 'tier-2') ? '2' : (tier || '1');
11
+ }
12
+
13
+ /**
14
+ * Call an API endpoint with automatic tier-2 expansion on NO_CAPACITY_MATCH.
15
+ * Returns the deployment object on success; throws or calls process.exit on failure.
16
+ *
17
+ * @param {string} endpoint - '/run' or '/serve'
18
+ * @param {object} callOpts - { apiKey, baseUrl }
19
+ * @param {function} buildBody - (tierOverride?) => body object
20
+ * @param {string} effectiveTier
21
+ * @param {object} chalk
22
+ * @param {object} labels - { thing: 'job'|'endpoint', cmd: 'badgr run'|'badgr serve' }
23
+ */
24
+ export async function callWithFallback(endpoint, callOpts, buildBody, effectiveTier, chalk, labels) {
25
+ const { callApi } = await import('./api.js');
26
+ const thing = labels?.thing ?? 'job';
27
+ const cmd = labels?.cmd ?? 'badgr run';
28
+
29
+ async function attempt(body) {
30
+ return callApi(endpoint, { method: 'POST', ...callOpts, body });
31
+ }
32
+
33
+ function handleErr(err, isFallback) {
34
+ const d = err.errorData;
35
+ if (d?.code === 'NO_CAPACITY_MATCH') {
36
+ console.error(chalk.red('\n ✗ No suitable GPU capacity available right now.\n'));
37
+ console.error(chalk.dim(' Try `badgr capacity` to see what\'s available, or try again shortly.'));
38
+ } else if (d?.code === 'PROVISIONING_FAILED' || d?.code === 'PROVIDER_ADAPTER_ERROR') {
39
+ if (d?.low_cost_provider_failed) {
40
+ console.error(chalk.red(`\n ✗ No suitable capacity available right now. Try again shortly.\n`));
41
+ } else {
42
+ console.error(chalk.red(`\n ✗ Capacity found but ${thing} failed to start. Please try again.\n`));
43
+ if (process.env.BADGR_DEBUG === '1' || process.env.BADGR_DEBUG === 'true') {
44
+ if (d?.debug_error) console.error(chalk.dim(` Detail: ${d.debug_error}`));
45
+ } else {
46
+ console.error(chalk.dim(` Run BADGR_DEBUG=1 ${cmd} … for a full trace.\n`));
47
+ }
48
+ }
49
+ } else {
50
+ console.error(chalk.red(`\n ✗ Could not start ${thing}: ${err.message}\n`));
51
+ console.error(chalk.dim(` Run BADGR_DEBUG=1 ${cmd} … for a full trace.`));
52
+ if (!isFallback) console.error(chalk.dim(` Check config: badgr config\n`));
53
+ }
54
+ process.exit(1);
55
+ }
56
+
57
+ try {
58
+ return await attempt(buildBody());
59
+ } catch (err) {
60
+ const d = err.errorData;
61
+
62
+ if (d?.code === 'NO_CAPACITY_MATCH' && effectiveTier !== '2') {
63
+ console.log(chalk.dim('\n Primary capacity unavailable — expanding search...\n'));
64
+ try {
65
+ return await attempt(buildBody('2'));
66
+ } catch (err2) {
67
+ handleErr(err2, true);
68
+ }
69
+ }
70
+
71
+ if (err.isPaymentRequired) throw err; // let caller handle payment errors
72
+ handleErr(err, false);
73
+ }
74
+ }
75
+
76
+ // ── GPU fallback prompt ───────────────────────────────────────────────────────
77
+
3
78
  // GPU descriptions for display only — no scoring logic lives here.
4
79
  // Ranking is computed server-side and returned as `rank` on each alternative.
5
80
  const GPU_DISPLAY = {