@genex-ai/cli-demo 1.35.0-dev.741 → 1.35.0-dev.743

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -23288,7 +23288,7 @@ import path35 from "path";
23288
23288
  import fs33 from "fs/promises";
23289
23289
  import path34 from "path";
23290
23290
  import { randomUUID as randomUUID2 } from "crypto";
23291
- var SUBS3 = ["models", "bench", "price"];
23291
+ var SUBS3 = ["models", "bench", "price", "status", "cancel"];
23292
23292
  var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN coin \u2014 your wallet, not a player's, and one this build's asset allowance cannot see. Re-run with --max-coins <n> --user-approved only after the person at the keyboard agreed to the number.";
23293
23293
  var LLM_LANE_OFF_LINE = "The runtime LLM lane is off on this stand \u2014 in-game generate() answers 404 here, and nothing can be benchmarked.";
23294
23294
  var LLM_TOOLS_CONVERT_HINT = [
@@ -23297,14 +23297,34 @@ var LLM_TOOLS_CONVERT_HINT = [
23297
23297
  "Ask the user first; on their yes run `npx genex init --convert`, then load the genex-tool-llm card."
23298
23298
  ];
23299
23299
  var LLM_BENCH_FILE = ".genex/llm-bench.json";
23300
+ function serverSlotCounts(body) {
23301
+ if (!body || typeof body.slotsHeld !== "number" || typeof body.slotLimit !== "number" || body.slotLimit <= 0) return null;
23302
+ return { held: body.slotsHeld, limit: body.slotLimit };
23303
+ }
23304
+ var PENDING_SLOT_RELEASE_MINUTES = 10;
23305
+ var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so its coin stays held until the bill resolves; such a row stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`;
23306
+ var LLM_STATUS_NO_COUNT_LINE = "This stand does not report how many slots are held across your projects, so no count is shown.";
23307
+ var LLM_CANCEL_AWAITING_BILL = "this call already stopped; its bill is awaiting the provider";
23308
+ var LLM_CANCEL_ALREADY_FINAL = "this call is already final";
23309
+ function benchSlotsHeldSentence(counts) {
23310
+ const count = counts === null ? "every ad-hoc slot is" : `${counts.held} of ${counts.limit} ad-hoc slots are`;
23311
+ return `refused before it started, nothing spent: ${count} held on this stand \u2014 \`genex llm status\` shows the calls holding them; wait for a bill to resolve or cancel an active one, then re-run.`;
23312
+ }
23313
+ var LLM_PROVIDER_KEY_REJECTED_LINE = "This stand's provider key is refused \u2014 every in-game call answers provider_http_401 until an operator replaces it.";
23300
23314
  var DEFAULT_BENCH_SAMPLES = 3;
23301
23315
  var DEFAULT_SAMPLE_TIMEOUT_SEC = 180;
23316
+ function parseProviderKey(value) {
23317
+ return value === "ok" || value === "rejected" || value === "unknown" ? value : null;
23318
+ }
23302
23319
  async function readRuntimeLane(apiUrl, token) {
23303
23320
  const empty = (state) => ({
23304
23321
  state,
23305
23322
  models: null,
23306
23323
  recommendedDeclaredHeadroomBps: null,
23307
- externalProviders: null
23324
+ externalProviders: null,
23325
+ total: null,
23326
+ featuredCount: null,
23327
+ providerKey: null
23308
23328
  });
23309
23329
  try {
23310
23330
  const res = await apiFetch(`${apiUrl}/api/runtime/models`, {
@@ -23320,7 +23340,10 @@ async function readRuntimeLane(apiUrl, token) {
23320
23340
  state: "live",
23321
23341
  models: body.models,
23322
23342
  recommendedDeclaredHeadroomBps: typeof body.recommendedDeclaredHeadroomBps === "number" && Number.isFinite(body.recommendedDeclaredHeadroomBps) ? body.recommendedDeclaredHeadroomBps : null,
23323
- externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null
23343
+ externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null,
23344
+ total: typeof body.total === "number" && Number.isFinite(body.total) ? body.total : null,
23345
+ featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null,
23346
+ providerKey: parseProviderKey(body.providerKey)
23324
23347
  };
23325
23348
  } catch {
23326
23349
  return empty("unknown");
@@ -23363,6 +23386,11 @@ async function runLlm(opts = {}) {
23363
23386
  }
23364
23387
  const ws = await resolveWorkspace(cwd);
23365
23388
  const toolsOnly = ws.mode === "tools" && !ws.hosted;
23389
+ const projectBound = sub === "bench" || sub === "status" || sub === "cancel";
23390
+ if (sub === "cancel" && !opts.generationId?.trim()) {
23391
+ fail4("llm cancel", "`genex llm cancel` needs the id of the call to stop, e.g. genex llm cancel <id> \u2014 `genex llm status` lists the open ones.");
23392
+ return;
23393
+ }
23366
23394
  if (sub === "bench") {
23367
23395
  if (opts.maxCoins === void 0 || opts.userApproved !== true) {
23368
23396
  fail4("llm bench", LLM_BENCH_APPROVAL_REQUIRED);
@@ -23380,22 +23408,22 @@ async function runLlm(opts = {}) {
23380
23408
  fail4("llm bench", "Pass --json-output or --text, not both.");
23381
23409
  return;
23382
23410
  }
23383
- if (toolsOnly) {
23384
- if (opts.json) writeJsonLine({ command: "llm bench", status: "failed", error: "`genex llm bench` is not part of Genex Tools." });
23385
- else {
23386
- reportPlatformRefused(log, "llm bench");
23387
- log.plain(` ${c.cyan("\u2192")} A model that runs while people PLAY is a platform feature: the player pays and approves it.`);
23388
- log.plain(` Ask the user, and on their yes: ${c.cyan("npx genex init --convert")} \u2014 then the ${c.cyan("genex-tool-llm")} card.`);
23389
- }
23390
- process.exitCode = 1;
23391
- return;
23411
+ }
23412
+ if (projectBound && toolsOnly) {
23413
+ if (opts.json) writeJsonLine({ command: `llm ${sub}`, status: "failed", error: `\`genex llm ${sub}\` is not part of Genex Tools.` });
23414
+ else {
23415
+ reportPlatformRefused(log, `llm ${sub}`);
23416
+ log.plain(` ${c.cyan("\u2192")} A model that runs while people PLAY is a platform feature: the player pays and approves it.`);
23417
+ log.plain(` Ask the user, and on their yes: ${c.cyan("npx genex init --convert")} \u2014 then the ${c.cyan("genex-tool-llm")} card.`);
23392
23418
  }
23419
+ process.exitCode = 1;
23420
+ return;
23393
23421
  }
23394
23422
  const meta = await readProject(cwd);
23395
- if (sub === "bench" && !meta?.id) {
23423
+ if (projectBound && !meta?.id) {
23396
23424
  fail4(
23397
- "llm bench",
23398
- "This folder isn't linked to a game \u2014 a benchmark is billed to a project you own."
23425
+ `llm ${sub}`,
23426
+ sub === "bench" ? "This folder isn't linked to a game \u2014 a benchmark is billed to a project you own." : `This folder isn't linked to a game \u2014 \`genex llm ${sub}\` works on the development calls of a project you own.`
23399
23427
  );
23400
23428
  if (!opts.json) {
23401
23429
  log.dim(` Run ${c.cyan("genex link")} (or ${c.cyan("genex list")} to find the slug) first.`);
@@ -23432,6 +23460,14 @@ async function runLlm(opts = {}) {
23432
23460
  await reportModels(apiUrl, token, opts, log, toolsOnly);
23433
23461
  return;
23434
23462
  }
23463
+ if (sub === "status") {
23464
+ await reportStatus({ apiUrl, token, projectId: meta.id, opts, log });
23465
+ return;
23466
+ }
23467
+ if (sub === "cancel") {
23468
+ await cancelCall({ apiUrl, token, projectId: meta.id, id: opts.generationId.trim(), opts, log });
23469
+ return;
23470
+ }
23435
23471
  await runBench({ apiUrl, token, projectId: meta.id, cwd, opts, log });
23436
23472
  }
23437
23473
  async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
@@ -23457,6 +23493,12 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23457
23493
  models: null,
23458
23494
  externalProviders: null,
23459
23495
  recommendedDeclaredHeadroomBps: null,
23496
+ total: null,
23497
+ featuredCount: null,
23498
+ // The stand was never asked, so the verdict is unknown — but the key is
23499
+ // present, like every other lane key, so a reader tells "no verdict"
23500
+ // from "a CLI that predates the probe".
23501
+ providerKey: null,
23460
23502
  convertHint: toolsOnly ? LLM_TOOLS_CONVERT_HINT.join(" ") : null
23461
23503
  });
23462
23504
  }
@@ -23470,9 +23512,16 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23470
23512
  status: lane.state,
23471
23513
  error: null,
23472
23514
  enabled: lane.state === "live",
23515
+ // --json ALWAYS carries every row with its `featured` flag. The short
23516
+ // list is a rendering decision for a person reading a terminal; a program
23517
+ // reading this object gets the whole catalog and decides for itself.
23473
23518
  models: lane.models,
23474
23519
  externalProviders: lane.externalProviders,
23475
23520
  recommendedDeclaredHeadroomBps: lane.recommendedDeclaredHeadroomBps,
23521
+ total: lane.total ?? lane.models?.length ?? null,
23522
+ featuredCount: lane.featuredCount ?? (lane.models ? lane.models.filter((m) => m.featured).length : null),
23523
+ // `null` when the lane is not live or the server predates the probe.
23524
+ providerKey: lane.providerKey,
23476
23525
  // Always present, `null` off a tools workspace — the same
23477
23526
  // every-key-always-present rule the other keys follow, so a reader can
23478
23527
  // tell "no hint" from "a CLI that predates the hint".
@@ -23507,16 +23556,29 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23507
23556
  process.exitCode = 1;
23508
23557
  return;
23509
23558
  }
23510
- log.plain(` ${lane.models.length} model${lane.models.length === 1 ? "" : "s"} live:`);
23511
- for (const m of lane.models) {
23559
+ const featured = lane.models.filter((m) => m.featured);
23560
+ const shown = opts.all || featured.length === 0 ? lane.models : featured;
23561
+ const hiddenCount = lane.models.length - shown.length;
23562
+ log.plain(
23563
+ shown === lane.models ? ` ${lane.models.length} model${lane.models.length === 1 ? "" : "s"} live:` : ` ${shown.length} featured model${shown.length === 1 ? "" : "s"} of ${lane.models.length} live:`
23564
+ );
23565
+ for (const m of shown) {
23566
+ const plan = m.personalPlan ? ` \xB7 personal plan: ${m.personalPlan}` : "";
23512
23567
  log.plain(
23513
- ` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens`
23568
+ ` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens${plan}`
23514
23569
  );
23515
23570
  }
23571
+ if (hiddenCount > 0) {
23572
+ log.dim(` ${hiddenCount} more available \u2014 ${c.cyan("genex llm models --all")} to list them.`);
23573
+ }
23516
23574
  if (lane.externalProviders && lane.externalProviders.length > 0) {
23517
23575
  log.plain("");
23518
23576
  log.plain(` Personal plans offered here: ${lane.externalProviders.map((p) => p.label).join(", ")}`);
23519
23577
  }
23578
+ if (lane.providerKey === "rejected") {
23579
+ log.plain("");
23580
+ log.warn(` ${LLM_PROVIDER_KEY_REJECTED_LINE}`);
23581
+ }
23520
23582
  log.plain("");
23521
23583
  if (lane.recommendedDeclaredHeadroomBps === null) {
23522
23584
  log.dim(" This stand serves no recommended headroom, so no price can be recommended from a bench.");
@@ -23530,6 +23592,32 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23530
23592
  log.dim(" Those per-million rates are the PROVIDER's; what your game is charged is the platform");
23531
23593
  log.dim(` tariff on top, which only a real call reveals: ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')}`);
23532
23594
  }
23595
+ var ACTIVE_STATUSES = /* @__PURE__ */ new Set(["requires_confirmation", "queued", "dispatching", "awaiting_external"]);
23596
+ function rowState(row) {
23597
+ if (row.status && ACTIVE_STATUSES.has(row.status)) return "active";
23598
+ if (row.billingStatus === "pending") return "awaiting bill";
23599
+ return "done";
23600
+ }
23601
+ function rowHoldsSlot(row) {
23602
+ if (typeof row.slotHeld === "boolean") return row.slotHeld;
23603
+ const state = rowState(row);
23604
+ return state === "active" || state === "awaiting bill" && (row.reservedCoins ?? 0) > 0;
23605
+ }
23606
+ function providerRefusalStatus(error) {
23607
+ const m = /^provider_http_(\d{3})$/.exec(error ?? "");
23608
+ return m ? Number(m[1]) : null;
23609
+ }
23610
+ function formatAge(iso, now = Date.now()) {
23611
+ const t = iso ? Date.parse(iso) : Number.NaN;
23612
+ if (!Number.isFinite(t)) return "\u2014";
23613
+ const sec = Math.max(0, Math.floor((now - t) / 1e3));
23614
+ if (sec < 60) return `${sec}s`;
23615
+ const min = Math.floor(sec / 60);
23616
+ if (min < 60) return `${min}m`;
23617
+ const hr = Math.floor(min / 60);
23618
+ if (hr < 48) return `${hr}h`;
23619
+ return `${Math.floor(hr / 24)}d`;
23620
+ }
23533
23621
  async function runBench(args) {
23534
23622
  const { apiUrl, token, projectId, cwd, opts, log } = args;
23535
23623
  const maxCoins = opts.maxCoins;
@@ -23573,6 +23661,14 @@ async function runBench(args) {
23573
23661
  benchFailed(opts, log, `This stand does not serve ${modelId}. Run \`genex llm models\` for the ids it does.`);
23574
23662
  return;
23575
23663
  }
23664
+ if (lane.providerKey === "rejected") {
23665
+ benchFailed(
23666
+ opts,
23667
+ log,
23668
+ `${LLM_PROVIDER_KEY_REJECTED_LINE} A bench against a refused key is ${samples} refusal${samples === 1 ? "" : "s"} and a slot lock-out, not a measurement \u2014 nothing was sent, nothing was spent. Tell the operator; re-run once \`genex llm models\` stops saying so.`
23669
+ );
23670
+ return;
23671
+ }
23576
23672
  const balanceBefore = await readCoinBalance(apiUrl, token);
23577
23673
  if (!opts.json) {
23578
23674
  log.plain(c.bold("genex llm bench"));
@@ -23588,6 +23684,8 @@ async function runBench(args) {
23588
23684
  }
23589
23685
  const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
23590
23686
  const results = [];
23687
+ let refusal = null;
23688
+ let providerRefused = 0;
23591
23689
  let inflight = null;
23592
23690
  const onSigint = () => {
23593
23691
  const target = inflight;
@@ -23618,13 +23716,14 @@ async function runBench(args) {
23618
23716
  })
23619
23717
  });
23620
23718
  if (!res.ok) {
23621
- const stop2 = await explainBenchRefusal(res, log, opts, i, results.length);
23622
- if (stop2) break;
23719
+ const verdict = await explainBenchRefusal(res, call, base, log, opts, i, results.length);
23720
+ if (verdict.refusal) refusal = verdict.refusal;
23721
+ if (verdict.stop) break;
23623
23722
  continue;
23624
23723
  }
23625
23724
  const started = await res.json().catch(() => ({}));
23626
23725
  if (!started.id) {
23627
- results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id" });
23726
+ results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id", providerMessage: null });
23628
23727
  continue;
23629
23728
  }
23630
23729
  inflight = started.id;
@@ -23636,19 +23735,23 @@ async function runBench(args) {
23636
23735
  billingStatus: settled?.billingStatus ?? null,
23637
23736
  chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
23638
23737
  costUsd: typeof settled?.usage?.costUsd === "number" ? settled.usage.costUsd : null,
23639
- error: settled?.error ?? (settled ? null : "timed_out")
23738
+ error: settled?.error ?? (settled ? null : "timed_out"),
23739
+ providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null
23640
23740
  };
23641
23741
  results.push(row);
23742
+ const refusedAt = providerRefusalStatus(row.error);
23743
+ if (refusedAt !== null) providerRefused++;
23642
23744
  if (!opts.json) {
23643
23745
  log.plain(
23644
23746
  ` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${row.error ? ` \xB7 ${row.error}` : ""}`
23645
23747
  );
23748
+ if (refusedAt !== null) printProviderRefusal(log, refusedAt, row.providerMessage, row.status === "unknown");
23646
23749
  }
23647
23750
  }
23648
23751
  } finally {
23649
23752
  process.removeListener("SIGINT", onSigint);
23650
23753
  }
23651
- const charged = results.filter((r) => r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
23754
+ const charged = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
23652
23755
  const settledCount = charged.length;
23653
23756
  const p50 = percentile2(charged, 50);
23654
23757
  const p95 = percentile2(charged, 95);
@@ -23693,6 +23796,9 @@ async function runBench(args) {
23693
23796
  balanceBefore,
23694
23797
  balanceAfter,
23695
23798
  results,
23799
+ // The create-time refusal that ended the run, or null. Beside the
23800
+ // samples, never among them: no attempt existed and nothing was spent.
23801
+ refusal,
23696
23802
  chargedCoins: { p50, p95, max },
23697
23803
  recommendedDeclaredHeadroomBps: headroomBps,
23698
23804
  recommendedEstimateCoins: recommended,
@@ -23704,7 +23810,9 @@ async function runBench(args) {
23704
23810
  }
23705
23811
  log.plain("");
23706
23812
  if (charged.length === 0) {
23707
- log.error(" No sample settled, so there is nothing to price from.");
23813
+ log.error(
23814
+ providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to price from." : " No sample settled, so there is nothing to price from."
23815
+ );
23708
23816
  process.exitCode = 1;
23709
23817
  return;
23710
23818
  }
@@ -23738,31 +23846,53 @@ function printRecommendation(log, record) {
23738
23846
  log.dim(" The ceiling is the price's room to be wrong; declaring one under the price is refused.");
23739
23847
  }
23740
23848
  }
23741
- async function explainBenchRefusal(res, log, opts, index, settledSoFar) {
23742
- if (printedStructuredError(res)) return true;
23849
+ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoFar) {
23850
+ if (printedStructuredError(res)) return { stop: true, refusal: { status: res.status, error: null, slotsHeld: null, slotLimit: null } };
23743
23851
  const body = await res.json().catch(() => ({}));
23744
- if (opts.json) return res.status !== 429;
23852
+ const refusal = { status: res.status, error: body.error ?? null, slotsHeld: null, slotLimit: null };
23853
+ if (res.status === 429 && body.error === "generation_limit") {
23854
+ const counts = serverSlotCounts(body) ?? await readSlotCounts(call, base);
23855
+ refusal.slotsHeld = counts?.held ?? null;
23856
+ refusal.slotLimit = counts?.limit ?? null;
23857
+ if (!opts.json) log.error(` Sample ${index + 1} ${benchSlotsHeldSentence(counts)}`);
23858
+ return { stop: true, refusal };
23859
+ }
23860
+ if (opts.json) return { stop: res.status !== 429, refusal };
23745
23861
  if (res.status === 404 && body.error === "not_found") {
23746
23862
  log.error(` ${LLM_LANE_OFF_LINE}`);
23747
- return true;
23863
+ return { stop: true, refusal };
23748
23864
  }
23749
23865
  if (res.status === 403 && body.error === "credential_scope") {
23750
23866
  log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
23751
23867
  log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
23752
- return true;
23868
+ return { stop: true, refusal };
23753
23869
  }
23754
23870
  if (res.status === 403 && body.error === "project_owner_required") {
23755
23871
  log.error(" A benchmark is billed to a project YOU own, and this one isn't yours.");
23756
- return true;
23872
+ return { stop: true, refusal };
23757
23873
  }
23758
23874
  if (res.status === 402) {
23759
23875
  log.error(" Not enough coin to start the attempt.");
23760
- return true;
23876
+ return { stop: true, refusal };
23761
23877
  }
23762
23878
  log.error(
23763
23879
  ` Sample ${index + 1} was refused: ${body.message ?? body.error ?? `HTTP ${res.status}`}${settledSoFar > 0 ? " (earlier samples still count)" : ""}`
23764
23880
  );
23765
- return res.status >= 500 || res.status === 401 || res.status === 403;
23881
+ return { stop: res.status >= 500 || res.status === 401 || res.status === 403, refusal };
23882
+ }
23883
+ function printProviderRefusal(log, status2, message, awaitingBill = false) {
23884
+ if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so this call's coin stays held until it resolves (see \`genex llm status\`); it is not a sample.`);
23885
+ else log.dim(" The provider refused this call at its door \u2014 no inference ran, it cost nothing, and it is not a sample.");
23886
+ if (message) log.dim(` Provider said: ${message}`);
23887
+ if (status2 === 401 || status2 === 403) {
23888
+ log.dim(` A ${status2} here is this stand's provider configuration refusing the model \u2014 a message for the operator, not something to fix in the game.`);
23889
+ }
23890
+ }
23891
+ async function readSlotCounts(call, base) {
23892
+ const res = await call(`${base}?scope=open`).catch(() => null);
23893
+ if (!res?.ok) return null;
23894
+ const body = await res.json().catch(() => null);
23895
+ return serverSlotCounts(body);
23766
23896
  }
23767
23897
  async function pollSettled(call, url, timeoutSec) {
23768
23898
  const deadline = Date.now() + timeoutSec * 1e3;
@@ -23820,6 +23950,7 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
23820
23950
  balanceBefore: null,
23821
23951
  balanceAfter: null,
23822
23952
  results: [],
23953
+ refusal: null,
23823
23954
  chargedCoins: { p50: null, p95: null, max: null },
23824
23955
  recommendedDeclaredHeadroomBps: null,
23825
23956
  recommendedEstimateCoins: null,
@@ -23827,6 +23958,166 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
23827
23958
  savedTo: null
23828
23959
  };
23829
23960
  }
23961
+ async function reportStatus(args) {
23962
+ const { apiUrl, token, projectId, opts, log } = args;
23963
+ const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
23964
+ const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
23965
+ const res = await call(`${base}?scope=open`).catch(() => null);
23966
+ if (!res || !res.ok) {
23967
+ if (res && printedStructuredError(res)) {
23968
+ if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", `http_${res.status}`, [], null, null));
23969
+ process.exitCode = 1;
23970
+ return;
23971
+ }
23972
+ const body2 = res ? await res.json().catch(() => ({})) : {};
23973
+ const error = body2.error ?? (res ? `http_${res.status}` : "unreachable");
23974
+ if (opts.json) {
23975
+ writeJsonLine(statusJson(apiUrl, projectId, res?.status === 404 && body2.error === "not_found" ? "off" : "failed", error, [], null, null));
23976
+ if (!(res?.status === 404 && body2.error === "not_found")) process.exitCode = 1;
23977
+ return;
23978
+ }
23979
+ if (res?.status === 404 && body2.error === "not_found") {
23980
+ log.plain(LLM_LANE_OFF_LINE);
23981
+ return;
23982
+ }
23983
+ if (res?.status === 403 && body2.error === "credential_scope") {
23984
+ log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
23985
+ log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
23986
+ } else if (res?.status === 403 && body2.error === "project_owner_required") {
23987
+ log.error(" Development calls belong to a project YOU own, and this one isn't yours.");
23988
+ } else if (!res) {
23989
+ log.error(` Couldn't reach ${apiUrl}.`);
23990
+ log.dim(` Check the network, then ${c.cyan("genex doctor")}.`);
23991
+ } else {
23992
+ log.error(` Couldn't read the open calls: ${error}.`);
23993
+ }
23994
+ process.exitCode = 1;
23995
+ return;
23996
+ }
23997
+ const body = await res.json().catch(() => null);
23998
+ if (!body || !Array.isArray(body.generations)) {
23999
+ if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", "unreadable_response", [], null, null));
24000
+ else log.error(` ${apiUrl} answered, but not with a list of calls.`);
24001
+ process.exitCode = 1;
24002
+ return;
24003
+ }
24004
+ const rows = body.generations;
24005
+ const counts = serverSlotCounts(body);
24006
+ if (opts.json) {
24007
+ writeJsonLine(statusJson(apiUrl, projectId, "ok", null, rows, counts?.held ?? null, counts?.limit ?? null));
24008
+ return;
24009
+ }
24010
+ log.plain(c.bold("genex llm status"));
24011
+ log.dim(` ${apiUrl} \xB7 project ${projectId}`);
24012
+ log.plain("");
24013
+ if (rows.length === 0) {
24014
+ log.plain(" No open calls in this project \u2014 nothing active, nothing awaiting a bill.");
24015
+ } else {
24016
+ const now = Date.now();
24017
+ for (const row of rows) {
24018
+ const state = rowState(row);
24019
+ const mark = state === "active" ? c.cyan("\u25CF") : state === "awaiting bill" ? c.yellow("\u25D0") : c.green("\u25CB");
24020
+ const reserved = typeof row.reservedCoins === "number" ? `${row.reservedCoins} coin reserved` : "reserved \u2014";
24021
+ log.plain(
24022
+ ` ${mark} ${row.id ?? "?"} ${row.modelId ?? "?"} ${state.padEnd(13)} ${formatAge(row.createdAt, now).padStart(4)} old ${reserved}${row.error ? ` ${row.error}` : ""}${rowHoldsSlot(row) ? "" : c.dim(" (no slot)")}`
24023
+ );
24024
+ if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
24025
+ }
24026
+ }
24027
+ log.plain("");
24028
+ log.dim(` ${LLM_STATUS_LEGEND}`);
24029
+ if (counts === null) {
24030
+ log.dim(` ${LLM_STATUS_NO_COUNT_LINE}`);
24031
+ return;
24032
+ }
24033
+ log.plain(` ${counts.held} of ${counts.limit} slots held${counts.held > rows.filter(rowHoldsSlot).length ? " \u2014 across every project, this one's rows included" : ""}`);
24034
+ if (counts.held >= counts.limit) {
24035
+ log.dim(` Every slot is held, so a new ad-hoc call answers generation_limit until one frees: ${c.cyan("genex llm cancel <id>")} stops an active one; a stopped one frees on its own.`);
24036
+ }
24037
+ }
24038
+ function statusJson(apiUrl, projectId, status2, error, rows, slotsHeld, slotLimit) {
24039
+ return {
24040
+ command: "llm status",
24041
+ status: status2,
24042
+ error,
24043
+ apiUrl,
24044
+ projectId,
24045
+ // Every row verbatim, plus the CLI's own reading of it — so a program
24046
+ // gets both the server's fields and the three words a person sees.
24047
+ generations: rows.map((row) => ({ ...row, state: rowState(row), holdsSlot: rowHoldsSlot(row) })),
24048
+ slotsHeld,
24049
+ slotLimit,
24050
+ pendingSlotReleaseMinutes: PENDING_SLOT_RELEASE_MINUTES
24051
+ };
24052
+ }
24053
+ async function cancelCall(args) {
24054
+ const { apiUrl, token, projectId, id, opts, log } = args;
24055
+ const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
24056
+ const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
24057
+ const rowUrl = `${base}/${encodeURIComponent(id)}`;
24058
+ const done = (status2, error, reason, row2) => {
24059
+ if (opts.json) writeJsonLine({ command: "llm cancel", status: status2, error, id, cancelled: reason === "cancelled", reason, generation: row2 });
24060
+ if (status2 === "failed") process.exitCode = 1;
24061
+ };
24062
+ const read = await call(rowUrl).catch(() => null);
24063
+ if (!read || !read.ok) {
24064
+ if (read && printedStructuredError(read)) return done("failed", `http_${read.status}`, null, null);
24065
+ const body = read ? await read.json().catch(() => ({})) : {};
24066
+ const error = body.error ?? (read ? `http_${read.status}` : "unreachable");
24067
+ if (!opts.json) {
24068
+ if (read?.status === 404) {
24069
+ log.error(` No call ${id} in this project \u2014 ${c.cyan("genex llm status")} lists the open ones.`);
24070
+ } else if (read?.status === 403 && body.error === "credential_scope") {
24071
+ log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
24072
+ log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
24073
+ } else if (!read) {
24074
+ log.error(` Couldn't reach ${apiUrl}.`);
24075
+ } else {
24076
+ log.error(` Couldn't read call ${id}: ${error}.`);
24077
+ }
24078
+ }
24079
+ return done("failed", error, null, null);
24080
+ }
24081
+ const row = await read.json().catch(() => null);
24082
+ if (!row?.id) {
24083
+ if (!opts.json) log.error(` ${apiUrl} answered, but not with a call.`);
24084
+ return done("failed", "unreadable_response", null, null);
24085
+ }
24086
+ const state = rowState(row);
24087
+ if (state !== "active") {
24088
+ const sentence = state === "awaiting bill" ? LLM_CANCEL_AWAITING_BILL : LLM_CANCEL_ALREADY_FINAL;
24089
+ if (!opts.json) {
24090
+ log.warn(` ${row.id} (${row.status ?? "?"}) \u2014 ${sentence}.`);
24091
+ if (state === "awaiting bill") {
24092
+ log.dim(` Nothing to cancel: its coin releases when the bill resolves, and it stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`);
24093
+ if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
24094
+ } else {
24095
+ log.dim(` ${typeof row.chargedCoins === "number" ? `${row.chargedCoins} coin charged` : "charge unknown"}${row.error ? ` \xB7 ${row.error}` : ""}.`);
24096
+ }
24097
+ }
24098
+ return done("ok", null, state === "awaiting bill" ? "awaiting_bill" : "already_final", row);
24099
+ }
24100
+ const res = await call(`${rowUrl}/cancel`, { method: "POST" }).catch(() => null);
24101
+ if (!res || !res.ok) {
24102
+ if (res && printedStructuredError(res)) return done("failed", `http_${res.status}`, null, row);
24103
+ const body = res ? await res.json().catch(() => ({})) : {};
24104
+ const error = body.error ?? (res ? `http_${res.status}` : "unreachable");
24105
+ if (!opts.json) {
24106
+ if (res?.status === 409 && body.error === "already_dispatched") {
24107
+ log.error(` ${row.id} is already running at the provider and cannot be cancelled now; it settles on its own \u2014 ${c.cyan("genex llm status")} shows it.`);
24108
+ } else {
24109
+ log.error(` Couldn't cancel ${row.id}: ${error}.`);
24110
+ }
24111
+ }
24112
+ return done("failed", error, null, row);
24113
+ }
24114
+ const after = await res.json().catch(() => null) ?? row;
24115
+ if (!opts.json) {
24116
+ log.success(` Cancelled ${row.id} (was ${row.status ?? "active"}).`);
24117
+ log.dim(` Its coin releases once the bill settles; ${c.cyan("genex llm status")} shows it until then.`);
24118
+ }
24119
+ return done("ok", null, "cancelled", after);
24120
+ }
23830
24121
  async function reportSavedPrice(cwd, opts, log) {
23831
24122
  let record = null;
23832
24123
  try {
@@ -24139,10 +24430,11 @@ function runtimeRow(state, lane, toolsOnly = false) {
24139
24430
  switch (lane?.state) {
24140
24431
  case "live": {
24141
24432
  const n = lane.models?.length ?? 0;
24433
+ const keyRefused = lane.providerKey === "rejected";
24142
24434
  return {
24143
24435
  label: "Runtime",
24144
- value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}`,
24145
- fix: toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
24436
+ value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}${keyRefused ? ` \xB7 ${c.yellow("provider key refused")}` : ""}`,
24437
+ fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
24146
24438
  };
24147
24439
  }
24148
24440
  case "off":
@@ -26076,17 +26368,20 @@ ${c.bold("Usage")}
26076
26368
  the asset table against it. --assets <credits>
26077
26369
  --user-approved raises it, only once the player
26078
26370
  has agreed to the number.
26079
- genex llm <sub> [options] In-game model calls: models | bench | price.
26080
- "models" says whether this stand serves the
26081
- runtime LLM lane at all, and at what provider
26082
- rates. "bench" runs the real model on YOUR OWN
26083
- coin and reports what each attempt was charged;
26084
- it needs --max-coins <n> --user-approved, like
26085
- any spend the player has to agree to. "price"
26086
- reprints the last run's recommended
26087
- estimateCoins. Declare a game's price from a
26088
- bench, never from a guess: a started attempt is
26089
- charged in full, including one that fails.
26371
+ genex llm <sub> [options] In-game model calls: models | bench | price |
26372
+ status | cancel. "models" says whether this
26373
+ stand serves the runtime LLM lane at all, and
26374
+ at what provider rates. "bench" runs the real
26375
+ model on YOUR OWN coin and reports what each
26376
+ attempt was charged; it needs --max-coins <n>
26377
+ --user-approved, like any spend the player has
26378
+ to agree to. "price" reprints the last run's
26379
+ recommended estimateCoins. "status" lists this
26380
+ project's open calls and the slots they hold;
26381
+ "cancel <id>" stops an active one. Declare a
26382
+ game's price from a bench, never from a guess:
26383
+ a started attempt is charged in full, including
26384
+ one that fails.
26090
26385
  genex player <sub> [options] Answer the in-game model requests you approve, on your
26091
26386
  OWN Claude or ChatGPT subscription: install | run |
26092
26387
  status | stop | uninstall. "install" signs this machine
@@ -26520,8 +26815,11 @@ ${c.bold("Examples")}
26520
26815
  genex shop test sku_123
26521
26816
  genex shop remove sku_123
26522
26817
  genex llm models
26818
+ genex llm models --all
26523
26819
  genex llm bench "Reply with one short taunt." --samples 5 --max-coins 20 --user-approved
26524
26820
  genex llm price
26821
+ genex llm status
26822
+ genex llm cancel <id>
26525
26823
  genex player install
26526
26824
  genex player status --json
26527
26825
  genex player uninstall
@@ -26758,11 +27056,13 @@ ${c.bold("How the allowance works")}
26758
27056
  llm: `${c.bold("genex llm")} \u2014 in-game model calls: is the lane live, what does one call cost, what should the game declare?
26759
27057
 
26760
27058
  ${c.bold("Usage")}
26761
- genex llm models [--json] Which models this stand serves, their provider rates per
27059
+ genex llm models [--all] [--json] Which models this stand serves, their provider rates per
26762
27060
  million tokens, and the headroom it recommends over a
26763
- benchmark. Off on this stand: it says so and exits 0 \u2014
26764
- build the feature without a model rather than promising
26765
- a 404.
27061
+ benchmark. The catalog is synced from OpenRouter, so the
27062
+ FEATURED short list is printed by default and --all
27063
+ prints every row (--json always carries every row). Off
27064
+ on this stand: it says so and exits 0 \u2014 build the feature
27065
+ without a model rather than promising a 404.
26766
27066
  genex llm bench "<prompt>" --max-coins <n> --user-approved [options]
26767
27067
  Run the real model <n> times through the development
26768
27068
  lane and report what each attempt was CHARGED. Refused
@@ -26771,6 +27071,14 @@ ${c.bold("Usage")}
26771
27071
  cannot see. Needs a folder linked to a game you own.
26772
27072
  genex llm price [--json] The recommendation from the last bench in this folder.
26773
27073
  Reads a file; spends nothing.
27074
+ genex llm status [--json] This project's open development calls: active, or
27075
+ stopped but awaiting the provider's bill \u2014 with the
27076
+ provider's own message when it refused one \u2014 and
27077
+ how many of the ad-hoc slots they hold. A bench
27078
+ refused as generation_limit sends you here.
27079
+ genex llm cancel <id> [--json] Stop one ACTIVE call. A call that already stopped
27080
+ is reported as such: awaiting its bill (the hold
27081
+ resolves on its own) or already final.
26774
27082
 
26775
27083
  ${c.bold("Bench options")}
26776
27084
  --model <id> One of the ids \`genex llm models\` prints (default: the first one).
@@ -27280,7 +27588,11 @@ function parseArgs(argv) {
27280
27588
  (parsed.options.selectors ??= []).push(arg);
27281
27589
  }
27282
27590
  } else if (parsed.command === "llm") {
27283
- parsed.options.benchPrompt = parsed.options.benchPrompt ? `${parsed.options.benchPrompt} ${arg}` : arg;
27591
+ if (parsed.options.name === "cancel" && parsed.options.generationId === void 0) {
27592
+ parsed.options.generationId = arg;
27593
+ } else {
27594
+ parsed.options.benchPrompt = parsed.options.benchPrompt ? `${parsed.options.benchPrompt} ${arg}` : arg;
27595
+ }
27284
27596
  } else if (parsed.command === "shop") {
27285
27597
  if (!parsed.options.hostname) parsed.options.hostname = arg;
27286
27598
  else {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@genex-ai/cli-demo",
3
- "version": "1.35.0-dev.741",
3
+ "version": "1.35.0-dev.743",
4
4
  "description": "Set up your project's agent workspace (.claude/.codex/.cursor in the game folder), authorize, create a game project, generate AI assets, and publish (genex CLI).",
5
5
  "type": "module",
6
6
  "bin": {
@@ -31,24 +31,36 @@ or two per session means a grant.
31
31
  npx genex llm models
32
32
  ```
33
33
 
34
- - **live** — it prints the models this stand serves. Build the feature.
34
+ - **live** — it prints the FEATURED models this stand serves, and how many more
35
+ there are. Build the feature.
35
36
  - **off on this stand** — the routes answer 404. Build the feature behind a
36
37
  graceful `unavailable` state (the NPC uses its authored lines, the quest falls
37
38
  back to the written one) and **say so plainly in the handoff**. Never promise
38
39
  the player something that 404s.
39
40
  - **misconfigured** — say that too; it is an operator fix, not a game bug.
40
41
 
42
+ The stand's catalog is synced and filtered, so it can hold far more rows than
43
+ anyone wants to read. **Offer the featured set. Run `npx genex llm models --all`
44
+ only when the user asks for more**, and put the full list in front of them
45
+ rather than picking an obscure row for them.
46
+
41
47
  Model ids come from that command and from `getGenerationModels()` at runtime.
42
48
  Never write one into the game's source: they differ per stand, and a hardcoded
43
- id is a feature that dies on somebody else's environment.
49
+ id is a feature that dies on somebody else's environment. **A picker's label
50
+ must be the model's own `label`** — a name you invent for a row ("Fast", "Smart")
51
+ is a label that differs from the model the player is billed for.
44
52
 
45
53
  ## The SDK surface (exact — do not invent methods)
46
54
 
47
55
  From `@genex-ai/embed-sdk`, already installed. `initEmbed()` must have run and
48
56
  identity must be resolved first — `$genex-threejs-embed-auth`.
49
57
 
50
- - `getGenerationModels()` — the models this stand serves. Any picker renders
51
- from this, never from a list you wrote.
58
+ - `getGenerationModels()` — `{ models, featuredCount, … }`, featured first. Any
59
+ picker renders from this, never from a list you wrote: show the rows where
60
+ `featured` is true (or all of them when `featuredCount` is 0) and offer the
61
+ rest only if the player asks. Each row carries `label`, `vendor`,
62
+ `contextLength`, `personalPlan` and `structuredOutputs`; render `label`
63
+ verbatim, so the name on screen is the model that gets billed.
52
64
  - `generate({ modelId, prompt, outputFormat, schema?, estimateCoins,
53
65
  allowExternal?, idempotencyKey?, grantId?, timeoutMs? })` →
54
66
  `{ status, generationId, output, source, error }` plus billing fields
@@ -108,6 +120,41 @@ the price and makes `requestSpendGrant()` refuse before it reaches the network.
108
120
  loop into disclosure numbers, re-benchmarking after a prompt change — is in
109
121
  [references/pricing.md](references/pricing.md).
110
122
 
123
+ Only samples that **succeeded and settled** are priced from. A sample the
124
+ provider refused at its door (`provider_http_<status>`) ran no inference and
125
+ cost nothing; the bench prints the code, the provider's own message and, for a
126
+ 401 or 403, that this is the stand's provider configuration refusing the model
127
+ — an operator's problem, never something to fix in the game. A sample refused
128
+ as `generation_limit` never started: you already have the lane's three ad-hoc
129
+ calls open, or recently stopped with their bill still pending. The bench says
130
+ how many slots are held and stops. Do not re-run it into the same refusal —
131
+ read `npx genex llm status`, then wait for a bill to resolve or
132
+ `npx genex llm cancel <id>` an active call.
133
+
134
+ ```bash
135
+ npx genex llm status # this project's open calls: active, or awaiting their bill; N of 3 slots held
136
+ npx genex llm cancel <id> # stop an ACTIVE call; a stopped one is reported, not cancelled
137
+ ```
138
+
139
+ ## The schema dialect (exact — anything else is refused)
140
+
141
+ `schema` is validated by the platform before the call, and a refused schema is
142
+ `invalid_schema` at the door — nothing is charged, nothing runs. The accepted
143
+ subset, and it is the whole subset:
144
+
145
+ - `type`: `object`, `array`, `string`, `number`, `integer`, `boolean`, `null`
146
+ - objects: `properties`, `required`, and `additionalProperties: false` on
147
+ **every** object (required, not optional)
148
+ - arrays: `items`
149
+ - `enum` (strings, numbers, booleans, `null`)
150
+ - `minimum` / `maximum`, `minLength` / `maxLength`, `minItems` / `maxItems`
151
+ - `description` and `title`, on any node
152
+
153
+ Everything else is refused, including `$schema`, `default`, `examples`,
154
+ `pattern`, `format`, `anyOf` / `oneOf` / `allOf` and `$ref`. Keep the schema
155
+ in one file (`./answer.schema.json`), benchmark with that file, and ship the
156
+ same object — a schema that passed the bench passes the game.
157
+
111
158
  ## Standing budgets
112
159
 
113
160
  ```ts
@@ -183,7 +230,9 @@ There is no compensation lane, so this is all work you do before the call:
183
230
 
184
231
  - **Validate inputs first** — a malformed prompt is still charged.
185
232
  - **Always set `schema` for `outputFormat: 'json'`** — unschema'd JSON is the
186
- commonest way a call is charged and the result is unusable.
233
+ commonest way a call is charged and the result is unusable. Write it in the
234
+ accepted dialect above; a refused schema is `invalid_schema` and costs
235
+ nothing, but it is a feature that never runs.
187
236
  - **Keep prompts short.** Long context is the price.
188
237
  - **Never loop `generate()` without a grant**, and never retry in a loop — each
189
238
  attempt is a separate charge.
@@ -222,7 +271,10 @@ These are source contracts, not a claim that every stand runs this lane —
222
271
 
223
272
  - [ ] `npx genex llm models` was run and its verdict is in the handoff
224
273
  - [ ] Model ids come from `getGenerationModels()`, never from source
274
+ - [ ] The picker offers the featured set, labelled with the server's own `label`
225
275
  - [ ] `estimateCoins` came from `npx genex llm bench`, not from judgement
276
+ - [ ] The schema uses only the accepted dialect (no `$ref`, `pattern`, `format`, `anyOf`, `default`)
277
+ - [ ] A bench refused as `generation_limit` was answered with `npx genex llm status`, never a retry
226
278
  - [ ] `generate()` / `requestSpendGrant()` is the first statement of a click handler
227
279
  - [ ] A repeated-call feature uses a grant; a one-off uses `generate()`
228
280
  - [ ] Disclosure numbers derive from the real loop and are written in `DESIGN.md`
@@ -256,6 +308,25 @@ cause rather than showing it for both.
256
308
  **`grant_price_unreasonable`** — the declared per-call price is far above what
257
309
  that prompt can cost on that model. Re-benchmark and declare what it prints.
258
310
 
311
+ **`generation_limit`** — three ad-hoc calls are already in flight for this
312
+ account, or recently stopped with their bill still pending; a pending call
313
+ stops counting ten minutes after it was dispatched. From the bench: run
314
+ `npx genex llm status`, then wait or `npx genex llm cancel <id>` an active one —
315
+ never re-run into the same refusal. In the game: the player is clicking faster
316
+ than one-off calls are meant for, which is the signal that this feature wants a
317
+ grant.
318
+
319
+ **`provider_http_<status>`** — the provider refused the call at its door, before
320
+ any inference: it cost nothing and is not a sample. Read the provider's own
321
+ message (`npx genex llm status` prints it under the row). A 401 or 403 is this
322
+ stand's provider configuration refusing the model — tell the operator, and
323
+ build nothing around it in the game.
324
+
325
+ **`invalid_schema`** — the schema uses a keyword outside the accepted dialect
326
+ (`$ref`, `pattern`, `format`, `anyOf`, `default`, `examples`, `$schema`), or an
327
+ object without `additionalProperties: false`. Nothing was charged. Rewrite it in
328
+ the subset above; `description` and `title` are allowed.
329
+
259
330
  **`grant_concurrency` / `grant_rate_limited`** — the game calls faster than the
260
331
  grant's own limits. Batch and cache; do not raise the limits to hide it.
261
332
 
@@ -49,7 +49,19 @@ npx genex llm bench "<the frozen prompt, one real example filled in>" \
49
49
  ## 3. Read the output
50
50
 
51
51
  Each sample prints what it actually charged. The aggregate prints p50, p95 and
52
- max of the charged coins, plus one recommendation.
52
+ max of the charged coins **over the samples that succeeded and settled**, plus
53
+ one recommendation. A sample that failed is printed with its code and stays
54
+ out of the numbers; a sample the provider refused at its door
55
+ (`provider_http_<status>`) ran no inference, cost nothing, and prints the
56
+ provider's own message under its row — on a 401 or 403 that is the stand's
57
+ provider configuration refusing the model, which is the operator's to fix. A
58
+ sample refused as `generation_limit` never started and ends the run: the
59
+ account's three ad-hoc calls are open or recently stopped with a pending
60
+ bill. `npx genex llm status` lists them with what each holds;
61
+ `npx genex llm cancel <id>` stops an active one; a stopped one frees on its
62
+ own once its bill resolves, and stops holding a slot ten minutes after
63
+ dispatch. Re-running the bench into the same refusal spends nothing and
64
+ learns nothing.
53
65
 
54
66
  - **p50** is what a typical call costs. It is the number to reason about when
55
67
  you ask "can the game afford this loop?" — multiply it by the calls per
@@ -53,7 +53,11 @@ expand the offer into a pitch — one line, one question.
53
53
  which models. It answers in this folder as it is, before anything is
54
54
  converted, so it comes first: if it says the lane is off, in-game calls
55
55
  answer 404 here — tell the user so plainly, build the graceful fallback, and
56
- do not convert a folder for a feature the stand does not serve.
56
+ do not convert a folder for a feature the stand does not serve. It prints
57
+ the FEATURED models; `--all` lists the whole catalog, and you run that only
58
+ when the user asks for more. Whatever they pick, the picker in the game
59
+ shows the server's own label for it — a name you invent is a name that
60
+ differs from the model they are billed for.
57
61
  2. `npx genex init --convert` — it connects this folder to a hosted Genex game
58
62
  in place. The code, the files and this toolkit stay exactly as they are, and
59
63
  generations still land in `./assets`. It is the user's yes that runs it, so