@genex-ai/cli-demo 1.35.0-dev.742 → 1.35.0-dev.743

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -23288,7 +23288,7 @@ import path35 from "path";
23288
23288
  import fs33 from "fs/promises";
23289
23289
  import path34 from "path";
23290
23290
  import { randomUUID as randomUUID2 } from "crypto";
23291
- var SUBS3 = ["models", "bench", "price"];
23291
+ var SUBS3 = ["models", "bench", "price", "status", "cancel"];
23292
23292
  var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN coin \u2014 your wallet, not a player's, and one this build's asset allowance cannot see. Re-run with --max-coins <n> --user-approved only after the person at the keyboard agreed to the number.";
23293
23293
  var LLM_LANE_OFF_LINE = "The runtime LLM lane is off on this stand \u2014 in-game generate() answers 404 here, and nothing can be benchmarked.";
23294
23294
  var LLM_TOOLS_CONVERT_HINT = [
@@ -23297,8 +23297,25 @@ var LLM_TOOLS_CONVERT_HINT = [
23297
23297
  "Ask the user first; on their yes run `npx genex init --convert`, then load the genex-tool-llm card."
23298
23298
  ];
23299
23299
  var LLM_BENCH_FILE = ".genex/llm-bench.json";
23300
+ function serverSlotCounts(body) {
23301
+ if (!body || typeof body.slotsHeld !== "number" || typeof body.slotLimit !== "number" || body.slotLimit <= 0) return null;
23302
+ return { held: body.slotsHeld, limit: body.slotLimit };
23303
+ }
23304
+ var PENDING_SLOT_RELEASE_MINUTES = 10;
23305
+ var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so its coin stays held until the bill resolves; such a row stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`;
23306
+ var LLM_STATUS_NO_COUNT_LINE = "This stand does not report how many slots are held across your projects, so no count is shown.";
23307
+ var LLM_CANCEL_AWAITING_BILL = "this call already stopped; its bill is awaiting the provider";
23308
+ var LLM_CANCEL_ALREADY_FINAL = "this call is already final";
23309
+ function benchSlotsHeldSentence(counts) {
23310
+ const count = counts === null ? "every ad-hoc slot is" : `${counts.held} of ${counts.limit} ad-hoc slots are`;
23311
+ return `refused before it started, nothing spent: ${count} held on this stand \u2014 \`genex llm status\` shows the calls holding them; wait for a bill to resolve or cancel an active one, then re-run.`;
23312
+ }
23313
+ var LLM_PROVIDER_KEY_REJECTED_LINE = "This stand's provider key is refused \u2014 every in-game call answers provider_http_401 until an operator replaces it.";
23300
23314
  var DEFAULT_BENCH_SAMPLES = 3;
23301
23315
  var DEFAULT_SAMPLE_TIMEOUT_SEC = 180;
23316
+ function parseProviderKey(value) {
23317
+ return value === "ok" || value === "rejected" || value === "unknown" ? value : null;
23318
+ }
23302
23319
  async function readRuntimeLane(apiUrl, token) {
23303
23320
  const empty = (state) => ({
23304
23321
  state,
@@ -23306,7 +23323,8 @@ async function readRuntimeLane(apiUrl, token) {
23306
23323
  recommendedDeclaredHeadroomBps: null,
23307
23324
  externalProviders: null,
23308
23325
  total: null,
23309
- featuredCount: null
23326
+ featuredCount: null,
23327
+ providerKey: null
23310
23328
  });
23311
23329
  try {
23312
23330
  const res = await apiFetch(`${apiUrl}/api/runtime/models`, {
@@ -23324,7 +23342,8 @@ async function readRuntimeLane(apiUrl, token) {
23324
23342
  recommendedDeclaredHeadroomBps: typeof body.recommendedDeclaredHeadroomBps === "number" && Number.isFinite(body.recommendedDeclaredHeadroomBps) ? body.recommendedDeclaredHeadroomBps : null,
23325
23343
  externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null,
23326
23344
  total: typeof body.total === "number" && Number.isFinite(body.total) ? body.total : null,
23327
- featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null
23345
+ featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null,
23346
+ providerKey: parseProviderKey(body.providerKey)
23328
23347
  };
23329
23348
  } catch {
23330
23349
  return empty("unknown");
@@ -23367,6 +23386,11 @@ async function runLlm(opts = {}) {
23367
23386
  }
23368
23387
  const ws = await resolveWorkspace(cwd);
23369
23388
  const toolsOnly = ws.mode === "tools" && !ws.hosted;
23389
+ const projectBound = sub === "bench" || sub === "status" || sub === "cancel";
23390
+ if (sub === "cancel" && !opts.generationId?.trim()) {
23391
+ fail4("llm cancel", "`genex llm cancel` needs the id of the call to stop, e.g. genex llm cancel <id> \u2014 `genex llm status` lists the open ones.");
23392
+ return;
23393
+ }
23370
23394
  if (sub === "bench") {
23371
23395
  if (opts.maxCoins === void 0 || opts.userApproved !== true) {
23372
23396
  fail4("llm bench", LLM_BENCH_APPROVAL_REQUIRED);
@@ -23384,22 +23408,22 @@ async function runLlm(opts = {}) {
23384
23408
  fail4("llm bench", "Pass --json-output or --text, not both.");
23385
23409
  return;
23386
23410
  }
23387
- if (toolsOnly) {
23388
- if (opts.json) writeJsonLine({ command: "llm bench", status: "failed", error: "`genex llm bench` is not part of Genex Tools." });
23389
- else {
23390
- reportPlatformRefused(log, "llm bench");
23391
- log.plain(` ${c.cyan("\u2192")} A model that runs while people PLAY is a platform feature: the player pays and approves it.`);
23392
- log.plain(` Ask the user, and on their yes: ${c.cyan("npx genex init --convert")} \u2014 then the ${c.cyan("genex-tool-llm")} card.`);
23393
- }
23394
- process.exitCode = 1;
23395
- return;
23411
+ }
23412
+ if (projectBound && toolsOnly) {
23413
+ if (opts.json) writeJsonLine({ command: `llm ${sub}`, status: "failed", error: `\`genex llm ${sub}\` is not part of Genex Tools.` });
23414
+ else {
23415
+ reportPlatformRefused(log, `llm ${sub}`);
23416
+ log.plain(` ${c.cyan("\u2192")} A model that runs while people PLAY is a platform feature: the player pays and approves it.`);
23417
+ log.plain(` Ask the user, and on their yes: ${c.cyan("npx genex init --convert")} \u2014 then the ${c.cyan("genex-tool-llm")} card.`);
23396
23418
  }
23419
+ process.exitCode = 1;
23420
+ return;
23397
23421
  }
23398
23422
  const meta = await readProject(cwd);
23399
- if (sub === "bench" && !meta?.id) {
23423
+ if (projectBound && !meta?.id) {
23400
23424
  fail4(
23401
- "llm bench",
23402
- "This folder isn't linked to a game \u2014 a benchmark is billed to a project you own."
23425
+ `llm ${sub}`,
23426
+ sub === "bench" ? "This folder isn't linked to a game \u2014 a benchmark is billed to a project you own." : `This folder isn't linked to a game \u2014 \`genex llm ${sub}\` works on the development calls of a project you own.`
23403
23427
  );
23404
23428
  if (!opts.json) {
23405
23429
  log.dim(` Run ${c.cyan("genex link")} (or ${c.cyan("genex list")} to find the slug) first.`);
@@ -23436,6 +23460,14 @@ async function runLlm(opts = {}) {
23436
23460
  await reportModels(apiUrl, token, opts, log, toolsOnly);
23437
23461
  return;
23438
23462
  }
23463
+ if (sub === "status") {
23464
+ await reportStatus({ apiUrl, token, projectId: meta.id, opts, log });
23465
+ return;
23466
+ }
23467
+ if (sub === "cancel") {
23468
+ await cancelCall({ apiUrl, token, projectId: meta.id, id: opts.generationId.trim(), opts, log });
23469
+ return;
23470
+ }
23439
23471
  await runBench({ apiUrl, token, projectId: meta.id, cwd, opts, log });
23440
23472
  }
23441
23473
  async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
@@ -23463,6 +23495,10 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23463
23495
  recommendedDeclaredHeadroomBps: null,
23464
23496
  total: null,
23465
23497
  featuredCount: null,
23498
+ // The stand was never asked, so the verdict is unknown — but the key is
23499
+ // present, like every other lane key, so a reader tells "no verdict"
23500
+ // from "a CLI that predates the probe".
23501
+ providerKey: null,
23466
23502
  convertHint: toolsOnly ? LLM_TOOLS_CONVERT_HINT.join(" ") : null
23467
23503
  });
23468
23504
  }
@@ -23484,6 +23520,8 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23484
23520
  recommendedDeclaredHeadroomBps: lane.recommendedDeclaredHeadroomBps,
23485
23521
  total: lane.total ?? lane.models?.length ?? null,
23486
23522
  featuredCount: lane.featuredCount ?? (lane.models ? lane.models.filter((m) => m.featured).length : null),
23523
+ // `null` when the lane is not live or the server predates the probe.
23524
+ providerKey: lane.providerKey,
23487
23525
  // Always present, `null` off a tools workspace — the same
23488
23526
  // every-key-always-present rule the other keys follow, so a reader can
23489
23527
  // tell "no hint" from "a CLI that predates the hint".
@@ -23537,6 +23575,10 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23537
23575
  log.plain("");
23538
23576
  log.plain(` Personal plans offered here: ${lane.externalProviders.map((p) => p.label).join(", ")}`);
23539
23577
  }
23578
+ if (lane.providerKey === "rejected") {
23579
+ log.plain("");
23580
+ log.warn(` ${LLM_PROVIDER_KEY_REJECTED_LINE}`);
23581
+ }
23540
23582
  log.plain("");
23541
23583
  if (lane.recommendedDeclaredHeadroomBps === null) {
23542
23584
  log.dim(" This stand serves no recommended headroom, so no price can be recommended from a bench.");
@@ -23550,6 +23592,32 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
23550
23592
  log.dim(" Those per-million rates are the PROVIDER's; what your game is charged is the platform");
23551
23593
  log.dim(` tariff on top, which only a real call reveals: ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')}`);
23552
23594
  }
23595
+ var ACTIVE_STATUSES = /* @__PURE__ */ new Set(["requires_confirmation", "queued", "dispatching", "awaiting_external"]);
23596
+ function rowState(row) {
23597
+ if (row.status && ACTIVE_STATUSES.has(row.status)) return "active";
23598
+ if (row.billingStatus === "pending") return "awaiting bill";
23599
+ return "done";
23600
+ }
23601
+ function rowHoldsSlot(row) {
23602
+ if (typeof row.slotHeld === "boolean") return row.slotHeld;
23603
+ const state = rowState(row);
23604
+ return state === "active" || state === "awaiting bill" && (row.reservedCoins ?? 0) > 0;
23605
+ }
23606
+ function providerRefusalStatus(error) {
23607
+ const m = /^provider_http_(\d{3})$/.exec(error ?? "");
23608
+ return m ? Number(m[1]) : null;
23609
+ }
23610
+ function formatAge(iso, now = Date.now()) {
23611
+ const t = iso ? Date.parse(iso) : Number.NaN;
23612
+ if (!Number.isFinite(t)) return "\u2014";
23613
+ const sec = Math.max(0, Math.floor((now - t) / 1e3));
23614
+ if (sec < 60) return `${sec}s`;
23615
+ const min = Math.floor(sec / 60);
23616
+ if (min < 60) return `${min}m`;
23617
+ const hr = Math.floor(min / 60);
23618
+ if (hr < 48) return `${hr}h`;
23619
+ return `${Math.floor(hr / 24)}d`;
23620
+ }
23553
23621
  async function runBench(args) {
23554
23622
  const { apiUrl, token, projectId, cwd, opts, log } = args;
23555
23623
  const maxCoins = opts.maxCoins;
@@ -23593,6 +23661,14 @@ async function runBench(args) {
23593
23661
  benchFailed(opts, log, `This stand does not serve ${modelId}. Run \`genex llm models\` for the ids it does.`);
23594
23662
  return;
23595
23663
  }
23664
+ if (lane.providerKey === "rejected") {
23665
+ benchFailed(
23666
+ opts,
23667
+ log,
23668
+ `${LLM_PROVIDER_KEY_REJECTED_LINE} A bench against a refused key is ${samples} refusal${samples === 1 ? "" : "s"} and a slot lock-out, not a measurement \u2014 nothing was sent, nothing was spent. Tell the operator; re-run once \`genex llm models\` stops saying so.`
23669
+ );
23670
+ return;
23671
+ }
23596
23672
  const balanceBefore = await readCoinBalance(apiUrl, token);
23597
23673
  if (!opts.json) {
23598
23674
  log.plain(c.bold("genex llm bench"));
@@ -23608,6 +23684,8 @@ async function runBench(args) {
23608
23684
  }
23609
23685
  const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
23610
23686
  const results = [];
23687
+ let refusal = null;
23688
+ let providerRefused = 0;
23611
23689
  let inflight = null;
23612
23690
  const onSigint = () => {
23613
23691
  const target = inflight;
@@ -23638,13 +23716,14 @@ async function runBench(args) {
23638
23716
  })
23639
23717
  });
23640
23718
  if (!res.ok) {
23641
- const stop2 = await explainBenchRefusal(res, log, opts, i, results.length);
23642
- if (stop2) break;
23719
+ const verdict = await explainBenchRefusal(res, call, base, log, opts, i, results.length);
23720
+ if (verdict.refusal) refusal = verdict.refusal;
23721
+ if (verdict.stop) break;
23643
23722
  continue;
23644
23723
  }
23645
23724
  const started = await res.json().catch(() => ({}));
23646
23725
  if (!started.id) {
23647
- results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id" });
23726
+ results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id", providerMessage: null });
23648
23727
  continue;
23649
23728
  }
23650
23729
  inflight = started.id;
@@ -23656,19 +23735,23 @@ async function runBench(args) {
23656
23735
  billingStatus: settled?.billingStatus ?? null,
23657
23736
  chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
23658
23737
  costUsd: typeof settled?.usage?.costUsd === "number" ? settled.usage.costUsd : null,
23659
- error: settled?.error ?? (settled ? null : "timed_out")
23738
+ error: settled?.error ?? (settled ? null : "timed_out"),
23739
+ providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null
23660
23740
  };
23661
23741
  results.push(row);
23742
+ const refusedAt = providerRefusalStatus(row.error);
23743
+ if (refusedAt !== null) providerRefused++;
23662
23744
  if (!opts.json) {
23663
23745
  log.plain(
23664
23746
  ` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${row.error ? ` \xB7 ${row.error}` : ""}`
23665
23747
  );
23748
+ if (refusedAt !== null) printProviderRefusal(log, refusedAt, row.providerMessage, row.status === "unknown");
23666
23749
  }
23667
23750
  }
23668
23751
  } finally {
23669
23752
  process.removeListener("SIGINT", onSigint);
23670
23753
  }
23671
- const charged = results.filter((r) => r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
23754
+ const charged = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
23672
23755
  const settledCount = charged.length;
23673
23756
  const p50 = percentile2(charged, 50);
23674
23757
  const p95 = percentile2(charged, 95);
@@ -23713,6 +23796,9 @@ async function runBench(args) {
23713
23796
  balanceBefore,
23714
23797
  balanceAfter,
23715
23798
  results,
23799
+ // The create-time refusal that ended the run, or null. Beside the
23800
+ // samples, never among them: no attempt existed and nothing was spent.
23801
+ refusal,
23716
23802
  chargedCoins: { p50, p95, max },
23717
23803
  recommendedDeclaredHeadroomBps: headroomBps,
23718
23804
  recommendedEstimateCoins: recommended,
@@ -23724,7 +23810,9 @@ async function runBench(args) {
23724
23810
  }
23725
23811
  log.plain("");
23726
23812
  if (charged.length === 0) {
23727
- log.error(" No sample settled, so there is nothing to price from.");
23813
+ log.error(
23814
+ providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to price from." : " No sample settled, so there is nothing to price from."
23815
+ );
23728
23816
  process.exitCode = 1;
23729
23817
  return;
23730
23818
  }
@@ -23758,31 +23846,53 @@ function printRecommendation(log, record) {
23758
23846
  log.dim(" The ceiling is the price's room to be wrong; declaring one under the price is refused.");
23759
23847
  }
23760
23848
  }
23761
- async function explainBenchRefusal(res, log, opts, index, settledSoFar) {
23762
- if (printedStructuredError(res)) return true;
23849
+ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoFar) {
23850
+ if (printedStructuredError(res)) return { stop: true, refusal: { status: res.status, error: null, slotsHeld: null, slotLimit: null } };
23763
23851
  const body = await res.json().catch(() => ({}));
23764
- if (opts.json) return res.status !== 429;
23852
+ const refusal = { status: res.status, error: body.error ?? null, slotsHeld: null, slotLimit: null };
23853
+ if (res.status === 429 && body.error === "generation_limit") {
23854
+ const counts = serverSlotCounts(body) ?? await readSlotCounts(call, base);
23855
+ refusal.slotsHeld = counts?.held ?? null;
23856
+ refusal.slotLimit = counts?.limit ?? null;
23857
+ if (!opts.json) log.error(` Sample ${index + 1} ${benchSlotsHeldSentence(counts)}`);
23858
+ return { stop: true, refusal };
23859
+ }
23860
+ if (opts.json) return { stop: res.status !== 429, refusal };
23765
23861
  if (res.status === 404 && body.error === "not_found") {
23766
23862
  log.error(` ${LLM_LANE_OFF_LINE}`);
23767
- return true;
23863
+ return { stop: true, refusal };
23768
23864
  }
23769
23865
  if (res.status === 403 && body.error === "credential_scope") {
23770
23866
  log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
23771
23867
  log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
23772
- return true;
23868
+ return { stop: true, refusal };
23773
23869
  }
23774
23870
  if (res.status === 403 && body.error === "project_owner_required") {
23775
23871
  log.error(" A benchmark is billed to a project YOU own, and this one isn't yours.");
23776
- return true;
23872
+ return { stop: true, refusal };
23777
23873
  }
23778
23874
  if (res.status === 402) {
23779
23875
  log.error(" Not enough coin to start the attempt.");
23780
- return true;
23876
+ return { stop: true, refusal };
23781
23877
  }
23782
23878
  log.error(
23783
23879
  ` Sample ${index + 1} was refused: ${body.message ?? body.error ?? `HTTP ${res.status}`}${settledSoFar > 0 ? " (earlier samples still count)" : ""}`
23784
23880
  );
23785
- return res.status >= 500 || res.status === 401 || res.status === 403;
23881
+ return { stop: res.status >= 500 || res.status === 401 || res.status === 403, refusal };
23882
+ }
23883
+ function printProviderRefusal(log, status2, message, awaitingBill = false) {
23884
+ if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so this call's coin stays held until it resolves (see \`genex llm status\`); it is not a sample.`);
23885
+ else log.dim(" The provider refused this call at its door \u2014 no inference ran, it cost nothing, and it is not a sample.");
23886
+ if (message) log.dim(` Provider said: ${message}`);
23887
+ if (status2 === 401 || status2 === 403) {
23888
+ log.dim(` A ${status2} here is this stand's provider configuration refusing the model \u2014 a message for the operator, not something to fix in the game.`);
23889
+ }
23890
+ }
23891
+ async function readSlotCounts(call, base) {
23892
+ const res = await call(`${base}?scope=open`).catch(() => null);
23893
+ if (!res?.ok) return null;
23894
+ const body = await res.json().catch(() => null);
23895
+ return serverSlotCounts(body);
23786
23896
  }
23787
23897
  async function pollSettled(call, url, timeoutSec) {
23788
23898
  const deadline = Date.now() + timeoutSec * 1e3;
@@ -23840,6 +23950,7 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
23840
23950
  balanceBefore: null,
23841
23951
  balanceAfter: null,
23842
23952
  results: [],
23953
+ refusal: null,
23843
23954
  chargedCoins: { p50: null, p95: null, max: null },
23844
23955
  recommendedDeclaredHeadroomBps: null,
23845
23956
  recommendedEstimateCoins: null,
@@ -23847,6 +23958,166 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
23847
23958
  savedTo: null
23848
23959
  };
23849
23960
  }
23961
+ async function reportStatus(args) {
23962
+ const { apiUrl, token, projectId, opts, log } = args;
23963
+ const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
23964
+ const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
23965
+ const res = await call(`${base}?scope=open`).catch(() => null);
23966
+ if (!res || !res.ok) {
23967
+ if (res && printedStructuredError(res)) {
23968
+ if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", `http_${res.status}`, [], null, null));
23969
+ process.exitCode = 1;
23970
+ return;
23971
+ }
23972
+ const body2 = res ? await res.json().catch(() => ({})) : {};
23973
+ const error = body2.error ?? (res ? `http_${res.status}` : "unreachable");
23974
+ if (opts.json) {
23975
+ writeJsonLine(statusJson(apiUrl, projectId, res?.status === 404 && body2.error === "not_found" ? "off" : "failed", error, [], null, null));
23976
+ if (!(res?.status === 404 && body2.error === "not_found")) process.exitCode = 1;
23977
+ return;
23978
+ }
23979
+ if (res?.status === 404 && body2.error === "not_found") {
23980
+ log.plain(LLM_LANE_OFF_LINE);
23981
+ return;
23982
+ }
23983
+ if (res?.status === 403 && body2.error === "credential_scope") {
23984
+ log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
23985
+ log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
23986
+ } else if (res?.status === 403 && body2.error === "project_owner_required") {
23987
+ log.error(" Development calls belong to a project YOU own, and this one isn't yours.");
23988
+ } else if (!res) {
23989
+ log.error(` Couldn't reach ${apiUrl}.`);
23990
+ log.dim(` Check the network, then ${c.cyan("genex doctor")}.`);
23991
+ } else {
23992
+ log.error(` Couldn't read the open calls: ${error}.`);
23993
+ }
23994
+ process.exitCode = 1;
23995
+ return;
23996
+ }
23997
+ const body = await res.json().catch(() => null);
23998
+ if (!body || !Array.isArray(body.generations)) {
23999
+ if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", "unreadable_response", [], null, null));
24000
+ else log.error(` ${apiUrl} answered, but not with a list of calls.`);
24001
+ process.exitCode = 1;
24002
+ return;
24003
+ }
24004
+ const rows = body.generations;
24005
+ const counts = serverSlotCounts(body);
24006
+ if (opts.json) {
24007
+ writeJsonLine(statusJson(apiUrl, projectId, "ok", null, rows, counts?.held ?? null, counts?.limit ?? null));
24008
+ return;
24009
+ }
24010
+ log.plain(c.bold("genex llm status"));
24011
+ log.dim(` ${apiUrl} \xB7 project ${projectId}`);
24012
+ log.plain("");
24013
+ if (rows.length === 0) {
24014
+ log.plain(" No open calls in this project \u2014 nothing active, nothing awaiting a bill.");
24015
+ } else {
24016
+ const now = Date.now();
24017
+ for (const row of rows) {
24018
+ const state = rowState(row);
24019
+ const mark = state === "active" ? c.cyan("\u25CF") : state === "awaiting bill" ? c.yellow("\u25D0") : c.green("\u25CB");
24020
+ const reserved = typeof row.reservedCoins === "number" ? `${row.reservedCoins} coin reserved` : "reserved \u2014";
24021
+ log.plain(
24022
+ ` ${mark} ${row.id ?? "?"} ${row.modelId ?? "?"} ${state.padEnd(13)} ${formatAge(row.createdAt, now).padStart(4)} old ${reserved}${row.error ? ` ${row.error}` : ""}${rowHoldsSlot(row) ? "" : c.dim(" (no slot)")}`
24023
+ );
24024
+ if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
24025
+ }
24026
+ }
24027
+ log.plain("");
24028
+ log.dim(` ${LLM_STATUS_LEGEND}`);
24029
+ if (counts === null) {
24030
+ log.dim(` ${LLM_STATUS_NO_COUNT_LINE}`);
24031
+ return;
24032
+ }
24033
+ log.plain(` ${counts.held} of ${counts.limit} slots held${counts.held > rows.filter(rowHoldsSlot).length ? " \u2014 across every project, this one's rows included" : ""}`);
24034
+ if (counts.held >= counts.limit) {
24035
+ log.dim(` Every slot is held, so a new ad-hoc call answers generation_limit until one frees: ${c.cyan("genex llm cancel <id>")} stops an active one; a stopped one frees on its own.`);
24036
+ }
24037
+ }
24038
+ function statusJson(apiUrl, projectId, status2, error, rows, slotsHeld, slotLimit) {
24039
+ return {
24040
+ command: "llm status",
24041
+ status: status2,
24042
+ error,
24043
+ apiUrl,
24044
+ projectId,
24045
+ // Every row verbatim, plus the CLI's own reading of it — so a program
24046
+ // gets both the server's fields and the three words a person sees.
24047
+ generations: rows.map((row) => ({ ...row, state: rowState(row), holdsSlot: rowHoldsSlot(row) })),
24048
+ slotsHeld,
24049
+ slotLimit,
24050
+ pendingSlotReleaseMinutes: PENDING_SLOT_RELEASE_MINUTES
24051
+ };
24052
+ }
24053
+ async function cancelCall(args) {
24054
+ const { apiUrl, token, projectId, id, opts, log } = args;
24055
+ const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
24056
+ const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
24057
+ const rowUrl = `${base}/${encodeURIComponent(id)}`;
24058
+ const done = (status2, error, reason, row2) => {
24059
+ if (opts.json) writeJsonLine({ command: "llm cancel", status: status2, error, id, cancelled: reason === "cancelled", reason, generation: row2 });
24060
+ if (status2 === "failed") process.exitCode = 1;
24061
+ };
24062
+ const read = await call(rowUrl).catch(() => null);
24063
+ if (!read || !read.ok) {
24064
+ if (read && printedStructuredError(read)) return done("failed", `http_${read.status}`, null, null);
24065
+ const body = read ? await read.json().catch(() => ({})) : {};
24066
+ const error = body.error ?? (read ? `http_${read.status}` : "unreachable");
24067
+ if (!opts.json) {
24068
+ if (read?.status === 404) {
24069
+ log.error(` No call ${id} in this project \u2014 ${c.cyan("genex llm status")} lists the open ones.`);
24070
+ } else if (read?.status === 403 && body.error === "credential_scope") {
24071
+ log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
24072
+ log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
24073
+ } else if (!read) {
24074
+ log.error(` Couldn't reach ${apiUrl}.`);
24075
+ } else {
24076
+ log.error(` Couldn't read call ${id}: ${error}.`);
24077
+ }
24078
+ }
24079
+ return done("failed", error, null, null);
24080
+ }
24081
+ const row = await read.json().catch(() => null);
24082
+ if (!row?.id) {
24083
+ if (!opts.json) log.error(` ${apiUrl} answered, but not with a call.`);
24084
+ return done("failed", "unreadable_response", null, null);
24085
+ }
24086
+ const state = rowState(row);
24087
+ if (state !== "active") {
24088
+ const sentence = state === "awaiting bill" ? LLM_CANCEL_AWAITING_BILL : LLM_CANCEL_ALREADY_FINAL;
24089
+ if (!opts.json) {
24090
+ log.warn(` ${row.id} (${row.status ?? "?"}) \u2014 ${sentence}.`);
24091
+ if (state === "awaiting bill") {
24092
+ log.dim(` Nothing to cancel: its coin releases when the bill resolves, and it stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`);
24093
+ if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
24094
+ } else {
24095
+ log.dim(` ${typeof row.chargedCoins === "number" ? `${row.chargedCoins} coin charged` : "charge unknown"}${row.error ? ` \xB7 ${row.error}` : ""}.`);
24096
+ }
24097
+ }
24098
+ return done("ok", null, state === "awaiting bill" ? "awaiting_bill" : "already_final", row);
24099
+ }
24100
+ const res = await call(`${rowUrl}/cancel`, { method: "POST" }).catch(() => null);
24101
+ if (!res || !res.ok) {
24102
+ if (res && printedStructuredError(res)) return done("failed", `http_${res.status}`, null, row);
24103
+ const body = res ? await res.json().catch(() => ({})) : {};
24104
+ const error = body.error ?? (res ? `http_${res.status}` : "unreachable");
24105
+ if (!opts.json) {
24106
+ if (res?.status === 409 && body.error === "already_dispatched") {
24107
+ log.error(` ${row.id} is already running at the provider and cannot be cancelled now; it settles on its own \u2014 ${c.cyan("genex llm status")} shows it.`);
24108
+ } else {
24109
+ log.error(` Couldn't cancel ${row.id}: ${error}.`);
24110
+ }
24111
+ }
24112
+ return done("failed", error, null, row);
24113
+ }
24114
+ const after = await res.json().catch(() => null) ?? row;
24115
+ if (!opts.json) {
24116
+ log.success(` Cancelled ${row.id} (was ${row.status ?? "active"}).`);
24117
+ log.dim(` Its coin releases once the bill settles; ${c.cyan("genex llm status")} shows it until then.`);
24118
+ }
24119
+ return done("ok", null, "cancelled", after);
24120
+ }
23850
24121
  async function reportSavedPrice(cwd, opts, log) {
23851
24122
  let record = null;
23852
24123
  try {
@@ -24159,10 +24430,11 @@ function runtimeRow(state, lane, toolsOnly = false) {
24159
24430
  switch (lane?.state) {
24160
24431
  case "live": {
24161
24432
  const n = lane.models?.length ?? 0;
24433
+ const keyRefused = lane.providerKey === "rejected";
24162
24434
  return {
24163
24435
  label: "Runtime",
24164
- value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}`,
24165
- fix: toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
24436
+ value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}${keyRefused ? ` \xB7 ${c.yellow("provider key refused")}` : ""}`,
24437
+ fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
24166
24438
  };
24167
24439
  }
24168
24440
  case "off":
@@ -26096,17 +26368,20 @@ ${c.bold("Usage")}
26096
26368
  the asset table against it. --assets <credits>
26097
26369
  --user-approved raises it, only once the player
26098
26370
  has agreed to the number.
26099
- genex llm <sub> [options] In-game model calls: models | bench | price.
26100
- "models" says whether this stand serves the
26101
- runtime LLM lane at all, and at what provider
26102
- rates. "bench" runs the real model on YOUR OWN
26103
- coin and reports what each attempt was charged;
26104
- it needs --max-coins <n> --user-approved, like
26105
- any spend the player has to agree to. "price"
26106
- reprints the last run's recommended
26107
- estimateCoins. Declare a game's price from a
26108
- bench, never from a guess: a started attempt is
26109
- charged in full, including one that fails.
26371
+ genex llm <sub> [options] In-game model calls: models | bench | price |
26372
+ status | cancel. "models" says whether this
26373
+ stand serves the runtime LLM lane at all, and
26374
+ at what provider rates. "bench" runs the real
26375
+ model on YOUR OWN coin and reports what each
26376
+ attempt was charged; it needs --max-coins <n>
26377
+ --user-approved, like any spend the player has
26378
+ to agree to. "price" reprints the last run's
26379
+ recommended estimateCoins. "status" lists this
26380
+ project's open calls and the slots they hold;
26381
+ "cancel <id>" stops an active one. Declare a
26382
+ game's price from a bench, never from a guess:
26383
+ a started attempt is charged in full, including
26384
+ one that fails.
26110
26385
  genex player <sub> [options] Answer the in-game model requests you approve, on your
26111
26386
  OWN Claude or ChatGPT subscription: install | run |
26112
26387
  status | stop | uninstall. "install" signs this machine
@@ -26543,6 +26818,8 @@ ${c.bold("Examples")}
26543
26818
  genex llm models --all
26544
26819
  genex llm bench "Reply with one short taunt." --samples 5 --max-coins 20 --user-approved
26545
26820
  genex llm price
26821
+ genex llm status
26822
+ genex llm cancel <id>
26546
26823
  genex player install
26547
26824
  genex player status --json
26548
26825
  genex player uninstall
@@ -26794,6 +27071,14 @@ ${c.bold("Usage")}
26794
27071
  cannot see. Needs a folder linked to a game you own.
26795
27072
  genex llm price [--json] The recommendation from the last bench in this folder.
26796
27073
  Reads a file; spends nothing.
27074
+ genex llm status [--json] This project's open development calls: active, or
27075
+ stopped but awaiting the provider's bill \u2014 with the
27076
+ provider's own message when it refused one \u2014 and
27077
+ how many of the ad-hoc slots they hold. A bench
27078
+ refused as generation_limit sends you here.
27079
+ genex llm cancel <id> [--json] Stop one ACTIVE call. A call that already stopped
27080
+ is reported as such: awaiting its bill (the hold
27081
+ resolves on its own) or already final.
26797
27082
 
26798
27083
  ${c.bold("Bench options")}
26799
27084
  --model <id> One of the ids \`genex llm models\` prints (default: the first one).
@@ -27303,7 +27588,11 @@ function parseArgs(argv) {
27303
27588
  (parsed.options.selectors ??= []).push(arg);
27304
27589
  }
27305
27590
  } else if (parsed.command === "llm") {
27306
- parsed.options.benchPrompt = parsed.options.benchPrompt ? `${parsed.options.benchPrompt} ${arg}` : arg;
27591
+ if (parsed.options.name === "cancel" && parsed.options.generationId === void 0) {
27592
+ parsed.options.generationId = arg;
27593
+ } else {
27594
+ parsed.options.benchPrompt = parsed.options.benchPrompt ? `${parsed.options.benchPrompt} ${arg}` : arg;
27595
+ }
27307
27596
  } else if (parsed.command === "shop") {
27308
27597
  if (!parsed.options.hostname) parsed.options.hostname = arg;
27309
27598
  else {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@genex-ai/cli-demo",
3
- "version": "1.35.0-dev.742",
3
+ "version": "1.35.0-dev.743",
4
4
  "description": "Set up your project's agent workspace (.claude/.codex/.cursor in the game folder), authorize, create a game project, generate AI assets, and publish (genex CLI).",
5
5
  "type": "module",
6
6
  "bin": {
@@ -120,6 +120,41 @@ the price and makes `requestSpendGrant()` refuse before it reaches the network.
120
120
  loop into disclosure numbers, re-benchmarking after a prompt change — is in
121
121
  [references/pricing.md](references/pricing.md).
122
122
 
123
+ Only samples that **succeeded and settled** are priced from. A sample the
124
+ provider refused at its door (`provider_http_<status>`) ran no inference and
125
+ cost nothing; the bench prints the code, the provider's own message and, for a
126
+ 401 or 403, that this is the stand's provider configuration refusing the model
127
+ — an operator's problem, never something to fix in the game. A sample refused
128
+ as `generation_limit` never started: you already have the lane's three ad-hoc
129
+ calls open, or recently stopped with their bill still pending. The bench says
130
+ how many slots are held and stops. Do not re-run it into the same refusal —
131
+ read `npx genex llm status`, then wait for a bill to resolve or
132
+ `npx genex llm cancel <id>` an active call.
133
+
134
+ ```bash
135
+ npx genex llm status # this project's open calls: active, or awaiting their bill; N of 3 slots held
136
+ npx genex llm cancel <id> # stop an ACTIVE call; a stopped one is reported, not cancelled
137
+ ```
138
+
139
+ ## The schema dialect (exact — anything else is refused)
140
+
141
+ `schema` is validated by the platform before the call, and a refused schema is
142
+ `invalid_schema` at the door — nothing is charged, nothing runs. The accepted
143
+ subset, and it is the whole subset:
144
+
145
+ - `type`: `object`, `array`, `string`, `number`, `integer`, `boolean`, `null`
146
+ - objects: `properties`, `required`, and `additionalProperties: false` on
147
+ **every** object (required, not optional)
148
+ - arrays: `items`
149
+ - `enum` (strings, numbers, booleans, `null`)
150
+ - `minimum` / `maximum`, `minLength` / `maxLength`, `minItems` / `maxItems`
151
+ - `description` and `title`, on any node
152
+
153
+ Everything else is refused, including `$schema`, `default`, `examples`,
154
+ `pattern`, `format`, `anyOf` / `oneOf` / `allOf` and `$ref`. Keep the schema
155
+ in one file (`./answer.schema.json`), benchmark with that file, and ship the
156
+ same object — a schema that passed the bench passes the game.
157
+
123
158
  ## Standing budgets
124
159
 
125
160
  ```ts
@@ -195,7 +230,9 @@ There is no compensation lane, so this is all work you do before the call:
195
230
 
196
231
  - **Validate inputs first** — a malformed prompt is still charged.
197
232
  - **Always set `schema` for `outputFormat: 'json'`** — unschema'd JSON is the
198
- commonest way a call is charged and the result is unusable.
233
+ commonest way a call is charged and the result is unusable. Write it in the
234
+ accepted dialect above; a refused schema is `invalid_schema` and costs
235
+ nothing, but it is a feature that never runs.
199
236
  - **Keep prompts short.** Long context is the price.
200
237
  - **Never loop `generate()` without a grant**, and never retry in a loop — each
201
238
  attempt is a separate charge.
@@ -236,6 +273,8 @@ These are source contracts, not a claim that every stand runs this lane —
236
273
  - [ ] Model ids come from `getGenerationModels()`, never from source
237
274
  - [ ] The picker offers the featured set, labelled with the server's own `label`
238
275
  - [ ] `estimateCoins` came from `npx genex llm bench`, not from judgement
276
+ - [ ] The schema uses only the accepted dialect (no `$ref`, `pattern`, `format`, `anyOf`, `default`)
277
+ - [ ] A bench refused as `generation_limit` was answered with `npx genex llm status`, never a retry
239
278
  - [ ] `generate()` / `requestSpendGrant()` is the first statement of a click handler
240
279
  - [ ] A repeated-call feature uses a grant; a one-off uses `generate()`
241
280
  - [ ] Disclosure numbers derive from the real loop and are written in `DESIGN.md`
@@ -269,6 +308,25 @@ cause rather than showing it for both.
269
308
  **`grant_price_unreasonable`** — the declared per-call price is far above what
270
309
  that prompt can cost on that model. Re-benchmark and declare what it prints.
271
310
 
311
+ **`generation_limit`** — three ad-hoc calls are already in flight for this
312
+ account, or recently stopped with their bill still pending; a pending call
313
+ stops counting ten minutes after it was dispatched. From the bench: run
314
+ `npx genex llm status`, then wait or `npx genex llm cancel <id>` an active one —
315
+ never re-run into the same refusal. In the game: the player is clicking faster
316
+ than one-off calls are meant for, which is the signal that this feature wants a
317
+ grant.
318
+
319
+ **`provider_http_<status>`** — the provider refused the call at its door, before
320
+ any inference: it cost nothing and is not a sample. Read the provider's own
321
+ message (`npx genex llm status` prints it under the row). A 401 or 403 is this
322
+ stand's provider configuration refusing the model — tell the operator, and
323
+ build nothing around it in the game.
324
+
325
+ **`invalid_schema`** — the schema uses a keyword outside the accepted dialect
326
+ (`$ref`, `pattern`, `format`, `anyOf`, `default`, `examples`, `$schema`), or an
327
+ object without `additionalProperties: false`. Nothing was charged. Rewrite it in
328
+ the subset above; `description` and `title` are allowed.
329
+
272
330
  **`grant_concurrency` / `grant_rate_limited`** — the game calls faster than the
273
331
  grant's own limits. Batch and cache; do not raise the limits to hide it.
274
332
 
@@ -49,7 +49,19 @@ npx genex llm bench "<the frozen prompt, one real example filled in>" \
49
49
  ## 3. Read the output
50
50
 
51
51
  Each sample prints what it actually charged. The aggregate prints p50, p95 and
52
- max of the charged coins, plus one recommendation.
52
+ max of the charged coins **over the samples that succeeded and settled**, plus
53
+ one recommendation. A sample that failed is printed with its code and stays
54
+ out of the numbers; a sample the provider refused at its door
55
+ (`provider_http_<status>`) ran no inference, cost nothing, and prints the
56
+ provider's own message under its row — on a 401 or 403 that is the stand's
57
+ provider configuration refusing the model, which is the operator's to fix. A
58
+ sample refused as `generation_limit` never started and ends the run: the
59
+ account's three ad-hoc calls are open or recently stopped with a pending
60
+ bill. `npx genex llm status` lists them with what each holds;
61
+ `npx genex llm cancel <id>` stops an active one; a stopped one frees on its
62
+ own once its bill resolves, and stops holding a slot ten minutes after
63
+ dispatch. Re-running the bench into the same refusal spends nothing and
64
+ learns nothing.
53
65
 
54
66
  - **p50** is what a typical call costs. It is the number to reason about when
55
67
  you ask "can the game afford this loop?" — multiply it by the calls per