@genex-ai/cli-demo 1.35.0-dev.741 → 1.35.0-dev.743
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js
CHANGED
|
@@ -23288,7 +23288,7 @@ import path35 from "path";
|
|
|
23288
23288
|
import fs33 from "fs/promises";
|
|
23289
23289
|
import path34 from "path";
|
|
23290
23290
|
import { randomUUID as randomUUID2 } from "crypto";
|
|
23291
|
-
var SUBS3 = ["models", "bench", "price"];
|
|
23291
|
+
var SUBS3 = ["models", "bench", "price", "status", "cancel"];
|
|
23292
23292
|
var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN coin \u2014 your wallet, not a player's, and one this build's asset allowance cannot see. Re-run with --max-coins <n> --user-approved only after the person at the keyboard agreed to the number.";
|
|
23293
23293
|
var LLM_LANE_OFF_LINE = "The runtime LLM lane is off on this stand \u2014 in-game generate() answers 404 here, and nothing can be benchmarked.";
|
|
23294
23294
|
var LLM_TOOLS_CONVERT_HINT = [
|
|
@@ -23297,14 +23297,34 @@ var LLM_TOOLS_CONVERT_HINT = [
|
|
|
23297
23297
|
"Ask the user first; on their yes run `npx genex init --convert`, then load the genex-tool-llm card."
|
|
23298
23298
|
];
|
|
23299
23299
|
var LLM_BENCH_FILE = ".genex/llm-bench.json";
|
|
23300
|
+
function serverSlotCounts(body) {
|
|
23301
|
+
if (!body || typeof body.slotsHeld !== "number" || typeof body.slotLimit !== "number" || body.slotLimit <= 0) return null;
|
|
23302
|
+
return { held: body.slotsHeld, limit: body.slotLimit };
|
|
23303
|
+
}
|
|
23304
|
+
var PENDING_SLOT_RELEASE_MINUTES = 10;
|
|
23305
|
+
var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so its coin stays held until the bill resolves; such a row stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`;
|
|
23306
|
+
var LLM_STATUS_NO_COUNT_LINE = "This stand does not report how many slots are held across your projects, so no count is shown.";
|
|
23307
|
+
var LLM_CANCEL_AWAITING_BILL = "this call already stopped; its bill is awaiting the provider";
|
|
23308
|
+
var LLM_CANCEL_ALREADY_FINAL = "this call is already final";
|
|
23309
|
+
function benchSlotsHeldSentence(counts) {
|
|
23310
|
+
const count = counts === null ? "every ad-hoc slot is" : `${counts.held} of ${counts.limit} ad-hoc slots are`;
|
|
23311
|
+
return `refused before it started, nothing spent: ${count} held on this stand \u2014 \`genex llm status\` shows the calls holding them; wait for a bill to resolve or cancel an active one, then re-run.`;
|
|
23312
|
+
}
|
|
23313
|
+
var LLM_PROVIDER_KEY_REJECTED_LINE = "This stand's provider key is refused \u2014 every in-game call answers provider_http_401 until an operator replaces it.";
|
|
23300
23314
|
var DEFAULT_BENCH_SAMPLES = 3;
|
|
23301
23315
|
var DEFAULT_SAMPLE_TIMEOUT_SEC = 180;
|
|
23316
|
+
function parseProviderKey(value) {
|
|
23317
|
+
return value === "ok" || value === "rejected" || value === "unknown" ? value : null;
|
|
23318
|
+
}
|
|
23302
23319
|
async function readRuntimeLane(apiUrl, token) {
|
|
23303
23320
|
const empty = (state) => ({
|
|
23304
23321
|
state,
|
|
23305
23322
|
models: null,
|
|
23306
23323
|
recommendedDeclaredHeadroomBps: null,
|
|
23307
|
-
externalProviders: null
|
|
23324
|
+
externalProviders: null,
|
|
23325
|
+
total: null,
|
|
23326
|
+
featuredCount: null,
|
|
23327
|
+
providerKey: null
|
|
23308
23328
|
});
|
|
23309
23329
|
try {
|
|
23310
23330
|
const res = await apiFetch(`${apiUrl}/api/runtime/models`, {
|
|
@@ -23320,7 +23340,10 @@ async function readRuntimeLane(apiUrl, token) {
|
|
|
23320
23340
|
state: "live",
|
|
23321
23341
|
models: body.models,
|
|
23322
23342
|
recommendedDeclaredHeadroomBps: typeof body.recommendedDeclaredHeadroomBps === "number" && Number.isFinite(body.recommendedDeclaredHeadroomBps) ? body.recommendedDeclaredHeadroomBps : null,
|
|
23323
|
-
externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null
|
|
23343
|
+
externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null,
|
|
23344
|
+
total: typeof body.total === "number" && Number.isFinite(body.total) ? body.total : null,
|
|
23345
|
+
featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null,
|
|
23346
|
+
providerKey: parseProviderKey(body.providerKey)
|
|
23324
23347
|
};
|
|
23325
23348
|
} catch {
|
|
23326
23349
|
return empty("unknown");
|
|
@@ -23363,6 +23386,11 @@ async function runLlm(opts = {}) {
|
|
|
23363
23386
|
}
|
|
23364
23387
|
const ws = await resolveWorkspace(cwd);
|
|
23365
23388
|
const toolsOnly = ws.mode === "tools" && !ws.hosted;
|
|
23389
|
+
const projectBound = sub === "bench" || sub === "status" || sub === "cancel";
|
|
23390
|
+
if (sub === "cancel" && !opts.generationId?.trim()) {
|
|
23391
|
+
fail4("llm cancel", "`genex llm cancel` needs the id of the call to stop, e.g. genex llm cancel <id> \u2014 `genex llm status` lists the open ones.");
|
|
23392
|
+
return;
|
|
23393
|
+
}
|
|
23366
23394
|
if (sub === "bench") {
|
|
23367
23395
|
if (opts.maxCoins === void 0 || opts.userApproved !== true) {
|
|
23368
23396
|
fail4("llm bench", LLM_BENCH_APPROVAL_REQUIRED);
|
|
@@ -23380,22 +23408,22 @@ async function runLlm(opts = {}) {
|
|
|
23380
23408
|
fail4("llm bench", "Pass --json-output or --text, not both.");
|
|
23381
23409
|
return;
|
|
23382
23410
|
}
|
|
23383
|
-
|
|
23384
|
-
|
|
23385
|
-
|
|
23386
|
-
|
|
23387
|
-
|
|
23388
|
-
|
|
23389
|
-
}
|
|
23390
|
-
process.exitCode = 1;
|
|
23391
|
-
return;
|
|
23411
|
+
}
|
|
23412
|
+
if (projectBound && toolsOnly) {
|
|
23413
|
+
if (opts.json) writeJsonLine({ command: `llm ${sub}`, status: "failed", error: `\`genex llm ${sub}\` is not part of Genex Tools.` });
|
|
23414
|
+
else {
|
|
23415
|
+
reportPlatformRefused(log, `llm ${sub}`);
|
|
23416
|
+
log.plain(` ${c.cyan("\u2192")} A model that runs while people PLAY is a platform feature: the player pays and approves it.`);
|
|
23417
|
+
log.plain(` Ask the user, and on their yes: ${c.cyan("npx genex init --convert")} \u2014 then the ${c.cyan("genex-tool-llm")} card.`);
|
|
23392
23418
|
}
|
|
23419
|
+
process.exitCode = 1;
|
|
23420
|
+
return;
|
|
23393
23421
|
}
|
|
23394
23422
|
const meta = await readProject(cwd);
|
|
23395
|
-
if (
|
|
23423
|
+
if (projectBound && !meta?.id) {
|
|
23396
23424
|
fail4(
|
|
23397
|
-
|
|
23398
|
-
"This folder isn't linked to a game \u2014 a benchmark is billed to a project you own."
|
|
23425
|
+
`llm ${sub}`,
|
|
23426
|
+
sub === "bench" ? "This folder isn't linked to a game \u2014 a benchmark is billed to a project you own." : `This folder isn't linked to a game \u2014 \`genex llm ${sub}\` works on the development calls of a project you own.`
|
|
23399
23427
|
);
|
|
23400
23428
|
if (!opts.json) {
|
|
23401
23429
|
log.dim(` Run ${c.cyan("genex link")} (or ${c.cyan("genex list")} to find the slug) first.`);
|
|
@@ -23432,6 +23460,14 @@ async function runLlm(opts = {}) {
|
|
|
23432
23460
|
await reportModels(apiUrl, token, opts, log, toolsOnly);
|
|
23433
23461
|
return;
|
|
23434
23462
|
}
|
|
23463
|
+
if (sub === "status") {
|
|
23464
|
+
await reportStatus({ apiUrl, token, projectId: meta.id, opts, log });
|
|
23465
|
+
return;
|
|
23466
|
+
}
|
|
23467
|
+
if (sub === "cancel") {
|
|
23468
|
+
await cancelCall({ apiUrl, token, projectId: meta.id, id: opts.generationId.trim(), opts, log });
|
|
23469
|
+
return;
|
|
23470
|
+
}
|
|
23435
23471
|
await runBench({ apiUrl, token, projectId: meta.id, cwd, opts, log });
|
|
23436
23472
|
}
|
|
23437
23473
|
async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
@@ -23457,6 +23493,12 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23457
23493
|
models: null,
|
|
23458
23494
|
externalProviders: null,
|
|
23459
23495
|
recommendedDeclaredHeadroomBps: null,
|
|
23496
|
+
total: null,
|
|
23497
|
+
featuredCount: null,
|
|
23498
|
+
// The stand was never asked, so the verdict is unknown — but the key is
|
|
23499
|
+
// present, like every other lane key, so a reader tells "no verdict"
|
|
23500
|
+
// from "a CLI that predates the probe".
|
|
23501
|
+
providerKey: null,
|
|
23460
23502
|
convertHint: toolsOnly ? LLM_TOOLS_CONVERT_HINT.join(" ") : null
|
|
23461
23503
|
});
|
|
23462
23504
|
}
|
|
@@ -23470,9 +23512,16 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23470
23512
|
status: lane.state,
|
|
23471
23513
|
error: null,
|
|
23472
23514
|
enabled: lane.state === "live",
|
|
23515
|
+
// --json ALWAYS carries every row with its `featured` flag. The short
|
|
23516
|
+
// list is a rendering decision for a person reading a terminal; a program
|
|
23517
|
+
// reading this object gets the whole catalog and decides for itself.
|
|
23473
23518
|
models: lane.models,
|
|
23474
23519
|
externalProviders: lane.externalProviders,
|
|
23475
23520
|
recommendedDeclaredHeadroomBps: lane.recommendedDeclaredHeadroomBps,
|
|
23521
|
+
total: lane.total ?? lane.models?.length ?? null,
|
|
23522
|
+
featuredCount: lane.featuredCount ?? (lane.models ? lane.models.filter((m) => m.featured).length : null),
|
|
23523
|
+
// `null` when the lane is not live or the server predates the probe.
|
|
23524
|
+
providerKey: lane.providerKey,
|
|
23476
23525
|
// Always present, `null` off a tools workspace — the same
|
|
23477
23526
|
// every-key-always-present rule the other keys follow, so a reader can
|
|
23478
23527
|
// tell "no hint" from "a CLI that predates the hint".
|
|
@@ -23507,16 +23556,29 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23507
23556
|
process.exitCode = 1;
|
|
23508
23557
|
return;
|
|
23509
23558
|
}
|
|
23510
|
-
|
|
23511
|
-
|
|
23559
|
+
const featured = lane.models.filter((m) => m.featured);
|
|
23560
|
+
const shown = opts.all || featured.length === 0 ? lane.models : featured;
|
|
23561
|
+
const hiddenCount = lane.models.length - shown.length;
|
|
23562
|
+
log.plain(
|
|
23563
|
+
shown === lane.models ? ` ${lane.models.length} model${lane.models.length === 1 ? "" : "s"} live:` : ` ${shown.length} featured model${shown.length === 1 ? "" : "s"} of ${lane.models.length} live:`
|
|
23564
|
+
);
|
|
23565
|
+
for (const m of shown) {
|
|
23566
|
+
const plan = m.personalPlan ? ` \xB7 personal plan: ${m.personalPlan}` : "";
|
|
23512
23567
|
log.plain(
|
|
23513
|
-
` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens`
|
|
23568
|
+
` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens${plan}`
|
|
23514
23569
|
);
|
|
23515
23570
|
}
|
|
23571
|
+
if (hiddenCount > 0) {
|
|
23572
|
+
log.dim(` ${hiddenCount} more available \u2014 ${c.cyan("genex llm models --all")} to list them.`);
|
|
23573
|
+
}
|
|
23516
23574
|
if (lane.externalProviders && lane.externalProviders.length > 0) {
|
|
23517
23575
|
log.plain("");
|
|
23518
23576
|
log.plain(` Personal plans offered here: ${lane.externalProviders.map((p) => p.label).join(", ")}`);
|
|
23519
23577
|
}
|
|
23578
|
+
if (lane.providerKey === "rejected") {
|
|
23579
|
+
log.plain("");
|
|
23580
|
+
log.warn(` ${LLM_PROVIDER_KEY_REJECTED_LINE}`);
|
|
23581
|
+
}
|
|
23520
23582
|
log.plain("");
|
|
23521
23583
|
if (lane.recommendedDeclaredHeadroomBps === null) {
|
|
23522
23584
|
log.dim(" This stand serves no recommended headroom, so no price can be recommended from a bench.");
|
|
@@ -23530,6 +23592,32 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23530
23592
|
log.dim(" Those per-million rates are the PROVIDER's; what your game is charged is the platform");
|
|
23531
23593
|
log.dim(` tariff on top, which only a real call reveals: ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')}`);
|
|
23532
23594
|
}
|
|
23595
|
+
var ACTIVE_STATUSES = /* @__PURE__ */ new Set(["requires_confirmation", "queued", "dispatching", "awaiting_external"]);
|
|
23596
|
+
function rowState(row) {
|
|
23597
|
+
if (row.status && ACTIVE_STATUSES.has(row.status)) return "active";
|
|
23598
|
+
if (row.billingStatus === "pending") return "awaiting bill";
|
|
23599
|
+
return "done";
|
|
23600
|
+
}
|
|
23601
|
+
function rowHoldsSlot(row) {
|
|
23602
|
+
if (typeof row.slotHeld === "boolean") return row.slotHeld;
|
|
23603
|
+
const state = rowState(row);
|
|
23604
|
+
return state === "active" || state === "awaiting bill" && (row.reservedCoins ?? 0) > 0;
|
|
23605
|
+
}
|
|
23606
|
+
function providerRefusalStatus(error) {
|
|
23607
|
+
const m = /^provider_http_(\d{3})$/.exec(error ?? "");
|
|
23608
|
+
return m ? Number(m[1]) : null;
|
|
23609
|
+
}
|
|
23610
|
+
function formatAge(iso, now = Date.now()) {
|
|
23611
|
+
const t = iso ? Date.parse(iso) : Number.NaN;
|
|
23612
|
+
if (!Number.isFinite(t)) return "\u2014";
|
|
23613
|
+
const sec = Math.max(0, Math.floor((now - t) / 1e3));
|
|
23614
|
+
if (sec < 60) return `${sec}s`;
|
|
23615
|
+
const min = Math.floor(sec / 60);
|
|
23616
|
+
if (min < 60) return `${min}m`;
|
|
23617
|
+
const hr = Math.floor(min / 60);
|
|
23618
|
+
if (hr < 48) return `${hr}h`;
|
|
23619
|
+
return `${Math.floor(hr / 24)}d`;
|
|
23620
|
+
}
|
|
23533
23621
|
async function runBench(args) {
|
|
23534
23622
|
const { apiUrl, token, projectId, cwd, opts, log } = args;
|
|
23535
23623
|
const maxCoins = opts.maxCoins;
|
|
@@ -23573,6 +23661,14 @@ async function runBench(args) {
|
|
|
23573
23661
|
benchFailed(opts, log, `This stand does not serve ${modelId}. Run \`genex llm models\` for the ids it does.`);
|
|
23574
23662
|
return;
|
|
23575
23663
|
}
|
|
23664
|
+
if (lane.providerKey === "rejected") {
|
|
23665
|
+
benchFailed(
|
|
23666
|
+
opts,
|
|
23667
|
+
log,
|
|
23668
|
+
`${LLM_PROVIDER_KEY_REJECTED_LINE} A bench against a refused key is ${samples} refusal${samples === 1 ? "" : "s"} and a slot lock-out, not a measurement \u2014 nothing was sent, nothing was spent. Tell the operator; re-run once \`genex llm models\` stops saying so.`
|
|
23669
|
+
);
|
|
23670
|
+
return;
|
|
23671
|
+
}
|
|
23576
23672
|
const balanceBefore = await readCoinBalance(apiUrl, token);
|
|
23577
23673
|
if (!opts.json) {
|
|
23578
23674
|
log.plain(c.bold("genex llm bench"));
|
|
@@ -23588,6 +23684,8 @@ async function runBench(args) {
|
|
|
23588
23684
|
}
|
|
23589
23685
|
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
23590
23686
|
const results = [];
|
|
23687
|
+
let refusal = null;
|
|
23688
|
+
let providerRefused = 0;
|
|
23591
23689
|
let inflight = null;
|
|
23592
23690
|
const onSigint = () => {
|
|
23593
23691
|
const target = inflight;
|
|
@@ -23618,13 +23716,14 @@ async function runBench(args) {
|
|
|
23618
23716
|
})
|
|
23619
23717
|
});
|
|
23620
23718
|
if (!res.ok) {
|
|
23621
|
-
const
|
|
23622
|
-
if (
|
|
23719
|
+
const verdict = await explainBenchRefusal(res, call, base, log, opts, i, results.length);
|
|
23720
|
+
if (verdict.refusal) refusal = verdict.refusal;
|
|
23721
|
+
if (verdict.stop) break;
|
|
23623
23722
|
continue;
|
|
23624
23723
|
}
|
|
23625
23724
|
const started = await res.json().catch(() => ({}));
|
|
23626
23725
|
if (!started.id) {
|
|
23627
|
-
results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id" });
|
|
23726
|
+
results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id", providerMessage: null });
|
|
23628
23727
|
continue;
|
|
23629
23728
|
}
|
|
23630
23729
|
inflight = started.id;
|
|
@@ -23636,19 +23735,23 @@ async function runBench(args) {
|
|
|
23636
23735
|
billingStatus: settled?.billingStatus ?? null,
|
|
23637
23736
|
chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
|
|
23638
23737
|
costUsd: typeof settled?.usage?.costUsd === "number" ? settled.usage.costUsd : null,
|
|
23639
|
-
error: settled?.error ?? (settled ? null : "timed_out")
|
|
23738
|
+
error: settled?.error ?? (settled ? null : "timed_out"),
|
|
23739
|
+
providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null
|
|
23640
23740
|
};
|
|
23641
23741
|
results.push(row);
|
|
23742
|
+
const refusedAt = providerRefusalStatus(row.error);
|
|
23743
|
+
if (refusedAt !== null) providerRefused++;
|
|
23642
23744
|
if (!opts.json) {
|
|
23643
23745
|
log.plain(
|
|
23644
23746
|
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${row.error ? ` \xB7 ${row.error}` : ""}`
|
|
23645
23747
|
);
|
|
23748
|
+
if (refusedAt !== null) printProviderRefusal(log, refusedAt, row.providerMessage, row.status === "unknown");
|
|
23646
23749
|
}
|
|
23647
23750
|
}
|
|
23648
23751
|
} finally {
|
|
23649
23752
|
process.removeListener("SIGINT", onSigint);
|
|
23650
23753
|
}
|
|
23651
|
-
const charged = results.filter((r) => r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
|
|
23754
|
+
const charged = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
|
|
23652
23755
|
const settledCount = charged.length;
|
|
23653
23756
|
const p50 = percentile2(charged, 50);
|
|
23654
23757
|
const p95 = percentile2(charged, 95);
|
|
@@ -23693,6 +23796,9 @@ async function runBench(args) {
|
|
|
23693
23796
|
balanceBefore,
|
|
23694
23797
|
balanceAfter,
|
|
23695
23798
|
results,
|
|
23799
|
+
// The create-time refusal that ended the run, or null. Beside the
|
|
23800
|
+
// samples, never among them: no attempt existed and nothing was spent.
|
|
23801
|
+
refusal,
|
|
23696
23802
|
chargedCoins: { p50, p95, max },
|
|
23697
23803
|
recommendedDeclaredHeadroomBps: headroomBps,
|
|
23698
23804
|
recommendedEstimateCoins: recommended,
|
|
@@ -23704,7 +23810,9 @@ async function runBench(args) {
|
|
|
23704
23810
|
}
|
|
23705
23811
|
log.plain("");
|
|
23706
23812
|
if (charged.length === 0) {
|
|
23707
|
-
log.error(
|
|
23813
|
+
log.error(
|
|
23814
|
+
providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to price from." : " No sample settled, so there is nothing to price from."
|
|
23815
|
+
);
|
|
23708
23816
|
process.exitCode = 1;
|
|
23709
23817
|
return;
|
|
23710
23818
|
}
|
|
@@ -23738,31 +23846,53 @@ function printRecommendation(log, record) {
|
|
|
23738
23846
|
log.dim(" The ceiling is the price's room to be wrong; declaring one under the price is refused.");
|
|
23739
23847
|
}
|
|
23740
23848
|
}
|
|
23741
|
-
async function explainBenchRefusal(res, log, opts, index, settledSoFar) {
|
|
23742
|
-
if (printedStructuredError(res)) return true;
|
|
23849
|
+
async function explainBenchRefusal(res, call, base, log, opts, index, settledSoFar) {
|
|
23850
|
+
if (printedStructuredError(res)) return { stop: true, refusal: { status: res.status, error: null, slotsHeld: null, slotLimit: null } };
|
|
23743
23851
|
const body = await res.json().catch(() => ({}));
|
|
23744
|
-
|
|
23852
|
+
const refusal = { status: res.status, error: body.error ?? null, slotsHeld: null, slotLimit: null };
|
|
23853
|
+
if (res.status === 429 && body.error === "generation_limit") {
|
|
23854
|
+
const counts = serverSlotCounts(body) ?? await readSlotCounts(call, base);
|
|
23855
|
+
refusal.slotsHeld = counts?.held ?? null;
|
|
23856
|
+
refusal.slotLimit = counts?.limit ?? null;
|
|
23857
|
+
if (!opts.json) log.error(` Sample ${index + 1} ${benchSlotsHeldSentence(counts)}`);
|
|
23858
|
+
return { stop: true, refusal };
|
|
23859
|
+
}
|
|
23860
|
+
if (opts.json) return { stop: res.status !== 429, refusal };
|
|
23745
23861
|
if (res.status === 404 && body.error === "not_found") {
|
|
23746
23862
|
log.error(` ${LLM_LANE_OFF_LINE}`);
|
|
23747
|
-
return true;
|
|
23863
|
+
return { stop: true, refusal };
|
|
23748
23864
|
}
|
|
23749
23865
|
if (res.status === 403 && body.error === "credential_scope") {
|
|
23750
23866
|
log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
|
|
23751
23867
|
log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
|
|
23752
|
-
return true;
|
|
23868
|
+
return { stop: true, refusal };
|
|
23753
23869
|
}
|
|
23754
23870
|
if (res.status === 403 && body.error === "project_owner_required") {
|
|
23755
23871
|
log.error(" A benchmark is billed to a project YOU own, and this one isn't yours.");
|
|
23756
|
-
return true;
|
|
23872
|
+
return { stop: true, refusal };
|
|
23757
23873
|
}
|
|
23758
23874
|
if (res.status === 402) {
|
|
23759
23875
|
log.error(" Not enough coin to start the attempt.");
|
|
23760
|
-
return true;
|
|
23876
|
+
return { stop: true, refusal };
|
|
23761
23877
|
}
|
|
23762
23878
|
log.error(
|
|
23763
23879
|
` Sample ${index + 1} was refused: ${body.message ?? body.error ?? `HTTP ${res.status}`}${settledSoFar > 0 ? " (earlier samples still count)" : ""}`
|
|
23764
23880
|
);
|
|
23765
|
-
return res.status >= 500 || res.status === 401 || res.status === 403;
|
|
23881
|
+
return { stop: res.status >= 500 || res.status === 401 || res.status === 403, refusal };
|
|
23882
|
+
}
|
|
23883
|
+
function printProviderRefusal(log, status2, message, awaitingBill = false) {
|
|
23884
|
+
if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so this call's coin stays held until it resolves (see \`genex llm status\`); it is not a sample.`);
|
|
23885
|
+
else log.dim(" The provider refused this call at its door \u2014 no inference ran, it cost nothing, and it is not a sample.");
|
|
23886
|
+
if (message) log.dim(` Provider said: ${message}`);
|
|
23887
|
+
if (status2 === 401 || status2 === 403) {
|
|
23888
|
+
log.dim(` A ${status2} here is this stand's provider configuration refusing the model \u2014 a message for the operator, not something to fix in the game.`);
|
|
23889
|
+
}
|
|
23890
|
+
}
|
|
23891
|
+
async function readSlotCounts(call, base) {
|
|
23892
|
+
const res = await call(`${base}?scope=open`).catch(() => null);
|
|
23893
|
+
if (!res?.ok) return null;
|
|
23894
|
+
const body = await res.json().catch(() => null);
|
|
23895
|
+
return serverSlotCounts(body);
|
|
23766
23896
|
}
|
|
23767
23897
|
async function pollSettled(call, url, timeoutSec) {
|
|
23768
23898
|
const deadline = Date.now() + timeoutSec * 1e3;
|
|
@@ -23820,6 +23950,7 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
|
|
|
23820
23950
|
balanceBefore: null,
|
|
23821
23951
|
balanceAfter: null,
|
|
23822
23952
|
results: [],
|
|
23953
|
+
refusal: null,
|
|
23823
23954
|
chargedCoins: { p50: null, p95: null, max: null },
|
|
23824
23955
|
recommendedDeclaredHeadroomBps: null,
|
|
23825
23956
|
recommendedEstimateCoins: null,
|
|
@@ -23827,6 +23958,166 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
|
|
|
23827
23958
|
savedTo: null
|
|
23828
23959
|
};
|
|
23829
23960
|
}
|
|
23961
|
+
async function reportStatus(args) {
|
|
23962
|
+
const { apiUrl, token, projectId, opts, log } = args;
|
|
23963
|
+
const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
|
|
23964
|
+
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
23965
|
+
const res = await call(`${base}?scope=open`).catch(() => null);
|
|
23966
|
+
if (!res || !res.ok) {
|
|
23967
|
+
if (res && printedStructuredError(res)) {
|
|
23968
|
+
if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", `http_${res.status}`, [], null, null));
|
|
23969
|
+
process.exitCode = 1;
|
|
23970
|
+
return;
|
|
23971
|
+
}
|
|
23972
|
+
const body2 = res ? await res.json().catch(() => ({})) : {};
|
|
23973
|
+
const error = body2.error ?? (res ? `http_${res.status}` : "unreachable");
|
|
23974
|
+
if (opts.json) {
|
|
23975
|
+
writeJsonLine(statusJson(apiUrl, projectId, res?.status === 404 && body2.error === "not_found" ? "off" : "failed", error, [], null, null));
|
|
23976
|
+
if (!(res?.status === 404 && body2.error === "not_found")) process.exitCode = 1;
|
|
23977
|
+
return;
|
|
23978
|
+
}
|
|
23979
|
+
if (res?.status === 404 && body2.error === "not_found") {
|
|
23980
|
+
log.plain(LLM_LANE_OFF_LINE);
|
|
23981
|
+
return;
|
|
23982
|
+
}
|
|
23983
|
+
if (res?.status === 403 && body2.error === "credential_scope") {
|
|
23984
|
+
log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
|
|
23985
|
+
log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
|
|
23986
|
+
} else if (res?.status === 403 && body2.error === "project_owner_required") {
|
|
23987
|
+
log.error(" Development calls belong to a project YOU own, and this one isn't yours.");
|
|
23988
|
+
} else if (!res) {
|
|
23989
|
+
log.error(` Couldn't reach ${apiUrl}.`);
|
|
23990
|
+
log.dim(` Check the network, then ${c.cyan("genex doctor")}.`);
|
|
23991
|
+
} else {
|
|
23992
|
+
log.error(` Couldn't read the open calls: ${error}.`);
|
|
23993
|
+
}
|
|
23994
|
+
process.exitCode = 1;
|
|
23995
|
+
return;
|
|
23996
|
+
}
|
|
23997
|
+
const body = await res.json().catch(() => null);
|
|
23998
|
+
if (!body || !Array.isArray(body.generations)) {
|
|
23999
|
+
if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", "unreadable_response", [], null, null));
|
|
24000
|
+
else log.error(` ${apiUrl} answered, but not with a list of calls.`);
|
|
24001
|
+
process.exitCode = 1;
|
|
24002
|
+
return;
|
|
24003
|
+
}
|
|
24004
|
+
const rows = body.generations;
|
|
24005
|
+
const counts = serverSlotCounts(body);
|
|
24006
|
+
if (opts.json) {
|
|
24007
|
+
writeJsonLine(statusJson(apiUrl, projectId, "ok", null, rows, counts?.held ?? null, counts?.limit ?? null));
|
|
24008
|
+
return;
|
|
24009
|
+
}
|
|
24010
|
+
log.plain(c.bold("genex llm status"));
|
|
24011
|
+
log.dim(` ${apiUrl} \xB7 project ${projectId}`);
|
|
24012
|
+
log.plain("");
|
|
24013
|
+
if (rows.length === 0) {
|
|
24014
|
+
log.plain(" No open calls in this project \u2014 nothing active, nothing awaiting a bill.");
|
|
24015
|
+
} else {
|
|
24016
|
+
const now = Date.now();
|
|
24017
|
+
for (const row of rows) {
|
|
24018
|
+
const state = rowState(row);
|
|
24019
|
+
const mark = state === "active" ? c.cyan("\u25CF") : state === "awaiting bill" ? c.yellow("\u25D0") : c.green("\u25CB");
|
|
24020
|
+
const reserved = typeof row.reservedCoins === "number" ? `${row.reservedCoins} coin reserved` : "reserved \u2014";
|
|
24021
|
+
log.plain(
|
|
24022
|
+
` ${mark} ${row.id ?? "?"} ${row.modelId ?? "?"} ${state.padEnd(13)} ${formatAge(row.createdAt, now).padStart(4)} old ${reserved}${row.error ? ` ${row.error}` : ""}${rowHoldsSlot(row) ? "" : c.dim(" (no slot)")}`
|
|
24023
|
+
);
|
|
24024
|
+
if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
|
|
24025
|
+
}
|
|
24026
|
+
}
|
|
24027
|
+
log.plain("");
|
|
24028
|
+
log.dim(` ${LLM_STATUS_LEGEND}`);
|
|
24029
|
+
if (counts === null) {
|
|
24030
|
+
log.dim(` ${LLM_STATUS_NO_COUNT_LINE}`);
|
|
24031
|
+
return;
|
|
24032
|
+
}
|
|
24033
|
+
log.plain(` ${counts.held} of ${counts.limit} slots held${counts.held > rows.filter(rowHoldsSlot).length ? " \u2014 across every project, this one's rows included" : ""}`);
|
|
24034
|
+
if (counts.held >= counts.limit) {
|
|
24035
|
+
log.dim(` Every slot is held, so a new ad-hoc call answers generation_limit until one frees: ${c.cyan("genex llm cancel <id>")} stops an active one; a stopped one frees on its own.`);
|
|
24036
|
+
}
|
|
24037
|
+
}
|
|
24038
|
+
function statusJson(apiUrl, projectId, status2, error, rows, slotsHeld, slotLimit) {
|
|
24039
|
+
return {
|
|
24040
|
+
command: "llm status",
|
|
24041
|
+
status: status2,
|
|
24042
|
+
error,
|
|
24043
|
+
apiUrl,
|
|
24044
|
+
projectId,
|
|
24045
|
+
// Every row verbatim, plus the CLI's own reading of it — so a program
|
|
24046
|
+
// gets both the server's fields and the three words a person sees.
|
|
24047
|
+
generations: rows.map((row) => ({ ...row, state: rowState(row), holdsSlot: rowHoldsSlot(row) })),
|
|
24048
|
+
slotsHeld,
|
|
24049
|
+
slotLimit,
|
|
24050
|
+
pendingSlotReleaseMinutes: PENDING_SLOT_RELEASE_MINUTES
|
|
24051
|
+
};
|
|
24052
|
+
}
|
|
24053
|
+
async function cancelCall(args) {
|
|
24054
|
+
const { apiUrl, token, projectId, id, opts, log } = args;
|
|
24055
|
+
const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
|
|
24056
|
+
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
24057
|
+
const rowUrl = `${base}/${encodeURIComponent(id)}`;
|
|
24058
|
+
const done = (status2, error, reason, row2) => {
|
|
24059
|
+
if (opts.json) writeJsonLine({ command: "llm cancel", status: status2, error, id, cancelled: reason === "cancelled", reason, generation: row2 });
|
|
24060
|
+
if (status2 === "failed") process.exitCode = 1;
|
|
24061
|
+
};
|
|
24062
|
+
const read = await call(rowUrl).catch(() => null);
|
|
24063
|
+
if (!read || !read.ok) {
|
|
24064
|
+
if (read && printedStructuredError(read)) return done("failed", `http_${read.status}`, null, null);
|
|
24065
|
+
const body = read ? await read.json().catch(() => ({})) : {};
|
|
24066
|
+
const error = body.error ?? (read ? `http_${read.status}` : "unreachable");
|
|
24067
|
+
if (!opts.json) {
|
|
24068
|
+
if (read?.status === 404) {
|
|
24069
|
+
log.error(` No call ${id} in this project \u2014 ${c.cyan("genex llm status")} lists the open ones.`);
|
|
24070
|
+
} else if (read?.status === 403 && body.error === "credential_scope") {
|
|
24071
|
+
log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
|
|
24072
|
+
log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
|
|
24073
|
+
} else if (!read) {
|
|
24074
|
+
log.error(` Couldn't reach ${apiUrl}.`);
|
|
24075
|
+
} else {
|
|
24076
|
+
log.error(` Couldn't read call ${id}: ${error}.`);
|
|
24077
|
+
}
|
|
24078
|
+
}
|
|
24079
|
+
return done("failed", error, null, null);
|
|
24080
|
+
}
|
|
24081
|
+
const row = await read.json().catch(() => null);
|
|
24082
|
+
if (!row?.id) {
|
|
24083
|
+
if (!opts.json) log.error(` ${apiUrl} answered, but not with a call.`);
|
|
24084
|
+
return done("failed", "unreadable_response", null, null);
|
|
24085
|
+
}
|
|
24086
|
+
const state = rowState(row);
|
|
24087
|
+
if (state !== "active") {
|
|
24088
|
+
const sentence = state === "awaiting bill" ? LLM_CANCEL_AWAITING_BILL : LLM_CANCEL_ALREADY_FINAL;
|
|
24089
|
+
if (!opts.json) {
|
|
24090
|
+
log.warn(` ${row.id} (${row.status ?? "?"}) \u2014 ${sentence}.`);
|
|
24091
|
+
if (state === "awaiting bill") {
|
|
24092
|
+
log.dim(` Nothing to cancel: its coin releases when the bill resolves, and it stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`);
|
|
24093
|
+
if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
|
|
24094
|
+
} else {
|
|
24095
|
+
log.dim(` ${typeof row.chargedCoins === "number" ? `${row.chargedCoins} coin charged` : "charge unknown"}${row.error ? ` \xB7 ${row.error}` : ""}.`);
|
|
24096
|
+
}
|
|
24097
|
+
}
|
|
24098
|
+
return done("ok", null, state === "awaiting bill" ? "awaiting_bill" : "already_final", row);
|
|
24099
|
+
}
|
|
24100
|
+
const res = await call(`${rowUrl}/cancel`, { method: "POST" }).catch(() => null);
|
|
24101
|
+
if (!res || !res.ok) {
|
|
24102
|
+
if (res && printedStructuredError(res)) return done("failed", `http_${res.status}`, null, row);
|
|
24103
|
+
const body = res ? await res.json().catch(() => ({})) : {};
|
|
24104
|
+
const error = body.error ?? (res ? `http_${res.status}` : "unreachable");
|
|
24105
|
+
if (!opts.json) {
|
|
24106
|
+
if (res?.status === 409 && body.error === "already_dispatched") {
|
|
24107
|
+
log.error(` ${row.id} is already running at the provider and cannot be cancelled now; it settles on its own \u2014 ${c.cyan("genex llm status")} shows it.`);
|
|
24108
|
+
} else {
|
|
24109
|
+
log.error(` Couldn't cancel ${row.id}: ${error}.`);
|
|
24110
|
+
}
|
|
24111
|
+
}
|
|
24112
|
+
return done("failed", error, null, row);
|
|
24113
|
+
}
|
|
24114
|
+
const after = await res.json().catch(() => null) ?? row;
|
|
24115
|
+
if (!opts.json) {
|
|
24116
|
+
log.success(` Cancelled ${row.id} (was ${row.status ?? "active"}).`);
|
|
24117
|
+
log.dim(` Its coin releases once the bill settles; ${c.cyan("genex llm status")} shows it until then.`);
|
|
24118
|
+
}
|
|
24119
|
+
return done("ok", null, "cancelled", after);
|
|
24120
|
+
}
|
|
23830
24121
|
async function reportSavedPrice(cwd, opts, log) {
|
|
23831
24122
|
let record = null;
|
|
23832
24123
|
try {
|
|
@@ -24139,10 +24430,11 @@ function runtimeRow(state, lane, toolsOnly = false) {
|
|
|
24139
24430
|
switch (lane?.state) {
|
|
24140
24431
|
case "live": {
|
|
24141
24432
|
const n = lane.models?.length ?? 0;
|
|
24433
|
+
const keyRefused = lane.providerKey === "rejected";
|
|
24142
24434
|
return {
|
|
24143
24435
|
label: "Runtime",
|
|
24144
|
-
value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}`,
|
|
24145
|
-
fix: toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
|
|
24436
|
+
value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}${keyRefused ? ` \xB7 ${c.yellow("provider key refused")}` : ""}`,
|
|
24437
|
+
fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
|
|
24146
24438
|
};
|
|
24147
24439
|
}
|
|
24148
24440
|
case "off":
|
|
@@ -26076,17 +26368,20 @@ ${c.bold("Usage")}
|
|
|
26076
26368
|
the asset table against it. --assets <credits>
|
|
26077
26369
|
--user-approved raises it, only once the player
|
|
26078
26370
|
has agreed to the number.
|
|
26079
|
-
genex llm <sub> [options] In-game model calls: models | bench | price
|
|
26080
|
-
"models" says whether this
|
|
26081
|
-
runtime LLM lane at all, and
|
|
26082
|
-
rates. "bench" runs the real
|
|
26083
|
-
coin and reports what each
|
|
26084
|
-
it needs --max-coins <n>
|
|
26085
|
-
any spend the player has
|
|
26086
|
-
reprints the last run's
|
|
26087
|
-
estimateCoins.
|
|
26088
|
-
|
|
26089
|
-
|
|
26371
|
+
genex llm <sub> [options] In-game model calls: models | bench | price |
|
|
26372
|
+
status | cancel. "models" says whether this
|
|
26373
|
+
stand serves the runtime LLM lane at all, and
|
|
26374
|
+
at what provider rates. "bench" runs the real
|
|
26375
|
+
model on YOUR OWN coin and reports what each
|
|
26376
|
+
attempt was charged; it needs --max-coins <n>
|
|
26377
|
+
--user-approved, like any spend the player has
|
|
26378
|
+
to agree to. "price" reprints the last run's
|
|
26379
|
+
recommended estimateCoins. "status" lists this
|
|
26380
|
+
project's open calls and the slots they hold;
|
|
26381
|
+
"cancel <id>" stops an active one. Declare a
|
|
26382
|
+
game's price from a bench, never from a guess:
|
|
26383
|
+
a started attempt is charged in full, including
|
|
26384
|
+
one that fails.
|
|
26090
26385
|
genex player <sub> [options] Answer the in-game model requests you approve, on your
|
|
26091
26386
|
OWN Claude or ChatGPT subscription: install | run |
|
|
26092
26387
|
status | stop | uninstall. "install" signs this machine
|
|
@@ -26520,8 +26815,11 @@ ${c.bold("Examples")}
|
|
|
26520
26815
|
genex shop test sku_123
|
|
26521
26816
|
genex shop remove sku_123
|
|
26522
26817
|
genex llm models
|
|
26818
|
+
genex llm models --all
|
|
26523
26819
|
genex llm bench "Reply with one short taunt." --samples 5 --max-coins 20 --user-approved
|
|
26524
26820
|
genex llm price
|
|
26821
|
+
genex llm status
|
|
26822
|
+
genex llm cancel <id>
|
|
26525
26823
|
genex player install
|
|
26526
26824
|
genex player status --json
|
|
26527
26825
|
genex player uninstall
|
|
@@ -26758,11 +27056,13 @@ ${c.bold("How the allowance works")}
|
|
|
26758
27056
|
llm: `${c.bold("genex llm")} \u2014 in-game model calls: is the lane live, what does one call cost, what should the game declare?
|
|
26759
27057
|
|
|
26760
27058
|
${c.bold("Usage")}
|
|
26761
|
-
genex llm models [--json]
|
|
27059
|
+
genex llm models [--all] [--json] Which models this stand serves, their provider rates per
|
|
26762
27060
|
million tokens, and the headroom it recommends over a
|
|
26763
|
-
benchmark.
|
|
26764
|
-
|
|
26765
|
-
|
|
27061
|
+
benchmark. The catalog is synced from OpenRouter, so the
|
|
27062
|
+
FEATURED short list is printed by default and --all
|
|
27063
|
+
prints every row (--json always carries every row). Off
|
|
27064
|
+
on this stand: it says so and exits 0 \u2014 build the feature
|
|
27065
|
+
without a model rather than promising a 404.
|
|
26766
27066
|
genex llm bench "<prompt>" --max-coins <n> --user-approved [options]
|
|
26767
27067
|
Run the real model <n> times through the development
|
|
26768
27068
|
lane and report what each attempt was CHARGED. Refused
|
|
@@ -26771,6 +27071,14 @@ ${c.bold("Usage")}
|
|
|
26771
27071
|
cannot see. Needs a folder linked to a game you own.
|
|
26772
27072
|
genex llm price [--json] The recommendation from the last bench in this folder.
|
|
26773
27073
|
Reads a file; spends nothing.
|
|
27074
|
+
genex llm status [--json] This project's open development calls: active, or
|
|
27075
|
+
stopped but awaiting the provider's bill \u2014 with the
|
|
27076
|
+
provider's own message when it refused one \u2014 and
|
|
27077
|
+
how many of the ad-hoc slots they hold. A bench
|
|
27078
|
+
refused as generation_limit sends you here.
|
|
27079
|
+
genex llm cancel <id> [--json] Stop one ACTIVE call. A call that already stopped
|
|
27080
|
+
is reported as such: awaiting its bill (the hold
|
|
27081
|
+
resolves on its own) or already final.
|
|
26774
27082
|
|
|
26775
27083
|
${c.bold("Bench options")}
|
|
26776
27084
|
--model <id> One of the ids \`genex llm models\` prints (default: the first one).
|
|
@@ -27280,7 +27588,11 @@ function parseArgs(argv) {
|
|
|
27280
27588
|
(parsed.options.selectors ??= []).push(arg);
|
|
27281
27589
|
}
|
|
27282
27590
|
} else if (parsed.command === "llm") {
|
|
27283
|
-
parsed.options.
|
|
27591
|
+
if (parsed.options.name === "cancel" && parsed.options.generationId === void 0) {
|
|
27592
|
+
parsed.options.generationId = arg;
|
|
27593
|
+
} else {
|
|
27594
|
+
parsed.options.benchPrompt = parsed.options.benchPrompt ? `${parsed.options.benchPrompt} ${arg}` : arg;
|
|
27595
|
+
}
|
|
27284
27596
|
} else if (parsed.command === "shop") {
|
|
27285
27597
|
if (!parsed.options.hostname) parsed.options.hostname = arg;
|
|
27286
27598
|
else {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@genex-ai/cli-demo",
|
|
3
|
-
"version": "1.35.0-dev.
|
|
3
|
+
"version": "1.35.0-dev.743",
|
|
4
4
|
"description": "Set up your project's agent workspace (.claude/.codex/.cursor in the game folder), authorize, create a game project, generate AI assets, and publish (genex CLI).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -31,24 +31,36 @@ or two per session means a grant.
|
|
|
31
31
|
npx genex llm models
|
|
32
32
|
```
|
|
33
33
|
|
|
34
|
-
- **live** — it prints the models this stand serves
|
|
34
|
+
- **live** — it prints the FEATURED models this stand serves, and how many more
|
|
35
|
+
there are. Build the feature.
|
|
35
36
|
- **off on this stand** — the routes answer 404. Build the feature behind a
|
|
36
37
|
graceful `unavailable` state (the NPC uses its authored lines, the quest falls
|
|
37
38
|
back to the written one) and **say so plainly in the handoff**. Never promise
|
|
38
39
|
the player something that 404s.
|
|
39
40
|
- **misconfigured** — say that too; it is an operator fix, not a game bug.
|
|
40
41
|
|
|
42
|
+
The stand's catalog is synced and filtered, so it can hold far more rows than
|
|
43
|
+
anyone wants to read. **Offer the featured set. Run `npx genex llm models --all`
|
|
44
|
+
only when the user asks for more**, and put the full list in front of them
|
|
45
|
+
rather than picking an obscure row for them.
|
|
46
|
+
|
|
41
47
|
Model ids come from that command and from `getGenerationModels()` at runtime.
|
|
42
48
|
Never write one into the game's source: they differ per stand, and a hardcoded
|
|
43
|
-
id is a feature that dies on somebody else's environment.
|
|
49
|
+
id is a feature that dies on somebody else's environment. **A picker's label
|
|
50
|
+
must be the model's own `label`** — a name you invent for a row ("Fast", "Smart")
|
|
51
|
+
is a label that differs from the model the player is billed for.
|
|
44
52
|
|
|
45
53
|
## The SDK surface (exact — do not invent methods)
|
|
46
54
|
|
|
47
55
|
From `@genex-ai/embed-sdk`, already installed. `initEmbed()` must have run and
|
|
48
56
|
identity must be resolved first — `$genex-threejs-embed-auth`.
|
|
49
57
|
|
|
50
|
-
- `getGenerationModels()` —
|
|
51
|
-
from this, never from a list you wrote
|
|
58
|
+
- `getGenerationModels()` — `{ models, featuredCount, … }`, featured first. Any
|
|
59
|
+
picker renders from this, never from a list you wrote: show the rows where
|
|
60
|
+
`featured` is true (or all of them when `featuredCount` is 0) and offer the
|
|
61
|
+
rest only if the player asks. Each row carries `label`, `vendor`,
|
|
62
|
+
`contextLength`, `personalPlan` and `structuredOutputs`; render `label`
|
|
63
|
+
verbatim, so the name on screen is the model that gets billed.
|
|
52
64
|
- `generate({ modelId, prompt, outputFormat, schema?, estimateCoins,
|
|
53
65
|
allowExternal?, idempotencyKey?, grantId?, timeoutMs? })` →
|
|
54
66
|
`{ status, generationId, output, source, error }` plus billing fields
|
|
@@ -108,6 +120,41 @@ the price and makes `requestSpendGrant()` refuse before it reaches the network.
|
|
|
108
120
|
loop into disclosure numbers, re-benchmarking after a prompt change — is in
|
|
109
121
|
[references/pricing.md](references/pricing.md).
|
|
110
122
|
|
|
123
|
+
Only samples that **succeeded and settled** are priced from. A sample the
|
|
124
|
+
provider refused at its door (`provider_http_<status>`) ran no inference and
|
|
125
|
+
cost nothing; the bench prints the code, the provider's own message and, for a
|
|
126
|
+
401 or 403, that this is the stand's provider configuration refusing the model
|
|
127
|
+
— an operator's problem, never something to fix in the game. A sample refused
|
|
128
|
+
as `generation_limit` never started: you already have the lane's three ad-hoc
|
|
129
|
+
calls open, or recently stopped with their bill still pending. The bench says
|
|
130
|
+
how many slots are held and stops. Do not re-run it into the same refusal —
|
|
131
|
+
read `npx genex llm status`, then wait for a bill to resolve or
|
|
132
|
+
`npx genex llm cancel <id>` an active call.
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
npx genex llm status # this project's open calls: active, or awaiting their bill; N of 3 slots held
|
|
136
|
+
npx genex llm cancel <id> # stop an ACTIVE call; a stopped one is reported, not cancelled
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
## The schema dialect (exact — anything else is refused)
|
|
140
|
+
|
|
141
|
+
`schema` is validated by the platform before the call, and a refused schema is
|
|
142
|
+
`invalid_schema` at the door — nothing is charged, nothing runs. The accepted
|
|
143
|
+
subset, and it is the whole subset:
|
|
144
|
+
|
|
145
|
+
- `type`: `object`, `array`, `string`, `number`, `integer`, `boolean`, `null`
|
|
146
|
+
- objects: `properties`, `required`, and `additionalProperties: false` on
|
|
147
|
+
**every** object (required, not optional)
|
|
148
|
+
- arrays: `items`
|
|
149
|
+
- `enum` (strings, numbers, booleans, `null`)
|
|
150
|
+
- `minimum` / `maximum`, `minLength` / `maxLength`, `minItems` / `maxItems`
|
|
151
|
+
- `description` and `title`, on any node
|
|
152
|
+
|
|
153
|
+
Everything else is refused, including `$schema`, `default`, `examples`,
|
|
154
|
+
`pattern`, `format`, `anyOf` / `oneOf` / `allOf` and `$ref`. Keep the schema
|
|
155
|
+
in one file (`./answer.schema.json`), benchmark with that file, and ship the
|
|
156
|
+
same object — a schema that passed the bench passes the game.
|
|
157
|
+
|
|
111
158
|
## Standing budgets
|
|
112
159
|
|
|
113
160
|
```ts
|
|
@@ -183,7 +230,9 @@ There is no compensation lane, so this is all work you do before the call:
|
|
|
183
230
|
|
|
184
231
|
- **Validate inputs first** — a malformed prompt is still charged.
|
|
185
232
|
- **Always set `schema` for `outputFormat: 'json'`** — unschema'd JSON is the
|
|
186
|
-
commonest way a call is charged and the result is unusable.
|
|
233
|
+
commonest way a call is charged and the result is unusable. Write it in the
|
|
234
|
+
accepted dialect above; a refused schema is `invalid_schema` and costs
|
|
235
|
+
nothing, but it is a feature that never runs.
|
|
187
236
|
- **Keep prompts short.** Long context is the price.
|
|
188
237
|
- **Never loop `generate()` without a grant**, and never retry in a loop — each
|
|
189
238
|
attempt is a separate charge.
|
|
@@ -222,7 +271,10 @@ These are source contracts, not a claim that every stand runs this lane —
|
|
|
222
271
|
|
|
223
272
|
- [ ] `npx genex llm models` was run and its verdict is in the handoff
|
|
224
273
|
- [ ] Model ids come from `getGenerationModels()`, never from source
|
|
274
|
+
- [ ] The picker offers the featured set, labelled with the server's own `label`
|
|
225
275
|
- [ ] `estimateCoins` came from `npx genex llm bench`, not from judgement
|
|
276
|
+
- [ ] The schema uses only the accepted dialect (no `$ref`, `pattern`, `format`, `anyOf`, `default`)
|
|
277
|
+
- [ ] A bench refused as `generation_limit` was answered with `npx genex llm status`, never a retry
|
|
226
278
|
- [ ] `generate()` / `requestSpendGrant()` is the first statement of a click handler
|
|
227
279
|
- [ ] A repeated-call feature uses a grant; a one-off uses `generate()`
|
|
228
280
|
- [ ] Disclosure numbers derive from the real loop and are written in `DESIGN.md`
|
|
@@ -256,6 +308,25 @@ cause rather than showing it for both.
|
|
|
256
308
|
**`grant_price_unreasonable`** — the declared per-call price is far above what
|
|
257
309
|
that prompt can cost on that model. Re-benchmark and declare what it prints.
|
|
258
310
|
|
|
311
|
+
**`generation_limit`** — three ad-hoc calls are already in flight for this
|
|
312
|
+
account, or recently stopped with their bill still pending; a pending call
|
|
313
|
+
stops counting ten minutes after it was dispatched. From the bench: run
|
|
314
|
+
`npx genex llm status`, then wait or `npx genex llm cancel <id>` an active one —
|
|
315
|
+
never re-run into the same refusal. In the game: the player is clicking faster
|
|
316
|
+
than one-off calls are meant for, which is the signal that this feature wants a
|
|
317
|
+
grant.
|
|
318
|
+
|
|
319
|
+
**`provider_http_<status>`** — the provider refused the call at its door, before
|
|
320
|
+
any inference: it cost nothing and is not a sample. Read the provider's own
|
|
321
|
+
message (`npx genex llm status` prints it under the row). A 401 or 403 is this
|
|
322
|
+
stand's provider configuration refusing the model — tell the operator, and
|
|
323
|
+
build nothing around it in the game.
|
|
324
|
+
|
|
325
|
+
**`invalid_schema`** — the schema uses a keyword outside the accepted dialect
|
|
326
|
+
(`$ref`, `pattern`, `format`, `anyOf`, `default`, `examples`, `$schema`), or an
|
|
327
|
+
object without `additionalProperties: false`. Nothing was charged. Rewrite it in
|
|
328
|
+
the subset above; `description` and `title` are allowed.
|
|
329
|
+
|
|
259
330
|
**`grant_concurrency` / `grant_rate_limited`** — the game calls faster than the
|
|
260
331
|
grant's own limits. Batch and cache; do not raise the limits to hide it.
|
|
261
332
|
|
|
@@ -49,7 +49,19 @@ npx genex llm bench "<the frozen prompt, one real example filled in>" \
|
|
|
49
49
|
## 3. Read the output
|
|
50
50
|
|
|
51
51
|
Each sample prints what it actually charged. The aggregate prints p50, p95 and
|
|
52
|
-
max of the charged coins
|
|
52
|
+
max of the charged coins **over the samples that succeeded and settled**, plus
|
|
53
|
+
one recommendation. A sample that failed is printed with its code and stays
|
|
54
|
+
out of the numbers; a sample the provider refused at its door
|
|
55
|
+
(`provider_http_<status>`) ran no inference, cost nothing, and prints the
|
|
56
|
+
provider's own message under its row — on a 401 or 403 that is the stand's
|
|
57
|
+
provider configuration refusing the model, which is the operator's to fix. A
|
|
58
|
+
sample refused as `generation_limit` never started and ends the run: the
|
|
59
|
+
account's three ad-hoc calls are open or recently stopped with a pending
|
|
60
|
+
bill. `npx genex llm status` lists them with what each holds;
|
|
61
|
+
`npx genex llm cancel <id>` stops an active one; a stopped one frees on its
|
|
62
|
+
own once its bill resolves, and stops holding a slot ten minutes after
|
|
63
|
+
dispatch. Re-running the bench into the same refusal spends nothing and
|
|
64
|
+
learns nothing.
|
|
53
65
|
|
|
54
66
|
- **p50** is what a typical call costs. It is the number to reason about when
|
|
55
67
|
you ask "can the game afford this loop?" — multiply it by the calls per
|
|
@@ -53,7 +53,11 @@ expand the offer into a pitch — one line, one question.
|
|
|
53
53
|
which models. It answers in this folder as it is, before anything is
|
|
54
54
|
converted, so it comes first: if it says the lane is off, in-game calls
|
|
55
55
|
answer 404 here — tell the user so plainly, build the graceful fallback, and
|
|
56
|
-
do not convert a folder for a feature the stand does not serve.
|
|
56
|
+
do not convert a folder for a feature the stand does not serve. It prints
|
|
57
|
+
the FEATURED models; `--all` lists the whole catalog, and you run that only
|
|
58
|
+
when the user asks for more. Whatever they pick, the picker in the game
|
|
59
|
+
shows the server's own label for it — a name you invent is a name that
|
|
60
|
+
differs from the model they are billed for.
|
|
57
61
|
2. `npx genex init --convert` — it connects this folder to a hosted Genex game
|
|
58
62
|
in place. The code, the files and this toolkit stay exactly as they are, and
|
|
59
63
|
generations still land in `./assets`. It is the user's yes that runs it, so
|