@genex-ai/cli-demo 1.35.0-dev.742 → 1.35.0-dev.744
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js
CHANGED
|
@@ -23288,7 +23288,7 @@ import path35 from "path";
|
|
|
23288
23288
|
import fs33 from "fs/promises";
|
|
23289
23289
|
import path34 from "path";
|
|
23290
23290
|
import { randomUUID as randomUUID2 } from "crypto";
|
|
23291
|
-
var SUBS3 = ["models", "bench", "price"];
|
|
23291
|
+
var SUBS3 = ["models", "bench", "price", "status", "cancel"];
|
|
23292
23292
|
var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN coin \u2014 your wallet, not a player's, and one this build's asset allowance cannot see. Re-run with --max-coins <n> --user-approved only after the person at the keyboard agreed to the number.";
|
|
23293
23293
|
var LLM_LANE_OFF_LINE = "The runtime LLM lane is off on this stand \u2014 in-game generate() answers 404 here, and nothing can be benchmarked.";
|
|
23294
23294
|
var LLM_TOOLS_CONVERT_HINT = [
|
|
@@ -23297,8 +23297,25 @@ var LLM_TOOLS_CONVERT_HINT = [
|
|
|
23297
23297
|
"Ask the user first; on their yes run `npx genex init --convert`, then load the genex-tool-llm card."
|
|
23298
23298
|
];
|
|
23299
23299
|
var LLM_BENCH_FILE = ".genex/llm-bench.json";
|
|
23300
|
+
function serverSlotCounts(body) {
|
|
23301
|
+
if (!body || typeof body.slotsHeld !== "number" || typeof body.slotLimit !== "number" || body.slotLimit <= 0) return null;
|
|
23302
|
+
return { held: body.slotsHeld, limit: body.slotLimit };
|
|
23303
|
+
}
|
|
23304
|
+
var PENDING_SLOT_RELEASE_MINUTES = 10;
|
|
23305
|
+
var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so its coin stays held until the bill resolves; such a row stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`;
|
|
23306
|
+
var LLM_STATUS_NO_COUNT_LINE = "This stand does not report how many slots are held across your projects, so no count is shown.";
|
|
23307
|
+
var LLM_CANCEL_AWAITING_BILL = "this call already stopped; its bill is awaiting the provider";
|
|
23308
|
+
var LLM_CANCEL_ALREADY_FINAL = "this call is already final";
|
|
23309
|
+
function benchSlotsHeldSentence(counts) {
|
|
23310
|
+
const count = counts === null ? "every ad-hoc slot is" : `${counts.held} of ${counts.limit} ad-hoc slots are`;
|
|
23311
|
+
return `refused before it started, nothing spent: ${count} held on this stand \u2014 \`genex llm status\` shows the calls holding them; wait for a bill to resolve or cancel an active one, then re-run.`;
|
|
23312
|
+
}
|
|
23313
|
+
var LLM_PROVIDER_KEY_REJECTED_LINE = "This stand's provider key is refused \u2014 every in-game call answers provider_http_401 until an operator replaces it.";
|
|
23300
23314
|
var DEFAULT_BENCH_SAMPLES = 3;
|
|
23301
23315
|
var DEFAULT_SAMPLE_TIMEOUT_SEC = 180;
|
|
23316
|
+
function parseProviderKey(value) {
|
|
23317
|
+
return value === "ok" || value === "rejected" || value === "unknown" ? value : null;
|
|
23318
|
+
}
|
|
23302
23319
|
async function readRuntimeLane(apiUrl, token) {
|
|
23303
23320
|
const empty = (state) => ({
|
|
23304
23321
|
state,
|
|
@@ -23306,7 +23323,8 @@ async function readRuntimeLane(apiUrl, token) {
|
|
|
23306
23323
|
recommendedDeclaredHeadroomBps: null,
|
|
23307
23324
|
externalProviders: null,
|
|
23308
23325
|
total: null,
|
|
23309
|
-
featuredCount: null
|
|
23326
|
+
featuredCount: null,
|
|
23327
|
+
providerKey: null
|
|
23310
23328
|
});
|
|
23311
23329
|
try {
|
|
23312
23330
|
const res = await apiFetch(`${apiUrl}/api/runtime/models`, {
|
|
@@ -23324,7 +23342,8 @@ async function readRuntimeLane(apiUrl, token) {
|
|
|
23324
23342
|
recommendedDeclaredHeadroomBps: typeof body.recommendedDeclaredHeadroomBps === "number" && Number.isFinite(body.recommendedDeclaredHeadroomBps) ? body.recommendedDeclaredHeadroomBps : null,
|
|
23325
23343
|
externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null,
|
|
23326
23344
|
total: typeof body.total === "number" && Number.isFinite(body.total) ? body.total : null,
|
|
23327
|
-
featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null
|
|
23345
|
+
featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null,
|
|
23346
|
+
providerKey: parseProviderKey(body.providerKey)
|
|
23328
23347
|
};
|
|
23329
23348
|
} catch {
|
|
23330
23349
|
return empty("unknown");
|
|
@@ -23367,6 +23386,11 @@ async function runLlm(opts = {}) {
|
|
|
23367
23386
|
}
|
|
23368
23387
|
const ws = await resolveWorkspace(cwd);
|
|
23369
23388
|
const toolsOnly = ws.mode === "tools" && !ws.hosted;
|
|
23389
|
+
const projectBound = sub === "bench" || sub === "status" || sub === "cancel";
|
|
23390
|
+
if (sub === "cancel" && !opts.generationId?.trim()) {
|
|
23391
|
+
fail4("llm cancel", "`genex llm cancel` needs the id of the call to stop, e.g. genex llm cancel <id> \u2014 `genex llm status` lists the open ones.");
|
|
23392
|
+
return;
|
|
23393
|
+
}
|
|
23370
23394
|
if (sub === "bench") {
|
|
23371
23395
|
if (opts.maxCoins === void 0 || opts.userApproved !== true) {
|
|
23372
23396
|
fail4("llm bench", LLM_BENCH_APPROVAL_REQUIRED);
|
|
@@ -23384,22 +23408,22 @@ async function runLlm(opts = {}) {
|
|
|
23384
23408
|
fail4("llm bench", "Pass --json-output or --text, not both.");
|
|
23385
23409
|
return;
|
|
23386
23410
|
}
|
|
23387
|
-
|
|
23388
|
-
|
|
23389
|
-
|
|
23390
|
-
|
|
23391
|
-
|
|
23392
|
-
|
|
23393
|
-
}
|
|
23394
|
-
process.exitCode = 1;
|
|
23395
|
-
return;
|
|
23411
|
+
}
|
|
23412
|
+
if (projectBound && toolsOnly) {
|
|
23413
|
+
if (opts.json) writeJsonLine({ command: `llm ${sub}`, status: "failed", error: `\`genex llm ${sub}\` is not part of Genex Tools.` });
|
|
23414
|
+
else {
|
|
23415
|
+
reportPlatformRefused(log, `llm ${sub}`);
|
|
23416
|
+
log.plain(` ${c.cyan("\u2192")} A model that runs while people PLAY is a platform feature: the player pays and approves it.`);
|
|
23417
|
+
log.plain(` Ask the user, and on their yes: ${c.cyan("npx genex init --convert")} \u2014 then the ${c.cyan("genex-tool-llm")} card.`);
|
|
23396
23418
|
}
|
|
23419
|
+
process.exitCode = 1;
|
|
23420
|
+
return;
|
|
23397
23421
|
}
|
|
23398
23422
|
const meta = await readProject(cwd);
|
|
23399
|
-
if (
|
|
23423
|
+
if (projectBound && !meta?.id) {
|
|
23400
23424
|
fail4(
|
|
23401
|
-
|
|
23402
|
-
"This folder isn't linked to a game \u2014 a benchmark is billed to a project you own."
|
|
23425
|
+
`llm ${sub}`,
|
|
23426
|
+
sub === "bench" ? "This folder isn't linked to a game \u2014 a benchmark is billed to a project you own." : `This folder isn't linked to a game \u2014 \`genex llm ${sub}\` works on the development calls of a project you own.`
|
|
23403
23427
|
);
|
|
23404
23428
|
if (!opts.json) {
|
|
23405
23429
|
log.dim(` Run ${c.cyan("genex link")} (or ${c.cyan("genex list")} to find the slug) first.`);
|
|
@@ -23436,6 +23460,14 @@ async function runLlm(opts = {}) {
|
|
|
23436
23460
|
await reportModels(apiUrl, token, opts, log, toolsOnly);
|
|
23437
23461
|
return;
|
|
23438
23462
|
}
|
|
23463
|
+
if (sub === "status") {
|
|
23464
|
+
await reportStatus({ apiUrl, token, projectId: meta.id, opts, log });
|
|
23465
|
+
return;
|
|
23466
|
+
}
|
|
23467
|
+
if (sub === "cancel") {
|
|
23468
|
+
await cancelCall({ apiUrl, token, projectId: meta.id, id: opts.generationId.trim(), opts, log });
|
|
23469
|
+
return;
|
|
23470
|
+
}
|
|
23439
23471
|
await runBench({ apiUrl, token, projectId: meta.id, cwd, opts, log });
|
|
23440
23472
|
}
|
|
23441
23473
|
async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
@@ -23463,6 +23495,10 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23463
23495
|
recommendedDeclaredHeadroomBps: null,
|
|
23464
23496
|
total: null,
|
|
23465
23497
|
featuredCount: null,
|
|
23498
|
+
// The stand was never asked, so the verdict is unknown — but the key is
|
|
23499
|
+
// present, like every other lane key, so a reader tells "no verdict"
|
|
23500
|
+
// from "a CLI that predates the probe".
|
|
23501
|
+
providerKey: null,
|
|
23466
23502
|
convertHint: toolsOnly ? LLM_TOOLS_CONVERT_HINT.join(" ") : null
|
|
23467
23503
|
});
|
|
23468
23504
|
}
|
|
@@ -23484,6 +23520,8 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23484
23520
|
recommendedDeclaredHeadroomBps: lane.recommendedDeclaredHeadroomBps,
|
|
23485
23521
|
total: lane.total ?? lane.models?.length ?? null,
|
|
23486
23522
|
featuredCount: lane.featuredCount ?? (lane.models ? lane.models.filter((m) => m.featured).length : null),
|
|
23523
|
+
// `null` when the lane is not live or the server predates the probe.
|
|
23524
|
+
providerKey: lane.providerKey,
|
|
23487
23525
|
// Always present, `null` off a tools workspace — the same
|
|
23488
23526
|
// every-key-always-present rule the other keys follow, so a reader can
|
|
23489
23527
|
// tell "no hint" from "a CLI that predates the hint".
|
|
@@ -23537,6 +23575,10 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23537
23575
|
log.plain("");
|
|
23538
23576
|
log.plain(` Personal plans offered here: ${lane.externalProviders.map((p) => p.label).join(", ")}`);
|
|
23539
23577
|
}
|
|
23578
|
+
if (lane.providerKey === "rejected") {
|
|
23579
|
+
log.plain("");
|
|
23580
|
+
log.warn(` ${LLM_PROVIDER_KEY_REJECTED_LINE}`);
|
|
23581
|
+
}
|
|
23540
23582
|
log.plain("");
|
|
23541
23583
|
if (lane.recommendedDeclaredHeadroomBps === null) {
|
|
23542
23584
|
log.dim(" This stand serves no recommended headroom, so no price can be recommended from a bench.");
|
|
@@ -23550,6 +23592,32 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23550
23592
|
log.dim(" Those per-million rates are the PROVIDER's; what your game is charged is the platform");
|
|
23551
23593
|
log.dim(` tariff on top, which only a real call reveals: ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')}`);
|
|
23552
23594
|
}
|
|
23595
|
+
var ACTIVE_STATUSES = /* @__PURE__ */ new Set(["requires_confirmation", "queued", "dispatching", "awaiting_external"]);
|
|
23596
|
+
function rowState(row) {
|
|
23597
|
+
if (row.status && ACTIVE_STATUSES.has(row.status)) return "active";
|
|
23598
|
+
if (row.billingStatus === "pending") return "awaiting bill";
|
|
23599
|
+
return "done";
|
|
23600
|
+
}
|
|
23601
|
+
function rowHoldsSlot(row) {
|
|
23602
|
+
if (typeof row.slotHeld === "boolean") return row.slotHeld;
|
|
23603
|
+
const state = rowState(row);
|
|
23604
|
+
return state === "active" || state === "awaiting bill" && (row.reservedCoins ?? 0) > 0;
|
|
23605
|
+
}
|
|
23606
|
+
function providerRefusalStatus(error) {
|
|
23607
|
+
const m = /^provider_http_(\d{3})$/.exec(error ?? "");
|
|
23608
|
+
return m ? Number(m[1]) : null;
|
|
23609
|
+
}
|
|
23610
|
+
function formatAge(iso, now = Date.now()) {
|
|
23611
|
+
const t = iso ? Date.parse(iso) : Number.NaN;
|
|
23612
|
+
if (!Number.isFinite(t)) return "\u2014";
|
|
23613
|
+
const sec = Math.max(0, Math.floor((now - t) / 1e3));
|
|
23614
|
+
if (sec < 60) return `${sec}s`;
|
|
23615
|
+
const min = Math.floor(sec / 60);
|
|
23616
|
+
if (min < 60) return `${min}m`;
|
|
23617
|
+
const hr = Math.floor(min / 60);
|
|
23618
|
+
if (hr < 48) return `${hr}h`;
|
|
23619
|
+
return `${Math.floor(hr / 24)}d`;
|
|
23620
|
+
}
|
|
23553
23621
|
async function runBench(args) {
|
|
23554
23622
|
const { apiUrl, token, projectId, cwd, opts, log } = args;
|
|
23555
23623
|
const maxCoins = opts.maxCoins;
|
|
@@ -23593,6 +23661,14 @@ async function runBench(args) {
|
|
|
23593
23661
|
benchFailed(opts, log, `This stand does not serve ${modelId}. Run \`genex llm models\` for the ids it does.`);
|
|
23594
23662
|
return;
|
|
23595
23663
|
}
|
|
23664
|
+
if (lane.providerKey === "rejected") {
|
|
23665
|
+
benchFailed(
|
|
23666
|
+
opts,
|
|
23667
|
+
log,
|
|
23668
|
+
`${LLM_PROVIDER_KEY_REJECTED_LINE} A bench against a refused key is ${samples} refusal${samples === 1 ? "" : "s"} and a slot lock-out, not a measurement \u2014 nothing was sent, nothing was spent. Tell the operator; re-run once \`genex llm models\` stops saying so.`
|
|
23669
|
+
);
|
|
23670
|
+
return;
|
|
23671
|
+
}
|
|
23596
23672
|
const balanceBefore = await readCoinBalance(apiUrl, token);
|
|
23597
23673
|
if (!opts.json) {
|
|
23598
23674
|
log.plain(c.bold("genex llm bench"));
|
|
@@ -23608,6 +23684,8 @@ async function runBench(args) {
|
|
|
23608
23684
|
}
|
|
23609
23685
|
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
23610
23686
|
const results = [];
|
|
23687
|
+
let refusal = null;
|
|
23688
|
+
let providerRefused = 0;
|
|
23611
23689
|
let inflight = null;
|
|
23612
23690
|
const onSigint = () => {
|
|
23613
23691
|
const target = inflight;
|
|
@@ -23638,13 +23716,14 @@ async function runBench(args) {
|
|
|
23638
23716
|
})
|
|
23639
23717
|
});
|
|
23640
23718
|
if (!res.ok) {
|
|
23641
|
-
const
|
|
23642
|
-
if (
|
|
23719
|
+
const verdict = await explainBenchRefusal(res, call, base, log, opts, i, results.length);
|
|
23720
|
+
if (verdict.refusal) refusal = verdict.refusal;
|
|
23721
|
+
if (verdict.stop) break;
|
|
23643
23722
|
continue;
|
|
23644
23723
|
}
|
|
23645
23724
|
const started = await res.json().catch(() => ({}));
|
|
23646
23725
|
if (!started.id) {
|
|
23647
|
-
results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id" });
|
|
23726
|
+
results.push({ id: null, status: null, billingStatus: null, chargedCoins: null, costUsd: null, error: "no_generation_id", providerMessage: null });
|
|
23648
23727
|
continue;
|
|
23649
23728
|
}
|
|
23650
23729
|
inflight = started.id;
|
|
@@ -23656,19 +23735,23 @@ async function runBench(args) {
|
|
|
23656
23735
|
billingStatus: settled?.billingStatus ?? null,
|
|
23657
23736
|
chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
|
|
23658
23737
|
costUsd: typeof settled?.usage?.costUsd === "number" ? settled.usage.costUsd : null,
|
|
23659
|
-
error: settled?.error ?? (settled ? null : "timed_out")
|
|
23738
|
+
error: settled?.error ?? (settled ? null : "timed_out"),
|
|
23739
|
+
providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null
|
|
23660
23740
|
};
|
|
23661
23741
|
results.push(row);
|
|
23742
|
+
const refusedAt = providerRefusalStatus(row.error);
|
|
23743
|
+
if (refusedAt !== null) providerRefused++;
|
|
23662
23744
|
if (!opts.json) {
|
|
23663
23745
|
log.plain(
|
|
23664
23746
|
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${row.error ? ` \xB7 ${row.error}` : ""}`
|
|
23665
23747
|
);
|
|
23748
|
+
if (refusedAt !== null) printProviderRefusal(log, refusedAt, row.providerMessage, row.status === "unknown");
|
|
23666
23749
|
}
|
|
23667
23750
|
}
|
|
23668
23751
|
} finally {
|
|
23669
23752
|
process.removeListener("SIGINT", onSigint);
|
|
23670
23753
|
}
|
|
23671
|
-
const charged = results.filter((r) => r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
|
|
23754
|
+
const charged = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
|
|
23672
23755
|
const settledCount = charged.length;
|
|
23673
23756
|
const p50 = percentile2(charged, 50);
|
|
23674
23757
|
const p95 = percentile2(charged, 95);
|
|
@@ -23713,6 +23796,9 @@ async function runBench(args) {
|
|
|
23713
23796
|
balanceBefore,
|
|
23714
23797
|
balanceAfter,
|
|
23715
23798
|
results,
|
|
23799
|
+
// The create-time refusal that ended the run, or null. Beside the
|
|
23800
|
+
// samples, never among them: no attempt existed and nothing was spent.
|
|
23801
|
+
refusal,
|
|
23716
23802
|
chargedCoins: { p50, p95, max },
|
|
23717
23803
|
recommendedDeclaredHeadroomBps: headroomBps,
|
|
23718
23804
|
recommendedEstimateCoins: recommended,
|
|
@@ -23724,7 +23810,9 @@ async function runBench(args) {
|
|
|
23724
23810
|
}
|
|
23725
23811
|
log.plain("");
|
|
23726
23812
|
if (charged.length === 0) {
|
|
23727
|
-
log.error(
|
|
23813
|
+
log.error(
|
|
23814
|
+
providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to price from." : " No sample settled, so there is nothing to price from."
|
|
23815
|
+
);
|
|
23728
23816
|
process.exitCode = 1;
|
|
23729
23817
|
return;
|
|
23730
23818
|
}
|
|
@@ -23758,31 +23846,53 @@ function printRecommendation(log, record) {
|
|
|
23758
23846
|
log.dim(" The ceiling is the price's room to be wrong; declaring one under the price is refused.");
|
|
23759
23847
|
}
|
|
23760
23848
|
}
|
|
23761
|
-
async function explainBenchRefusal(res, log, opts, index, settledSoFar) {
|
|
23762
|
-
if (printedStructuredError(res)) return true;
|
|
23849
|
+
async function explainBenchRefusal(res, call, base, log, opts, index, settledSoFar) {
|
|
23850
|
+
if (printedStructuredError(res)) return { stop: true, refusal: { status: res.status, error: null, slotsHeld: null, slotLimit: null } };
|
|
23763
23851
|
const body = await res.json().catch(() => ({}));
|
|
23764
|
-
|
|
23852
|
+
const refusal = { status: res.status, error: body.error ?? null, slotsHeld: null, slotLimit: null };
|
|
23853
|
+
if (res.status === 429 && body.error === "generation_limit") {
|
|
23854
|
+
const counts = serverSlotCounts(body) ?? await readSlotCounts(call, base);
|
|
23855
|
+
refusal.slotsHeld = counts?.held ?? null;
|
|
23856
|
+
refusal.slotLimit = counts?.limit ?? null;
|
|
23857
|
+
if (!opts.json) log.error(` Sample ${index + 1} ${benchSlotsHeldSentence(counts)}`);
|
|
23858
|
+
return { stop: true, refusal };
|
|
23859
|
+
}
|
|
23860
|
+
if (opts.json) return { stop: res.status !== 429, refusal };
|
|
23765
23861
|
if (res.status === 404 && body.error === "not_found") {
|
|
23766
23862
|
log.error(` ${LLM_LANE_OFF_LINE}`);
|
|
23767
|
-
return true;
|
|
23863
|
+
return { stop: true, refusal };
|
|
23768
23864
|
}
|
|
23769
23865
|
if (res.status === 403 && body.error === "credential_scope") {
|
|
23770
23866
|
log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
|
|
23771
23867
|
log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
|
|
23772
|
-
return true;
|
|
23868
|
+
return { stop: true, refusal };
|
|
23773
23869
|
}
|
|
23774
23870
|
if (res.status === 403 && body.error === "project_owner_required") {
|
|
23775
23871
|
log.error(" A benchmark is billed to a project YOU own, and this one isn't yours.");
|
|
23776
|
-
return true;
|
|
23872
|
+
return { stop: true, refusal };
|
|
23777
23873
|
}
|
|
23778
23874
|
if (res.status === 402) {
|
|
23779
23875
|
log.error(" Not enough coin to start the attempt.");
|
|
23780
|
-
return true;
|
|
23876
|
+
return { stop: true, refusal };
|
|
23781
23877
|
}
|
|
23782
23878
|
log.error(
|
|
23783
23879
|
` Sample ${index + 1} was refused: ${body.message ?? body.error ?? `HTTP ${res.status}`}${settledSoFar > 0 ? " (earlier samples still count)" : ""}`
|
|
23784
23880
|
);
|
|
23785
|
-
return res.status >= 500 || res.status === 401 || res.status === 403;
|
|
23881
|
+
return { stop: res.status >= 500 || res.status === 401 || res.status === 403, refusal };
|
|
23882
|
+
}
|
|
23883
|
+
function printProviderRefusal(log, status2, message, awaitingBill = false) {
|
|
23884
|
+
if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so this call's coin stays held until it resolves (see \`genex llm status\`); it is not a sample.`);
|
|
23885
|
+
else log.dim(" The provider refused this call at its door \u2014 no inference ran, it cost nothing, and it is not a sample.");
|
|
23886
|
+
if (message) log.dim(` Provider said: ${message}`);
|
|
23887
|
+
if (status2 === 401 || status2 === 403) {
|
|
23888
|
+
log.dim(` A ${status2} here is this stand's provider configuration refusing the model \u2014 a message for the operator, not something to fix in the game.`);
|
|
23889
|
+
}
|
|
23890
|
+
}
|
|
23891
|
+
async function readSlotCounts(call, base) {
|
|
23892
|
+
const res = await call(`${base}?scope=open`).catch(() => null);
|
|
23893
|
+
if (!res?.ok) return null;
|
|
23894
|
+
const body = await res.json().catch(() => null);
|
|
23895
|
+
return serverSlotCounts(body);
|
|
23786
23896
|
}
|
|
23787
23897
|
async function pollSettled(call, url, timeoutSec) {
|
|
23788
23898
|
const deadline = Date.now() + timeoutSec * 1e3;
|
|
@@ -23840,6 +23950,7 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
|
|
|
23840
23950
|
balanceBefore: null,
|
|
23841
23951
|
balanceAfter: null,
|
|
23842
23952
|
results: [],
|
|
23953
|
+
refusal: null,
|
|
23843
23954
|
chargedCoins: { p50: null, p95: null, max: null },
|
|
23844
23955
|
recommendedDeclaredHeadroomBps: null,
|
|
23845
23956
|
recommendedEstimateCoins: null,
|
|
@@ -23847,6 +23958,166 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
|
|
|
23847
23958
|
savedTo: null
|
|
23848
23959
|
};
|
|
23849
23960
|
}
|
|
23961
|
+
async function reportStatus(args) {
|
|
23962
|
+
const { apiUrl, token, projectId, opts, log } = args;
|
|
23963
|
+
const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
|
|
23964
|
+
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
23965
|
+
const res = await call(`${base}?scope=open`).catch(() => null);
|
|
23966
|
+
if (!res || !res.ok) {
|
|
23967
|
+
if (res && printedStructuredError(res)) {
|
|
23968
|
+
if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", `http_${res.status}`, [], null, null));
|
|
23969
|
+
process.exitCode = 1;
|
|
23970
|
+
return;
|
|
23971
|
+
}
|
|
23972
|
+
const body2 = res ? await res.json().catch(() => ({})) : {};
|
|
23973
|
+
const error = body2.error ?? (res ? `http_${res.status}` : "unreachable");
|
|
23974
|
+
if (opts.json) {
|
|
23975
|
+
writeJsonLine(statusJson(apiUrl, projectId, res?.status === 404 && body2.error === "not_found" ? "off" : "failed", error, [], null, null));
|
|
23976
|
+
if (!(res?.status === 404 && body2.error === "not_found")) process.exitCode = 1;
|
|
23977
|
+
return;
|
|
23978
|
+
}
|
|
23979
|
+
if (res?.status === 404 && body2.error === "not_found") {
|
|
23980
|
+
log.plain(LLM_LANE_OFF_LINE);
|
|
23981
|
+
return;
|
|
23982
|
+
}
|
|
23983
|
+
if (res?.status === 403 && body2.error === "credential_scope") {
|
|
23984
|
+
log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
|
|
23985
|
+
log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
|
|
23986
|
+
} else if (res?.status === 403 && body2.error === "project_owner_required") {
|
|
23987
|
+
log.error(" Development calls belong to a project YOU own, and this one isn't yours.");
|
|
23988
|
+
} else if (!res) {
|
|
23989
|
+
log.error(` Couldn't reach ${apiUrl}.`);
|
|
23990
|
+
log.dim(` Check the network, then ${c.cyan("genex doctor")}.`);
|
|
23991
|
+
} else {
|
|
23992
|
+
log.error(` Couldn't read the open calls: ${error}.`);
|
|
23993
|
+
}
|
|
23994
|
+
process.exitCode = 1;
|
|
23995
|
+
return;
|
|
23996
|
+
}
|
|
23997
|
+
const body = await res.json().catch(() => null);
|
|
23998
|
+
if (!body || !Array.isArray(body.generations)) {
|
|
23999
|
+
if (opts.json) writeJsonLine(statusJson(apiUrl, projectId, "failed", "unreadable_response", [], null, null));
|
|
24000
|
+
else log.error(` ${apiUrl} answered, but not with a list of calls.`);
|
|
24001
|
+
process.exitCode = 1;
|
|
24002
|
+
return;
|
|
24003
|
+
}
|
|
24004
|
+
const rows = body.generations;
|
|
24005
|
+
const counts = serverSlotCounts(body);
|
|
24006
|
+
if (opts.json) {
|
|
24007
|
+
writeJsonLine(statusJson(apiUrl, projectId, "ok", null, rows, counts?.held ?? null, counts?.limit ?? null));
|
|
24008
|
+
return;
|
|
24009
|
+
}
|
|
24010
|
+
log.plain(c.bold("genex llm status"));
|
|
24011
|
+
log.dim(` ${apiUrl} \xB7 project ${projectId}`);
|
|
24012
|
+
log.plain("");
|
|
24013
|
+
if (rows.length === 0) {
|
|
24014
|
+
log.plain(" No open calls in this project \u2014 nothing active, nothing awaiting a bill.");
|
|
24015
|
+
} else {
|
|
24016
|
+
const now = Date.now();
|
|
24017
|
+
for (const row of rows) {
|
|
24018
|
+
const state = rowState(row);
|
|
24019
|
+
const mark = state === "active" ? c.cyan("\u25CF") : state === "awaiting bill" ? c.yellow("\u25D0") : c.green("\u25CB");
|
|
24020
|
+
const reserved = typeof row.reservedCoins === "number" ? `${row.reservedCoins} coin reserved` : "reserved \u2014";
|
|
24021
|
+
log.plain(
|
|
24022
|
+
` ${mark} ${row.id ?? "?"} ${row.modelId ?? "?"} ${state.padEnd(13)} ${formatAge(row.createdAt, now).padStart(4)} old ${reserved}${row.error ? ` ${row.error}` : ""}${rowHoldsSlot(row) ? "" : c.dim(" (no slot)")}`
|
|
24023
|
+
);
|
|
24024
|
+
if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
|
|
24025
|
+
}
|
|
24026
|
+
}
|
|
24027
|
+
log.plain("");
|
|
24028
|
+
log.dim(` ${LLM_STATUS_LEGEND}`);
|
|
24029
|
+
if (counts === null) {
|
|
24030
|
+
log.dim(` ${LLM_STATUS_NO_COUNT_LINE}`);
|
|
24031
|
+
return;
|
|
24032
|
+
}
|
|
24033
|
+
log.plain(` ${counts.held} of ${counts.limit} slots held${counts.held > rows.filter(rowHoldsSlot).length ? " \u2014 across every project, this one's rows included" : ""}`);
|
|
24034
|
+
if (counts.held >= counts.limit) {
|
|
24035
|
+
log.dim(` Every slot is held, so a new ad-hoc call answers generation_limit until one frees: ${c.cyan("genex llm cancel <id>")} stops an active one; a stopped one frees on its own.`);
|
|
24036
|
+
}
|
|
24037
|
+
}
|
|
24038
|
+
function statusJson(apiUrl, projectId, status2, error, rows, slotsHeld, slotLimit) {
|
|
24039
|
+
return {
|
|
24040
|
+
command: "llm status",
|
|
24041
|
+
status: status2,
|
|
24042
|
+
error,
|
|
24043
|
+
apiUrl,
|
|
24044
|
+
projectId,
|
|
24045
|
+
// Every row verbatim, plus the CLI's own reading of it — so a program
|
|
24046
|
+
// gets both the server's fields and the three words a person sees.
|
|
24047
|
+
generations: rows.map((row) => ({ ...row, state: rowState(row), holdsSlot: rowHoldsSlot(row) })),
|
|
24048
|
+
slotsHeld,
|
|
24049
|
+
slotLimit,
|
|
24050
|
+
pendingSlotReleaseMinutes: PENDING_SLOT_RELEASE_MINUTES
|
|
24051
|
+
};
|
|
24052
|
+
}
|
|
24053
|
+
async function cancelCall(args) {
|
|
24054
|
+
const { apiUrl, token, projectId, id, opts, log } = args;
|
|
24055
|
+
const call = (url, init2) => apiFetch(url, { ...init2, headers: { Authorization: `Bearer ${token}`, ...init2?.headers ?? {} } });
|
|
24056
|
+
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
24057
|
+
const rowUrl = `${base}/${encodeURIComponent(id)}`;
|
|
24058
|
+
const done = (status2, error, reason, row2) => {
|
|
24059
|
+
if (opts.json) writeJsonLine({ command: "llm cancel", status: status2, error, id, cancelled: reason === "cancelled", reason, generation: row2 });
|
|
24060
|
+
if (status2 === "failed") process.exitCode = 1;
|
|
24061
|
+
};
|
|
24062
|
+
const read = await call(rowUrl).catch(() => null);
|
|
24063
|
+
if (!read || !read.ok) {
|
|
24064
|
+
if (read && printedStructuredError(read)) return done("failed", `http_${read.status}`, null, null);
|
|
24065
|
+
const body = read ? await read.json().catch(() => ({})) : {};
|
|
24066
|
+
const error = body.error ?? (read ? `http_${read.status}` : "unreachable");
|
|
24067
|
+
if (!opts.json) {
|
|
24068
|
+
if (read?.status === 404) {
|
|
24069
|
+
log.error(` No call ${id} in this project \u2014 ${c.cyan("genex llm status")} lists the open ones.`);
|
|
24070
|
+
} else if (read?.status === 403 && body.error === "credential_scope") {
|
|
24071
|
+
log.error(" This credential can't reach the runtime lane \u2014 an API key or MCP token is scoped out.");
|
|
24072
|
+
log.dim(` Sign in with ${c.cyan("genex auth")} and re-run.`);
|
|
24073
|
+
} else if (!read) {
|
|
24074
|
+
log.error(` Couldn't reach ${apiUrl}.`);
|
|
24075
|
+
} else {
|
|
24076
|
+
log.error(` Couldn't read call ${id}: ${error}.`);
|
|
24077
|
+
}
|
|
24078
|
+
}
|
|
24079
|
+
return done("failed", error, null, null);
|
|
24080
|
+
}
|
|
24081
|
+
const row = await read.json().catch(() => null);
|
|
24082
|
+
if (!row?.id) {
|
|
24083
|
+
if (!opts.json) log.error(` ${apiUrl} answered, but not with a call.`);
|
|
24084
|
+
return done("failed", "unreadable_response", null, null);
|
|
24085
|
+
}
|
|
24086
|
+
const state = rowState(row);
|
|
24087
|
+
if (state !== "active") {
|
|
24088
|
+
const sentence = state === "awaiting bill" ? LLM_CANCEL_AWAITING_BILL : LLM_CANCEL_ALREADY_FINAL;
|
|
24089
|
+
if (!opts.json) {
|
|
24090
|
+
log.warn(` ${row.id} (${row.status ?? "?"}) \u2014 ${sentence}.`);
|
|
24091
|
+
if (state === "awaiting bill") {
|
|
24092
|
+
log.dim(` Nothing to cancel: its coin releases when the bill resolves, and it stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`);
|
|
24093
|
+
if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
|
|
24094
|
+
} else {
|
|
24095
|
+
log.dim(` ${typeof row.chargedCoins === "number" ? `${row.chargedCoins} coin charged` : "charge unknown"}${row.error ? ` \xB7 ${row.error}` : ""}.`);
|
|
24096
|
+
}
|
|
24097
|
+
}
|
|
24098
|
+
return done("ok", null, state === "awaiting bill" ? "awaiting_bill" : "already_final", row);
|
|
24099
|
+
}
|
|
24100
|
+
const res = await call(`${rowUrl}/cancel`, { method: "POST" }).catch(() => null);
|
|
24101
|
+
if (!res || !res.ok) {
|
|
24102
|
+
if (res && printedStructuredError(res)) return done("failed", `http_${res.status}`, null, row);
|
|
24103
|
+
const body = res ? await res.json().catch(() => ({})) : {};
|
|
24104
|
+
const error = body.error ?? (res ? `http_${res.status}` : "unreachable");
|
|
24105
|
+
if (!opts.json) {
|
|
24106
|
+
if (res?.status === 409 && body.error === "already_dispatched") {
|
|
24107
|
+
log.error(` ${row.id} is already running at the provider and cannot be cancelled now; it settles on its own \u2014 ${c.cyan("genex llm status")} shows it.`);
|
|
24108
|
+
} else {
|
|
24109
|
+
log.error(` Couldn't cancel ${row.id}: ${error}.`);
|
|
24110
|
+
}
|
|
24111
|
+
}
|
|
24112
|
+
return done("failed", error, null, row);
|
|
24113
|
+
}
|
|
24114
|
+
const after = await res.json().catch(() => null) ?? row;
|
|
24115
|
+
if (!opts.json) {
|
|
24116
|
+
log.success(` Cancelled ${row.id} (was ${row.status ?? "active"}).`);
|
|
24117
|
+
log.dim(` Its coin releases once the bill settles; ${c.cyan("genex llm status")} shows it until then.`);
|
|
24118
|
+
}
|
|
24119
|
+
return done("ok", null, "cancelled", after);
|
|
24120
|
+
}
|
|
23850
24121
|
async function reportSavedPrice(cwd, opts, log) {
|
|
23851
24122
|
let record = null;
|
|
23852
24123
|
try {
|
|
@@ -24159,10 +24430,11 @@ function runtimeRow(state, lane, toolsOnly = false) {
|
|
|
24159
24430
|
switch (lane?.state) {
|
|
24160
24431
|
case "live": {
|
|
24161
24432
|
const n = lane.models?.length ?? 0;
|
|
24433
|
+
const keyRefused = lane.providerKey === "rejected";
|
|
24162
24434
|
return {
|
|
24163
24435
|
label: "Runtime",
|
|
24164
|
-
value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}`,
|
|
24165
|
-
fix: toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
|
|
24436
|
+
value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}${keyRefused ? ` \xB7 ${c.yellow("provider key refused")}` : ""}`,
|
|
24437
|
+
fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
|
|
24166
24438
|
};
|
|
24167
24439
|
}
|
|
24168
24440
|
case "off":
|
|
@@ -26096,17 +26368,20 @@ ${c.bold("Usage")}
|
|
|
26096
26368
|
the asset table against it. --assets <credits>
|
|
26097
26369
|
--user-approved raises it, only once the player
|
|
26098
26370
|
has agreed to the number.
|
|
26099
|
-
genex llm <sub> [options] In-game model calls: models | bench | price
|
|
26100
|
-
"models" says whether this
|
|
26101
|
-
runtime LLM lane at all, and
|
|
26102
|
-
rates. "bench" runs the real
|
|
26103
|
-
coin and reports what each
|
|
26104
|
-
it needs --max-coins <n>
|
|
26105
|
-
any spend the player has
|
|
26106
|
-
reprints the last run's
|
|
26107
|
-
estimateCoins.
|
|
26108
|
-
|
|
26109
|
-
|
|
26371
|
+
genex llm <sub> [options] In-game model calls: models | bench | price |
|
|
26372
|
+
status | cancel. "models" says whether this
|
|
26373
|
+
stand serves the runtime LLM lane at all, and
|
|
26374
|
+
at what provider rates. "bench" runs the real
|
|
26375
|
+
model on YOUR OWN coin and reports what each
|
|
26376
|
+
attempt was charged; it needs --max-coins <n>
|
|
26377
|
+
--user-approved, like any spend the player has
|
|
26378
|
+
to agree to. "price" reprints the last run's
|
|
26379
|
+
recommended estimateCoins. "status" lists this
|
|
26380
|
+
project's open calls and the slots they hold;
|
|
26381
|
+
"cancel <id>" stops an active one. Declare a
|
|
26382
|
+
game's price from a bench, never from a guess:
|
|
26383
|
+
a started attempt is charged in full, including
|
|
26384
|
+
one that fails.
|
|
26110
26385
|
genex player <sub> [options] Answer the in-game model requests you approve, on your
|
|
26111
26386
|
OWN Claude or ChatGPT subscription: install | run |
|
|
26112
26387
|
status | stop | uninstall. "install" signs this machine
|
|
@@ -26543,6 +26818,8 @@ ${c.bold("Examples")}
|
|
|
26543
26818
|
genex llm models --all
|
|
26544
26819
|
genex llm bench "Reply with one short taunt." --samples 5 --max-coins 20 --user-approved
|
|
26545
26820
|
genex llm price
|
|
26821
|
+
genex llm status
|
|
26822
|
+
genex llm cancel <id>
|
|
26546
26823
|
genex player install
|
|
26547
26824
|
genex player status --json
|
|
26548
26825
|
genex player uninstall
|
|
@@ -26794,6 +27071,14 @@ ${c.bold("Usage")}
|
|
|
26794
27071
|
cannot see. Needs a folder linked to a game you own.
|
|
26795
27072
|
genex llm price [--json] The recommendation from the last bench in this folder.
|
|
26796
27073
|
Reads a file; spends nothing.
|
|
27074
|
+
genex llm status [--json] This project's open development calls: active, or
|
|
27075
|
+
stopped but awaiting the provider's bill \u2014 with the
|
|
27076
|
+
provider's own message when it refused one \u2014 and
|
|
27077
|
+
how many of the ad-hoc slots they hold. A bench
|
|
27078
|
+
refused as generation_limit sends you here.
|
|
27079
|
+
genex llm cancel <id> [--json] Stop one ACTIVE call. A call that already stopped
|
|
27080
|
+
is reported as such: awaiting its bill (the hold
|
|
27081
|
+
resolves on its own) or already final.
|
|
26797
27082
|
|
|
26798
27083
|
${c.bold("Bench options")}
|
|
26799
27084
|
--model <id> One of the ids \`genex llm models\` prints (default: the first one).
|
|
@@ -27303,7 +27588,11 @@ function parseArgs(argv) {
|
|
|
27303
27588
|
(parsed.options.selectors ??= []).push(arg);
|
|
27304
27589
|
}
|
|
27305
27590
|
} else if (parsed.command === "llm") {
|
|
27306
|
-
parsed.options.
|
|
27591
|
+
if (parsed.options.name === "cancel" && parsed.options.generationId === void 0) {
|
|
27592
|
+
parsed.options.generationId = arg;
|
|
27593
|
+
} else {
|
|
27594
|
+
parsed.options.benchPrompt = parsed.options.benchPrompt ? `${parsed.options.benchPrompt} ${arg}` : arg;
|
|
27595
|
+
}
|
|
27307
27596
|
} else if (parsed.command === "shop") {
|
|
27308
27597
|
if (!parsed.options.hostname) parsed.options.hostname = arg;
|
|
27309
27598
|
else {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@genex-ai/cli-demo",
|
|
3
|
-
"version": "1.35.0-dev.
|
|
3
|
+
"version": "1.35.0-dev.744",
|
|
4
4
|
"description": "Set up your project's agent workspace (.claude/.codex/.cursor in the game folder), authorize, create a game project, generate AI assets, and publish (genex CLI).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -120,6 +120,41 @@ the price and makes `requestSpendGrant()` refuse before it reaches the network.
|
|
|
120
120
|
loop into disclosure numbers, re-benchmarking after a prompt change — is in
|
|
121
121
|
[references/pricing.md](references/pricing.md).
|
|
122
122
|
|
|
123
|
+
Only samples that **succeeded and settled** are priced from. A sample the
|
|
124
|
+
provider refused at its door (`provider_http_<status>`) ran no inference and
|
|
125
|
+
cost nothing; the bench prints the code, the provider's own message and, for a
|
|
126
|
+
401 or 403, that this is the stand's provider configuration refusing the model
|
|
127
|
+
— an operator's problem, never something to fix in the game. A sample refused
|
|
128
|
+
as `generation_limit` never started: you already have the lane's three ad-hoc
|
|
129
|
+
calls open, or recently stopped with their bill still pending. The bench says
|
|
130
|
+
how many slots are held and stops. Do not re-run it into the same refusal —
|
|
131
|
+
read `npx genex llm status`, then wait for a bill to resolve or
|
|
132
|
+
`npx genex llm cancel <id>` an active call.
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
npx genex llm status # this project's open calls: active, or awaiting their bill; N of 3 slots held
|
|
136
|
+
npx genex llm cancel <id> # stop an ACTIVE call; a stopped one is reported, not cancelled
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
## The schema dialect (exact — anything else is refused)
|
|
140
|
+
|
|
141
|
+
`schema` is validated by the platform before the call, and a refused schema is
|
|
142
|
+
`invalid_schema` at the door — nothing is charged, nothing runs. The accepted
|
|
143
|
+
subset, and it is the whole subset:
|
|
144
|
+
|
|
145
|
+
- `type`: `object`, `array`, `string`, `number`, `integer`, `boolean`, `null`
|
|
146
|
+
- objects: `properties`, `required`, and `additionalProperties: false` on
|
|
147
|
+
**every** object (required, not optional)
|
|
148
|
+
- arrays: `items`
|
|
149
|
+
- `enum` (strings, numbers, booleans, `null`)
|
|
150
|
+
- `minimum` / `maximum`, `minLength` / `maxLength`, `minItems` / `maxItems`
|
|
151
|
+
- `description` and `title`, on any node
|
|
152
|
+
|
|
153
|
+
Everything else is refused, including `$schema`, `default`, `examples`,
|
|
154
|
+
`pattern`, `format`, `anyOf` / `oneOf` / `allOf` and `$ref`. Keep the schema
|
|
155
|
+
in one file (`./answer.schema.json`), benchmark with that file, and ship the
|
|
156
|
+
same object — a schema that passed the bench passes the game.
|
|
157
|
+
|
|
123
158
|
## Standing budgets
|
|
124
159
|
|
|
125
160
|
```ts
|
|
@@ -195,7 +230,9 @@ There is no compensation lane, so this is all work you do before the call:
|
|
|
195
230
|
|
|
196
231
|
- **Validate inputs first** — a malformed prompt is still charged.
|
|
197
232
|
- **Always set `schema` for `outputFormat: 'json'`** — unschema'd JSON is the
|
|
198
|
-
commonest way a call is charged and the result is unusable.
|
|
233
|
+
commonest way a call is charged and the result is unusable. Write it in the
|
|
234
|
+
accepted dialect above; a refused schema is `invalid_schema` and costs
|
|
235
|
+
nothing, but it is a feature that never runs.
|
|
199
236
|
- **Keep prompts short.** Long context is the price.
|
|
200
237
|
- **Never loop `generate()` without a grant**, and never retry in a loop — each
|
|
201
238
|
attempt is a separate charge.
|
|
@@ -236,6 +273,8 @@ These are source contracts, not a claim that every stand runs this lane —
|
|
|
236
273
|
- [ ] Model ids come from `getGenerationModels()`, never from source
|
|
237
274
|
- [ ] The picker offers the featured set, labelled with the server's own `label`
|
|
238
275
|
- [ ] `estimateCoins` came from `npx genex llm bench`, not from judgement
|
|
276
|
+
- [ ] The schema uses only the accepted dialect (no `$ref`, `pattern`, `format`, `anyOf`, `default`)
|
|
277
|
+
- [ ] A bench refused as `generation_limit` was answered with `npx genex llm status`, never a retry
|
|
239
278
|
- [ ] `generate()` / `requestSpendGrant()` is the first statement of a click handler
|
|
240
279
|
- [ ] A repeated-call feature uses a grant; a one-off uses `generate()`
|
|
241
280
|
- [ ] Disclosure numbers derive from the real loop and are written in `DESIGN.md`
|
|
@@ -269,6 +308,25 @@ cause rather than showing it for both.
|
|
|
269
308
|
**`grant_price_unreasonable`** — the declared per-call price is far above what
|
|
270
309
|
that prompt can cost on that model. Re-benchmark and declare what it prints.
|
|
271
310
|
|
|
311
|
+
**`generation_limit`** — three ad-hoc calls are already in flight for this
|
|
312
|
+
account, or recently stopped with their bill still pending; a pending call
|
|
313
|
+
stops counting ten minutes after it was dispatched. From the bench: run
|
|
314
|
+
`npx genex llm status`, then wait or `npx genex llm cancel <id>` an active one —
|
|
315
|
+
never re-run into the same refusal. In the game: the player is clicking faster
|
|
316
|
+
than one-off calls are meant for, which is the signal that this feature wants a
|
|
317
|
+
grant.
|
|
318
|
+
|
|
319
|
+
**`provider_http_<status>`** — the provider refused the call at its door, before
|
|
320
|
+
any inference: it cost nothing and is not a sample. Read the provider's own
|
|
321
|
+
message (`npx genex llm status` prints it under the row). A 401 or 403 is this
|
|
322
|
+
stand's provider configuration refusing the model — tell the operator, and
|
|
323
|
+
build nothing around it in the game.
|
|
324
|
+
|
|
325
|
+
**`invalid_schema`** — the schema uses a keyword outside the accepted dialect
|
|
326
|
+
(`$ref`, `pattern`, `format`, `anyOf`, `default`, `examples`, `$schema`), or an
|
|
327
|
+
object without `additionalProperties: false`. Nothing was charged. Rewrite it in
|
|
328
|
+
the subset above; `description` and `title` are allowed.
|
|
329
|
+
|
|
272
330
|
**`grant_concurrency` / `grant_rate_limited`** — the game calls faster than the
|
|
273
331
|
grant's own limits. Batch and cache; do not raise the limits to hide it.
|
|
274
332
|
|
|
@@ -49,7 +49,19 @@ npx genex llm bench "<the frozen prompt, one real example filled in>" \
|
|
|
49
49
|
## 3. Read the output
|
|
50
50
|
|
|
51
51
|
Each sample prints what it actually charged. The aggregate prints p50, p95 and
|
|
52
|
-
max of the charged coins
|
|
52
|
+
max of the charged coins **over the samples that succeeded and settled**, plus
|
|
53
|
+
one recommendation. A sample that failed is printed with its code and stays
|
|
54
|
+
out of the numbers; a sample the provider refused at its door
|
|
55
|
+
(`provider_http_<status>`) ran no inference, cost nothing, and prints the
|
|
56
|
+
provider's own message under its row — on a 401 or 403 that is the stand's
|
|
57
|
+
provider configuration refusing the model, which is the operator's to fix. A
|
|
58
|
+
sample refused as `generation_limit` never started and ends the run: the
|
|
59
|
+
account's three ad-hoc calls are open or recently stopped with a pending
|
|
60
|
+
bill. `npx genex llm status` lists them with what each holds;
|
|
61
|
+
`npx genex llm cancel <id>` stops an active one; a stopped one frees on its
|
|
62
|
+
own once its bill resolves, and stops holding a slot ten minutes after
|
|
63
|
+
dispatch. Re-running the bench into the same refusal spends nothing and
|
|
64
|
+
learns nothing.
|
|
53
65
|
|
|
54
66
|
- **p50** is what a typical call costs. It is the number to reason about when
|
|
55
67
|
you ask "can the game afford this loop?" — multiply it by the calls per
|