@genex-ai/cli-demo 1.36.3-dev.782 → 1.36.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{blender-mcp-WSURPZET.js → blender-mcp-QVEGHE5L.js} +2 -2
- package/dist/{blender-serve-EH2MOW3U.js → blender-serve-66OHC4ML.js} +1 -1
- package/dist/{chunk-EABIVAGT.js → chunk-4BBTPRJW.js} +2 -2
- package/dist/{chunk-K4EHXOZK.js → chunk-QI7FIYBY.js} +1 -1
- package/dist/index.js +216 -448
- package/package.json +1 -1
- package/templates/skills/genex-game-director/SKILL.md +1 -1
- package/templates/skills/genex-getting-started/SKILL.md +2 -2
- package/templates/skills/genex-llm-in-games/SKILL.md +72 -209
- package/templates/skills/genex-llm-in-games/references/pricing.md +85 -126
- package/templates/skills/genex-tool-llm/SKILL.md +9 -11
- package/templates/skills/genex-updates/SKILL.md +1 -1
package/dist/index.js
CHANGED
|
@@ -57,7 +57,7 @@ import {
|
|
|
57
57
|
writeSecretFile,
|
|
58
58
|
writeUserToken,
|
|
59
59
|
writeWorkspace
|
|
60
|
-
} from "./chunk-
|
|
60
|
+
} from "./chunk-4BBTPRJW.js";
|
|
61
61
|
import {
|
|
62
62
|
CLI_CHANNEL,
|
|
63
63
|
DEFAULT_API_URL,
|
|
@@ -78,7 +78,7 @@ import {
|
|
|
78
78
|
normalizeApiOrigin,
|
|
79
79
|
originKey,
|
|
80
80
|
resolveAgentTargets
|
|
81
|
-
} from "./chunk-
|
|
81
|
+
} from "./chunk-QI7FIYBY.js";
|
|
82
82
|
|
|
83
83
|
// src/instrument.ts
|
|
84
84
|
import * as Sentry from "@sentry/node";
|
|
@@ -981,7 +981,7 @@ Important note: put soul into your creations, with many details and love. Aim to
|
|
|
981
981
|
21. A turn ends in exactly one of two ways: on a question the player must answer before the next step can be chosen, or on a handoff \u2014 the draft link, what changed, what to try, and what is still open. Never close a turn on a promise. "I'll keep building", "while I keep working", "adding that now" are things you say and then DO before you stop; if you are stopping, say that you have stopped and what remains. While the Build plan's \`Now:\` line has open work and no answer is needed from the player, the turn is not over: preview, hand off, and start the next milestone in the same turn, until the requested outcome is reached or the platform's own budget ends the session. If a stop was forced on you anyway, the first line of your next turn is the promise you left and where it stands.
|
|
982
982
|
22. **What the player sees must read as the thing it is**, in this game's own style: a building as that building, a person as a person, a prop as that prop, a surface as its material \u2014 and a coloured box, a capsule, or a flat grey block standing in for one of them is never its finished version. Generation is the default route to that bar wherever code will not honestly reach it: the player's character, whatever the request names, whatever the player walks up to, enters, or interacts with, and the music and sound underneath \u2014 reach for the belt on your own judgment, without stopping to ask; a status line saying what you queued is enough, and \`--no-wait\` keeps you building while it lands. Procedural code is a first-class engine for what is structural, repeated, distant, or parametric \u2014 terrain, sky, fences, paving, modular kits, filler \u2014 and for anything else only when the result meets the same bar and you have looked at it in a capture before its row says \`landed\`. Start with one reusable implementation per distinct required gameplay or visual role. Background populations default to reusable procedural bodies and motion or compatible existing rigged assets. Catalog clips need a compatible skeleton and may be paid. Reserve paid custom bodies for the player and characters examined or interacted with closely, while honoring explicit user requests. Add variants only for an unmet requirement; unspent allowance is not unfinished work. Allowance questions use timeoutPolicy no-consent: silence does not raise the allowance. Record role, reuse and completion criteria in the existing Assets table. Decide per object, not per category, and write the route on each Assets row in \`DESIGN.md\` \u2014 the paid flow for generated pieces, the procedural flow with its local path for code-built ones. An Assets table with no generation in it is a decision, not a default: record it as \`Generation: none \u2014 <why code alone reaches the bar here>\` or generate.
|
|
983
983
|
23. **The screenshots you take to look at your own work go in \`.genex/scratch/\`.** Captures, render comparisons, before/after strips, traces and metrics dumps are how you SEE what you built \u2014 they are not part of what you built, and \`genex preview\` pushes the whole folder to the game's source, every time, forever. \`.genex/scratch/\` already exists for this and is already ignored, so there is nothing to set up: write the capture as \`.genex/scratch/arena-before.png\` and read it back from there. Never put them in a folder of your own at the top level: \`progress/\`, \`reports/\`, \`shots/\` and their kind are pushed like source and become permanent weight in the game and in every remix of it \u2014 one real game reached 11 GB and 10,332 committed screenshots exactly this way. If you find such a folder already there, add it to \`.gitignore\`; never delete the player's files (law 18). One file in \`.genex/scratch/\` IS read, by the platform: \`.genex/scratch/cover.png\` is the game's cover \u2014 the frame you would put on its poster: real gameplay at its signature moment, well lit, no menus, popups or busy HUD, landscape 16:9, and never any text (the page sets the title beside it). The next \`genex preview\` makes it the cover; save a better moment over it whenever one comes along, and if your browser tool can set the cover itself, use that instead. Before the first publish, and whenever the game's look changes, load \`$genex-cover\`: it chooses, stages, lights and judges that frame.
|
|
984
|
-
24. **Benchmark before declaring an in-game LLM
|
|
984
|
+
24. **Benchmark before declaring an in-game LLM price.** If the game calls a language model while the player plays \u2014 an NPC answering in its own words, a quest written for this save, a judge reading what the player typed \u2014 the PLAYER pays for it, and \`estimateCoins\` is the FIXED price of a started attempt: charged whether it succeeds, fails, is canceled, or stops at its own budget. Never pick that number by judgement or from a vendor's rate card. Run the real prompt on your own coins \u2014 \`npx genex llm bench "<the prompt>" --schema <file> --samples 3 --max-coins <n> --user-approved\` \u2014 and declare the figure it recommends, recorded in \`DESIGN.md\` with its model and date; re-benchmark whenever the prompt, the schema or the model changes, because a prompt edit is a price change. Check the lane first with \`npx genex llm models\`: where it is off the routes 404, so build the feature behind a graceful unavailable state and say so in the handoff. A repeated call is a standing budget the player approves once (\`requestSpendGrant()\`, then \`generate({ grantId })\`), never a popup per turn, and its disclosed call rate is computed from the game's own loop. Model output is data the game validates against its own expectation \u2014 never executed, and never authority over coin, items, entitlements or rewards. Load \`$genex-llm-in-games\` before writing any of it.
|
|
985
985
|
${CONTRACT_END}
|
|
986
986
|
`;
|
|
987
987
|
var TOOLS_CONTRACT_HEAD = `# Genex Tools (always in effect in this folder)
|
|
@@ -995,7 +995,7 @@ var TOOLS_CONTRACT_OFFER = `5. **When the game is built and playable \u2014 neve
|
|
|
995
995
|
var TOOLS_CONTRACT_HOSTED_LAWS = `5. **This game is hosted on Genex now, and publishing is glue \u2014 never a rebuild.** The game ships exactly as the user built it: \`initEmbed()\` first in the boot code (the \`genex-threejs-embed-auth\` card has the call; the renderer draws before any \`await waitForPlayer()\`), a static build into \`dist/\` with relative asset paths, then \`npx genex preview\`. Never restructure, reformat or "improve" the game to ship it, never scaffold a new app around it, and never start a design document or a build plan for it \u2014 the user's own process is theirs. Load the \`genex-tool-publish\` card before the first preview; it owns the vocabulary, the links and the limits.
|
|
996
996
|
6. **Ship first, then report.** \`preview\` prints a preflight \u2014 phone memory, a missing volume slider, an asset nobody wired, a viewport line. It never blocks a deploy, so push the build as it is, hand over the link, THEN relay each preflight line to the user in one plain sentence with an offer to fix it, and fix one only on their yes. The preflight is a report for the user, not a to-do list for you.
|
|
997
997
|
7. **The link is the game's page, and there are two versions.** After every preview give the user \`<dashboard>/draft/<slug>\` \u2014 \`dashboardOrigins[0]\` and \`slug\` from \`.genex/project.json\` \u2014 never localhost, a file path or the bare play origin. \`preview\` updates the draft and never touches what players are on; the first release is \`npx genex publish\` (it lists the game); after that "publish it", "update it" and "yes" all mean \`npx genex promote\` \u2014 the exact draft build, no rebuild. Ask once per round of work, in one line, and keep working while you wait.`;
|
|
998
|
-
var TOOLS_CONTRACT_LLM_OFFER = `A model running while people play is built into the Genex platform \u2014 the player pays, with
|
|
998
|
+
var TOOLS_CONTRACT_LLM_OFFER = `A model running while people play is built into the Genex platform \u2014 the player pays, with Genex coins or their own Claude/ChatGPT subscription, and approves it on a Genex sheet; your game just calls \`generate()\`. Want it that way?`;
|
|
999
999
|
function toolsContractLlmLaw(n, hosted) {
|
|
1000
1000
|
const onYes = hosted ? "On a yes, load `$genex-tool-llm` and then `$genex-llm-in-games`, which owns the build." : "On a yes, load `$genex-tool-llm`, which owns the rest: it checks the lane with `npx genex llm models` first, then runs `npx genex init --convert` (a hosted game is a static browser build \u2014 a local server holding a key can never ship).";
|
|
1001
1001
|
return `${n}. **A model running while people PLAY is a platform feature \u2014 never something you wire onto the user's own meter.** When a request implies one (NPCs that talk or decide in their own words, content written from what the player types, a prompt box in the game, "let the player pick a model"), recognise it, say in ONE line: "${TOOLS_CONTRACT_LLM_OFFER}" \u2014 then ASK and wait. ${onYes} On a no, build the authored version instead. Either way, never ship a game that calls a model on a key or an account of the user's own: every visitor would spend their money with nobody approving it, and the credential is readable in the bundle. Player-funded generation is text and JSON only \u2014 3D, images, video and audio stay the asset lanes above, on the user's meter.`;
|
|
@@ -22943,7 +22943,7 @@ async function runBlender(opts) {
|
|
|
22943
22943
|
return 1;
|
|
22944
22944
|
}
|
|
22945
22945
|
if (sub === "serve") {
|
|
22946
|
-
const { serveLocalBlender } = await import("./blender-serve-
|
|
22946
|
+
const { serveLocalBlender } = await import("./blender-serve-66OHC4ML.js");
|
|
22947
22947
|
const port = Number(process.env.GENEX_BLENDER_PORT ?? 8088);
|
|
22948
22948
|
log.step(`Starting a local Blender service on port ${port}`);
|
|
22949
22949
|
log.plain(
|
|
@@ -22952,7 +22952,7 @@ async function runBlender(opts) {
|
|
|
22952
22952
|
return serveLocalBlender({ port, log });
|
|
22953
22953
|
}
|
|
22954
22954
|
if (sub === "mcp") {
|
|
22955
|
-
const { runBlenderMcp } = await import("./blender-mcp-
|
|
22955
|
+
const { runBlenderMcp } = await import("./blender-mcp-QVEGHE5L.js");
|
|
22956
22956
|
return runBlenderMcp();
|
|
22957
22957
|
}
|
|
22958
22958
|
if (sub === "seat") {
|
|
@@ -23460,12 +23460,7 @@ import fs35 from "fs/promises";
|
|
|
23460
23460
|
import path36 from "path";
|
|
23461
23461
|
import { randomUUID as randomUUID2 } from "crypto";
|
|
23462
23462
|
var SUBS3 = ["models", "bench", "price", "status", "cancel"];
|
|
23463
|
-
var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN
|
|
23464
|
-
var LLM_MAX_COINS_DEPRECATED = "--max-coins is deprecated: it is read as --max-credits, the same number \u2014 a bench spends credits now.";
|
|
23465
|
-
function benchBalanceShortLine(spendable, maxCredits) {
|
|
23466
|
-
return `Your spendable credits (${spendable}) cannot cover one attempt at --max-credits ${maxCredits} \u2014 nothing was sent, nothing was spent. Add credits, or re-run with a lower --max-credits once the person at the keyboard agrees to it.`;
|
|
23467
|
-
}
|
|
23468
|
-
var LLM_BENCH_UNVERIFIED_LINE = "Credits pay only for a verified email, and this account's is not verified yet (credits_unverified) \u2014 nothing was sent, nothing was spent. Verify it, then re-run.";
|
|
23463
|
+
var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN coin \u2014 your wallet, not a player's, and one this build's asset allowance cannot see. Re-run with --max-coins <n> --user-approved only after the person at the keyboard agreed to the number.";
|
|
23469
23464
|
var LLM_LANE_OFF_LINE = "The runtime LLM lane is off on this stand \u2014 in-game generate() answers 404 here, and nothing can be benchmarked.";
|
|
23470
23465
|
var LLM_TOOLS_CONVERT_HINT = [
|
|
23471
23466
|
"In-game model calls need a hosted game. This folder is a Genex Tools workspace, so nothing here",
|
|
@@ -23478,7 +23473,7 @@ function serverSlotCounts(body) {
|
|
|
23478
23473
|
return { held: body.slotsHeld, limit: body.slotLimit };
|
|
23479
23474
|
}
|
|
23480
23475
|
var PENDING_SLOT_RELEASE_MINUTES = 10;
|
|
23481
|
-
var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so
|
|
23476
|
+
var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so its coin stays held until the bill resolves; such a row stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`;
|
|
23482
23477
|
var LLM_STATUS_NO_COUNT_LINE = "This stand does not report how many slots are held across your projects, so no count is shown.";
|
|
23483
23478
|
var LLM_CANCEL_AWAITING_BILL = "this call already stopped; its bill is awaiting the provider";
|
|
23484
23479
|
var LLM_CANCEL_ALREADY_FINAL = "this call is already final";
|
|
@@ -23492,33 +23487,11 @@ var DEFAULT_SAMPLE_TIMEOUT_SEC = 180;
|
|
|
23492
23487
|
function parseProviderKey(value) {
|
|
23493
23488
|
return value === "ok" || value === "rejected" || value === "unknown" ? value : null;
|
|
23494
23489
|
}
|
|
23495
|
-
function bpsOrNull(value) {
|
|
23496
|
-
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
23497
|
-
}
|
|
23498
|
-
function standingCreditGrantOrNull(value) {
|
|
23499
|
-
if (!value || typeof value !== "object") return null;
|
|
23500
|
-
const v = value;
|
|
23501
|
-
const whole = (x) => typeof x === "number" && Number.isSafeInteger(x) && x >= 0;
|
|
23502
|
-
if (!whole(v.minCredits) || !whole(v.maxCredits) || !whole(v.stepCredits) || !whole(v.defaultCredits)) return null;
|
|
23503
|
-
return {
|
|
23504
|
-
minCredits: v.minCredits,
|
|
23505
|
-
maxCredits: v.maxCredits,
|
|
23506
|
-
stepCredits: v.stepCredits,
|
|
23507
|
-
defaultCredits: v.defaultCredits,
|
|
23508
|
-
...whole(v.minUsdCents) ? { minUsdCents: v.minUsdCents } : {},
|
|
23509
|
-
...whole(v.maxUsdCents) ? { maxUsdCents: v.maxUsdCents } : {},
|
|
23510
|
-
...whole(v.stepUsdCents) ? { stepUsdCents: v.stepUsdCents } : {}
|
|
23511
|
-
};
|
|
23512
|
-
}
|
|
23513
23490
|
async function readRuntimeLane(apiUrl, token) {
|
|
23514
23491
|
const empty = (state) => ({
|
|
23515
23492
|
state,
|
|
23516
23493
|
models: null,
|
|
23517
23494
|
recommendedDeclaredHeadroomBps: null,
|
|
23518
|
-
currency: null,
|
|
23519
|
-
feeBps: null,
|
|
23520
|
-
usdCentsPerCredit: null,
|
|
23521
|
-
standingCreditGrant: null,
|
|
23522
23495
|
externalProviders: null,
|
|
23523
23496
|
total: null,
|
|
23524
23497
|
featuredCount: null,
|
|
@@ -23537,11 +23510,7 @@ async function readRuntimeLane(apiUrl, token) {
|
|
|
23537
23510
|
return {
|
|
23538
23511
|
state: "live",
|
|
23539
23512
|
models: body.models,
|
|
23540
|
-
recommendedDeclaredHeadroomBps:
|
|
23541
|
-
currency: body.currency === "credits" || body.currency === "coins" ? body.currency : null,
|
|
23542
|
-
feeBps: bpsOrNull(body.feeBps),
|
|
23543
|
-
usdCentsPerCredit: typeof body.usdCentsPerCredit === "number" && Number.isSafeInteger(body.usdCentsPerCredit) && body.usdCentsPerCredit >= 1 ? body.usdCentsPerCredit : null,
|
|
23544
|
-
standingCreditGrant: standingCreditGrantOrNull(body.standingCreditGrant),
|
|
23513
|
+
recommendedDeclaredHeadroomBps: typeof body.recommendedDeclaredHeadroomBps === "number" && Number.isFinite(body.recommendedDeclaredHeadroomBps) ? body.recommendedDeclaredHeadroomBps : null,
|
|
23545
23514
|
externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null,
|
|
23546
23515
|
total: typeof body.total === "number" && Number.isFinite(body.total) ? body.total : null,
|
|
23547
23516
|
featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null,
|
|
@@ -23557,45 +23526,18 @@ function percentile2(values, p) {
|
|
|
23557
23526
|
const rank2 = Math.ceil(p / 100 * sorted.length);
|
|
23558
23527
|
return sorted[Math.min(sorted.length - 1, Math.max(0, rank2 - 1))];
|
|
23559
23528
|
}
|
|
23560
|
-
function
|
|
23561
|
-
|
|
23562
|
-
const sorted = [...values].sort((a, b) => a < b ? -1 : a > b ? 1 : 0);
|
|
23563
|
-
const rank2 = Math.ceil(p / 100 * sorted.length);
|
|
23564
|
-
return sorted[Math.min(sorted.length - 1, Math.max(0, rank2 - 1))];
|
|
23529
|
+
function recommendEstimateCoins(p95, headroomBps) {
|
|
23530
|
+
return Math.max(1, Math.ceil(p95 * (1e4 + headroomBps) / 1e4));
|
|
23565
23531
|
}
|
|
23566
|
-
|
|
23567
|
-
|
|
23568
|
-
if (costPicos <= 0n) return 1;
|
|
23569
|
-
const num = costPicos * BigInt(1e4 + terms.feeBps) * BigInt(1e4 + terms.headroomBps);
|
|
23570
|
-
const den = 100000000n * BigInt(terms.usdCentsPerCredit) * PICOS_PER_CENT;
|
|
23571
|
-
return Math.max(1, Number((num + den - 1n) / den));
|
|
23572
|
-
}
|
|
23573
|
-
function averageCreditsPerCall(costsPicos, terms) {
|
|
23574
|
-
if (costsPicos.length === 0) return null;
|
|
23575
|
-
const sum = costsPicos.reduce((a, b) => a + b, 0n);
|
|
23576
|
-
return Number(sum * BigInt(1e4 + terms.feeBps)) / Number(BigInt(costsPicos.length) * 10000n * BigInt(terms.usdCentsPerCredit) * PICOS_PER_CENT);
|
|
23577
|
-
}
|
|
23578
|
-
function perCallEstimateCreditsFor(costsPicos, terms) {
|
|
23579
|
-
if (costsPicos.length === 0) return null;
|
|
23580
|
-
const sum = costsPicos.reduce((a, b) => a + b, 0n);
|
|
23581
|
-
const num = sum * BigInt(1e4 + terms.feeBps);
|
|
23582
|
-
const den = BigInt(costsPicos.length) * 10000n * BigInt(terms.usdCentsPerCredit) * PICOS_PER_CENT;
|
|
23583
|
-
return Math.max(1, Number((num + den - 1n) / den));
|
|
23584
|
-
}
|
|
23585
|
-
function formatCredits(value) {
|
|
23586
|
-
if (!Number.isFinite(value)) return "\u2014";
|
|
23587
|
-
if (value >= 100) return String(Math.round(value));
|
|
23588
|
-
return String(Number(value.toPrecision(value < 1 ? 2 : 3)));
|
|
23589
|
-
}
|
|
23590
|
-
function usdFromPicos(picos) {
|
|
23591
|
-
if (picos === null || picos === void 0) return "unknown";
|
|
23592
|
-
const value = typeof picos === "bigint" ? picos : /^\d+$/.test(picos) ? BigInt(picos) : null;
|
|
23593
|
-
if (value === null) return "unknown";
|
|
23594
|
-
return `$${(Number(value) / 1e12).toFixed(6)}`;
|
|
23532
|
+
function recommendPerCallMaxCoins(p95, max, headroomBps) {
|
|
23533
|
+
return Math.max(recommendEstimateCoins(p95, headroomBps), recommendEstimateCoins(max, headroomBps));
|
|
23595
23534
|
}
|
|
23596
23535
|
function usdPerMillion(value) {
|
|
23597
23536
|
return `$${value.toFixed(value < 1 ? 4 : 2)}`;
|
|
23598
23537
|
}
|
|
23538
|
+
function usd(value) {
|
|
23539
|
+
return value === null ? "unknown" : `$${value.toFixed(6)}`;
|
|
23540
|
+
}
|
|
23599
23541
|
async function runLlm(opts = {}) {
|
|
23600
23542
|
const log = createLogger({ quiet: opts.quiet || opts.json });
|
|
23601
23543
|
const cwd = opts.cwd ?? process.cwd();
|
|
@@ -23621,23 +23563,16 @@ async function runLlm(opts = {}) {
|
|
|
23621
23563
|
return;
|
|
23622
23564
|
}
|
|
23623
23565
|
if (sub === "bench") {
|
|
23624
|
-
if (opts.maxCoins
|
|
23625
|
-
`);
|
|
23626
|
-
if (opts.maxCoins !== void 0 && opts.maxCredits !== void 0 && opts.maxCoins !== opts.maxCredits) {
|
|
23627
|
-
fail5("llm bench", `Pass --max-credits alone \u2014 --max-coins is its deprecated name, and the two disagree (${opts.maxCredits} vs ${opts.maxCoins}).`);
|
|
23628
|
-
return;
|
|
23629
|
-
}
|
|
23630
|
-
const approvedCredits = opts.maxCredits ?? opts.maxCoins;
|
|
23631
|
-
if (approvedCredits === void 0 || opts.userApproved !== true) {
|
|
23566
|
+
if (opts.maxCoins === void 0 || opts.userApproved !== true) {
|
|
23632
23567
|
fail5("llm bench", LLM_BENCH_APPROVAL_REQUIRED);
|
|
23633
23568
|
return;
|
|
23634
23569
|
}
|
|
23635
|
-
if (!Number.isInteger(
|
|
23636
|
-
fail5("llm bench", `--max-
|
|
23570
|
+
if (!Number.isInteger(opts.maxCoins) || opts.maxCoins < 1) {
|
|
23571
|
+
fail5("llm bench", `--max-coins takes a whole number of coin, 1 or more (got ${String(opts.maxCoins)}).`);
|
|
23637
23572
|
return;
|
|
23638
23573
|
}
|
|
23639
23574
|
if (!opts.benchPrompt?.trim()) {
|
|
23640
|
-
fail5("llm bench", '`genex llm bench` needs the prompt your game would send, e.g. genex llm bench "<prompt>" --max-
|
|
23575
|
+
fail5("llm bench", '`genex llm bench` needs the prompt your game would send, e.g. genex llm bench "<prompt>" --max-coins <n> --user-approved.');
|
|
23641
23576
|
return;
|
|
23642
23577
|
}
|
|
23643
23578
|
if (opts.jsonOutput && opts.textOutput) {
|
|
@@ -23729,10 +23664,6 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23729
23664
|
models: null,
|
|
23730
23665
|
externalProviders: null,
|
|
23731
23666
|
recommendedDeclaredHeadroomBps: null,
|
|
23732
|
-
currency: null,
|
|
23733
|
-
feeBps: null,
|
|
23734
|
-
usdCentsPerCredit: null,
|
|
23735
|
-
standingCreditGrant: null,
|
|
23736
23667
|
total: null,
|
|
23737
23668
|
featuredCount: null,
|
|
23738
23669
|
// The stand was never asked, so the verdict is unknown — but the key is
|
|
@@ -23758,12 +23689,6 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23758
23689
|
models: lane.models,
|
|
23759
23690
|
externalProviders: lane.externalProviders,
|
|
23760
23691
|
recommendedDeclaredHeadroomBps: lane.recommendedDeclaredHeadroomBps,
|
|
23761
|
-
// The credit terms a call is billed on, the server's — null on a stand
|
|
23762
|
-
// that predates credit billing, never a number the CLI filled in.
|
|
23763
|
-
currency: lane.currency,
|
|
23764
|
-
feeBps: lane.feeBps,
|
|
23765
|
-
usdCentsPerCredit: lane.usdCentsPerCredit,
|
|
23766
|
-
standingCreditGrant: lane.standingCreditGrant,
|
|
23767
23692
|
total: lane.total ?? lane.models?.length ?? null,
|
|
23768
23693
|
featuredCount: lane.featuredCount ?? (lane.models ? lane.models.filter((m) => m.featured).length : null),
|
|
23769
23694
|
// `null` when the lane is not live or the server predates the probe.
|
|
@@ -23810,9 +23735,8 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23810
23735
|
);
|
|
23811
23736
|
for (const m of shown) {
|
|
23812
23737
|
const plan = m.personalPlan ? ` \xB7 personal plan: ${m.personalPlan}` : "";
|
|
23813
|
-
const kind = m.kind === "typesafe" ? " \xB7 judge (classifier): judge() under a budget, not generate() or bench" : "";
|
|
23814
23738
|
log.plain(
|
|
23815
|
-
` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens${plan}
|
|
23739
|
+
` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens${plan}`
|
|
23816
23740
|
);
|
|
23817
23741
|
}
|
|
23818
23742
|
if (hiddenCount > 0) {
|
|
@@ -23828,104 +23752,74 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23828
23752
|
}
|
|
23829
23753
|
log.plain("");
|
|
23830
23754
|
if (lane.recommendedDeclaredHeadroomBps === null) {
|
|
23831
|
-
log.dim(" This stand serves no recommended headroom, so no
|
|
23755
|
+
log.dim(" This stand serves no recommended headroom, so no price can be recommended from a bench.");
|
|
23832
23756
|
} else {
|
|
23833
|
-
log.dim(` Recommended headroom over
|
|
23834
|
-
}
|
|
23835
|
-
if (lane.feeBps === null) {
|
|
23836
|
-
log.dim(" This stand serves no platform fee \u2014 it predates credit billing, so no ceiling can be recommended from a bench.");
|
|
23837
|
-
} else {
|
|
23838
|
-
log.dim(` In-game calls are billed in the player's credits, as used: the provider's cost plus this stand's platform fee (${lane.feeBps} bps).`);
|
|
23757
|
+
log.dim(` Recommended headroom over benchmarked coins on this stand: ${lane.recommendedDeclaredHeadroomBps} bps.`);
|
|
23839
23758
|
}
|
|
23840
23759
|
if (toolsOnly) {
|
|
23841
23760
|
printConvertHint();
|
|
23842
23761
|
return;
|
|
23843
23762
|
}
|
|
23844
|
-
log.dim(" Those per-million rates are the PROVIDER's
|
|
23845
|
-
log.dim(`
|
|
23763
|
+
log.dim(" Those per-million rates are the PROVIDER's; what your game is charged is the platform");
|
|
23764
|
+
log.dim(` tariff on top, which only a real call reveals: ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')}`);
|
|
23846
23765
|
}
|
|
23847
23766
|
function wholeOrNull(value) {
|
|
23848
23767
|
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
23849
23768
|
}
|
|
23850
|
-
function costPicosOf(usage) {
|
|
23851
|
-
if (typeof usage?.costUsdPicos === "string" && /^\d+$/.test(usage.costUsdPicos)) return usage.costUsdPicos;
|
|
23852
|
-
if (typeof usage?.costUsd === "number" && Number.isFinite(usage.costUsd) && usage.costUsd >= 0) {
|
|
23853
|
-
return String(BigInt(Math.round(usage.costUsd * 1e12)));
|
|
23854
|
-
}
|
|
23855
|
-
return null;
|
|
23856
|
-
}
|
|
23857
23769
|
function sampleFrom(id, settled) {
|
|
23858
23770
|
const usage = settled?.usage;
|
|
23859
|
-
const picos = costPicosOf(usage);
|
|
23860
23771
|
return {
|
|
23861
23772
|
id,
|
|
23862
23773
|
status: settled?.status ?? null,
|
|
23863
23774
|
billingStatus: settled?.billingStatus ?? null,
|
|
23864
|
-
|
|
23865
|
-
costUsd: typeof usage?.costUsd === "number" ? usage.costUsd :
|
|
23866
|
-
costUsdPicos: picos,
|
|
23775
|
+
chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
|
|
23776
|
+
costUsd: typeof usage?.costUsd === "number" ? usage.costUsd : null,
|
|
23867
23777
|
error: settled?.error ?? (settled ? null : "timed_out"),
|
|
23868
23778
|
providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null,
|
|
23869
23779
|
inputTokens: wholeOrNull(usage?.inputTokens),
|
|
23870
23780
|
outputTokens: wholeOrNull(usage?.outputTokens),
|
|
23871
23781
|
maxOutputTokens: wholeOrNull(usage?.maxOutputTokens),
|
|
23872
|
-
|
|
23782
|
+
fitCoins: typeof usage?.fitCoins === "number" && Number.isSafeInteger(usage.fitCoins) && usage.fitCoins >= 1 ? usage.fitCoins : null,
|
|
23873
23783
|
fitExceedsCap: typeof usage?.fitExceedsCap === "boolean" ? usage.fitExceedsCap : null,
|
|
23874
23784
|
truncated: typeof usage?.truncated === "boolean" ? usage.truncated : settled?.error === "provider_token_limit" ? true : null
|
|
23875
23785
|
};
|
|
23876
23786
|
}
|
|
23877
23787
|
var PROVIDER_TOKEN_LIMIT = "provider_token_limit";
|
|
23878
|
-
function benchRecommendation(results,
|
|
23879
|
-
const settled = results.filter(
|
|
23880
|
-
|
|
23881
|
-
);
|
|
23882
|
-
const
|
|
23883
|
-
const
|
|
23884
|
-
const fits = settled.map((r) => r.fitCredits).filter((v) => typeof v === "number");
|
|
23885
|
-
const costP95 = percentilePicos(costs, 95);
|
|
23886
|
-
const costMax = costs.length ? costs.reduce((a, b) => b > a ? b : a) : null;
|
|
23788
|
+
function benchRecommendation(results, headroomBps) {
|
|
23789
|
+
const settled = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number");
|
|
23790
|
+
const charged = settled.map((r) => r.chargedCoins);
|
|
23791
|
+
const fits = settled.map((r) => r.fitCoins).filter((v) => typeof v === "number");
|
|
23792
|
+
const p95 = percentile2(charged, 95);
|
|
23793
|
+
const max = charged.length ? Math.max(...charged) : null;
|
|
23887
23794
|
const fitP95 = percentile2(fits, 95);
|
|
23888
23795
|
const fitMax = fits.length ? Math.max(...fits) : null;
|
|
23889
23796
|
const exceedsStandCap = results.some((r) => r.fitExceedsCap === true);
|
|
23890
23797
|
const cutOffSamples = results.filter((r) => r.error === PROVIDER_TOKEN_LIMIT && r.fitExceedsCap !== true).length;
|
|
23891
23798
|
const noRecommendationReason = exceedsStandCap ? "answer_exceeds_stand_cap" : cutOffSamples > 0 ? "answer_cut_off" : null;
|
|
23892
|
-
const
|
|
23893
|
-
const
|
|
23894
|
-
const
|
|
23895
|
-
const billing = feeBps !== null && usdCentsPerCredit !== null ? { feeBps, usdCentsPerCredit } : null;
|
|
23896
|
-
const costBased = full !== null && costP95 !== null ? ceilingCreditsFor(costP95, full) : null;
|
|
23897
|
-
const recommended = costBased === null || noRecommendationReason !== null ? null : Math.max(costBased, fitP95 ?? 0);
|
|
23898
|
-
const ceiling = recommended === null || costMax === null || full === null ? null : Math.max(ceilingCreditsFor(costMax, full), recommended, fitMax ?? 0);
|
|
23899
|
-
const average = billing !== null ? averageCreditsPerCall(costs, billing) : null;
|
|
23900
|
-
const estimate = ceiling === null || billing === null ? null : Math.min(ceiling, perCallEstimateCreditsFor(costs, billing) ?? 1);
|
|
23799
|
+
const chargedBased = p95 !== null && headroomBps !== null ? recommendEstimateCoins(p95, headroomBps) : null;
|
|
23800
|
+
const recommended = chargedBased === null || noRecommendationReason !== null ? null : Math.max(chargedBased, fitP95 ?? 0);
|
|
23801
|
+
const ceiling = recommended === null || p95 === null || max === null || headroomBps === null ? null : Math.max(recommendPerCallMaxCoins(p95, max, headroomBps), recommended, fitMax ?? 0);
|
|
23901
23802
|
return {
|
|
23902
|
-
costs,
|
|
23903
23803
|
charged,
|
|
23904
|
-
|
|
23905
|
-
|
|
23906
|
-
|
|
23907
|
-
chargedP50: percentile2(charged, 50),
|
|
23908
|
-
chargedP95: percentile2(charged, 95),
|
|
23909
|
-
chargedMax: charged.length ? Math.max(...charged) : null,
|
|
23804
|
+
p50: percentile2(charged, 50),
|
|
23805
|
+
p95,
|
|
23806
|
+
max,
|
|
23910
23807
|
fits,
|
|
23911
23808
|
fitP95,
|
|
23912
23809
|
fitMax,
|
|
23913
|
-
|
|
23914
|
-
|
|
23915
|
-
|
|
23916
|
-
|
|
23917
|
-
recommendedPerCallEstimateCredits: estimate,
|
|
23918
|
-
lengthRaisedCeiling: recommended !== null && costBased !== null && recommended > costBased,
|
|
23810
|
+
chargedBasedEstimateCoins: chargedBased,
|
|
23811
|
+
recommendedEstimateCoins: recommended,
|
|
23812
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
23813
|
+
lengthRaisedPrice: recommended !== null && chargedBased !== null && recommended > chargedBased,
|
|
23919
23814
|
cutOffSamples,
|
|
23920
23815
|
exceedsStandCap,
|
|
23921
|
-
noRecommendationReason
|
|
23922
|
-
missingTerms
|
|
23816
|
+
noRecommendationReason
|
|
23923
23817
|
};
|
|
23924
23818
|
}
|
|
23925
|
-
var LLM_LENGTH_RAISED_LINE = "The
|
|
23926
|
-
var LLM_EXCEEDS_STAND_CAP_LINE = "No
|
|
23927
|
-
function benchCutOffLine(cutOff,
|
|
23928
|
-
return `No
|
|
23819
|
+
var LLM_LENGTH_RAISED_LINE = "The declared price also pays for how long the answer may be: at the charged-based price the answer would be cut off, so declare this one.";
|
|
23820
|
+
var LLM_EXCEEDS_STAND_CAP_LINE = "No price recommended: this answer is longer than one call on this stand may produce. Ask for a shorter answer \u2014 fewer fields, shorter strings, a length the prompt states \u2014 and benchmark again.";
|
|
23821
|
+
function benchCutOffLine(cutOff, maxCoins) {
|
|
23822
|
+
return `No price recommended: ${cutOff} sample${cutOff === 1 ? " was" : "s were"} cut off at this ceiling (--max-coins ${maxCoins}) before the answer was finished, so its real length is unknown. Re-run with a higher --max-coins.`;
|
|
23929
23823
|
}
|
|
23930
23824
|
var ACTIVE_STATUSES = /* @__PURE__ */ new Set(["requires_confirmation", "queued", "dispatching", "awaiting_external"]);
|
|
23931
23825
|
function rowState(row) {
|
|
@@ -23936,22 +23830,7 @@ function rowState(row) {
|
|
|
23936
23830
|
function rowHoldsSlot(row) {
|
|
23937
23831
|
if (typeof row.slotHeld === "boolean") return row.slotHeld;
|
|
23938
23832
|
const state = rowState(row);
|
|
23939
|
-
return state === "active" || state === "awaiting bill" && (
|
|
23940
|
-
}
|
|
23941
|
-
function rowMoney(row) {
|
|
23942
|
-
const credits = row.currency === "credits" || row.currency === void 0 && typeof row.reservedCredits === "number" && typeof row.reservedCoins !== "number";
|
|
23943
|
-
if (credits) {
|
|
23944
|
-
return {
|
|
23945
|
-
unit: "credits",
|
|
23946
|
-
reserved: typeof row.reservedCredits === "number" ? row.reservedCredits : null,
|
|
23947
|
-
charged: typeof row.chargedCredits === "number" ? row.chargedCredits : null
|
|
23948
|
-
};
|
|
23949
|
-
}
|
|
23950
|
-
return {
|
|
23951
|
-
unit: "coin",
|
|
23952
|
-
reserved: typeof row.reservedCoins === "number" ? row.reservedCoins : null,
|
|
23953
|
-
charged: typeof row.chargedCoins === "number" ? row.chargedCoins : null
|
|
23954
|
-
};
|
|
23833
|
+
return state === "active" || state === "awaiting bill" && (row.reservedCoins ?? 0) > 0;
|
|
23955
23834
|
}
|
|
23956
23835
|
function providerRefusalStatus(error) {
|
|
23957
23836
|
const m = /^provider_http_(\d{3})$/.exec(error ?? "");
|
|
@@ -23970,7 +23849,7 @@ function formatAge(iso, now = Date.now()) {
|
|
|
23970
23849
|
}
|
|
23971
23850
|
async function runBench(args) {
|
|
23972
23851
|
const { apiUrl, token, projectId, cwd, opts, log } = args;
|
|
23973
|
-
const
|
|
23852
|
+
const maxCoins = opts.maxCoins;
|
|
23974
23853
|
const samples = opts.samples ?? DEFAULT_BENCH_SAMPLES;
|
|
23975
23854
|
const prompt = opts.benchPrompt.trim();
|
|
23976
23855
|
const auth = { Authorization: `Bearer ${token}`, "Content-Type": "application/json" };
|
|
@@ -23992,7 +23871,7 @@ async function runBench(args) {
|
|
|
23992
23871
|
const lane = await readRuntimeLane(apiUrl, token);
|
|
23993
23872
|
if (lane.state !== "live" || !lane.models || lane.models.length === 0) {
|
|
23994
23873
|
if (lane.state === "off") {
|
|
23995
|
-
if (opts.json) writeJsonLine({ ...emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples,
|
|
23874
|
+
if (opts.json) writeJsonLine({ ...emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoins, opts), status: "off", error: null });
|
|
23996
23875
|
else {
|
|
23997
23876
|
log.plain(LLM_LANE_OFF_LINE);
|
|
23998
23877
|
log.dim(" Nothing was spent.");
|
|
@@ -24019,27 +23898,17 @@ async function runBench(args) {
|
|
|
24019
23898
|
);
|
|
24020
23899
|
return;
|
|
24021
23900
|
}
|
|
24022
|
-
const
|
|
24023
|
-
if (walletBefore?.emailVerified === false) {
|
|
24024
|
-
benchFailed(opts, log, LLM_BENCH_UNVERIFIED_LINE);
|
|
24025
|
-
return;
|
|
24026
|
-
}
|
|
24027
|
-
if (walletBefore !== null && walletBefore.spendable < maxCredits) {
|
|
24028
|
-
benchFailed(opts, log, benchBalanceShortLine(walletBefore.spendable, maxCredits));
|
|
24029
|
-
return;
|
|
24030
|
-
}
|
|
24031
|
-
const balanceBefore = walletBefore?.spendable ?? null;
|
|
23901
|
+
const balanceBefore = await readCoinBalance(apiUrl, token);
|
|
24032
23902
|
if (!opts.json) {
|
|
24033
23903
|
log.plain(c.bold("genex llm bench"));
|
|
24034
23904
|
log.dim(` ${apiUrl}`);
|
|
24035
23905
|
log.plain("");
|
|
24036
23906
|
log.plain(` Model ${c.cyan(modelId)}`);
|
|
24037
23907
|
log.plain(` Samples ${samples} real attempt${samples === 1 ? "" : "s"}, ${outputFormat} output`);
|
|
24038
|
-
log.plain(` Approved ${
|
|
23908
|
+
log.plain(` Approved ${maxCoins} coin per attempt \u2014 at worst ${maxCoins * samples} coin for this run`);
|
|
24039
23909
|
log.plain(
|
|
24040
|
-
` Balance ${balanceBefore === null ? "couldn't be read" : `${balanceBefore}
|
|
23910
|
+
` Balance ${balanceBefore === null ? "couldn't be read" : `${balanceBefore} coin spendable`}`
|
|
24041
23911
|
);
|
|
24042
|
-
log.dim(" Billed as used: each attempt is charged its real cost plus the platform fee, rounded up to a whole credit.");
|
|
24043
23912
|
log.plain("");
|
|
24044
23913
|
}
|
|
24045
23914
|
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
@@ -24065,7 +23934,7 @@ async function runBench(args) {
|
|
|
24065
23934
|
const res = await call(base, {
|
|
24066
23935
|
method: "POST",
|
|
24067
23936
|
body: JSON.stringify({
|
|
24068
|
-
|
|
23937
|
+
maxCoins,
|
|
24069
23938
|
request: {
|
|
24070
23939
|
idempotencyKey: `bench-${randomUUID2()}`,
|
|
24071
23940
|
modelId,
|
|
@@ -24095,14 +23964,13 @@ async function runBench(args) {
|
|
|
24095
23964
|
if (refusedAt !== null) providerRefused++;
|
|
24096
23965
|
if (!opts.json) {
|
|
24097
23966
|
const length = row.outputTokens !== null ? ` \xB7 ${row.outputTokens}${row.maxOutputTokens !== null ? ` of ${row.maxOutputTokens}` : ""} tokens out` : "";
|
|
24098
|
-
const charged = row.chargedCredits === null ? "\u2014" : String(row.chargedCredits);
|
|
24099
23967
|
log.plain(
|
|
24100
|
-
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${
|
|
23968
|
+
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${length}${row.error ? ` \xB7 ${row.error}` : ""}`
|
|
24101
23969
|
);
|
|
24102
23970
|
if (refusedAt !== null) printProviderRefusal(log, refusedAt, row.providerMessage, row.status === "unknown");
|
|
24103
23971
|
if (row.error === PROVIDER_TOKEN_LIMIT) {
|
|
24104
23972
|
log.dim(
|
|
24105
|
-
row.fitExceedsCap === true ? " The answer was cut off at this stand's own output limit \u2014 no
|
|
23973
|
+
row.fitExceedsCap === true ? " The answer was cut off at this stand's own output limit \u2014 no price makes room for it; it is not a sample." : ` The answer was cut off at this ceiling (--max-coins ${maxCoins}) before it was finished \u2014 it was charged, and it is not a sample.`
|
|
24106
23974
|
);
|
|
24107
23975
|
}
|
|
24108
23976
|
}
|
|
@@ -24110,19 +23978,16 @@ async function runBench(args) {
|
|
|
24110
23978
|
} finally {
|
|
24111
23979
|
process.removeListener("SIGINT", onSigint);
|
|
24112
23980
|
}
|
|
24113
|
-
const
|
|
24114
|
-
|
|
24115
|
-
|
|
24116
|
-
|
|
24117
|
-
|
|
24118
|
-
const
|
|
24119
|
-
const settledCount = rec.charged.length;
|
|
23981
|
+
const headroomBps = lane.recommendedDeclaredHeadroomBps;
|
|
23982
|
+
const rec = benchRecommendation(results, headroomBps);
|
|
23983
|
+
const { charged, p50, p95, max } = rec;
|
|
23984
|
+
const settledCount = charged.length;
|
|
23985
|
+
const recommended = rec.recommendedEstimateCoins;
|
|
23986
|
+
const ceiling = rec.recommendedPerCallMaxCoins;
|
|
24120
23987
|
const lengthBlocked = rec.noRecommendationReason !== null;
|
|
24121
|
-
const balanceAfter =
|
|
24122
|
-
const picosOrNull = (v) => v === null ? null : String(v);
|
|
23988
|
+
const balanceAfter = await readCoinBalance(apiUrl, token);
|
|
24123
23989
|
const record = {
|
|
24124
|
-
v:
|
|
24125
|
-
currency: "credits",
|
|
23990
|
+
v: 1,
|
|
24126
23991
|
ranAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
24127
23992
|
apiUrl,
|
|
24128
23993
|
projectId,
|
|
@@ -24131,34 +23996,28 @@ async function runBench(args) {
|
|
|
24131
23996
|
outputFormat,
|
|
24132
23997
|
samples,
|
|
24133
23998
|
completed: settledCount,
|
|
24134
|
-
|
|
24135
|
-
|
|
24136
|
-
|
|
24137
|
-
|
|
24138
|
-
|
|
24139
|
-
|
|
23999
|
+
chargedCoins: charged,
|
|
24000
|
+
p50,
|
|
24001
|
+
p95,
|
|
24002
|
+
max,
|
|
24003
|
+
recommendedDeclaredHeadroomBps: headroomBps,
|
|
24004
|
+
recommendedEstimateCoins: recommended,
|
|
24005
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
24006
|
+
maxCoinsPerSample: maxCoins,
|
|
24007
|
+
fitCoins: rec.fits,
|
|
24140
24008
|
fitP95: rec.fitP95,
|
|
24141
|
-
|
|
24142
|
-
|
|
24143
|
-
feeBps: terms.feeBps,
|
|
24144
|
-
usdCentsPerCredit: terms.usdCentsPerCredit,
|
|
24145
|
-
costBasedMaxCredits: rec.costBasedMaxCredits,
|
|
24146
|
-
recommendedMaxCredits: rec.recommendedMaxCredits,
|
|
24147
|
-
recommendedPerCallMaxCredits: rec.recommendedPerCallMaxCredits,
|
|
24148
|
-
averageCreditsPerCall: rec.averageCreditsPerCall,
|
|
24149
|
-
recommendedPerCallEstimateCredits: rec.recommendedPerCallEstimateCredits,
|
|
24150
|
-
lengthRaisedCeiling: rec.lengthRaisedCeiling,
|
|
24009
|
+
chargedBasedEstimateCoins: rec.chargedBasedEstimateCoins,
|
|
24010
|
+
lengthRaisedPrice: rec.lengthRaisedPrice,
|
|
24151
24011
|
cutOffSamples: rec.cutOffSamples,
|
|
24152
24012
|
noRecommendationReason: rec.noRecommendationReason
|
|
24153
24013
|
};
|
|
24154
24014
|
const savedTo = await saveBench(cwd, record);
|
|
24155
24015
|
if (opts.json) {
|
|
24156
|
-
const ok =
|
|
24016
|
+
const ok = charged.length > 0 && !lengthBlocked;
|
|
24157
24017
|
writeJsonLine({
|
|
24158
24018
|
command: "llm bench",
|
|
24159
24019
|
status: ok ? "ok" : "failed",
|
|
24160
24020
|
error: ok ? null : rec.noRecommendationReason ?? "no_sample_settled",
|
|
24161
|
-
currency: "credits",
|
|
24162
24021
|
apiUrl,
|
|
24163
24022
|
projectId,
|
|
24164
24023
|
modelId,
|
|
@@ -24167,54 +24026,35 @@ async function runBench(args) {
|
|
|
24167
24026
|
schemaPath: opts.schemaPath ?? null,
|
|
24168
24027
|
samples,
|
|
24169
24028
|
completed: settledCount,
|
|
24170
|
-
|
|
24171
|
-
// Spendable credits before and after the run; null when unreadable.
|
|
24029
|
+
maxCoinsPerSample: maxCoins,
|
|
24172
24030
|
balanceBefore,
|
|
24173
24031
|
balanceAfter,
|
|
24174
24032
|
results,
|
|
24175
24033
|
// The create-time refusal that ended the run, or null. Beside the
|
|
24176
24034
|
// samples, never among them: no attempt existed and nothing was spent.
|
|
24177
24035
|
refusal: refusal2,
|
|
24178
|
-
|
|
24179
|
-
//
|
|
24180
|
-
costUsdPicos: record.cost,
|
|
24181
|
-
costUsd: {
|
|
24182
|
-
p50: rec.costP50 === null ? null : Number(rec.costP50) / 1e12,
|
|
24183
|
-
p95: rec.costP95 === null ? null : Number(rec.costP95) / 1e12,
|
|
24184
|
-
max: rec.costMax === null ? null : Number(rec.costMax) / 1e12
|
|
24185
|
-
},
|
|
24186
|
-
chargedCredits: record.charged,
|
|
24187
|
-
// The answer's LENGTH, priced by the server per sample (`fitCredits`,
|
|
24036
|
+
chargedCoins: { p50, p95, max },
|
|
24037
|
+
// The answer's LENGTH, priced by the server per sample (`fitCoins`,
|
|
24188
24038
|
// headroom already on the tokens), over the settled samples.
|
|
24189
|
-
|
|
24190
|
-
|
|
24191
|
-
|
|
24192
|
-
usdCentsPerCredit: terms.usdCentsPerCredit,
|
|
24193
|
-
missingTerms: rec.missingTerms,
|
|
24194
|
-
costBasedMaxCredits: rec.costBasedMaxCredits,
|
|
24195
|
-
lengthRaisedCeiling: rec.lengthRaisedCeiling,
|
|
24039
|
+
fitCoins: { p95: rec.fitP95, max: rec.fitMax },
|
|
24040
|
+
chargedBasedEstimateCoins: rec.chargedBasedEstimateCoins,
|
|
24041
|
+
lengthRaisedPrice: rec.lengthRaisedPrice,
|
|
24196
24042
|
cutOffSamples: rec.cutOffSamples,
|
|
24197
24043
|
exceedsStandCap: rec.exceedsStandCap,
|
|
24198
24044
|
noRecommendationReason: rec.noRecommendationReason,
|
|
24199
|
-
|
|
24200
|
-
|
|
24201
|
-
|
|
24202
|
-
recommendedPerCallEstimateCredits: rec.recommendedPerCallEstimateCredits,
|
|
24045
|
+
recommendedDeclaredHeadroomBps: headroomBps,
|
|
24046
|
+
recommendedEstimateCoins: recommended,
|
|
24047
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
24203
24048
|
savedTo
|
|
24204
24049
|
});
|
|
24205
24050
|
if (!ok) process.exitCode = 1;
|
|
24206
24051
|
return;
|
|
24207
24052
|
}
|
|
24208
24053
|
log.plain("");
|
|
24209
|
-
const printMeasured = () => {
|
|
24210
|
-
log.plain(c.bold(" Real cost per call"));
|
|
24211
|
-
log.plain(` p50 ${usdFromPicos(rec.costP50)} p95 ${usdFromPicos(rec.costP95)} max ${usdFromPicos(rec.costMax)} (${settledCount} of ${samples} settled)`);
|
|
24212
|
-
log.plain(c.bold(" Charged credits"));
|
|
24213
|
-
log.plain(` p50 ${rec.chargedP50} p95 ${rec.chargedP95} max ${rec.chargedMax} (whole credits, rounded up per call)`);
|
|
24214
|
-
};
|
|
24215
24054
|
if (lengthBlocked) {
|
|
24216
|
-
if (
|
|
24217
|
-
|
|
24055
|
+
if (charged.length > 0) {
|
|
24056
|
+
log.plain(c.bold(" Charged coins"));
|
|
24057
|
+
log.plain(` p50 ${p50} p95 ${p95} max ${max} (${charged.length} of ${samples} settled)`);
|
|
24218
24058
|
log.plain("");
|
|
24219
24059
|
}
|
|
24220
24060
|
printRecommendation(log, record);
|
|
@@ -24222,16 +24062,17 @@ async function runBench(args) {
|
|
|
24222
24062
|
process.exitCode = 1;
|
|
24223
24063
|
return;
|
|
24224
24064
|
}
|
|
24225
|
-
if (
|
|
24065
|
+
if (charged.length === 0) {
|
|
24226
24066
|
log.error(
|
|
24227
|
-
providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to
|
|
24067
|
+
providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to price from." : " No sample settled, so there is nothing to price from."
|
|
24228
24068
|
);
|
|
24229
24069
|
process.exitCode = 1;
|
|
24230
24070
|
return;
|
|
24231
24071
|
}
|
|
24232
|
-
|
|
24072
|
+
log.plain(c.bold(" Charged coins"));
|
|
24073
|
+
log.plain(` p50 ${p50} p95 ${p95} max ${max} (${charged.length} of ${samples} settled)`);
|
|
24233
24074
|
if (balanceBefore !== null && balanceAfter !== null) {
|
|
24234
|
-
log.plain(` Balance ${balanceBefore} \u2192 ${balanceAfter}
|
|
24075
|
+
log.plain(` Balance ${balanceBefore} \u2192 ${balanceAfter} coin spendable`);
|
|
24235
24076
|
}
|
|
24236
24077
|
log.plain("");
|
|
24237
24078
|
printRecommendation(log, record);
|
|
@@ -24243,58 +24084,43 @@ function printRecommendation(log, record) {
|
|
|
24243
24084
|
return;
|
|
24244
24085
|
}
|
|
24245
24086
|
if (record.noRecommendationReason === "answer_cut_off") {
|
|
24246
|
-
log.warn(` ${benchCutOffLine(record.cutOffSamples
|
|
24087
|
+
log.warn(` ${benchCutOffLine(record.cutOffSamples ?? 1, record.maxCoinsPerSample ?? 0)}`);
|
|
24247
24088
|
return;
|
|
24248
24089
|
}
|
|
24249
|
-
|
|
24250
|
-
|
|
24251
|
-
log.warn(" No ceiling recommended.");
|
|
24090
|
+
if (record.recommendedEstimateCoins === null || record.p95 === null) {
|
|
24091
|
+
log.warn(" No price recommended.");
|
|
24252
24092
|
log.dim(
|
|
24253
|
-
record.
|
|
24093
|
+
record.recommendedDeclaredHeadroomBps === null ? " This stand served no recommended headroom, and the multiplier is the server's to set \u2014" : " No sample settled, so there is no p95 to build a price on \u2014"
|
|
24254
24094
|
);
|
|
24255
24095
|
log.dim(" declaring a number from this run would be a guess dressed as a measurement.");
|
|
24256
|
-
printAverage(log, record, attempts);
|
|
24257
24096
|
return;
|
|
24258
24097
|
}
|
|
24259
|
-
log.plain(c.bold(` Declare
|
|
24260
|
-
if (record.
|
|
24098
|
+
log.plain(c.bold(` Declare estimateCoins: ${record.recommendedEstimateCoins}`));
|
|
24099
|
+
if (record.lengthRaisedPrice) {
|
|
24261
24100
|
log.plain(` ${LLM_LENGTH_RAISED_LINE}`);
|
|
24262
24101
|
log.dim(
|
|
24263
|
-
` = the smallest
|
|
24102
|
+
` = the smallest price whose answer allowance holds the answer plus this stand's headroom (p95 ${record.fitP95 ?? "\u2014"}), above the`
|
|
24264
24103
|
);
|
|
24265
24104
|
log.dim(
|
|
24266
|
-
`
|
|
24105
|
+
` charged-based ${record.chargedBasedEstimateCoins ?? "\u2014"} (p95 of charged coins ${record.p95} plus ${record.recommendedDeclaredHeadroomBps} bps), over ${record.completed} settled attempt${record.completed === 1 ? "" : "s"} on ${record.modelId}.`
|
|
24267
24106
|
);
|
|
24268
24107
|
} else {
|
|
24269
24108
|
log.dim(
|
|
24270
|
-
` = p95
|
|
24109
|
+
` = p95 of charged coins (${record.p95}) plus this stand's recommended headroom (${record.recommendedDeclaredHeadroomBps} bps),`
|
|
24271
24110
|
);
|
|
24272
|
-
log.dim(` measured over ${
|
|
24111
|
+
log.dim(` measured over ${record.completed} settled attempt${record.completed === 1 ? "" : "s"} on ${record.modelId}.`);
|
|
24273
24112
|
}
|
|
24274
|
-
log.dim(" That number is a
|
|
24275
|
-
|
|
24276
|
-
|
|
24277
|
-
log.plain(c.bold(` Grant perCallMaxCredits: ${record.recommendedPerCallMaxCredits}`));
|
|
24113
|
+
log.dim(" That number is a PRICE: a started attempt is charged in full, including one that fails.");
|
|
24114
|
+
if (record.recommendedPerCallMaxCoins != null) {
|
|
24115
|
+
log.plain(c.bold(` Grant perCallMaxCoins: ${record.recommendedPerCallMaxCoins}`));
|
|
24278
24116
|
log.dim(
|
|
24279
|
-
record.
|
|
24117
|
+
record.lengthRaisedPrice ? ` = the same headroom over the worst sample (${record.max}) or the longest answer's price, whichever is larger, and never below the price above it.` : ` = the same headroom over the worst sample (${record.max}), and never below the price above it.`
|
|
24280
24118
|
);
|
|
24281
|
-
|
|
24282
|
-
printAverage(log, record, attempts);
|
|
24283
|
-
}
|
|
24284
|
-
function printAverage(log, record, attempts) {
|
|
24285
|
-
if (record.averageCreditsPerCall === null) return;
|
|
24286
|
-
log.plain(` Per-call estimate: about ${formatCredits(record.averageCreditsPerCall)} credits per call`);
|
|
24287
|
-
log.dim(` = the average real cost with the platform fee, no headroom, over ${attempts}.`);
|
|
24288
|
-
if (record.recommendedPerCallEstimateCredits !== null) {
|
|
24289
|
-
log.plain(c.bold(` Grant perCallEstimateCredits: ${record.recommendedPerCallEstimateCredits}`));
|
|
24290
|
-
log.dim(` = that average rounded up to a whole credit. disclosure.estimatedCreditsPerPeriod = your calls per period \xD7 ${formatCredits(record.averageCreditsPerCall)}, rounded up.`);
|
|
24119
|
+
log.dim(" The ceiling is the price's room to be wrong; declaring one under the price is refused.");
|
|
24291
24120
|
}
|
|
24292
24121
|
}
|
|
24293
24122
|
async function explainBenchRefusal(res, call, base, log, opts, index, settledSoFar) {
|
|
24294
|
-
if (printedStructuredError(res)) {
|
|
24295
|
-
const printed = await res.json().catch(() => ({}));
|
|
24296
|
-
return { stop: true, refusal: { status: res.status, error: typeof printed.error === "string" ? printed.error : null, slotsHeld: null, slotLimit: null } };
|
|
24297
|
-
}
|
|
24123
|
+
if (printedStructuredError(res)) return { stop: true, refusal: { status: res.status, error: null, slotsHeld: null, slotLimit: null } };
|
|
24298
24124
|
const body = await res.json().catch(() => ({}));
|
|
24299
24125
|
const refusal2 = { status: res.status, error: body.error ?? null, slotsHeld: null, slotLimit: null };
|
|
24300
24126
|
if (res.status === 429 && body.error === "generation_limit") {
|
|
@@ -24304,10 +24130,6 @@ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoF
|
|
|
24304
24130
|
if (!opts.json) log.error(` Sample ${index + 1} ${benchSlotsHeldSentence(counts)}`);
|
|
24305
24131
|
return { stop: true, refusal: refusal2 };
|
|
24306
24132
|
}
|
|
24307
|
-
if (res.status === 403 && body.error === "credits_unverified") {
|
|
24308
|
-
if (!opts.json) log.error(` ${LLM_BENCH_UNVERIFIED_LINE}`);
|
|
24309
|
-
return { stop: true, refusal: refusal2 };
|
|
24310
|
-
}
|
|
24311
24133
|
if (opts.json) return { stop: res.status !== 429, refusal: refusal2 };
|
|
24312
24134
|
if (res.status === 404 && body.error === "not_found") {
|
|
24313
24135
|
log.error(` ${LLM_LANE_OFF_LINE}`);
|
|
@@ -24323,7 +24145,7 @@ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoF
|
|
|
24323
24145
|
return { stop: true, refusal: refusal2 };
|
|
24324
24146
|
}
|
|
24325
24147
|
if (res.status === 402) {
|
|
24326
|
-
log.error(" Not enough
|
|
24148
|
+
log.error(" Not enough coin to start the attempt.");
|
|
24327
24149
|
return { stop: true, refusal: refusal2 };
|
|
24328
24150
|
}
|
|
24329
24151
|
log.error(
|
|
@@ -24332,7 +24154,7 @@ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoF
|
|
|
24332
24154
|
return { stop: res.status >= 500 || res.status === 401 || res.status === 403, refusal: refusal2 };
|
|
24333
24155
|
}
|
|
24334
24156
|
function printProviderRefusal(log, status2, message, awaitingBill = false) {
|
|
24335
|
-
if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so
|
|
24157
|
+
if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so this call's coin stays held until it resolves (see \`genex llm status\`); it is not a sample.`);
|
|
24336
24158
|
else log.dim(" The provider refused this call at its door \u2014 no inference ran, it cost nothing, and it is not a sample.");
|
|
24337
24159
|
if (message) log.dim(` Provider said: ${message}`);
|
|
24338
24160
|
if (status2 === 401 || status2 === 403) {
|
|
@@ -24358,6 +24180,19 @@ async function pollSettled(call, url, timeoutSec) {
|
|
|
24358
24180
|
}
|
|
24359
24181
|
return last;
|
|
24360
24182
|
}
|
|
24183
|
+
async function readCoinBalance(apiUrl, token) {
|
|
24184
|
+
try {
|
|
24185
|
+
const res = await apiFetch(`${apiUrl}/api/coin/balance`, {
|
|
24186
|
+
headers: { Authorization: `Bearer ${token}` },
|
|
24187
|
+
signal: AbortSignal.timeout(6e3)
|
|
24188
|
+
});
|
|
24189
|
+
if (!res.ok) return null;
|
|
24190
|
+
const body = await res.json().catch(() => null);
|
|
24191
|
+
return typeof body?.spendable === "number" ? body.spendable : null;
|
|
24192
|
+
} catch {
|
|
24193
|
+
return null;
|
|
24194
|
+
}
|
|
24195
|
+
}
|
|
24361
24196
|
async function saveBench(cwd, record) {
|
|
24362
24197
|
try {
|
|
24363
24198
|
const file = path36.join(cwd, LLM_BENCH_FILE);
|
|
@@ -24373,10 +24208,9 @@ function benchFailed(opts, log, message) {
|
|
|
24373
24208
|
else log.error(message);
|
|
24374
24209
|
process.exitCode = 1;
|
|
24375
24210
|
}
|
|
24376
|
-
function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples,
|
|
24211
|
+
function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoins, opts) {
|
|
24377
24212
|
return {
|
|
24378
24213
|
command: "llm bench",
|
|
24379
|
-
currency: "credits",
|
|
24380
24214
|
apiUrl,
|
|
24381
24215
|
projectId,
|
|
24382
24216
|
modelId: opts.modelId ?? null,
|
|
@@ -24385,28 +24219,21 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCre
|
|
|
24385
24219
|
schemaPath: opts.schemaPath ?? null,
|
|
24386
24220
|
samples,
|
|
24387
24221
|
completed: 0,
|
|
24388
|
-
|
|
24222
|
+
maxCoinsPerSample: maxCoins,
|
|
24389
24223
|
balanceBefore: null,
|
|
24390
24224
|
balanceAfter: null,
|
|
24391
24225
|
results: [],
|
|
24392
24226
|
refusal: null,
|
|
24393
|
-
|
|
24394
|
-
|
|
24395
|
-
|
|
24396
|
-
|
|
24397
|
-
recommendedDeclaredHeadroomBps: null,
|
|
24398
|
-
feeBps: null,
|
|
24399
|
-
usdCentsPerCredit: null,
|
|
24400
|
-
missingTerms: null,
|
|
24401
|
-
costBasedMaxCredits: null,
|
|
24402
|
-
lengthRaisedCeiling: false,
|
|
24227
|
+
chargedCoins: { p50: null, p95: null, max: null },
|
|
24228
|
+
fitCoins: { p95: null, max: null },
|
|
24229
|
+
chargedBasedEstimateCoins: null,
|
|
24230
|
+
lengthRaisedPrice: false,
|
|
24403
24231
|
cutOffSamples: 0,
|
|
24404
24232
|
exceedsStandCap: false,
|
|
24405
24233
|
noRecommendationReason: null,
|
|
24406
|
-
|
|
24407
|
-
|
|
24408
|
-
|
|
24409
|
-
recommendedPerCallEstimateCredits: null,
|
|
24234
|
+
recommendedDeclaredHeadroomBps: null,
|
|
24235
|
+
recommendedEstimateCoins: null,
|
|
24236
|
+
recommendedPerCallMaxCoins: null,
|
|
24410
24237
|
savedTo: null
|
|
24411
24238
|
};
|
|
24412
24239
|
}
|
|
@@ -24469,8 +24296,7 @@ async function reportStatus(args) {
|
|
|
24469
24296
|
for (const row of rows) {
|
|
24470
24297
|
const state = rowState(row);
|
|
24471
24298
|
const mark = state === "active" ? c.cyan("\u25CF") : state === "awaiting bill" ? c.yellow("\u25D0") : c.green("\u25CB");
|
|
24472
|
-
const
|
|
24473
|
-
const reserved = money2.reserved === null ? "reserved \u2014" : `${money2.reserved} ${money2.unit} reserved`;
|
|
24299
|
+
const reserved = typeof row.reservedCoins === "number" ? `${row.reservedCoins} coin reserved` : "reserved \u2014";
|
|
24474
24300
|
log.plain(
|
|
24475
24301
|
` ${mark} ${row.id ?? "?"} ${row.modelId ?? "?"} ${state.padEnd(13)} ${formatAge(row.createdAt, now).padStart(4)} old ${reserved}${row.error ? ` ${row.error}` : ""}${rowHoldsSlot(row) ? "" : c.dim(" (no slot)")}`
|
|
24476
24302
|
);
|
|
@@ -24542,11 +24368,10 @@ async function cancelCall(args) {
|
|
|
24542
24368
|
if (!opts.json) {
|
|
24543
24369
|
log.warn(` ${row.id} (${row.status ?? "?"}) \u2014 ${sentence}.`);
|
|
24544
24370
|
if (state === "awaiting bill") {
|
|
24545
|
-
log.dim(` Nothing to cancel:
|
|
24371
|
+
log.dim(` Nothing to cancel: its coin releases when the bill resolves, and it stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`);
|
|
24546
24372
|
if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
|
|
24547
24373
|
} else {
|
|
24548
|
-
|
|
24549
|
-
log.dim(` ${money2.charged === null ? "charge unknown" : `${money2.charged} ${money2.unit} charged`}${row.error ? ` \xB7 ${row.error}` : ""}.`);
|
|
24374
|
+
log.dim(` ${typeof row.chargedCoins === "number" ? `${row.chargedCoins} coin charged` : "charge unknown"}${row.error ? ` \xB7 ${row.error}` : ""}.`);
|
|
24550
24375
|
}
|
|
24551
24376
|
}
|
|
24552
24377
|
return done("ok", null, state === "awaiting bill" ? "awaiting_bill" : "already_final", row);
|
|
@@ -24568,126 +24393,78 @@ async function cancelCall(args) {
|
|
|
24568
24393
|
const after = await res.json().catch(() => null) ?? row;
|
|
24569
24394
|
if (!opts.json) {
|
|
24570
24395
|
log.success(` Cancelled ${row.id} (was ${row.status ?? "active"}).`);
|
|
24571
|
-
log.dim(`
|
|
24396
|
+
log.dim(` Its coin releases once the bill settles; ${c.cyan("genex llm status")} shows it until then.`);
|
|
24572
24397
|
}
|
|
24573
24398
|
return done("ok", null, "cancelled", after);
|
|
24574
24399
|
}
|
|
24575
|
-
async function
|
|
24576
|
-
let
|
|
24400
|
+
async function reportSavedPrice(cwd, opts, log) {
|
|
24401
|
+
let record = null;
|
|
24577
24402
|
try {
|
|
24578
|
-
|
|
24403
|
+
record = JSON.parse(await fs35.readFile(path36.join(cwd, LLM_BENCH_FILE), "utf8"));
|
|
24579
24404
|
} catch {
|
|
24580
|
-
|
|
24405
|
+
record = null;
|
|
24581
24406
|
}
|
|
24582
|
-
if (!
|
|
24583
|
-
const v = parsed;
|
|
24584
|
-
if (v.v === 2 && v.currency === "credits" && v.cost && v.charged) return { kind: "credits", record: parsed };
|
|
24585
|
-
return { kind: "coins", record: parsed };
|
|
24586
|
-
}
|
|
24587
|
-
var LLM_PRICE_COIN_TERMS_LINE = "This bench was measured under the old COIN terms, when a game declared a fixed price per call. In-game calls are now billed in the player's credits, as used, and a game declares a per-call CEILING (maxCredits) instead \u2014 a coin price is not that number. Re-run the bench for the figures to declare.";
|
|
24588
|
-
async function reportSavedPrice(cwd, opts, log) {
|
|
24589
|
-
const saved = await readSavedBench(cwd);
|
|
24590
|
-
const rerun = 'genex llm bench "<prompt>" --max-credits <n> --user-approved';
|
|
24591
|
-
const creditKeys = (record2) => ({
|
|
24592
|
-
costUsdPicos: record2?.cost ?? { p50: null, p95: null, max: null },
|
|
24593
|
-
chargedCredits: record2?.charged ?? { p50: null, p95: null, max: null },
|
|
24594
|
-
fitP95: record2?.fitP95 ?? null,
|
|
24595
|
-
fitMax: record2?.fitMax ?? null,
|
|
24596
|
-
recommendedDeclaredHeadroomBps: record2?.recommendedDeclaredHeadroomBps ?? null,
|
|
24597
|
-
feeBps: record2?.feeBps ?? null,
|
|
24598
|
-
usdCentsPerCredit: record2?.usdCentsPerCredit ?? null,
|
|
24599
|
-
costBasedMaxCredits: record2?.costBasedMaxCredits ?? null,
|
|
24600
|
-
lengthRaisedCeiling: record2?.lengthRaisedCeiling ?? false,
|
|
24601
|
-
cutOffSamples: record2?.cutOffSamples ?? 0,
|
|
24602
|
-
noRecommendationReason: record2?.noRecommendationReason ?? null,
|
|
24603
|
-
recommendedMaxCredits: record2?.recommendedMaxCredits ?? null,
|
|
24604
|
-
recommendedPerCallMaxCredits: record2?.recommendedPerCallMaxCredits ?? null,
|
|
24605
|
-
averageCreditsPerCall: record2?.averageCreditsPerCall ?? null,
|
|
24606
|
-
recommendedPerCallEstimateCredits: record2?.recommendedPerCallEstimateCredits ?? null
|
|
24607
|
-
});
|
|
24608
|
-
if (!saved) {
|
|
24407
|
+
if (!record) {
|
|
24609
24408
|
if (opts.json) {
|
|
24610
24409
|
writeJsonLine({
|
|
24611
24410
|
command: "llm price",
|
|
24612
24411
|
status: "failed",
|
|
24613
24412
|
error: "no_bench",
|
|
24614
|
-
currency: null,
|
|
24615
24413
|
ranAt: null,
|
|
24616
24414
|
modelId: null,
|
|
24617
24415
|
samples: null,
|
|
24618
24416
|
completed: null,
|
|
24619
|
-
|
|
24620
|
-
|
|
24417
|
+
p50: null,
|
|
24418
|
+
p95: null,
|
|
24419
|
+
max: null,
|
|
24420
|
+
fitP95: null,
|
|
24421
|
+
chargedBasedEstimateCoins: null,
|
|
24422
|
+
lengthRaisedPrice: false,
|
|
24423
|
+
cutOffSamples: 0,
|
|
24424
|
+
noRecommendationReason: null,
|
|
24425
|
+
recommendedDeclaredHeadroomBps: null,
|
|
24426
|
+
recommendedEstimateCoins: null,
|
|
24427
|
+
recommendedPerCallMaxCoins: null,
|
|
24621
24428
|
savedTo: null
|
|
24622
24429
|
});
|
|
24623
24430
|
} else {
|
|
24624
24431
|
log.error("Nothing has been benchmarked in this folder yet.");
|
|
24625
|
-
log.dim(` Run ${c.cyan(
|
|
24626
|
-
}
|
|
24627
|
-
process.exitCode = 1;
|
|
24628
|
-
return;
|
|
24629
|
-
}
|
|
24630
|
-
if (saved.kind === "coins") {
|
|
24631
|
-
const old = saved.record;
|
|
24632
|
-
const legacy = {
|
|
24633
|
-
p50: old.p50 ?? null,
|
|
24634
|
-
p95: old.p95 ?? null,
|
|
24635
|
-
max: old.max ?? null,
|
|
24636
|
-
recommendedEstimateCoins: old.recommendedEstimateCoins ?? null,
|
|
24637
|
-
recommendedPerCallMaxCoins: old.recommendedPerCallMaxCoins ?? null,
|
|
24638
|
-
maxCoinsPerSample: old.maxCoinsPerSample ?? null,
|
|
24639
|
-
noRecommendationReason: old.noRecommendationReason ?? null
|
|
24640
|
-
};
|
|
24641
|
-
if (opts.json) {
|
|
24642
|
-
writeJsonLine({
|
|
24643
|
-
command: "llm price",
|
|
24644
|
-
status: "failed",
|
|
24645
|
-
error: "coin_terms",
|
|
24646
|
-
currency: "coins",
|
|
24647
|
-
ranAt: old.ranAt ?? null,
|
|
24648
|
-
modelId: old.modelId ?? null,
|
|
24649
|
-
samples: old.samples ?? null,
|
|
24650
|
-
completed: old.completed ?? null,
|
|
24651
|
-
...creditKeys(null),
|
|
24652
|
-
legacyCoinTerms: legacy,
|
|
24653
|
-
savedTo: LLM_BENCH_FILE
|
|
24654
|
-
});
|
|
24655
|
-
} else {
|
|
24656
|
-
log.plain(c.bold("genex llm price"));
|
|
24657
|
-
log.dim(` from ${LLM_BENCH_FILE}, benchmarked ${old.ranAt ?? "at an unknown time"}${old.modelId ? ` on ${old.modelId}` : ""}`);
|
|
24658
|
-
log.plain("");
|
|
24659
|
-
log.warn(` ${LLM_PRICE_COIN_TERMS_LINE}`);
|
|
24660
|
-
const measured = legacy.p95 === null ? "no settled sample" : `charged coins p50 ${legacy.p50} \xB7 p95 ${legacy.p95} \xB7 max ${legacy.max}`;
|
|
24661
|
-
const recommended = legacy.recommendedEstimateCoins === null ? "no price recommended" : `it recommended estimateCoins ${legacy.recommendedEstimateCoins}${legacy.recommendedPerCallMaxCoins != null ? ` and perCallMaxCoins ${legacy.recommendedPerCallMaxCoins}` : ""}`;
|
|
24662
|
-
log.dim(` What it measured then: ${measured}; ${recommended}.`);
|
|
24663
|
-
log.dim(` Re-run: ${c.cyan(rerun)}`);
|
|
24432
|
+
log.dim(` Run ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')} first.`);
|
|
24664
24433
|
}
|
|
24665
24434
|
process.exitCode = 1;
|
|
24666
24435
|
return;
|
|
24667
24436
|
}
|
|
24668
|
-
const record = saved.record;
|
|
24669
24437
|
if (opts.json) {
|
|
24670
24438
|
writeJsonLine({
|
|
24671
24439
|
command: "llm price",
|
|
24672
|
-
status: record.
|
|
24673
|
-
error: record.
|
|
24674
|
-
currency: "credits",
|
|
24440
|
+
status: record.recommendedEstimateCoins === null ? "failed" : "ok",
|
|
24441
|
+
error: record.recommendedEstimateCoins === null ? record.noRecommendationReason ?? "no_recommendation" : null,
|
|
24675
24442
|
ranAt: record.ranAt ?? null,
|
|
24676
24443
|
modelId: record.modelId ?? null,
|
|
24677
24444
|
samples: record.samples ?? null,
|
|
24678
24445
|
completed: record.completed ?? null,
|
|
24679
|
-
|
|
24680
|
-
|
|
24446
|
+
p50: record.p50 ?? null,
|
|
24447
|
+
p95: record.p95 ?? null,
|
|
24448
|
+
max: record.max ?? null,
|
|
24449
|
+
// Absent from a file an older CLI wrote; null / false / 0 then, never missing.
|
|
24450
|
+
fitP95: record.fitP95 ?? null,
|
|
24451
|
+
chargedBasedEstimateCoins: record.chargedBasedEstimateCoins ?? null,
|
|
24452
|
+
lengthRaisedPrice: record.lengthRaisedPrice ?? false,
|
|
24453
|
+
cutOffSamples: record.cutOffSamples ?? 0,
|
|
24454
|
+
noRecommendationReason: record.noRecommendationReason ?? null,
|
|
24455
|
+
recommendedDeclaredHeadroomBps: record.recommendedDeclaredHeadroomBps ?? null,
|
|
24456
|
+
recommendedEstimateCoins: record.recommendedEstimateCoins ?? null,
|
|
24457
|
+
recommendedPerCallMaxCoins: record.recommendedPerCallMaxCoins ?? null,
|
|
24681
24458
|
savedTo: LLM_BENCH_FILE
|
|
24682
24459
|
});
|
|
24683
|
-
if (record.
|
|
24460
|
+
if (record.recommendedEstimateCoins === null) process.exitCode = 1;
|
|
24684
24461
|
return;
|
|
24685
24462
|
}
|
|
24686
24463
|
log.plain(c.bold("genex llm price"));
|
|
24687
24464
|
log.dim(` from ${LLM_BENCH_FILE}, benchmarked ${record.ranAt}`);
|
|
24688
24465
|
log.plain("");
|
|
24689
24466
|
printRecommendation(log, record);
|
|
24690
|
-
if (record.
|
|
24467
|
+
if (record.recommendedEstimateCoins === null) process.exitCode = 1;
|
|
24691
24468
|
}
|
|
24692
24469
|
function writeJsonLine(value) {
|
|
24693
24470
|
process.stdout.write(`${JSON.stringify(value)}
|
|
@@ -24947,7 +24724,7 @@ function runtimeRow(state, lane, toolsOnly = false) {
|
|
|
24947
24724
|
return {
|
|
24948
24725
|
label: "Runtime",
|
|
24949
24726
|
value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}${keyRefused ? ` \xB7 ${c.yellow("provider key refused")}` : ""}`,
|
|
24950
|
-
fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : '
|
|
24727
|
+
fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
|
|
24951
24728
|
};
|
|
24952
24729
|
}
|
|
24953
24730
|
case "off":
|
|
@@ -26903,16 +26680,16 @@ ${c.bold("Usage")}
|
|
|
26903
26680
|
status | cancel. "models" says whether this
|
|
26904
26681
|
stand serves the runtime LLM lane at all, and
|
|
26905
26682
|
at what provider rates. "bench" runs the real
|
|
26906
|
-
model on YOUR OWN
|
|
26907
|
-
attempt
|
|
26683
|
+
model on YOUR OWN coin and reports what each
|
|
26684
|
+
attempt was charged; it needs --max-coins <n>
|
|
26908
26685
|
--user-approved, like any spend the player has
|
|
26909
26686
|
to agree to. "price" reprints the last run's
|
|
26910
|
-
recommended
|
|
26687
|
+
recommended estimateCoins. "status" lists this
|
|
26911
26688
|
project's open calls and the slots they hold;
|
|
26912
26689
|
"cancel <id>" stops an active one. Declare a
|
|
26913
|
-
game's
|
|
26914
|
-
|
|
26915
|
-
|
|
26690
|
+
game's price from a bench, never from a guess:
|
|
26691
|
+
a started attempt is charged in full, including
|
|
26692
|
+
one that fails.
|
|
26916
26693
|
genex player <sub> [options] Answer the in-game model requests you approve, on your
|
|
26917
26694
|
OWN Claude or ChatGPT subscription: install | run |
|
|
26918
26695
|
status | stop | uninstall. "install" signs this machine
|
|
@@ -27367,7 +27144,7 @@ ${c.bold("Examples")}
|
|
|
27367
27144
|
genex shop remove sku_123
|
|
27368
27145
|
genex llm models
|
|
27369
27146
|
genex llm models --all
|
|
27370
|
-
genex llm bench "Reply with one short taunt." --samples 5 --max-
|
|
27147
|
+
genex llm bench "Reply with one short taunt." --samples 5 --max-coins 20 --user-approved
|
|
27371
27148
|
genex llm price
|
|
27372
27149
|
genex llm status
|
|
27373
27150
|
genex llm cancel <id>
|
|
@@ -27604,24 +27381,22 @@ ${c.bold("How the allowance works")}
|
|
|
27604
27381
|
|
|
27605
27382
|
Every number printed is live from your account; prices can change without a deploy.
|
|
27606
27383
|
`,
|
|
27607
|
-
llm: `${c.bold("genex llm")} \u2014 in-game model calls: is the lane live, what does one call cost, what
|
|
27384
|
+
llm: `${c.bold("genex llm")} \u2014 in-game model calls: is the lane live, what does one call cost, what should the game declare?
|
|
27608
27385
|
|
|
27609
27386
|
${c.bold("Usage")}
|
|
27610
27387
|
genex llm models [--all] [--json] Which models this stand serves, their provider rates per
|
|
27611
|
-
million tokens, the
|
|
27612
|
-
|
|
27613
|
-
|
|
27614
|
-
|
|
27615
|
-
|
|
27616
|
-
|
|
27617
|
-
|
|
27618
|
-
|
|
27619
|
-
|
|
27620
|
-
|
|
27621
|
-
|
|
27622
|
-
|
|
27623
|
-
allowance does not count. Needs a folder linked to a
|
|
27624
|
-
game you own, and a balance that covers one attempt.
|
|
27388
|
+
million tokens, and the headroom it recommends over a
|
|
27389
|
+
benchmark. The catalog is synced from OpenRouter, so the
|
|
27390
|
+
FEATURED short list is printed by default and --all
|
|
27391
|
+
prints every row (--json always carries every row). Off
|
|
27392
|
+
on this stand: it says so and exits 0 \u2014 build the feature
|
|
27393
|
+
without a model rather than promising a 404.
|
|
27394
|
+
genex llm bench "<prompt>" --max-coins <n> --user-approved [options]
|
|
27395
|
+
Run the real model <n> times through the development
|
|
27396
|
+
lane and report what each attempt was CHARGED. Refused
|
|
27397
|
+
without both flags, before any network call: this spends
|
|
27398
|
+
YOUR OWN COIN, a wallet the build's asset allowance
|
|
27399
|
+
cannot see. Needs a folder linked to a game you own.
|
|
27625
27400
|
genex llm price [--json] The recommendation from the last bench in this folder.
|
|
27626
27401
|
Reads a file; spends nothing.
|
|
27627
27402
|
genex llm status [--json] This project's open development calls: active, or
|
|
@@ -27638,32 +27413,29 @@ ${c.bold("Bench options")}
|
|
|
27638
27413
|
--samples <n> How many real attempts (default ${DEFAULT_BENCH_SAMPLES}). More samples, better p95.
|
|
27639
27414
|
--schema <file> A JSON file holding the output schema; implies JSON output.
|
|
27640
27415
|
--json-output Ask for JSON without a schema. --text Ask for text (the default).
|
|
27641
|
-
--max-
|
|
27416
|
+
--max-coins <n> The per-attempt ceiling YOU approved \u2014 a bench runs on your own coin,
|
|
27642
27417
|
never a player's. The run's worst case is that number times --samples,
|
|
27643
|
-
and it is printed before anything starts.
|
|
27644
|
-
name, read as the same number.)
|
|
27418
|
+
and it is printed before anything starts.
|
|
27645
27419
|
--timeout <seconds> Bounds the wait for each attempt.
|
|
27646
27420
|
--json One machine-readable object; every key is always present, null when
|
|
27647
27421
|
the CLI could not learn it.
|
|
27648
27422
|
|
|
27649
27423
|
${c.bold("Why a benchmark and not an estimate")}
|
|
27650
|
-
|
|
27651
|
-
|
|
27652
|
-
|
|
27653
|
-
|
|
27654
|
-
|
|
27655
|
-
|
|
27656
|
-
|
|
27657
|
-
|
|
27658
|
-
|
|
27659
|
-
recommendation, because those multipliers are not the CLI's to invent.
|
|
27424
|
+
\`estimateCoins\` is a PRICE, not a guess: declaring it charges it, in full, even when the
|
|
27425
|
+
attempt fails, is cancelled, or stops at its budget. And three layers sit between the
|
|
27426
|
+
provider's rate card and that number \u2014 the provider's cost, the platform's tariff (already
|
|
27427
|
+
inside every charged coin), and your headroom. The bench itself is billed to YOU, on the
|
|
27428
|
+
development lane; the player-funded lane is what the published game uses. Adding a margin
|
|
27429
|
+
to the provider's rate silently drops the tariff and under-prices every call, so the
|
|
27430
|
+
recommendation is built from CHARGED COINS and the headroom comes from the server. A stand
|
|
27431
|
+
that serves no headroom gets no recommendation, because the multiplier is not the CLI's to
|
|
27432
|
+
invent.
|
|
27660
27433
|
|
|
27661
|
-
|
|
27662
|
-
|
|
27663
|
-
A sample cut off at your --max-
|
|
27664
|
-
|
|
27665
|
-
|
|
27666
|
-
fee included, which is the honest estimate a budget's disclosure is built on.
|
|
27434
|
+
The declared price also decides how LONG the answer may be: it sizes each call's output
|
|
27435
|
+
allowance. So the recommendation is never below the smallest price that leaves room for
|
|
27436
|
+
the benchmarked answer, as the server computes it. A sample cut off at your --max-coins
|
|
27437
|
+
(provider_token_limit) is not a sample \u2014 re-run with a higher --max-coins; an answer longer
|
|
27438
|
+
than one call on this stand may produce gets no price at all \u2014 ask for a shorter one.
|
|
27667
27439
|
|
|
27668
27440
|
Results land in ${LLM_BENCH_FILE} \u2014 gitignored, machine-local, re-read by \`genex llm price\`.
|
|
27669
27441
|
`,
|
|
@@ -27878,13 +27650,11 @@ function parseArgs(argv) {
|
|
|
27878
27650
|
"--limit",
|
|
27879
27651
|
// `genex budget --assets <credits>` — the allowance the player approved.
|
|
27880
27652
|
"--assets",
|
|
27881
|
-
// `genex llm bench` value flags. `--max-
|
|
27882
|
-
// shape as `--assets
|
|
27883
|
-
// count a bench. `--max-coins` is its deprecated name (same number).
|
|
27653
|
+
// `genex llm bench` value flags. `--max-coins` is the COIN half of the
|
|
27654
|
+
// same approval shape as `--assets` (a different economy, its own gate).
|
|
27884
27655
|
"--model",
|
|
27885
27656
|
"--samples",
|
|
27886
27657
|
"--schema",
|
|
27887
|
-
"--max-credits",
|
|
27888
27658
|
"--max-coins",
|
|
27889
27659
|
// `genex player install --label <name>` — what this machine is called in
|
|
27890
27660
|
// the dashboard. Its own key rather than the generic 2nd positional,
|
|
@@ -28399,14 +28169,12 @@ function applyValueFlag(options, flag, value) {
|
|
|
28399
28169
|
options.samples = n;
|
|
28400
28170
|
break;
|
|
28401
28171
|
}
|
|
28402
|
-
case "--max-credits":
|
|
28403
28172
|
case "--max-coins": {
|
|
28404
28173
|
const n = Number(value);
|
|
28405
28174
|
if (!Number.isInteger(n) || n < 1) {
|
|
28406
|
-
throw new Error(`Invalid
|
|
28175
|
+
throw new Error(`Invalid --max-coins value: ${value} (whole coin, 1 or more)`);
|
|
28407
28176
|
}
|
|
28408
|
-
|
|
28409
|
-
else options.maxCoins = n;
|
|
28177
|
+
options.maxCoins = n;
|
|
28410
28178
|
break;
|
|
28411
28179
|
}
|
|
28412
28180
|
case "--timeout": {
|
|
@@ -28698,9 +28466,9 @@ async function main() {
|
|
|
28698
28466
|
await runBudget(parsed.options);
|
|
28699
28467
|
break;
|
|
28700
28468
|
// The in-game model lane: is it live here, what does one call really
|
|
28701
|
-
// cost, what
|
|
28702
|
-
//
|
|
28703
|
-
//
|
|
28469
|
+
// cost, what should the game declare. `bench` is the only CLI command
|
|
28470
|
+
// that spends COIN, so it carries its own --max-coins/--user-approved
|
|
28471
|
+
// gate — the credits allowance cannot see that wallet.
|
|
28704
28472
|
case "llm":
|
|
28705
28473
|
await runLlm(parsed.options);
|
|
28706
28474
|
break;
|