@genex-ai/cli-demo 1.36.0-dev.772 → 1.36.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{blender-mcp-ZRCMYTXI.js → blender-mcp-NRJ5FI22.js} +2 -2
- package/dist/{blender-serve-EH2MOW3U.js → blender-serve-66OHC4ML.js} +1 -1
- package/dist/{chunk-Y6WO5FJZ.js → chunk-GOP5FRSJ.js} +2 -2
- package/dist/{chunk-K4EHXOZK.js → chunk-QI7FIYBY.js} +1 -1
- package/dist/index.js +216 -448
- package/package.json +1 -1
- package/templates/skills/genex-game-director/SKILL.md +1 -1
- package/templates/skills/genex-getting-started/SKILL.md +2 -2
- package/templates/skills/genex-llm-in-games/SKILL.md +72 -209
- package/templates/skills/genex-llm-in-games/references/pricing.md +85 -126
- package/templates/skills/genex-tool-llm/SKILL.md +9 -11
- package/templates/skills/genex-updates/SKILL.md +1 -1
package/dist/index.js
CHANGED
|
@@ -44,7 +44,7 @@ import {
|
|
|
44
44
|
writeSecretFile,
|
|
45
45
|
writeUserToken,
|
|
46
46
|
writeWorkspace
|
|
47
|
-
} from "./chunk-
|
|
47
|
+
} from "./chunk-GOP5FRSJ.js";
|
|
48
48
|
import {
|
|
49
49
|
CLI_CHANNEL,
|
|
50
50
|
DEFAULT_API_URL,
|
|
@@ -65,7 +65,7 @@ import {
|
|
|
65
65
|
normalizeApiOrigin,
|
|
66
66
|
originKey,
|
|
67
67
|
resolveAgentTargets
|
|
68
|
-
} from "./chunk-
|
|
68
|
+
} from "./chunk-QI7FIYBY.js";
|
|
69
69
|
|
|
70
70
|
// src/instrument.ts
|
|
71
71
|
import * as Sentry from "@sentry/node";
|
|
@@ -904,7 +904,7 @@ Important note: put soul into your creations, with many details and love. Aim to
|
|
|
904
904
|
21. A turn ends in exactly one of two ways: on a question the player must answer before the next step can be chosen, or on a handoff \u2014 the draft link, what changed, what to try, and what is still open. Never close a turn on a promise. "I'll keep building", "while I keep working", "adding that now" are things you say and then DO before you stop; if you are stopping, say that you have stopped and what remains. While the Build plan's \`Now:\` line has open work and no answer is needed from the player, the turn is not over: preview, hand off, and start the next milestone in the same turn, until the requested outcome is reached or the platform's own budget ends the session. If a stop was forced on you anyway, the first line of your next turn is the promise you left and where it stands.
|
|
905
905
|
22. **What the player sees must read as the thing it is**, in this game's own style: a building as that building, a person as a person, a prop as that prop, a surface as its material \u2014 and a coloured box, a capsule, or a flat grey block standing in for one of them is never its finished version. Generation is the default route to that bar wherever code will not honestly reach it: the player's character, whatever the request names, whatever the player walks up to, enters, or interacts with, and the music and sound underneath \u2014 reach for the belt on your own judgment, without stopping to ask; a status line saying what you queued is enough, and \`--no-wait\` keeps you building while it lands. Procedural code is a first-class engine for what is structural, repeated, distant, or parametric \u2014 terrain, sky, fences, paving, modular kits, filler \u2014 and for anything else only when the result meets the same bar and you have looked at it in a capture before its row says \`landed\`. Start with one reusable implementation per distinct required gameplay or visual role. Background populations default to reusable procedural bodies and motion or compatible existing rigged assets. Catalog clips need a compatible skeleton and may be paid. Reserve paid custom bodies for the player and characters examined or interacted with closely, while honoring explicit user requests. Add variants only for an unmet requirement; unspent allowance is not unfinished work. Allowance questions use timeoutPolicy no-consent: silence does not raise the allowance. Record role, reuse and completion criteria in the existing Assets table. Decide per object, not per category, and write the route on each Assets row in \`DESIGN.md\` \u2014 the paid flow for generated pieces, the procedural flow with its local path for code-built ones. An Assets table with no generation in it is a decision, not a default: record it as \`Generation: none \u2014 <why code alone reaches the bar here>\` or generate.
|
|
906
906
|
23. **The screenshots you take to look at your own work go in \`.genex/scratch/\`.** Captures, render comparisons, before/after strips, traces and metrics dumps are how you SEE what you built \u2014 they are not part of what you built, and \`genex preview\` pushes the whole folder to the game's source, every time, forever. \`.genex/scratch/\` already exists for this and is already ignored, so there is nothing to set up: write the capture as \`.genex/scratch/arena-before.png\` and read it back from there. Never put them in a folder of your own at the top level: \`progress/\`, \`reports/\`, \`shots/\` and their kind are pushed like source and become permanent weight in the game and in every remix of it \u2014 one real game reached 11 GB and 10,332 committed screenshots exactly this way. If you find such a folder already there, add it to \`.gitignore\`; never delete the player's files (law 18). One file in \`.genex/scratch/\` IS read, by the platform: \`.genex/scratch/cover.png\` is the game's cover \u2014 the frame you would put on its poster: real gameplay at its signature moment, well lit, no menus, popups or busy HUD, landscape 16:9, and never any text (the page sets the title beside it). The next \`genex preview\` makes it the cover; save a better moment over it whenever one comes along, and if your browser tool can set the cover itself, use that instead.
|
|
907
|
-
24. **Benchmark before declaring an in-game LLM
|
|
907
|
+
24. **Benchmark before declaring an in-game LLM price.** If the game calls a language model while the player plays \u2014 an NPC answering in its own words, a quest written for this save, a judge reading what the player typed \u2014 the PLAYER pays for it, and \`estimateCoins\` is the FIXED price of a started attempt: charged whether it succeeds, fails, is canceled, or stops at its own budget. Never pick that number by judgement or from a vendor's rate card. Run the real prompt on your own coins \u2014 \`npx genex llm bench "<the prompt>" --schema <file> --samples 3 --max-coins <n> --user-approved\` \u2014 and declare the figure it recommends, recorded in \`DESIGN.md\` with its model and date; re-benchmark whenever the prompt, the schema or the model changes, because a prompt edit is a price change. Check the lane first with \`npx genex llm models\`: where it is off the routes 404, so build the feature behind a graceful unavailable state and say so in the handoff. A repeated call is a standing budget the player approves once (\`requestSpendGrant()\`, then \`generate({ grantId })\`), never a popup per turn, and its disclosed call rate is computed from the game's own loop. Model output is data the game validates against its own expectation \u2014 never executed, and never authority over coin, items, entitlements or rewards. Load \`$genex-llm-in-games\` before writing any of it.
|
|
908
908
|
${CONTRACT_END}
|
|
909
909
|
`;
|
|
910
910
|
var TOOLS_CONTRACT_HEAD = `# Genex Tools (always in effect in this folder)
|
|
@@ -918,7 +918,7 @@ var TOOLS_CONTRACT_OFFER = `5. **When the game is built and playable \u2014 neve
|
|
|
918
918
|
var TOOLS_CONTRACT_HOSTED_LAWS = `5. **This game is hosted on Genex now, and publishing is glue \u2014 never a rebuild.** The game ships exactly as the user built it: \`initEmbed()\` first in the boot code (the \`genex-threejs-embed-auth\` card has the call; the renderer draws before any \`await waitForPlayer()\`), a static build into \`dist/\` with relative asset paths, then \`npx genex preview\`. Never restructure, reformat or "improve" the game to ship it, never scaffold a new app around it, and never start a design document or a build plan for it \u2014 the user's own process is theirs. Load the \`genex-tool-publish\` card before the first preview; it owns the vocabulary, the links and the limits.
|
|
919
919
|
6. **Ship first, then report.** \`preview\` prints a preflight \u2014 phone memory, a missing volume slider, an asset nobody wired, a viewport line. It never blocks a deploy, so push the build as it is, hand over the link, THEN relay each preflight line to the user in one plain sentence with an offer to fix it, and fix one only on their yes. The preflight is a report for the user, not a to-do list for you.
|
|
920
920
|
7. **The link is the game's page, and there are two versions.** After every preview give the user \`<dashboard>/draft/<slug>\` \u2014 \`dashboardOrigins[0]\` and \`slug\` from \`.genex/project.json\` \u2014 never localhost, a file path or the bare play origin. \`preview\` updates the draft and never touches what players are on; the first release is \`npx genex publish\` (it lists the game); after that "publish it", "update it" and "yes" all mean \`npx genex promote\` \u2014 the exact draft build, no rebuild. Ask once per round of work, in one line, and keep working while you wait.`;
|
|
921
|
-
var TOOLS_CONTRACT_LLM_OFFER = `A model running while people play is built into the Genex platform \u2014 the player pays, with
|
|
921
|
+
var TOOLS_CONTRACT_LLM_OFFER = `A model running while people play is built into the Genex platform \u2014 the player pays, with Genex coins or their own Claude/ChatGPT subscription, and approves it on a Genex sheet; your game just calls \`generate()\`. Want it that way?`;
|
|
922
922
|
function toolsContractLlmLaw(n, hosted) {
|
|
923
923
|
const onYes = hosted ? "On a yes, load `$genex-tool-llm` and then `$genex-llm-in-games`, which owns the build." : "On a yes, load `$genex-tool-llm`, which owns the rest: it checks the lane with `npx genex llm models` first, then runs `npx genex init --convert` (a hosted game is a static browser build \u2014 a local server holding a key can never ship).";
|
|
924
924
|
return `${n}. **A model running while people PLAY is a platform feature \u2014 never something you wire onto the user's own meter.** When a request implies one (NPCs that talk or decide in their own words, content written from what the player types, a prompt box in the game, "let the player pick a model"), recognise it, say in ONE line: "${TOOLS_CONTRACT_LLM_OFFER}" \u2014 then ASK and wait. ${onYes} On a no, build the authored version instead. Either way, never ship a game that calls a model on a key or an account of the user's own: every visitor would spend their money with nobody approving it, and the credential is readable in the bundle. Player-funded generation is text and JSON only \u2014 3D, images, video and audio stay the asset lanes above, on the user's meter.`;
|
|
@@ -23140,7 +23140,7 @@ async function runBlender(opts) {
|
|
|
23140
23140
|
return 1;
|
|
23141
23141
|
}
|
|
23142
23142
|
if (sub === "serve") {
|
|
23143
|
-
const { serveLocalBlender } = await import("./blender-serve-
|
|
23143
|
+
const { serveLocalBlender } = await import("./blender-serve-66OHC4ML.js");
|
|
23144
23144
|
const port = Number(process.env.GENEX_BLENDER_PORT ?? 8088);
|
|
23145
23145
|
log.step(`Starting a local Blender service on port ${port}`);
|
|
23146
23146
|
log.plain(
|
|
@@ -23149,7 +23149,7 @@ async function runBlender(opts) {
|
|
|
23149
23149
|
return serveLocalBlender({ port, log });
|
|
23150
23150
|
}
|
|
23151
23151
|
if (sub === "mcp") {
|
|
23152
|
-
const { runBlenderMcp } = await import("./blender-mcp-
|
|
23152
|
+
const { runBlenderMcp } = await import("./blender-mcp-NRJ5FI22.js");
|
|
23153
23153
|
return runBlenderMcp();
|
|
23154
23154
|
}
|
|
23155
23155
|
if (sub === "seat") {
|
|
@@ -23657,12 +23657,7 @@ import fs35 from "fs/promises";
|
|
|
23657
23657
|
import path36 from "path";
|
|
23658
23658
|
import { randomUUID as randomUUID2 } from "crypto";
|
|
23659
23659
|
var SUBS3 = ["models", "bench", "price", "status", "cancel"];
|
|
23660
|
-
var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN
|
|
23661
|
-
var LLM_MAX_COINS_DEPRECATED = "--max-coins is deprecated: it is read as --max-credits, the same number \u2014 a bench spends credits now.";
|
|
23662
|
-
function benchBalanceShortLine(spendable, maxCredits) {
|
|
23663
|
-
return `Your spendable credits (${spendable}) cannot cover one attempt at --max-credits ${maxCredits} \u2014 nothing was sent, nothing was spent. Add credits, or re-run with a lower --max-credits once the person at the keyboard agrees to it.`;
|
|
23664
|
-
}
|
|
23665
|
-
var LLM_BENCH_UNVERIFIED_LINE = "Credits pay only for a verified email, and this account's is not verified yet (credits_unverified) \u2014 nothing was sent, nothing was spent. Verify it, then re-run.";
|
|
23660
|
+
var LLM_BENCH_APPROVAL_REQUIRED = "STOP: `genex llm bench` runs the real model and spends YOUR OWN coin \u2014 your wallet, not a player's, and one this build's asset allowance cannot see. Re-run with --max-coins <n> --user-approved only after the person at the keyboard agreed to the number.";
|
|
23666
23661
|
var LLM_LANE_OFF_LINE = "The runtime LLM lane is off on this stand \u2014 in-game generate() answers 404 here, and nothing can be benchmarked.";
|
|
23667
23662
|
var LLM_TOOLS_CONVERT_HINT = [
|
|
23668
23663
|
"In-game model calls need a hosted game. This folder is a Genex Tools workspace, so nothing here",
|
|
@@ -23675,7 +23670,7 @@ function serverSlotCounts(body) {
|
|
|
23675
23670
|
return { held: body.slotsHeld, limit: body.slotLimit };
|
|
23676
23671
|
}
|
|
23677
23672
|
var PENDING_SLOT_RELEASE_MINUTES = 10;
|
|
23678
|
-
var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so
|
|
23673
|
+
var LLM_STATUS_LEGEND = `"awaiting bill" = the call has stopped but the provider's bill is not final, so its coin stays held until the bill resolves; such a row stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`;
|
|
23679
23674
|
var LLM_STATUS_NO_COUNT_LINE = "This stand does not report how many slots are held across your projects, so no count is shown.";
|
|
23680
23675
|
var LLM_CANCEL_AWAITING_BILL = "this call already stopped; its bill is awaiting the provider";
|
|
23681
23676
|
var LLM_CANCEL_ALREADY_FINAL = "this call is already final";
|
|
@@ -23689,33 +23684,11 @@ var DEFAULT_SAMPLE_TIMEOUT_SEC = 180;
|
|
|
23689
23684
|
function parseProviderKey(value) {
|
|
23690
23685
|
return value === "ok" || value === "rejected" || value === "unknown" ? value : null;
|
|
23691
23686
|
}
|
|
23692
|
-
function bpsOrNull(value) {
|
|
23693
|
-
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
23694
|
-
}
|
|
23695
|
-
function standingCreditGrantOrNull(value) {
|
|
23696
|
-
if (!value || typeof value !== "object") return null;
|
|
23697
|
-
const v = value;
|
|
23698
|
-
const whole = (x) => typeof x === "number" && Number.isSafeInteger(x) && x >= 0;
|
|
23699
|
-
if (!whole(v.minCredits) || !whole(v.maxCredits) || !whole(v.stepCredits) || !whole(v.defaultCredits)) return null;
|
|
23700
|
-
return {
|
|
23701
|
-
minCredits: v.minCredits,
|
|
23702
|
-
maxCredits: v.maxCredits,
|
|
23703
|
-
stepCredits: v.stepCredits,
|
|
23704
|
-
defaultCredits: v.defaultCredits,
|
|
23705
|
-
...whole(v.minUsdCents) ? { minUsdCents: v.minUsdCents } : {},
|
|
23706
|
-
...whole(v.maxUsdCents) ? { maxUsdCents: v.maxUsdCents } : {},
|
|
23707
|
-
...whole(v.stepUsdCents) ? { stepUsdCents: v.stepUsdCents } : {}
|
|
23708
|
-
};
|
|
23709
|
-
}
|
|
23710
23687
|
async function readRuntimeLane(apiUrl, token) {
|
|
23711
23688
|
const empty = (state) => ({
|
|
23712
23689
|
state,
|
|
23713
23690
|
models: null,
|
|
23714
23691
|
recommendedDeclaredHeadroomBps: null,
|
|
23715
|
-
currency: null,
|
|
23716
|
-
feeBps: null,
|
|
23717
|
-
usdCentsPerCredit: null,
|
|
23718
|
-
standingCreditGrant: null,
|
|
23719
23692
|
externalProviders: null,
|
|
23720
23693
|
total: null,
|
|
23721
23694
|
featuredCount: null,
|
|
@@ -23734,11 +23707,7 @@ async function readRuntimeLane(apiUrl, token) {
|
|
|
23734
23707
|
return {
|
|
23735
23708
|
state: "live",
|
|
23736
23709
|
models: body.models,
|
|
23737
|
-
recommendedDeclaredHeadroomBps:
|
|
23738
|
-
currency: body.currency === "credits" || body.currency === "coins" ? body.currency : null,
|
|
23739
|
-
feeBps: bpsOrNull(body.feeBps),
|
|
23740
|
-
usdCentsPerCredit: typeof body.usdCentsPerCredit === "number" && Number.isSafeInteger(body.usdCentsPerCredit) && body.usdCentsPerCredit >= 1 ? body.usdCentsPerCredit : null,
|
|
23741
|
-
standingCreditGrant: standingCreditGrantOrNull(body.standingCreditGrant),
|
|
23710
|
+
recommendedDeclaredHeadroomBps: typeof body.recommendedDeclaredHeadroomBps === "number" && Number.isFinite(body.recommendedDeclaredHeadroomBps) ? body.recommendedDeclaredHeadroomBps : null,
|
|
23742
23711
|
externalProviders: Array.isArray(body.externalProviders) ? body.externalProviders : null,
|
|
23743
23712
|
total: typeof body.total === "number" && Number.isFinite(body.total) ? body.total : null,
|
|
23744
23713
|
featuredCount: typeof body.featuredCount === "number" && Number.isFinite(body.featuredCount) ? body.featuredCount : null,
|
|
@@ -23754,45 +23723,18 @@ function percentile2(values, p) {
|
|
|
23754
23723
|
const rank2 = Math.ceil(p / 100 * sorted.length);
|
|
23755
23724
|
return sorted[Math.min(sorted.length - 1, Math.max(0, rank2 - 1))];
|
|
23756
23725
|
}
|
|
23757
|
-
function
|
|
23758
|
-
|
|
23759
|
-
const sorted = [...values].sort((a, b) => a < b ? -1 : a > b ? 1 : 0);
|
|
23760
|
-
const rank2 = Math.ceil(p / 100 * sorted.length);
|
|
23761
|
-
return sorted[Math.min(sorted.length - 1, Math.max(0, rank2 - 1))];
|
|
23726
|
+
function recommendEstimateCoins(p95, headroomBps) {
|
|
23727
|
+
return Math.max(1, Math.ceil(p95 * (1e4 + headroomBps) / 1e4));
|
|
23762
23728
|
}
|
|
23763
|
-
|
|
23764
|
-
|
|
23765
|
-
if (costPicos <= 0n) return 1;
|
|
23766
|
-
const num = costPicos * BigInt(1e4 + terms.feeBps) * BigInt(1e4 + terms.headroomBps);
|
|
23767
|
-
const den = 100000000n * BigInt(terms.usdCentsPerCredit) * PICOS_PER_CENT;
|
|
23768
|
-
return Math.max(1, Number((num + den - 1n) / den));
|
|
23769
|
-
}
|
|
23770
|
-
function averageCreditsPerCall(costsPicos, terms) {
|
|
23771
|
-
if (costsPicos.length === 0) return null;
|
|
23772
|
-
const sum = costsPicos.reduce((a, b) => a + b, 0n);
|
|
23773
|
-
return Number(sum * BigInt(1e4 + terms.feeBps)) / Number(BigInt(costsPicos.length) * 10000n * BigInt(terms.usdCentsPerCredit) * PICOS_PER_CENT);
|
|
23774
|
-
}
|
|
23775
|
-
function perCallEstimateCreditsFor(costsPicos, terms) {
|
|
23776
|
-
if (costsPicos.length === 0) return null;
|
|
23777
|
-
const sum = costsPicos.reduce((a, b) => a + b, 0n);
|
|
23778
|
-
const num = sum * BigInt(1e4 + terms.feeBps);
|
|
23779
|
-
const den = BigInt(costsPicos.length) * 10000n * BigInt(terms.usdCentsPerCredit) * PICOS_PER_CENT;
|
|
23780
|
-
return Math.max(1, Number((num + den - 1n) / den));
|
|
23781
|
-
}
|
|
23782
|
-
function formatCredits(value) {
|
|
23783
|
-
if (!Number.isFinite(value)) return "\u2014";
|
|
23784
|
-
if (value >= 100) return String(Math.round(value));
|
|
23785
|
-
return String(Number(value.toPrecision(value < 1 ? 2 : 3)));
|
|
23786
|
-
}
|
|
23787
|
-
function usdFromPicos(picos) {
|
|
23788
|
-
if (picos === null || picos === void 0) return "unknown";
|
|
23789
|
-
const value = typeof picos === "bigint" ? picos : /^\d+$/.test(picos) ? BigInt(picos) : null;
|
|
23790
|
-
if (value === null) return "unknown";
|
|
23791
|
-
return `$${(Number(value) / 1e12).toFixed(6)}`;
|
|
23729
|
+
function recommendPerCallMaxCoins(p95, max, headroomBps) {
|
|
23730
|
+
return Math.max(recommendEstimateCoins(p95, headroomBps), recommendEstimateCoins(max, headroomBps));
|
|
23792
23731
|
}
|
|
23793
23732
|
function usdPerMillion(value) {
|
|
23794
23733
|
return `$${value.toFixed(value < 1 ? 4 : 2)}`;
|
|
23795
23734
|
}
|
|
23735
|
+
function usd(value) {
|
|
23736
|
+
return value === null ? "unknown" : `$${value.toFixed(6)}`;
|
|
23737
|
+
}
|
|
23796
23738
|
async function runLlm(opts = {}) {
|
|
23797
23739
|
const log = createLogger({ quiet: opts.quiet || opts.json });
|
|
23798
23740
|
const cwd = opts.cwd ?? process.cwd();
|
|
@@ -23818,23 +23760,16 @@ async function runLlm(opts = {}) {
|
|
|
23818
23760
|
return;
|
|
23819
23761
|
}
|
|
23820
23762
|
if (sub === "bench") {
|
|
23821
|
-
if (opts.maxCoins
|
|
23822
|
-
`);
|
|
23823
|
-
if (opts.maxCoins !== void 0 && opts.maxCredits !== void 0 && opts.maxCoins !== opts.maxCredits) {
|
|
23824
|
-
fail5("llm bench", `Pass --max-credits alone \u2014 --max-coins is its deprecated name, and the two disagree (${opts.maxCredits} vs ${opts.maxCoins}).`);
|
|
23825
|
-
return;
|
|
23826
|
-
}
|
|
23827
|
-
const approvedCredits = opts.maxCredits ?? opts.maxCoins;
|
|
23828
|
-
if (approvedCredits === void 0 || opts.userApproved !== true) {
|
|
23763
|
+
if (opts.maxCoins === void 0 || opts.userApproved !== true) {
|
|
23829
23764
|
fail5("llm bench", LLM_BENCH_APPROVAL_REQUIRED);
|
|
23830
23765
|
return;
|
|
23831
23766
|
}
|
|
23832
|
-
if (!Number.isInteger(
|
|
23833
|
-
fail5("llm bench", `--max-
|
|
23767
|
+
if (!Number.isInteger(opts.maxCoins) || opts.maxCoins < 1) {
|
|
23768
|
+
fail5("llm bench", `--max-coins takes a whole number of coin, 1 or more (got ${String(opts.maxCoins)}).`);
|
|
23834
23769
|
return;
|
|
23835
23770
|
}
|
|
23836
23771
|
if (!opts.benchPrompt?.trim()) {
|
|
23837
|
-
fail5("llm bench", '`genex llm bench` needs the prompt your game would send, e.g. genex llm bench "<prompt>" --max-
|
|
23772
|
+
fail5("llm bench", '`genex llm bench` needs the prompt your game would send, e.g. genex llm bench "<prompt>" --max-coins <n> --user-approved.');
|
|
23838
23773
|
return;
|
|
23839
23774
|
}
|
|
23840
23775
|
if (opts.jsonOutput && opts.textOutput) {
|
|
@@ -23926,10 +23861,6 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23926
23861
|
models: null,
|
|
23927
23862
|
externalProviders: null,
|
|
23928
23863
|
recommendedDeclaredHeadroomBps: null,
|
|
23929
|
-
currency: null,
|
|
23930
|
-
feeBps: null,
|
|
23931
|
-
usdCentsPerCredit: null,
|
|
23932
|
-
standingCreditGrant: null,
|
|
23933
23864
|
total: null,
|
|
23934
23865
|
featuredCount: null,
|
|
23935
23866
|
// The stand was never asked, so the verdict is unknown — but the key is
|
|
@@ -23955,12 +23886,6 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23955
23886
|
models: lane.models,
|
|
23956
23887
|
externalProviders: lane.externalProviders,
|
|
23957
23888
|
recommendedDeclaredHeadroomBps: lane.recommendedDeclaredHeadroomBps,
|
|
23958
|
-
// The credit terms a call is billed on, the server's — null on a stand
|
|
23959
|
-
// that predates credit billing, never a number the CLI filled in.
|
|
23960
|
-
currency: lane.currency,
|
|
23961
|
-
feeBps: lane.feeBps,
|
|
23962
|
-
usdCentsPerCredit: lane.usdCentsPerCredit,
|
|
23963
|
-
standingCreditGrant: lane.standingCreditGrant,
|
|
23964
23889
|
total: lane.total ?? lane.models?.length ?? null,
|
|
23965
23890
|
featuredCount: lane.featuredCount ?? (lane.models ? lane.models.filter((m) => m.featured).length : null),
|
|
23966
23891
|
// `null` when the lane is not live or the server predates the probe.
|
|
@@ -24007,9 +23932,8 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
24007
23932
|
);
|
|
24008
23933
|
for (const m of shown) {
|
|
24009
23934
|
const plan = m.personalPlan ? ` \xB7 personal plan: ${m.personalPlan}` : "";
|
|
24010
|
-
const kind = m.kind === "typesafe" ? " \xB7 judge (classifier): judge() under a budget, not generate() or bench" : "";
|
|
24011
23935
|
log.plain(
|
|
24012
|
-
` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens${plan}
|
|
23936
|
+
` ${c.cyan(m.id)} ${m.label} \u2014 ${usdPerMillion(m.inputUsdPerMillion)} in / ${usdPerMillion(m.outputUsdPerMillion)} out per million tokens${plan}`
|
|
24013
23937
|
);
|
|
24014
23938
|
}
|
|
24015
23939
|
if (hiddenCount > 0) {
|
|
@@ -24025,104 +23949,74 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
24025
23949
|
}
|
|
24026
23950
|
log.plain("");
|
|
24027
23951
|
if (lane.recommendedDeclaredHeadroomBps === null) {
|
|
24028
|
-
log.dim(" This stand serves no recommended headroom, so no
|
|
23952
|
+
log.dim(" This stand serves no recommended headroom, so no price can be recommended from a bench.");
|
|
24029
23953
|
} else {
|
|
24030
|
-
log.dim(` Recommended headroom over
|
|
24031
|
-
}
|
|
24032
|
-
if (lane.feeBps === null) {
|
|
24033
|
-
log.dim(" This stand serves no platform fee \u2014 it predates credit billing, so no ceiling can be recommended from a bench.");
|
|
24034
|
-
} else {
|
|
24035
|
-
log.dim(` In-game calls are billed in the player's credits, as used: the provider's cost plus this stand's platform fee (${lane.feeBps} bps).`);
|
|
23954
|
+
log.dim(` Recommended headroom over benchmarked coins on this stand: ${lane.recommendedDeclaredHeadroomBps} bps.`);
|
|
24036
23955
|
}
|
|
24037
23956
|
if (toolsOnly) {
|
|
24038
23957
|
printConvertHint();
|
|
24039
23958
|
return;
|
|
24040
23959
|
}
|
|
24041
|
-
log.dim(" Those per-million rates are the PROVIDER's
|
|
24042
|
-
log.dim(`
|
|
23960
|
+
log.dim(" Those per-million rates are the PROVIDER's; what your game is charged is the platform");
|
|
23961
|
+
log.dim(` tariff on top, which only a real call reveals: ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')}`);
|
|
24043
23962
|
}
|
|
24044
23963
|
function wholeOrNull(value) {
|
|
24045
23964
|
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
24046
23965
|
}
|
|
24047
|
-
function costPicosOf(usage) {
|
|
24048
|
-
if (typeof usage?.costUsdPicos === "string" && /^\d+$/.test(usage.costUsdPicos)) return usage.costUsdPicos;
|
|
24049
|
-
if (typeof usage?.costUsd === "number" && Number.isFinite(usage.costUsd) && usage.costUsd >= 0) {
|
|
24050
|
-
return String(BigInt(Math.round(usage.costUsd * 1e12)));
|
|
24051
|
-
}
|
|
24052
|
-
return null;
|
|
24053
|
-
}
|
|
24054
23966
|
function sampleFrom(id, settled) {
|
|
24055
23967
|
const usage = settled?.usage;
|
|
24056
|
-
const picos = costPicosOf(usage);
|
|
24057
23968
|
return {
|
|
24058
23969
|
id,
|
|
24059
23970
|
status: settled?.status ?? null,
|
|
24060
23971
|
billingStatus: settled?.billingStatus ?? null,
|
|
24061
|
-
|
|
24062
|
-
costUsd: typeof usage?.costUsd === "number" ? usage.costUsd :
|
|
24063
|
-
costUsdPicos: picos,
|
|
23972
|
+
chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
|
|
23973
|
+
costUsd: typeof usage?.costUsd === "number" ? usage.costUsd : null,
|
|
24064
23974
|
error: settled?.error ?? (settled ? null : "timed_out"),
|
|
24065
23975
|
providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null,
|
|
24066
23976
|
inputTokens: wholeOrNull(usage?.inputTokens),
|
|
24067
23977
|
outputTokens: wholeOrNull(usage?.outputTokens),
|
|
24068
23978
|
maxOutputTokens: wholeOrNull(usage?.maxOutputTokens),
|
|
24069
|
-
|
|
23979
|
+
fitCoins: typeof usage?.fitCoins === "number" && Number.isSafeInteger(usage.fitCoins) && usage.fitCoins >= 1 ? usage.fitCoins : null,
|
|
24070
23980
|
fitExceedsCap: typeof usage?.fitExceedsCap === "boolean" ? usage.fitExceedsCap : null,
|
|
24071
23981
|
truncated: typeof usage?.truncated === "boolean" ? usage.truncated : settled?.error === "provider_token_limit" ? true : null
|
|
24072
23982
|
};
|
|
24073
23983
|
}
|
|
24074
23984
|
var PROVIDER_TOKEN_LIMIT = "provider_token_limit";
|
|
24075
|
-
function benchRecommendation(results,
|
|
24076
|
-
const settled = results.filter(
|
|
24077
|
-
|
|
24078
|
-
);
|
|
24079
|
-
const
|
|
24080
|
-
const
|
|
24081
|
-
const fits = settled.map((r) => r.fitCredits).filter((v) => typeof v === "number");
|
|
24082
|
-
const costP95 = percentilePicos(costs, 95);
|
|
24083
|
-
const costMax = costs.length ? costs.reduce((a, b) => b > a ? b : a) : null;
|
|
23985
|
+
function benchRecommendation(results, headroomBps) {
|
|
23986
|
+
const settled = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number");
|
|
23987
|
+
const charged = settled.map((r) => r.chargedCoins);
|
|
23988
|
+
const fits = settled.map((r) => r.fitCoins).filter((v) => typeof v === "number");
|
|
23989
|
+
const p95 = percentile2(charged, 95);
|
|
23990
|
+
const max = charged.length ? Math.max(...charged) : null;
|
|
24084
23991
|
const fitP95 = percentile2(fits, 95);
|
|
24085
23992
|
const fitMax = fits.length ? Math.max(...fits) : null;
|
|
24086
23993
|
const exceedsStandCap = results.some((r) => r.fitExceedsCap === true);
|
|
24087
23994
|
const cutOffSamples = results.filter((r) => r.error === PROVIDER_TOKEN_LIMIT && r.fitExceedsCap !== true).length;
|
|
24088
23995
|
const noRecommendationReason = exceedsStandCap ? "answer_exceeds_stand_cap" : cutOffSamples > 0 ? "answer_cut_off" : null;
|
|
24089
|
-
const
|
|
24090
|
-
const
|
|
24091
|
-
const
|
|
24092
|
-
const billing = feeBps !== null && usdCentsPerCredit !== null ? { feeBps, usdCentsPerCredit } : null;
|
|
24093
|
-
const costBased = full !== null && costP95 !== null ? ceilingCreditsFor(costP95, full) : null;
|
|
24094
|
-
const recommended = costBased === null || noRecommendationReason !== null ? null : Math.max(costBased, fitP95 ?? 0);
|
|
24095
|
-
const ceiling = recommended === null || costMax === null || full === null ? null : Math.max(ceilingCreditsFor(costMax, full), recommended, fitMax ?? 0);
|
|
24096
|
-
const average = billing !== null ? averageCreditsPerCall(costs, billing) : null;
|
|
24097
|
-
const estimate = ceiling === null || billing === null ? null : Math.min(ceiling, perCallEstimateCreditsFor(costs, billing) ?? 1);
|
|
23996
|
+
const chargedBased = p95 !== null && headroomBps !== null ? recommendEstimateCoins(p95, headroomBps) : null;
|
|
23997
|
+
const recommended = chargedBased === null || noRecommendationReason !== null ? null : Math.max(chargedBased, fitP95 ?? 0);
|
|
23998
|
+
const ceiling = recommended === null || p95 === null || max === null || headroomBps === null ? null : Math.max(recommendPerCallMaxCoins(p95, max, headroomBps), recommended, fitMax ?? 0);
|
|
24098
23999
|
return {
|
|
24099
|
-
costs,
|
|
24100
24000
|
charged,
|
|
24101
|
-
|
|
24102
|
-
|
|
24103
|
-
|
|
24104
|
-
chargedP50: percentile2(charged, 50),
|
|
24105
|
-
chargedP95: percentile2(charged, 95),
|
|
24106
|
-
chargedMax: charged.length ? Math.max(...charged) : null,
|
|
24001
|
+
p50: percentile2(charged, 50),
|
|
24002
|
+
p95,
|
|
24003
|
+
max,
|
|
24107
24004
|
fits,
|
|
24108
24005
|
fitP95,
|
|
24109
24006
|
fitMax,
|
|
24110
|
-
|
|
24111
|
-
|
|
24112
|
-
|
|
24113
|
-
|
|
24114
|
-
recommendedPerCallEstimateCredits: estimate,
|
|
24115
|
-
lengthRaisedCeiling: recommended !== null && costBased !== null && recommended > costBased,
|
|
24007
|
+
chargedBasedEstimateCoins: chargedBased,
|
|
24008
|
+
recommendedEstimateCoins: recommended,
|
|
24009
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
24010
|
+
lengthRaisedPrice: recommended !== null && chargedBased !== null && recommended > chargedBased,
|
|
24116
24011
|
cutOffSamples,
|
|
24117
24012
|
exceedsStandCap,
|
|
24118
|
-
noRecommendationReason
|
|
24119
|
-
missingTerms
|
|
24013
|
+
noRecommendationReason
|
|
24120
24014
|
};
|
|
24121
24015
|
}
|
|
24122
|
-
var LLM_LENGTH_RAISED_LINE = "The
|
|
24123
|
-
var LLM_EXCEEDS_STAND_CAP_LINE = "No
|
|
24124
|
-
function benchCutOffLine(cutOff,
|
|
24125
|
-
return `No
|
|
24016
|
+
var LLM_LENGTH_RAISED_LINE = "The declared price also pays for how long the answer may be: at the charged-based price the answer would be cut off, so declare this one.";
|
|
24017
|
+
var LLM_EXCEEDS_STAND_CAP_LINE = "No price recommended: this answer is longer than one call on this stand may produce. Ask for a shorter answer \u2014 fewer fields, shorter strings, a length the prompt states \u2014 and benchmark again.";
|
|
24018
|
+
function benchCutOffLine(cutOff, maxCoins) {
|
|
24019
|
+
return `No price recommended: ${cutOff} sample${cutOff === 1 ? " was" : "s were"} cut off at this ceiling (--max-coins ${maxCoins}) before the answer was finished, so its real length is unknown. Re-run with a higher --max-coins.`;
|
|
24126
24020
|
}
|
|
24127
24021
|
var ACTIVE_STATUSES = /* @__PURE__ */ new Set(["requires_confirmation", "queued", "dispatching", "awaiting_external"]);
|
|
24128
24022
|
function rowState(row) {
|
|
@@ -24133,22 +24027,7 @@ function rowState(row) {
|
|
|
24133
24027
|
function rowHoldsSlot(row) {
|
|
24134
24028
|
if (typeof row.slotHeld === "boolean") return row.slotHeld;
|
|
24135
24029
|
const state = rowState(row);
|
|
24136
|
-
return state === "active" || state === "awaiting bill" && (
|
|
24137
|
-
}
|
|
24138
|
-
function rowMoney(row) {
|
|
24139
|
-
const credits = row.currency === "credits" || row.currency === void 0 && typeof row.reservedCredits === "number" && typeof row.reservedCoins !== "number";
|
|
24140
|
-
if (credits) {
|
|
24141
|
-
return {
|
|
24142
|
-
unit: "credits",
|
|
24143
|
-
reserved: typeof row.reservedCredits === "number" ? row.reservedCredits : null,
|
|
24144
|
-
charged: typeof row.chargedCredits === "number" ? row.chargedCredits : null
|
|
24145
|
-
};
|
|
24146
|
-
}
|
|
24147
|
-
return {
|
|
24148
|
-
unit: "coin",
|
|
24149
|
-
reserved: typeof row.reservedCoins === "number" ? row.reservedCoins : null,
|
|
24150
|
-
charged: typeof row.chargedCoins === "number" ? row.chargedCoins : null
|
|
24151
|
-
};
|
|
24030
|
+
return state === "active" || state === "awaiting bill" && (row.reservedCoins ?? 0) > 0;
|
|
24152
24031
|
}
|
|
24153
24032
|
function providerRefusalStatus(error) {
|
|
24154
24033
|
const m = /^provider_http_(\d{3})$/.exec(error ?? "");
|
|
@@ -24167,7 +24046,7 @@ function formatAge(iso, now = Date.now()) {
|
|
|
24167
24046
|
}
|
|
24168
24047
|
async function runBench(args) {
|
|
24169
24048
|
const { apiUrl, token, projectId, cwd, opts, log } = args;
|
|
24170
|
-
const
|
|
24049
|
+
const maxCoins = opts.maxCoins;
|
|
24171
24050
|
const samples = opts.samples ?? DEFAULT_BENCH_SAMPLES;
|
|
24172
24051
|
const prompt = opts.benchPrompt.trim();
|
|
24173
24052
|
const auth = { Authorization: `Bearer ${token}`, "Content-Type": "application/json" };
|
|
@@ -24189,7 +24068,7 @@ async function runBench(args) {
|
|
|
24189
24068
|
const lane = await readRuntimeLane(apiUrl, token);
|
|
24190
24069
|
if (lane.state !== "live" || !lane.models || lane.models.length === 0) {
|
|
24191
24070
|
if (lane.state === "off") {
|
|
24192
|
-
if (opts.json) writeJsonLine({ ...emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples,
|
|
24071
|
+
if (opts.json) writeJsonLine({ ...emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoins, opts), status: "off", error: null });
|
|
24193
24072
|
else {
|
|
24194
24073
|
log.plain(LLM_LANE_OFF_LINE);
|
|
24195
24074
|
log.dim(" Nothing was spent.");
|
|
@@ -24216,27 +24095,17 @@ async function runBench(args) {
|
|
|
24216
24095
|
);
|
|
24217
24096
|
return;
|
|
24218
24097
|
}
|
|
24219
|
-
const
|
|
24220
|
-
if (walletBefore?.emailVerified === false) {
|
|
24221
|
-
benchFailed(opts, log, LLM_BENCH_UNVERIFIED_LINE);
|
|
24222
|
-
return;
|
|
24223
|
-
}
|
|
24224
|
-
if (walletBefore !== null && walletBefore.spendable < maxCredits) {
|
|
24225
|
-
benchFailed(opts, log, benchBalanceShortLine(walletBefore.spendable, maxCredits));
|
|
24226
|
-
return;
|
|
24227
|
-
}
|
|
24228
|
-
const balanceBefore = walletBefore?.spendable ?? null;
|
|
24098
|
+
const balanceBefore = await readCoinBalance(apiUrl, token);
|
|
24229
24099
|
if (!opts.json) {
|
|
24230
24100
|
log.plain(c.bold("genex llm bench"));
|
|
24231
24101
|
log.dim(` ${apiUrl}`);
|
|
24232
24102
|
log.plain("");
|
|
24233
24103
|
log.plain(` Model ${c.cyan(modelId)}`);
|
|
24234
24104
|
log.plain(` Samples ${samples} real attempt${samples === 1 ? "" : "s"}, ${outputFormat} output`);
|
|
24235
|
-
log.plain(` Approved ${
|
|
24105
|
+
log.plain(` Approved ${maxCoins} coin per attempt \u2014 at worst ${maxCoins * samples} coin for this run`);
|
|
24236
24106
|
log.plain(
|
|
24237
|
-
` Balance ${balanceBefore === null ? "couldn't be read" : `${balanceBefore}
|
|
24107
|
+
` Balance ${balanceBefore === null ? "couldn't be read" : `${balanceBefore} coin spendable`}`
|
|
24238
24108
|
);
|
|
24239
|
-
log.dim(" Billed as used: each attempt is charged its real cost plus the platform fee, rounded up to a whole credit.");
|
|
24240
24109
|
log.plain("");
|
|
24241
24110
|
}
|
|
24242
24111
|
const base = `${apiUrl}/api/runtime/development/projects/${encodeURIComponent(projectId)}/generations`;
|
|
@@ -24262,7 +24131,7 @@ async function runBench(args) {
|
|
|
24262
24131
|
const res = await call(base, {
|
|
24263
24132
|
method: "POST",
|
|
24264
24133
|
body: JSON.stringify({
|
|
24265
|
-
|
|
24134
|
+
maxCoins,
|
|
24266
24135
|
request: {
|
|
24267
24136
|
idempotencyKey: `bench-${randomUUID2()}`,
|
|
24268
24137
|
modelId,
|
|
@@ -24292,14 +24161,13 @@ async function runBench(args) {
|
|
|
24292
24161
|
if (refusedAt !== null) providerRefused++;
|
|
24293
24162
|
if (!opts.json) {
|
|
24294
24163
|
const length = row.outputTokens !== null ? ` \xB7 ${row.outputTokens}${row.maxOutputTokens !== null ? ` of ${row.maxOutputTokens}` : ""} tokens out` : "";
|
|
24295
|
-
const charged = row.chargedCredits === null ? "\u2014" : String(row.chargedCredits);
|
|
24296
24164
|
log.plain(
|
|
24297
|
-
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${
|
|
24165
|
+
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${length}${row.error ? ` \xB7 ${row.error}` : ""}`
|
|
24298
24166
|
);
|
|
24299
24167
|
if (refusedAt !== null) printProviderRefusal(log, refusedAt, row.providerMessage, row.status === "unknown");
|
|
24300
24168
|
if (row.error === PROVIDER_TOKEN_LIMIT) {
|
|
24301
24169
|
log.dim(
|
|
24302
|
-
row.fitExceedsCap === true ? " The answer was cut off at this stand's own output limit \u2014 no
|
|
24170
|
+
row.fitExceedsCap === true ? " The answer was cut off at this stand's own output limit \u2014 no price makes room for it; it is not a sample." : ` The answer was cut off at this ceiling (--max-coins ${maxCoins}) before it was finished \u2014 it was charged, and it is not a sample.`
|
|
24303
24171
|
);
|
|
24304
24172
|
}
|
|
24305
24173
|
}
|
|
@@ -24307,19 +24175,16 @@ async function runBench(args) {
|
|
|
24307
24175
|
} finally {
|
|
24308
24176
|
process.removeListener("SIGINT", onSigint);
|
|
24309
24177
|
}
|
|
24310
|
-
const
|
|
24311
|
-
|
|
24312
|
-
|
|
24313
|
-
|
|
24314
|
-
|
|
24315
|
-
const
|
|
24316
|
-
const settledCount = rec.charged.length;
|
|
24178
|
+
const headroomBps = lane.recommendedDeclaredHeadroomBps;
|
|
24179
|
+
const rec = benchRecommendation(results, headroomBps);
|
|
24180
|
+
const { charged, p50, p95, max } = rec;
|
|
24181
|
+
const settledCount = charged.length;
|
|
24182
|
+
const recommended = rec.recommendedEstimateCoins;
|
|
24183
|
+
const ceiling = rec.recommendedPerCallMaxCoins;
|
|
24317
24184
|
const lengthBlocked = rec.noRecommendationReason !== null;
|
|
24318
|
-
const balanceAfter =
|
|
24319
|
-
const picosOrNull = (v) => v === null ? null : String(v);
|
|
24185
|
+
const balanceAfter = await readCoinBalance(apiUrl, token);
|
|
24320
24186
|
const record = {
|
|
24321
|
-
v:
|
|
24322
|
-
currency: "credits",
|
|
24187
|
+
v: 1,
|
|
24323
24188
|
ranAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
24324
24189
|
apiUrl,
|
|
24325
24190
|
projectId,
|
|
@@ -24328,34 +24193,28 @@ async function runBench(args) {
|
|
|
24328
24193
|
outputFormat,
|
|
24329
24194
|
samples,
|
|
24330
24195
|
completed: settledCount,
|
|
24331
|
-
|
|
24332
|
-
|
|
24333
|
-
|
|
24334
|
-
|
|
24335
|
-
|
|
24336
|
-
|
|
24196
|
+
chargedCoins: charged,
|
|
24197
|
+
p50,
|
|
24198
|
+
p95,
|
|
24199
|
+
max,
|
|
24200
|
+
recommendedDeclaredHeadroomBps: headroomBps,
|
|
24201
|
+
recommendedEstimateCoins: recommended,
|
|
24202
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
24203
|
+
maxCoinsPerSample: maxCoins,
|
|
24204
|
+
fitCoins: rec.fits,
|
|
24337
24205
|
fitP95: rec.fitP95,
|
|
24338
|
-
|
|
24339
|
-
|
|
24340
|
-
feeBps: terms.feeBps,
|
|
24341
|
-
usdCentsPerCredit: terms.usdCentsPerCredit,
|
|
24342
|
-
costBasedMaxCredits: rec.costBasedMaxCredits,
|
|
24343
|
-
recommendedMaxCredits: rec.recommendedMaxCredits,
|
|
24344
|
-
recommendedPerCallMaxCredits: rec.recommendedPerCallMaxCredits,
|
|
24345
|
-
averageCreditsPerCall: rec.averageCreditsPerCall,
|
|
24346
|
-
recommendedPerCallEstimateCredits: rec.recommendedPerCallEstimateCredits,
|
|
24347
|
-
lengthRaisedCeiling: rec.lengthRaisedCeiling,
|
|
24206
|
+
chargedBasedEstimateCoins: rec.chargedBasedEstimateCoins,
|
|
24207
|
+
lengthRaisedPrice: rec.lengthRaisedPrice,
|
|
24348
24208
|
cutOffSamples: rec.cutOffSamples,
|
|
24349
24209
|
noRecommendationReason: rec.noRecommendationReason
|
|
24350
24210
|
};
|
|
24351
24211
|
const savedTo = await saveBench(cwd, record);
|
|
24352
24212
|
if (opts.json) {
|
|
24353
|
-
const ok =
|
|
24213
|
+
const ok = charged.length > 0 && !lengthBlocked;
|
|
24354
24214
|
writeJsonLine({
|
|
24355
24215
|
command: "llm bench",
|
|
24356
24216
|
status: ok ? "ok" : "failed",
|
|
24357
24217
|
error: ok ? null : rec.noRecommendationReason ?? "no_sample_settled",
|
|
24358
|
-
currency: "credits",
|
|
24359
24218
|
apiUrl,
|
|
24360
24219
|
projectId,
|
|
24361
24220
|
modelId,
|
|
@@ -24364,54 +24223,35 @@ async function runBench(args) {
|
|
|
24364
24223
|
schemaPath: opts.schemaPath ?? null,
|
|
24365
24224
|
samples,
|
|
24366
24225
|
completed: settledCount,
|
|
24367
|
-
|
|
24368
|
-
// Spendable credits before and after the run; null when unreadable.
|
|
24226
|
+
maxCoinsPerSample: maxCoins,
|
|
24369
24227
|
balanceBefore,
|
|
24370
24228
|
balanceAfter,
|
|
24371
24229
|
results,
|
|
24372
24230
|
// The create-time refusal that ended the run, or null. Beside the
|
|
24373
24231
|
// samples, never among them: no attempt existed and nothing was spent.
|
|
24374
24232
|
refusal: refusal2,
|
|
24375
|
-
|
|
24376
|
-
//
|
|
24377
|
-
costUsdPicos: record.cost,
|
|
24378
|
-
costUsd: {
|
|
24379
|
-
p50: rec.costP50 === null ? null : Number(rec.costP50) / 1e12,
|
|
24380
|
-
p95: rec.costP95 === null ? null : Number(rec.costP95) / 1e12,
|
|
24381
|
-
max: rec.costMax === null ? null : Number(rec.costMax) / 1e12
|
|
24382
|
-
},
|
|
24383
|
-
chargedCredits: record.charged,
|
|
24384
|
-
// The answer's LENGTH, priced by the server per sample (`fitCredits`,
|
|
24233
|
+
chargedCoins: { p50, p95, max },
|
|
24234
|
+
// The answer's LENGTH, priced by the server per sample (`fitCoins`,
|
|
24385
24235
|
// headroom already on the tokens), over the settled samples.
|
|
24386
|
-
|
|
24387
|
-
|
|
24388
|
-
|
|
24389
|
-
usdCentsPerCredit: terms.usdCentsPerCredit,
|
|
24390
|
-
missingTerms: rec.missingTerms,
|
|
24391
|
-
costBasedMaxCredits: rec.costBasedMaxCredits,
|
|
24392
|
-
lengthRaisedCeiling: rec.lengthRaisedCeiling,
|
|
24236
|
+
fitCoins: { p95: rec.fitP95, max: rec.fitMax },
|
|
24237
|
+
chargedBasedEstimateCoins: rec.chargedBasedEstimateCoins,
|
|
24238
|
+
lengthRaisedPrice: rec.lengthRaisedPrice,
|
|
24393
24239
|
cutOffSamples: rec.cutOffSamples,
|
|
24394
24240
|
exceedsStandCap: rec.exceedsStandCap,
|
|
24395
24241
|
noRecommendationReason: rec.noRecommendationReason,
|
|
24396
|
-
|
|
24397
|
-
|
|
24398
|
-
|
|
24399
|
-
recommendedPerCallEstimateCredits: rec.recommendedPerCallEstimateCredits,
|
|
24242
|
+
recommendedDeclaredHeadroomBps: headroomBps,
|
|
24243
|
+
recommendedEstimateCoins: recommended,
|
|
24244
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
24400
24245
|
savedTo
|
|
24401
24246
|
});
|
|
24402
24247
|
if (!ok) process.exitCode = 1;
|
|
24403
24248
|
return;
|
|
24404
24249
|
}
|
|
24405
24250
|
log.plain("");
|
|
24406
|
-
const printMeasured = () => {
|
|
24407
|
-
log.plain(c.bold(" Real cost per call"));
|
|
24408
|
-
log.plain(` p50 ${usdFromPicos(rec.costP50)} p95 ${usdFromPicos(rec.costP95)} max ${usdFromPicos(rec.costMax)} (${settledCount} of ${samples} settled)`);
|
|
24409
|
-
log.plain(c.bold(" Charged credits"));
|
|
24410
|
-
log.plain(` p50 ${rec.chargedP50} p95 ${rec.chargedP95} max ${rec.chargedMax} (whole credits, rounded up per call)`);
|
|
24411
|
-
};
|
|
24412
24251
|
if (lengthBlocked) {
|
|
24413
|
-
if (
|
|
24414
|
-
|
|
24252
|
+
if (charged.length > 0) {
|
|
24253
|
+
log.plain(c.bold(" Charged coins"));
|
|
24254
|
+
log.plain(` p50 ${p50} p95 ${p95} max ${max} (${charged.length} of ${samples} settled)`);
|
|
24415
24255
|
log.plain("");
|
|
24416
24256
|
}
|
|
24417
24257
|
printRecommendation(log, record);
|
|
@@ -24419,16 +24259,17 @@ async function runBench(args) {
|
|
|
24419
24259
|
process.exitCode = 1;
|
|
24420
24260
|
return;
|
|
24421
24261
|
}
|
|
24422
|
-
if (
|
|
24262
|
+
if (charged.length === 0) {
|
|
24423
24263
|
log.error(
|
|
24424
|
-
providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to
|
|
24264
|
+
providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to price from." : " No sample settled, so there is nothing to price from."
|
|
24425
24265
|
);
|
|
24426
24266
|
process.exitCode = 1;
|
|
24427
24267
|
return;
|
|
24428
24268
|
}
|
|
24429
|
-
|
|
24269
|
+
log.plain(c.bold(" Charged coins"));
|
|
24270
|
+
log.plain(` p50 ${p50} p95 ${p95} max ${max} (${charged.length} of ${samples} settled)`);
|
|
24430
24271
|
if (balanceBefore !== null && balanceAfter !== null) {
|
|
24431
|
-
log.plain(` Balance ${balanceBefore} \u2192 ${balanceAfter}
|
|
24272
|
+
log.plain(` Balance ${balanceBefore} \u2192 ${balanceAfter} coin spendable`);
|
|
24432
24273
|
}
|
|
24433
24274
|
log.plain("");
|
|
24434
24275
|
printRecommendation(log, record);
|
|
@@ -24440,58 +24281,43 @@ function printRecommendation(log, record) {
|
|
|
24440
24281
|
return;
|
|
24441
24282
|
}
|
|
24442
24283
|
if (record.noRecommendationReason === "answer_cut_off") {
|
|
24443
|
-
log.warn(` ${benchCutOffLine(record.cutOffSamples
|
|
24284
|
+
log.warn(` ${benchCutOffLine(record.cutOffSamples ?? 1, record.maxCoinsPerSample ?? 0)}`);
|
|
24444
24285
|
return;
|
|
24445
24286
|
}
|
|
24446
|
-
|
|
24447
|
-
|
|
24448
|
-
log.warn(" No ceiling recommended.");
|
|
24287
|
+
if (record.recommendedEstimateCoins === null || record.p95 === null) {
|
|
24288
|
+
log.warn(" No price recommended.");
|
|
24449
24289
|
log.dim(
|
|
24450
|
-
record.
|
|
24290
|
+
record.recommendedDeclaredHeadroomBps === null ? " This stand served no recommended headroom, and the multiplier is the server's to set \u2014" : " No sample settled, so there is no p95 to build a price on \u2014"
|
|
24451
24291
|
);
|
|
24452
24292
|
log.dim(" declaring a number from this run would be a guess dressed as a measurement.");
|
|
24453
|
-
printAverage(log, record, attempts);
|
|
24454
24293
|
return;
|
|
24455
24294
|
}
|
|
24456
|
-
log.plain(c.bold(` Declare
|
|
24457
|
-
if (record.
|
|
24295
|
+
log.plain(c.bold(` Declare estimateCoins: ${record.recommendedEstimateCoins}`));
|
|
24296
|
+
if (record.lengthRaisedPrice) {
|
|
24458
24297
|
log.plain(` ${LLM_LENGTH_RAISED_LINE}`);
|
|
24459
24298
|
log.dim(
|
|
24460
|
-
` = the smallest
|
|
24299
|
+
` = the smallest price whose answer allowance holds the answer plus this stand's headroom (p95 ${record.fitP95 ?? "\u2014"}), above the`
|
|
24461
24300
|
);
|
|
24462
24301
|
log.dim(
|
|
24463
|
-
`
|
|
24302
|
+
` charged-based ${record.chargedBasedEstimateCoins ?? "\u2014"} (p95 of charged coins ${record.p95} plus ${record.recommendedDeclaredHeadroomBps} bps), over ${record.completed} settled attempt${record.completed === 1 ? "" : "s"} on ${record.modelId}.`
|
|
24464
24303
|
);
|
|
24465
24304
|
} else {
|
|
24466
24305
|
log.dim(
|
|
24467
|
-
` = p95
|
|
24306
|
+
` = p95 of charged coins (${record.p95}) plus this stand's recommended headroom (${record.recommendedDeclaredHeadroomBps} bps),`
|
|
24468
24307
|
);
|
|
24469
|
-
log.dim(` measured over ${
|
|
24308
|
+
log.dim(` measured over ${record.completed} settled attempt${record.completed === 1 ? "" : "s"} on ${record.modelId}.`);
|
|
24470
24309
|
}
|
|
24471
|
-
log.dim(" That number is a
|
|
24472
|
-
|
|
24473
|
-
|
|
24474
|
-
log.plain(c.bold(` Grant perCallMaxCredits: ${record.recommendedPerCallMaxCredits}`));
|
|
24310
|
+
log.dim(" That number is a PRICE: a started attempt is charged in full, including one that fails.");
|
|
24311
|
+
if (record.recommendedPerCallMaxCoins != null) {
|
|
24312
|
+
log.plain(c.bold(` Grant perCallMaxCoins: ${record.recommendedPerCallMaxCoins}`));
|
|
24475
24313
|
log.dim(
|
|
24476
|
-
record.
|
|
24314
|
+
record.lengthRaisedPrice ? ` = the same headroom over the worst sample (${record.max}) or the longest answer's price, whichever is larger, and never below the price above it.` : ` = the same headroom over the worst sample (${record.max}), and never below the price above it.`
|
|
24477
24315
|
);
|
|
24478
|
-
|
|
24479
|
-
printAverage(log, record, attempts);
|
|
24480
|
-
}
|
|
24481
|
-
function printAverage(log, record, attempts) {
|
|
24482
|
-
if (record.averageCreditsPerCall === null) return;
|
|
24483
|
-
log.plain(` Per-call estimate: about ${formatCredits(record.averageCreditsPerCall)} credits per call`);
|
|
24484
|
-
log.dim(` = the average real cost with the platform fee, no headroom, over ${attempts}.`);
|
|
24485
|
-
if (record.recommendedPerCallEstimateCredits !== null) {
|
|
24486
|
-
log.plain(c.bold(` Grant perCallEstimateCredits: ${record.recommendedPerCallEstimateCredits}`));
|
|
24487
|
-
log.dim(` = that average rounded up to a whole credit. disclosure.estimatedCreditsPerPeriod = your calls per period \xD7 ${formatCredits(record.averageCreditsPerCall)}, rounded up.`);
|
|
24316
|
+
log.dim(" The ceiling is the price's room to be wrong; declaring one under the price is refused.");
|
|
24488
24317
|
}
|
|
24489
24318
|
}
|
|
24490
24319
|
async function explainBenchRefusal(res, call, base, log, opts, index, settledSoFar) {
|
|
24491
|
-
if (printedStructuredError(res)) {
|
|
24492
|
-
const printed2 = await res.json().catch(() => ({}));
|
|
24493
|
-
return { stop: true, refusal: { status: res.status, error: typeof printed2.error === "string" ? printed2.error : null, slotsHeld: null, slotLimit: null } };
|
|
24494
|
-
}
|
|
24320
|
+
if (printedStructuredError(res)) return { stop: true, refusal: { status: res.status, error: null, slotsHeld: null, slotLimit: null } };
|
|
24495
24321
|
const body = await res.json().catch(() => ({}));
|
|
24496
24322
|
const refusal2 = { status: res.status, error: body.error ?? null, slotsHeld: null, slotLimit: null };
|
|
24497
24323
|
if (res.status === 429 && body.error === "generation_limit") {
|
|
@@ -24501,10 +24327,6 @@ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoF
|
|
|
24501
24327
|
if (!opts.json) log.error(` Sample ${index + 1} ${benchSlotsHeldSentence(counts)}`);
|
|
24502
24328
|
return { stop: true, refusal: refusal2 };
|
|
24503
24329
|
}
|
|
24504
|
-
if (res.status === 403 && body.error === "credits_unverified") {
|
|
24505
|
-
if (!opts.json) log.error(` ${LLM_BENCH_UNVERIFIED_LINE}`);
|
|
24506
|
-
return { stop: true, refusal: refusal2 };
|
|
24507
|
-
}
|
|
24508
24330
|
if (opts.json) return { stop: res.status !== 429, refusal: refusal2 };
|
|
24509
24331
|
if (res.status === 404 && body.error === "not_found") {
|
|
24510
24332
|
log.error(` ${LLM_LANE_OFF_LINE}`);
|
|
@@ -24520,7 +24342,7 @@ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoF
|
|
|
24520
24342
|
return { stop: true, refusal: refusal2 };
|
|
24521
24343
|
}
|
|
24522
24344
|
if (res.status === 402) {
|
|
24523
|
-
log.error(" Not enough
|
|
24345
|
+
log.error(" Not enough coin to start the attempt.");
|
|
24524
24346
|
return { stop: true, refusal: refusal2 };
|
|
24525
24347
|
}
|
|
24526
24348
|
log.error(
|
|
@@ -24529,7 +24351,7 @@ async function explainBenchRefusal(res, call, base, log, opts, index, settledSoF
|
|
|
24529
24351
|
return { stop: res.status >= 500 || res.status === 401 || res.status === 403, refusal: refusal2 };
|
|
24530
24352
|
}
|
|
24531
24353
|
function printProviderRefusal(log, status2, message, awaitingBill = false) {
|
|
24532
|
-
if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so
|
|
24354
|
+
if (awaitingBill) log.dim(` The provider answered ${status2} after routing \u2014 its bill is not final, so this call's coin stays held until it resolves (see \`genex llm status\`); it is not a sample.`);
|
|
24533
24355
|
else log.dim(" The provider refused this call at its door \u2014 no inference ran, it cost nothing, and it is not a sample.");
|
|
24534
24356
|
if (message) log.dim(` Provider said: ${message}`);
|
|
24535
24357
|
if (status2 === 401 || status2 === 403) {
|
|
@@ -24555,6 +24377,19 @@ async function pollSettled(call, url, timeoutSec) {
|
|
|
24555
24377
|
}
|
|
24556
24378
|
return last;
|
|
24557
24379
|
}
|
|
24380
|
+
async function readCoinBalance(apiUrl, token) {
|
|
24381
|
+
try {
|
|
24382
|
+
const res = await apiFetch(`${apiUrl}/api/coin/balance`, {
|
|
24383
|
+
headers: { Authorization: `Bearer ${token}` },
|
|
24384
|
+
signal: AbortSignal.timeout(6e3)
|
|
24385
|
+
});
|
|
24386
|
+
if (!res.ok) return null;
|
|
24387
|
+
const body = await res.json().catch(() => null);
|
|
24388
|
+
return typeof body?.spendable === "number" ? body.spendable : null;
|
|
24389
|
+
} catch {
|
|
24390
|
+
return null;
|
|
24391
|
+
}
|
|
24392
|
+
}
|
|
24558
24393
|
async function saveBench(cwd, record) {
|
|
24559
24394
|
try {
|
|
24560
24395
|
const file = path36.join(cwd, LLM_BENCH_FILE);
|
|
@@ -24570,10 +24405,9 @@ function benchFailed(opts, log, message) {
|
|
|
24570
24405
|
else log.error(message);
|
|
24571
24406
|
process.exitCode = 1;
|
|
24572
24407
|
}
|
|
24573
|
-
function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples,
|
|
24408
|
+
function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoins, opts) {
|
|
24574
24409
|
return {
|
|
24575
24410
|
command: "llm bench",
|
|
24576
|
-
currency: "credits",
|
|
24577
24411
|
apiUrl,
|
|
24578
24412
|
projectId,
|
|
24579
24413
|
modelId: opts.modelId ?? null,
|
|
@@ -24582,28 +24416,21 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCre
|
|
|
24582
24416
|
schemaPath: opts.schemaPath ?? null,
|
|
24583
24417
|
samples,
|
|
24584
24418
|
completed: 0,
|
|
24585
|
-
|
|
24419
|
+
maxCoinsPerSample: maxCoins,
|
|
24586
24420
|
balanceBefore: null,
|
|
24587
24421
|
balanceAfter: null,
|
|
24588
24422
|
results: [],
|
|
24589
24423
|
refusal: null,
|
|
24590
|
-
|
|
24591
|
-
|
|
24592
|
-
|
|
24593
|
-
|
|
24594
|
-
recommendedDeclaredHeadroomBps: null,
|
|
24595
|
-
feeBps: null,
|
|
24596
|
-
usdCentsPerCredit: null,
|
|
24597
|
-
missingTerms: null,
|
|
24598
|
-
costBasedMaxCredits: null,
|
|
24599
|
-
lengthRaisedCeiling: false,
|
|
24424
|
+
chargedCoins: { p50: null, p95: null, max: null },
|
|
24425
|
+
fitCoins: { p95: null, max: null },
|
|
24426
|
+
chargedBasedEstimateCoins: null,
|
|
24427
|
+
lengthRaisedPrice: false,
|
|
24600
24428
|
cutOffSamples: 0,
|
|
24601
24429
|
exceedsStandCap: false,
|
|
24602
24430
|
noRecommendationReason: null,
|
|
24603
|
-
|
|
24604
|
-
|
|
24605
|
-
|
|
24606
|
-
recommendedPerCallEstimateCredits: null,
|
|
24431
|
+
recommendedDeclaredHeadroomBps: null,
|
|
24432
|
+
recommendedEstimateCoins: null,
|
|
24433
|
+
recommendedPerCallMaxCoins: null,
|
|
24607
24434
|
savedTo: null
|
|
24608
24435
|
};
|
|
24609
24436
|
}
|
|
@@ -24666,8 +24493,7 @@ async function reportStatus(args) {
|
|
|
24666
24493
|
for (const row of rows) {
|
|
24667
24494
|
const state = rowState(row);
|
|
24668
24495
|
const mark = state === "active" ? c.cyan("\u25CF") : state === "awaiting bill" ? c.yellow("\u25D0") : c.green("\u25CB");
|
|
24669
|
-
const
|
|
24670
|
-
const reserved = money2.reserved === null ? "reserved \u2014" : `${money2.reserved} ${money2.unit} reserved`;
|
|
24496
|
+
const reserved = typeof row.reservedCoins === "number" ? `${row.reservedCoins} coin reserved` : "reserved \u2014";
|
|
24671
24497
|
log.plain(
|
|
24672
24498
|
` ${mark} ${row.id ?? "?"} ${row.modelId ?? "?"} ${state.padEnd(13)} ${formatAge(row.createdAt, now).padStart(4)} old ${reserved}${row.error ? ` ${row.error}` : ""}${rowHoldsSlot(row) ? "" : c.dim(" (no slot)")}`
|
|
24673
24499
|
);
|
|
@@ -24739,11 +24565,10 @@ async function cancelCall(args) {
|
|
|
24739
24565
|
if (!opts.json) {
|
|
24740
24566
|
log.warn(` ${row.id} (${row.status ?? "?"}) \u2014 ${sentence}.`);
|
|
24741
24567
|
if (state === "awaiting bill") {
|
|
24742
|
-
log.dim(` Nothing to cancel:
|
|
24568
|
+
log.dim(` Nothing to cancel: its coin releases when the bill resolves, and it stops holding a slot ${PENDING_SLOT_RELEASE_MINUTES} minutes after dispatch.`);
|
|
24743
24569
|
if (row.providerMessage?.trim()) log.dim(` Provider said: ${row.providerMessage.trim()}`);
|
|
24744
24570
|
} else {
|
|
24745
|
-
|
|
24746
|
-
log.dim(` ${money2.charged === null ? "charge unknown" : `${money2.charged} ${money2.unit} charged`}${row.error ? ` \xB7 ${row.error}` : ""}.`);
|
|
24571
|
+
log.dim(` ${typeof row.chargedCoins === "number" ? `${row.chargedCoins} coin charged` : "charge unknown"}${row.error ? ` \xB7 ${row.error}` : ""}.`);
|
|
24747
24572
|
}
|
|
24748
24573
|
}
|
|
24749
24574
|
return done("ok", null, state === "awaiting bill" ? "awaiting_bill" : "already_final", row);
|
|
@@ -24765,126 +24590,78 @@ async function cancelCall(args) {
|
|
|
24765
24590
|
const after = await res.json().catch(() => null) ?? row;
|
|
24766
24591
|
if (!opts.json) {
|
|
24767
24592
|
log.success(` Cancelled ${row.id} (was ${row.status ?? "active"}).`);
|
|
24768
|
-
log.dim(`
|
|
24593
|
+
log.dim(` Its coin releases once the bill settles; ${c.cyan("genex llm status")} shows it until then.`);
|
|
24769
24594
|
}
|
|
24770
24595
|
return done("ok", null, "cancelled", after);
|
|
24771
24596
|
}
|
|
24772
|
-
async function
|
|
24773
|
-
let
|
|
24597
|
+
async function reportSavedPrice(cwd, opts, log) {
|
|
24598
|
+
let record = null;
|
|
24774
24599
|
try {
|
|
24775
|
-
|
|
24600
|
+
record = JSON.parse(await fs35.readFile(path36.join(cwd, LLM_BENCH_FILE), "utf8"));
|
|
24776
24601
|
} catch {
|
|
24777
|
-
|
|
24602
|
+
record = null;
|
|
24778
24603
|
}
|
|
24779
|
-
if (!
|
|
24780
|
-
const v = parsed;
|
|
24781
|
-
if (v.v === 2 && v.currency === "credits" && v.cost && v.charged) return { kind: "credits", record: parsed };
|
|
24782
|
-
return { kind: "coins", record: parsed };
|
|
24783
|
-
}
|
|
24784
|
-
var LLM_PRICE_COIN_TERMS_LINE = "This bench was measured under the old COIN terms, when a game declared a fixed price per call. In-game calls are now billed in the player's credits, as used, and a game declares a per-call CEILING (maxCredits) instead \u2014 a coin price is not that number. Re-run the bench for the figures to declare.";
|
|
24785
|
-
async function reportSavedPrice(cwd, opts, log) {
|
|
24786
|
-
const saved = await readSavedBench(cwd);
|
|
24787
|
-
const rerun = 'genex llm bench "<prompt>" --max-credits <n> --user-approved';
|
|
24788
|
-
const creditKeys = (record2) => ({
|
|
24789
|
-
costUsdPicos: record2?.cost ?? { p50: null, p95: null, max: null },
|
|
24790
|
-
chargedCredits: record2?.charged ?? { p50: null, p95: null, max: null },
|
|
24791
|
-
fitP95: record2?.fitP95 ?? null,
|
|
24792
|
-
fitMax: record2?.fitMax ?? null,
|
|
24793
|
-
recommendedDeclaredHeadroomBps: record2?.recommendedDeclaredHeadroomBps ?? null,
|
|
24794
|
-
feeBps: record2?.feeBps ?? null,
|
|
24795
|
-
usdCentsPerCredit: record2?.usdCentsPerCredit ?? null,
|
|
24796
|
-
costBasedMaxCredits: record2?.costBasedMaxCredits ?? null,
|
|
24797
|
-
lengthRaisedCeiling: record2?.lengthRaisedCeiling ?? false,
|
|
24798
|
-
cutOffSamples: record2?.cutOffSamples ?? 0,
|
|
24799
|
-
noRecommendationReason: record2?.noRecommendationReason ?? null,
|
|
24800
|
-
recommendedMaxCredits: record2?.recommendedMaxCredits ?? null,
|
|
24801
|
-
recommendedPerCallMaxCredits: record2?.recommendedPerCallMaxCredits ?? null,
|
|
24802
|
-
averageCreditsPerCall: record2?.averageCreditsPerCall ?? null,
|
|
24803
|
-
recommendedPerCallEstimateCredits: record2?.recommendedPerCallEstimateCredits ?? null
|
|
24804
|
-
});
|
|
24805
|
-
if (!saved) {
|
|
24604
|
+
if (!record) {
|
|
24806
24605
|
if (opts.json) {
|
|
24807
24606
|
writeJsonLine({
|
|
24808
24607
|
command: "llm price",
|
|
24809
24608
|
status: "failed",
|
|
24810
24609
|
error: "no_bench",
|
|
24811
|
-
currency: null,
|
|
24812
24610
|
ranAt: null,
|
|
24813
24611
|
modelId: null,
|
|
24814
24612
|
samples: null,
|
|
24815
24613
|
completed: null,
|
|
24816
|
-
|
|
24817
|
-
|
|
24614
|
+
p50: null,
|
|
24615
|
+
p95: null,
|
|
24616
|
+
max: null,
|
|
24617
|
+
fitP95: null,
|
|
24618
|
+
chargedBasedEstimateCoins: null,
|
|
24619
|
+
lengthRaisedPrice: false,
|
|
24620
|
+
cutOffSamples: 0,
|
|
24621
|
+
noRecommendationReason: null,
|
|
24622
|
+
recommendedDeclaredHeadroomBps: null,
|
|
24623
|
+
recommendedEstimateCoins: null,
|
|
24624
|
+
recommendedPerCallMaxCoins: null,
|
|
24818
24625
|
savedTo: null
|
|
24819
24626
|
});
|
|
24820
24627
|
} else {
|
|
24821
24628
|
log.error("Nothing has been benchmarked in this folder yet.");
|
|
24822
|
-
log.dim(` Run ${c.cyan(
|
|
24823
|
-
}
|
|
24824
|
-
process.exitCode = 1;
|
|
24825
|
-
return;
|
|
24826
|
-
}
|
|
24827
|
-
if (saved.kind === "coins") {
|
|
24828
|
-
const old = saved.record;
|
|
24829
|
-
const legacy = {
|
|
24830
|
-
p50: old.p50 ?? null,
|
|
24831
|
-
p95: old.p95 ?? null,
|
|
24832
|
-
max: old.max ?? null,
|
|
24833
|
-
recommendedEstimateCoins: old.recommendedEstimateCoins ?? null,
|
|
24834
|
-
recommendedPerCallMaxCoins: old.recommendedPerCallMaxCoins ?? null,
|
|
24835
|
-
maxCoinsPerSample: old.maxCoinsPerSample ?? null,
|
|
24836
|
-
noRecommendationReason: old.noRecommendationReason ?? null
|
|
24837
|
-
};
|
|
24838
|
-
if (opts.json) {
|
|
24839
|
-
writeJsonLine({
|
|
24840
|
-
command: "llm price",
|
|
24841
|
-
status: "failed",
|
|
24842
|
-
error: "coin_terms",
|
|
24843
|
-
currency: "coins",
|
|
24844
|
-
ranAt: old.ranAt ?? null,
|
|
24845
|
-
modelId: old.modelId ?? null,
|
|
24846
|
-
samples: old.samples ?? null,
|
|
24847
|
-
completed: old.completed ?? null,
|
|
24848
|
-
...creditKeys(null),
|
|
24849
|
-
legacyCoinTerms: legacy,
|
|
24850
|
-
savedTo: LLM_BENCH_FILE
|
|
24851
|
-
});
|
|
24852
|
-
} else {
|
|
24853
|
-
log.plain(c.bold("genex llm price"));
|
|
24854
|
-
log.dim(` from ${LLM_BENCH_FILE}, benchmarked ${old.ranAt ?? "at an unknown time"}${old.modelId ? ` on ${old.modelId}` : ""}`);
|
|
24855
|
-
log.plain("");
|
|
24856
|
-
log.warn(` ${LLM_PRICE_COIN_TERMS_LINE}`);
|
|
24857
|
-
const measured = legacy.p95 === null ? "no settled sample" : `charged coins p50 ${legacy.p50} \xB7 p95 ${legacy.p95} \xB7 max ${legacy.max}`;
|
|
24858
|
-
const recommended = legacy.recommendedEstimateCoins === null ? "no price recommended" : `it recommended estimateCoins ${legacy.recommendedEstimateCoins}${legacy.recommendedPerCallMaxCoins != null ? ` and perCallMaxCoins ${legacy.recommendedPerCallMaxCoins}` : ""}`;
|
|
24859
|
-
log.dim(` What it measured then: ${measured}; ${recommended}.`);
|
|
24860
|
-
log.dim(` Re-run: ${c.cyan(rerun)}`);
|
|
24629
|
+
log.dim(` Run ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')} first.`);
|
|
24861
24630
|
}
|
|
24862
24631
|
process.exitCode = 1;
|
|
24863
24632
|
return;
|
|
24864
24633
|
}
|
|
24865
|
-
const record = saved.record;
|
|
24866
24634
|
if (opts.json) {
|
|
24867
24635
|
writeJsonLine({
|
|
24868
24636
|
command: "llm price",
|
|
24869
|
-
status: record.
|
|
24870
|
-
error: record.
|
|
24871
|
-
currency: "credits",
|
|
24637
|
+
status: record.recommendedEstimateCoins === null ? "failed" : "ok",
|
|
24638
|
+
error: record.recommendedEstimateCoins === null ? record.noRecommendationReason ?? "no_recommendation" : null,
|
|
24872
24639
|
ranAt: record.ranAt ?? null,
|
|
24873
24640
|
modelId: record.modelId ?? null,
|
|
24874
24641
|
samples: record.samples ?? null,
|
|
24875
24642
|
completed: record.completed ?? null,
|
|
24876
|
-
|
|
24877
|
-
|
|
24643
|
+
p50: record.p50 ?? null,
|
|
24644
|
+
p95: record.p95 ?? null,
|
|
24645
|
+
max: record.max ?? null,
|
|
24646
|
+
// Absent from a file an older CLI wrote; null / false / 0 then, never missing.
|
|
24647
|
+
fitP95: record.fitP95 ?? null,
|
|
24648
|
+
chargedBasedEstimateCoins: record.chargedBasedEstimateCoins ?? null,
|
|
24649
|
+
lengthRaisedPrice: record.lengthRaisedPrice ?? false,
|
|
24650
|
+
cutOffSamples: record.cutOffSamples ?? 0,
|
|
24651
|
+
noRecommendationReason: record.noRecommendationReason ?? null,
|
|
24652
|
+
recommendedDeclaredHeadroomBps: record.recommendedDeclaredHeadroomBps ?? null,
|
|
24653
|
+
recommendedEstimateCoins: record.recommendedEstimateCoins ?? null,
|
|
24654
|
+
recommendedPerCallMaxCoins: record.recommendedPerCallMaxCoins ?? null,
|
|
24878
24655
|
savedTo: LLM_BENCH_FILE
|
|
24879
24656
|
});
|
|
24880
|
-
if (record.
|
|
24657
|
+
if (record.recommendedEstimateCoins === null) process.exitCode = 1;
|
|
24881
24658
|
return;
|
|
24882
24659
|
}
|
|
24883
24660
|
log.plain(c.bold("genex llm price"));
|
|
24884
24661
|
log.dim(` from ${LLM_BENCH_FILE}, benchmarked ${record.ranAt}`);
|
|
24885
24662
|
log.plain("");
|
|
24886
24663
|
printRecommendation(log, record);
|
|
24887
|
-
if (record.
|
|
24664
|
+
if (record.recommendedEstimateCoins === null) process.exitCode = 1;
|
|
24888
24665
|
}
|
|
24889
24666
|
function writeJsonLine(value) {
|
|
24890
24667
|
process.stdout.write(`${JSON.stringify(value)}
|
|
@@ -25144,7 +24921,7 @@ function runtimeRow(state, lane, toolsOnly = false) {
|
|
|
25144
24921
|
return {
|
|
25145
24922
|
label: "Runtime",
|
|
25146
24923
|
value: `in-game LLM lane live \xB7 ${n} model${n === 1 ? "" : "s"}${keyRefused ? ` \xB7 ${c.yellow("provider key refused")}` : ""}`,
|
|
25147
|
-
fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : '
|
|
24924
|
+
fix: keyRefused ? LLM_PROVIDER_KEY_REJECTED_LINE : toolsOnly ? convertFix : 'Price a call before a game declares one: `npx genex llm bench "<prompt>" --max-coins <n> --user-approved`.'
|
|
25148
24925
|
};
|
|
25149
24926
|
}
|
|
25150
24927
|
case "off":
|
|
@@ -27082,16 +26859,16 @@ ${c.bold("Usage")}
|
|
|
27082
26859
|
status | cancel. "models" says whether this
|
|
27083
26860
|
stand serves the runtime LLM lane at all, and
|
|
27084
26861
|
at what provider rates. "bench" runs the real
|
|
27085
|
-
model on YOUR OWN
|
|
27086
|
-
attempt
|
|
26862
|
+
model on YOUR OWN coin and reports what each
|
|
26863
|
+
attempt was charged; it needs --max-coins <n>
|
|
27087
26864
|
--user-approved, like any spend the player has
|
|
27088
26865
|
to agree to. "price" reprints the last run's
|
|
27089
|
-
recommended
|
|
26866
|
+
recommended estimateCoins. "status" lists this
|
|
27090
26867
|
project's open calls and the slots they hold;
|
|
27091
26868
|
"cancel <id>" stops an active one. Declare a
|
|
27092
|
-
game's
|
|
27093
|
-
|
|
27094
|
-
|
|
26869
|
+
game's price from a bench, never from a guess:
|
|
26870
|
+
a started attempt is charged in full, including
|
|
26871
|
+
one that fails.
|
|
27095
26872
|
genex player <sub> [options] Answer the in-game model requests you approve, on your
|
|
27096
26873
|
OWN Claude or ChatGPT subscription: install | run |
|
|
27097
26874
|
status | stop | uninstall. "install" signs this machine
|
|
@@ -27546,7 +27323,7 @@ ${c.bold("Examples")}
|
|
|
27546
27323
|
genex shop remove sku_123
|
|
27547
27324
|
genex llm models
|
|
27548
27325
|
genex llm models --all
|
|
27549
|
-
genex llm bench "Reply with one short taunt." --samples 5 --max-
|
|
27326
|
+
genex llm bench "Reply with one short taunt." --samples 5 --max-coins 20 --user-approved
|
|
27550
27327
|
genex llm price
|
|
27551
27328
|
genex llm status
|
|
27552
27329
|
genex llm cancel <id>
|
|
@@ -27783,24 +27560,22 @@ ${c.bold("How the allowance works")}
|
|
|
27783
27560
|
|
|
27784
27561
|
Every number printed is live from your account; prices can change without a deploy.
|
|
27785
27562
|
`,
|
|
27786
|
-
llm: `${c.bold("genex llm")} \u2014 in-game model calls: is the lane live, what does one call cost, what
|
|
27563
|
+
llm: `${c.bold("genex llm")} \u2014 in-game model calls: is the lane live, what does one call cost, what should the game declare?
|
|
27787
27564
|
|
|
27788
27565
|
${c.bold("Usage")}
|
|
27789
27566
|
genex llm models [--all] [--json] Which models this stand serves, their provider rates per
|
|
27790
|
-
million tokens, the
|
|
27791
|
-
|
|
27792
|
-
|
|
27793
|
-
|
|
27794
|
-
|
|
27795
|
-
|
|
27796
|
-
|
|
27797
|
-
|
|
27798
|
-
|
|
27799
|
-
|
|
27800
|
-
|
|
27801
|
-
|
|
27802
|
-
allowance does not count. Needs a folder linked to a
|
|
27803
|
-
game you own, and a balance that covers one attempt.
|
|
27567
|
+
million tokens, and the headroom it recommends over a
|
|
27568
|
+
benchmark. The catalog is synced from OpenRouter, so the
|
|
27569
|
+
FEATURED short list is printed by default and --all
|
|
27570
|
+
prints every row (--json always carries every row). Off
|
|
27571
|
+
on this stand: it says so and exits 0 \u2014 build the feature
|
|
27572
|
+
without a model rather than promising a 404.
|
|
27573
|
+
genex llm bench "<prompt>" --max-coins <n> --user-approved [options]
|
|
27574
|
+
Run the real model <n> times through the development
|
|
27575
|
+
lane and report what each attempt was CHARGED. Refused
|
|
27576
|
+
without both flags, before any network call: this spends
|
|
27577
|
+
YOUR OWN COIN, a wallet the build's asset allowance
|
|
27578
|
+
cannot see. Needs a folder linked to a game you own.
|
|
27804
27579
|
genex llm price [--json] The recommendation from the last bench in this folder.
|
|
27805
27580
|
Reads a file; spends nothing.
|
|
27806
27581
|
genex llm status [--json] This project's open development calls: active, or
|
|
@@ -27817,32 +27592,29 @@ ${c.bold("Bench options")}
|
|
|
27817
27592
|
--samples <n> How many real attempts (default ${DEFAULT_BENCH_SAMPLES}). More samples, better p95.
|
|
27818
27593
|
--schema <file> A JSON file holding the output schema; implies JSON output.
|
|
27819
27594
|
--json-output Ask for JSON without a schema. --text Ask for text (the default).
|
|
27820
|
-
--max-
|
|
27595
|
+
--max-coins <n> The per-attempt ceiling YOU approved \u2014 a bench runs on your own coin,
|
|
27821
27596
|
never a player's. The run's worst case is that number times --samples,
|
|
27822
|
-
and it is printed before anything starts.
|
|
27823
|
-
name, read as the same number.)
|
|
27597
|
+
and it is printed before anything starts.
|
|
27824
27598
|
--timeout <seconds> Bounds the wait for each attempt.
|
|
27825
27599
|
--json One machine-readable object; every key is always present, null when
|
|
27826
27600
|
the CLI could not learn it.
|
|
27827
27601
|
|
|
27828
27602
|
${c.bold("Why a benchmark and not an estimate")}
|
|
27829
|
-
|
|
27830
|
-
|
|
27831
|
-
|
|
27832
|
-
|
|
27833
|
-
|
|
27834
|
-
|
|
27835
|
-
|
|
27836
|
-
|
|
27837
|
-
|
|
27838
|
-
recommendation, because those multipliers are not the CLI's to invent.
|
|
27603
|
+
\`estimateCoins\` is a PRICE, not a guess: declaring it charges it, in full, even when the
|
|
27604
|
+
attempt fails, is cancelled, or stops at its budget. And three layers sit between the
|
|
27605
|
+
provider's rate card and that number \u2014 the provider's cost, the platform's tariff (already
|
|
27606
|
+
inside every charged coin), and your headroom. The bench itself is billed to YOU, on the
|
|
27607
|
+
development lane; the player-funded lane is what the published game uses. Adding a margin
|
|
27608
|
+
to the provider's rate silently drops the tariff and under-prices every call, so the
|
|
27609
|
+
recommendation is built from CHARGED COINS and the headroom comes from the server. A stand
|
|
27610
|
+
that serves no headroom gets no recommendation, because the multiplier is not the CLI's to
|
|
27611
|
+
invent.
|
|
27839
27612
|
|
|
27840
|
-
|
|
27841
|
-
|
|
27842
|
-
A sample cut off at your --max-
|
|
27843
|
-
|
|
27844
|
-
|
|
27845
|
-
fee included, which is the honest estimate a budget's disclosure is built on.
|
|
27613
|
+
The declared price also decides how LONG the answer may be: it sizes each call's output
|
|
27614
|
+
allowance. So the recommendation is never below the smallest price that leaves room for
|
|
27615
|
+
the benchmarked answer, as the server computes it. A sample cut off at your --max-coins
|
|
27616
|
+
(provider_token_limit) is not a sample \u2014 re-run with a higher --max-coins; an answer longer
|
|
27617
|
+
than one call on this stand may produce gets no price at all \u2014 ask for a shorter one.
|
|
27846
27618
|
|
|
27847
27619
|
Results land in ${LLM_BENCH_FILE} \u2014 gitignored, machine-local, re-read by \`genex llm price\`.
|
|
27848
27620
|
`,
|
|
@@ -28057,13 +27829,11 @@ function parseArgs(argv) {
|
|
|
28057
27829
|
"--limit",
|
|
28058
27830
|
// `genex budget --assets <credits>` — the allowance the player approved.
|
|
28059
27831
|
"--assets",
|
|
28060
|
-
// `genex llm bench` value flags. `--max-
|
|
28061
|
-
// shape as `--assets
|
|
28062
|
-
// count a bench. `--max-coins` is its deprecated name (same number).
|
|
27832
|
+
// `genex llm bench` value flags. `--max-coins` is the COIN half of the
|
|
27833
|
+
// same approval shape as `--assets` (a different economy, its own gate).
|
|
28063
27834
|
"--model",
|
|
28064
27835
|
"--samples",
|
|
28065
27836
|
"--schema",
|
|
28066
|
-
"--max-credits",
|
|
28067
27837
|
"--max-coins",
|
|
28068
27838
|
// `genex player install --label <name>` — what this machine is called in
|
|
28069
27839
|
// the dashboard. Its own key rather than the generic 2nd positional,
|
|
@@ -28578,14 +28348,12 @@ function applyValueFlag(options, flag, value) {
|
|
|
28578
28348
|
options.samples = n;
|
|
28579
28349
|
break;
|
|
28580
28350
|
}
|
|
28581
|
-
case "--max-credits":
|
|
28582
28351
|
case "--max-coins": {
|
|
28583
28352
|
const n = Number(value);
|
|
28584
28353
|
if (!Number.isInteger(n) || n < 1) {
|
|
28585
|
-
throw new Error(`Invalid
|
|
28354
|
+
throw new Error(`Invalid --max-coins value: ${value} (whole coin, 1 or more)`);
|
|
28586
28355
|
}
|
|
28587
|
-
|
|
28588
|
-
else options.maxCoins = n;
|
|
28356
|
+
options.maxCoins = n;
|
|
28589
28357
|
break;
|
|
28590
28358
|
}
|
|
28591
28359
|
case "--timeout": {
|
|
@@ -28877,9 +28645,9 @@ async function main() {
|
|
|
28877
28645
|
await runBudget(parsed.options);
|
|
28878
28646
|
break;
|
|
28879
28647
|
// The in-game model lane: is it live here, what does one call really
|
|
28880
|
-
// cost, what
|
|
28881
|
-
//
|
|
28882
|
-
//
|
|
28648
|
+
// cost, what should the game declare. `bench` is the only CLI command
|
|
28649
|
+
// that spends COIN, so it carries its own --max-coins/--user-approved
|
|
28650
|
+
// gate — the credits allowance cannot see that wallet.
|
|
28883
28651
|
case "llm":
|
|
28884
28652
|
await runLlm(parsed.options);
|
|
28885
28653
|
break;
|