@genex-ai/cli-demo 1.35.0-dev.743 → 1.35.0-dev.746
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js
CHANGED
|
@@ -23592,6 +23592,64 @@ async function reportModels(apiUrl, token, opts, log, toolsOnly = false) {
|
|
|
23592
23592
|
log.dim(" Those per-million rates are the PROVIDER's; what your game is charged is the platform");
|
|
23593
23593
|
log.dim(` tariff on top, which only a real call reveals: ${c.cyan('genex llm bench "<prompt>" --max-coins <n> --user-approved')}`);
|
|
23594
23594
|
}
|
|
23595
|
+
function wholeOrNull(value) {
|
|
23596
|
+
return typeof value === "number" && Number.isSafeInteger(value) && value >= 0 ? value : null;
|
|
23597
|
+
}
|
|
23598
|
+
function sampleFrom(id, settled) {
|
|
23599
|
+
const usage = settled?.usage;
|
|
23600
|
+
return {
|
|
23601
|
+
id,
|
|
23602
|
+
status: settled?.status ?? null,
|
|
23603
|
+
billingStatus: settled?.billingStatus ?? null,
|
|
23604
|
+
chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
|
|
23605
|
+
costUsd: typeof usage?.costUsd === "number" ? usage.costUsd : null,
|
|
23606
|
+
error: settled?.error ?? (settled ? null : "timed_out"),
|
|
23607
|
+
providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null,
|
|
23608
|
+
inputTokens: wholeOrNull(usage?.inputTokens),
|
|
23609
|
+
outputTokens: wholeOrNull(usage?.outputTokens),
|
|
23610
|
+
maxOutputTokens: wholeOrNull(usage?.maxOutputTokens),
|
|
23611
|
+
fitCoins: typeof usage?.fitCoins === "number" && Number.isSafeInteger(usage.fitCoins) && usage.fitCoins >= 1 ? usage.fitCoins : null,
|
|
23612
|
+
fitExceedsCap: typeof usage?.fitExceedsCap === "boolean" ? usage.fitExceedsCap : null,
|
|
23613
|
+
truncated: typeof usage?.truncated === "boolean" ? usage.truncated : settled?.error === "provider_token_limit" ? true : null
|
|
23614
|
+
};
|
|
23615
|
+
}
|
|
23616
|
+
var PROVIDER_TOKEN_LIMIT = "provider_token_limit";
|
|
23617
|
+
function benchRecommendation(results, headroomBps) {
|
|
23618
|
+
const settled = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number");
|
|
23619
|
+
const charged = settled.map((r) => r.chargedCoins);
|
|
23620
|
+
const fits = settled.map((r) => r.fitCoins).filter((v) => typeof v === "number");
|
|
23621
|
+
const p95 = percentile2(charged, 95);
|
|
23622
|
+
const max = charged.length ? Math.max(...charged) : null;
|
|
23623
|
+
const fitP95 = percentile2(fits, 95);
|
|
23624
|
+
const fitMax = fits.length ? Math.max(...fits) : null;
|
|
23625
|
+
const exceedsStandCap = results.some((r) => r.fitExceedsCap === true);
|
|
23626
|
+
const cutOffSamples = results.filter((r) => r.error === PROVIDER_TOKEN_LIMIT && r.fitExceedsCap !== true).length;
|
|
23627
|
+
const noRecommendationReason = exceedsStandCap ? "answer_exceeds_stand_cap" : cutOffSamples > 0 ? "answer_cut_off" : null;
|
|
23628
|
+
const chargedBased = p95 !== null && headroomBps !== null ? recommendEstimateCoins(p95, headroomBps) : null;
|
|
23629
|
+
const recommended = chargedBased === null || noRecommendationReason !== null ? null : Math.max(chargedBased, fitP95 ?? 0);
|
|
23630
|
+
const ceiling = recommended === null || p95 === null || max === null || headroomBps === null ? null : Math.max(recommendPerCallMaxCoins(p95, max, headroomBps), recommended, fitMax ?? 0);
|
|
23631
|
+
return {
|
|
23632
|
+
charged,
|
|
23633
|
+
p50: percentile2(charged, 50),
|
|
23634
|
+
p95,
|
|
23635
|
+
max,
|
|
23636
|
+
fits,
|
|
23637
|
+
fitP95,
|
|
23638
|
+
fitMax,
|
|
23639
|
+
chargedBasedEstimateCoins: chargedBased,
|
|
23640
|
+
recommendedEstimateCoins: recommended,
|
|
23641
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
23642
|
+
lengthRaisedPrice: recommended !== null && chargedBased !== null && recommended > chargedBased,
|
|
23643
|
+
cutOffSamples,
|
|
23644
|
+
exceedsStandCap,
|
|
23645
|
+
noRecommendationReason
|
|
23646
|
+
};
|
|
23647
|
+
}
|
|
23648
|
+
var LLM_LENGTH_RAISED_LINE = "The declared price also pays for how long the answer may be: at the charged-based price the answer would be cut off, so declare this one.";
|
|
23649
|
+
var LLM_EXCEEDS_STAND_CAP_LINE = "No price recommended: this answer is longer than one call on this stand may produce. Ask for a shorter answer \u2014 fewer fields, shorter strings, a length the prompt states \u2014 and benchmark again.";
|
|
23650
|
+
function benchCutOffLine(cutOff, maxCoins) {
|
|
23651
|
+
return `No price recommended: ${cutOff} sample${cutOff === 1 ? " was" : "s were"} cut off at this ceiling (--max-coins ${maxCoins}) before the answer was finished, so its real length is unknown. Re-run with a higher --max-coins.`;
|
|
23652
|
+
}
|
|
23595
23653
|
var ACTIVE_STATUSES = /* @__PURE__ */ new Set(["requires_confirmation", "queued", "dispatching", "awaiting_external"]);
|
|
23596
23654
|
function rowState(row) {
|
|
23597
23655
|
if (row.status && ACTIVE_STATUSES.has(row.status)) return "active";
|
|
@@ -23723,42 +23781,39 @@ async function runBench(args) {
|
|
|
23723
23781
|
}
|
|
23724
23782
|
const started = await res.json().catch(() => ({}));
|
|
23725
23783
|
if (!started.id) {
|
|
23726
|
-
results.push({
|
|
23784
|
+
results.push({ ...sampleFrom("", null), id: null, error: "no_generation_id" });
|
|
23727
23785
|
continue;
|
|
23728
23786
|
}
|
|
23729
23787
|
inflight = started.id;
|
|
23730
23788
|
const settled = await pollSettled(call, `${base}/${encodeURIComponent(started.id)}`, opts.timeoutSec ?? DEFAULT_SAMPLE_TIMEOUT_SEC);
|
|
23731
23789
|
inflight = null;
|
|
23732
|
-
const row =
|
|
23733
|
-
id: started.id,
|
|
23734
|
-
status: settled?.status ?? null,
|
|
23735
|
-
billingStatus: settled?.billingStatus ?? null,
|
|
23736
|
-
chargedCoins: typeof settled?.chargedCoins === "number" ? settled.chargedCoins : null,
|
|
23737
|
-
costUsd: typeof settled?.usage?.costUsd === "number" ? settled.usage.costUsd : null,
|
|
23738
|
-
error: settled?.error ?? (settled ? null : "timed_out"),
|
|
23739
|
-
providerMessage: typeof settled?.providerMessage === "string" && settled.providerMessage.trim() ? settled.providerMessage.trim() : null
|
|
23740
|
-
};
|
|
23790
|
+
const row = sampleFrom(started.id, settled);
|
|
23741
23791
|
results.push(row);
|
|
23742
23792
|
const refusedAt = providerRefusalStatus(row.error);
|
|
23743
23793
|
if (refusedAt !== null) providerRefused++;
|
|
23744
23794
|
if (!opts.json) {
|
|
23795
|
+
const length = row.outputTokens !== null ? ` \xB7 ${row.outputTokens}${row.maxOutputTokens !== null ? ` of ${row.maxOutputTokens}` : ""} tokens out` : "";
|
|
23745
23796
|
log.plain(
|
|
23746
|
-
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${row.error ? ` \xB7 ${row.error}` : ""}`
|
|
23797
|
+
` ${row.status === "succeeded" ? c.green("\u2713") : c.yellow("!")} sample ${i + 1} ${String(row.chargedCoins ?? "\u2014").padStart(4)} coin charged \xB7 provider ${usd(row.costUsd)}${length}${row.error ? ` \xB7 ${row.error}` : ""}`
|
|
23747
23798
|
);
|
|
23748
23799
|
if (refusedAt !== null) printProviderRefusal(log, refusedAt, row.providerMessage, row.status === "unknown");
|
|
23800
|
+
if (row.error === PROVIDER_TOKEN_LIMIT) {
|
|
23801
|
+
log.dim(
|
|
23802
|
+
row.fitExceedsCap === true ? " The answer was cut off at this stand's own output limit \u2014 no price makes room for it; it is not a sample." : ` The answer was cut off at this ceiling (--max-coins ${maxCoins}) before it was finished \u2014 it was charged, and it is not a sample.`
|
|
23803
|
+
);
|
|
23804
|
+
}
|
|
23749
23805
|
}
|
|
23750
23806
|
}
|
|
23751
23807
|
} finally {
|
|
23752
23808
|
process.removeListener("SIGINT", onSigint);
|
|
23753
23809
|
}
|
|
23754
|
-
const charged = results.filter((r) => r.status === "succeeded" && r.billingStatus === "final" && typeof r.chargedCoins === "number").map((r) => r.chargedCoins);
|
|
23755
|
-
const settledCount = charged.length;
|
|
23756
|
-
const p50 = percentile2(charged, 50);
|
|
23757
|
-
const p95 = percentile2(charged, 95);
|
|
23758
|
-
const max = charged.length ? Math.max(...charged) : null;
|
|
23759
23810
|
const headroomBps = lane.recommendedDeclaredHeadroomBps;
|
|
23760
|
-
const
|
|
23761
|
-
const
|
|
23811
|
+
const rec = benchRecommendation(results, headroomBps);
|
|
23812
|
+
const { charged, p50, p95, max } = rec;
|
|
23813
|
+
const settledCount = charged.length;
|
|
23814
|
+
const recommended = rec.recommendedEstimateCoins;
|
|
23815
|
+
const ceiling = rec.recommendedPerCallMaxCoins;
|
|
23816
|
+
const lengthBlocked = rec.noRecommendationReason !== null;
|
|
23762
23817
|
const balanceAfter = await readCoinBalance(apiUrl, token);
|
|
23763
23818
|
const record = {
|
|
23764
23819
|
v: 1,
|
|
@@ -23776,14 +23831,22 @@ async function runBench(args) {
|
|
|
23776
23831
|
max,
|
|
23777
23832
|
recommendedDeclaredHeadroomBps: headroomBps,
|
|
23778
23833
|
recommendedEstimateCoins: recommended,
|
|
23779
|
-
recommendedPerCallMaxCoins: ceiling
|
|
23834
|
+
recommendedPerCallMaxCoins: ceiling,
|
|
23835
|
+
maxCoinsPerSample: maxCoins,
|
|
23836
|
+
fitCoins: rec.fits,
|
|
23837
|
+
fitP95: rec.fitP95,
|
|
23838
|
+
chargedBasedEstimateCoins: rec.chargedBasedEstimateCoins,
|
|
23839
|
+
lengthRaisedPrice: rec.lengthRaisedPrice,
|
|
23840
|
+
cutOffSamples: rec.cutOffSamples,
|
|
23841
|
+
noRecommendationReason: rec.noRecommendationReason
|
|
23780
23842
|
};
|
|
23781
23843
|
const savedTo = await saveBench(cwd, record);
|
|
23782
23844
|
if (opts.json) {
|
|
23845
|
+
const ok = charged.length > 0 && !lengthBlocked;
|
|
23783
23846
|
writeJsonLine({
|
|
23784
23847
|
command: "llm bench",
|
|
23785
|
-
status:
|
|
23786
|
-
error:
|
|
23848
|
+
status: ok ? "ok" : "failed",
|
|
23849
|
+
error: ok ? null : rec.noRecommendationReason ?? "no_sample_settled",
|
|
23787
23850
|
apiUrl,
|
|
23788
23851
|
projectId,
|
|
23789
23852
|
modelId,
|
|
@@ -23800,15 +23863,34 @@ async function runBench(args) {
|
|
|
23800
23863
|
// samples, never among them: no attempt existed and nothing was spent.
|
|
23801
23864
|
refusal,
|
|
23802
23865
|
chargedCoins: { p50, p95, max },
|
|
23866
|
+
// The answer's LENGTH, priced by the server per sample (`fitCoins`,
|
|
23867
|
+
// headroom already on the tokens), over the settled samples.
|
|
23868
|
+
fitCoins: { p95: rec.fitP95, max: rec.fitMax },
|
|
23869
|
+
chargedBasedEstimateCoins: rec.chargedBasedEstimateCoins,
|
|
23870
|
+
lengthRaisedPrice: rec.lengthRaisedPrice,
|
|
23871
|
+
cutOffSamples: rec.cutOffSamples,
|
|
23872
|
+
exceedsStandCap: rec.exceedsStandCap,
|
|
23873
|
+
noRecommendationReason: rec.noRecommendationReason,
|
|
23803
23874
|
recommendedDeclaredHeadroomBps: headroomBps,
|
|
23804
23875
|
recommendedEstimateCoins: recommended,
|
|
23805
23876
|
recommendedPerCallMaxCoins: ceiling,
|
|
23806
23877
|
savedTo
|
|
23807
23878
|
});
|
|
23808
|
-
if (
|
|
23879
|
+
if (!ok) process.exitCode = 1;
|
|
23809
23880
|
return;
|
|
23810
23881
|
}
|
|
23811
23882
|
log.plain("");
|
|
23883
|
+
if (lengthBlocked) {
|
|
23884
|
+
if (charged.length > 0) {
|
|
23885
|
+
log.plain(c.bold(" Charged coins"));
|
|
23886
|
+
log.plain(` p50 ${p50} p95 ${p95} max ${max} (${charged.length} of ${samples} settled)`);
|
|
23887
|
+
log.plain("");
|
|
23888
|
+
}
|
|
23889
|
+
printRecommendation(log, record);
|
|
23890
|
+
if (savedTo) log.dim(` Saved to ${savedTo}.`);
|
|
23891
|
+
process.exitCode = 1;
|
|
23892
|
+
return;
|
|
23893
|
+
}
|
|
23812
23894
|
if (charged.length === 0) {
|
|
23813
23895
|
log.error(
|
|
23814
23896
|
providerRefused > 0 && providerRefused === results.length ? " No sample ran \u2014 the provider refused every attempt at its door \u2014 so there is nothing to price from." : " No sample settled, so there is nothing to price from."
|
|
@@ -23826,6 +23908,14 @@ async function runBench(args) {
|
|
|
23826
23908
|
if (savedTo) log.dim(` Saved to ${savedTo} \u2014 re-read it any time with ${c.cyan("genex llm price")}.`);
|
|
23827
23909
|
}
|
|
23828
23910
|
function printRecommendation(log, record) {
|
|
23911
|
+
if (record.noRecommendationReason === "answer_exceeds_stand_cap") {
|
|
23912
|
+
log.warn(` ${LLM_EXCEEDS_STAND_CAP_LINE}`);
|
|
23913
|
+
return;
|
|
23914
|
+
}
|
|
23915
|
+
if (record.noRecommendationReason === "answer_cut_off") {
|
|
23916
|
+
log.warn(` ${benchCutOffLine(record.cutOffSamples ?? 1, record.maxCoinsPerSample ?? 0)}`);
|
|
23917
|
+
return;
|
|
23918
|
+
}
|
|
23829
23919
|
if (record.recommendedEstimateCoins === null || record.p95 === null) {
|
|
23830
23920
|
log.warn(" No price recommended.");
|
|
23831
23921
|
log.dim(
|
|
@@ -23835,14 +23925,26 @@ function printRecommendation(log, record) {
|
|
|
23835
23925
|
return;
|
|
23836
23926
|
}
|
|
23837
23927
|
log.plain(c.bold(` Declare estimateCoins: ${record.recommendedEstimateCoins}`));
|
|
23838
|
-
|
|
23839
|
-
`
|
|
23840
|
-
|
|
23841
|
-
|
|
23928
|
+
if (record.lengthRaisedPrice) {
|
|
23929
|
+
log.plain(` ${LLM_LENGTH_RAISED_LINE}`);
|
|
23930
|
+
log.dim(
|
|
23931
|
+
` = the smallest price whose answer allowance holds the answer plus this stand's headroom (p95 ${record.fitP95 ?? "\u2014"}), above the`
|
|
23932
|
+
);
|
|
23933
|
+
log.dim(
|
|
23934
|
+
` charged-based ${record.chargedBasedEstimateCoins ?? "\u2014"} (p95 of charged coins ${record.p95} plus ${record.recommendedDeclaredHeadroomBps} bps), over ${record.completed} settled attempt${record.completed === 1 ? "" : "s"} on ${record.modelId}.`
|
|
23935
|
+
);
|
|
23936
|
+
} else {
|
|
23937
|
+
log.dim(
|
|
23938
|
+
` = p95 of charged coins (${record.p95}) plus this stand's recommended headroom (${record.recommendedDeclaredHeadroomBps} bps),`
|
|
23939
|
+
);
|
|
23940
|
+
log.dim(` measured over ${record.completed} settled attempt${record.completed === 1 ? "" : "s"} on ${record.modelId}.`);
|
|
23941
|
+
}
|
|
23842
23942
|
log.dim(" That number is a PRICE: a started attempt is charged in full, including one that fails.");
|
|
23843
23943
|
if (record.recommendedPerCallMaxCoins != null) {
|
|
23844
23944
|
log.plain(c.bold(` Grant perCallMaxCoins: ${record.recommendedPerCallMaxCoins}`));
|
|
23845
|
-
log.dim(
|
|
23945
|
+
log.dim(
|
|
23946
|
+
record.lengthRaisedPrice ? ` = the same headroom over the worst sample (${record.max}) or the longest answer's price, whichever is larger, and never below the price above it.` : ` = the same headroom over the worst sample (${record.max}), and never below the price above it.`
|
|
23947
|
+
);
|
|
23846
23948
|
log.dim(" The ceiling is the price's room to be wrong; declaring one under the price is refused.");
|
|
23847
23949
|
}
|
|
23848
23950
|
}
|
|
@@ -23952,6 +24054,12 @@ function emptyBenchJson(apiUrl, projectId, prompt, outputFormat, samples, maxCoi
|
|
|
23952
24054
|
results: [],
|
|
23953
24055
|
refusal: null,
|
|
23954
24056
|
chargedCoins: { p50: null, p95: null, max: null },
|
|
24057
|
+
fitCoins: { p95: null, max: null },
|
|
24058
|
+
chargedBasedEstimateCoins: null,
|
|
24059
|
+
lengthRaisedPrice: false,
|
|
24060
|
+
cutOffSamples: 0,
|
|
24061
|
+
exceedsStandCap: false,
|
|
24062
|
+
noRecommendationReason: null,
|
|
23955
24063
|
recommendedDeclaredHeadroomBps: null,
|
|
23956
24064
|
recommendedEstimateCoins: null,
|
|
23957
24065
|
recommendedPerCallMaxCoins: null,
|
|
@@ -24138,6 +24246,11 @@ async function reportSavedPrice(cwd, opts, log) {
|
|
|
24138
24246
|
p50: null,
|
|
24139
24247
|
p95: null,
|
|
24140
24248
|
max: null,
|
|
24249
|
+
fitP95: null,
|
|
24250
|
+
chargedBasedEstimateCoins: null,
|
|
24251
|
+
lengthRaisedPrice: false,
|
|
24252
|
+
cutOffSamples: 0,
|
|
24253
|
+
noRecommendationReason: null,
|
|
24141
24254
|
recommendedDeclaredHeadroomBps: null,
|
|
24142
24255
|
recommendedEstimateCoins: null,
|
|
24143
24256
|
recommendedPerCallMaxCoins: null,
|
|
@@ -24154,7 +24267,7 @@ async function reportSavedPrice(cwd, opts, log) {
|
|
|
24154
24267
|
writeJsonLine({
|
|
24155
24268
|
command: "llm price",
|
|
24156
24269
|
status: record.recommendedEstimateCoins === null ? "failed" : "ok",
|
|
24157
|
-
error: record.recommendedEstimateCoins === null ? "no_recommendation" : null,
|
|
24270
|
+
error: record.recommendedEstimateCoins === null ? record.noRecommendationReason ?? "no_recommendation" : null,
|
|
24158
24271
|
ranAt: record.ranAt ?? null,
|
|
24159
24272
|
modelId: record.modelId ?? null,
|
|
24160
24273
|
samples: record.samples ?? null,
|
|
@@ -24162,6 +24275,12 @@ async function reportSavedPrice(cwd, opts, log) {
|
|
|
24162
24275
|
p50: record.p50 ?? null,
|
|
24163
24276
|
p95: record.p95 ?? null,
|
|
24164
24277
|
max: record.max ?? null,
|
|
24278
|
+
// Absent from a file an older CLI wrote; null / false / 0 then, never missing.
|
|
24279
|
+
fitP95: record.fitP95 ?? null,
|
|
24280
|
+
chargedBasedEstimateCoins: record.chargedBasedEstimateCoins ?? null,
|
|
24281
|
+
lengthRaisedPrice: record.lengthRaisedPrice ?? false,
|
|
24282
|
+
cutOffSamples: record.cutOffSamples ?? 0,
|
|
24283
|
+
noRecommendationReason: record.noRecommendationReason ?? null,
|
|
24165
24284
|
recommendedDeclaredHeadroomBps: record.recommendedDeclaredHeadroomBps ?? null,
|
|
24166
24285
|
recommendedEstimateCoins: record.recommendedEstimateCoins ?? null,
|
|
24167
24286
|
recommendedPerCallMaxCoins: record.recommendedPerCallMaxCoins ?? null,
|
|
@@ -27103,6 +27222,12 @@ ${c.bold("Why a benchmark and not an estimate")}
|
|
|
27103
27222
|
that serves no headroom gets no recommendation, because the multiplier is not the CLI's to
|
|
27104
27223
|
invent.
|
|
27105
27224
|
|
|
27225
|
+
The declared price also decides how LONG the answer may be: it sizes each call's output
|
|
27226
|
+
allowance. So the recommendation is never below the smallest price that leaves room for
|
|
27227
|
+
the benchmarked answer, as the server computes it. A sample cut off at your --max-coins
|
|
27228
|
+
(provider_token_limit) is not a sample \u2014 re-run with a higher --max-coins; an answer longer
|
|
27229
|
+
than one call on this stand may produce gets no price at all \u2014 ask for a shorter one.
|
|
27230
|
+
|
|
27106
27231
|
Results land in ${LLM_BENCH_FILE} \u2014 gitignored, machine-local, re-read by \`genex llm price\`.
|
|
27107
27232
|
`,
|
|
27108
27233
|
player: `${c.bold("genex player")} \u2014 answer the in-game model requests you approve, on your own Claude or ChatGPT subscription.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@genex-ai/cli-demo",
|
|
3
|
-
"version": "1.35.0-dev.
|
|
3
|
+
"version": "1.35.0-dev.746",
|
|
4
4
|
"description": "Set up your project's agent workspace (.claude/.codex/.cursor in the game folder), authorize, create a game project, generate AI assets, and publish (genex CLI).",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -120,6 +120,14 @@ the price and makes `requestSpendGrant()` refuse before it reaches the network.
|
|
|
120
120
|
loop into disclosure numbers, re-benchmarking after a prompt change — is in
|
|
121
121
|
[references/pricing.md](references/pricing.md).
|
|
122
122
|
|
|
123
|
+
**The declared price also decides how long the answer may be.** Each call's
|
|
124
|
+
room to answer is funded from the price it declares, so a price that covers
|
|
125
|
+
what an answer cost can still cut it off. The bench's recommendation already
|
|
126
|
+
leaves that room — always declare what it prints, never the bare charged
|
|
127
|
+
number. A sample cut off at your `--max-coins` is not a sample: re-run with a
|
|
128
|
+
higher one. When the bench says the answer is longer than one call on this
|
|
129
|
+
stand may produce, ask for a shorter answer.
|
|
130
|
+
|
|
123
131
|
Only samples that **succeeded and settled** are priced from. A sample the
|
|
124
132
|
provider refused at its door (`provider_http_<status>`) ran no inference and
|
|
125
133
|
cost nothing; the bench prints the code, the provider's own message and, for a
|
|
@@ -272,7 +280,7 @@ These are source contracts, not a claim that every stand runs this lane —
|
|
|
272
280
|
- [ ] `npx genex llm models` was run and its verdict is in the handoff
|
|
273
281
|
- [ ] Model ids come from `getGenerationModels()`, never from source
|
|
274
282
|
- [ ] The picker offers the featured set, labelled with the server's own `label`
|
|
275
|
-
- [ ] `estimateCoins`
|
|
283
|
+
- [ ] `estimateCoins` is the figure `npx genex llm bench` printed, verbatim — never the bare charged number
|
|
276
284
|
- [ ] The schema uses only the accepted dialect (no `$ref`, `pattern`, `format`, `anyOf`, `default`)
|
|
277
285
|
- [ ] A bench refused as `generation_limit` was answered with `npx genex llm status`, never a retry
|
|
278
286
|
- [ ] `generate()` / `requestSpendGrant()` is the first statement of a click handler
|
|
@@ -322,6 +330,13 @@ message (`npx genex llm status` prints it under the row). A 401 or 403 is this
|
|
|
322
330
|
stand's provider configuration refusing the model — tell the operator, and
|
|
323
331
|
build nothing around it in the game.
|
|
324
332
|
|
|
333
|
+
**`provider_token_limit`** — the answer was cut off: the model ran out of room
|
|
334
|
+
before it finished, and the attempt is still charged. The declared price is too
|
|
335
|
+
low for the answer's length — re-benchmark and declare what the bench prints,
|
|
336
|
+
never the bare charged number — or the answer is longer than one call on this
|
|
337
|
+
stand may produce, and the fix is a shorter answer (fewer fields, shorter
|
|
338
|
+
strings, a length the prompt states).
|
|
339
|
+
|
|
325
340
|
**`invalid_schema`** — the schema uses a keyword outside the accepted dialect
|
|
326
341
|
(`$ref`, `pattern`, `format`, `anyOf`, `default`, `examples`, `$schema`), or an
|
|
327
342
|
object without `additionalProperties: false`. Nothing was charged. Rewrite it in
|
|
@@ -63,6 +63,13 @@ own once its bill resolves, and stops holding a slot ten minutes after
|
|
|
63
63
|
dispatch. Re-running the bench into the same refusal spends nothing and
|
|
64
64
|
learns nothing.
|
|
65
65
|
|
|
66
|
+
A sample the model had to stop writing (`provider_token_limit`) was cut off at
|
|
67
|
+
your `--max-coins`: it was charged, it is not a sample, and its real length is
|
|
68
|
+
unknown — so the run recommends no price and asks you to re-run with a higher
|
|
69
|
+
`--max-coins`. When the stand itself cannot hold the answer, the run says the
|
|
70
|
+
answer is longer than one call there may produce; no price fixes that, a
|
|
71
|
+
shorter answer does.
|
|
72
|
+
|
|
66
73
|
- **p50** is what a typical call costs. It is the number to reason about when
|
|
67
74
|
you ask "can the game afford this loop?" — multiply it by the calls per
|
|
68
75
|
minute you are about to disclose.
|
|
@@ -75,7 +82,14 @@ learns nothing.
|
|
|
75
82
|
Fix the prompt rather than declaring a bigger number.
|
|
76
83
|
|
|
77
84
|
The recommendation line already applies the **server's own recommended
|
|
78
|
-
headroom** on top of p95.
|
|
85
|
+
headroom** on top of p95. It also covers the answer's **length**: the declared
|
|
86
|
+
price decides how long each call's answer may be, because the room to answer is
|
|
87
|
+
funded from it, so a price built from charged coins alone can cut the answer
|
|
88
|
+
off in the game while the bench — run under a larger `--max-coins` — never saw
|
|
89
|
+
it. The recommendation is never below the smallest price that leaves room for
|
|
90
|
+
the benchmarked answer, and when the length is what set it the run says so in
|
|
91
|
+
one sentence. Either way, declare that figure verbatim — never the bare charged
|
|
92
|
+
number:
|
|
79
93
|
|
|
80
94
|
```ts
|
|
81
95
|
const NPC_CALL_PRICE = <the recommended figure>; // from `npx genex llm bench`, <date>
|