@trazum/cli 1.50.4 → 1.50.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/i18n/en.d.ts.map +1 -1
- package/dist/i18n/en.js +58 -0
- package/dist/i18n/en.js.map +1 -1
- package/dist/i18n/es.d.ts.map +1 -1
- package/dist/i18n/es.js +58 -0
- package/dist/i18n/es.js.map +1 -1
- package/dist/i18n/types.d.ts +31 -0
- package/dist/i18n/types.d.ts.map +1 -1
- package/dist/index.js +156 -1
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/i18n/en.ts +72 -0
- package/src/i18n/es.ts +72 -0
- package/src/i18n/types.ts +27 -0
- package/src/index.ts +208 -0
package/dist/index.js
CHANGED
|
@@ -4,7 +4,7 @@ import { open, readdir, readFile, stat, writeFile } from 'node:fs/promises';
|
|
|
4
4
|
import { dirname, join, resolve as resolvePath } from 'node:path';
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { gunzipSync } from 'node:zlib';
|
|
7
|
-
import { applyRewrites, BASELINE_FILENAME, BASELINE_VERSION, breaches, cacheableMinimum, analyzeCachePrefix, billLevers, bucketedCacheEconomics, bucketedProfile, buildHistory, buildPlan, connectorFor, CONNECTORS, normalizeAnthropicUsage, normalizeOpenAIUsage, bucketsFromRecords, evaluateWatch, firedKey, pruneRecords, recordsFromBuckets, storeInventory, storedReportFrom, verifyPlan, cacheEconomics, cacheHitRate, contextPressure, comparePrompts, compareToBaseline, computeSavings, countTokensAnthropic, DEFAULT_USAGE, budgetPositions, conform, outcomeReport, FAILURE_POLICIES, detectFromSource, matchLocale, parsePlanDocument, waiverDay, waiverHistory, proposeInit, MIN_RATE_DAYS, parseConfig, coverageDrift, driversBetween, explainGateFailure, assignSources, fleetRollup, labelCoverage, measuredUsage, gateMargin, GATE_MARGIN_TIGHT, estimateTokens, evaluate, extractPrompts, findExamples, formatBaseline, formatSignedUsd, formatUsd, getMessages, getModel, hasMarker, LOCALES, MAX_BASELINE_BYTES, moneyIsComparable, mostSpecificMatch, nearestName, optimize, parseBaseline, PHRASE_LANGUAGES, plannedCalls, profilePrompt, profileToCsv, profileUsage, promptId, providerFromEnv, pruneExamples, refineWithLlm, rejectionText, reorderForCache, repriceProfile, reviewAgeDays, reviewExamples, RULES, sharedPrefixes, sharesOf, SOURCE_EXTENSIONS, suggestRewrites, toOtlpMetrics, toPromptfoo, TTL_1H_MS, UNLABELLED, withExactTokenCounts, } from '@trazum/core';
|
|
7
|
+
import { applyRewrites, BASELINE_FILENAME, BASELINE_VERSION, breaches, cacheableMinimum, analyzeCachePrefix, billLevers, bucketedCacheEconomics, bucketedProfile, buildHistory, buildPlan, connectorFor, CONNECTORS, normalizeAnthropicUsage, normalizeOpenAIUsage, bucketsFromRecords, evaluateWatch, firedKey, pruneRecords, recordsFromBuckets, storeInventory, storedReportFrom, verifyPlan, cacheEconomics, cacheHitRate, contextPressure, comparePrompts, compareToBaseline, computeSavings, countTokensAnthropic, DEFAULT_USAGE, budgetPositions, conform, BREAK_EVEN_BAND, ladderPosition, validateLadder, outcomeReport, rankPerOutcome, FAILURE_POLICIES, detectFromSource, matchLocale, parsePlanDocument, waiverDay, waiverHistory, proposeInit, MIN_RATE_DAYS, parseConfig, coverageDrift, driversBetween, explainGateFailure, assignSources, fleetRollup, labelCoverage, measuredUsage, gateMargin, GATE_MARGIN_TIGHT, estimateTokens, evaluate, extractPrompts, findExamples, formatBaseline, formatSignedUsd, formatUsd, getMessages, getModel, hasMarker, LOCALES, MAX_BASELINE_BYTES, moneyIsComparable, mostSpecificMatch, nearestName, optimize, parseBaseline, PHRASE_LANGUAGES, plannedCalls, profilePrompt, profileToCsv, profileUsage, promptId, providerFromEnv, pruneExamples, refineWithLlm, rejectionText, reorderForCache, repriceProfile, reviewAgeDays, reviewExamples, RULES, sharedPrefixes, sharesOf, SOURCE_EXTENSIONS, suggestRewrites, toOtlpMetrics, toPromptfoo, TTL_1H_MS, UNLABELLED, withExactTokenCounts, } from '@trazum/core';
|
|
8
8
|
import { cacheDir, cacheStats, cachingProvider, clearCache } from './suggest-cache.js';
|
|
9
9
|
import { dayOf, formatGap, median, spanDays } from './time.js';
|
|
10
10
|
// Everything that reads the filesystem, on its own entry point so the web
|
|
@@ -319,6 +319,7 @@ const COMMAND_FLAGS = {
|
|
|
319
319
|
conform: ['contract', 'json'],
|
|
320
320
|
feedback: [],
|
|
321
321
|
gateway: ['on-cannot-tell', 'port', 'socket', 'pricing', 'pricing-live'],
|
|
322
|
+
ladder: ['pricing', 'pricing-live', 'since', 'until', 'label'],
|
|
322
323
|
where: [],
|
|
323
324
|
rules: [],
|
|
324
325
|
blame: ['limit', 'model', 'calls', 'output-tokens', 'batch', 'prompt', 'markdown-out'],
|
|
@@ -1665,6 +1666,109 @@ async function commandGateway(args, config, configDir, pricing, t) {
|
|
|
1665
1666
|
console.log(` ${c.dim(wrap(t.gateway.policy(policyFlag), 74, ' '))}`);
|
|
1666
1667
|
console.log();
|
|
1667
1668
|
}
|
|
1669
|
+
/**
|
|
1670
|
+
* `trazum ladder <log>` — is the ladder saving money, or is it a bill?
|
|
1671
|
+
*
|
|
1672
|
+
* The one number this command exists to print is the **break-even escalation
|
|
1673
|
+
* rate**. "We route to the cheap model first" describes a policy that saves
|
|
1674
|
+
* money and a policy that costs money equally well; only the rate separates
|
|
1675
|
+
* them, and nobody works it out in their head because the shape of the
|
|
1676
|
+
* arithmetic is not obvious — an escalation pays twice, since the cheap
|
|
1677
|
+
* attempt is not refunded.
|
|
1678
|
+
*/
|
|
1679
|
+
async function commandLadder(args, config, pricing, t) {
|
|
1680
|
+
const path = args.positional[0];
|
|
1681
|
+
if (path === undefined) {
|
|
1682
|
+
throw new Error(t.errors.missingInputFile());
|
|
1683
|
+
}
|
|
1684
|
+
const report = profileUsage(await readUsageLog(path, t), { catalogue: pricing });
|
|
1685
|
+
const ladders = config.ladders ?? {};
|
|
1686
|
+
const n = (value) => value.toLocaleString(t.numberLocale);
|
|
1687
|
+
const pct = (value) => `${(value * 100).toFixed(1)}%`;
|
|
1688
|
+
console.log();
|
|
1689
|
+
console.log(c.bold(t.ladder.heading()));
|
|
1690
|
+
if (Object.keys(ladders).length === 0) {
|
|
1691
|
+
console.log(` ${c.dim(wrap(t.ladder.noLadders(), 74, ' '))}`);
|
|
1692
|
+
console.log();
|
|
1693
|
+
return;
|
|
1694
|
+
}
|
|
1695
|
+
console.log(` ${c.dim(wrap(t.ladder.theDoubleSpend(), 74, ' '))}`);
|
|
1696
|
+
console.log();
|
|
1697
|
+
const vocabulary = config.outcomes ?? null;
|
|
1698
|
+
let anyProblem = false;
|
|
1699
|
+
for (const [label, policy] of Object.entries(ladders)) {
|
|
1700
|
+
/**
|
|
1701
|
+
* Validated before it is measured, and loudly.
|
|
1702
|
+
*
|
|
1703
|
+
* A ladder that escalates on a value declared a *success* pays twice for
|
|
1704
|
+
* work that already worked, on every call, while looking exactly like a
|
|
1705
|
+
* cost-saving measure in the config. Printing its measured position first
|
|
1706
|
+
* would bury that under a number.
|
|
1707
|
+
*/
|
|
1708
|
+
const problems = validateLadder(policy, vocabulary, pricing);
|
|
1709
|
+
if (problems.length > 0) {
|
|
1710
|
+
anyProblem = true;
|
|
1711
|
+
console.log(` ${c.red('✗')} ${c.bold(t.ladder.problemsHeading(label))}`);
|
|
1712
|
+
for (const problem of problems) {
|
|
1713
|
+
const detail = 'value' in problem
|
|
1714
|
+
? problem.value
|
|
1715
|
+
: 'model' in problem
|
|
1716
|
+
? problem.model
|
|
1717
|
+
: String(problem.tiers);
|
|
1718
|
+
console.log(` ${wrap(t.ladder.problem(problem.kind, detail), 70, ' ')}`);
|
|
1719
|
+
}
|
|
1720
|
+
console.log();
|
|
1721
|
+
continue;
|
|
1722
|
+
}
|
|
1723
|
+
const slice = report.outcomeTallyByLabel.find((entry) => entry.label === label);
|
|
1724
|
+
const breakdown = report.byLabel.find((entry) => entry.label === label);
|
|
1725
|
+
/**
|
|
1726
|
+
* The shape of the work comes from the measured calls, so the break-even
|
|
1727
|
+
* rate is priced against what this workload actually sends rather than
|
|
1728
|
+
* against a token count somebody guessed at.
|
|
1729
|
+
*/
|
|
1730
|
+
const calls = breakdown?.breakdown.calls ?? 0;
|
|
1731
|
+
const shape = breakdown === undefined || calls === 0
|
|
1732
|
+
? { inputTokens: 0, outputTokens: 0 }
|
|
1733
|
+
: {
|
|
1734
|
+
inputTokens: Math.round((breakdown.breakdown.inputTokens +
|
|
1735
|
+
breakdown.breakdown.cacheReadTokens +
|
|
1736
|
+
breakdown.breakdown.cacheWriteTokens) /
|
|
1737
|
+
calls),
|
|
1738
|
+
outputTokens: Math.round(breakdown.breakdown.outputTokens / calls),
|
|
1739
|
+
};
|
|
1740
|
+
const empty = { byValue: [], recorded: 0, parsed: 0, unrecordedUsd: 0 };
|
|
1741
|
+
const position = ladderPosition(policy, slice?.tally ?? empty, shape, vocabulary, pricing);
|
|
1742
|
+
console.log(` ${c.bold(t.ladder.workload(label))} ${c.dim(policy.tiers.join(' → '))}`);
|
|
1743
|
+
console.log(` ${c.dim(t.ladder.arithmetic(formatUsd(position.arithmetic.cheapUsd), formatUsd(position.arithmetic.dearUsd), position.arithmetic.breakEvenRate === null ? '—' : pct(position.arithmetic.breakEvenRate)))}`);
|
|
1744
|
+
if (position.verdict === 'cannot-tell') {
|
|
1745
|
+
console.log(` ${c.yellow('?')} ${wrap(t.ladder.cannotTell(position.unknown ?? '', n(position.calls)), 70, ' ')}`);
|
|
1746
|
+
}
|
|
1747
|
+
else {
|
|
1748
|
+
console.log(` ${t.ladder.measured(pct(position.measuredRate ?? 0), n(position.escalations), n(position.calls))}`);
|
|
1749
|
+
const delta = formatUsd(Math.abs(position.deltaUsdPerCall ?? 0));
|
|
1750
|
+
if (position.verdict === 'saving') {
|
|
1751
|
+
console.log(` ${c.green('✓')} ${wrap(t.ladder.saving(delta), 70, ' ')}`);
|
|
1752
|
+
}
|
|
1753
|
+
else if (position.verdict === 'costing') {
|
|
1754
|
+
console.log(` ${c.red('✗')} ${wrap(t.ladder.costing(delta), 70, ' ')}`);
|
|
1755
|
+
}
|
|
1756
|
+
else {
|
|
1757
|
+
console.log(` ${c.dim('·')} ${wrap(t.ladder.atBreakEven(pct(BREAK_EVEN_BAND)), 70, ' ')}`);
|
|
1758
|
+
}
|
|
1759
|
+
}
|
|
1760
|
+
console.log();
|
|
1761
|
+
}
|
|
1762
|
+
console.log(` ${c.dim(wrap(t.ladder.notExecuted(), 74, ' '))}`);
|
|
1763
|
+
console.log();
|
|
1764
|
+
/**
|
|
1765
|
+
* A misconfigured ladder fails the command, because it is the one finding
|
|
1766
|
+
* here that is wrong *now* rather than a measurement somebody should look
|
|
1767
|
+
* at. Everything else exits 0: this is a survey, like `doctor`.
|
|
1768
|
+
*/
|
|
1769
|
+
if (anyProblem)
|
|
1770
|
+
process.exitCode = 1;
|
|
1771
|
+
}
|
|
1668
1772
|
function commandModels(t, pricing) {
|
|
1669
1773
|
const n = (value) => value.toLocaleString(t.numberLocale);
|
|
1670
1774
|
const col = t.models.columns;
|
|
@@ -5268,6 +5372,54 @@ async function commandProfile(args, config, configDir, pricing, t) {
|
|
|
5268
5372
|
if (outcomes.coverage.unrecordedUsd > 0 && report.total.totalUsd > 0) {
|
|
5269
5373
|
console.log(` ${c.yellow('!')} ${wrap(t.profile.outcomeUnrecorded(pct(outcomes.coverage.unrecordedUsd / report.total.totalUsd), formatUsd(outcomes.coverage.unrecordedUsd)), 74, ' ')}`);
|
|
5270
5374
|
}
|
|
5375
|
+
/**
|
|
5376
|
+
* What an outcome costs, per workload — the finding a total cannot make.
|
|
5377
|
+
*
|
|
5378
|
+
* Printed only when at least one workload has enough recorded outcomes
|
|
5379
|
+
* to say something, and every row that cannot state a figure says which
|
|
5380
|
+
* of the five reasons applies rather than showing a blank.
|
|
5381
|
+
*/
|
|
5382
|
+
const ranking = rankPerOutcome(report.outcomeTallyByLabel.map((entry) => ({
|
|
5383
|
+
key: entry.label,
|
|
5384
|
+
calls: entry.calls,
|
|
5385
|
+
totalUsd: entry.totalUsd,
|
|
5386
|
+
tally: entry.tally,
|
|
5387
|
+
})), config.outcomes ?? null);
|
|
5388
|
+
if (ranking.byCall.length > 0) {
|
|
5389
|
+
console.log();
|
|
5390
|
+
console.log(c.bold(t.profile.perOutcomeHeading()));
|
|
5391
|
+
const pcol = t.profile.perOutcomeColumns;
|
|
5392
|
+
const prows = ranking.byCall.map((slice) => ({
|
|
5393
|
+
key: slice.key,
|
|
5394
|
+
perCall: formatUsd(slice.usdPerCall),
|
|
5395
|
+
perOutcome: slice.per.usdPerSuccess !== null
|
|
5396
|
+
? formatUsd(slice.per.usdPerSuccess)
|
|
5397
|
+
: t.profile.perOutcomeWithheld(slice.per.withheld ?? 'no-vocabulary', n(slice.per.successes), pct(slice.per.coverage)),
|
|
5398
|
+
recorded: pct(slice.per.coverage),
|
|
5399
|
+
}));
|
|
5400
|
+
const pw = {
|
|
5401
|
+
key: Math.max(...prows.map((r) => r.key.length), pcol.workload.length),
|
|
5402
|
+
perCall: Math.max(...prows.map((r) => r.perCall.length), pcol.perCall.length),
|
|
5403
|
+
perOutcome: Math.max(...prows.map((r) => r.perOutcome.length), pcol.perOutcome.length),
|
|
5404
|
+
recorded: Math.max(...prows.map((r) => r.recorded.length), pcol.recorded.length),
|
|
5405
|
+
};
|
|
5406
|
+
console.log(c.dim(` ${pcol.workload.padEnd(pw.key)} ${pcol.perCall.padStart(pw.perCall)} ` +
|
|
5407
|
+
`${pcol.perOutcome.padStart(pw.perOutcome)} ${pcol.recorded.padStart(pw.recorded)}`));
|
|
5408
|
+
for (const row of prows) {
|
|
5409
|
+
console.log(` ${row.key.padEnd(pw.key)} ${row.perCall.padStart(pw.perCall)} ` +
|
|
5410
|
+
`${row.perOutcome.padStart(pw.perOutcome)} ${c.dim(row.recorded.padStart(pw.recorded))}`);
|
|
5411
|
+
}
|
|
5412
|
+
console.log();
|
|
5413
|
+
console.log(` ${c.dim(wrap(t.profile.perOutcomeNumerator(), 74, ' '))}`);
|
|
5414
|
+
// The disagreement between the two orders, which is itself the finding.
|
|
5415
|
+
if (ranking.disagreements.length > 0) {
|
|
5416
|
+
console.log();
|
|
5417
|
+
console.log(` ${c.dim(wrap(t.profile.perOutcomeBothOrders(), 74, ' '))}`);
|
|
5418
|
+
for (const d of ranking.disagreements) {
|
|
5419
|
+
console.log(` ${c.yellow('→')} ${wrap(t.profile.perOutcomeDisagreement(d.slice.key, n(d.callRank + 1), n(d.outcomeRank + 1)), 74, ' ')}`);
|
|
5420
|
+
}
|
|
5421
|
+
}
|
|
5422
|
+
}
|
|
5271
5423
|
if (outcomes.undeclared.length > 0) {
|
|
5272
5424
|
console.log(` ${c.yellow('!')} ${wrap(t.profile.outcomeUndeclared(outcomes.undeclared.map((s) => s.value).join(', ')), 74, ' ')}`);
|
|
5273
5425
|
}
|
|
@@ -6664,6 +6816,9 @@ async function main() {
|
|
|
6664
6816
|
case 'models':
|
|
6665
6817
|
commandModels(t, pricing);
|
|
6666
6818
|
break;
|
|
6819
|
+
case 'ladder':
|
|
6820
|
+
await commandLadder(args, config, pricing, t);
|
|
6821
|
+
break;
|
|
6667
6822
|
case 'gateway':
|
|
6668
6823
|
await commandGateway(args, config, configDir, pricing, t);
|
|
6669
6824
|
break;
|