@trazum/cli 1.50.7 → 1.50.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/i18n/en.d.ts.map +1 -1
- package/dist/i18n/en.js +42 -0
- package/dist/i18n/en.js.map +1 -1
- package/dist/i18n/es.d.ts.map +1 -1
- package/dist/i18n/es.js +42 -0
- package/dist/i18n/es.js.map +1 -1
- package/dist/i18n/types.d.ts +15 -0
- package/dist/i18n/types.d.ts.map +1 -1
- package/dist/index.js +116 -1
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/i18n/en.ts +52 -0
- package/src/i18n/es.ts +52 -0
- package/src/i18n/types.ts +16 -0
- package/src/index.ts +153 -0
package/dist/index.js
CHANGED
|
@@ -4,7 +4,7 @@ import { open, readdir, readFile, stat, writeFile } from 'node:fs/promises';
|
|
|
4
4
|
import { dirname, join, resolve as resolvePath } from 'node:path';
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { gunzipSync } from 'node:zlib';
|
|
7
|
-
import { applyRewrites, BASELINE_FILENAME, BASELINE_VERSION, breaches, cacheableMinimum, analyzeCachePrefix, billLevers, bucketedCacheEconomics, bucketedProfile, buildHistory, buildPlan, connectorFor, CONNECTORS, normalizeAnthropicUsage, normalizeOpenAIUsage, bucketsFromRecords, evaluateWatch, firedKey, pruneRecords, recordsFromBuckets, storeInventory, storedReportFrom, verifyPlan, cacheEconomics, cacheHitRate, contextPressure, comparePrompts, compareToBaseline, computeSavings, countTokensAnthropic, DEFAULT_USAGE, budgetPositions, conform, BREAK_EVEN_BAND, runExperiment, ladderPosition, validateLadder, outcomeReport, rankPerOutcome, FAILURE_POLICIES, detectFromSource, matchLocale, parsePlanDocument, waiverDay, waiverHistory, proposeInit, MIN_RATE_DAYS, parseConfig, coverageDrift, driversBetween, explainGateFailure, assignSources, fleetRollup, labelCoverage, measuredUsage, gateMargin, GATE_MARGIN_TIGHT, estimateTokens, evaluate, extractPrompts, findExamples, formatBaseline, formatSignedUsd, formatUsd, getMessages, getModel, hasMarker, LOCALES, MAX_BASELINE_BYTES, moneyIsComparable, mostSpecificMatch, nearestName, optimize, parseBaseline, PHRASE_LANGUAGES, plannedCalls, profilePrompt, profileToCsv, profileUsage, promptId, providerFromEnv, pruneExamples, refineWithLlm, rejectionText, reorderForCache, repriceProfile, reviewAgeDays, reviewExamples, RULES, sharedPrefixes, sharesOf, SOURCE_EXTENSIONS, suggestRewrites, toOtlpMetrics, toPromptfoo, TTL_1H_MS, UNLABELLED, withExactTokenCounts, } from '@trazum/core';
|
|
7
|
+
import { applyRewrites, BASELINE_FILENAME, BASELINE_VERSION, breaches, cacheableMinimum, analyzeCachePrefix, billLevers, bucketedCacheEconomics, bucketedProfile, buildHistory, buildPlan, connectorFor, CONNECTORS, normalizeAnthropicUsage, normalizeOpenAIUsage, bucketsFromRecords, evaluateWatch, firedKey, pruneRecords, recordsFromBuckets, storeInventory, storedReportFrom, verifyPlan, cacheEconomics, cacheHitRate, contextPressure, comparePrompts, compareToBaseline, computeSavings, countTokensAnthropic, DEFAULT_USAGE, budgetPositions, conform, BREAK_EVEN_BAND, runExperiment, qualityGate, ladderPosition, validateLadder, outcomeReport, rankPerOutcome, FAILURE_POLICIES, detectFromSource, matchLocale, parsePlanDocument, waiverDay, waiverHistory, proposeInit, MIN_RATE_DAYS, parseConfig, coverageDrift, driversBetween, explainGateFailure, assignSources, fleetRollup, labelCoverage, measuredUsage, gateMargin, GATE_MARGIN_TIGHT, estimateTokens, evaluate, extractPrompts, findExamples, formatBaseline, formatSignedUsd, formatUsd, getMessages, getModel, hasMarker, LOCALES, MAX_BASELINE_BYTES, moneyIsComparable, mostSpecificMatch, nearestName, optimize, parseBaseline, PHRASE_LANGUAGES, plannedCalls, profilePrompt, profileToCsv, profileUsage, promptId, providerFromEnv, pruneExamples, refineWithLlm, rejectionText, reorderForCache, repriceProfile, reviewAgeDays, reviewExamples, RULES, sharedPrefixes, sharesOf, SOURCE_EXTENSIONS, suggestRewrites, toOtlpMetrics, toPromptfoo, TTL_1H_MS, UNLABELLED, withExactTokenCounts, } from '@trazum/core';
|
|
8
8
|
import { cacheDir, cacheStats, cachingProvider, clearCache } from './suggest-cache.js';
|
|
9
9
|
import { dayOf, formatGap, median, spanDays } from './time.js';
|
|
10
10
|
// Everything that reads the filesystem, on its own entry point so the web
|
|
@@ -33,6 +33,7 @@ const c = {
|
|
|
33
33
|
};
|
|
34
34
|
const VALUE_FLAGS = new Set([
|
|
35
35
|
'a',
|
|
36
|
+
'at',
|
|
36
37
|
'b',
|
|
37
38
|
'min-outcomes',
|
|
38
39
|
'against',
|
|
@@ -324,6 +325,7 @@ const COMMAND_FLAGS = {
|
|
|
324
325
|
gateway: ['on-cannot-tell', 'port', 'socket', 'pricing', 'pricing-live'],
|
|
325
326
|
ladder: ['pricing', 'pricing-live', 'since', 'until', 'label'],
|
|
326
327
|
experiment: ['a', 'b', 'min-outcomes', 'pricing', 'pricing-live'],
|
|
328
|
+
quality: ['label', 'at', 'gate', 'pricing', 'pricing-live'],
|
|
327
329
|
where: [],
|
|
328
330
|
rules: [],
|
|
329
331
|
blame: ['limit', 'model', 'calls', 'output-tokens', 'batch', 'prompt', 'markdown-out'],
|
|
@@ -1854,6 +1856,116 @@ async function commandExperiment(args, config, pricing, t) {
|
|
|
1854
1856
|
console.log(` ${c.dim(wrap(t.experiment.neverPromotes(), 74, ' '))}`);
|
|
1855
1857
|
console.log();
|
|
1856
1858
|
}
|
|
1859
|
+
/**
|
|
1860
|
+
* `trazum quality <log> --label <name> --at <iso> [--gate]`
|
|
1861
|
+
*
|
|
1862
|
+
* The failure that actually matters: a prompt edit that quietly made the
|
|
1863
|
+
* product worse. CI has been able to fail a build for tokens since 1.4 and for
|
|
1864
|
+
* dollars since 1.21, and this has never been gateable — so every saving this
|
|
1865
|
+
* tool has ever recommended went into a repository with its most important
|
|
1866
|
+
* consequence unmeasured.
|
|
1867
|
+
*
|
|
1868
|
+
* **Named `quality` rather than `check --against-outcomes`, which is what the
|
|
1869
|
+
* plan called for.** `check` reads *prompt files* and gates on tokens; it has
|
|
1870
|
+
* never opened a usage log, and a command that takes either a prompt or a log
|
|
1871
|
+
* depending on a flag is two commands wearing one name. The split-by-time this
|
|
1872
|
+
* needs is also not a `check` idea — there is nothing in a prompt file with a
|
|
1873
|
+
* timestamp on it.
|
|
1874
|
+
*/
|
|
1875
|
+
async function commandQuality(args, config, pricing, t) {
|
|
1876
|
+
const path = args.positional[0];
|
|
1877
|
+
if (path === undefined)
|
|
1878
|
+
throw new Error(t.errors.missingInputFile());
|
|
1879
|
+
const label = stringFlag(args, 'label');
|
|
1880
|
+
if (label === undefined)
|
|
1881
|
+
throw new Error(t.quality.needsLabel());
|
|
1882
|
+
const atRaw = stringFlag(args, 'at');
|
|
1883
|
+
const atMs = atRaw === undefined ? Number.NaN : Date.parse(atRaw);
|
|
1884
|
+
if (!Number.isFinite(atMs))
|
|
1885
|
+
throw new Error(t.quality.needsAt());
|
|
1886
|
+
/**
|
|
1887
|
+
* Two profiles over the same file, split at the boundary — rather than one
|
|
1888
|
+
* profile the caller has to slice.
|
|
1889
|
+
*
|
|
1890
|
+
* The alternative is asking somebody for two logs, which invites the mistake
|
|
1891
|
+
* this whole module exists to avoid: two files gathered under conditions
|
|
1892
|
+
* nobody wrote down.
|
|
1893
|
+
*/
|
|
1894
|
+
const raw = await readUsageLog(path, t);
|
|
1895
|
+
const sideOf = (since, until) => {
|
|
1896
|
+
const report = profileUsage(raw, { catalogue: pricing, label, sinceMs: since, untilMs: until });
|
|
1897
|
+
const slice = report.outcomeTallyByLabel.find((entry) => entry.label === label);
|
|
1898
|
+
return {
|
|
1899
|
+
arm: {
|
|
1900
|
+
name: label,
|
|
1901
|
+
totalUsd: report.total.totalUsd,
|
|
1902
|
+
tally: slice?.tally ?? { byValue: [], recorded: 0, parsed: 0, unrecordedUsd: 0 },
|
|
1903
|
+
},
|
|
1904
|
+
calls: report.total.calls,
|
|
1905
|
+
usdByModel: report.byModel.map((entry) => ({ model: entry.model, usd: entry.breakdown.totalUsd })),
|
|
1906
|
+
};
|
|
1907
|
+
};
|
|
1908
|
+
const result = qualityGate(sideOf(undefined, atMs), sideOf(atMs, undefined), config.outcomes ?? null);
|
|
1909
|
+
const pct = (value) => `${(value * 100).toFixed(1)}%`;
|
|
1910
|
+
const n = (value) => value.toLocaleString(t.numberLocale);
|
|
1911
|
+
console.log();
|
|
1912
|
+
console.log(c.bold(t.quality.heading(label)));
|
|
1913
|
+
console.log(` ${c.dim(wrap(t.quality.notRandomised(), 74, ' '))}`);
|
|
1914
|
+
console.log();
|
|
1915
|
+
console.log(` ${t.quality.sides(result.before.rate === null ? '—' : pct(result.before.rate), result.after.rate === null ? '—' : pct(result.after.rate), n(result.outcomes.before), n(result.outcomes.after))}`);
|
|
1916
|
+
console.log();
|
|
1917
|
+
if (result.verdict === 'dropped') {
|
|
1918
|
+
const cost = result.cost === null
|
|
1919
|
+
? ''
|
|
1920
|
+
: result.cost.deltaUsdPerCall < 0
|
|
1921
|
+
? `saves ${formatUsd(-result.cost.deltaUsdPerCall)} a call`
|
|
1922
|
+
: `costs ${formatUsd(result.cost.deltaUsdPerCall)} a call more`;
|
|
1923
|
+
console.log(` ${c.red('✗')} ${wrap(t.quality.dropped(pct(result.before.rate ?? 0), pct(result.after.rate ?? 0), n(result.outcomes.before + result.outcomes.after), cost), 74, ' ')}`);
|
|
1924
|
+
}
|
|
1925
|
+
else if (result.verdict === 'held') {
|
|
1926
|
+
console.log(` ${c.green('✓')} ${wrap(t.quality.held(pct(result.before.rate ?? 0), pct(result.after.rate ?? 0), n(result.outcomes.before + result.outcomes.after)), 74, ' ')}`);
|
|
1927
|
+
}
|
|
1928
|
+
else {
|
|
1929
|
+
const need = result.unknown === 'too-few-before' ? n(result.outcomes.before) : n(result.outcomes.after);
|
|
1930
|
+
console.log(` ${c.yellow('?')} ${wrap(t.quality.cannotTell(result.unknown ?? '', need), 74, ' ')}`);
|
|
1931
|
+
}
|
|
1932
|
+
/**
|
|
1933
|
+
* Confounders print on **every** verdict, not only on `cannot-tell`.
|
|
1934
|
+
*
|
|
1935
|
+
* A rate that held while the model changed underneath is not evidence that
|
|
1936
|
+
* the prompt is fine either, and hiding the confounder on a green result is
|
|
1937
|
+
* how a gate teaches people to trust it in exactly the case it should not be
|
|
1938
|
+
* trusted.
|
|
1939
|
+
*/
|
|
1940
|
+
if (result.confounders.length > 0) {
|
|
1941
|
+
console.log();
|
|
1942
|
+
console.log(` ${c.bold(t.quality.confoundersHeading())}`);
|
|
1943
|
+
for (const confounder of result.confounders) {
|
|
1944
|
+
const detail = confounder.kind === 'model-mix-moved'
|
|
1945
|
+
? `${pct(confounder.drift)} (${confounder.model})`
|
|
1946
|
+
: confounder.kind === 'volume-moved'
|
|
1947
|
+
? `${n(confounder.beforeCalls)} → ${n(confounder.afterCalls)} calls`
|
|
1948
|
+
: `${pct(confounder.before)} → ${pct(confounder.after)}`;
|
|
1949
|
+
console.log(` ${c.yellow('!')} ${wrap(t.quality.confounder(confounder.kind, detail), 70, ' ')}`);
|
|
1950
|
+
}
|
|
1951
|
+
}
|
|
1952
|
+
console.log();
|
|
1953
|
+
console.log(` ${c.dim(wrap(t.quality.cannotSee(), 74, ' '))}`);
|
|
1954
|
+
if (boolFlag(args, 'gate')) {
|
|
1955
|
+
console.log();
|
|
1956
|
+
if (result.verdict === 'dropped') {
|
|
1957
|
+
console.log(` ${c.red(t.quality.gateFailed())}`);
|
|
1958
|
+
process.exitCode = 1;
|
|
1959
|
+
}
|
|
1960
|
+
else if (result.verdict === 'cannot-tell') {
|
|
1961
|
+
// Three outcomes, never two. `cannot tell` holds the claim open rather
|
|
1962
|
+
// than exiting green, the posture `verify --gate` has had since 1.39.
|
|
1963
|
+
console.log(` ${c.yellow(t.quality.gateHeldOpen())}`);
|
|
1964
|
+
process.exitCode = 2;
|
|
1965
|
+
}
|
|
1966
|
+
}
|
|
1967
|
+
console.log();
|
|
1968
|
+
}
|
|
1857
1969
|
function commandModels(t, pricing) {
|
|
1858
1970
|
const n = (value) => value.toLocaleString(t.numberLocale);
|
|
1859
1971
|
const col = t.models.columns;
|
|
@@ -6901,6 +7013,9 @@ async function main() {
|
|
|
6901
7013
|
case 'models':
|
|
6902
7014
|
commandModels(t, pricing);
|
|
6903
7015
|
break;
|
|
7016
|
+
case 'quality':
|
|
7017
|
+
await commandQuality(args, config, pricing, t);
|
|
7018
|
+
break;
|
|
6904
7019
|
case 'experiment':
|
|
6905
7020
|
await commandExperiment(args, config, pricing, t);
|
|
6906
7021
|
break;
|