@pi-in-go/pigpen-pi-typesafe 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CREDITS.md +14 -0
- package/LICENSE +22 -0
- package/README.md +45 -0
- package/extensions/pi-typesafe/branches_test.go +185 -0
- package/extensions/pi-typesafe/command.go +319 -0
- package/extensions/pi-typesafe/export_test.go +9 -0
- package/extensions/pi-typesafe/extension.go +188 -0
- package/extensions/pi-typesafe/extension_test.go +321 -0
- package/extensions/pi-typesafe/fakehost_test.go +548 -0
- package/extensions/pi-typesafe/format.go +191 -0
- package/extensions/pi-typesafe/format_test.go +75 -0
- package/extensions/pi-typesafe/go.mod +10 -0
- package/extensions/pi-typesafe/go.sum +2 -0
- package/extensions/pi-typesafe/go.work +11 -0
- package/extensions/pi-typesafe/harness_test.go +200 -0
- package/extensions/pi-typesafe/ownmodel_test.go +100 -0
- package/extensions/pi-typesafe/review_test.go +134 -0
- package/extensions/pi-typesafe/tool.go +193 -0
- package/extensions/pi-typesafe/twin_test.go +28 -0
- package/libs/pi-typesafe-api/CREDITS.md +14 -0
- package/libs/pi-typesafe-api/LICENSE +22 -0
- package/libs/pi-typesafe-api/README.md +30 -0
- package/libs/pi-typesafe-api/ask.go +62 -0
- package/libs/pi-typesafe-api/ask_test.go +76 -0
- package/libs/pi-typesafe-api/auth.go +249 -0
- package/libs/pi-typesafe-api/auth_test.go +131 -0
- package/libs/pi-typesafe-api/backends.go +336 -0
- package/libs/pi-typesafe-api/backends_test.go +404 -0
- package/libs/pi-typesafe-api/batch.go +202 -0
- package/libs/pi-typesafe-api/batch_test.go +202 -0
- package/libs/pi-typesafe-api/battery_test.go +41 -0
- package/libs/pi-typesafe-api/calibrate.go +354 -0
- package/libs/pi-typesafe-api/calibrate_test.go +186 -0
- package/libs/pi-typesafe-api/client.go +615 -0
- package/libs/pi-typesafe-api/client_test.go +490 -0
- package/libs/pi-typesafe-api/credentials.go +252 -0
- package/libs/pi-typesafe-api/credentials_test.go +216 -0
- package/libs/pi-typesafe-api/doc.go +14 -0
- package/libs/pi-typesafe-api/errors.go +143 -0
- package/libs/pi-typesafe-api/evaluation.go +86 -0
- package/libs/pi-typesafe-api/evaluation_schema.json +264 -0
- package/libs/pi-typesafe-api/gaps_test.go +77 -0
- package/libs/pi-typesafe-api/go.mod +9 -0
- package/libs/pi-typesafe-api/go.sum +2 -0
- package/libs/pi-typesafe-api/helpers_test.go +169 -0
- package/libs/pi-typesafe-api/hostmodel/hostmodel.go +87 -0
- package/libs/pi-typesafe-api/json.go +299 -0
- package/libs/pi-typesafe-api/json_test.go +92 -0
- package/libs/pi-typesafe-api/ownmodel_test.go +79 -0
- package/libs/pi-typesafe-api/package.json +40 -0
- package/libs/pi-typesafe-api/provenance.json +18 -0
- package/libs/pi-typesafe-api/review_test.go +23 -0
- package/libs/pi-typesafe-api/schema.go +473 -0
- package/libs/pi-typesafe-api/schema_test.go +262 -0
- package/libs/pi-typesafe-api/testdata/tools/typebox-messages.mts +5 -0
- package/libs/pi-typesafe-api/testdata/typebox-messages.json +285 -0
- package/libs/pi-typesafe-api/twin_test.go +28 -0
- package/libs/pi-typesafe-api/ui/fakehost_test.go +548 -0
- package/libs/pi-typesafe-api/ui/keyprompt.go +115 -0
- package/libs/pi-typesafe-api/ui/login.go +106 -0
- package/libs/pi-typesafe-api/ui/twin_test.go +28 -0
- package/libs/pi-typesafe-api/ui/ui_test.go +285 -0
- package/libs/pi-typesafe-api/usage.go +366 -0
- package/libs/pi-typesafe-api/usage_test.go +139 -0
- package/libs/typesafe/CONTRACT.md +125 -0
- package/libs/typesafe/CREDITS.md +37 -0
- package/libs/typesafe/LICENSE +23 -0
- package/libs/typesafe/README.md +19 -0
- package/libs/typesafe/go.mod +3 -0
- package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
- package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
- package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
- package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
- package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
- package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
- package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
- package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
- package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
- package/libs/typesafe/libraries/ownmodel/run.go +288 -0
- package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
- package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
- package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
- package/libs/typesafe/libraries/typesafe/answers.go +268 -0
- package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
- package/libs/typesafe/libraries/typesafe/batch.go +80 -0
- package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
- package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
- package/libs/typesafe/libraries/typesafe/client.go +561 -0
- package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
- package/libs/typesafe/libraries/typesafe/doc.go +27 -0
- package/libs/typesafe/libraries/typesafe/entry.go +142 -0
- package/libs/typesafe/libraries/typesafe/env.go +11 -0
- package/libs/typesafe/libraries/typesafe/errors.go +310 -0
- package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
- package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
- package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
- package/libs/typesafe/libraries/typesafe/logging.go +160 -0
- package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
- package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
- package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
- package/libs/typesafe/libraries/typesafe/questions.go +490 -0
- package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
- package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
- package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
- package/libs/typesafe/libraries/typesafe/retry.go +350 -0
- package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
- package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
- package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
- package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
- package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
- package/libs/typesafe/libraries/typesafe/version.go +10 -0
- package/libs/typesafe/package.json +37 -0
- package/libs/typesafe/provenance.json +49 -0
- package/package.json +42 -0
- package/port/PORT.md +98 -0
- package/port/accepted-gaps.json +3 -0
- package/port/golden/enable-confirm.jsonl +11 -0
- package/port/golden/enable-decline.jsonl +20 -0
- package/port/golden/enable-missing-key.jsonl +4 -0
- package/port/golden/login-shadow.jsonl +4 -0
- package/port/golden/logout-env-key.jsonl +6 -0
- package/port/golden/playground-cancel.jsonl +4 -0
- package/port/golden/playground-invalid-json.jsonl +5 -0
- package/port/golden/playground-invalid-questions.jsonl +5 -0
- package/port/golden/status-env-key.jsonl +6 -0
- package/port/golden/status-no-key.jsonl +6 -0
- package/port/golden/tool-disabled.jsonl +18 -0
- package/port/golden/trailing-words.jsonl +10 -0
- package/port/library-mutations.py +44 -0
- package/port/mutations.json +302 -0
- package/port/oracle/.env.example +4 -0
- package/port/oracle/CHANGELOG.md +91 -0
- package/port/oracle/CONTRIBUTING.md +35 -0
- package/port/oracle/LICENSE +21 -0
- package/port/oracle/README.md +159 -0
- package/port/oracle/docs/api.md +143 -0
- package/port/oracle/docs/ci-cd.md +97 -0
- package/port/oracle/examples/decision-extension.ts +41 -0
- package/port/oracle/extensions/index.js +2 -0
- package/port/oracle/package.json +89 -0
- package/port/oracle/scripts/dev-pi.mjs +23 -0
- package/port/oracle/scripts/live-smoke.mjs +35 -0
- package/port/oracle/src/ask.ts +42 -0
- package/port/oracle/src/auth.ts +171 -0
- package/port/oracle/src/backends.ts +196 -0
- package/port/oracle/src/batch.ts +170 -0
- package/port/oracle/src/calibrate.ts +237 -0
- package/port/oracle/src/client.ts +310 -0
- package/port/oracle/src/credentials.ts +136 -0
- package/port/oracle/src/errors.ts +53 -0
- package/port/oracle/src/extension.ts +204 -0
- package/port/oracle/src/index.ts +31 -0
- package/port/oracle/src/key-prompt.ts +51 -0
- package/port/oracle/src/login.ts +60 -0
- package/port/oracle/src/schema.ts +158 -0
- package/port/oracle/src/ui.ts +4 -0
- package/port/oracle/src/usage.ts +258 -0
- package/port/oracle/tests/ask.test.ts +63 -0
- package/port/oracle/tests/auth.test.ts +141 -0
- package/port/oracle/tests/backends.test.ts +380 -0
- package/port/oracle/tests/batch.test.ts +156 -0
- package/port/oracle/tests/calibrate.test.ts +144 -0
- package/port/oracle/tests/client.test.ts +499 -0
- package/port/oracle/tests/credentials.test.ts +144 -0
- package/port/oracle/tests/extension.test.ts +276 -0
- package/port/oracle/tests/key-prompt.test.ts +47 -0
- package/port/oracle/tests/login.test.ts +101 -0
- package/port/oracle/tests/schema.test.ts +85 -0
- package/port/oracle/tests/usage.test.ts +106 -0
- package/port/oracle/tsconfig.build.json +10 -0
- package/port/oracle/tsconfig.json +14 -0
- package/port/scenarios/enable-confirm.json +5 -0
- package/port/scenarios/enable-decline.json +3 -0
- package/port/scenarios/enable-missing-key.json +2 -0
- package/port/scenarios/login-shadow.json +2 -0
- package/port/scenarios/logout-env-key.json +3 -0
- package/port/scenarios/playground-cancel.json +2 -0
- package/port/scenarios/playground-invalid-json.json +2 -0
- package/port/scenarios/playground-invalid-questions.json +2 -0
- package/port/scenarios/status-env-key.json +3 -0
- package/port/scenarios/status-no-key.json +3 -0
- package/port/scenarios/tool-disabled.json +2 -0
- package/port/scenarios/trailing-words.json +5 -0
- package/port/upstream-tests.json +160 -0
- package/provenance.json +18 -0
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
import { fanOut } from "./batch.js";
|
|
2
|
+
import type { Settled } from "./batch.js";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* A judge-tuning kit: label a set of cases, score them with Jev, and read off AUC and threshold behaviour. It carries
|
|
6
|
+
* no domain knowledge — a case is anything a scorer can turn into a number — so the same toolkit fits an action guard,
|
|
7
|
+
* a triage rule, or a prose check. `scripts/calibrate-action.mjs` in pi-warden is the worked example it came from.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** One labelled observation: the truth about the case, and the number the judge assigned to it. */
|
|
11
|
+
export interface ScoredSample {
|
|
12
|
+
readonly label: boolean;
|
|
13
|
+
readonly score: number;
|
|
14
|
+
/** Optional name; used in the missed and flagged listings. */
|
|
15
|
+
readonly id?: string;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/** Outcome counts for one threshold. `precision` and `recall` are undefined when their denominator is empty. */
|
|
19
|
+
export interface ThresholdRow {
|
|
20
|
+
readonly threshold: number;
|
|
21
|
+
readonly flagged: number;
|
|
22
|
+
readonly tp: number;
|
|
23
|
+
readonly fp: number;
|
|
24
|
+
readonly fn: number;
|
|
25
|
+
readonly tn: number;
|
|
26
|
+
readonly precision?: number;
|
|
27
|
+
readonly recall?: number;
|
|
28
|
+
/** Share of all cases the threshold selects. */
|
|
29
|
+
readonly flagRate: number;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export interface Calibration {
|
|
33
|
+
readonly name: string;
|
|
34
|
+
readonly scored: number;
|
|
35
|
+
readonly positives: number;
|
|
36
|
+
readonly negatives: number;
|
|
37
|
+
readonly errors: number;
|
|
38
|
+
/** Rank-based AUC (Mann–Whitney, ties count half); undefined when one class is empty. */
|
|
39
|
+
readonly auc?: number;
|
|
40
|
+
readonly rows: readonly ThresholdRow[];
|
|
41
|
+
/** The lowest threshold meeting the requested precision and recall floors, when one exists. */
|
|
42
|
+
readonly recommended?: ThresholdRow;
|
|
43
|
+
/** Positive cases the recommended threshold misses. */
|
|
44
|
+
readonly missed: readonly ScoredSample[];
|
|
45
|
+
/** Negative cases the recommended threshold flags. */
|
|
46
|
+
readonly flagged: readonly ScoredSample[];
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
export interface CalibrateOptions {
|
|
50
|
+
/** Thresholds to evaluate. Default: every distinct score, ascending (at most 64 rows). */
|
|
51
|
+
thresholds?: readonly number[];
|
|
52
|
+
/** Precision floor for the recommendation. */
|
|
53
|
+
minPrecision?: number;
|
|
54
|
+
/** Recall floor for the recommendation. */
|
|
55
|
+
minRecall?: number;
|
|
56
|
+
/** Cases that could not be scored; reported and excluded from the metrics. */
|
|
57
|
+
errors?: number;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export function auc(samples: readonly ScoredSample[]): number | undefined {
|
|
61
|
+
const positives = samples.filter(sample => sample.label).map(sample => sample.score);
|
|
62
|
+
const negatives = samples.filter(sample => !sample.label).map(sample => sample.score);
|
|
63
|
+
if (!positives.length || !negatives.length) return undefined;
|
|
64
|
+
let wins = 0;
|
|
65
|
+
for (const positive of positives) {
|
|
66
|
+
for (const negative of negatives) wins += positive > negative ? 1 : positive === negative ? 0.5 : 0;
|
|
67
|
+
}
|
|
68
|
+
return wins / (positives.length * negatives.length);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function ratio(numerator: number, denominator: number): number | undefined {
|
|
72
|
+
return denominator ? numerator / denominator : undefined;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Counts at one threshold: a case is flagged when its score is at least the threshold. */
|
|
76
|
+
export function metricsAt(samples: readonly ScoredSample[], threshold: number): ThresholdRow {
|
|
77
|
+
let tp = 0, fp = 0, fn = 0, tn = 0;
|
|
78
|
+
for (const sample of samples) {
|
|
79
|
+
const flagged = sample.score >= threshold;
|
|
80
|
+
if (flagged && sample.label) tp++;
|
|
81
|
+
else if (flagged) fp++;
|
|
82
|
+
else if (sample.label) fn++;
|
|
83
|
+
else tn++;
|
|
84
|
+
}
|
|
85
|
+
const flagged = tp + fp;
|
|
86
|
+
const precision = ratio(tp, tp + fp);
|
|
87
|
+
const recall = ratio(tp, tp + fn);
|
|
88
|
+
return {
|
|
89
|
+
threshold,
|
|
90
|
+
flagged,
|
|
91
|
+
tp,
|
|
92
|
+
fp,
|
|
93
|
+
fn,
|
|
94
|
+
tn,
|
|
95
|
+
...(precision === undefined ? {} : { precision }),
|
|
96
|
+
...(recall === undefined ? {} : { recall }),
|
|
97
|
+
flagRate: samples.length ? flagged / samples.length : 0,
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export function sweep(samples: readonly ScoredSample[], thresholds: readonly number[]): ThresholdRow[] {
|
|
102
|
+
return thresholds.map(threshold => metricsAt(samples, threshold));
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** The distinct scores, ascending, as a threshold grid: every point where the counts can change. */
|
|
106
|
+
export function defaultThresholds(samples: readonly ScoredSample[], limit = 64): number[] {
|
|
107
|
+
const distinct = [...new Set(samples.map(sample => sample.score))].sort((a, b) => a - b);
|
|
108
|
+
if (distinct.length <= limit) return distinct;
|
|
109
|
+
const step = (distinct.length - 1) / (limit - 1);
|
|
110
|
+
return Array.from({ length: limit }, (_, index) => distinct[Math.round(index * step)] as number);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* The lowest threshold that clears the precision and recall floors. With no floors, the best F1 among the rows;
|
|
115
|
+
* undefined when nothing clears them.
|
|
116
|
+
*/
|
|
117
|
+
export function pickThreshold(rows: readonly ThresholdRow[], options: { minPrecision?: number; minRecall?: number } = {}): ThresholdRow | undefined {
|
|
118
|
+
const { minPrecision, minRecall } = options;
|
|
119
|
+
if (minPrecision === undefined && minRecall === undefined) {
|
|
120
|
+
let best: ThresholdRow | undefined;
|
|
121
|
+
let bestF1 = -1;
|
|
122
|
+
for (const row of rows) {
|
|
123
|
+
if (row.precision === undefined || row.recall === undefined) continue;
|
|
124
|
+
const f1 = row.precision + row.recall === 0 ? 0 : 2 * row.precision * row.recall / (row.precision + row.recall);
|
|
125
|
+
if (f1 > bestF1) { bestF1 = f1; best = row; }
|
|
126
|
+
}
|
|
127
|
+
return best;
|
|
128
|
+
}
|
|
129
|
+
const candidates = rows
|
|
130
|
+
.filter(row => (minPrecision === undefined || (row.precision ?? 0) >= minPrecision) && (minRecall === undefined || (row.recall ?? 0) >= minRecall))
|
|
131
|
+
.sort((a, b) => a.threshold - b.threshold);
|
|
132
|
+
return candidates[0];
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** Label, score, and read the numbers: AUC, the threshold sweep, and one recommendation. */
|
|
136
|
+
export function calibrate(name: string, samples: readonly ScoredSample[], options: CalibrateOptions = {}): Calibration {
|
|
137
|
+
const thresholds = options.thresholds ?? defaultThresholds(samples);
|
|
138
|
+
const rows = sweep(samples, thresholds);
|
|
139
|
+
const recommendation = pickThreshold(rows, {
|
|
140
|
+
...(options.minPrecision === undefined ? {} : { minPrecision: options.minPrecision }),
|
|
141
|
+
...(options.minRecall === undefined ? {} : { minRecall: options.minRecall }),
|
|
142
|
+
});
|
|
143
|
+
const rank = auc(samples);
|
|
144
|
+
const threshold = recommendation?.threshold;
|
|
145
|
+
return {
|
|
146
|
+
name,
|
|
147
|
+
scored: samples.length,
|
|
148
|
+
positives: samples.filter(sample => sample.label).length,
|
|
149
|
+
negatives: samples.filter(sample => !sample.label).length,
|
|
150
|
+
errors: options.errors ?? 0,
|
|
151
|
+
...(rank === undefined ? {} : { auc: rank }),
|
|
152
|
+
rows,
|
|
153
|
+
...(recommendation === undefined ? {} : { recommended: recommendation }),
|
|
154
|
+
missed: threshold === undefined ? [] : samples.filter(sample => sample.label && sample.score < threshold),
|
|
155
|
+
flagged: threshold === undefined ? [] : samples.filter(sample => !sample.label && sample.score >= threshold),
|
|
156
|
+
};
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
const percent = (value: number | undefined) => value === undefined ? "-" : `${(value * 100).toFixed(0)}%`;
|
|
160
|
+
|
|
161
|
+
/** Plain text, no colour: safe to write to a report file or a log. */
|
|
162
|
+
export function formatCalibration(calibration: Calibration): string {
|
|
163
|
+
const lines = [
|
|
164
|
+
`${calibration.name}: ${calibration.scored} scored, ${calibration.positives} positives, ${calibration.negatives} negatives${calibration.errors ? `, ${calibration.errors} errors` : ""}`,
|
|
165
|
+
`AUC ${calibration.auc === undefined ? "-" : calibration.auc.toFixed(3)}`,
|
|
166
|
+
"threshold flagged TP FP FN TN precision recall",
|
|
167
|
+
];
|
|
168
|
+
for (const row of calibration.rows) {
|
|
169
|
+
lines.push(`${row.threshold.toFixed(2).padStart(9)} ${String(row.flagged).padStart(7)} ${String(row.tp).padStart(2)} ${String(row.fp).padStart(2)} ${String(row.fn).padStart(2)} ${String(row.tn).padStart(2)} ${percent(row.precision).padStart(9)} ${percent(row.recall).padStart(6)}`);
|
|
170
|
+
}
|
|
171
|
+
const recommended = calibration.recommended;
|
|
172
|
+
lines.push(recommended === undefined
|
|
173
|
+
? "recommended: none (no threshold clears the floors)"
|
|
174
|
+
: `recommended ${recommended.threshold.toFixed(2)}: precision ${percent(recommended.precision)}, recall ${percent(recommended.recall)}, flags ${percent(recommended.flagRate)}`);
|
|
175
|
+
if (calibration.missed.length) lines.push(`missed positives (${calibration.missed.length}): ${calibration.missed.map(sample => sample.id ?? sample.score.toFixed(2)).join(", ").slice(0, 300)}`);
|
|
176
|
+
if (calibration.flagged.length) lines.push(`flagged negatives (${calibration.flagged.length}): ${calibration.flagged.map(sample => sample.id ?? sample.score.toFixed(2)).join(", ").slice(0, 300)}`);
|
|
177
|
+
return lines.join("\n");
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** One labelled replay case: the truth, plus whatever the scorer needs to judge it. */
|
|
181
|
+
export interface ReplayCase<T> {
|
|
182
|
+
readonly id: string;
|
|
183
|
+
readonly label: boolean;
|
|
184
|
+
readonly data: T;
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
export interface ReplayResult<T> {
|
|
188
|
+
readonly id: string;
|
|
189
|
+
readonly label: boolean;
|
|
190
|
+
readonly data: T;
|
|
191
|
+
/** The judge's number, absent when the case could not be scored. */
|
|
192
|
+
readonly score?: number;
|
|
193
|
+
/** The failure message, absent on success. Carries no upstream body. */
|
|
194
|
+
readonly error?: string;
|
|
195
|
+
/** True when the case was never submitted (abort or a stopped batch). */
|
|
196
|
+
readonly skipped: boolean;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
export interface ReplayOptions {
|
|
200
|
+
/** Cases in flight at once. Default: DEFAULT_CONCURRENCY (4). */
|
|
201
|
+
concurrency?: number;
|
|
202
|
+
signal?: AbortSignal;
|
|
203
|
+
/** Stop launching new cases once this returns true for a failure, e.g. a `budget` error. */
|
|
204
|
+
stopOn?: (error: unknown) => boolean;
|
|
205
|
+
/** Turns a thrown scorer error into the reported message. Default: the error's own message. */
|
|
206
|
+
describeError?: (error: unknown) => string;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Replay labelled cases through a scorer with bounded concurrency, keeping order and capturing per-case failures.
|
|
211
|
+
* The scorer is usually one Jev question; a thrown error is recorded rather than aborting the run, so one bad case
|
|
212
|
+
* cannot destroy a long calibration. Results feed straight into `samplesOf` and `calibrate`.
|
|
213
|
+
*/
|
|
214
|
+
export async function replay<T>(cases: readonly ReplayCase<T>[], score: (data: T, index: number) => Promise<number>, options: ReplayOptions = {}): Promise<ReplayResult<T>[]> {
|
|
215
|
+
const settled: Settled<number>[] = await fanOut(cases, (item, index) => score(item.data, index), {
|
|
216
|
+
...(options.concurrency === undefined ? {} : { concurrency: options.concurrency }),
|
|
217
|
+
...(options.signal === undefined ? {} : { signal: options.signal }),
|
|
218
|
+
...(options.stopOn === undefined ? {} : { stopOn: options.stopOn }),
|
|
219
|
+
});
|
|
220
|
+
return settled.map((result, index) => {
|
|
221
|
+
const item = cases[index] as ReplayCase<T>;
|
|
222
|
+
if (result.ok) return { id: item.id, label: item.label, data: item.data, score: result.value, skipped: false };
|
|
223
|
+
const error = options.describeError ? options.describeError(result.error) : result.error instanceof Error ? result.error.message : "The scorer failed.";
|
|
224
|
+
return { id: item.id, label: item.label, data: item.data, error, skipped: result.skipped };
|
|
225
|
+
});
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/** The scored cases of a replay, in replay order. Unscored cases are excluded and counted as errors. */
|
|
229
|
+
export function samplesOf<T>(results: readonly ReplayResult<T>[]): { samples: ScoredSample[]; errors: number } {
|
|
230
|
+
const samples: ScoredSample[] = [];
|
|
231
|
+
let errors = 0;
|
|
232
|
+
for (const result of results) {
|
|
233
|
+
if (result.score === undefined || !Number.isFinite(result.score)) { errors++; continue; }
|
|
234
|
+
samples.push({ label: result.label, score: result.score, id: result.id });
|
|
235
|
+
}
|
|
236
|
+
return { samples, errors };
|
|
237
|
+
}
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
import { TypeSafeClient } from "@typesafe-ai/sdk";
|
|
2
|
+
import type { Fetch, Questions, SystemOneRequest, SystemOneResult } from "@typesafe-ai/sdk";
|
|
3
|
+
import { recordAuthFailure, recordAuthVerified } from "./auth.js";
|
|
4
|
+
import { TYPESAFE_KEY_ENV, backendModelId, resolveBackend, usesTypesafeKey } from "./backends.js";
|
|
5
|
+
import type { BackendConfig, BackendSpec, ResolvedBackend } from "./backends.js";
|
|
6
|
+
import type { BatchEvaluation, BatchOptions } from "./batch.js";
|
|
7
|
+
import { evaluateAll, evaluateMany } from "./batch.js";
|
|
8
|
+
import { keySituation } from "./credentials.js";
|
|
9
|
+
import { TypeSafeIntegrationError, safeError } from "./errors.js";
|
|
10
|
+
import { DEFAULT_MAX_INPUT_BYTES, assertWithinByteLimit, prepareEvaluationRequest } from "./schema.js";
|
|
11
|
+
import { DEFAULT_USD_PER_MTOK, capsFromEnvironment, estimateUsd, mergeCaps, openUsageLedger } from "./usage.js";
|
|
12
|
+
import type { BlockedCap, SpendCaps, UsageLedger, UsageReport } from "./usage.js";
|
|
13
|
+
|
|
14
|
+
export { backendHost, resolveBackend, DECISIONS_BACKENDS, DEFAULT_BACKEND } from "./backends.js";
|
|
15
|
+
export type { BackendConfig, BackendEndpoint, BackendSpec, ResolvedBackend, TypeSafeBackend } from "./backends.js";
|
|
16
|
+
|
|
17
|
+
/** The paths the SDK appends to whatever base URL it is given. */
|
|
18
|
+
const SDK_PATH = "/v1/systemone";
|
|
19
|
+
const SDK_MODELS_PATH = "/v1/models";
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Send the SDK's fixed paths to the backend's own, preserving any caller-supplied transport. A backend that serves
|
|
23
|
+
* its model list under another path also gets that list renamed to the field the SDK reads.
|
|
24
|
+
*/
|
|
25
|
+
function backendFetch(backend: BackendConfig, inner: Fetch = fetch): Fetch {
|
|
26
|
+
const { path, modelsPath, modelsField } = backend;
|
|
27
|
+
return async (input, init) => {
|
|
28
|
+
const url = String(input);
|
|
29
|
+
const models = modelsPath !== undefined && url.includes(SDK_MODELS_PATH);
|
|
30
|
+
const rewrite = models ? modelsPath : path;
|
|
31
|
+
const response = await inner(rewrite === undefined ? url : url.replace(models ? SDK_MODELS_PATH : SDK_PATH, rewrite), init);
|
|
32
|
+
return models && modelsField !== undefined ? translateModels(response, backend) : response;
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Hand the SDK the list it expects: the field it reads, and the entry value callers pass as `model:` when the backend
|
|
38
|
+
* labels models differently. Status and headers survive; a body without the declared field is passed through
|
|
39
|
+
* unchanged, so the SDK still reports its own shape error.
|
|
40
|
+
*/
|
|
41
|
+
async function translateModels(response: Response, backend: BackendConfig): Promise<Response> {
|
|
42
|
+
const { modelsField, modelsIdField } = backend;
|
|
43
|
+
const text = await response.text();
|
|
44
|
+
let wire: unknown;
|
|
45
|
+
try { wire = JSON.parse(text); } catch { wire = undefined; }
|
|
46
|
+
const list = modelsField !== undefined && wire !== null && typeof wire === "object" ? (wire as Record<string, unknown>)[modelsField] : undefined;
|
|
47
|
+
const headers = new Headers(response.headers);
|
|
48
|
+
// The body is replaced, so a copied length would describe the old one.
|
|
49
|
+
headers.delete("content-length");
|
|
50
|
+
headers.delete("content-encoding");
|
|
51
|
+
const send = (body: string) => new Response(body, { status: response.status, statusText: response.statusText, headers });
|
|
52
|
+
if (!Array.isArray(list)) return send(text);
|
|
53
|
+
const models = list.map(entry => {
|
|
54
|
+
if (modelsIdField === undefined || entry === null || typeof entry !== "object") return entry;
|
|
55
|
+
const id = (entry as Record<string, unknown>)[modelsIdField];
|
|
56
|
+
return typeof id === "string" && id.length > 0 ? { ...entry, name: id } : entry;
|
|
57
|
+
});
|
|
58
|
+
return send(JSON.stringify({ models }));
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export interface TypeSafeOptions {
|
|
62
|
+
/** Defaults to TYPESAFE_API_KEY, then the key saved by `/typesafe login`; never returned. */
|
|
63
|
+
apiKey?: string;
|
|
64
|
+
/** Judgment backend: a registry name or a caller-supplied endpoint. When omitted, routes to the default TypeSafe host. */
|
|
65
|
+
backend?: BackendSpec;
|
|
66
|
+
/** Defaults to the backend's own default (`jev-latest`, `typesafe/jev-1.13` on OpenRouter); a bare Jev id is mapped to the backend's id form before sending. No model is inferred from submitted content. */
|
|
67
|
+
model?: string;
|
|
68
|
+
/** Per request. Default: 15 seconds. No automatic retries. */
|
|
69
|
+
timeoutMs?: number;
|
|
70
|
+
/** UTF-8 JSON bytes, including model/questions. Default: 64 KiB. Not a token limit. */
|
|
71
|
+
maxInputBytes?: number;
|
|
72
|
+
/** Attempts per client instance, including failed network requests. Default: 20. */
|
|
73
|
+
maxRequests?: number;
|
|
74
|
+
/** Requests per local day, counted across processes and restarts. Unlimited by default. */
|
|
75
|
+
maxRequestsPerDay?: number;
|
|
76
|
+
/** Input tokens per local day. Unlimited by default. */
|
|
77
|
+
maxInputTokensPerDay?: number;
|
|
78
|
+
/** Estimated spend per local day, in US dollars. Unlimited by default. */
|
|
79
|
+
maxUsdPerDay?: number;
|
|
80
|
+
/** Price used for the cost estimate and the USD cap. Default: DEFAULT_USD_PER_MTOK. */
|
|
81
|
+
usdPerMTok?: number;
|
|
82
|
+
/** The usage ledger; defaults to the store next to the key. Injected by tests. */
|
|
83
|
+
ledger?: UsageLedger;
|
|
84
|
+
/** Transport injection for extension authors and offline tests. */
|
|
85
|
+
fetch?: Fetch;
|
|
86
|
+
}
|
|
87
|
+
export interface EvaluationOptions { signal?: AbortSignal }
|
|
88
|
+
export type Evaluation<Q extends Questions> = SystemOneResult<Q> & { readonly elapsedMs: number };
|
|
89
|
+
|
|
90
|
+
/** Session counters for one client instance, plus the cost estimate they add up to. */
|
|
91
|
+
export interface UsageSnapshot {
|
|
92
|
+
readonly requestsStarted: number;
|
|
93
|
+
readonly requestsSucceeded: number;
|
|
94
|
+
readonly requestsFailed: number;
|
|
95
|
+
readonly inputTokens: number;
|
|
96
|
+
readonly outputTokens: number;
|
|
97
|
+
readonly estimatedUsd: number;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** Session counters, today's persisted counters, the caps in force, and the cap that is currently reached. */
|
|
101
|
+
export interface SpendReport {
|
|
102
|
+
readonly session: UsageSnapshot;
|
|
103
|
+
readonly today: UsageReport;
|
|
104
|
+
readonly caps: SpendCaps;
|
|
105
|
+
readonly usdPerMTok: number;
|
|
106
|
+
readonly blocked?: BlockedCap;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export interface TypeSafe {
|
|
110
|
+
evaluate<Q extends Questions>(request: SystemOneRequest<Q>, options?: EvaluationOptions): Promise<Evaluation<Q>>;
|
|
111
|
+
/** Several requests with bounded concurrency; answers, usage, and model merged. Never throws. */
|
|
112
|
+
evaluateMany<Q extends Questions>(requests: readonly SystemOneRequest<Q>[], options?: BatchOptions): Promise<BatchEvaluation<Q>>;
|
|
113
|
+
/** One state, any number of questions: chunk to the per-request limit, then fan out. Never throws. */
|
|
114
|
+
evaluateAll<Q extends Questions>(request: SystemOneRequest<Q>, options?: BatchOptions & { maxQuestions?: number }): Promise<BatchEvaluation<Q>>;
|
|
115
|
+
/** Model names available to the account. Verifies the key; does not count toward maxRequests. */
|
|
116
|
+
listModels(options?: EvaluationOptions): Promise<string[]>;
|
|
117
|
+
/** This client's session counters. */
|
|
118
|
+
getUsage(): UsageSnapshot;
|
|
119
|
+
/** Session counters, today's persisted totals, and the caps that stop the next request. */
|
|
120
|
+
getSpend(): SpendReport;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/** Default attempts per client instance; the extension quotes the same number in its consent copy. */
|
|
124
|
+
export const DEFAULT_MAX_REQUESTS = 20;
|
|
125
|
+
|
|
126
|
+
function positiveInteger(value: number, label: string): number {
|
|
127
|
+
if (!Number.isSafeInteger(value) || value <= 0) {
|
|
128
|
+
throw new TypeSafeIntegrationError("configuration", `${label} must be a positive safe integer.`);
|
|
129
|
+
}
|
|
130
|
+
return value;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
function positiveNumber(value: number, label: string): number {
|
|
134
|
+
if (!Number.isFinite(value) || value <= 0) {
|
|
135
|
+
throw new TypeSafeIntegrationError("configuration", `${label} must be a positive number.`);
|
|
136
|
+
}
|
|
137
|
+
return value;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
function validResult<Q extends Questions>(result: SystemOneResult<Q>, questions: Q): boolean {
|
|
141
|
+
const probability = (value: number) => Number.isFinite(value) && value >= 0 && value <= 1;
|
|
142
|
+
if (!result || typeof result.model !== "string" || !result.model || !result.usage || !result.answers) return false;
|
|
143
|
+
for (const count of [result.usage.input_tokens, result.usage.output_tokens]) {
|
|
144
|
+
if (!Number.isSafeInteger(count) || count < 0) return false;
|
|
145
|
+
}
|
|
146
|
+
if (Object.keys(result.answers).length !== Object.keys(questions).length) return false;
|
|
147
|
+
for (const [id, question] of Object.entries(questions)) {
|
|
148
|
+
const answer = result.answers[id];
|
|
149
|
+
if (!answer || answer.type !== question.type) return false;
|
|
150
|
+
if (answer.type === "noul") {
|
|
151
|
+
if (!probability(answer.noul)) return false;
|
|
152
|
+
continue;
|
|
153
|
+
}
|
|
154
|
+
if (!probability(answer.confidence) || !answer.probabilities) return false;
|
|
155
|
+
const keys = question.type === "choice" ? Object.keys(question.criteria)
|
|
156
|
+
: question.type === "score" ? question.criteria.map((_, index) => String(index)) : [];
|
|
157
|
+
const probabilities = new Map(Object.entries(answer.probabilities));
|
|
158
|
+
if (probabilities.size !== keys.length || keys.some(key => !probability(probabilities.get(key) ?? NaN))) return false;
|
|
159
|
+
if (answer.type === "choice" && !keys.includes(answer.choice)) return false;
|
|
160
|
+
if (answer.type === "score" && (!Number.isFinite(answer.score) || answer.score < 0 || answer.score > keys.length - 1 || !answer.legend)) return false;
|
|
161
|
+
}
|
|
162
|
+
return true;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
const CAP_LABELS: Record<BlockedCap["cap"], string> = {
|
|
166
|
+
requestsPerDay: "daily request cap",
|
|
167
|
+
inputTokensPerDay: "daily input-token cap",
|
|
168
|
+
usdPerDay: "daily spend cap",
|
|
169
|
+
};
|
|
170
|
+
|
|
171
|
+
function capsDescription(caps: SpendCaps): string {
|
|
172
|
+
const parts = [
|
|
173
|
+
caps.maxRequests === undefined ? undefined : `${caps.maxRequests} per session`,
|
|
174
|
+
caps.maxRequestsPerDay === undefined ? undefined : `${caps.maxRequestsPerDay} requests per day`,
|
|
175
|
+
caps.maxInputTokensPerDay === undefined ? undefined : `${caps.maxInputTokensPerDay} input tokens per day`,
|
|
176
|
+
caps.maxUsdPerDay === undefined ? undefined : `$${caps.maxUsdPerDay} per day`,
|
|
177
|
+
].filter((part): part is string => part !== undefined);
|
|
178
|
+
return parts.length ? `Caps: ${parts.join(", ")}.` : "No request or spend caps are set.";
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/** A bounded, server-side TypeSafe client independent of Pi's runtime. */
|
|
182
|
+
export function createTypeSafe(options: TypeSafeOptions = {}): TypeSafe {
|
|
183
|
+
const backend: ResolvedBackend = resolveBackend(options.backend);
|
|
184
|
+
let apiKey = options.apiKey?.trim();
|
|
185
|
+
const baseURL = backend.host;
|
|
186
|
+
// Only a registry backend maps ids; a caller-supplied endpoint's model is sent as the caller wrote it.
|
|
187
|
+
const mapModel = (model: string): string => (backend.name === undefined ? model : backendModelId(backend.name, model));
|
|
188
|
+
|
|
189
|
+
if (!apiKey) {
|
|
190
|
+
// The same resolution that authState() and ensureApiKey() report, so the status line and the request agree.
|
|
191
|
+
const situation = keySituation(options.backend);
|
|
192
|
+
if (situation.kind === "unusable") throw new TypeSafeIntegrationError("configuration", situation.reason);
|
|
193
|
+
if (situation.kind === "environment" || situation.kind === "stored") apiKey = situation.key;
|
|
194
|
+
}
|
|
195
|
+
if (!apiKey) {
|
|
196
|
+
const how = usesTypesafeKey(backend) ? `Run /typesafe login in Pi, or set ${TYPESAFE_KEY_ENV}` : `Set ${backend.keyEnv}`;
|
|
197
|
+
throw new TypeSafeIntegrationError("configuration", `No API key. ${how} in the environment.`);
|
|
198
|
+
}
|
|
199
|
+
const timeout = positiveInteger(options.timeoutMs ?? 15_000, "timeoutMs");
|
|
200
|
+
const maxInputBytes = positiveInteger(options.maxInputBytes ?? DEFAULT_MAX_INPUT_BYTES, "maxInputBytes");
|
|
201
|
+
const maxRequests = positiveInteger(options.maxRequests ?? DEFAULT_MAX_REQUESTS, "maxRequests");
|
|
202
|
+
const usdPerMTok = positiveNumber(options.usdPerMTok ?? DEFAULT_USD_PER_MTOK, "usdPerMTok");
|
|
203
|
+
const caps = mergeCaps({
|
|
204
|
+
maxRequests,
|
|
205
|
+
...(options.maxRequestsPerDay === undefined ? {} : { maxRequestsPerDay: positiveInteger(options.maxRequestsPerDay, "maxRequestsPerDay") }),
|
|
206
|
+
...(options.maxInputTokensPerDay === undefined ? {} : { maxInputTokensPerDay: positiveInteger(options.maxInputTokensPerDay, "maxInputTokensPerDay") }),
|
|
207
|
+
...(options.maxUsdPerDay === undefined ? {} : { maxUsdPerDay: positiveNumber(options.maxUsdPerDay, "maxUsdPerDay") }),
|
|
208
|
+
}, capsFromEnvironment());
|
|
209
|
+
// A backend that serves its own paths gets a transport that rewrites them; the default backend keeps the caller's.
|
|
210
|
+
const transport = backend.path !== undefined || backend.modelsPath !== undefined ? backendFetch(backend, options.fetch) : options.fetch;
|
|
211
|
+
// The caller's input is validated as written, then mapped to the backend's id form; omitting it sends the backend's
|
|
212
|
+
// own default, which the mapping leaves unchanged.
|
|
213
|
+
const requested = options.model ?? backend.defaultModel;
|
|
214
|
+
if (requested === undefined) throw new TypeSafeIntegrationError("configuration", `Backend "${backend.label}" names no defaultModel; pass model to createTypeSafe.`);
|
|
215
|
+
if (typeof requested !== "string" || !requested.trim() || requested.length > 100) throw new TypeSafeIntegrationError("configuration", "model must be a nonempty string of at most 100 characters.");
|
|
216
|
+
const model = mapModel(requested);
|
|
217
|
+
// Do not inherit SDK debug logging or alternate destinations from the environment.
|
|
218
|
+
const client = new TypeSafeClient({
|
|
219
|
+
apiKey,
|
|
220
|
+
defaultModel: model,
|
|
221
|
+
baseURL,
|
|
222
|
+
timeout,
|
|
223
|
+
retry: { maxRetries: 0 },
|
|
224
|
+
logLevel: "off",
|
|
225
|
+
...(transport ? { fetch: transport } : {}),
|
|
226
|
+
});
|
|
227
|
+
const ledger = options.ledger ?? openUsageLedger({ usdPerMTok });
|
|
228
|
+
const usage = { requestsStarted: 0, requestsSucceeded: 0, requestsFailed: 0, inputTokens: 0, outputTokens: 0 };
|
|
229
|
+
// Auth state writes are deduplicated: one verification record per process, one record per distinct failure.
|
|
230
|
+
let verificationRecorded = false;
|
|
231
|
+
let lastFailureRecorded: string | undefined;
|
|
232
|
+
|
|
233
|
+
const snapshot = (): UsageSnapshot => ({ ...usage, estimatedUsd: estimateUsd(usage.inputTokens, usdPerMTok) });
|
|
234
|
+
const blocked = (): BlockedCap | undefined => (usage.requestsStarted >= maxRequests ? undefined : ledger.blocked(caps));
|
|
235
|
+
|
|
236
|
+
const typesafe: TypeSafe = {
|
|
237
|
+
getUsage: snapshot,
|
|
238
|
+
getSpend: () => {
|
|
239
|
+
const reached = blocked();
|
|
240
|
+
return {
|
|
241
|
+
session: snapshot(),
|
|
242
|
+
today: ledger.today(),
|
|
243
|
+
caps,
|
|
244
|
+
usdPerMTok,
|
|
245
|
+
...(reached === undefined ? {} : { blocked: reached }),
|
|
246
|
+
};
|
|
247
|
+
},
|
|
248
|
+
async listModels(callOptions: EvaluationOptions = {}): Promise<string[]> {
|
|
249
|
+
try {
|
|
250
|
+
const models = await client.models.list(callOptions);
|
|
251
|
+
if (!Array.isArray(models)) throw new TypeSafeIntegrationError("response", "TypeSafe returned an unexpected model list.");
|
|
252
|
+
// A backend that serves its list publicly accepts any key, so a success there proves nothing about one.
|
|
253
|
+
if (backend.modelsVerifyKey && !verificationRecorded) {
|
|
254
|
+
verificationRecorded = true;
|
|
255
|
+
recordAuthVerified();
|
|
256
|
+
}
|
|
257
|
+
return models.map(card => card?.name).filter((name): name is string => typeof name === "string" && name.length > 0 && name.length <= 100);
|
|
258
|
+
} catch (error) {
|
|
259
|
+
throw safeError(error, backend);
|
|
260
|
+
}
|
|
261
|
+
},
|
|
262
|
+
async evaluate<Q extends Questions>(input: SystemOneRequest<Q>, callOptions: EvaluationOptions = {}): Promise<Evaluation<Q>> {
|
|
263
|
+
const validated = prepareEvaluationRequest(input, { maxInputBytes });
|
|
264
|
+
// A per-request model meets the same mapping as the client default; the schema already limited the caller's own id.
|
|
265
|
+
const body = JSON.stringify({ ...validated, model: validated.model === undefined ? model : mapModel(validated.model) });
|
|
266
|
+
assertWithinByteLimit(body, maxInputBytes);
|
|
267
|
+
if (callOptions.signal?.aborted) throw new TypeSafeIntegrationError("aborted", "TypeSafe request cancelled before submission.");
|
|
268
|
+
if (usage.requestsStarted >= maxRequests) {
|
|
269
|
+
throw new TypeSafeIntegrationError("budget", `TypeSafe request limit reached (${maxRequests} attempts per client instance). ${capsDescription(caps)}`);
|
|
270
|
+
}
|
|
271
|
+
const reached = ledger.blocked(caps);
|
|
272
|
+
if (reached) {
|
|
273
|
+
// Name the cap, the used amount, and the day, so a long run stops loudly instead of burning tokens unnoticed.
|
|
274
|
+
throw new TypeSafeIntegrationError("budget", `TypeSafe ${CAP_LABELS[reached.cap]} reached (${reached.used} of ${reached.limit} on ${reached.day}); no request was submitted. Requests resume after the local day rolls over, or raise the cap deliberately.`);
|
|
275
|
+
}
|
|
276
|
+
// Snapshot before awaiting so later mutations cannot change the request or validation.
|
|
277
|
+
const request = JSON.parse(body) as SystemOneRequest<Q>;
|
|
278
|
+
usage.requestsStarted += 1;
|
|
279
|
+
ledger.recordStart();
|
|
280
|
+
const start = performance.now();
|
|
281
|
+
try {
|
|
282
|
+
const result = await client.systemOne(request, callOptions);
|
|
283
|
+
if (!validResult(result, request.questions)) throw new TypeSafeIntegrationError("response", "TypeSafe returned an unexpected answer or usage format.");
|
|
284
|
+
usage.requestsSucceeded += 1;
|
|
285
|
+
usage.inputTokens += result.usage.input_tokens;
|
|
286
|
+
usage.outputTokens += result.usage.output_tokens;
|
|
287
|
+
ledger.recordSuccess(result.usage.input_tokens, result.usage.output_tokens);
|
|
288
|
+
if (!verificationRecorded) {
|
|
289
|
+
verificationRecorded = true;
|
|
290
|
+
recordAuthVerified();
|
|
291
|
+
}
|
|
292
|
+
return { ...result, elapsedMs: Math.round(performance.now() - start) };
|
|
293
|
+
} catch (error) {
|
|
294
|
+
const safe = safeError(error, backend);
|
|
295
|
+
// The request was submitted, so it counts even when it fails; the reason stays visible in `authState()`.
|
|
296
|
+
usage.requestsFailed += 1;
|
|
297
|
+
ledger.recordFailure();
|
|
298
|
+
const fingerprint = `${safe.code}:${safe.status ?? ""}:${safe.message}`;
|
|
299
|
+
if (lastFailureRecorded !== fingerprint) {
|
|
300
|
+
lastFailureRecorded = fingerprint;
|
|
301
|
+
recordAuthFailure(safe);
|
|
302
|
+
}
|
|
303
|
+
throw safe;
|
|
304
|
+
}
|
|
305
|
+
},
|
|
306
|
+
evaluateMany: <Q extends Questions>(requests: readonly SystemOneRequest<Q>[], batchOptions: BatchOptions = {}) => evaluateMany(typesafe, requests, batchOptions),
|
|
307
|
+
evaluateAll: <Q extends Questions>(request: SystemOneRequest<Q>, batchOptions: BatchOptions & { maxQuestions?: number } = {}) => evaluateAll(typesafe, request, batchOptions),
|
|
308
|
+
};
|
|
309
|
+
return typesafe;
|
|
310
|
+
}
|