auto-model-router 0.30.3 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/eval/calibrate.ts +47 -12
- package/src/eval/run.ts +18 -2
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +109 -46
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +73 -16
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
package/test/tier-plan.test.ts
CHANGED
|
@@ -2,7 +2,6 @@ import { describe, expect, test } from "bun:test";
|
|
|
2
2
|
|
|
3
3
|
import { joinBenchmarks, normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
|
|
4
4
|
import type { CatalogModel, CatalogSnapshot } from "../src/catalog/types.ts";
|
|
5
|
-
import type { Ledger } from "../src/cost/types.ts";
|
|
6
5
|
import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
|
|
7
6
|
import { buildCandidates } from "../src/router/candidates.ts";
|
|
8
7
|
import { extractFeatures } from "../src/router/features.ts";
|
|
@@ -48,7 +47,7 @@ function snapshot(list: CatalogModel[]): CatalogSnapshot {
|
|
|
48
47
|
}
|
|
49
48
|
|
|
50
49
|
describe("joinBenchmarks", () => {
|
|
51
|
-
test("copies benchmarks onto key-scoped records matched by id", () => {
|
|
50
|
+
test("copies benchmarks onto key-scoped records matched by id", async () => {
|
|
52
51
|
const keyScoped = [{ id: "a/one" }, { id: "b/two" }];
|
|
53
52
|
const pub = [
|
|
54
53
|
{ id: "a/one", benchmarks: { artificial_analysis: { coding_index: 70 } } },
|
|
@@ -59,13 +58,13 @@ describe("joinBenchmarks", () => {
|
|
|
59
58
|
expect(keyScoped[1]).not.toHaveProperty("benchmarks");
|
|
60
59
|
});
|
|
61
60
|
|
|
62
|
-
test("falls back to canonical_slug", () => {
|
|
61
|
+
test("falls back to canonical_slug", async () => {
|
|
63
62
|
const keyScoped = [{ id: "vendor/model-preview", canonical_slug: "vendor/model" }];
|
|
64
63
|
const pub = [{ id: "vendor/model", benchmarks: { artificial_analysis: { coding_index: 55 } } }];
|
|
65
64
|
expect(joinBenchmarks(keyScoped, pub)).toBe(1);
|
|
66
65
|
});
|
|
67
66
|
|
|
68
|
-
test("strips a leading ~ from an alias id", () => {
|
|
67
|
+
test("strips a leading ~ from an alias id", async () => {
|
|
69
68
|
const keyScoped = [{ id: "~vendor/model-latest" }];
|
|
70
69
|
const pub = [{ id: "vendor/model-latest", benchmarks: { artificial_analysis: { coding_index: 60 } } }];
|
|
71
70
|
expect(joinBenchmarks(keyScoped, pub)).toBe(1);
|
|
@@ -77,14 +76,14 @@ describe("joinBenchmarks", () => {
|
|
|
77
76
|
return rec.benchmarks?.artificial_analysis?.coding_index;
|
|
78
77
|
}
|
|
79
78
|
|
|
80
|
-
test("never overwrites benchmarks that are already present", () => {
|
|
79
|
+
test("never overwrites benchmarks that are already present", async () => {
|
|
81
80
|
const keyScoped: unknown[] = [{ id: "a/one", benchmarks: { artificial_analysis: { coding_index: 1 } } }];
|
|
82
81
|
const pub: unknown[] = [{ id: "a/one", benchmarks: { artificial_analysis: { coding_index: 99 } } }];
|
|
83
82
|
expect(joinBenchmarks(keyScoped, pub)).toBe(0);
|
|
84
83
|
expect(codingOf(keyScoped[0])).toBe(1);
|
|
85
84
|
});
|
|
86
85
|
|
|
87
|
-
test("a real id beats an alias target for the same key", () => {
|
|
86
|
+
test("a real id beats an alias target for the same key", async () => {
|
|
88
87
|
const keyScoped: unknown[] = [{ id: "vendor/model" }];
|
|
89
88
|
const pub: unknown[] = [
|
|
90
89
|
{ id: "other/model", canonical_slug: "vendor/model", benchmarks: { artificial_analysis: { coding_index: 10 } } },
|
|
@@ -94,11 +93,11 @@ describe("joinBenchmarks", () => {
|
|
|
94
93
|
expect(codingOf(keyScoped[0])).toBe(80);
|
|
95
94
|
});
|
|
96
95
|
|
|
97
|
-
test("tolerates junk records on both sides", () => {
|
|
96
|
+
test("tolerates junk records on both sides", async () => {
|
|
98
97
|
expect(joinBenchmarks([null, 7, "x"], [null, { id: "a" }])).toBe(0);
|
|
99
98
|
});
|
|
100
99
|
|
|
101
|
-
test("normalizing a joined record yields a scored model", () => {
|
|
100
|
+
test("normalizing a joined record yields a scored model", async () => {
|
|
102
101
|
const keyScoped: unknown[] = [raw("a/one", null, 1)];
|
|
103
102
|
joinBenchmarks(keyScoped, [raw("a/one", 66, 1)]);
|
|
104
103
|
const model = normalizeCatalogModel(keyScoped[0]);
|
|
@@ -107,7 +106,7 @@ describe("joinBenchmarks", () => {
|
|
|
107
106
|
});
|
|
108
107
|
|
|
109
108
|
describe("computeTierPlan", () => {
|
|
110
|
-
test("bands ascend across tiers", () => {
|
|
109
|
+
test("bands ascend across tiers", async () => {
|
|
111
110
|
const plan = computeTierPlan(
|
|
112
111
|
models([
|
|
113
112
|
["a/1", 10, 0.1],
|
|
@@ -124,7 +123,7 @@ describe("computeTierPlan", () => {
|
|
|
124
123
|
expect(f.hard).toBe(70);
|
|
125
124
|
});
|
|
126
125
|
|
|
127
|
-
test("an all-unscored catalog yields zero floors, never an imputed score", () => {
|
|
126
|
+
test("an all-unscored catalog yields zero floors, never an imputed score", async () => {
|
|
128
127
|
const plan = computeTierPlan(
|
|
129
128
|
models([
|
|
130
129
|
["a/1", null, 0.1],
|
|
@@ -136,7 +135,7 @@ describe("computeTierPlan", () => {
|
|
|
136
135
|
expect(plan.scoredCount.coding).toBe(0);
|
|
137
136
|
});
|
|
138
137
|
|
|
139
|
-
test("every tier floor is met by at least one available model", () => {
|
|
138
|
+
test("every tier floor is met by at least one available model", async () => {
|
|
140
139
|
const list = models([
|
|
141
140
|
["a/1", 12, 0.1],
|
|
142
141
|
["a/2", 44, 0.2],
|
|
@@ -151,7 +150,7 @@ describe("computeTierPlan", () => {
|
|
|
151
150
|
}
|
|
152
151
|
});
|
|
153
152
|
|
|
154
|
-
test("excludes built-in denials from the ranking", () => {
|
|
153
|
+
test("excludes built-in denials from the ranking", async () => {
|
|
155
154
|
// The batch entry is cheap and scored, but selection can never pick it,
|
|
156
155
|
// so it must not drag the bands down.
|
|
157
156
|
const plan = computeTierPlan(
|
|
@@ -166,12 +165,12 @@ describe("computeTierPlan", () => {
|
|
|
166
165
|
expect(plan.floors.coding.trivial).toBe(70);
|
|
167
166
|
});
|
|
168
167
|
|
|
169
|
-
test("a single scored model puts that model in every tier", () => {
|
|
168
|
+
test("a single scored model puts that model in every tier", async () => {
|
|
170
169
|
const plan = computeTierPlan(models([["a/1", 42, 0.1]]), BASE);
|
|
171
170
|
for (const tier of TIER_ORDER) expect(plan.floors.coding[tier]).toBe(42);
|
|
172
171
|
});
|
|
173
172
|
|
|
174
|
-
test("scores each axis independently", () => {
|
|
173
|
+
test("scores each axis independently", async () => {
|
|
175
174
|
const plan = computeTierPlan(models([["a/1", 30, 0.1]]), BASE);
|
|
176
175
|
expect(plan.scoredCount.coding).toBe(1);
|
|
177
176
|
expect(plan.scoredCount.agentic).toBe(1);
|
|
@@ -190,26 +189,26 @@ describe("effectiveQualityFloor", () => {
|
|
|
190
189
|
BASE,
|
|
191
190
|
);
|
|
192
191
|
|
|
193
|
-
test("relaxes a floor the catalog cannot meet", () => {
|
|
192
|
+
test("relaxes a floor the catalog cannot meet", async () => {
|
|
194
193
|
expect(effectiveQualityFloor(95, "hard", "coding", plan)).toBe(80);
|
|
195
194
|
});
|
|
196
195
|
|
|
197
|
-
test("never tightens a floor the catalog exceeds", () => {
|
|
196
|
+
test("never tightens a floor the catalog exceeds", async () => {
|
|
198
197
|
expect(effectiveQualityFloor(10, "hard", "coding", plan)).toBe(10);
|
|
199
198
|
});
|
|
200
199
|
|
|
201
|
-
test("is a no-op when configured equals adaptive", () => {
|
|
200
|
+
test("is a no-op when configured equals adaptive", async () => {
|
|
202
201
|
expect(effectiveQualityFloor(80, "hard", "coding", plan)).toBe(80);
|
|
203
202
|
});
|
|
204
203
|
});
|
|
205
204
|
|
|
206
205
|
describe("tierPlanFor", () => {
|
|
207
|
-
test("memoizes per snapshot object", () => {
|
|
206
|
+
test("memoizes per snapshot object", async () => {
|
|
208
207
|
const snap = snapshot(models([["a/1", 50, 0.1]]));
|
|
209
208
|
expect(tierPlanFor(snap, BASE)).toBe(tierPlanFor(snap, BASE));
|
|
210
209
|
});
|
|
211
210
|
|
|
212
|
-
test("a new snapshot recomputes, so a refresh tracks availability", () => {
|
|
211
|
+
test("a new snapshot recomputes, so a refresh tracks availability", async () => {
|
|
213
212
|
const first = snapshot(models([["a/1", 50, 0.1]]));
|
|
214
213
|
const second = snapshot(models([["a/1", 50, 0.1], ["a/2", 90, 0.1]]));
|
|
215
214
|
expect(tierPlanFor(second, BASE)).not.toBe(tierPlanFor(first, BASE));
|
|
@@ -245,26 +244,25 @@ describe("adaptive floors in candidate selection", () => {
|
|
|
245
244
|
tier: "hard",
|
|
246
245
|
task: "coding",
|
|
247
246
|
snapshot: lowCatalog,
|
|
248
|
-
ledger: null,
|
|
249
247
|
cfg,
|
|
250
248
|
expectedCompletionTokens: 512,
|
|
251
249
|
warmSlug: null,
|
|
252
250
|
});
|
|
253
251
|
}
|
|
254
252
|
|
|
255
|
-
test("hard is empty with adaptive floors off", () => {
|
|
253
|
+
test("hard is empty with adaptive floors off", async () => {
|
|
256
254
|
const { candidates } = build({ ...BASE, adaptiveTierFloors: false });
|
|
257
255
|
expect(candidates).toHaveLength(0);
|
|
258
256
|
});
|
|
259
257
|
|
|
260
|
-
test("hard still selects the best available with adaptive floors on", () => {
|
|
258
|
+
test("hard still selects the best available with adaptive floors on", async () => {
|
|
261
259
|
const { candidates } = build({ ...BASE, adaptiveTierFloors: true });
|
|
262
260
|
expect(candidates.length).toBeGreaterThan(0);
|
|
263
261
|
// The top band is the best-scoring model, not the cheapest.
|
|
264
262
|
expect(candidates.some((c) => c.model.slug === "a/4")).toBe(true);
|
|
265
263
|
});
|
|
266
264
|
|
|
267
|
-
test("adaptive floors still order the tiers apart", () => {
|
|
265
|
+
test("adaptive floors still order the tiers apart", async () => {
|
|
268
266
|
const cfg = { ...BASE, adaptiveTierFloors: true };
|
|
269
267
|
const best = (tier: "trivial" | "hard"): number => {
|
|
270
268
|
const { candidates } = buildCandidates({
|
|
@@ -273,7 +271,6 @@ describe("adaptive floors in candidate selection", () => {
|
|
|
273
271
|
tier,
|
|
274
272
|
task: "coding",
|
|
275
273
|
snapshot: lowCatalog,
|
|
276
|
-
ledger: null,
|
|
277
274
|
cfg,
|
|
278
275
|
expectedCompletionTokens: 512,
|
|
279
276
|
warmSlug: null,
|
|
@@ -284,7 +281,7 @@ describe("adaptive floors in candidate selection", () => {
|
|
|
284
281
|
expect(best("hard")).toBeGreaterThanOrEqual(best("trivial"));
|
|
285
282
|
});
|
|
286
283
|
|
|
287
|
-
test("excludeSlugs removes a model from the candidate set", () => {
|
|
284
|
+
test("excludeSlugs removes a model from the candidate set", async () => {
|
|
288
285
|
const cfg = { ...BASE, adaptiveTierFloors: true };
|
|
289
286
|
const all = build(cfg).candidates.map((c) => c.model.slug);
|
|
290
287
|
const target = all[0];
|
|
@@ -295,7 +292,6 @@ describe("adaptive floors in candidate selection", () => {
|
|
|
295
292
|
tier: "hard",
|
|
296
293
|
task: "coding",
|
|
297
294
|
snapshot: lowCatalog,
|
|
298
|
-
ledger: null,
|
|
299
295
|
cfg,
|
|
300
296
|
expectedCompletionTokens: 512,
|
|
301
297
|
warmSlug: null,
|
|
@@ -323,12 +319,12 @@ describe("adaptive price ceilings", () => {
|
|
|
323
319
|
);
|
|
324
320
|
const features = extractFeatures(req, 100);
|
|
325
321
|
|
|
326
|
-
test("computeTierPlan derives per-tier price bands from the catalog", () => {
|
|
322
|
+
test("computeTierPlan derives per-tier price bands from the catalog", async () => {
|
|
327
323
|
const plan = computeTierPlan(priced, BASE);
|
|
328
324
|
expect(plan.priceCeilings).toEqual({ trivial: 1, simple: 2, moderate: 3, hard: 4 });
|
|
329
325
|
});
|
|
330
326
|
|
|
331
|
-
test("effectivePriceCeiling: band when on, tighter of config/band, config when off", () => {
|
|
327
|
+
test("effectivePriceCeiling: band when on, tighter of config/band, config when off", async () => {
|
|
332
328
|
const plan = computeTierPlan(priced, BASE);
|
|
333
329
|
expect(effectivePriceCeiling(undefined, "moderate", plan, true)).toBe(3); // band
|
|
334
330
|
expect(effectivePriceCeiling(2, "moderate", plan, true)).toBe(2); // config tightens
|
|
@@ -337,7 +333,7 @@ describe("adaptive price ceilings", () => {
|
|
|
337
333
|
expect(effectivePriceCeiling(undefined, "hard", plan, false)).toBeUndefined();
|
|
338
334
|
});
|
|
339
335
|
|
|
340
|
-
test("a model above the adaptive band is dropped in candidate selection", () => {
|
|
336
|
+
test("a model above the adaptive band is dropped in candidate selection", async () => {
|
|
341
337
|
const snap = snapshot(priced);
|
|
342
338
|
const run = (adaptivePriceCeilings: boolean) =>
|
|
343
339
|
buildCandidates({
|
|
@@ -346,7 +342,6 @@ describe("adaptive price ceilings", () => {
|
|
|
346
342
|
tier: "moderate",
|
|
347
343
|
task: "coding",
|
|
348
344
|
snapshot: snap,
|
|
349
|
-
ledger: null,
|
|
350
345
|
cfg: { ...BASE, adaptivePriceCeilings },
|
|
351
346
|
expectedCompletionTokens: 512,
|
|
352
347
|
warmSlug: null,
|
|
@@ -359,14 +354,14 @@ describe("adaptive price ceilings", () => {
|
|
|
359
354
|
expect(on.rejected.some((r) => r.slug === "a/4" && r.reason === "over_price_ceiling")).toBe(true);
|
|
360
355
|
});
|
|
361
356
|
|
|
362
|
-
test("a price ceiling judges the BIASED price, so prepaid capacity is not thrown out on list", () => {
|
|
357
|
+
test("a price ceiling judges the BIASED price, so prepaid capacity is not thrown out on list", async () => {
|
|
363
358
|
// A subscription upstream inherits its OpenRouter twin's list price ($4/Mtok here) and is
|
|
364
359
|
// discounted by `costBias` because the capacity is already paid for. Judging the ceiling on
|
|
365
360
|
// list threw it out before the bias was ever read, which made the bias entirely inert.
|
|
366
361
|
const snap = snapshot(priced);
|
|
367
362
|
const biased = { ...snap, providerBias: { [snap.models[0]!.provider]: 0.1 } };
|
|
368
363
|
const run = (s: typeof snap) =>
|
|
369
|
-
buildCandidates({ req, features, tier: "moderate", task: "coding", snapshot: s,
|
|
364
|
+
buildCandidates({ req, features, tier: "moderate", task: "coding", snapshot: s, cfg: { ...BASE, adaptivePriceCeilings: true }, expectedCompletionTokens: 512, warmSlug: null });
|
|
370
365
|
// Unbiased: the band tightens moderate to $3 and a/4 is over it.
|
|
371
366
|
expect(run(snap).rejected.some((r) => r.slug === "a/4" && r.reason === "over_price_ceiling")).toBe(true);
|
|
372
367
|
// Biased ×0.1: $4 list is $0.40 to this deployment, so it clears the same ceiling.
|
|
@@ -375,7 +370,7 @@ describe("adaptive price ceilings", () => {
|
|
|
375
370
|
expect(on.rejected.some((r) => r.slug === "a/4")).toBe(false);
|
|
376
371
|
});
|
|
377
372
|
|
|
378
|
-
test("a tool turn excludes tool-incapable models on the cheap tiers and ranks on agentic above them", () => {
|
|
373
|
+
test("a tool turn excludes tool-incapable models on the cheap tiers and ranks on agentic above them", async () => {
|
|
379
374
|
// Two Ollama-priced models: the cheap one cannot drive a tool loop (agentic 1.4, as
|
|
380
375
|
// ollama/gpt-oss:20b really scores), the dearer one can (agentic 51.2, glm-5.3-flash).
|
|
381
376
|
const mk = (slug: string, price: number, intelligence: number, agentic: number) => ({
|
|
@@ -391,7 +386,7 @@ describe("adaptive price ceilings", () => {
|
|
|
391
386
|
const snap = { models: [weak, capable], fetchedAtMs: Date.now(), keyScoped: false };
|
|
392
387
|
const run = (tier: "trivial" | "moderate", min: number) =>
|
|
393
388
|
buildCandidates({
|
|
394
|
-
req, features, tier, task: "chat", snapshot: snap,
|
|
389
|
+
req, features, tier, task: "chat", snapshot: snap,
|
|
395
390
|
cfg: { ...BASE, filters: { ...BASE.filters, minAgenticForToolTurns: min } },
|
|
396
391
|
expectedCompletionTokens: 512, warmSlug: null,
|
|
397
392
|
});
|
|
@@ -440,20 +435,19 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
|
|
|
440
435
|
tier: "hard",
|
|
441
436
|
task: "coding",
|
|
442
437
|
snapshot: spread,
|
|
443
|
-
ledger: null,
|
|
444
438
|
cfg: { ...BASE, tiers: { ...BASE.tiers, hard: { ...BASE.tiers.hard, ...tierOverride } } },
|
|
445
439
|
expectedCompletionTokens: 512,
|
|
446
440
|
warmSlug: null,
|
|
447
441
|
});
|
|
448
442
|
|
|
449
|
-
test("raw scoring at the shipped exponent picks the cheapest ELIGIBLE model", () => {
|
|
443
|
+
test("raw scoring at the shipped exponent picks the cheapest ELIGIBLE model", async () => {
|
|
450
444
|
const { candidates, rejected } = run({ qualityExponent: 3 });
|
|
451
445
|
expect(candidates[0]?.model.slug).toBe("mid/2");
|
|
452
446
|
// cheap/1 is under the hard floor of 72 and never competes.
|
|
453
447
|
expect(rejected.some((r) => r.slug === "cheap/1" && r.reason === "below_quality_floor")).toBe(true);
|
|
454
448
|
});
|
|
455
449
|
|
|
456
|
-
test("normalization lets a single-digit exponent buy the best model, which raw cannot", () => {
|
|
450
|
+
test("normalization lets a single-digit exponent buy the best model, which raw cannot", async () => {
|
|
457
451
|
// Raw at the same exponent still cannot reach it: that is the defect.
|
|
458
452
|
expect(run({ qualityExponent: 12 }).candidates[0]?.model.slug).toBe("mid/2");
|
|
459
453
|
// Normalised, the same 12 selects the top-quality model.
|
|
@@ -462,7 +456,7 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
|
|
|
462
456
|
expect(normalised.candidates[0]?.reasons.some((r) => r.includes("quality normalised"))).toBe(true);
|
|
463
457
|
});
|
|
464
458
|
|
|
465
|
-
test("normalization is monotone in the exponent: higher never picks a weaker model", () => {
|
|
459
|
+
test("normalization is monotone in the exponent: higher never picks a weaker model", async () => {
|
|
466
460
|
let lastQuality = 0;
|
|
467
461
|
for (const qualityExponent of [1, 4, 8, 12, 20]) {
|
|
468
462
|
const top = run({ qualityExponent, qualityNormalization: true }).candidates[0];
|
|
@@ -472,7 +466,7 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
|
|
|
472
466
|
}
|
|
473
467
|
});
|
|
474
468
|
|
|
475
|
-
test("capability floor takes the best model inside the cap, ignoring the ratio", () => {
|
|
469
|
+
test("capability floor takes the best model inside the cap, ignoring the ratio", async () => {
|
|
476
470
|
// mid/2 costs ~$0.0016 and good/3 ~$0.0049, so this cap admits both but
|
|
477
471
|
// excludes best/4 (~$0.0082). The ranked winner is mid/2 (cheapest).
|
|
478
472
|
const cap = 0.005;
|
|
@@ -485,14 +479,14 @@ describe("quality normalization and capability floor (benchmark findings 4/6)",
|
|
|
485
479
|
expect(top?.reasons.some((r) => r.includes("capability floor"))).toBe(true);
|
|
486
480
|
});
|
|
487
481
|
|
|
488
|
-
test("capability floor is strictly an upgrade: an unaffordable cap changes nothing", () => {
|
|
482
|
+
test("capability floor is strictly an upgrade: an unaffordable cap changes nothing", async () => {
|
|
489
483
|
const base = run({}).candidates.map((c) => c.model.slug);
|
|
490
484
|
// A cap below every candidate's cost promotes nobody.
|
|
491
485
|
const tiny = run({ capabilityFloorUsd: 1e-9 }).candidates.map((c) => c.model.slug);
|
|
492
486
|
expect(tiny).toEqual(base);
|
|
493
487
|
});
|
|
494
488
|
|
|
495
|
-
test("both modes stay inert by default, so shipped behaviour is unchanged", () => {
|
|
489
|
+
test("both modes stay inert by default, so shipped behaviour is unchanged", async () => {
|
|
496
490
|
const shipped = run({});
|
|
497
491
|
expect(shipped.candidates[0]?.model.slug).toBe("mid/2");
|
|
498
492
|
expect(shipped.candidates.every((c) => !c.reasons.some((r) => r.includes("normalised")))).toBe(true);
|
|
@@ -524,26 +518,26 @@ describe("thinness-gated relaxation (review 2026-09-05 §1)", () => {
|
|
|
524
518
|
BASE,
|
|
525
519
|
);
|
|
526
520
|
|
|
527
|
-
test("the bands sit below the configured floors on a wide catalog", () => {
|
|
521
|
+
test("the bands sit below the configured floors on a wide catalog", async () => {
|
|
528
522
|
// The premise the gate exists for: unconditional min() would relax here.
|
|
529
523
|
expect(wide.floors.coding.moderate).toBeLessThan(60);
|
|
530
524
|
expect(wide.floors.coding.hard).toBeLessThan(72);
|
|
531
525
|
});
|
|
532
526
|
|
|
533
|
-
test("a configured floor that three or more models meet stands as written", () => {
|
|
527
|
+
test("a configured floor that three or more models meet stands as written", async () => {
|
|
534
528
|
expect(effectiveQualityFloor(60, "moderate", "coding", wide)).toBe(60); // 62,70,74,76,78 meet it
|
|
535
529
|
expect(effectiveQualityFloor(72, "hard", "coding", wide)).toBe(72); // 74,76,78 meet it
|
|
536
530
|
expect(effectiveQualityFloor(40, "simple", "coding", wide)).toBe(40);
|
|
537
531
|
});
|
|
538
532
|
|
|
539
|
-
test("a floor fewer than three models meet is relaxed to the band", () => {
|
|
533
|
+
test("a floor fewer than three models meet is relaxed to the band", async () => {
|
|
540
534
|
// Only 76 and 78 clear 75: thin, so the hard band applies.
|
|
541
535
|
expect(effectiveQualityFloor(75, "hard", "coding", wide)).toBe(Math.min(75, wide.floors.coding.hard));
|
|
542
536
|
// Nothing clears 90: relaxed as before.
|
|
543
537
|
expect(effectiveQualityFloor(90, "hard", "coding", wide)).toBe(wide.floors.coding.hard);
|
|
544
538
|
});
|
|
545
539
|
|
|
546
|
-
test("countAdmitted counts scores at or above the floor", () => {
|
|
540
|
+
test("countAdmitted counts scores at or above the floor", async () => {
|
|
547
541
|
expect(countAdmitted([10, 20, 30, 40], 25)).toBe(2);
|
|
548
542
|
expect(countAdmitted([10, 20, 30, 40], 40)).toBe(1);
|
|
549
543
|
expect(countAdmitted([10, 20, 30, 40], 41)).toBe(0);
|
|
@@ -551,7 +545,7 @@ describe("thinness-gated relaxation (review 2026-09-05 §1)", () => {
|
|
|
551
545
|
expect(countAdmitted([], 0)).toBe(0);
|
|
552
546
|
});
|
|
553
547
|
|
|
554
|
-
test("in candidate selection the wide catalog keeps weak models out of moderate", () => {
|
|
548
|
+
test("in candidate selection the wide catalog keeps weak models out of moderate", async () => {
|
|
555
549
|
const req = parseChatRequest(
|
|
556
550
|
{
|
|
557
551
|
model: "auto",
|
|
@@ -576,7 +570,6 @@ describe("thinness-gated relaxation (review 2026-09-05 §1)", () => {
|
|
|
576
570
|
tier: "moderate",
|
|
577
571
|
task: "coding",
|
|
578
572
|
snapshot: snap,
|
|
579
|
-
ledger: null,
|
|
580
573
|
cfg: { ...BASE, adaptiveTierFloors: true },
|
|
581
574
|
expectedCompletionTokens: 512,
|
|
582
575
|
warmSlug: null,
|
|
@@ -610,17 +603,6 @@ describe("escalation-cost term (review 2026-09-05 §2)", () => {
|
|
|
610
603
|
slug === "cheap/flaky"
|
|
611
604
|
? { slug, attempts: 100, escalations: 4, errors: 0, successRate: 0.96, meanCostError: 0 }
|
|
612
605
|
: { slug, attempts: 100, escalations: 0, errors: 4, successRate: 0.96, meanCostError: 0 };
|
|
613
|
-
const ledger: Ledger = {
|
|
614
|
-
record: () => {},
|
|
615
|
-
conversationSpend: () => 0,
|
|
616
|
-
spendSince: () => 0,
|
|
617
|
-
blendedRate: () => null,
|
|
618
|
-
trust: (slug) => trustOf(slug),
|
|
619
|
-
allTrust: () => [],
|
|
620
|
-
latency: () => null,
|
|
621
|
-
tokenRatio: () => null,
|
|
622
|
-
recentEntries: () => [],
|
|
623
|
-
};
|
|
624
606
|
function build(weight: number, usdPerPromptToken?: number) {
|
|
625
607
|
return buildCandidates({
|
|
626
608
|
req,
|
|
@@ -628,7 +610,8 @@ describe("escalation-cost term (review 2026-09-05 §2)", () => {
|
|
|
628
610
|
tier: "trivial",
|
|
629
611
|
task: "coding",
|
|
630
612
|
snapshot: snap,
|
|
631
|
-
|
|
613
|
+
// Scoring reads trust out of the prefetched signals.
|
|
614
|
+
signals: new Map(snap.models.map((m) => [m.slug, { trust: trustOf(m.slug), latency: null }])),
|
|
632
615
|
cfg: { ...BASE, filters: { ...BASE.filters, escalationCostWeight: weight } },
|
|
633
616
|
expectedCompletionTokens: 512,
|
|
634
617
|
warmSlug: null,
|
|
@@ -636,21 +619,21 @@ describe("escalation-cost term (review 2026-09-05 §2)", () => {
|
|
|
636
619
|
});
|
|
637
620
|
}
|
|
638
621
|
|
|
639
|
-
test("with the term off, nothing separates them and the tie falls lexically to the flaky model", () => {
|
|
622
|
+
test("with the term off, nothing separates them and the tie falls lexically to the flaky model", async () => {
|
|
640
623
|
const { candidates } = build(0, 1e-6);
|
|
641
624
|
// Same success rate, same trust divisor: escalations are invisible.
|
|
642
625
|
expect(candidates[0]!.model.slug).toBe("cheap/flaky");
|
|
643
626
|
expect(candidates[0]!.reasons.some((r) => r.startsWith("escalation risk"))).toBe(false);
|
|
644
627
|
});
|
|
645
628
|
|
|
646
|
-
test("priced at what an escalated retry actually bills, the flaky model loses", () => {
|
|
629
|
+
test("priced at what an escalated retry actually bills, the flaky model loses", async () => {
|
|
647
630
|
const { candidates } = build(1, 1e-6); // $1/Mtok of escalated-retry cost
|
|
648
631
|
expect(candidates[0]!.model.slug).toBe("cheap/solid");
|
|
649
632
|
const flaky = candidates.find((c) => c.model.slug === "cheap/flaky")!;
|
|
650
633
|
expect(flaky.reasons.some((r) => r.startsWith("escalation risk"))).toBe(true);
|
|
651
634
|
});
|
|
652
635
|
|
|
653
|
-
test("inert until the ledger can measure the retry cost", () => {
|
|
636
|
+
test("inert until the ledger can measure the retry cost", async () => {
|
|
654
637
|
const { candidates } = build(1);
|
|
655
638
|
expect(candidates[0]!.model.slug).toBe("cheap/flaky");
|
|
656
639
|
});
|
package/test/toast-logic.test.ts
CHANGED
|
@@ -18,7 +18,7 @@ describe("resolveRouterUrl", () => {
|
|
|
18
18
|
const resolve = (env: string | undefined, text: string | null): string =>
|
|
19
19
|
resolveRouterUrl(env, text, parseYaml);
|
|
20
20
|
|
|
21
|
-
test("the embed port file wins over AUTO_MODEL_ROUTER_PORT and the config", () => {
|
|
21
|
+
test("the embed port file wins over AUTO_MODEL_ROUTER_PORT and the config", async () => {
|
|
22
22
|
// The embedded router binds a free OS-assigned port and writes it to the
|
|
23
23
|
// port file; the toast must poll that actual address, not a stale config.
|
|
24
24
|
expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "8812", 45678)).toBe(
|
|
@@ -26,57 +26,57 @@ describe("resolveRouterUrl", () => {
|
|
|
26
26
|
);
|
|
27
27
|
});
|
|
28
28
|
|
|
29
|
-
test("AUTO_MODEL_ROUTER_URL still beats the embed port file", () => {
|
|
29
|
+
test("AUTO_MODEL_ROUTER_URL still beats the embed port file", async () => {
|
|
30
30
|
expect(resolveRouterUrl("http://host:9999", "server:\n port: 8788\n", parseYaml, "8812", 45678)).toBe(
|
|
31
31
|
"http://host:9999",
|
|
32
32
|
);
|
|
33
33
|
});
|
|
34
34
|
|
|
35
|
-
test("an embed port file of null falls back to AUTO_MODEL_ROUTER_PORT", () => {
|
|
35
|
+
test("an embed port file of null falls back to AUTO_MODEL_ROUTER_PORT", async () => {
|
|
36
36
|
expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "8812", null)).toBe("http://127.0.0.1:8812");
|
|
37
37
|
});
|
|
38
38
|
|
|
39
|
-
test("AUTO_MODEL_ROUTER_URL still beats AUTO_MODEL_ROUTER_PORT", () => {
|
|
39
|
+
test("AUTO_MODEL_ROUTER_URL still beats AUTO_MODEL_ROUTER_PORT", async () => {
|
|
40
40
|
expect(resolveRouterUrl("http://host:9999", "server:\n port: 8788\n", parseYaml, "8812")).toBe("http://host:9999");
|
|
41
41
|
});
|
|
42
42
|
|
|
43
|
-
test("an invalid AUTO_MODEL_ROUTER_PORT falls back to the config port", () => {
|
|
43
|
+
test("an invalid AUTO_MODEL_ROUTER_PORT falls back to the config port", async () => {
|
|
44
44
|
expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "notaport")).toBe("http://127.0.0.1:8788");
|
|
45
45
|
expect(resolveRouterUrl(undefined, "server:\n port: 8788\n", parseYaml, "70000")).toBe("http://127.0.0.1:8788");
|
|
46
46
|
});
|
|
47
47
|
|
|
48
|
-
test("AUTO_MODEL_ROUTER_PORT with no config uses loopback", () => {
|
|
48
|
+
test("AUTO_MODEL_ROUTER_PORT with no config uses loopback", async () => {
|
|
49
49
|
expect(resolveRouterUrl(undefined, null, parseYaml, "8812")).toBe("http://127.0.0.1:8812");
|
|
50
50
|
});
|
|
51
51
|
|
|
52
|
-
test("reads host and port from the router's own config", () => {
|
|
52
|
+
test("reads host and port from the router's own config", async () => {
|
|
53
53
|
// The bug this prevents: defaulting to 8788 polls whatever else owns that
|
|
54
54
|
// port once the router has been moved, and toasts silently never appear.
|
|
55
55
|
expect(resolve(undefined, "server:\n host: 127.0.0.1\n port: 8788\n")).toBe("http://127.0.0.1:8788");
|
|
56
56
|
});
|
|
57
57
|
|
|
58
|
-
test("a port-only config keeps the loopback default host", () => {
|
|
58
|
+
test("a port-only config keeps the loopback default host", async () => {
|
|
59
59
|
expect(resolve(undefined, "server:\n port: 8790\n")).toBe("http://127.0.0.1:8790");
|
|
60
60
|
});
|
|
61
61
|
|
|
62
|
-
test("a wildcard listen address becomes loopback", () => {
|
|
62
|
+
test("a wildcard listen address becomes loopback", async () => {
|
|
63
63
|
expect(resolve(undefined, "server:\n host: 0.0.0.0\n port: 8788\n")).toBe("http://127.0.0.1:8788");
|
|
64
64
|
expect(resolve(undefined, "server:\n host: '::'\n port: 8788\n")).toBe("http://127.0.0.1:8788");
|
|
65
65
|
});
|
|
66
66
|
|
|
67
|
-
test("falls back when there is no config, no server block, or junk", () => {
|
|
67
|
+
test("falls back when there is no config, no server block, or junk", async () => {
|
|
68
68
|
expect(resolve(undefined, null)).toBe(DEFAULT_ROUTER_URL);
|
|
69
69
|
expect(resolve(undefined, "")).toBe(DEFAULT_ROUTER_URL);
|
|
70
70
|
expect(resolve(undefined, "logLevel: debug\n")).toBe(DEFAULT_ROUTER_URL);
|
|
71
71
|
expect(resolve(undefined, "server: 5\n")).toBe(DEFAULT_ROUTER_URL);
|
|
72
72
|
});
|
|
73
73
|
|
|
74
|
-
test("ignores a non-integer or non-positive port", () => {
|
|
74
|
+
test("ignores a non-integer or non-positive port", async () => {
|
|
75
75
|
expect(resolve(undefined, "server:\n port: 0\n")).toBe(DEFAULT_ROUTER_URL);
|
|
76
76
|
expect(resolve(undefined, "server:\n port: notaport\n")).toBe(DEFAULT_ROUTER_URL);
|
|
77
77
|
});
|
|
78
78
|
|
|
79
|
-
test("an empty env override does not shadow the config", () => {
|
|
79
|
+
test("an empty env override does not shadow the config", async () => {
|
|
80
80
|
expect(resolve("", "server:\n port: 8788\n")).toBe("http://127.0.0.1:8788");
|
|
81
81
|
});
|
|
82
82
|
});
|
|
@@ -95,12 +95,12 @@ function dec(partial: Partial<ToastDecision>): ToastDecision {
|
|
|
95
95
|
}
|
|
96
96
|
|
|
97
97
|
describe("selectToasts", () => {
|
|
98
|
-
test("toasts nothing on the first tick (lastSeenId null)", () => {
|
|
98
|
+
test("toasts nothing on the first tick (lastSeenId null)", async () => {
|
|
99
99
|
const entries = [dec({ id: "a" }), dec({ id: "b" })];
|
|
100
100
|
expect(selectToasts(entries, null)).toEqual([]);
|
|
101
101
|
});
|
|
102
102
|
|
|
103
|
-
test("toasts only entries newer than the last-seen id, oldest first", () => {
|
|
103
|
+
test("toasts only entries newer than the last-seen id, oldest first", async () => {
|
|
104
104
|
// newest-first order: d3 is newest, d1 oldest
|
|
105
105
|
const entries = [dec({ id: "d3", slug: "x/c" }), dec({ id: "d2", slug: "x/b" }), dec({ id: "d1", slug: "x/a" })];
|
|
106
106
|
const toasts = selectToasts(entries, "d1");
|
|
@@ -110,7 +110,7 @@ describe("selectToasts", () => {
|
|
|
110
110
|
expect(toasts[1]?.model).toBe("x/c");
|
|
111
111
|
});
|
|
112
112
|
|
|
113
|
-
test("skips wasted (abandoned escalation) entries", () => {
|
|
113
|
+
test("skips wasted (abandoned escalation) entries", async () => {
|
|
114
114
|
const entries = [dec({ id: "d2", slug: "served", wasted: false }), dec({ id: "d1", wasted: true })];
|
|
115
115
|
// both newer than lastSeenId ""; only the non-wasted one toasts
|
|
116
116
|
expect(selectToasts(entries, "")).toHaveLength(1);
|
|
@@ -120,12 +120,12 @@ describe("selectToasts", () => {
|
|
|
120
120
|
expect(out[0]?.model).toBe("real");
|
|
121
121
|
});
|
|
122
122
|
|
|
123
|
-
test("empty input yields no toasts and null newest id", () => {
|
|
123
|
+
test("empty input yields no toasts and null newest id", async () => {
|
|
124
124
|
expect(selectToasts([], "x")).toEqual([]);
|
|
125
125
|
expect(newestId([])).toBeNull();
|
|
126
126
|
});
|
|
127
127
|
|
|
128
|
-
test("filters to the requesting harness when one is set", () => {
|
|
128
|
+
test("filters to the requesting harness when one is set", async () => {
|
|
129
129
|
const entries = [
|
|
130
130
|
dec({ id: "d3", slug: "mine", harnessId: "harness-a" }),
|
|
131
131
|
dec({ id: "d2", slug: "other", harnessId: "harness-b" }),
|
|
@@ -137,7 +137,7 @@ describe("selectToasts", () => {
|
|
|
137
137
|
expect(toasts[0]?.model).toBe("mine");
|
|
138
138
|
});
|
|
139
139
|
|
|
140
|
-
test("empty harness id toasts every harness", () => {
|
|
140
|
+
test("empty harness id toasts every harness", async () => {
|
|
141
141
|
const entries = [
|
|
142
142
|
dec({ id: "d2", slug: "a", harnessId: "harness-a" }),
|
|
143
143
|
dec({ id: "d1", slug: "b", harnessId: "harness-b" }),
|
|
@@ -145,7 +145,7 @@ describe("selectToasts", () => {
|
|
|
145
145
|
expect(selectToasts(entries, "", "")).toHaveLength(2);
|
|
146
146
|
});
|
|
147
147
|
|
|
148
|
-
test("filters to the requesting omp session when one is set", () => {
|
|
148
|
+
test("filters to the requesting omp session when one is set", async () => {
|
|
149
149
|
// Two interactive omp sessions sharing one router's ledger: session-a's
|
|
150
150
|
// toast must not surface session-b's decisions.
|
|
151
151
|
const entries = [
|
|
@@ -158,7 +158,7 @@ describe("selectToasts", () => {
|
|
|
158
158
|
expect(toasts[0]?.model).toBe("mine");
|
|
159
159
|
});
|
|
160
160
|
|
|
161
|
-
test("empty omp session id toasts every session", () => {
|
|
161
|
+
test("empty omp session id toasts every session", async () => {
|
|
162
162
|
const entries = [
|
|
163
163
|
dec({ id: "d2", slug: "a", ompSessionId: "sess-a" }),
|
|
164
164
|
dec({ id: "d1", slug: "b", ompSessionId: "sess-b" }),
|
|
@@ -166,7 +166,7 @@ describe("selectToasts", () => {
|
|
|
166
166
|
expect(selectToasts(entries, "", "", "")).toHaveLength(2);
|
|
167
167
|
});
|
|
168
168
|
|
|
169
|
-
test("harness and session filters compose", () => {
|
|
169
|
+
test("harness and session filters compose", async () => {
|
|
170
170
|
const entries = [
|
|
171
171
|
dec({ id: "d3", slug: "keep", harnessId: "h", ompSessionId: "sess-a" }),
|
|
172
172
|
dec({ id: "d2", slug: "wrong-session", harnessId: "h", ompSessionId: "sess-b" }),
|
|
@@ -179,21 +179,21 @@ describe("selectToasts", () => {
|
|
|
179
179
|
});
|
|
180
180
|
|
|
181
181
|
describe("toToastText", () => {
|
|
182
|
-
test("prefers servedSlug when present, else slug", () => {
|
|
182
|
+
test("prefers servedSlug when present, else slug", async () => {
|
|
183
183
|
expect(toToastText(dec({ slug: "s/one", servedSlug: "s/real" }))).toContain("s/real");
|
|
184
184
|
expect(toToastText(dec({ slug: "s/one", servedSlug: null }))).toContain("s/one");
|
|
185
185
|
});
|
|
186
186
|
|
|
187
|
-
test("includes cost when reported, omits otherwise", () => {
|
|
187
|
+
test("includes cost when reported, omits otherwise", async () => {
|
|
188
188
|
expect(toToastText(dec({ reportedUsd: 0.5 }))).toContain("$0.50000");
|
|
189
189
|
expect(toToastText(dec({ reportedUsd: null }))).not.toContain("$");
|
|
190
190
|
});
|
|
191
191
|
|
|
192
|
-
test("renders provider · model [tier]", () => {
|
|
192
|
+
test("renders provider · model [tier]", async () => {
|
|
193
193
|
expect(toToastText(dec({ slug: "q/w", tier: "hard", reportedUsd: null }))).toBe("openrouter · q/w [hard]");
|
|
194
194
|
});
|
|
195
195
|
|
|
196
|
-
test("an Ollama slug is labelled with its provider and shown without the prefix", () => {
|
|
196
|
+
test("an Ollama slug is labelled with its provider and shown without the prefix", async () => {
|
|
197
197
|
expect(toToastText(dec({ slug: "ollama/glm-5.3-flash", servedSlug: "ollama/glm-5.3-flash", tier: "moderate", reportedUsd: 0.0007 }))).toBe(
|
|
198
198
|
"ollama · glm-5.3-flash [moderate] · $0.00070",
|
|
199
199
|
);
|
|
@@ -219,7 +219,7 @@ describe("verbose toast", () => {
|
|
|
219
219
|
ttftMs: 2100,
|
|
220
220
|
});
|
|
221
221
|
|
|
222
|
-
test("the headline keeps its shape and the body explains the choice", () => {
|
|
222
|
+
test("the headline keeps its shape and the body explains the choice", async () => {
|
|
223
223
|
const lines = toToastText(FULL).split(String.fromCharCode(10));
|
|
224
224
|
expect(lines[0]).toBe("ollama · gpt-oss:20b [trivial] · $0.00031");
|
|
225
225
|
expect(lines[1]).toBe("why: policy: pinned to ollama/gpt-oss:20b");
|
|
@@ -227,12 +227,12 @@ describe("verbose toast", () => {
|
|
|
227
227
|
expect(lines[3]).toBe("22.9k prompt · 12.8k compacted · 11 tools · tool continuation · attempt 2 · 2.1s to first token");
|
|
228
228
|
});
|
|
229
229
|
|
|
230
|
-
test("compact is the old single line", () => {
|
|
230
|
+
test("compact is the old single line", async () => {
|
|
231
231
|
expect(toToastText(FULL, false)).toBe("ollama · gpt-oss:20b [trivial] · $0.00031");
|
|
232
232
|
expect(toToastText(FULL, false).includes(String.fromCharCode(10))).toBe(false);
|
|
233
233
|
});
|
|
234
234
|
|
|
235
|
-
test("a surprise outranks ordinary ranking, and ranking shows when nothing surprised", () => {
|
|
235
|
+
test("a surprise outranks ordinary ranking, and ranking shows when nothing surprised", async () => {
|
|
236
236
|
// "cheapest above the quality floor" is ordinary: a failover and a hold win.
|
|
237
237
|
expect(whyReasons(["cheapest above the quality floor", "failover: x/y empty_completion; retrying a/b"])[0]).toContain("failover");
|
|
238
238
|
expect(whyReasons(["cheapest above the quality floor", "held from the previous turn: switch margin not cleared"])[0]).toContain("held");
|
|
@@ -246,17 +246,17 @@ describe("verbose toast", () => {
|
|
|
246
246
|
expect(whyReasons(["policy: pinned to ollama/gpt-oss:20b", "pinned to ollama/gpt-oss:20b by session override"])).toEqual(["policy: pinned to ollama/gpt-oss:20b"]);
|
|
247
247
|
});
|
|
248
248
|
|
|
249
|
-
test("an unreported cost falls back to the prediction, and thin decisions stay short", () => {
|
|
249
|
+
test("an unreported cost falls back to the prediction, and thin decisions stay short", async () => {
|
|
250
250
|
expect(toToastText(dec({ reportedUsd: null, predictedUsd: 0.00042, reasons: [], features: null }))).toBe("openrouter · meta/muse-glimmer-30b [trivial] · ~$0.00042");
|
|
251
251
|
expect(toToastText(dec({ reportedUsd: null, features: null }))).toBe("openrouter · meta/muse-glimmer-30b [trivial]");
|
|
252
252
|
});
|
|
253
253
|
|
|
254
|
-
test("facts skip what a reader does not need", () => {
|
|
254
|
+
test("facts skip what a reader does not need", async () => {
|
|
255
255
|
expect(factsOf(dec({ features: { promptTokens: 0, toolCount: 0 }, attempt: 0, task: "coding" }))).toEqual([]);
|
|
256
256
|
expect(factsOf(dec({ features: { promptTokens: 900 }, task: "vision" }))).toEqual(["900 prompt", "vision"]);
|
|
257
257
|
});
|
|
258
258
|
|
|
259
|
-
test("selectToasts renders compact when asked", () => {
|
|
259
|
+
test("selectToasts renders compact when asked", async () => {
|
|
260
260
|
const entries = [dec({ id: "d2", slug: "x/b", reasons: ["failover: nope"] }), dec({ id: "d1" })];
|
|
261
261
|
expect(selectToasts(entries, "d1", "", "", false)[0]?.text.includes(String.fromCharCode(10))).toBe(false);
|
|
262
262
|
expect(selectToasts(entries, "d1")[0]?.text.includes(String.fromCharCode(10))).toBe(true);
|