auto-model-router 0.31.0 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.omp-plugin/marketplace.json +2 -2
- package/README.md +32 -2
- package/omp-extension/router-configure.ts +9 -7
- package/package.json +1 -1
- package/src/cli/config-cmd.ts +8 -7
- package/src/cli/explain.ts +10 -5
- package/src/cli/export.ts +6 -5
- package/src/cli/models.ts +10 -7
- package/src/cli/report.ts +6 -1
- package/src/cli/stats.ts +7 -7
- package/src/config/load.ts +10 -1
- package/src/config/types.ts +10 -1
- package/src/context/bridge.ts +7 -7
- package/src/context/index.ts +3 -3
- package/src/context/store.ts +39 -56
- package/src/context/types.ts +7 -6
- package/src/cost/blended.ts +28 -7
- package/src/cost/feedback.ts +33 -37
- package/src/cost/ledger-sql.ts +547 -0
- package/src/cost/ledger.ts +30 -459
- package/src/cost/report.ts +171 -129
- package/src/cost/retention.ts +10 -10
- package/src/cost/summary.ts +15 -10
- package/src/cost/types.ts +43 -62
- package/src/cost/views.ts +79 -49
- package/src/lib.ts +6 -2
- package/src/router/candidates.ts +7 -15
- package/src/router/classify.ts +6 -4
- package/src/router/index.ts +95 -9
- package/src/router/select.ts +38 -21
- package/src/router/state.ts +90 -102
- package/src/router/types.ts +11 -5
- package/src/server/advise.ts +6 -4
- package/src/server/compaction-digest.ts +1 -1
- package/src/server/digest.ts +9 -10
- package/src/server/http.ts +101 -41
- package/src/server/providers.ts +18 -4
- package/src/server/turn.ts +32 -9
- package/src/tokens/estimate.ts +16 -6
- package/src/upstream/ollama-usage.ts +21 -11
- package/src/util/schema.ts +201 -0
- package/src/util/sql.ts +246 -0
- package/src/wire/anthropic/messages.ts +3 -4
- package/src/wire/openai/request.ts +1 -0
- package/src/wire/types.ts +7 -0
- package/test/anthropic-wire.test.ts +9 -9
- package/test/benchmark-feeds.test.ts +7 -7
- package/test/cache-control.test.ts +7 -7
- package/test/cache-estimate.test.ts +5 -5
- package/test/catalog-view.test.ts +4 -4
- package/test/catalog.test.ts +11 -11
- package/test/classify.test.ts +24 -24
- package/test/compaction.test.ts +20 -20
- package/test/config-wizard.test.ts +32 -32
- package/test/config.test.ts +10 -10
- package/test/connect-harnesses.test.ts +11 -11
- package/test/context-bridge.test.ts +40 -30
- package/test/context-prune.test.ts +43 -36
- package/test/context-query.test.ts +8 -8
- package/test/controls.test.ts +54 -27
- package/test/cost.test.ts +12 -12
- package/test/digest.test.ts +55 -44
- package/test/embed-lifecycle.test.ts +5 -5
- package/test/embed-logic.test.ts +26 -26
- package/test/escalate.test.ts +17 -17
- package/test/eval.test.ts +13 -13
- package/test/executable.test.ts +6 -6
- package/test/exploration.test.ts +19 -20
- package/test/failover.test.ts +22 -21
- package/test/fakes.ts +105 -0
- package/test/features.test.ts +21 -21
- package/test/harness-requests.test.ts +3 -3
- package/test/harness-switch.test.ts +5 -5
- package/test/hold-exploration.test.ts +13 -13
- package/test/hot-reload.test.ts +5 -5
- package/test/learned.test.ts +5 -5
- package/test/ledger-sql.test.ts +342 -0
- package/test/mcp-entry.test.ts +5 -5
- package/test/migrations.test.ts +28 -22
- package/test/models-yml.test.ts +18 -18
- package/test/ollama.test.ts +40 -34
- package/test/omp-credentials.test.ts +16 -16
- package/test/policy.test.ts +3 -3
- package/test/reconfigure.test.ts +4 -4
- package/test/redaction.test.ts +41 -35
- package/test/remote.test.ts +12 -12
- package/test/report-logic.test.ts +8 -8
- package/test/report.test.ts +95 -87
- package/test/retention.test.ts +79 -66
- package/test/schema.test.ts +123 -0
- package/test/scope.test.ts +8 -8
- package/test/select.test.ts +216 -257
- package/test/skills.test.ts +3 -3
- package/test/sql-shim.test.ts +154 -0
- package/test/state.test.ts +43 -36
- package/test/summary.test.ts +38 -27
- package/test/tier-plan.test.ts +45 -62
- package/test/toast-logic.test.ts +31 -31
- package/test/tokens.test.ts +95 -80
- package/test/trust-attribution.test.ts +217 -187
- package/test/trust-window.test.ts +37 -32
- package/test/turn.test.ts +55 -23
- package/test/upstreams.test.ts +13 -13
- package/test/views.test.ts +81 -59
- package/test/wire-request.test.ts +17 -17
- package/test/wire-responses.test.ts +4 -4
- package/tools/agentdox-e2e.ts +5 -2
- package/tools/export-benchmarks.ts +5 -5
- package/tools/ledger-parity.ts +266 -0
- package/tools/replay.ts +16 -8
package/test/select.test.ts
CHANGED
|
@@ -1,11 +1,14 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
|
+
import { fakeLedger } from "./fakes.ts";
|
|
3
|
+
import type { ModelLatency } from "../src/cost/types.ts";
|
|
4
|
+
import { prefetchTurnReads } from "../src/router/index.ts";
|
|
5
|
+
import type { AsyncLedger } from "../src/cost/types.ts";
|
|
2
6
|
|
|
3
7
|
import { normalizeCatalogModel } from "../src/catalog/openrouter-catalog.ts";
|
|
4
8
|
import type { CatalogModel, CatalogSnapshot } from "../src/catalog/types.ts";
|
|
5
9
|
import { DEFAULT_CONFIG } from "../src/config/defaults.ts";
|
|
6
10
|
import { loadConfig } from "../src/config/load.ts";
|
|
7
11
|
import type { ProfileConfig, RouterConfig } from "../src/config/types.ts";
|
|
8
|
-
import type { Ledger } from "../src/cost/types.ts";
|
|
9
12
|
import { extractFeatures } from "../src/router/features.ts";
|
|
10
13
|
import { scoreHeuristic } from "../src/router/classify.ts";
|
|
11
14
|
import { latencyWeightFor } from "../src/router/candidates.ts";
|
|
@@ -74,13 +77,14 @@ function state(over: Partial<ConversationState> = {}): ConversationState {
|
|
|
74
77
|
};
|
|
75
78
|
}
|
|
76
79
|
|
|
77
|
-
|
|
80
|
+
|
|
81
|
+
async function run(opts: {
|
|
78
82
|
userText?: string;
|
|
79
83
|
promptTokens?: number;
|
|
80
84
|
cfg?: RouterConfig;
|
|
81
85
|
st?: ConversationState;
|
|
82
86
|
tier?: Tier;
|
|
83
|
-
ledger?:
|
|
87
|
+
ledger?: AsyncLedger | null;
|
|
84
88
|
harnessId?: string;
|
|
85
89
|
maxTokens?: number;
|
|
86
90
|
}) {
|
|
@@ -90,23 +94,28 @@ function run(opts: {
|
|
|
90
94
|
const features = extractFeatures(req, opts.promptTokens ?? 4000);
|
|
91
95
|
const heuristic = scoreHeuristic(features, cfg);
|
|
92
96
|
const classification = opts.tier === undefined ? heuristic : { ...heuristic, tier: opts.tier };
|
|
97
|
+
const finalReq = opts.harnessId === undefined ? req : { ...req, harnessId: opts.harnessId };
|
|
98
|
+
// `select` takes the ledger's answers as data; the fakes below are read
|
|
99
|
+
// through the same prefetch the router uses, so the tests exercise the real
|
|
100
|
+
// path rather than a second one.
|
|
101
|
+
const reads = await prefetchTurnReads(opts.ledger ?? null, finalReq, PROFILE, cfg, SNAPSHOT, classification.task);
|
|
93
102
|
return select({
|
|
94
|
-
req:
|
|
103
|
+
req: finalReq,
|
|
95
104
|
features,
|
|
96
105
|
classification,
|
|
97
106
|
profile: PROFILE,
|
|
98
107
|
state: opts.st ?? state(),
|
|
99
108
|
snapshot: SNAPSHOT,
|
|
100
|
-
|
|
109
|
+
reads,
|
|
101
110
|
cfg,
|
|
102
111
|
nowMs: Date.now(),
|
|
103
112
|
});
|
|
104
113
|
}
|
|
105
114
|
|
|
106
115
|
describe("hard exclusions", () => {
|
|
107
|
-
test("never selects a meta-router, floating alias, batch endpoint, or cloaked model", () => {
|
|
116
|
+
test("never selects a meta-router, floating alias, batch endpoint, or cloaked model", async () => {
|
|
108
117
|
for (const tier of ["trivial", "simple", "moderate", "hard"] as Tier[]) {
|
|
109
|
-
const d = run({ tier });
|
|
118
|
+
const d = await run({ tier });
|
|
110
119
|
expect(d.slug.startsWith("openrouter/")).toBe(false);
|
|
111
120
|
expect(d.slug.startsWith("~")).toBe(false);
|
|
112
121
|
expect(d.slug.endsWith(":batch")).toBe(false);
|
|
@@ -119,23 +128,23 @@ describe("hard exclusions", () => {
|
|
|
119
128
|
}
|
|
120
129
|
});
|
|
121
130
|
|
|
122
|
-
test("only offers tool-capable models when the request offers tools", () => {
|
|
131
|
+
test("only offers tool-capable models when the request offers tools", async () => {
|
|
123
132
|
for (const tier of ["trivial", "simple", "moderate", "hard"] as Tier[]) {
|
|
124
|
-
const d = run({ tier });
|
|
133
|
+
const d = await run({ tier });
|
|
125
134
|
for (const c of d.considered) expect(c.model.supportsTools).toBe(true);
|
|
126
135
|
}
|
|
127
136
|
});
|
|
128
137
|
|
|
129
|
-
test("excludes free models by default", () => {
|
|
130
|
-
const d = run({ tier: "trivial" });
|
|
138
|
+
test("excludes free models by default", async () => {
|
|
139
|
+
const d = await run({ tier: "trivial" });
|
|
131
140
|
for (const c of d.considered) expect(c.model.isFree).toBe(false);
|
|
132
141
|
});
|
|
133
142
|
});
|
|
134
143
|
|
|
135
144
|
describe("quality floor", () => {
|
|
136
|
-
test("an unscored model never satisfies a tier with a floor above zero", () => {
|
|
145
|
+
test("an unscored model never satisfies a tier with a floor above zero", async () => {
|
|
137
146
|
for (const tier of ["simple", "moderate", "hard"] as Tier[]) {
|
|
138
|
-
const d = run({ tier });
|
|
147
|
+
const d = await run({ tier });
|
|
139
148
|
for (const c of d.considered) {
|
|
140
149
|
const q = c.model.quality;
|
|
141
150
|
const unscored = q.coding === undefined && q.agentic === undefined && q.intelligence === undefined;
|
|
@@ -144,15 +153,15 @@ describe("quality floor", () => {
|
|
|
144
153
|
}
|
|
145
154
|
});
|
|
146
155
|
|
|
147
|
-
test("unscored models are eligible in the trivial tier, whose floor is zero", () => {
|
|
148
|
-
const d = run({ tier: "trivial" });
|
|
156
|
+
test("unscored models are eligible in the trivial tier, whose floor is zero", async () => {
|
|
157
|
+
const d = await run({ tier: "trivial" });
|
|
149
158
|
expect(BASE.tiers.trivial.minQuality).toBe(0);
|
|
150
159
|
expect(d.considered.length).toBeGreaterThan(0);
|
|
151
160
|
});
|
|
152
161
|
|
|
153
|
-
test("a higher tier selects a higher-quality model than a lower tier", () => {
|
|
154
|
-
const cheap = run({ tier: "trivial" });
|
|
155
|
-
const dear = run({ tier: "hard" });
|
|
162
|
+
test("a higher tier selects a higher-quality model than a lower tier", async () => {
|
|
163
|
+
const cheap = await run({ tier: "trivial" });
|
|
164
|
+
const dear = await run({ tier: "hard" });
|
|
156
165
|
const cheapModel = MODELS.find((m) => m.slug === cheap.slug);
|
|
157
166
|
const dearModel = MODELS.find((m) => m.slug === dear.slug);
|
|
158
167
|
expect(cheapModel).toBeDefined();
|
|
@@ -162,17 +171,17 @@ describe("quality floor", () => {
|
|
|
162
171
|
});
|
|
163
172
|
|
|
164
173
|
describe("context window", () => {
|
|
165
|
-
test("rejects models whose context cannot hold the prompt", () => {
|
|
174
|
+
test("rejects models whose context cannot hold the prompt", async () => {
|
|
166
175
|
// Far larger than the small-context models in the catalog can take.
|
|
167
|
-
const d = run({ tier: "trivial", promptTokens: 300_000 });
|
|
176
|
+
const d = await run({ tier: "trivial", promptTokens: 300_000 });
|
|
168
177
|
expect(d.rejected.some((r) => r.reason === "context_too_small")).toBe(true);
|
|
169
178
|
const chosen = MODELS.find((m) => m.slug === d.slug);
|
|
170
179
|
expect(chosen).toBeDefined();
|
|
171
180
|
expect(chosen?.contextLength ?? 0).toBeGreaterThan(300_000);
|
|
172
181
|
});
|
|
173
182
|
|
|
174
|
-
test("applies headroom so a token-estimate error cannot overflow the window", () => {
|
|
175
|
-
const d = run({ tier: "trivial", promptTokens: 100_000 });
|
|
183
|
+
test("applies headroom so a token-estimate error cannot overflow the window", async () => {
|
|
184
|
+
const d = await run({ tier: "trivial", promptTokens: 100_000 });
|
|
176
185
|
const chosen = MODELS.find((m) => m.slug === d.slug);
|
|
177
186
|
expect(chosen?.contextLength ?? 0).toBeGreaterThanOrEqual(100_000 * BASE.filters.contextHeadroom);
|
|
178
187
|
});
|
|
@@ -183,9 +192,9 @@ describe("cache-aware switching", () => {
|
|
|
183
192
|
// choice against a better option, or the switch logic is never exercised.
|
|
184
193
|
const warmSlug = "x-ai/grok-4.6";
|
|
185
194
|
|
|
186
|
-
test("keeps the warm model when switching does not clear the margin", () => {
|
|
195
|
+
test("keeps the warm model when switching does not clear the margin", async () => {
|
|
187
196
|
const cfg: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, switchMargin: 1e6 } };
|
|
188
|
-
const d = run({
|
|
197
|
+
const d = await run({
|
|
189
198
|
tier: "hard",
|
|
190
199
|
promptTokens: 80_000,
|
|
191
200
|
cfg,
|
|
@@ -201,8 +210,8 @@ describe("cache-aware switching", () => {
|
|
|
201
210
|
expect(d.sticky).toBe(true);
|
|
202
211
|
});
|
|
203
212
|
|
|
204
|
-
test("abandons a warm cache whose TTL has expired", () => {
|
|
205
|
-
const d = run({
|
|
213
|
+
test("abandons a warm cache whose TTL has expired", async () => {
|
|
214
|
+
const d = await run({
|
|
206
215
|
tier: "hard",
|
|
207
216
|
promptTokens: 80_000,
|
|
208
217
|
st: state({
|
|
@@ -219,7 +228,7 @@ describe("cache-aware switching", () => {
|
|
|
219
228
|
});
|
|
220
229
|
|
|
221
230
|
describe("budget guard", () => {
|
|
222
|
-
test("downgrades when the cold forecast breaches the per-turn cap", () => {
|
|
231
|
+
test("downgrades when the cold forecast breaches the per-turn cap", async () => {
|
|
223
232
|
// A hard-tier turn at this size forecasts ~$0.02 cold, while cheaper
|
|
224
233
|
// tiers land well under a cent, so a $0.005 cap is breachable AND
|
|
225
234
|
// satisfiable further down.
|
|
@@ -227,57 +236,57 @@ describe("budget guard", () => {
|
|
|
227
236
|
...BASE,
|
|
228
237
|
budget: { ...BASE.budget, perTurnUsd: 0.005, onExceeded: "downgrade" },
|
|
229
238
|
};
|
|
230
|
-
const d = run({ tier: "hard", promptTokens: 50_000, cfg });
|
|
239
|
+
const d = await run({ tier: "hard", promptTokens: 50_000, cfg });
|
|
231
240
|
expect(d.budgetDowngraded).toBe(true);
|
|
232
241
|
expect(d.forecast.coldUsd).toBeLessThanOrEqual(0.005);
|
|
233
242
|
});
|
|
234
243
|
|
|
235
|
-
test("throws in downgrade mode when no candidate at any tier fits", () => {
|
|
244
|
+
test("throws in downgrade mode when no candidate at any tier fits", async () => {
|
|
236
245
|
// Failing loudly beats silently spending past an impossible cap.
|
|
237
246
|
const cfg: RouterConfig = {
|
|
238
247
|
...BASE,
|
|
239
248
|
budget: { ...BASE.budget, perTurnUsd: 1e-9, onExceeded: "downgrade" },
|
|
240
249
|
};
|
|
241
|
-
expect(
|
|
250
|
+
await expect(run({ tier: "hard", promptTokens: 50_000, cfg })).rejects.toThrow(BudgetExceededError);
|
|
242
251
|
});
|
|
243
252
|
|
|
244
|
-
test("rejects outright when configured to", () => {
|
|
253
|
+
test("rejects outright when configured to", async () => {
|
|
245
254
|
const cfg: RouterConfig = {
|
|
246
255
|
...BASE,
|
|
247
256
|
budget: { ...BASE.budget, perTurnUsd: 1e-9, onExceeded: "reject" },
|
|
248
257
|
};
|
|
249
|
-
expect(
|
|
258
|
+
await expect(run({ tier: "hard", promptTokens: 50_000, cfg })).rejects.toThrow(BudgetExceededError);
|
|
250
259
|
});
|
|
251
260
|
|
|
252
|
-
test("a satisfiable budget does not downgrade", () => {
|
|
261
|
+
test("a satisfiable budget does not downgrade", async () => {
|
|
253
262
|
const cfg: RouterConfig = { ...BASE, budget: { ...BASE.budget, perTurnUsd: 100, onExceeded: "reject" } };
|
|
254
|
-
const d = run({ tier: "moderate", promptTokens: 5000, cfg });
|
|
263
|
+
const d = await run({ tier: "moderate", promptTokens: 5000, cfg });
|
|
255
264
|
expect(d.budgetDowngraded).toBe(false);
|
|
256
265
|
});
|
|
257
266
|
|
|
258
|
-
test("scopes the daily budget to the requesting harness", () => {
|
|
267
|
+
test("scopes the daily budget to the requesting harness", async () => {
|
|
259
268
|
// Harness A has already spent the whole daily cap; harness B has spent
|
|
260
269
|
// nothing. A request from B must NOT be budget-blocked by A's spend.
|
|
261
270
|
const spendByHarness: Record<string, number> = { "harness-a": 1.0 };
|
|
262
|
-
const ledger:
|
|
263
|
-
record: () => {},
|
|
264
|
-
conversationSpend: () => 0,
|
|
265
|
-
spendSince: (_sinceMs, harnessId) => (harnessId === undefined ? 1.0 : spendByHarness[harnessId] ?? 0),
|
|
266
|
-
blendedRate: () => null,
|
|
267
|
-
latency: () => null,
|
|
268
|
-
trust: () => null,
|
|
269
|
-
allTrust: () => [],
|
|
270
|
-
tokenRatio: () => null,
|
|
271
|
-
recentEntries: () => [],
|
|
272
|
-
};
|
|
271
|
+
const ledger: AsyncLedger = fakeLedger({
|
|
272
|
+
record: async () => {},
|
|
273
|
+
conversationSpend: async () => 0,
|
|
274
|
+
spendSince: async (_sinceMs, harnessId) => (harnessId === undefined ? 1.0 : spendByHarness[harnessId] ?? 0),
|
|
275
|
+
blendedRate: async () => null,
|
|
276
|
+
latency: async () => null,
|
|
277
|
+
trust: async () => null,
|
|
278
|
+
allTrust: async () => [],
|
|
279
|
+
tokenRatio: async () => null,
|
|
280
|
+
recentEntries: async () => [],
|
|
281
|
+
});
|
|
273
282
|
const cfg: RouterConfig = { ...BASE, budget: { ...BASE.budget, perDayUsd: 0.5, onExceeded: "reject" } };
|
|
274
283
|
|
|
275
284
|
// Harness A is over its daily cap → rejected.
|
|
276
|
-
expect(
|
|
285
|
+
await expect(run({ tier: "hard", promptTokens: 50_000, cfg, ledger, harnessId: "harness-a" })).rejects.toThrow(
|
|
277
286
|
BudgetExceededError,
|
|
278
287
|
);
|
|
279
288
|
// Harness B has spent nothing → not blocked by A's spend.
|
|
280
|
-
const d = run({ tier: "hard", promptTokens: 50_000, cfg, ledger, harnessId: "harness-b" });
|
|
289
|
+
const d = await run({ tier: "hard", promptTokens: 50_000, cfg, ledger, harnessId: "harness-b" });
|
|
281
290
|
expect(d.budgetDowngraded).toBe(false);
|
|
282
291
|
});
|
|
283
292
|
});
|
|
@@ -286,68 +295,48 @@ describe("per-harness trust scoping", () => {
|
|
|
286
295
|
// When filters.trustScopedByHarness is on, trust is read from the requesting
|
|
287
296
|
// harness's own ledger rows, so one harness's flaky-model demotion does not
|
|
288
297
|
// leak into another's routing. Off (default), trust is shared.
|
|
289
|
-
|
|
290
|
-
record: () => {},
|
|
291
|
-
conversationSpend: () => 0,
|
|
292
|
-
spendSince: () => 0,
|
|
293
|
-
blendedRate: () => null,
|
|
294
|
-
latency: () => null,
|
|
295
|
-
trust: (_slug, harnessId) => {
|
|
296
|
-
// Harness A has burned the model; harness B has never tried it.
|
|
297
|
-
if (harnessId === "harness-a") {
|
|
298
|
-
return { slug: "x", attempts: 40, escalations: 30, errors: 30, successRate: 0.1, meanCostError: 0.2 };
|
|
299
|
-
}
|
|
300
|
-
return null; // harness B / shared → unmeasured
|
|
301
|
-
},
|
|
302
|
-
allTrust: () => [],
|
|
303
|
-
tokenRatio: () => null,
|
|
304
|
-
recentEntries: () => [],
|
|
305
|
-
});
|
|
306
|
-
|
|
307
|
-
test("scoped trust passes the harness id into the ledger trust query", () => {
|
|
298
|
+
test("scoped trust passes the harness id into the ledger trust query", async () => {
|
|
308
299
|
// The feature's contract is that the router's trust lookup is scoped to
|
|
309
|
-
// the requesting harness when enabled.
|
|
310
|
-
//
|
|
300
|
+
// the requesting harness when enabled. The lookup is the prefetch, so
|
|
301
|
+
// assert the harness id arrives there.
|
|
311
302
|
let queriedWith: string | undefined;
|
|
312
|
-
const ledger:
|
|
313
|
-
|
|
314
|
-
trust: (_slug, harnessId) => {
|
|
303
|
+
const ledger: AsyncLedger = fakeLedger({
|
|
304
|
+
signals: async (slugs, harnessId) => {
|
|
315
305
|
queriedWith = harnessId;
|
|
316
|
-
return null;
|
|
306
|
+
return new Map(slugs.map((slug) => [slug, { trust: null, latency: null }]));
|
|
317
307
|
},
|
|
318
|
-
};
|
|
308
|
+
});
|
|
319
309
|
const cfg: RouterConfig = {
|
|
320
310
|
...BASE,
|
|
321
311
|
filters: { ...BASE.filters, trustScopedByHarness: true },
|
|
322
312
|
};
|
|
323
|
-
run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
|
|
313
|
+
await run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
|
|
324
314
|
expect(queriedWith).toBe("harness-a");
|
|
325
315
|
});
|
|
326
316
|
|
|
327
|
-
test("shared trust (default) reads the whole ledger, not per-harness", () => {
|
|
317
|
+
test("shared trust (default) reads the whole ledger, not per-harness", async () => {
|
|
328
318
|
// With scoping off, the trust lookup must NOT carry the harness id, so
|
|
329
319
|
// harness A's flaky history is visible globally (shared reliability).
|
|
330
320
|
let queriedWith: string | undefined;
|
|
331
|
-
const ledger:
|
|
332
|
-
|
|
333
|
-
trust: (_slug, harnessId) => {
|
|
321
|
+
const ledger: AsyncLedger = fakeLedger({
|
|
322
|
+
signals: async (slugs, harnessId) => {
|
|
334
323
|
queriedWith = harnessId;
|
|
335
|
-
return null;
|
|
324
|
+
return new Map(slugs.map((slug) => [slug, { trust: null, latency: null }]));
|
|
336
325
|
},
|
|
337
|
-
};
|
|
326
|
+
});
|
|
338
327
|
const cfg: RouterConfig = {
|
|
339
328
|
...BASE,
|
|
340
329
|
filters: { ...BASE.filters, trustScopedByHarness: false },
|
|
341
330
|
};
|
|
342
|
-
run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
|
|
331
|
+
await run({ tier: "simple", cfg, ledger, harnessId: "harness-a" });
|
|
343
332
|
// The trust lookup must NOT carry the harness id when scoping is off.
|
|
344
333
|
expect(queriedWith).toBeUndefined();
|
|
345
334
|
});
|
|
346
335
|
});
|
|
347
336
|
|
|
348
337
|
describe("decision shape", () => {
|
|
349
|
-
test("clamps max tokens to the chosen model's published ceiling", () => {
|
|
350
|
-
const d = run({ tier: "moderate" });
|
|
338
|
+
test("clamps max tokens to the chosen model's published ceiling", async () => {
|
|
339
|
+
const d = await run({ tier: "moderate" });
|
|
351
340
|
const chosen = MODELS.find((m) => m.slug === d.slug);
|
|
352
341
|
const ceiling = chosen?.maxCompletionTokens;
|
|
353
342
|
if (ceiling !== undefined && d.maxTokens !== undefined) {
|
|
@@ -355,7 +344,7 @@ describe("decision shape", () => {
|
|
|
355
344
|
}
|
|
356
345
|
});
|
|
357
346
|
|
|
358
|
-
test("a reasoning model gets the completion floor; a direct one keeps the caller's cap", () => {
|
|
347
|
+
test("a reasoning model gets the completion floor; a direct one keeps the caller's cap", async () => {
|
|
359
348
|
const usable = (m: (typeof MODELS)[number]): boolean => m.supportsTools && m.contextLength >= 32_000 && (m.maxCompletionTokens ?? 100_000) >= 4096;
|
|
360
349
|
const thinker = MODELS.find((m) => usable(m) && m.supportsReasoning);
|
|
361
350
|
const direct = MODELS.find((m) => usable(m) && !m.supportsReasoning && !m.reasoningMandatory);
|
|
@@ -365,29 +354,29 @@ describe("decision shape", () => {
|
|
|
365
354
|
|
|
366
355
|
// omp asks for a dozen tokens for a title; a reasoning model would spend them thinking
|
|
367
356
|
// and return nothing, so the dispatch is raised.
|
|
368
|
-
const raised = run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 12 });
|
|
357
|
+
const raised = await run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 12 });
|
|
369
358
|
expect(raised.slug).toBe(thinker!.slug);
|
|
370
359
|
expect(raised.maxTokens).toBe(512);
|
|
371
360
|
expect(raised.reasons.some((r) => r.includes("reasons before it answers"))).toBe(true);
|
|
372
361
|
|
|
373
362
|
// A model that answers directly is untouched: its cap is the caller's.
|
|
374
|
-
const kept = run({ tier: "trivial", cfg: withFloor(direct!.slug, 512), maxTokens: 12 });
|
|
363
|
+
const kept = await run({ tier: "trivial", cfg: withFloor(direct!.slug, 512), maxTokens: 12 });
|
|
375
364
|
expect(kept.slug).toBe(direct!.slug);
|
|
376
365
|
expect(kept.maxTokens).toBe(12);
|
|
377
366
|
|
|
378
367
|
// The floor never raises past what the caller already asked for, and 0 disables it.
|
|
379
|
-
expect(run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 4000 }).maxTokens).toBe(4000);
|
|
380
|
-
expect(run({ tier: "trivial", cfg: withFloor(thinker!.slug, 0), maxTokens: 12 }).maxTokens).toBe(12);
|
|
368
|
+
expect((await run({ tier: "trivial", cfg: withFloor(thinker!.slug, 512), maxTokens: 4000 })).maxTokens).toBe(4000);
|
|
369
|
+
expect((await run({ tier: "trivial", cfg: withFloor(thinker!.slug, 0), maxTokens: 12 })).maxTokens).toBe(12);
|
|
381
370
|
});
|
|
382
371
|
|
|
383
|
-
test("plans a probe for cheap tiers and leaves the top tier unprobed", () => {
|
|
384
|
-
expect(run({ tier: "trivial" }).probe.enabled).toBe(true);
|
|
372
|
+
test("plans a probe for cheap tiers and leaves the top tier unprobed", async () => {
|
|
373
|
+
expect((await run({ tier: "trivial" })).probe.enabled).toBe(true);
|
|
385
374
|
// Nothing above `hard` to escalate into, so probing it would only add latency.
|
|
386
|
-
expect(run({ tier: "hard" }).probe.enabled).toBe(false);
|
|
375
|
+
expect((await run({ tier: "hard" })).probe.enabled).toBe(false);
|
|
387
376
|
});
|
|
388
377
|
|
|
389
|
-
test("carries the session id, features, and a reasoning trail", () => {
|
|
390
|
-
const d = run({ tier: "simple" });
|
|
378
|
+
test("carries the session id, features, and a reasoning trail", async () => {
|
|
379
|
+
const d = await run({ tier: "simple" });
|
|
391
380
|
expect(d.sessionId.startsWith("omp-")).toBe(true);
|
|
392
381
|
// `d.reasons` holds decision-level notes — a widening, a hysteresis hold — and is
|
|
393
382
|
// legitimately empty when a tier serves the turn without incident. The trail that is
|
|
@@ -397,7 +386,7 @@ describe("decision shape", () => {
|
|
|
397
386
|
expect(d.considered.length).toBeGreaterThan(0);
|
|
398
387
|
});
|
|
399
388
|
|
|
400
|
-
test("respects a profile that caps the tier", () => {
|
|
389
|
+
test("respects a profile that caps the tier", async () => {
|
|
401
390
|
const req = request("redesign the whole architecture and explain the race condition root cause");
|
|
402
391
|
const features = extractFeatures(req, 4000);
|
|
403
392
|
const d = select({
|
|
@@ -407,7 +396,6 @@ describe("decision shape", () => {
|
|
|
407
396
|
profile: { ...PROFILE, id: "auto-cheap", maxTier: "simple" },
|
|
408
397
|
state: state(),
|
|
409
398
|
snapshot: SNAPSHOT,
|
|
410
|
-
ledger: null,
|
|
411
399
|
cfg: BASE,
|
|
412
400
|
nowMs: Date.now(),
|
|
413
401
|
});
|
|
@@ -431,10 +419,11 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
|
|
|
431
419
|
keyScoped: true,
|
|
432
420
|
};
|
|
433
421
|
|
|
434
|
-
function runConstrained(ledger:
|
|
422
|
+
async function runConstrained(ledger: AsyncLedger | null = null) {
|
|
435
423
|
const req = request("refactor the service layer and explain the cache coherence contract");
|
|
436
424
|
const features = extractFeatures(req, 4000);
|
|
437
425
|
const heuristic = scoreHeuristic(features, BASE);
|
|
426
|
+
const reads = await prefetchTurnReads(ledger, req, PROFILE, BASE, constrained, heuristic.task);
|
|
438
427
|
return select({
|
|
439
428
|
req,
|
|
440
429
|
features,
|
|
@@ -442,47 +431,35 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
|
|
|
442
431
|
profile: PROFILE,
|
|
443
432
|
state: state(),
|
|
444
433
|
snapshot: constrained,
|
|
445
|
-
|
|
434
|
+
reads,
|
|
446
435
|
cfg: BASE,
|
|
447
436
|
nowMs: Date.now(),
|
|
448
437
|
});
|
|
449
438
|
}
|
|
450
439
|
|
|
451
440
|
/** Every model is probed-and-failed: below the trust floor at every tier. */
|
|
452
|
-
function untrustedLedger():
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
trust: (slug) => ({
|
|
460
|
-
slug,
|
|
461
|
-
attempts: 40,
|
|
462
|
-
escalations: 30,
|
|
463
|
-
errors: 30,
|
|
464
|
-
successRate: 0.1,
|
|
465
|
-
meanCostError: 0.2,
|
|
466
|
-
}),
|
|
467
|
-
allTrust: () => [],
|
|
468
|
-
tokenRatio: () => null,
|
|
469
|
-
recentEntries: () => [],
|
|
470
|
-
};
|
|
441
|
+
function untrustedLedger(): AsyncLedger {
|
|
442
|
+
const burned = (slug: string) => ({ slug, attempts: 40, escalations: 30, errors: 30, successRate: 0.1, meanCostError: 0.2 });
|
|
443
|
+
return fakeLedger({
|
|
444
|
+
trust: async (slug) => burned(slug),
|
|
445
|
+
// Candidate scoring reads `signals`, which is what the prefetch fills.
|
|
446
|
+
signals: async (slugs) => new Map(slugs.map((slug) => [slug, { trust: burned(slug), latency: null }])),
|
|
447
|
+
});
|
|
471
448
|
}
|
|
472
449
|
|
|
473
|
-
test("rescues a model instead of throwing when no strict tier admits the catalog", () => {
|
|
474
|
-
const d = runConstrained(untrustedLedger());
|
|
450
|
+
test("rescues a model instead of throwing when no strict tier admits the catalog", async () => {
|
|
451
|
+
const d = await runConstrained(untrustedLedger());
|
|
475
452
|
// It must pick one of the available models, not throw `catalog exhausted`.
|
|
476
453
|
expect(constrained.models.some((m) => m.slug === d.slug)).toBe(true);
|
|
477
454
|
});
|
|
478
455
|
|
|
479
|
-
test("records the rescue in the reasoning trail", () => {
|
|
480
|
-
const d = runConstrained(untrustedLedger());
|
|
456
|
+
test("records the rescue in the reasoning trail", async () => {
|
|
457
|
+
const d = await runConstrained(untrustedLedger());
|
|
481
458
|
expect(d.reasons.some((r) => r.startsWith("tier rescue:"))).toBe(true);
|
|
482
459
|
});
|
|
483
460
|
|
|
484
|
-
test("the rescue chooses the cheapest available model when quality is secondary", () => {
|
|
485
|
-
const d = runConstrained(untrustedLedger());
|
|
461
|
+
test("the rescue chooses the cheapest available model when quality is secondary", async () => {
|
|
462
|
+
const d = await runConstrained(untrustedLedger());
|
|
486
463
|
const chosen = MODELS.find((m) => m.slug === d.slug);
|
|
487
464
|
expect(chosen).toBeDefined();
|
|
488
465
|
// Price ceilings are relaxed first; the cheapest surviving model wins.
|
|
@@ -490,17 +467,17 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
|
|
|
490
467
|
expect(d.slug).toBe(cheapest.slug);
|
|
491
468
|
});
|
|
492
469
|
|
|
493
|
-
test("a guardrail that leaves every model below the trust bar is rescued by relaxing it", () => {
|
|
470
|
+
test("a guardrail that leaves every model below the trust bar is rescued by relaxing it", async () => {
|
|
494
471
|
// Reproduces the real failure: a tiny guardrail catalog whose models are
|
|
495
472
|
// all marked untrusted (probed and failed). The trust floor (minTrust 0.7
|
|
496
473
|
// over minTrustSamples 12) excludes them at EVERY tier, so strict widening
|
|
497
474
|
// finds nothing; the rescue relaxes trust and picks a model.
|
|
498
|
-
const d = runConstrained(untrustedLedger());
|
|
475
|
+
const d = await runConstrained(untrustedLedger());
|
|
499
476
|
expect(constrained.models.some((m) => m.slug === d.slug)).toBe(true);
|
|
500
477
|
expect(d.reasons.some((r) => r.startsWith("tier rescue:"))).toBe(true);
|
|
501
478
|
});
|
|
502
479
|
|
|
503
|
-
test("still throws when the catalog is empty after relaxing all economic constraints", () => {
|
|
480
|
+
test("still throws when the catalog is empty after relaxing all economic constraints", async () => {
|
|
504
481
|
const empty: CatalogSnapshot = { models: [], fetchedAtMs: Date.now(), keyScoped: true };
|
|
505
482
|
const req = request("anything");
|
|
506
483
|
const features = extractFeatures(req, 4000);
|
|
@@ -513,7 +490,6 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
|
|
|
513
490
|
profile: PROFILE,
|
|
514
491
|
state: state(),
|
|
515
492
|
snapshot: empty,
|
|
516
|
-
ledger: null,
|
|
517
493
|
cfg: BASE,
|
|
518
494
|
nowMs: Date.now(),
|
|
519
495
|
}),
|
|
@@ -522,7 +498,7 @@ describe("tier rescue under a guardrail-constrained catalog", () => {
|
|
|
522
498
|
});
|
|
523
499
|
|
|
524
500
|
describe("task-type routing", () => {
|
|
525
|
-
test("a vision task only considers image-capable models", () => {
|
|
501
|
+
test("a vision task only considers image-capable models", async () => {
|
|
526
502
|
// Force the vision task and a tier; every considered candidate must
|
|
527
503
|
// support image input.
|
|
528
504
|
const req = request("describe this image");
|
|
@@ -535,7 +511,6 @@ describe("task-type routing", () => {
|
|
|
535
511
|
profile: PROFILE,
|
|
536
512
|
state: state(),
|
|
537
513
|
snapshot: SNAPSHOT,
|
|
538
|
-
ledger: null,
|
|
539
514
|
cfg: BASE,
|
|
540
515
|
nowMs: Date.now(),
|
|
541
516
|
});
|
|
@@ -543,7 +518,7 @@ describe("task-type routing", () => {
|
|
|
543
518
|
for (const c of d.considered) expect(c.model.inputModalities.includes("image")).toBe(true);
|
|
544
519
|
});
|
|
545
520
|
|
|
546
|
-
test("the task config's quality floor overrides the tier floor when higher", () => {
|
|
521
|
+
test("the task config's quality floor overrides the tier floor when higher", async () => {
|
|
547
522
|
// A coding task with a high minQuality must not admit models below it,
|
|
548
523
|
// even in a tier whose own floor is lower.
|
|
549
524
|
const cfg: RouterConfig = {
|
|
@@ -560,7 +535,6 @@ describe("task-type routing", () => {
|
|
|
560
535
|
profile: PROFILE,
|
|
561
536
|
state: state(),
|
|
562
537
|
snapshot: SNAPSHOT,
|
|
563
|
-
ledger: null,
|
|
564
538
|
cfg,
|
|
565
539
|
nowMs: Date.now(),
|
|
566
540
|
});
|
|
@@ -576,23 +550,18 @@ describe("task-type routing", () => {
|
|
|
576
550
|
});
|
|
577
551
|
|
|
578
552
|
describe("latency scoring", () => {
|
|
579
|
-
function ledgerWithLatency(bySlug: Record<string, { ttftMs: number; samples: number; tokensPerSec?: number }>):
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
trust: () => null,
|
|
586
|
-
allTrust: () => [],
|
|
587
|
-
latency: (slug) => {
|
|
588
|
-
const v = bySlug[slug];
|
|
589
|
-
// Default throughput is fast, so these cases isolate the TTFT axis
|
|
590
|
-
// unless a test sets tokensPerSec explicitly.
|
|
591
|
-
return v === undefined ? null : { slug, samples: v.samples, ttftMs: v.ttftMs, tokensPerSec: v.tokensPerSec ?? 1000 };
|
|
592
|
-
},
|
|
593
|
-
tokenRatio: () => null,
|
|
594
|
-
recentEntries: () => [],
|
|
553
|
+
function ledgerWithLatency(bySlug: Record<string, { ttftMs: number; samples: number; tokensPerSec?: number }>): AsyncLedger {
|
|
554
|
+
const latencyOf = (slug: string): ModelLatency | null => {
|
|
555
|
+
const v = bySlug[slug];
|
|
556
|
+
// Default throughput is fast, so these cases isolate the TTFT axis
|
|
557
|
+
// unless a test sets tokensPerSec explicitly.
|
|
558
|
+
return v === undefined ? null : { slug, samples: v.samples, ttftMs: v.ttftMs, tokensPerSec: v.tokensPerSec ?? 1000 };
|
|
595
559
|
};
|
|
560
|
+
return fakeLedger({
|
|
561
|
+
latency: async (slug) => latencyOf(slug),
|
|
562
|
+
// Scoring reads latency out of the prefetched signals.
|
|
563
|
+
signals: async (slugs) => new Map(slugs.map((slug) => [slug, { trust: null, latency: latencyOf(slug) }])),
|
|
564
|
+
});
|
|
596
565
|
}
|
|
597
566
|
|
|
598
567
|
const withWeight = (latencyWeight: number): RouterConfig => ({
|
|
@@ -600,30 +569,30 @@ describe("latency scoring", () => {
|
|
|
600
569
|
filters: { ...BASE.filters, latencyWeight, latencyReferenceMs: 5000, latencyMinSamples: 20 },
|
|
601
570
|
});
|
|
602
571
|
|
|
603
|
-
test("penalises a chronically slow model out of the top slot", () => {
|
|
604
|
-
const slow = run({ tier: "simple" }).slug;
|
|
572
|
+
test("penalises a chronically slow model out of the top slot", async () => {
|
|
573
|
+
const slow = (await run({ tier: "simple" })).slug;
|
|
605
574
|
const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
|
|
606
|
-
const d = run({ tier: "simple", cfg: withWeight(2), ledger });
|
|
575
|
+
const d = await run({ tier: "simple", cfg: withWeight(2), ledger });
|
|
607
576
|
expect(d.slug).not.toBe(slow);
|
|
608
577
|
});
|
|
609
578
|
|
|
610
|
-
test("latencyWeight 0 disables the penalty", () => {
|
|
611
|
-
const slow = run({ tier: "simple" }).slug;
|
|
579
|
+
test("latencyWeight 0 disables the penalty", async () => {
|
|
580
|
+
const slow = (await run({ tier: "simple" })).slug;
|
|
612
581
|
const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
|
|
613
|
-
expect(run({ tier: "simple", cfg: withWeight(0), ledger }).slug).toBe(slow);
|
|
582
|
+
expect((await run({ tier: "simple", cfg: withWeight(0), ledger })).slug).toBe(slow);
|
|
614
583
|
});
|
|
615
584
|
|
|
616
|
-
test("a model with too few samples is not penalised", () => {
|
|
617
|
-
const slow = run({ tier: "simple" }).slug;
|
|
585
|
+
test("a model with too few samples is not penalised", async () => {
|
|
586
|
+
const slow = (await run({ tier: "simple" })).slug;
|
|
618
587
|
const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 5 } });
|
|
619
|
-
expect(run({ tier: "simple", cfg: withWeight(2), ledger }).slug).toBe(slow);
|
|
588
|
+
expect((await run({ tier: "simple", cfg: withWeight(2), ledger })).slug).toBe(slow);
|
|
620
589
|
});
|
|
621
590
|
|
|
622
|
-
test("penalises a model that starts fast but streams slowly", () => {
|
|
591
|
+
test("penalises a model that starts fast but streams slowly", async () => {
|
|
623
592
|
// The case TTFT-only scoring misses: quick first token, slow body.
|
|
624
|
-
const slow = run({ tier: "simple" }).slug;
|
|
593
|
+
const slow = (await run({ tier: "simple" })).slug;
|
|
625
594
|
const ledger = ledgerWithLatency({ [slow]: { ttftMs: 1500, samples: 50, tokensPerSec: 12 } });
|
|
626
|
-
const d = run({ tier: "simple", cfg: withWeight(2), ledger });
|
|
595
|
+
const d = await run({ tier: "simple", cfg: withWeight(2), ledger });
|
|
627
596
|
expect(d.slug).not.toBe(slow);
|
|
628
597
|
});
|
|
629
598
|
|
|
@@ -632,24 +601,24 @@ describe("latency scoring", () => {
|
|
|
632
601
|
filters: { ...BASE.filters, latencyWeight, latencyReferenceMs: 5000, latencyMinSamples: 20, maxExpectedWaitMs },
|
|
633
602
|
});
|
|
634
603
|
|
|
635
|
-
test("ceiling hard-drops a proven-slow model the penalty cannot, even at weight 0", () => {
|
|
636
|
-
const slow = run({ tier: "simple" }).slug;
|
|
604
|
+
test("ceiling hard-drops a proven-slow model the penalty cannot, even at weight 0", async () => {
|
|
605
|
+
const slow = (await run({ tier: "simple" })).slug;
|
|
637
606
|
const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
|
|
638
607
|
// latencyWeight 0 → the multiplier is inert; only the hard ceiling can act.
|
|
639
|
-
const d = run({ tier: "simple", cfg: withCeiling(20_000), ledger });
|
|
608
|
+
const d = await run({ tier: "simple", cfg: withCeiling(20_000), ledger });
|
|
640
609
|
expect(d.slug).not.toBe(slow);
|
|
641
610
|
});
|
|
642
611
|
|
|
643
|
-
test("ceiling spares an under-sampled slow model (cold-start grace)", () => {
|
|
644
|
-
const slow = run({ tier: "simple" }).slug;
|
|
612
|
+
test("ceiling spares an under-sampled slow model (cold-start grace)", async () => {
|
|
613
|
+
const slow = (await run({ tier: "simple" })).slug;
|
|
645
614
|
const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 5 } });
|
|
646
|
-
expect(run({ tier: "simple", cfg: withCeiling(20_000), ledger }).slug).toBe(slow);
|
|
615
|
+
expect((await run({ tier: "simple", cfg: withCeiling(20_000), ledger })).slug).toBe(slow);
|
|
647
616
|
});
|
|
648
617
|
|
|
649
|
-
test("ceiling unset ⇒ no latency gate (proven-slow model still wins on price)", () => {
|
|
650
|
-
const slow = run({ tier: "simple" }).slug;
|
|
618
|
+
test("ceiling unset ⇒ no latency gate (proven-slow model still wins on price)", async () => {
|
|
619
|
+
const slow = (await run({ tier: "simple" })).slug;
|
|
651
620
|
const ledger = ledgerWithLatency({ [slow]: { ttftMs: 60_000, samples: 50 } });
|
|
652
|
-
expect(run({ tier: "simple", cfg: withWeight(0), ledger }).slug).toBe(slow);
|
|
621
|
+
expect((await run({ tier: "simple", cfg: withWeight(0), ledger })).slug).toBe(slow);
|
|
653
622
|
});
|
|
654
623
|
});
|
|
655
624
|
|
|
@@ -688,7 +657,7 @@ describe("context compaction", () => {
|
|
|
688
657
|
);
|
|
689
658
|
}
|
|
690
659
|
|
|
691
|
-
test("an over-budget turn produces a compaction plan and records savings", () => {
|
|
660
|
+
test("an over-budget turn produces a compaction plan and records savings", async () => {
|
|
692
661
|
const req = loopReq();
|
|
693
662
|
const features = extractFeatures(req, 5_000); // over budgetTokens=1000
|
|
694
663
|
const d = select({
|
|
@@ -698,7 +667,6 @@ describe("context compaction", () => {
|
|
|
698
667
|
profile: PROFILE,
|
|
699
668
|
state: state(),
|
|
700
669
|
snapshot: SNAPSHOT,
|
|
701
|
-
ledger: null,
|
|
702
670
|
cfg: COMPACT_CFG,
|
|
703
671
|
nowMs: Date.now(),
|
|
704
672
|
});
|
|
@@ -707,7 +675,7 @@ describe("context compaction", () => {
|
|
|
707
675
|
expect(d.reasons.some((r) => r.startsWith("compaction:"))).toBe(true);
|
|
708
676
|
});
|
|
709
677
|
|
|
710
|
-
test("a small turn is left untouched", () => {
|
|
678
|
+
test("a small turn is left untouched", async () => {
|
|
711
679
|
const req = loopReq();
|
|
712
680
|
const features = extractFeatures(req, 500); // under budgetTokens=1000
|
|
713
681
|
const d = select({
|
|
@@ -717,7 +685,6 @@ describe("context compaction", () => {
|
|
|
717
685
|
profile: PROFILE,
|
|
718
686
|
state: state(),
|
|
719
687
|
snapshot: SNAPSHOT,
|
|
720
|
-
ledger: null,
|
|
721
688
|
cfg: COMPACT_CFG,
|
|
722
689
|
nowMs: Date.now(),
|
|
723
690
|
});
|
|
@@ -725,7 +692,7 @@ describe("context compaction", () => {
|
|
|
725
692
|
expect(d.promptTokensSaved).toBe(0);
|
|
726
693
|
});
|
|
727
694
|
|
|
728
|
-
test("a carried plan is re-applied even when the turn is now under budget", () => {
|
|
695
|
+
test("a carried plan is re-applied even when the turn is now under budget", async () => {
|
|
729
696
|
// The prompt cache is a byte-prefix cache: dropping an edit that was
|
|
730
697
|
// already dispatched rewrites history the upstream had cached, and
|
|
731
698
|
// re-sends the tokens the edit saved. So a carried plan survives a turn
|
|
@@ -739,7 +706,6 @@ describe("context compaction", () => {
|
|
|
739
706
|
profile: PROFILE,
|
|
740
707
|
state: state(),
|
|
741
708
|
snapshot: SNAPSHOT,
|
|
742
|
-
ledger: null,
|
|
743
709
|
cfg: COMPACT_CFG,
|
|
744
710
|
nowMs: Date.now(),
|
|
745
711
|
});
|
|
@@ -753,7 +719,6 @@ describe("context compaction", () => {
|
|
|
753
719
|
profile: PROFILE,
|
|
754
720
|
state: state({ compactionPlan: first.compactionPlan }),
|
|
755
721
|
snapshot: SNAPSHOT,
|
|
756
|
-
ledger: null,
|
|
757
722
|
cfg: COMPACT_CFG,
|
|
758
723
|
nowMs: Date.now(),
|
|
759
724
|
});
|
|
@@ -761,7 +726,7 @@ describe("context compaction", () => {
|
|
|
761
726
|
expect(second.promptTokensSaved).toBeGreaterThan(0);
|
|
762
727
|
});
|
|
763
728
|
|
|
764
|
-
test("floorRatio below 1 compacts strictly past the budget so the plan holds longer", () => {
|
|
729
|
+
test("floorRatio below 1 compacts strictly past the budget so the plan holds longer", async () => {
|
|
765
730
|
// Each plan change rewrites cached prompt bytes, so compaction overshoots
|
|
766
731
|
// deliberately: eliding more now buys byte-stable turns later.
|
|
767
732
|
const req = parseChatRequest(
|
|
@@ -791,7 +756,6 @@ describe("context compaction", () => {
|
|
|
791
756
|
profile: PROFILE,
|
|
792
757
|
state: state(),
|
|
793
758
|
snapshot: SNAPSHOT,
|
|
794
|
-
ledger: null,
|
|
795
759
|
cfg: { ...COMPACT_CFG, compaction: { ...COMPACT_CFG.compaction, floorRatio } },
|
|
796
760
|
nowMs: Date.now(),
|
|
797
761
|
});
|
|
@@ -830,10 +794,10 @@ describe("hysteresis.breakHoldOnMechanical", () => {
|
|
|
830
794
|
function decide(req: NormRequest, breakHold: boolean) {
|
|
831
795
|
const cfg: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, breakHoldOnMechanical: breakHold } };
|
|
832
796
|
const features = extractFeatures(req, 4_000);
|
|
833
|
-
return { d: select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: held, snapshot: SNAPSHOT,
|
|
797
|
+
return { d: select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: held, snapshot: SNAPSHOT, cfg, nowMs: Date.now() }), features };
|
|
834
798
|
}
|
|
835
799
|
|
|
836
|
-
test("off by default, so a hold still pins the tier", () => {
|
|
800
|
+
test("off by default, so a hold still pins the tier", async () => {
|
|
837
801
|
expect(DEFAULT_CONFIG.hysteresis.breakHoldOnMechanical).toBe(false); // SHIPPED default, not the live config.yml (machine-dependent)
|
|
838
802
|
const { d, features } = decide(continuation(), false);
|
|
839
803
|
expect(features.isToolResultContinuation).toBe(true);
|
|
@@ -841,14 +805,14 @@ describe("hysteresis.breakHoldOnMechanical", () => {
|
|
|
841
805
|
expect(d.classification.source).toBe("sticky");
|
|
842
806
|
});
|
|
843
807
|
|
|
844
|
-
test("on, a mechanical continuation escapes the hold", () => {
|
|
808
|
+
test("on, a mechanical continuation escapes the hold", async () => {
|
|
845
809
|
const { d } = decide(continuation(), true);
|
|
846
810
|
expect(d.tier).not.toBe("hard");
|
|
847
811
|
expect(d.classification.source).not.toBe("sticky");
|
|
848
812
|
expect(d.reasons.some((r) => /hold hard broken/.test(r))).toBe(true);
|
|
849
813
|
});
|
|
850
814
|
|
|
851
|
-
test("a NON-mechanical turn still gets the hold, so flap protection survives", () => {
|
|
815
|
+
test("a NON-mechanical turn still gets the hold, so flap protection survives", async () => {
|
|
852
816
|
// This is the case hysteresis exists for: fresh user work mid-conversation
|
|
853
817
|
// must not bounce the model and cold-start its cache.
|
|
854
818
|
const { d, features } = decide(request("now refactor the retry helper"), true);
|
|
@@ -857,7 +821,7 @@ describe("hysteresis.breakHoldOnMechanical", () => {
|
|
|
857
821
|
expect(d.classification.source).toBe("sticky");
|
|
858
822
|
});
|
|
859
823
|
|
|
860
|
-
test("the downgrade clamp still applies, so quality steps rather than falls", () => {
|
|
824
|
+
test("the downgrade clamp still applies, so quality steps rather than falls", async () => {
|
|
861
825
|
const cfg: RouterConfig = {
|
|
862
826
|
...BASE,
|
|
863
827
|
hysteresis: { ...BASE.hysteresis, breakHoldOnMechanical: true, maxDowngradePerTurn: 1 },
|
|
@@ -872,7 +836,6 @@ describe("hysteresis.breakHoldOnMechanical", () => {
|
|
|
872
836
|
profile: PROFILE,
|
|
873
837
|
state: held,
|
|
874
838
|
snapshot: SNAPSHOT,
|
|
875
|
-
ledger: null,
|
|
876
839
|
cfg,
|
|
877
840
|
nowMs: Date.now(),
|
|
878
841
|
});
|
|
@@ -928,25 +891,24 @@ describe("hysteresis.switchHorizonTurns (review 2026-09-05 §4)", () => {
|
|
|
928
891
|
profile: PROFILE,
|
|
929
892
|
state: state({ currentSlug: "test/warm-dear", currentTier: "moderate", cacheWarmSlug: "test/warm-dear", cacheWarmAtMs: Date.now(), lastPromptTokens: promptTokens }),
|
|
930
893
|
snapshot: snap,
|
|
931
|
-
ledger: null,
|
|
932
894
|
cfg,
|
|
933
895
|
nowMs: Date.now(),
|
|
934
896
|
});
|
|
935
897
|
}
|
|
936
898
|
|
|
937
|
-
test("the ranked winner is the cheaper cold model", () => {
|
|
899
|
+
test("the ranked winner is the cheaper cold model", async () => {
|
|
938
900
|
const d = decide(1);
|
|
939
901
|
expect(d.considered[0]!.model.slug).toBe("test/winner");
|
|
940
902
|
});
|
|
941
903
|
|
|
942
|
-
test("a one-turn horizon keeps the dear model warm (the shipped behaviour)", () => {
|
|
904
|
+
test("a one-turn horizon keeps the dear model warm (the shipped behaviour)", async () => {
|
|
943
905
|
const d = decide(1);
|
|
944
906
|
expect(d.slug).toBe("test/warm-dear");
|
|
945
907
|
expect(d.sticky).toBe(true);
|
|
946
908
|
expect(d.reasons.some((r) => r.startsWith("cache: keeping warm test/warm-dear"))).toBe(true);
|
|
947
909
|
});
|
|
948
910
|
|
|
949
|
-
test("amortised over a run of turns, the switch is taken", () => {
|
|
911
|
+
test("amortised over a run of turns, the switch is taken", async () => {
|
|
950
912
|
const d = decide(8);
|
|
951
913
|
expect(d.slug).toBe("test/winner");
|
|
952
914
|
expect(d.sticky).toBe(false);
|
|
@@ -990,17 +952,17 @@ describe("compaction.replanGrowthRatio (review 2026-09-05 §7)", () => {
|
|
|
990
952
|
}
|
|
991
953
|
function decide(req: NormRequest, cfg: RouterConfig, st: ConversationState, promptTokens: number) {
|
|
992
954
|
const features = extractFeatures(req, promptTokens);
|
|
993
|
-
return select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: st, snapshot: SNAPSHOT,
|
|
955
|
+
return select({ req, features, classification: scoreHeuristic(features, cfg), profile: PROFILE, state: st, snapshot: SNAPSHOT, cfg, nowMs: Date.now() });
|
|
994
956
|
}
|
|
995
957
|
|
|
996
|
-
test("the plan records the compacted size it was made at", () => {
|
|
958
|
+
test("the plan records the compacted size it was made at", async () => {
|
|
997
959
|
const first = decide(turn(2), cfgWith(1), state(), 4_000);
|
|
998
960
|
expect(first.compactionPlan.length).toBe(1);
|
|
999
961
|
expect(first.compactionPlanTokens).toBe(4_000 - first.promptTokensSaved);
|
|
1000
962
|
expect(first.compactionSavedBytes).toBeGreaterThan(0);
|
|
1001
963
|
});
|
|
1002
964
|
|
|
1003
|
-
test("at 1 (shipped) a newly eligible result is compacted on the very next turn", () => {
|
|
965
|
+
test("at 1 (shipped) a newly eligible result is compacted on the very next turn", async () => {
|
|
1004
966
|
const first = decide(turn(2), cfgWith(1), state(), 4_000);
|
|
1005
967
|
const carried = state({ compactionPlan: first.compactionPlan, compactionPlanTokens: first.compactionPlanTokens });
|
|
1006
968
|
// One more round: the prompt grew ~25%, one more result aged out.
|
|
@@ -1009,7 +971,7 @@ describe("compaction.replanGrowthRatio (review 2026-09-05 §7)", () => {
|
|
|
1009
971
|
expect(second.reasons.some((r) => r.includes("(1 carried, 1 new)"))).toBe(true);
|
|
1010
972
|
});
|
|
1011
973
|
|
|
1012
|
-
test("above 1, an existing plan holds until the compacted prompt has grown by the ratio", () => {
|
|
974
|
+
test("above 1, an existing plan holds until the compacted prompt has grown by the ratio", async () => {
|
|
1013
975
|
const first = decide(turn(2), cfgWith(2), state(), 4_000);
|
|
1014
976
|
const carried = state({ compactionPlan: first.compactionPlan, compactionPlanTokens: first.compactionPlanTokens });
|
|
1015
977
|
// The raw prompt grew 25% and the COMPACTED prompt ~60% (the carried
|
|
@@ -1032,9 +994,11 @@ describe("hysteresis.confirmUpgradesBelowConfidence", () => {
|
|
|
1032
994
|
// A low-confidence heuristic upgrade from a warm model waits one turn.
|
|
1033
995
|
// Measured: 65 of 67 moderate→hard upgrades in a week bounced back within
|
|
1034
996
|
// 3 turns, each paying a cold hard-tier read of a ~120k prompt.
|
|
1035
|
-
|
|
1036
|
-
|
|
997
|
+
// Resolved per test: a describe body cannot await.
|
|
998
|
+
const warmSlugOf = async (): Promise<string> => (await run({ tier: "moderate" })).slug;
|
|
999
|
+
async function upgrade(opts: { confidence?: number; source?: "heuristic" | "escalation"; st?: Partial<ConversationState>; cfg?: RouterConfig; lastToolFailed?: boolean }) {
|
|
1037
1000
|
const cfg = opts.cfg ?? BASE;
|
|
1001
|
+
const warmSlug = await warmSlugOf();
|
|
1038
1002
|
const req = request("now rework the whole scheduler");
|
|
1039
1003
|
const base = extractFeatures(req, 120_000);
|
|
1040
1004
|
const features = opts.lastToolFailed === true ? { ...base, lastToolFailed: true } : base;
|
|
@@ -1046,47 +1010,47 @@ describe("hysteresis.confirmUpgradesBelowConfidence", () => {
|
|
|
1046
1010
|
profile: PROFILE,
|
|
1047
1011
|
state: state({ turn: 4, currentTier: "moderate", currentSlug: warmSlug, cacheWarmSlug: warmSlug, cacheWarmAtMs: Date.now(), lastPromptTokens: 110_000, ...opts.st }),
|
|
1048
1012
|
snapshot: SNAPSHOT,
|
|
1049
|
-
ledger: null,
|
|
1050
1013
|
cfg,
|
|
1051
1014
|
nowMs: Date.now(),
|
|
1052
1015
|
});
|
|
1053
1016
|
}
|
|
1054
1017
|
|
|
1055
|
-
test("a low-confidence upgrade from a warm model is deferred to the held tier", () => {
|
|
1056
|
-
const d = upgrade({});
|
|
1018
|
+
test("a low-confidence upgrade from a warm model is deferred to the held tier", async () => {
|
|
1019
|
+
const d = await upgrade({});
|
|
1057
1020
|
expect(d.tier).toBe("moderate");
|
|
1058
1021
|
expect(d.upgradeDeferred).toBe("hard");
|
|
1059
1022
|
expect(d.reasons.some((r) => r.includes("upgrade moderate → hard deferred one turn"))).toBe(true);
|
|
1060
1023
|
});
|
|
1061
1024
|
|
|
1062
|
-
test("a second consecutive upgrade classification confirms it", () => {
|
|
1063
|
-
const d = upgrade({ st: { upgradeDeferredTier: "hard" } });
|
|
1025
|
+
test("a second consecutive upgrade classification confirms it", async () => {
|
|
1026
|
+
const d = await upgrade({ st: { upgradeDeferredTier: "hard" } });
|
|
1064
1027
|
expect(d.tier).toBe("hard");
|
|
1065
1028
|
expect(d.upgradeDeferred).toBeNull();
|
|
1066
1029
|
expect(d.reasons.some((r) => r.includes("upgrade moderate → hard confirmed"))).toBe(true);
|
|
1067
1030
|
});
|
|
1068
1031
|
|
|
1069
|
-
test("confident classifications, cold caches, escalations, failing tools and the off switch all upgrade at once", () => {
|
|
1070
|
-
expect(upgrade({ confidence: 0.9 }).tier).toBe("hard");
|
|
1071
|
-
expect(upgrade({ st: { cacheWarmAtMs: Date.now() - 3_600_000 } }).tier).toBe("hard");
|
|
1072
|
-
expect(upgrade({ source: "escalation" }).tier).toBe("hard");
|
|
1073
|
-
expect(upgrade({ lastToolFailed: true }).tier).toBe("hard");
|
|
1032
|
+
test("confident classifications, cold caches, escalations, failing tools and the off switch all upgrade at once", async () => {
|
|
1033
|
+
expect((await upgrade({ confidence: 0.9 })).tier).toBe("hard");
|
|
1034
|
+
expect((await upgrade({ st: { cacheWarmAtMs: Date.now() - 3_600_000 } })).tier).toBe("hard");
|
|
1035
|
+
expect((await upgrade({ source: "escalation" })).tier).toBe("hard");
|
|
1036
|
+
expect((await upgrade({ lastToolFailed: true })).tier).toBe("hard");
|
|
1074
1037
|
const off: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, confirmUpgradesBelowConfidence: 0 } };
|
|
1075
|
-
expect(upgrade({ cfg: off }).tier).toBe("hard");
|
|
1076
|
-
for (const d of [upgrade({ confidence: 0.9 }), upgrade({ source: "escalation" })]) expect(d.upgradeDeferred).toBeNull();
|
|
1038
|
+
expect((await upgrade({ cfg: off })).tier).toBe("hard");
|
|
1039
|
+
for (const d of [await upgrade({ confidence: 0.9 }), await upgrade({ source: "escalation" })]) expect(d.upgradeDeferred).toBeNull();
|
|
1077
1040
|
});
|
|
1078
1041
|
|
|
1079
|
-
test("a downgrade or a same-tier turn is never deferred", () => {
|
|
1080
|
-
const
|
|
1042
|
+
test("a downgrade or a same-tier turn is never deferred", async () => {
|
|
1043
|
+
const warmSlug = await warmSlugOf();
|
|
1044
|
+
const d = await run({ tier: "simple", st: state({ turn: 4, currentTier: "moderate", currentSlug: warmSlug, cacheWarmSlug: warmSlug, cacheWarmAtMs: Date.now() }) });
|
|
1081
1045
|
expect(d.upgradeDeferred).toBeNull();
|
|
1082
1046
|
});
|
|
1083
1047
|
});
|
|
1084
1048
|
|
|
1085
1049
|
describe("recorded forecast is the expected price, not the cold worst case", () => {
|
|
1086
1050
|
const warmSlug = "x-ai/grok-4.6";
|
|
1087
|
-
test("a warm stay prices the previous prompt as cache reads; coldUsd keeps the cold figure", () => {
|
|
1051
|
+
test("a warm stay prices the previous prompt as cache reads; coldUsd keeps the cold figure", async () => {
|
|
1088
1052
|
const cfg: RouterConfig = { ...BASE, hysteresis: { ...BASE.hysteresis, switchMargin: 1e6 } };
|
|
1089
|
-
const d = run({
|
|
1053
|
+
const d = await run({
|
|
1090
1054
|
tier: "hard",
|
|
1091
1055
|
promptTokens: 80_000,
|
|
1092
1056
|
cfg,
|
|
@@ -1099,8 +1063,8 @@ describe("recorded forecast is the expected price, not the cold worst case", ()
|
|
|
1099
1063
|
expect(d.forecast.breakdown.cacheRead).toBeGreaterThan(0);
|
|
1100
1064
|
});
|
|
1101
1065
|
|
|
1102
|
-
test("a cold turn records the cold price", () => {
|
|
1103
|
-
const d = run({ tier: "hard", promptTokens: 80_000 });
|
|
1066
|
+
test("a cold turn records the cold price", async () => {
|
|
1067
|
+
const d = await run({ tier: "hard", promptTokens: 80_000 });
|
|
1104
1068
|
expect(d.forecast.assumedCacheHitRate).toBe(0);
|
|
1105
1069
|
expect(d.forecast.expectedUsd).toBeLessThanOrEqual(d.forecast.coldUsd);
|
|
1106
1070
|
});
|
|
@@ -1108,22 +1072,18 @@ describe("recorded forecast is the expected price, not the cold worst case", ()
|
|
|
1108
1072
|
|
|
1109
1073
|
describe("cache reliability in the stay/switch comparison", () => {
|
|
1110
1074
|
const warmSlug = "x-ai/grok-4.6";
|
|
1111
|
-
function ledgerWithReliability(rate: number | null, samples = 50):
|
|
1112
|
-
return {
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
tokenRatio: () => null,
|
|
1121
|
-
recentEntries: () => [],
|
|
1122
|
-
cacheReliability: (slug) => (rate === null || slug !== warmSlug ? null : { slug, samples, hitRate: rate }),
|
|
1123
|
-
};
|
|
1075
|
+
function ledgerWithReliability(rate: number | null, samples = 50): AsyncLedger {
|
|
1076
|
+
return fakeLedger({
|
|
1077
|
+
// Only the warm model has a measured rate; everything else is unmeasured,
|
|
1078
|
+
// which is what the stay/switch comparison treats as "assume reliable".
|
|
1079
|
+
cacheReliability: async (slugs) =>
|
|
1080
|
+
rate === null
|
|
1081
|
+
? new Map()
|
|
1082
|
+
: new Map(slugs.filter((s) => s === warmSlug).map((slug) => [slug, { slug, samples, hitRate: rate }])),
|
|
1083
|
+
});
|
|
1124
1084
|
}
|
|
1125
|
-
const stayCostOf = (ledger:
|
|
1126
|
-
const d = run({
|
|
1085
|
+
const stayCostOf = async (ledger: AsyncLedger, cfg: RouterConfig = BASE): Promise<number> => {
|
|
1086
|
+
const d = await run({
|
|
1127
1087
|
tier: "hard",
|
|
1128
1088
|
promptTokens: 80_000,
|
|
1129
1089
|
cfg,
|
|
@@ -1135,23 +1095,23 @@ describe("cache reliability in the stay/switch comparison", () => {
|
|
|
1135
1095
|
return Number(m[1]);
|
|
1136
1096
|
};
|
|
1137
1097
|
|
|
1138
|
-
test("an unreliable cache prices staying at the fresh rate, a reliable one at the cached rate", () => {
|
|
1139
|
-
const reliable = stayCostOf(ledgerWithReliability(1));
|
|
1140
|
-
const flaky = stayCostOf(ledgerWithReliability(0));
|
|
1141
|
-
const unknown = stayCostOf(ledgerWithReliability(null));
|
|
1098
|
+
test("an unreliable cache prices staying at the fresh rate, a reliable one at the cached rate", async () => {
|
|
1099
|
+
const reliable = await stayCostOf(ledgerWithReliability(1));
|
|
1100
|
+
const flaky = await stayCostOf(ledgerWithReliability(0));
|
|
1101
|
+
const unknown = await stayCostOf(ledgerWithReliability(null));
|
|
1142
1102
|
expect(flaky).toBeGreaterThan(reliable);
|
|
1143
1103
|
expect(unknown).toBeCloseTo(reliable, 6);
|
|
1144
1104
|
});
|
|
1145
1105
|
|
|
1146
|
-
test("too few samples, or the feature off, assume a reliable cache", () => {
|
|
1147
|
-
const reliable = stayCostOf(ledgerWithReliability(1));
|
|
1148
|
-
expect(stayCostOf(ledgerWithReliability(0, 3))).toBeCloseTo(reliable, 6);
|
|
1106
|
+
test("too few samples, or the feature off, assume a reliable cache", async () => {
|
|
1107
|
+
const reliable = await stayCostOf(ledgerWithReliability(1));
|
|
1108
|
+
expect(await stayCostOf(ledgerWithReliability(0, 3))).toBeCloseTo(reliable, 6);
|
|
1149
1109
|
const off: RouterConfig = { ...BASE, filters: { ...BASE.filters, cacheReliabilityMinSamples: 0 } };
|
|
1150
|
-
expect(stayCostOf(ledgerWithReliability(0), off)).toBeCloseTo(reliable, 6);
|
|
1110
|
+
expect(await stayCostOf(ledgerWithReliability(0), off)).toBeCloseTo(reliable, 6);
|
|
1151
1111
|
});
|
|
1152
1112
|
|
|
1153
|
-
test("the reason names the measured hit rate", () => {
|
|
1154
|
-
const d = run({
|
|
1113
|
+
test("the reason names the measured hit rate", async () => {
|
|
1114
|
+
const d = await run({
|
|
1155
1115
|
tier: "hard",
|
|
1156
1116
|
promptTokens: 80_000,
|
|
1157
1117
|
ledger: ledgerWithReliability(0.5, 40),
|
|
@@ -1162,9 +1122,9 @@ describe("cache reliability in the stay/switch comparison", () => {
|
|
|
1162
1122
|
});
|
|
1163
1123
|
|
|
1164
1124
|
describe("session pin (forceSlug)", () => {
|
|
1165
|
-
test("a pinned catalog model wins over ranking and the warm model; an unknown pin is ignored with a reason", () => {
|
|
1166
|
-
const warmSlug = run({ tier: "moderate" }).slug;
|
|
1167
|
-
const pinSlug = run({ tier: "hard" }).slug; // a real, differently-ranked model
|
|
1125
|
+
test("a pinned catalog model wins over ranking and the warm model; an unknown pin is ignored with a reason", async () => {
|
|
1126
|
+
const warmSlug = (await run({ tier: "moderate" })).slug;
|
|
1127
|
+
const pinSlug = (await run({ tier: "hard" })).slug; // a real, differently-ranked model
|
|
1168
1128
|
const req = request("tidy the retry helper");
|
|
1169
1129
|
const features = extractFeatures(req, 50_000);
|
|
1170
1130
|
const base = {
|
|
@@ -1174,7 +1134,6 @@ describe("session pin (forceSlug)", () => {
|
|
|
1174
1134
|
profile: PROFILE,
|
|
1175
1135
|
state: state({ currentSlug: warmSlug, currentTier: "moderate", cacheWarmSlug: warmSlug, cacheWarmAtMs: Date.now(), lastPromptTokens: 50_000 }),
|
|
1176
1136
|
snapshot: SNAPSHOT,
|
|
1177
|
-
ledger: null,
|
|
1178
1137
|
cfg: { ...BASE, hysteresis: { ...BASE.hysteresis, switchMargin: 1e6 } },
|
|
1179
1138
|
nowMs: Date.now(),
|
|
1180
1139
|
};
|
|
@@ -1189,7 +1148,7 @@ describe("session pin (forceSlug)", () => {
|
|
|
1189
1148
|
});
|
|
1190
1149
|
|
|
1191
1150
|
describe("budget.perMonthUsd pacing", () => {
|
|
1192
|
-
test("monthPace spreads what is left over the days left, today included", () => {
|
|
1151
|
+
test("monthPace spreads what is left over the days left, today included", async () => {
|
|
1193
1152
|
const sep7 = Date.UTC(2026, 8, 7, 12);
|
|
1194
1153
|
expect(monthStartMs(sep7)).toBe(Date.UTC(2026, 8, 1));
|
|
1195
1154
|
const p = monthPace(sep7, 60, 30);
|
|
@@ -1199,28 +1158,28 @@ describe("budget.perMonthUsd pacing", () => {
|
|
|
1199
1158
|
expect(monthPace(Date.UTC(2026, 8, 30, 12), 60, 0).daysLeft).toBe(1);
|
|
1200
1159
|
});
|
|
1201
1160
|
|
|
1202
|
-
test("a month running ahead of pace tightens the daily cap and says so", () => {
|
|
1203
|
-
const ledger:
|
|
1204
|
-
record: () => {},
|
|
1205
|
-
conversationSpend: () => 0,
|
|
1206
|
-
spendSince: (sinceMs) => (sinceMs <= monthStartMs(Date.now()) + 1 ? 59.99 : 0), // month-to-date $59.99, last 24h $0
|
|
1207
|
-
blendedRate: () => null,
|
|
1208
|
-
trust: () => null,
|
|
1209
|
-
allTrust: () => [],
|
|
1210
|
-
latency: () => null,
|
|
1211
|
-
tokenRatio: () => null,
|
|
1212
|
-
recentEntries: () => [],
|
|
1213
|
-
};
|
|
1161
|
+
test("a month running ahead of pace tightens the daily cap and says so", async () => {
|
|
1162
|
+
const ledger: AsyncLedger = fakeLedger({
|
|
1163
|
+
record: async () => {},
|
|
1164
|
+
conversationSpend: async () => 0,
|
|
1165
|
+
spendSince: async (sinceMs) => (sinceMs <= monthStartMs(Date.now()) + 1 ? 59.99 : 0), // month-to-date $59.99, last 24h $0
|
|
1166
|
+
blendedRate: async () => null,
|
|
1167
|
+
trust: async () => null,
|
|
1168
|
+
allTrust: async () => [],
|
|
1169
|
+
latency: async () => null,
|
|
1170
|
+
tokenRatio: async () => null,
|
|
1171
|
+
recentEntries: async () => [],
|
|
1172
|
+
});
|
|
1214
1173
|
const cfg: RouterConfig = { ...BASE, budget: { ...BASE.budget, perMonthUsd: 60, onExceeded: "reject" } };
|
|
1215
|
-
expect(
|
|
1174
|
+
await expect(run({ tier: "hard", promptTokens: 50_000, cfg, ledger })).rejects.toThrow(/month pacing: \$59\.99 of \$60 spent/);
|
|
1216
1175
|
// Under pace: the cap is generous and nothing breaches.
|
|
1217
|
-
const easy:
|
|
1218
|
-
expect(run({ tier: "hard", promptTokens: 50_000, cfg, ledger: easy }).budgetDowngraded).toBe(false);
|
|
1176
|
+
const easy: AsyncLedger = { ...ledger, spendSince: async () => 1 };
|
|
1177
|
+
expect((await run({ tier: "hard", promptTokens: 50_000, cfg, ledger: easy })).budgetDowngraded).toBe(false);
|
|
1219
1178
|
});
|
|
1220
1179
|
});
|
|
1221
1180
|
|
|
1222
1181
|
describe("filters.latencyWeightContinuation", () => {
|
|
1223
|
-
test("applies only to tool-result continuations, and only when set", () => {
|
|
1182
|
+
test("applies only to tool-result continuations, and only when set", async () => {
|
|
1224
1183
|
const f = { ...BASE.filters, latencyWeight: 0.75 };
|
|
1225
1184
|
expect(latencyWeightFor(f, false)).toBe(0.75);
|
|
1226
1185
|
expect(latencyWeightFor(f, true)).toBe(0.75);
|