auto-model-router 0.31.0 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (110) hide show
  1. package/.omp-plugin/marketplace.json +2 -2
  2. package/README.md +32 -2
  3. package/omp-extension/router-configure.ts +9 -7
  4. package/package.json +1 -1
  5. package/src/cli/config-cmd.ts +8 -7
  6. package/src/cli/explain.ts +10 -5
  7. package/src/cli/export.ts +6 -5
  8. package/src/cli/models.ts +10 -7
  9. package/src/cli/report.ts +6 -1
  10. package/src/cli/stats.ts +7 -7
  11. package/src/config/load.ts +10 -1
  12. package/src/config/types.ts +10 -1
  13. package/src/context/bridge.ts +7 -7
  14. package/src/context/index.ts +3 -3
  15. package/src/context/store.ts +39 -56
  16. package/src/context/types.ts +7 -6
  17. package/src/cost/blended.ts +28 -7
  18. package/src/cost/feedback.ts +33 -37
  19. package/src/cost/ledger-sql.ts +547 -0
  20. package/src/cost/ledger.ts +30 -459
  21. package/src/cost/report.ts +171 -129
  22. package/src/cost/retention.ts +10 -10
  23. package/src/cost/summary.ts +15 -10
  24. package/src/cost/types.ts +43 -62
  25. package/src/cost/views.ts +79 -49
  26. package/src/lib.ts +6 -2
  27. package/src/router/candidates.ts +7 -15
  28. package/src/router/classify.ts +6 -4
  29. package/src/router/index.ts +95 -9
  30. package/src/router/select.ts +38 -21
  31. package/src/router/state.ts +90 -102
  32. package/src/router/types.ts +11 -5
  33. package/src/server/advise.ts +6 -4
  34. package/src/server/compaction-digest.ts +1 -1
  35. package/src/server/digest.ts +9 -10
  36. package/src/server/http.ts +101 -41
  37. package/src/server/providers.ts +18 -4
  38. package/src/server/turn.ts +32 -9
  39. package/src/tokens/estimate.ts +16 -6
  40. package/src/upstream/ollama-usage.ts +21 -11
  41. package/src/util/schema.ts +201 -0
  42. package/src/util/sql.ts +246 -0
  43. package/src/wire/anthropic/messages.ts +3 -4
  44. package/src/wire/openai/request.ts +1 -0
  45. package/src/wire/types.ts +7 -0
  46. package/test/anthropic-wire.test.ts +9 -9
  47. package/test/benchmark-feeds.test.ts +7 -7
  48. package/test/cache-control.test.ts +7 -7
  49. package/test/cache-estimate.test.ts +5 -5
  50. package/test/catalog-view.test.ts +4 -4
  51. package/test/catalog.test.ts +11 -11
  52. package/test/classify.test.ts +24 -24
  53. package/test/compaction.test.ts +20 -20
  54. package/test/config-wizard.test.ts +32 -32
  55. package/test/config.test.ts +10 -10
  56. package/test/connect-harnesses.test.ts +11 -11
  57. package/test/context-bridge.test.ts +40 -30
  58. package/test/context-prune.test.ts +43 -36
  59. package/test/context-query.test.ts +8 -8
  60. package/test/controls.test.ts +54 -27
  61. package/test/cost.test.ts +12 -12
  62. package/test/digest.test.ts +55 -44
  63. package/test/embed-lifecycle.test.ts +5 -5
  64. package/test/embed-logic.test.ts +26 -26
  65. package/test/escalate.test.ts +17 -17
  66. package/test/eval.test.ts +13 -13
  67. package/test/executable.test.ts +6 -6
  68. package/test/exploration.test.ts +19 -20
  69. package/test/failover.test.ts +22 -21
  70. package/test/fakes.ts +105 -0
  71. package/test/features.test.ts +21 -21
  72. package/test/harness-requests.test.ts +3 -3
  73. package/test/harness-switch.test.ts +5 -5
  74. package/test/hold-exploration.test.ts +13 -13
  75. package/test/hot-reload.test.ts +5 -5
  76. package/test/learned.test.ts +5 -5
  77. package/test/ledger-sql.test.ts +342 -0
  78. package/test/mcp-entry.test.ts +5 -5
  79. package/test/migrations.test.ts +28 -22
  80. package/test/models-yml.test.ts +18 -18
  81. package/test/ollama.test.ts +40 -34
  82. package/test/omp-credentials.test.ts +16 -16
  83. package/test/policy.test.ts +3 -3
  84. package/test/reconfigure.test.ts +4 -4
  85. package/test/redaction.test.ts +41 -35
  86. package/test/remote.test.ts +12 -12
  87. package/test/report-logic.test.ts +8 -8
  88. package/test/report.test.ts +95 -87
  89. package/test/retention.test.ts +79 -66
  90. package/test/schema.test.ts +123 -0
  91. package/test/scope.test.ts +8 -8
  92. package/test/select.test.ts +216 -257
  93. package/test/skills.test.ts +3 -3
  94. package/test/sql-shim.test.ts +154 -0
  95. package/test/state.test.ts +43 -36
  96. package/test/summary.test.ts +38 -27
  97. package/test/tier-plan.test.ts +45 -62
  98. package/test/toast-logic.test.ts +31 -31
  99. package/test/tokens.test.ts +95 -80
  100. package/test/trust-attribution.test.ts +217 -187
  101. package/test/trust-window.test.ts +37 -32
  102. package/test/turn.test.ts +55 -23
  103. package/test/upstreams.test.ts +13 -13
  104. package/test/views.test.ts +81 -59
  105. package/test/wire-request.test.ts +17 -17
  106. package/test/wire-responses.test.ts +4 -4
  107. package/tools/agentdox-e2e.ts +5 -2
  108. package/tools/export-benchmarks.ts +5 -5
  109. package/tools/ledger-parity.ts +266 -0
  110. package/tools/replay.ts +16 -8
@@ -43,19 +43,19 @@ function loop(turns: number): NormRequest {
43
43
  }
44
44
 
45
45
  describe("planCacheBreakpoints", () => {
46
- test("marks the tail so the next turn can read this turn's whole prompt", () => {
46
+ test("marks the tail so the next turn can read this turn's whole prompt", async () => {
47
47
  const req = loop(12);
48
48
  const picks = planCacheBreakpoints(req, MODEL, cfg());
49
49
  expect(picks).toContain(req.messages.length - 1);
50
50
  });
51
51
 
52
- test("marks the system prefix", () => {
52
+ test("marks the system prefix", async () => {
53
53
  const req = loop(12);
54
54
  const picks = planCacheBreakpoints(req, MODEL, cfg());
55
55
  expect(picks).toContain(0);
56
56
  });
57
57
 
58
- test("mid-history boundaries are stable as the conversation grows", () => {
58
+ test("mid-history boundaries are stable as the conversation grows", async () => {
59
59
  // Uncapped so the comparison is about placement, not slot eviction.
60
60
  const uncapped = cfg({ maxBreakpoints: 64, milestoneTokens: 4_000 });
61
61
  const mid = (turns: number): number[] => {
@@ -72,7 +72,7 @@ describe("planCacheBreakpoints", () => {
72
72
  }
73
73
  });
74
74
 
75
- test("boundaries are spaced by the milestone size, not by message position", () => {
75
+ test("boundaries are spaced by the milestone size, not by message position", async () => {
76
76
  const req = loop(30);
77
77
  const tail = req.messages.length - 1;
78
78
  const coarse = planCacheBreakpoints(req, MODEL, cfg({ maxBreakpoints: 64, milestoneTokens: 20_000 })).filter(
@@ -84,13 +84,13 @@ describe("planCacheBreakpoints", () => {
84
84
  expect(fine.length).toBeGreaterThan(coarse.length);
85
85
  });
86
86
 
87
- test("keeps the system prefix and the tail when slots are scarce", () => {
87
+ test("keeps the system prefix and the tail when slots are scarce", async () => {
88
88
  const req = loop(30);
89
89
  const picks = planCacheBreakpoints(req, MODEL, cfg({ maxBreakpoints: 2, milestoneTokens: 4_000 }));
90
90
  expect(picks).toEqual([0, req.messages.length - 1]);
91
91
  });
92
92
 
93
- test("milestones follow post-compaction sizes", () => {
93
+ test("milestones follow post-compaction sizes", async () => {
94
94
  const req = loop(30);
95
95
  const tail = req.messages.length - 1;
96
96
  const plan = planCompaction(req.messages, { ...BASE.compaction, enabled: true }, req.promptBytes * 0.3, req.promptBytes);
@@ -103,7 +103,7 @@ describe("planCacheBreakpoints", () => {
103
103
  expect(Math.min(...compacted)).toBeGreaterThan(Math.min(...raw));
104
104
  });
105
105
 
106
- test("injects nothing below the minimum prompt size, or when disabled", () => {
106
+ test("injects nothing below the minimum prompt size, or when disabled", async () => {
107
107
  const small = parseChatRequest({ model: "auto", messages: [{ role: "user", content: "hi" }] }, new Headers());
108
108
  expect(planCacheBreakpoints(small, MODEL, cfg())).toEqual([]);
109
109
  expect(planCacheBreakpoints(loop(12), MODEL, cfg({ injectBreakpoints: false }))).toEqual([]);
@@ -15,18 +15,18 @@ const base = (over: Partial<UsageCounts> = {}): UsageCounts => ({ ...EMPTY_USAGE
15
15
  const ctx = { previousSlug: "ollama/glm", previousPromptTokens: 100_000, previousAtMs: 1_000_000, servedSlug: "ollama/glm", nowMs: 1_060_000, cacheWarmTtlMs: 300_000 };
16
16
 
17
17
  describe("estimateUnreportedCache", () => {
18
- test("same model within the TTL: the previous prompt is the cached prefix, flagged as estimated", () => {
18
+ test("same model within the TTL: the previous prompt is the cached prefix, flagged as estimated", async () => {
19
19
  const out = estimateUnreportedCache(base(), ctx);
20
20
  expect(out.cachedTokens).toBe(100_000);
21
21
  expect(out.cachedEstimated).toBe(true);
22
22
  expect(out.promptTokens).toBe(120_000);
23
23
  });
24
24
 
25
- test("a shorter prompt than the previous one caps the cached count at the prompt", () => {
25
+ test("a shorter prompt than the previous one caps the cached count at the prompt", async () => {
26
26
  expect(estimateUnreportedCache(base({ promptTokens: 40_000 }), ctx).cachedTokens).toBe(40_000);
27
27
  });
28
28
 
29
- test("first turn, model switch, or idle past the TTL count as cold", () => {
29
+ test("first turn, model switch, or idle past the TTL count as cold", async () => {
30
30
  expect(estimateUnreportedCache(base(), { ...ctx, previousSlug: null }).cachedTokens).toBe(0);
31
31
  expect(estimateUnreportedCache(base(), { ...ctx, previousSlug: "ollama/other" }).cachedTokens).toBe(0);
32
32
  expect(estimateUnreportedCache(base(), { ...ctx, nowMs: ctx.previousAtMs + 300_001 }).cachedTokens).toBe(0);
@@ -34,14 +34,14 @@ describe("estimateUnreportedCache", () => {
34
34
  expect(estimateUnreportedCache(base(), { ...ctx, cacheWarmTtlMs: 0 }).cachedTokens).toBe(0);
35
35
  });
36
36
 
37
- test("provider-reported cache counts are never overwritten", () => {
37
+ test("provider-reported cache counts are never overwritten", async () => {
38
38
  const reported = base({ cachedTokens: 5_000 });
39
39
  expect(estimateUnreportedCache(reported, ctx)).toBe(reported);
40
40
  const written = base({ cacheWriteTokens: 5_000 });
41
41
  expect(estimateUnreportedCache(written, ctx)).toBe(written);
42
42
  });
43
43
 
44
- test("no prompt tokens: nothing to estimate", () => {
44
+ test("no prompt tokens: nothing to estimate", async () => {
45
45
  const empty = base({ promptTokens: 0 });
46
46
  expect(estimateUnreportedCache(empty, ctx)).toBe(empty);
47
47
  });
@@ -63,7 +63,7 @@ const served = () => null;
63
63
  const bySlug = (view: CatalogView): Map<string, CatalogViewModel> => new Map(view.models.map((m) => [m.slug, m]));
64
64
 
65
65
  describe("catalogView", () => {
66
- test("sorted by slug, prices per million, vendor per slug, and no verdict without a policy", () => {
66
+ test("sorted by slug, prices per million, vendor per slug, and no verdict without a policy", async () => {
67
67
  const view = catalogView({ models: MODELS, fetchedAtMs: 123, unserved: served });
68
68
  expect(view.fetchedAtMs).toBe(123);
69
69
  expect(view.models.map((m) => m.slug)).toEqual([...MODELS.map((m) => m.slug)].sort());
@@ -91,7 +91,7 @@ describe("catalogView", () => {
91
91
  expect(bySlug(view).get("azure-eu/gpt-4o-deploy")!.price).toEqual({ prompt: 2.5, completion: 10, cacheRead: 1.25 });
92
92
  });
93
93
 
94
- test("vendor: the namespace before the first slash; a named upstream's model id may carry its own", () => {
94
+ test("vendor: the namespace before the first slash; a named upstream's model id may carry its own", async () => {
95
95
  expect(vendorOf({ slug: "anthropic/claude-sonnet-5", provider: "openrouter" })).toBe("anthropic");
96
96
  expect(vendorOf({ slug: "ollama/glm-5.3-flash", provider: "ollama" })).toBe("ollama");
97
97
  expect(vendorOf({ slug: "vllm/meta-llama/Llama-3", provider: "vllm" })).toBe("meta-llama");
@@ -101,7 +101,7 @@ describe("catalogView", () => {
101
101
  expect(view.get("azure-eu/gpt-4o-deploy")!.vendor).toBe("azure-eu");
102
102
  });
103
103
 
104
- test("an empty policy still judges: the router's own filters and the upstream's state", () => {
104
+ test("an empty policy still judges: the router's own filters and the upstream's state", async () => {
105
105
  const unserved = (p: string) => (p === "azure-eu" ? "upstream azure-eu is disabled" : null);
106
106
  const view = bySlug(catalogView({ models: MODELS, fetchedAtMs: 0, verdict: { filters: DEFAULT_CONFIG.filters }, unserved }));
107
107
  expect(view.get("openai/gpt-5")).toMatchObject({ admitted: true });
@@ -118,7 +118,7 @@ describe("catalogView", () => {
118
118
  expect(open.get("tencent/translator")!.admitted).toBe(true);
119
119
  });
120
120
 
121
- test("allow list, deny glob and pin, in the order a turn applies them", () => {
121
+ test("allow list, deny glob and pin, in the order a turn applies them", async () => {
122
122
  const filters = { ...DEFAULT_CONFIG.filters, allow: ["anthropic/*", "vllm/*"], deny: ["*haiku*"] };
123
123
  const view = bySlug(catalogView({ models: MODELS, fetchedAtMs: 0, verdict: { filters }, unserved: served }));
124
124
  expect(view.get("openai/gpt-5")).toMatchObject({ admitted: false, reason: "not in the allow list" });
@@ -12,7 +12,7 @@ function rawFor(slug: string): unknown {
12
12
  }
13
13
 
14
14
  describe("normalizeCatalogModel", () => {
15
- test("survives every record in the real catalog without throwing", () => {
15
+ test("survives every record in the real catalog without throwing", async () => {
16
16
  let normalized = 0;
17
17
  for (const raw of RAW) {
18
18
  const model = normalizeCatalogModel(raw);
@@ -22,7 +22,7 @@ describe("normalizeCatalogModel", () => {
22
22
  expect(normalized).toBeGreaterThan(RAW.length * 0.9);
23
23
  });
24
24
 
25
- test("every normalized model has strictly positive prompt and completion prices", () => {
25
+ test("every normalized model has strictly positive prompt and completion prices", async () => {
26
26
  for (const raw of RAW) {
27
27
  const model = normalizeCatalogModel(raw);
28
28
  if (model === null) continue;
@@ -33,7 +33,7 @@ describe("normalizeCatalogModel", () => {
33
33
  }
34
34
  });
35
35
 
36
- test("rejects the openrouter meta-routers, whose -1 pricing means unknown", () => {
36
+ test("rejects the openrouter meta-routers, whose -1 pricing means unknown", async () => {
37
37
  // Routing to another router is both out of scope and uncostable: a -1
38
38
  // price would otherwise be read as free and win every tier outright.
39
39
  for (const slug of ["openrouter/auto", "openrouter/pareto-code", "openrouter/fusion"]) {
@@ -41,7 +41,7 @@ describe("normalizeCatalogModel", () => {
41
41
  }
42
42
  });
43
43
 
44
- test("does not mistake unknown (-1) pricing for free pricing", () => {
44
+ test("does not mistake unknown (-1) pricing for free pricing", async () => {
45
45
  const free = normalizeCatalogModel(rawFor("openai/gpt-oss-20b:free"));
46
46
  expect(free).not.toBeNull();
47
47
  expect(free?.isFree).toBe(true);
@@ -49,7 +49,7 @@ describe("normalizeCatalogModel", () => {
49
49
  expect(normalizeCatalogModel(rawFor("openrouter/auto"))).toBeNull();
50
50
  });
51
51
 
52
- test("preserves published quality scores and never imputes missing ones", () => {
52
+ test("preserves published quality scores and never imputes missing ones", async () => {
53
53
  const scoredInFixture = RAW.filter(
54
54
  (m) =>
55
55
  typeof m === "object" &&
@@ -74,7 +74,7 @@ describe("normalizeCatalogModel", () => {
74
74
  expect(unscored).toBeGreaterThan(0);
75
75
  });
76
76
 
77
- test("reads long-context override tiers, sorted ascending", () => {
77
+ test("reads long-context override tiers, sorted ascending", async () => {
78
78
  const sonnet = normalizeCatalogModel(rawFor("anthropic/claude-sonnet-4.5"));
79
79
  expect(sonnet).not.toBeNull();
80
80
  expect(sonnet?.priceTiers.length).toBeGreaterThan(0);
@@ -91,7 +91,7 @@ describe("normalizeCatalogModel", () => {
91
91
  expect(first?.price.prompt).toBeGreaterThan(sonnet?.price.prompt ?? 0);
92
92
  });
93
93
 
94
- test("an override tier inherits components it does not restate", () => {
94
+ test("an override tier inherits components it does not restate", async () => {
95
95
  for (const raw of RAW) {
96
96
  const model = normalizeCatalogModel(raw);
97
97
  if (model === null || model.priceTiers.length === 0) continue;
@@ -103,7 +103,7 @@ describe("normalizeCatalogModel", () => {
103
103
  }
104
104
  });
105
105
 
106
- test("strips the floating-alias marker from the author segment", () => {
106
+ test("strips the floating-alias marker from the author segment", async () => {
107
107
  const alias = normalizeCatalogModel(rawFor("~x-ai/grok-latest"));
108
108
  expect(alias).not.toBeNull();
109
109
  expect(alias?.author).toBe("x-ai");
@@ -111,7 +111,7 @@ describe("normalizeCatalogModel", () => {
111
111
  expect(alias?.slug.startsWith("~")).toBe(true);
112
112
  });
113
113
 
114
- test("derives capability flags from supported_parameters", () => {
114
+ test("derives capability flags from supported_parameters", async () => {
115
115
  const sonnet = normalizeCatalogModel(rawFor("anthropic/claude-sonnet-4.5"));
116
116
  expect(sonnet?.supportsTools).toBe(true);
117
117
  expect(sonnet?.supportsToolChoice).toBe(true);
@@ -122,7 +122,7 @@ describe("normalizeCatalogModel", () => {
122
122
  expect(toolless?.supportsTools).toBe(false);
123
123
  });
124
124
 
125
- test("rejects records missing the fields routing depends on", () => {
125
+ test("rejects records missing the fields routing depends on", async () => {
126
126
  expect(normalizeCatalogModel({})).toBeNull();
127
127
  expect(normalizeCatalogModel(null)).toBeNull();
128
128
  expect(normalizeCatalogModel({ id: "x/y" })).toBeNull();
@@ -220,7 +220,7 @@ describe("createCatalog key-scoped availability", () => {
220
220
  const db = openDb(":memory:");
221
221
  const catalog = createCatalog(cfg, upstream, db);
222
222
 
223
- await expect(catalog.get()).rejects.toThrow("Unauthorized");
223
+ (await expect(catalog.get())).rejects.toThrow("Unauthorized");
224
224
  db.close();
225
225
  });
226
226
 
@@ -104,7 +104,7 @@ function scriptedUpstream(behaviour: () => Promise<CompletionResult>): UpstreamC
104
104
 
105
105
  const tierIdx = (t: Tier): number => TIER_ORDER.indexOf(t);
106
106
  describe("scoreHeuristic", () => {
107
- test("a mechanical tool-result continuation scores cheaper than a fresh architecture question", () => {
107
+ test("a mechanical tool-result continuation scores cheaper than a fresh architecture question", async () => {
108
108
  // The single most valuable signal in agent traffic: most turns are
109
109
  // post-tool-result continuations, and they do not need a frontier model.
110
110
  const continuation = scoreHeuristic(
@@ -135,7 +135,7 @@ describe("scoreHeuristic", () => {
135
135
  expect(tierIdx(continuation.tier)).toBeLessThan(tierIdx(architecture.tier));
136
136
  });
137
137
 
138
- test("a failing tool result raises the tier above a clean one", () => {
138
+ test("a failing tool result raises the tier above a clean one", async () => {
139
139
  const clean = scoreHeuristic(
140
140
  featuresFor([
141
141
  SYSTEM,
@@ -165,7 +165,7 @@ describe("scoreHeuristic", () => {
165
165
  expect(failed.score).toBeGreaterThan(clean.score);
166
166
  });
167
167
 
168
- test("a requested high reasoning effort raises the score", () => {
168
+ test("a requested high reasoning effort raises the score", async () => {
169
169
  const plain = scoreHeuristic(featuresFor([SYSTEM, { role: "user", content: "tidy this up" }]), BASE);
170
170
  const thinking = scoreHeuristic(
171
171
  extractFeatures(
@@ -180,7 +180,7 @@ describe("scoreHeuristic", () => {
180
180
  expect(thinking.score).toBeGreaterThan(plain.score);
181
181
  });
182
182
 
183
- test("the reasoning weight is configurable, so a session-wide level can be discounted", () => {
183
+ test("the reasoning weight is configurable, so a session-wide level can be discounted", async () => {
184
184
  // A harness that pins one reasoning level for a whole session turns this
185
185
  // "signal" into a constant that lifts every turn's score. Measured live:
186
186
  // the level never changed within 111 of 115 conversations, and 64 of 119
@@ -205,11 +205,11 @@ describe("scoreHeuristic", () => {
205
205
  expect(discounted.reasons.some((r) => /requested reasoning/.test(r))).toBe(false);
206
206
  });
207
207
 
208
- test("ships with the historical weights, so enabling a discount is opt-in", () => {
208
+ test("ships with the historical weights, so enabling a discount is opt-in", async () => {
209
209
  expect(DEFAULT_CONFIG.classifier.reasoningWeights).toEqual({ medium: 0.14, high: 0.24, xhigh: 0.3, max: 0.34 });
210
210
  });
211
211
 
212
- test("always produces a bounded score, a real tier, and its reasoning", () => {
212
+ test("always produces a bounded score, a real tier, and its reasoning", async () => {
213
213
  const c = scoreHeuristic(featuresFor([SYSTEM, { role: "user", content: "hello" }]), BASE);
214
214
  expect(c.score).toBeGreaterThanOrEqual(0);
215
215
  expect(c.score).toBeLessThanOrEqual(1);
@@ -220,12 +220,12 @@ describe("scoreHeuristic", () => {
220
220
  expect(c.confidence).toBeLessThanOrEqual(1);
221
221
  });
222
222
 
223
- test("a shallow tool-result continuation stays trivial", () => {
223
+ test("a shallow tool-result continuation stays trivial", async () => {
224
224
  const shallow = scoreHeuristic(contFeatures(2), BASE);
225
225
  expect(shallow.tier).toBe("trivial");
226
226
  });
227
227
 
228
- test("a sustained autonomous loop climbs out of trivial", () => {
228
+ test("a sustained autonomous loop climbs out of trivial", async () => {
229
229
  // The failure mode this fixes: a long coding loop pinned to the cheapest
230
230
  // tier for dozens of turns because agentic complexity never accumulated.
231
231
  const shallow = scoreHeuristic(contFeatures(2), BASE);
@@ -234,7 +234,7 @@ describe("scoreHeuristic", () => {
234
234
  expect(deep.tier).not.toBe("trivial");
235
235
  });
236
236
 
237
- test("score increases monotonically with loop depth past the agentic threshold", () => {
237
+ test("score increases monotonically with loop depth past the agentic threshold", async () => {
238
238
  const depths = [4, 6, 8, 10, 15, 20, 30];
239
239
  let prev = -1;
240
240
  for (const d of depths) {
@@ -244,7 +244,7 @@ describe("scoreHeuristic", () => {
244
244
  }
245
245
  });
246
246
 
247
- test("pure loop depth never reaches hard on its own, however runaway", () => {
247
+ test("pure loop depth never reaches hard on its own, however runaway", async () => {
248
248
  // A sustained-but-not-runaway loop tops out in moderate: the calibrated
249
249
  // ramp ceiling for ordinary deep work.
250
250
  const midRange = scoreHeuristic(contFeatures(30), BASE);
@@ -264,7 +264,7 @@ describe("scoreHeuristic", () => {
264
264
  expect(scoreHeuristic(contFeatures(400), BASE).tier).toBe("moderate");
265
265
  });
266
266
 
267
- test("a circular tool call on a FRESH turn escalates to hard", () => {
267
+ test("a circular tool call on a FRESH turn escalates to hard", async () => {
268
268
  // Off a continuation the stuck signal keeps full weight: the user is
269
269
  // watching a live loop and a pricier model may actually break it.
270
270
  const deepCircular = scoreHeuristic(
@@ -274,7 +274,7 @@ describe("scoreHeuristic", () => {
274
274
  expect(deepCircular.tier).toBe("hard");
275
275
  });
276
276
 
277
- test("a circular tool call on a mechanical continuation is damped, not hard", () => {
277
+ test("a circular tool call on a mechanical continuation is damped, not hard", async () => {
278
278
  // Measured: hard escalations on circular calls never shortened the loop
279
279
  // (chain means identical, 5.74 turns, hard vs moderate). 22 of 27 such
280
280
  // hard turns were mechanical continuations paying up to 6x for nothing.
@@ -285,7 +285,7 @@ describe("scoreHeuristic", () => {
285
285
  expect(retry.score - plain.score).toBeCloseTo(BASE.classifier.mechanicalRetryFactor * 0.24, 5);
286
286
  });
287
287
 
288
- test("a failing tool result on a deep loop is at least simple", () => {
288
+ test("a failing tool result on a deep loop is at least simple", async () => {
289
289
  // Was 'at least moderate' before the mechanical-retry damp: the flat
290
290
  // +0.26 pushed deep mechanical retry loops into hard. A damped retry
291
291
  // still clears trivial.
@@ -293,7 +293,7 @@ describe("scoreHeuristic", () => {
293
293
  expect(tierIdx(deepAndFailing.tier)).toBeGreaterThanOrEqual(tierIdx("simple"));
294
294
  });
295
295
 
296
- test("a failed-tool retry on a mechanical continuation is damped, not hard", () => {
296
+ test("a failed-tool retry on a mechanical continuation is damped, not hard", async () => {
297
297
  // A retry after a failed tool call is the most mechanical turn there is;
298
298
  // the flat +0.26 let automated retry loops buy the hard tier ($7.02 of one
299
299
  // measured day vs $0.19 for the same rows as moderate picks). The
@@ -313,14 +313,14 @@ describe("scoreHeuristic", () => {
313
313
  });
314
314
 
315
315
  describe("pickQualityAxis", () => {
316
- test("tools imply the coding axis, plain chat the chat axis", () => {
316
+ test("tools imply the coding axis, plain chat the chat axis", async () => {
317
317
  expect(pickQualityAxis(featuresFor([SYSTEM, { role: "user", content: "fix it" }]), BASE)).toBe(BASE.classifier.toolAxis);
318
318
  expect(pickQualityAxis(featuresFor([SYSTEM, { role: "user", content: "hello" }], []), BASE)).toBe(
319
319
  BASE.classifier.chatAxis,
320
320
  );
321
321
  });
322
322
 
323
- test("a deep tool loop switches to the agentic axis", () => {
323
+ test("a deep tool loop switches to the agentic axis", async () => {
324
324
  const deep = featuresFor([
325
325
  SYSTEM,
326
326
  { role: "user", content: "go" },
@@ -395,12 +395,12 @@ describe("classify", () => {
395
395
  });
396
396
 
397
397
  describe("classifyTask", () => {
398
- test("image input is a vision task", () => {
398
+ test("image input is a vision task", async () => {
399
399
  const f = featuresFor([SYSTEM, { role: "user", content: [{ type: "image_url", image_url: { url: "data:image/png;base64,xxx" } }] }], []);
400
400
  expect(classifyTask(f)).toBe("vision");
401
401
  });
402
402
 
403
- test("a stale image on a tool continuation is coding, not vision", () => {
403
+ test("a stale image on a tool continuation is coding, not vision", async () => {
404
404
  const f = featuresFor(
405
405
  [
406
406
  SYSTEM,
@@ -421,7 +421,7 @@ describe("classifyTask", () => {
421
421
  expect(classifyTask(f)).toBe("coding");
422
422
  });
423
423
 
424
- test("a freshly supplied image mid-loop is vision", () => {
424
+ test("a freshly supplied image mid-loop is vision", async () => {
425
425
  const f = featuresFor(
426
426
  [
427
427
  SYSTEM,
@@ -442,26 +442,26 @@ describe("classifyTask", () => {
442
442
  expect(classifyTask(f)).toBe("vision");
443
443
  });
444
444
 
445
- test("code blocks and diffs are coding tasks", () => {
445
+ test("code blocks and diffs are coding tasks", async () => {
446
446
  expect(classifyTask(featuresFor([SYSTEM, { role: "user", content: "```ts\nconst x = 1;\n```" }], []))).toBe("coding");
447
447
  expect(classifyTask(featuresFor([SYSTEM, { role: "user", content: "diff --git a/x b/x\n@@ -1 +1 @@\n-old\n+new" }], []))).toBe("coding");
448
448
  });
449
449
 
450
- test("tools offered is a coding task", () => {
450
+ test("tools offered is a coding task", async () => {
451
451
  expect(classifyTask(featuresFor([SYSTEM, { role: "user", content: "read the file" }], TOOLS))).toBe("coding");
452
452
  });
453
453
 
454
- test("bare chat with no tools or code is a chat task", () => {
454
+ test("bare chat with no tools or code is a chat task", async () => {
455
455
  expect(classifyTask(featuresFor([SYSTEM, { role: "user", content: "hello, how are you?" }], []))).toBe("chat");
456
456
  });
457
457
 
458
- test("design/architecture prose is a documentation task", () => {
458
+ test("design/architecture prose is a documentation task", async () => {
459
459
  expect(classifyTask(featuresFor([SYSTEM, { role: "user", content: "explain the architecture of the system" }], []))).toBe("documentation");
460
460
  });
461
461
  });
462
462
 
463
463
  describe("classifier.readOnlyToolWeight", () => {
464
- test("subtracts only when enabled and the tail is a read-only loop", () => {
464
+ test("subtracts only when enabled and the tail is a read-only loop", async () => {
465
465
  const base = { ...featuresFor([{ role: "user", content: "look" }]), isToolResultContinuation: true, readOnlyToolTail: true };
466
466
  const off = scoreHeuristic(base, DEFAULT_CONFIG);
467
467
  const cfg = structuredClone(DEFAULT_CONFIG);
@@ -34,12 +34,12 @@ const PAD: NormMessage[] = [asst("z1", "bash", '{"command":"ls"}'), toolMsg("z1"
34
34
  const big = (marker: string): string => `${marker}:${"x".repeat(200)}`;
35
35
 
36
36
  describe("planCompaction", () => {
37
- test("is a no-op when disabled", () => {
37
+ test("is a no-op when disabled", async () => {
38
38
  const msgs = [user("go"), asst("c1", "read", '{"path":"a"}'), toolMsg("c1", "read", big("A")), ...PAD];
39
39
  expect(planCompaction(msgs, { ...CFG, enabled: false }, 0, 10_000).edits).toEqual([]);
40
40
  });
41
41
 
42
- test("truncates a large stale tool result, protecting recent turns", () => {
42
+ test("truncates a large stale tool result, protecting recent turns", async () => {
43
43
  const msgs = [
44
44
  user("go"),
45
45
  asst("c1", "read", '{"path":"a.ts"}'),
@@ -54,7 +54,7 @@ describe("planCompaction", () => {
54
54
  expect(edits[0]?.mode).toBe("truncate");
55
55
  });
56
56
 
57
- test("collapses byte-identical duplicate results, keeping the last", () => {
57
+ test("collapses byte-identical duplicate results, keeping the last", async () => {
58
58
  const msgs = [
59
59
  user("go"),
60
60
  asst("c1", "read", '{"path":"a.ts"}'),
@@ -69,7 +69,7 @@ describe("planCompaction", () => {
69
69
  expect(edits[0]?.mode).toBe("stub");
70
70
  });
71
71
 
72
- test("elides a read superseded by a newer call to the same resource", () => {
72
+ test("elides a read superseded by a newer call to the same resource", async () => {
73
73
  const msgs = [
74
74
  user("go"),
75
75
  asst("c1", "read", '{"path":"a.ts"}'),
@@ -83,7 +83,7 @@ describe("planCompaction", () => {
83
83
  expect(edits[0]?.mode).toBe("stub");
84
84
  });
85
85
 
86
- test("different resources are not superseded", () => {
86
+ test("different resources are not superseded", async () => {
87
87
  const msgs = [
88
88
  user("go"),
89
89
  asst("c1", "read", '{"path":"a.ts"}'),
@@ -95,7 +95,7 @@ describe("planCompaction", () => {
95
95
  expect(planCompaction(msgs, CFG, 10_000, 10_000).edits).toEqual([]);
96
96
  });
97
97
 
98
- test("is deterministic and idempotent on stable input", () => {
98
+ test("is deterministic and idempotent on stable input", async () => {
99
99
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("OLD")), ...PAD];
100
100
  const a = planCompaction(msgs, CFG, 1, 10_000);
101
101
  const b = planCompaction(msgs, CFG, 1, 10_000);
@@ -106,7 +106,7 @@ describe("planCompaction", () => {
106
106
  // bytes for the rest of the conversation, and every later edit lands AFTER
107
107
  // it. Anything else rewrites already-cached history and forces a full
108
108
  // re-read of the prefix on the next turn.
109
- test("the edit set only ever extends forward as the conversation grows", () => {
109
+ test("the edit set only ever extends forward as the conversation grows", async () => {
110
110
  // Sizes GROW with age-descending order (newest results are the biggest), so
111
111
  // a size-ordered planner selects newest-first and its later additions move
112
112
  // BACKWARD into already-cached history. Equal-sized results would make
@@ -154,7 +154,7 @@ describe("renderUpstreamBody applies compaction", () => {
154
154
  stripAssistantReasoning: false,
155
155
  };
156
156
 
157
- test("truncates content in place with a recoverable breadcrumb, preserving pairing", () => {
157
+ test("truncates content in place with a recoverable breadcrumb, preserving pairing", async () => {
158
158
  const raw = [
159
159
  { role: "user", content: "go" },
160
160
  { role: "assistant", content: null, tool_calls: [{ id: "c1", type: "function", function: { name: "read", arguments: "{}" } }] },
@@ -173,7 +173,7 @@ describe("renderUpstreamBody applies compaction", () => {
173
173
  expect(content as string).toEndWith("TAIL");
174
174
  });
175
175
 
176
- test("stub replaces the whole content with a breadcrumb", () => {
176
+ test("stub replaces the whole content with a breadcrumb", async () => {
177
177
  const raw = [
178
178
  { role: "user", content: "go" },
179
179
  { role: "assistant", content: null, tool_calls: [{ id: "c1", type: "function", function: { name: "read", arguments: "{}" } }] },
@@ -191,20 +191,20 @@ describe("plan byte-stability across turns", () => {
191
191
  // invalidates everything after it. So an edit, once dispatched, must be
192
192
  // re-emitted identically on every later turn — which means the planner has
193
193
  // to be told what it already did rather than re-deriving it.
194
- test("edits carry their original byte length for persistence", () => {
194
+ test("edits carry their original byte length for persistence", async () => {
195
195
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("A")), ...PAD];
196
196
  const { edits } = planCompaction(msgs, CFG, 1, 10_000);
197
197
  expect(edits).toHaveLength(1);
198
198
  expect(edits[0]?.bytes).toBe(Buffer.byteLength(big("A")));
199
199
  });
200
200
 
201
- test("validatePlan keeps edits whose target is byte-identical and role-correct", () => {
201
+ test("validatePlan keeps edits whose target is byte-identical and role-correct", async () => {
202
202
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("A")), ...PAD];
203
203
  const { edits } = planCompaction(msgs, CFG, 1, 10_000);
204
204
  expect(validatePlan(edits, msgs)).toEqual(edits);
205
205
  });
206
206
 
207
- test("validatePlan drops edits when history changed under them", () => {
207
+ test("validatePlan drops edits when history changed under them", async () => {
208
208
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("A")), ...PAD];
209
209
  const { edits } = planCompaction(msgs, CFG, 1, 10_000);
210
210
  // Client re-wrote history: the tool result is a different length now.
@@ -212,14 +212,14 @@ describe("plan byte-stability across turns", () => {
212
212
  expect(validatePlan(edits, rewritten)).toEqual([]);
213
213
  });
214
214
 
215
- test("validatePlan drops edits that fall off the message array", () => {
215
+ test("validatePlan drops edits that fall off the message array", async () => {
216
216
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("A")), ...PAD];
217
217
  const { edits } = planCompaction(msgs, CFG, 1, 10_000);
218
218
  // Conversation compacted away client-side: index 2 no longer exists.
219
219
  expect(validatePlan(edits, [user("go"), ...PAD.slice(1)])).toEqual([]);
220
220
  });
221
221
 
222
- test("a carried plan produces identical edits to a fresh plan over the same bytes", () => {
222
+ test("a carried plan produces identical edits to a fresh plan over the same bytes", async () => {
223
223
  // Determinism contract: re-planning over unchanged bytes re-derives the
224
224
  // persisted plan, so the merge in select.ts is a no-op, not a rewrite.
225
225
  const msgs = [
@@ -235,7 +235,7 @@ describe("plan byte-stability across turns", () => {
235
235
  expect(again.edits).toEqual(first.edits);
236
236
  });
237
237
 
238
- test("a carried edit is re-emitted verbatim even when nothing new is eligible", () => {
238
+ test("a carried edit is re-emitted verbatim even when nothing new is eligible", async () => {
239
239
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("A")), ...PAD];
240
240
  const carried = planCompaction(msgs, CFG, 1, 10_000).edits;
241
241
  // Target already met, so a stateless planner would emit nothing at all.
@@ -244,7 +244,7 @@ describe("plan byte-stability across turns", () => {
244
244
  expect(next.savedBytes).toBeGreaterThan(0);
245
245
  });
246
246
 
247
- test("carried savings count toward the target, so the planner only adds what is still needed", () => {
247
+ test("carried savings count toward the target, so the planner only adds what is still needed", async () => {
248
248
  const msgs = [
249
249
  user("go"),
250
250
  asst("c1", "read", '{"path":"a.ts"}'),
@@ -262,7 +262,7 @@ describe("plan byte-stability across turns", () => {
262
262
  expect(next.edits.map((e) => e.index)).toEqual(carried.map((e) => e.index));
263
263
  });
264
264
 
265
- test("a carried edit is never re-planned into a different shape", () => {
265
+ test("a carried edit is never re-planned into a different shape", async () => {
266
266
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("A")), ...PAD];
267
267
  // Carried as a stub; a fresh plan would have chosen truncate.
268
268
  const carried = [{ index: 2, mode: "stub" as const, keepHead: 0, keepTail: 0, note: "carried", bytes: Buffer.byteLength(big("A")) }];
@@ -276,7 +276,7 @@ describe("summarising compaction (edit.digest)", () => {
276
276
  const MUT = { slug: "x/y", fallbacks: [], sessionId: "s", cacheBreakpointMessageIndices: [], reasoning: undefined, maxTokens: undefined, stripAssistantReasoning: false };
277
277
  const DIGEST = "[digest: read output 208 bytes → 40 chars by cheap/model. Full output: re-run read {}]\nA: two hundred x's.";
278
278
 
279
- test("a digested edit replaces the content with the digest, whatever its mode", () => {
279
+ test("a digested edit replaces the content with the digest, whatever its mode", async () => {
280
280
  const raw = [
281
281
  { role: "user", content: "go" },
282
282
  { role: "assistant", content: null, tool_calls: [{ id: "c1", type: "function", function: { name: "read", arguments: "{}" } }] },
@@ -289,13 +289,13 @@ describe("summarising compaction (edit.digest)", () => {
289
289
  }
290
290
  });
291
291
 
292
- test("compactedBytes sizes a digested edit by its digest, never above the original", () => {
292
+ test("compactedBytes sizes a digested edit by its digest, never above the original", async () => {
293
293
  const plain = { index: 2, mode: "truncate" as const, keepHead: 10, keepTail: 10, note: "n", bytes: 5_000 };
294
294
  expect(compactedBytes(5_000, { ...plain, digest: DIGEST })).toBe(Buffer.byteLength(DIGEST));
295
295
  expect(compactedBytes(5_000, { ...plain, digest: "y".repeat(9_000) })).toBe(5_000);
296
296
  });
297
297
 
298
- test("the digest survives validation and a re-plan, so the bytes stay stable", () => {
298
+ test("the digest survives validation and a re-plan, so the bytes stay stable", async () => {
299
299
  const msgs = [user("go"), asst("c1", "read", '{"path":"a.ts"}'), toolMsg("c1", "read", big("A")), ...PAD];
300
300
  const { edits } = planCompaction(msgs, CFG, 1, 10_000);
301
301
  const digested = edits.map((e) => ({ ...e, digest: DIGEST }));