world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,64 @@
1
+ """Per-endpoint serving settings: `endpoint.toml`, next to the model's `policy.json`.
2
+
3
+ Why a file of its own rather than a key in the model's `config.toml`: that file is the world
4
+ model's BUILD configuration, rewritten by every build, and this is a live serving control an
5
+ operator (or the platform's slider) turns between builds. Keeping them apart means turning the
6
+ dial can never race a rebuild, and a rebuild can never quietly reset the dial.
7
+
8
+ Why not on `policy.json`: the policy is the optimizer's OUTPUT, and `wmo optimize route tune`
9
+ does write the dial into it. This file is the serving-side override for a policy the operator
10
+ does not want to rewrite (the common case for the platform, which serves artifacts it did not
11
+ fit). At mount time the file wins; with no file the policy is served exactly as fitted.
12
+
13
+ # .wmo/models/support-endpoint/endpoint.toml
14
+ cost_quality = 0.6
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import tomllib
20
+ from pathlib import Path
21
+
22
+ import tomli_w
23
+ from pydantic import BaseModel, ConfigDict, Field
24
+
25
+ ENDPOINT_CONFIG_FILENAME = "endpoint.toml"
26
+
27
+
28
+ class EndpointConfig(BaseModel):
29
+ """What an operator can set per endpoint without refitting anything.
30
+
31
+ `cost_quality` is the one dial (0.0 = max quality, 1.0 = max savings; see
32
+ `wmo.optimize.knn.apply_cost_quality`). None means "serve the policy as fitted", which is
33
+ also what an absent file means.
34
+
35
+ `extra="forbid"` for the same reason `PoolEntry` forbids it: a typo like `cost_qualty` must
36
+ fail at load with the key named, not be silently ignored and leave an operator staring at an
37
+ endpoint that ignored the dial they set.
38
+ """
39
+
40
+ model_config = ConfigDict(extra="forbid")
41
+
42
+ cost_quality: float | None = Field(default=None, ge=0.0, le=1.0)
43
+
44
+ @classmethod
45
+ def load(cls, path: Path) -> EndpointConfig:
46
+ """Read `endpoint.toml`; a missing file is the empty config, not an error."""
47
+ if not path.is_file():
48
+ return cls()
49
+ try:
50
+ data = tomllib.loads(path.read_text(encoding="utf-8"))
51
+ except tomllib.TOMLDecodeError as error:
52
+ raise ValueError(
53
+ f"invalid endpoint config at {path}: {error}. Expected TOML with at most a "
54
+ "`cost_quality` key between 0.0 and 1.0, and no other keys"
55
+ ) from error
56
+ return cls.model_validate(data)
57
+
58
+ def save(self, path: Path) -> None:
59
+ """Write the config atomically (a half-written dial must not be loadable)."""
60
+ path.parent.mkdir(parents=True, exist_ok=True)
61
+ staging = path.with_name(f"{path.name}.partial")
62
+ with staging.open("wb") as handle:
63
+ tomli_w.dump(self.model_dump(exclude_none=True), handle)
64
+ staging.replace(path)
wmo/serving/savings.py ADDED
@@ -0,0 +1,250 @@
1
+ """What an endpoint has saved so far, computed from its own request log.
2
+
3
+ The customer-facing counterpart of the cost/quality dial: the dial says what the endpoint is
4
+ TRYING to do, this says what it has actually done since it started serving. Cost is the honest
5
+ part (logged dollars against a priced counterfactual), latency is an estimate calibrated on the
6
+ endpoint's own traffic, and quality is a fitted expectation carried over from the offline
7
+ measurement, clearly labeled as such because live quality needs a feedback signal nobody is
8
+ sending yet.
9
+
10
+ Everything is recomputed from the persisted JSONL rows rather than accumulated in memory, so a
11
+ restart does not reset a customer's savings and a number can always be traced back to the rows
12
+ that produced it.
13
+
14
+ Every estimate carries its basis as a sentence in `estimate_basis`. Those strings render verbatim
15
+ in the customer UI, so they are written for a customer: no knob names, no internals, and no
16
+ number without a stated basis.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import statistics
22
+ from datetime import UTC, datetime, timedelta
23
+ from typing import TYPE_CHECKING, Literal
24
+
25
+ from pydantic import BaseModel, Field
26
+
27
+ from wmo.optimize.knn import COST_QUALITY_ANCHORS, COST_QUALITY_BALANCED, cost_quality_knobs
28
+ from wmo.providers.base import TokenUsage
29
+
30
+ if TYPE_CHECKING:
31
+ from collections.abc import Sequence
32
+
33
+ from wmo.optimize.policy import RoutingPolicy
34
+ from wmo.serving.chat import RequestLogRecord
35
+
36
+ SavingsWindow = Literal["all_time", "7d"]
37
+
38
+ WINDOW_DAYS = 7
39
+
40
+ # The sentences the response ships. Written as customer copy on purpose (see module docstring).
41
+ BASIS_COUNTERFACTUAL = (
42
+ "Savings compare what you were billed against what the same requests would have cost on "
43
+ "{fallback} alone, priced on the same number of tokens each request actually used, including "
44
+ "crediting {fallback} with the cached reads the model that served earned. Both assumptions "
45
+ "understate the saving rather than inflate it."
46
+ )
47
+ BASIS_NO_TRAFFIC = "This endpoint has not served any requests yet, so there is nothing to compare."
48
+ BASIS_LATENCY_SELF = (
49
+ "Time saved is an estimate: it compares each request that used a different model against "
50
+ "the median response time of this endpoint's own {fallback} requests, so it becomes more "
51
+ "accurate as the endpoint serves more traffic."
52
+ )
53
+ BASIS_LATENCY_NO_BASELINE = (
54
+ "Time saved is not shown yet: this endpoint has not served enough requests on {fallback} to "
55
+ "establish a response time to compare against."
56
+ )
57
+ BASIS_QUALITY_ANCHOR = (
58
+ "The quality figure is a fitted expectation from offline evaluation of this setting, not a "
59
+ "live measurement of your traffic."
60
+ )
61
+ BASIS_QUALITY_INTERPOLATED = (
62
+ "The quality figure is interpolated between the two nearest evaluated settings, so treat it "
63
+ "as a direction rather than a precise value."
64
+ )
65
+ BASIS_QUALITY_AS_FITTED = (
66
+ "This endpoint is serving the settings it was optimized with, which are our balanced "
67
+ "setting, so the quality figure is that setting's fitted expectation from offline "
68
+ "evaluation rather than a live measurement of your traffic."
69
+ )
70
+ BASIS_QUALITY_UNKNOWN = (
71
+ "No quality figure is shown: this endpoint was tuned by hand to settings we have not "
72
+ "evaluated, so there is no fitted expectation to quote for it."
73
+ )
74
+ BASIS_QUALITY_NO_DIAL = (
75
+ "No quality figure is shown: this endpoint sends every request to one model, so there is no "
76
+ "cost and quality setting to compare against."
77
+ )
78
+ BASIS_BILLING = "Your invoices remain the record of what you were charged."
79
+
80
+
81
+ class EndpointSavings(BaseModel):
82
+ """One endpoint's savings over one window, as the platform card renders it.
83
+
84
+ `requests_served` is a count of successfully served requests; a card with 0 there is the
85
+ empty state, and every other field is zero rather than null so a client never has to
86
+ special-case a missing key. `cost_saved_usd` and `cost_saved_pct` come from logged dollars
87
+ against the priced counterfactual; the two accounting fields behind them are included so the
88
+ subtraction is auditable. `time_saved_s_estimate` and `expected_quality_delta_pt` are
89
+ estimates, named so, and every estimate's basis is a sentence in `estimate_basis`.
90
+ """
91
+
92
+ requests_served: int = Field(ge=0)
93
+ cost_saved_usd: float
94
+ cost_saved_pct: float
95
+ time_saved_s_estimate: float
96
+ expected_quality_delta_pt: float
97
+ estimate_basis: list[str]
98
+ window: SavingsWindow
99
+ # The two sums the cost saving is the difference of: logged spend, and the counterfactual.
100
+ actual_cost_usd: float = Field(ge=0.0)
101
+ baseline_cost_estimate_usd: float = Field(ge=0.0)
102
+
103
+
104
+ def _in_window(record: RequestLogRecord, *, window: SavingsWindow, now: datetime) -> bool:
105
+ if window == "all_time":
106
+ return True
107
+ try:
108
+ stamped = datetime.fromisoformat(record.ts)
109
+ except ValueError:
110
+ # An unparseable timestamp cannot be placed in a bounded window. It still counts toward
111
+ # all-time, where no placement is needed.
112
+ return False
113
+ if stamped.tzinfo is None:
114
+ stamped = stamped.replace(tzinfo=UTC)
115
+ return stamped >= now - timedelta(days=WINDOW_DAYS)
116
+
117
+
118
+ def _expected_quality(policy: RoutingPolicy) -> tuple[float, str]:
119
+ """The fitted quality expectation for the endpoint's dial position, and its basis.
120
+
121
+ Exactly on an anchor: that anchor's measured delta. Between two anchors: a linear
122
+ interpolation, labeled as one. Dial never set: the balanced anchor, because a policy fitted
123
+ with the shipped defaults IS the balanced setting, which is checked against ALL FOUR knobs
124
+ the dial controls rather than assumed. The coverage knob matters most here: `floor_q` is the
125
+ only thing separating the balanced setting from the quality-max one, so a fit that set it
126
+ differently is a different operating point no matter how the other three read. A policy off
127
+ the dial, or one whose `floor_q` was never recorded, gets no figure at all: quoting the
128
+ balanced number for an endpoint someone tuned elsewhere would be a claim about an evaluation
129
+ that never ran.
130
+ """
131
+ dial = policy.cost_quality
132
+ balanced = next(
133
+ anchor
134
+ for anchor in COST_QUALITY_ANCHORS
135
+ if abs(anchor.cost_quality - COST_QUALITY_BALANCED) < 1e-9
136
+ )
137
+ if policy.kind != "knn":
138
+ return 0.0, BASIS_QUALITY_NO_DIAL
139
+ if dial is None:
140
+ default_knobs = cost_quality_knobs(COST_QUALITY_BALANCED)
141
+ as_fitted_is_balanced = (
142
+ policy.floor_q is not None
143
+ and abs(policy.floor_q - default_knobs.floor_q) < 1e-9
144
+ and policy.knn_z == default_knobs.knn_z
145
+ and policy.pick_lam == default_knobs.pick_lam
146
+ and policy.guard_mode == default_knobs.guard_mode
147
+ )
148
+ if as_fitted_is_balanced:
149
+ return balanced.quality_delta_points, BASIS_QUALITY_AS_FITTED
150
+ return 0.0, BASIS_QUALITY_UNKNOWN
151
+ anchors = sorted(COST_QUALITY_ANCHORS, key=lambda anchor: anchor.cost_quality)
152
+ for anchor in anchors:
153
+ if abs(anchor.cost_quality - dial) < 1e-9:
154
+ return anchor.quality_delta_points, BASIS_QUALITY_ANCHOR
155
+ below = [anchor for anchor in anchors if anchor.cost_quality < dial]
156
+ above = [anchor for anchor in anchors if anchor.cost_quality > dial]
157
+ if not below or not above:
158
+ # Unreachable while the anchors span the full dial (a test pins that they do); if that
159
+ # ever changes, say nothing rather than extrapolate off the end of the measurement.
160
+ return 0.0, BASIS_QUALITY_UNKNOWN
161
+ low, high = below[-1], above[0]
162
+ span = high.cost_quality - low.cost_quality
163
+ weight = (dial - low.cost_quality) / span
164
+ delta = (1.0 - weight) * low.quality_delta_points + weight * high.quality_delta_points
165
+ return delta, BASIS_QUALITY_INTERPOLATED
166
+
167
+
168
+ def compute_savings(
169
+ records: Sequence[RequestLogRecord],
170
+ policy: RoutingPolicy,
171
+ *,
172
+ window: SavingsWindow = "all_time",
173
+ now: datetime | None = None,
174
+ ) -> EndpointSavings:
175
+ """Total up what this endpoint saved over `window`, from its logged rows.
176
+
177
+ The counterfactual is `policy.guard_model` (the fallback the endpoint would have served
178
+ every request on without a policy), priced by its own pool entry on each request's ACTUAL
179
+ token counts. That is an assumption, not a measurement: a different model would have emitted
180
+ a different number of output tokens, and its prompt cache would have been its own. It is the
181
+ assumption the response states, and it is the conservative direction for a router that
182
+ routes toward cheaper models, since those models tend to be the wordier ones.
183
+
184
+ Latency has no counterfactual price list, so its baseline is measured instead: the median
185
+ latency of this endpoint's OWN fallback-served requests. That self-calibrates as traffic
186
+ accrues, and until there are fallback requests to take a median of, no figure is reported.
187
+ Differences are summed signed, so a routed model that ran slower subtracts. Requests that
188
+ failed are excluded from every total: nobody was served.
189
+ """
190
+ entries = {entry.name: entry for entry in policy.pool}
191
+ fallback = policy.guard_model or policy.default_model
192
+ served = [
193
+ record
194
+ for record in records
195
+ if record.status == "ok" and _in_window(record, window=window, now=now or datetime.now(UTC))
196
+ ]
197
+ basis: list[str] = []
198
+ if not served:
199
+ return EndpointSavings(
200
+ requests_served=0,
201
+ cost_saved_usd=0.0,
202
+ cost_saved_pct=0.0,
203
+ time_saved_s_estimate=0.0,
204
+ expected_quality_delta_pt=0.0,
205
+ estimate_basis=[BASIS_NO_TRAFFIC, BASIS_BILLING],
206
+ window=window,
207
+ actual_cost_usd=0.0,
208
+ baseline_cost_estimate_usd=0.0,
209
+ )
210
+
211
+ actual = sum(record.cost_usd for record in served)
212
+ baseline_entry = entries.get(fallback)
213
+ baseline = actual
214
+ if baseline_entry is not None:
215
+ baseline = sum(
216
+ baseline_entry.cost_usd(
217
+ TokenUsage(
218
+ input_tokens=record.input_tokens,
219
+ output_tokens=record.output_tokens,
220
+ cached_input_tokens=record.cached_tokens,
221
+ )
222
+ )
223
+ for record in served
224
+ )
225
+ basis.append(BASIS_COUNTERFACTUAL.format(fallback=fallback))
226
+
227
+ on_fallback = [record.latency_ms for record in served if record.model == fallback]
228
+ routed_away = [record for record in served if record.model != fallback]
229
+ time_saved = 0.0
230
+ if on_fallback and routed_away:
231
+ fallback_p50 = statistics.median(on_fallback)
232
+ time_saved = sum(fallback_p50 - record.latency_ms for record in routed_away) / 1000.0
233
+ basis.append(BASIS_LATENCY_SELF.format(fallback=fallback))
234
+ elif routed_away:
235
+ basis.append(BASIS_LATENCY_NO_BASELINE.format(fallback=fallback))
236
+
237
+ quality, quality_basis = _expected_quality(policy)
238
+ basis.append(quality_basis)
239
+ basis.append(BASIS_BILLING)
240
+ return EndpointSavings(
241
+ requests_served=len(served),
242
+ cost_saved_usd=baseline - actual,
243
+ cost_saved_pct=((baseline - actual) / baseline * 100.0) if baseline > 0.0 else 0.0,
244
+ time_saved_s_estimate=time_saved,
245
+ expected_quality_delta_pt=quality,
246
+ estimate_basis=basis,
247
+ window=window,
248
+ actual_cost_usd=actual,
249
+ baseline_cost_estimate_usd=baseline,
250
+ )