world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Per-endpoint serving settings: `endpoint.toml`, next to the model's `policy.json`.
|
|
2
|
+
|
|
3
|
+
Why a file of its own rather than a key in the model's `config.toml`: that file is the world
|
|
4
|
+
model's BUILD configuration, rewritten by every build, and this is a live serving control an
|
|
5
|
+
operator (or the platform's slider) turns between builds. Keeping them apart means turning the
|
|
6
|
+
dial can never race a rebuild, and a rebuild can never quietly reset the dial.
|
|
7
|
+
|
|
8
|
+
Why not on `policy.json`: the policy is the optimizer's OUTPUT, and `wmo optimize route tune`
|
|
9
|
+
does write the dial into it. This file is the serving-side override for a policy the operator
|
|
10
|
+
does not want to rewrite (the common case for the platform, which serves artifacts it did not
|
|
11
|
+
fit). At mount time the file wins; with no file the policy is served exactly as fitted.
|
|
12
|
+
|
|
13
|
+
# .wmo/models/support-endpoint/endpoint.toml
|
|
14
|
+
cost_quality = 0.6
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import tomllib
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
import tomli_w
|
|
23
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
24
|
+
|
|
25
|
+
ENDPOINT_CONFIG_FILENAME = "endpoint.toml"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class EndpointConfig(BaseModel):
|
|
29
|
+
"""What an operator can set per endpoint without refitting anything.
|
|
30
|
+
|
|
31
|
+
`cost_quality` is the one dial (0.0 = max quality, 1.0 = max savings; see
|
|
32
|
+
`wmo.optimize.knn.apply_cost_quality`). None means "serve the policy as fitted", which is
|
|
33
|
+
also what an absent file means.
|
|
34
|
+
|
|
35
|
+
`extra="forbid"` for the same reason `PoolEntry` forbids it: a typo like `cost_qualty` must
|
|
36
|
+
fail at load with the key named, not be silently ignored and leave an operator staring at an
|
|
37
|
+
endpoint that ignored the dial they set.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
model_config = ConfigDict(extra="forbid")
|
|
41
|
+
|
|
42
|
+
cost_quality: float | None = Field(default=None, ge=0.0, le=1.0)
|
|
43
|
+
|
|
44
|
+
@classmethod
|
|
45
|
+
def load(cls, path: Path) -> EndpointConfig:
|
|
46
|
+
"""Read `endpoint.toml`; a missing file is the empty config, not an error."""
|
|
47
|
+
if not path.is_file():
|
|
48
|
+
return cls()
|
|
49
|
+
try:
|
|
50
|
+
data = tomllib.loads(path.read_text(encoding="utf-8"))
|
|
51
|
+
except tomllib.TOMLDecodeError as error:
|
|
52
|
+
raise ValueError(
|
|
53
|
+
f"invalid endpoint config at {path}: {error}. Expected TOML with at most a "
|
|
54
|
+
"`cost_quality` key between 0.0 and 1.0, and no other keys"
|
|
55
|
+
) from error
|
|
56
|
+
return cls.model_validate(data)
|
|
57
|
+
|
|
58
|
+
def save(self, path: Path) -> None:
|
|
59
|
+
"""Write the config atomically (a half-written dial must not be loadable)."""
|
|
60
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
61
|
+
staging = path.with_name(f"{path.name}.partial")
|
|
62
|
+
with staging.open("wb") as handle:
|
|
63
|
+
tomli_w.dump(self.model_dump(exclude_none=True), handle)
|
|
64
|
+
staging.replace(path)
|
wmo/serving/savings.py
ADDED
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
"""What an endpoint has saved so far, computed from its own request log.
|
|
2
|
+
|
|
3
|
+
The customer-facing counterpart of the cost/quality dial: the dial says what the endpoint is
|
|
4
|
+
TRYING to do, this says what it has actually done since it started serving. Cost is the honest
|
|
5
|
+
part (logged dollars against a priced counterfactual), latency is an estimate calibrated on the
|
|
6
|
+
endpoint's own traffic, and quality is a fitted expectation carried over from the offline
|
|
7
|
+
measurement, clearly labeled as such because live quality needs a feedback signal nobody is
|
|
8
|
+
sending yet.
|
|
9
|
+
|
|
10
|
+
Everything is recomputed from the persisted JSONL rows rather than accumulated in memory, so a
|
|
11
|
+
restart does not reset a customer's savings and a number can always be traced back to the rows
|
|
12
|
+
that produced it.
|
|
13
|
+
|
|
14
|
+
Every estimate carries its basis as a sentence in `estimate_basis`. Those strings render verbatim
|
|
15
|
+
in the customer UI, so they are written for a customer: no knob names, no internals, and no
|
|
16
|
+
number without a stated basis.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import statistics
|
|
22
|
+
from datetime import UTC, datetime, timedelta
|
|
23
|
+
from typing import TYPE_CHECKING, Literal
|
|
24
|
+
|
|
25
|
+
from pydantic import BaseModel, Field
|
|
26
|
+
|
|
27
|
+
from wmo.optimize.knn import COST_QUALITY_ANCHORS, COST_QUALITY_BALANCED, cost_quality_knobs
|
|
28
|
+
from wmo.providers.base import TokenUsage
|
|
29
|
+
|
|
30
|
+
if TYPE_CHECKING:
|
|
31
|
+
from collections.abc import Sequence
|
|
32
|
+
|
|
33
|
+
from wmo.optimize.policy import RoutingPolicy
|
|
34
|
+
from wmo.serving.chat import RequestLogRecord
|
|
35
|
+
|
|
36
|
+
SavingsWindow = Literal["all_time", "7d"]
|
|
37
|
+
|
|
38
|
+
WINDOW_DAYS = 7
|
|
39
|
+
|
|
40
|
+
# The sentences the response ships. Written as customer copy on purpose (see module docstring).
|
|
41
|
+
BASIS_COUNTERFACTUAL = (
|
|
42
|
+
"Savings compare what you were billed against what the same requests would have cost on "
|
|
43
|
+
"{fallback} alone, priced on the same number of tokens each request actually used, including "
|
|
44
|
+
"crediting {fallback} with the cached reads the model that served earned. Both assumptions "
|
|
45
|
+
"understate the saving rather than inflate it."
|
|
46
|
+
)
|
|
47
|
+
BASIS_NO_TRAFFIC = "This endpoint has not served any requests yet, so there is nothing to compare."
|
|
48
|
+
BASIS_LATENCY_SELF = (
|
|
49
|
+
"Time saved is an estimate: it compares each request that used a different model against "
|
|
50
|
+
"the median response time of this endpoint's own {fallback} requests, so it becomes more "
|
|
51
|
+
"accurate as the endpoint serves more traffic."
|
|
52
|
+
)
|
|
53
|
+
BASIS_LATENCY_NO_BASELINE = (
|
|
54
|
+
"Time saved is not shown yet: this endpoint has not served enough requests on {fallback} to "
|
|
55
|
+
"establish a response time to compare against."
|
|
56
|
+
)
|
|
57
|
+
BASIS_QUALITY_ANCHOR = (
|
|
58
|
+
"The quality figure is a fitted expectation from offline evaluation of this setting, not a "
|
|
59
|
+
"live measurement of your traffic."
|
|
60
|
+
)
|
|
61
|
+
BASIS_QUALITY_INTERPOLATED = (
|
|
62
|
+
"The quality figure is interpolated between the two nearest evaluated settings, so treat it "
|
|
63
|
+
"as a direction rather than a precise value."
|
|
64
|
+
)
|
|
65
|
+
BASIS_QUALITY_AS_FITTED = (
|
|
66
|
+
"This endpoint is serving the settings it was optimized with, which are our balanced "
|
|
67
|
+
"setting, so the quality figure is that setting's fitted expectation from offline "
|
|
68
|
+
"evaluation rather than a live measurement of your traffic."
|
|
69
|
+
)
|
|
70
|
+
BASIS_QUALITY_UNKNOWN = (
|
|
71
|
+
"No quality figure is shown: this endpoint was tuned by hand to settings we have not "
|
|
72
|
+
"evaluated, so there is no fitted expectation to quote for it."
|
|
73
|
+
)
|
|
74
|
+
BASIS_QUALITY_NO_DIAL = (
|
|
75
|
+
"No quality figure is shown: this endpoint sends every request to one model, so there is no "
|
|
76
|
+
"cost and quality setting to compare against."
|
|
77
|
+
)
|
|
78
|
+
BASIS_BILLING = "Your invoices remain the record of what you were charged."
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class EndpointSavings(BaseModel):
|
|
82
|
+
"""One endpoint's savings over one window, as the platform card renders it.
|
|
83
|
+
|
|
84
|
+
`requests_served` is a count of successfully served requests; a card with 0 there is the
|
|
85
|
+
empty state, and every other field is zero rather than null so a client never has to
|
|
86
|
+
special-case a missing key. `cost_saved_usd` and `cost_saved_pct` come from logged dollars
|
|
87
|
+
against the priced counterfactual; the two accounting fields behind them are included so the
|
|
88
|
+
subtraction is auditable. `time_saved_s_estimate` and `expected_quality_delta_pt` are
|
|
89
|
+
estimates, named so, and every estimate's basis is a sentence in `estimate_basis`.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
requests_served: int = Field(ge=0)
|
|
93
|
+
cost_saved_usd: float
|
|
94
|
+
cost_saved_pct: float
|
|
95
|
+
time_saved_s_estimate: float
|
|
96
|
+
expected_quality_delta_pt: float
|
|
97
|
+
estimate_basis: list[str]
|
|
98
|
+
window: SavingsWindow
|
|
99
|
+
# The two sums the cost saving is the difference of: logged spend, and the counterfactual.
|
|
100
|
+
actual_cost_usd: float = Field(ge=0.0)
|
|
101
|
+
baseline_cost_estimate_usd: float = Field(ge=0.0)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _in_window(record: RequestLogRecord, *, window: SavingsWindow, now: datetime) -> bool:
|
|
105
|
+
if window == "all_time":
|
|
106
|
+
return True
|
|
107
|
+
try:
|
|
108
|
+
stamped = datetime.fromisoformat(record.ts)
|
|
109
|
+
except ValueError:
|
|
110
|
+
# An unparseable timestamp cannot be placed in a bounded window. It still counts toward
|
|
111
|
+
# all-time, where no placement is needed.
|
|
112
|
+
return False
|
|
113
|
+
if stamped.tzinfo is None:
|
|
114
|
+
stamped = stamped.replace(tzinfo=UTC)
|
|
115
|
+
return stamped >= now - timedelta(days=WINDOW_DAYS)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _expected_quality(policy: RoutingPolicy) -> tuple[float, str]:
|
|
119
|
+
"""The fitted quality expectation for the endpoint's dial position, and its basis.
|
|
120
|
+
|
|
121
|
+
Exactly on an anchor: that anchor's measured delta. Between two anchors: a linear
|
|
122
|
+
interpolation, labeled as one. Dial never set: the balanced anchor, because a policy fitted
|
|
123
|
+
with the shipped defaults IS the balanced setting, which is checked against ALL FOUR knobs
|
|
124
|
+
the dial controls rather than assumed. The coverage knob matters most here: `floor_q` is the
|
|
125
|
+
only thing separating the balanced setting from the quality-max one, so a fit that set it
|
|
126
|
+
differently is a different operating point no matter how the other three read. A policy off
|
|
127
|
+
the dial, or one whose `floor_q` was never recorded, gets no figure at all: quoting the
|
|
128
|
+
balanced number for an endpoint someone tuned elsewhere would be a claim about an evaluation
|
|
129
|
+
that never ran.
|
|
130
|
+
"""
|
|
131
|
+
dial = policy.cost_quality
|
|
132
|
+
balanced = next(
|
|
133
|
+
anchor
|
|
134
|
+
for anchor in COST_QUALITY_ANCHORS
|
|
135
|
+
if abs(anchor.cost_quality - COST_QUALITY_BALANCED) < 1e-9
|
|
136
|
+
)
|
|
137
|
+
if policy.kind != "knn":
|
|
138
|
+
return 0.0, BASIS_QUALITY_NO_DIAL
|
|
139
|
+
if dial is None:
|
|
140
|
+
default_knobs = cost_quality_knobs(COST_QUALITY_BALANCED)
|
|
141
|
+
as_fitted_is_balanced = (
|
|
142
|
+
policy.floor_q is not None
|
|
143
|
+
and abs(policy.floor_q - default_knobs.floor_q) < 1e-9
|
|
144
|
+
and policy.knn_z == default_knobs.knn_z
|
|
145
|
+
and policy.pick_lam == default_knobs.pick_lam
|
|
146
|
+
and policy.guard_mode == default_knobs.guard_mode
|
|
147
|
+
)
|
|
148
|
+
if as_fitted_is_balanced:
|
|
149
|
+
return balanced.quality_delta_points, BASIS_QUALITY_AS_FITTED
|
|
150
|
+
return 0.0, BASIS_QUALITY_UNKNOWN
|
|
151
|
+
anchors = sorted(COST_QUALITY_ANCHORS, key=lambda anchor: anchor.cost_quality)
|
|
152
|
+
for anchor in anchors:
|
|
153
|
+
if abs(anchor.cost_quality - dial) < 1e-9:
|
|
154
|
+
return anchor.quality_delta_points, BASIS_QUALITY_ANCHOR
|
|
155
|
+
below = [anchor for anchor in anchors if anchor.cost_quality < dial]
|
|
156
|
+
above = [anchor for anchor in anchors if anchor.cost_quality > dial]
|
|
157
|
+
if not below or not above:
|
|
158
|
+
# Unreachable while the anchors span the full dial (a test pins that they do); if that
|
|
159
|
+
# ever changes, say nothing rather than extrapolate off the end of the measurement.
|
|
160
|
+
return 0.0, BASIS_QUALITY_UNKNOWN
|
|
161
|
+
low, high = below[-1], above[0]
|
|
162
|
+
span = high.cost_quality - low.cost_quality
|
|
163
|
+
weight = (dial - low.cost_quality) / span
|
|
164
|
+
delta = (1.0 - weight) * low.quality_delta_points + weight * high.quality_delta_points
|
|
165
|
+
return delta, BASIS_QUALITY_INTERPOLATED
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def compute_savings(
|
|
169
|
+
records: Sequence[RequestLogRecord],
|
|
170
|
+
policy: RoutingPolicy,
|
|
171
|
+
*,
|
|
172
|
+
window: SavingsWindow = "all_time",
|
|
173
|
+
now: datetime | None = None,
|
|
174
|
+
) -> EndpointSavings:
|
|
175
|
+
"""Total up what this endpoint saved over `window`, from its logged rows.
|
|
176
|
+
|
|
177
|
+
The counterfactual is `policy.guard_model` (the fallback the endpoint would have served
|
|
178
|
+
every request on without a policy), priced by its own pool entry on each request's ACTUAL
|
|
179
|
+
token counts. That is an assumption, not a measurement: a different model would have emitted
|
|
180
|
+
a different number of output tokens, and its prompt cache would have been its own. It is the
|
|
181
|
+
assumption the response states, and it is the conservative direction for a router that
|
|
182
|
+
routes toward cheaper models, since those models tend to be the wordier ones.
|
|
183
|
+
|
|
184
|
+
Latency has no counterfactual price list, so its baseline is measured instead: the median
|
|
185
|
+
latency of this endpoint's OWN fallback-served requests. That self-calibrates as traffic
|
|
186
|
+
accrues, and until there are fallback requests to take a median of, no figure is reported.
|
|
187
|
+
Differences are summed signed, so a routed model that ran slower subtracts. Requests that
|
|
188
|
+
failed are excluded from every total: nobody was served.
|
|
189
|
+
"""
|
|
190
|
+
entries = {entry.name: entry for entry in policy.pool}
|
|
191
|
+
fallback = policy.guard_model or policy.default_model
|
|
192
|
+
served = [
|
|
193
|
+
record
|
|
194
|
+
for record in records
|
|
195
|
+
if record.status == "ok" and _in_window(record, window=window, now=now or datetime.now(UTC))
|
|
196
|
+
]
|
|
197
|
+
basis: list[str] = []
|
|
198
|
+
if not served:
|
|
199
|
+
return EndpointSavings(
|
|
200
|
+
requests_served=0,
|
|
201
|
+
cost_saved_usd=0.0,
|
|
202
|
+
cost_saved_pct=0.0,
|
|
203
|
+
time_saved_s_estimate=0.0,
|
|
204
|
+
expected_quality_delta_pt=0.0,
|
|
205
|
+
estimate_basis=[BASIS_NO_TRAFFIC, BASIS_BILLING],
|
|
206
|
+
window=window,
|
|
207
|
+
actual_cost_usd=0.0,
|
|
208
|
+
baseline_cost_estimate_usd=0.0,
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
actual = sum(record.cost_usd for record in served)
|
|
212
|
+
baseline_entry = entries.get(fallback)
|
|
213
|
+
baseline = actual
|
|
214
|
+
if baseline_entry is not None:
|
|
215
|
+
baseline = sum(
|
|
216
|
+
baseline_entry.cost_usd(
|
|
217
|
+
TokenUsage(
|
|
218
|
+
input_tokens=record.input_tokens,
|
|
219
|
+
output_tokens=record.output_tokens,
|
|
220
|
+
cached_input_tokens=record.cached_tokens,
|
|
221
|
+
)
|
|
222
|
+
)
|
|
223
|
+
for record in served
|
|
224
|
+
)
|
|
225
|
+
basis.append(BASIS_COUNTERFACTUAL.format(fallback=fallback))
|
|
226
|
+
|
|
227
|
+
on_fallback = [record.latency_ms for record in served if record.model == fallback]
|
|
228
|
+
routed_away = [record for record in served if record.model != fallback]
|
|
229
|
+
time_saved = 0.0
|
|
230
|
+
if on_fallback and routed_away:
|
|
231
|
+
fallback_p50 = statistics.median(on_fallback)
|
|
232
|
+
time_saved = sum(fallback_p50 - record.latency_ms for record in routed_away) / 1000.0
|
|
233
|
+
basis.append(BASIS_LATENCY_SELF.format(fallback=fallback))
|
|
234
|
+
elif routed_away:
|
|
235
|
+
basis.append(BASIS_LATENCY_NO_BASELINE.format(fallback=fallback))
|
|
236
|
+
|
|
237
|
+
quality, quality_basis = _expected_quality(policy)
|
|
238
|
+
basis.append(quality_basis)
|
|
239
|
+
basis.append(BASIS_BILLING)
|
|
240
|
+
return EndpointSavings(
|
|
241
|
+
requests_served=len(served),
|
|
242
|
+
cost_saved_usd=baseline - actual,
|
|
243
|
+
cost_saved_pct=((baseline - actual) / baseline * 100.0) if baseline > 0.0 else 0.0,
|
|
244
|
+
time_saved_s_estimate=time_saved,
|
|
245
|
+
expected_quality_delta_pt=quality,
|
|
246
|
+
estimate_basis=basis,
|
|
247
|
+
window=window,
|
|
248
|
+
actual_cost_usd=actual,
|
|
249
|
+
baseline_cost_estimate_usd=baseline,
|
|
250
|
+
)
|