world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
llm_waterfall/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Experiential Labs
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""llm-waterfall: stateless LLM failover across an ordered chain of backends.
|
|
2
|
+
|
|
3
|
+
Capacity errors (throttling, transient 5xx, timeouts) spill to the next backend; real client
|
|
4
|
+
errors raise immediately. Every call returns which backend served it, token usage, USD cost, and
|
|
5
|
+
the full attempt trail.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from llm_waterfall.classify import is_capacity_error, outcome_for
|
|
9
|
+
from llm_waterfall.pricing import ModelPrice, cost_usd, price_for
|
|
10
|
+
from llm_waterfall.types import (
|
|
11
|
+
Attempt,
|
|
12
|
+
Backend,
|
|
13
|
+
ChatMaxTokensField,
|
|
14
|
+
ChatRequest,
|
|
15
|
+
ChatResponse,
|
|
16
|
+
ChatResult,
|
|
17
|
+
CompletionResult,
|
|
18
|
+
EmbeddingResult,
|
|
19
|
+
EmbeddingsUnsupported,
|
|
20
|
+
Message,
|
|
21
|
+
RetryPolicy,
|
|
22
|
+
TokenUsage,
|
|
23
|
+
ToolCallingUnsupported,
|
|
24
|
+
VerifyResult,
|
|
25
|
+
WaterfallExhausted,
|
|
26
|
+
)
|
|
27
|
+
from llm_waterfall.waterfall import Waterfall
|
|
28
|
+
|
|
29
|
+
__version__ = "0.1.4"
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"Attempt",
|
|
33
|
+
"Backend",
|
|
34
|
+
"ChatMaxTokensField",
|
|
35
|
+
"ChatRequest",
|
|
36
|
+
"ChatResponse",
|
|
37
|
+
"ChatResult",
|
|
38
|
+
"CompletionResult",
|
|
39
|
+
"EmbeddingResult",
|
|
40
|
+
"EmbeddingsUnsupported",
|
|
41
|
+
"Message",
|
|
42
|
+
"ModelPrice",
|
|
43
|
+
"RetryPolicy",
|
|
44
|
+
"TokenUsage",
|
|
45
|
+
"ToolCallingUnsupported",
|
|
46
|
+
"VerifyResult",
|
|
47
|
+
"Waterfall",
|
|
48
|
+
"WaterfallExhausted",
|
|
49
|
+
"cost_usd",
|
|
50
|
+
"is_capacity_error",
|
|
51
|
+
"outcome_for",
|
|
52
|
+
"price_for",
|
|
53
|
+
]
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Adapter registry: map a Backend's provider name to its adapter class.
|
|
2
|
+
|
|
3
|
+
Adapter modules import at module scope (they gate their SDK imports internally, so this package
|
|
4
|
+
imports with zero SDKs installed); only the SDKs themselves are lazy.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
|
|
11
|
+
from llm_waterfall.adapters.anthropic import AnthropicAdapter
|
|
12
|
+
from llm_waterfall.adapters.aws_mantle import AwsMantleAdapter
|
|
13
|
+
from llm_waterfall.adapters.azure_openai import AzureOpenAIAdapter
|
|
14
|
+
from llm_waterfall.adapters.base import Adapter
|
|
15
|
+
from llm_waterfall.adapters.bedrock import BedrockAdapter
|
|
16
|
+
from llm_waterfall.adapters.openai import OpenAIAdapter
|
|
17
|
+
from llm_waterfall.types import PROVIDERS, Backend
|
|
18
|
+
|
|
19
|
+
_ADAPTERS: dict[str, Callable[[Backend], Adapter]] = {
|
|
20
|
+
"bedrock": BedrockAdapter,
|
|
21
|
+
"openai": OpenAIAdapter,
|
|
22
|
+
"anthropic": AnthropicAdapter,
|
|
23
|
+
"azure_openai": AzureOpenAIAdapter, # real; construction validates endpoint/api_version
|
|
24
|
+
"aws_mantle": AwsMantleAdapter, # stub: raises NotImplementedError at construction
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def build_adapter(backend: Backend) -> Adapter:
|
|
29
|
+
"""Construct the adapter for `backend`. Unimplemented providers fail here, fast."""
|
|
30
|
+
try:
|
|
31
|
+
factory = _ADAPTERS[backend.provider]
|
|
32
|
+
except KeyError:
|
|
33
|
+
raise ValueError(
|
|
34
|
+
f"unknown provider {backend.provider!r}; expected one of {', '.join(PROVIDERS)}"
|
|
35
|
+
) from None
|
|
36
|
+
return factory(backend)
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Anthropic direct-API adapter. Reads ANTHROPIC_API_KEY from the environment."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import threading
|
|
6
|
+
from typing import TYPE_CHECKING, cast
|
|
7
|
+
|
|
8
|
+
from llm_waterfall.adapters.base import missing_sdk_error
|
|
9
|
+
from llm_waterfall.types import (
|
|
10
|
+
Backend,
|
|
11
|
+
ChatRequest,
|
|
12
|
+
ChatResponse,
|
|
13
|
+
EmbeddingsUnsupported,
|
|
14
|
+
Message,
|
|
15
|
+
TokenUsage,
|
|
16
|
+
ToolCallingUnsupported,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
if TYPE_CHECKING:
|
|
20
|
+
from anthropic import Anthropic
|
|
21
|
+
from anthropic.types import MessageParam
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class AnthropicAdapter:
|
|
25
|
+
"""Claude via the direct Anthropic Messages API."""
|
|
26
|
+
|
|
27
|
+
def __init__(self, backend: Backend) -> None:
|
|
28
|
+
self.backend = backend
|
|
29
|
+
self._client: Anthropic | None = None
|
|
30
|
+
self._lock = threading.Lock()
|
|
31
|
+
|
|
32
|
+
def _get_client(self) -> Anthropic:
|
|
33
|
+
if self._client is None:
|
|
34
|
+
with self._lock:
|
|
35
|
+
if self._client is None:
|
|
36
|
+
try:
|
|
37
|
+
import httpx
|
|
38
|
+
from anthropic import Anthropic
|
|
39
|
+
except ModuleNotFoundError as exc:
|
|
40
|
+
raise missing_sdk_error("anthropic", "anthropic") from exc
|
|
41
|
+
|
|
42
|
+
# max_retries=0: the waterfall owns retry policy. Granular httpx.Timeout so
|
|
43
|
+
# a dead endpoint fails over after connect_timeout_s, not read_timeout_s.
|
|
44
|
+
self._client = Anthropic(
|
|
45
|
+
max_retries=0,
|
|
46
|
+
timeout=httpx.Timeout(
|
|
47
|
+
self.backend.read_timeout_s,
|
|
48
|
+
connect=self.backend.connect_timeout_s,
|
|
49
|
+
),
|
|
50
|
+
)
|
|
51
|
+
return self._client
|
|
52
|
+
|
|
53
|
+
def complete(
|
|
54
|
+
self,
|
|
55
|
+
system: str,
|
|
56
|
+
messages: list[Message],
|
|
57
|
+
*,
|
|
58
|
+
temperature: float | None,
|
|
59
|
+
max_tokens: int,
|
|
60
|
+
) -> tuple[str, TokenUsage]:
|
|
61
|
+
"""One Messages API call; system is a top-level arg (Anthropic-native)."""
|
|
62
|
+
api_messages = [
|
|
63
|
+
cast("MessageParam", {"role": m.role, "content": m.content}) for m in messages
|
|
64
|
+
]
|
|
65
|
+
client_messages = self._get_client().messages
|
|
66
|
+
# Claude 4.7+ rejects sampling params; only forward temperature when explicitly set.
|
|
67
|
+
if temperature is None:
|
|
68
|
+
response = client_messages.create(
|
|
69
|
+
model=self.backend.model,
|
|
70
|
+
system=system,
|
|
71
|
+
messages=api_messages,
|
|
72
|
+
max_tokens=max_tokens,
|
|
73
|
+
)
|
|
74
|
+
else:
|
|
75
|
+
response = client_messages.create(
|
|
76
|
+
model=self.backend.model,
|
|
77
|
+
system=system,
|
|
78
|
+
messages=api_messages,
|
|
79
|
+
max_tokens=max_tokens,
|
|
80
|
+
temperature=temperature,
|
|
81
|
+
)
|
|
82
|
+
text = "".join(block.text for block in response.content if block.type == "text")
|
|
83
|
+
usage = TokenUsage(
|
|
84
|
+
input_tokens=response.usage.input_tokens,
|
|
85
|
+
output_tokens=response.usage.output_tokens,
|
|
86
|
+
)
|
|
87
|
+
return text, usage
|
|
88
|
+
|
|
89
|
+
def complete_chat(self, request: ChatRequest) -> ChatResponse:
|
|
90
|
+
"""Direct Anthropic structured mapping is not implemented yet."""
|
|
91
|
+
del request
|
|
92
|
+
raise ToolCallingUnsupported(
|
|
93
|
+
"the anthropic adapter does not yet map OpenAI-compatible tool calls; "
|
|
94
|
+
"put an openai, azure_openai, or bedrock backend in the chain"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
def embed_model_id(self) -> str | None:
|
|
98
|
+
"""Anthropic has no embeddings API."""
|
|
99
|
+
return None
|
|
100
|
+
|
|
101
|
+
def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
|
|
102
|
+
raise EmbeddingsUnsupported(
|
|
103
|
+
"Anthropic has no embeddings API; put a bedrock or openai backend in the chain "
|
|
104
|
+
"for embeddings (the waterfall skips this backend for embed calls)."
|
|
105
|
+
)
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""AWS Mantle adapter — stub that fails at construction.
|
|
2
|
+
|
|
3
|
+
TODO: implement against the Bedrock Mantle client (`anthropic.AnthropicBedrockMantle` — the
|
|
4
|
+
Messages-API Bedrock endpoint) with `aws_region=backend.region`, credentials via
|
|
5
|
+
`boto3.Session(profile_name=backend.profile)`, `max_retries=0`, and the backend's timeouts.
|
|
6
|
+
Model ids take an `anthropic.` prefix. Classification already covers its error shapes (the
|
|
7
|
+
anthropic SDK exception types + botocore codes).
|
|
8
|
+
|
|
9
|
+
Failing in `__init__` (not at call time) is deliberate: an unimplemented rung must break
|
|
10
|
+
`Waterfall(...)` construction loudly, never abort a live call mid-chain.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from llm_waterfall.types import Backend, ChatRequest, ChatResponse, Message, TokenUsage
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class AwsMantleAdapter:
|
|
19
|
+
"""Not implemented yet; constructing one fails fast with the workaround."""
|
|
20
|
+
|
|
21
|
+
backend: Backend # satisfies the Adapter protocol; never assigned (init raises)
|
|
22
|
+
|
|
23
|
+
def __init__(self, backend: Backend) -> None:
|
|
24
|
+
raise NotImplementedError(
|
|
25
|
+
"the aws_mantle adapter is not implemented yet; use a 'bedrock' backend for AWS "
|
|
26
|
+
"traffic, or contribute the adapter (see the TODO in "
|
|
27
|
+
"llm_waterfall/adapters/aws_mantle.py)."
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
def complete(
|
|
31
|
+
self,
|
|
32
|
+
system: str,
|
|
33
|
+
messages: list[Message],
|
|
34
|
+
*,
|
|
35
|
+
temperature: float | None,
|
|
36
|
+
max_tokens: int,
|
|
37
|
+
) -> tuple[str, TokenUsage]:
|
|
38
|
+
raise NotImplementedError # pragma: no cover - unreachable (init raises)
|
|
39
|
+
|
|
40
|
+
def complete_chat(self, request: ChatRequest) -> ChatResponse:
|
|
41
|
+
raise NotImplementedError # pragma: no cover - unreachable (init raises)
|
|
42
|
+
|
|
43
|
+
def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
|
|
44
|
+
raise NotImplementedError # pragma: no cover - unreachable (init raises)
|
|
45
|
+
|
|
46
|
+
def embed_model_id(self) -> str | None:
|
|
47
|
+
raise NotImplementedError # pragma: no cover - unreachable (init raises)
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""Azure OpenAI adapter: the OpenAI wire mapping behind an AzureOpenAI client.
|
|
2
|
+
|
|
3
|
+
Requests route by DEPLOYMENT name (`Backend.deployment`, defaulting to `Backend.model`); the key
|
|
4
|
+
is read from AZURE_OPENAI_API_KEY. Construction fails fast when `endpoint`/`api_version` are
|
|
5
|
+
missing — a misconfigured rung must break `Waterfall(...)`, not a live call mid-chain.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import TYPE_CHECKING
|
|
11
|
+
|
|
12
|
+
from llm_waterfall.adapters.base import missing_sdk_error
|
|
13
|
+
from llm_waterfall.adapters.openai import OpenAIAdapter
|
|
14
|
+
from llm_waterfall.types import Backend, EmbeddingsUnsupported, TokenUsage
|
|
15
|
+
|
|
16
|
+
if TYPE_CHECKING:
|
|
17
|
+
from openai import OpenAI
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class AzureOpenAIAdapter(OpenAIAdapter):
|
|
21
|
+
"""GPT-5.x Azure deployments; embeddings via an Azure embedding deployment."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, backend: Backend) -> None:
|
|
24
|
+
if not backend.endpoint:
|
|
25
|
+
raise ValueError(
|
|
26
|
+
"azure_openai backends need endpoint= (e.g. https://<resource>.openai.azure.com)"
|
|
27
|
+
)
|
|
28
|
+
if not backend.api_version:
|
|
29
|
+
raise ValueError("azure_openai backends need api_version= (e.g. 2024-12-01-preview)")
|
|
30
|
+
# Validated non-None copies (the dataclass fields stay Optional for other providers).
|
|
31
|
+
self._azure_endpoint: str = backend.endpoint
|
|
32
|
+
self._api_version: str = backend.api_version
|
|
33
|
+
super().__init__(backend)
|
|
34
|
+
|
|
35
|
+
def _get_client(self) -> OpenAI:
|
|
36
|
+
if self._client is None:
|
|
37
|
+
with self._lock:
|
|
38
|
+
if self._client is None:
|
|
39
|
+
try:
|
|
40
|
+
import httpx
|
|
41
|
+
from openai import AzureOpenAI
|
|
42
|
+
except ModuleNotFoundError as exc:
|
|
43
|
+
raise missing_sdk_error("openai", "azure") from exc
|
|
44
|
+
|
|
45
|
+
# Same policy as every adapter: the waterfall owns retries; bound connects.
|
|
46
|
+
self._client = AzureOpenAI(
|
|
47
|
+
azure_endpoint=self._azure_endpoint,
|
|
48
|
+
api_version=self._api_version,
|
|
49
|
+
max_retries=0,
|
|
50
|
+
timeout=httpx.Timeout(
|
|
51
|
+
self.backend.read_timeout_s,
|
|
52
|
+
connect=self.backend.connect_timeout_s,
|
|
53
|
+
),
|
|
54
|
+
)
|
|
55
|
+
return self._client
|
|
56
|
+
|
|
57
|
+
def _request_model(self) -> str:
|
|
58
|
+
# Azure routes by deployment name, not the base model id.
|
|
59
|
+
return self.backend.deployment or self.backend.model
|
|
60
|
+
|
|
61
|
+
def embed_model_id(self) -> str | None:
|
|
62
|
+
# Azure embeddings need their own deployment; there is no meaningful default.
|
|
63
|
+
return self.backend.embed_model
|
|
64
|
+
|
|
65
|
+
def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
|
|
66
|
+
if self.backend.embed_model is None:
|
|
67
|
+
raise EmbeddingsUnsupported(
|
|
68
|
+
"this azure_openai backend has no embed_model= (an Azure embedding deployment); "
|
|
69
|
+
"the waterfall skips it for embed calls"
|
|
70
|
+
)
|
|
71
|
+
return super().embed(texts)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""The Adapter protocol every provider backend implements.
|
|
2
|
+
|
|
3
|
+
An adapter owns exactly one backend's SDK client and wire mapping. It raises its SDK's native
|
|
4
|
+
exceptions — classification (capacity vs client error) happens centrally in
|
|
5
|
+
`llm_waterfall.classify`, and the failover loop in `llm_waterfall.waterfall` owns retry policy.
|
|
6
|
+
Adapters must not retry internally.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Protocol, runtime_checkable
|
|
12
|
+
|
|
13
|
+
from llm_waterfall.types import Backend, ChatRequest, ChatResponse, Message, TokenUsage
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def missing_sdk_error(package: str, extra: str) -> ModuleNotFoundError:
|
|
17
|
+
"""A ModuleNotFoundError that tells the user which extra installs the missing SDK."""
|
|
18
|
+
return ModuleNotFoundError(
|
|
19
|
+
f"the '{package}' package is required for this backend; "
|
|
20
|
+
f'install it with: pip install "llm-waterfall[{extra}]"'
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@runtime_checkable
|
|
25
|
+
class Adapter(Protocol):
|
|
26
|
+
"""One backend's completion + embedding implementation."""
|
|
27
|
+
|
|
28
|
+
backend: Backend
|
|
29
|
+
|
|
30
|
+
def complete(
|
|
31
|
+
self,
|
|
32
|
+
system: str,
|
|
33
|
+
messages: list[Message],
|
|
34
|
+
*,
|
|
35
|
+
temperature: float | None,
|
|
36
|
+
max_tokens: int,
|
|
37
|
+
) -> tuple[str, TokenUsage]:
|
|
38
|
+
"""Run one completion; return (text, usage). Raises the SDK's native errors."""
|
|
39
|
+
...
|
|
40
|
+
|
|
41
|
+
def complete_chat(self, request: ChatRequest) -> ChatResponse:
|
|
42
|
+
"""Run one structured tool-calling completion."""
|
|
43
|
+
...
|
|
44
|
+
|
|
45
|
+
def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
|
|
46
|
+
"""Embed texts; return (vectors, usage). Raises EmbeddingsUnsupported if N/A."""
|
|
47
|
+
...
|
|
48
|
+
|
|
49
|
+
def embed_model_id(self) -> str | None:
|
|
50
|
+
"""The model embed() resolves to (None if unsupported) — owns embed attribution."""
|
|
51
|
+
...
|