world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
llm_waterfall/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Experiential Labs
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,53 @@
1
+ """llm-waterfall: stateless LLM failover across an ordered chain of backends.
2
+
3
+ Capacity errors (throttling, transient 5xx, timeouts) spill to the next backend; real client
4
+ errors raise immediately. Every call returns which backend served it, token usage, USD cost, and
5
+ the full attempt trail.
6
+ """
7
+
8
+ from llm_waterfall.classify import is_capacity_error, outcome_for
9
+ from llm_waterfall.pricing import ModelPrice, cost_usd, price_for
10
+ from llm_waterfall.types import (
11
+ Attempt,
12
+ Backend,
13
+ ChatMaxTokensField,
14
+ ChatRequest,
15
+ ChatResponse,
16
+ ChatResult,
17
+ CompletionResult,
18
+ EmbeddingResult,
19
+ EmbeddingsUnsupported,
20
+ Message,
21
+ RetryPolicy,
22
+ TokenUsage,
23
+ ToolCallingUnsupported,
24
+ VerifyResult,
25
+ WaterfallExhausted,
26
+ )
27
+ from llm_waterfall.waterfall import Waterfall
28
+
29
+ __version__ = "0.1.4"
30
+
31
+ __all__ = [
32
+ "Attempt",
33
+ "Backend",
34
+ "ChatMaxTokensField",
35
+ "ChatRequest",
36
+ "ChatResponse",
37
+ "ChatResult",
38
+ "CompletionResult",
39
+ "EmbeddingResult",
40
+ "EmbeddingsUnsupported",
41
+ "Message",
42
+ "ModelPrice",
43
+ "RetryPolicy",
44
+ "TokenUsage",
45
+ "ToolCallingUnsupported",
46
+ "VerifyResult",
47
+ "Waterfall",
48
+ "WaterfallExhausted",
49
+ "cost_usd",
50
+ "is_capacity_error",
51
+ "outcome_for",
52
+ "price_for",
53
+ ]
@@ -0,0 +1,36 @@
1
+ """Adapter registry: map a Backend's provider name to its adapter class.
2
+
3
+ Adapter modules import at module scope (they gate their SDK imports internally, so this package
4
+ imports with zero SDKs installed); only the SDKs themselves are lazy.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from collections.abc import Callable
10
+
11
+ from llm_waterfall.adapters.anthropic import AnthropicAdapter
12
+ from llm_waterfall.adapters.aws_mantle import AwsMantleAdapter
13
+ from llm_waterfall.adapters.azure_openai import AzureOpenAIAdapter
14
+ from llm_waterfall.adapters.base import Adapter
15
+ from llm_waterfall.adapters.bedrock import BedrockAdapter
16
+ from llm_waterfall.adapters.openai import OpenAIAdapter
17
+ from llm_waterfall.types import PROVIDERS, Backend
18
+
19
+ _ADAPTERS: dict[str, Callable[[Backend], Adapter]] = {
20
+ "bedrock": BedrockAdapter,
21
+ "openai": OpenAIAdapter,
22
+ "anthropic": AnthropicAdapter,
23
+ "azure_openai": AzureOpenAIAdapter, # real; construction validates endpoint/api_version
24
+ "aws_mantle": AwsMantleAdapter, # stub: raises NotImplementedError at construction
25
+ }
26
+
27
+
28
+ def build_adapter(backend: Backend) -> Adapter:
29
+ """Construct the adapter for `backend`. Unimplemented providers fail here, fast."""
30
+ try:
31
+ factory = _ADAPTERS[backend.provider]
32
+ except KeyError:
33
+ raise ValueError(
34
+ f"unknown provider {backend.provider!r}; expected one of {', '.join(PROVIDERS)}"
35
+ ) from None
36
+ return factory(backend)
@@ -0,0 +1,105 @@
1
+ """Anthropic direct-API adapter. Reads ANTHROPIC_API_KEY from the environment."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import threading
6
+ from typing import TYPE_CHECKING, cast
7
+
8
+ from llm_waterfall.adapters.base import missing_sdk_error
9
+ from llm_waterfall.types import (
10
+ Backend,
11
+ ChatRequest,
12
+ ChatResponse,
13
+ EmbeddingsUnsupported,
14
+ Message,
15
+ TokenUsage,
16
+ ToolCallingUnsupported,
17
+ )
18
+
19
+ if TYPE_CHECKING:
20
+ from anthropic import Anthropic
21
+ from anthropic.types import MessageParam
22
+
23
+
24
+ class AnthropicAdapter:
25
+ """Claude via the direct Anthropic Messages API."""
26
+
27
+ def __init__(self, backend: Backend) -> None:
28
+ self.backend = backend
29
+ self._client: Anthropic | None = None
30
+ self._lock = threading.Lock()
31
+
32
+ def _get_client(self) -> Anthropic:
33
+ if self._client is None:
34
+ with self._lock:
35
+ if self._client is None:
36
+ try:
37
+ import httpx
38
+ from anthropic import Anthropic
39
+ except ModuleNotFoundError as exc:
40
+ raise missing_sdk_error("anthropic", "anthropic") from exc
41
+
42
+ # max_retries=0: the waterfall owns retry policy. Granular httpx.Timeout so
43
+ # a dead endpoint fails over after connect_timeout_s, not read_timeout_s.
44
+ self._client = Anthropic(
45
+ max_retries=0,
46
+ timeout=httpx.Timeout(
47
+ self.backend.read_timeout_s,
48
+ connect=self.backend.connect_timeout_s,
49
+ ),
50
+ )
51
+ return self._client
52
+
53
+ def complete(
54
+ self,
55
+ system: str,
56
+ messages: list[Message],
57
+ *,
58
+ temperature: float | None,
59
+ max_tokens: int,
60
+ ) -> tuple[str, TokenUsage]:
61
+ """One Messages API call; system is a top-level arg (Anthropic-native)."""
62
+ api_messages = [
63
+ cast("MessageParam", {"role": m.role, "content": m.content}) for m in messages
64
+ ]
65
+ client_messages = self._get_client().messages
66
+ # Claude 4.7+ rejects sampling params; only forward temperature when explicitly set.
67
+ if temperature is None:
68
+ response = client_messages.create(
69
+ model=self.backend.model,
70
+ system=system,
71
+ messages=api_messages,
72
+ max_tokens=max_tokens,
73
+ )
74
+ else:
75
+ response = client_messages.create(
76
+ model=self.backend.model,
77
+ system=system,
78
+ messages=api_messages,
79
+ max_tokens=max_tokens,
80
+ temperature=temperature,
81
+ )
82
+ text = "".join(block.text for block in response.content if block.type == "text")
83
+ usage = TokenUsage(
84
+ input_tokens=response.usage.input_tokens,
85
+ output_tokens=response.usage.output_tokens,
86
+ )
87
+ return text, usage
88
+
89
+ def complete_chat(self, request: ChatRequest) -> ChatResponse:
90
+ """Direct Anthropic structured mapping is not implemented yet."""
91
+ del request
92
+ raise ToolCallingUnsupported(
93
+ "the anthropic adapter does not yet map OpenAI-compatible tool calls; "
94
+ "put an openai, azure_openai, or bedrock backend in the chain"
95
+ )
96
+
97
+ def embed_model_id(self) -> str | None:
98
+ """Anthropic has no embeddings API."""
99
+ return None
100
+
101
+ def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
102
+ raise EmbeddingsUnsupported(
103
+ "Anthropic has no embeddings API; put a bedrock or openai backend in the chain "
104
+ "for embeddings (the waterfall skips this backend for embed calls)."
105
+ )
@@ -0,0 +1,47 @@
1
+ """AWS Mantle adapter — stub that fails at construction.
2
+
3
+ TODO: implement against the Bedrock Mantle client (`anthropic.AnthropicBedrockMantle` — the
4
+ Messages-API Bedrock endpoint) with `aws_region=backend.region`, credentials via
5
+ `boto3.Session(profile_name=backend.profile)`, `max_retries=0`, and the backend's timeouts.
6
+ Model ids take an `anthropic.` prefix. Classification already covers its error shapes (the
7
+ anthropic SDK exception types + botocore codes).
8
+
9
+ Failing in `__init__` (not at call time) is deliberate: an unimplemented rung must break
10
+ `Waterfall(...)` construction loudly, never abort a live call mid-chain.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from llm_waterfall.types import Backend, ChatRequest, ChatResponse, Message, TokenUsage
16
+
17
+
18
+ class AwsMantleAdapter:
19
+ """Not implemented yet; constructing one fails fast with the workaround."""
20
+
21
+ backend: Backend # satisfies the Adapter protocol; never assigned (init raises)
22
+
23
+ def __init__(self, backend: Backend) -> None:
24
+ raise NotImplementedError(
25
+ "the aws_mantle adapter is not implemented yet; use a 'bedrock' backend for AWS "
26
+ "traffic, or contribute the adapter (see the TODO in "
27
+ "llm_waterfall/adapters/aws_mantle.py)."
28
+ )
29
+
30
+ def complete(
31
+ self,
32
+ system: str,
33
+ messages: list[Message],
34
+ *,
35
+ temperature: float | None,
36
+ max_tokens: int,
37
+ ) -> tuple[str, TokenUsage]:
38
+ raise NotImplementedError # pragma: no cover - unreachable (init raises)
39
+
40
+ def complete_chat(self, request: ChatRequest) -> ChatResponse:
41
+ raise NotImplementedError # pragma: no cover - unreachable (init raises)
42
+
43
+ def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
44
+ raise NotImplementedError # pragma: no cover - unreachable (init raises)
45
+
46
+ def embed_model_id(self) -> str | None:
47
+ raise NotImplementedError # pragma: no cover - unreachable (init raises)
@@ -0,0 +1,71 @@
1
+ """Azure OpenAI adapter: the OpenAI wire mapping behind an AzureOpenAI client.
2
+
3
+ Requests route by DEPLOYMENT name (`Backend.deployment`, defaulting to `Backend.model`); the key
4
+ is read from AZURE_OPENAI_API_KEY. Construction fails fast when `endpoint`/`api_version` are
5
+ missing — a misconfigured rung must break `Waterfall(...)`, not a live call mid-chain.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from typing import TYPE_CHECKING
11
+
12
+ from llm_waterfall.adapters.base import missing_sdk_error
13
+ from llm_waterfall.adapters.openai import OpenAIAdapter
14
+ from llm_waterfall.types import Backend, EmbeddingsUnsupported, TokenUsage
15
+
16
+ if TYPE_CHECKING:
17
+ from openai import OpenAI
18
+
19
+
20
+ class AzureOpenAIAdapter(OpenAIAdapter):
21
+ """GPT-5.x Azure deployments; embeddings via an Azure embedding deployment."""
22
+
23
+ def __init__(self, backend: Backend) -> None:
24
+ if not backend.endpoint:
25
+ raise ValueError(
26
+ "azure_openai backends need endpoint= (e.g. https://<resource>.openai.azure.com)"
27
+ )
28
+ if not backend.api_version:
29
+ raise ValueError("azure_openai backends need api_version= (e.g. 2024-12-01-preview)")
30
+ # Validated non-None copies (the dataclass fields stay Optional for other providers).
31
+ self._azure_endpoint: str = backend.endpoint
32
+ self._api_version: str = backend.api_version
33
+ super().__init__(backend)
34
+
35
+ def _get_client(self) -> OpenAI:
36
+ if self._client is None:
37
+ with self._lock:
38
+ if self._client is None:
39
+ try:
40
+ import httpx
41
+ from openai import AzureOpenAI
42
+ except ModuleNotFoundError as exc:
43
+ raise missing_sdk_error("openai", "azure") from exc
44
+
45
+ # Same policy as every adapter: the waterfall owns retries; bound connects.
46
+ self._client = AzureOpenAI(
47
+ azure_endpoint=self._azure_endpoint,
48
+ api_version=self._api_version,
49
+ max_retries=0,
50
+ timeout=httpx.Timeout(
51
+ self.backend.read_timeout_s,
52
+ connect=self.backend.connect_timeout_s,
53
+ ),
54
+ )
55
+ return self._client
56
+
57
+ def _request_model(self) -> str:
58
+ # Azure routes by deployment name, not the base model id.
59
+ return self.backend.deployment or self.backend.model
60
+
61
+ def embed_model_id(self) -> str | None:
62
+ # Azure embeddings need their own deployment; there is no meaningful default.
63
+ return self.backend.embed_model
64
+
65
+ def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
66
+ if self.backend.embed_model is None:
67
+ raise EmbeddingsUnsupported(
68
+ "this azure_openai backend has no embed_model= (an Azure embedding deployment); "
69
+ "the waterfall skips it for embed calls"
70
+ )
71
+ return super().embed(texts)
@@ -0,0 +1,51 @@
1
+ """The Adapter protocol every provider backend implements.
2
+
3
+ An adapter owns exactly one backend's SDK client and wire mapping. It raises its SDK's native
4
+ exceptions — classification (capacity vs client error) happens centrally in
5
+ `llm_waterfall.classify`, and the failover loop in `llm_waterfall.waterfall` owns retry policy.
6
+ Adapters must not retry internally.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from typing import Protocol, runtime_checkable
12
+
13
+ from llm_waterfall.types import Backend, ChatRequest, ChatResponse, Message, TokenUsage
14
+
15
+
16
+ def missing_sdk_error(package: str, extra: str) -> ModuleNotFoundError:
17
+ """A ModuleNotFoundError that tells the user which extra installs the missing SDK."""
18
+ return ModuleNotFoundError(
19
+ f"the '{package}' package is required for this backend; "
20
+ f'install it with: pip install "llm-waterfall[{extra}]"'
21
+ )
22
+
23
+
24
+ @runtime_checkable
25
+ class Adapter(Protocol):
26
+ """One backend's completion + embedding implementation."""
27
+
28
+ backend: Backend
29
+
30
+ def complete(
31
+ self,
32
+ system: str,
33
+ messages: list[Message],
34
+ *,
35
+ temperature: float | None,
36
+ max_tokens: int,
37
+ ) -> tuple[str, TokenUsage]:
38
+ """Run one completion; return (text, usage). Raises the SDK's native errors."""
39
+ ...
40
+
41
+ def complete_chat(self, request: ChatRequest) -> ChatResponse:
42
+ """Run one structured tool-calling completion."""
43
+ ...
44
+
45
+ def embed(self, texts: list[str]) -> tuple[list[list[float]], TokenUsage]:
46
+ """Embed texts; return (vectors, usage). Raises EmbeddingsUnsupported if N/A."""
47
+ ...
48
+
49
+ def embed_model_id(self) -> str | None:
50
+ """The model embed() resolves to (None if unsupported) — owns embed attribution."""
51
+ ...