world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/distill/deadlines.py
ADDED
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
"""Hard wall-clock deadlines around Tinker SDK calls.
|
|
2
|
+
|
|
3
|
+
A wedged Tinker session can block forever inside the SDK's internal
|
|
4
|
+
JWT-refresh/heartbeat retry loop: the call neither returns nor raises, so
|
|
5
|
+
provider retry wrappers and trial/episode timeouts never engage (one live run
|
|
6
|
+
hung 33 minutes this way while a fresh connection worked at the same moment).
|
|
7
|
+
The `Sdk*` adapters therefore bound every SDK call with a per-call-kind
|
|
8
|
+
deadline. Expiry raises `TinkerDeadlineError`, whose message deliberately
|
|
9
|
+
reads as a timeout so llm-waterfall classifies it as a capacity (transient)
|
|
10
|
+
error: callers retry with a fresh session instead of hanging.
|
|
11
|
+
|
|
12
|
+
Deadlines must scale with MODEL SIZE and TOKEN VOLUME, not just with call
|
|
13
|
+
kind. The original defaults were derived from small students on short
|
|
14
|
+
contexts (sample mean 2.8s / p95 7.7s / max 10.2s; compute_logprobs max
|
|
15
|
+
1.5s; forward_backward ~1.5s), and every one of them proved too tight once a
|
|
16
|
+
120B student ran a 240k-token context budget: `forward_backward` needed 900s
|
|
17
|
+
under Super plus top-k, and a 48-episode probe wave produced 10 `sample`
|
|
18
|
+
expiries against the old 120s. A sample that legitimately needs 150s gets
|
|
19
|
+
killed, retried on a fresh session, killed again, and finally surfaces as
|
|
20
|
+
`provider_error` — i.e. OUR timeout is recorded as a scaffold loss and
|
|
21
|
+
pollutes the very metric that is supposed to measure the agent loop. The
|
|
22
|
+
current values are therefore sized so the EPISODE wall (`episode_timeout_s`,
|
|
23
|
+
1800s) is what cuts a slow episode, never a single call's deadline, while a
|
|
24
|
+
genuinely wedged session still cannot hang forever.
|
|
25
|
+
|
|
26
|
+
`optim_step` and `save_state` were the SECOND lesson, and they killed a live
|
|
27
|
+
run: they scale with the number of DATUMS in the batch, which is a function of
|
|
28
|
+
the LOSS, not of the model. `topk_ce` replicates every datum k times, so the
|
|
29
|
+
same 64-episode batch became **512 datums against `importance_sampling`'s 62**,
|
|
30
|
+
and the optimizer step blew the old 120s while the reverse-KL arm — 8x lighter
|
|
31
|
+
on the identical batch — sailed through. Anything sized from small-model
|
|
32
|
+
timings (the "~0.5-3s" below) is therefore a floor, not a guide.
|
|
33
|
+
|
|
34
|
+
Remaining follow-up: make the sample/compute_logprobs deadlines a function of
|
|
35
|
+
`(prompt_tokens + max_tokens)`, and optim_step/save_state a function of datum
|
|
36
|
+
count, rather than flat constants, so a 4B student on 8k contexts is not
|
|
37
|
+
waiting 300s to discover a wedged session.
|
|
38
|
+
|
|
39
|
+
Historical measurements for the smaller-model regime (optim_step ~0.5s;
|
|
40
|
+
save_state ~2.4s; save_weights_for_sampler mean
|
|
41
|
+
4.4s / max 18s with one observed 80s outlier). `load_state` gets the same
|
|
42
|
+
600s as save_weights_for_sampler for a different reason: restoring a large
|
|
43
|
+
student's weights plus optimizer state exceeded 120s on a live 120B resume,
|
|
44
|
+
and the call cannot be retried on the same client (tinker refuses LoadWeights
|
|
45
|
+
once anything initialized the model, see `SdkTrainingClient`), so its deadline
|
|
46
|
+
is the whole budget rather than the first of two attempts. Each default is
|
|
47
|
+
overridable via one env var per kind, `WMO_TINKER_DEADLINE_<KIND>` in seconds:
|
|
48
|
+
|
|
49
|
+
- sample: 300s (WMO_TINKER_DEADLINE_SAMPLE)
|
|
50
|
+
- compute_logprobs: 300s (WMO_TINKER_DEADLINE_COMPUTE_LOGPROBS)
|
|
51
|
+
- forward_backward: 900s (WMO_TINKER_DEADLINE_FORWARD_BACKWARD)
|
|
52
|
+
- optim_step: 600s (WMO_TINKER_DEADLINE_OPTIM_STEP)
|
|
53
|
+
- save_state: 600s (WMO_TINKER_DEADLINE_SAVE_STATE)
|
|
54
|
+
- load_state: 600s (WMO_TINKER_DEADLINE_LOAD_STATE)
|
|
55
|
+
- save_weights_for_sampler: 600s (WMO_TINKER_DEADLINE_SAVE_WEIGHTS_FOR_SAMPLER)
|
|
56
|
+
- connect: 60s (WMO_TINKER_DEADLINE_CONNECT)
|
|
57
|
+
|
|
58
|
+
"connect" covers every synchronous call with no future to wait on: service
|
|
59
|
+
client construction, sampling/training client creation, and tokenizer
|
|
60
|
+
fetches. Overrides must be a positive finite number of seconds; anything
|
|
61
|
+
unparsable or non-positive raises immediately (a silently ignored override
|
|
62
|
+
would defeat the whole point), and sub-millisecond values clamp up to 1ms.
|
|
63
|
+
|
|
64
|
+
Two waiting strategies, chosen per call site:
|
|
65
|
+
|
|
66
|
+
- `wait_with_deadline` uses the SDK future's own `result(timeout=...)`. The
|
|
67
|
+
abandoned request keeps polling on the SDK's background loop, but the
|
|
68
|
+
caller proceeds immediately.
|
|
69
|
+
- `call_with_deadline` runs a fully blocking call (no timeout parameter in
|
|
70
|
+
the SDK) on a dedicated daemon thread and waits at most the deadline. On
|
|
71
|
+
expiry the thread is abandoned and may linger inside the SDK; a daemon
|
|
72
|
+
thread never blocks interpreter exit, and the run proceeds with a typed,
|
|
73
|
+
retryable error.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
from __future__ import annotations
|
|
77
|
+
|
|
78
|
+
import logging
|
|
79
|
+
import math
|
|
80
|
+
import os
|
|
81
|
+
import threading
|
|
82
|
+
import time
|
|
83
|
+
from collections.abc import Callable
|
|
84
|
+
from typing import Literal, Protocol
|
|
85
|
+
|
|
86
|
+
logger = logging.getLogger(__name__)
|
|
87
|
+
|
|
88
|
+
TinkerCallKind = Literal[
|
|
89
|
+
"sample",
|
|
90
|
+
"compute_logprobs",
|
|
91
|
+
"forward_backward",
|
|
92
|
+
"optim_step",
|
|
93
|
+
"save_state",
|
|
94
|
+
"load_state",
|
|
95
|
+
"save_weights_for_sampler",
|
|
96
|
+
"connect",
|
|
97
|
+
]
|
|
98
|
+
|
|
99
|
+
DEADLINE_ENV_PREFIX = "WMO_TINKER_DEADLINE_"
|
|
100
|
+
|
|
101
|
+
DEFAULT_DEADLINES_S: dict[TinkerCallKind, float] = {
|
|
102
|
+
"sample": 300.0,
|
|
103
|
+
"compute_logprobs": 300.0,
|
|
104
|
+
"forward_backward": 900.0,
|
|
105
|
+
"optim_step": 600.0,
|
|
106
|
+
"save_state": 600.0,
|
|
107
|
+
"load_state": 600.0,
|
|
108
|
+
"save_weights_for_sampler": 600.0,
|
|
109
|
+
"connect": 60.0,
|
|
110
|
+
}
|
|
111
|
+
"""Per-kind default deadlines; see the module docstring for their derivation."""
|
|
112
|
+
|
|
113
|
+
_MIN_DEADLINE_S = 0.001
|
|
114
|
+
"""Floor for overrides: a sub-millisecond deadline clamps up to this."""
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class DeadlineFuture[T](Protocol):
|
|
118
|
+
"""The slice of a tinker `APIFuture` the deadline wait consumes."""
|
|
119
|
+
|
|
120
|
+
def result(self, timeout: float | None = None) -> T:
|
|
121
|
+
"""Block for the result, raising `TimeoutError` after `timeout` seconds."""
|
|
122
|
+
...
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class TinkerDeadlineError(TimeoutError):
|
|
126
|
+
"""A Tinker SDK call blew its wall-clock deadline; the session is likely wedged.
|
|
127
|
+
|
|
128
|
+
The call was abandoned, so the caller can proceed; the remedy is a retry
|
|
129
|
+
through a fresh session (the wmo adapters drop and rebuild their cached
|
|
130
|
+
clients where they own them). Subclasses `TimeoutError` and keeps the
|
|
131
|
+
phrase "timed out" in its message on purpose: that is what makes
|
|
132
|
+
llm-waterfall's `is_capacity_error` classify it as transient, so the
|
|
133
|
+
provider retry wrapper re-attempts instead of propagating.
|
|
134
|
+
|
|
135
|
+
Attributes:
|
|
136
|
+
kind: Which call kind expired.
|
|
137
|
+
elapsed_s: Wall-clock seconds actually waited.
|
|
138
|
+
deadline_s: The deadline that was in force.
|
|
139
|
+
"""
|
|
140
|
+
|
|
141
|
+
def __init__(self, kind: TinkerCallKind, *, elapsed_s: float, deadline_s: float) -> None:
|
|
142
|
+
self.kind = kind
|
|
143
|
+
self.elapsed_s = elapsed_s
|
|
144
|
+
self.deadline_s = deadline_s
|
|
145
|
+
super().__init__(
|
|
146
|
+
f"tinker {kind} timed out after {elapsed_s:.1f}s (deadline {deadline_s:.1f}s, "
|
|
147
|
+
f"override via {env_var_for(kind)}); the session is likely wedged inside the "
|
|
148
|
+
"SDK's internal retry loop. The call was abandoned; retry with a fresh "
|
|
149
|
+
"session (the wmo adapters rebuild their own clients on the next attempt)"
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def env_var_for(kind: TinkerCallKind) -> str:
|
|
154
|
+
"""The env var that overrides one call kind's deadline."""
|
|
155
|
+
return DEADLINE_ENV_PREFIX + kind.upper()
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def deadline_for(kind: TinkerCallKind) -> float:
|
|
159
|
+
"""The effective deadline for one call kind, in seconds.
|
|
160
|
+
|
|
161
|
+
Reads the kind's env var on every call so an override set mid-process
|
|
162
|
+
(e.g. by a test) takes effect immediately.
|
|
163
|
+
|
|
164
|
+
Raises:
|
|
165
|
+
ValueError: If the override is unparsable, non-finite, or not
|
|
166
|
+
positive; the message names the env var and the fix.
|
|
167
|
+
"""
|
|
168
|
+
env_var = env_var_for(kind)
|
|
169
|
+
raw = os.environ.get(env_var)
|
|
170
|
+
if raw is None:
|
|
171
|
+
return DEFAULT_DEADLINES_S[kind]
|
|
172
|
+
try:
|
|
173
|
+
value = float(raw)
|
|
174
|
+
except ValueError as exc:
|
|
175
|
+
raise ValueError(
|
|
176
|
+
f"invalid {env_var}={raw!r}: set a positive number of seconds "
|
|
177
|
+
f"(e.g. {env_var}={DEFAULT_DEADLINES_S[kind]:g}), or unset it for the default"
|
|
178
|
+
) from exc
|
|
179
|
+
if not math.isfinite(value) or value <= 0:
|
|
180
|
+
raise ValueError(
|
|
181
|
+
f"invalid {env_var}={raw!r}: the deadline must be a positive finite number "
|
|
182
|
+
f"of seconds (default {DEFAULT_DEADLINES_S[kind]:g}); unset it for the default"
|
|
183
|
+
)
|
|
184
|
+
return max(value, _MIN_DEADLINE_S)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def wait_with_deadline[T](kind: TinkerCallKind, future: DeadlineFuture[T]) -> T:
|
|
188
|
+
"""Wait for an SDK future under the kind's deadline.
|
|
189
|
+
|
|
190
|
+
Uses the future's own `result(timeout=...)`, so no extra thread is
|
|
191
|
+
needed; on expiry the abandoned request keeps polling on the SDK's
|
|
192
|
+
background loop while this caller proceeds.
|
|
193
|
+
|
|
194
|
+
Args:
|
|
195
|
+
kind: The call kind, for the deadline lookup and the error message.
|
|
196
|
+
future: The SDK future to wait on.
|
|
197
|
+
|
|
198
|
+
Returns:
|
|
199
|
+
The future's result.
|
|
200
|
+
|
|
201
|
+
Raises:
|
|
202
|
+
TinkerDeadlineError: If the deadline expires before the result lands.
|
|
203
|
+
"""
|
|
204
|
+
deadline = deadline_for(kind)
|
|
205
|
+
started = time.monotonic()
|
|
206
|
+
try:
|
|
207
|
+
return future.result(timeout=deadline)
|
|
208
|
+
except TinkerDeadlineError:
|
|
209
|
+
raise
|
|
210
|
+
except TimeoutError as exc:
|
|
211
|
+
raise TinkerDeadlineError(
|
|
212
|
+
kind, elapsed_s=time.monotonic() - started, deadline_s=deadline
|
|
213
|
+
) from exc
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def call_with_deadline[T](kind: TinkerCallKind, call: Callable[[], T]) -> T:
|
|
217
|
+
"""Run a fully blocking SDK call under the kind's deadline.
|
|
218
|
+
|
|
219
|
+
For SDK calls with no timeout parameter and no future (client and
|
|
220
|
+
tokenizer construction): the call runs on a dedicated daemon thread and
|
|
221
|
+
the caller waits at most the deadline. On expiry the thread is abandoned;
|
|
222
|
+
it may linger blocked inside the SDK, but daemon threads never block
|
|
223
|
+
interpreter exit and the run proceeds.
|
|
224
|
+
|
|
225
|
+
Args:
|
|
226
|
+
kind: The call kind, for the deadline lookup and the error message.
|
|
227
|
+
call: The zero-argument blocking call.
|
|
228
|
+
|
|
229
|
+
Returns:
|
|
230
|
+
Whatever `call` returns.
|
|
231
|
+
|
|
232
|
+
Raises:
|
|
233
|
+
TinkerDeadlineError: If the deadline expires before the call returns.
|
|
234
|
+
Exception: Anything `call` itself raises, re-raised on this thread.
|
|
235
|
+
"""
|
|
236
|
+
deadline = deadline_for(kind)
|
|
237
|
+
outcome: list[T] = []
|
|
238
|
+
failure: list[BaseException] = []
|
|
239
|
+
|
|
240
|
+
def _run() -> None:
|
|
241
|
+
try:
|
|
242
|
+
outcome.append(call())
|
|
243
|
+
except BaseException as exc: # noqa: BLE001 - re-raised on the caller thread below
|
|
244
|
+
failure.append(exc)
|
|
245
|
+
|
|
246
|
+
thread = threading.Thread(target=_run, name=f"wmo-tinker-{kind}", daemon=True)
|
|
247
|
+
started = time.monotonic()
|
|
248
|
+
thread.start()
|
|
249
|
+
thread.join(deadline)
|
|
250
|
+
if thread.is_alive():
|
|
251
|
+
raise TinkerDeadlineError(kind, elapsed_s=time.monotonic() - started, deadline_s=deadline)
|
|
252
|
+
if failure:
|
|
253
|
+
raise failure[0]
|
|
254
|
+
return outcome[0]
|