world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/engine/autoconfig.py
ADDED
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
"""Max-fidelity auto-configuration: find the agentic config that best fits THIS corpus.
|
|
2
|
+
|
|
3
|
+
The lever matrix is empirical and task-dependent (measured across tau/terminal/swe: reasoning
|
|
4
|
+
wins on tool-call APIs, live fetch on web-heavy shells, the verify self-check on hard content
|
|
5
|
+
prediction — and no blanket setting wins everywhere). `wmo build --max-fidelity` automates that
|
|
6
|
+
search: each candidate configuration is replay-scored on the build's held-out split (leak-free,
|
|
7
|
+
same judge and demos as `wmo eval`), the winner's flags are persisted to the artifact's
|
|
8
|
+
config.toml, and serving picks them up automatically. The default build stays plain RAG — the
|
|
9
|
+
search is strictly opt-in, and `--fidelity-budget` chooses how deep it goes.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections.abc import Callable, Sequence
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from statistics import fmean
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel, Field
|
|
19
|
+
|
|
20
|
+
from wmo.core.types import Trace
|
|
21
|
+
from wmo.engine.grounding import FetchGrounder, Grounder, SourceResolver, extract_get_url
|
|
22
|
+
from wmo.engine.knowledge import seeded_knowledge_text
|
|
23
|
+
from wmo.engine.replay import replay
|
|
24
|
+
from wmo.engine.workspace import RepoTreeResolver
|
|
25
|
+
from wmo.optimize.judge import Judge
|
|
26
|
+
from wmo.providers.base import Embedder, Provider
|
|
27
|
+
from wmo.retrieval import EmbeddingRetriever
|
|
28
|
+
|
|
29
|
+
# Held-out traces scored per candidate by default: small enough that the search costs a fraction
|
|
30
|
+
# of the GEPA build, large enough to separate candidates beyond judge noise on most corpora.
|
|
31
|
+
DEFAULT_VAL_CAP = 8
|
|
32
|
+
|
|
33
|
+
# A challenger must beat the incumbent's mean fidelity by more than this to displace it. Sized
|
|
34
|
+
# above the per-cell judge/selection noise measured in the D37 ladder (±0.003–0.014): without
|
|
35
|
+
# it, a fluke +0.002 on the selection sample promoted a config that then LOST on test, which is
|
|
36
|
+
# exactly how the old independent-per-tier search went non-monotonic (tau high 0.886 < medium
|
|
37
|
+
# 0.891). The incumbent floor makes each tier improve-or-hold, never regress.
|
|
38
|
+
_NOISE_MARGIN = 0.01
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class CandidateConfig:
|
|
43
|
+
"""One agentic configuration the search can select (maps 1:1 onto HarnessConfig flags)."""
|
|
44
|
+
|
|
45
|
+
label: str
|
|
46
|
+
reasoning: bool = False
|
|
47
|
+
knowledge: bool = False
|
|
48
|
+
verify: bool = False
|
|
49
|
+
grounder: str = "none"
|
|
50
|
+
# Workspace grounding (pinned source files + repo tree). Research-measurable today; enters
|
|
51
|
+
# build's search only once serve-side activation lands (a winner the runtime can't serve
|
|
52
|
+
# would be a lie in auto_fidelity.json).
|
|
53
|
+
workspace: bool = False
|
|
54
|
+
# Retrieval-depth overrides (None = the engine defaults: top_k 5, no demo cap). The measured
|
|
55
|
+
# "rag-deep" config: on record-heavy corpora more demos put more of the database in context
|
|
56
|
+
# (tau full-slice: base 0.939 -> 0.955 at k=20+cap2000, +0.016, replicating PR #72's +0.015);
|
|
57
|
+
# the cap keeps verbose corpora affordable.
|
|
58
|
+
top_k: int | None = None
|
|
59
|
+
demo_obs_cap: int | None = None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# Ordered by MEASURED serve cost, cheapest first (swe $/run: base 5.82, reason 6.20, workspace
|
|
63
|
+
# 6.48, kb 11.50, verify 12.73) — ties go to the earlier candidate, so the price-performance
|
|
64
|
+
# frontier wins: a cheap grounding config beats an expensive deliberation config that only
|
|
65
|
+
# matches it. Grounding-class candidates join right after `reason` because test-time ground
|
|
66
|
+
# truth is nearly free and measured as the largest lift class (fetch +0.040, workspace +0.065).
|
|
67
|
+
DEFAULT_CANDIDATES: tuple[CandidateConfig, ...] = (
|
|
68
|
+
CandidateConfig(label="base"),
|
|
69
|
+
CandidateConfig(label="reason", reasoning=True),
|
|
70
|
+
CandidateConfig(label="reason+kb", reasoning=True, knowledge=True),
|
|
71
|
+
CandidateConfig(label="reason+verify", reasoning=True, verify=True),
|
|
72
|
+
)
|
|
73
|
+
# Grounding-class candidates (cheap, corpus-gated). workspace needs instance pins (auto-detected
|
|
74
|
+
# next to the traces file); fetch is non-hermetic (hits the real web during the search) and is
|
|
75
|
+
# considered only when the corpus actually contains fetchable curl GETs.
|
|
76
|
+
WORKSPACE_CANDIDATE = CandidateConfig(label="reason+workspace", reasoning=True, workspace=True)
|
|
77
|
+
FETCH_CANDIDATE = CandidateConfig(label="reason+fetch", reasoning=True, grounder="fetch")
|
|
78
|
+
# Deep retrieval: 4x the demos, each observation capped (PR #72's optimized RAG, replicated on
|
|
79
|
+
# this protocol: tau +0.016 full-slice, terminal +0.004, swe +0.001 per #72). ~2x serve cost —
|
|
80
|
+
# it sits in the expensive tail, not the cheap frontier.
|
|
81
|
+
RAG_DEEP_CANDIDATE = CandidateConfig(label="rag-deep", top_k=20, demo_obs_cap=2000)
|
|
82
|
+
# The ladder's expensive tail: levers that ~2x the serve bill (kb rebuilds context, verify
|
|
83
|
+
# doubles completions). The medium tier's cheap-frontier search stops before these.
|
|
84
|
+
_EXPENSIVE_LABELS = frozenset({"rag-deep", "reason+kb", "reason+verify"})
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass(frozen=True)
|
|
88
|
+
class CorpusSignature:
|
|
89
|
+
"""Zero-token corpus features that predict which levers can pay off.
|
|
90
|
+
|
|
91
|
+
Measured reference points (healthy corpora, 2026-07-02): tau-bench curl=0.00/obs=414/
|
|
92
|
+
tool=1.00 (winner: reason), terminal-tasks 0.43/1236/0.00 (winner: reason+fetch),
|
|
93
|
+
swe-bench 0.00/889/0.00 (winner: reason+verify).
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
curl_get_share: float # steps whose action is a read-only curl GET
|
|
97
|
+
mean_obs_chars: float # content-heaviness of observations
|
|
98
|
+
tool_call_share: float # structured tool-call API vs free-form bash
|
|
99
|
+
|
|
100
|
+
@classmethod
|
|
101
|
+
def from_traces(cls, traces: list[Trace]) -> CorpusSignature:
|
|
102
|
+
steps = [s for t in traces for s in t.steps]
|
|
103
|
+
if not steps:
|
|
104
|
+
return cls(curl_get_share=0.0, mean_obs_chars=0.0, tool_call_share=0.0)
|
|
105
|
+
return cls(
|
|
106
|
+
curl_get_share=fmean(1.0 if extract_get_url(s.action) else 0.0 for s in steps),
|
|
107
|
+
mean_obs_chars=fmean(len(s.observation.content) for s in steps),
|
|
108
|
+
tool_call_share=fmean(0.0 if s.action.name == "bash" else 1.0 for s in steps),
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def signature_estimate(signature: CorpusSignature, *, has_pins: bool = False) -> CandidateConfig:
|
|
113
|
+
"""The single strongest config the measured lever matrix predicts for this corpus — free.
|
|
114
|
+
|
|
115
|
+
This is the `low` tier's shipped config and the incumbent FLOOR every searching tier seeds
|
|
116
|
+
from, so the ladder starts from a strong prior and can only improve. Deterministic from the
|
|
117
|
+
signature, so separate `wmo build --fidelity {medium,high,max}` invocations all carry the
|
|
118
|
+
identical floor. Rules are the D27 recommendations:
|
|
119
|
+
- tool-call APIs (tau-like): reasoning alone won (.899 -> .919).
|
|
120
|
+
- curl-heavy shells (terminal-like): live fetch of the action's own URL won (+0.040).
|
|
121
|
+
- pinned code repos (swe-like): workspace grounding won (+0.065).
|
|
122
|
+
- free-form, content-heavy bash otherwise: the knowledge base helped most.
|
|
123
|
+
- anything else: reasoning is the safe, cheap default.
|
|
124
|
+
"""
|
|
125
|
+
if signature.tool_call_share >= 0.5:
|
|
126
|
+
return DEFAULT_CANDIDATES[1] # reason
|
|
127
|
+
if signature.curl_get_share >= 0.10:
|
|
128
|
+
return FETCH_CANDIDATE
|
|
129
|
+
if has_pins:
|
|
130
|
+
return WORKSPACE_CANDIDATE
|
|
131
|
+
if signature.mean_obs_chars >= 600:
|
|
132
|
+
return DEFAULT_CANDIDATES[2] # reason+kb
|
|
133
|
+
return DEFAULT_CANDIDATES[1] # reason
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def select_candidates(
|
|
137
|
+
signature: CorpusSignature,
|
|
138
|
+
*,
|
|
139
|
+
full_ladder: bool = False,
|
|
140
|
+
has_pins: bool = False,
|
|
141
|
+
cheap_only: bool = False,
|
|
142
|
+
) -> tuple[CandidateConfig, ...]:
|
|
143
|
+
"""Choose which candidates are worth spending tokens on for THIS corpus.
|
|
144
|
+
|
|
145
|
+
Price sets the ORDER, never the menu: candidates are laddered cheapest-first (so a
|
|
146
|
+
truncated budget spends on cheap tricks first, and the winner tie-break favors the cheaper
|
|
147
|
+
config), but a candidate is dropped only when the corpus signature says it CANNOT matter
|
|
148
|
+
here — never because a cheaper lever is also available. Fidelity picks the winner; the
|
|
149
|
+
tier (cheap search vs `full_ladder`) only decides how hard we look.
|
|
150
|
+
|
|
151
|
+
Signature gates (from the measured lever matrix, not intuition):
|
|
152
|
+
- knowledge/verify: free-form (bash-like) environments; verify additionally wants
|
|
153
|
+
content-heavy observations (it only ever paid off where content prediction is hardest).
|
|
154
|
+
- fetch: a meaningful share of read-only curl GETs (nothing to prefetch = byte-identical
|
|
155
|
+
to `reason`); workspace: instance pins exist (same no-op logic).
|
|
156
|
+
`full_ladder` (the max tier) keeps only the no-op gates. `cheap_only` (the medium tier)
|
|
157
|
+
truncates the ladder before the expensive deliberation levers — grounding serves at ~base
|
|
158
|
+
cost, so even a budget tier can afford to discover a workspace/fetch win.
|
|
159
|
+
"""
|
|
160
|
+
bash_like = signature.tool_call_share < 0.5
|
|
161
|
+
fetchable = signature.curl_get_share >= 0.10
|
|
162
|
+
# PRICE ORDER: base -> reason -> grounding class (workspace/fetch, ~free) -> kb -> verify.
|
|
163
|
+
if full_ladder:
|
|
164
|
+
chosen = [DEFAULT_CANDIDATES[0], DEFAULT_CANDIDATES[1]]
|
|
165
|
+
if has_pins:
|
|
166
|
+
chosen.append(WORKSPACE_CANDIDATE)
|
|
167
|
+
if fetchable:
|
|
168
|
+
chosen.append(FETCH_CANDIDATE)
|
|
169
|
+
chosen.append(RAG_DEEP_CANDIDATE)
|
|
170
|
+
chosen.extend([DEFAULT_CANDIDATES[2], DEFAULT_CANDIDATES[3]])
|
|
171
|
+
return _maybe_cheap(tuple(chosen), cheap_only)
|
|
172
|
+
chosen = [DEFAULT_CANDIDATES[0], DEFAULT_CANDIDATES[1]] # base, reason
|
|
173
|
+
if has_pins:
|
|
174
|
+
chosen.append(WORKSPACE_CANDIDATE) # cheapest strong lever when a repo pin exists
|
|
175
|
+
if fetchable:
|
|
176
|
+
chosen.append(FETCH_CANDIDATE)
|
|
177
|
+
chosen.append(RAG_DEEP_CANDIDATE) # never signature-gated: it never hurt anywhere measured
|
|
178
|
+
if bash_like:
|
|
179
|
+
chosen.append(DEFAULT_CANDIDATES[2]) # reason+kb
|
|
180
|
+
if signature.mean_obs_chars >= 600:
|
|
181
|
+
chosen.append(DEFAULT_CANDIDATES[3]) # reason+verify
|
|
182
|
+
return _maybe_cheap(tuple(chosen), cheap_only)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _maybe_cheap(
|
|
186
|
+
candidates: tuple[CandidateConfig, ...], cheap_only: bool
|
|
187
|
+
) -> tuple[CandidateConfig, ...]:
|
|
188
|
+
if not cheap_only:
|
|
189
|
+
return candidates
|
|
190
|
+
return tuple(c for c in candidates if c.label not in _EXPENSIVE_LABELS)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
class WinnerSpec(BaseModel):
|
|
194
|
+
"""The winning candidate's resolved flags, persisted so old artifacts stay self-describing.
|
|
195
|
+
|
|
196
|
+
Without this, `winner` is a foreign key into the in-code candidate tuple — and the ladder
|
|
197
|
+
churns (this PR alone added three candidates), so a rename would break `--max-fidelity`
|
|
198
|
+
loads of every previously built artifact.
|
|
199
|
+
"""
|
|
200
|
+
|
|
201
|
+
label: str
|
|
202
|
+
reasoning: bool = False
|
|
203
|
+
knowledge: bool = False
|
|
204
|
+
verify: bool = False
|
|
205
|
+
grounder: str = "none"
|
|
206
|
+
workspace: bool = False
|
|
207
|
+
top_k: int | None = None
|
|
208
|
+
demo_obs_cap: int | None = None
|
|
209
|
+
|
|
210
|
+
@classmethod
|
|
211
|
+
def from_candidate(cls, candidate: CandidateConfig) -> WinnerSpec:
|
|
212
|
+
return cls(
|
|
213
|
+
label=candidate.label,
|
|
214
|
+
reasoning=candidate.reasoning,
|
|
215
|
+
knowledge=candidate.knowledge,
|
|
216
|
+
verify=candidate.verify,
|
|
217
|
+
grounder=candidate.grounder,
|
|
218
|
+
workspace=candidate.workspace,
|
|
219
|
+
top_k=candidate.top_k,
|
|
220
|
+
demo_obs_cap=candidate.demo_obs_cap,
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
def to_candidate(self) -> CandidateConfig:
|
|
224
|
+
return CandidateConfig(
|
|
225
|
+
label=self.label,
|
|
226
|
+
reasoning=self.reasoning,
|
|
227
|
+
knowledge=self.knowledge,
|
|
228
|
+
verify=self.verify,
|
|
229
|
+
grounder=self.grounder,
|
|
230
|
+
workspace=self.workspace,
|
|
231
|
+
top_k=self.top_k,
|
|
232
|
+
demo_obs_cap=self.demo_obs_cap,
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
class AutoFidelityReport(BaseModel):
|
|
237
|
+
"""The search's outcome, persisted into the artifact for provenance."""
|
|
238
|
+
|
|
239
|
+
winner_label: str
|
|
240
|
+
scores: dict[str, float] = Field(default_factory=dict)
|
|
241
|
+
val_traces: int = 0
|
|
242
|
+
considered: list[str] = Field(default_factory=list) # candidate labels after pruning
|
|
243
|
+
# The winner's resolved flags (None only in pre-WinnerSpec artifacts, which fall back to
|
|
244
|
+
# the in-code label lookup).
|
|
245
|
+
winner_spec: WinnerSpec | None = None
|
|
246
|
+
|
|
247
|
+
@property
|
|
248
|
+
def winner(self) -> CandidateConfig:
|
|
249
|
+
if self.winner_spec is not None:
|
|
250
|
+
return self.winner_spec.to_candidate()
|
|
251
|
+
for candidate in (
|
|
252
|
+
*DEFAULT_CANDIDATES,
|
|
253
|
+
FETCH_CANDIDATE,
|
|
254
|
+
WORKSPACE_CANDIDATE,
|
|
255
|
+
RAG_DEEP_CANDIDATE,
|
|
256
|
+
):
|
|
257
|
+
if candidate.label == self.winner_label:
|
|
258
|
+
return candidate
|
|
259
|
+
raise ValueError(f"unknown winner label {self.winner_label!r}")
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def search_max_fidelity(
|
|
263
|
+
prompt: str,
|
|
264
|
+
train: list[Trace],
|
|
265
|
+
val: list[Trace],
|
|
266
|
+
provider: Provider,
|
|
267
|
+
judge: Judge,
|
|
268
|
+
embedder: Embedder | None,
|
|
269
|
+
*,
|
|
270
|
+
val_cap: int = DEFAULT_VAL_CAP,
|
|
271
|
+
top_k: int = 5,
|
|
272
|
+
seed: int = 0,
|
|
273
|
+
concurrency: int = 4,
|
|
274
|
+
candidates: Sequence[CandidateConfig] | None = None,
|
|
275
|
+
full_ladder: bool = False,
|
|
276
|
+
cheap_only: bool = False,
|
|
277
|
+
incumbent: CandidateConfig | None = None,
|
|
278
|
+
knowledge_text: str | None = None,
|
|
279
|
+
source_pins: str | None = None,
|
|
280
|
+
on_candidate_start: Callable[[str], None] | None = None,
|
|
281
|
+
on_candidate_done: Callable[[str, float], None] | None = None,
|
|
282
|
+
) -> AutoFidelityReport:
|
|
283
|
+
"""Replay-score the candidate configs on (a cap of) the held-out split; return the winner.
|
|
284
|
+
|
|
285
|
+
`candidates=None` computes the corpus signature (zero tokens) and prunes the ladder to the
|
|
286
|
+
levers that can matter for this corpus (`full_ladder=True` skips the pruning — the max
|
|
287
|
+
tier's "be certain" mode). Leak-free by construction: demos and the candidate knowledge
|
|
288
|
+
base both come from `train` only, and the scored `val` traces are the build's held-out
|
|
289
|
+
split.
|
|
290
|
+
|
|
291
|
+
`incumbent` is the floor the search may improve on but never regress below (the lower tier's
|
|
292
|
+
winner / the `low`-tier signature estimate). It is always scored on the same sample, and a
|
|
293
|
+
challenger only displaces it when it beats it by more than `_NOISE_MARGIN` — this is what
|
|
294
|
+
makes the tier ladder monotonic (improve-or-hold) instead of chasing selection noise. With
|
|
295
|
+
no incumbent the winner is just the highest mean fidelity, cheapest-config tie-break.
|
|
296
|
+
"""
|
|
297
|
+
scored_val = val[:val_cap]
|
|
298
|
+
if candidates is None:
|
|
299
|
+
signature = CorpusSignature.from_traces(train)
|
|
300
|
+
candidates = select_candidates(
|
|
301
|
+
signature,
|
|
302
|
+
full_ladder=full_ladder,
|
|
303
|
+
has_pins=source_pins is not None,
|
|
304
|
+
cheap_only=cheap_only,
|
|
305
|
+
)
|
|
306
|
+
if incumbent is not None and not any(c.label == incumbent.label for c in candidates):
|
|
307
|
+
candidates = (incumbent, *candidates)
|
|
308
|
+
source = SourceResolver.from_file(source_pins) if source_pins is not None else None
|
|
309
|
+
tree = RepoTreeResolver(source.pins) if source is not None else None
|
|
310
|
+
# The candidate KB, seeded once (train-only) and reused for every knowledge candidate.
|
|
311
|
+
# `knowledge_text` lets the caller supply the EXACT text the artifact will serve (build
|
|
312
|
+
# seeds into the artifact dir first), so the winning score was measured on the KB that
|
|
313
|
+
# ships — a second independent extraction would be a different nondeterministic text.
|
|
314
|
+
kb_text = knowledge_text
|
|
315
|
+
if kb_text is None and any(c.knowledge for c in candidates):
|
|
316
|
+
kb_text = seeded_knowledge_text(train, provider)
|
|
317
|
+
|
|
318
|
+
scores: dict[str, float] = {}
|
|
319
|
+
best: CandidateConfig = incumbent if incumbent is not None else candidates[0]
|
|
320
|
+
best_score = -1.0
|
|
321
|
+
for candidate in candidates:
|
|
322
|
+
if on_candidate_start is not None:
|
|
323
|
+
on_candidate_start(candidate.label)
|
|
324
|
+
grounder: Grounder | None = FetchGrounder() if candidate.grounder == "fetch" else None
|
|
325
|
+
use_ws = candidate.workspace and source is not None
|
|
326
|
+
report = replay(
|
|
327
|
+
prompt,
|
|
328
|
+
scored_val,
|
|
329
|
+
provider,
|
|
330
|
+
judge,
|
|
331
|
+
retriever=EmbeddingRetriever(embedder) if embedder is not None else None,
|
|
332
|
+
train=train if embedder is not None else None,
|
|
333
|
+
top_k=candidate.top_k if candidate.top_k is not None else top_k,
|
|
334
|
+
max_retrieved_observation_chars=candidate.demo_obs_cap,
|
|
335
|
+
sample_turns="sampled",
|
|
336
|
+
seed=seed,
|
|
337
|
+
concurrency=concurrency,
|
|
338
|
+
knowledge=kb_text if candidate.knowledge else None,
|
|
339
|
+
reasoning=candidate.reasoning,
|
|
340
|
+
verify=candidate.verify,
|
|
341
|
+
grounder=grounder,
|
|
342
|
+
source=source if use_ws else None,
|
|
343
|
+
source_annotate_stale=use_ws,
|
|
344
|
+
tree=tree if use_ws else None,
|
|
345
|
+
)
|
|
346
|
+
scores[candidate.label] = report.mean_score
|
|
347
|
+
if on_candidate_done is not None:
|
|
348
|
+
on_candidate_done(candidate.label, report.mean_score)
|
|
349
|
+
if report.mean_score > best_score:
|
|
350
|
+
best, best_score = candidate, report.mean_score
|
|
351
|
+
|
|
352
|
+
if incumbent is not None:
|
|
353
|
+
# Improve-or-hold: keep the incumbent unless a challenger clears it by > the noise band.
|
|
354
|
+
incumbent_score = scores.get(incumbent.label, -1.0)
|
|
355
|
+
if best.label != incumbent.label and best_score <= incumbent_score + _NOISE_MARGIN:
|
|
356
|
+
# Defensive default: the incumbent is prepended into `candidates` above, so the
|
|
357
|
+
# lookup always finds it today — but a future caller passing an explicit
|
|
358
|
+
# `candidates` omitting it should fall back to the incumbent, not StopIteration.
|
|
359
|
+
best = next((c for c in candidates if c.label == incumbent.label), incumbent)
|
|
360
|
+
|
|
361
|
+
return AutoFidelityReport(
|
|
362
|
+
winner_label=best.label,
|
|
363
|
+
scores=scores,
|
|
364
|
+
val_traces=len(scored_val),
|
|
365
|
+
considered=[c.label for c in candidates],
|
|
366
|
+
winner_spec=WinnerSpec.from_candidate(best),
|
|
367
|
+
)
|