world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Exact task selection for the harbor scorer.
|
|
2
|
+
|
|
3
|
+
Harbor's own `DatasetConfig.task_names` filter uses fnmatch semantics, so a task id containing a
|
|
4
|
+
glob character would silently over-match, and an optimizer's train/heldout split firewall relies
|
|
5
|
+
on exact selection. This module resolves a dataset once, post-filters by exact id, downloads any
|
|
6
|
+
remote (git/package) tasks ONCE, and returns pinned `TaskConfig`s that candidate jobs run
|
|
7
|
+
directly (`tasks=[...]`, `overwrite=False`): per-candidate jobs must never re-clone or clobber
|
|
8
|
+
the shared task cache that concurrent jobs are reading from.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from collections.abc import Sequence
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Protocol
|
|
17
|
+
|
|
18
|
+
from harbor.models.job.config import DatasetConfig
|
|
19
|
+
from harbor.models.trial.config import TaskConfig
|
|
20
|
+
from harbor.tasks.client import BatchDownloadResult, TaskClient, TaskIdType
|
|
21
|
+
|
|
22
|
+
_GIT_COMMIT_PATTERN = re.compile(r"^[0-9a-fA-F]{40}$|^[0-9a-fA-F]{64}$")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class HarborTaskDownloader(Protocol):
|
|
26
|
+
"""The task-download slice of harbor's TaskClient (fakes replace it in tests)."""
|
|
27
|
+
|
|
28
|
+
async def download_tasks(
|
|
29
|
+
self,
|
|
30
|
+
task_ids: list[TaskIdType],
|
|
31
|
+
overwrite: bool = False,
|
|
32
|
+
output_dir: Path | None = None,
|
|
33
|
+
) -> BatchDownloadResult: ...
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
async def resolve_harbor_tasks(
|
|
37
|
+
dataset: DatasetConfig | Path,
|
|
38
|
+
task_ids: Sequence[str],
|
|
39
|
+
*,
|
|
40
|
+
task_client: HarborTaskDownloader | None = None,
|
|
41
|
+
) -> list[TaskConfig]:
|
|
42
|
+
"""Resolve `task_ids` from a harbor dataset (or local task dir) by exact identity.
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
dataset: A harbor `DatasetConfig`, or a local directory of task dirs (shorthand for
|
|
46
|
+
`DatasetConfig(path=...)`).
|
|
47
|
+
task_ids: Exact task names to select, in the order the caller wants them evaluated.
|
|
48
|
+
task_client: Download seam; defaults to harbor's `TaskClient`.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
One pinned `TaskConfig` per requested id, in request order. Git and package tasks are
|
|
52
|
+
downloaded here exactly once (git with `overwrite=True`, since harbor's cache does not
|
|
53
|
+
verify that an existing checkout still matches the requested commit) and returned with
|
|
54
|
+
the resolved commit pinned and `overwrite=False`, so every candidate job reuses the
|
|
55
|
+
resolved local bytes instead of re-cloning or clobbering the shared cache.
|
|
56
|
+
|
|
57
|
+
Raises:
|
|
58
|
+
ValueError: On empty/duplicate ids, a dataset that resolves duplicate task names, ids
|
|
59
|
+
the dataset does not contain, or a git download whose commit cannot be pinned.
|
|
60
|
+
"""
|
|
61
|
+
requested = list(task_ids)
|
|
62
|
+
if not requested or any(not task_id for task_id in requested):
|
|
63
|
+
raise ValueError("task_ids must be nonempty strings")
|
|
64
|
+
if len(requested) != len(set(requested)):
|
|
65
|
+
raise ValueError("task_ids must be unique")
|
|
66
|
+
|
|
67
|
+
if isinstance(dataset, Path):
|
|
68
|
+
dataset = DatasetConfig(path=dataset)
|
|
69
|
+
# Resolve the dataset WITHOUT harbor's filters, then select by exact id ourselves:
|
|
70
|
+
# task_names is an fnmatch pattern list, not an exact-selection API.
|
|
71
|
+
resolved = DatasetConfig.model_validate(dataset.model_dump(mode="python"))
|
|
72
|
+
resolved.task_names = None
|
|
73
|
+
resolved.exclude_task_names = None
|
|
74
|
+
resolved.n_tasks = None
|
|
75
|
+
configs = await resolved.get_task_configs()
|
|
76
|
+
|
|
77
|
+
by_id: dict[str, TaskConfig] = {}
|
|
78
|
+
for config in configs:
|
|
79
|
+
task_id = config.get_task_id().get_name()
|
|
80
|
+
if task_id in by_id:
|
|
81
|
+
raise ValueError(f"harbor dataset resolved duplicate task {task_id!r}")
|
|
82
|
+
by_id[task_id] = config
|
|
83
|
+
missing = sorted(set(requested) - set(by_id))
|
|
84
|
+
if missing:
|
|
85
|
+
raise ValueError(
|
|
86
|
+
f"harbor task selection was not exact: missing={missing}; "
|
|
87
|
+
"check the ids against the dataset's task names"
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
selected = [by_id[task_id].model_copy(deep=True) for task_id in requested]
|
|
91
|
+
return await _pin_remote_tasks(
|
|
92
|
+
selected,
|
|
93
|
+
dataset=resolved,
|
|
94
|
+
task_client=task_client or TaskClient(),
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
async def _pin_remote_tasks(
|
|
99
|
+
selected: list[TaskConfig],
|
|
100
|
+
*,
|
|
101
|
+
dataset: DatasetConfig,
|
|
102
|
+
task_client: HarborTaskDownloader,
|
|
103
|
+
) -> list[TaskConfig]:
|
|
104
|
+
"""Download remote tasks once and pin their provenance with `overwrite=False`."""
|
|
105
|
+
remote_indexes = [
|
|
106
|
+
index
|
|
107
|
+
for index, config in enumerate(selected)
|
|
108
|
+
if config.is_git_task() or config.is_package_task()
|
|
109
|
+
]
|
|
110
|
+
if not remote_indexes:
|
|
111
|
+
return selected
|
|
112
|
+
downloads = await task_client.download_tasks(
|
|
113
|
+
[selected[index].get_task_id() for index in remote_indexes],
|
|
114
|
+
# Refresh git checkouts once, here: harbor's cache does not verify that an existing
|
|
115
|
+
# checkout still came from the requested commit.
|
|
116
|
+
overwrite=dataset.overwrite
|
|
117
|
+
or any(selected[index].is_git_task() for index in remote_indexes),
|
|
118
|
+
output_dir=dataset.download_dir,
|
|
119
|
+
)
|
|
120
|
+
if len(downloads.results) != len(remote_indexes):
|
|
121
|
+
raise ValueError("harbor returned an incomplete task download result")
|
|
122
|
+
for index, download in zip(remote_indexes, downloads.results, strict=True):
|
|
123
|
+
config = selected[index]
|
|
124
|
+
updates: dict[str, object] = {"overwrite": False}
|
|
125
|
+
if config.is_git_task():
|
|
126
|
+
commit = download.resolved_git_commit_id
|
|
127
|
+
if commit is None or _GIT_COMMIT_PATTERN.fullmatch(commit) is None:
|
|
128
|
+
raise ValueError(
|
|
129
|
+
"harbor did not resolve a git commit for task "
|
|
130
|
+
f"{config.get_task_id().get_name()!r}"
|
|
131
|
+
)
|
|
132
|
+
requested_commit = config.git_commit_id
|
|
133
|
+
if requested_commit is not None and requested_commit.lower() != commit.lower():
|
|
134
|
+
raise ValueError(
|
|
135
|
+
"harbor resolved a different git commit for task "
|
|
136
|
+
f"{config.get_task_id().get_name()!r}"
|
|
137
|
+
)
|
|
138
|
+
updates["git_commit_id"] = commit.lower()
|
|
139
|
+
selected[index] = config.model_copy(update=updates)
|
|
140
|
+
return selected
|
wmo/evals/open_loop.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""Open-loop evaluation: reconstruction fidelity over trace files (the default `wmo eval` mode).
|
|
2
|
+
|
|
3
|
+
`replay` (in `wmo.engine.replay`) scores one corpus of held-out steps teacher-forced. This
|
|
4
|
+
orchestration layer is what `wmo eval` calls: it loads one or more OTel trace files, splits each
|
|
5
|
+
into train/holdout, replays the holdout through a world-model prompt with leak-free RAG, and
|
|
6
|
+
aggregates a per-file + overall scorecard. Its closed-loop counterpart
|
|
7
|
+
(`wmo eval --mode closed-loop`, `wmo.evals.closed_loop`) runs a live agent instead of replaying;
|
|
8
|
+
both implement the `Evaluation` interface in `wmo.evals.base`.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from statistics import fmean, pstdev
|
|
15
|
+
|
|
16
|
+
from pydantic import BaseModel, Field
|
|
17
|
+
|
|
18
|
+
from wmo.engine.build import split_traces, split_traces_3way
|
|
19
|
+
from wmo.engine.knowledge import seeded_knowledge_text
|
|
20
|
+
from wmo.engine.replay import ReplayReport, replay, valid_scores
|
|
21
|
+
from wmo.ingest import get_adapter
|
|
22
|
+
from wmo.optimize.judge import Judge
|
|
23
|
+
from wmo.providers.base import Embedder, Provider
|
|
24
|
+
from wmo.retrieval import EmbeddingRetriever
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class EvalReport(BaseModel):
|
|
28
|
+
"""Per-file fidelity reports plus the step-weighted overall mean ± std.
|
|
29
|
+
|
|
30
|
+
`per_file` maps a trace file's clean name to its `ReplayReport` (per-step `StepResult`s), and
|
|
31
|
+
`overall_fidelity`/`overall_std` are the step-weighted aggregates across files.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
per_file: dict[str, ReplayReport] = Field(default_factory=dict)
|
|
35
|
+
overall_fidelity: float = 0.0 # step-weighted mean of valid per-step scores across all files
|
|
36
|
+
overall_std: float = 0.0 # std of valid per-step scores across all files
|
|
37
|
+
total_steps: int = 0 # all steps attempted, including judge-invalid ones
|
|
38
|
+
total_invalid: int = 0 # judge failures across files; excluded from fidelity/std
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def headline(self) -> float:
|
|
42
|
+
"""The `EvalResult` headline: per-step reconstruction fidelity."""
|
|
43
|
+
return self.overall_fidelity
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def total_valid(self) -> int:
|
|
47
|
+
"""Steps that actually back the fidelity mean (judge-invalid ones excluded)."""
|
|
48
|
+
return self.total_steps - self.total_invalid
|
|
49
|
+
|
|
50
|
+
def summary(self) -> str:
|
|
51
|
+
invalid = f", {self.total_invalid} judge-invalid excluded" if self.total_invalid else ""
|
|
52
|
+
return (
|
|
53
|
+
f"fidelity={self.overall_fidelity:.3f}±{self.overall_std:.3f} "
|
|
54
|
+
f"({self.total_steps} steps, {len(self.per_file)} file(s){invalid})"
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def evaluate_files(
|
|
59
|
+
files: list[Path],
|
|
60
|
+
prompt: str,
|
|
61
|
+
provider: Provider,
|
|
62
|
+
judge: Judge,
|
|
63
|
+
*,
|
|
64
|
+
embedder: Embedder | None = None,
|
|
65
|
+
train_split: float = 0.7,
|
|
66
|
+
val_frac: float | None = None,
|
|
67
|
+
top_k: int = 5,
|
|
68
|
+
sample_turns: str = "all",
|
|
69
|
+
seed: int = 0,
|
|
70
|
+
adapter_name: str = "otel-genai",
|
|
71
|
+
max_holdout_traces: int | None = None,
|
|
72
|
+
knowledge: bool = False,
|
|
73
|
+
reasoning: bool = False,
|
|
74
|
+
) -> EvalReport:
|
|
75
|
+
"""Replay-score each trace file's held-out split. `embedder=None` -> zero-shot (no retrieval).
|
|
76
|
+
|
|
77
|
+
Each file is split deterministically; tiny corpora with no held-out trace fall back to scoring
|
|
78
|
+
every trace. RAG, when enabled, retrieves from that file's own train split only (leak-free).
|
|
79
|
+
`sample_turns`/`seed` are forwarded to `replay` (see its docstring). `max_holdout_traces` caps
|
|
80
|
+
how many held-out traces are scored per file (a deterministic prefix by trace_id) - for cheap
|
|
81
|
+
dry-runs; the train side stays full so retrieval is unaffected.
|
|
82
|
+
|
|
83
|
+
`val_frac` makes the split leak-free against a GEPA-evolved prompt: when it is a positive
|
|
84
|
+
fraction, the traces are cut 3-way (`train`/`val`/`test`) on the same hash line GEPA used, and
|
|
85
|
+
only the reserved `test` band is scored, so a prompt selected on `val` is never graded on those
|
|
86
|
+
same traces. Retrieval still draws from `train` only. `val_frac` of `None` or `0` keeps the
|
|
87
|
+
plain 2-way `train`/held-out split (`split_traces_3way` requires a strictly positive band).
|
|
88
|
+
|
|
89
|
+
`knowledge` seeds an ephemeral knowledge base from each file's TRAIN split (never the holdout -
|
|
90
|
+
the same leak-free discipline as RAG) and renders it into every prediction; `reasoning`
|
|
91
|
+
switches predictions to the deliberate-then-answer contract. Both mirror the serving engine's
|
|
92
|
+
agentic mode. Closed-loop evals get agentic mode from the ARTIFACT instead (the served
|
|
93
|
+
WorldModel's config / --max-fidelity winner), not from these flags.
|
|
94
|
+
"""
|
|
95
|
+
adapter = get_adapter(adapter_name)
|
|
96
|
+
per_file: dict[str, ReplayReport] = {}
|
|
97
|
+
for path in files:
|
|
98
|
+
traces = adapter.from_file(str(path))
|
|
99
|
+
if not traces:
|
|
100
|
+
continue
|
|
101
|
+
if val_frac: # truthy (a positive val band); None or 0.0 keeps the plain 2-way split
|
|
102
|
+
train, _val, holdout = split_traces_3way(traces, train_split, val_frac)
|
|
103
|
+
else:
|
|
104
|
+
train, holdout = split_traces(traces, train_split)
|
|
105
|
+
if not holdout: # tiny corpus: evaluate on everything
|
|
106
|
+
train, holdout = traces, traces
|
|
107
|
+
if max_holdout_traces is not None:
|
|
108
|
+
holdout = sorted(holdout, key=lambda t: t.trace_id)[:max_holdout_traces]
|
|
109
|
+
retriever = EmbeddingRetriever(embedder) if embedder is not None else None
|
|
110
|
+
# Ephemeral, per-file, train-only KB: rendered text only — nothing under models/ is read
|
|
111
|
+
# or written, so eval can never leak a serve-time learned.md into scoring.
|
|
112
|
+
knowledge_text = seeded_knowledge_text(train, provider) if knowledge else None
|
|
113
|
+
name = _display_name(path)
|
|
114
|
+
per_file[name] = replay(
|
|
115
|
+
prompt,
|
|
116
|
+
holdout,
|
|
117
|
+
provider,
|
|
118
|
+
judge,
|
|
119
|
+
retriever=retriever,
|
|
120
|
+
train=train if embedder is not None else None,
|
|
121
|
+
top_k=top_k,
|
|
122
|
+
sample_turns=sample_turns,
|
|
123
|
+
seed=seed,
|
|
124
|
+
knowledge=knowledge_text,
|
|
125
|
+
reasoning=reasoning,
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
# Step-weighted aggregate over every validly-judged step across files (judge failures are
|
|
129
|
+
# counted in total_invalid, never as spurious zeros — see replay.valid_scores).
|
|
130
|
+
step_scores = valid_scores(r for rep in per_file.values() for r in rep.results)
|
|
131
|
+
overall = fmean(step_scores) if step_scores else 0.0
|
|
132
|
+
overall_std = pstdev(step_scores) if len(step_scores) > 1 else 0.0
|
|
133
|
+
return EvalReport(
|
|
134
|
+
per_file=per_file,
|
|
135
|
+
overall_fidelity=overall,
|
|
136
|
+
overall_std=overall_std,
|
|
137
|
+
total_steps=sum(rep.n_steps for rep in per_file.values()),
|
|
138
|
+
total_invalid=sum(rep.n_invalid for rep in per_file.values()),
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _display_name(path: Path) -> str:
|
|
143
|
+
"""Human label for a corpus, using the example folder name for `traces.otel.jsonl`."""
|
|
144
|
+
name = path.name.removesuffix(".jsonl").removesuffix(".otel")
|
|
145
|
+
return path.parent.name if name == "traces" else name
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
class OpenLoopEval:
|
|
149
|
+
"""The open-loop `Evaluation`: teacher-forced replay of held-out trace steps."""
|
|
150
|
+
|
|
151
|
+
def __init__(
|
|
152
|
+
self,
|
|
153
|
+
files: list[Path],
|
|
154
|
+
prompt: str,
|
|
155
|
+
provider: Provider,
|
|
156
|
+
judge: Judge,
|
|
157
|
+
*,
|
|
158
|
+
embedder: Embedder | None = None,
|
|
159
|
+
train_split: float = 0.7,
|
|
160
|
+
top_k: int = 5,
|
|
161
|
+
sample_turns: str = "all",
|
|
162
|
+
seed: int = 0,
|
|
163
|
+
adapter_name: str = "otel-genai",
|
|
164
|
+
knowledge: bool = False,
|
|
165
|
+
reasoning: bool = False,
|
|
166
|
+
) -> None:
|
|
167
|
+
self._files = files
|
|
168
|
+
self._prompt = prompt
|
|
169
|
+
self._provider = provider
|
|
170
|
+
self._judge = judge
|
|
171
|
+
self._embedder = embedder
|
|
172
|
+
self._train_split = train_split
|
|
173
|
+
self._top_k = top_k
|
|
174
|
+
self._sample_turns = sample_turns
|
|
175
|
+
self._seed = seed
|
|
176
|
+
self._adapter_name = adapter_name
|
|
177
|
+
self._knowledge = knowledge
|
|
178
|
+
self._reasoning = reasoning
|
|
179
|
+
|
|
180
|
+
def run(self) -> EvalReport:
|
|
181
|
+
return evaluate_files(
|
|
182
|
+
self._files,
|
|
183
|
+
self._prompt,
|
|
184
|
+
self._provider,
|
|
185
|
+
self._judge,
|
|
186
|
+
embedder=self._embedder,
|
|
187
|
+
train_split=self._train_split,
|
|
188
|
+
top_k=self._top_k,
|
|
189
|
+
sample_turns=self._sample_turns,
|
|
190
|
+
seed=self._seed,
|
|
191
|
+
adapter_name=self._adapter_name,
|
|
192
|
+
knowledge=self._knowledge,
|
|
193
|
+
reasoning=self._reasoning,
|
|
194
|
+
)
|
wmo/evals/tasks.py
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Task specs for closed-loop evaluation: an instruction plus gold assertions that define success.
|
|
2
|
+
|
|
3
|
+
Gold assertions are semantic post-conditions the `GoldJudge` checks against the run transcript —
|
|
4
|
+
conditions on the final state, made robust to wording by an LLM judge instead of brittle exact
|
|
5
|
+
matching. Tasks are typically derived from the same benchmark the world model's traces came from —
|
|
6
|
+
`Trace.metadata` already carries gold assertions for traces captured with them.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, Field, field_validator
|
|
14
|
+
|
|
15
|
+
from wmo.core.text import normalize_durable_text
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TaskSpec(BaseModel):
|
|
19
|
+
"""One task: what the agent must do, and the assertions that must hold afterwards."""
|
|
20
|
+
|
|
21
|
+
task_id: str
|
|
22
|
+
instruction: str
|
|
23
|
+
gold: list[str] = Field(default_factory=list) # assertions that define success
|
|
24
|
+
|
|
25
|
+
@field_validator("task_id", "instruction")
|
|
26
|
+
@classmethod
|
|
27
|
+
def _normalize_scalar_text(cls, value: str) -> str:
|
|
28
|
+
return normalize_durable_text(value)
|
|
29
|
+
|
|
30
|
+
@field_validator("gold")
|
|
31
|
+
@classmethod
|
|
32
|
+
def _normalize_gold(cls, value: list[str]) -> list[str]:
|
|
33
|
+
return [normalize_durable_text(assertion) for assertion in value]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def load_tasks(path: str | Path) -> list[TaskSpec]:
|
|
37
|
+
"""Read a JSONL task file (one TaskSpec per line; blank lines ignored).
|
|
38
|
+
|
|
39
|
+
Duplicate `task_id`s are an error: reports key outcomes by task_id, so a duplicate would run
|
|
40
|
+
(and cost) k passes twice while silently keeping only the last outcome.
|
|
41
|
+
"""
|
|
42
|
+
tasks: list[TaskSpec] = []
|
|
43
|
+
for line in Path(path).read_text(encoding="utf-8").splitlines():
|
|
44
|
+
stripped = line.strip()
|
|
45
|
+
if stripped:
|
|
46
|
+
tasks.append(TaskSpec.model_validate_json(stripped))
|
|
47
|
+
if not tasks:
|
|
48
|
+
raise ValueError(f"no tasks in {path}")
|
|
49
|
+
ids = [t.task_id for t in tasks]
|
|
50
|
+
duplicates = sorted({i for i in ids if ids.count(i) > 1})
|
|
51
|
+
if duplicates:
|
|
52
|
+
raise ValueError(f"duplicate task_id(s) in {path}: {duplicates}")
|
|
53
|
+
return tasks
|
wmo/harness/__init__.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""The agent harness: the scaffold a live agent runs with, and the machinery to improve it.
|
|
2
|
+
|
|
3
|
+
A minimal, fixed agent loop (`AgentRuntime`) drives one action at a time against an
|
|
4
|
+
`AgentEnvironment` — an interface, not a backend: closed-loop eval (`wmo.evals.closed_loop`) binds
|
|
5
|
+
it to the world model, and a real execution backend can bind the same loop to reality. What the
|
|
6
|
+
loop runs with is a `HarnessDoc` — a typed document of identity-keyed surfaces (prompt sections,
|
|
7
|
+
tool policy, loop params, skills) stored as immutable versions with movable aliases
|
|
8
|
+
(`wmo.harness.store`) and updated through audited `HarnessDelta`s (`wmo.harness.delta`,
|
|
9
|
+
docs/reference/harness_delta.md) proposed by a meta-agent and gated on non-regression
|
|
10
|
+
(`wmo.harness.create`, the `wmo optimize` search).
|
|
11
|
+
|
|
12
|
+
`create` and `mutate` are imported directly (not re-exported here): they depend on
|
|
13
|
+
`wmo.evals.closed_loop`, which itself binds to this package's runtime — re-exporting them would
|
|
14
|
+
make `import wmo.evals` observe a partially initialized module.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from wmo.harness.delta import (
|
|
18
|
+
FailureSignature,
|
|
19
|
+
GateRecord,
|
|
20
|
+
HarnessDelta,
|
|
21
|
+
SurfaceOp,
|
|
22
|
+
apply_delta,
|
|
23
|
+
)
|
|
24
|
+
from wmo.harness.doc import HarnessDoc, Surface, SurfaceKind
|
|
25
|
+
from wmo.harness.environment import AgentEnvironment, is_env_action
|
|
26
|
+
from wmo.harness.runtime import AgentRuntime, RunResult, StopReason
|
|
27
|
+
from wmo.harness.skills import Skill, SkillLibrary
|
|
28
|
+
from wmo.harness.store import HarnessStore
|
|
29
|
+
from wmo.harness.tools import TOOL_REGISTRY, ToolCall, parse_tool_call
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"TOOL_REGISTRY",
|
|
33
|
+
"AgentEnvironment",
|
|
34
|
+
"AgentRuntime",
|
|
35
|
+
"FailureSignature",
|
|
36
|
+
"GateRecord",
|
|
37
|
+
"HarnessDelta",
|
|
38
|
+
"HarnessDoc",
|
|
39
|
+
"HarnessStore",
|
|
40
|
+
"RunResult",
|
|
41
|
+
"Skill",
|
|
42
|
+
"SkillLibrary",
|
|
43
|
+
"StopReason",
|
|
44
|
+
"Surface",
|
|
45
|
+
"SurfaceKind",
|
|
46
|
+
"SurfaceOp",
|
|
47
|
+
"ToolCall",
|
|
48
|
+
"apply_delta",
|
|
49
|
+
"is_env_action",
|
|
50
|
+
"parse_tool_call",
|
|
51
|
+
]
|