world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,495 @@
|
|
|
1
|
+
"""`PiRuntime`: run the vendored pi agent (a real multi-file TypeScript harness) as an episode.
|
|
2
|
+
|
|
3
|
+
The harness under search is the pi agent's own source: each file is a `code:` surface carrying a
|
|
4
|
+
`path`. To run one task the runtime materializes those files into a checkout on a runner box,
|
|
5
|
+
starts a local shim, and drives pi headless through it (`wmo/harness/pi_entry/entry.ts`):
|
|
6
|
+
|
|
7
|
+
- pi's LLM calls hit the shim's OpenAI-compatible `/v1/chat/completions`; the shim validates the
|
|
8
|
+
structured request and delegates it to the caller's tool-calling provider. Provider-owned auth,
|
|
9
|
+
routing, translation, retries, and waterfall failover stay on the control host.
|
|
10
|
+
- pi's task tools POST `/tool`, which the runtime answers from the `AgentEnvironment` (the world
|
|
11
|
+
model in simulation, the real backend in the transfer check). These calls are the recorded
|
|
12
|
+
transcript the judge grades.
|
|
13
|
+
- `submit` POSTs `/done`; the runtime returns a `RunResult` shaped exactly like the other runtimes.
|
|
14
|
+
|
|
15
|
+
The runner is remote (node lives on a separate box, never the control host), reached over SSH with
|
|
16
|
+
a reverse tunnel so the runner's node process can call back to the shim. The environment budget is
|
|
17
|
+
enforced kit-style: past the cap, `/tool` returns an error observation and the episode ends.
|
|
18
|
+
|
|
19
|
+
Concurrency note: episodes are serialized on one runner directory + port. Parallel rollouts must
|
|
20
|
+
pass distinct `port`/`workdir` (a per-episode caller responsibility); the default is a single
|
|
21
|
+
sequential lane, which is what the current search driver uses.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import json
|
|
27
|
+
import os
|
|
28
|
+
import re
|
|
29
|
+
import subprocess
|
|
30
|
+
import threading
|
|
31
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
32
|
+
|
|
33
|
+
from llm_waterfall import ChatRequest
|
|
34
|
+
from pydantic import JsonValue
|
|
35
|
+
|
|
36
|
+
from wmo.core.types import Action, ActionKind, EnvState, JsonObject, Observation, Step
|
|
37
|
+
from wmo.harness.environment import AgentEnvironment, is_env_action
|
|
38
|
+
from wmo.harness.runner_link import params_schema, provider_context_window, stop_reason_for_done
|
|
39
|
+
from wmo.harness.runtime import (
|
|
40
|
+
DEFAULT_EVAL_EPISODE_TIMEOUT_S,
|
|
41
|
+
DEFAULT_MAX_OUTPUT_TOKENS,
|
|
42
|
+
DEFAULT_MAX_TURNS,
|
|
43
|
+
RunResult,
|
|
44
|
+
StopReason,
|
|
45
|
+
validate_episode_timeout_s,
|
|
46
|
+
)
|
|
47
|
+
from wmo.harness.skills import SkillLibrary
|
|
48
|
+
from wmo.harness.tools import READ_SKILL, ToolSpec
|
|
49
|
+
from wmo.providers.base import UNPARSED_TOOL_CALLS_KEY, Provider, ToolCallingProvider
|
|
50
|
+
|
|
51
|
+
# The runner: node runs here, reached over SSH. The checkout keeps pi's node_modules; per-episode
|
|
52
|
+
# source is overwritten from the harness surfaces.
|
|
53
|
+
PI_RUNNER_HOST = os.environ.get("PI_RUNNER_HOST", "kion@nucbox.local")
|
|
54
|
+
PI_RUNNER_DIR = os.environ.get("PI_RUNNER_DIR", "~/pi-run")
|
|
55
|
+
DEFAULT_MAX_ENV_ACTIONS = 40
|
|
56
|
+
_PI_ENTRY_DIR = os.path.join(os.path.dirname(__file__), "pi_entry")
|
|
57
|
+
_ENTRY_TS = os.path.join(_PI_ENTRY_DIR, "entry.ts")
|
|
58
|
+
# entry.ts imports the shared classify + nudge policy, so it must be materialized alongside it.
|
|
59
|
+
_TERMINATION_TS = os.path.join(_PI_ENTRY_DIR, "runner_termination.ts")
|
|
60
|
+
# Cleanup headroom past the node wall budget: one SSH round trip plus process teardown.
|
|
61
|
+
_NODE_TEARDOWN_GRACE_S = 60.0
|
|
62
|
+
# Runner paths are interpolated into remote shell commands, so restrict them to characters that
|
|
63
|
+
# cannot break out of the command (allows `~` expansion; rejects spaces, quotes, `;`, `$`, etc.).
|
|
64
|
+
_SAFE_REMOTE_PATH = re.compile(r"^[A-Za-z0-9_./~-]+$")
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class _MaterializeError(RuntimeError):
|
|
68
|
+
"""Remote source materialization failed; the episode must not run stale files."""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class _Episode:
|
|
72
|
+
"""Mutable per-run state the shim handlers share."""
|
|
73
|
+
|
|
74
|
+
def __init__(
|
|
75
|
+
self,
|
|
76
|
+
*,
|
|
77
|
+
instruction: str,
|
|
78
|
+
system_prompt: str,
|
|
79
|
+
tools: list[ToolSpec],
|
|
80
|
+
provider: ToolCallingProvider,
|
|
81
|
+
environment: AgentEnvironment,
|
|
82
|
+
temperature: float,
|
|
83
|
+
skills: SkillLibrary,
|
|
84
|
+
max_env_actions: int,
|
|
85
|
+
max_turns: int,
|
|
86
|
+
max_output_tokens: int,
|
|
87
|
+
context_window: int | None = None,
|
|
88
|
+
) -> None:
|
|
89
|
+
self.instruction = instruction
|
|
90
|
+
self.system_prompt = system_prompt
|
|
91
|
+
self.tools = tools
|
|
92
|
+
self.provider = provider
|
|
93
|
+
self.environment = environment
|
|
94
|
+
self.temperature = temperature
|
|
95
|
+
self.skills = skills
|
|
96
|
+
self.max_env_actions = max_env_actions
|
|
97
|
+
self.max_turns = max_turns
|
|
98
|
+
self.max_output_tokens = max_output_tokens
|
|
99
|
+
self.context_window = context_window
|
|
100
|
+
self.steps: list[Step] = []
|
|
101
|
+
self.answer: str = ""
|
|
102
|
+
self.proxy_error: str = ""
|
|
103
|
+
self.done_reason: str = ""
|
|
104
|
+
# The host's view of the most recent worker completion, which entry.ts reads back through
|
|
105
|
+
# GET /signal so its termination classifier sees the same evidence the frame runners get.
|
|
106
|
+
self.finish_reason: str = ""
|
|
107
|
+
self.unparsed_tool_calls: list[str] = []
|
|
108
|
+
self.tool_call_turns: int = 0
|
|
109
|
+
self.done = threading.Event()
|
|
110
|
+
self._env_calls = 0
|
|
111
|
+
|
|
112
|
+
def task_json(self) -> JsonObject:
|
|
113
|
+
return {
|
|
114
|
+
"instruction": self.instruction,
|
|
115
|
+
"system": self.system_prompt,
|
|
116
|
+
"max_turns": self.max_turns,
|
|
117
|
+
"max_output_tokens": self.max_output_tokens,
|
|
118
|
+
"context_window": self.context_window,
|
|
119
|
+
"tools": [
|
|
120
|
+
{"name": t.name, "description": t.description, "parameters": _params_schema(t)}
|
|
121
|
+
for t in self.tools
|
|
122
|
+
],
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
def signal_json(self) -> JsonObject:
|
|
126
|
+
"""The host's view of the last worker completion, for entry.ts's classifier."""
|
|
127
|
+
return {
|
|
128
|
+
"finish_reason": self.finish_reason,
|
|
129
|
+
"unparsed_tool_calls": list(self.unparsed_tool_calls),
|
|
130
|
+
"provider_error": self.proxy_error,
|
|
131
|
+
"tool_call_turns": self.tool_call_turns,
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
def run_tool(self, name: str, arguments: JsonObject) -> JsonObject:
|
|
135
|
+
action = Action(kind=ActionKind.TOOL_CALL, name=name, arguments=arguments)
|
|
136
|
+
if name not in {t.name for t in self.tools}:
|
|
137
|
+
obs = Observation(content=f"tool {name!r} not available", is_error=True)
|
|
138
|
+
elif name == READ_SKILL.name:
|
|
139
|
+
raw_name = arguments.get("name")
|
|
140
|
+
skill_name = raw_name if isinstance(raw_name, str) else ""
|
|
141
|
+
skill = self.skills.get(skill_name)
|
|
142
|
+
if skill is None:
|
|
143
|
+
obs = Observation(content=f"no skill named {skill_name!r}", is_error=True)
|
|
144
|
+
else:
|
|
145
|
+
obs = Observation(content=skill.body)
|
|
146
|
+
elif self._env_calls >= self.max_env_actions:
|
|
147
|
+
obs = Observation(content="environment action budget exhausted", is_error=True)
|
|
148
|
+
elif not is_env_action(action):
|
|
149
|
+
obs = Observation(content=f"tool {name!r} not available", is_error=True)
|
|
150
|
+
else:
|
|
151
|
+
self._env_calls += 1
|
|
152
|
+
obs = self.environment.execute(action)
|
|
153
|
+
self.steps.append(
|
|
154
|
+
Step(action=action, observation=obs, state_before=EnvState(), task=self.instruction)
|
|
155
|
+
)
|
|
156
|
+
return {"content": obs.content, "is_error": obs.is_error}
|
|
157
|
+
|
|
158
|
+
def worker_request(self, body: JsonObject) -> ChatRequest:
|
|
159
|
+
"""Apply the document sampling policy to one runner-authored structured request."""
|
|
160
|
+
request_body = dict(body)
|
|
161
|
+
request_body["temperature"] = self.temperature
|
|
162
|
+
return ChatRequest.model_validate(request_body)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
# The tool `parameters` schema builder lives in runner_link (shared with the frame transport);
|
|
166
|
+
# re-exported here under its old private name so existing callers and tests keep working.
|
|
167
|
+
_params_schema = params_schema
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
class _ShimServer(ThreadingHTTPServer):
|
|
171
|
+
"""A threading HTTP server that carries the current episode for its handlers."""
|
|
172
|
+
|
|
173
|
+
# Environment calls mutate evaluator-owned state, so server_close() must join every active
|
|
174
|
+
# handler: no environment write may land after the episode is declared finished (against a
|
|
175
|
+
# real execution environment a late write would mutate state the evaluator is already
|
|
176
|
+
# verifying; with the world model it was merely cosmetic). The join is bounded by whatever
|
|
177
|
+
# the slowest handler is blocked on, worst case a completion handler waiting out the provider
|
|
178
|
+
# SDK's timeout and retries (minutes during an outage), not just a tool command's budget. A
|
|
179
|
+
# slow close is the accepted price of a trustworthy verdict.
|
|
180
|
+
daemon_threads = False
|
|
181
|
+
episode: _Episode
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
class _ShimHandler(BaseHTTPRequestHandler):
|
|
185
|
+
# HTTP/1.1 so the OpenAI SDK's keep-alive works; the SSE handler forces a fresh socket per
|
|
186
|
+
# turn (see _serve_completion) to avoid mis-framing the pipelined next request.
|
|
187
|
+
protocol_version = "HTTP/1.1"
|
|
188
|
+
|
|
189
|
+
def log_message(self, format: str, *args: object) -> None: # noqa: A002 - base API name
|
|
190
|
+
return # silence per-request stderr spam
|
|
191
|
+
|
|
192
|
+
@property
|
|
193
|
+
def _ep(self) -> _Episode:
|
|
194
|
+
assert isinstance(self.server, _ShimServer)
|
|
195
|
+
return self.server.episode
|
|
196
|
+
|
|
197
|
+
def _read_body(self) -> JsonObject:
|
|
198
|
+
length = int(self.headers.get("Content-Length", 0))
|
|
199
|
+
raw = self.rfile.read(length) if length else b"{}"
|
|
200
|
+
return json.loads(raw or b"{}")
|
|
201
|
+
|
|
202
|
+
def _send_json(self, obj: JsonObject, status: int = 200) -> None:
|
|
203
|
+
body = json.dumps(obj).encode("utf-8")
|
|
204
|
+
self.send_response(status)
|
|
205
|
+
self.send_header("Content-Type", "application/json")
|
|
206
|
+
self.send_header("Content-Length", str(len(body)))
|
|
207
|
+
self.end_headers()
|
|
208
|
+
self.wfile.write(body)
|
|
209
|
+
|
|
210
|
+
def do_GET(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
|
|
211
|
+
path = self.path.rstrip("/")
|
|
212
|
+
if path == "/task":
|
|
213
|
+
self._send_json(self._ep.task_json())
|
|
214
|
+
elif path == "/signal":
|
|
215
|
+
self._send_json(self._ep.signal_json())
|
|
216
|
+
else:
|
|
217
|
+
self._send_json({"error": "not found"}, status=404)
|
|
218
|
+
|
|
219
|
+
def do_POST(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
|
|
220
|
+
path = self.path.rstrip("/")
|
|
221
|
+
if path == "/v1/chat/completions":
|
|
222
|
+
self._serve_completion(self._read_body())
|
|
223
|
+
elif path == "/tool":
|
|
224
|
+
body = self._read_body()
|
|
225
|
+
name = body.get("name")
|
|
226
|
+
args = body.get("arguments")
|
|
227
|
+
self._send_json(
|
|
228
|
+
self._ep.run_tool(
|
|
229
|
+
name if isinstance(name, str) else "",
|
|
230
|
+
args if isinstance(args, dict) else {},
|
|
231
|
+
)
|
|
232
|
+
)
|
|
233
|
+
elif path == "/done":
|
|
234
|
+
body = self._read_body()
|
|
235
|
+
answer = body.get("answer")
|
|
236
|
+
reason = body.get("reason")
|
|
237
|
+
self._ep.answer = answer if isinstance(answer, str) else ""
|
|
238
|
+
self._ep.done_reason = reason if isinstance(reason, str) else ""
|
|
239
|
+
self._send_json({})
|
|
240
|
+
self._ep.done.set()
|
|
241
|
+
else:
|
|
242
|
+
self._send_json({"error": "not found"}, status=404)
|
|
243
|
+
|
|
244
|
+
def _serve_completion(self, body: JsonObject) -> None:
|
|
245
|
+
"""Delegate pi's structured request to the provider and synthesize OpenAI SSE."""
|
|
246
|
+
self.send_response(200)
|
|
247
|
+
self.send_header("Content-Type", "text/event-stream")
|
|
248
|
+
self.send_header("Connection", "close")
|
|
249
|
+
self.end_headers()
|
|
250
|
+
self.close_connection = True
|
|
251
|
+
try:
|
|
252
|
+
completion = self._ep.provider.complete_chat(self._ep.worker_request(body))
|
|
253
|
+
choice = completion.choices[0]
|
|
254
|
+
message = choice.message
|
|
255
|
+
# Record the host's view of this turn BEFORE streaming it: entry.ts reads it back to
|
|
256
|
+
# classify why the episode ended (truncated at the cap vs unparsed vs prose only).
|
|
257
|
+
self._ep.proxy_error = ""
|
|
258
|
+
self._ep.finish_reason = choice.finish_reason or "stop"
|
|
259
|
+
unparsed = (choice.model_extra or {}).get(UNPARSED_TOOL_CALLS_KEY)
|
|
260
|
+
self._ep.unparsed_tool_calls = (
|
|
261
|
+
[str(item) for item in unparsed] if isinstance(unparsed, list) else []
|
|
262
|
+
)
|
|
263
|
+
if message.tool_calls:
|
|
264
|
+
self._ep.tool_call_turns += 1
|
|
265
|
+
content = message.content if isinstance(message.content, str) else ""
|
|
266
|
+
delta: dict[str, JsonValue] = {"role": "assistant", "content": content}
|
|
267
|
+
if message.tool_calls:
|
|
268
|
+
delta["tool_calls"] = [
|
|
269
|
+
{
|
|
270
|
+
"index": i,
|
|
271
|
+
**tool_call.model_dump(mode="json"),
|
|
272
|
+
}
|
|
273
|
+
for i, tool_call in enumerate(message.tool_calls)
|
|
274
|
+
]
|
|
275
|
+
first = {"choices": [{"index": 0, "delta": delta, "finish_reason": None}]}
|
|
276
|
+
last = {
|
|
277
|
+
"choices": [
|
|
278
|
+
{"index": 0, "delta": {}, "finish_reason": choice.finish_reason or "stop"}
|
|
279
|
+
]
|
|
280
|
+
}
|
|
281
|
+
self.wfile.write(f"data: {json.dumps(first)}\n\n".encode())
|
|
282
|
+
self.wfile.write(f"data: {json.dumps(last)}\n\n".encode())
|
|
283
|
+
self.wfile.write(b"data: [DONE]\n\n")
|
|
284
|
+
except Exception as exc: # noqa: BLE001 - never crash the shim
|
|
285
|
+
self._ep.proxy_error = str(exc)
|
|
286
|
+
err = json.dumps({"error": {"message": f"agent provider failed: {exc}"}})
|
|
287
|
+
self.wfile.write(f"data: {err}\n\ndata: [DONE]\n\n".encode())
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
class PiRuntime:
|
|
291
|
+
"""Runs one episode of the vendored pi harness against an `AgentEnvironment`."""
|
|
292
|
+
|
|
293
|
+
def __init__(
|
|
294
|
+
self,
|
|
295
|
+
provider: Provider,
|
|
296
|
+
*,
|
|
297
|
+
files: dict[str, str],
|
|
298
|
+
tools: list[ToolSpec],
|
|
299
|
+
temperature: float = 0.7,
|
|
300
|
+
skills: SkillLibrary | None = None,
|
|
301
|
+
system_prompt: str = "",
|
|
302
|
+
port: int = 8891,
|
|
303
|
+
workdir: str | None = None,
|
|
304
|
+
max_env_actions: int = DEFAULT_MAX_ENV_ACTIONS,
|
|
305
|
+
max_turns: int = DEFAULT_MAX_TURNS,
|
|
306
|
+
max_output_tokens: int = DEFAULT_MAX_OUTPUT_TOKENS,
|
|
307
|
+
episode_timeout_s: float = DEFAULT_EVAL_EPISODE_TIMEOUT_S,
|
|
308
|
+
context_window: int | None = None,
|
|
309
|
+
) -> None:
|
|
310
|
+
if not isinstance(provider, ToolCallingProvider):
|
|
311
|
+
raise TypeError("PiRuntime needs a ToolCallingProvider")
|
|
312
|
+
self._provider = provider
|
|
313
|
+
self._files = files
|
|
314
|
+
self._skills = skills if skills is not None else SkillLibrary()
|
|
315
|
+
self._tools = list(tools)
|
|
316
|
+
if len(self._skills) and READ_SKILL.name not in {tool.name for tool in self._tools}:
|
|
317
|
+
self._tools.append(READ_SKILL)
|
|
318
|
+
if not 0.0 <= temperature <= 2.0:
|
|
319
|
+
raise ValueError("temperature must be in [0, 2]")
|
|
320
|
+
self._temperature = temperature
|
|
321
|
+
self._system_prompt = system_prompt
|
|
322
|
+
self._port = port
|
|
323
|
+
self._workdir = workdir or f"{PI_RUNNER_DIR}/ep-{port}"
|
|
324
|
+
self._max_env_actions = max_env_actions
|
|
325
|
+
if max_turns < 1:
|
|
326
|
+
raise ValueError("max_turns must be >= 1")
|
|
327
|
+
if max_output_tokens < 1:
|
|
328
|
+
raise ValueError("max_output_tokens must be >= 1")
|
|
329
|
+
self._max_turns = max_turns
|
|
330
|
+
self._max_output_tokens = max_output_tokens
|
|
331
|
+
# The SSH path used to hardcode `timeout 300 node`, so every configured wall budget was
|
|
332
|
+
# silently 300s and 30% of long TerminalBench-2 trials died on it.
|
|
333
|
+
self._episode_timeout_s = validate_episode_timeout_s(episode_timeout_s)
|
|
334
|
+
self._context_window = (
|
|
335
|
+
context_window if context_window is not None else provider_context_window(provider)
|
|
336
|
+
)
|
|
337
|
+
for label, path in (("PI_RUNNER_DIR", PI_RUNNER_DIR), ("workdir", self._workdir)):
|
|
338
|
+
if not _SAFE_REMOTE_PATH.match(path):
|
|
339
|
+
raise ValueError(
|
|
340
|
+
f"unsafe remote {label} {path!r}: only [A-Za-z0-9_./~-] allowed "
|
|
341
|
+
"(it is interpolated into a remote shell command)"
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
def run(self, task_id: str, instruction: str, environment: AgentEnvironment) -> RunResult:
|
|
345
|
+
episode = _Episode(
|
|
346
|
+
instruction=instruction,
|
|
347
|
+
system_prompt=self._system_prompt,
|
|
348
|
+
tools=self._tools,
|
|
349
|
+
provider=self._provider,
|
|
350
|
+
environment=environment,
|
|
351
|
+
temperature=self._temperature,
|
|
352
|
+
skills=self._skills,
|
|
353
|
+
max_env_actions=self._max_env_actions,
|
|
354
|
+
max_turns=self._max_turns,
|
|
355
|
+
max_output_tokens=self._max_output_tokens,
|
|
356
|
+
context_window=self._context_window,
|
|
357
|
+
)
|
|
358
|
+
server = _ShimServer(("127.0.0.1", self._port), _ShimHandler)
|
|
359
|
+
server.episode = episode
|
|
360
|
+
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
|
361
|
+
thread.start()
|
|
362
|
+
try:
|
|
363
|
+
try:
|
|
364
|
+
self._materialize()
|
|
365
|
+
except _MaterializeError as exc:
|
|
366
|
+
# Remote write failed; do not run node against stale files from a prior episode.
|
|
367
|
+
return self._error_result(task_id, episode, instruction, str(exc), StopReason.ERROR)
|
|
368
|
+
code, note = self._run_node()
|
|
369
|
+
finally:
|
|
370
|
+
server.shutdown()
|
|
371
|
+
server.server_close()
|
|
372
|
+
if not episode.done.is_set():
|
|
373
|
+
stop = StopReason.ERROR if code != 0 else StopReason.MAX_TURNS
|
|
374
|
+
return self._error_result(
|
|
375
|
+
task_id, episode, instruction, note or "episode ended without submit", stop
|
|
376
|
+
)
|
|
377
|
+
if episode.proxy_error:
|
|
378
|
+
# The worker LLM proxy failed (auth/outage/HTTP error); entry.ts still POSTs /done, but
|
|
379
|
+
# this is infrastructure failure, not an agent submission, so never count it as
|
|
380
|
+
# SUBMITTED.
|
|
381
|
+
return self._error_result(
|
|
382
|
+
task_id,
|
|
383
|
+
episode,
|
|
384
|
+
instruction,
|
|
385
|
+
f"worker LLM proxy error: {episode.proxy_error}",
|
|
386
|
+
StopReason.PROVIDER_ERROR,
|
|
387
|
+
)
|
|
388
|
+
# entry.ts reports WHY it finished; only an explicit submit is a completion.
|
|
389
|
+
stop_reason = stop_reason_for_done(episode.done_reason)
|
|
390
|
+
return RunResult(
|
|
391
|
+
task_id=task_id,
|
|
392
|
+
steps=episode.steps,
|
|
393
|
+
stop_reason=stop_reason,
|
|
394
|
+
answer=episode.answer,
|
|
395
|
+
turns=len(episode.steps),
|
|
396
|
+
)
|
|
397
|
+
|
|
398
|
+
@staticmethod
|
|
399
|
+
def _error_result(
|
|
400
|
+
task_id: str, episode: _Episode, instruction: str, note: str, stop: StopReason
|
|
401
|
+
) -> RunResult:
|
|
402
|
+
episode.steps.append(
|
|
403
|
+
Step(
|
|
404
|
+
action=Action(kind=ActionKind.MESSAGE, content="(pi runtime)"),
|
|
405
|
+
observation=Observation(content=note, is_error=True),
|
|
406
|
+
state_before=EnvState(),
|
|
407
|
+
task=instruction,
|
|
408
|
+
)
|
|
409
|
+
)
|
|
410
|
+
return RunResult(
|
|
411
|
+
task_id=task_id,
|
|
412
|
+
steps=episode.steps,
|
|
413
|
+
stop_reason=stop,
|
|
414
|
+
answer="",
|
|
415
|
+
turns=len(episode.steps),
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
def _materialize(self) -> None:
|
|
419
|
+
"""Write the harness's code surfaces + entry.ts into the runner checkout via SSH.
|
|
420
|
+
|
|
421
|
+
The files stream as one JSON blob into a python materializer on the runner (one SSH round
|
|
422
|
+
trip, no per-file scp), with node_modules symlinked from the persistent checkout.
|
|
423
|
+
"""
|
|
424
|
+
blob = json.dumps(
|
|
425
|
+
{
|
|
426
|
+
"entry.ts": _read(_ENTRY_TS),
|
|
427
|
+
"runner_termination.ts": _read(_TERMINATION_TS),
|
|
428
|
+
**self._files,
|
|
429
|
+
}
|
|
430
|
+
)
|
|
431
|
+
writer = (
|
|
432
|
+
"import json,sys,os\n"
|
|
433
|
+
"d=json.load(sys.stdin)\n"
|
|
434
|
+
"for p,c in d.items():\n"
|
|
435
|
+
" os.makedirs(os.path.dirname(p) or '.',exist_ok=True)\n"
|
|
436
|
+
" open(p,'w').write(c)\n"
|
|
437
|
+
)
|
|
438
|
+
remote = (
|
|
439
|
+
f"mkdir -p {self._workdir}"
|
|
440
|
+
f" && ln -sfn {PI_RUNNER_DIR}/node_modules {self._workdir}/node_modules"
|
|
441
|
+
f" && cd {self._workdir} && python3 -c {_shq(writer)}"
|
|
442
|
+
)
|
|
443
|
+
result = _ssh(remote, input_bytes=blob.encode("utf-8"))
|
|
444
|
+
if result.returncode != 0:
|
|
445
|
+
detail = (result.stderr or b"").decode("utf-8", "replace").strip()[-300:]
|
|
446
|
+
raise _MaterializeError(f"remote materialize failed (rc={result.returncode}): {detail}")
|
|
447
|
+
|
|
448
|
+
def _run_node(self) -> tuple[int, str]:
|
|
449
|
+
"""Run entry.ts on the runner with a reverse tunnel back to the local shim.
|
|
450
|
+
|
|
451
|
+
The node wall budget is the configured episode timeout, not a fixed 300s: TerminalBench-2
|
|
452
|
+
tasks compile toolchains and boot VMs, and the old constant killed 30% of them mid-turn.
|
|
453
|
+
"""
|
|
454
|
+
url = f"http://127.0.0.1:{self._port}"
|
|
455
|
+
node_timeout_s = int(self._episode_timeout_s)
|
|
456
|
+
remote_cmd = (
|
|
457
|
+
f"cd {self._workdir} && PI_SHIM_URL={url} "
|
|
458
|
+
f"timeout {node_timeout_s} node --experimental-strip-types entry.ts"
|
|
459
|
+
)
|
|
460
|
+
proc = subprocess.run(
|
|
461
|
+
[
|
|
462
|
+
"ssh",
|
|
463
|
+
"-o",
|
|
464
|
+
"ConnectTimeout=10",
|
|
465
|
+
"-o",
|
|
466
|
+
"BatchMode=yes",
|
|
467
|
+
"-R",
|
|
468
|
+
f"{self._port}:127.0.0.1:{self._port}",
|
|
469
|
+
PI_RUNNER_HOST,
|
|
470
|
+
remote_cmd,
|
|
471
|
+
],
|
|
472
|
+
capture_output=True,
|
|
473
|
+
text=True,
|
|
474
|
+
timeout=self._episode_timeout_s + _NODE_TEARDOWN_GRACE_S,
|
|
475
|
+
)
|
|
476
|
+
return proc.returncode, (proc.stderr or "").strip()[-500:]
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def _ssh(remote_cmd: str, input_bytes: bytes | None = None) -> subprocess.CompletedProcess[bytes]:
|
|
480
|
+
return subprocess.run(
|
|
481
|
+
["ssh", "-o", "ConnectTimeout=10", "-o", "BatchMode=yes", PI_RUNNER_HOST, remote_cmd],
|
|
482
|
+
input=input_bytes,
|
|
483
|
+
capture_output=True,
|
|
484
|
+
timeout=120,
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _read(path: str) -> str:
|
|
489
|
+
with open(path, encoding="utf-8") as fh:
|
|
490
|
+
return fh.read()
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def _shq(text: str) -> str:
|
|
494
|
+
"""Single-quote a string for a remote shell (the python -c body)."""
|
|
495
|
+
return "'" + text.replace("'", "'\\''") + "'"
|
wmo/harness/pi_vendor.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""The vendored pi agent, and the seam that turns it into a searchable harness.
|
|
2
|
+
|
|
3
|
+
`wmo/harness/vendor/pi-agent/` is a byte-exact copy of `packages/agent` from
|
|
4
|
+
earendil-works/pi at v0.80.3 (commit a23abe4a695df8b69b613f73e9fdda2a8af894d4). The pin, the
|
|
5
|
+
license attribution, and the integrity ledger live beside it: `vendor/pi-agent/VENDOR.md`,
|
|
6
|
+
`vendor/pi-agent/LICENSE`, and `vendor/manifest.sha256` (regenerate/verify with
|
|
7
|
+
`wmo/harness/vendor/vendor_pi.sh`).
|
|
8
|
+
|
|
9
|
+
This module is the ONLY place wmo reads that tree, and it reads it straight from disk:
|
|
10
|
+
`pi_agent_code_surfaces()` loads pi's own TypeScript source into `code:` surfaces so the
|
|
11
|
+
meta-agent searches over the real agent's source, and `PiRuntime` materializes those surfaces to
|
|
12
|
+
run pi headless. Nothing here fetches pi over the network or from a scratch checkout — the
|
|
13
|
+
committed vendored copy is the sole source of truth. The whole 56-file package is vendored on disk
|
|
14
|
+
(byte-checked against upstream); the 25 runnable `src/**/*.ts` files become the searchable
|
|
15
|
+
surfaces (fixtures, docs, and the package's own vitest specs are vendored but not surfaced).
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
# code_surface_id lives with the Surface grammar in doc.py; imported (and re-exported) here for
|
|
23
|
+
# the existing pi-vendor call sites.
|
|
24
|
+
from wmo.harness.doc import Surface, SurfaceKind, code_surface_id
|
|
25
|
+
|
|
26
|
+
# The committed, byte-exact vendored copy (see VENDOR.md for the upstream pin).
|
|
27
|
+
PI_AGENT_ROOT = Path(__file__).parent / "vendor" / "pi-agent"
|
|
28
|
+
# pi's runnable TypeScript source — the harness the meta-agent searches over.
|
|
29
|
+
_SOURCE_GLOB = "src/**/*.ts"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def pi_agent_source_paths() -> list[Path]:
|
|
33
|
+
"""Every runnable pi source file under the vendored tree, sorted, excluding vitest specs."""
|
|
34
|
+
return sorted(
|
|
35
|
+
p
|
|
36
|
+
for p in PI_AGENT_ROOT.glob(_SOURCE_GLOB)
|
|
37
|
+
if p.is_file() and not p.name.endswith(".test.ts")
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def pi_agent_code_surfaces() -> list[Surface]:
|
|
42
|
+
"""pi's vendored source as `code:` surfaces, each carrying its path under the package root.
|
|
43
|
+
|
|
44
|
+
Read straight from `PI_AGENT_ROOT` (the committed copy) — never from a network fetch or a
|
|
45
|
+
scratch checkout. Raises if the vendored tree is missing so a broken vendoring fails loudly
|
|
46
|
+
instead of silently running an empty harness.
|
|
47
|
+
"""
|
|
48
|
+
paths = pi_agent_source_paths()
|
|
49
|
+
if not paths:
|
|
50
|
+
raise FileNotFoundError(
|
|
51
|
+
f"no pi source under {PI_AGENT_ROOT}; is the vendored copy present? "
|
|
52
|
+
"regenerate with wmo/harness/vendor/vendor_pi.sh"
|
|
53
|
+
)
|
|
54
|
+
surfaces: list[Surface] = []
|
|
55
|
+
for p in paths:
|
|
56
|
+
rel = p.relative_to(PI_AGENT_ROOT).as_posix()
|
|
57
|
+
surfaces.append(
|
|
58
|
+
Surface(
|
|
59
|
+
id=code_surface_id(rel),
|
|
60
|
+
kind=SurfaceKind.CODE,
|
|
61
|
+
path=rel,
|
|
62
|
+
content=p.read_text(encoding="utf-8"),
|
|
63
|
+
)
|
|
64
|
+
)
|
|
65
|
+
return surfaces
|