world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/harness/doc.py
ADDED
|
@@ -0,0 +1,556 @@
|
|
|
1
|
+
"""`HarnessDoc`: a harness as a typed document of identity-keyed surfaces.
|
|
2
|
+
|
|
3
|
+
A harness is not a directory of files — it is a set of named **surfaces**, each an independently
|
|
4
|
+
addressable unit of behavior: prompt sections, the tool policy, scalar loop parameters, and skills.
|
|
5
|
+
Files (`SYSTEM.md`, `config.toml`, `skills/*.md`) are a *render target* the store exports for
|
|
6
|
+
running the harness elsewhere; the document is the interface everything else programs against.
|
|
7
|
+
|
|
8
|
+
Why surfaces instead of files:
|
|
9
|
+
- **Identity.** Every surface has a stable id (`prompt:core`, `skill:count-words`). An update names
|
|
10
|
+
its target; nothing is ever addressed by position or filename, so "which thing changed" is never
|
|
11
|
+
inferred.
|
|
12
|
+
- **Content addressing.** Each surface has a content hash, and the document has a hash over its
|
|
13
|
+
surfaces. "The score of harness X" is well-defined because X is a hash; an update can assert
|
|
14
|
+
exactly what it believes it is editing.
|
|
15
|
+
- **Typed validation.** A document validates as a whole (tools resolve, `submit` present, params in
|
|
16
|
+
range, budgets respected) the moment it is constructed — an invalid harness cannot exist as a
|
|
17
|
+
value, so nothing downstream re-checks.
|
|
18
|
+
|
|
19
|
+
Surface *content* stays a free-form string on purpose: structure lives in the envelope (ids, kinds,
|
|
20
|
+
hashes, budgets), not in the payload, so richer surface kinds can be added without changing how
|
|
21
|
+
updates work.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import hashlib
|
|
27
|
+
import os
|
|
28
|
+
import re
|
|
29
|
+
from collections.abc import Callable
|
|
30
|
+
from enum import StrEnum
|
|
31
|
+
from pathlib import PurePosixPath
|
|
32
|
+
from typing import TYPE_CHECKING
|
|
33
|
+
|
|
34
|
+
from pydantic import BaseModel, Field, field_validator, model_validator
|
|
35
|
+
|
|
36
|
+
from wmo.core.text import validate_durable_text
|
|
37
|
+
from wmo.harness.code_runtime import (
|
|
38
|
+
DEFAULT_RUNTIME_CODE,
|
|
39
|
+
CodeRuntime,
|
|
40
|
+
compile_harness_code,
|
|
41
|
+
)
|
|
42
|
+
from wmo.harness.runtime import (
|
|
43
|
+
DEFAULT_EVAL_EPISODE_TIMEOUT_S,
|
|
44
|
+
DEFAULT_MAX_OUTPUT_TOKENS,
|
|
45
|
+
DEFAULT_MAX_TURNS,
|
|
46
|
+
DEFAULT_SYSTEM_PROMPT,
|
|
47
|
+
AgentRuntime,
|
|
48
|
+
Runtime,
|
|
49
|
+
strip_json_protocol_clause,
|
|
50
|
+
validate_episode_timeout_s,
|
|
51
|
+
)
|
|
52
|
+
from wmo.harness.skills import Skill, SkillLibrary
|
|
53
|
+
from wmo.harness.tools import DEFAULT_TOOLS, READ_SKILL, render_tools, resolve_tools
|
|
54
|
+
from wmo.providers.base import Provider, ToolCallingProvider
|
|
55
|
+
|
|
56
|
+
if TYPE_CHECKING:
|
|
57
|
+
# Import-time neutral: pi_e2b (the optional e2b extra's consumer) is imported lazily inside
|
|
58
|
+
# runtime(); this name exists only for the e2b_pool annotation.
|
|
59
|
+
from wmo.harness.pi_e2b import E2BSandboxPool
|
|
60
|
+
|
|
61
|
+
_SLUG_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
|
|
62
|
+
|
|
63
|
+
# Well-known surface ids. The tool policy and scalar parameters are singletons; prompt and skill
|
|
64
|
+
# surfaces may be added freely (an update can split `prompt:core` into finer sections).
|
|
65
|
+
TOOL_POLICY_ID = "tool_policy:main"
|
|
66
|
+
MAX_TURNS_ID = "param:max-turns"
|
|
67
|
+
MAX_OUTPUT_TOKENS_ID = "param:max-output-tokens"
|
|
68
|
+
TEMPERATURE_ID = "param:temperature"
|
|
69
|
+
RUNTIME_KIND_ID = "param:runtime-kind" # absent/"kit-python" -> in-process; "pi-node" -> PiRuntime
|
|
70
|
+
CODE_RUNTIME_ID = "code:runtime"
|
|
71
|
+
|
|
72
|
+
_SAFE_PATH_RE = re.compile(r"^[A-Za-z0-9._-]+(?:/[A-Za-z0-9._-]+)*$")
|
|
73
|
+
|
|
74
|
+
# Store metadata filenames a pathful code surface may never materialize to: on render they
|
|
75
|
+
# would shadow the harness store's own authority files.
|
|
76
|
+
_STORE_METADATA_FILES = frozenset({"doc.json", "aliases.toml"})
|
|
77
|
+
MAX_SURFACE_PATH_BYTES = 1_024
|
|
78
|
+
|
|
79
|
+
DEFAULT_TEMPERATURE = 0.7
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def code_surface_id(relpath: str) -> str:
|
|
83
|
+
"""A stable surface id from a path (`src/agent-loop.ts` -> `code:src-agent-loop-ts`)."""
|
|
84
|
+
return "code:" + relpath.replace("/", "-").replace(".", "-")
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class SurfaceKind(StrEnum):
|
|
88
|
+
PROMPT = "prompt" # a section of the system prompt (joined in id order)
|
|
89
|
+
SKILL = "skill" # one skill: frontmatter (name, description) + body
|
|
90
|
+
TOOL_POLICY = "tool_policy" # the tool list, one tool name per line
|
|
91
|
+
PARAM = "param" # a scalar loop knob, serialized as its string form
|
|
92
|
+
CODE = "code" # the agent loop itself: a module defining `run(kit)` (see code_runtime)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class Surface(BaseModel):
|
|
96
|
+
"""One named, independently addressable unit of harness behavior."""
|
|
97
|
+
|
|
98
|
+
id: str # "<kind>:<slug>"
|
|
99
|
+
kind: SurfaceKind
|
|
100
|
+
content: str
|
|
101
|
+
# For CODE surfaces of a vendored multi-file harness: the file path the content materializes
|
|
102
|
+
# to (relative, no traversal). A path-less CODE surface is the legacy in-process
|
|
103
|
+
# `code:runtime` module.
|
|
104
|
+
path: str | None = None
|
|
105
|
+
# Optional size budget (characters). Enforced at construction: a surface that exceeds its
|
|
106
|
+
# budget is invalid, so context cost is a schema property rather than a runtime surprise.
|
|
107
|
+
budget: int | None = Field(default=None, ge=1)
|
|
108
|
+
|
|
109
|
+
@model_validator(mode="after")
|
|
110
|
+
def _validate(self) -> Surface:
|
|
111
|
+
validate_durable_text(self.content, field=f"surface {self.id!r} content")
|
|
112
|
+
prefix, sep, slug = self.id.partition(":")
|
|
113
|
+
if not sep or prefix != self.kind.value or not _SLUG_RE.fullmatch(slug):
|
|
114
|
+
raise ValueError(
|
|
115
|
+
f"surface id {self.id!r} must be '{self.kind.value}:<kebab-slug>' matching its kind"
|
|
116
|
+
)
|
|
117
|
+
if self.path is not None:
|
|
118
|
+
if self.kind is not SurfaceKind.CODE:
|
|
119
|
+
raise ValueError(f"surface {self.id!r}: only code surfaces may carry a path")
|
|
120
|
+
candidate = PurePosixPath(self.path)
|
|
121
|
+
if (
|
|
122
|
+
not _SAFE_PATH_RE.fullmatch(self.path)
|
|
123
|
+
or not candidate.parts
|
|
124
|
+
or candidate.as_posix() != self.path
|
|
125
|
+
or ".." in candidate.parts
|
|
126
|
+
or len(self.path.encode("utf-8")) > MAX_SURFACE_PATH_BYTES
|
|
127
|
+
):
|
|
128
|
+
raise ValueError(
|
|
129
|
+
f"surface {self.id!r}: unsafe path {self.path!r}; a code surface path must "
|
|
130
|
+
f"be a canonical relative POSIX path of at most {MAX_SURFACE_PATH_BYTES} "
|
|
131
|
+
"UTF-8 bytes with no '.' or '..' segments"
|
|
132
|
+
)
|
|
133
|
+
if self.path in _STORE_METADATA_FILES:
|
|
134
|
+
raise ValueError(
|
|
135
|
+
f"surface {self.id!r}: path {self.path!r} would shadow a harness store "
|
|
136
|
+
"metadata file; rename the file"
|
|
137
|
+
)
|
|
138
|
+
if self.id == CODE_RUNTIME_ID:
|
|
139
|
+
raise ValueError(
|
|
140
|
+
f"surface {CODE_RUNTIME_ID!r} is the in-process runtime module and must not "
|
|
141
|
+
f"carry a path (got {self.path!r}); rename the file so it maps to its own id"
|
|
142
|
+
)
|
|
143
|
+
expected_id = code_surface_id(self.path)
|
|
144
|
+
if not _SLUG_RE.fullmatch(expected_id.partition(":")[2]):
|
|
145
|
+
raise ValueError(
|
|
146
|
+
f"code surface path {self.path!r} maps to surface id {expected_id!r}, which "
|
|
147
|
+
"is not a valid 'code:<kebab-slug>' id; use lowercase [a-z0-9] runs separated "
|
|
148
|
+
"by single '/', '.', or '-' characters (for example src/agent-loop.ts)"
|
|
149
|
+
)
|
|
150
|
+
if self.id != expected_id:
|
|
151
|
+
raise ValueError(
|
|
152
|
+
f"surface {self.id!r} does not match its path {self.path!r}: a pathful code "
|
|
153
|
+
f"surface must use id {expected_id!r} (code_surface_id of its path)"
|
|
154
|
+
)
|
|
155
|
+
if self.budget is not None and len(self.content) > self.budget:
|
|
156
|
+
raise ValueError(
|
|
157
|
+
f"surface {self.id!r} content is {len(self.content)} chars, "
|
|
158
|
+
f"over its budget of {self.budget}"
|
|
159
|
+
)
|
|
160
|
+
return self
|
|
161
|
+
|
|
162
|
+
@property
|
|
163
|
+
def slug(self) -> str:
|
|
164
|
+
return self.id.partition(":")[2]
|
|
165
|
+
|
|
166
|
+
@property
|
|
167
|
+
def content_hash(self) -> str:
|
|
168
|
+
return _digest(self.content)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
class HarnessDoc(BaseModel):
|
|
172
|
+
"""A complete, validated harness: the value the runtime runs and updates are applied to."""
|
|
173
|
+
|
|
174
|
+
name: str
|
|
175
|
+
version: int = Field(default=0, ge=0) # assigned by the store on save; 0 = unsaved
|
|
176
|
+
surfaces: list[Surface]
|
|
177
|
+
|
|
178
|
+
@field_validator("surfaces")
|
|
179
|
+
@classmethod
|
|
180
|
+
def _canonical_order(cls, v: list[Surface]) -> list[Surface]:
|
|
181
|
+
return sorted(v, key=lambda s: s.id)
|
|
182
|
+
|
|
183
|
+
@model_validator(mode="after")
|
|
184
|
+
def _validate_document(self) -> HarnessDoc:
|
|
185
|
+
ids = [s.id for s in self.surfaces]
|
|
186
|
+
duplicates = sorted({i for i in ids if ids.count(i) > 1})
|
|
187
|
+
if duplicates:
|
|
188
|
+
raise ValueError(f"duplicate surface id(s): {duplicates}")
|
|
189
|
+
if not any(s.kind is SurfaceKind.PROMPT for s in self.surfaces):
|
|
190
|
+
raise ValueError("a harness needs at least one prompt surface")
|
|
191
|
+
# These validations construct the derived values; failures surface here, at the boundary.
|
|
192
|
+
self.tools()
|
|
193
|
+
self.max_turns()
|
|
194
|
+
self.max_output_tokens()
|
|
195
|
+
self.temperature()
|
|
196
|
+
for surface in self.surfaces:
|
|
197
|
+
if surface.kind is SurfaceKind.SKILL:
|
|
198
|
+
skill = Skill.from_markdown(surface.content)
|
|
199
|
+
if skill.name != surface.slug:
|
|
200
|
+
raise ValueError(
|
|
201
|
+
f"skill surface {surface.id!r} declares frontmatter name "
|
|
202
|
+
f"{skill.name!r}; the slug and frontmatter name must match"
|
|
203
|
+
)
|
|
204
|
+
elif surface.kind is SurfaceKind.CODE:
|
|
205
|
+
if surface.path is None:
|
|
206
|
+
# The legacy in-process runtime module: a singleton, compile-checked here.
|
|
207
|
+
if surface.id != CODE_RUNTIME_ID:
|
|
208
|
+
raise ValueError(
|
|
209
|
+
f"path-less code surface must be {CODE_RUNTIME_ID!r} "
|
|
210
|
+
f"(got {surface.id!r}); vendored files carry a `path`"
|
|
211
|
+
)
|
|
212
|
+
compile_harness_code(surface.content)
|
|
213
|
+
paths = [s.path for s in self.surfaces if s.path is not None]
|
|
214
|
+
dup_paths = sorted({p for p in paths if paths.count(p) > 1})
|
|
215
|
+
if dup_paths:
|
|
216
|
+
raise ValueError(f"duplicate code surface path(s): {dup_paths}")
|
|
217
|
+
return self
|
|
218
|
+
|
|
219
|
+
# -- surface access ---------------------------------------------------------------------
|
|
220
|
+
|
|
221
|
+
def surface(self, surface_id: str) -> Surface | None:
|
|
222
|
+
for s in self.surfaces:
|
|
223
|
+
if s.id == surface_id:
|
|
224
|
+
return s
|
|
225
|
+
return None
|
|
226
|
+
|
|
227
|
+
def surface_hashes(self) -> dict[str, str]:
|
|
228
|
+
return {s.id: s.content_hash for s in self.surfaces}
|
|
229
|
+
|
|
230
|
+
@property
|
|
231
|
+
def doc_hash(self) -> str:
|
|
232
|
+
"""Identity of every surface field that can change materialized execution.
|
|
233
|
+
|
|
234
|
+
Display metadata and validation-only budgets do not affect execution. A code surface's
|
|
235
|
+
destination path does, even when its id and content stay unchanged.
|
|
236
|
+
"""
|
|
237
|
+
joined = "\n".join(
|
|
238
|
+
f"{surface.id}\x00{surface.content_hash}"
|
|
239
|
+
+ (f"\x00path={surface.path}" if surface.path is not None else "")
|
|
240
|
+
for surface in self.surfaces
|
|
241
|
+
)
|
|
242
|
+
return _digest(joined)
|
|
243
|
+
|
|
244
|
+
@property
|
|
245
|
+
def legacy_doc_hash(self) -> str:
|
|
246
|
+
"""The pre-path-inclusion document identity (surface id + content hash only).
|
|
247
|
+
|
|
248
|
+
Exists ONLY so `wmo pull` can integrity-check harness versions the platform recorded
|
|
249
|
+
before `doc_hash` covered materialized paths. Never use it anywhere else: not for new
|
|
250
|
+
records, dedupe, or caching.
|
|
251
|
+
"""
|
|
252
|
+
joined = "\n".join(f"{s.id}\x00{s.content_hash}" for s in self.surfaces)
|
|
253
|
+
return _digest(joined)
|
|
254
|
+
|
|
255
|
+
# -- derived, validated views ------------------------------------------------------------
|
|
256
|
+
|
|
257
|
+
def system_prompt(self) -> str:
|
|
258
|
+
"""All prompt surfaces joined in id order (a single `prompt:core` is the common case)."""
|
|
259
|
+
parts = [s.content for s in self.surfaces if s.kind is SurfaceKind.PROMPT]
|
|
260
|
+
return "\n\n".join(parts)
|
|
261
|
+
|
|
262
|
+
def tools(self) -> list[str]:
|
|
263
|
+
policy = self.surface(TOOL_POLICY_ID)
|
|
264
|
+
if policy is None:
|
|
265
|
+
return list(DEFAULT_TOOLS)
|
|
266
|
+
names = [line.strip() for line in policy.content.splitlines() if line.strip()]
|
|
267
|
+
resolve_tools(names) # raises on unknown tools / missing submit
|
|
268
|
+
return names
|
|
269
|
+
|
|
270
|
+
def max_turns(self) -> int:
|
|
271
|
+
raw = self.surface(MAX_TURNS_ID)
|
|
272
|
+
if raw is None:
|
|
273
|
+
return DEFAULT_MAX_TURNS
|
|
274
|
+
try:
|
|
275
|
+
value = int(raw.content.strip())
|
|
276
|
+
except ValueError as exc:
|
|
277
|
+
raise ValueError(f"{MAX_TURNS_ID} must be an integer, got {raw.content!r}") from exc
|
|
278
|
+
if value < 1:
|
|
279
|
+
raise ValueError(f"{MAX_TURNS_ID} must be >= 1, got {value}")
|
|
280
|
+
return value
|
|
281
|
+
|
|
282
|
+
def max_output_tokens(self) -> int:
|
|
283
|
+
"""Return the per-model-call output cap carried by every pi execution mode."""
|
|
284
|
+
raw = self.surface(MAX_OUTPUT_TOKENS_ID)
|
|
285
|
+
if raw is None:
|
|
286
|
+
return DEFAULT_MAX_OUTPUT_TOKENS
|
|
287
|
+
try:
|
|
288
|
+
value = int(raw.content.strip())
|
|
289
|
+
except ValueError as exc:
|
|
290
|
+
raise ValueError(
|
|
291
|
+
f"{MAX_OUTPUT_TOKENS_ID} must be an integer, got {raw.content!r}"
|
|
292
|
+
) from exc
|
|
293
|
+
if value < 1:
|
|
294
|
+
raise ValueError(f"{MAX_OUTPUT_TOKENS_ID} must be >= 1, got {value}")
|
|
295
|
+
return value
|
|
296
|
+
|
|
297
|
+
def temperature(self) -> float:
|
|
298
|
+
raw = self.surface(TEMPERATURE_ID)
|
|
299
|
+
if raw is None:
|
|
300
|
+
return DEFAULT_TEMPERATURE
|
|
301
|
+
try:
|
|
302
|
+
value = float(raw.content.strip())
|
|
303
|
+
except ValueError as exc:
|
|
304
|
+
raise ValueError(f"{TEMPERATURE_ID} must be a float, got {raw.content!r}") from exc
|
|
305
|
+
if not 0.0 <= value <= 2.0:
|
|
306
|
+
raise ValueError(f"{TEMPERATURE_ID} must be in [0, 2], got {value}")
|
|
307
|
+
return value
|
|
308
|
+
|
|
309
|
+
def skills(self) -> list[Skill]:
|
|
310
|
+
return [
|
|
311
|
+
Skill.from_markdown(s.content) for s in self.surfaces if s.kind is SurfaceKind.SKILL
|
|
312
|
+
]
|
|
313
|
+
|
|
314
|
+
def runtime_kind(self) -> str:
|
|
315
|
+
raw = self.surface(RUNTIME_KIND_ID)
|
|
316
|
+
return raw.content.strip() if raw is not None else "kit-python"
|
|
317
|
+
|
|
318
|
+
def code_files(self) -> list[Surface]:
|
|
319
|
+
"""The vendored code surfaces (those carrying a file path), in id order."""
|
|
320
|
+
return [s for s in self.surfaces if s.kind is SurfaceKind.CODE and s.path is not None]
|
|
321
|
+
|
|
322
|
+
def runtime(
|
|
323
|
+
self,
|
|
324
|
+
provider: Provider,
|
|
325
|
+
*,
|
|
326
|
+
backend: str = "local",
|
|
327
|
+
e2b_template: str | None = None,
|
|
328
|
+
e2b_pool: E2BSandboxPool | None = None,
|
|
329
|
+
episode_timeout_s: float | None = None,
|
|
330
|
+
context_window: int | None = None,
|
|
331
|
+
transport_retries: int | None = None,
|
|
332
|
+
should_cancel: Callable[[], bool] | None = None,
|
|
333
|
+
) -> Runtime:
|
|
334
|
+
"""The configured agent runtime this document describes.
|
|
335
|
+
|
|
336
|
+
`backend` chooses WHERE the harness process executes; the ENVIRONMENT its tool calls hit
|
|
337
|
+
is whatever `AgentEnvironment` the eval binds (normally the world-model simulation),
|
|
338
|
+
regardless of backend. `local` runs in/from this process. `e2b` runs the harness process
|
|
339
|
+
in E2B sandboxes the runtime owns — only meaningful for `param:runtime-kind` = "pi-node"
|
|
340
|
+
(the vendored pi agent, whose real context management is the point of running it); any
|
|
341
|
+
other kind raises, because its loop already runs in-process and "e2b" would silently mean
|
|
342
|
+
nothing. `e2b_template` names a prebaked sandbox template whose bootstrap (node 22 + pi's
|
|
343
|
+
npm deps) is already done; default is $WMO_E2B_TEMPLATE. Under `local`, "pi-node" uses
|
|
344
|
+
the SSH shim (or the RunnerLink frame transport when PI_TRANSPORT=link); otherwise a
|
|
345
|
+
`code:runtime` surface drives episodes with the harness's own in-process program; with
|
|
346
|
+
neither, the fixed baseline loop runs. `episode_timeout_s` is the pi-node episode wall
|
|
347
|
+
budget, host-enforced on the e2b and link transports and applied as the remote node timeout
|
|
348
|
+
on the SSH transport; omitting it preserves the 300-second default. `context_window` is the
|
|
349
|
+
served context window the runner calibrates pi's context guard to; omitting it asks the
|
|
350
|
+
provider (`ContextWindowProvider`) and otherwise leaves the runner's documented fallback,
|
|
351
|
+
because a wrong window is worse than none. `transport_retries`
|
|
352
|
+
controls whole-episode replay after an E2B transport death; omitting it preserves that
|
|
353
|
+
runtime's one-retry default, and a side-effectful real environment passes 0. Other
|
|
354
|
+
execution modes reject both instead of silently ignoring them. `should_cancel` is
|
|
355
|
+
honored cooperatively by the e2b and link pi-node runtimes; the local SSH pi-node
|
|
356
|
+
runtime has no cancellation hook, so there it is accepted but best-effort only (an
|
|
357
|
+
in-flight episode runs to its own node/SSH bound). All expose the same
|
|
358
|
+
`run(task_id, instruction, environment) -> RunResult` shape closed-loop eval drives.
|
|
359
|
+
"""
|
|
360
|
+
if backend not in ("local", "e2b"):
|
|
361
|
+
raise ValueError(f"unknown backend {backend!r}; choose local or e2b")
|
|
362
|
+
if transport_retries is not None and (
|
|
363
|
+
isinstance(transport_retries, bool)
|
|
364
|
+
or not isinstance(transport_retries, int)
|
|
365
|
+
or transport_retries < 0
|
|
366
|
+
):
|
|
367
|
+
raise ValueError("transport_retries must be a nonnegative integer")
|
|
368
|
+
runtime_kind = self.runtime_kind()
|
|
369
|
+
if episode_timeout_s is not None:
|
|
370
|
+
episode_timeout_s = validate_episode_timeout_s(episode_timeout_s)
|
|
371
|
+
if runtime_kind != "pi-node":
|
|
372
|
+
raise ValueError("episode_timeout_s applies only to pi-node execution")
|
|
373
|
+
if context_window is not None and runtime_kind != "pi-node":
|
|
374
|
+
raise ValueError("context_window applies only to pi-node execution")
|
|
375
|
+
if transport_retries is not None and (backend != "e2b" or runtime_kind != "pi-node"):
|
|
376
|
+
raise ValueError("transport_retries applies only to e2b pi-node execution")
|
|
377
|
+
if runtime_kind == "pi-node":
|
|
378
|
+
skills = SkillLibrary(self.skills())
|
|
379
|
+
code_files = {s.path: s.content for s in self.code_files() if s.path is not None}
|
|
380
|
+
tool_names = self.tools()
|
|
381
|
+
# Progressive disclosure is runtime plumbing, not a burden on every persisted tool
|
|
382
|
+
# policy. Keep pi-node behavior aligned with AgentRuntime: a skill-bearing document
|
|
383
|
+
# always exposes read_skill, even when the authored policy lists only env tools.
|
|
384
|
+
if len(skills) and READ_SKILL.name not in tool_names:
|
|
385
|
+
tool_names.append(READ_SKILL.name)
|
|
386
|
+
tools = resolve_tools(tool_names)
|
|
387
|
+
structured_provider = provider if isinstance(provider, ToolCallingProvider) else None
|
|
388
|
+
if (
|
|
389
|
+
backend == "e2b" or os.environ.get("PI_TRANSPORT") == "link"
|
|
390
|
+
) and structured_provider is None:
|
|
391
|
+
raise TypeError(
|
|
392
|
+
"pi-node link/e2b execution needs a ToolCallingProvider; "
|
|
393
|
+
"use a structured provider or WaterfallProvider"
|
|
394
|
+
)
|
|
395
|
+
if backend == "e2b":
|
|
396
|
+
# Lazy: the e2b backend is an optional extra; `local` must import none of it.
|
|
397
|
+
from wmo.harness.pi_e2b import E2BPiRuntime
|
|
398
|
+
|
|
399
|
+
assert structured_provider is not None
|
|
400
|
+
return E2BPiRuntime(
|
|
401
|
+
provider=structured_provider,
|
|
402
|
+
files=code_files,
|
|
403
|
+
tools=tools,
|
|
404
|
+
system_prompt=self.assembled_prompt(skills, structured_tools=True),
|
|
405
|
+
temperature=self.temperature(),
|
|
406
|
+
skills=skills,
|
|
407
|
+
template=e2b_template,
|
|
408
|
+
pool=e2b_pool,
|
|
409
|
+
max_turns=self.max_turns(),
|
|
410
|
+
max_output_tokens=self.max_output_tokens(),
|
|
411
|
+
episode_timeout_s=(
|
|
412
|
+
DEFAULT_EVAL_EPISODE_TIMEOUT_S
|
|
413
|
+
if episode_timeout_s is None
|
|
414
|
+
else episode_timeout_s
|
|
415
|
+
),
|
|
416
|
+
context_window=context_window,
|
|
417
|
+
transport_retries=(1 if transport_retries is None else transport_retries),
|
|
418
|
+
should_cancel=should_cancel,
|
|
419
|
+
)
|
|
420
|
+
# PI_TRANSPORT=link routes pi to the RunnerLink frame transport (a persistent runner the
|
|
421
|
+
# host set via runner_link.set_active_channel) instead of the per-episode SSH shim; the
|
|
422
|
+
# default (unset / "ssh") keeps PiRuntime. The worker LLM reads the same PI_AGENT_* env.
|
|
423
|
+
if os.environ.get("PI_TRANSPORT") == "link":
|
|
424
|
+
from wmo.harness.runner_link import (
|
|
425
|
+
RunnerLink,
|
|
426
|
+
active_channel,
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
channel = active_channel()
|
|
430
|
+
if channel is None:
|
|
431
|
+
raise RuntimeError(
|
|
432
|
+
"PI_TRANSPORT=link but no active runner channel; call "
|
|
433
|
+
"runner_link.set_active_channel(channel) before running episodes"
|
|
434
|
+
)
|
|
435
|
+
assert structured_provider is not None
|
|
436
|
+
return RunnerLink(
|
|
437
|
+
channel,
|
|
438
|
+
tools=tools,
|
|
439
|
+
provider=structured_provider,
|
|
440
|
+
system_prompt=self.assembled_prompt(skills, structured_tools=True),
|
|
441
|
+
files=code_files,
|
|
442
|
+
temperature=self.temperature(),
|
|
443
|
+
skills=skills,
|
|
444
|
+
max_turns=self.max_turns(),
|
|
445
|
+
max_output_tokens=self.max_output_tokens(),
|
|
446
|
+
episode_timeout_s=episode_timeout_s,
|
|
447
|
+
context_window=context_window,
|
|
448
|
+
should_cancel=should_cancel,
|
|
449
|
+
)
|
|
450
|
+
from wmo.harness.pi_runtime import PiRuntime # circular: pi_runtime imports doc
|
|
451
|
+
|
|
452
|
+
return PiRuntime(
|
|
453
|
+
provider,
|
|
454
|
+
files=code_files,
|
|
455
|
+
tools=tools,
|
|
456
|
+
temperature=self.temperature(),
|
|
457
|
+
skills=skills,
|
|
458
|
+
system_prompt=self.assembled_prompt(skills, structured_tools=True),
|
|
459
|
+
max_turns=self.max_turns(),
|
|
460
|
+
max_output_tokens=self.max_output_tokens(),
|
|
461
|
+
episode_timeout_s=(
|
|
462
|
+
DEFAULT_EVAL_EPISODE_TIMEOUT_S
|
|
463
|
+
if episode_timeout_s is None
|
|
464
|
+
else episode_timeout_s
|
|
465
|
+
),
|
|
466
|
+
context_window=context_window,
|
|
467
|
+
)
|
|
468
|
+
if backend == "e2b":
|
|
469
|
+
raise ValueError(
|
|
470
|
+
f"backend='e2b' runs the pi-node harness process in a sandbox; this harness's "
|
|
471
|
+
f"runtime kind is {self.runtime_kind()!r}, which already runs in-process — "
|
|
472
|
+
"use backend='local'"
|
|
473
|
+
)
|
|
474
|
+
code = self.surface(CODE_RUNTIME_ID)
|
|
475
|
+
skills = SkillLibrary(self.skills())
|
|
476
|
+
if code is not None:
|
|
477
|
+
return CodeRuntime(
|
|
478
|
+
provider,
|
|
479
|
+
code=code.content,
|
|
480
|
+
tools=resolve_tools(self.tools()),
|
|
481
|
+
temperature=self.temperature(),
|
|
482
|
+
skills=skills,
|
|
483
|
+
system_prompt=self.assembled_prompt(skills),
|
|
484
|
+
)
|
|
485
|
+
return AgentRuntime(
|
|
486
|
+
provider,
|
|
487
|
+
system_prompt=self.system_prompt(),
|
|
488
|
+
tools=self.tools(),
|
|
489
|
+
max_turns=self.max_turns(),
|
|
490
|
+
temperature=self.temperature(),
|
|
491
|
+
skills=skills,
|
|
492
|
+
)
|
|
493
|
+
|
|
494
|
+
def assembled_prompt(
|
|
495
|
+
self, skills: SkillLibrary | None = None, *, structured_tools: bool = False
|
|
496
|
+
) -> str:
|
|
497
|
+
"""Return the system prompt shared by episode and project session runtimes.
|
|
498
|
+
|
|
499
|
+
Args:
|
|
500
|
+
skills: The resolved skill library; defaults to this document's skills.
|
|
501
|
+
structured_tools: True when the runtime passes STRUCTURED tool schemas to the model and
|
|
502
|
+
its own renderer defines the calling convention (every pi runtime). The
|
|
503
|
+
JSON-action clause is then dropped: carrying it declares a protocol that is not in
|
|
504
|
+
use, and in the Nemotron-3 runs 14.7% of trials spent reasoning on a JSON envelope
|
|
505
|
+
the renderer never parses. Only the runtimes that really call `parse_tool_call`
|
|
506
|
+
(`AgentRuntime`, `CodeRuntime`) leave it in.
|
|
507
|
+
|
|
508
|
+
Returns:
|
|
509
|
+
The assembled system prompt.
|
|
510
|
+
"""
|
|
511
|
+
resolved_skills = skills if skills is not None else SkillLibrary(self.skills())
|
|
512
|
+
tool_names = self.tools()
|
|
513
|
+
if len(resolved_skills) and READ_SKILL.name not in tool_names:
|
|
514
|
+
tool_names.append(READ_SKILL.name)
|
|
515
|
+
core = self.system_prompt()
|
|
516
|
+
if structured_tools:
|
|
517
|
+
core = strip_json_protocol_clause(core)
|
|
518
|
+
prompt = f"{core}\n\n## Tools\n{render_tools(resolve_tools(tool_names))}"
|
|
519
|
+
index = resolved_skills.render_index()
|
|
520
|
+
if index:
|
|
521
|
+
prompt += f"\n\n## Your skills (read a body with read_skill)\n{index}"
|
|
522
|
+
return prompt
|
|
523
|
+
|
|
524
|
+
@classmethod
|
|
525
|
+
def baseline(cls, name: str = "baseline") -> HarnessDoc:
|
|
526
|
+
"""The default harness: one core prompt, the default tools, default loop params."""
|
|
527
|
+
return cls(
|
|
528
|
+
name=name,
|
|
529
|
+
surfaces=[
|
|
530
|
+
Surface(id="prompt:core", kind=SurfaceKind.PROMPT, content=DEFAULT_SYSTEM_PROMPT),
|
|
531
|
+
Surface(
|
|
532
|
+
id=TOOL_POLICY_ID,
|
|
533
|
+
kind=SurfaceKind.TOOL_POLICY,
|
|
534
|
+
content="\n".join(DEFAULT_TOOLS),
|
|
535
|
+
),
|
|
536
|
+
Surface(id=MAX_TURNS_ID, kind=SurfaceKind.PARAM, content=str(DEFAULT_MAX_TURNS)),
|
|
537
|
+
Surface(
|
|
538
|
+
id=TEMPERATURE_ID, kind=SurfaceKind.PARAM, content=str(DEFAULT_TEMPERATURE)
|
|
539
|
+
),
|
|
540
|
+
],
|
|
541
|
+
)
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def code_baseline(name: str = "baseline") -> HarnessDoc:
|
|
545
|
+
"""The baseline harness with its loop as an editable `code:runtime` surface.
|
|
546
|
+
|
|
547
|
+
Behaviorally equivalent to `HarnessDoc.baseline()` — same prompt, tools, and one-call-per-turn
|
|
548
|
+
loop, but the loop is data, so `wmo optimize` can propose structural changes to it.
|
|
549
|
+
"""
|
|
550
|
+
base = HarnessDoc.baseline(name)
|
|
551
|
+
code = Surface(id=CODE_RUNTIME_ID, kind=SurfaceKind.CODE, content=DEFAULT_RUNTIME_CODE)
|
|
552
|
+
return HarnessDoc(name=name, surfaces=[*base.surfaces, code])
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _digest(text: str) -> str:
|
|
556
|
+
return hashlib.blake2b(text.encode("utf-8"), digest_size=16).hexdigest()
|