world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Why a pi episode finished, and what to do about it before giving up.
|
|
3
|
+
*
|
|
4
|
+
* pi's `agent.prompt()` resolves as soon as an assistant turn carries no tool calls. That single
|
|
5
|
+
* event covers four completely different things: the model called `submit`, it emitted prose, its
|
|
6
|
+
* tool call was cut off at the output-token cap, or the renderer could not parse the tool call it
|
|
7
|
+
* did emit. The episode runners used to report all four as one `done` frame, which the host mapped
|
|
8
|
+
* to `submitted`, so every one of them scored as a clean completion with reward 0.
|
|
9
|
+
*
|
|
10
|
+
* This module is the shared classifier + nudge policy the three episode runners
|
|
11
|
+
* (`runner_stdio.ts`, `runner_service.ts`, `entry.ts`) drive:
|
|
12
|
+
*
|
|
13
|
+
* - `TurnSignal` is updated by each runner's LLM bridge from the host's completion frame, so the
|
|
14
|
+
* runner can see the finish_reason, whether tool calls came back, and any tool-call parse
|
|
15
|
+
* errors the host's renderer reported (`wmo_unparsed_tool_calls`).
|
|
16
|
+
* - `classifyEnd` turns that plus the runner's own flags into a `DoneReason`.
|
|
17
|
+
* - `nudgeFor` is the observation fed back to the model instead of ending the episode, modeled
|
|
18
|
+
* on the reference terminus-2 agent's behavior (report the parser's complaint, tell the model
|
|
19
|
+
* to re-issue in smaller chunks, and ask it to either act or submit).
|
|
20
|
+
* - `shouldNudge` bounds that to MAX_NONACTION_TURNS consecutive non-action turns.
|
|
21
|
+
*
|
|
22
|
+
* `DoneReason` values are the wire vocabulary `wmo/harness/runner_link.py` maps onto distinct
|
|
23
|
+
* `StopReason`s, and MAX_NONACTION_TURNS mirrors `wmo.harness.runtime.MAX_NONACTION_TURNS`.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
/** The `done` frame's `reason`: exactly why this episode stopped. */
|
|
27
|
+
export type DoneReason =
|
|
28
|
+
| "submit"
|
|
29
|
+
| "no_tool_call"
|
|
30
|
+
| "output_truncated"
|
|
31
|
+
| "unparsed_tool_call"
|
|
32
|
+
| "provider_error"
|
|
33
|
+
| "max_turns";
|
|
34
|
+
|
|
35
|
+
/** Consecutive non-action turns a runner nudges through before reporting `done`. */
|
|
36
|
+
export const MAX_NONACTION_TURNS = 3;
|
|
37
|
+
|
|
38
|
+
/** What the last host completion frame carried, as the bridge observed it. */
|
|
39
|
+
export interface TurnSignal {
|
|
40
|
+
/** finish_reason of the most recent completion ("length" means the output cap was hit). */
|
|
41
|
+
finishReason: string;
|
|
42
|
+
/** Tool-call parse errors the host's renderer reported for the most recent completion. */
|
|
43
|
+
unparsedToolCallErrors: string[];
|
|
44
|
+
/** The most recent host-side worker error (context overflow, outage), or "". */
|
|
45
|
+
providerError: string;
|
|
46
|
+
/** Cumulative count of completions that carried at least one tool call. */
|
|
47
|
+
toolCallTurns: number;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** A fresh signal, before any completion has come back. */
|
|
51
|
+
export function newTurnSignal(): TurnSignal {
|
|
52
|
+
return { finishReason: "", unparsedToolCallErrors: [], providerError: "", toolCallTurns: 0 };
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Record one host completion frame into `signal`.
|
|
57
|
+
*
|
|
58
|
+
* Called by each runner's LLM bridge with the raw `llm_request` reply, so classification sees the
|
|
59
|
+
* host's own view of the turn rather than re-deriving it from pi's message log.
|
|
60
|
+
*/
|
|
61
|
+
export function observeCompletion(signal: TurnSignal, reply: Record<string, any>): void {
|
|
62
|
+
if (reply.error) {
|
|
63
|
+
signal.providerError = String(reply.error);
|
|
64
|
+
return;
|
|
65
|
+
}
|
|
66
|
+
const choice = reply.completion?.choices?.[0] ?? {};
|
|
67
|
+
const message = choice.message ?? {};
|
|
68
|
+
signal.providerError = "";
|
|
69
|
+
signal.finishReason = String(choice.finish_reason ?? "stop");
|
|
70
|
+
const unparsed = choice.wmo_unparsed_tool_calls;
|
|
71
|
+
signal.unparsedToolCallErrors = Array.isArray(unparsed) ? unparsed.map((e: unknown) => String(e)) : [];
|
|
72
|
+
if (Array.isArray(message.tool_calls) && message.tool_calls.length > 0) {
|
|
73
|
+
signal.toolCallTurns += 1;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Runner-owned flags `classifyEnd` needs beyond the last completion. */
|
|
78
|
+
export interface EndContext {
|
|
79
|
+
/** The runner aborted the agent because the turn cap was reached. */
|
|
80
|
+
hitTurnCap: boolean;
|
|
81
|
+
/** pi's own terminal error message, if any (`agent.state.errorMessage`). */
|
|
82
|
+
agentError: string;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Why `agent.prompt()` returned, for an episode where `submit` was NOT called.
|
|
87
|
+
*
|
|
88
|
+
* Order matters: a dead provider explains everything downstream of it, an explicit parse error is
|
|
89
|
+
* more specific than "no tool call", and truncation at the cap is more specific still.
|
|
90
|
+
*/
|
|
91
|
+
export function classifyEnd(signal: TurnSignal, context: EndContext): DoneReason {
|
|
92
|
+
if (signal.providerError) return "provider_error";
|
|
93
|
+
if (signal.unparsedToolCallErrors.length > 0) return "unparsed_tool_call";
|
|
94
|
+
if (signal.finishReason === "length") return "output_truncated";
|
|
95
|
+
if (context.hitTurnCap) return "max_turns";
|
|
96
|
+
if (context.agentError) return "provider_error";
|
|
97
|
+
return "no_tool_call";
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Whether the runner should nudge instead of reporting `done`.
|
|
102
|
+
*
|
|
103
|
+
* A provider that is failing every call and a turn cap that has already fired are terminal: more
|
|
104
|
+
* prompts would only re-pay for the same failure. Everything else gets up to MAX_NONACTION_TURNS
|
|
105
|
+
* consecutive attempts to act or submit.
|
|
106
|
+
*/
|
|
107
|
+
export function shouldNudge(
|
|
108
|
+
reason: DoneReason,
|
|
109
|
+
consecutiveNonActionTurns: number,
|
|
110
|
+
turns: number,
|
|
111
|
+
maxTurns: number,
|
|
112
|
+
): boolean {
|
|
113
|
+
if (reason === "provider_error" || reason === "max_turns") return false;
|
|
114
|
+
if (consecutiveNonActionTurns >= MAX_NONACTION_TURNS) return false;
|
|
115
|
+
return turns < maxTurns;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const ACT_OR_SUBMIT =
|
|
119
|
+
"Continue the task: either call exactly one tool now, or call `submit` with your final answer " +
|
|
120
|
+
"if the task is already complete. Do not reply with prose alone.";
|
|
121
|
+
|
|
122
|
+
/** The observation fed back to the model in place of ending the episode. */
|
|
123
|
+
export function nudgeFor(reason: DoneReason, signal: TurnSignal, maxOutputTokens: number): string {
|
|
124
|
+
if (reason === "output_truncated") {
|
|
125
|
+
return (
|
|
126
|
+
`[ERROR] NONE of the actions you just requested were performed: your reply exceeded ` +
|
|
127
|
+
`${maxOutputTokens} output tokens and was cut off mid-emission. Re-issue the request, ` +
|
|
128
|
+
`breaking it into chunks each of which is well under ${maxOutputTokens} tokens. ` +
|
|
129
|
+
ACT_OR_SUBMIT
|
|
130
|
+
);
|
|
131
|
+
}
|
|
132
|
+
if (reason === "unparsed_tool_call") {
|
|
133
|
+
const warnings = signal.unparsedToolCallErrors.join("; ");
|
|
134
|
+
return (
|
|
135
|
+
`[ERROR] your tool call could not be parsed, so NOTHING was executed. Parser ` +
|
|
136
|
+
`warnings from your last reply: ${warnings}. Emit the call again in exactly the ` +
|
|
137
|
+
`documented format, closing every block you open. ` +
|
|
138
|
+
ACT_OR_SUBMIT
|
|
139
|
+
);
|
|
140
|
+
}
|
|
141
|
+
return `[ERROR] your last reply contained no tool call, so nothing happened. ${ACT_OR_SUBMIT}`;
|
|
142
|
+
}
|
wmo/harness/pi_local.py
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
"""Run the vendored pi live-session peer as a local Node.js process.
|
|
2
|
+
|
|
3
|
+
The platform remains the credential boundary: the Node peer receives worker
|
|
4
|
+
completions over the existing stdio frame protocol and never receives provider
|
|
5
|
+
keys. Unlike the E2B backend, this module deliberately runs the harness process
|
|
6
|
+
on the user's machine. The CLI presents an explicit consent prompt before it
|
|
7
|
+
reaches this boundary.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import base64
|
|
13
|
+
import contextlib
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import queue
|
|
17
|
+
import re
|
|
18
|
+
import shutil
|
|
19
|
+
import subprocess
|
|
20
|
+
import tempfile
|
|
21
|
+
import threading
|
|
22
|
+
from collections import deque
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import TYPE_CHECKING, Protocol, TextIO, cast
|
|
25
|
+
|
|
26
|
+
from wmo.core.types import JsonObject
|
|
27
|
+
from wmo.harness.pi_e2b import (
|
|
28
|
+
HELLO_TIMEOUT_S,
|
|
29
|
+
PI_NPM_PACKAGES,
|
|
30
|
+
TRANSPORT_KEEPALIVE_TYPE,
|
|
31
|
+
session_entry_files,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
if TYPE_CHECKING:
|
|
35
|
+
from collections.abc import Callable
|
|
36
|
+
|
|
37
|
+
_PI_VERSION = "0.80.3"
|
|
38
|
+
_MIN_NODE = (22, 19, 0)
|
|
39
|
+
_PACKAGE_JSON = '{"name":"wmo-pi-local","private":true,"type":"module"}\n'
|
|
40
|
+
_INSTALL_MARKER = ".wmo-pi-dependencies"
|
|
41
|
+
_STDERR_LINES = 50
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class _CompletedCommand(Protocol):
|
|
45
|
+
"""The subprocess result slice runtime bootstrap consumes."""
|
|
46
|
+
|
|
47
|
+
stdout: str
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class _TextProcess(Protocol):
|
|
51
|
+
"""The text-mode Popen slice used by the local frame channel."""
|
|
52
|
+
|
|
53
|
+
stdin: TextIO | None
|
|
54
|
+
stdout: TextIO | None
|
|
55
|
+
stderr: TextIO | None
|
|
56
|
+
|
|
57
|
+
def poll(self) -> int | None: ...
|
|
58
|
+
|
|
59
|
+
def wait(self, timeout: float | None = None) -> int: ...
|
|
60
|
+
|
|
61
|
+
def terminate(self) -> None: ...
|
|
62
|
+
|
|
63
|
+
def kill(self) -> None: ...
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class _Eof:
|
|
67
|
+
"""Reader-thread sentinel for a closed runner stdout stream."""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
_EOF = _Eof()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def parse_node_version(output: str) -> tuple[int, int, int]:
|
|
74
|
+
"""Parse ``node --version`` output into a semantic-version triple."""
|
|
75
|
+
match = re.fullmatch(r"v(\d+)\.(\d+)\.(\d+)\s*", output)
|
|
76
|
+
if match is None:
|
|
77
|
+
raise RuntimeError(f"could not parse Node.js version output: {output.strip()!r}")
|
|
78
|
+
major, minor, patch = match.groups()
|
|
79
|
+
return int(major), int(minor), int(patch)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def default_local_pi_runtime_dir() -> Path:
|
|
83
|
+
"""Return the user cache directory for the pinned local pi runtime."""
|
|
84
|
+
cache = Path(os.environ.get("XDG_CACHE_HOME", Path.home() / ".cache"))
|
|
85
|
+
return cache / "wmo" / "pi" / _PI_VERSION
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def ensure_local_pi_runtime(
|
|
89
|
+
runtime_dir: Path,
|
|
90
|
+
*,
|
|
91
|
+
node: str,
|
|
92
|
+
npm: str,
|
|
93
|
+
run_command: Callable[..., _CompletedCommand] = subprocess.run,
|
|
94
|
+
) -> Path:
|
|
95
|
+
"""Refresh the live runner and install its pinned npm dependencies once."""
|
|
96
|
+
version = run_command(
|
|
97
|
+
[node, "--version"],
|
|
98
|
+
capture_output=True,
|
|
99
|
+
text=True,
|
|
100
|
+
check=True,
|
|
101
|
+
).stdout
|
|
102
|
+
parsed = parse_node_version(version)
|
|
103
|
+
if parsed < _MIN_NODE:
|
|
104
|
+
required = ".".join(str(part) for part in _MIN_NODE)
|
|
105
|
+
found = ".".join(str(part) for part in parsed)
|
|
106
|
+
raise RuntimeError(f"local pi requires Node.js {required} or newer (found {found})")
|
|
107
|
+
|
|
108
|
+
runtime_dir.mkdir(parents=True, exist_ok=True)
|
|
109
|
+
(runtime_dir / "package.json").write_text(_PACKAGE_JSON, encoding="utf-8")
|
|
110
|
+
for name, content in session_entry_files().items():
|
|
111
|
+
(runtime_dir / name).write_text(content, encoding="utf-8")
|
|
112
|
+
|
|
113
|
+
marker = runtime_dir / _INSTALL_MARKER
|
|
114
|
+
expected = "\n".join(PI_NPM_PACKAGES) + "\n"
|
|
115
|
+
if marker.is_file() and marker.read_text(encoding="utf-8") == expected:
|
|
116
|
+
return runtime_dir
|
|
117
|
+
run_command(
|
|
118
|
+
[npm, "install", "--no-audit", "--no-fund", "--ignore-scripts", *PI_NPM_PACKAGES],
|
|
119
|
+
cwd=runtime_dir,
|
|
120
|
+
check=True,
|
|
121
|
+
capture_output=True,
|
|
122
|
+
text=True,
|
|
123
|
+
timeout=600,
|
|
124
|
+
)
|
|
125
|
+
marker.write_text(expected, encoding="utf-8")
|
|
126
|
+
return runtime_dir
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class LocalStdioChannel:
|
|
130
|
+
"""A live-session frame channel over a local Node child process."""
|
|
131
|
+
|
|
132
|
+
def __init__(
|
|
133
|
+
self,
|
|
134
|
+
process: _TextProcess,
|
|
135
|
+
*,
|
|
136
|
+
stderr_lines: int = _STDERR_LINES,
|
|
137
|
+
cleanup_dir: Path | None = None,
|
|
138
|
+
) -> None:
|
|
139
|
+
"""Start bounded stdout/stderr reader threads for ``process``."""
|
|
140
|
+
if process.stdin is None or process.stdout is None or process.stderr is None:
|
|
141
|
+
raise RuntimeError("local pi process must expose stdin, stdout, and stderr pipes")
|
|
142
|
+
self._process = process
|
|
143
|
+
self._stdin = process.stdin
|
|
144
|
+
self._stdout = process.stdout
|
|
145
|
+
self._stderr_stream = process.stderr
|
|
146
|
+
self._frames: queue.Queue[JsonObject | _Eof] = queue.Queue()
|
|
147
|
+
self._stderr: deque[str] = deque(maxlen=stderr_lines)
|
|
148
|
+
self._cleanup_dir = cleanup_dir
|
|
149
|
+
self._closed = False
|
|
150
|
+
threading.Thread(target=self._read_stdout, name="pi-local-stdout", daemon=True).start()
|
|
151
|
+
threading.Thread(target=self._read_stderr, name="pi-local-stderr", daemon=True).start()
|
|
152
|
+
|
|
153
|
+
def send(self, frame: JsonObject) -> None:
|
|
154
|
+
"""Write one base64(JSON) frame to the child process."""
|
|
155
|
+
line = base64.b64encode(json.dumps(frame).encode()).decode() + "\n"
|
|
156
|
+
self._stdin.write(line)
|
|
157
|
+
self._stdin.flush()
|
|
158
|
+
|
|
159
|
+
def recv(self, timeout: float | None = None) -> JsonObject | None:
|
|
160
|
+
"""Return the next decoded frame, optionally bounded by ``timeout``."""
|
|
161
|
+
try:
|
|
162
|
+
item = self._frames.get(timeout=timeout)
|
|
163
|
+
except queue.Empty:
|
|
164
|
+
message = f"no frame from local pi within {timeout}s{self._stderr_suffix()}"
|
|
165
|
+
raise TimeoutError(message) from None
|
|
166
|
+
if isinstance(item, _Eof):
|
|
167
|
+
self._frames.put(item)
|
|
168
|
+
if self._closed:
|
|
169
|
+
return None
|
|
170
|
+
raise RuntimeError(f"local pi process exited unexpectedly{self._stderr_suffix()}")
|
|
171
|
+
return item
|
|
172
|
+
|
|
173
|
+
def close(self) -> None:
|
|
174
|
+
"""Shut down the child process without leaving a local runner behind."""
|
|
175
|
+
if self._closed:
|
|
176
|
+
return
|
|
177
|
+
self._closed = True
|
|
178
|
+
with contextlib.suppress(Exception):
|
|
179
|
+
self.send({"type": "shutdown"})
|
|
180
|
+
with contextlib.suppress(Exception):
|
|
181
|
+
self._process.wait(timeout=2)
|
|
182
|
+
if self._process.poll() is None:
|
|
183
|
+
with contextlib.suppress(Exception):
|
|
184
|
+
self._process.terminate()
|
|
185
|
+
with contextlib.suppress(Exception):
|
|
186
|
+
self._process.wait(timeout=2)
|
|
187
|
+
if self._process.poll() is None:
|
|
188
|
+
with contextlib.suppress(Exception):
|
|
189
|
+
self._process.kill()
|
|
190
|
+
if self._cleanup_dir is not None:
|
|
191
|
+
with contextlib.suppress(OSError):
|
|
192
|
+
shutil.rmtree(self._cleanup_dir)
|
|
193
|
+
|
|
194
|
+
def _read_stdout(self) -> None:
|
|
195
|
+
try:
|
|
196
|
+
for raw in self._stdout:
|
|
197
|
+
text = raw.strip()
|
|
198
|
+
if not text:
|
|
199
|
+
continue
|
|
200
|
+
try:
|
|
201
|
+
frame = json.loads(base64.b64decode(text, validate=True))
|
|
202
|
+
except ValueError:
|
|
203
|
+
self._stderr.append(f"[stdout] {text}")
|
|
204
|
+
continue
|
|
205
|
+
if isinstance(frame, dict):
|
|
206
|
+
if frame.get("type") == TRANSPORT_KEEPALIVE_TYPE:
|
|
207
|
+
continue
|
|
208
|
+
self._frames.put(cast("JsonObject", frame))
|
|
209
|
+
else:
|
|
210
|
+
self._stderr.append(f"[stdout] {text}")
|
|
211
|
+
finally:
|
|
212
|
+
self._frames.put(_EOF)
|
|
213
|
+
|
|
214
|
+
def _read_stderr(self) -> None:
|
|
215
|
+
for raw in self._stderr_stream:
|
|
216
|
+
text = raw.rstrip()
|
|
217
|
+
if text:
|
|
218
|
+
self._stderr.append(text)
|
|
219
|
+
|
|
220
|
+
def _stderr_suffix(self) -> str:
|
|
221
|
+
tail = "\n".join(self._stderr)
|
|
222
|
+
return f"; recent stderr:\n{tail}" if tail else ""
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def start_local_live_runner(
|
|
226
|
+
*,
|
|
227
|
+
runtime_dir: Path | None = None,
|
|
228
|
+
hello_timeout: float = HELLO_TIMEOUT_S,
|
|
229
|
+
) -> LocalStdioChannel:
|
|
230
|
+
"""Bootstrap and start the local pi peer, returning a hello-verified channel."""
|
|
231
|
+
node = shutil.which("node")
|
|
232
|
+
npm = shutil.which("npm")
|
|
233
|
+
if node is None or npm is None:
|
|
234
|
+
raise RuntimeError("local pi requires Node.js 22.19+ and npm on PATH")
|
|
235
|
+
root = ensure_local_pi_runtime(
|
|
236
|
+
runtime_dir or default_local_pi_runtime_dir(), node=node, npm=npm
|
|
237
|
+
)
|
|
238
|
+
# Each process gets a private cwd for materialized champion code. Node still
|
|
239
|
+
# resolves dependencies from the cached runner's parent directory.
|
|
240
|
+
process_root = Path(tempfile.mkdtemp(prefix="run-", dir=root))
|
|
241
|
+
try:
|
|
242
|
+
process = subprocess.Popen( # noqa: S603 - fixed executable/args, no shell
|
|
243
|
+
[node, "--experimental-strip-types", str(root / "runner_live.ts")],
|
|
244
|
+
cwd=process_root,
|
|
245
|
+
stdin=subprocess.PIPE,
|
|
246
|
+
stdout=subprocess.PIPE,
|
|
247
|
+
stderr=subprocess.PIPE,
|
|
248
|
+
text=True,
|
|
249
|
+
bufsize=1,
|
|
250
|
+
)
|
|
251
|
+
except BaseException:
|
|
252
|
+
shutil.rmtree(process_root, ignore_errors=True)
|
|
253
|
+
raise
|
|
254
|
+
channel = LocalStdioChannel(cast("_TextProcess", process), cleanup_dir=process_root)
|
|
255
|
+
try:
|
|
256
|
+
frame = channel.recv(timeout=hello_timeout)
|
|
257
|
+
if frame is None or frame.get("type") != "hello":
|
|
258
|
+
raise RuntimeError("local pi did not send its hello frame")
|
|
259
|
+
except BaseException:
|
|
260
|
+
channel.close()
|
|
261
|
+
raise
|
|
262
|
+
return channel
|