world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Headless pi-agent entrypoint driven by the Python world-model shim.
|
|
3
|
+
*
|
|
4
|
+
* Run: PI_SHIM_URL=http://127.0.0.1:$PORT node --experimental-strip-types entry.ts
|
|
5
|
+
*
|
|
6
|
+
* Flow:
|
|
7
|
+
* 1. GET $PI_SHIM_URL/task -> {instruction, system, tools[]}
|
|
8
|
+
* 2. Build a pi Agent whose Model.baseUrl = $PI_SHIM_URL + "/v1" and
|
|
9
|
+
* api = "openai-completions" (so streamSimple hits the shim's SSE endpoint).
|
|
10
|
+
* 3. Register each task tool as an AgentTool whose execute() POSTs /tool.
|
|
11
|
+
* 4. Register a `submit` tool whose execute() POSTs /done {answer, reason:"submit"} and
|
|
12
|
+
* terminates the loop (AgentToolResult.terminate = true).
|
|
13
|
+
* 5. agent.prompt(instruction); a turn WITHOUT tool calls is not a completion, so read the
|
|
14
|
+
* shim's GET /signal (the host's view of the last completion) and nudge, bounded, before
|
|
15
|
+
* POSTing /done with the classified reason. See runner_termination.ts.
|
|
16
|
+
*
|
|
17
|
+
* Lives next to src/ when materialized on the runner (import paths "./src/agent.ts",
|
|
18
|
+
* "./runner_termination.ts").
|
|
19
|
+
*/
|
|
20
|
+
import { Agent } from "./src/agent.ts";
|
|
21
|
+
import type { AgentTool, AgentToolResult } from "./src/types.ts";
|
|
22
|
+
import type { Model } from "@earendil-works/pi-ai";
|
|
23
|
+
import {
|
|
24
|
+
classifyEnd,
|
|
25
|
+
newTurnSignal,
|
|
26
|
+
nudgeFor,
|
|
27
|
+
shouldNudge,
|
|
28
|
+
type DoneReason,
|
|
29
|
+
} from "./runner_termination.ts";
|
|
30
|
+
|
|
31
|
+
const SHIM = process.env.PI_SHIM_URL;
|
|
32
|
+
if (!SHIM) {
|
|
33
|
+
console.error("PI_SHIM_URL not set");
|
|
34
|
+
process.exit(2);
|
|
35
|
+
}
|
|
36
|
+
const BASE = SHIM.replace(/\/$/, "");
|
|
37
|
+
const DEFAULT_MAX_TURNS = 20;
|
|
38
|
+
// Last-resort model context window when /task carries none. The host resolves the REAL served
|
|
39
|
+
// window (provider/SDK model info) and reports it as context_window; never assume a size here.
|
|
40
|
+
const DEFAULT_CONTEXT_WINDOW = 128000;
|
|
41
|
+
|
|
42
|
+
interface TaskTool {
|
|
43
|
+
name: string;
|
|
44
|
+
description: string;
|
|
45
|
+
parameters: any;
|
|
46
|
+
}
|
|
47
|
+
interface Task {
|
|
48
|
+
instruction: string;
|
|
49
|
+
system?: string;
|
|
50
|
+
tools: TaskTool[];
|
|
51
|
+
max_turns?: number;
|
|
52
|
+
max_output_tokens?: number;
|
|
53
|
+
context_window?: number;
|
|
54
|
+
}
|
|
55
|
+
/** The shim's host-side view of the most recent worker completion. */
|
|
56
|
+
interface ShimSignal {
|
|
57
|
+
finish_reason?: string;
|
|
58
|
+
unparsed_tool_calls?: string[];
|
|
59
|
+
provider_error?: string;
|
|
60
|
+
tool_call_turns?: number;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
async function getJson<T>(path: string): Promise<T> {
|
|
64
|
+
const res = await fetch(BASE + path);
|
|
65
|
+
if (!res.ok) throw new Error(`GET ${path} -> ${res.status}`);
|
|
66
|
+
return (await res.json()) as T;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
async function postJson<T>(path: string, body: unknown): Promise<T> {
|
|
70
|
+
const res = await fetch(BASE + path, {
|
|
71
|
+
method: "POST",
|
|
72
|
+
headers: { "Content-Type": "application/json" },
|
|
73
|
+
body: JSON.stringify(body),
|
|
74
|
+
});
|
|
75
|
+
if (!res.ok) throw new Error(`POST ${path} -> ${res.status}`);
|
|
76
|
+
return (await res.json()) as T;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
let doneSent = false;
|
|
80
|
+
async function sendDone(answer: string | null, reason: DoneReason): Promise<void> {
|
|
81
|
+
if (doneSent) return;
|
|
82
|
+
doneSent = true;
|
|
83
|
+
await postJson("/done", { answer, reason });
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** Pull the host's view of the last completion into the shared TurnSignal shape. */
|
|
87
|
+
async function readSignal(toolCallTurns: number): Promise<ReturnType<typeof newTurnSignal>> {
|
|
88
|
+
const signal = newTurnSignal();
|
|
89
|
+
signal.toolCallTurns = toolCallTurns;
|
|
90
|
+
try {
|
|
91
|
+
const shim = await getJson<ShimSignal>("/signal");
|
|
92
|
+
signal.finishReason = String(shim.finish_reason ?? "");
|
|
93
|
+
signal.unparsedToolCallErrors = Array.isArray(shim.unparsed_tool_calls)
|
|
94
|
+
? shim.unparsed_tool_calls.map((e) => String(e))
|
|
95
|
+
: [];
|
|
96
|
+
signal.providerError = String(shim.provider_error ?? "");
|
|
97
|
+
signal.toolCallTurns = Number.isInteger(shim.tool_call_turns)
|
|
98
|
+
? Number(shim.tool_call_turns)
|
|
99
|
+
: toolCallTurns;
|
|
100
|
+
} catch (e) {
|
|
101
|
+
// An older shim has no /signal endpoint. Classification then degrades to "no_tool_call",
|
|
102
|
+
// which is still honest (it is never reported as a submission).
|
|
103
|
+
console.error(`[entry] /signal unavailable: ${e}`);
|
|
104
|
+
}
|
|
105
|
+
return signal;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// pi often ends by writing its final answer as a normal assistant message rather than calling
|
|
109
|
+
// `submit`. Capture the latest assistant text so we can use it as the answer if the loop exits
|
|
110
|
+
// without a submit call (otherwise the answer would be empty).
|
|
111
|
+
let lastAssistantText = "";
|
|
112
|
+
function assistantText(msg: any): string {
|
|
113
|
+
if (!msg || msg.role !== "assistant" || !Array.isArray(msg.content)) return "";
|
|
114
|
+
return msg.content
|
|
115
|
+
.filter((c: any) => c?.type === "text")
|
|
116
|
+
.map((c: any) => String(c.text ?? ""))
|
|
117
|
+
.join("")
|
|
118
|
+
.trim();
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function makeShimTool(t: TaskTool): AgentTool<any> {
|
|
122
|
+
return {
|
|
123
|
+
name: t.name,
|
|
124
|
+
label: t.name,
|
|
125
|
+
description: t.description,
|
|
126
|
+
parameters: t.parameters,
|
|
127
|
+
execute: async (_id, params): Promise<AgentToolResult<any>> => {
|
|
128
|
+
const r = await postJson<{ content: string; is_error?: boolean }>("/tool", {
|
|
129
|
+
name: t.name,
|
|
130
|
+
arguments: params,
|
|
131
|
+
});
|
|
132
|
+
return {
|
|
133
|
+
content: [{ type: "text", text: String(r.content ?? "") }],
|
|
134
|
+
details: r,
|
|
135
|
+
terminate: false,
|
|
136
|
+
};
|
|
137
|
+
},
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function makeSubmitTool(): AgentTool<any> {
|
|
142
|
+
return {
|
|
143
|
+
name: "submit",
|
|
144
|
+
label: "submit",
|
|
145
|
+
description: "Submit the final answer and finish the task.",
|
|
146
|
+
parameters: {
|
|
147
|
+
type: "object",
|
|
148
|
+
properties: { answer: { type: "string" } },
|
|
149
|
+
required: ["answer"],
|
|
150
|
+
},
|
|
151
|
+
execute: async (_id, params: { answer: string }): Promise<AgentToolResult<any>> => {
|
|
152
|
+
await sendDone(params.answer ?? "", "submit");
|
|
153
|
+
return {
|
|
154
|
+
content: [{ type: "text", text: "submitted" }],
|
|
155
|
+
details: { answer: params.answer },
|
|
156
|
+
terminate: true, // stop the agent loop after this tool batch
|
|
157
|
+
};
|
|
158
|
+
},
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
async function main(): Promise<void> {
|
|
163
|
+
const task = await getJson<Task>("/task");
|
|
164
|
+
const configuredMaxTurns = task.max_turns;
|
|
165
|
+
const maxTurns =
|
|
166
|
+
configuredMaxTurns !== undefined &&
|
|
167
|
+
Number.isInteger(configuredMaxTurns) &&
|
|
168
|
+
configuredMaxTurns >= 1
|
|
169
|
+
? configuredMaxTurns
|
|
170
|
+
: DEFAULT_MAX_TURNS;
|
|
171
|
+
const configuredMaxOutputTokens = task.max_output_tokens;
|
|
172
|
+
const maxOutputTokens =
|
|
173
|
+
configuredMaxOutputTokens !== undefined &&
|
|
174
|
+
Number.isInteger(configuredMaxOutputTokens) &&
|
|
175
|
+
configuredMaxOutputTokens >= 1
|
|
176
|
+
? configuredMaxOutputTokens
|
|
177
|
+
: 4096;
|
|
178
|
+
const configuredContextWindow = task.context_window;
|
|
179
|
+
const contextWindow =
|
|
180
|
+
configuredContextWindow !== undefined &&
|
|
181
|
+
Number.isInteger(configuredContextWindow) &&
|
|
182
|
+
configuredContextWindow >= 1024
|
|
183
|
+
? configuredContextWindow
|
|
184
|
+
: DEFAULT_CONTEXT_WINDOW;
|
|
185
|
+
|
|
186
|
+
const model: Model<"openai-completions"> = {
|
|
187
|
+
id: "stub-model",
|
|
188
|
+
name: "stub-model",
|
|
189
|
+
api: "openai-completions",
|
|
190
|
+
provider: "shim", // non-builtin provider -> uses model.baseUrl directly
|
|
191
|
+
baseUrl: BASE + "/v1",
|
|
192
|
+
reasoning: false,
|
|
193
|
+
input: ["text"],
|
|
194
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
195
|
+
contextWindow,
|
|
196
|
+
maxTokens: maxOutputTokens,
|
|
197
|
+
};
|
|
198
|
+
|
|
199
|
+
// `submit` is provided by entry.ts (it drives /done + loop termination); drop any
|
|
200
|
+
// task-supplied `submit` so the tool list pi sends the model has unique names.
|
|
201
|
+
const envTools = task.tools.filter((t) => t.name !== "submit");
|
|
202
|
+
const tools: AgentTool<any>[] = [...envTools.map(makeShimTool), makeSubmitTool()];
|
|
203
|
+
|
|
204
|
+
const agent = new Agent({
|
|
205
|
+
initialState: {
|
|
206
|
+
systemPrompt: task.system ?? "",
|
|
207
|
+
model,
|
|
208
|
+
tools,
|
|
209
|
+
},
|
|
210
|
+
// apiKey passed via stream options; shim ignores it but SDK requires non-empty.
|
|
211
|
+
getApiKey: () => "x",
|
|
212
|
+
});
|
|
213
|
+
|
|
214
|
+
// Hard turn cap: use the harness document's per-episode value.
|
|
215
|
+
let turnCount = 0;
|
|
216
|
+
let hitTurnCap = false;
|
|
217
|
+
agent.subscribe((event) => {
|
|
218
|
+
if (event.type === "turn_end" || event.type === "message_end") {
|
|
219
|
+
const t = assistantText((event as any).message);
|
|
220
|
+
if (t) lastAssistantText = t;
|
|
221
|
+
}
|
|
222
|
+
if (event.type === "turn_end") {
|
|
223
|
+
turnCount += 1;
|
|
224
|
+
if (turnCount >= maxTurns) {
|
|
225
|
+
hitTurnCap = true;
|
|
226
|
+
agent.abort();
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
});
|
|
230
|
+
|
|
231
|
+
await agent.prompt(task.instruction);
|
|
232
|
+
|
|
233
|
+
// A turn without tool calls is NOT a completion. Nudge (bounded) before ending, then report
|
|
234
|
+
// exactly why this episode stopped so the host never records it as a submission.
|
|
235
|
+
let signal = await readSignal(0);
|
|
236
|
+
let reason: DoneReason = classifyEnd(signal, {
|
|
237
|
+
hitTurnCap,
|
|
238
|
+
agentError: String(agent.state.errorMessage ?? ""),
|
|
239
|
+
});
|
|
240
|
+
let consecutiveNonAction = 1;
|
|
241
|
+
while (!doneSent && shouldNudge(reason, consecutiveNonAction, turnCount, maxTurns)) {
|
|
242
|
+
const before = signal.toolCallTurns;
|
|
243
|
+
await agent.prompt(nudgeFor(reason, signal, maxOutputTokens));
|
|
244
|
+
signal = await readSignal(before);
|
|
245
|
+
consecutiveNonAction = signal.toolCallTurns > before ? 1 : consecutiveNonAction + 1;
|
|
246
|
+
reason = classifyEnd(signal, {
|
|
247
|
+
hitTurnCap,
|
|
248
|
+
agentError: String(agent.state.errorMessage ?? ""),
|
|
249
|
+
});
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
// Ensure /done was sent. If pi never called submit, fall back to its last assistant message
|
|
253
|
+
// text (its de-facto answer) rather than reporting empty.
|
|
254
|
+
if (!doneSent) {
|
|
255
|
+
const err = agent.state.errorMessage;
|
|
256
|
+
await sendDone(err ? null : lastAssistantText, reason);
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
console.error(
|
|
260
|
+
`[entry] done sent=${doneSent} reason=${reason} turns=${turnCount} err=${agent.state.errorMessage ?? ""}`,
|
|
261
|
+
);
|
|
262
|
+
process.exit(0);
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
main().catch((e) => {
|
|
266
|
+
console.error("[entry] fatal", e);
|
|
267
|
+
process.exit(1);
|
|
268
|
+
});
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Length-prefixed JSON frame client for the RunnerLink transport (the Node peer of
|
|
3
|
+
* wmo/harness/runner_link.py). One TCP socket to the host carries every episode; requests get a
|
|
4
|
+
* fresh `req_id` and resolve when the matching response frame arrives, while server-pushed frames
|
|
5
|
+
* (episode_start, cancel, ping) fire registered handlers. Wire format matches runner_link.py
|
|
6
|
+
* exactly: 4-byte big-endian length prefix + UTF-8 JSON body.
|
|
7
|
+
*/
|
|
8
|
+
import net from "node:net";
|
|
9
|
+
|
|
10
|
+
export type Frame = Record<string, any>;
|
|
11
|
+
|
|
12
|
+
export class FrameConn {
|
|
13
|
+
private sock: net.Socket;
|
|
14
|
+
private buf: Buffer = Buffer.alloc(0);
|
|
15
|
+
private waiters = new Map<number, (f: Frame) => void>();
|
|
16
|
+
private handlers = new Map<string, (f: Frame) => void>();
|
|
17
|
+
private reqSeq = 0;
|
|
18
|
+
private closed = false;
|
|
19
|
+
|
|
20
|
+
constructor(sock: net.Socket) {
|
|
21
|
+
this.sock = sock;
|
|
22
|
+
sock.on("data", (d: Buffer) => this._onData(d));
|
|
23
|
+
// If the channel drops while a request is in flight, settle every pending waiter with an
|
|
24
|
+
// error frame — otherwise awaiting llm_request/tool_request promises hang forever and the
|
|
25
|
+
// episode never returns a done/episode_error.
|
|
26
|
+
sock.on("close", () => this._settleAll("runner channel closed"));
|
|
27
|
+
sock.on("error", (e: Error) => this._settleAll(`runner channel error: ${e.message}`));
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
private _settleAll(reason: string): void {
|
|
31
|
+
this.closed = true;
|
|
32
|
+
const pending = [...this.waiters.values()];
|
|
33
|
+
this.waiters.clear();
|
|
34
|
+
for (const resolve of pending) {
|
|
35
|
+
resolve({ error: reason, content: reason, is_error: true });
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Register a handler for a server-pushed frame type (no req_id): episode_start, cancel, ping. */
|
|
40
|
+
on(type: string, handler: (f: Frame) => void): void {
|
|
41
|
+
this.handlers.set(type, handler);
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
send(frame: Frame): void {
|
|
45
|
+
const body = Buffer.from(JSON.stringify(frame), "utf8");
|
|
46
|
+
const hdr = Buffer.alloc(4);
|
|
47
|
+
hdr.writeUInt32BE(body.length, 0);
|
|
48
|
+
this.sock.write(Buffer.concat([hdr, body]));
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Send a request frame with a fresh req_id; resolve with the matching response frame. */
|
|
52
|
+
request(type: string, payload: Frame): Promise<Frame> {
|
|
53
|
+
if (this.closed) {
|
|
54
|
+
const reason = "runner channel closed";
|
|
55
|
+
return Promise.resolve({ error: reason, content: reason, is_error: true });
|
|
56
|
+
}
|
|
57
|
+
const req_id = ++this.reqSeq;
|
|
58
|
+
return new Promise((resolve) => {
|
|
59
|
+
this.waiters.set(req_id, resolve);
|
|
60
|
+
this.send({ type, req_id, ...payload });
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
private _onData(chunk: Buffer): void {
|
|
65
|
+
this.buf = Buffer.concat([this.buf, chunk]);
|
|
66
|
+
while (this.buf.length >= 4) {
|
|
67
|
+
const n = this.buf.readUInt32BE(0);
|
|
68
|
+
if (this.buf.length < 4 + n) break;
|
|
69
|
+
const body = this.buf.subarray(4, 4 + n);
|
|
70
|
+
this.buf = this.buf.subarray(4 + n);
|
|
71
|
+
let frame: Frame;
|
|
72
|
+
try {
|
|
73
|
+
frame = JSON.parse(body.toString("utf8"));
|
|
74
|
+
} catch {
|
|
75
|
+
continue;
|
|
76
|
+
}
|
|
77
|
+
this._dispatch(frame);
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
private _dispatch(frame: Frame): void {
|
|
82
|
+
const rid = frame.req_id;
|
|
83
|
+
if (typeof rid === "number" && this.waiters.has(rid)) {
|
|
84
|
+
const resolve = this.waiters.get(rid);
|
|
85
|
+
this.waiters.delete(rid);
|
|
86
|
+
resolve?.(frame);
|
|
87
|
+
return;
|
|
88
|
+
}
|
|
89
|
+
const handler = this.handlers.get(frame.type);
|
|
90
|
+
if (handler) handler(frame);
|
|
91
|
+
}
|
|
92
|
+
}
|