world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/connect/store.py
ADDED
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""Context bundle persistence and rendering under `<project>/.wmo/context/`.
|
|
2
|
+
|
|
3
|
+
A bundle is one pull's replayable artifact: `manifest.json` (what was pulled, when, from where)
|
|
4
|
+
plus `items.jsonl` (one normalized `ContextItem` per line). "Filesystem as DB", like the model
|
|
5
|
+
store: loading a bundle is just reading its folder. `ContextStore.save`/`load` persist and read
|
|
6
|
+
bundles, and `render_markdown` turns a bundle into a deterministic markdown document callers can
|
|
7
|
+
write into a model's knowledge dir.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import shutil
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from pydantic import BaseModel
|
|
16
|
+
|
|
17
|
+
from wmo.config.config import ARTIFACT_DIR
|
|
18
|
+
from wmo.config.store import validate_name
|
|
19
|
+
from wmo.connect.types import ContextItem, PullQuery
|
|
20
|
+
|
|
21
|
+
_MANIFEST_FILENAME = "manifest.json"
|
|
22
|
+
_ITEMS_FILENAME = "items.jsonl"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class BundleManifest(BaseModel):
|
|
26
|
+
"""Provenance for one saved bundle: what was pulled, when, by which connector.
|
|
27
|
+
|
|
28
|
+
Attributes:
|
|
29
|
+
name: Bundle name (the directory name under `.wmo/context/`).
|
|
30
|
+
connector: The connector that produced the bundle.
|
|
31
|
+
query: The exact `PullQuery` used, kept for replayable re-pulls.
|
|
32
|
+
pulled_at: ISO-8601 timestamp of the pull.
|
|
33
|
+
item_count: Number of items in `items.jsonl`.
|
|
34
|
+
account: Human-readable identity the pull ran as, when known.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
name: str
|
|
38
|
+
connector: str
|
|
39
|
+
query: PullQuery
|
|
40
|
+
pulled_at: str
|
|
41
|
+
item_count: int
|
|
42
|
+
account: str | None = None
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class ContextStore:
|
|
46
|
+
"""Named context bundles on disk under `<root>/.wmo/context/<name>/`.
|
|
47
|
+
|
|
48
|
+
`root` is the PROJECT directory (the parent of `.wmo/`), defaulting to the current working
|
|
49
|
+
directory; tests pass a tmp path. Note this differs from `WorldModelStore`, whose root is
|
|
50
|
+
the `.wmo` artifact dir itself.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
def __init__(self, root: str | Path | None = None) -> None:
|
|
54
|
+
self.root = Path(root) if root is not None else Path.cwd()
|
|
55
|
+
self.context_dir = self.root / ARTIFACT_DIR / "context"
|
|
56
|
+
|
|
57
|
+
def bundle_dir(self, name: str) -> Path:
|
|
58
|
+
"""The directory a bundle named `name` lives in (may not exist)."""
|
|
59
|
+
return self.context_dir / _validated_bundle_name(name)
|
|
60
|
+
|
|
61
|
+
def save(
|
|
62
|
+
self, manifest: BundleManifest, items: list[ContextItem], *, overwrite: bool = False
|
|
63
|
+
) -> Path:
|
|
64
|
+
"""Write one bundle (`manifest.json` + `items.jsonl`); returns its directory.
|
|
65
|
+
|
|
66
|
+
Raises:
|
|
67
|
+
FileExistsError: When the bundle already exists and `overwrite` is False.
|
|
68
|
+
ValueError: When the bundle name is not a safe single path segment.
|
|
69
|
+
"""
|
|
70
|
+
directory = self.bundle_dir(manifest.name)
|
|
71
|
+
if directory.exists():
|
|
72
|
+
if not overwrite:
|
|
73
|
+
raise FileExistsError(
|
|
74
|
+
f"context bundle {manifest.name!r} already exists at {directory}; "
|
|
75
|
+
"pass overwrite=True to replace it"
|
|
76
|
+
)
|
|
77
|
+
shutil.rmtree(directory)
|
|
78
|
+
directory.mkdir(parents=True)
|
|
79
|
+
manifest_text = manifest.model_dump_json(indent=2) + "\n"
|
|
80
|
+
(directory / _MANIFEST_FILENAME).write_text(manifest_text, encoding="utf-8")
|
|
81
|
+
lines = "".join(item.model_dump_json() + "\n" for item in items)
|
|
82
|
+
(directory / _ITEMS_FILENAME).write_text(lines, encoding="utf-8")
|
|
83
|
+
return directory
|
|
84
|
+
|
|
85
|
+
def load(self, name: str) -> tuple[BundleManifest, list[ContextItem]]:
|
|
86
|
+
"""Read one bundle back as (manifest, items).
|
|
87
|
+
|
|
88
|
+
Raises:
|
|
89
|
+
FileNotFoundError: When no bundle named `name` exists.
|
|
90
|
+
"""
|
|
91
|
+
directory = self.bundle_dir(name)
|
|
92
|
+
manifest_path = directory / _MANIFEST_FILENAME
|
|
93
|
+
if not manifest_path.exists():
|
|
94
|
+
raise FileNotFoundError(
|
|
95
|
+
f"no context bundle named {name!r} under {self.context_dir}; "
|
|
96
|
+
"pull a bundle and persist it with ContextStore.save first"
|
|
97
|
+
)
|
|
98
|
+
manifest = BundleManifest.model_validate_json(manifest_path.read_text(encoding="utf-8"))
|
|
99
|
+
items_text = (directory / _ITEMS_FILENAME).read_text(encoding="utf-8")
|
|
100
|
+
items = [
|
|
101
|
+
ContextItem.model_validate_json(line)
|
|
102
|
+
for line in items_text.splitlines()
|
|
103
|
+
if line.strip()
|
|
104
|
+
]
|
|
105
|
+
return manifest, items
|
|
106
|
+
|
|
107
|
+
def list_bundles(self) -> list[BundleManifest]:
|
|
108
|
+
"""Manifests of every saved bundle, sorted by directory name."""
|
|
109
|
+
if not self.context_dir.exists():
|
|
110
|
+
return []
|
|
111
|
+
manifests: list[BundleManifest] = []
|
|
112
|
+
for child in sorted(self.context_dir.iterdir()):
|
|
113
|
+
manifest_path = child / _MANIFEST_FILENAME
|
|
114
|
+
if child.is_dir() and manifest_path.exists():
|
|
115
|
+
manifests.append(
|
|
116
|
+
BundleManifest.model_validate_json(manifest_path.read_text(encoding="utf-8"))
|
|
117
|
+
)
|
|
118
|
+
return manifests
|
|
119
|
+
|
|
120
|
+
def delete(self, name: str) -> bool:
|
|
121
|
+
"""Remove one bundle directory; returns whether it existed."""
|
|
122
|
+
directory = self.bundle_dir(name)
|
|
123
|
+
if not directory.exists():
|
|
124
|
+
return False
|
|
125
|
+
shutil.rmtree(directory)
|
|
126
|
+
return True
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def render_markdown(
|
|
130
|
+
manifest: BundleManifest, items: list[ContextItem], *, max_chars: int | None = None
|
|
131
|
+
) -> str:
|
|
132
|
+
"""Render a bundle as one deterministic markdown document.
|
|
133
|
+
|
|
134
|
+
A provenance header (connector, pulled_at, query) is followed by one `## title` section per
|
|
135
|
+
item (a kind/date/url fact line, then the body). When the result would exceed `max_chars`,
|
|
136
|
+
whole items are dropped from the tail and a final "... n items omitted" line makes the
|
|
137
|
+
truncation visible, never silent.
|
|
138
|
+
"""
|
|
139
|
+
header = _render_header(manifest)
|
|
140
|
+
sections = [_render_item(item) for item in items]
|
|
141
|
+
full = "\n\n".join([header, *sections]) + "\n"
|
|
142
|
+
if max_chars is None or len(full) <= max_chars:
|
|
143
|
+
return full
|
|
144
|
+
candidate = full
|
|
145
|
+
for kept in range(len(items) - 1, -1, -1):
|
|
146
|
+
omitted = len(items) - kept
|
|
147
|
+
tail = f"... {omitted} items omitted"
|
|
148
|
+
candidate = "\n\n".join([header, *sections[:kept], tail]) + "\n"
|
|
149
|
+
if len(candidate) <= max_chars:
|
|
150
|
+
return candidate
|
|
151
|
+
# Even the header alone is over budget; return the loud minimal form anyway.
|
|
152
|
+
return candidate
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _validated_bundle_name(name: str) -> str:
|
|
156
|
+
"""Reject unsafe bundle names with bundle-specific wording (same rules as model names)."""
|
|
157
|
+
try:
|
|
158
|
+
return validate_name(name)
|
|
159
|
+
except ValueError as exc:
|
|
160
|
+
raise ValueError(
|
|
161
|
+
f"invalid context bundle name {name!r}: use letters, digits, '.', '_', '-' "
|
|
162
|
+
"(must start with a letter or digit, no path separators)"
|
|
163
|
+
) from exc
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _render_header(manifest: BundleManifest) -> str:
|
|
167
|
+
"""The provenance block: bundle name, connector, pull time, identity, query, item count."""
|
|
168
|
+
query_parts = [
|
|
169
|
+
f"{field}={value}"
|
|
170
|
+
for field, value in manifest.query.model_dump(mode="json").items()
|
|
171
|
+
if value is not None
|
|
172
|
+
]
|
|
173
|
+
lines = [
|
|
174
|
+
f"# Context bundle: {manifest.name}",
|
|
175
|
+
"",
|
|
176
|
+
f"- connector: {manifest.connector}",
|
|
177
|
+
f"- pulled_at: {manifest.pulled_at}",
|
|
178
|
+
]
|
|
179
|
+
if manifest.account:
|
|
180
|
+
lines.append(f"- account: {manifest.account}")
|
|
181
|
+
lines.append(f"- query: {', '.join(query_parts)}")
|
|
182
|
+
lines.append(f"- items: {manifest.item_count}")
|
|
183
|
+
return "\n".join(lines)
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _render_item(item: ContextItem) -> str:
|
|
187
|
+
"""One `## title` section: a kind/date/url fact line, then the body."""
|
|
188
|
+
facts = [item.kind.value]
|
|
189
|
+
if item.created_at:
|
|
190
|
+
facts.append(f"created {item.created_at}")
|
|
191
|
+
if item.updated_at:
|
|
192
|
+
facts.append(f"updated {item.updated_at}")
|
|
193
|
+
if item.url:
|
|
194
|
+
facts.append(item.url)
|
|
195
|
+
section = f"## {item.title}\n\n{' | '.join(facts)}"
|
|
196
|
+
body = item.body.strip()
|
|
197
|
+
if body:
|
|
198
|
+
section += f"\n\n{body}"
|
|
199
|
+
return section
|
wmo/connect/types.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Normalized types for context connectors.
|
|
2
|
+
|
|
3
|
+
Connectors (github, google, slack, notion, ...) authenticate against a service and pull content
|
|
4
|
+
into these vendor-agnostic shapes. Everything downstream (the bundle store, markdown rendering,
|
|
5
|
+
knowledge attachment) operates on `ContextItem`, never on raw vendor payloads. The `opt_str`,
|
|
6
|
+
`capped`, and `strip_html` helpers are the shared coercions for raw vendor JSON and fetched
|
|
7
|
+
content bodies.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import html
|
|
13
|
+
import re
|
|
14
|
+
from collections.abc import Iterator
|
|
15
|
+
from contextlib import contextmanager
|
|
16
|
+
from enum import StrEnum
|
|
17
|
+
from typing import Literal
|
|
18
|
+
|
|
19
|
+
import httpx
|
|
20
|
+
from pydantic import BaseModel, Field
|
|
21
|
+
|
|
22
|
+
from wmo.core.types import JsonObject, JsonValue
|
|
23
|
+
|
|
24
|
+
# Fetched content is capped so one huge document or page cannot blow up a bundle.
|
|
25
|
+
CONTENT_CAP_CHARS = 200_000
|
|
26
|
+
TRUNCATION_MARKER = f"\n[content truncated at {CONTENT_CAP_CHARS} characters]"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def opt_str(value: JsonValue | None) -> str | None:
|
|
30
|
+
"""`value` when it is a non-empty string, else None (vendor JSON field coercion)."""
|
|
31
|
+
return value if isinstance(value, str) and value else None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def capped(text: str) -> str:
|
|
35
|
+
"""Cap fetched content, appending a loud truncation marker when anything was cut."""
|
|
36
|
+
if len(text) <= CONTENT_CAP_CHARS:
|
|
37
|
+
return text
|
|
38
|
+
return text[:CONTENT_CAP_CHARS] + TRUNCATION_MARKER
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def strip_html(markup: str) -> str:
|
|
42
|
+
"""A small HTML-to-text fallback: drop script/style, break on block ends, strip tags."""
|
|
43
|
+
text = re.sub(r"(?is)<(script|style)\b.*?</\1>", " ", markup)
|
|
44
|
+
text = re.sub(r"(?is)<br\s*/?>|</p>|</div>", "\n", text)
|
|
45
|
+
text = re.sub(r"(?s)<[^>]*>", " ", text)
|
|
46
|
+
text = html.unescape(text)
|
|
47
|
+
lines = [re.sub(r"[ \t]+", " ", line).strip() for line in text.splitlines()]
|
|
48
|
+
return "\n".join(line for line in lines if line)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class ConnectError(RuntimeError):
|
|
52
|
+
"""A connector operation failed; messages say what went wrong and what to do about it."""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@contextmanager
|
|
56
|
+
def transport_errors(host: str) -> Iterator[None]:
|
|
57
|
+
"""Turn httpx transport failures inside the block into actionable ConnectErrors.
|
|
58
|
+
|
|
59
|
+
Connectors wrap their HTTP calls with this so network-level failures (DNS errors, refused
|
|
60
|
+
connections, timeouts) honor the ConnectError contract instead of escaping to callers as
|
|
61
|
+
raw httpx tracebacks.
|
|
62
|
+
|
|
63
|
+
Args:
|
|
64
|
+
host: The host the block talks to, named in the error message.
|
|
65
|
+
|
|
66
|
+
Raises:
|
|
67
|
+
ConnectError: For any `httpx.HTTPError` raised inside the block.
|
|
68
|
+
"""
|
|
69
|
+
try:
|
|
70
|
+
yield
|
|
71
|
+
except httpx.HTTPError as exc:
|
|
72
|
+
detail = str(exc) or type(exc).__name__
|
|
73
|
+
raise ConnectError(
|
|
74
|
+
f"could not reach {host} ({detail}); check your network connection and retry"
|
|
75
|
+
) from exc
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class ItemKind(StrEnum):
|
|
79
|
+
"""The normalized kind of one pulled content item."""
|
|
80
|
+
|
|
81
|
+
DOCUMENT = "document"
|
|
82
|
+
PAGE = "page"
|
|
83
|
+
ISSUE = "issue"
|
|
84
|
+
PULL_REQUEST = "pull_request"
|
|
85
|
+
MESSAGE = "message"
|
|
86
|
+
THREAD = "thread"
|
|
87
|
+
EMAIL = "email"
|
|
88
|
+
EVENT = "event"
|
|
89
|
+
FILE = "file"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class ContextItem(BaseModel):
|
|
93
|
+
"""One normalized piece of pulled content (an issue, a page, a message, ...).
|
|
94
|
+
|
|
95
|
+
Attributes:
|
|
96
|
+
id: Stable identifier within the source service (issue number, page id, message ts).
|
|
97
|
+
source: The connector name that produced the item (e.g. "github").
|
|
98
|
+
kind: What the item is, from the normalized `ItemKind` vocabulary.
|
|
99
|
+
title: Short human title; shown as the item's markdown section heading.
|
|
100
|
+
body: The content itself, plain text or markdown.
|
|
101
|
+
url: Canonical link back to the item, when the service has one.
|
|
102
|
+
created_at: ISO-8601 creation timestamp, when known.
|
|
103
|
+
updated_at: ISO-8601 last-modified timestamp, when known.
|
|
104
|
+
metadata: Connector-specific extras (labels, authors, channel ids) as arbitrary JSON.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
id: str
|
|
108
|
+
source: str
|
|
109
|
+
kind: ItemKind
|
|
110
|
+
title: str
|
|
111
|
+
body: str
|
|
112
|
+
url: str | None = None
|
|
113
|
+
created_at: str | None = None
|
|
114
|
+
updated_at: str | None = None
|
|
115
|
+
metadata: JsonObject = Field(default_factory=dict)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class PullQuery(BaseModel):
|
|
119
|
+
"""What to pull: the parameters every connector's `pull` accepts.
|
|
120
|
+
|
|
121
|
+
Attributes:
|
|
122
|
+
target: Service-specific container: a repo "owner/name", a channel name, a calendar id,
|
|
123
|
+
a drive folder.
|
|
124
|
+
query: Free-text or service search-syntax filter.
|
|
125
|
+
since: ISO-8601 date or datetime lower bound on item time.
|
|
126
|
+
until: ISO-8601 date or datetime upper bound on item time.
|
|
127
|
+
limit: Maximum number of items a connector may fetch (connectors must cap at this).
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
target: str | None = None
|
|
131
|
+
query: str | None = None
|
|
132
|
+
since: str | None = None
|
|
133
|
+
until: str | None = None
|
|
134
|
+
limit: int = 100
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class ConnectorAuth(BaseModel):
|
|
138
|
+
"""A stored credential for one connector.
|
|
139
|
+
|
|
140
|
+
Attributes:
|
|
141
|
+
kind: "oauth" for browser/device OAuth grants, "token" for pasted or env-injected tokens.
|
|
142
|
+
access_token: The bearer credential API calls send.
|
|
143
|
+
refresh_token: OAuth refresh token, when the provider issued one.
|
|
144
|
+
expires_at: ISO-8601 absolute expiry of `access_token`, when known.
|
|
145
|
+
scopes: The granted OAuth scopes.
|
|
146
|
+
account: Human-readable identity captured at connect time (e.g. "octocat").
|
|
147
|
+
extra: Connector-specific extras (e.g. a slack team id) as arbitrary JSON.
|
|
148
|
+
"""
|
|
149
|
+
|
|
150
|
+
kind: Literal["oauth", "token"]
|
|
151
|
+
access_token: str
|
|
152
|
+
refresh_token: str | None = None
|
|
153
|
+
expires_at: str | None = None
|
|
154
|
+
scopes: list[str] = Field(default_factory=list)
|
|
155
|
+
account: str | None = None
|
|
156
|
+
extra: JsonObject = Field(default_factory=dict)
|
wmo/core/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Core data types shared across the harness."""
|
|
2
|
+
|
|
3
|
+
from wmo.core.types import (
|
|
4
|
+
Action,
|
|
5
|
+
ActionKind,
|
|
6
|
+
EnvState,
|
|
7
|
+
Observation,
|
|
8
|
+
Session,
|
|
9
|
+
Step,
|
|
10
|
+
Trace,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"Action",
|
|
15
|
+
"ActionKind",
|
|
16
|
+
"EnvState",
|
|
17
|
+
"Observation",
|
|
18
|
+
"Session",
|
|
19
|
+
"Step",
|
|
20
|
+
"Trace",
|
|
21
|
+
]
|
wmo/core/parsing.py
ADDED
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
"""Robust parsing of model completions into structured values.
|
|
2
|
+
|
|
3
|
+
Two concerns live here because both the serving engine and the optimizer need them, and `wmo.core`
|
|
4
|
+
has no dependencies (so neither imports the other):
|
|
5
|
+
|
|
6
|
+
- `extract_json_object`: pull the first complete JSON object out of a noisy LLM reply.
|
|
7
|
+
- `parse_observation`: turn a world-model completion into a structured `Observation`.
|
|
8
|
+
|
|
9
|
+
The world-model output contract (see `wmo.core.render.build_env_prompt`) asks the model to reply
|
|
10
|
+
with a JSON object ``{"output": str, "is_error": bool, "state_note": str}``. `parse_observation`
|
|
11
|
+
is lenient: a reply that is not JSON is treated as a plain-text observation, so a model that ignores
|
|
12
|
+
the contract still produces a usable (non-error) observation rather than crashing the step.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import re
|
|
19
|
+
|
|
20
|
+
from pydantic import BaseModel, ValidationError, field_validator
|
|
21
|
+
|
|
22
|
+
from wmo.core.types import JsonObject, JsonValue, Observation
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def accepted_confidence(value: float | int | str | bool) -> float | None:
|
|
26
|
+
"""The one definition of a usable stated confidence: a finite number in [0, 1], else None.
|
|
27
|
+
|
|
28
|
+
Gates and calibration both consume this, so the acceptance rule must not fork: booleans are
|
|
29
|
+
not confidences (JSON `true` is not 1.0), NaN/inf are garbage, and OUT-OF-RANGE numerics
|
|
30
|
+
degrade to "not stated" rather than clamping — a model answering 85 (percent) or 7 (out of
|
|
31
|
+
10) has violated the 0.0-1.0 contract, and clamping such a reply to 1.0 would record maximal
|
|
32
|
+
certainty on exactly the steps where the model is off the rails. Missing conservatively
|
|
33
|
+
gates as LOW; malformed must never gate as certain.
|
|
34
|
+
"""
|
|
35
|
+
if isinstance(value, bool):
|
|
36
|
+
return None
|
|
37
|
+
try:
|
|
38
|
+
parsed = float(value)
|
|
39
|
+
except ValueError:
|
|
40
|
+
return None
|
|
41
|
+
if not (0.0 <= parsed <= 1.0): # also rejects NaN (all comparisons false) and +/-inf
|
|
42
|
+
return None
|
|
43
|
+
return parsed
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def extract_json_object(text: str) -> str | None:
|
|
47
|
+
"""Return the first complete JSON object substring in `text`, or None if there is none.
|
|
48
|
+
|
|
49
|
+
Scans from the first ``{`` to its balanced closing ``}``, tracking string literals and escapes.
|
|
50
|
+
This tolerates ```json fences, surrounding prose, nested objects, and multiple objects (the
|
|
51
|
+
first is returned) — cases a greedy/lazy regex gets wrong.
|
|
52
|
+
"""
|
|
53
|
+
start = text.find("{")
|
|
54
|
+
if start == -1:
|
|
55
|
+
return None
|
|
56
|
+
depth = 0
|
|
57
|
+
in_string = False
|
|
58
|
+
escaped = False
|
|
59
|
+
for i in range(start, len(text)):
|
|
60
|
+
ch = text[i]
|
|
61
|
+
if in_string:
|
|
62
|
+
if escaped:
|
|
63
|
+
escaped = False
|
|
64
|
+
elif ch == "\\":
|
|
65
|
+
escaped = True
|
|
66
|
+
elif ch == '"':
|
|
67
|
+
in_string = False
|
|
68
|
+
continue
|
|
69
|
+
if ch == '"':
|
|
70
|
+
in_string = True
|
|
71
|
+
elif ch == "{":
|
|
72
|
+
depth += 1
|
|
73
|
+
elif ch == "}":
|
|
74
|
+
depth -= 1
|
|
75
|
+
if depth == 0:
|
|
76
|
+
return text[start : i + 1]
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class _RawObservation(BaseModel):
|
|
81
|
+
"""Lenient view of the world-model JSON contract before normalization.
|
|
82
|
+
|
|
83
|
+
The reasoning-mode fields (`reasoning`, `kb_note`, `ground_query` — see
|
|
84
|
+
`wmo.core.render.output_contract`) default to empty so base-contract replies parse unchanged.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
reasoning: str = ""
|
|
88
|
+
output: str = ""
|
|
89
|
+
is_error: bool = False
|
|
90
|
+
state_note: str = ""
|
|
91
|
+
kb_note: str = ""
|
|
92
|
+
ground_query: str = ""
|
|
93
|
+
state_update: str = ""
|
|
94
|
+
# Verbalized confidence (WS-A6): None when the model didn't state one. Lenient like the rest
|
|
95
|
+
# of the contract — an off-contract value degrades to "no stated confidence", never a crash.
|
|
96
|
+
confidence: float | None = None
|
|
97
|
+
confidence_why: str = ""
|
|
98
|
+
|
|
99
|
+
@field_validator("confidence", mode="before")
|
|
100
|
+
@classmethod
|
|
101
|
+
def _lenient_confidence(cls, value: JsonValue) -> float | None:
|
|
102
|
+
"""Coerce the raw JSON field through `accepted_confidence` (off-contract -> None)."""
|
|
103
|
+
if isinstance(value, int | float | str):
|
|
104
|
+
return accepted_confidence(value)
|
|
105
|
+
return None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
# The keys that mark a reply as following the observation contract (any one present is enough).
|
|
109
|
+
# Used to tell a real — possibly empty — contract response apart from arbitrary JSON that happens
|
|
110
|
+
# to validate against `_RawObservation`'s all-defaulted fields. Deliberately ONLY the core keys:
|
|
111
|
+
# every complete contract reply (base or reasoning mode) carries `output`/`is_error`, while a
|
|
112
|
+
# reasoning-mode superset key alone (e.g. off-contract JSON with a "reasoning" field but no
|
|
113
|
+
# "output") must fall through to the plain-text fallback, not become an empty observation.
|
|
114
|
+
# Confidence-mode keys are deliberately excluded too: an arbitrary API payload with its own
|
|
115
|
+
# "confidence" field must not be mistaken for a contract reply.
|
|
116
|
+
_CONTRACT_KEYS = frozenset({"output", "is_error", "state_note"})
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def parse_observation(text: str) -> Observation:
|
|
120
|
+
"""Parse a world-model completion into a structured Observation.
|
|
121
|
+
|
|
122
|
+
Prefers the JSON contract ``{"output", "is_error", "state_note"}`` and its reasoning-mode
|
|
123
|
+
superset (``reasoning``/``kb_note``/``ground_query``). ``output`` becomes the observation the
|
|
124
|
+
agent sees; every other populated field is carried in ``metadata`` (``state_note`` feeds the
|
|
125
|
+
session scratchpad, ``kb_note`` the cross-session knowledge base, ``ground_query`` the
|
|
126
|
+
grounder, ``reasoning`` is kept for inspection only). Falls back to treating the whole reply
|
|
127
|
+
as plain observation text when it is not the expected JSON, so an off-contract model still
|
|
128
|
+
yields a usable observation.
|
|
129
|
+
"""
|
|
130
|
+
raw = extract_json_object(text)
|
|
131
|
+
if raw is not None:
|
|
132
|
+
try:
|
|
133
|
+
obj: object = json.loads(raw)
|
|
134
|
+
except json.JSONDecodeError:
|
|
135
|
+
obj = None
|
|
136
|
+
# Recognize the contract by the PRESENCE of its keys, not by truthy values.
|
|
137
|
+
# `_RawObservation` defaults every field, so arbitrary JSON like `{"foo": 1}` would validate
|
|
138
|
+
# to an all-empty observation; requiring a contract key keeps that falling through to raw
|
|
139
|
+
# text. But a legitimate silent success `{"output": "", "is_error": false, ...}` (many shell
|
|
140
|
+
# writes/redirects print nothing) MUST be honored as an empty observation, not re-serialized
|
|
141
|
+
# as visible JSON text — closed-loop rollouts would otherwise show spurious output.
|
|
142
|
+
if isinstance(obj, dict) and _CONTRACT_KEYS.intersection(obj):
|
|
143
|
+
try:
|
|
144
|
+
parsed = _RawObservation.model_validate(obj)
|
|
145
|
+
except ValidationError:
|
|
146
|
+
parsed = None
|
|
147
|
+
if parsed is not None:
|
|
148
|
+
metadata: JsonObject = {}
|
|
149
|
+
for key, value in (
|
|
150
|
+
("state_note", parsed.state_note),
|
|
151
|
+
("reasoning", parsed.reasoning),
|
|
152
|
+
("kb_note", parsed.kb_note),
|
|
153
|
+
("ground_query", parsed.ground_query),
|
|
154
|
+
("state_update", parsed.state_update),
|
|
155
|
+
("confidence_why", parsed.confidence_why),
|
|
156
|
+
):
|
|
157
|
+
if value:
|
|
158
|
+
metadata[key] = value
|
|
159
|
+
# Separate from the truthiness loop: a stated confidence of 0.0 must survive.
|
|
160
|
+
if parsed.confidence is not None:
|
|
161
|
+
metadata["confidence"] = parsed.confidence
|
|
162
|
+
return Observation(
|
|
163
|
+
content=parsed.output, is_error=parsed.is_error, metadata=metadata
|
|
164
|
+
)
|
|
165
|
+
# A contract reply cut off mid-generation is not valid JSON at all: salvage the fields the
|
|
166
|
+
# text already contains rather than surfacing the raw truncated JSON as the observation.
|
|
167
|
+
salvaged = _salvage_truncated_contract(text)
|
|
168
|
+
if salvaged is not None:
|
|
169
|
+
return salvaged
|
|
170
|
+
return Observation(content=text.strip())
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _salvage_truncated_contract(text: str) -> Observation | None:
|
|
174
|
+
"""Recover a contract reply whose JSON never closed (token-budget truncation).
|
|
175
|
+
|
|
176
|
+
Long deliberations plus long escaped observations can blow the completion budget mid-string;
|
|
177
|
+
without this, the ENTIRE raw contract text (reasoning included) becomes the observation the
|
|
178
|
+
agent sees — observed live as a catastrophic 0.26-fidelity step. Conservative trigger: the
|
|
179
|
+
text must look like a contract object (starts with ``{`` and names an ``"output"`` key) and
|
|
180
|
+
must NOT have parsed as complete JSON (callers try that first). Recovered string fields are
|
|
181
|
+
unescaped up to the truncation point.
|
|
182
|
+
"""
|
|
183
|
+
stripped = text.strip()
|
|
184
|
+
if not stripped.startswith("{") or '"output"' not in stripped:
|
|
185
|
+
return None
|
|
186
|
+
output = _string_field_value(stripped, "output")
|
|
187
|
+
if output is None:
|
|
188
|
+
return None
|
|
189
|
+
metadata: JsonObject = {}
|
|
190
|
+
# Recover every metadata-carried contract field the truncated text still contains —
|
|
191
|
+
# dropping state_note/state_update here would silently stall the scratchpad and belief
|
|
192
|
+
# profile for the rest of the session.
|
|
193
|
+
salvage_keys = (
|
|
194
|
+
"reasoning",
|
|
195
|
+
"state_note",
|
|
196
|
+
"kb_note",
|
|
197
|
+
"ground_query",
|
|
198
|
+
"state_update",
|
|
199
|
+
"confidence_why",
|
|
200
|
+
)
|
|
201
|
+
for key in salvage_keys:
|
|
202
|
+
value = _string_field_value(stripped, key)
|
|
203
|
+
if value:
|
|
204
|
+
metadata[key] = value
|
|
205
|
+
# Salvage a stated confidence too: truncation correlates with HARD steps, so silently
|
|
206
|
+
# dropping their confidences would bias any calibration analysis toward the easy ones.
|
|
207
|
+
# Same acceptance rule as the validator — the two paths must not fork.
|
|
208
|
+
match = _CONFIDENCE_VALUE.search(stripped)
|
|
209
|
+
if match is not None:
|
|
210
|
+
confidence = accepted_confidence(match.group(1))
|
|
211
|
+
if confidence is not None:
|
|
212
|
+
metadata["confidence"] = confidence
|
|
213
|
+
is_error = re.search(r'"is_error"\s*:\s*true', stripped) is not None
|
|
214
|
+
return Observation(content=output, is_error=is_error, metadata=metadata)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _string_field_value(text: str, key: str) -> str | None:
|
|
218
|
+
"""Extract `key`'s JSON string value from possibly-truncated JSON, unescaping as we go."""
|
|
219
|
+
marker = f'"{key}"'
|
|
220
|
+
at = text.find(marker)
|
|
221
|
+
if at == -1:
|
|
222
|
+
return None
|
|
223
|
+
i = at + len(marker)
|
|
224
|
+
while i < len(text) and text[i] in ": \t\n":
|
|
225
|
+
i += 1
|
|
226
|
+
if i >= len(text) or text[i] != '"':
|
|
227
|
+
return None
|
|
228
|
+
i += 1
|
|
229
|
+
out: list[str] = []
|
|
230
|
+
escaped = False
|
|
231
|
+
while i < len(text):
|
|
232
|
+
ch = text[i]
|
|
233
|
+
if escaped:
|
|
234
|
+
if ch == "u" and i + 4 < len(text):
|
|
235
|
+
# \uXXXX escape: decode the four hex digits (accents/box-drawing chars are
|
|
236
|
+
# common in terminal corpora; dropping the backslash rendered 'u00e9' garbage).
|
|
237
|
+
hex_digits = text[i + 1 : i + 5]
|
|
238
|
+
try:
|
|
239
|
+
out.append(chr(int(hex_digits, 16)))
|
|
240
|
+
i += 4
|
|
241
|
+
except ValueError:
|
|
242
|
+
out.append(ch)
|
|
243
|
+
else:
|
|
244
|
+
out.append(_UNESCAPE.get(ch, ch))
|
|
245
|
+
escaped = False
|
|
246
|
+
elif ch == "\\":
|
|
247
|
+
escaped = True
|
|
248
|
+
elif ch == '"':
|
|
249
|
+
break # properly terminated string
|
|
250
|
+
else:
|
|
251
|
+
out.append(ch)
|
|
252
|
+
i += 1
|
|
253
|
+
return "".join(out)
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
_UNESCAPE = {"n": "\n", "t": "\t", "r": "\r", '"': '"', "\\": "\\", "/": "/"}
|
|
257
|
+
|
|
258
|
+
# The numeric confidence value in possibly-truncated contract text (salvage path only; complete
|
|
259
|
+
# JSON goes through `_RawObservation`).
|
|
260
|
+
_CONFIDENCE_VALUE = re.compile(r'"confidence"\s*:\s*([0-9]+(?:\.[0-9]+)?)')
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def dumps_observation_contract(observation: Observation) -> str:
|
|
264
|
+
"""Render an Observation back into the JSON output contract (used to seed/demo the format).
|
|
265
|
+
|
|
266
|
+
Carries the confidence fields when present (key order mirroring the contract:
|
|
267
|
+
justification before the number, both after `is_error`) — the verify pass embeds this as
|
|
268
|
+
the draft, and a draft missing the field the contract demands invites the reviser to drop
|
|
269
|
+
it too, thinning stated confidence exactly on the verified (low-confidence) population.
|
|
270
|
+
"""
|
|
271
|
+
payload: JsonObject = {"output": observation.content, "is_error": observation.is_error}
|
|
272
|
+
why = observation.metadata.get("confidence_why")
|
|
273
|
+
if isinstance(why, str) and why:
|
|
274
|
+
payload["confidence_why"] = why
|
|
275
|
+
confidence = observation.metadata.get("confidence")
|
|
276
|
+
if isinstance(confidence, int | float) and not isinstance(confidence, bool):
|
|
277
|
+
payload["confidence"] = float(confidence)
|
|
278
|
+
note = observation.metadata.get("state_note")
|
|
279
|
+
if isinstance(note, str) and note:
|
|
280
|
+
payload["state_note"] = note
|
|
281
|
+
return json.dumps(payload)
|