world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/connect/store.py ADDED
@@ -0,0 +1,199 @@
1
+ """Context bundle persistence and rendering under `<project>/.wmo/context/`.
2
+
3
+ A bundle is one pull's replayable artifact: `manifest.json` (what was pulled, when, from where)
4
+ plus `items.jsonl` (one normalized `ContextItem` per line). "Filesystem as DB", like the model
5
+ store: loading a bundle is just reading its folder. `ContextStore.save`/`load` persist and read
6
+ bundles, and `render_markdown` turns a bundle into a deterministic markdown document callers can
7
+ write into a model's knowledge dir.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import shutil
13
+ from pathlib import Path
14
+
15
+ from pydantic import BaseModel
16
+
17
+ from wmo.config.config import ARTIFACT_DIR
18
+ from wmo.config.store import validate_name
19
+ from wmo.connect.types import ContextItem, PullQuery
20
+
21
+ _MANIFEST_FILENAME = "manifest.json"
22
+ _ITEMS_FILENAME = "items.jsonl"
23
+
24
+
25
+ class BundleManifest(BaseModel):
26
+ """Provenance for one saved bundle: what was pulled, when, by which connector.
27
+
28
+ Attributes:
29
+ name: Bundle name (the directory name under `.wmo/context/`).
30
+ connector: The connector that produced the bundle.
31
+ query: The exact `PullQuery` used, kept for replayable re-pulls.
32
+ pulled_at: ISO-8601 timestamp of the pull.
33
+ item_count: Number of items in `items.jsonl`.
34
+ account: Human-readable identity the pull ran as, when known.
35
+ """
36
+
37
+ name: str
38
+ connector: str
39
+ query: PullQuery
40
+ pulled_at: str
41
+ item_count: int
42
+ account: str | None = None
43
+
44
+
45
+ class ContextStore:
46
+ """Named context bundles on disk under `<root>/.wmo/context/<name>/`.
47
+
48
+ `root` is the PROJECT directory (the parent of `.wmo/`), defaulting to the current working
49
+ directory; tests pass a tmp path. Note this differs from `WorldModelStore`, whose root is
50
+ the `.wmo` artifact dir itself.
51
+ """
52
+
53
+ def __init__(self, root: str | Path | None = None) -> None:
54
+ self.root = Path(root) if root is not None else Path.cwd()
55
+ self.context_dir = self.root / ARTIFACT_DIR / "context"
56
+
57
+ def bundle_dir(self, name: str) -> Path:
58
+ """The directory a bundle named `name` lives in (may not exist)."""
59
+ return self.context_dir / _validated_bundle_name(name)
60
+
61
+ def save(
62
+ self, manifest: BundleManifest, items: list[ContextItem], *, overwrite: bool = False
63
+ ) -> Path:
64
+ """Write one bundle (`manifest.json` + `items.jsonl`); returns its directory.
65
+
66
+ Raises:
67
+ FileExistsError: When the bundle already exists and `overwrite` is False.
68
+ ValueError: When the bundle name is not a safe single path segment.
69
+ """
70
+ directory = self.bundle_dir(manifest.name)
71
+ if directory.exists():
72
+ if not overwrite:
73
+ raise FileExistsError(
74
+ f"context bundle {manifest.name!r} already exists at {directory}; "
75
+ "pass overwrite=True to replace it"
76
+ )
77
+ shutil.rmtree(directory)
78
+ directory.mkdir(parents=True)
79
+ manifest_text = manifest.model_dump_json(indent=2) + "\n"
80
+ (directory / _MANIFEST_FILENAME).write_text(manifest_text, encoding="utf-8")
81
+ lines = "".join(item.model_dump_json() + "\n" for item in items)
82
+ (directory / _ITEMS_FILENAME).write_text(lines, encoding="utf-8")
83
+ return directory
84
+
85
+ def load(self, name: str) -> tuple[BundleManifest, list[ContextItem]]:
86
+ """Read one bundle back as (manifest, items).
87
+
88
+ Raises:
89
+ FileNotFoundError: When no bundle named `name` exists.
90
+ """
91
+ directory = self.bundle_dir(name)
92
+ manifest_path = directory / _MANIFEST_FILENAME
93
+ if not manifest_path.exists():
94
+ raise FileNotFoundError(
95
+ f"no context bundle named {name!r} under {self.context_dir}; "
96
+ "pull a bundle and persist it with ContextStore.save first"
97
+ )
98
+ manifest = BundleManifest.model_validate_json(manifest_path.read_text(encoding="utf-8"))
99
+ items_text = (directory / _ITEMS_FILENAME).read_text(encoding="utf-8")
100
+ items = [
101
+ ContextItem.model_validate_json(line)
102
+ for line in items_text.splitlines()
103
+ if line.strip()
104
+ ]
105
+ return manifest, items
106
+
107
+ def list_bundles(self) -> list[BundleManifest]:
108
+ """Manifests of every saved bundle, sorted by directory name."""
109
+ if not self.context_dir.exists():
110
+ return []
111
+ manifests: list[BundleManifest] = []
112
+ for child in sorted(self.context_dir.iterdir()):
113
+ manifest_path = child / _MANIFEST_FILENAME
114
+ if child.is_dir() and manifest_path.exists():
115
+ manifests.append(
116
+ BundleManifest.model_validate_json(manifest_path.read_text(encoding="utf-8"))
117
+ )
118
+ return manifests
119
+
120
+ def delete(self, name: str) -> bool:
121
+ """Remove one bundle directory; returns whether it existed."""
122
+ directory = self.bundle_dir(name)
123
+ if not directory.exists():
124
+ return False
125
+ shutil.rmtree(directory)
126
+ return True
127
+
128
+
129
+ def render_markdown(
130
+ manifest: BundleManifest, items: list[ContextItem], *, max_chars: int | None = None
131
+ ) -> str:
132
+ """Render a bundle as one deterministic markdown document.
133
+
134
+ A provenance header (connector, pulled_at, query) is followed by one `## title` section per
135
+ item (a kind/date/url fact line, then the body). When the result would exceed `max_chars`,
136
+ whole items are dropped from the tail and a final "... n items omitted" line makes the
137
+ truncation visible, never silent.
138
+ """
139
+ header = _render_header(manifest)
140
+ sections = [_render_item(item) for item in items]
141
+ full = "\n\n".join([header, *sections]) + "\n"
142
+ if max_chars is None or len(full) <= max_chars:
143
+ return full
144
+ candidate = full
145
+ for kept in range(len(items) - 1, -1, -1):
146
+ omitted = len(items) - kept
147
+ tail = f"... {omitted} items omitted"
148
+ candidate = "\n\n".join([header, *sections[:kept], tail]) + "\n"
149
+ if len(candidate) <= max_chars:
150
+ return candidate
151
+ # Even the header alone is over budget; return the loud minimal form anyway.
152
+ return candidate
153
+
154
+
155
+ def _validated_bundle_name(name: str) -> str:
156
+ """Reject unsafe bundle names with bundle-specific wording (same rules as model names)."""
157
+ try:
158
+ return validate_name(name)
159
+ except ValueError as exc:
160
+ raise ValueError(
161
+ f"invalid context bundle name {name!r}: use letters, digits, '.', '_', '-' "
162
+ "(must start with a letter or digit, no path separators)"
163
+ ) from exc
164
+
165
+
166
+ def _render_header(manifest: BundleManifest) -> str:
167
+ """The provenance block: bundle name, connector, pull time, identity, query, item count."""
168
+ query_parts = [
169
+ f"{field}={value}"
170
+ for field, value in manifest.query.model_dump(mode="json").items()
171
+ if value is not None
172
+ ]
173
+ lines = [
174
+ f"# Context bundle: {manifest.name}",
175
+ "",
176
+ f"- connector: {manifest.connector}",
177
+ f"- pulled_at: {manifest.pulled_at}",
178
+ ]
179
+ if manifest.account:
180
+ lines.append(f"- account: {manifest.account}")
181
+ lines.append(f"- query: {', '.join(query_parts)}")
182
+ lines.append(f"- items: {manifest.item_count}")
183
+ return "\n".join(lines)
184
+
185
+
186
+ def _render_item(item: ContextItem) -> str:
187
+ """One `## title` section: a kind/date/url fact line, then the body."""
188
+ facts = [item.kind.value]
189
+ if item.created_at:
190
+ facts.append(f"created {item.created_at}")
191
+ if item.updated_at:
192
+ facts.append(f"updated {item.updated_at}")
193
+ if item.url:
194
+ facts.append(item.url)
195
+ section = f"## {item.title}\n\n{' | '.join(facts)}"
196
+ body = item.body.strip()
197
+ if body:
198
+ section += f"\n\n{body}"
199
+ return section
wmo/connect/types.py ADDED
@@ -0,0 +1,156 @@
1
+ """Normalized types for context connectors.
2
+
3
+ Connectors (github, google, slack, notion, ...) authenticate against a service and pull content
4
+ into these vendor-agnostic shapes. Everything downstream (the bundle store, markdown rendering,
5
+ knowledge attachment) operates on `ContextItem`, never on raw vendor payloads. The `opt_str`,
6
+ `capped`, and `strip_html` helpers are the shared coercions for raw vendor JSON and fetched
7
+ content bodies.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import html
13
+ import re
14
+ from collections.abc import Iterator
15
+ from contextlib import contextmanager
16
+ from enum import StrEnum
17
+ from typing import Literal
18
+
19
+ import httpx
20
+ from pydantic import BaseModel, Field
21
+
22
+ from wmo.core.types import JsonObject, JsonValue
23
+
24
+ # Fetched content is capped so one huge document or page cannot blow up a bundle.
25
+ CONTENT_CAP_CHARS = 200_000
26
+ TRUNCATION_MARKER = f"\n[content truncated at {CONTENT_CAP_CHARS} characters]"
27
+
28
+
29
+ def opt_str(value: JsonValue | None) -> str | None:
30
+ """`value` when it is a non-empty string, else None (vendor JSON field coercion)."""
31
+ return value if isinstance(value, str) and value else None
32
+
33
+
34
+ def capped(text: str) -> str:
35
+ """Cap fetched content, appending a loud truncation marker when anything was cut."""
36
+ if len(text) <= CONTENT_CAP_CHARS:
37
+ return text
38
+ return text[:CONTENT_CAP_CHARS] + TRUNCATION_MARKER
39
+
40
+
41
+ def strip_html(markup: str) -> str:
42
+ """A small HTML-to-text fallback: drop script/style, break on block ends, strip tags."""
43
+ text = re.sub(r"(?is)<(script|style)\b.*?</\1>", " ", markup)
44
+ text = re.sub(r"(?is)<br\s*/?>|</p>|</div>", "\n", text)
45
+ text = re.sub(r"(?s)<[^>]*>", " ", text)
46
+ text = html.unescape(text)
47
+ lines = [re.sub(r"[ \t]+", " ", line).strip() for line in text.splitlines()]
48
+ return "\n".join(line for line in lines if line)
49
+
50
+
51
+ class ConnectError(RuntimeError):
52
+ """A connector operation failed; messages say what went wrong and what to do about it."""
53
+
54
+
55
+ @contextmanager
56
+ def transport_errors(host: str) -> Iterator[None]:
57
+ """Turn httpx transport failures inside the block into actionable ConnectErrors.
58
+
59
+ Connectors wrap their HTTP calls with this so network-level failures (DNS errors, refused
60
+ connections, timeouts) honor the ConnectError contract instead of escaping to callers as
61
+ raw httpx tracebacks.
62
+
63
+ Args:
64
+ host: The host the block talks to, named in the error message.
65
+
66
+ Raises:
67
+ ConnectError: For any `httpx.HTTPError` raised inside the block.
68
+ """
69
+ try:
70
+ yield
71
+ except httpx.HTTPError as exc:
72
+ detail = str(exc) or type(exc).__name__
73
+ raise ConnectError(
74
+ f"could not reach {host} ({detail}); check your network connection and retry"
75
+ ) from exc
76
+
77
+
78
+ class ItemKind(StrEnum):
79
+ """The normalized kind of one pulled content item."""
80
+
81
+ DOCUMENT = "document"
82
+ PAGE = "page"
83
+ ISSUE = "issue"
84
+ PULL_REQUEST = "pull_request"
85
+ MESSAGE = "message"
86
+ THREAD = "thread"
87
+ EMAIL = "email"
88
+ EVENT = "event"
89
+ FILE = "file"
90
+
91
+
92
+ class ContextItem(BaseModel):
93
+ """One normalized piece of pulled content (an issue, a page, a message, ...).
94
+
95
+ Attributes:
96
+ id: Stable identifier within the source service (issue number, page id, message ts).
97
+ source: The connector name that produced the item (e.g. "github").
98
+ kind: What the item is, from the normalized `ItemKind` vocabulary.
99
+ title: Short human title; shown as the item's markdown section heading.
100
+ body: The content itself, plain text or markdown.
101
+ url: Canonical link back to the item, when the service has one.
102
+ created_at: ISO-8601 creation timestamp, when known.
103
+ updated_at: ISO-8601 last-modified timestamp, when known.
104
+ metadata: Connector-specific extras (labels, authors, channel ids) as arbitrary JSON.
105
+ """
106
+
107
+ id: str
108
+ source: str
109
+ kind: ItemKind
110
+ title: str
111
+ body: str
112
+ url: str | None = None
113
+ created_at: str | None = None
114
+ updated_at: str | None = None
115
+ metadata: JsonObject = Field(default_factory=dict)
116
+
117
+
118
+ class PullQuery(BaseModel):
119
+ """What to pull: the parameters every connector's `pull` accepts.
120
+
121
+ Attributes:
122
+ target: Service-specific container: a repo "owner/name", a channel name, a calendar id,
123
+ a drive folder.
124
+ query: Free-text or service search-syntax filter.
125
+ since: ISO-8601 date or datetime lower bound on item time.
126
+ until: ISO-8601 date or datetime upper bound on item time.
127
+ limit: Maximum number of items a connector may fetch (connectors must cap at this).
128
+ """
129
+
130
+ target: str | None = None
131
+ query: str | None = None
132
+ since: str | None = None
133
+ until: str | None = None
134
+ limit: int = 100
135
+
136
+
137
+ class ConnectorAuth(BaseModel):
138
+ """A stored credential for one connector.
139
+
140
+ Attributes:
141
+ kind: "oauth" for browser/device OAuth grants, "token" for pasted or env-injected tokens.
142
+ access_token: The bearer credential API calls send.
143
+ refresh_token: OAuth refresh token, when the provider issued one.
144
+ expires_at: ISO-8601 absolute expiry of `access_token`, when known.
145
+ scopes: The granted OAuth scopes.
146
+ account: Human-readable identity captured at connect time (e.g. "octocat").
147
+ extra: Connector-specific extras (e.g. a slack team id) as arbitrary JSON.
148
+ """
149
+
150
+ kind: Literal["oauth", "token"]
151
+ access_token: str
152
+ refresh_token: str | None = None
153
+ expires_at: str | None = None
154
+ scopes: list[str] = Field(default_factory=list)
155
+ account: str | None = None
156
+ extra: JsonObject = Field(default_factory=dict)
wmo/core/__init__.py ADDED
@@ -0,0 +1,21 @@
1
+ """Core data types shared across the harness."""
2
+
3
+ from wmo.core.types import (
4
+ Action,
5
+ ActionKind,
6
+ EnvState,
7
+ Observation,
8
+ Session,
9
+ Step,
10
+ Trace,
11
+ )
12
+
13
+ __all__ = [
14
+ "Action",
15
+ "ActionKind",
16
+ "EnvState",
17
+ "Observation",
18
+ "Session",
19
+ "Step",
20
+ "Trace",
21
+ ]
wmo/core/parsing.py ADDED
@@ -0,0 +1,281 @@
1
+ """Robust parsing of model completions into structured values.
2
+
3
+ Two concerns live here because both the serving engine and the optimizer need them, and `wmo.core`
4
+ has no dependencies (so neither imports the other):
5
+
6
+ - `extract_json_object`: pull the first complete JSON object out of a noisy LLM reply.
7
+ - `parse_observation`: turn a world-model completion into a structured `Observation`.
8
+
9
+ The world-model output contract (see `wmo.core.render.build_env_prompt`) asks the model to reply
10
+ with a JSON object ``{"output": str, "is_error": bool, "state_note": str}``. `parse_observation`
11
+ is lenient: a reply that is not JSON is treated as a plain-text observation, so a model that ignores
12
+ the contract still produces a usable (non-error) observation rather than crashing the step.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import re
19
+
20
+ from pydantic import BaseModel, ValidationError, field_validator
21
+
22
+ from wmo.core.types import JsonObject, JsonValue, Observation
23
+
24
+
25
+ def accepted_confidence(value: float | int | str | bool) -> float | None:
26
+ """The one definition of a usable stated confidence: a finite number in [0, 1], else None.
27
+
28
+ Gates and calibration both consume this, so the acceptance rule must not fork: booleans are
29
+ not confidences (JSON `true` is not 1.0), NaN/inf are garbage, and OUT-OF-RANGE numerics
30
+ degrade to "not stated" rather than clamping — a model answering 85 (percent) or 7 (out of
31
+ 10) has violated the 0.0-1.0 contract, and clamping such a reply to 1.0 would record maximal
32
+ certainty on exactly the steps where the model is off the rails. Missing conservatively
33
+ gates as LOW; malformed must never gate as certain.
34
+ """
35
+ if isinstance(value, bool):
36
+ return None
37
+ try:
38
+ parsed = float(value)
39
+ except ValueError:
40
+ return None
41
+ if not (0.0 <= parsed <= 1.0): # also rejects NaN (all comparisons false) and +/-inf
42
+ return None
43
+ return parsed
44
+
45
+
46
+ def extract_json_object(text: str) -> str | None:
47
+ """Return the first complete JSON object substring in `text`, or None if there is none.
48
+
49
+ Scans from the first ``{`` to its balanced closing ``}``, tracking string literals and escapes.
50
+ This tolerates ```json fences, surrounding prose, nested objects, and multiple objects (the
51
+ first is returned) — cases a greedy/lazy regex gets wrong.
52
+ """
53
+ start = text.find("{")
54
+ if start == -1:
55
+ return None
56
+ depth = 0
57
+ in_string = False
58
+ escaped = False
59
+ for i in range(start, len(text)):
60
+ ch = text[i]
61
+ if in_string:
62
+ if escaped:
63
+ escaped = False
64
+ elif ch == "\\":
65
+ escaped = True
66
+ elif ch == '"':
67
+ in_string = False
68
+ continue
69
+ if ch == '"':
70
+ in_string = True
71
+ elif ch == "{":
72
+ depth += 1
73
+ elif ch == "}":
74
+ depth -= 1
75
+ if depth == 0:
76
+ return text[start : i + 1]
77
+ return None
78
+
79
+
80
+ class _RawObservation(BaseModel):
81
+ """Lenient view of the world-model JSON contract before normalization.
82
+
83
+ The reasoning-mode fields (`reasoning`, `kb_note`, `ground_query` — see
84
+ `wmo.core.render.output_contract`) default to empty so base-contract replies parse unchanged.
85
+ """
86
+
87
+ reasoning: str = ""
88
+ output: str = ""
89
+ is_error: bool = False
90
+ state_note: str = ""
91
+ kb_note: str = ""
92
+ ground_query: str = ""
93
+ state_update: str = ""
94
+ # Verbalized confidence (WS-A6): None when the model didn't state one. Lenient like the rest
95
+ # of the contract — an off-contract value degrades to "no stated confidence", never a crash.
96
+ confidence: float | None = None
97
+ confidence_why: str = ""
98
+
99
+ @field_validator("confidence", mode="before")
100
+ @classmethod
101
+ def _lenient_confidence(cls, value: JsonValue) -> float | None:
102
+ """Coerce the raw JSON field through `accepted_confidence` (off-contract -> None)."""
103
+ if isinstance(value, int | float | str):
104
+ return accepted_confidence(value)
105
+ return None
106
+
107
+
108
+ # The keys that mark a reply as following the observation contract (any one present is enough).
109
+ # Used to tell a real — possibly empty — contract response apart from arbitrary JSON that happens
110
+ # to validate against `_RawObservation`'s all-defaulted fields. Deliberately ONLY the core keys:
111
+ # every complete contract reply (base or reasoning mode) carries `output`/`is_error`, while a
112
+ # reasoning-mode superset key alone (e.g. off-contract JSON with a "reasoning" field but no
113
+ # "output") must fall through to the plain-text fallback, not become an empty observation.
114
+ # Confidence-mode keys are deliberately excluded too: an arbitrary API payload with its own
115
+ # "confidence" field must not be mistaken for a contract reply.
116
+ _CONTRACT_KEYS = frozenset({"output", "is_error", "state_note"})
117
+
118
+
119
+ def parse_observation(text: str) -> Observation:
120
+ """Parse a world-model completion into a structured Observation.
121
+
122
+ Prefers the JSON contract ``{"output", "is_error", "state_note"}`` and its reasoning-mode
123
+ superset (``reasoning``/``kb_note``/``ground_query``). ``output`` becomes the observation the
124
+ agent sees; every other populated field is carried in ``metadata`` (``state_note`` feeds the
125
+ session scratchpad, ``kb_note`` the cross-session knowledge base, ``ground_query`` the
126
+ grounder, ``reasoning`` is kept for inspection only). Falls back to treating the whole reply
127
+ as plain observation text when it is not the expected JSON, so an off-contract model still
128
+ yields a usable observation.
129
+ """
130
+ raw = extract_json_object(text)
131
+ if raw is not None:
132
+ try:
133
+ obj: object = json.loads(raw)
134
+ except json.JSONDecodeError:
135
+ obj = None
136
+ # Recognize the contract by the PRESENCE of its keys, not by truthy values.
137
+ # `_RawObservation` defaults every field, so arbitrary JSON like `{"foo": 1}` would validate
138
+ # to an all-empty observation; requiring a contract key keeps that falling through to raw
139
+ # text. But a legitimate silent success `{"output": "", "is_error": false, ...}` (many shell
140
+ # writes/redirects print nothing) MUST be honored as an empty observation, not re-serialized
141
+ # as visible JSON text — closed-loop rollouts would otherwise show spurious output.
142
+ if isinstance(obj, dict) and _CONTRACT_KEYS.intersection(obj):
143
+ try:
144
+ parsed = _RawObservation.model_validate(obj)
145
+ except ValidationError:
146
+ parsed = None
147
+ if parsed is not None:
148
+ metadata: JsonObject = {}
149
+ for key, value in (
150
+ ("state_note", parsed.state_note),
151
+ ("reasoning", parsed.reasoning),
152
+ ("kb_note", parsed.kb_note),
153
+ ("ground_query", parsed.ground_query),
154
+ ("state_update", parsed.state_update),
155
+ ("confidence_why", parsed.confidence_why),
156
+ ):
157
+ if value:
158
+ metadata[key] = value
159
+ # Separate from the truthiness loop: a stated confidence of 0.0 must survive.
160
+ if parsed.confidence is not None:
161
+ metadata["confidence"] = parsed.confidence
162
+ return Observation(
163
+ content=parsed.output, is_error=parsed.is_error, metadata=metadata
164
+ )
165
+ # A contract reply cut off mid-generation is not valid JSON at all: salvage the fields the
166
+ # text already contains rather than surfacing the raw truncated JSON as the observation.
167
+ salvaged = _salvage_truncated_contract(text)
168
+ if salvaged is not None:
169
+ return salvaged
170
+ return Observation(content=text.strip())
171
+
172
+
173
+ def _salvage_truncated_contract(text: str) -> Observation | None:
174
+ """Recover a contract reply whose JSON never closed (token-budget truncation).
175
+
176
+ Long deliberations plus long escaped observations can blow the completion budget mid-string;
177
+ without this, the ENTIRE raw contract text (reasoning included) becomes the observation the
178
+ agent sees — observed live as a catastrophic 0.26-fidelity step. Conservative trigger: the
179
+ text must look like a contract object (starts with ``{`` and names an ``"output"`` key) and
180
+ must NOT have parsed as complete JSON (callers try that first). Recovered string fields are
181
+ unescaped up to the truncation point.
182
+ """
183
+ stripped = text.strip()
184
+ if not stripped.startswith("{") or '"output"' not in stripped:
185
+ return None
186
+ output = _string_field_value(stripped, "output")
187
+ if output is None:
188
+ return None
189
+ metadata: JsonObject = {}
190
+ # Recover every metadata-carried contract field the truncated text still contains —
191
+ # dropping state_note/state_update here would silently stall the scratchpad and belief
192
+ # profile for the rest of the session.
193
+ salvage_keys = (
194
+ "reasoning",
195
+ "state_note",
196
+ "kb_note",
197
+ "ground_query",
198
+ "state_update",
199
+ "confidence_why",
200
+ )
201
+ for key in salvage_keys:
202
+ value = _string_field_value(stripped, key)
203
+ if value:
204
+ metadata[key] = value
205
+ # Salvage a stated confidence too: truncation correlates with HARD steps, so silently
206
+ # dropping their confidences would bias any calibration analysis toward the easy ones.
207
+ # Same acceptance rule as the validator — the two paths must not fork.
208
+ match = _CONFIDENCE_VALUE.search(stripped)
209
+ if match is not None:
210
+ confidence = accepted_confidence(match.group(1))
211
+ if confidence is not None:
212
+ metadata["confidence"] = confidence
213
+ is_error = re.search(r'"is_error"\s*:\s*true', stripped) is not None
214
+ return Observation(content=output, is_error=is_error, metadata=metadata)
215
+
216
+
217
+ def _string_field_value(text: str, key: str) -> str | None:
218
+ """Extract `key`'s JSON string value from possibly-truncated JSON, unescaping as we go."""
219
+ marker = f'"{key}"'
220
+ at = text.find(marker)
221
+ if at == -1:
222
+ return None
223
+ i = at + len(marker)
224
+ while i < len(text) and text[i] in ": \t\n":
225
+ i += 1
226
+ if i >= len(text) or text[i] != '"':
227
+ return None
228
+ i += 1
229
+ out: list[str] = []
230
+ escaped = False
231
+ while i < len(text):
232
+ ch = text[i]
233
+ if escaped:
234
+ if ch == "u" and i + 4 < len(text):
235
+ # \uXXXX escape: decode the four hex digits (accents/box-drawing chars are
236
+ # common in terminal corpora; dropping the backslash rendered 'u00e9' garbage).
237
+ hex_digits = text[i + 1 : i + 5]
238
+ try:
239
+ out.append(chr(int(hex_digits, 16)))
240
+ i += 4
241
+ except ValueError:
242
+ out.append(ch)
243
+ else:
244
+ out.append(_UNESCAPE.get(ch, ch))
245
+ escaped = False
246
+ elif ch == "\\":
247
+ escaped = True
248
+ elif ch == '"':
249
+ break # properly terminated string
250
+ else:
251
+ out.append(ch)
252
+ i += 1
253
+ return "".join(out)
254
+
255
+
256
+ _UNESCAPE = {"n": "\n", "t": "\t", "r": "\r", '"': '"', "\\": "\\", "/": "/"}
257
+
258
+ # The numeric confidence value in possibly-truncated contract text (salvage path only; complete
259
+ # JSON goes through `_RawObservation`).
260
+ _CONFIDENCE_VALUE = re.compile(r'"confidence"\s*:\s*([0-9]+(?:\.[0-9]+)?)')
261
+
262
+
263
+ def dumps_observation_contract(observation: Observation) -> str:
264
+ """Render an Observation back into the JSON output contract (used to seed/demo the format).
265
+
266
+ Carries the confidence fields when present (key order mirroring the contract:
267
+ justification before the number, both after `is_error`) — the verify pass embeds this as
268
+ the draft, and a draft missing the field the contract demands invites the reviser to drop
269
+ it too, thinning stated confidence exactly on the verified (low-confidence) population.
270
+ """
271
+ payload: JsonObject = {"output": observation.content, "is_error": observation.is_error}
272
+ why = observation.metadata.get("confidence_why")
273
+ if isinstance(why, str) and why:
274
+ payload["confidence_why"] = why
275
+ confidence = observation.metadata.get("confidence")
276
+ if isinstance(confidence, int | float) and not isinstance(confidence, bool):
277
+ payload["confidence"] = float(confidence)
278
+ note = observation.metadata.get("state_note")
279
+ if isinstance(note, str) and note:
280
+ payload["state_note"] = note
281
+ return json.dumps(payload)