world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,346 @@
1
+ """Teacher-side rendering: the same conversation under the teacher's chat template.
2
+
3
+ The teacher cannot score the student's token ids (different vocabulary), so it
4
+ scores its own tokenization of the same conversation. This module renders the
5
+ canonical message list with the TEACHER's chat template and reports which
6
+ teacher token ranges cover byte-identical message content, which is what the
7
+ chunk aligner pairs against the student's sampled tokens.
8
+
9
+ Only message CONTENT can be compared. The two templates frame turns
10
+ differently (Qwen writes `<|im_start|>assistant`, GLM writes `<|assistant|>`;
11
+ Qwen writes tool calls as `<function=bash><parameter=command>`, GLM as
12
+ `<tool_call>bash<arg_key>command</arg_key><arg_value>`), so framing tokens have
13
+ no counterpart on the other side and are deliberately left uncovered. Measured
14
+ on the headline run's real spans, framing is 4.5% of sampled tokens, so ~95.5%
15
+ remains scoreable.
16
+
17
+ Two properties of GLM-5.2's template drive the render options:
18
+
19
+ - `clear_thinking=False` keeps `reasoning_content` on EVERY assistant turn.
20
+ The default drops it from historical turns (`loop.index0 > ns.last_user_index`),
21
+ which would make 73% of the run's think tokens unscoreable and, worse, would
22
+ have the teacher condition on a history the student never saw. Passing
23
+ `clear_thinking=False` also short-circuits that `last_user_index` test, which
24
+ makes the render prefix-stable as a side benefit.
25
+ - The template STRIPS visible content (`content.strip()`) and splits
26
+ `<think>...</think>` out of content on its own. Islands are therefore located
27
+ by searching for the stripped form; leading and trailing whitespace the
28
+ student sampled has no counterpart and stays uncovered.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import json
34
+ import logging
35
+ from typing import Literal, Protocol
36
+
37
+ from llm_waterfall.types import ChatMessage, ChatTool
38
+ from pydantic import BaseModel, ConfigDict, Field
39
+
40
+ from wmo.distill.xtoken.byte_offsets import span_byte_ends
41
+
42
+ logger = logging.getLogger(__name__)
43
+
44
+ IslandKind = Literal["reasoning", "text", "tool_argument"]
45
+ """Which part of an assistant message an island covers."""
46
+
47
+
48
+ class TemplateTokenizer(Protocol):
49
+ """The tokenizer slice teacher rendering needs.
50
+
51
+ HuggingFace fast tokenizers satisfy this structurally; the tests use small
52
+ deterministic fakes.
53
+ """
54
+
55
+ def apply_chat_template(
56
+ self,
57
+ conversation: list[dict[str, object]],
58
+ *,
59
+ tools: list[dict[str, object]] | None = ...,
60
+ tokenize: bool = ...,
61
+ add_generation_prompt: bool = ...,
62
+ clear_thinking: bool = ...,
63
+ ) -> str:
64
+ """Render a conversation to a string with the model's chat template."""
65
+ ...
66
+
67
+ def encode(self, text: str, add_special_tokens: bool = ...) -> list[int]:
68
+ """Encode text to token ids."""
69
+ ...
70
+
71
+ def convert_ids_to_tokens(self, ids: list[int]) -> list[str | None]:
72
+ """The surface form of each token id; needed for exact byte offsets."""
73
+ ...
74
+
75
+ def decode(self, token_ids: list[int]) -> str:
76
+ """Decode token ids back to text; used to verify the byte reconstruction."""
77
+ ...
78
+
79
+
80
+ class ContentIsland(BaseModel):
81
+ """One byte-identical content region, located in the teacher's token stream."""
82
+
83
+ model_config = ConfigDict(frozen=True, extra="forbid")
84
+
85
+ kind: IslandKind
86
+ message_index: int = Field(ge=0)
87
+ """Index into the rendered conversation's message list."""
88
+
89
+ text: str = Field(min_length=1)
90
+ """The content bytes this island covers, identical on both sides."""
91
+
92
+ teacher_start: int = Field(ge=1)
93
+ """First teacher token fully inside the island. Position 0 can never be
94
+ scored (no context), so an island never starts there."""
95
+
96
+ teacher_end: int = Field(gt=1)
97
+ """One past the last teacher token fully inside the island."""
98
+
99
+ byte_start: int = Field(ge=0)
100
+ """Island start as a byte offset into the rendered text."""
101
+
102
+ byte_end: int = Field(gt=0)
103
+ """Island end as a byte offset into the rendered text."""
104
+
105
+
106
+ class TeacherRender(BaseModel):
107
+ """A conversation rendered and tokenized under the teacher's template."""
108
+
109
+ model_config = ConfigDict(frozen=True, extra="forbid")
110
+
111
+ token_ids: list[int]
112
+ islands: list[ContentIsland] = Field(default_factory=list)
113
+ unmatched: list[str] = Field(default_factory=list)
114
+ """Content pieces the render did not contain verbatim (the template
115
+ transformed them), reported so the caller can meter the fallback rate
116
+ instead of silently losing signal."""
117
+
118
+
119
+ def _byte_prefixes(text: str) -> list[int]:
120
+ """Byte offset of each character index, plus a final total.
121
+
122
+ Entry i is the number of UTF-8 bytes in `text[:i]`, so a character range
123
+ `[a, b)` maps to the byte range `[out[a], out[b])`. Built in one pass
124
+ because doing `len(text[:i].encode())` per token is quadratic.
125
+ """
126
+ out = [0] * (len(text) + 1)
127
+ total = 0
128
+ for index, char in enumerate(text):
129
+ total += len(char.encode("utf-8"))
130
+ out[index + 1] = total
131
+ return out
132
+
133
+
134
+ def _content_pieces(message: ChatMessage) -> list[tuple[IslandKind, str]]:
135
+ """The comparable content of one assistant message, in render order.
136
+
137
+ Reasoning comes first (the template emits `<think>` before visible text),
138
+ then visible text, then each tool call's argument VALUES. Argument values
139
+ are compared, not the JSON around them: both templates emit the values
140
+ with raw newlines and no JSON escaping (verified for Qwen3.6 and GLM-5.2),
141
+ but the surrounding syntax differs entirely.
142
+ """
143
+ pieces: list[tuple[IslandKind, str]] = []
144
+ raw = message.content if isinstance(message.content, str) else ""
145
+ if "</think>" in raw:
146
+ head, _, tail = raw.partition("</think>")
147
+ reasoning = head.split("<think>")[-1]
148
+ if reasoning.strip():
149
+ pieces.append(("reasoning", reasoning.strip()))
150
+ visible = tail
151
+ else:
152
+ visible = raw
153
+ if visible.strip():
154
+ pieces.append(("text", visible.strip()))
155
+ for call in message.tool_calls or []:
156
+ arguments = call.function.arguments
157
+ if not isinstance(arguments, str):
158
+ continue
159
+ # The template renders each argument VALUE; the JSON envelope differs
160
+ # per template so it is never an island.
161
+ for value in _argument_values(arguments):
162
+ if value.strip():
163
+ pieces.append(("tool_argument", value))
164
+ return pieces
165
+
166
+
167
+ def _argument_values(arguments: str) -> list[str]:
168
+ """The string values of a tool call's JSON arguments object, in order.
169
+
170
+ Non-string values are skipped: templates render them through their own
171
+ formatting (`true` vs `True`, number spacing), so byte identity is not
172
+ guaranteed and a mismatched island is worse than an absent one.
173
+ """
174
+ try:
175
+ parsed = json.loads(arguments)
176
+ except (TypeError, ValueError):
177
+ logger.debug("tool call arguments are not valid JSON; no islands from them")
178
+ return []
179
+ if not isinstance(parsed, dict):
180
+ return []
181
+ return [value for value in parsed.values() if isinstance(value, str)]
182
+
183
+
184
+ def _renderer_messages(messages: list[ChatMessage]) -> list[dict[str, object]]:
185
+ """Convert canonical chat messages into the dict shape chat templates expect.
186
+
187
+ Tool call arguments are passed as a parsed dict: both Qwen3.6's and
188
+ GLM-5.2's templates iterate `arguments.items()` and raise on a string.
189
+ """
190
+ out: list[dict[str, object]] = []
191
+ for message in messages:
192
+ entry: dict[str, object] = {
193
+ "role": message.role,
194
+ "content": message.content if isinstance(message.content, str) else "",
195
+ }
196
+ if message.tool_calls:
197
+ calls: list[dict[str, object]] = []
198
+ for call in message.tool_calls:
199
+ try:
200
+ arguments = json.loads(call.function.arguments)
201
+ except (TypeError, ValueError):
202
+ arguments = {}
203
+ calls.append(
204
+ {
205
+ "type": "function",
206
+ "function": {"name": call.function.name, "arguments": arguments},
207
+ }
208
+ )
209
+ entry["tool_calls"] = calls
210
+ out.append(entry)
211
+ return out
212
+
213
+
214
+ def _tool_specs(tools: list[ChatTool]) -> list[dict[str, object]]:
215
+ """Convert canonical tool definitions into chat-template tool dicts."""
216
+ return [
217
+ {
218
+ "type": "function",
219
+ "function": {
220
+ "name": tool.function.name,
221
+ "description": tool.function.description,
222
+ "parameters": dict(tool.function.parameters),
223
+ },
224
+ }
225
+ for tool in tools
226
+ ]
227
+
228
+
229
+ def render_for_teacher(
230
+ tokenizer: TemplateTokenizer,
231
+ messages: list[ChatMessage],
232
+ tools: list[ChatTool] | None = None,
233
+ ) -> TeacherRender:
234
+ """Render a conversation with the teacher's template and locate content islands.
235
+
236
+ Islands are found by scanning the rendered text forward with a monotonic
237
+ cursor, so a content string that also appears earlier can never be matched
238
+ to the wrong occurrence. Only teacher tokens FULLY inside an island are
239
+ reported: a token straddling an island edge mixes content bytes with
240
+ framing bytes, so its logprob is not comparable.
241
+
242
+ Args:
243
+ tokenizer: The teacher's tokenizer, with its chat template.
244
+ messages: The canonical conversation, in order.
245
+ tools: Tool schemas the conversation was generated with.
246
+
247
+ Returns:
248
+ The teacher token ids plus the located islands, and the content pieces
249
+ the render did not contain verbatim.
250
+ """
251
+ rendered = tokenizer.apply_chat_template(
252
+ _renderer_messages(messages),
253
+ tools=_tool_specs(tools) if tools else None,
254
+ tokenize=False,
255
+ add_generation_prompt=False,
256
+ clear_thinking=False,
257
+ )
258
+ token_ids = tokenizer.encode(rendered, add_special_tokens=False)
259
+ token_byte_ends = _token_byte_ends(tokenizer, token_ids, rendered)
260
+ if token_byte_ends is None:
261
+ return TeacherRender(token_ids=token_ids, islands=[], unmatched=[])
262
+
263
+ prefixes = _byte_prefixes(rendered)
264
+ islands: list[ContentIsland] = []
265
+ unmatched: list[str] = []
266
+ cursor = 0
267
+ for message_index, message in enumerate(messages):
268
+ if message.role != "assistant":
269
+ continue
270
+ for kind, piece in _content_pieces(message):
271
+ found = rendered.find(piece, cursor)
272
+ if found < 0:
273
+ unmatched.append(piece)
274
+ logger.debug(
275
+ "content piece of kind %s in message %d is not present verbatim in "
276
+ "the teacher render; it will not be scored",
277
+ kind,
278
+ message_index,
279
+ )
280
+ continue
281
+ cursor = found + len(piece)
282
+ byte_start = prefixes[found]
283
+ byte_end = prefixes[found + len(piece)]
284
+ span = _tokens_inside(token_byte_ends, byte_start, byte_end)
285
+ if span is None:
286
+ unmatched.append(piece)
287
+ continue
288
+ start, end = span
289
+ islands.append(
290
+ ContentIsland(
291
+ kind=kind,
292
+ message_index=message_index,
293
+ text=piece,
294
+ teacher_start=start,
295
+ teacher_end=end,
296
+ byte_start=byte_start,
297
+ byte_end=byte_end,
298
+ )
299
+ )
300
+ return TeacherRender(token_ids=token_ids, islands=islands, unmatched=unmatched)
301
+
302
+
303
+ def _token_byte_ends(
304
+ tokenizer: TemplateTokenizer, token_ids: list[int], rendered: str
305
+ ) -> list[int] | None:
306
+ """Cumulative byte-end offset per teacher token over the rendered text.
307
+
308
+ Uses the same exact reconstruction as the student side so both sides'
309
+ offsets live in one byte space.
310
+ """
311
+ result = span_byte_ends(tokenizer, token_ids)
312
+ if result is None:
313
+ return None
314
+ ends, span = result
315
+ if span != rendered.encode("utf-8"):
316
+ logger.warning(
317
+ "re-encoding the teacher render does not reproduce it byte for byte "
318
+ "(%d vs %d bytes), so island offsets would be wrong; nothing is scored",
319
+ len(span),
320
+ len(rendered.encode("utf-8")),
321
+ )
322
+ return None
323
+ return ends
324
+
325
+
326
+ def _tokens_inside(
327
+ token_byte_ends: list[int], byte_start: int, byte_end: int
328
+ ) -> tuple[int, int] | None:
329
+ """The half-open token range fully contained in a byte range.
330
+
331
+ A token spans `[ends[i - 1], ends[i])`. Only tokens whose whole span lies
332
+ inside `[byte_start, byte_end)` qualify, and position 0 is excluded because
333
+ it has no context and can never carry a logprob.
334
+ """
335
+ start: int | None = None
336
+ end: int | None = None
337
+ previous = 0
338
+ for index, boundary in enumerate(token_byte_ends):
339
+ if index >= 1 and previous >= byte_start and boundary <= byte_end:
340
+ if start is None:
341
+ start = index
342
+ end = index + 1
343
+ previous = boundary
344
+ if start is None or end is None:
345
+ return None
346
+ return start, end
wmo/engine/__init__.py ADDED
@@ -0,0 +1,28 @@
1
+ """The world-model engine: prompt assembly, the WorldModel, the build pipeline, demo, play.
2
+
3
+ Evaluation of a built world model (open-loop replay fidelity + closed-loop task success) lives in
4
+ `wmo.evals`."""
5
+
6
+ from wmo.engine.build import build, ingest, split_traces, split_traces_3way
7
+ from wmo.engine.demo import DemoReplay, DemoStep, run_demo
8
+ from wmo.engine.loader import load_world_model
9
+ from wmo.engine.play import PlayTurn, parse_action, play_turn
10
+ from wmo.engine.reporting import BuildReporter, NullReporter
11
+ from wmo.engine.world_model import WorldModel
12
+
13
+ __all__ = [
14
+ "build",
15
+ "ingest",
16
+ "split_traces",
17
+ "split_traces_3way",
18
+ "DemoReplay",
19
+ "DemoStep",
20
+ "run_demo",
21
+ "load_world_model",
22
+ "PlayTurn",
23
+ "parse_action",
24
+ "play_turn",
25
+ "BuildReporter",
26
+ "NullReporter",
27
+ "WorldModel",
28
+ ]