world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/ingest/mastra.py ADDED
@@ -0,0 +1,330 @@
1
+ """Mastra adapter: turn a Mastra AI-tracing export into `Trace`s.
2
+
3
+ [Mastra](https://mastra.ai) (a TypeScript agent framework) records agent runs as **AI-tracing
4
+ spans** (`ExportedSpan`), typed by `type`, that share a `traceId`. The span id field is `id`, times
5
+ are `startTime`/`endTime`, and errors ride on a structured `errorInfo` object:
6
+
7
+ {"traceId": "t1", "id": "s2", "parentSpanId": "s1",
8
+ "name": "modelGeneration", "type": "model_generation",
9
+ "input": [{"role": "user", "content": "what's the weather in Paris?"}],
10
+ "output": {"role": "assistant",
11
+ "toolCalls": [{"toolCallId": "c1", "toolName": "getWeather",
12
+ "input": {"city": "Paris"}}]}, # AI SDK v5 uses `input` (v4: `args`)
13
+ "attributes": {"model": "gpt-4o"}, "startTime": "2026-01-01T00:00:01.000Z"}
14
+ {"traceId": "t1", "id": "s3", "name": "getWeather", "type": "tool_call",
15
+ "input": {"city": "Paris"}, "output": "18C and sunny", "startTime": "..."}
16
+
17
+ Because this is not an OTLP/OpenInference span shape, the adapter overrides `spans_from_payload`
18
+ (like `wmo.ingest.langfuse`) and emits `SpanRecord`s in the **OTel-GenAI vocabulary** so the shared
19
+ classifier/normalizer (`wmo.ingest.normalize`) does the pairing/state/metadata work:
20
+
21
+ - a `model_generation` span (or the pre-rename `llm_generation`) whose `output` carries tool calls
22
+ -> one `chat` action span per call (`gen_ai.tool.name` + `gen_ai.tool.call.arguments`). Mastra
23
+ exposes tool calls as either the AI SDK `toolCalls` (`{toolCallId, toolName, input}` on v5;
24
+ `args` on v4) or the OpenAI `tool_calls` (`{id, function:{name, arguments}}`); both are handled.
25
+ (Tool calls may instead appear only as separate `tool_call` spans — that case is covered below.)
26
+ - a `tool_call` / `mcp_tool_call` span -> an `execute_tool` result span (`gen_ai.tool.message`
27
+ from `output`), also carrying name/args (from `name`/`input`) so a standalone tool span still
28
+ pairs; the normalizer backfills the action's name/args from here if the model span lacked them.
29
+ - a `model_generation` with no tool call -> a plain `chat` message span (`gen_ai.completion`).
30
+ - `agent_run` / `workflow_*` / `model_chunk` / `model_step` / `generic` container/noise spans are
31
+ skipped (no `(action) -> observation` step); an `agent_run`/`model_generation` input supplies
32
+ the task.
33
+ - a span with an `errorInfo` (or an error status) sets `status_error=True`.
34
+
35
+ Spans order by `startTime`/`startedAt` (ISO-8601 or datetime -> a monotonic ordinal, index if none).
36
+
37
+ Accepted file shapes (`from_file`): a single span, a JSON array of spans, a wrapper
38
+ (`{"spans": [...]}` / `{"traces": [...]}` / `{"data": [...]}`), or JSONL. Grouping is by `traceId`.
39
+
40
+ Pull: `_pull_payloads` fetches from a running Mastra server's observability API
41
+ (`{base}/api/observability/traces`), with the base URL passed as `--project` (or `$MASTRA_URL`). The
42
+ response is handed to the same flexible span extractor, so it tolerates the server's wrapper shape.
43
+ """
44
+
45
+ from __future__ import annotations
46
+
47
+ import os
48
+
49
+ import httpx
50
+ from pydantic import JsonValue
51
+
52
+ from wmo.core.types import JsonObject
53
+ from wmo.ingest.adapter import VendorPull, register_adapter
54
+ from wmo.ingest.base import BaseTraceAdapter
55
+ from wmo.ingest.normalize import SpanRecord, as_text, iso_to_ordinal
56
+
57
+ # Mastra self-hosts, so the "vendor" is a server base URL (dev default http://localhost:4111).
58
+ _MASTRA_URL_ENV = "MASTRA_URL"
59
+
60
+ # `type` values (normalized lowercase). LLM/agent turns vs tool executions; the rest are
61
+ # containers/noise that carry no standalone step. Mastra renamed LLM spans to "model" spans
62
+ # (changelog 2025-11-01): `model_generation` is current, `llm_generation` is the pre-rename name we
63
+ # still accept. `model_chunk`/`model_step`/`agent_run`/`workflow_*`/`generic` are not steps.
64
+ _LLM_TYPES = frozenset({"model_generation", "llm_generation"})
65
+ _TOOL_TYPES = frozenset({"tool_call", "mcp_tool_call"})
66
+
67
+
68
+ def _as_str(value: JsonValue) -> str:
69
+ return value if isinstance(value, str) else ""
70
+
71
+
72
+ def _span_type(span: JsonObject) -> str:
73
+ for key in ("spanType", "span_type", "type"):
74
+ value = span.get(key)
75
+ if isinstance(value, str) and value:
76
+ return value.lower()
77
+ return ""
78
+
79
+
80
+ def _start_ordinal(span: JsonObject, fallback: int) -> int:
81
+ """Monotonic ordering key from the span's start time (shared helper; UTC-safe)."""
82
+ for key in ("startTime", "startedAt", "start_time"):
83
+ value = span.get(key)
84
+ if value is not None:
85
+ return iso_to_ordinal(value, fallback)
86
+ return fallback
87
+
88
+
89
+ def _is_error(span: JsonObject) -> bool:
90
+ """A non-empty `errorInfo`/`error`, or an error status, marks the span as failed."""
91
+ for key in ("errorInfo", "error"):
92
+ value = span.get(key)
93
+ if isinstance(value, str) and value.strip():
94
+ return True
95
+ if isinstance(value, dict) and value:
96
+ return True
97
+ status = span.get("status")
98
+ return isinstance(status, str) and status.upper() in {"ERROR", "FAILED"}
99
+
100
+
101
+ def _tool_calls(output: JsonValue) -> list[JsonObject]:
102
+ """Extract tool calls from an llm span `output` (AI SDK `toolCalls` or OpenAI `tool_calls`)."""
103
+ calls: list[JsonObject] = []
104
+ candidates: list[JsonValue] = [output]
105
+ if isinstance(output, dict):
106
+ # Mastra may nest the assistant message under output.message / output.response.
107
+ for key in ("message", "response"):
108
+ nested = output.get(key)
109
+ if nested is not None:
110
+ candidates.append(nested)
111
+ for candidate in candidates:
112
+ if isinstance(candidate, dict):
113
+ for key in ("toolCalls", "tool_calls"):
114
+ raw = candidate.get(key)
115
+ if isinstance(raw, list):
116
+ calls.extend(tc for tc in raw if isinstance(tc, dict))
117
+ elif isinstance(candidate, list):
118
+ for message in candidate:
119
+ if isinstance(message, dict):
120
+ for key in ("toolCalls", "tool_calls"):
121
+ raw = message.get(key)
122
+ if isinstance(raw, list):
123
+ calls.extend(tc for tc in raw if isinstance(tc, dict))
124
+ return calls
125
+
126
+
127
+ def _call_name_args(tool_call: JsonObject) -> tuple[str, str]:
128
+ """(name, raw-arguments-json) from a Mastra AI-SDK or OpenAI-shaped tool call."""
129
+ fn = tool_call.get("function")
130
+ if isinstance(fn, dict): # OpenAI shape: {"function": {"name", "arguments": "<json str>"}}
131
+ name = fn.get("name")
132
+ args = fn.get("arguments")
133
+ else: # AI SDK shape: v5 {"toolName", "input": {...}, "toolCallId"}; v4 used "args".
134
+ name = tool_call.get("toolName") or tool_call.get("name")
135
+ args = tool_call.get("input") # AI SDK v5
136
+ if args is None:
137
+ args = tool_call.get("args") # AI SDK v4
138
+ if args is None:
139
+ args = tool_call.get("arguments")
140
+ name_s = name if isinstance(name, str) else ""
141
+ args_s = args if isinstance(args, str) else as_text(args)
142
+ return name_s, args_s
143
+
144
+
145
+ def _tool_span_name(span: JsonObject) -> str:
146
+ """A tool name for a tool_call span: explicit attribute first, else the span name."""
147
+ attributes = span.get("attributes")
148
+ if isinstance(attributes, dict):
149
+ for key in ("toolName", "tool_name"):
150
+ value = attributes.get(key)
151
+ if isinstance(value, str) and value:
152
+ return value
153
+ name = span.get("name")
154
+ return name if isinstance(name, str) else ""
155
+
156
+
157
+ def _first_user_text(value: JsonValue) -> str | None:
158
+ """First user message text from a span `input` (a messages list), else the input as text."""
159
+ if isinstance(value, list):
160
+ for message in value:
161
+ if isinstance(message, dict) and _as_str(message.get("role")).lower() in {
162
+ "user",
163
+ "human",
164
+ }:
165
+ content = message.get("content")
166
+ if content is not None:
167
+ return as_text(content)
168
+ return None
169
+ if isinstance(value, dict):
170
+ for key in ("prompt", "input", "query", "message"):
171
+ inner = value.get(key)
172
+ if isinstance(inner, str) and inner:
173
+ return inner
174
+ return None
175
+ if isinstance(value, str) and value:
176
+ return value
177
+ return None
178
+
179
+
180
+ def _completion_text(output: JsonValue) -> str:
181
+ """Render an llm span `output` to text: an assistant message content, else the whole output."""
182
+ if isinstance(output, dict):
183
+ for key in ("text", "content"):
184
+ value = output.get(key)
185
+ if isinstance(value, str) and value:
186
+ return value
187
+ return as_text(output)
188
+
189
+
190
+ class MastraAdapter(BaseTraceAdapter):
191
+ """Map a Mastra AI-tracing export into normalized `Trace`s. No SDK."""
192
+
193
+ name = "mastra"
194
+
195
+ def spans_from_payload(self, payload: JsonValue) -> list[SpanRecord]:
196
+ raw_spans = self._spans(payload)
197
+ by_trace: dict[str, list[JsonObject]] = {}
198
+ for span in raw_spans:
199
+ by_trace.setdefault(self._trace_id(span), []).append(span)
200
+ spans: list[SpanRecord] = []
201
+ for trace_id, trace_spans in by_trace.items():
202
+ spans.extend(self._spans_for_trace(trace_id, trace_spans))
203
+ return spans
204
+
205
+ def _spans(self, payload: JsonValue) -> list[JsonObject]:
206
+ """Normalize a payload into a flat list of Mastra span objects.
207
+
208
+ Accepts a single span, a bare list, or a wrapper (`{"spans"|"traces"|"data": [...]}`). A
209
+ wrapper item may itself be a trace object holding its own `spans`, so we recurse.
210
+ """
211
+ if isinstance(payload, list):
212
+ out: list[JsonObject] = []
213
+ for item in payload:
214
+ out.extend(self._spans(item))
215
+ return out
216
+ if not isinstance(payload, dict):
217
+ return []
218
+ for wrapper_key in ("spans", "traces", "data"):
219
+ inner = payload.get(wrapper_key)
220
+ if isinstance(inner, list):
221
+ out = []
222
+ for item in inner:
223
+ out.extend(self._spans(item))
224
+ return out
225
+ # A bare span carries a trace id and a type (Mastra's span id field is `id`).
226
+ if any(key in payload for key in ("traceId", "type", "spanType", "id", "spanId")):
227
+ return [payload]
228
+ return []
229
+
230
+ def _trace_id(self, span: JsonObject) -> str:
231
+ for key in ("traceId", "trace_id"):
232
+ value = span.get(key)
233
+ if isinstance(value, str) and value:
234
+ return value
235
+ sid = span.get("id") or span.get("spanId") or span.get("span_id")
236
+ if isinstance(sid, str) and sid:
237
+ return sid
238
+ import hashlib
239
+
240
+ return hashlib.sha256(as_text(span).encode()).hexdigest()[:32]
241
+
242
+ def _spans_for_trace(self, trace_id: str, raw_spans: list[JsonObject]) -> list[SpanRecord]:
243
+ indexed = list(enumerate(raw_spans))
244
+ indexed.sort(key=lambda pair: (_start_ordinal(pair[1], pair[0]), pair[0]))
245
+
246
+ task: str | None = None
247
+ for _, span in indexed:
248
+ task = _first_user_text(span.get("input"))
249
+ if task is not None:
250
+ break
251
+
252
+ spans: list[SpanRecord] = []
253
+ ordinal = 0
254
+
255
+ def emit(attrs: JsonObject, *, tool: bool, error: bool = False) -> None:
256
+ nonlocal ordinal
257
+ if ordinal == 0 and task is not None:
258
+ attrs.setdefault("gen_ai.prompt", task)
259
+ spans.append(
260
+ SpanRecord(
261
+ trace_id=trace_id,
262
+ span_id=f"{trace_id[:12]}{ordinal:06x}{'t' if tool else 'a'}",
263
+ name="execute_tool" if tool else "chat",
264
+ start_nano=ordinal,
265
+ attributes={
266
+ "gen_ai.operation.name": "execute_tool" if tool else "chat",
267
+ **attrs,
268
+ },
269
+ status_error=error,
270
+ )
271
+ )
272
+ ordinal += 1
273
+
274
+ for _, span in indexed:
275
+ stype = _span_type(span)
276
+ error = _is_error(span)
277
+ if stype in _LLM_TYPES:
278
+ calls = _tool_calls(span.get("output"))
279
+ if calls:
280
+ for tool_call in calls:
281
+ name, args = _call_name_args(tool_call)
282
+ emit(
283
+ {"gen_ai.tool.name": name, "gen_ai.tool.call.arguments": args},
284
+ tool=False,
285
+ error=error,
286
+ )
287
+ else:
288
+ emit(
289
+ {"gen_ai.completion": _completion_text(span.get("output"))},
290
+ tool=False,
291
+ error=error,
292
+ )
293
+ elif stype in _TOOL_TYPES:
294
+ emit(
295
+ {
296
+ "gen_ai.tool.name": _tool_span_name(span),
297
+ "gen_ai.tool.call.arguments": as_text(span.get("input")),
298
+ "gen_ai.tool.message": as_text(span.get("output")),
299
+ },
300
+ tool=True,
301
+ error=error,
302
+ )
303
+ # agent_run / workflow_* / llm_chunk / generic -> no standalone step.
304
+ return spans
305
+
306
+ def _pull_payloads(self, pull: VendorPull) -> list[JsonValue]:
307
+ """Fetch AI-tracing spans from a running Mastra server's observability API.
308
+
309
+ `pull.project` (else `$MASTRA_URL`) is the Mastra server base URL, e.g.
310
+ `http://localhost:4111`. Fetches `{base}/api/observability/traces` and hands the response to
311
+ the flexible span extractor (which tolerates the server's `{traces|spans: [...]}` wrapper).
312
+ """
313
+ base = (pull.project or os.environ.get(_MASTRA_URL_ENV) or "").rstrip("/")
314
+ if not base:
315
+ raise ValueError(
316
+ f"mastra pull needs the server URL: pass --project <base-url> or set "
317
+ f"${_MASTRA_URL_ENV}"
318
+ )
319
+ headers = {"Authorization": f"Bearer {pull.api_key}"} if pull.api_key else {}
320
+ params: dict[str, str] = {}
321
+ if pull.limit is not None:
322
+ params["perPage"] = str(pull.limit)
323
+ resp = httpx.get(
324
+ f"{base}/api/observability/traces", headers=headers, params=params, timeout=60.0
325
+ )
326
+ resp.raise_for_status()
327
+ return [resp.json()]
328
+
329
+
330
+ register_adapter(MastraAdapter())
wmo/ingest/messages.py ADDED
@@ -0,0 +1,170 @@
1
+ """Chat / tool-call converter: turn recorded LLM conversations into `Trace`s (no SDK, no spans).
2
+
3
+ Not every source is a span exporter. The most universal trace people already have is a list of chat
4
+ messages with tool calls — the OpenAI Chat Completions shape, which LangChain, the Anthropic SDK
5
+ (after a light dump), and most agent frameworks can emit:
6
+
7
+ {"messages": [
8
+ {"role": "user", "content": "what's the weather in Paris?"},
9
+ {"role": "assistant", "content": "let me check",
10
+ "tool_calls": [{"id": "c1", "function": {"name": "get_weather",
11
+ "arguments": "{\"city\": \"Paris\"}"}}]},
12
+ {"role": "tool", "tool_call_id": "c1", "content": "18C and sunny"},
13
+ {"role": "assistant", "content": "It's 18C and sunny in Paris."}
14
+ ]}
15
+
16
+ This adapter maps each assistant **tool call** to an Action and the matching `role:"tool"` message
17
+ (by `tool_call_id`, else the next tool message in order) to its Observation — exactly the
18
+ `(action) -> observation` step the harness scores. A trailing assistant message with no tool call
19
+ becomes a final message Step with an empty observation. The first user message is the trace `task`.
20
+
21
+ It builds `SpanRecord`s in the OTel-GenAI vocabulary and hands them to the shared normalizer, so it
22
+ reuses the same pairing/state/metadata logic as every other adapter rather than re-implementing it.
23
+
24
+ Accepted file shapes (`from_file`):
25
+ - a single conversation object `{"messages": [...]}` (optionally `{"id"/"trace_id", "metadata"}`)
26
+ - a JSON array of such conversation objects
27
+ - JSONL: one conversation object per line
28
+ - a bare list of messages `[{"role": ...}, ...]` (treated as one conversation)
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import json
34
+
35
+ from pydantic import JsonValue
36
+
37
+ from wmo.core.types import JsonObject
38
+ from wmo.ingest.adapter import VendorPull, register_adapter
39
+ from wmo.ingest.base import BaseTraceAdapter
40
+ from wmo.ingest.normalize import SpanRecord, as_text, openai_call_name_args
41
+
42
+
43
+ def _hash_id(*parts: str) -> str:
44
+ import hashlib
45
+
46
+ return hashlib.sha256("|".join(parts).encode()).hexdigest()[:32]
47
+
48
+
49
+ def _tool_calls(message: JsonObject) -> list[JsonObject]:
50
+ raw = message.get("tool_calls")
51
+ if not isinstance(raw, list):
52
+ return []
53
+ return [tc for tc in raw if isinstance(tc, dict)]
54
+
55
+
56
+ def _spans_for_conversation(
57
+ messages: list[JsonValue], trace_id: str, metadata: JsonObject
58
+ ) -> list[SpanRecord]:
59
+ """Build ordered action/observation SpanRecords (GenAI vocab) for one conversation."""
60
+ # Index tool results by tool_call_id; fall back to consumption in order for results lacking one.
61
+ results_by_id: dict[str, str] = {}
62
+ ordered_results: list[str] = []
63
+ task: str | None = None
64
+ for m in messages:
65
+ if not isinstance(m, dict):
66
+ continue
67
+ role = m.get("role")
68
+ if role == "user" and task is None:
69
+ task = as_text(m.get("content"))
70
+ if role == "tool":
71
+ content = as_text(m.get("content"))
72
+ tcid = m.get("tool_call_id")
73
+ if isinstance(tcid, str):
74
+ results_by_id[tcid] = content
75
+ ordered_results.append(content)
76
+
77
+ spans: list[SpanRecord] = []
78
+ ordinal = 0
79
+ unmatched = list(ordered_results)
80
+
81
+ def _emit(attrs: JsonObject, *, tool: bool, error: bool = False) -> None:
82
+ nonlocal ordinal
83
+ if ordinal == 0 and task is not None:
84
+ attrs.setdefault("gen_ai.prompt", task)
85
+ if ordinal == 0 and metadata:
86
+ attrs.setdefault("wmo.trace.metadata", json.dumps(metadata))
87
+ spans.append(
88
+ SpanRecord(
89
+ trace_id=trace_id,
90
+ span_id=f"{trace_id[:12]}{ordinal:06x}{'t' if tool else 'a'}",
91
+ name="execute_tool" if tool else "chat",
92
+ start_nano=ordinal,
93
+ attributes={"gen_ai.operation.name": "execute_tool" if tool else "chat", **attrs},
94
+ status_error=error,
95
+ )
96
+ )
97
+ ordinal += 1
98
+
99
+ for m in messages:
100
+ if not isinstance(m, dict) or m.get("role") != "assistant":
101
+ continue
102
+ calls = _tool_calls(m)
103
+ if calls:
104
+ for tc in calls:
105
+ name, args = openai_call_name_args(tc)
106
+ _emit({"gen_ai.tool.name": name, "gen_ai.tool.call.arguments": args}, tool=False)
107
+ tcid = tc.get("id")
108
+ if isinstance(tcid, str) and tcid in results_by_id:
109
+ result = results_by_id[tcid]
110
+ if result in unmatched:
111
+ unmatched.remove(result)
112
+ elif unmatched:
113
+ result = unmatched.pop(0)
114
+ else:
115
+ result = ""
116
+ _emit({"gen_ai.tool.message": result}, tool=True)
117
+ else:
118
+ # A plain assistant message turn (e.g. the final answer): a message Action, no tool obs.
119
+ content = m.get("content")
120
+ if content is not None:
121
+ _emit({"gen_ai.completion": as_text(content)}, tool=False)
122
+ return spans
123
+
124
+
125
+ def _conversation_records(payload: JsonValue) -> list[tuple[str, list[JsonValue], JsonObject]]:
126
+ """Normalize a payload into a list of (trace_id, messages, metadata) conversations."""
127
+ out: list[tuple[str, list[JsonValue], JsonObject]] = []
128
+ if isinstance(payload, list):
129
+ # Either a list of conversation objects, or a bare message list (one conversation).
130
+ if payload and all(isinstance(x, dict) and "role" in x for x in payload):
131
+ out.append((_hash_id(as_text(payload)), payload, {}))
132
+ return out
133
+ for item in payload:
134
+ out.extend(_conversation_records(item))
135
+ return out
136
+ if not isinstance(payload, dict):
137
+ return out
138
+ messages = payload.get("messages")
139
+ if not isinstance(messages, list):
140
+ return out
141
+ tid = payload.get("trace_id") or payload.get("id")
142
+ trace_id = tid if isinstance(tid, str) and tid else _hash_id(as_text(messages))
143
+ # 32-hex normalize so trace ids are uniform regardless of source id format.
144
+ if len(trace_id) != 32 or any(c not in "0123456789abcdef" for c in trace_id.lower()):
145
+ trace_id = _hash_id(trace_id)
146
+ meta = payload.get("metadata")
147
+ metadata: JsonObject = meta if isinstance(meta, dict) else {}
148
+ out.append((trace_id, messages, metadata))
149
+ return out
150
+
151
+
152
+ class ChatMessagesAdapter(BaseTraceAdapter):
153
+ """Convert recorded chat/tool-call conversations (OpenAI-style) into `Trace`s. No SDK."""
154
+
155
+ name = "chat-json"
156
+
157
+ def spans_from_payload(self, payload: JsonValue) -> list[SpanRecord]:
158
+ spans: list[SpanRecord] = []
159
+ for trace_id, messages, metadata in _conversation_records(payload):
160
+ spans.extend(_spans_for_conversation(messages, trace_id, metadata))
161
+ return spans
162
+
163
+ def _pull_payloads(self, pull: VendorPull) -> list[JsonValue]:
164
+ raise ValueError(
165
+ "chat-json converts local conversation files; it has no vendor API. "
166
+ "Use `from_file` with an exported messages JSON/JSONL."
167
+ )
168
+
169
+
170
+ register_adapter(ChatMessagesAdapter())