world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/ingest/mastra.py
ADDED
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
"""Mastra adapter: turn a Mastra AI-tracing export into `Trace`s.
|
|
2
|
+
|
|
3
|
+
[Mastra](https://mastra.ai) (a TypeScript agent framework) records agent runs as **AI-tracing
|
|
4
|
+
spans** (`ExportedSpan`), typed by `type`, that share a `traceId`. The span id field is `id`, times
|
|
5
|
+
are `startTime`/`endTime`, and errors ride on a structured `errorInfo` object:
|
|
6
|
+
|
|
7
|
+
{"traceId": "t1", "id": "s2", "parentSpanId": "s1",
|
|
8
|
+
"name": "modelGeneration", "type": "model_generation",
|
|
9
|
+
"input": [{"role": "user", "content": "what's the weather in Paris?"}],
|
|
10
|
+
"output": {"role": "assistant",
|
|
11
|
+
"toolCalls": [{"toolCallId": "c1", "toolName": "getWeather",
|
|
12
|
+
"input": {"city": "Paris"}}]}, # AI SDK v5 uses `input` (v4: `args`)
|
|
13
|
+
"attributes": {"model": "gpt-4o"}, "startTime": "2026-01-01T00:00:01.000Z"}
|
|
14
|
+
{"traceId": "t1", "id": "s3", "name": "getWeather", "type": "tool_call",
|
|
15
|
+
"input": {"city": "Paris"}, "output": "18C and sunny", "startTime": "..."}
|
|
16
|
+
|
|
17
|
+
Because this is not an OTLP/OpenInference span shape, the adapter overrides `spans_from_payload`
|
|
18
|
+
(like `wmo.ingest.langfuse`) and emits `SpanRecord`s in the **OTel-GenAI vocabulary** so the shared
|
|
19
|
+
classifier/normalizer (`wmo.ingest.normalize`) does the pairing/state/metadata work:
|
|
20
|
+
|
|
21
|
+
- a `model_generation` span (or the pre-rename `llm_generation`) whose `output` carries tool calls
|
|
22
|
+
-> one `chat` action span per call (`gen_ai.tool.name` + `gen_ai.tool.call.arguments`). Mastra
|
|
23
|
+
exposes tool calls as either the AI SDK `toolCalls` (`{toolCallId, toolName, input}` on v5;
|
|
24
|
+
`args` on v4) or the OpenAI `tool_calls` (`{id, function:{name, arguments}}`); both are handled.
|
|
25
|
+
(Tool calls may instead appear only as separate `tool_call` spans — that case is covered below.)
|
|
26
|
+
- a `tool_call` / `mcp_tool_call` span -> an `execute_tool` result span (`gen_ai.tool.message`
|
|
27
|
+
from `output`), also carrying name/args (from `name`/`input`) so a standalone tool span still
|
|
28
|
+
pairs; the normalizer backfills the action's name/args from here if the model span lacked them.
|
|
29
|
+
- a `model_generation` with no tool call -> a plain `chat` message span (`gen_ai.completion`).
|
|
30
|
+
- `agent_run` / `workflow_*` / `model_chunk` / `model_step` / `generic` container/noise spans are
|
|
31
|
+
skipped (no `(action) -> observation` step); an `agent_run`/`model_generation` input supplies
|
|
32
|
+
the task.
|
|
33
|
+
- a span with an `errorInfo` (or an error status) sets `status_error=True`.
|
|
34
|
+
|
|
35
|
+
Spans order by `startTime`/`startedAt` (ISO-8601 or datetime -> a monotonic ordinal, index if none).
|
|
36
|
+
|
|
37
|
+
Accepted file shapes (`from_file`): a single span, a JSON array of spans, a wrapper
|
|
38
|
+
(`{"spans": [...]}` / `{"traces": [...]}` / `{"data": [...]}`), or JSONL. Grouping is by `traceId`.
|
|
39
|
+
|
|
40
|
+
Pull: `_pull_payloads` fetches from a running Mastra server's observability API
|
|
41
|
+
(`{base}/api/observability/traces`), with the base URL passed as `--project` (or `$MASTRA_URL`). The
|
|
42
|
+
response is handed to the same flexible span extractor, so it tolerates the server's wrapper shape.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import os
|
|
48
|
+
|
|
49
|
+
import httpx
|
|
50
|
+
from pydantic import JsonValue
|
|
51
|
+
|
|
52
|
+
from wmo.core.types import JsonObject
|
|
53
|
+
from wmo.ingest.adapter import VendorPull, register_adapter
|
|
54
|
+
from wmo.ingest.base import BaseTraceAdapter
|
|
55
|
+
from wmo.ingest.normalize import SpanRecord, as_text, iso_to_ordinal
|
|
56
|
+
|
|
57
|
+
# Mastra self-hosts, so the "vendor" is a server base URL (dev default http://localhost:4111).
|
|
58
|
+
_MASTRA_URL_ENV = "MASTRA_URL"
|
|
59
|
+
|
|
60
|
+
# `type` values (normalized lowercase). LLM/agent turns vs tool executions; the rest are
|
|
61
|
+
# containers/noise that carry no standalone step. Mastra renamed LLM spans to "model" spans
|
|
62
|
+
# (changelog 2025-11-01): `model_generation` is current, `llm_generation` is the pre-rename name we
|
|
63
|
+
# still accept. `model_chunk`/`model_step`/`agent_run`/`workflow_*`/`generic` are not steps.
|
|
64
|
+
_LLM_TYPES = frozenset({"model_generation", "llm_generation"})
|
|
65
|
+
_TOOL_TYPES = frozenset({"tool_call", "mcp_tool_call"})
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _as_str(value: JsonValue) -> str:
|
|
69
|
+
return value if isinstance(value, str) else ""
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _span_type(span: JsonObject) -> str:
|
|
73
|
+
for key in ("spanType", "span_type", "type"):
|
|
74
|
+
value = span.get(key)
|
|
75
|
+
if isinstance(value, str) and value:
|
|
76
|
+
return value.lower()
|
|
77
|
+
return ""
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _start_ordinal(span: JsonObject, fallback: int) -> int:
|
|
81
|
+
"""Monotonic ordering key from the span's start time (shared helper; UTC-safe)."""
|
|
82
|
+
for key in ("startTime", "startedAt", "start_time"):
|
|
83
|
+
value = span.get(key)
|
|
84
|
+
if value is not None:
|
|
85
|
+
return iso_to_ordinal(value, fallback)
|
|
86
|
+
return fallback
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _is_error(span: JsonObject) -> bool:
|
|
90
|
+
"""A non-empty `errorInfo`/`error`, or an error status, marks the span as failed."""
|
|
91
|
+
for key in ("errorInfo", "error"):
|
|
92
|
+
value = span.get(key)
|
|
93
|
+
if isinstance(value, str) and value.strip():
|
|
94
|
+
return True
|
|
95
|
+
if isinstance(value, dict) and value:
|
|
96
|
+
return True
|
|
97
|
+
status = span.get("status")
|
|
98
|
+
return isinstance(status, str) and status.upper() in {"ERROR", "FAILED"}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _tool_calls(output: JsonValue) -> list[JsonObject]:
|
|
102
|
+
"""Extract tool calls from an llm span `output` (AI SDK `toolCalls` or OpenAI `tool_calls`)."""
|
|
103
|
+
calls: list[JsonObject] = []
|
|
104
|
+
candidates: list[JsonValue] = [output]
|
|
105
|
+
if isinstance(output, dict):
|
|
106
|
+
# Mastra may nest the assistant message under output.message / output.response.
|
|
107
|
+
for key in ("message", "response"):
|
|
108
|
+
nested = output.get(key)
|
|
109
|
+
if nested is not None:
|
|
110
|
+
candidates.append(nested)
|
|
111
|
+
for candidate in candidates:
|
|
112
|
+
if isinstance(candidate, dict):
|
|
113
|
+
for key in ("toolCalls", "tool_calls"):
|
|
114
|
+
raw = candidate.get(key)
|
|
115
|
+
if isinstance(raw, list):
|
|
116
|
+
calls.extend(tc for tc in raw if isinstance(tc, dict))
|
|
117
|
+
elif isinstance(candidate, list):
|
|
118
|
+
for message in candidate:
|
|
119
|
+
if isinstance(message, dict):
|
|
120
|
+
for key in ("toolCalls", "tool_calls"):
|
|
121
|
+
raw = message.get(key)
|
|
122
|
+
if isinstance(raw, list):
|
|
123
|
+
calls.extend(tc for tc in raw if isinstance(tc, dict))
|
|
124
|
+
return calls
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _call_name_args(tool_call: JsonObject) -> tuple[str, str]:
|
|
128
|
+
"""(name, raw-arguments-json) from a Mastra AI-SDK or OpenAI-shaped tool call."""
|
|
129
|
+
fn = tool_call.get("function")
|
|
130
|
+
if isinstance(fn, dict): # OpenAI shape: {"function": {"name", "arguments": "<json str>"}}
|
|
131
|
+
name = fn.get("name")
|
|
132
|
+
args = fn.get("arguments")
|
|
133
|
+
else: # AI SDK shape: v5 {"toolName", "input": {...}, "toolCallId"}; v4 used "args".
|
|
134
|
+
name = tool_call.get("toolName") or tool_call.get("name")
|
|
135
|
+
args = tool_call.get("input") # AI SDK v5
|
|
136
|
+
if args is None:
|
|
137
|
+
args = tool_call.get("args") # AI SDK v4
|
|
138
|
+
if args is None:
|
|
139
|
+
args = tool_call.get("arguments")
|
|
140
|
+
name_s = name if isinstance(name, str) else ""
|
|
141
|
+
args_s = args if isinstance(args, str) else as_text(args)
|
|
142
|
+
return name_s, args_s
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _tool_span_name(span: JsonObject) -> str:
|
|
146
|
+
"""A tool name for a tool_call span: explicit attribute first, else the span name."""
|
|
147
|
+
attributes = span.get("attributes")
|
|
148
|
+
if isinstance(attributes, dict):
|
|
149
|
+
for key in ("toolName", "tool_name"):
|
|
150
|
+
value = attributes.get(key)
|
|
151
|
+
if isinstance(value, str) and value:
|
|
152
|
+
return value
|
|
153
|
+
name = span.get("name")
|
|
154
|
+
return name if isinstance(name, str) else ""
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _first_user_text(value: JsonValue) -> str | None:
|
|
158
|
+
"""First user message text from a span `input` (a messages list), else the input as text."""
|
|
159
|
+
if isinstance(value, list):
|
|
160
|
+
for message in value:
|
|
161
|
+
if isinstance(message, dict) and _as_str(message.get("role")).lower() in {
|
|
162
|
+
"user",
|
|
163
|
+
"human",
|
|
164
|
+
}:
|
|
165
|
+
content = message.get("content")
|
|
166
|
+
if content is not None:
|
|
167
|
+
return as_text(content)
|
|
168
|
+
return None
|
|
169
|
+
if isinstance(value, dict):
|
|
170
|
+
for key in ("prompt", "input", "query", "message"):
|
|
171
|
+
inner = value.get(key)
|
|
172
|
+
if isinstance(inner, str) and inner:
|
|
173
|
+
return inner
|
|
174
|
+
return None
|
|
175
|
+
if isinstance(value, str) and value:
|
|
176
|
+
return value
|
|
177
|
+
return None
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _completion_text(output: JsonValue) -> str:
|
|
181
|
+
"""Render an llm span `output` to text: an assistant message content, else the whole output."""
|
|
182
|
+
if isinstance(output, dict):
|
|
183
|
+
for key in ("text", "content"):
|
|
184
|
+
value = output.get(key)
|
|
185
|
+
if isinstance(value, str) and value:
|
|
186
|
+
return value
|
|
187
|
+
return as_text(output)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
class MastraAdapter(BaseTraceAdapter):
|
|
191
|
+
"""Map a Mastra AI-tracing export into normalized `Trace`s. No SDK."""
|
|
192
|
+
|
|
193
|
+
name = "mastra"
|
|
194
|
+
|
|
195
|
+
def spans_from_payload(self, payload: JsonValue) -> list[SpanRecord]:
|
|
196
|
+
raw_spans = self._spans(payload)
|
|
197
|
+
by_trace: dict[str, list[JsonObject]] = {}
|
|
198
|
+
for span in raw_spans:
|
|
199
|
+
by_trace.setdefault(self._trace_id(span), []).append(span)
|
|
200
|
+
spans: list[SpanRecord] = []
|
|
201
|
+
for trace_id, trace_spans in by_trace.items():
|
|
202
|
+
spans.extend(self._spans_for_trace(trace_id, trace_spans))
|
|
203
|
+
return spans
|
|
204
|
+
|
|
205
|
+
def _spans(self, payload: JsonValue) -> list[JsonObject]:
|
|
206
|
+
"""Normalize a payload into a flat list of Mastra span objects.
|
|
207
|
+
|
|
208
|
+
Accepts a single span, a bare list, or a wrapper (`{"spans"|"traces"|"data": [...]}`). A
|
|
209
|
+
wrapper item may itself be a trace object holding its own `spans`, so we recurse.
|
|
210
|
+
"""
|
|
211
|
+
if isinstance(payload, list):
|
|
212
|
+
out: list[JsonObject] = []
|
|
213
|
+
for item in payload:
|
|
214
|
+
out.extend(self._spans(item))
|
|
215
|
+
return out
|
|
216
|
+
if not isinstance(payload, dict):
|
|
217
|
+
return []
|
|
218
|
+
for wrapper_key in ("spans", "traces", "data"):
|
|
219
|
+
inner = payload.get(wrapper_key)
|
|
220
|
+
if isinstance(inner, list):
|
|
221
|
+
out = []
|
|
222
|
+
for item in inner:
|
|
223
|
+
out.extend(self._spans(item))
|
|
224
|
+
return out
|
|
225
|
+
# A bare span carries a trace id and a type (Mastra's span id field is `id`).
|
|
226
|
+
if any(key in payload for key in ("traceId", "type", "spanType", "id", "spanId")):
|
|
227
|
+
return [payload]
|
|
228
|
+
return []
|
|
229
|
+
|
|
230
|
+
def _trace_id(self, span: JsonObject) -> str:
|
|
231
|
+
for key in ("traceId", "trace_id"):
|
|
232
|
+
value = span.get(key)
|
|
233
|
+
if isinstance(value, str) and value:
|
|
234
|
+
return value
|
|
235
|
+
sid = span.get("id") or span.get("spanId") or span.get("span_id")
|
|
236
|
+
if isinstance(sid, str) and sid:
|
|
237
|
+
return sid
|
|
238
|
+
import hashlib
|
|
239
|
+
|
|
240
|
+
return hashlib.sha256(as_text(span).encode()).hexdigest()[:32]
|
|
241
|
+
|
|
242
|
+
def _spans_for_trace(self, trace_id: str, raw_spans: list[JsonObject]) -> list[SpanRecord]:
|
|
243
|
+
indexed = list(enumerate(raw_spans))
|
|
244
|
+
indexed.sort(key=lambda pair: (_start_ordinal(pair[1], pair[0]), pair[0]))
|
|
245
|
+
|
|
246
|
+
task: str | None = None
|
|
247
|
+
for _, span in indexed:
|
|
248
|
+
task = _first_user_text(span.get("input"))
|
|
249
|
+
if task is not None:
|
|
250
|
+
break
|
|
251
|
+
|
|
252
|
+
spans: list[SpanRecord] = []
|
|
253
|
+
ordinal = 0
|
|
254
|
+
|
|
255
|
+
def emit(attrs: JsonObject, *, tool: bool, error: bool = False) -> None:
|
|
256
|
+
nonlocal ordinal
|
|
257
|
+
if ordinal == 0 and task is not None:
|
|
258
|
+
attrs.setdefault("gen_ai.prompt", task)
|
|
259
|
+
spans.append(
|
|
260
|
+
SpanRecord(
|
|
261
|
+
trace_id=trace_id,
|
|
262
|
+
span_id=f"{trace_id[:12]}{ordinal:06x}{'t' if tool else 'a'}",
|
|
263
|
+
name="execute_tool" if tool else "chat",
|
|
264
|
+
start_nano=ordinal,
|
|
265
|
+
attributes={
|
|
266
|
+
"gen_ai.operation.name": "execute_tool" if tool else "chat",
|
|
267
|
+
**attrs,
|
|
268
|
+
},
|
|
269
|
+
status_error=error,
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
ordinal += 1
|
|
273
|
+
|
|
274
|
+
for _, span in indexed:
|
|
275
|
+
stype = _span_type(span)
|
|
276
|
+
error = _is_error(span)
|
|
277
|
+
if stype in _LLM_TYPES:
|
|
278
|
+
calls = _tool_calls(span.get("output"))
|
|
279
|
+
if calls:
|
|
280
|
+
for tool_call in calls:
|
|
281
|
+
name, args = _call_name_args(tool_call)
|
|
282
|
+
emit(
|
|
283
|
+
{"gen_ai.tool.name": name, "gen_ai.tool.call.arguments": args},
|
|
284
|
+
tool=False,
|
|
285
|
+
error=error,
|
|
286
|
+
)
|
|
287
|
+
else:
|
|
288
|
+
emit(
|
|
289
|
+
{"gen_ai.completion": _completion_text(span.get("output"))},
|
|
290
|
+
tool=False,
|
|
291
|
+
error=error,
|
|
292
|
+
)
|
|
293
|
+
elif stype in _TOOL_TYPES:
|
|
294
|
+
emit(
|
|
295
|
+
{
|
|
296
|
+
"gen_ai.tool.name": _tool_span_name(span),
|
|
297
|
+
"gen_ai.tool.call.arguments": as_text(span.get("input")),
|
|
298
|
+
"gen_ai.tool.message": as_text(span.get("output")),
|
|
299
|
+
},
|
|
300
|
+
tool=True,
|
|
301
|
+
error=error,
|
|
302
|
+
)
|
|
303
|
+
# agent_run / workflow_* / llm_chunk / generic -> no standalone step.
|
|
304
|
+
return spans
|
|
305
|
+
|
|
306
|
+
def _pull_payloads(self, pull: VendorPull) -> list[JsonValue]:
|
|
307
|
+
"""Fetch AI-tracing spans from a running Mastra server's observability API.
|
|
308
|
+
|
|
309
|
+
`pull.project` (else `$MASTRA_URL`) is the Mastra server base URL, e.g.
|
|
310
|
+
`http://localhost:4111`. Fetches `{base}/api/observability/traces` and hands the response to
|
|
311
|
+
the flexible span extractor (which tolerates the server's `{traces|spans: [...]}` wrapper).
|
|
312
|
+
"""
|
|
313
|
+
base = (pull.project or os.environ.get(_MASTRA_URL_ENV) or "").rstrip("/")
|
|
314
|
+
if not base:
|
|
315
|
+
raise ValueError(
|
|
316
|
+
f"mastra pull needs the server URL: pass --project <base-url> or set "
|
|
317
|
+
f"${_MASTRA_URL_ENV}"
|
|
318
|
+
)
|
|
319
|
+
headers = {"Authorization": f"Bearer {pull.api_key}"} if pull.api_key else {}
|
|
320
|
+
params: dict[str, str] = {}
|
|
321
|
+
if pull.limit is not None:
|
|
322
|
+
params["perPage"] = str(pull.limit)
|
|
323
|
+
resp = httpx.get(
|
|
324
|
+
f"{base}/api/observability/traces", headers=headers, params=params, timeout=60.0
|
|
325
|
+
)
|
|
326
|
+
resp.raise_for_status()
|
|
327
|
+
return [resp.json()]
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
register_adapter(MastraAdapter())
|
wmo/ingest/messages.py
ADDED
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""Chat / tool-call converter: turn recorded LLM conversations into `Trace`s (no SDK, no spans).
|
|
2
|
+
|
|
3
|
+
Not every source is a span exporter. The most universal trace people already have is a list of chat
|
|
4
|
+
messages with tool calls — the OpenAI Chat Completions shape, which LangChain, the Anthropic SDK
|
|
5
|
+
(after a light dump), and most agent frameworks can emit:
|
|
6
|
+
|
|
7
|
+
{"messages": [
|
|
8
|
+
{"role": "user", "content": "what's the weather in Paris?"},
|
|
9
|
+
{"role": "assistant", "content": "let me check",
|
|
10
|
+
"tool_calls": [{"id": "c1", "function": {"name": "get_weather",
|
|
11
|
+
"arguments": "{\"city\": \"Paris\"}"}}]},
|
|
12
|
+
{"role": "tool", "tool_call_id": "c1", "content": "18C and sunny"},
|
|
13
|
+
{"role": "assistant", "content": "It's 18C and sunny in Paris."}
|
|
14
|
+
]}
|
|
15
|
+
|
|
16
|
+
This adapter maps each assistant **tool call** to an Action and the matching `role:"tool"` message
|
|
17
|
+
(by `tool_call_id`, else the next tool message in order) to its Observation — exactly the
|
|
18
|
+
`(action) -> observation` step the harness scores. A trailing assistant message with no tool call
|
|
19
|
+
becomes a final message Step with an empty observation. The first user message is the trace `task`.
|
|
20
|
+
|
|
21
|
+
It builds `SpanRecord`s in the OTel-GenAI vocabulary and hands them to the shared normalizer, so it
|
|
22
|
+
reuses the same pairing/state/metadata logic as every other adapter rather than re-implementing it.
|
|
23
|
+
|
|
24
|
+
Accepted file shapes (`from_file`):
|
|
25
|
+
- a single conversation object `{"messages": [...]}` (optionally `{"id"/"trace_id", "metadata"}`)
|
|
26
|
+
- a JSON array of such conversation objects
|
|
27
|
+
- JSONL: one conversation object per line
|
|
28
|
+
- a bare list of messages `[{"role": ...}, ...]` (treated as one conversation)
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import json
|
|
34
|
+
|
|
35
|
+
from pydantic import JsonValue
|
|
36
|
+
|
|
37
|
+
from wmo.core.types import JsonObject
|
|
38
|
+
from wmo.ingest.adapter import VendorPull, register_adapter
|
|
39
|
+
from wmo.ingest.base import BaseTraceAdapter
|
|
40
|
+
from wmo.ingest.normalize import SpanRecord, as_text, openai_call_name_args
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _hash_id(*parts: str) -> str:
|
|
44
|
+
import hashlib
|
|
45
|
+
|
|
46
|
+
return hashlib.sha256("|".join(parts).encode()).hexdigest()[:32]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _tool_calls(message: JsonObject) -> list[JsonObject]:
|
|
50
|
+
raw = message.get("tool_calls")
|
|
51
|
+
if not isinstance(raw, list):
|
|
52
|
+
return []
|
|
53
|
+
return [tc for tc in raw if isinstance(tc, dict)]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _spans_for_conversation(
|
|
57
|
+
messages: list[JsonValue], trace_id: str, metadata: JsonObject
|
|
58
|
+
) -> list[SpanRecord]:
|
|
59
|
+
"""Build ordered action/observation SpanRecords (GenAI vocab) for one conversation."""
|
|
60
|
+
# Index tool results by tool_call_id; fall back to consumption in order for results lacking one.
|
|
61
|
+
results_by_id: dict[str, str] = {}
|
|
62
|
+
ordered_results: list[str] = []
|
|
63
|
+
task: str | None = None
|
|
64
|
+
for m in messages:
|
|
65
|
+
if not isinstance(m, dict):
|
|
66
|
+
continue
|
|
67
|
+
role = m.get("role")
|
|
68
|
+
if role == "user" and task is None:
|
|
69
|
+
task = as_text(m.get("content"))
|
|
70
|
+
if role == "tool":
|
|
71
|
+
content = as_text(m.get("content"))
|
|
72
|
+
tcid = m.get("tool_call_id")
|
|
73
|
+
if isinstance(tcid, str):
|
|
74
|
+
results_by_id[tcid] = content
|
|
75
|
+
ordered_results.append(content)
|
|
76
|
+
|
|
77
|
+
spans: list[SpanRecord] = []
|
|
78
|
+
ordinal = 0
|
|
79
|
+
unmatched = list(ordered_results)
|
|
80
|
+
|
|
81
|
+
def _emit(attrs: JsonObject, *, tool: bool, error: bool = False) -> None:
|
|
82
|
+
nonlocal ordinal
|
|
83
|
+
if ordinal == 0 and task is not None:
|
|
84
|
+
attrs.setdefault("gen_ai.prompt", task)
|
|
85
|
+
if ordinal == 0 and metadata:
|
|
86
|
+
attrs.setdefault("wmo.trace.metadata", json.dumps(metadata))
|
|
87
|
+
spans.append(
|
|
88
|
+
SpanRecord(
|
|
89
|
+
trace_id=trace_id,
|
|
90
|
+
span_id=f"{trace_id[:12]}{ordinal:06x}{'t' if tool else 'a'}",
|
|
91
|
+
name="execute_tool" if tool else "chat",
|
|
92
|
+
start_nano=ordinal,
|
|
93
|
+
attributes={"gen_ai.operation.name": "execute_tool" if tool else "chat", **attrs},
|
|
94
|
+
status_error=error,
|
|
95
|
+
)
|
|
96
|
+
)
|
|
97
|
+
ordinal += 1
|
|
98
|
+
|
|
99
|
+
for m in messages:
|
|
100
|
+
if not isinstance(m, dict) or m.get("role") != "assistant":
|
|
101
|
+
continue
|
|
102
|
+
calls = _tool_calls(m)
|
|
103
|
+
if calls:
|
|
104
|
+
for tc in calls:
|
|
105
|
+
name, args = openai_call_name_args(tc)
|
|
106
|
+
_emit({"gen_ai.tool.name": name, "gen_ai.tool.call.arguments": args}, tool=False)
|
|
107
|
+
tcid = tc.get("id")
|
|
108
|
+
if isinstance(tcid, str) and tcid in results_by_id:
|
|
109
|
+
result = results_by_id[tcid]
|
|
110
|
+
if result in unmatched:
|
|
111
|
+
unmatched.remove(result)
|
|
112
|
+
elif unmatched:
|
|
113
|
+
result = unmatched.pop(0)
|
|
114
|
+
else:
|
|
115
|
+
result = ""
|
|
116
|
+
_emit({"gen_ai.tool.message": result}, tool=True)
|
|
117
|
+
else:
|
|
118
|
+
# A plain assistant message turn (e.g. the final answer): a message Action, no tool obs.
|
|
119
|
+
content = m.get("content")
|
|
120
|
+
if content is not None:
|
|
121
|
+
_emit({"gen_ai.completion": as_text(content)}, tool=False)
|
|
122
|
+
return spans
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _conversation_records(payload: JsonValue) -> list[tuple[str, list[JsonValue], JsonObject]]:
|
|
126
|
+
"""Normalize a payload into a list of (trace_id, messages, metadata) conversations."""
|
|
127
|
+
out: list[tuple[str, list[JsonValue], JsonObject]] = []
|
|
128
|
+
if isinstance(payload, list):
|
|
129
|
+
# Either a list of conversation objects, or a bare message list (one conversation).
|
|
130
|
+
if payload and all(isinstance(x, dict) and "role" in x for x in payload):
|
|
131
|
+
out.append((_hash_id(as_text(payload)), payload, {}))
|
|
132
|
+
return out
|
|
133
|
+
for item in payload:
|
|
134
|
+
out.extend(_conversation_records(item))
|
|
135
|
+
return out
|
|
136
|
+
if not isinstance(payload, dict):
|
|
137
|
+
return out
|
|
138
|
+
messages = payload.get("messages")
|
|
139
|
+
if not isinstance(messages, list):
|
|
140
|
+
return out
|
|
141
|
+
tid = payload.get("trace_id") or payload.get("id")
|
|
142
|
+
trace_id = tid if isinstance(tid, str) and tid else _hash_id(as_text(messages))
|
|
143
|
+
# 32-hex normalize so trace ids are uniform regardless of source id format.
|
|
144
|
+
if len(trace_id) != 32 or any(c not in "0123456789abcdef" for c in trace_id.lower()):
|
|
145
|
+
trace_id = _hash_id(trace_id)
|
|
146
|
+
meta = payload.get("metadata")
|
|
147
|
+
metadata: JsonObject = meta if isinstance(meta, dict) else {}
|
|
148
|
+
out.append((trace_id, messages, metadata))
|
|
149
|
+
return out
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class ChatMessagesAdapter(BaseTraceAdapter):
|
|
153
|
+
"""Convert recorded chat/tool-call conversations (OpenAI-style) into `Trace`s. No SDK."""
|
|
154
|
+
|
|
155
|
+
name = "chat-json"
|
|
156
|
+
|
|
157
|
+
def spans_from_payload(self, payload: JsonValue) -> list[SpanRecord]:
|
|
158
|
+
spans: list[SpanRecord] = []
|
|
159
|
+
for trace_id, messages, metadata in _conversation_records(payload):
|
|
160
|
+
spans.extend(_spans_for_conversation(messages, trace_id, metadata))
|
|
161
|
+
return spans
|
|
162
|
+
|
|
163
|
+
def _pull_payloads(self, pull: VendorPull) -> list[JsonValue]:
|
|
164
|
+
raise ValueError(
|
|
165
|
+
"chat-json converts local conversation files; it has no vendor API. "
|
|
166
|
+
"Use `from_file` with an exported messages JSON/JSONL."
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
register_adapter(ChatMessagesAdapter())
|