world-model-optimizer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_waterfall/LICENSE +21 -0
- llm_waterfall/__init__.py +53 -0
- llm_waterfall/adapters/__init__.py +36 -0
- llm_waterfall/adapters/anthropic.py +105 -0
- llm_waterfall/adapters/aws_mantle.py +47 -0
- llm_waterfall/adapters/azure_openai.py +71 -0
- llm_waterfall/adapters/base.py +51 -0
- llm_waterfall/adapters/bedrock.py +309 -0
- llm_waterfall/adapters/openai.py +130 -0
- llm_waterfall/classify.py +184 -0
- llm_waterfall/pricing.py +110 -0
- llm_waterfall/py.typed +0 -0
- llm_waterfall/types.py +295 -0
- llm_waterfall/waterfall.py +255 -0
- wmo/__init__.py +38 -0
- wmo/agents/__init__.py +7 -0
- wmo/agents/default.py +29 -0
- wmo/agents/meta.py +55 -0
- wmo/agents/optimizer.py +55 -0
- wmo/agents/project.py +928 -0
- wmo/cli/__init__.py +5 -0
- wmo/cli/agent_session.py +1123 -0
- wmo/cli/app.py +2489 -0
- wmo/cli/e2b_cmds.py +212 -0
- wmo/cli/eval_closed_loop.py +207 -0
- wmo/cli/harness_app.py +1147 -0
- wmo/cli/harness_distill.py +659 -0
- wmo/cli/hosted_session.py +880 -0
- wmo/cli/ingest_cmd.py +165 -0
- wmo/cli/model_roles.py +82 -0
- wmo/cli/platform_cmds.py +372 -0
- wmo/cli/route_app.py +274 -0
- wmo/cli/session_state.py +243 -0
- wmo/cli/ui.py +1107 -0
- wmo/cli/workspace_sync.py +504 -0
- wmo/config/__init__.py +60 -0
- wmo/config/card.py +129 -0
- wmo/config/config.py +367 -0
- wmo/config/dotenv.py +67 -0
- wmo/config/settings.py +128 -0
- wmo/config/store.py +177 -0
- wmo/conftest.py +19 -0
- wmo/connect/__init__.py +88 -0
- wmo/connect/apps.py +78 -0
- wmo/connect/brave.py +284 -0
- wmo/connect/connector.py +79 -0
- wmo/connect/credentials.py +164 -0
- wmo/connect/github.py +321 -0
- wmo/connect/google.py +627 -0
- wmo/connect/notion.py +790 -0
- wmo/connect/oauth.py +461 -0
- wmo/connect/slack.py +555 -0
- wmo/connect/store.py +199 -0
- wmo/connect/types.py +156 -0
- wmo/core/__init__.py +21 -0
- wmo/core/parsing.py +281 -0
- wmo/core/render.py +271 -0
- wmo/core/text.py +40 -0
- wmo/core/types.py +116 -0
- wmo/distill/__init__.py +14 -0
- wmo/distill/agents.py +140 -0
- wmo/distill/config.py +1006 -0
- wmo/distill/cost.py +437 -0
- wmo/distill/data.py +921 -0
- wmo/distill/deadlines.py +254 -0
- wmo/distill/fake_tinker.py +734 -0
- wmo/distill/gate.py +122 -0
- wmo/distill/loop.py +3499 -0
- wmo/distill/renderers.py +399 -0
- wmo/distill/rendering.py +620 -0
- wmo/distill/rollouts.py +726 -0
- wmo/distill/samples.py +195 -0
- wmo/distill/store.py +829 -0
- wmo/distill/teacher.py +714 -0
- wmo/distill/tokens.py +535 -0
- wmo/distill/tracking.py +552 -0
- wmo/distill/tripwire.py +411 -0
- wmo/distill/xtoken/byte_offsets.py +152 -0
- wmo/distill/xtoken/chunks.py +457 -0
- wmo/distill/xtoken/prompt_logprobs.py +475 -0
- wmo/distill/xtoken/teacher_render.py +346 -0
- wmo/engine/__init__.py +28 -0
- wmo/engine/autoconfig.py +367 -0
- wmo/engine/build.py +346 -0
- wmo/engine/demo.py +77 -0
- wmo/engine/eval_suites.py +245 -0
- wmo/engine/grounding.py +491 -0
- wmo/engine/knowledge.py +291 -0
- wmo/engine/loader.py +36 -0
- wmo/engine/play.py +92 -0
- wmo/engine/prompts.py +99 -0
- wmo/engine/replay.py +443 -0
- wmo/engine/reporting.py +58 -0
- wmo/engine/workspace.py +468 -0
- wmo/engine/world_model.py +568 -0
- wmo/env/__init__.py +22 -0
- wmo/env/base.py +121 -0
- wmo/env/closed_loop.py +229 -0
- wmo/env/episode.py +107 -0
- wmo/env/llm_agent.py +93 -0
- wmo/env/scenarios.py +73 -0
- wmo/evals/__init__.py +52 -0
- wmo/evals/agreement.py +110 -0
- wmo/evals/base.py +45 -0
- wmo/evals/closed_loop.py +480 -0
- wmo/evals/failover.py +96 -0
- wmo/evals/gold.py +127 -0
- wmo/evals/grid.py +394 -0
- wmo/evals/grid_plot.py +205 -0
- wmo/evals/harbor/__init__.py +27 -0
- wmo/evals/harbor/agent.py +573 -0
- wmo/evals/harbor/ctrf.py +171 -0
- wmo/evals/harbor/e2b_environment.py +587 -0
- wmo/evals/harbor/e2b_template_policy.py +144 -0
- wmo/evals/harbor/scorer.py +875 -0
- wmo/evals/harbor/tasks.py +140 -0
- wmo/evals/open_loop.py +194 -0
- wmo/evals/tasks.py +53 -0
- wmo/harness/__init__.py +51 -0
- wmo/harness/code_runtime.py +288 -0
- wmo/harness/create.py +1191 -0
- wmo/harness/delta.py +220 -0
- wmo/harness/doc.py +556 -0
- wmo/harness/e2b_ledger.py +342 -0
- wmo/harness/e2b_reap.py +476 -0
- wmo/harness/e2b_sandbox.py +350 -0
- wmo/harness/environment.py +35 -0
- wmo/harness/live_session.py +543 -0
- wmo/harness/mutate.py +343 -0
- wmo/harness/pi_e2b.py +1710 -0
- wmo/harness/pi_entry/entry.ts +268 -0
- wmo/harness/pi_entry/runner_frames.ts +92 -0
- wmo/harness/pi_entry/runner_live.ts +587 -0
- wmo/harness/pi_entry/runner_service.ts +270 -0
- wmo/harness/pi_entry/runner_stdio.ts +374 -0
- wmo/harness/pi_entry/runner_termination.ts +142 -0
- wmo/harness/pi_local.py +262 -0
- wmo/harness/pi_runtime.py +495 -0
- wmo/harness/pi_vendor.py +65 -0
- wmo/harness/population.py +509 -0
- wmo/harness/project_proposer.py +569 -0
- wmo/harness/proposer.py +977 -0
- wmo/harness/runner_link.py +619 -0
- wmo/harness/runtime.py +389 -0
- wmo/harness/scoring.py +247 -0
- wmo/harness/skills.py +116 -0
- wmo/harness/source_tree.py +319 -0
- wmo/harness/store.py +176 -0
- wmo/harness/tools.py +105 -0
- wmo/harness/vendor/manifest.sha256 +58 -0
- wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
- wmo/harness/vendor/pi-agent/LICENSE +21 -0
- wmo/harness/vendor/pi-agent/README.md +488 -0
- wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
- wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
- wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
- wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
- wmo/harness/vendor/pi-agent/docs/models.md +966 -0
- wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
- wmo/harness/vendor/pi-agent/package.json +60 -0
- wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
- wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
- wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
- wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
- wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
- wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
- wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
- wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
- wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
- wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
- wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
- wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
- wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
- wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
- wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
- wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
- wmo/harness/vendor/pi-agent/src/index.ts +44 -0
- wmo/harness/vendor/pi-agent/src/node.ts +2 -0
- wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
- wmo/harness/vendor/pi-agent/src/types.ts +428 -0
- wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
- wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
- wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
- wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
- wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
- wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
- wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
- wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
- wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
- wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
- wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
- wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
- wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
- wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
- wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
- wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
- wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
- wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
- wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
- wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
- wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
- wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
- wmo/harness/vendor/vendor_pi.sh +59 -0
- wmo/harness/workspace_patch.py +270 -0
- wmo/ingest/__init__.py +47 -0
- wmo/ingest/adapter.py +72 -0
- wmo/ingest/base.py +114 -0
- wmo/ingest/braintrust.py +339 -0
- wmo/ingest/detect.py +126 -0
- wmo/ingest/langfuse.py +291 -0
- wmo/ingest/langsmith.py +444 -0
- wmo/ingest/mastra.py +330 -0
- wmo/ingest/messages.py +170 -0
- wmo/ingest/normalize.py +679 -0
- wmo/ingest/otel_genai.py +69 -0
- wmo/ingest/otel_writer.py +100 -0
- wmo/ingest/phoenix.py +150 -0
- wmo/ingest/postgres.py +246 -0
- wmo/ingest/posthog.py +320 -0
- wmo/ingest/quality.py +28 -0
- wmo/ingest/stream.py +209 -0
- wmo/ingest/testdata/sample_otlp.json +60 -0
- wmo/ingest/testdata/sample_spans.jsonl +3 -0
- wmo/optimize/__init__.py +25 -0
- wmo/optimize/base.py +143 -0
- wmo/optimize/gepa.py +806 -0
- wmo/optimize/judge.py +262 -0
- wmo/optimize/judge_quality.py +359 -0
- wmo/optimize/knn.py +468 -0
- wmo/optimize/numeric.py +152 -0
- wmo/optimize/outcomes.py +103 -0
- wmo/optimize/policy.py +669 -0
- wmo/optimize/report.py +231 -0
- wmo/optimize/reward.py +129 -0
- wmo/optimize/routing.py +373 -0
- wmo/platform/__init__.py +6 -0
- wmo/platform/auth.py +115 -0
- wmo/platform/client.py +551 -0
- wmo/platform/credentials.py +126 -0
- wmo/platform/transfer.py +158 -0
- wmo/providers/__init__.py +40 -0
- wmo/providers/_bedrock_chat.py +155 -0
- wmo/providers/_openai_common.py +182 -0
- wmo/providers/_responses_common.py +472 -0
- wmo/providers/anthropic.py +134 -0
- wmo/providers/azure_openai.py +296 -0
- wmo/providers/base.py +300 -0
- wmo/providers/bedrock.py +312 -0
- wmo/providers/models.py +205 -0
- wmo/providers/openai.py +143 -0
- wmo/providers/openai_responses.py +240 -0
- wmo/providers/pool.py +170 -0
- wmo/providers/registry.py +73 -0
- wmo/providers/retry.py +151 -0
- wmo/providers/tinker.py +936 -0
- wmo/providers/waterfall.py +336 -0
- wmo/research/__init__.py +81 -0
- wmo/research/ablation.py +133 -0
- wmo/research/concurrency_plot.py +523 -0
- wmo/research/concurrency_run.py +240 -0
- wmo/research/concurrency_scaling.py +270 -0
- wmo/research/gepa_scaling.py +274 -0
- wmo/research/pipeline.py +198 -0
- wmo/research/scaling_split.py +82 -0
- wmo/research/scenario_fidelity.py +198 -0
- wmo/research/scenario_recovery.py +92 -0
- wmo/research/seed_stability.py +90 -0
- wmo/research/trace_scaling.py +348 -0
- wmo/retrieval/__init__.py +6 -0
- wmo/retrieval/embedders.py +105 -0
- wmo/retrieval/leakfree.py +52 -0
- wmo/retrieval/retriever.py +173 -0
- wmo/scenarios/__init__.py +58 -0
- wmo/scenarios/builder.py +152 -0
- wmo/scenarios/mining/__init__.py +27 -0
- wmo/scenarios/mining/clustering.py +171 -0
- wmo/scenarios/mining/facets.py +226 -0
- wmo/scenarios/mining/selection.py +220 -0
- wmo/scenarios/synthesis/__init__.py +6 -0
- wmo/scenarios/synthesis/scenario_set.py +63 -0
- wmo/scenarios/synthesis/synthesizer.py +85 -0
- wmo/scenarios/verification/__init__.py +17 -0
- wmo/scenarios/verification/judge.py +97 -0
- wmo/scenarios/verification/verify.py +135 -0
- wmo/serving/__init__.py +5 -0
- wmo/serving/builds.py +451 -0
- wmo/serving/chat.py +878 -0
- wmo/serving/endpoint_config.py +64 -0
- wmo/serving/savings.py +250 -0
- wmo/serving/server.py +553 -0
- wmo/serving/traces_source.py +206 -0
- wmo/telemetry.py +213 -0
- wmo/tracking/__init__.py +36 -0
- wmo/tracking/clock.py +24 -0
- wmo/tracking/metered.py +125 -0
- wmo/tracking/pricing.py +99 -0
- wmo/tracking/store.py +31 -0
- wmo/tracking/tracker.py +149 -0
- world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
- world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
- world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
- world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
wmo/ingest/langsmith.py
ADDED
|
@@ -0,0 +1,444 @@
|
|
|
1
|
+
"""LangSmith adapter: turn a LangSmith run-tree export into `Trace`s.
|
|
2
|
+
|
|
3
|
+
LangSmith (LangChain's tracing product) does NOT export OTLP spans. It models a trace as a tree of
|
|
4
|
+
**runs**, where each run is a node typed by `run_type`
|
|
5
|
+
(`tool | chain | llm | retriever | embedding | prompt | parser`). An exported run (from
|
|
6
|
+
`POST /api/v1/runs/query` — the list endpoint is a POST with a JSON filter body, NOT a GET — or the
|
|
7
|
+
SDK `Client.list_runs`) looks roughly like:
|
|
8
|
+
|
|
9
|
+
{"id": "<run uuid>", "trace_id": "<trace grouping uuid>", "parent_run_id": "<uuid|null>",
|
|
10
|
+
"run_type": "llm" | "tool" | "chain" | "retriever" | "embedding" | "prompt" | "parser",
|
|
11
|
+
"name": "ChatOpenAI",
|
|
12
|
+
"inputs": {...}, "outputs": {...},
|
|
13
|
+
"start_time": "2026-01-01T00:00:00.000000", "end_time": "...",
|
|
14
|
+
"error": null | "<traceback str>", "extra": {...}}
|
|
15
|
+
|
|
16
|
+
Because this is not an OTLP/OpenInference span shape, the adapter overrides `spans_from_payload`
|
|
17
|
+
(like `wmo.ingest.messages` / `wmo.ingest.langfuse`) and emits `SpanRecord`s in the **OTel-GenAI
|
|
18
|
+
vocabulary** so the shared classifier/normalizer (`wmo.ingest.normalize`) does the pairing /
|
|
19
|
+
state / metadata work. Each run maps to zero or more spans:
|
|
20
|
+
|
|
21
|
+
- `run_type == "llm"`: if the outputs carry tool calls, emit one `chat` ACTION span per call with
|
|
22
|
+
`{"gen_ai.tool.name", "gen_ai.tool.call.arguments"}`; otherwise emit one plain `chat` message
|
|
23
|
+
span with `{"gen_ai.completion": <text>}`.
|
|
24
|
+
- `run_type == "tool"`: emit one `execute_tool` RESULT span with
|
|
25
|
+
`{"gen_ai.operation.name": "execute_tool", "gen_ai.tool.name": <name>,
|
|
26
|
+
"gen_ai.tool.message": <outputs as text>}`. The normalizer pairs it with the preceding action
|
|
27
|
+
span (the llm run's tool call), backfilling name/args from here if the action lacked them.
|
|
28
|
+
- `run_type in {"chain", "retriever", ...}`: skipped (not directly actionable). A run we cannot
|
|
29
|
+
interpret is skipped rather than crashing the ingest.
|
|
30
|
+
|
|
31
|
+
Where LangChain hides tool calls (these paths are version-dependent and best-effort; we dig all of
|
|
32
|
+
them and degrade gracefully):
|
|
33
|
+
- `outputs["generations"][i]["message"]["kwargs"]["tool_calls"]` (LCEL ChatGeneration dump)
|
|
34
|
+
- `outputs["generations"][i]["message"]["kwargs"]["additional_kwargs"]["tool_calls"]` (OpenAI)
|
|
35
|
+
- `outputs["generations"][i][j]["message"]...` (nested list-of-lists generations)
|
|
36
|
+
- `outputs["tool_calls"]` / `outputs["message"]...` (flatter dumps)
|
|
37
|
+
A LangChain `tool_calls` entry is either the normalized shape `{"name", "args": {...}, "id"}` or the
|
|
38
|
+
OpenAI shape `{"id", "function": {"name", "arguments": "<json str>"}}`; both are handled.
|
|
39
|
+
|
|
40
|
+
The LLM completion text is dug from `generations[i].text`, `generations[i].message.kwargs.content`,
|
|
41
|
+
or `outputs["output"|"content"|"text"]`. A tool run's output is dug from
|
|
42
|
+
`outputs["output"]`, else the whole `outputs`. The task (`gen_ai.prompt`) is the first human/user
|
|
43
|
+
input, dug from a run's `inputs` (chat `messages`, a `messages` list, or `inputs["input"]`).
|
|
44
|
+
|
|
45
|
+
Ordering: runs are ordered by `start_time` (ISO-8601 -> epoch microseconds); only monotonicity
|
|
46
|
+
within a trace matters (the normalizer sorts by `start_nano`), so an absent/unparseable timestamp
|
|
47
|
+
degrades to the run's list index.
|
|
48
|
+
|
|
49
|
+
Accepted file shapes (`from_file`): a single run object, a JSON array of runs, a `{"runs": [...]}`
|
|
50
|
+
wrapper, or JSONL (one run per line). Grouping is by `trace_id` (falling back to `id` for a root
|
|
51
|
+
run that omits it).
|
|
52
|
+
|
|
53
|
+
Pull: `from_vendor` fetches runs live over the REST API with plain httpx (no SDK); see
|
|
54
|
+
`_pull_payloads`. File exports remain fully supported via `from_file`.
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
from __future__ import annotations
|
|
58
|
+
|
|
59
|
+
import os
|
|
60
|
+
|
|
61
|
+
import httpx
|
|
62
|
+
from pydantic import JsonValue
|
|
63
|
+
|
|
64
|
+
from wmo.core.types import JsonObject
|
|
65
|
+
from wmo.ingest.adapter import VendorPull, register_adapter
|
|
66
|
+
from wmo.ingest.base import BaseTraceAdapter
|
|
67
|
+
from wmo.ingest.normalize import SpanRecord, as_text, iso_to_ordinal
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _as_str(value: JsonValue) -> str:
|
|
71
|
+
return value if isinstance(value, str) else ""
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _is_error(run: JsonObject) -> bool:
|
|
75
|
+
"""A non-null/non-empty `error` marks the run as failed; an empty string is NOT an error.
|
|
76
|
+
|
|
77
|
+
Some LangSmith dumps set `error: ""` on a successful run, so a bare `is not None` check would
|
|
78
|
+
misclassify it as a failure.
|
|
79
|
+
"""
|
|
80
|
+
error = run.get("error")
|
|
81
|
+
if error is None:
|
|
82
|
+
return False
|
|
83
|
+
if isinstance(error, str):
|
|
84
|
+
return bool(error.strip())
|
|
85
|
+
return bool(error)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _start_ordinal(run: JsonObject, fallback: int) -> int:
|
|
89
|
+
"""Monotonic ordering key from the run's `start_time` (shared helper; UTC-safe)."""
|
|
90
|
+
return iso_to_ordinal(run.get("start_time"), fallback)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _generations(outputs: JsonObject) -> list[JsonObject]:
|
|
94
|
+
"""Flatten `outputs["generations"]` (a list, or a list-of-lists) into generation dicts."""
|
|
95
|
+
raw = outputs.get("generations")
|
|
96
|
+
if not isinstance(raw, list):
|
|
97
|
+
return []
|
|
98
|
+
flat: list[JsonObject] = []
|
|
99
|
+
for item in raw:
|
|
100
|
+
if isinstance(item, dict):
|
|
101
|
+
flat.append(item)
|
|
102
|
+
elif isinstance(item, list): # list-of-lists (one inner list per prompt)
|
|
103
|
+
flat.extend(g for g in item if isinstance(g, dict))
|
|
104
|
+
return flat
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _message_kwargs(generation: JsonObject) -> JsonObject:
|
|
108
|
+
"""The `message.kwargs` dict of a ChatGeneration dump (empty when absent)."""
|
|
109
|
+
message = generation.get("message")
|
|
110
|
+
if isinstance(message, dict):
|
|
111
|
+
kwargs = message.get("kwargs")
|
|
112
|
+
if isinstance(kwargs, dict):
|
|
113
|
+
return kwargs
|
|
114
|
+
return {}
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _tool_calls_in(container: JsonObject) -> list[JsonObject]:
|
|
118
|
+
"""Pull `tool_calls` from a dict, checking `tool_calls` then `additional_kwargs.tool_calls`."""
|
|
119
|
+
calls: list[JsonObject] = []
|
|
120
|
+
raw = container.get("tool_calls")
|
|
121
|
+
if isinstance(raw, list):
|
|
122
|
+
calls.extend(tc for tc in raw if isinstance(tc, dict))
|
|
123
|
+
extra = container.get("additional_kwargs")
|
|
124
|
+
if isinstance(extra, dict):
|
|
125
|
+
raw_extra = extra.get("tool_calls")
|
|
126
|
+
if isinstance(raw_extra, list):
|
|
127
|
+
calls.extend(tc for tc in raw_extra if isinstance(tc, dict))
|
|
128
|
+
return calls
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _llm_tool_calls(outputs: JsonObject) -> list[JsonObject]:
|
|
132
|
+
"""Dig tool calls out of an llm run's `outputs`, across the common LangChain locations."""
|
|
133
|
+
calls: list[JsonObject] = []
|
|
134
|
+
for generation in _generations(outputs):
|
|
135
|
+
calls.extend(_tool_calls_in(_message_kwargs(generation)))
|
|
136
|
+
# Flatter dumps: outputs.tool_calls or outputs.message.kwargs.tool_calls.
|
|
137
|
+
calls.extend(_tool_calls_in(outputs))
|
|
138
|
+
message = outputs.get("message")
|
|
139
|
+
if isinstance(message, dict):
|
|
140
|
+
kwargs = message.get("kwargs")
|
|
141
|
+
if isinstance(kwargs, dict):
|
|
142
|
+
calls.extend(_tool_calls_in(kwargs))
|
|
143
|
+
return calls
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _call_name_args(tool_call: JsonObject) -> tuple[str, str]:
|
|
147
|
+
"""(name, raw-arguments-json) from a LangChain-normalized or OpenAI-shaped tool call."""
|
|
148
|
+
fn = tool_call.get("function")
|
|
149
|
+
if isinstance(fn, dict): # OpenAI shape: {"function": {"name", "arguments": "<json str>"}}
|
|
150
|
+
name = fn.get("name")
|
|
151
|
+
args = fn.get("arguments")
|
|
152
|
+
else: # LangChain-normalized shape: {"name", "args": {...}, "id"}
|
|
153
|
+
name = tool_call.get("name")
|
|
154
|
+
args = tool_call.get("args")
|
|
155
|
+
if args is None:
|
|
156
|
+
args = tool_call.get("arguments")
|
|
157
|
+
name_s = name if isinstance(name, str) else ""
|
|
158
|
+
args_s = args if isinstance(args, str) else as_text(args)
|
|
159
|
+
return name_s, args_s
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _llm_completion(outputs: JsonObject) -> str:
|
|
163
|
+
"""Dig the assistant text out of an llm run's `outputs` (best-effort across dump shapes)."""
|
|
164
|
+
for generation in _generations(outputs):
|
|
165
|
+
text = generation.get("text")
|
|
166
|
+
if isinstance(text, str) and text:
|
|
167
|
+
return text
|
|
168
|
+
content = _message_kwargs(generation).get("content")
|
|
169
|
+
if isinstance(content, str) and content:
|
|
170
|
+
return content
|
|
171
|
+
for key in ("output", "content", "text"):
|
|
172
|
+
value = outputs.get(key)
|
|
173
|
+
if isinstance(value, str) and value:
|
|
174
|
+
return value
|
|
175
|
+
return as_text(outputs)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _tool_output_text(outputs: JsonValue) -> str:
|
|
179
|
+
"""A tool run's result text: `outputs["output"]` if present, else the whole outputs."""
|
|
180
|
+
if isinstance(outputs, dict):
|
|
181
|
+
value = outputs.get("output")
|
|
182
|
+
if value is not None:
|
|
183
|
+
return as_text(value)
|
|
184
|
+
return as_text(outputs)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _tool_run_name(run: JsonObject) -> str:
|
|
188
|
+
"""A tool name for a `tool` run: explicit fields first, else the run name."""
|
|
189
|
+
for key in ("tool_name", "name"):
|
|
190
|
+
value = run.get(key)
|
|
191
|
+
if isinstance(value, str) and value:
|
|
192
|
+
return value
|
|
193
|
+
return ""
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _first_user_text(inputs: JsonValue) -> str | None:
|
|
197
|
+
"""Dig the first human/user input out of a run's `inputs` (the trace task)."""
|
|
198
|
+
if isinstance(inputs, str):
|
|
199
|
+
return inputs or None
|
|
200
|
+
if not isinstance(inputs, dict):
|
|
201
|
+
return None
|
|
202
|
+
messages = inputs.get("messages")
|
|
203
|
+
text = _first_user_in_messages(messages)
|
|
204
|
+
if text is not None:
|
|
205
|
+
return text
|
|
206
|
+
for key in ("input", "question", "query", "text"):
|
|
207
|
+
value = inputs.get(key)
|
|
208
|
+
if isinstance(value, str) and value:
|
|
209
|
+
return value
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _first_user_in_messages(messages: JsonValue) -> str | None:
|
|
214
|
+
"""First human/user message content in a (possibly nested) LangChain messages list."""
|
|
215
|
+
if not isinstance(messages, list):
|
|
216
|
+
return None
|
|
217
|
+
for message in messages:
|
|
218
|
+
# Nested list-of-lists (one inner list per prompt) — recurse.
|
|
219
|
+
if isinstance(message, list):
|
|
220
|
+
found = _first_user_in_messages(message)
|
|
221
|
+
if found is not None:
|
|
222
|
+
return found
|
|
223
|
+
continue
|
|
224
|
+
if not isinstance(message, dict):
|
|
225
|
+
continue
|
|
226
|
+
role = _message_role(message)
|
|
227
|
+
content = _message_content(message)
|
|
228
|
+
if role in {"human", "user"} and content:
|
|
229
|
+
return content
|
|
230
|
+
return None
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _message_role(message: JsonObject) -> str:
|
|
234
|
+
"""Role of a LangChain/OpenAI message dict (`role`, `type`, or a serialized class id)."""
|
|
235
|
+
for key in ("role", "type"):
|
|
236
|
+
value = message.get(key)
|
|
237
|
+
if isinstance(value, str) and value:
|
|
238
|
+
return value.lower()
|
|
239
|
+
# Serialized LangChain message: {"id": [..., "HumanMessage"], "kwargs": {...}}.
|
|
240
|
+
cid = message.get("id")
|
|
241
|
+
if isinstance(cid, list) and cid:
|
|
242
|
+
last = cid[-1]
|
|
243
|
+
if isinstance(last, str):
|
|
244
|
+
return last.lower().removesuffix("message")
|
|
245
|
+
return ""
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _message_content(message: JsonObject) -> str:
|
|
249
|
+
"""Text content of a LangChain/OpenAI message dict (top-level or under `kwargs`)."""
|
|
250
|
+
content = message.get("content")
|
|
251
|
+
if isinstance(content, str) and content:
|
|
252
|
+
return content
|
|
253
|
+
kwargs = message.get("kwargs")
|
|
254
|
+
if isinstance(kwargs, dict):
|
|
255
|
+
nested = kwargs.get("content")
|
|
256
|
+
if isinstance(nested, str) and nested:
|
|
257
|
+
return nested
|
|
258
|
+
return ""
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
_API_KEY_ENVS = ("LANGCHAIN_API_KEY", "LANGSMITH_API_KEY")
|
|
262
|
+
_ENDPOINT_ENV = "LANGCHAIN_ENDPOINT"
|
|
263
|
+
_DEFAULT_ENDPOINT = "https://api.smith.langchain.com"
|
|
264
|
+
_PAGE_SIZE = 100
|
|
265
|
+
# Backstop on unbounded pulls (no --limit): runs, not traces, are what the API pages by.
|
|
266
|
+
_MAX_RUNS = 2000
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
class LangSmithAdapter(BaseTraceAdapter):
|
|
270
|
+
"""Map a LangSmith run-tree export into normalized `Trace`s. No SDK; pure JSON."""
|
|
271
|
+
|
|
272
|
+
name = "langsmith"
|
|
273
|
+
|
|
274
|
+
def _pull_payloads(self, pull: VendorPull) -> list[JsonValue]:
|
|
275
|
+
"""Fetch runs live from the LangSmith REST API (plain httpx, no SDK).
|
|
276
|
+
|
|
277
|
+
`pull.api_key` (else `$LANGCHAIN_API_KEY`/`$LANGSMITH_API_KEY`) auths; `pull.project` is
|
|
278
|
+
the project (session) UUID passed to `POST /api/v1/runs/query`: the list endpoint is a
|
|
279
|
+
POST with a JSON filter body). Pagination follows `cursors.next` until exhausted, `--limit`
|
|
280
|
+
traces are covered, or a run-count backstop trips.
|
|
281
|
+
"""
|
|
282
|
+
api_key = pull.api_key or next(
|
|
283
|
+
(value for env in _API_KEY_ENVS if (value := os.environ.get(env))), None
|
|
284
|
+
)
|
|
285
|
+
if not api_key:
|
|
286
|
+
raise ValueError(
|
|
287
|
+
f"langsmith pull needs an API key: pass --api-key or set ${_API_KEY_ENVS[0]}"
|
|
288
|
+
)
|
|
289
|
+
endpoint = (os.environ.get(_ENDPOINT_ENV) or _DEFAULT_ENDPOINT).rstrip("/")
|
|
290
|
+
base_body: JsonObject = {"limit": _PAGE_SIZE}
|
|
291
|
+
if pull.project:
|
|
292
|
+
base_body["session"] = [pull.project]
|
|
293
|
+
if pull.since is not None:
|
|
294
|
+
base_body["start_time"] = pull.since
|
|
295
|
+
|
|
296
|
+
runs: list[JsonValue] = []
|
|
297
|
+
trace_ids: set[str] = set()
|
|
298
|
+
cursor: str | None = None
|
|
299
|
+
while True:
|
|
300
|
+
body = dict(base_body)
|
|
301
|
+
# The cursor key is present only when following a page: cursor-based APIs
|
|
302
|
+
# commonly reject an explicit null on the first request.
|
|
303
|
+
if cursor is not None:
|
|
304
|
+
body["cursor"] = cursor
|
|
305
|
+
resp = httpx.post(
|
|
306
|
+
f"{endpoint}/api/v1/runs/query",
|
|
307
|
+
headers={"x-api-key": api_key},
|
|
308
|
+
json=body,
|
|
309
|
+
timeout=60.0,
|
|
310
|
+
)
|
|
311
|
+
resp.raise_for_status()
|
|
312
|
+
page = resp.json()
|
|
313
|
+
page_runs = page.get("runs") if isinstance(page, dict) else None
|
|
314
|
+
if not isinstance(page_runs, list) or not page_runs:
|
|
315
|
+
break
|
|
316
|
+
for run in page_runs:
|
|
317
|
+
if isinstance(run, dict):
|
|
318
|
+
runs.append(run)
|
|
319
|
+
trace_ids.add(self._trace_id(run))
|
|
320
|
+
if pull.limit is not None and len(trace_ids) >= pull.limit:
|
|
321
|
+
break
|
|
322
|
+
if len(runs) >= _MAX_RUNS:
|
|
323
|
+
break
|
|
324
|
+
cursors = page.get("cursors")
|
|
325
|
+
cursor = cursors.get("next") if isinstance(cursors, dict) else None
|
|
326
|
+
if not isinstance(cursor, str) or not cursor:
|
|
327
|
+
break
|
|
328
|
+
return [{"runs": runs}]
|
|
329
|
+
|
|
330
|
+
def spans_from_payload(self, payload: JsonValue) -> list[SpanRecord]:
|
|
331
|
+
"""Map one payload (single run, list, `{runs:[...]}`) to ordered `SpanRecord`s by trace."""
|
|
332
|
+
runs = self._runs(payload)
|
|
333
|
+
# Group by trace_id so we can assign a per-trace monotonic ordinal and set the task once.
|
|
334
|
+
by_trace: dict[str, list[JsonObject]] = {}
|
|
335
|
+
for run in runs:
|
|
336
|
+
by_trace.setdefault(self._trace_id(run), []).append(run)
|
|
337
|
+
|
|
338
|
+
spans: list[SpanRecord] = []
|
|
339
|
+
for trace_id, trace_runs in by_trace.items():
|
|
340
|
+
spans.extend(self._spans_for_trace(trace_id, trace_runs))
|
|
341
|
+
return spans
|
|
342
|
+
|
|
343
|
+
def _runs(self, payload: JsonValue) -> list[JsonObject]:
|
|
344
|
+
"""Normalize a payload into a flat list of run objects.
|
|
345
|
+
|
|
346
|
+
Accepts a single run, a bare list of runs, or a `{"runs": [...]}` wrapper.
|
|
347
|
+
"""
|
|
348
|
+
if isinstance(payload, list):
|
|
349
|
+
out: list[JsonObject] = []
|
|
350
|
+
for item in payload:
|
|
351
|
+
out.extend(self._runs(item))
|
|
352
|
+
return out
|
|
353
|
+
if not isinstance(payload, dict):
|
|
354
|
+
return []
|
|
355
|
+
wrapped = payload.get("runs")
|
|
356
|
+
if isinstance(wrapped, list) and "run_type" not in payload:
|
|
357
|
+
out = []
|
|
358
|
+
for item in wrapped:
|
|
359
|
+
out.extend(self._runs(item))
|
|
360
|
+
return out
|
|
361
|
+
# A run object: it has an id (and usually run_type). Be permissive.
|
|
362
|
+
if "id" in payload or "run_type" in payload:
|
|
363
|
+
return [payload]
|
|
364
|
+
return []
|
|
365
|
+
|
|
366
|
+
def _spans_for_trace(self, trace_id: str, runs: list[JsonObject]) -> list[SpanRecord]:
|
|
367
|
+
# Order by start_time; ties (or absent timestamps) keep input order via the index fallback.
|
|
368
|
+
indexed = list(enumerate(runs))
|
|
369
|
+
indexed.sort(key=lambda pair: (_start_ordinal(pair[1], pair[0]), pair[0]))
|
|
370
|
+
|
|
371
|
+
task = self._trace_task([run for _, run in indexed])
|
|
372
|
+
|
|
373
|
+
spans: list[SpanRecord] = []
|
|
374
|
+
ordinal = 0
|
|
375
|
+
|
|
376
|
+
def emit(attrs: JsonObject, *, tool: bool, error: bool = False) -> None:
|
|
377
|
+
nonlocal ordinal
|
|
378
|
+
if ordinal == 0 and task is not None:
|
|
379
|
+
attrs.setdefault("gen_ai.prompt", task)
|
|
380
|
+
spans.append(
|
|
381
|
+
SpanRecord(
|
|
382
|
+
trace_id=trace_id,
|
|
383
|
+
span_id=f"{trace_id[:12]}{ordinal:06x}{'t' if tool else 'a'}",
|
|
384
|
+
name="execute_tool" if tool else "chat",
|
|
385
|
+
start_nano=ordinal,
|
|
386
|
+
attributes={
|
|
387
|
+
"gen_ai.operation.name": "execute_tool" if tool else "chat",
|
|
388
|
+
**attrs,
|
|
389
|
+
},
|
|
390
|
+
status_error=error,
|
|
391
|
+
)
|
|
392
|
+
)
|
|
393
|
+
ordinal += 1
|
|
394
|
+
|
|
395
|
+
for _, run in indexed:
|
|
396
|
+
run_type = _as_str(run.get("run_type")).lower()
|
|
397
|
+
error = _is_error(run)
|
|
398
|
+
outputs = run.get("outputs")
|
|
399
|
+
out_obj: JsonObject = outputs if isinstance(outputs, dict) else {}
|
|
400
|
+
|
|
401
|
+
if run_type == "llm":
|
|
402
|
+
calls = _llm_tool_calls(out_obj)
|
|
403
|
+
if calls:
|
|
404
|
+
for tool_call in calls:
|
|
405
|
+
name, args = _call_name_args(tool_call)
|
|
406
|
+
emit(
|
|
407
|
+
{"gen_ai.tool.name": name, "gen_ai.tool.call.arguments": args},
|
|
408
|
+
tool=False,
|
|
409
|
+
error=error,
|
|
410
|
+
)
|
|
411
|
+
else:
|
|
412
|
+
emit({"gen_ai.completion": _llm_completion(out_obj)}, tool=False, error=error)
|
|
413
|
+
elif run_type == "tool":
|
|
414
|
+
emit(
|
|
415
|
+
{
|
|
416
|
+
"gen_ai.tool.name": _tool_run_name(run),
|
|
417
|
+
"gen_ai.tool.message": _tool_output_text(outputs),
|
|
418
|
+
},
|
|
419
|
+
tool=True,
|
|
420
|
+
error=error,
|
|
421
|
+
)
|
|
422
|
+
# chain / retriever / unknown run types are not directly actionable -> skipped.
|
|
423
|
+
return spans
|
|
424
|
+
|
|
425
|
+
def _trace_task(self, runs: list[JsonObject]) -> str | None:
|
|
426
|
+
"""First human/user input across a trace's runs (ordered) -> the task text."""
|
|
427
|
+
for run in runs:
|
|
428
|
+
text = _first_user_text(run.get("inputs"))
|
|
429
|
+
if text is not None:
|
|
430
|
+
return text
|
|
431
|
+
return None
|
|
432
|
+
|
|
433
|
+
def _trace_id(self, run: JsonObject) -> str:
|
|
434
|
+
"""Grouping key: `trace_id`, else the run's own `id` (a root run may omit trace_id)."""
|
|
435
|
+
for key in ("trace_id", "id"):
|
|
436
|
+
value = run.get(key)
|
|
437
|
+
if isinstance(value, str) and value:
|
|
438
|
+
return value
|
|
439
|
+
import hashlib
|
|
440
|
+
|
|
441
|
+
return hashlib.sha256(as_text(run).encode()).hexdigest()[:32]
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
register_adapter(LangSmithAdapter())
|