world-model-optimizer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (308) hide show
  1. llm_waterfall/LICENSE +21 -0
  2. llm_waterfall/__init__.py +53 -0
  3. llm_waterfall/adapters/__init__.py +36 -0
  4. llm_waterfall/adapters/anthropic.py +105 -0
  5. llm_waterfall/adapters/aws_mantle.py +47 -0
  6. llm_waterfall/adapters/azure_openai.py +71 -0
  7. llm_waterfall/adapters/base.py +51 -0
  8. llm_waterfall/adapters/bedrock.py +309 -0
  9. llm_waterfall/adapters/openai.py +130 -0
  10. llm_waterfall/classify.py +184 -0
  11. llm_waterfall/pricing.py +110 -0
  12. llm_waterfall/py.typed +0 -0
  13. llm_waterfall/types.py +295 -0
  14. llm_waterfall/waterfall.py +255 -0
  15. wmo/__init__.py +38 -0
  16. wmo/agents/__init__.py +7 -0
  17. wmo/agents/default.py +29 -0
  18. wmo/agents/meta.py +55 -0
  19. wmo/agents/optimizer.py +55 -0
  20. wmo/agents/project.py +928 -0
  21. wmo/cli/__init__.py +5 -0
  22. wmo/cli/agent_session.py +1123 -0
  23. wmo/cli/app.py +2489 -0
  24. wmo/cli/e2b_cmds.py +212 -0
  25. wmo/cli/eval_closed_loop.py +207 -0
  26. wmo/cli/harness_app.py +1147 -0
  27. wmo/cli/harness_distill.py +659 -0
  28. wmo/cli/hosted_session.py +880 -0
  29. wmo/cli/ingest_cmd.py +165 -0
  30. wmo/cli/model_roles.py +82 -0
  31. wmo/cli/platform_cmds.py +372 -0
  32. wmo/cli/route_app.py +274 -0
  33. wmo/cli/session_state.py +243 -0
  34. wmo/cli/ui.py +1107 -0
  35. wmo/cli/workspace_sync.py +504 -0
  36. wmo/config/__init__.py +60 -0
  37. wmo/config/card.py +129 -0
  38. wmo/config/config.py +367 -0
  39. wmo/config/dotenv.py +67 -0
  40. wmo/config/settings.py +128 -0
  41. wmo/config/store.py +177 -0
  42. wmo/conftest.py +19 -0
  43. wmo/connect/__init__.py +88 -0
  44. wmo/connect/apps.py +78 -0
  45. wmo/connect/brave.py +284 -0
  46. wmo/connect/connector.py +79 -0
  47. wmo/connect/credentials.py +164 -0
  48. wmo/connect/github.py +321 -0
  49. wmo/connect/google.py +627 -0
  50. wmo/connect/notion.py +790 -0
  51. wmo/connect/oauth.py +461 -0
  52. wmo/connect/slack.py +555 -0
  53. wmo/connect/store.py +199 -0
  54. wmo/connect/types.py +156 -0
  55. wmo/core/__init__.py +21 -0
  56. wmo/core/parsing.py +281 -0
  57. wmo/core/render.py +271 -0
  58. wmo/core/text.py +40 -0
  59. wmo/core/types.py +116 -0
  60. wmo/distill/__init__.py +14 -0
  61. wmo/distill/agents.py +140 -0
  62. wmo/distill/config.py +1006 -0
  63. wmo/distill/cost.py +437 -0
  64. wmo/distill/data.py +921 -0
  65. wmo/distill/deadlines.py +254 -0
  66. wmo/distill/fake_tinker.py +734 -0
  67. wmo/distill/gate.py +122 -0
  68. wmo/distill/loop.py +3499 -0
  69. wmo/distill/renderers.py +399 -0
  70. wmo/distill/rendering.py +620 -0
  71. wmo/distill/rollouts.py +726 -0
  72. wmo/distill/samples.py +195 -0
  73. wmo/distill/store.py +829 -0
  74. wmo/distill/teacher.py +714 -0
  75. wmo/distill/tokens.py +535 -0
  76. wmo/distill/tracking.py +552 -0
  77. wmo/distill/tripwire.py +411 -0
  78. wmo/distill/xtoken/byte_offsets.py +152 -0
  79. wmo/distill/xtoken/chunks.py +457 -0
  80. wmo/distill/xtoken/prompt_logprobs.py +475 -0
  81. wmo/distill/xtoken/teacher_render.py +346 -0
  82. wmo/engine/__init__.py +28 -0
  83. wmo/engine/autoconfig.py +367 -0
  84. wmo/engine/build.py +346 -0
  85. wmo/engine/demo.py +77 -0
  86. wmo/engine/eval_suites.py +245 -0
  87. wmo/engine/grounding.py +491 -0
  88. wmo/engine/knowledge.py +291 -0
  89. wmo/engine/loader.py +36 -0
  90. wmo/engine/play.py +92 -0
  91. wmo/engine/prompts.py +99 -0
  92. wmo/engine/replay.py +443 -0
  93. wmo/engine/reporting.py +58 -0
  94. wmo/engine/workspace.py +468 -0
  95. wmo/engine/world_model.py +568 -0
  96. wmo/env/__init__.py +22 -0
  97. wmo/env/base.py +121 -0
  98. wmo/env/closed_loop.py +229 -0
  99. wmo/env/episode.py +107 -0
  100. wmo/env/llm_agent.py +93 -0
  101. wmo/env/scenarios.py +73 -0
  102. wmo/evals/__init__.py +52 -0
  103. wmo/evals/agreement.py +110 -0
  104. wmo/evals/base.py +45 -0
  105. wmo/evals/closed_loop.py +480 -0
  106. wmo/evals/failover.py +96 -0
  107. wmo/evals/gold.py +127 -0
  108. wmo/evals/grid.py +394 -0
  109. wmo/evals/grid_plot.py +205 -0
  110. wmo/evals/harbor/__init__.py +27 -0
  111. wmo/evals/harbor/agent.py +573 -0
  112. wmo/evals/harbor/ctrf.py +171 -0
  113. wmo/evals/harbor/e2b_environment.py +587 -0
  114. wmo/evals/harbor/e2b_template_policy.py +144 -0
  115. wmo/evals/harbor/scorer.py +875 -0
  116. wmo/evals/harbor/tasks.py +140 -0
  117. wmo/evals/open_loop.py +194 -0
  118. wmo/evals/tasks.py +53 -0
  119. wmo/harness/__init__.py +51 -0
  120. wmo/harness/code_runtime.py +288 -0
  121. wmo/harness/create.py +1191 -0
  122. wmo/harness/delta.py +220 -0
  123. wmo/harness/doc.py +556 -0
  124. wmo/harness/e2b_ledger.py +342 -0
  125. wmo/harness/e2b_reap.py +476 -0
  126. wmo/harness/e2b_sandbox.py +350 -0
  127. wmo/harness/environment.py +35 -0
  128. wmo/harness/live_session.py +543 -0
  129. wmo/harness/mutate.py +343 -0
  130. wmo/harness/pi_e2b.py +1710 -0
  131. wmo/harness/pi_entry/entry.ts +268 -0
  132. wmo/harness/pi_entry/runner_frames.ts +92 -0
  133. wmo/harness/pi_entry/runner_live.ts +587 -0
  134. wmo/harness/pi_entry/runner_service.ts +270 -0
  135. wmo/harness/pi_entry/runner_stdio.ts +374 -0
  136. wmo/harness/pi_entry/runner_termination.ts +142 -0
  137. wmo/harness/pi_local.py +262 -0
  138. wmo/harness/pi_runtime.py +495 -0
  139. wmo/harness/pi_vendor.py +65 -0
  140. wmo/harness/population.py +509 -0
  141. wmo/harness/project_proposer.py +569 -0
  142. wmo/harness/proposer.py +977 -0
  143. wmo/harness/runner_link.py +619 -0
  144. wmo/harness/runtime.py +389 -0
  145. wmo/harness/scoring.py +247 -0
  146. wmo/harness/skills.py +116 -0
  147. wmo/harness/source_tree.py +319 -0
  148. wmo/harness/store.py +176 -0
  149. wmo/harness/tools.py +105 -0
  150. wmo/harness/vendor/manifest.sha256 +58 -0
  151. wmo/harness/vendor/pi-agent/CHANGELOG.md +556 -0
  152. wmo/harness/vendor/pi-agent/LICENSE +21 -0
  153. wmo/harness/vendor/pi-agent/README.md +488 -0
  154. wmo/harness/vendor/pi-agent/VENDOR.md +39 -0
  155. wmo/harness/vendor/pi-agent/docs/agent-harness.md +486 -0
  156. wmo/harness/vendor/pi-agent/docs/durable-harness.md +212 -0
  157. wmo/harness/vendor/pi-agent/docs/hooks.md +445 -0
  158. wmo/harness/vendor/pi-agent/docs/models.md +966 -0
  159. wmo/harness/vendor/pi-agent/docs/observability.md +376 -0
  160. wmo/harness/vendor/pi-agent/package.json +60 -0
  161. wmo/harness/vendor/pi-agent/src/agent-loop.ts +748 -0
  162. wmo/harness/vendor/pi-agent/src/agent.ts +575 -0
  163. wmo/harness/vendor/pi-agent/src/harness/agent-harness.ts +1029 -0
  164. wmo/harness/vendor/pi-agent/src/harness/compaction/branch-summarization.ts +261 -0
  165. wmo/harness/vendor/pi-agent/src/harness/compaction/compaction.ts +747 -0
  166. wmo/harness/vendor/pi-agent/src/harness/compaction/utils.ts +144 -0
  167. wmo/harness/vendor/pi-agent/src/harness/env/nodejs.ts +550 -0
  168. wmo/harness/vendor/pi-agent/src/harness/messages.ts +164 -0
  169. wmo/harness/vendor/pi-agent/src/harness/prompt-templates.ts +267 -0
  170. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-repo.ts +177 -0
  171. wmo/harness/vendor/pi-agent/src/harness/session/jsonl-storage.ts +293 -0
  172. wmo/harness/vendor/pi-agent/src/harness/session/memory-repo.ts +50 -0
  173. wmo/harness/vendor/pi-agent/src/harness/session/memory-storage.ts +131 -0
  174. wmo/harness/vendor/pi-agent/src/harness/session/repo-utils.ts +51 -0
  175. wmo/harness/vendor/pi-agent/src/harness/session/session.ts +267 -0
  176. wmo/harness/vendor/pi-agent/src/harness/session/uuid.ts +54 -0
  177. wmo/harness/vendor/pi-agent/src/harness/skills.ts +375 -0
  178. wmo/harness/vendor/pi-agent/src/harness/system-prompt.ts +34 -0
  179. wmo/harness/vendor/pi-agent/src/harness/types.ts +836 -0
  180. wmo/harness/vendor/pi-agent/src/harness/utils/shell-output.ts +135 -0
  181. wmo/harness/vendor/pi-agent/src/harness/utils/truncate.ts +344 -0
  182. wmo/harness/vendor/pi-agent/src/index.ts +44 -0
  183. wmo/harness/vendor/pi-agent/src/node.ts +2 -0
  184. wmo/harness/vendor/pi-agent/src/proxy.ts +367 -0
  185. wmo/harness/vendor/pi-agent/src/types.ts +428 -0
  186. wmo/harness/vendor/pi-agent/test/agent-loop.test.ts +1351 -0
  187. wmo/harness/vendor/pi-agent/test/agent.test.ts +699 -0
  188. wmo/harness/vendor/pi-agent/test/e2e.test.ts +404 -0
  189. wmo/harness/vendor/pi-agent/test/harness/agent-harness-stream.test.ts +213 -0
  190. wmo/harness/vendor/pi-agent/test/harness/agent-harness.test.ts +608 -0
  191. wmo/harness/vendor/pi-agent/test/harness/compaction.test.ts +655 -0
  192. wmo/harness/vendor/pi-agent/test/harness/nodejs-env.test.ts +321 -0
  193. wmo/harness/vendor/pi-agent/test/harness/prompt-templates.test.ts +90 -0
  194. wmo/harness/vendor/pi-agent/test/harness/repo.test.ts +68 -0
  195. wmo/harness/vendor/pi-agent/test/harness/resource-formatting.test.ts +24 -0
  196. wmo/harness/vendor/pi-agent/test/harness/session-test-utils.ts +55 -0
  197. wmo/harness/vendor/pi-agent/test/harness/session-uuid.test.ts +50 -0
  198. wmo/harness/vendor/pi-agent/test/harness/session.test.ts +156 -0
  199. wmo/harness/vendor/pi-agent/test/harness/skills.test.ts +116 -0
  200. wmo/harness/vendor/pi-agent/test/harness/storage.test.ts +299 -0
  201. wmo/harness/vendor/pi-agent/test/harness/system-prompt.test.ts +66 -0
  202. wmo/harness/vendor/pi-agent/test/harness/truncate.test.ts +169 -0
  203. wmo/harness/vendor/pi-agent/test/scratch/simple.ts +72 -0
  204. wmo/harness/vendor/pi-agent/test/utils/calculate.ts +32 -0
  205. wmo/harness/vendor/pi-agent/test/utils/get-current-time.ts +46 -0
  206. wmo/harness/vendor/pi-agent/tsconfig.build.json +13 -0
  207. wmo/harness/vendor/pi-agent/vitest.config.ts +19 -0
  208. wmo/harness/vendor/pi-agent/vitest.harness.config.ts +28 -0
  209. wmo/harness/vendor/vendor_pi.sh +59 -0
  210. wmo/harness/workspace_patch.py +270 -0
  211. wmo/ingest/__init__.py +47 -0
  212. wmo/ingest/adapter.py +72 -0
  213. wmo/ingest/base.py +114 -0
  214. wmo/ingest/braintrust.py +339 -0
  215. wmo/ingest/detect.py +126 -0
  216. wmo/ingest/langfuse.py +291 -0
  217. wmo/ingest/langsmith.py +444 -0
  218. wmo/ingest/mastra.py +330 -0
  219. wmo/ingest/messages.py +170 -0
  220. wmo/ingest/normalize.py +679 -0
  221. wmo/ingest/otel_genai.py +69 -0
  222. wmo/ingest/otel_writer.py +100 -0
  223. wmo/ingest/phoenix.py +150 -0
  224. wmo/ingest/postgres.py +246 -0
  225. wmo/ingest/posthog.py +320 -0
  226. wmo/ingest/quality.py +28 -0
  227. wmo/ingest/stream.py +209 -0
  228. wmo/ingest/testdata/sample_otlp.json +60 -0
  229. wmo/ingest/testdata/sample_spans.jsonl +3 -0
  230. wmo/optimize/__init__.py +25 -0
  231. wmo/optimize/base.py +143 -0
  232. wmo/optimize/gepa.py +806 -0
  233. wmo/optimize/judge.py +262 -0
  234. wmo/optimize/judge_quality.py +359 -0
  235. wmo/optimize/knn.py +468 -0
  236. wmo/optimize/numeric.py +152 -0
  237. wmo/optimize/outcomes.py +103 -0
  238. wmo/optimize/policy.py +669 -0
  239. wmo/optimize/report.py +231 -0
  240. wmo/optimize/reward.py +129 -0
  241. wmo/optimize/routing.py +373 -0
  242. wmo/platform/__init__.py +6 -0
  243. wmo/platform/auth.py +115 -0
  244. wmo/platform/client.py +551 -0
  245. wmo/platform/credentials.py +126 -0
  246. wmo/platform/transfer.py +158 -0
  247. wmo/providers/__init__.py +40 -0
  248. wmo/providers/_bedrock_chat.py +155 -0
  249. wmo/providers/_openai_common.py +182 -0
  250. wmo/providers/_responses_common.py +472 -0
  251. wmo/providers/anthropic.py +134 -0
  252. wmo/providers/azure_openai.py +296 -0
  253. wmo/providers/base.py +300 -0
  254. wmo/providers/bedrock.py +312 -0
  255. wmo/providers/models.py +205 -0
  256. wmo/providers/openai.py +143 -0
  257. wmo/providers/openai_responses.py +240 -0
  258. wmo/providers/pool.py +170 -0
  259. wmo/providers/registry.py +73 -0
  260. wmo/providers/retry.py +151 -0
  261. wmo/providers/tinker.py +936 -0
  262. wmo/providers/waterfall.py +336 -0
  263. wmo/research/__init__.py +81 -0
  264. wmo/research/ablation.py +133 -0
  265. wmo/research/concurrency_plot.py +523 -0
  266. wmo/research/concurrency_run.py +240 -0
  267. wmo/research/concurrency_scaling.py +270 -0
  268. wmo/research/gepa_scaling.py +274 -0
  269. wmo/research/pipeline.py +198 -0
  270. wmo/research/scaling_split.py +82 -0
  271. wmo/research/scenario_fidelity.py +198 -0
  272. wmo/research/scenario_recovery.py +92 -0
  273. wmo/research/seed_stability.py +90 -0
  274. wmo/research/trace_scaling.py +348 -0
  275. wmo/retrieval/__init__.py +6 -0
  276. wmo/retrieval/embedders.py +105 -0
  277. wmo/retrieval/leakfree.py +52 -0
  278. wmo/retrieval/retriever.py +173 -0
  279. wmo/scenarios/__init__.py +58 -0
  280. wmo/scenarios/builder.py +152 -0
  281. wmo/scenarios/mining/__init__.py +27 -0
  282. wmo/scenarios/mining/clustering.py +171 -0
  283. wmo/scenarios/mining/facets.py +226 -0
  284. wmo/scenarios/mining/selection.py +220 -0
  285. wmo/scenarios/synthesis/__init__.py +6 -0
  286. wmo/scenarios/synthesis/scenario_set.py +63 -0
  287. wmo/scenarios/synthesis/synthesizer.py +85 -0
  288. wmo/scenarios/verification/__init__.py +17 -0
  289. wmo/scenarios/verification/judge.py +97 -0
  290. wmo/scenarios/verification/verify.py +135 -0
  291. wmo/serving/__init__.py +5 -0
  292. wmo/serving/builds.py +451 -0
  293. wmo/serving/chat.py +878 -0
  294. wmo/serving/endpoint_config.py +64 -0
  295. wmo/serving/savings.py +250 -0
  296. wmo/serving/server.py +553 -0
  297. wmo/serving/traces_source.py +206 -0
  298. wmo/telemetry.py +213 -0
  299. wmo/tracking/__init__.py +36 -0
  300. wmo/tracking/clock.py +24 -0
  301. wmo/tracking/metered.py +125 -0
  302. wmo/tracking/pricing.py +99 -0
  303. wmo/tracking/store.py +31 -0
  304. wmo/tracking/tracker.py +149 -0
  305. world_model_optimizer-0.2.0.dist-info/METADATA +203 -0
  306. world_model_optimizer-0.2.0.dist-info/RECORD +308 -0
  307. world_model_optimizer-0.2.0.dist-info/WHEEL +4 -0
  308. world_model_optimizer-0.2.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,444 @@
1
+ """LangSmith adapter: turn a LangSmith run-tree export into `Trace`s.
2
+
3
+ LangSmith (LangChain's tracing product) does NOT export OTLP spans. It models a trace as a tree of
4
+ **runs**, where each run is a node typed by `run_type`
5
+ (`tool | chain | llm | retriever | embedding | prompt | parser`). An exported run (from
6
+ `POST /api/v1/runs/query` — the list endpoint is a POST with a JSON filter body, NOT a GET — or the
7
+ SDK `Client.list_runs`) looks roughly like:
8
+
9
+ {"id": "<run uuid>", "trace_id": "<trace grouping uuid>", "parent_run_id": "<uuid|null>",
10
+ "run_type": "llm" | "tool" | "chain" | "retriever" | "embedding" | "prompt" | "parser",
11
+ "name": "ChatOpenAI",
12
+ "inputs": {...}, "outputs": {...},
13
+ "start_time": "2026-01-01T00:00:00.000000", "end_time": "...",
14
+ "error": null | "<traceback str>", "extra": {...}}
15
+
16
+ Because this is not an OTLP/OpenInference span shape, the adapter overrides `spans_from_payload`
17
+ (like `wmo.ingest.messages` / `wmo.ingest.langfuse`) and emits `SpanRecord`s in the **OTel-GenAI
18
+ vocabulary** so the shared classifier/normalizer (`wmo.ingest.normalize`) does the pairing /
19
+ state / metadata work. Each run maps to zero or more spans:
20
+
21
+ - `run_type == "llm"`: if the outputs carry tool calls, emit one `chat` ACTION span per call with
22
+ `{"gen_ai.tool.name", "gen_ai.tool.call.arguments"}`; otherwise emit one plain `chat` message
23
+ span with `{"gen_ai.completion": <text>}`.
24
+ - `run_type == "tool"`: emit one `execute_tool` RESULT span with
25
+ `{"gen_ai.operation.name": "execute_tool", "gen_ai.tool.name": <name>,
26
+ "gen_ai.tool.message": <outputs as text>}`. The normalizer pairs it with the preceding action
27
+ span (the llm run's tool call), backfilling name/args from here if the action lacked them.
28
+ - `run_type in {"chain", "retriever", ...}`: skipped (not directly actionable). A run we cannot
29
+ interpret is skipped rather than crashing the ingest.
30
+
31
+ Where LangChain hides tool calls (these paths are version-dependent and best-effort; we dig all of
32
+ them and degrade gracefully):
33
+ - `outputs["generations"][i]["message"]["kwargs"]["tool_calls"]` (LCEL ChatGeneration dump)
34
+ - `outputs["generations"][i]["message"]["kwargs"]["additional_kwargs"]["tool_calls"]` (OpenAI)
35
+ - `outputs["generations"][i][j]["message"]...` (nested list-of-lists generations)
36
+ - `outputs["tool_calls"]` / `outputs["message"]...` (flatter dumps)
37
+ A LangChain `tool_calls` entry is either the normalized shape `{"name", "args": {...}, "id"}` or the
38
+ OpenAI shape `{"id", "function": {"name", "arguments": "<json str>"}}`; both are handled.
39
+
40
+ The LLM completion text is dug from `generations[i].text`, `generations[i].message.kwargs.content`,
41
+ or `outputs["output"|"content"|"text"]`. A tool run's output is dug from
42
+ `outputs["output"]`, else the whole `outputs`. The task (`gen_ai.prompt`) is the first human/user
43
+ input, dug from a run's `inputs` (chat `messages`, a `messages` list, or `inputs["input"]`).
44
+
45
+ Ordering: runs are ordered by `start_time` (ISO-8601 -> epoch microseconds); only monotonicity
46
+ within a trace matters (the normalizer sorts by `start_nano`), so an absent/unparseable timestamp
47
+ degrades to the run's list index.
48
+
49
+ Accepted file shapes (`from_file`): a single run object, a JSON array of runs, a `{"runs": [...]}`
50
+ wrapper, or JSONL (one run per line). Grouping is by `trace_id` (falling back to `id` for a root
51
+ run that omits it).
52
+
53
+ Pull: `from_vendor` fetches runs live over the REST API with plain httpx (no SDK); see
54
+ `_pull_payloads`. File exports remain fully supported via `from_file`.
55
+ """
56
+
57
+ from __future__ import annotations
58
+
59
+ import os
60
+
61
+ import httpx
62
+ from pydantic import JsonValue
63
+
64
+ from wmo.core.types import JsonObject
65
+ from wmo.ingest.adapter import VendorPull, register_adapter
66
+ from wmo.ingest.base import BaseTraceAdapter
67
+ from wmo.ingest.normalize import SpanRecord, as_text, iso_to_ordinal
68
+
69
+
70
+ def _as_str(value: JsonValue) -> str:
71
+ return value if isinstance(value, str) else ""
72
+
73
+
74
+ def _is_error(run: JsonObject) -> bool:
75
+ """A non-null/non-empty `error` marks the run as failed; an empty string is NOT an error.
76
+
77
+ Some LangSmith dumps set `error: ""` on a successful run, so a bare `is not None` check would
78
+ misclassify it as a failure.
79
+ """
80
+ error = run.get("error")
81
+ if error is None:
82
+ return False
83
+ if isinstance(error, str):
84
+ return bool(error.strip())
85
+ return bool(error)
86
+
87
+
88
+ def _start_ordinal(run: JsonObject, fallback: int) -> int:
89
+ """Monotonic ordering key from the run's `start_time` (shared helper; UTC-safe)."""
90
+ return iso_to_ordinal(run.get("start_time"), fallback)
91
+
92
+
93
+ def _generations(outputs: JsonObject) -> list[JsonObject]:
94
+ """Flatten `outputs["generations"]` (a list, or a list-of-lists) into generation dicts."""
95
+ raw = outputs.get("generations")
96
+ if not isinstance(raw, list):
97
+ return []
98
+ flat: list[JsonObject] = []
99
+ for item in raw:
100
+ if isinstance(item, dict):
101
+ flat.append(item)
102
+ elif isinstance(item, list): # list-of-lists (one inner list per prompt)
103
+ flat.extend(g for g in item if isinstance(g, dict))
104
+ return flat
105
+
106
+
107
+ def _message_kwargs(generation: JsonObject) -> JsonObject:
108
+ """The `message.kwargs` dict of a ChatGeneration dump (empty when absent)."""
109
+ message = generation.get("message")
110
+ if isinstance(message, dict):
111
+ kwargs = message.get("kwargs")
112
+ if isinstance(kwargs, dict):
113
+ return kwargs
114
+ return {}
115
+
116
+
117
+ def _tool_calls_in(container: JsonObject) -> list[JsonObject]:
118
+ """Pull `tool_calls` from a dict, checking `tool_calls` then `additional_kwargs.tool_calls`."""
119
+ calls: list[JsonObject] = []
120
+ raw = container.get("tool_calls")
121
+ if isinstance(raw, list):
122
+ calls.extend(tc for tc in raw if isinstance(tc, dict))
123
+ extra = container.get("additional_kwargs")
124
+ if isinstance(extra, dict):
125
+ raw_extra = extra.get("tool_calls")
126
+ if isinstance(raw_extra, list):
127
+ calls.extend(tc for tc in raw_extra if isinstance(tc, dict))
128
+ return calls
129
+
130
+
131
+ def _llm_tool_calls(outputs: JsonObject) -> list[JsonObject]:
132
+ """Dig tool calls out of an llm run's `outputs`, across the common LangChain locations."""
133
+ calls: list[JsonObject] = []
134
+ for generation in _generations(outputs):
135
+ calls.extend(_tool_calls_in(_message_kwargs(generation)))
136
+ # Flatter dumps: outputs.tool_calls or outputs.message.kwargs.tool_calls.
137
+ calls.extend(_tool_calls_in(outputs))
138
+ message = outputs.get("message")
139
+ if isinstance(message, dict):
140
+ kwargs = message.get("kwargs")
141
+ if isinstance(kwargs, dict):
142
+ calls.extend(_tool_calls_in(kwargs))
143
+ return calls
144
+
145
+
146
+ def _call_name_args(tool_call: JsonObject) -> tuple[str, str]:
147
+ """(name, raw-arguments-json) from a LangChain-normalized or OpenAI-shaped tool call."""
148
+ fn = tool_call.get("function")
149
+ if isinstance(fn, dict): # OpenAI shape: {"function": {"name", "arguments": "<json str>"}}
150
+ name = fn.get("name")
151
+ args = fn.get("arguments")
152
+ else: # LangChain-normalized shape: {"name", "args": {...}, "id"}
153
+ name = tool_call.get("name")
154
+ args = tool_call.get("args")
155
+ if args is None:
156
+ args = tool_call.get("arguments")
157
+ name_s = name if isinstance(name, str) else ""
158
+ args_s = args if isinstance(args, str) else as_text(args)
159
+ return name_s, args_s
160
+
161
+
162
+ def _llm_completion(outputs: JsonObject) -> str:
163
+ """Dig the assistant text out of an llm run's `outputs` (best-effort across dump shapes)."""
164
+ for generation in _generations(outputs):
165
+ text = generation.get("text")
166
+ if isinstance(text, str) and text:
167
+ return text
168
+ content = _message_kwargs(generation).get("content")
169
+ if isinstance(content, str) and content:
170
+ return content
171
+ for key in ("output", "content", "text"):
172
+ value = outputs.get(key)
173
+ if isinstance(value, str) and value:
174
+ return value
175
+ return as_text(outputs)
176
+
177
+
178
+ def _tool_output_text(outputs: JsonValue) -> str:
179
+ """A tool run's result text: `outputs["output"]` if present, else the whole outputs."""
180
+ if isinstance(outputs, dict):
181
+ value = outputs.get("output")
182
+ if value is not None:
183
+ return as_text(value)
184
+ return as_text(outputs)
185
+
186
+
187
+ def _tool_run_name(run: JsonObject) -> str:
188
+ """A tool name for a `tool` run: explicit fields first, else the run name."""
189
+ for key in ("tool_name", "name"):
190
+ value = run.get(key)
191
+ if isinstance(value, str) and value:
192
+ return value
193
+ return ""
194
+
195
+
196
+ def _first_user_text(inputs: JsonValue) -> str | None:
197
+ """Dig the first human/user input out of a run's `inputs` (the trace task)."""
198
+ if isinstance(inputs, str):
199
+ return inputs or None
200
+ if not isinstance(inputs, dict):
201
+ return None
202
+ messages = inputs.get("messages")
203
+ text = _first_user_in_messages(messages)
204
+ if text is not None:
205
+ return text
206
+ for key in ("input", "question", "query", "text"):
207
+ value = inputs.get(key)
208
+ if isinstance(value, str) and value:
209
+ return value
210
+ return None
211
+
212
+
213
+ def _first_user_in_messages(messages: JsonValue) -> str | None:
214
+ """First human/user message content in a (possibly nested) LangChain messages list."""
215
+ if not isinstance(messages, list):
216
+ return None
217
+ for message in messages:
218
+ # Nested list-of-lists (one inner list per prompt) — recurse.
219
+ if isinstance(message, list):
220
+ found = _first_user_in_messages(message)
221
+ if found is not None:
222
+ return found
223
+ continue
224
+ if not isinstance(message, dict):
225
+ continue
226
+ role = _message_role(message)
227
+ content = _message_content(message)
228
+ if role in {"human", "user"} and content:
229
+ return content
230
+ return None
231
+
232
+
233
+ def _message_role(message: JsonObject) -> str:
234
+ """Role of a LangChain/OpenAI message dict (`role`, `type`, or a serialized class id)."""
235
+ for key in ("role", "type"):
236
+ value = message.get(key)
237
+ if isinstance(value, str) and value:
238
+ return value.lower()
239
+ # Serialized LangChain message: {"id": [..., "HumanMessage"], "kwargs": {...}}.
240
+ cid = message.get("id")
241
+ if isinstance(cid, list) and cid:
242
+ last = cid[-1]
243
+ if isinstance(last, str):
244
+ return last.lower().removesuffix("message")
245
+ return ""
246
+
247
+
248
+ def _message_content(message: JsonObject) -> str:
249
+ """Text content of a LangChain/OpenAI message dict (top-level or under `kwargs`)."""
250
+ content = message.get("content")
251
+ if isinstance(content, str) and content:
252
+ return content
253
+ kwargs = message.get("kwargs")
254
+ if isinstance(kwargs, dict):
255
+ nested = kwargs.get("content")
256
+ if isinstance(nested, str) and nested:
257
+ return nested
258
+ return ""
259
+
260
+
261
+ _API_KEY_ENVS = ("LANGCHAIN_API_KEY", "LANGSMITH_API_KEY")
262
+ _ENDPOINT_ENV = "LANGCHAIN_ENDPOINT"
263
+ _DEFAULT_ENDPOINT = "https://api.smith.langchain.com"
264
+ _PAGE_SIZE = 100
265
+ # Backstop on unbounded pulls (no --limit): runs, not traces, are what the API pages by.
266
+ _MAX_RUNS = 2000
267
+
268
+
269
+ class LangSmithAdapter(BaseTraceAdapter):
270
+ """Map a LangSmith run-tree export into normalized `Trace`s. No SDK; pure JSON."""
271
+
272
+ name = "langsmith"
273
+
274
+ def _pull_payloads(self, pull: VendorPull) -> list[JsonValue]:
275
+ """Fetch runs live from the LangSmith REST API (plain httpx, no SDK).
276
+
277
+ `pull.api_key` (else `$LANGCHAIN_API_KEY`/`$LANGSMITH_API_KEY`) auths; `pull.project` is
278
+ the project (session) UUID passed to `POST /api/v1/runs/query`: the list endpoint is a
279
+ POST with a JSON filter body). Pagination follows `cursors.next` until exhausted, `--limit`
280
+ traces are covered, or a run-count backstop trips.
281
+ """
282
+ api_key = pull.api_key or next(
283
+ (value for env in _API_KEY_ENVS if (value := os.environ.get(env))), None
284
+ )
285
+ if not api_key:
286
+ raise ValueError(
287
+ f"langsmith pull needs an API key: pass --api-key or set ${_API_KEY_ENVS[0]}"
288
+ )
289
+ endpoint = (os.environ.get(_ENDPOINT_ENV) or _DEFAULT_ENDPOINT).rstrip("/")
290
+ base_body: JsonObject = {"limit": _PAGE_SIZE}
291
+ if pull.project:
292
+ base_body["session"] = [pull.project]
293
+ if pull.since is not None:
294
+ base_body["start_time"] = pull.since
295
+
296
+ runs: list[JsonValue] = []
297
+ trace_ids: set[str] = set()
298
+ cursor: str | None = None
299
+ while True:
300
+ body = dict(base_body)
301
+ # The cursor key is present only when following a page: cursor-based APIs
302
+ # commonly reject an explicit null on the first request.
303
+ if cursor is not None:
304
+ body["cursor"] = cursor
305
+ resp = httpx.post(
306
+ f"{endpoint}/api/v1/runs/query",
307
+ headers={"x-api-key": api_key},
308
+ json=body,
309
+ timeout=60.0,
310
+ )
311
+ resp.raise_for_status()
312
+ page = resp.json()
313
+ page_runs = page.get("runs") if isinstance(page, dict) else None
314
+ if not isinstance(page_runs, list) or not page_runs:
315
+ break
316
+ for run in page_runs:
317
+ if isinstance(run, dict):
318
+ runs.append(run)
319
+ trace_ids.add(self._trace_id(run))
320
+ if pull.limit is not None and len(trace_ids) >= pull.limit:
321
+ break
322
+ if len(runs) >= _MAX_RUNS:
323
+ break
324
+ cursors = page.get("cursors")
325
+ cursor = cursors.get("next") if isinstance(cursors, dict) else None
326
+ if not isinstance(cursor, str) or not cursor:
327
+ break
328
+ return [{"runs": runs}]
329
+
330
+ def spans_from_payload(self, payload: JsonValue) -> list[SpanRecord]:
331
+ """Map one payload (single run, list, `{runs:[...]}`) to ordered `SpanRecord`s by trace."""
332
+ runs = self._runs(payload)
333
+ # Group by trace_id so we can assign a per-trace monotonic ordinal and set the task once.
334
+ by_trace: dict[str, list[JsonObject]] = {}
335
+ for run in runs:
336
+ by_trace.setdefault(self._trace_id(run), []).append(run)
337
+
338
+ spans: list[SpanRecord] = []
339
+ for trace_id, trace_runs in by_trace.items():
340
+ spans.extend(self._spans_for_trace(trace_id, trace_runs))
341
+ return spans
342
+
343
+ def _runs(self, payload: JsonValue) -> list[JsonObject]:
344
+ """Normalize a payload into a flat list of run objects.
345
+
346
+ Accepts a single run, a bare list of runs, or a `{"runs": [...]}` wrapper.
347
+ """
348
+ if isinstance(payload, list):
349
+ out: list[JsonObject] = []
350
+ for item in payload:
351
+ out.extend(self._runs(item))
352
+ return out
353
+ if not isinstance(payload, dict):
354
+ return []
355
+ wrapped = payload.get("runs")
356
+ if isinstance(wrapped, list) and "run_type" not in payload:
357
+ out = []
358
+ for item in wrapped:
359
+ out.extend(self._runs(item))
360
+ return out
361
+ # A run object: it has an id (and usually run_type). Be permissive.
362
+ if "id" in payload or "run_type" in payload:
363
+ return [payload]
364
+ return []
365
+
366
+ def _spans_for_trace(self, trace_id: str, runs: list[JsonObject]) -> list[SpanRecord]:
367
+ # Order by start_time; ties (or absent timestamps) keep input order via the index fallback.
368
+ indexed = list(enumerate(runs))
369
+ indexed.sort(key=lambda pair: (_start_ordinal(pair[1], pair[0]), pair[0]))
370
+
371
+ task = self._trace_task([run for _, run in indexed])
372
+
373
+ spans: list[SpanRecord] = []
374
+ ordinal = 0
375
+
376
+ def emit(attrs: JsonObject, *, tool: bool, error: bool = False) -> None:
377
+ nonlocal ordinal
378
+ if ordinal == 0 and task is not None:
379
+ attrs.setdefault("gen_ai.prompt", task)
380
+ spans.append(
381
+ SpanRecord(
382
+ trace_id=trace_id,
383
+ span_id=f"{trace_id[:12]}{ordinal:06x}{'t' if tool else 'a'}",
384
+ name="execute_tool" if tool else "chat",
385
+ start_nano=ordinal,
386
+ attributes={
387
+ "gen_ai.operation.name": "execute_tool" if tool else "chat",
388
+ **attrs,
389
+ },
390
+ status_error=error,
391
+ )
392
+ )
393
+ ordinal += 1
394
+
395
+ for _, run in indexed:
396
+ run_type = _as_str(run.get("run_type")).lower()
397
+ error = _is_error(run)
398
+ outputs = run.get("outputs")
399
+ out_obj: JsonObject = outputs if isinstance(outputs, dict) else {}
400
+
401
+ if run_type == "llm":
402
+ calls = _llm_tool_calls(out_obj)
403
+ if calls:
404
+ for tool_call in calls:
405
+ name, args = _call_name_args(tool_call)
406
+ emit(
407
+ {"gen_ai.tool.name": name, "gen_ai.tool.call.arguments": args},
408
+ tool=False,
409
+ error=error,
410
+ )
411
+ else:
412
+ emit({"gen_ai.completion": _llm_completion(out_obj)}, tool=False, error=error)
413
+ elif run_type == "tool":
414
+ emit(
415
+ {
416
+ "gen_ai.tool.name": _tool_run_name(run),
417
+ "gen_ai.tool.message": _tool_output_text(outputs),
418
+ },
419
+ tool=True,
420
+ error=error,
421
+ )
422
+ # chain / retriever / unknown run types are not directly actionable -> skipped.
423
+ return spans
424
+
425
+ def _trace_task(self, runs: list[JsonObject]) -> str | None:
426
+ """First human/user input across a trace's runs (ordered) -> the task text."""
427
+ for run in runs:
428
+ text = _first_user_text(run.get("inputs"))
429
+ if text is not None:
430
+ return text
431
+ return None
432
+
433
+ def _trace_id(self, run: JsonObject) -> str:
434
+ """Grouping key: `trace_id`, else the run's own `id` (a root run may omit trace_id)."""
435
+ for key in ("trace_id", "id"):
436
+ value = run.get(key)
437
+ if isinstance(value, str) and value:
438
+ return value
439
+ import hashlib
440
+
441
+ return hashlib.sha256(as_text(run).encode()).hexdigest()[:32]
442
+
443
+
444
+ register_adapter(LangSmithAdapter())