synth-ai 0.2.6.dev1__py3-none-any.whl → 0.4.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (738) hide show
  1. synth_ai/__init__.py +44 -24
  2. synth_ai/__main__.py +30 -3
  3. synth_ai/cli/__init__.py +103 -48
  4. synth_ai/cli/__main__.py +42 -0
  5. synth_ai/cli/_internal/__init__.py +5 -0
  6. synth_ai/cli/_internal/modal_wrapper.py +31 -0
  7. synth_ai/cli/_internal/storage.py +20 -0
  8. synth_ai/cli/_internal/typer_patch.py +47 -0
  9. synth_ai/cli/_internal/validate_task_app.py +29 -0
  10. synth_ai/cli/agents/__init__.py +17 -0
  11. synth_ai/cli/agents/claude.py +77 -0
  12. synth_ai/cli/agents/codex.py +265 -0
  13. synth_ai/cli/agents/opencode.py +253 -0
  14. synth_ai/cli/commands/__init__.py +18 -0
  15. synth_ai/cli/commands/artifacts/__init__.py +13 -0
  16. synth_ai/cli/commands/artifacts/client.py +119 -0
  17. synth_ai/cli/commands/artifacts/config.py +57 -0
  18. synth_ai/cli/commands/artifacts/core.py +24 -0
  19. synth_ai/cli/commands/artifacts/download.py +188 -0
  20. synth_ai/cli/commands/artifacts/export.py +186 -0
  21. synth_ai/cli/commands/artifacts/list.py +156 -0
  22. synth_ai/cli/commands/artifacts/parsing.py +250 -0
  23. synth_ai/cli/commands/artifacts/show.py +336 -0
  24. synth_ai/cli/commands/demo/__init__.py +3 -0
  25. synth_ai/cli/commands/demo/core.py +153 -0
  26. synth_ai/cli/commands/eval/__init__.py +10 -0
  27. synth_ai/cli/commands/eval/config.py +338 -0
  28. synth_ai/cli/commands/eval/core.py +256 -0
  29. synth_ai/cli/commands/eval/runner.py +704 -0
  30. synth_ai/cli/commands/eval/validation.py +60 -0
  31. synth_ai/cli/commands/filter/__init__.py +12 -0
  32. synth_ai/cli/commands/filter/core.py +424 -0
  33. synth_ai/cli/commands/filter/errors.py +55 -0
  34. synth_ai/cli/commands/filter/validation.py +77 -0
  35. synth_ai/cli/commands/help/__init__.py +185 -0
  36. synth_ai/cli/commands/help/core.py +72 -0
  37. synth_ai/cli/commands/scan/__init__.py +19 -0
  38. synth_ai/cli/commands/scan/cloudflare_scanner.py +403 -0
  39. synth_ai/cli/commands/scan/core.py +344 -0
  40. synth_ai/cli/commands/scan/health_checker.py +242 -0
  41. synth_ai/cli/commands/scan/local_scanner.py +278 -0
  42. synth_ai/cli/commands/scan/models.py +83 -0
  43. synth_ai/cli/commands/smoke/__init__.py +7 -0
  44. synth_ai/cli/commands/smoke/core.py +1428 -0
  45. synth_ai/cli/commands/status/__init__.py +3 -0
  46. synth_ai/cli/commands/status/client.py +91 -0
  47. synth_ai/cli/commands/status/config.py +12 -0
  48. synth_ai/cli/commands/status/errors.py +11 -0
  49. synth_ai/cli/commands/status/subcommands/__init__.py +3 -0
  50. synth_ai/cli/commands/status/subcommands/config.py +13 -0
  51. synth_ai/cli/commands/status/subcommands/files.py +34 -0
  52. synth_ai/cli/commands/status/subcommands/jobs.py +51 -0
  53. synth_ai/cli/commands/status/subcommands/models.py +35 -0
  54. synth_ai/cli/commands/status/subcommands/runs.py +34 -0
  55. synth_ai/cli/commands/status/subcommands/session.py +77 -0
  56. synth_ai/cli/commands/status/subcommands/summary.py +39 -0
  57. synth_ai/cli/commands/status/subcommands/utils.py +41 -0
  58. synth_ai/cli/commands/status/utils.py +23 -0
  59. synth_ai/cli/commands/train/__init__.py +53 -0
  60. synth_ai/cli/commands/train/core.py +22 -0
  61. synth_ai/cli/commands/train/errors.py +117 -0
  62. synth_ai/cli/commands/train/judge_schemas.py +201 -0
  63. synth_ai/cli/commands/train/judge_validation.py +305 -0
  64. synth_ai/cli/commands/train/prompt_learning_validation.py +633 -0
  65. synth_ai/cli/commands/train/validation.py +392 -0
  66. synth_ai/cli/demo_apps/__init__.py +10 -0
  67. synth_ai/cli/demo_apps/core/__init__.py +28 -0
  68. synth_ai/cli/demo_apps/core/cli.py +1735 -0
  69. synth_ai/cli/demo_apps/crafter/__init__.py +1 -0
  70. synth_ai/cli/demo_apps/crafter/crafter_fft_4b.toml +55 -0
  71. synth_ai/cli/demo_apps/crafter/grpo_crafter_task_app.py +186 -0
  72. synth_ai/cli/demo_apps/crafter/rl_from_base_qwen4b.toml +74 -0
  73. synth_ai/cli/demo_apps/demo_registry.py +176 -0
  74. synth_ai/cli/demo_apps/demo_task_apps/__init__.py +7 -0
  75. synth_ai/{demos → cli/demo_apps}/demo_task_apps/core.py +117 -51
  76. synth_ai/cli/demo_apps/demo_task_apps/crafter/__init__.py +1 -0
  77. synth_ai/cli/demo_apps/demo_task_apps/crafter/configs/crafter_fft_4b.toml +53 -0
  78. synth_ai/cli/demo_apps/demo_task_apps/crafter/configs/rl_from_base_qwen4b.toml +73 -0
  79. synth_ai/cli/demo_apps/demo_task_apps/crafter/grpo_crafter_task_app.py +185 -0
  80. synth_ai/cli/demo_apps/demo_task_apps/math/_common.py +16 -0
  81. synth_ai/{demos → cli/demo_apps}/demo_task_apps/math/app.py +2 -1
  82. synth_ai/cli/demo_apps/demo_task_apps/math/config.toml +73 -0
  83. synth_ai/{demos → cli/demo_apps}/demo_task_apps/math/deploy_modal.py +3 -6
  84. synth_ai/cli/demo_apps/demo_task_apps/math/modal_task_app.py +738 -0
  85. synth_ai/cli/demo_apps/demo_task_apps/math/task_app_entry.py +39 -0
  86. synth_ai/cli/demo_apps/math/__init__.py +1 -0
  87. synth_ai/cli/demo_apps/math/_common.py +16 -0
  88. synth_ai/cli/demo_apps/math/app.py +38 -0
  89. synth_ai/cli/demo_apps/math/config.toml +75 -0
  90. synth_ai/cli/demo_apps/math/deploy_modal.py +54 -0
  91. synth_ai/cli/demo_apps/math/modal_task_app.py +698 -0
  92. synth_ai/cli/demo_apps/math/task_app_entry.py +53 -0
  93. synth_ai/cli/demo_apps/mipro/main.py +271 -0
  94. synth_ai/cli/demo_apps/mipro/task_app.py +922 -0
  95. synth_ai/cli/demo_apps/mipro/train_cfg.toml +92 -0
  96. synth_ai/cli/demos/__init__.py +12 -0
  97. synth_ai/cli/demos/demo.py +32 -0
  98. synth_ai/cli/demos/rl_demo.py +254 -0
  99. synth_ai/cli/deploy.py +216 -0
  100. synth_ai/cli/infra/__init__.py +14 -0
  101. synth_ai/cli/{balance.py → infra/balance.py} +21 -3
  102. synth_ai/cli/infra/mcp.py +35 -0
  103. synth_ai/cli/infra/modal_app.py +36 -0
  104. synth_ai/cli/infra/setup.py +69 -0
  105. synth_ai/cli/infra/status.py +16 -0
  106. synth_ai/cli/infra/turso.py +77 -0
  107. synth_ai/cli/lib/__init__.py +10 -0
  108. synth_ai/cli/lib/agents.py +76 -0
  109. synth_ai/cli/lib/apps/modal_app.py +101 -0
  110. synth_ai/cli/lib/apps/task_app.py +642 -0
  111. synth_ai/cli/lib/bin.py +39 -0
  112. synth_ai/cli/lib/env.py +375 -0
  113. synth_ai/cli/lib/errors.py +85 -0
  114. synth_ai/cli/lib/modal.py +315 -0
  115. synth_ai/cli/lib/plotting.py +126 -0
  116. synth_ai/cli/lib/prompt_args.py +39 -0
  117. synth_ai/cli/lib/prompts.py +284 -0
  118. synth_ai/cli/lib/sqld.py +122 -0
  119. synth_ai/cli/lib/task_app_discovery.py +884 -0
  120. synth_ai/cli/lib/task_app_env.py +295 -0
  121. synth_ai/cli/lib/train_cfgs.py +300 -0
  122. synth_ai/cli/lib/tunnel_records.py +207 -0
  123. synth_ai/cli/local/__init__.py +14 -0
  124. synth_ai/cli/local/experiment_queue/__init__.py +72 -0
  125. synth_ai/cli/local/experiment_queue/api_schemas.py +221 -0
  126. synth_ai/cli/local/experiment_queue/celery_app.py +208 -0
  127. synth_ai/cli/local/experiment_queue/config.py +128 -0
  128. synth_ai/cli/local/experiment_queue/config_utils.py +272 -0
  129. synth_ai/cli/local/experiment_queue/database.py +175 -0
  130. synth_ai/cli/local/experiment_queue/dispatcher.py +119 -0
  131. synth_ai/cli/local/experiment_queue/models.py +231 -0
  132. synth_ai/cli/local/experiment_queue/progress_info.py +160 -0
  133. synth_ai/cli/local/experiment_queue/results.py +373 -0
  134. synth_ai/cli/local/experiment_queue/schemas.py +131 -0
  135. synth_ai/cli/local/experiment_queue/service.py +344 -0
  136. synth_ai/cli/local/experiment_queue/status.py +372 -0
  137. synth_ai/cli/local/experiment_queue/status_tracker.py +360 -0
  138. synth_ai/cli/local/experiment_queue/tasks.py +1984 -0
  139. synth_ai/cli/local/experiment_queue/trace_storage.py +65 -0
  140. synth_ai/cli/local/experiment_queue/validation.py +157 -0
  141. synth_ai/cli/local/session/__init__.py +92 -0
  142. synth_ai/cli/local/session/client.py +383 -0
  143. synth_ai/cli/local/session/constants.py +63 -0
  144. synth_ai/cli/local/session/exceptions.py +105 -0
  145. synth_ai/cli/local/session/manager.py +139 -0
  146. synth_ai/cli/local/session/models.py +89 -0
  147. synth_ai/cli/local/session/query.py +110 -0
  148. synth_ai/cli/root.py +150 -102
  149. synth_ai/cli/task_apps/__init__.py +37 -0
  150. synth_ai/cli/task_apps/commands.py +3145 -0
  151. synth_ai/cli/task_apps/deploy.py +7 -0
  152. synth_ai/cli/task_apps/list.py +26 -0
  153. synth_ai/cli/task_apps/main.py +36 -0
  154. synth_ai/cli/task_apps/modal_serve.py +11 -0
  155. synth_ai/cli/task_apps/serve.py +11 -0
  156. synth_ai/cli/training/__init__.py +8 -0
  157. synth_ai/cli/training/train.py +5 -0
  158. synth_ai/cli/training/train_cfg.py +34 -0
  159. synth_ai/cli/{watch.py → training/watch.py} +13 -18
  160. synth_ai/cli/turso.py +52 -0
  161. synth_ai/cli/utils/__init__.py +8 -0
  162. synth_ai/cli/utils/experiments.py +235 -0
  163. synth_ai/cli/utils/queue.py +504 -0
  164. synth_ai/cli/{recent.py → utils/recent.py} +13 -7
  165. synth_ai/cli/{traces.py → utils/traces.py} +9 -5
  166. synth_ai/contracts/__init__.py +67 -0
  167. synth_ai/core/__init__.py +100 -0
  168. synth_ai/core/_utils/__init__.py +54 -0
  169. synth_ai/core/_utils/base_url.py +10 -0
  170. synth_ai/core/_utils/http.py +10 -0
  171. synth_ai/core/_utils/prompts.py +14 -0
  172. synth_ai/core/_utils/task_app_state.py +12 -0
  173. synth_ai/core/_utils/user_config.py +10 -0
  174. synth_ai/core/apps/common.py +116 -0
  175. synth_ai/core/auth.py +95 -0
  176. synth_ai/core/cfgs.py +240 -0
  177. synth_ai/core/config/__init__.py +16 -0
  178. synth_ai/core/config/base.py +168 -0
  179. synth_ai/core/config/resolver.py +89 -0
  180. synth_ai/core/env.py +231 -0
  181. synth_ai/core/errors.py +126 -0
  182. synth_ai/core/http.py +230 -0
  183. synth_ai/core/integrations/__init__.py +11 -0
  184. synth_ai/core/integrations/cloudflare.py +1710 -0
  185. synth_ai/core/integrations/mcp/__init__.py +6 -0
  186. synth_ai/core/integrations/mcp/__main__.py +8 -0
  187. synth_ai/core/integrations/mcp/claude.py +36 -0
  188. synth_ai/core/integrations/mcp/main.py +254 -0
  189. synth_ai/core/integrations/mcp/setup.py +100 -0
  190. synth_ai/core/integrations/modal.py +277 -0
  191. synth_ai/core/json.py +72 -0
  192. synth_ai/core/log_filter.py +99 -0
  193. synth_ai/core/logging.py +82 -0
  194. synth_ai/core/paths.py +107 -0
  195. synth_ai/core/pricing.py +109 -0
  196. synth_ai/core/process.py +233 -0
  197. synth_ai/core/ssl.py +25 -0
  198. synth_ai/core/storage/__init__.py +71 -0
  199. synth_ai/core/task_app_state.py +318 -0
  200. synth_ai/core/telemetry.py +282 -0
  201. synth_ai/{tracing_v3 → core/tracing_v3}/__init__.py +5 -1
  202. synth_ai/{tracing_v3 → core/tracing_v3}/abstractions.py +21 -4
  203. synth_ai/core/tracing_v3/config.py +229 -0
  204. synth_ai/core/tracing_v3/constants.py +21 -0
  205. synth_ai/{tracing_v3 → core/tracing_v3}/db_config.py +42 -29
  206. synth_ai/{tracing_v3 → core/tracing_v3}/decorators.py +80 -45
  207. synth_ai/{tracing_v3 → core/tracing_v3}/examples/basic_usage.py +15 -9
  208. synth_ai/{tracing_v3 → core/tracing_v3}/hooks.py +6 -4
  209. synth_ai/{tracing_v3 → core/tracing_v3}/llm_call_record_helpers.py +161 -61
  210. synth_ai/{tracing_v3 → core/tracing_v3}/migration_helper.py +1 -2
  211. synth_ai/{tracing_v3 → core/tracing_v3}/replica_sync.py +12 -7
  212. synth_ai/core/tracing_v3/serialization.py +130 -0
  213. synth_ai/{tracing_v3 → core/tracing_v3}/session_tracer.py +88 -21
  214. synth_ai/{tracing_v3 → core/tracing_v3}/storage/base.py +99 -12
  215. synth_ai/core/tracing_v3/storage/config.py +109 -0
  216. synth_ai/{tracing_v3 → core/tracing_v3}/storage/factory.py +11 -9
  217. synth_ai/{tracing_v3 → core/tracing_v3}/storage/utils.py +15 -11
  218. synth_ai/core/tracing_v3/trace_utils.py +326 -0
  219. synth_ai/core/tracing_v3/turso/__init__.py +12 -0
  220. synth_ai/core/tracing_v3/turso/daemon.py +278 -0
  221. synth_ai/{tracing_v3 → core/tracing_v3}/turso/models.py +7 -3
  222. synth_ai/core/tracing_v3/turso/native_manager.py +1385 -0
  223. synth_ai/{tracing_v3 → core/tracing_v3}/utils.py +5 -4
  224. synth_ai/core/urls.py +18 -0
  225. synth_ai/core/user_config.py +137 -0
  226. synth_ai/core/uvicorn.py +222 -0
  227. synth_ai/data/__init__.py +83 -0
  228. synth_ai/data/enums.py +123 -0
  229. synth_ai/data/rewards.py +152 -0
  230. synth_ai/data/traces.py +35 -0
  231. synth_ai/products/__init__.py +6 -0
  232. synth_ai/products/graph_evolve/__init__.py +46 -0
  233. synth_ai/products/graph_evolve/client.py +226 -0
  234. synth_ai/products/graph_evolve/config.py +591 -0
  235. synth_ai/products/graph_evolve/converters/__init__.py +42 -0
  236. synth_ai/products/graph_evolve/converters/openai_sft.py +484 -0
  237. synth_ai/products/graph_evolve/examples/hotpotqa/config.toml +109 -0
  238. synth_ai/products/graph_evolve/run.py +222 -0
  239. synth_ai/products/graph_gepa/__init__.py +23 -0
  240. synth_ai/products/graph_gepa/converters/__init__.py +19 -0
  241. synth_ai/products/graph_gepa/converters/openai_sft.py +29 -0
  242. synth_ai/sdk/__init__.py +123 -0
  243. synth_ai/sdk/api/__init__.py +1 -0
  244. synth_ai/sdk/api/models/supported.py +514 -0
  245. synth_ai/sdk/api/research_agent/__init__.py +296 -0
  246. synth_ai/sdk/api/train/__init__.py +85 -0
  247. synth_ai/sdk/api/train/builders.py +895 -0
  248. synth_ai/sdk/api/train/cli.py +2199 -0
  249. synth_ai/sdk/api/train/config_finder.py +267 -0
  250. synth_ai/sdk/api/train/configs/__init__.py +65 -0
  251. synth_ai/sdk/api/train/configs/prompt_learning.py +1706 -0
  252. synth_ai/sdk/api/train/configs/rl.py +187 -0
  253. synth_ai/sdk/api/train/configs/sft.py +99 -0
  254. synth_ai/sdk/api/train/configs/shared.py +81 -0
  255. synth_ai/sdk/api/train/context_learning.py +312 -0
  256. synth_ai/sdk/api/train/env_resolver.py +418 -0
  257. synth_ai/sdk/api/train/graph_validators.py +216 -0
  258. synth_ai/sdk/api/train/graphgen.py +984 -0
  259. synth_ai/sdk/api/train/graphgen_models.py +823 -0
  260. synth_ai/sdk/api/train/graphgen_validators.py +109 -0
  261. synth_ai/sdk/api/train/local_api.py +10 -0
  262. synth_ai/sdk/api/train/pollers.py +124 -0
  263. synth_ai/sdk/api/train/progress/__init__.py +97 -0
  264. synth_ai/sdk/api/train/progress/dataclasses.py +569 -0
  265. synth_ai/sdk/api/train/progress/events.py +326 -0
  266. synth_ai/sdk/api/train/progress/results.py +428 -0
  267. synth_ai/sdk/api/train/progress/tracker.py +641 -0
  268. synth_ai/sdk/api/train/prompt_learning.py +469 -0
  269. synth_ai/sdk/api/train/rl.py +441 -0
  270. synth_ai/sdk/api/train/sft.py +396 -0
  271. synth_ai/sdk/api/train/summary.py +522 -0
  272. synth_ai/sdk/api/train/supported_algos.py +147 -0
  273. synth_ai/sdk/api/train/task_app.py +351 -0
  274. synth_ai/sdk/api/train/utils.py +279 -0
  275. synth_ai/sdk/api/train/validators.py +2424 -0
  276. synth_ai/sdk/graphs/__init__.py +15 -0
  277. synth_ai/sdk/graphs/completions.py +570 -0
  278. synth_ai/{inference → sdk/inference}/__init__.py +0 -1
  279. synth_ai/sdk/inference/client.py +128 -0
  280. synth_ai/sdk/jobs/__init__.py +16 -0
  281. synth_ai/sdk/jobs/client.py +371 -0
  282. synth_ai/sdk/judging/__init__.py +14 -0
  283. synth_ai/sdk/judging/base.py +24 -0
  284. synth_ai/sdk/judging/client.py +40 -0
  285. synth_ai/sdk/judging/schemas.py +222 -0
  286. synth_ai/sdk/judging/types.py +42 -0
  287. synth_ai/sdk/learning/__init__.py +99 -0
  288. synth_ai/sdk/learning/algorithms.py +14 -0
  289. synth_ai/{learning → sdk/learning}/client.py +121 -30
  290. synth_ai/sdk/learning/config.py +5 -0
  291. synth_ai/{learning → sdk/learning}/constants.py +0 -2
  292. synth_ai/sdk/learning/context_learning_client.py +531 -0
  293. synth_ai/sdk/learning/context_learning_types.py +292 -0
  294. synth_ai/sdk/learning/ft_client.py +7 -0
  295. synth_ai/{learning → sdk/learning}/health.py +15 -9
  296. synth_ai/{learning → sdk/learning}/jobs.py +44 -47
  297. synth_ai/sdk/learning/prompt_extraction.py +334 -0
  298. synth_ai/sdk/learning/prompt_learning_client.py +455 -0
  299. synth_ai/sdk/learning/prompt_learning_types.py +186 -0
  300. synth_ai/{rl → sdk/learning/rl}/__init__.py +13 -8
  301. synth_ai/{learning/rl_client.py → sdk/learning/rl/client.py} +89 -77
  302. synth_ai/sdk/learning/rl/config.py +31 -0
  303. synth_ai/{rl → sdk/learning/rl}/contracts.py +5 -14
  304. synth_ai/{rl → sdk/learning/rl}/env_keys.py +45 -16
  305. synth_ai/sdk/learning/rl/secrets.py +13 -0
  306. synth_ai/sdk/learning/rl_client.py +5 -0
  307. synth_ai/sdk/learning/sft/__init__.py +29 -0
  308. synth_ai/sdk/learning/sft/client.py +95 -0
  309. synth_ai/sdk/learning/sft/config.py +270 -0
  310. synth_ai/sdk/learning/sft/data.py +698 -0
  311. synth_ai/sdk/learning/sse.py +57 -0
  312. synth_ai/sdk/learning/validators.py +52 -0
  313. synth_ai/sdk/localapi/__init__.py +40 -0
  314. synth_ai/sdk/localapi/apps/__init__.py +28 -0
  315. synth_ai/sdk/localapi/client.py +10 -0
  316. synth_ai/sdk/localapi/contracts.py +10 -0
  317. synth_ai/sdk/localapi/helpers.py +519 -0
  318. synth_ai/sdk/localapi/rollouts.py +87 -0
  319. synth_ai/sdk/localapi/server.py +29 -0
  320. synth_ai/sdk/localapi/template.py +70 -0
  321. synth_ai/sdk/streaming/__init__.py +35 -0
  322. synth_ai/sdk/streaming/config.py +94 -0
  323. synth_ai/sdk/streaming/handlers.py +1997 -0
  324. synth_ai/sdk/streaming/streamer.py +713 -0
  325. synth_ai/sdk/streaming/types.py +112 -0
  326. synth_ai/sdk/task/__init__.py +164 -0
  327. synth_ai/sdk/task/apps/__init__.py +169 -0
  328. synth_ai/sdk/task/auth.py +165 -0
  329. synth_ai/sdk/task/client.py +175 -0
  330. synth_ai/sdk/task/config.py +257 -0
  331. synth_ai/sdk/task/contracts.py +219 -0
  332. synth_ai/sdk/task/datasets.py +108 -0
  333. synth_ai/sdk/task/errors.py +50 -0
  334. synth_ai/sdk/task/health.py +34 -0
  335. synth_ai/sdk/task/in_process.py +1190 -0
  336. synth_ai/sdk/task/in_process_runner.py +314 -0
  337. synth_ai/sdk/task/inference_api.py +299 -0
  338. synth_ai/sdk/task/json.py +111 -0
  339. synth_ai/sdk/task/proxy.py +287 -0
  340. synth_ai/sdk/task/rubrics/__init__.py +55 -0
  341. synth_ai/sdk/task/rubrics/loaders.py +156 -0
  342. synth_ai/sdk/task/rubrics/models.py +57 -0
  343. synth_ai/sdk/task/rubrics/scoring.py +116 -0
  344. synth_ai/sdk/task/rubrics/strict.py +149 -0
  345. synth_ai/sdk/task/rubrics.py +219 -0
  346. synth_ai/sdk/task/server.py +631 -0
  347. synth_ai/sdk/task/trace_correlation_helpers.py +539 -0
  348. synth_ai/sdk/task/tracing_utils.py +95 -0
  349. synth_ai/sdk/task/validators.py +441 -0
  350. synth_ai/sdk/task/vendors.py +59 -0
  351. synth_ai/sdk/training/__init__.py +102 -0
  352. synth_ai/sdk/tunnels/__init__.py +83 -0
  353. synth_ai/sdk/tunnels/cleanup.py +83 -0
  354. synth_ai/sdk/tunnels/ports.py +120 -0
  355. synth_ai/utils/__init__.py +213 -0
  356. synth_ai-0.4.3.dist-info/METADATA +262 -0
  357. synth_ai-0.4.3.dist-info/RECORD +370 -0
  358. {synth_ai-0.2.6.dev1.dist-info → synth_ai-0.4.3.dist-info}/entry_points.txt +0 -1
  359. synth_ai/cli/calc.py +0 -69
  360. synth_ai/cli/demo.py +0 -131
  361. synth_ai/cli/legacy_root_backup.py +0 -470
  362. synth_ai/cli/man.py +0 -106
  363. synth_ai/cli/rl_demo.py +0 -137
  364. synth_ai/cli/status.py +0 -133
  365. synth_ai/config/base_url.py +0 -98
  366. synth_ai/core/experiment.py +0 -15
  367. synth_ai/core/system.py +0 -15
  368. synth_ai/demos/core/__init__.py +0 -1
  369. synth_ai/demos/core/cli.py +0 -685
  370. synth_ai/demos/demo_task_apps/__init__.py +0 -1
  371. synth_ai/demos/demo_task_apps/math/config.toml +0 -44
  372. synth_ai/demos/demo_task_apps/math/deploy_task_app.sh +0 -22
  373. synth_ai/environments/__init__.py +0 -31
  374. synth_ai/environments/environment/__init__.py +0 -1
  375. synth_ai/environments/environment/artifacts/__init__.py +0 -1
  376. synth_ai/environments/environment/artifacts/base.py +0 -52
  377. synth_ai/environments/environment/core.py +0 -67
  378. synth_ai/environments/environment/db/__init__.py +0 -1
  379. synth_ai/environments/environment/db/sqlite.py +0 -45
  380. synth_ai/environments/environment/registry.py +0 -233
  381. synth_ai/environments/environment/resources/sqlite.py +0 -45
  382. synth_ai/environments/environment/results.py +0 -1
  383. synth_ai/environments/environment/rewards/__init__.py +0 -1
  384. synth_ai/environments/environment/rewards/core.py +0 -29
  385. synth_ai/environments/environment/shared_engine.py +0 -26
  386. synth_ai/environments/environment/tools/__init__.py +0 -200
  387. synth_ai/environments/examples/__init__.py +0 -1
  388. synth_ai/environments/examples/bandit/__init__.py +0 -33
  389. synth_ai/environments/examples/bandit/engine.py +0 -294
  390. synth_ai/environments/examples/bandit/environment.py +0 -194
  391. synth_ai/environments/examples/bandit/taskset.py +0 -200
  392. synth_ai/environments/examples/crafter_classic/__init__.py +0 -8
  393. synth_ai/environments/examples/crafter_classic/agent_demos/analyze_semantic_words_markdown.py +0 -250
  394. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_comprehensive_evaluation.py +0 -59
  395. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_evaluation_browser.py +0 -152
  396. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_evaluation_config.toml +0 -24
  397. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_evaluation_framework.py +0 -1194
  398. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/crafter_synth_config.toml +0 -56
  399. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/filter_config_modal.toml +0 -32
  400. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/filter_traces_sft_turso.py +0 -724
  401. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/kick_off_ft_modal.py +0 -384
  402. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/analyze_action_results.py +0 -53
  403. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/analyze_agent_actions.py +0 -178
  404. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/analyze_latest_run.py +0 -222
  405. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/analyze_lm_traces.py +0 -183
  406. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/analyze_no_rewards.py +0 -210
  407. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/analyze_trace_issue.py +0 -206
  408. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/check_db_schema.py +0 -49
  409. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/check_latest_results.py +0 -64
  410. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/debug_agent_responses.py +0 -88
  411. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_modal_ft/old/quick_trace_check.py +0 -77
  412. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/compare_experiments.py +0 -324
  413. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/filter_traces_sft_turso.py +0 -580
  414. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/kick_off_ft_oai.py +0 -362
  415. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/multi_model_config.toml +0 -49
  416. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/analyze_enhanced_hooks.py +0 -332
  417. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/analyze_hook_events.py +0 -97
  418. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/analyze_hook_results.py +0 -217
  419. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/check_hook_storage.py +0 -87
  420. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/check_seeds.py +0 -88
  421. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/compare_seed_performance.py +0 -195
  422. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/custom_eval_pipelines.py +0 -400
  423. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/plot_hook_frequency.py +0 -195
  424. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/old/seed_analysis_summary.py +0 -56
  425. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_openai_ft/run_rollouts_for_models_and_compare_v3.py +0 -858
  426. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_quick_evaluation.py +0 -52
  427. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_react_agent.py +0 -874
  428. synth_ai/environments/examples/crafter_classic/agent_demos/crafter_trace_evaluation.py +0 -1412
  429. synth_ai/environments/examples/crafter_classic/agent_demos/example_v3_usage.py +0 -216
  430. synth_ai/environments/examples/crafter_classic/agent_demos/old/compare_traces.py +0 -296
  431. synth_ai/environments/examples/crafter_classic/agent_demos/old/crafter_comprehensive_evaluation.py +0 -58
  432. synth_ai/environments/examples/crafter_classic/agent_demos/old/crafter_env_serialization.py +0 -464
  433. synth_ai/environments/examples/crafter_classic/agent_demos/old/crafter_evaluation_browser.py +0 -152
  434. synth_ai/environments/examples/crafter_classic/agent_demos/old/crafter_quick_evaluation.py +0 -51
  435. synth_ai/environments/examples/crafter_classic/agent_demos/old/crafter_trace_evaluation.py +0 -1412
  436. synth_ai/environments/examples/crafter_classic/agent_demos/old/debug_player_loss.py +0 -112
  437. synth_ai/environments/examples/crafter_classic/agent_demos/old/diagnose_service.py +0 -203
  438. synth_ai/environments/examples/crafter_classic/agent_demos/old/diagnose_slowness.py +0 -305
  439. synth_ai/environments/examples/crafter_classic/agent_demos/old/eval_by_difficulty.py +0 -126
  440. synth_ai/environments/examples/crafter_classic/agent_demos/old/eval_example.py +0 -94
  441. synth_ai/environments/examples/crafter_classic/agent_demos/old/explore_saved_states.py +0 -142
  442. synth_ai/environments/examples/crafter_classic/agent_demos/old/filter_traces_sft.py +0 -26
  443. synth_ai/environments/examples/crafter_classic/agent_demos/old/filter_traces_sft_OLD.py +0 -984
  444. synth_ai/environments/examples/crafter_classic/agent_demos/old/generate_ft_data_gemini.py +0 -724
  445. synth_ai/environments/examples/crafter_classic/agent_demos/old/generate_ft_data_modal.py +0 -386
  446. synth_ai/environments/examples/crafter_classic/agent_demos/old/generate_ft_metadata.py +0 -205
  447. synth_ai/environments/examples/crafter_classic/agent_demos/old/kick_off_ft_gemini.py +0 -150
  448. synth_ai/environments/examples/crafter_classic/agent_demos/old/kick_off_ft_modal.py +0 -283
  449. synth_ai/environments/examples/crafter_classic/agent_demos/old/prepare_vertex_ft.py +0 -280
  450. synth_ai/environments/examples/crafter_classic/agent_demos/old/profile_env_slowness.py +0 -456
  451. synth_ai/environments/examples/crafter_classic/agent_demos/old/replicate_issue.py +0 -166
  452. synth_ai/environments/examples/crafter_classic/agent_demos/old/run_and_eval.py +0 -102
  453. synth_ai/environments/examples/crafter_classic/agent_demos/old/run_comparison.py +0 -128
  454. synth_ai/environments/examples/crafter_classic/agent_demos/old/run_qwen_rollouts.py +0 -655
  455. synth_ai/environments/examples/crafter_classic/agent_demos/old/trace_eval_OLD.py +0 -202
  456. synth_ai/environments/examples/crafter_classic/agent_demos/old/validate_openai_format.py +0 -166
  457. synth_ai/environments/examples/crafter_classic/config_logging.py +0 -111
  458. synth_ai/environments/examples/crafter_classic/debug_translation.py +0 -0
  459. synth_ai/environments/examples/crafter_classic/engine.py +0 -579
  460. synth_ai/environments/examples/crafter_classic/engine_deterministic_patch.py +0 -64
  461. synth_ai/environments/examples/crafter_classic/engine_helpers/action_map.py +0 -6
  462. synth_ai/environments/examples/crafter_classic/engine_helpers/serialization.py +0 -75
  463. synth_ai/environments/examples/crafter_classic/engine_serialization_patch_v3.py +0 -267
  464. synth_ai/environments/examples/crafter_classic/environment.py +0 -404
  465. synth_ai/environments/examples/crafter_classic/taskset.py +0 -233
  466. synth_ai/environments/examples/crafter_classic/trace_hooks_v3.py +0 -228
  467. synth_ai/environments/examples/crafter_classic/world_config_patch_simple.py +0 -299
  468. synth_ai/environments/examples/crafter_custom/__init__.py +0 -4
  469. synth_ai/environments/examples/crafter_custom/agent_demos/__init__.py +0 -1
  470. synth_ai/environments/examples/crafter_custom/agent_demos/trace_eval.py +0 -202
  471. synth_ai/environments/examples/crafter_custom/crafter/__init__.py +0 -7
  472. synth_ai/environments/examples/crafter_custom/crafter/config.py +0 -182
  473. synth_ai/environments/examples/crafter_custom/crafter/constants.py +0 -8
  474. synth_ai/environments/examples/crafter_custom/crafter/engine.py +0 -269
  475. synth_ai/environments/examples/crafter_custom/crafter/env.py +0 -262
  476. synth_ai/environments/examples/crafter_custom/crafter/objects.py +0 -417
  477. synth_ai/environments/examples/crafter_custom/crafter/recorder.py +0 -187
  478. synth_ai/environments/examples/crafter_custom/crafter/worldgen.py +0 -118
  479. synth_ai/environments/examples/crafter_custom/dataset_builder.py +0 -373
  480. synth_ai/environments/examples/crafter_custom/environment.py +0 -312
  481. synth_ai/environments/examples/crafter_custom/old/analyze_diamond_issue.py +0 -159
  482. synth_ai/environments/examples/crafter_custom/old/analyze_diamond_spawning.py +0 -158
  483. synth_ai/environments/examples/crafter_custom/old/compare_worlds.py +0 -71
  484. synth_ai/environments/examples/crafter_custom/old/dataset_stats.py +0 -105
  485. synth_ai/environments/examples/crafter_custom/old/diamond_spawning_summary.py +0 -119
  486. synth_ai/environments/examples/crafter_custom/old/example_dataset_usage.py +0 -52
  487. synth_ai/environments/examples/crafter_custom/run_dataset.py +0 -305
  488. synth_ai/environments/examples/enron/art_helpers/email_search_tools.py +0 -156
  489. synth_ai/environments/examples/enron/art_helpers/local_email_db.py +0 -281
  490. synth_ai/environments/examples/enron/art_helpers/types_enron.py +0 -25
  491. synth_ai/environments/examples/enron/engine.py +0 -295
  492. synth_ai/environments/examples/enron/environment.py +0 -166
  493. synth_ai/environments/examples/enron/taskset.py +0 -112
  494. synth_ai/environments/examples/enron/units/keyword_stats.py +0 -112
  495. synth_ai/environments/examples/minigrid/__init__.py +0 -48
  496. synth_ai/environments/examples/minigrid/agent_demos/minigrid_evaluation_framework.py +0 -1188
  497. synth_ai/environments/examples/minigrid/agent_demos/minigrid_quick_evaluation.py +0 -48
  498. synth_ai/environments/examples/minigrid/agent_demos/minigrid_react_agent.py +0 -562
  499. synth_ai/environments/examples/minigrid/agent_demos/minigrid_trace_evaluation.py +0 -221
  500. synth_ai/environments/examples/minigrid/engine.py +0 -589
  501. synth_ai/environments/examples/minigrid/environment.py +0 -274
  502. synth_ai/environments/examples/minigrid/environment_mapping.py +0 -242
  503. synth_ai/environments/examples/minigrid/puzzle_loader.py +0 -417
  504. synth_ai/environments/examples/minigrid/taskset.py +0 -583
  505. synth_ai/environments/examples/nethack/__init__.py +0 -7
  506. synth_ai/environments/examples/nethack/achievements.py +0 -337
  507. synth_ai/environments/examples/nethack/agent_demos/nethack_evaluation_framework.py +0 -981
  508. synth_ai/environments/examples/nethack/agent_demos/nethack_quick_evaluation.py +0 -74
  509. synth_ai/environments/examples/nethack/agent_demos/nethack_react_agent.py +0 -831
  510. synth_ai/environments/examples/nethack/engine.py +0 -739
  511. synth_ai/environments/examples/nethack/environment.py +0 -256
  512. synth_ai/environments/examples/nethack/helpers/__init__.py +0 -41
  513. synth_ai/environments/examples/nethack/helpers/action_mapping.py +0 -301
  514. synth_ai/environments/examples/nethack/helpers/nle_wrapper.py +0 -402
  515. synth_ai/environments/examples/nethack/helpers/observation_utils.py +0 -433
  516. synth_ai/environments/examples/nethack/helpers/recording_wrapper.py +0 -200
  517. synth_ai/environments/examples/nethack/helpers/trajectory_recorder.py +0 -269
  518. synth_ai/environments/examples/nethack/helpers/visualization/replay_viewer.py +0 -308
  519. synth_ai/environments/examples/nethack/helpers/visualization/visualizer.py +0 -431
  520. synth_ai/environments/examples/nethack/taskset.py +0 -323
  521. synth_ai/environments/examples/red/__init__.py +0 -7
  522. synth_ai/environments/examples/red/agent_demos/__init__.py +0 -1
  523. synth_ai/environments/examples/red/config_logging.py +0 -110
  524. synth_ai/environments/examples/red/engine.py +0 -694
  525. synth_ai/environments/examples/red/engine_helpers/__init__.py +0 -1
  526. synth_ai/environments/examples/red/engine_helpers/memory_map.py +0 -28
  527. synth_ai/environments/examples/red/engine_helpers/reward_components.py +0 -276
  528. synth_ai/environments/examples/red/engine_helpers/reward_library/__init__.py +0 -142
  529. synth_ai/environments/examples/red/engine_helpers/reward_library/adaptive_rewards.py +0 -57
  530. synth_ai/environments/examples/red/engine_helpers/reward_library/battle_rewards.py +0 -284
  531. synth_ai/environments/examples/red/engine_helpers/reward_library/composite_rewards.py +0 -150
  532. synth_ai/environments/examples/red/engine_helpers/reward_library/economy_rewards.py +0 -138
  533. synth_ai/environments/examples/red/engine_helpers/reward_library/efficiency_rewards.py +0 -57
  534. synth_ai/environments/examples/red/engine_helpers/reward_library/exploration_rewards.py +0 -331
  535. synth_ai/environments/examples/red/engine_helpers/reward_library/novelty_rewards.py +0 -121
  536. synth_ai/environments/examples/red/engine_helpers/reward_library/pallet_town_rewards.py +0 -559
  537. synth_ai/environments/examples/red/engine_helpers/reward_library/pokemon_rewards.py +0 -313
  538. synth_ai/environments/examples/red/engine_helpers/reward_library/social_rewards.py +0 -148
  539. synth_ai/environments/examples/red/engine_helpers/reward_library/story_rewards.py +0 -247
  540. synth_ai/environments/examples/red/engine_helpers/screen_analysis.py +0 -368
  541. synth_ai/environments/examples/red/engine_helpers/state_extraction.py +0 -140
  542. synth_ai/environments/examples/red/environment.py +0 -238
  543. synth_ai/environments/examples/red/taskset.py +0 -79
  544. synth_ai/environments/examples/red/units/__init__.py +0 -1
  545. synth_ai/environments/examples/sokoban/__init__.py +0 -1
  546. synth_ai/environments/examples/sokoban/agent_demos/sokoban_full_eval.py +0 -899
  547. synth_ai/environments/examples/sokoban/engine.py +0 -678
  548. synth_ai/environments/examples/sokoban/engine_helpers/__init__.py +0 -1
  549. synth_ai/environments/examples/sokoban/engine_helpers/room_utils.py +0 -657
  550. synth_ai/environments/examples/sokoban/engine_helpers/vendored/__init__.py +0 -18
  551. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/__init__.py +0 -3
  552. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/boxoban_env.py +0 -131
  553. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/render_utils.py +0 -370
  554. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/room_utils.py +0 -332
  555. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/sokoban_env.py +0 -306
  556. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/sokoban_env_fixed_targets.py +0 -67
  557. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/sokoban_env_pull.py +0 -115
  558. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/sokoban_env_two_player.py +0 -123
  559. synth_ai/environments/examples/sokoban/engine_helpers/vendored/envs/sokoban_env_variations.py +0 -394
  560. synth_ai/environments/examples/sokoban/environment.py +0 -229
  561. synth_ai/environments/examples/sokoban/generate_verified_puzzles.py +0 -440
  562. synth_ai/environments/examples/sokoban/puzzle_loader.py +0 -312
  563. synth_ai/environments/examples/sokoban/taskset.py +0 -428
  564. synth_ai/environments/examples/sokoban/units/astar_common.py +0 -95
  565. synth_ai/environments/examples/tictactoe/__init__.py +0 -1
  566. synth_ai/environments/examples/tictactoe/engine.py +0 -368
  567. synth_ai/environments/examples/tictactoe/environment.py +0 -240
  568. synth_ai/environments/examples/tictactoe/taskset.py +0 -215
  569. synth_ai/environments/examples/verilog/__init__.py +0 -10
  570. synth_ai/environments/examples/verilog/engine.py +0 -329
  571. synth_ai/environments/examples/verilog/environment.py +0 -350
  572. synth_ai/environments/examples/verilog/taskset.py +0 -420
  573. synth_ai/environments/examples/wordle/__init__.py +0 -29
  574. synth_ai/environments/examples/wordle/engine.py +0 -398
  575. synth_ai/environments/examples/wordle/environment.py +0 -159
  576. synth_ai/environments/examples/wordle/helpers/generate_instances_wordfreq.py +0 -75
  577. synth_ai/environments/examples/wordle/taskset.py +0 -230
  578. synth_ai/environments/reproducibility/core.py +0 -42
  579. synth_ai/environments/reproducibility/helpers.py +0 -0
  580. synth_ai/environments/reproducibility/tree.py +0 -364
  581. synth_ai/environments/service/app.py +0 -91
  582. synth_ai/environments/service/core_routes.py +0 -1020
  583. synth_ai/environments/service/external_registry.py +0 -56
  584. synth_ai/environments/service/registry.py +0 -9
  585. synth_ai/environments/stateful/__init__.py +0 -1
  586. synth_ai/environments/stateful/core.py +0 -163
  587. synth_ai/environments/stateful/engine.py +0 -21
  588. synth_ai/environments/stateful/state.py +0 -7
  589. synth_ai/environments/tasks/api.py +0 -19
  590. synth_ai/environments/tasks/core.py +0 -80
  591. synth_ai/environments/tasks/filters.py +0 -41
  592. synth_ai/environments/tasks/utils.py +0 -91
  593. synth_ai/environments/v0_observability/history.py +0 -3
  594. synth_ai/environments/v0_observability/log.py +0 -2
  595. synth_ai/evals/base.py +0 -15
  596. synth_ai/experimental/synth_oss.py +0 -446
  597. synth_ai/http.py +0 -102
  598. synth_ai/inference/client.py +0 -20
  599. synth_ai/install_sqld.sh +0 -40
  600. synth_ai/jobs/client.py +0 -246
  601. synth_ai/learning/__init__.py +0 -24
  602. synth_ai/learning/config.py +0 -43
  603. synth_ai/learning/filtering.py +0 -0
  604. synth_ai/learning/ft_client.py +0 -59
  605. synth_ai/learning/offline/dpo.py +0 -0
  606. synth_ai/learning/offline/providers.py +0 -7
  607. synth_ai/learning/offline/sft.py +0 -0
  608. synth_ai/learning/offline/shared.py +0 -0
  609. synth_ai/learning/online/grpo.py +0 -0
  610. synth_ai/learning/online/irft.py +0 -0
  611. synth_ai/learning/prompts/banking77_injection_eval.py +0 -168
  612. synth_ai/learning/prompts/gepa.py +0 -0
  613. synth_ai/learning/prompts/hello_world_in_context_injection_ex.py +0 -213
  614. synth_ai/learning/prompts/mipro.py +0 -289
  615. synth_ai/learning/prompts/random_search.py +0 -246
  616. synth_ai/learning/prompts/run_mipro_banking77.py +0 -172
  617. synth_ai/learning/prompts/run_random_search_banking77.py +0 -324
  618. synth_ai/learning/sse.py +0 -58
  619. synth_ai/learning/validators.py +0 -48
  620. synth_ai/lm/__init__.py +0 -51
  621. synth_ai/lm/caching/constants.py +0 -6
  622. synth_ai/lm/caching/dbs.py +0 -0
  623. synth_ai/lm/caching/ephemeral.py +0 -102
  624. synth_ai/lm/caching/handler.py +0 -137
  625. synth_ai/lm/caching/initialize.py +0 -11
  626. synth_ai/lm/caching/persistent.py +0 -114
  627. synth_ai/lm/config.py +0 -110
  628. synth_ai/lm/constants.py +0 -32
  629. synth_ai/lm/core/__init__.py +0 -8
  630. synth_ai/lm/core/all.py +0 -73
  631. synth_ai/lm/core/exceptions.py +0 -7
  632. synth_ai/lm/core/main.py +0 -319
  633. synth_ai/lm/core/main_v3.py +0 -594
  634. synth_ai/lm/core/synth_models.py +0 -48
  635. synth_ai/lm/core/vendor_clients.py +0 -188
  636. synth_ai/lm/cost/__init__.py +0 -0
  637. synth_ai/lm/cost/monitor.py +0 -1
  638. synth_ai/lm/cost/statefulness.py +0 -1
  639. synth_ai/lm/injection.py +0 -80
  640. synth_ai/lm/overrides.py +0 -206
  641. synth_ai/lm/provider_support/__init__.py +0 -8
  642. synth_ai/lm/provider_support/anthropic.py +0 -972
  643. synth_ai/lm/provider_support/openai.py +0 -1139
  644. synth_ai/lm/provider_support/suppress_logging.py +0 -31
  645. synth_ai/lm/structured_outputs/__init__.py +0 -0
  646. synth_ai/lm/structured_outputs/handler.py +0 -440
  647. synth_ai/lm/structured_outputs/inject.py +0 -297
  648. synth_ai/lm/structured_outputs/rehabilitate.py +0 -185
  649. synth_ai/lm/tools/__init__.py +0 -3
  650. synth_ai/lm/tools/base.py +0 -172
  651. synth_ai/lm/unified_interface.py +0 -202
  652. synth_ai/lm/vendors/__init__.py +0 -0
  653. synth_ai/lm/vendors/base.py +0 -81
  654. synth_ai/lm/vendors/core/__init__.py +0 -0
  655. synth_ai/lm/vendors/core/anthropic_api.py +0 -387
  656. synth_ai/lm/vendors/core/gemini_api.py +0 -292
  657. synth_ai/lm/vendors/core/mistral_api.py +0 -322
  658. synth_ai/lm/vendors/core/openai_api.py +0 -220
  659. synth_ai/lm/vendors/core/synth_dev_api.py +0 -0
  660. synth_ai/lm/vendors/local/__init__.py +0 -0
  661. synth_ai/lm/vendors/local/ollama.py +0 -0
  662. synth_ai/lm/vendors/openai_standard.py +0 -780
  663. synth_ai/lm/vendors/openai_standard_responses.py +0 -256
  664. synth_ai/lm/vendors/retries.py +0 -22
  665. synth_ai/lm/vendors/supported/__init__.py +0 -0
  666. synth_ai/lm/vendors/supported/custom_endpoint.py +0 -417
  667. synth_ai/lm/vendors/supported/deepseek.py +0 -69
  668. synth_ai/lm/vendors/supported/grok.py +0 -75
  669. synth_ai/lm/vendors/supported/groq.py +0 -16
  670. synth_ai/lm/vendors/supported/ollama.py +0 -15
  671. synth_ai/lm/vendors/supported/openrouter.py +0 -74
  672. synth_ai/lm/vendors/supported/together.py +0 -11
  673. synth_ai/lm/vendors/synth_client.py +0 -808
  674. synth_ai/lm/warmup.py +0 -186
  675. synth_ai/rl/secrets.py +0 -19
  676. synth_ai/scripts/verify_rewards.py +0 -100
  677. synth_ai/task/__init__.py +0 -10
  678. synth_ai/task/contracts.py +0 -120
  679. synth_ai/task/health.py +0 -28
  680. synth_ai/task/validators.py +0 -12
  681. synth_ai/tracing/__init__.py +0 -30
  682. synth_ai/tracing_v1/__init__.py +0 -33
  683. synth_ai/tracing_v3/config.py +0 -84
  684. synth_ai/tracing_v3/storage/config.py +0 -62
  685. synth_ai/tracing_v3/turso/__init__.py +0 -25
  686. synth_ai/tracing_v3/turso/daemon.py +0 -144
  687. synth_ai/tracing_v3/turso/manager.py +0 -760
  688. synth_ai/v0/tracing/__init__.py +0 -0
  689. synth_ai/v0/tracing/abstractions.py +0 -224
  690. synth_ai/v0/tracing/base_client.py +0 -91
  691. synth_ai/v0/tracing/client_manager.py +0 -131
  692. synth_ai/v0/tracing/config.py +0 -140
  693. synth_ai/v0/tracing/context.py +0 -146
  694. synth_ai/v0/tracing/decorators.py +0 -680
  695. synth_ai/v0/tracing/events/__init__.py +0 -0
  696. synth_ai/v0/tracing/events/manage.py +0 -147
  697. synth_ai/v0/tracing/events/scope.py +0 -86
  698. synth_ai/v0/tracing/events/store.py +0 -228
  699. synth_ai/v0/tracing/immediate_client.py +0 -151
  700. synth_ai/v0/tracing/local.py +0 -18
  701. synth_ai/v0/tracing/log_client_base.py +0 -73
  702. synth_ai/v0/tracing/retry_queue.py +0 -186
  703. synth_ai/v0/tracing/trackers.py +0 -515
  704. synth_ai/v0/tracing/upload.py +0 -510
  705. synth_ai/v0/tracing/utils.py +0 -9
  706. synth_ai/v0/tracing_v1/__init__.py +0 -16
  707. synth_ai/v0/tracing_v1/abstractions.py +0 -224
  708. synth_ai/v0/tracing_v1/base_client.py +0 -91
  709. synth_ai/v0/tracing_v1/client_manager.py +0 -131
  710. synth_ai/v0/tracing_v1/config.py +0 -140
  711. synth_ai/v0/tracing_v1/context.py +0 -146
  712. synth_ai/v0/tracing_v1/decorators.py +0 -701
  713. synth_ai/v0/tracing_v1/events/__init__.py +0 -0
  714. synth_ai/v0/tracing_v1/events/manage.py +0 -147
  715. synth_ai/v0/tracing_v1/events/scope.py +0 -86
  716. synth_ai/v0/tracing_v1/events/store.py +0 -228
  717. synth_ai/v0/tracing_v1/immediate_client.py +0 -151
  718. synth_ai/v0/tracing_v1/local.py +0 -18
  719. synth_ai/v0/tracing_v1/log_client_base.py +0 -73
  720. synth_ai/v0/tracing_v1/retry_queue.py +0 -186
  721. synth_ai/v0/tracing_v1/trackers.py +0 -515
  722. synth_ai/v0/tracing_v1/upload.py +0 -525
  723. synth_ai/v0/tracing_v1/utils.py +0 -9
  724. synth_ai/zyk/__init__.py +0 -30
  725. synth_ai-0.2.6.dev1.dist-info/METADATA +0 -106
  726. synth_ai-0.2.6.dev1.dist-info/RECORD +0 -416
  727. /synth_ai/{demos → cli/demo_apps}/demo_task_apps/math/__init__.py +0 -0
  728. /synth_ai/{lm/caching → core/apps}/__init__.py +0 -0
  729. /synth_ai/{tracing_v3 → core/tracing_v3}/lm_call_record_abstractions.py +0 -0
  730. /synth_ai/{tracing_v3 → core/tracing_v3}/storage/__init__.py +0 -0
  731. /synth_ai/{tracing_v3 → core/tracing_v3}/storage/exceptions.py +0 -0
  732. /synth_ai/{tracing_v3 → core/tracing_v3}/storage/types.py +0 -0
  733. /synth_ai/{compound/cais.py → py.typed} +0 -0
  734. /synth_ai/{learning → sdk/learning}/core.py +0 -0
  735. /synth_ai/{learning → sdk/learning}/gateway.py +0 -0
  736. {synth_ai-0.2.6.dev1.dist-info → synth_ai-0.4.3.dist-info}/WHEEL +0 -0
  737. {synth_ai-0.2.6.dev1.dist-info → synth_ai-0.4.3.dist-info}/licenses/LICENSE +0 -0
  738. {synth_ai-0.2.6.dev1.dist-info → synth_ai-0.4.3.dist-info}/top_level.txt +0 -0
@@ -1,739 +0,0 @@
1
- """NetHack engine implementation with state management and NLE integration."""
2
-
3
- from __future__ import annotations
4
-
5
- import asyncio
6
- import base64
7
- import logging
8
- from dataclasses import dataclass, field
9
- from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, cast
10
-
11
- import numpy as np
12
-
13
- from synth_ai.environments.environment.rewards.core import RewardComponent, RewardStack
14
- from synth_ai.environments.environment.shared_engine import (
15
- GetObservationCallable,
16
- InternalObservation,
17
- )
18
- from synth_ai.environments.reproducibility.core import IReproducibleEngine
19
- from synth_ai.environments.stateful.engine import StatefulEngine, StatefulEngineSnapshot
20
- from synth_ai.environments.tasks.core import TaskInstance
21
-
22
- logger = logging.getLogger(__name__)
23
-
24
- # NLE imports are required
25
- try:
26
- from .achievements import NetHackAchievements, calculate_balrog_reward
27
- from .helpers.action_mapping import convert_action_to_nle
28
- from .helpers.nle_wrapper import NLEWrapper
29
- except ImportError as e:
30
- raise ImportError(
31
- "NLE (NetHack Learning Environment) is required but not installed. "
32
- "Please install it with: pip install nle"
33
- ) from e
34
-
35
- if TYPE_CHECKING:
36
- from .taskset import NetHackTaskInstanceMetadata
37
-
38
-
39
- @dataclass
40
- class NetHackPublicState:
41
- """State visible to the agent."""
42
-
43
- # Game state
44
- dungeon_level: int = 1
45
- character_stats: Dict[str, Any] = field(default_factory=dict)
46
- inventory: List[Dict[str, Any]] = field(default_factory=list)
47
- position: Tuple[int, int] = (0, 0)
48
-
49
- # Observation data
50
- ascii_map: str = ""
51
- message: str = ""
52
- cursor_position: Tuple[int, int] = (0, 0)
53
-
54
- # Meta information
55
- turn_count: int = 0
56
- max_turns: int = 10000
57
- last_action: str = ""
58
- terminated: bool = False
59
-
60
- # Game context
61
- in_menu: bool = False
62
- menu_items: List[str] = field(default_factory=list)
63
-
64
- # Achievements tracking
65
- achievements: NetHackAchievements = field(default_factory=NetHackAchievements)
66
- achievements_unlocked: Dict[str, bool] = field(default_factory=dict)
67
-
68
- def diff(self, prev_state: "NetHackPublicState") -> Dict[str, Any]:
69
- """Track changes between states."""
70
- differences = {}
71
-
72
- if self.dungeon_level != prev_state.dungeon_level:
73
- differences["dungeon_level"] = (
74
- prev_state.dungeon_level,
75
- self.dungeon_level,
76
- )
77
- if self.position != prev_state.position:
78
- differences["position"] = (prev_state.position, self.position)
79
- if self.message != prev_state.message:
80
- differences["message"] = (prev_state.message, self.message)
81
- if self.turn_count != prev_state.turn_count:
82
- differences["turn_count"] = (prev_state.turn_count, self.turn_count)
83
- if self.terminated != prev_state.terminated:
84
- differences["terminated"] = (prev_state.terminated, self.terminated)
85
- if self.last_action != prev_state.last_action:
86
- differences["last_action"] = (prev_state.last_action, self.last_action)
87
-
88
- return differences
89
-
90
- @property
91
- def map_text(self) -> str:
92
- """Formatted ASCII dungeon map."""
93
- return self.ascii_map
94
-
95
-
96
- @dataclass
97
- class NetHackPrivateState:
98
- """Internal state (rewards, termination flags)."""
99
-
100
- reward_last: float = 0.0
101
- total_reward: float = 0.0
102
- terminated: bool = False
103
- truncated: bool = False
104
-
105
- # Progress tracking
106
- score: int = 0
107
- depth_reached: int = 1
108
- experience_level: int = 1
109
- monsters_killed: int = 0
110
- items_collected: int = 0
111
-
112
- # Balrog reward tracking
113
- balrog_reward_last: float = 0.0
114
- balrog_total_reward: float = 0.0
115
-
116
- def diff(self, prev_state: "NetHackPrivateState") -> Dict[str, Any]:
117
- """Track reward/progress changes."""
118
- differences = {}
119
-
120
- if self.reward_last != prev_state.reward_last:
121
- differences["reward_last"] = (prev_state.reward_last, self.reward_last)
122
- if self.total_reward != prev_state.total_reward:
123
- differences["total_reward"] = (prev_state.total_reward, self.total_reward)
124
- if self.score != prev_state.score:
125
- differences["score"] = (prev_state.score, self.score)
126
- if self.depth_reached != prev_state.depth_reached:
127
- differences["depth_reached"] = (
128
- prev_state.depth_reached,
129
- self.depth_reached,
130
- )
131
-
132
- return differences
133
-
134
-
135
- @dataclass
136
- class NetHackEngineSnapshot(StatefulEngineSnapshot):
137
- """Serialization container for NetHack engine state."""
138
-
139
- task_instance_dict: Dict[str, Any]
140
- engine_snapshot: Dict[str, Any]
141
- nle_state: Optional[Dict[str, Any]] = None # NLE-specific state if available
142
-
143
-
144
- class NetHackSurvivalComponent(RewardComponent):
145
- """Reward component for staying alive."""
146
-
147
- async def score(self, state: NetHackPublicState, action: str) -> float:
148
- if state.terminated:
149
- return -1.0 # Penalty for death
150
- return 0.01 # Small reward for each turn survived
151
-
152
-
153
- class NetHackProgressComponent(RewardComponent):
154
- """Reward component for exploration and depth."""
155
-
156
- def __init__(self):
157
- self.last_depth = 1
158
-
159
- async def score(self, state: NetHackPublicState, action: str) -> float:
160
- reward = 0.0
161
-
162
- # Reward for reaching new dungeon levels
163
- if state.dungeon_level > self.last_depth:
164
- reward += 1.0 * (state.dungeon_level - self.last_depth)
165
- self.last_depth = state.dungeon_level
166
-
167
- return reward
168
-
169
-
170
- class NetHackScoreComponent(RewardComponent):
171
- """Reward component based on game score."""
172
-
173
- def __init__(self):
174
- self.last_score = 0
175
-
176
- async def score(self, state: NetHackPublicState, action: str) -> float:
177
- # Get score from character stats - require it exists
178
- current_score = state.character_stats["score"]
179
-
180
- # Calculate score delta
181
- score_delta = current_score - self.last_score
182
- self.last_score = current_score
183
-
184
- # Scale the score reward (NLE scores can be large)
185
- return score_delta / 100.0 if score_delta > 0 else 0.0
186
-
187
-
188
- class NetHackAchievementComponent(RewardComponent):
189
- """Reward component for unlocking achievements."""
190
-
191
- def __init__(self):
192
- self.last_unlocked = set()
193
-
194
- async def score(self, state: NetHackPublicState, action: str) -> float:
195
- reward = 0.0
196
-
197
- # Count newly unlocked achievements
198
- current_unlocked = set(k for k, v in state.achievements_unlocked.items() if v)
199
- new_achievements = current_unlocked - self.last_unlocked
200
-
201
- # Give rewards for different achievement types
202
- for achievement in new_achievements:
203
- if "first_" in achievement:
204
- reward += 1.0 # First-time achievements
205
- elif "reached_dlvl_" in achievement:
206
- reward += 2.0 # Depth achievements
207
- elif "killed_" in achievement and "monsters" in achievement:
208
- reward += 0.5 # Kill milestones
209
- elif "collected_" in achievement and "gold" in achievement:
210
- reward += 0.5 # Gold milestones
211
- elif "reached_level_" in achievement:
212
- reward += 1.5 # Experience level milestones
213
- elif "minetown" in achievement or "castle" in achievement:
214
- reward += 5.0 # Special locations
215
- elif "quest" in achievement:
216
- reward += 10.0 # Quest achievements
217
- else:
218
- reward += 0.5 # Default reward
219
-
220
- self.last_unlocked = current_unlocked
221
- return reward
222
-
223
-
224
- class NetHackEngine(StatefulEngine, IReproducibleEngine):
225
- """NetHack game engine with NLE backend."""
226
-
227
- def __init__(self, task_instance: TaskInstance):
228
- self.task_instance = task_instance
229
-
230
- # Require proper metadata
231
- from .taskset import NetHackTaskInstanceMetadata
232
-
233
- if not isinstance(task_instance.metadata, NetHackTaskInstanceMetadata):
234
- raise TypeError(
235
- f"Expected NetHackTaskInstanceMetadata, got {type(task_instance.metadata).__name__}"
236
- )
237
-
238
- metadata = cast(NetHackTaskInstanceMetadata, task_instance.metadata)
239
- self.character_role = metadata.character_role
240
- self.max_turns = metadata.time_limit
241
-
242
- # Initialize NLE wrapper
243
- self.nle = NLEWrapper(character_role=self.character_role)
244
-
245
- # Initialize reward components with proper tracking - NO SURVIVAL NOISE
246
- self.progress_component = NetHackProgressComponent()
247
- self.score_component = NetHackScoreComponent()
248
- self.achievement_component = NetHackAchievementComponent()
249
-
250
- self.reward_stack = RewardStack(
251
- [
252
- self.progress_component, # Depth progress
253
- self.score_component, # Game score changes
254
- self.achievement_component, # Achievement unlocks
255
- ]
256
- )
257
-
258
- # State tracking
259
- self.public_state: Optional[NetHackPublicState] = None
260
- self.private_state: Optional[NetHackPrivateState] = None
261
-
262
- # NLE observation processing
263
- self.last_nle_obs = None
264
-
265
- async def _reset_engine(
266
- self, *, seed: int | None = None
267
- ) -> Tuple[NetHackPrivateState, NetHackPublicState]:
268
- """Reset to initial state using NLE."""
269
- # Reset NLE environment with seed
270
- obs = await asyncio.to_thread(self.nle.reset, seed)
271
- self.last_nle_obs = obs
272
-
273
- # Log what we actually got from NLE
274
- logger.info(f"NLE reset returned observation keys: {list(obs.keys())}")
275
- if "player_stats" in obs:
276
- logger.info(f"Player stats keys: {list(obs['player_stats'].keys())}")
277
-
278
- # Initialize private state - require all fields
279
- player_stats = obs["player_stats"] # Will KeyError if missing
280
- self.private_state = NetHackPrivateState(
281
- reward_last=0.0,
282
- total_reward=0.0,
283
- terminated=False,
284
- truncated=False,
285
- score=player_stats["score"],
286
- depth_reached=player_stats["depth"],
287
- experience_level=player_stats["experience_level"],
288
- monsters_killed=0,
289
- items_collected=0,
290
- balrog_reward_last=0.0,
291
- balrog_total_reward=0.0,
292
- )
293
-
294
- # Initialize public state from NLE observation - no fallbacks
295
- self.public_state = NetHackPublicState(
296
- dungeon_level=player_stats["depth"],
297
- character_stats={
298
- "hp": player_stats["hp"],
299
- "max_hp": player_stats["max_hp"],
300
- "strength": player_stats["strength"],
301
- "dexterity": player_stats["dexterity"],
302
- "constitution": player_stats["constitution"],
303
- "intelligence": player_stats["intelligence"],
304
- "wisdom": player_stats["wisdom"],
305
- "charisma": player_stats["charisma"],
306
- "gold": player_stats["gold"],
307
- "experience": player_stats["experience_points"],
308
- "level": player_stats["experience_level"],
309
- "ac": player_stats["ac"],
310
- },
311
- inventory=self._process_inventory(obs["inventory"]) if "inventory" in obs else [],
312
- position=(player_stats["y"], player_stats["x"]),
313
- ascii_map=obs["ascii_map"],
314
- message=obs["message"],
315
- cursor_position=obs.get(
316
- "cursor", (player_stats["y"], player_stats["x"])
317
- ), # Cursor might not be in processed obs
318
- turn_count=0,
319
- max_turns=self.max_turns,
320
- last_action="",
321
- terminated=False,
322
- in_menu=obs.get("in_menu", False), # Menu detection is heuristic-based
323
- menu_items=obs.get("menu_text", []), # Menu text only present when in menu
324
- achievements=NetHackAchievements(),
325
- achievements_unlocked={},
326
- )
327
-
328
- # Reset reward components
329
- self.progress_component.last_depth = self.public_state.dungeon_level
330
- self.score_component.last_score = self.private_state.score
331
-
332
- return self.private_state, self.public_state
333
-
334
- def _process_inventory(self, inventory_items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
335
- """Process NLE inventory format to our format."""
336
- processed_items = []
337
- for item in inventory_items:
338
- processed_items.append(
339
- {
340
- "name": item["description"],
341
- "count": 1, # NLE doesn't always provide count
342
- "letter": item["letter"],
343
- }
344
- )
345
- return processed_items
346
-
347
- async def _step_engine(self, action: str) -> Tuple[NetHackPrivateState, NetHackPublicState]:
348
- """Execute one step/action using NLE."""
349
- # print(f"===== NetHack Engine _step_engine called with action: {action} =====")
350
- if self.public_state is None or self.private_state is None:
351
- raise RuntimeError("Engine not initialized. Call _reset_engine first.")
352
-
353
- # Validate action
354
- if action not in self.nle.action_map and action not in ["terminate"]:
355
- # Try to handle menu selections and special cases
356
- if len(action) == 1 and (action.isalpha() or action.isdigit()):
357
- # Single character actions are likely menu selections
358
- pass
359
- else:
360
- raise ValueError(
361
- f"Invalid action: {action}. Valid actions: {list(self.nle.action_map.keys())}"
362
- )
363
-
364
- # Update turn count
365
- self.public_state.turn_count += 1
366
- self.public_state.last_action = action
367
-
368
- # Define non-turn-consuming actions
369
- non_turn_actions = [
370
- "look",
371
- "farlook",
372
- "whatis",
373
- "identify",
374
- "discoveries",
375
- "conduct",
376
- "attributes",
377
- "help",
378
- "version",
379
- "history",
380
- ]
381
-
382
- # Warn about non-advancing actions
383
- if action in non_turn_actions:
384
- logger.warning(f"Action '{action}' is a free action that doesn't advance game time!")
385
- # If we're repeatedly using non-advancing actions, force a wait
386
- if hasattr(self, "_consecutive_free_actions"):
387
- self._consecutive_free_actions += 1
388
- if self._consecutive_free_actions >= 3:
389
- logger.warning(
390
- f"Too many consecutive free actions ({self._consecutive_free_actions}), forcing 'wait'"
391
- )
392
- action = "wait"
393
- self._consecutive_free_actions = 0
394
- else:
395
- self._consecutive_free_actions = 1
396
- else:
397
- self._consecutive_free_actions = 0
398
-
399
- # Check for manual termination
400
- if action == "terminate":
401
- self.public_state.terminated = True
402
- self.private_state.terminated = True
403
- self.public_state.message = "Game terminated by agent."
404
- return self.private_state, self.public_state
405
-
406
- # Check for timeout
407
- if self.public_state.turn_count >= self.public_state.max_turns:
408
- self.public_state.terminated = True
409
- self.private_state.terminated = True
410
- self.private_state.truncated = True
411
- self.public_state.message = "Time limit reached. Game over!"
412
- return self.private_state, self.public_state
413
-
414
- # Execute action in NLE
415
- try:
416
- # Save previous observation BEFORE stepping
417
- prev_obs = self.last_nle_obs
418
-
419
- obs, reward, done, info = await asyncio.to_thread(self.nle.step, action)
420
- logger.debug(f"NLE step returned - reward: {reward}, done: {done}, info: {info}")
421
- except Exception as e:
422
- logger.error(f"NLE step failed for action '{action}': {e}")
423
- raise
424
-
425
- # Log observation structure on first few steps for debugging
426
- if self.public_state.turn_count < 3:
427
- logger.info(f"Turn {self.public_state.turn_count} observation keys: {list(obs.keys())}")
428
-
429
- # Update state from NLE observation - no defensive coding
430
- player_stats = obs["player_stats"] # Will KeyError if missing
431
-
432
- # Track previous values for reward calculation
433
- prev_score = self.private_state.score
434
- prev_depth = self.private_state.depth_reached
435
-
436
- # Update private state
437
- self.private_state.score = player_stats["score"]
438
- self.private_state.depth_reached = max(
439
- self.private_state.depth_reached, player_stats["depth"]
440
- )
441
- self.private_state.experience_level = player_stats["experience_level"]
442
-
443
- # Update public state
444
- self.public_state.dungeon_level = player_stats["depth"]
445
- self.public_state.position = (player_stats["y"], player_stats["x"])
446
- self.public_state.ascii_map = obs["ascii_map"]
447
- self.public_state.message = obs["message"]
448
- self.public_state.cursor_position = obs.get(
449
- "cursor", (player_stats["y"], player_stats["x"])
450
- )
451
- self.public_state.in_menu = obs.get("in_menu", False)
452
- self.public_state.menu_items = obs.get("menu_text", [])
453
-
454
- # Update character stats - require all fields
455
- self.public_state.character_stats = {
456
- "hp": player_stats["hp"],
457
- "max_hp": player_stats["max_hp"],
458
- "strength": player_stats["strength"],
459
- "dexterity": player_stats["dexterity"],
460
- "constitution": player_stats["constitution"],
461
- "intelligence": player_stats["intelligence"],
462
- "wisdom": player_stats["wisdom"],
463
- "charisma": player_stats["charisma"],
464
- "gold": player_stats["gold"],
465
- "experience": player_stats["experience_points"],
466
- "level": player_stats["experience_level"],
467
- "ac": player_stats["ac"],
468
- "score": player_stats["score"],
469
- }
470
-
471
- # Update inventory
472
- self.public_state.inventory = (
473
- self._process_inventory(obs["inventory"]) if "inventory" in obs else []
474
- )
475
-
476
- # Handle termination from NLE
477
- if done:
478
- self.public_state.terminated = True
479
- self.private_state.terminated = True
480
- # Log info to understand structure
481
- logger.info(f"Game ended - info: {info}")
482
- if "end_status" in info and info["end_status"] == 0: # 0 means death
483
- self.public_state.message = info.get(
484
- "death_reason", "You died!"
485
- ) # death_reason might not always exist
486
- else:
487
- self.public_state.message = "Game ended."
488
-
489
- # Update achievements before calculating rewards
490
- newly_unlocked = self.public_state.achievements.update_from_observation(obs, prev_obs)
491
- self.public_state.achievements_unlocked.update(
492
- self.public_state.achievements.get_unlocked_achievements()
493
- )
494
-
495
- # Log newly unlocked achievements
496
- if newly_unlocked:
497
- logger.info(f"Achievements unlocked: {list(newly_unlocked.keys())}")
498
-
499
- # Calculate rewards
500
- # Base reward from NLE
501
- nle_reward = reward
502
-
503
- # Additional reward shaping
504
- step_reward = await self.reward_stack.step_reward(self.public_state, action)
505
-
506
- self.private_state.reward_last = nle_reward + step_reward
507
- self.private_state.total_reward += self.private_state.reward_last
508
-
509
- # Calculate Balrog-style reward
510
- self.private_state.balrog_reward_last = calculate_balrog_reward(obs, prev_obs)
511
- self.private_state.balrog_total_reward += self.private_state.balrog_reward_last
512
-
513
- # Log balrog reward changes with context
514
- if self.private_state.balrog_reward_last > 0:
515
- print(
516
- f"🏆 BALROG REWARD: +{self.private_state.balrog_reward_last:.3f} (total: {self.private_state.balrog_total_reward:.3f})"
517
- )
518
- balrog_score = self.public_state.achievements.balrog_progress.percent
519
- print(
520
- f" Balrog score: {balrog_score}% (dungeon: {self.public_state.achievements.balrog_progress.dungeon_progression}, exp: {self.public_state.achievements.balrog_progress.experience_progression})"
521
- )
522
-
523
- # NOW update last_nle_obs for next step
524
- self.last_nle_obs = obs
525
-
526
- return self.private_state, self.public_state
527
-
528
- def __del__(self):
529
- """Cleanup NLE environment on deletion."""
530
- if hasattr(self, "nle"):
531
- self.nle.close()
532
-
533
- async def _serialize_engine(self) -> NetHackEngineSnapshot:
534
- """Serialize current state."""
535
- if self.public_state is None or self.private_state is None:
536
- raise RuntimeError("Cannot serialize uninitialized engine")
537
-
538
- # Get NLE state
539
- nle_state = None
540
- try:
541
- nle_state_bytes = await asyncio.to_thread(self.nle.get_state)
542
- # Convert bytes to base64 string for JSON serialization
543
- nle_state = base64.b64encode(nle_state_bytes).decode("ascii")
544
- except Exception as e:
545
- logger.warning(f"Failed to serialize NLE state: {e}")
546
-
547
- task_dict = await self.task_instance.serialize()
548
- logger.debug(f"Serialized task instance: {task_dict}")
549
-
550
- return NetHackEngineSnapshot(
551
- task_instance_dict=task_dict,
552
- engine_snapshot={
553
- "public_state": {
554
- "dungeon_level": self.public_state.dungeon_level,
555
- "character_stats": self.public_state.character_stats,
556
- "inventory": self.public_state.inventory,
557
- "position": self.public_state.position,
558
- "ascii_map": self.public_state.ascii_map,
559
- "message": self.public_state.message,
560
- "cursor_position": self.public_state.cursor_position,
561
- "turn_count": self.public_state.turn_count,
562
- "max_turns": self.public_state.max_turns,
563
- "last_action": self.public_state.last_action,
564
- "terminated": self.public_state.terminated,
565
- "in_menu": self.public_state.in_menu,
566
- "menu_items": self.public_state.menu_items,
567
- },
568
- "private_state": {
569
- "reward_last": self.private_state.reward_last,
570
- "total_reward": self.private_state.total_reward,
571
- "terminated": self.private_state.terminated,
572
- "truncated": self.private_state.truncated,
573
- "score": self.private_state.score,
574
- "depth_reached": self.private_state.depth_reached,
575
- "experience_level": self.private_state.experience_level,
576
- "monsters_killed": self.private_state.monsters_killed,
577
- "items_collected": self.private_state.items_collected,
578
- },
579
- "character_role": self.character_role,
580
- "progress_last_depth": self.progress_component.last_depth,
581
- "score_last_score": self.score_component.last_score,
582
- },
583
- nle_state=nle_state,
584
- )
585
-
586
- @classmethod
587
- async def _deserialize_engine(cls, snapshot: NetHackEngineSnapshot) -> "NetHackEngine":
588
- """Restore from serialized state."""
589
- from .taskset import NetHackTaskInstance
590
-
591
- task_instance = await NetHackTaskInstance.deserialize(snapshot.task_instance_dict)
592
- if task_instance is None:
593
- raise ValueError("Failed to deserialize task instance")
594
- engine = cls(task_instance)
595
-
596
- # Restore state
597
- engine_data = snapshot.engine_snapshot
598
- pub_data = engine_data["public_state"]
599
- priv_data = engine_data["private_state"]
600
-
601
- engine.public_state = NetHackPublicState(
602
- dungeon_level=pub_data["dungeon_level"],
603
- character_stats=pub_data["character_stats"],
604
- inventory=pub_data["inventory"],
605
- position=(pub_data["position"][0], pub_data["position"][1]),
606
- ascii_map=pub_data["ascii_map"],
607
- message=pub_data["message"],
608
- cursor_position=(
609
- pub_data["cursor_position"][0],
610
- pub_data["cursor_position"][1],
611
- ),
612
- turn_count=pub_data["turn_count"],
613
- max_turns=pub_data["max_turns"],
614
- last_action=pub_data["last_action"],
615
- terminated=pub_data["terminated"],
616
- in_menu=pub_data["in_menu"],
617
- menu_items=pub_data["menu_items"],
618
- )
619
-
620
- engine.private_state = NetHackPrivateState(
621
- reward_last=priv_data["reward_last"],
622
- total_reward=priv_data["total_reward"],
623
- terminated=priv_data["terminated"],
624
- truncated=priv_data["truncated"],
625
- score=priv_data["score"],
626
- depth_reached=priv_data["depth_reached"],
627
- experience_level=priv_data["experience_level"],
628
- monsters_killed=priv_data["monsters_killed"],
629
- items_collected=priv_data["items_collected"],
630
- )
631
-
632
- engine.character_role = engine_data["character_role"]
633
-
634
- # Restore reward component states
635
- engine.progress_component.last_depth = engine_data["progress_last_depth"]
636
- engine.score_component.last_score = engine_data["score_last_score"]
637
-
638
- # Restore NLE state if available
639
- if snapshot.nle_state:
640
- try:
641
- nle_state_bytes = base64.b64decode(snapshot.nle_state)
642
- await asyncio.to_thread(engine.nle.set_state, nle_state_bytes)
643
- except Exception as e:
644
- logger.warning(f"Failed to restore NLE state: {e}")
645
- # If we can't restore NLE state, reset it
646
- await asyncio.to_thread(engine.nle.reset)
647
-
648
- return engine
649
-
650
- def get_current_states_for_observation(
651
- self,
652
- ) -> Tuple[NetHackPrivateState, NetHackPublicState]:
653
- """Get current states without advancing."""
654
- if self.public_state is None or self.private_state is None:
655
- raise RuntimeError("Engine not initialized")
656
- return self.private_state, self.public_state
657
-
658
-
659
- class NetHackObservationCallable(GetObservationCallable):
660
- """Standard observation callable for NetHack."""
661
-
662
- async def get_observation(
663
- self, pub: NetHackPublicState, priv: NetHackPrivateState
664
- ) -> InternalObservation:
665
- observation = {
666
- "ascii_map": pub.ascii_map,
667
- "message": pub.message,
668
- "character_stats": pub.character_stats,
669
- "inventory_summary": self._format_inventory(pub.inventory),
670
- "dungeon_level": pub.dungeon_level,
671
- "position": pub.position,
672
- "turn_count": pub.turn_count,
673
- "last_action": pub.last_action,
674
- "reward_last": priv.reward_last,
675
- "total_reward": priv.total_reward,
676
- "balrog_reward_last": priv.balrog_reward_last,
677
- "balrog_total_reward": priv.balrog_total_reward,
678
- "score": priv.score,
679
- "experience_level": priv.experience_level,
680
- "terminated": priv.terminated,
681
- "in_menu": pub.in_menu,
682
- "menu_items": pub.menu_items if pub.in_menu else [],
683
- "achievements_unlocked": pub.achievements_unlocked,
684
- "achievements_summary": self._format_achievements(pub.achievements_unlocked),
685
- }
686
- return observation # type: ignore[return-value]
687
-
688
- def _format_inventory(self, inventory: List[Dict[str, Any]]) -> str:
689
- """Format inventory for display."""
690
- if not inventory:
691
- return "Your inventory is empty."
692
-
693
- items = []
694
- for item in inventory:
695
- items.append(f"- {item['name']} (count: {item.get('count', 1)})")
696
- return "\n".join(items)
697
-
698
- def _format_achievements(self, achievements: Dict[str, bool]) -> str:
699
- """Format achievements for display."""
700
- unlocked = [name for name, status in achievements.items() if status]
701
- if not unlocked:
702
- return "None unlocked yet"
703
- if len(unlocked) <= 5:
704
- return ", ".join(unlocked)
705
- else:
706
- return f"{', '.join(unlocked[:5])} and {len(unlocked) - 5} more"
707
-
708
-
709
- class NetHackCheckpointObservationCallable(GetObservationCallable):
710
- """Checkpoint observation callable for NetHack."""
711
-
712
- async def get_observation(
713
- self, pub: NetHackPublicState, priv: NetHackPrivateState
714
- ) -> InternalObservation:
715
- observation = {
716
- "final_score": priv.score,
717
- "max_depth": priv.depth_reached,
718
- "experience_level": priv.experience_level,
719
- "monsters_killed": priv.monsters_killed,
720
- "items_collected": priv.items_collected,
721
- "turn_count_final": pub.turn_count,
722
- "total_reward": priv.total_reward,
723
- "balrog_total_reward": priv.balrog_total_reward,
724
- "terminated": priv.terminated,
725
- "truncated": priv.truncated,
726
- "character_role": pub.character_stats.get("role", "unknown"),
727
- "achievements_unlocked": list(pub.achievements_unlocked.keys()),
728
- "achievements_count": len([v for v in pub.achievements_unlocked.values() if v]),
729
- "achievement_stats": {
730
- "depth_reached": pub.achievements.depth_reached,
731
- "monsters_killed": pub.achievements.monsters_killed,
732
- "gold_collected": pub.achievements.gold_collected,
733
- "items_collected": pub.achievements.items_picked_up,
734
- "max_level": pub.achievements.max_level_reached,
735
- "turns_survived": pub.achievements.turns_survived,
736
- "balrog_score": pub.achievements.balrog_progress.percent,
737
- },
738
- }
739
- return observation # type: ignore[return-value]