agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,616 @@
1
+ """The runtime a generated world runs on.
2
+
3
+ A generated world is a database plus one handler per tool. The handler decides what a call does;
4
+ this decides what a handler is allowed to be, what happens when one fails, and what the world
5
+ looks like afterwards. Keeping that here means a generated file stays small enough to read and
6
+ correct, and the parts that must be exact are not regenerated every time.
7
+
8
+ The contract with the rest of the platform is ``EnvironmentAdapter``: ``reset`` publishes the
9
+ tools and the starting state, ``handle_tool_call`` executes one call, and the state afterwards is
10
+ what the checks grade. A world is therefore drivable by any loop that already drives an
11
+ environment, which is the whole reason we generate against this interface rather than inventing
12
+ one.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import copy
18
+ import json
19
+ import re
20
+ import sqlite3
21
+ import time
22
+ from dataclasses import dataclass, field
23
+ from pathlib import Path
24
+ from typing import Any, Mapping, Sequence
25
+
26
+ from ..environment import EnvironmentAdapter, EnvironmentSnapshot, ToolExecutionResult
27
+
28
+
29
+ class ToolError(Exception):
30
+ """A tool refusing for a real reason the agent should see and recover from.
31
+
32
+ Distinct from a crash. A refusal is the world working: the id does not exist, the item is
33
+ unavailable, the argument is outside what the tool accepts. A crash is our bug, and the two
34
+ must never look the same to a caller deciding whether the agent behaved correctly.
35
+ """
36
+
37
+
38
+ @dataclass
39
+ class Db:
40
+ """The handle a handler gets. Deliberately small: query, execute, one.
41
+
42
+ Handlers get a database, not a filesystem and not a network. Anything a handler can reach is
43
+ something a generated world could depend on, and a world that depends on the outside is not
44
+ reproducible.
45
+ """
46
+
47
+ # Whatever this agent's records live in. A handler's statements are written in that store's
48
+ # own language, so this passes them through rather than interpreting them.
49
+ store: Any
50
+ # The agent's own state object, where its tools keep what they act on in memory rather than
51
+ # in a database. Their code is the thing that shapes it, so the world holds it and does not
52
+ # interpret it: freezing it is a serialisation, and restoring it is the reverse.
53
+ state: Any = None
54
+
55
+ def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
56
+ return self.store.query(sql, params)
57
+
58
+ def one(self, sql: str, params: Sequence[Any] = ()) -> dict[str, Any] | None:
59
+ rows = self.query(sql, params)
60
+ return rows[0] if rows else None
61
+
62
+ def execute(self, sql: str, params: Sequence[Any] = ()) -> int:
63
+ return self.store.execute(sql, params)
64
+
65
+ # -- reading without a query language ---------------------------------------------
66
+ #
67
+ # Not every agent has a database. One whose state lives in services and files gets a world
68
+ # whose collections the harness invented, and there is no dialect to write a SELECT in. A
69
+ # handler that could only issue SQL would be unable to read the world it was given at all.
70
+
71
+ def collections(self) -> list[str]:
72
+ """Every collection this world holds, by name."""
73
+ return list(self.store.collections())
74
+
75
+ def records(self, collection: str) -> list[dict[str, Any]]:
76
+ """Every record in one collection. The store-agnostic way to read."""
77
+ return list(self.store.records(collection))
78
+
79
+ def find(self, collection: str, **fields: Any) -> list[dict[str, Any]]:
80
+ """The records in a collection whose fields all match what was asked for."""
81
+ return [
82
+ record
83
+ for record in self.records(collection)
84
+ if all(record.get(field) == value for field, value in fields.items())
85
+ ]
86
+
87
+ def add(self, collection: str, record: Mapping[str, Any]) -> int | dict[str, Any]:
88
+ return self.store.add(collection, record)
89
+
90
+
91
+ def settled(value: Any) -> Any:
92
+ """The value, with a coroutine run to completion first.
93
+
94
+ A tool the agent wrote may well be async: every framework-decorated tool is. Handlers here
95
+ are synchronous, and the build stage is itself inside a running event loop, so ``asyncio.run``
96
+ cannot be called directly. Running it on a worker thread gives it a loop of its own and keeps
97
+ the handler contract unchanged.
98
+ """
99
+ import asyncio
100
+ import inspect
101
+
102
+ if not inspect.isawaitable(value):
103
+ return value
104
+ from concurrent.futures import ThreadPoolExecutor
105
+
106
+ with ThreadPoolExecutor(max_workers=1) as pool:
107
+ return pool.submit(asyncio.run, value).result()
108
+
109
+
110
+ def _is_refusal(raised: BaseException, refusal_signature: str = "") -> bool:
111
+ """Whether an exception is the world saying no, rather than the world falling over.
112
+
113
+ Matched by name as well as by identity. A generated handler often declares its own
114
+ ``ToolError`` rather than using the one already in scope, which is defensive and sensible
115
+ from where it sits, and would otherwise turn every deliberate refusal into a reported crash.
116
+ Relying on an invisible convention being followed is not a way to decide something this
117
+ load-bearing.
118
+ """
119
+ if isinstance(raised, ToolError):
120
+ return True
121
+ classes = [base.__name__ for base in type(raised).__mro__]
122
+ if "ToolError" in classes:
123
+ return True
124
+ # Submitted tools commonly use their own semantic exception type (for example
125
+ # ``LookupError`` for a missing account) rather than importing the harness's ToolError.
126
+ # Treat it as a refusal only when source understanding explicitly recorded that exception
127
+ # in the contract. Broadly classifying LookupError/ValueError would hide handler defects;
128
+ # grounding this in the refusal signature preserves the refusal-versus-crash boundary.
129
+ described = refusal_signature or ""
130
+ semantic_classes = [
131
+ name
132
+ for name in classes
133
+ if name not in {"BaseException", "Exception"} and name.endswith("Error")
134
+ ]
135
+ return any(
136
+ re.search(rf"\b{re.escape(name)}\b", described) is not None
137
+ for name in semantic_classes
138
+ )
139
+
140
+
141
+ @dataclass
142
+ class Call:
143
+ """One tool call and what the world did with it."""
144
+
145
+ name: str
146
+ arguments: dict[str, Any]
147
+ result: Any = None
148
+ ok: bool = True
149
+ error: str = ""
150
+ refused: bool = False
151
+ # When it happened, seconds since the epoch. What lets a recording and a list of calls be
152
+ # read as one thing: without it the UI can show what the agent did but not when, and "when"
153
+ # is the whole question for a spoken run.
154
+ at: float = 0.0
155
+
156
+
157
+ class GeneratedWorld(EnvironmentAdapter):
158
+ """A database-backed world whose tools are generated per agent.
159
+
160
+ Subclasses declare ``name``, ``tools`` and ``handlers``. Everything about execution,
161
+ refusal, and state reporting is here so that a generated subclass carries only the parts
162
+ that are specific to one agent.
163
+ """
164
+
165
+ name = "generated"
166
+ tools: list[dict[str, Any]] = []
167
+ handlers: dict[str, str] = {}
168
+
169
+ # Where the agent's own code lives, so a handler that binds to one of its tools can import
170
+ # it. Empty when the world implements the tools itself.
171
+ source_root: str = ""
172
+ # The agent's own in-memory state, for tools that take it as an argument instead of
173
+ # connecting to anything. Held opaquely: their code gives it shape.
174
+ state_object: Any = None
175
+ # How this agent says no in a returned value. A tool that answers "Error: no such order" is
176
+ # refusing, and recording that as a success would hide the very behaviour worth testing.
177
+ refusal_signature: str = ""
178
+
179
+ def __init__(
180
+ self, database: str | Path = ":memory:", *, store: Any = None, kind: str = ""
181
+ ) -> None:
182
+ from .stores import open_store
183
+
184
+ # Generated subclasses declare these as class-level templates. Every world must own
185
+ # its copies: binding and probing mutate them, and sharing the template lets one
186
+ # agent's tools leak into a later agent authored by the same worker process.
187
+ self.tools = copy.deepcopy(type(self).tools)
188
+ self.handlers = dict(type(self).handlers)
189
+ self.state_object = copy.deepcopy(type(self).state_object)
190
+ self.database = str(database)
191
+ # Where this agent's records live. Given rather than assumed, because the harness
192
+ # writes statements in whatever the agent's own store speaks and they have to reach
193
+ # it. A world with no store of its own gets one that says so.
194
+ self.store = store or open_store(kind or "sqlite", database=self.database)
195
+ # A world is usable as soon as it is constructed. SQLite happens to open its
196
+ # connection in __init__, which used to hide this missing lifecycle step; container
197
+ # stores (Postgres, MySQL, …) do not have an address until start() is called.
198
+ # Starting here also makes snapshot.restore() safe, since load_from() immediately
199
+ # writes the frozen rows into the newly-created store.
200
+ self.store.start()
201
+ self.calls: list[Call] = []
202
+
203
+ @property
204
+ def connection(self) -> Any:
205
+ """The store's own connection, where it has one.
206
+
207
+ Kept so that code written when every world was a SQLite file still works. Anything
208
+ new should go through the store, or through put, change and drop, so it holds for a
209
+ world whose records are somewhere else.
210
+ """
211
+ found = getattr(self.store, "connection", None)
212
+ if found is None:
213
+ raise AttributeError(
214
+ f"this world's store ({getattr(self.store, 'key', 'unknown')}) has no "
215
+ "connection. Use the store, or put, change and drop."
216
+ )
217
+ return found
218
+
219
+ def reach(self, source_root: str) -> None:
220
+ """Make the agent's own code importable, so a binding can call it rather than copy it.
221
+
222
+ Two directories go on the path, not one. An agent pointed at flatly is imported from where
223
+ it sits, but an agent laid out as a package is nearly always pointed at the part under
224
+ test rather than at its root: `tau_bench/envs/retail` is where the agent is, while
225
+ `tau_bench.envs.retail.data` only resolves from the repository above it. Adding just the
226
+ directory named makes every import the agent's own code writes fail, which arrives as
227
+ "No module named tau_bench" and reads as the package being absent rather than as us
228
+ having pointed at the middle of it.
229
+
230
+ The package root is found the way Python finds it: walk up while each directory is itself
231
+ a package, and stop at the first that is not.
232
+ """
233
+ import sys
234
+
235
+ self.source_root = str(source_root or "")
236
+ for path in self._import_roots(self.source_root):
237
+ if path not in sys.path:
238
+ sys.path.insert(0, path)
239
+
240
+ @staticmethod
241
+ def _import_roots(source_root: str) -> list[str]:
242
+ """Where the agent's code can be imported from: where it sits, and its package root."""
243
+ if not source_root:
244
+ return []
245
+ roots = [source_root]
246
+ here = Path(source_root)
247
+ # Bounded by the filesystem root: `parents` stops there, so a source outside any package
248
+ # simply never enters the loop.
249
+ while (here / "__init__.py").exists() and here.parent != here:
250
+ here = here.parent
251
+ if str(here) not in roots:
252
+ roots.append(str(here))
253
+ return roots
254
+
255
+ # -- EnvironmentAdapter ----------------------------------------------------------
256
+
257
+ def reset(self, **_context: Any) -> EnvironmentSnapshot:
258
+ self.calls = []
259
+ return EnvironmentSnapshot(tools=list(self.tools), state=self.state())
260
+
261
+ def observe(self, **_context: Any) -> EnvironmentSnapshot:
262
+ return EnvironmentSnapshot(tools=list(self.tools), state=self.state())
263
+
264
+ def handle_tool_call(
265
+ self, tool_call: Mapping[str, Any], **_context: Any
266
+ ) -> ToolExecutionResult | None:
267
+ name = str(
268
+ tool_call.get("name") or (tool_call.get("function") or {}).get("name") or ""
269
+ )
270
+ call_id = tool_call.get("id") or tool_call.get("tool_call_id")
271
+ arguments = tool_call.get("arguments") or tool_call.get("args") or {}
272
+ if not isinstance(arguments, Mapping):
273
+ arguments = {}
274
+
275
+ call = self.call(name, arguments)
276
+ content = (
277
+ json.dumps(call.result, default=str)
278
+ if not isinstance(call.result, str)
279
+ else call.result
280
+ )
281
+ return ToolExecutionResult(
282
+ tool_call_id=call_id,
283
+ tool_name=name or "unknown",
284
+ content=call.error if not call.ok else content,
285
+ result=call.result,
286
+ success=call.ok,
287
+ error=call.error or None,
288
+ state_updates=self.state(),
289
+ )
290
+
291
+ # -- execution -------------------------------------------------------------------
292
+
293
+ def call(self, name: str, arguments: Mapping[str, Any] | None = None) -> Call:
294
+ """Execute one call and record it. Never raises: a failure is an outcome, not an event.
295
+
296
+ An unknown tool is a refusal rather than a silent success. An agent reaching for a tool
297
+ that does not exist is a finding, and answering it with an acknowledgement is how a test
298
+ passes something it should have caught.
299
+ """
300
+ args = dict(arguments or {})
301
+ if name not in self.handlers:
302
+ return self._record(
303
+ Call(
304
+ name=name,
305
+ arguments=args,
306
+ ok=False,
307
+ refused=True,
308
+ error=(
309
+ f"no such tool {name!r}; this agent has "
310
+ f"{', '.join(sorted(self.handlers)) or 'none'}"
311
+ ),
312
+ )
313
+ )
314
+
315
+ namespace: dict[str, Any] = {"ToolError": ToolError, "json": json}
316
+ try:
317
+ exec(compile(self.handlers[name], f"<handler:{name}>", "exec"), namespace)
318
+ handle = namespace.get("handle")
319
+ if not callable(handle):
320
+ raise RuntimeError("handler defines no handle(args, db)")
321
+ value = handle(args, Db(self.store, self.state_object))
322
+ except Exception as raised:
323
+ if _is_refusal(raised, self.refusal_signature):
324
+ return self._record(
325
+ Call(
326
+ name=name,
327
+ arguments=args,
328
+ ok=False,
329
+ refused=True,
330
+ error=str(raised),
331
+ )
332
+ )
333
+ # Our bug, not the agent's. Labelled differently so a run is never scored
334
+ # against a world that fell over.
335
+ return self._record(
336
+ Call(
337
+ name=name,
338
+ arguments=args,
339
+ ok=False,
340
+ error=f"{type(raised).__name__}: {raised}",
341
+ )
342
+ )
343
+ # A tool of the agent's own may refuse by returning rather than by raising, which is
344
+ # ordinary in code that was never written to be tested. Recording that as a success
345
+ # would hide exactly the behaviour worth measuring, so the agent's own convention
346
+ # decides. Only the recording differs: the value still reaches the agent unchanged.
347
+ if self._refused_by_value(value):
348
+ return self._record(
349
+ Call(
350
+ name=name,
351
+ arguments=args,
352
+ result=value,
353
+ ok=False,
354
+ refused=True,
355
+ error=str(value)[:400],
356
+ )
357
+ )
358
+ return self._record(Call(name=name, arguments=args, result=value))
359
+
360
+ def _refused_by_value(self, value: Any) -> bool:
361
+ """Whether a returned value is this agent's way of saying no.
362
+
363
+ The convention is recorded as a description, because that is what somebody reading the
364
+ agent's code can actually write: "strings starting with Error:". So the marker is taken
365
+ from inside it rather than treating the whole sentence as a prefix, which would match
366
+ nothing and quietly record every refusal as a success.
367
+ """
368
+ if not isinstance(value, str) or not value:
369
+ return False
370
+ described = (self.refusal_signature or "").strip()
371
+ if not described:
372
+ return False
373
+ for marker in self._markers(described):
374
+ if value.lower().startswith(marker.lower()):
375
+ return True
376
+ return False
377
+
378
+ def _markers(self, described: str) -> list[str]:
379
+ """The literal markers named inside a described convention.
380
+
381
+ Anything quoted is taken as written, since that is how a convention gets spelled out. With
382
+ nothing quoted the whole description is treated as the marker, which is right when somebody
383
+ recorded just the prefix itself.
384
+ """
385
+ import re
386
+
387
+ # A convention written for people gets quoted the way people quote, and a model writing
388
+ # JSON often escapes those quotes. Left in, the backslash ends up inside the marker, so
389
+ # "Error:" is looked for as 'Error:\' and matches nothing at all. Every refusal is then
390
+ # recorded as a success, which is the failure this whole field exists to prevent.
391
+ plain = described.replace('\\"', '"').replace("\\'", "'")
392
+ quoted = re.findall(r"[\"'“”‘’`]([^\"'“”‘’`]{1,40})[\"'“”‘’`]", plain)
393
+ found = [one.strip().strip("\\").strip() for one in quoted]
394
+ # A convention that lists examples separates them, and the separator sits between one
395
+ # closing quote and the next opening one, so it is matched as though it were quoted too.
396
+ # A marker of "," would make any result beginning with a comma a refusal, so anything
397
+ # without a character a message could start with is dropped.
398
+ found = [one for one in found if any(char.isalnum() for char in one)]
399
+ return found or [plain.strip()]
400
+
401
+ def _record(self, call: Call) -> Call:
402
+ # Stamped here rather than by the caller, so every call is stamped and none of them
403
+ # depend on whoever made it remembering to.
404
+ call.at = call.at or time.time()
405
+ self.calls.append(call)
406
+ return call
407
+
408
+ # -- state -----------------------------------------------------------------------
409
+
410
+ def _settle(self) -> None:
411
+ """Close any transaction left open on the connection.
412
+
413
+ A handler that only reads still leaves an implicit read transaction behind, and SQLite
414
+ refuses to back up into a connection that has one open: "destination database is in
415
+ use". Left unsettled, the first read-only handler poisons every probe after it, and the
416
+ world can never be checked or saved.
417
+ """
418
+ connection = getattr(self.store, "connection", None)
419
+ if connection is None:
420
+ # Nothing to settle. A store with no transactions has no open one to close, and
421
+ # reaching for a connection it never had would fail every probe on such a world.
422
+ return
423
+ try:
424
+ connection.commit()
425
+ except sqlite3.Error:
426
+ connection.rollback()
427
+
428
+ def checkpoint(self) -> Any:
429
+ """A copy of everything the world holds, to come back to.
430
+
431
+ Probes and smoke calls mutate: ordering an item inserts a record, cancelling one changes
432
+ it. Without a way back, each runs against the debris of the ones before it, and a check
433
+ expecting three records finds seven.
434
+
435
+ Both halves are copied, and that matters more for an adopted world than a generated one.
436
+ A tool the agent wrote changes the structure it was given, in place. Backing up only the
437
+ store would leave those changes permanent, so a smoke call against one record would quietly
438
+ spend it, and whatever ran later against that same record would fail for a reason nothing
439
+ could see.
440
+ """
441
+ import copy as duplicate
442
+
443
+ self._settle()
444
+ # Through the store's own freeze rather than a SQLite backup, so a world whose records
445
+ # live somewhere else is revertible too. Every store knows how to go back; only some of
446
+ # them have a connection to copy.
447
+ store = self.store.freeze()
448
+ held = (
449
+ duplicate.deepcopy(self.state_object)
450
+ if self.state_object is not None
451
+ else None
452
+ )
453
+ return {"store": store, "state": held}
454
+
455
+ def revert(self, checkpoint: Any) -> None:
456
+ """Put everything back as it was when the checkpoint was taken."""
457
+ import copy as duplicate
458
+
459
+ self._settle()
460
+ # A bare connection is accepted so that anything written against the older shape of this
461
+ # method keeps working rather than reverting nothing at all, which would be silent.
462
+ if isinstance(checkpoint, sqlite3.Connection):
463
+ checkpoint.backup(self.connection)
464
+ return
465
+ held = (checkpoint or {}).get("store")
466
+ if held is not None:
467
+ self.store.restore(held)
468
+ if (checkpoint or {}).get("state") is not None:
469
+ self.state_object = duplicate.deepcopy(checkpoint["state"])
470
+
471
+ def state(self) -> dict[str, Any]:
472
+ """What the checks compare against after a run.
473
+
474
+ Tables and their rows, plus whatever the agent's own tools keep in memory. A world that
475
+ adopted the agent's code may have all of its state in the second of those, so a check has
476
+ to be able to see both without knowing which kind of world it is grading.
477
+ """
478
+ found: dict[str, Any] = {
479
+ name: self.store.records(name) for name in self.store.collections()
480
+ }
481
+ if isinstance(self.state_object, dict):
482
+ # Collections the agent's own code owns. Not merged blindly: a table and a key of
483
+ # the same name would silently shadow one another, and a check comparing the wrong
484
+ # one would be wrong in a way nobody could see.
485
+ for key, value in self.state_object.items():
486
+ found.setdefault(str(key), value)
487
+ elif self.state_object is not None:
488
+ found.setdefault("state", self.state_object)
489
+ return found
490
+
491
+ # -- changing the world, without naming what it is kept in ------------------------
492
+ #
493
+ # A scenario changes the world before it runs, and it must not have to know whether the world
494
+ # is a database, a mapping the agent's own code owns, or something else again. Speaking SQL
495
+ # here would write SQLite into every scenario ever written, and the store is the one thing
496
+ # this design expects to vary per agent.
497
+ #
498
+ # So the vocabulary is collections and records, which every store has under some name, and
499
+ # each method dispatches on what the collection actually is. The preferred way to change the
500
+ # world is still the agent's own tools, because anything they refuse would have refused the
501
+ # agent too; these are for the states no tool can produce.
502
+
503
+ def _table(self, collection: str) -> bool:
504
+ return bool(self.store.holds(collection))
505
+
506
+ def _held(self, collection: str) -> Any:
507
+ if isinstance(self.state_object, dict):
508
+ return self.state_object.get(collection)
509
+ return None
510
+
511
+ def put(self, collection: str, record: Mapping[str, Any], *, key: str = "") -> None:
512
+ """Add one record to a collection, whatever the collection is kept in."""
513
+ if self._table(collection):
514
+ self.store.add(collection, record)
515
+ return
516
+ held = self._held(collection)
517
+ if isinstance(held, dict):
518
+ if not key:
519
+ raise KeyError(
520
+ f"{collection} is keyed, so adding to it needs a key: "
521
+ "world.put(collection, record, key=...)"
522
+ )
523
+ held[key] = dict(record)
524
+ return
525
+ if isinstance(held, list):
526
+ held.append(dict(record))
527
+ return
528
+ # A collection nobody has created yet is made here rather than refused. An agent whose
529
+ # state lives in services and files has no store to declare tables in, so every collection
530
+ # the world needs is one the harness invents: refusing the first record leaves that agent
531
+ # with a world that cannot hold anything at all.
532
+ made = getattr(self.store, "start_collection", None)
533
+ if callable(made):
534
+ made(collection, keyed=bool(key))
535
+ self.store.add(collection, {**record, "_id": key} if key else record)
536
+ return
537
+ raise KeyError(
538
+ f"no collection called {collection!r}; this world has {sorted(self.state())}"
539
+ )
540
+
541
+ def change(
542
+ self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
543
+ ) -> int:
544
+ """Change records in a collection. Returns how many were changed.
545
+
546
+ ``by`` names the column a table is keyed on. A collection the agent's own code keeps is
547
+ keyed already, so it is not needed there.
548
+ """
549
+ if self._table(collection):
550
+ return self.store.amend(collection, key, changes, by=by)
551
+ held = self._held(collection)
552
+ if isinstance(held, dict) and key in held:
553
+ if isinstance(held[key], dict):
554
+ held[key].update(dict(changes))
555
+ else:
556
+ held[key] = dict(changes)
557
+ return 1
558
+ raise KeyError(f"nothing called {key!r} in {collection!r}")
559
+
560
+ def drop(self, collection: str, key: str = "", *, by: str = "") -> int:
561
+ """Remove a record, or the whole contents of a collection when no key is given."""
562
+ if self._table(collection):
563
+ return self.store.remove(collection, key, by=by)
564
+ held = self._held(collection)
565
+ if isinstance(held, dict):
566
+ if not key:
567
+ count = len(held)
568
+ held.clear()
569
+ return count
570
+ return 1 if held.pop(key, None) is not None else 0
571
+ if isinstance(held, list):
572
+ count = len(held)
573
+ del held[:]
574
+ return count
575
+ raise KeyError(
576
+ f"no collection called {collection!r}; this world has {sorted(self.state())}"
577
+ )
578
+
579
+ def shapes(self) -> str:
580
+ """What this world's collections actually are, in words.
581
+
582
+ Said wherever code written against the wrong shape fails. A table gives a list of records;
583
+ a collection the agent's own code keeps is often a mapping keyed by identifier, and
584
+ iterating that yields strings. No amount of general advice substitutes for naming which is
585
+ which, for the world in front of whoever got it wrong.
586
+ """
587
+ lines = []
588
+ for name, held in sorted(self.state().items()):
589
+ if isinstance(held, dict):
590
+ first = next(iter(held), None)
591
+ lines.append(
592
+ f" {name}: a mapping of {len(held)} records keyed by identifier"
593
+ + (f", e.g. {first!r}" if first is not None else "")
594
+ + ". Iterate .values(), or .items() when the key matters."
595
+ )
596
+ elif isinstance(held, list):
597
+ lines.append(
598
+ f" {name}: a list of {len(held)} records. Iterate it directly."
599
+ )
600
+ else:
601
+ lines.append(f" {name}: a single {type(held).__name__}.")
602
+ return "This world holds:\n" + ("\n".join(lines) or " nothing yet")
603
+
604
+ def close(self) -> None:
605
+ self.store.close()
606
+
607
+
608
+ @dataclass
609
+ class WorldSpec:
610
+ """What a generated world is, before it is written out."""
611
+
612
+ agent: str
613
+ schema_sql: str = ""
614
+ tools: list[dict[str, Any]] = field(default_factory=list)
615
+ handlers: dict[str, str] = field(default_factory=dict)
616
+ notes: str = ""