agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,288 @@
1
+ """Freezing a world, and starting every scenario from the same frozen copy.
2
+
3
+ The database is built once and snapshotted; that snapshot is the base state. A scenario restores
4
+ its own copy and layers on whatever it additionally needs, so scenarios cannot inherit each
5
+ other's leftovers and a run is repeatable a week later.
6
+
7
+ Which is why the overlay exists: a scenario that needs a customer with three open orders adds
8
+ those rows to a restored copy rather than editing the snapshot. The base world stays the shared
9
+ starting point instead of drifting toward whichever scenario was written last.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import shutil
16
+ import sqlite3
17
+ from pathlib import Path
18
+ from typing import Any, Mapping
19
+
20
+ from .runtime import GeneratedWorld
21
+
22
+ DATABASE = "world.sqlite"
23
+ HANDLERS = "handlers"
24
+ MANIFEST = "manifest.json"
25
+ STATE = "state.json"
26
+
27
+
28
+ def saved(path: str | Path | None) -> bool:
29
+ """Whether a world has been written here.
30
+
31
+ One function, because this question gets asked from six places: the build stage, the
32
+ conversation, the session listing, the CLI and the UI. Asked as "is there a world.sqlite"
33
+ each of those was really asking "is this a SQLite world", so an agent whose state lives in
34
+ services and files saved a world that scored 1.00 and was then invisible to all of them.
35
+ """
36
+ return bool(path) and (Path(path) / MANIFEST).exists()
37
+
38
+
39
+ WORLD_MODULE = "world.py"
40
+
41
+ _MODULE = '''"""Generated world for {agent}. Do not edit by hand; regenerate instead.
42
+
43
+ {notes}
44
+ """
45
+
46
+ from pathlib import Path
47
+
48
+ from fi.alk.harness.world.runtime import GeneratedWorld
49
+
50
+ _HERE = Path(__file__).parent
51
+
52
+ TOOLS = {tools}
53
+
54
+
55
+ class World(GeneratedWorld):
56
+ name = {agent!r}
57
+ tools = TOOLS
58
+ handlers = {{
59
+ name: (_HERE / "handlers" / f"{{name}}.py").read_text(encoding="utf-8")
60
+ for name in {handler_names}
61
+ }}
62
+
63
+
64
+ def load(database=None):
65
+ """This world, restored from the snapshot beside this file.
66
+
67
+ Through `restore` rather than by opening a database directly, because not every world has
68
+ one: an agent whose state lives in services and files keeps its records in the snapshot, and
69
+ naming a SQLite file would hand back an empty world instead of this one.
70
+ """
71
+ from fi.alk.harness.world.snapshot import restore
72
+
73
+ return restore(_HERE, into=database) if database else restore(_HERE)
74
+ '''
75
+
76
+
77
+ def save(
78
+ world: GeneratedWorld,
79
+ path: str | Path,
80
+ *,
81
+ notes: str = "",
82
+ sequences: list[dict[str, Any]] | None = None,
83
+ world_checks: Mapping[str, str] | None = None,
84
+ ) -> Path:
85
+ """Write the world out: the snapshot, the handlers, the module, and a manifest."""
86
+ root = Path(path)
87
+ (root / HANDLERS).mkdir(parents=True, exist_ok=True)
88
+
89
+ # Through the store, so a world whose records live somewhere other than a SQLite file, or
90
+ # nowhere at all, freezes by its own means rather than by one assumed here.
91
+ world.store.save_to(root)
92
+
93
+ for name, source in world.handlers.items():
94
+ (root / HANDLERS / f"{name}.py").write_text(source, encoding="utf-8")
95
+
96
+ (root / WORLD_MODULE).write_text(
97
+ _MODULE.format(
98
+ agent=world.name,
99
+ notes=notes or "Generated from the agent's contract.",
100
+ tools=json.dumps(world.tools, indent=4),
101
+ handler_names=json.dumps(sorted(world.handlers)),
102
+ ),
103
+ encoding="utf-8",
104
+ )
105
+
106
+ # The agent's own in-memory state, where its tools keep what they act on there rather
107
+ # than in the database. Frozen as JSON so restoring is the exact reverse, and so a
108
+ # person can read what the world starts from.
109
+ if world.state_object is not None:
110
+ # Round-tripped rather than only written. Every scenario restores from this file, so state
111
+ # that does not survive the trip would come back subtly different and every check after
112
+ # it would be grading something else. Better to fail here than to be wrong quietly.
113
+ frozen = json.dumps(world.state_object, indent=2, default=str)
114
+ if json.loads(frozen) != world.state_object:
115
+ raise ValueError(
116
+ "the agent's state does not survive being frozen as JSON, so restoring it would "
117
+ "not give back what was saved. Every scenario starts from that restore, so this "
118
+ "world cannot be trusted. What is in the state that is not plain JSON?"
119
+ )
120
+ (root / STATE).write_text(frozen, encoding="utf-8")
121
+
122
+ state = world.state()
123
+ (root / MANIFEST).write_text(
124
+ json.dumps(
125
+ {
126
+ "agent": world.name,
127
+ # Which store this world used, so restoring it opens the same one rather
128
+ # than assuming a database that may never have existed.
129
+ "store": getattr(world.store, "key", "sqlite"),
130
+ "tools": sorted(world.handlers),
131
+ # Written because restore reads it. Without it a restored world publishes no
132
+ # tool descriptions at all, and every later stage has to reconstruct them.
133
+ "tool_specs": list(world.tools),
134
+ "tables": {name: len(rows) for name, rows in state.items()},
135
+ # Kept because they are judgement about this agent, not something a schema
136
+ # implies. A world picked up again can be re-verified without redeclaring them.
137
+ "sequences": list(sequences or []),
138
+ # The world's own checks are judgement about this agent, so a world picked
139
+ # up again keeps them rather than having them rewritten from scratch.
140
+ "world_checks": dict(world_checks or {}),
141
+ # Where the agent's own code lives. Kept because a restored world has to
142
+ # be able to import the tools it was bound to, and a scenario run happens
143
+ # long after the build stage that found the path.
144
+ "source_root": world.source_root,
145
+ # A run refuses legacy/demo worlds whose handlers were authored by the harness.
146
+ # New worlds can only acquire handlers through adopt_tool, and a source root is
147
+ # required for that import to work.
148
+ "tool_implementation": "source" if world.source_root else "synthetic",
149
+ "runtime_tools": sorted(getattr(world, "runtime_tools", set())),
150
+ "external_runtime": bool(getattr(world, "external_runtime", False)),
151
+ # How this agent says no in a returned value. Without it a restored world
152
+ # cannot tell a refusal from a success, so every run records "Error: no such
153
+ # order" as if the call worked, and a check asking whether the agent was
154
+ # refused is answered wrongly rather than reported as unanswerable.
155
+ "refusal_signature": world.refusal_signature,
156
+ "notes": notes,
157
+ },
158
+ indent=2,
159
+ ensure_ascii=False,
160
+ ),
161
+ encoding="utf-8",
162
+ )
163
+ return root
164
+
165
+
166
+ def restore(path: str | Path, *, into: str | Path | None = None) -> GeneratedWorld:
167
+ """A fresh, independent copy of the frozen world.
168
+
169
+ In memory by default, because a scenario should not be able to write back into the snapshot
170
+ every later scenario depends on.
171
+ """
172
+ root = Path(path)
173
+ source = root / DATABASE
174
+ if not (root / MANIFEST).exists():
175
+ raise FileNotFoundError(f"no world snapshot at {root}")
176
+
177
+ manifest = read_manifest(root)
178
+ provisioned_http = False
179
+ if (root / "environment.json").exists():
180
+ from ..provision import ProvisionedEnvironment
181
+
182
+ environment = ProvisionedEnvironment.load(root)
183
+ provisioned_http = bool(
184
+ environment
185
+ and any(
186
+ value.startswith(("http://", "https://"))
187
+ for value in environment.overrides.values()
188
+ )
189
+ )
190
+ if provisioned_http:
191
+ if into is not None:
192
+ raise ValueError(
193
+ "a source-provisioned world restores into its submitted service, not a database "
194
+ "file path"
195
+ )
196
+ from ..understand import load as load_contract
197
+ from .provisioned import open_provisioned_world
198
+
199
+ contract = load_contract(root)
200
+ if contract is None:
201
+ raise FileNotFoundError(f"no contract beside source environment at {root}")
202
+ world = open_provisioned_world(
203
+ root,
204
+ contract,
205
+ source_root=str(manifest.get("source_root") or ""),
206
+ )
207
+ world.store.load_from(root)
208
+ world.tools = manifest.get("tool_specs", [])
209
+ world.refusal_signature = str(manifest.get("refusal_signature") or "")
210
+ return world
211
+ handlers = {
212
+ name: (root / HANDLERS / f"{name}.py").read_text(encoding="utf-8")
213
+ for name in manifest.get("tools", [])
214
+ if (root / HANDLERS / f"{name}.py").exists()
215
+ }
216
+
217
+ named = str(manifest.get("store") or "sqlite")
218
+ if into is None:
219
+ world = GeneratedWorld(":memory:", kind=named)
220
+ # Only where there is one. A world whose records the agent's own code keeps has no
221
+ # database file, and demanding one would make it unrestorable.
222
+ if source.exists() and getattr(world.store, "connection", None) is not None:
223
+ origin = sqlite3.connect(source)
224
+ with world.connection:
225
+ origin.backup(world.connection)
226
+ origin.close()
227
+ else:
228
+ # A store that keeps its records somewhere other than a SQLite file loads them its own
229
+ # way. Without this the world comes back with an empty store, and everything a check
230
+ # reads is whatever happened to land in the agent's state instead.
231
+ world.store.load_from(root)
232
+ else:
233
+ target = Path(into)
234
+ target.parent.mkdir(parents=True, exist_ok=True)
235
+ shutil.copyfile(source, target)
236
+ world = GeneratedWorld(target, kind=named)
237
+
238
+ world.name = manifest.get("agent", "generated")
239
+ world.handlers = handlers
240
+ world.runtime_tools = set(manifest.get("runtime_tools") or [])
241
+ world.external_runtime = bool(manifest.get("external_runtime", False))
242
+ world.tools = manifest.get("tool_specs", [])
243
+ world.refusal_signature = str(manifest.get("refusal_signature") or "")
244
+ # A world whose handlers bind to the agent's own code cannot run them unless that code
245
+ # is importable again, and the frozen state is what those tools act on.
246
+ reached = str(manifest.get("source_root") or "")
247
+ if reached:
248
+ world.reach(reached)
249
+ frozen_state = root / STATE
250
+ if frozen_state.exists():
251
+ world.state_object = json.loads(frozen_state.read_text(encoding="utf-8"))
252
+ return world
253
+
254
+
255
+ def read_manifest(path: str | Path) -> dict[str, Any]:
256
+ return json.loads((Path(path) / MANIFEST).read_text(encoding="utf-8"))
257
+
258
+
259
+ def require_source_implementation(path: str | Path) -> None:
260
+ """Refuse worlds that do not prove their tools came from the submitted source."""
261
+ manifest = read_manifest(path)
262
+ if manifest.get("tool_implementation") != "source":
263
+ raise RuntimeError(
264
+ "This environment has no source-implementation provenance. It was created by an "
265
+ "older/synthetic harness path and may contain reimplemented handlers. Rebuild it "
266
+ "from the agent repository; it cannot be used for a test run."
267
+ )
268
+
269
+
270
+ def apply_overlay(world: GeneratedWorld, overlay: Mapping[str, Any] | None) -> int:
271
+ """Layer one scenario's own rows onto a restored world.
272
+
273
+ ``{"table": [{"column": value}, ...]}``. The only sanctioned way a scenario adds data, so the
274
+ base world stays the shared starting point rather than drifting per scenario.
275
+ """
276
+ written = 0
277
+ for table, rows in (overlay or {}).items():
278
+ for row in rows or []:
279
+ if not isinstance(row, Mapping) or not row:
280
+ continue
281
+ columns = ", ".join(row)
282
+ marks = ", ".join("?" for _ in row)
283
+ world.connection.execute(
284
+ f"INSERT INTO {table} ({columns}) VALUES ({marks})", list(row.values())
285
+ )
286
+ written += 1
287
+ world.connection.commit()
288
+ return written
@@ -0,0 +1,305 @@
1
+ """The stores the harness can stand up for an agent, and what every one of them owes a world.
2
+
3
+ A store is the thing underneath an agent's tools: whatever really holds the records its queries
4
+ run against. It is never asked to execute a tool. It is asked to exist, to hold data, to say what
5
+ it holds, to let a scenario change a little of it, and to go back to how it was.
6
+
7
+ Which engine gets stood up is read off the agent, never chosen for it. Postgres and ClickHouse
8
+ disagree about dialect, types and what a transaction even means, so testing one against the other
9
+ grades an agent on queries it never runs. An engine the harness cannot stand up is an answer, not
10
+ a reason to substitute something that merely resembles it.
11
+
12
+ What a store owes falls into four groups, and most stores care about three:
13
+
14
+ lifecycle start, stop, dsn stand it up and say where it is
15
+ contents apply, execute, query statements, in whatever this engine speaks
16
+ records collections, holds, records, add, amend, remove
17
+ going back freeze, restore between scenarios
18
+ save_to, load_from to and from disk, for the base world
19
+
20
+ The records group is what keeps a scenario from ever naming a store. `world.put`, `world.change`
21
+ and `world.drop` land here, so the same scenario runs against SQLite, against Postgres in a
22
+ container, or against a structure the agent's own code holds, without a line of it changing.
23
+ `state()` comes free from that group, and `Records` provides it.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ from dataclasses import dataclass, field
29
+ from pathlib import Path
30
+ from typing import Any, Callable, Mapping, Protocol, Sequence, runtime_checkable
31
+
32
+
33
+ class StoreError(RuntimeError):
34
+ """The store could not be stood up, or could not answer.
35
+
36
+ Distinct from anything the agent did. A store that will not start is our problem and should
37
+ stop the run loudly, because every result after it would be measured against something that
38
+ is not there.
39
+ """
40
+
41
+
42
+ @dataclass
43
+ class Snapshot:
44
+ """Everything a store held at one moment, and what it takes to put it back.
45
+
46
+ ``rows`` is kept in the shape ``state()`` reports, so a check written against a world's state
47
+ reads a snapshot without knowing which engine produced it.
48
+
49
+ ``counters`` is whatever an engine hands out that is not itself a record: a Postgres sequence,
50
+ a MySQL auto-increment, anything that keeps counting after the rows are gone. Restoring rows
51
+ without restoring these gives the next scenario ids that continue from the last one, and a
52
+ check naming a specific id then fails for a reason that has nothing to do with the agent.
53
+ Engines that hand out nothing of the sort leave it empty, which is not a gap.
54
+ """
55
+
56
+ rows: dict[str, list[dict[str, Any]]] = field(default_factory=dict)
57
+ counters: dict[str, int] = field(default_factory=dict)
58
+
59
+ def counts(self) -> dict[str, int]:
60
+ return {name: len(rows) for name, rows in self.rows.items()}
61
+
62
+
63
+ @runtime_checkable
64
+ class Store(Protocol):
65
+ """A running store the world's records live in."""
66
+
67
+ # What this engine is. ``key`` is the same thing under the name a saved manifest already
68
+ # uses, so a world written before this split still reopens.
69
+ engine: str
70
+ key: str
71
+
72
+ def start(self) -> None: ...
73
+ def stop(self) -> None: ...
74
+ def dsn(self) -> str: ...
75
+
76
+ # Statements the harness wrote, in whatever this store speaks.
77
+ def apply(self, script: str) -> None: ...
78
+ def execute(self, statement: str, params: Sequence[Any] = ()) -> int: ...
79
+ def query(
80
+ self, statement: str, params: Sequence[Any] = ()
81
+ ) -> list[dict[str, Any]]: ...
82
+
83
+ # What a scenario and its checks need without writing a statement themselves.
84
+ def collections(self) -> list[str]: ...
85
+ def holds(self, collection: str) -> bool: ...
86
+ def records(self, collection: str) -> list[dict[str, Any]]: ...
87
+ def state(self) -> dict[str, list[dict[str, Any]]]: ...
88
+ def table(self, name: str) -> list[dict[str, Any]]: ...
89
+ def add(self, collection: str, record: Mapping[str, Any]) -> int | dict[str, Any]: ...
90
+ def amend(
91
+ self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
92
+ ) -> int: ...
93
+ def remove(self, collection: str, key: str = "", *, by: str = "") -> int: ...
94
+
95
+ # Between scenarios, in memory.
96
+ def freeze(self) -> Snapshot: ...
97
+ def restore(self, snapshot: Snapshot) -> None: ...
98
+
99
+ # To and from disk, so the base world outlives the process that built it.
100
+ def save_to(self, path: str | Path) -> None: ...
101
+ def load_from(self, path: str | Path) -> None: ...
102
+
103
+ def close(self) -> None: ...
104
+
105
+
106
+ class Records:
107
+ """``state`` from the record methods, for any store that has them.
108
+
109
+ Kept in one place because the two would otherwise drift, and they are the pair the gates
110
+ compare: the bite gate empties a store and reads ``state``, while a scenario changes it
111
+ through ``add`` and ``amend``. If those disagree about what a collection contains, a check
112
+ passes against something no scenario can produce.
113
+ """
114
+
115
+ def state(self) -> dict[str, list[dict[str, Any]]]:
116
+ return {name: self.records(name) for name in self.collections()} # type: ignore[attr-defined]
117
+
118
+
119
+ class Held:
120
+ """The record methods, and disk, for a store that already answers ``state``.
121
+
122
+ The mirror of ``Records``, for stores built the other way round: a container store reads
123
+ everything it holds in one go, and the per-collection questions follow from that. Saving to
124
+ disk is the snapshot as JSON, which works for any engine because a snapshot is already the
125
+ engine-independent shape.
126
+
127
+ ``add``, ``amend`` and ``remove`` are not derivable and are left to the engine. A store
128
+ without them refuses loudly rather than silently doing nothing, because the alternative is a
129
+ scenario whose setup appears to run and changes nothing, and a run then graded against a
130
+ world that was never set up.
131
+ """
132
+
133
+ engine: str = ""
134
+
135
+ @property
136
+ def key(self) -> str:
137
+ return self.engine
138
+
139
+ def collections(self) -> list[str]:
140
+ return sorted(self.state()) # type: ignore[attr-defined]
141
+
142
+ def holds(self, collection: str) -> bool:
143
+ return collection in self.state() # type: ignore[attr-defined]
144
+
145
+ def records(self, collection: str) -> list[dict[str, Any]]:
146
+ return self.state().get(collection, []) # type: ignore[attr-defined]
147
+
148
+ def execute(self, statement: str, params: Sequence[Any] = ()) -> int:
149
+ self.apply(statement) # type: ignore[attr-defined]
150
+ return 0
151
+
152
+ def query(self, statement: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
153
+ raise StoreError(
154
+ f"{self.engine} does not read back arbitrary statements. Read what it holds with "
155
+ "records() or state()."
156
+ )
157
+
158
+ def table(self, name: str) -> list[dict[str, Any]]:
159
+ raise StoreError(
160
+ f"{self.engine} does not read one table at a time. Read what it holds with "
161
+ "records() or state()."
162
+ )
163
+
164
+ def add(self, collection: str, record: Mapping[str, Any]) -> int | dict[str, Any]:
165
+ raise StoreError(_UNWRITABLE.format(engine=self.engine, verb="add to"))
166
+
167
+ def amend(
168
+ self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
169
+ ) -> int:
170
+ raise StoreError(_UNWRITABLE.format(engine=self.engine, verb="change"))
171
+
172
+ def remove(self, collection: str, key: str = "", *, by: str = "") -> int:
173
+ raise StoreError(_UNWRITABLE.format(engine=self.engine, verb="remove from"))
174
+
175
+ def clear(self) -> None:
176
+ """Empty it, by restoring a snapshot that holds nothing."""
177
+ self.restore(Snapshot()) # type: ignore[attr-defined]
178
+
179
+ def save_to(self, path: str | Path) -> None:
180
+ import json
181
+
182
+ root = Path(path)
183
+ root.mkdir(parents=True, exist_ok=True)
184
+ frozen = self.freeze() # type: ignore[attr-defined]
185
+ (root / SAVED).write_text(
186
+ json.dumps(
187
+ {
188
+ # The schema as the scripts that made it, because the rows alone cannot
189
+ # come back: a fresh engine has no tables to put them in.
190
+ "schema": list(getattr(self, "applied", [])),
191
+ "rows": frozen.rows,
192
+ "counters": frozen.counters,
193
+ },
194
+ indent=2,
195
+ default=str,
196
+ ),
197
+ encoding="utf-8",
198
+ )
199
+
200
+ def load_from(self, path: str | Path) -> None:
201
+ import json
202
+
203
+ held = Path(path) / SAVED
204
+ if not held.exists():
205
+ raise StoreError(f"no saved store at {held}")
206
+ kept = json.loads(held.read_text(encoding="utf-8"))
207
+ for script in kept.get("schema") or []:
208
+ self.apply(script) # type: ignore[attr-defined]
209
+ self.restore(
210
+ Snapshot(rows=kept.get("rows") or {}, counters=kept.get("counters") or {})
211
+ ) # type: ignore[attr-defined]
212
+
213
+ def close(self) -> None:
214
+ self.stop() # type: ignore[attr-defined]
215
+
216
+
217
+ # What a saved container store is written as. Not the engine's own dump format: a snapshot is
218
+ # already engine-independent, and a dump would tie the saved world to the version that wrote it.
219
+ SAVED = "store.json"
220
+
221
+ _UNWRITABLE = (
222
+ "{engine} has no way to {verb} a collection one record at a time, so a scenario cannot set "
223
+ "up on it. Give the store add, amend and remove in this engine's own language."
224
+ )
225
+
226
+ _REGISTRY: dict[str, Callable[..., Store]] = {}
227
+
228
+ # Names people and manifests actually write, pointing at the engine they mean. Kept explicit
229
+ # rather than normalised in code, because guessing which engine an unrecognised word meant is
230
+ # how an agent ends up graded against the wrong one.
231
+ _ALIASES = {
232
+ "": "in_process",
233
+ "none": "in_process",
234
+ "memory": "in_process",
235
+ "in-memory": "in_process",
236
+ "inprocess": "in_process",
237
+ }
238
+
239
+
240
+ def register_store(engine: str, factory: Callable[..., Store]) -> None:
241
+ """Teach the harness an engine. A class and this line.
242
+
243
+ The cost of this line is what decides whether "whatever the agent uses" is real or an
244
+ aspiration, which is why the shared work lives in ``ContainerStore`` and an engine
245
+ contributes only what genuinely differs.
246
+ """
247
+ _REGISTRY[engine] = factory
248
+
249
+
250
+ def supported() -> tuple[str, ...]:
251
+ return tuple(sorted(_REGISTRY))
252
+
253
+
254
+ def resolve(engine: str = "", **options: Any) -> Store:
255
+ """The store for an engine, or a refusal naming what there is.
256
+
257
+ Deliberately not a fallback. An agent on an engine nobody has taught the harness to run is a
258
+ gap worth reporting, and quietly handing it a different store would produce a green suite
259
+ about queries the agent never executes.
260
+ """
261
+ named = (engine or "").strip().lower()
262
+ named = _ALIASES.get(named, named)
263
+ if named not in _REGISTRY:
264
+ raise StoreError(
265
+ f"no store for engine {named!r}; the harness can stand up "
266
+ f"{', '.join(supported()) or 'nothing yet'}. Adding one is a class with the record "
267
+ "methods and a call to register_store, or write_store_ops for an engine in a container."
268
+ )
269
+ return _REGISTRY[named](**options)
270
+
271
+
272
+ # The name the rest of the harness has always called this by.
273
+ open_store = resolve
274
+
275
+
276
+ from .inprocess import InProcessStore # noqa: E402
277
+ from .sqlite import SqliteStore # noqa: E402
278
+
279
+ register_store(SqliteStore.engine, SqliteStore)
280
+ register_store(InProcessStore.engine, InProcessStore)
281
+
282
+ from .container import ContainerStore, docker, strays # noqa: E402
283
+ from .postgres import PostgresStore # noqa: E402
284
+
285
+ # Postgres is registered as the worked example, not as the supported list. An engine the harness
286
+ # has never seen is meant to be written at build time against ``ContainerStore`` and proved by
287
+ # the gates, rather than waiting for someone to ship a class for it.
288
+ register_store(PostgresStore.engine, PostgresStore)
289
+
290
+ __all__ = [
291
+ "ContainerStore",
292
+ "InProcessStore",
293
+ "PostgresStore",
294
+ "Records",
295
+ "Snapshot",
296
+ "SqliteStore",
297
+ "Store",
298
+ "StoreError",
299
+ "docker",
300
+ "open_store",
301
+ "register_store",
302
+ "resolve",
303
+ "strays",
304
+ "supported",
305
+ ]