agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,511 @@
1
+ """The tools that provision an environment, and the gate that decides it may be saved.
2
+
3
+ The difference from ``world/tools.py`` is the whole point of this path. There, the builder
4
+ writes a handler per tool and the world answers the agent's calls itself, so what gets tested
5
+ is a replica of the agent graded against a replica of its data. Here it writes none: it stands
6
+ up the engine the agent already uses, runs the agent's own migrations into it, and points the
7
+ agent at it. The agent's code, its client and its queries are untouched.
8
+
9
+ So there is deliberately no tool here that writes into the agent's repository. Not a guarded
10
+ one, not one that records what it changed. When a check fails there are two ways to make it
11
+ green -- fix the environment, or edit the agent until it stops failing -- and the second
12
+ produces a green suite about code nobody ships. That is not prevented by asking nicely in a
13
+ skill file. It is prevented by there being no verb for it.
14
+
15
+ The same three habits as the older surface, for the same reasons: execute immediately, say what
16
+ happened briefly, and never answer with nothing.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ from pathlib import Path
23
+ from typing import Any
24
+
25
+ from ..backends import tool, tool_server
26
+
27
+ from ..contract import AgentContract
28
+ from ..environment import (
29
+ SubGoal,
30
+ load_catalogue,
31
+ save_catalogue,
32
+ save_simulator_prompt,
33
+ validate_simulator_prompt,
34
+ validate_sub_goal,
35
+ )
36
+ from ..tools import schema
37
+ from .stores import Store, StoreError, resolve, supported
38
+ from .stores.prove import prove_checks_bite, prove_store
39
+
40
+ PROVISION_SERVER = "environment"
41
+
42
+ MANIFEST = "environment.json"
43
+
44
+ # What ``write_store_ops`` is actually given. Said here as well as in the skill because this is
45
+ # where the mistake surfaces: a store whose reset is wrong has a model reading *this* message.
46
+ from .stores.written import API as OPS_API # noqa: E402
47
+
48
+
49
+ def _ok(text: str) -> dict[str, Any]:
50
+ return {"content": [{"type": "text", "text": text}]}
51
+
52
+
53
+ def _err(text: str) -> dict[str, Any]:
54
+ return {"content": [{"type": "text", "text": text}], "is_error": True}
55
+
56
+
57
+ def _counts(state: dict[str, list[dict]]) -> str:
58
+ return (
59
+ ", ".join(f"{name}: {len(rows)}" for name, rows in sorted(state.items()))
60
+ or "nothing"
61
+ )
62
+
63
+
64
+ def provision_tools(
65
+ contract: AgentContract, destination: Path, source: str | Path | None = None
66
+ ) -> Any:
67
+ """A server exposing the provisioning surface for one agent.
68
+
69
+ ``source`` is the agent's repository. An in-memory store is reached by importing the
70
+ agent's own loader, so the path its modules live under has to be known here; for a store
71
+ behind a socket it is unused.
72
+ """
73
+ destination.mkdir(parents=True, exist_ok=True)
74
+ catalogue = load_catalogue(destination)
75
+
76
+ # Everything the stage builds up, held here until ``save_environment`` writes it out.
77
+ standing: dict[str, Any] = {
78
+ "store": None, # the running Store
79
+ "engine": "",
80
+ "version": "",
81
+ "seam": {}, # how the agent is pointed at it
82
+ "migrations": "",
83
+ "seed": "",
84
+ "mutation": "",
85
+ "proven": False,
86
+ }
87
+
88
+ def store() -> Store | None:
89
+ return standing["store"]
90
+
91
+ # -- what to stand up -------------------------------------------------------------
92
+
93
+ @tool(
94
+ "declare_engine",
95
+ "Stand up the engine this agent already uses, and record how the agent will be "
96
+ "pointed at it. Read the engine off the contract; never choose one for the agent. "
97
+ "The container starts immediately, so a bad image comes back now rather than at "
98
+ "save time.",
99
+ schema(
100
+ {
101
+ "engine": str,
102
+ "version": str,
103
+ "dsn_env": str,
104
+ "config_key": str,
105
+ "host": str,
106
+ "port": int,
107
+ "database": str,
108
+ "user": str,
109
+ "loader_module": str,
110
+ "loader_function": str,
111
+ },
112
+ ["engine"],
113
+ ),
114
+ )
115
+ async def declare_engine(args: dict[str, Any]) -> dict[str, Any]:
116
+ engine = str(args.get("engine") or "").strip()
117
+ if engine not in supported():
118
+ return _err(
119
+ f"no store for {engine!r} yet. The harness can stand up "
120
+ f"{', '.join(supported())}. Write one with write_store_ops first: it needs "
121
+ f"the image, the port it listens on, and how to read and reset it.\n{OPS_API}"
122
+ )
123
+ if standing["store"] is not None:
124
+ standing["store"].stop()
125
+
126
+ # What the agent already expects, matched rather than changed. A hardcoded host is
127
+ # redirected by a network alias; a hardcoded database name is simply the name we use.
128
+ options: dict[str, Any] = {}
129
+ if args.get("loader_module"):
130
+ # An in-memory store is reached by importing the agent's own loader, so it takes
131
+ # where to import from rather than anything to connect to.
132
+ options["module"] = str(args["loader_module"])
133
+ options["function"] = str(args.get("loader_function") or "load_data")
134
+ if source:
135
+ options["root"] = str(source)
136
+ else:
137
+ for key in ("database", "user"):
138
+ if args.get(key):
139
+ options[key] = str(args[key])
140
+ if args.get("version"):
141
+ options["version"] = str(args["version"])
142
+
143
+ try:
144
+ running = resolve(engine, **options)
145
+ running.start()
146
+ except StoreError as exc:
147
+ return _err(f"{engine} did not come up: {exc}")
148
+ except TypeError as exc:
149
+ return _err(
150
+ f"{engine} does not take those: {exc}. A store behind a socket takes version, "
151
+ "database and user; an in-memory one takes loader_module and loader_function."
152
+ )
153
+
154
+ standing.update(
155
+ store=running,
156
+ engine=engine,
157
+ version=str(args.get("version") or ""),
158
+ seam={
159
+ key: args[key]
160
+ for key in (
161
+ "dsn_env",
162
+ "config_key",
163
+ "host",
164
+ "port",
165
+ "database",
166
+ "user",
167
+ "loader_module",
168
+ "loader_function",
169
+ )
170
+ if args.get(key)
171
+ },
172
+ proven=False,
173
+ )
174
+ seam = standing["seam"]
175
+ if args.get("loader_module"):
176
+ return _ok(
177
+ f"{engine} is up, holding what {args['loader_module']}."
178
+ f"{options['function']} loaded: {_counts(running.state())}. Nothing connects "
179
+ "to it — the agent's tools read this structure directly. Seed it if the "
180
+ "contract carries data the loader does not."
181
+ )
182
+ pointed = (
183
+ f"${seam['dsn_env']}"
184
+ if seam.get("dsn_env")
185
+ else (
186
+ seam.get("config_key")
187
+ or "NOTHING RECORDED — say how the agent reaches it"
188
+ )
189
+ )
190
+ return _ok(
191
+ f"{engine} is up at {running.dsn()}. The agent will be pointed at it with "
192
+ f"{pointed}. Now run its own migrations."
193
+ )
194
+
195
+ @tool(
196
+ "write_store_ops",
197
+ "Teach the harness an engine it has never stood up: the image, the port, the boot "
198
+ "environment, and how to read and reset what it holds. Only needed for an engine not "
199
+ f"already known.\n\n{OPS_API}",
200
+ schema(
201
+ {
202
+ "engine": str,
203
+ "image": str,
204
+ "container_port": int,
205
+ "boot_env": dict,
206
+ "dsn_template": str,
207
+ "code": str,
208
+ },
209
+ ["engine", "image", "container_port", "code"],
210
+ ),
211
+ )
212
+ async def write_store_ops(args: dict[str, Any]) -> dict[str, Any]:
213
+ from .stores.written import register_written
214
+
215
+ try:
216
+ register_written(
217
+ engine=str(args["engine"]),
218
+ image=str(args["image"]),
219
+ container_port=int(args["container_port"]),
220
+ boot_env={
221
+ str(k): str(v) for k, v in (args.get("boot_env") or {}).items()
222
+ },
223
+ dsn_template=str(args.get("dsn_template") or ""),
224
+ code=str(args["code"]),
225
+ )
226
+ except (StoreError, SyntaxError, ValueError) as exc:
227
+ return _err(f"not registered: {exc}")
228
+ return _ok(
229
+ f"{args['engine']} registered. declare_engine can stand it up now; whether its "
230
+ "reset is right is decided by prove_environment, not by either of us."
231
+ )
232
+
233
+ # -- what goes in it --------------------------------------------------------------
234
+
235
+ @tool(
236
+ "run_migrations",
237
+ "Run the agent's OWN migrations into the store, so the schema is the agent's, spelled "
238
+ "the way the agent spells it. Never write a schema yourself: one we invented is a "
239
+ "guess, and every check written against it inherits the guess.",
240
+ schema({"script": str}, ["script"]),
241
+ )
242
+ async def run_migrations(args: dict[str, Any]) -> dict[str, Any]:
243
+ running = store()
244
+ if running is None:
245
+ return _err("nothing is standing yet. declare_engine first.")
246
+ try:
247
+ running.apply(str(args.get("script") or ""))
248
+ except Exception as exc: # noqa: BLE001 - the engine's complaint, reported as given
249
+ return _err(f"the migration was refused: {exc}")
250
+ standing["migrations"] = str(args.get("script") or "")
251
+ state = running.state()
252
+ if not state:
253
+ return _err(
254
+ "that ran but left no tables, so it was not the agent's schema. Find its "
255
+ "migrations, its models, or the DDL it ships."
256
+ )
257
+ return _ok(f"schema is up. {_counts(state)}")
258
+
259
+ @tool(
260
+ "seed",
261
+ "Put the agent's real starting data in, from the contract. Leave it in its natural "
262
+ "starting state: empty carts, no in-flight work. Scenarios add what they need.",
263
+ schema({"script": str}, ["script"]),
264
+ )
265
+ async def seed(args: dict[str, Any]) -> dict[str, Any]:
266
+ running = store()
267
+ if running is None:
268
+ return _err("nothing is standing yet. declare_engine first.")
269
+ try:
270
+ running.apply(str(args.get("script") or ""))
271
+ except Exception as exc: # noqa: BLE001
272
+ return _err(f"the seed was refused: {exc}")
273
+ standing["seed"] += "\n" + str(args.get("script") or "")
274
+ standing["proven"] = False
275
+ return _ok(f"seeded. {_counts(running.state())}")
276
+
277
+ @tool(
278
+ "inspect_environment",
279
+ "What is standing, and what it holds.",
280
+ schema({}, []),
281
+ )
282
+ async def inspect_environment(_args: dict[str, Any]) -> dict[str, Any]:
283
+ running = store()
284
+ if running is None:
285
+ # Said before anything is standing, because this is usually the first tool called
286
+ # and "nothing yet" answers nothing. A stage that does not know inprocess already
287
+ # exists goes looking for a server to put the agent's in-memory data in.
288
+ return _ok(
289
+ "nothing is standing yet. The harness can already stand up: "
290
+ f"{', '.join(supported())}.\n"
291
+ "'inprocess' is for an agent that holds its data in memory and whose tools read "
292
+ "that structure directly — it starts no server and the agent's own loader fills "
293
+ "it. Only write_store_ops for an engine genuinely absent from that list."
294
+ )
295
+ return _ok(
296
+ f"{standing['engine']} {standing['version']} at {running.dsn()}\n"
297
+ f"holds: {_counts(running.state())}\n"
298
+ f"agent reaches it by: {json.dumps(standing['seam']) or 'nothing recorded'}\n"
299
+ f"proven: {standing['proven']}"
300
+ )
301
+
302
+ # -- what it has to survive --------------------------------------------------------
303
+
304
+ @tool(
305
+ "add_sub_goal",
306
+ "Add a named thing this agent can be checked on, shared by every scenario that needs "
307
+ "it. Defined here, once, so results roll up: the same sub-goal failing in seven of "
308
+ "twelve scenarios is one sentence.\n\n"
309
+ "`check` is Python: define check(world, calls) returning a sentence when something is "
310
+ "wrong, or None when it held. `world` is the environment afterwards, and "
311
+ "world.state() gives every group and its rows; `calls` is every tool call made, each "
312
+ "with .name and .arguments — so a check can insist a call happened with the right "
313
+ "arguments, not merely that it happened.\n\n"
314
+ "Do not write a check that asks whether a tool refused. Many agents report a refusal "
315
+ "by returning an ordinary string, so nothing distinguishes it from success. Check what "
316
+ "the world holds afterwards, and the arguments the agent actually used.\n\n"
317
+ "`judged` is not a flag: it is the sentence saying what a model has to decide and why "
318
+ "code cannot. Leave it empty for anything code can settle, which is most things.",
319
+ schema(
320
+ {"name": str, "what": str, "check": str, "judged": str},
321
+ ["name", "what"],
322
+ ),
323
+ )
324
+ async def add_sub_goal(args: dict[str, Any]) -> dict[str, Any]:
325
+ one = SubGoal(
326
+ name=str(args["name"]),
327
+ what=str(args.get("what") or ""),
328
+ check=str(args.get("check") or ""),
329
+ judged=str(args.get("judged") or ""),
330
+ )
331
+ problems = validate_sub_goal(one)
332
+ if problems:
333
+ return _err(f"{one.name} not added:\n - " + "\n - ".join(problems))
334
+ catalogue.sub_goals = [
335
+ existing for existing in catalogue.sub_goals if existing.name != one.name
336
+ ]
337
+ catalogue.sub_goals.append(one)
338
+ settled = [g.name for g in catalogue.sub_goals if g.deterministic()]
339
+ standing["proven"] = False
340
+ return _ok(
341
+ f"{one.name} added. The catalogue has {len(catalogue.sub_goals)}, "
342
+ f"{len(settled)} settled by code: {', '.join(sorted(settled)) or 'none'}"
343
+ )
344
+
345
+ @tool(
346
+ "write_simulator_prompt",
347
+ "The person on the other side of this conversation, with a {{ instruction }} slot each "
348
+ "scenario fills. Only for a conversational agent.",
349
+ schema({"prompt": str}, ["prompt"]),
350
+ )
351
+ async def write_simulator_prompt_tool(args: dict[str, Any]) -> dict[str, Any]:
352
+ prompt = str(args.get("prompt") or "")
353
+ problems = validate_simulator_prompt(prompt)
354
+ if problems:
355
+ return _err("not saved:\n - " + "\n - ".join(problems))
356
+ path = save_simulator_prompt(prompt, destination)
357
+ return _ok(f"Saved to {path}.")
358
+
359
+ @tool(
360
+ "prove_environment",
361
+ "Put the environment through what it has to survive before anything is measured "
362
+ "against it. `mutation` is any statement this engine accepts that changes something -- "
363
+ "one insert is plenty -- and it is checked too, because a mutation that moves nothing "
364
+ "would make a broken reset look perfect.",
365
+ schema({"mutation": str}, ["mutation"]),
366
+ )
367
+ async def prove_environment(args: dict[str, Any]) -> dict[str, Any]:
368
+ running = store()
369
+ if running is None:
370
+ return _err("nothing is standing yet. declare_engine first.")
371
+ mutation = str(args.get("mutation") or "")
372
+
373
+ report = prove_store(running, mutation)
374
+ lines = [report.summary()]
375
+ failed = [one.name for one in report.results if not one.passed]
376
+
377
+ if not failed and catalogue.sub_goals:
378
+ bites = prove_checks_bite(running, _checks(catalogue))
379
+ lines.append(bites.summary())
380
+ failed += [one.name for one in bites.results if not one.passed]
381
+
382
+ standing["proven"] = not failed
383
+ standing["mutation"] = mutation
384
+ if failed:
385
+ return _err("\n".join(lines))
386
+ return _ok(
387
+ "\n".join(lines) + "\nThe environment holds. save_environment will keep it."
388
+ )
389
+
390
+ @tool(
391
+ "save_environment",
392
+ "Freeze the environment and write it out. Refused unless it passes its own gate, "
393
+ "which is re-run here rather than taken on trust.",
394
+ schema({"notes": str}, []),
395
+ )
396
+ async def save_environment(args: dict[str, Any]) -> dict[str, Any]:
397
+ running = store()
398
+ if running is None:
399
+ return _err("nothing is standing yet. declare_engine first.")
400
+ # An in-memory store has no migration step: the agent's loader is what creates the
401
+ # structure and fills it, and it already ran. Insisting on one here would be asking for
402
+ # a schema this agent does not have.
403
+ by_loader = bool(standing["seam"].get("loader_module"))
404
+ if not standing["migrations"] and not by_loader:
405
+ return _err(
406
+ "Not saved. The agent's own migrations were never run, so this schema is not "
407
+ "the agent's."
408
+ )
409
+ if not standing["seam"]:
410
+ return _err(
411
+ "Not saved. Nothing records how the agent reaches this store, so it cannot be "
412
+ "pointed at it. If the agent genuinely has no configuration seam, that is a "
413
+ "finding to report rather than something to work around."
414
+ )
415
+ if not catalogue.sub_goals:
416
+ return _err(
417
+ "Not saved. No sub-goals yet. They are defined here, once, and every scenario "
418
+ "names the ones it needs — that is what makes results add up across the suite."
419
+ )
420
+ if not [one for one in catalogue.sub_goals if one.deterministic()]:
421
+ return _err(
422
+ "Not saved. Every sub-goal is judged by a model. Most of what this agent does "
423
+ "leaves a trace in the store, and that should be settled by code."
424
+ )
425
+
426
+ # Re-run rather than trusting the flag: the builder does not get to declare its own
427
+ # environment sound, for the same reason it does not get to declare a scenario passed.
428
+ report = prove_store(running, standing["mutation"])
429
+ failed = [one.name for one in report.results if not one.passed]
430
+ if failed:
431
+ return _err(
432
+ f"Not saved, the environment does not hold up.\n{report.summary()}"
433
+ )
434
+
435
+ baseline = running.freeze()
436
+ manifest = {
437
+ "agent": contract.agent,
438
+ "engine": standing["engine"],
439
+ "version": standing["version"],
440
+ "seam": standing["seam"],
441
+ "migrations": standing["migrations"],
442
+ "seed": standing["seed"],
443
+ "mutation": standing["mutation"],
444
+ "notes": str(args.get("notes") or ""),
445
+ # The recipe, not the running container: run time stands the engine up again and
446
+ # replays these, which is what makes the environment reproducible rather than a
447
+ # thing that happened once on somebody's laptop.
448
+ "baseline": {"rows": baseline.rows, "counters": baseline.counters},
449
+ }
450
+ path = destination / MANIFEST
451
+ path.write_text(json.dumps(manifest, indent=2, default=str), encoding="utf-8")
452
+ save_catalogue(catalogue, destination)
453
+ return _ok(
454
+ f"Saved to {destination}. {standing['engine']} holding {_counts(baseline.rows)}, "
455
+ f"{len(catalogue.sub_goals)} sub-goals."
456
+ )
457
+
458
+ server = tool_server(
459
+ name=PROVISION_SERVER,
460
+ version="0.1.0",
461
+ tools=[
462
+ declare_engine,
463
+ write_store_ops,
464
+ run_migrations,
465
+ seed,
466
+ inspect_environment,
467
+ add_sub_goal,
468
+ write_simulator_prompt_tool,
469
+ prove_environment,
470
+ save_environment,
471
+ ],
472
+ )
473
+ return server, standing
474
+
475
+
476
+ def _checks(catalogue: Any) -> dict[str, Any]:
477
+ """The catalogue's code checks, as things that can be run against a store.
478
+
479
+ A check is written ``check(world, calls)``, and a store answers ``state()`` the same way a
480
+ world does, so the same code runs against either. At build time nothing has been called
481
+ yet, which is exactly the condition the bite gate wants: a check that still holds with no
482
+ calls and no rows is not checking anything.
483
+ """
484
+ out: dict[str, Any] = {}
485
+ for one in catalogue.sub_goals:
486
+ if not one.deterministic():
487
+ continue
488
+ out[one.name] = _compiled(one)
489
+ return out
490
+
491
+
492
+ def _compiled(one: Any) -> Any:
493
+ def run(store: Store) -> str | None:
494
+ namespace: dict[str, Any] = {}
495
+ exec(compile(one.check, f"<check:{one.name}>", "exec"), namespace) # nosec B102
496
+ return namespace["check"](store, [])
497
+
498
+ return run
499
+
500
+
501
+ TOOL_NAMES = (
502
+ "declare_engine",
503
+ "write_store_ops",
504
+ "run_migrations",
505
+ "seed",
506
+ "inspect_environment",
507
+ "add_sub_goal",
508
+ "write_simulator_prompt",
509
+ "prove_environment",
510
+ "save_environment",
511
+ )