agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1516 @@
1
+ """The tools that build a world, and the gate that decides it may be saved.
2
+
3
+ A deliberately narrow surface. The builder gets no generic file write, because a guardrail needs
4
+ something to sit behind: every action goes through a tool that can execute it, check it, and say
5
+ what went wrong. Interface design work on coding agents is consistent that this beats handing
6
+ over raw access and hoping.
7
+
8
+ Three habits throughout, for the same reason:
9
+
10
+ - **execute immediately.** A handler is run the moment it is defined, so a mistake comes back on
11
+ the next turn rather than at save time.
12
+ - **say what happened, briefly.** Counts and names, never dumps. More context measurably makes
13
+ agents worse at this.
14
+ - **never answer with nothing.** "0 rows inserted" is a result; an empty string is a puzzle.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import asyncio
20
+ import json
21
+ from pathlib import Path
22
+ from typing import Any
23
+
24
+ from ..backends import tool, tool_server
25
+
26
+ from ..amend import add_rule, drop_rule, fix_tool, set_modality, widen
27
+ from ..catalogue import (
28
+ SubGoal,
29
+ catalogue_problems,
30
+ load_catalogue,
31
+ save_catalogue,
32
+ compares_to_a_value,
33
+ validate_sub_goal,
34
+ weak_check_advisory,
35
+ )
36
+ from ..checks import run_check, run_world_check
37
+ from ..contract import AgentContract, is_data_free_conversation
38
+ from ..simulator import (
39
+ load_simulator_prompt,
40
+ save_simulator_prompt,
41
+ validate_simulator_prompt,
42
+ )
43
+ from ..tools import brief as _brief
44
+ from ..tools import schema
45
+ from .kinds import for_contract
46
+ from .mutate import UNDAMAGED, blind, unnoticed, unnoticed_in_place
47
+ from .probe import dirty_state, probe
48
+ from .runtime import GeneratedWorld
49
+ from .snapshot import MANIFEST, read_manifest, restore, save
50
+ from .stores.written import API as OPS_API
51
+
52
+ WORLD_SERVER = "world"
53
+
54
+ # What a handler is actually given. Said again here, and not only in the skill, because this is
55
+ # where the mistake surfaces: a handler that crashed has a model reading *this* message, and an
56
+ # error naming the failure without naming the API produces the same wrong guess again. Three
57
+ # identical attempts at one handler is what that costs.
58
+ DB_API = (
59
+ "Inside a handler, `db` reads the world two ways and has no cursors.\n\n"
60
+ "Works on every world, database or not:\n"
61
+ ' db.records("orders") -> every record in a collection, as dicts\n'
62
+ ' db.find("orders", status="new") -> the ones whose fields all match\n'
63
+ " db.collections() -> the collection names\n"
64
+ ' db.add("orders", {"id": "o1"}) -> put one record in\n\n'
65
+ "Only where this world has a query language, which not every agent does:\n"
66
+ ' db.query("SELECT * FROM t WHERE id = ?", [x]) -> list of dicts, [] if none\n'
67
+ ' db.one("SELECT * FROM t WHERE id = ?", [x]) -> one dict, or None\n'
68
+ ' db.execute("INSERT INTO t (a) VALUES (?)", [x]) -> number of rows changed\n\n'
69
+ "If this world has no connection, those three raise and the first four are what to use. "
70
+ "Records are dicts read by field name. db.execute returns a count, not a cursor, so calling "
71
+ ".fetchone(), .fetchall() or .lastrowid on any of these is a mistake. You also have `args`, "
72
+ "`ToolError` and `json`, and nothing else. Do not import anything."
73
+ )
74
+
75
+ # Below this, the world is not good enough to build tests on. Synthesis work that measures this
76
+ # converges on roughly this bar, and rejects a quarter to a third of what it generates.
77
+ ACCEPTABLE = 0.85
78
+ DRAFT = "build-draft.json"
79
+
80
+ # What a world check is, said where the mistake surfaces. A check that inspects nothing is
81
+ # the failure this whole mechanism exists to catch, so the answer says what "inspects
82
+ # something" means rather than only that the check was rejected.
83
+ WORLD_CHECK_HELP = (
84
+ "A world check is Python defining check(world), returning None when it holds or a "
85
+ "sentence saying what is wrong.\n\n"
86
+ "`world.state()` gives every collection this world has. **A collection is not always a list.**\n"
87
+ "A table gives a list of records. A collection the agent's own code keeps is often a mapping "
88
+ "keyed by identifier, and iterating that yields the keys, which are strings. Reading a field "
89
+ "off one of those is where a check written for the wrong shape fails.\n"
90
+ " held = world.state()['some_collection']\n"
91
+ " records = list(held.values()) if isinstance(held, dict) else held\n"
92
+ " wanted = [one for one in records if one.get('status') == 'pending']\n"
93
+ "The shapes this world actually has are listed below, so write for those rather than "
94
+ "guessing.\n\n"
95
+ "A check also has to inspect something that could be wrong. One that returns None without "
96
+ "reading the world passes forever, and it is rejected once the world is broken on purpose "
97
+ "and it stays green."
98
+ )
99
+
100
+
101
+ def _shapes(world: Any) -> str:
102
+ """What this world's collections are. Asked of the world, so every gate says the same thing."""
103
+ return world.shapes()
104
+
105
+
106
+ # What to read when a binding to the agent's own code will not run. The failure is nearly always
107
+ # the shape of the call rather than the code being unreachable, so the answer says what the
108
+ # shapes are instead of only reporting the exception.
109
+ BINDING_SCOPE = (
110
+ "Inside a binding, and inside a factory expression, these are the only names that exist:\n"
111
+ " args the arguments the agent passed, as a dict\n"
112
+ " db the world. db.state is the agent's own state, as adopt_state loaded it\n"
113
+ " ToolError to refuse\n"
114
+ " json\n"
115
+ "plus whatever the binding itself imports from the agent's source. There is no `userdata`, no "
116
+ "`state`, no framework context and no session: if the callable needs one of those, it has to "
117
+ "be constructed in the factory expression out of what is listed above, or it cannot be "
118
+ "reached from here at all."
119
+ )
120
+
121
+ ADOPT_HELP = (
122
+ "A binding is how one of the agent's own callables is reached. Four things can be wrong:\n"
123
+ " - the module path. It is imported from the agent's source root, so use the path its own "
124
+ "code would use, e.g. package.module.file, not a filesystem path\n"
125
+ " - the style. 'function' for a module-level def, 'staticmethod' for one on a class, "
126
+ "'method' when an instance has to exist first, in which case `factory` is the expression "
127
+ "that builds it\n"
128
+ " - first_arg. If the callable takes the agent's state as its first argument, name it here "
129
+ "and the world passes what adopt_state loaded. Leave it empty when the callable connects "
130
+ "for itself\n"
131
+ " - smoke_arguments. These are passed as keywords, so they have to match the callable's own "
132
+ "parameter names, and their values should be real: look at the world first and use an "
133
+ "identifier that exists, or the call only ever proves the tool can say no\n"
134
+ "If the tool genuinely cannot be reached without editing the agent, say so and ask. Do not "
135
+ "write a replacement for it."
136
+ )
137
+
138
+
139
+ def _size(value: Any) -> Any:
140
+ return len(value) if isinstance(value, (list, dict, tuple, str)) else value
141
+
142
+
143
+ def _base_data_problems(
144
+ state: dict[str, Any], source_state: dict[str, Any] | None = None
145
+ ) -> list[str]:
146
+ """Catch demo-shaped shared seed data before every scenario inherits it."""
147
+ problems: list[str] = []
148
+ weak_codes = {
149
+ "000000",
150
+ "111111",
151
+ "222222",
152
+ "333333",
153
+ "444444",
154
+ "555555",
155
+ "666666",
156
+ "777777",
157
+ "888888",
158
+ "999999",
159
+ "012345",
160
+ "123456",
161
+ "234567",
162
+ "345678",
163
+ "456789",
164
+ "987654",
165
+ "876543",
166
+ "765432",
167
+ "654321",
168
+ }
169
+ seen_codes: set[str] = set()
170
+ demo_card_endings: set[str] = set()
171
+ demo_identifiers: set[str] = set()
172
+
173
+ source_rows: dict[str, list[dict[str, Any]]] = {}
174
+ for collection, value in (source_state or {}).items():
175
+ rows = value if isinstance(value, list) else [value]
176
+ source_rows[str(collection)] = [row for row in rows if isinstance(row, dict)]
177
+
178
+ identity_keys = (
179
+ "id",
180
+ "code",
181
+ "booking_ref",
182
+ "booking_id",
183
+ "transaction_id",
184
+ "case_number",
185
+ "label",
186
+ "email",
187
+ "phone",
188
+ "name",
189
+ )
190
+
191
+ def submitted_record(
192
+ value: dict[str, Any], collection: str
193
+ ) -> dict[str, Any] | None:
194
+ # Environment setup may resolve source-relative values such as TODAY+3. Preserve
195
+ # provenance through that transformation by matching the stable record identity, then
196
+ # exempt only leaf values that are themselves unchanged from the submitted row.
197
+ for source_row in source_rows.get(collection, []):
198
+ if source_row and all(
199
+ value.get(key) == item for key, item in source_row.items()
200
+ ):
201
+ return source_row
202
+ if any(
203
+ key in source_row and key in value and source_row[key] == value[key]
204
+ for key in identity_keys
205
+ ):
206
+ return source_row
207
+ return None
208
+
209
+ def walk(
210
+ value: Any,
211
+ key: str = "",
212
+ collection: str = "",
213
+ source_record: dict[str, Any] | None = None,
214
+ ) -> None:
215
+ if isinstance(value, dict):
216
+ matched = submitted_record(value, collection) if collection else None
217
+ for child, item in value.items():
218
+ child_collection = collection or str(child)
219
+ walk(item, str(child), child_collection, matched or source_record)
220
+ elif isinstance(value, list):
221
+ for item in value:
222
+ walk(item, key, collection, source_record)
223
+ elif "otp" in key.lower() or key.lower() in {"verification_code", "code"}:
224
+ if source_record is not None and source_record.get(key) == value:
225
+ return
226
+ text = str(value)
227
+ if text in weak_codes:
228
+ seen_codes.add(text)
229
+ elif key.lower() in {"last4", "card_last4", "payment_last4"}:
230
+ if source_record is not None and source_record.get(key) == value:
231
+ return
232
+ text = str(value)
233
+ if text in {"0000", "1111", "1234", "4242", "4444"}:
234
+ demo_card_endings.add(text)
235
+ elif key.lower() in {"booking_ref", "booking_id", "transaction_id"}:
236
+ if source_record is not None and source_record.get(key) == value:
237
+ return
238
+ text = str(value).lower()
239
+ if text in {"ub12345678", "booking123", "booking_123", "test123"}:
240
+ demo_identifiers.add(str(value))
241
+
242
+ walk(state)
243
+ if seen_codes:
244
+ problems.append(
245
+ "predictable verification codes in shared data: "
246
+ + ", ".join(sorted(seen_codes))
247
+ )
248
+ if demo_card_endings:
249
+ problems.append(
250
+ "placeholder payment-card endings in shared data: "
251
+ + ", ".join(sorted(demo_card_endings))
252
+ )
253
+ if demo_identifiers:
254
+ problems.append(
255
+ "placeholder transaction identifiers in shared data: "
256
+ + ", ".join(sorted(demo_identifiers))
257
+ )
258
+ written = json.dumps(state, default=str).lower()
259
+ clichés = [
260
+ value
261
+ for value in ("test user", "john doe", "jane doe", "123 main street")
262
+ if value in written
263
+ ]
264
+ if clichés:
265
+ problems.append(
266
+ "placeholder identities/addresses in shared data: " + ", ".join(clichés)
267
+ )
268
+ return problems
269
+
270
+
271
+ # What a store file is called, when nobody has said where it is. Extensions rather than names,
272
+ # because the name is the agent's business and the extension is the convention.
273
+ STORE_SUFFIXES = (".db", ".sqlite", ".sqlite3", ".duckdb", ".dump", ".sql")
274
+
275
+
276
+ def _stores_here(source_root: str) -> str:
277
+ """Where the agent's code is, and which files under it look like a store.
278
+
279
+ Said rather than left to be guessed. A message that reports a path was wrong without saying
280
+ what the right ones are turns one call into a search, and the search is over a filesystem this
281
+ stage deliberately cannot list.
282
+ """
283
+ if not source_root:
284
+ return (
285
+ "This stage was not told where the agent's code lives, so a relative path has nothing "
286
+ "to resolve against. Give an absolute path, or say that the source root is missing."
287
+ )
288
+ root = Path(source_root)
289
+ seen: list[str] = []
290
+ for path in sorted(root.rglob("*")):
291
+ if len(seen) >= 12:
292
+ break
293
+ if path.is_file() and path.suffix.lower() in STORE_SUFFIXES:
294
+ size = path.stat().st_size
295
+ measure = f"{size // 1024} KB" if size else "empty"
296
+ seen.append(f" {path.relative_to(root)} ({measure})")
297
+ if not seen:
298
+ return (
299
+ f"The agent's code is at {root}, and nothing under it looks like a store. If it "
300
+ "builds or downloads one on first run, say so and ask rather than inventing data."
301
+ )
302
+ return (
303
+ "The agent's code is at "
304
+ + str(root)
305
+ + ", and these look like stores:\n"
306
+ + "\n".join(seen)
307
+ )
308
+
309
+
310
+ def _binding(
311
+ *, module: str, called: str, style: str, first_arg: str, factory: str
312
+ ) -> str:
313
+ """The handler that calls one of the agent's own callables.
314
+
315
+ Written as source rather than held as a closure, so it is saved with the world, readable by
316
+ whoever wants to know what actually ran, and restored exactly as every other handler is.
317
+
318
+ ``called`` may be a dotted path inside the module, which is how a staticmethod is reached:
319
+ ``CancelPendingOrder.invoke`` imports the class and calls the method on it. Only the first
320
+ segment is imported.
321
+ """
322
+ root = called.split(".")[0]
323
+ reach = f"from {module} import {root}" if module else ""
324
+ state = "db.state, " if first_arg else ""
325
+ # Their code may be async, which is true of every framework-decorated tool. The result is
326
+ # settled here rather than by the caller so that a handler stays synchronous, which is what
327
+ # every other part of the world already assumes.
328
+ if style == "method":
329
+ built = factory or f"{root}()"
330
+ attr = called.split(".", 1)[1] if "." in called else "__call__"
331
+ return (
332
+ f"{reach}\n"
333
+ "from fi.alk.harness.world.runtime import settled\n\n"
334
+ "def handle(args, db):\n"
335
+ f" instance = {built}\n"
336
+ f" return settled(instance.{attr}({state}**args))\n"
337
+ )
338
+ return (
339
+ f"{reach}\n"
340
+ "from fi.alk.harness.world.runtime import settled\n\n"
341
+ "def handle(args, db):\n"
342
+ f" return settled({called}({state}**args))\n"
343
+ )
344
+
345
+
346
+ def _ok(text: str) -> dict[str, Any]:
347
+ return {"content": [{"type": "text", "text": text}]}
348
+
349
+
350
+ def _err(text: str) -> dict[str, Any]:
351
+ return {"content": [{"type": "text", "text": text}], "is_error": True}
352
+
353
+
354
+ def world_tools(
355
+ contract: AgentContract,
356
+ destination: Path,
357
+ *,
358
+ source_root: str = "",
359
+ deferred_runtime: bool = False,
360
+ external_runtime: bool = False,
361
+ ) -> Any:
362
+ """A server exposing the world-building surface for one agent.
363
+
364
+ ``source_root`` is where the agent's own code lives. With it, a tool can be bound to the
365
+ agent's own implementation; without it the agent was given as a specification and its
366
+ tools have to be written here.
367
+ """
368
+ # An existing world is picked up rather than replaced. Amending one is the ordinary case
369
+ # once it has been built once, and starting empty every time would mean rebuilding a
370
+ # catalogue from scratch to add a single item to it.
371
+ existing = (destination / MANIFEST).exists()
372
+ # The store comes from what the contract found, not from a default. An agent whose tools keep
373
+ # their own state has no database, and opening one for it would be carrying something unused
374
+ # and describing the world as something it is not.
375
+ named = str(getattr(getattr(contract, "data_store", None), "kind", "") or "")
376
+ if existing:
377
+ world = restore(destination)
378
+ elif (destination / "environment.json").exists():
379
+ from ..provision import ProvisionedEnvironment, attached_postgres_store
380
+
381
+ provisioned = ProvisionedEnvironment.load(destination)
382
+ http_overrides = {
383
+ name: value
384
+ for name, value in (provisioned.overrides if provisioned else {}).items()
385
+ if value.startswith(("http://", "https://"))
386
+ }
387
+ if http_overrides:
388
+ from .provisioned import open_provisioned_world
389
+
390
+ world = open_provisioned_world(
391
+ destination,
392
+ contract,
393
+ source_root=source_root,
394
+ )
395
+ else:
396
+ # A harness-managed dependency-only environment (for example a Dockerfile plus
397
+ # Postgres) has no HTTP tool service to forward to. Its real database is still the
398
+ # world store; import/construct bindings execute the submitted tool code normally.
399
+ has_postgres = bool(
400
+ provisioned
401
+ and any(
402
+ "postgres" in service.lower() for service in provisioned.services
403
+ )
404
+ )
405
+ world = (
406
+ GeneratedWorld(store=attached_postgres_store(destination), kind=named)
407
+ if has_postgres
408
+ else GeneratedWorld(":memory:", kind=named)
409
+ )
410
+ if (
411
+ provisioned
412
+ and provisioned.runtime_services
413
+ and contract.modality == "voice"
414
+ ):
415
+ # These tools execute only through a real call to the submitted worker. They need no
416
+ # harness-authored handler and must not be smoke-called outside their captured RTC
417
+ # session state while the world is being constructed.
418
+ world.runtime_tools = set(contract.tool_names())
419
+ elif deferred_runtime or external_runtime:
420
+ # Hosted authoring runs on a control-plane worker without Docker. The repository's
421
+ # exact processes and declared datastore are compiled into Bundle V2 and started in
422
+ # Daytona; this lightweight store exists only to author baseline data, checks and
423
+ # scenarios before execution-time validation against those real processes.
424
+ world = GeneratedWorld(":memory:", kind="sqlite")
425
+ # This boundary applies to every submitted runtime, not only voice workers. Importing a
426
+ # chat agent's framework-decorated tools here would require installing untrusted target
427
+ # dependencies into the credentialed control process. It also tests them under a
428
+ # different interpreter than the one Bundle V2 will actually build. Mark them as
429
+ # runtime-owned instead: process_runtime installs the repository dependencies in its
430
+ # writable build tree, starts the target as svc-agent, and the real scenario calls prove
431
+ # the tool behavior and capture its tool trace there.
432
+ world.runtime_tools = set(contract.tool_names())
433
+ else:
434
+ world = GeneratedWorld(":memory:", kind=named)
435
+ world.name = contract.agent
436
+ world.external_runtime = external_runtime or bool(
437
+ getattr(world, "external_runtime", False)
438
+ )
439
+ world.refusal_signature = contract.refusal_signature
440
+ if source_root:
441
+ world.reach(source_root)
442
+ kind = for_contract(contract)
443
+ catalogue = load_catalogue(destination)
444
+ scores: list[float] = []
445
+ # The checks that decide whether this world is usable, written here rather than fixed in
446
+ # advance, because what makes a world usable is a judgement about this agent.
447
+ draft_path = destination / DRAFT
448
+ try:
449
+ draft = (
450
+ json.loads(draft_path.read_text(encoding="utf-8"))
451
+ if draft_path.exists()
452
+ else {}
453
+ )
454
+ except (OSError, json.JSONDecodeError):
455
+ draft = {}
456
+ world_checks: dict[str, str] = (
457
+ dict(read_manifest(destination).get("world_checks") or {})
458
+ if existing
459
+ else dict(draft.get("world_checks") or {})
460
+ )
461
+ # How many times each tool has been attempted, so a binding that cannot be made to work
462
+ # is told to stop rather than tried indefinitely.
463
+ tried: dict[str, int] = {}
464
+ sequences: list[dict[str, Any]] = (
465
+ list(read_manifest(destination).get("sequences") or [])
466
+ if existing
467
+ else list(draft.get("sequences") or [])
468
+ )
469
+
470
+ def _keep_draft() -> None:
471
+ destination.mkdir(parents=True, exist_ok=True)
472
+ draft_path.write_text(
473
+ json.dumps(
474
+ {"world_checks": world_checks, "sequences": sequences}, indent=2
475
+ ),
476
+ encoding="utf-8",
477
+ )
478
+
479
+ def _verified() -> tuple[list[str], list[str], dict[str, list[str]]]:
480
+ """How the world's own checks fare, and which of them cannot fail.
481
+
482
+ Run against the world as it stands, and then against worlds broken on purpose. A check
483
+ that stays green through every kind of damage is reported as blind: it is not verifying
484
+ anything, whatever it claims to inspect.
485
+ """
486
+ import tempfile
487
+
488
+ failing = [
489
+ name
490
+ for name, source in sorted(world_checks.items())
491
+ if not run_world_check(source, world, name=name).held
492
+ ]
493
+ if not world_checks:
494
+ return failing, [], {}
495
+ checks = sorted(world_checks.items())
496
+ if (destination / "environment.json").exists():
497
+ # The submitted Compose project owns this database. Its attached store can freeze
498
+ # and restore the exact live rows; exporting it as a generic temporary Postgres world
499
+ # discards that ownership and incorrectly demands generated schema.sql/driver setup.
500
+ survived = unnoticed_in_place(
501
+ world,
502
+ checks,
503
+ run=lambda source, broken: run_world_check(
504
+ source, broken, name="check"
505
+ ),
506
+ )
507
+ else:
508
+ # Snapshotted first so each mutation gets its own copy and none inherits another's
509
+ # damage. The world being built is never touched.
510
+ held = Path(tempfile.mkdtemp())
511
+ save(world, held, notes="mutation", sequences=sequences)
512
+ survived = unnoticed(
513
+ held,
514
+ checks,
515
+ run=lambda source, broken: run_world_check(
516
+ source, broken, name="check"
517
+ ),
518
+ restore=restore,
519
+ )
520
+ return failing, blind(survived), survived
521
+
522
+ @tool(
523
+ "create_schema",
524
+ "Run CREATE TABLE statements. Call once with the whole schema; call again to alter it.",
525
+ {"sql": str},
526
+ )
527
+ async def create_schema(args: dict[str, Any]) -> dict[str, Any]:
528
+ if external_runtime:
529
+ return _err(
530
+ "The connected provider owns its external state. A local schema would be a "
531
+ "shadow implementation and cannot affect the agent under test. Keep this "
532
+ "world empty."
533
+ )
534
+ try:
535
+ applies = getattr(world.store, "apply", None)
536
+ if applies is not None:
537
+ # A store-backed world speaks through its own engine — the
538
+ # sqlite-era connection only exists for worlds that are files.
539
+ # Off the loop: first touch boots the store's container.
540
+ await asyncio.to_thread(applies, args["sql"])
541
+ else:
542
+ world.connection.executescript(args["sql"])
543
+ world.connection.commit()
544
+ except Exception as failed:
545
+ return _err(f"schema rejected: {failed}")
546
+ tables = sorted(world.state())
547
+ return _ok(f"{len(tables)} tables: {', '.join(tables) or 'none'}")
548
+
549
+ @tool(
550
+ "seed",
551
+ "Put records into a collection. Rows is a list of objects whose keys are field names. "
552
+ "Works whether or not this world has a database: a collection that does not exist yet is "
553
+ "made, which is how an agent with no store of its own gets one.",
554
+ {"table": str, "rows": list},
555
+ )
556
+ async def seed(args: dict[str, Any]) -> dict[str, Any]:
557
+ if external_runtime:
558
+ return _err(
559
+ "The connected provider owns its external state. Local seed data cannot reach "
560
+ "that agent and would make the scenario proof false. Keep this world empty."
561
+ )
562
+ table, rows = str(args["table"]), args.get("rows") or []
563
+ written = 0
564
+ for row in rows:
565
+ if not isinstance(row, dict) or not row:
566
+ continue
567
+ try:
568
+ # Through the world rather than the connection, so this is the same call for a
569
+ # table, for a structure the agent's own code keeps, and for an agent that has no
570
+ # store at all and whose collections the harness is inventing.
571
+ world.put(table, row)
572
+ written += 1
573
+ except Exception as failed:
574
+ return _err(
575
+ f"{written} records written, then {table} rejected one: {failed}\n"
576
+ f"{_shapes(world)}"
577
+ )
578
+ total = len(world.state().get(table, []))
579
+ return _ok(f"{written} records put into {table}; {total} there now")
580
+
581
+ @tool(
582
+ "change_data",
583
+ "Change or remove rows already in the world: one UPDATE or DELETE statement. Seeding "
584
+ "only ever inserts, so without this a row put in wrong can never be taken out, and the "
585
+ "only way left to make a check pass is to change the contract, which is the wrong "
586
+ "repair. Use inspect_world to read; this is for changing.",
587
+ {"sql": str},
588
+ )
589
+ async def change_data(args: dict[str, Any]) -> dict[str, Any]:
590
+ if external_runtime:
591
+ return _err(
592
+ "There is no harness-controlled provider datastore to change in connect-only "
593
+ "mode. Keep this world empty."
594
+ )
595
+ statement = str(args.get("sql") or "").strip()
596
+ verb = statement.split(None, 1)[0].upper() if statement else ""
597
+ if verb not in ("UPDATE", "DELETE"):
598
+ return _err(
599
+ "this runs one UPDATE or DELETE. Use seed to add rows, create_schema to change "
600
+ "the shape of a table, and inspect_world to look."
601
+ )
602
+ try:
603
+ # The store is the portable mutation boundary. SQLite exposes a long-lived
604
+ # connection, while attached Postgres deliberately uses short autocommit
605
+ # connections; reaching through ``world.connection`` made cleanup impossible on
606
+ # exactly the provisioned worlds that most need it.
607
+ changed = world.store.execute(statement)
608
+ except Exception as failed:
609
+ return _err(f"rejected: {failed}")
610
+ counts = ", ".join(f"{n}: {len(r)}" for n, r in sorted(world.state().items()))
611
+ return _ok(f"{changed} rows changed. The world now holds {counts}")
612
+
613
+ @tool(
614
+ "adopt_state",
615
+ "Load the agent's own starting state by calling its own loader, so the world holds what "
616
+ "the agent really has rather than a copy of it. Give the module and the callable, for "
617
+ "example the function that reads its data files.",
618
+ schema({"module": str, "callable": str}, ["module", "callable"]),
619
+ )
620
+ async def adopt_state(args: dict[str, Any]) -> dict[str, Any]:
621
+ if external_runtime:
622
+ return _err(
623
+ "A source-free provider connection has no local state loader to adopt."
624
+ )
625
+ module = str(args["module"])
626
+ called = str(args["callable"])
627
+ world.reach(source_root)
628
+ try:
629
+ loaded = __import__(module, fromlist=[called])
630
+ factory = getattr(loaded, called)
631
+ world.state_object = factory()
632
+ except Exception as raised:
633
+ return _err(
634
+ f"could not load state with {module}.{called}: "
635
+ f"{type(raised).__name__}: {raised}\n{ADOPT_HELP}"
636
+ )
637
+ summary = (
638
+ {key: _size(value) for key, value in world.state_object.items()}
639
+ if isinstance(world.state_object, dict)
640
+ else type(world.state_object).__name__
641
+ )
642
+ return _ok(
643
+ f"state loaded from {module}.{called}: {json.dumps(summary, default=str)}"
644
+ )
645
+
646
+ @tool(
647
+ "adopt_store",
648
+ "Take the agent's own store as this world's starting data, so the world holds what the "
649
+ "agent really has. Give the path to it, relative to the agent's source or absolute. Use "
650
+ "this whenever the agent ships or builds a store of its own: seeding it by hand instead "
651
+ "produces a smaller, invented dataset that its real queries were never written against.",
652
+ schema({"path": str, "note": str}, ["path"]),
653
+ )
654
+ async def adopt_store(args: dict[str, Any]) -> dict[str, Any]:
655
+ if external_runtime:
656
+ return _err(
657
+ "A source-free provider connection has no local store to adopt."
658
+ )
659
+ given = str(args["path"]).strip()
660
+ found = Path(given)
661
+ if not found.is_absolute() and source_root:
662
+ found = Path(source_root) / given
663
+ if not found.exists() and source_root:
664
+ # An absolute path that is wrong is nearly always the agent's own repo-relative path
665
+ # read out of its source, so the same name under the real root is worth trying before
666
+ # reporting a miss.
667
+ under = Path(source_root) / Path(given).name
668
+ if under.exists():
669
+ found = under
670
+ if not found.exists():
671
+ return _err(f"nothing at {found}.\n{_stores_here(source_root)}")
672
+ if found.is_file() and found.stat().st_size == 0:
673
+ return _err(
674
+ f"{found} is empty, so there is nothing to adopt. If the agent builds or "
675
+ "downloads its store on first run, say so and ask rather than inventing data."
676
+ )
677
+ try:
678
+ # Off the event loop: taking a store can pull an image and boot a
679
+ # container, and the API must stay answerable meanwhile.
680
+ await asyncio.to_thread(world.store.take, found)
681
+ except AttributeError:
682
+ return _err(
683
+ f"a {world.store.engine} store cannot take another one yet. Seed it instead, or "
684
+ "say what it would need."
685
+ )
686
+ except Exception as raised:
687
+ return _err(f"could not take {found}: {type(raised).__name__}: {raised}")
688
+ state = world.state()
689
+ return _ok(
690
+ f"adopted {found.name}: "
691
+ + (
692
+ ", ".join(
693
+ f"{name}: {len(rows)}" for name, rows in sorted(state.items())
694
+ )
695
+ or "nothing"
696
+ )
697
+ )
698
+
699
+ @tool(
700
+ "adopt_tool",
701
+ "Bind one tool to the agent's own implementation, so its code runs rather than a "
702
+ "replacement. Give the module and the callable. `style` is how it is invoked: "
703
+ "'function' for a plain function, 'staticmethod' for one hanging off a class, "
704
+ "'method' when an instance has to be built first. `first_arg` names what the agent's "
705
+ "state is passed as, if its signature takes it. The binding runs immediately, so give "
706
+ "`smoke_arguments` that a real record in this world would satisfy: an identifier that is "
707
+ "actually there. A smoke call that refuses proves the binding can refuse, not that it "
708
+ "works.",
709
+ schema(
710
+ {
711
+ "tool_name": str,
712
+ "module": str,
713
+ "callable": str,
714
+ "style": str,
715
+ "first_arg": str,
716
+ "factory": str,
717
+ "binding": str,
718
+ "smoke_arguments": dict,
719
+ },
720
+ ["tool_name"],
721
+ ),
722
+ )
723
+ async def adopt_tool(args: dict[str, Any]) -> dict[str, Any]:
724
+ name = str(args["tool_name"])
725
+ if name not in contract.tool_names():
726
+ return _err(
727
+ f"{name!r} is not a tool this agent has. It has: "
728
+ f"{', '.join(sorted(contract.tool_names()))}"
729
+ )
730
+ if name in set(getattr(world, "runtime_tools", set())):
731
+ return _ok(
732
+ f"{name} is owned by the submitted runtime. Its dependencies and callable are "
733
+ "built and exercised inside the isolated execution process; no control-plane "
734
+ "binding is needed."
735
+ )
736
+ if not source_root:
737
+ return _err(
738
+ "there is no agent source on disk to bind to, so nothing can be adopted here. "
739
+ "Environment creation stops here: provide the repository containing the real "
740
+ "implementation. The harness will not write a substitute."
741
+ )
742
+ world.reach(source_root)
743
+ # A binding written here wins. The generated shapes cover a plain callable and a method
744
+ # on an object, which is most agents, but no set of shapes covers every framework, and
745
+ # guessing wrong is worse than letting whoever read the code write the two lines.
746
+ written = str(args.get("binding") or "").strip()
747
+ if written:
748
+ binding = written
749
+ elif not str(args.get("module") or ""):
750
+ return _err(
751
+ "give either a module and callable to bind to, or a binding of your own.\n\n"
752
+ + ADOPT_HELP
753
+ )
754
+ else:
755
+ binding = _binding(
756
+ module=str(args["module"]),
757
+ called=str(args["callable"]),
758
+ style=str(args.get("style") or "function"),
759
+ first_arg=str(args.get("first_arg") or ""),
760
+ factory=str(args.get("factory") or ""),
761
+ )
762
+ world.handlers[name] = binding
763
+ # Reverted after, because a smoke call against the agent's own code really does what the
764
+ # tool does: cancelling an order to prove the binding works would spend that order, and
765
+ # every scenario after it starts from this same world. Proving a tool works must not cost
766
+ # a record.
767
+ held = world.checkpoint()
768
+ call = world.call(name, args.get("smoke_arguments") or {})
769
+ world.revert(held)
770
+ if call.refused:
771
+ return _ok(
772
+ f"{name} adopted. Its own code answered with a refusal, which is it working: "
773
+ f"{call.error}"
774
+ )
775
+ if not call.ok:
776
+ del world.handlers[name]
777
+ tried[name] = tried.get(name, 0) + 1
778
+ said = f"{name} not adopted, the binding failed: {call.error}"
779
+ # A name that does not exist is the commonest way this fails, and the answer to it is
780
+ # the list of names that do, not a repeat of the general advice.
781
+ if "nameerror" in (call.error or "").lower():
782
+ said += f"\n\n{BINDING_SCOPE}"
783
+ if tried[name] >= 3:
784
+ said += (
785
+ f"\n\nThat is {tried[name]} attempts at this one. Some tools cannot be "
786
+ "reached without editing the agent: a framework may build them inside a "
787
+ "session that does not exist here. Stop and say so, naming this tool and what "
788
+ "it would need, and let the person decide. A tool nobody can run is a fact "
789
+ "worth reporting, and writing a stand-in instead is the one failure that "
790
+ "leaves no trace."
791
+ )
792
+ return _err(f"{said}\n\n{ADOPT_HELP}")
793
+ return _ok(
794
+ f"{name} adopted and ran, its own code. Returned {_brief(call.result)}"
795
+ )
796
+
797
+ @tool(
798
+ "run_tool",
799
+ "Call a defined tool and see what the world does. Use this to check a refusal works.",
800
+ schema({"tool_name": str, "arguments": dict}, ["tool_name"]),
801
+ )
802
+ async def run_tool(args: dict[str, Any]) -> dict[str, Any]:
803
+ call = world.call(str(args["tool_name"]), args.get("arguments") or {})
804
+ if call.refused:
805
+ return _ok(f"refused: {call.error}")
806
+ if not call.ok:
807
+ return _err(f"crashed: {call.error}")
808
+ return _ok(f"ok: {_brief(call.result)}")
809
+
810
+ @tool(
811
+ "declare_sequence",
812
+ "Declare a series of calls whose end state should hold, so consistency across calls is "
813
+ "checked. Each call is {tool, arguments}. expect_state keys are 'table.column' or "
814
+ "'table.count'. Declaring the same name again replaces it.\n\n"
815
+ "Every sequence runs on its own from the frozen world: the state is put back before each "
816
+ "one, so they never see each other's rows and expect_state is an absolute count, not a "
817
+ "running total. If a sequence fails, the fault is in that sequence, not in the ones "
818
+ "declared before it.",
819
+ schema({"name": str, "calls": list, "expect_state": dict}, ["name", "calls"]),
820
+ )
821
+ async def declare_sequence(args: dict[str, Any]) -> dict[str, Any]:
822
+ if external_runtime:
823
+ return _err(
824
+ "Provider tools execute only during the live conversation. Do not invent a "
825
+ "local reference sequence; use judged sub-goals and an empty solution."
826
+ )
827
+ name = str(args.get("name") or f"sequence-{len(sequences)}")
828
+ calls = args.get("calls") or []
829
+
830
+ # Checked here rather than at save time. A malformed sequence that only fails three
831
+ # tools later reads as a mystery, and there is nothing to learn from it in between.
832
+ problems: list[str] = []
833
+ if not calls:
834
+ problems.append("no calls: a sequence with no calls checks nothing")
835
+ for index, step in enumerate(calls):
836
+ if not isinstance(step, dict):
837
+ problems.append(
838
+ f"call {index} is not an object with a tool and arguments"
839
+ )
840
+ continue
841
+ called = str(step.get("tool") or "")
842
+ if not called:
843
+ problems.append(f"call {index} has no tool name")
844
+ elif called not in world.handlers:
845
+ problems.append(
846
+ f"call {index} names {called!r}, which has no handler yet. Defined: "
847
+ f"{', '.join(sorted(world.handlers)) or 'none'}"
848
+ )
849
+ if problems:
850
+ return _err(f"{name} not declared:\n - " + "\n - ".join(problems))
851
+
852
+ replaced = any(existing["name"] == name for existing in sequences)
853
+ sequences[:] = [existing for existing in sequences if existing["name"] != name]
854
+ sequences.append(
855
+ {
856
+ "name": name,
857
+ "calls": calls,
858
+ "expect_state": args.get("expect_state") or {},
859
+ }
860
+ )
861
+ _keep_draft()
862
+ verb = "replaced" if replaced else "declared"
863
+ return _ok(
864
+ f"{name} {verb}. {len(sequences)} sequences: {', '.join(s['name'] for s in sequences)}"
865
+ )
866
+
867
+ @tool(
868
+ "drop_sequence",
869
+ "Remove a declared sequence by name, or all of them with name '*'.",
870
+ {"name": str},
871
+ )
872
+ async def drop_sequence(args: dict[str, Any]) -> dict[str, Any]:
873
+ name = str(args.get("name") or "")
874
+ if name == "*":
875
+ sequences.clear()
876
+ _keep_draft()
877
+ return _ok("all sequences dropped")
878
+ before = len(sequences)
879
+ sequences[:] = [existing for existing in sequences if existing["name"] != name]
880
+ if len(sequences) == before:
881
+ return _err(
882
+ f"no sequence called {name!r}. Declared: "
883
+ f"{', '.join(s['name'] for s in sequences) or 'none'}"
884
+ )
885
+ _keep_draft()
886
+ return _ok(f"{name} dropped. {len(sequences)} left")
887
+
888
+ @tool(
889
+ "amend_contract",
890
+ "Let one of the agent's tools accept values it did not before. Use this when the world "
891
+ "holds something the agent has no way to name: an item added to the menu that item_id "
892
+ "does not list is dead data, and a scenario about it can only fail.\n\n"
893
+ "Only widen where the agent genuinely should accept the value. Say why in one line; it "
894
+ "is recorded on the contract, because a contract nobody can audit is worth nothing.",
895
+ {"tool_name": str, "argument": str, "values": list, "why": str},
896
+ )
897
+ async def amend_contract(args: dict[str, Any]) -> dict[str, Any]:
898
+ done, said = widen(
899
+ contract,
900
+ destination,
901
+ tool_name=str(args.get("tool_name") or ""),
902
+ argument=str(args.get("argument") or ""),
903
+ values=[str(value) for value in (args.get("values") or [])],
904
+ why=str(args.get("why") or ""),
905
+ )
906
+ return _ok(said) if done else _err(said)
907
+
908
+ @tool(
909
+ "add_rule",
910
+ "Give the agent a hard rule its source did not state, when the operator asks for one. "
911
+ "The agent under test is told every rule and the judge grades against them, so this "
912
+ "changes what is being tested. Say why in one line; it is recorded on the contract.",
913
+ {"rule": str, "why": str},
914
+ )
915
+ async def add_rule_tool(args: dict[str, Any]) -> dict[str, Any]:
916
+ done, said = add_rule(
917
+ contract,
918
+ destination,
919
+ rule=str(args.get("rule") or ""),
920
+ why=str(args.get("why") or ""),
921
+ )
922
+ return _ok(said) if done else _err(said)
923
+
924
+ @tool(
925
+ "set_modality",
926
+ "Correct how a person actually reaches this agent: voice, chat or browser. Modality "
927
+ "picks the world, the simulated person and the transport, so a wrong one does not weaken "
928
+ "a run, it runs a different test. Use it when the operator says where the agent is "
929
+ "deployed and the contract disagrees: an agent's code reads the same answering a chat "
930
+ "window or a phone call, so where it is deployed is something only they can settle.",
931
+ {"modality": str, "why": str},
932
+ )
933
+ async def set_modality_tool(args: dict[str, Any]) -> dict[str, Any]:
934
+ done, said = set_modality(
935
+ contract,
936
+ destination,
937
+ modality=str(args.get("modality") or ""),
938
+ why=str(args.get("why") or ""),
939
+ )
940
+ return _ok(said) if done else _err(said)
941
+
942
+ @tool(
943
+ "inspect_world",
944
+ "Look at what is in the world you are building. With no collection named, lists what "
945
+ "there is and how much is in each. With one, returns records from it. `matching` is plain "
946
+ "text and filters to records containing it, which is how you find a record in a large "
947
+ "collection without reading all of it.",
948
+ schema({"table": str, "limit": int, "matching": str}, []),
949
+ )
950
+ async def inspect_world(args: dict[str, Any]) -> dict[str, Any]:
951
+ state = world.state()
952
+ table = str(args.get("table") or "")
953
+ if not table:
954
+ return _ok(
955
+ "\n".join(
956
+ f"{name}: {_size(held)}" for name, held in sorted(state.items())
957
+ )
958
+ or "nothing in the world yet"
959
+ )
960
+ if table not in state:
961
+ return _err(
962
+ f"nothing called {table!r}; there is {', '.join(sorted(state)) or 'nothing'}"
963
+ )
964
+ held = state[table]
965
+ # A collection is a list of rows from a table, or a mapping the agent's own code keeps.
966
+ # Slicing the second one raises, so the shape is handled rather than assumed.
967
+ if isinstance(held, dict):
968
+ found = [
969
+ {"_key": key, **value}
970
+ if isinstance(value, dict)
971
+ else {"_key": key, "value": value}
972
+ for key, value in held.items()
973
+ ]
974
+ elif isinstance(held, list):
975
+ found = list(held)
976
+ else:
977
+ found = [held]
978
+ matching = str(args.get("matching") or "").strip().lower()
979
+ if matching:
980
+ narrowed = [
981
+ one for one in found if matching in json.dumps(one, default=str).lower()
982
+ ]
983
+ if not narrowed:
984
+ return _ok(
985
+ f"nothing in {table} contains {matching!r}, out of {len(found)} records."
986
+ )
987
+ found = narrowed
988
+ shown = found[: int(args.get("limit") or 5)]
989
+ return _ok(
990
+ f"{len(found)} records"
991
+ + (f" matching {matching!r}" if matching else "")
992
+ + f", showing {len(shown)}:\n"
993
+ + "\n".join(_brief(one) for one in shown)
994
+ )
995
+
996
+ @tool(
997
+ "drop_rule",
998
+ "Take away a hard rule the agent does not really have. A rule nobody has is worse than "
999
+ "a missing one: the agent is told to obey it and graded for not doing something it was "
1000
+ "never supposed to do. Say why.",
1001
+ {"rule": str, "why": str},
1002
+ )
1003
+ async def drop_rule_tool(args: dict[str, Any]) -> dict[str, Any]:
1004
+ done, said = drop_rule(
1005
+ contract,
1006
+ destination,
1007
+ rule=str(args.get("rule") or ""),
1008
+ why=str(args.get("why") or ""),
1009
+ )
1010
+ return _ok(said) if done else _err(said)
1011
+
1012
+ @tool(
1013
+ "fix_tool",
1014
+ "Correct a tool that was read wrong, or remove one the agent does not have. `args` "
1015
+ "replaces its argument names in order; `arg_types` and `description` update those. Set "
1016
+ "`remove` to take the tool away entirely. Everything downstream is built from these, so "
1017
+ "a wrong argument name produces a world that refuses everything. Say why.",
1018
+ schema(
1019
+ {
1020
+ "tool_name": str,
1021
+ "args": list,
1022
+ "arg_types": dict,
1023
+ "description": str,
1024
+ "remove": bool,
1025
+ "why": str,
1026
+ },
1027
+ ["tool_name", "why"],
1028
+ ),
1029
+ )
1030
+ async def fix_tool_tool(args: dict[str, Any]) -> dict[str, Any]:
1031
+ done, said = fix_tool(
1032
+ contract,
1033
+ destination,
1034
+ tool_name=str(args.get("tool_name") or ""),
1035
+ why=str(args.get("why") or ""),
1036
+ args=[str(a) for a in args["args"]] if args.get("args") else None,
1037
+ arg_types={
1038
+ str(k): str(v) for k, v in (args.get("arg_types") or {}).items()
1039
+ },
1040
+ description=str(args.get("description") or ""),
1041
+ remove=bool(args.get("remove")),
1042
+ )
1043
+ return _ok(said) if done else _err(said)
1044
+
1045
+ @tool(
1046
+ "write_simulator_prompt",
1047
+ "Write the prompt that drives the simulated user of this agent, for a conversational "
1048
+ "agent only. It is written once and every scenario fills its slots, so leave variables "
1049
+ "as {{ instruction }}, {{ persona }} and any others this agent needs.\n\n"
1050
+ "It has to cover how a person in this conversation actually behaves: that they are "
1051
+ "living the situation rather than describing it, that they speak one turn at a time, "
1052
+ "that they never break character or explain that they are testing anything, what they "
1053
+ "know and when they may say it, and when the conversation is over. Write it for this "
1054
+ "agent, not in general.",
1055
+ schema({"prompt": str}, ["prompt"]),
1056
+ )
1057
+ async def write_simulator_prompt(args: dict[str, Any]) -> dict[str, Any]:
1058
+ prompt = str(args.get("prompt") or "")
1059
+ problems = validate_simulator_prompt(
1060
+ prompt, require_persona=contract.conversational
1061
+ )
1062
+ if problems:
1063
+ return _err("Not saved:\n - " + "\n - ".join(problems))
1064
+ path = save_simulator_prompt(prompt, destination)
1065
+ from ..simulator import variables_in
1066
+
1067
+ return _ok(
1068
+ f"Saved to {path}. Scenarios must fill: "
1069
+ + ", ".join(sorted(variables_in(prompt)))
1070
+ )
1071
+
1072
+ @tool(
1073
+ "add_sub_goal",
1074
+ "Add a named thing this agent can be checked on, shared by every scenario that needs "
1075
+ "it. Defined here, once, so results roll up: the same sub-goal failing in seven of "
1076
+ "twelve scenarios is one sentence.\n\n"
1077
+ "`check` is Python: define check(world, calls) returning a sentence when something is "
1078
+ "wrong, or None when it held. `world` is the environment afterwards; `calls` is every "
1079
+ "tool call made, each with .name, .arguments, .ok and .refused — so a check can insist "
1080
+ "a call happened with the right arguments, not merely that it happened.\n\n"
1081
+ "Use `judged` only where nothing observable settles it, saying what a model has to "
1082
+ "decide and why code cannot.",
1083
+ schema(
1084
+ {"name": str, "what": str, "check": str, "judged": str}, ["name", "what"]
1085
+ ),
1086
+ )
1087
+ async def add_sub_goal(args: dict[str, Any]) -> dict[str, Any]:
1088
+ sub_goal = SubGoal(
1089
+ name=str(args.get("name") or ""),
1090
+ what=str(args.get("what") or ""),
1091
+ check=str(args.get("check") or ""),
1092
+ judged=str(args.get("judged") or ""),
1093
+ )
1094
+ problems = validate_sub_goal(sub_goal)
1095
+ if problems:
1096
+ return _err("Not added:\n - " + "\n - ".join(problems))
1097
+ # Run it here, the same way a handler is run the moment it is defined. A check that raises
1098
+ # is not a check, and accepting one now means every scenario that names it is refused later
1099
+ # for a reason that looks like the scenario's fault rather than this one's.
1100
+ if sub_goal.deterministic():
1101
+ outcome = run_check(
1102
+ sub_goal.check, world, list(world.calls), name=sub_goal.name
1103
+ )
1104
+ if outcome.broken:
1105
+ return _err(
1106
+ f"Not added. {sub_goal.name} is not a working check: {outcome.said}\n\n"
1107
+ f"{world.shapes()}\n\n"
1108
+ "It does not have to hold against the world as it stands, since a sub-goal is "
1109
+ "about what a run leaves behind. It does have to run without raising."
1110
+ )
1111
+ catalogue.sub_goals = [
1112
+ one for one in catalogue.sub_goals if one.name != sub_goal.name
1113
+ ]
1114
+ catalogue.sub_goals.append(sub_goal)
1115
+ save_catalogue(catalogue, destination)
1116
+ # Said on acceptance rather than as a refusal: a truthiness check is weak, not unusable,
1117
+ # and a gate the authoring loop cannot satisfy fails the run instead of improving it.
1118
+ advisory = weak_check_advisory(sub_goal)
1119
+ settled = sum(1 for one in catalogue.sub_goals if one.deterministic())
1120
+ return _ok(
1121
+ f"{sub_goal.name} added. The catalogue has {len(catalogue.sub_goals)}, "
1122
+ f"{settled} settled by code: "
1123
+ + ", ".join(sorted(catalogue.names()))
1124
+ + (f"\n\nWorth strengthening: {advisory}" if advisory else "")
1125
+ )
1126
+
1127
+ @tool(
1128
+ "write_env_file",
1129
+ "Write one file the environment is built from: a Dockerfile, a compose file, a schema, an "
1130
+ "entrypoint, whatever this agent needs. Paths are relative and stay inside the "
1131
+ "environment directory. Call it once per file, then build with run_env_command.",
1132
+ schema({"path": str, "contents": str}, ["path", "contents"]),
1133
+ )
1134
+ async def write_env_file(args: dict[str, Any]) -> dict[str, Any]:
1135
+ from .workspace import listing, write
1136
+
1137
+ try:
1138
+ written = write(destination, str(args["path"]), str(args["contents"]))
1139
+ except ValueError as refused:
1140
+ return _err(str(refused))
1141
+ lines = len(str(args["contents"]).splitlines())
1142
+ return _ok(
1143
+ f"wrote {written.name}, {lines} lines. The environment now has: "
1144
+ + ", ".join(listing(destination))
1145
+ )
1146
+
1147
+ @tool(
1148
+ "run_env_command",
1149
+ "Run one docker or docker compose command from the environment directory: build an image, "
1150
+ "bring a store up, run something inside a container. Only container commands run here, so "
1151
+ "anything the environment needs belongs in a file it builds from rather than in a "
1152
+ "command. Returns the exit code and the output.",
1153
+ schema({"command": str}, ["command"]),
1154
+ )
1155
+ async def run_env_command(args: dict[str, Any]) -> dict[str, Any]:
1156
+ from .workspace import run
1157
+
1158
+ # Off the event loop: a docker build takes minutes, and run synchronously
1159
+ # it deafens every API endpoint this server has until it finishes.
1160
+ code, output = await asyncio.to_thread(run, destination, str(args["command"]))
1161
+ shown = (
1162
+ output
1163
+ if len(output) <= 2500
1164
+ else output[:1200] + "\n...\n" + output[-1200:]
1165
+ )
1166
+ if code != 0:
1167
+ return _err(f"exit {code}\n{shown or '(no output)'}")
1168
+ return _ok(f"ok\n{shown or '(no output)'}")
1169
+
1170
+ @tool(
1171
+ "write_store_ops",
1172
+ "Teach the harness an engine it has never stood up: the image, the port it listens on, "
1173
+ "the environment it needs to boot, and how to read, reset and change what it holds. "
1174
+ "Only needed when the agent's engine is not one inspect_world already lists. Registering "
1175
+ "it says nothing about whether it works: that is decided by proving it, not by either "
1176
+ f"of us.\n\n{OPS_API}",
1177
+ schema(
1178
+ {
1179
+ "engine": str,
1180
+ "image": str,
1181
+ "container_port": int,
1182
+ "boot_env": dict,
1183
+ "dsn_template": str,
1184
+ "code": str,
1185
+ },
1186
+ ["engine", "image", "container_port", "code"],
1187
+ ),
1188
+ )
1189
+ async def write_store_ops(args: dict[str, Any]) -> dict[str, Any]:
1190
+ from .stores import StoreError, supported
1191
+ from .stores.written import register_written
1192
+
1193
+ try:
1194
+ register_written(
1195
+ engine=str(args["engine"]),
1196
+ image=str(args["image"]),
1197
+ container_port=int(args["container_port"]),
1198
+ boot_env={
1199
+ str(k): str(v) for k, v in (args.get("boot_env") or {}).items()
1200
+ },
1201
+ dsn_template=str(args.get("dsn_template") or ""),
1202
+ code=str(args["code"]),
1203
+ )
1204
+ except (StoreError, SyntaxError, ValueError) as exc:
1205
+ return _err(f"not registered: {exc}")
1206
+ return _ok(
1207
+ f"{args['engine']} registered, alongside {', '.join(supported())}. Whether its "
1208
+ "reset is right is decided when the environment is proved, not now."
1209
+ )
1210
+
1211
+ @tool(
1212
+ "add_world_check",
1213
+ "Add one check that decides whether this world is usable. Python defining "
1214
+ "check(world) which returns None when it holds, or a sentence saying what is wrong. "
1215
+ "`world.state()` gives every collection and its contents. It runs immediately, and it is "
1216
+ "later put through a world that has been broken on purpose: a check that stays green "
1217
+ "there is not checking anything.",
1218
+ schema({"name": str, "code": str, "what": str}, ["name", "code"]),
1219
+ )
1220
+ async def add_world_check(args: dict[str, Any]) -> dict[str, Any]:
1221
+ if (
1222
+ (is_data_free_conversation(contract) or external_runtime)
1223
+ and not world.state()
1224
+ and not world.handlers
1225
+ ):
1226
+ return _ok(
1227
+ "World-data checks are not applicable: this conversational agent has no "
1228
+ "business state or custom tools. Do not invent data or empty-world checks. "
1229
+ "Write conversational sub-goals and the simulator prompt, then save_world. "
1230
+ "The hosted runtime must still pass readiness and live conversation tests."
1231
+ )
1232
+ name = str(args["name"])
1233
+ source = str(args["code"])
1234
+ outcome = run_world_check(source, world, name=name)
1235
+ if outcome.broken:
1236
+ return _err(
1237
+ f"{name} is not a working check: {outcome.said}\n\n"
1238
+ f"{_shapes(world)}\n\n{WORLD_CHECK_HELP}"
1239
+ )
1240
+ world_checks[name] = source
1241
+ _keep_draft()
1242
+ held = "holds" if outcome.held else f"fails right now: {outcome.said}"
1243
+ return _ok(
1244
+ f"{name} added, {len(world_checks)} checks: {', '.join(sorted(world_checks))}.\n"
1245
+ f"Against the world as it stands it {held}."
1246
+ )
1247
+
1248
+ @tool(
1249
+ "check_world",
1250
+ "Exercise every tool with a valid call, a nonexistent id, and a missing argument, then "
1251
+ "run the declared sequences. Reports what is wrong without saving anything.\n\n"
1252
+ "Sequences are run independently from the frozen world, so a failure is never caused by "
1253
+ "another sequence. Fix the failures it names; declaring more sequences only adds more "
1254
+ "probes to pass.",
1255
+ {},
1256
+ )
1257
+ async def check_world(_args: dict[str, Any]) -> dict[str, Any]:
1258
+ report = probe(world, contract, sequences=sequences, kind=kind)
1259
+ scores.append(report.score)
1260
+ # Saying the score is going nowhere, rather than leaving it to be noticed. A stage that
1261
+ # has misdiagnosed something will otherwise keep applying the same non-fix, and every
1262
+ # round of that costs money and gets no closer.
1263
+ stuck = ""
1264
+ if len(scores) >= 3 and len({round(s, 2) for s in scores[-3:]}) == 1:
1265
+ stuck = (
1266
+ "\n\nThis is the third check with the same score. Whatever you are changing is "
1267
+ "not what is failing. Read the failures above literally and fix one of them, or "
1268
+ "say what you are stuck on."
1269
+ )
1270
+ failing, cannot_fail, survived = _verified()
1271
+ own = ""
1272
+ if world_checks:
1273
+ own = f"\n{len(world_checks) - len(failing)}/{len(world_checks)} of your own world checks hold"
1274
+ if failing:
1275
+ own += "\n failing: " + ", ".join(failing)
1276
+ if cannot_fail:
1277
+ own += (
1278
+ "\n these stayed green even with the world emptied and every tool "
1279
+ "silenced, so they are not checking anything: "
1280
+ + ", ".join(cannot_fail)
1281
+ )
1282
+ # Said out loud, because the alternative is a person rewriting checks that were right.
1283
+ for note in (survived or {}).get(UNDAMAGED, []):
1284
+ own += (
1285
+ f"\n the emptied test could not be run: {note}. Nothing is concluded from "
1286
+ "it, so this is ours to fix rather than yours."
1287
+ )
1288
+ else:
1289
+ own = (
1290
+ "\nWorld-data checks are not applicable to this empty conversational world."
1291
+ if is_data_free_conversation(contract)
1292
+ and not world.state()
1293
+ and not world.handlers
1294
+ else "\nNo world checks of your own yet. Add them with add_world_check."
1295
+ )
1296
+ return _ok(f"{report.summary()}\nscore {report.score:.2f}{own}{stuck}")
1297
+
1298
+ @tool(
1299
+ "save_world",
1300
+ "Freeze the world and write it out. Refused unless it passes its own checks.",
1301
+ schema({"notes": str}, []),
1302
+ )
1303
+ async def save_world(args: dict[str, Any]) -> dict[str, Any]:
1304
+ data_free = (
1305
+ (is_data_free_conversation(contract) or external_runtime)
1306
+ and not world.state()
1307
+ and not world.handlers
1308
+ )
1309
+ report = probe(world, contract, sequences=sequences, kind=kind)
1310
+ if report.score < ACCEPTABLE and not data_free:
1311
+ return _err(
1312
+ f"Not saved, the world does not hold up yet.\n{report.summary()}\n"
1313
+ f"score {report.score:.2f}, needs {ACCEPTABLE:.2f}"
1314
+ )
1315
+ runtime_tools = set(getattr(world, "runtime_tools", set()))
1316
+ runtime_only = bool(contract.tools) and set(contract.tool_names()).issubset(
1317
+ runtime_tools
1318
+ )
1319
+ # Sequences prove that a stateful tool surface remains coherent across multiple calls.
1320
+ # A conversational agent with no executable tools has no legal sequence to declare: an
1321
+ # empty sequence proves nothing, and every named call is necessarily fabricated. Such
1322
+ # agents can still have baseline data (for example provider dynamic variables), so
1323
+ # ``data_free`` alone is not enough to identify them. Keep requiring world checks for
1324
+ # that data, but do not deadlock authoring on a proof that cannot exist.
1325
+ requires_sequence = bool(contract.tools) and not runtime_only
1326
+ if not sequences and requires_sequence and not data_free:
1327
+ return _err(
1328
+ "Not saved. Declare at least one sequence first: a world whose calls each work "
1329
+ "alone can still forget what the previous one did."
1330
+ )
1331
+ if data_problems := _base_data_problems(
1332
+ world.state(), source_state=contract.base_environment
1333
+ ):
1334
+ return _err(
1335
+ "Not saved. The shared seed would make every scenario look like demo data:\n - "
1336
+ + "\n - ".join(data_problems)
1337
+ + "\nPreserve useful source records, but replace predictable credentials and add "
1338
+ "realistic varied baseline records through seed."
1339
+ )
1340
+ # The world has to prove itself, and the proof has to be capable of failing. Both halves
1341
+ # matter: checks nobody wrote verify nothing, and checks that pass a world with no data
1342
+ # and no working tools verify nothing either.
1343
+ if not world_checks and not data_free:
1344
+ return _err(
1345
+ "Not saved. This world has no checks of its own yet. Add them with "
1346
+ "add_world_check: what has to be true for this world to be worth testing "
1347
+ "against, as code.\n\n" + WORLD_CHECK_HELP
1348
+ )
1349
+ # Per-sub-goal validation cannot see the shape of the set, and the shape is what decides
1350
+ # whether the suite grades anything: a catalogue that is mostly judged reports opinions.
1351
+ if shape_problems := catalogue_problems(
1352
+ catalogue.sub_goals,
1353
+ world_is_observable=bool(world.state()) or bool(world.handlers),
1354
+ ):
1355
+ return _err("Not saved.\n - " + "\n - ".join(shape_problems))
1356
+ failing, cannot_fail, _survived = _verified()
1357
+ if failing:
1358
+ return _err(
1359
+ "Not saved. These of your own world checks do not hold: "
1360
+ + ", ".join(failing)
1361
+ + ".\nFix the world, or the check if the check is what is wrong."
1362
+ )
1363
+ if cannot_fail:
1364
+ return _err(
1365
+ "Not saved. These checks stayed green with the world emptied and every tool "
1366
+ "silenced, so they are not verifying anything: "
1367
+ + ", ".join(cannot_fail)
1368
+ + ".\nA check has to inspect something that could actually be wrong. Make each "
1369
+ "of them read the part of the world it claims to be about, and fail when it is "
1370
+ "missing."
1371
+ )
1372
+ # The environment is not only the world. Every scenario is a delta on what is built
1373
+ # here, so a catalogue nobody wrote means every scenario invents its own wording and
1374
+ # nothing rolls up across the suite.
1375
+ if not catalogue.sub_goals:
1376
+ return _err(
1377
+ "Not saved. No sub-goals yet. They are defined here, once, and every scenario "
1378
+ "names the ones it needs — that is what makes results add up across the suite. "
1379
+ "Add them with add_sub_goal."
1380
+ )
1381
+ settled = [one for one in catalogue.sub_goals if one.deterministic()]
1382
+ if not settled and not data_free:
1383
+ return _err(
1384
+ "Not saved. Every sub-goal is judged by a model. Most of what this agent does "
1385
+ "leaves a trace in the world or in its calls, and those should be settled by "
1386
+ "code; a judge is the fallback for what leaves none."
1387
+ )
1388
+ if contract.conversational and not load_simulator_prompt(destination):
1389
+ return _err(
1390
+ "Not saved. This agent is conversational, so it needs a simulator prompt for "
1391
+ "the person on the other side. Write it with write_simulator_prompt."
1392
+ )
1393
+ if contract.conversational:
1394
+ problems = validate_simulator_prompt(
1395
+ load_simulator_prompt(destination), require_persona=True
1396
+ )
1397
+ if problems:
1398
+ return _err(
1399
+ "Not saved. The simulator prompt is incomplete:\n - "
1400
+ + "\n - ".join(problems)
1401
+ )
1402
+ dirty = dirty_state(world, sequences, kind)
1403
+ if dirty:
1404
+ counts = world.state()
1405
+ listed = ", ".join(f"{name} ({len(counts[name])} rows)" for name in dirty)
1406
+ return _err(
1407
+ f"Not saved. These hold rows left over from building: {listed}.\n"
1408
+ "This is the state every scenario starts from, so those rows would appear in "
1409
+ "every test as somebody else's order already in the cart. Clear them with "
1410
+ "change_data (DELETE FROM ...), keep the catalogue, and save again."
1411
+ )
1412
+ # What the world publishes when something resets it. Without this a restored world
1413
+ # announces no tools at all, so anything driving it through the environment interface
1414
+ # sees an agent with nothing to call.
1415
+ world.tools = [
1416
+ {
1417
+ "name": spec.name,
1418
+ "description": spec.description,
1419
+ "parameters": {
1420
+ arg: {
1421
+ "type": spec.arg_types.get(arg, "str"),
1422
+ "values": spec.arg_values.get(arg),
1423
+ }
1424
+ for arg in spec.args
1425
+ },
1426
+ }
1427
+ for spec in contract.tools
1428
+ ]
1429
+ path = save(
1430
+ world,
1431
+ destination,
1432
+ notes=str(args.get("notes") or ""),
1433
+ sequences=sequences,
1434
+ # Written out with the world. They are judgement about this agent, and a world reopened
1435
+ # without them would have to have them rewritten before it could be saved again.
1436
+ world_checks=world_checks,
1437
+ )
1438
+ draft_path.unlink(missing_ok=True)
1439
+ tables = world.state()
1440
+ # How many checks judge a VALUE rather than the presence of one. Reported as a fact at the
1441
+ # point of saving, because the per-sub-goal note is easy to read past and this number is
1442
+ # what decides whether the suite would catch an agent that acted on a misheard detail.
1443
+ coded = [one for one in catalogue.sub_goals if one.deterministic()]
1444
+ strong = [one for one in coded if compares_to_a_value(one.check)]
1445
+ return _ok(
1446
+ f"Saved to {path}.\n"
1447
+ f"{len(world.handlers)} tools, {len(tables)} collections, "
1448
+ f"{sum(_size(held) for held in tables.values())} records, "
1449
+ f"{len(world_checks)} world checks.\n"
1450
+ f"{len(strong)} of {len(coded)} checks compare a value against an expectation; the "
1451
+ "rest only test that an argument was present, which an agent acting on a misheard "
1452
+ "detail would pass.\n"
1453
+ + (
1454
+ "Tool/data probes not applicable; conversational runtime proof remains required."
1455
+ if data_free
1456
+ else f"score {report.score:.2f}"
1457
+ )
1458
+ )
1459
+
1460
+ server = tool_server(
1461
+ name=WORLD_SERVER,
1462
+ version="0.1.0",
1463
+ tools=[
1464
+ create_schema,
1465
+ seed,
1466
+ change_data,
1467
+ run_tool,
1468
+ declare_sequence,
1469
+ drop_sequence,
1470
+ amend_contract,
1471
+ add_rule_tool,
1472
+ drop_rule_tool,
1473
+ fix_tool_tool,
1474
+ set_modality_tool,
1475
+ inspect_world,
1476
+ write_simulator_prompt,
1477
+ add_sub_goal,
1478
+ adopt_state,
1479
+ adopt_store,
1480
+ adopt_tool,
1481
+ write_store_ops,
1482
+ add_world_check,
1483
+ write_env_file,
1484
+ run_env_command,
1485
+ check_world,
1486
+ save_world,
1487
+ ],
1488
+ )
1489
+ return server, world
1490
+
1491
+
1492
+ TOOL_NAMES = (
1493
+ "create_schema",
1494
+ "seed",
1495
+ "change_data",
1496
+ "adopt_state",
1497
+ "adopt_store",
1498
+ "adopt_tool",
1499
+ "run_tool",
1500
+ "declare_sequence",
1501
+ "drop_sequence",
1502
+ "amend_contract",
1503
+ "add_rule",
1504
+ "drop_rule",
1505
+ "fix_tool",
1506
+ "set_modality",
1507
+ "inspect_world",
1508
+ "write_simulator_prompt",
1509
+ "add_sub_goal",
1510
+ "write_store_ops",
1511
+ "add_world_check",
1512
+ "write_env_file",
1513
+ "run_env_command",
1514
+ "check_world",
1515
+ "save_world",
1516
+ )