agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,606 @@
1
+ ---
2
+ name: write-scenarios
3
+ description: Write the scenarios an agent is tested with, each proved against the real world before it is kept. Use whenever scenarios, test cases or a suite are wanted for an agent whose contract and world have already been built.
4
+ ---
5
+
6
+ # Write the scenarios
7
+
8
+ When the accepted agent has no custom tools or business data, test its actual conversational
9
+ behavior. Leave reference tool actions and data setup empty; do not invent tool calls, database
10
+ records or capabilities from commented examples. Keep meaningful conduct/evaluation checks and
11
+ exercise the real agent over multiple turns. A lack of tool calls is expected for such an agent,
12
+ not evidence of failure. Runtime readiness and conversation proof still apply.
13
+
14
+ You are writing tests for an AI agent. The environment already exists: a world its tools really act
15
+ on, a prompt for the person it talks to, and a shared catalogue of the named things this agent can be
16
+ checked on. Your job is to write the individual tests, prove each one, and keep it.
17
+
18
+ Everything you need about the agent is in front of you. The contract above lists its tools with their
19
+ arguments, its hard rules, its data, its real use cases and how its tools report a refusal. A summary
20
+ of the world follows it. Do not restate those; read them.
21
+
22
+ ## Which job you have
23
+
24
+ These instructions are loaded by more than one kind of session. Work out which you are from what you
25
+ were asked, then follow only that part.
26
+
27
+ **You were asked for a number of scenarios.** You are planning a suite. Decide what it covers, then
28
+ hand it to `generate_suite`, which runs one writer per part of your plan. Separate planning
29
+ instructions follow this file when a suite is what was asked for. You do not write the scenarios
30
+ yourself.
31
+
32
+ **You were given one brief.** You are a writer. Somebody has already read the agent, decided which
33
+ pairings of thing-acted-on and thing-wanted are worth testing, and how many scenarios each earns.
34
+ Your brief is one of those. Write inside it, and:
35
+
36
+ - Do not widen the brief to take in something interesting you noticed. Say so when you finish and
37
+ let the plan decide.
38
+ - Do not write a second scenario because the person could be somebody else. The same test with a
39
+ different person is one test written twice.
40
+
41
+ **You were asked for one particular scenario**, or to replace one that came back wrong. Write that
42
+ one and nothing else.
43
+
44
+ If the agent's modality has its own instructions, they follow below. They add requirements; they do
45
+ not replace any of these.
46
+
47
+ ## What a scenario is
48
+
49
+ One test. It changes the world a little, gives a person a task, and names what must be true
50
+ afterwards. It is a whole conversation from first contact to a settled outcome, not a single step.
51
+
52
+ | Field | What it is | Required |
53
+ |---|---|---|
54
+ | `name` | Short identifier, lower case with hyphens or underscores. Becomes the scenario's folder name, so it is how a result is read later. | yes |
55
+ | `instruction` | What the person is trying to achieve, written to them, plus everything they hold. | yes |
56
+ | `sub_goals` | Names from the shared catalogue that must hold. Nothing else grades this scenario. | yes |
57
+ | `solution` | What a correct agent would do, as steps. Never run against the agent under test. | yes |
58
+ | `fixture` | A readable manifest of the data this scenario relies on. Its `origin` must be `seed`, `generated` or `mixed`. A fixture changes nothing by itself. | yes, when the world has data |
59
+ | `persona` | Who the person is, as structured fields. | yes, when the prompt asks for one |
60
+ | `setup_code` | Python defining `setup(world)`. This is what actually changes the world. | when the instruction presumes anything |
61
+ | `ready_code` | Python defining `ready(world)`. Returns `None` when the world is ready, or a sentence naming what is missing. | recommended |
62
+ | `use_case` | Which of the agent's use cases this belongs to, copied from the contract word for word. Results are grouped by matching this string exactly, so a rewording becomes a group of its own. | no |
63
+ | `branch` | What is true here that is not true of its siblings in the same use case. | no |
64
+ | `tests` | One line: the condition this scenario passes on. Shown to people as "passes when", so write it to complete that phrase. | no |
65
+ | `variables` | Extra values the prompt asks for, by slot name. Each is substituted into the prompt where its name appears. | when the prompt asks |
66
+ | `max_turns` | How many turns the conversation may take. Defaults to 10. | no |
67
+
68
+ `branch` and `tests` are different and easy to confuse. `branch` is the **condition**: what is
69
+ different about this scenario's world or request. `tests` is the **question**: what the run will find
70
+ out. For a scenario about an expired payment method, `branch` is "the method on file has expired" and
71
+ `tests` is "the agent notices before charging and offers another".
72
+
73
+ Write `branch` and `tests` about the agent's behaviour, never about how the scenario was built.
74
+ "Synthetic", "seeded", "setup_code" and "fixture" name your machinery, not anything the agent did,
75
+ and they are noise in a report.
76
+
77
+ **A solution step** is a tool name plus the arguments the agent would supply:
78
+
79
+ ```
80
+ {"tool": "get_account", "arguments": {"account_id": "..."}}
81
+ ```
82
+
83
+ Some agents insert trusted values between what the model chooses and what the underlying service
84
+ receives: resolved identifiers, prices, routes. Those must never appear as arguments the model
85
+ supposedly chose. Put them in `environment_arguments` on the same step, which the proof passes to the
86
+ service and the agent never sees.
87
+
88
+ ## How grading works
89
+
90
+ A **sub-goal** is one named thing the agent can be checked on, defined once for the agent and shared
91
+ by every scenario that names it. That sharing is what makes results add up: the same sub-goal failing
92
+ in seven of twelve scenarios is one sentence somebody can act on, rather than seven separate notes.
93
+
94
+ Every sub-goal is graded one of two ways, and **you choose which by whether you give it a check**:
95
+
96
+ - **Deterministic.** The sub-goal carries a `check`: Python receiving the world as the run left it and
97
+ every call the agent made with its arguments, returning nothing if it held or a sentence saying
98
+ what was wrong. This is what you want almost always.
99
+ - **Judged.** The sub-goal carries no check, so a model reads the transcript and decides. Reserve
100
+ this for things nothing observable can settle: whether a refusal was explained kindly, whether a
101
+ number was invented.
102
+
103
+ Prefer a check. You have the world afterwards and every call with its arguments, so most things worth
104
+ checking are visible in one of them. A judged sub-goal is reported as judged, and a suite of them
105
+ tells you less than it appears to.
106
+
107
+ **A check that only asks whether a tool was called is not a check.** It proves the plumbing worked, not
108
+ that the agent behaved: any agent that reaches the tool at all passes it, and no agent that behaves
109
+ correctly by another route can. Assert the arguments it was given, or the state the world was left in.
110
+ "The row now holds the value the caller gave" is a check. "the tool appears in the calls" is not.
111
+
112
+ **And do not make the mechanics of ending a call a sub-goal.** Whether a particular closing tool was
113
+ invoked is plumbing. What is worth checking is what the agent did before it stopped: that it left a
114
+ message naming who was calling and why, that it stopped asking questions once there was nobody to
115
+ answer, that it did not press on after being told to stop. A scenario that requires a named closing
116
+ tool fails an agent which closed the call correctly through another tool, scoring right behaviour as
117
+ wrong. Where one tool's documented effect
118
+ already covers another's, requiring both is asking for a redundant call.
119
+
120
+ Name entries that already exist. Do not restate one in your own words and do not invent a second name
121
+ for something already covered. If something genuinely needs checking and no entry covers it, add one
122
+ with `add_sub_goal`.
123
+
124
+ ## The tools, and the order to use them
125
+
126
+ | Tool | What it does |
127
+ |---|---|
128
+ | `inspect_world` | Lists the world's collections and their sizes; with a collection, returns records from it. `matching` is plain text, not SQL. |
129
+ | `inspect_scenario` | Reads one already-kept scenario in full. Use before replacing one, rather than reconstructing it from memory. |
130
+ | `try_calls` | Runs calls against a **throwaway copy** of the world and shows the state they leave. This is how you work out a solution and what its checks should assert. Nothing you do here is visible to anybody else. |
131
+ | `add_sub_goal` | Adds a named thing this agent can be checked on, with its check in code. |
132
+ | `submit_scenario` | Keeps one scenario, after validation and the three gates. |
133
+ | `drop_scenario` | Removes one by name, or all of them with `*`. |
134
+ | `aim_for` | Sets how many scenarios are wanted. Needed when reopening an existing suite to add more, because the target starts at what is already there. Not for saving a suite nobody asked for. |
135
+ | `save_scenarios` | Finishes the suite. Reports what is off about it as a whole. |
136
+ | `amend_contract`, `add_rule`, `drop_rule`, `fix_tool` | Correct the contract when it is wrong. See the last section. |
137
+
138
+ The order for one scenario:
139
+
140
+ 1. `inspect_world` with no collection, then look at the ones that matter. Read the sub-goals that
141
+ already exist.
142
+ 2. Read the agent's hard rules. Each one is a branch waiting to be written.
143
+ 3. Work out the solution with `try_calls`, passing your `setup_code` so you see the world the agent
144
+ will actually face. Confirm the sub-goals you intend to name respond to it.
145
+ 4. `submit_scenario`. Read what comes back: a refusal names exactly what is wrong.
146
+ 5. `save_scenarios` once you have what was asked for.
147
+
148
+ A scenario is written to disk the moment it is kept, so proved work survives a stopped turn. Submit
149
+ as you go rather than composing a whole suite before the first call.
150
+
151
+ If `REAL TOOLS` says `(none)`, this is a conversation-only target. Do not invent a tool or try to
152
+ call a chat endpoint as though it were an agent tool. Use `solution: []`, choose judged sub-goals,
153
+ and use setup/ready only to give the caller a private, valid fixture. In that lane the transcript is
154
+ the outcome evidence; the ready gate still proves the fixture, while tool-solution and no-op gates
155
+ do not pretend there was an environment action to replay.
156
+
157
+ ## The three gates
158
+
159
+ Every scenario is put through these when you submit it. Failing any one means it is not kept, and you
160
+ are told which.
161
+
162
+ **1. Ready.** The world is restored, your `setup_code` runs, then your `ready_code`. The world must
163
+ end up holding what your scenario presumes.
164
+
165
+ This is the gate people skip and the one that saves you. A scenario about the last five items in
166
+ stock is only a test of the agent if there really are five. If there are none, the agent fails for
167
+ something you got wrong and it reads as the agent's fault. `ready_code` makes that impossible.
168
+
169
+ **2. Solvable.** Your reference solution is played through that world, and the checks of every
170
+ sub-goal you named must pass. If they do not, either the scenario cannot be passed at all or a check
171
+ is wrong.
172
+
173
+ For a conversation-only target with no real tools, `solution: []` is intentional and behavioral
174
+ sub-goals are judged from the transcript. Never fabricate a tool call merely to satisfy this gate.
175
+
176
+ **3. Not vacuous.** The same checks run again with nothing done, and must fail. A check that passes
177
+ while the agent does nothing grades nothing while reporting a result.
178
+
179
+ Gate 3 has a common trap. If your scenario is about something that must **not** happen, checking the
180
+ world alone cannot show it: an untouched world looks exactly like one where the agent correctly
181
+ refused. Check the calls instead: that the agent tried, and that the attempt was refused rather than
182
+ succeeding.
183
+
184
+ ## What will be refused, and what to do
185
+
186
+ Validation runs before the gates, and every problem is reported at once, so fix them together.
187
+ **`references/refusals.md` lists every refusal, its cause and its fix.** Read it before your
188
+ first submission rather than after a refusal: most of what it names is cheaper to avoid than to
189
+ correct, and several entries are mistakes that look correct on the page.
190
+
191
+ The ones worth knowing before you write anything:
192
+
193
+ - A value the instruction tells the person to say back must exist in `setup_code` or the world.
194
+ Naming it in `fixture` only declares it.
195
+ - A reference solution of one call is refused, because nothing had to be established first.
196
+ - A scenario name may not contain the person's own name.
197
+ - A `fixture` whose `origin` is `generated` or `mixed` must actually create data.
198
+ - `personality`, `communication_style`, `accent` and `languages` must use offered values.
199
+
200
+ `save_scenarios` additionally reports what is off about the suite as a whole: too few distinct people,
201
+ opening lines repeated word for word, too few locations, verification codes reused between scenarios,
202
+ identical setup data, and for suites where the agent started the conversation, one awareness value
203
+ used for more than about two thirds of them. These are reported rather than refused. Read them and
204
+ fix what they name.
205
+
206
+ ## The bar every scenario has to clear
207
+
208
+ Four of these are enforced by validation. Two are your judgement, and no check can make them for you.
209
+
210
+ - **A competent agent could plausibly fail it.** *(judgement)* If any correct implementation passes
211
+ for free, it teaches nothing. Do not write it.
212
+ - **A real person could plausibly bring this situation.** *(judgement)* Nothing contrived.
213
+ - **Every concrete value is real**, taken from the contract or the world. *(enforced: values handed to
214
+ the person must exist)* An invented identifier makes the test worthless whatever else it does.
215
+ - **Check the path, not only the outcome.** *(enforced: a one-step solution is refused)* Where the
216
+ right answer depends on something the agent must find out first, the sub-goals cover that too.
217
+ - **The scenario seeds what it needs.** *(enforced: a fixture claiming data must create it)* Every
218
+ record whose state decides the outcome is created by this scenario's `setup_code`.
219
+ - **The name says what is tested.** *(enforced: the person's name may not appear in it)*
220
+
221
+ **What is not a scenario.** A person asks for the ordinary thing, the agent does it, both are polite,
222
+ it ends. Nothing was withheld, nothing contradicted, no rule was pressed, no state had to carry, and
223
+ any working agent passes. That is a demonstration. It costs a real run and real money and returns no
224
+ information about the agent. One scenario covers the ordinary path for a whole suite; everything else
225
+ has to earn its place by being able to fail.
226
+
227
+ ```
228
+ BAD solution [transfer_to_human(reason="Account suspended")]
229
+ sub_goals [transferred_to_human]
230
+ (an agent that hands off every request on arrival passes this. Whether it
231
+ looked the account up, and found the suspension, is never measured)
232
+
233
+ GOOD solution [find_account(identifier=...), get_account(account_id=...),
234
+ transfer_to_human(reason="Account suspended")]
235
+ sub_goals [account_identified, account_state_checked, transferred_to_human]
236
+ (the handoff now has to be reached by discovering the reason for it)
237
+ ```
238
+
239
+ ## Three parts that must never leak into each other
240
+
241
+ Getting this wrong is the most common way to write a scenario that looks fine and measures nothing.
242
+
243
+ | | What it is | What it must never contain |
244
+ |---|---|---|
245
+ | **instruction** | what the person is living through | the answer, the checks, facts they could not know, or anything the agent is expected to do |
246
+ | **setup** | the world's condition | anything the person is supposed to say |
247
+ | **checks** | the hidden pass or fail rules | anything the agent was told |
248
+
249
+ ## Writing the instruction
250
+
251
+ **The instruction is a circumstance, not a script.** Write it in the second person, as what this
252
+ person is living through: who they are, what is happening to them, and what they want. Never a list
253
+ of lines to say, and never the agent's turns.
254
+
255
+ ```
256
+ BAD Ask for <thing A>. Then change your mind and ask for <thing B> instead.
257
+ Confirm the total at the end.
258
+ (a stage direction. The person recites it, and the run measures whether the
259
+ agent can follow dictation. Nothing about the change of mind is tested,
260
+ because it arrives exactly when the script says so)
261
+
262
+ GOOD You want <thing A>, and you are not particular about <the detail the agent
263
+ has to settle>. Partway through, you realise <thing B> is what you actually
264
+ need, and you would rather swap than end up with both.
265
+ (a situation. What they say is theirs to work out, and the agent has to cope
266
+ with a change of mind arriving mid-conversation rather than on cue)
267
+ ```
268
+
269
+ Those placeholders are deliberate. Fill them from **this** agent's own data, never from a worked
270
+ example of another agent.
271
+
272
+ **Write the objective, not the history.** A person told what happened narrates it; a person told what
273
+ they want pursues it. Open with the goal in their own words, "get <the thing> put right", not with the
274
+ history that led to it. Then give them the facts they hold, the values they can be asked for, and
275
+ what they will only say when asked.
276
+
277
+ **What they know but will not volunteer goes in its own paragraph**, marked as such: *"You know the
278
+ reference for it, but you will only give it if asked."* Whether the agent asks is the whole point of
279
+ many scenarios. Put it in the instruction and the agent gets it for free; leave it out entirely and
280
+ the scenario cannot be completed.
281
+
282
+ **Knowing a value and volunteering it are separate choices.** The person must possess every value the
283
+ agent could legitimately ask for. Whether they offer it unprompted is the scenario's decision. Those
284
+ are two different sentences and only the second is optional.
285
+
286
+ **Never tell the person what the agent will do.** This is the single most common way a scenario stops
287
+ measuring anything. The agent's moves are what is being tested, so a person told to expect them plays
288
+ along whether or not they happen, and the check passes on a conversation that never earned it. Write
289
+ only what this person knows before the conversation begins.
290
+
291
+ ```
292
+ BAD The agent will tell you about <the condition>. Accept it and say yes when
293
+ asked to confirm.
294
+ (the scenario is testing whether the agent discloses <the condition>. A person
295
+ primed to accept it agrees even when the agent never says it, so the run
296
+ reports a pass for behaviour that did not occur)
297
+
298
+ GOOD You want <the outcome>. You will accept <the condition> if there is one, but
299
+ you want to know <the detail> before you agree to anything.
300
+ (the person's own position. If the agent discloses, they accept; if it does
301
+ not, they ask, and the transcript records which happened)
302
+ ```
303
+
304
+ The rule covers every phrasing: "the agent will send you <a value>", "they will offer you <an
305
+ option>", "they should hand you over". Give the person the value, the preference or the problem they
306
+ arrived with. What the agent does about it is the measurement, so it cannot also be part of the brief.
307
+
308
+ **The test that catches all of it: could this person say the sentence out loud?** A parenthetical
309
+ explaining where the agent should find a value is not a smaller version of the mistake, it is the same
310
+ mistake more quietly.
311
+
312
+ ```
313
+ BAD Your <destination>: <value> (the agent should find this from your <record>)
314
+ (the person has no idea the agent has records, let alone which one. The note is
315
+ written for whoever reads the scenario, and it names the mechanism being tested)
316
+
317
+ GOOD Your <destination> is the same one you used last time. You do not remember the
318
+ exact address and would rather not look it up.
319
+ (now the person has a reason to expect the agent to know, which is what makes
320
+ the lookup worth testing, without being told the lookup exists)
321
+ ```
322
+
323
+ Pre-agreeing to something the agent has not done yet is the most damaging form. "You have already
324
+ <completed the step> that the agent will <send>" hands the agent a pass. Write what the person has
325
+ done, never what they have done in response to an action the agent has not taken.
326
+
327
+ **Steps that happen outside the conversation need a state, not a response.** Some flows depend on the
328
+ person doing something the simulation cannot perform: following a link, checking another device,
329
+ reading a message. The temptation is to write their answer in advance, which is the pass-handing form
330
+ again, because the answer arrives whether or not the agent ever asked.
331
+
332
+ ```
333
+ BAD The agent will send you <the out-of-band thing>. Tell them you have
334
+ completed it when asked.
335
+ (the scenario is testing whether the agent sends it. This person confirms
336
+ completing it even in a run where nothing was ever sent)
337
+
338
+ GOOD You have your <device> with you and you are willing to follow anything you
339
+ are sent. You have not been sent anything yet.
340
+ (a state. If the agent sends it, this person can act on it and say so
341
+ truthfully. If the agent never does, they have nothing to confirm, and the
342
+ transcript shows the difference)
343
+ ```
344
+
345
+ The closing sentence matters: saying what has **not** happened yet is what stops the person assuming
346
+ it has. And only write such a step where the agent can observe it completing, because the person
347
+ saying they did it changes nothing the agent reads. If the agent confirms progress by checking state,
348
+ the world has to move when the person acts, or the agent polls something that never changes and the
349
+ scenario measures the world's gap instead of the agent.
350
+
351
+ The same applies to anything the agent can only offer. A check that passes only once the person
352
+ accepts an optional courtesy needs that willingness written in, because the agent can raise the offer
353
+ but cannot make them take it. Either give them a reason to accept, or check that the offer was made
354
+ rather than what followed it.
355
+
356
+ ### What this person is known by
357
+
358
+ Many agents establish who they are dealing with before they will act. Give that its own short section
359
+ at the end of the instruction, and **read every value out of the world with `inspect_world` first**.
360
+ Never invented, never carried from another scenario: the record has to be the one the agent's own
361
+ lookup will find.
362
+
363
+ Four rules, and each has cost a whole run:
364
+
365
+ **Cover every route, not the one you expect.** Where an agent can establish something more than one
366
+ way, which way it takes is not yours to choose. An instruction carrying values for one route is
367
+ complete until that route fails, and then the person cannot answer a question they plainly should be
368
+ able to answer.
369
+
370
+ **Say what each value is for.** Where two values share a shape but not a role, the current one and
371
+ the replacement, the account's and the order's, give both and name each role. Handed one, the person
372
+ offers it for the other purpose because it is the only such value they have. It is real, it is in the
373
+ instruction, and it still fails, which is harder to diagnose than a missing value.
374
+
375
+ **Take them all from one record.** Fields from two records describe somebody who does not exist, and
376
+ no lookup will find them.
377
+
378
+ ### The person, and why they are hard
379
+
380
+ `persona` is the structured profile of the person making the request: `name`, `gender`, `age_group`,
381
+ `occupation`, `location`, `personality`, `communication_style`, `keywords`, `languages`, `accent`,
382
+ `multilingual`, and free-form `metadata`. `personality`, `communication_style`, `accent` and
383
+ `languages` must use offered values, because each selects real behaviour downstream; a word of your
384
+ own renders fine and selects nothing.
385
+
386
+ Use the fields that change the risk being tested, and **make the person the reason the scenario is
387
+ hard**. If swapping in a calm, fully informed person would not change the outcome, the persona is
388
+ doing no work.
389
+
390
+ **A different name is not a different person.** Personas drift toward one temperament: cooperative,
391
+ articulate, patient, answering exactly what was asked. A suite of those tests a person the agent will
392
+ rarely encounter, and passes every scenario for the same reason. Vary `personality` and
393
+ `communication_style`, not just identity: somebody terse to the point of unhelpfulness, somebody who
394
+ volunteers three things at once, somebody distracted who has to be asked twice, somebody impatient who
395
+ pushes back early. Let the situation pick the temperament rather than attaching one at random.
396
+
397
+ Keep the person and the world's condition apart: the persona is who is asking, `setup_code` is what is
398
+ true of the world. A name that says one person in the persona and another in the instruction
399
+ misreports every result anybody reads.
400
+
401
+ ## When the agent started the conversation
402
+
403
+ Read `CALL DIRECTION` on the contract before writing a single instruction. The two directions need the
404
+ person written differently, and getting it wrong tests the wrong half of the exchange.
405
+
406
+ **Inbound: the person approached the agent.** Everything above assumes this. They have an errand, they
407
+ know why they are there, and they open by saying what they want.
408
+
409
+ **Outbound: the agent approached the person.** This inverts almost everything:
410
+
411
+ - They have **no errand of their own.** They were doing something else.
412
+ - They do **not know who this is** until the agent says so, and must not act as if they do.
413
+ - Their first turn is a bare greeting and nothing more. It answers the agent's opening, which still
414
+ comes first.
415
+ - They may be **suspicious.** An unexpected approach about their account is what a scam looks like,
416
+ so asking the agent to prove itself is correct behaviour, not obstruction.
417
+ - They may be **busy or unwilling.** Declining to talk now is a legitimate outcome worth testing.
418
+ - What this is about is the **agent's** purpose. The instruction says how the person reacts to it,
419
+ not what they wanted.
420
+
421
+ ```
422
+ BAD Book the premium tier from your home to your office.
423
+ (they never approached anyone; nothing prompts them to ask for this)
424
+
425
+ GOOD You are at home getting ready for work. If someone contacts you about the
426
+ booking you have on file, you would take it, leaving from home and going to
427
+ the office. You will not raise any of that yourself.
428
+ ```
429
+
430
+ An outbound instruction that opens with a request has been written as inbound, and the scenario then
431
+ tests an errand the agent never raised.
432
+
433
+ ### How much the person already knows
434
+
435
+ `caller_awareness` changes the whole exchange, so choose it deliberately. Each value needs
436
+ **different data** in the instruction:
437
+
438
+ | They are | What the instruction must carry |
439
+ |---|---|
440
+ | `expecting` | They know what it is about and roughly what they agreed, so they can be asked to confirm a detail. They must hold that detail, and their version may differ from the world's. |
441
+ | `partial` | They know something happened but not the detail: not the date, not the amount, not which of two things. Say what they do recall and what they have lost. |
442
+ | `unaware` | No context at all. The agent has to establish who they are and why it is contacting them before anything else. Give them the facts they hold about themselves and nothing about the reason. |
443
+
444
+ **Do not write every outbound scenario as `expecting`.** It is the easiest and least informative: a
445
+ person who was told to expect this can reasonably ask for what they want, so the scenario stops
446
+ testing how the agent opens something it started. At least one outbound scenario per use case must be
447
+ `unaware`, and where a use case gets only one or two, prefer `unaware`.
448
+
449
+ **But an `unaware` person still needs facts.** Somebody with no context and nothing to offer produces
450
+ a short, empty exchange, and that is a badly written scenario rather than a finding about the agent.
451
+ They hold their own details and a reaction to being contacted unexpectedly; what they must not hold is
452
+ the reason.
453
+
454
+ ## Making the world match the instruction
455
+
456
+ **Whatever the instruction presumes, setup has to make true.** This is where scenarios most often go
457
+ wrong: the instruction says the person is returning an order that has already shipped, setup leaves
458
+ every order pending, so the agent refuses correctly and the scenario fails it for being right.
459
+
460
+ Read your own instruction back, list every condition it assumes, and make sure `setup_code`
461
+ establishes each one and `ready_code` proves it. If the instruction hands the person a value to say
462
+ back, `setup_code` is what puts that exact value where the agent will look for it.
463
+
464
+ ### setup_code
465
+
466
+ Python defining `setup(world)`.
467
+
468
+ **Create the records this scenario turns on.** Every record whose state decides the outcome is made
469
+ here, with values belonging to this scenario: its own person, its own order, its own booking, its own
470
+ code. Shared reference data the whole world sits on, a product catalogue or a list of regions, can be
471
+ read as it is and used as a model for what a realistic new record looks like. What you must not do is
472
+ build the test on rows that were already there: another scenario may change them, two scenarios then
473
+ quietly test the same row, and neither describes a world it controls.
474
+
475
+ **Write every setup against the base world, never against another scenario.** At run time each
476
+ scenario restores its own copy of the frozen base and applies only its own setup, so nothing another
477
+ scenario did is there. Writers run at the same time and in any order, so there is no "before" to
478
+ depend on: if a scenario needs an order delivered, its own setup delivers it. The calls you make while
479
+ rehearsing with `try_calls` run on a throwaway copy and change nothing anybody else sees.
480
+
481
+ You have two ways to change things, and **neither names what the world is kept in**. A scenario that
482
+ wrote SQL would only work against a world that happened to use that engine, and the store varies more
483
+ between agents than anything else.
484
+
485
+ **Prefer the agent's own tools.** They go through the same path the agent will, so anything the world
486
+ would refuse you would have refused the agent too.
487
+
488
+ ```python
489
+ def setup(world):
490
+ world.call("add_to_stock", {"item_id": "widget", "quantity": 5})
491
+ ```
492
+
493
+ **Otherwise change the world directly.** Three calls cover it, and none of them names what the world
494
+ is kept in: `world.put(collection, record)` adds one, `world.change(collection, key, changes, by=...)`
495
+ alters one, `world.drop(collection, key, by=...)` removes one. Use the direct route only for states no
496
+ tool can produce: a record already in a condition the agent could never create itself.
497
+
498
+ **Collections are not all lists.** A collection held in a store gives a list of records; one the agent's own code
499
+ keeps is often a mapping, and iterating it yields keys rather than records. Look with `inspect_world`
500
+ before writing against one.
501
+
502
+ **Fill every field an existing record has.** Read one back with `inspect_world` and give your new record
503
+ the same fields, timestamps included. A column the store requires and you leave out, or set to nothing,
504
+ fails the insert and the scenario dies in its own setup:
505
+
506
+ ```
507
+ NotNullViolation: null value in column "issued_at" of relation "otp_codes" violates not-null constraint
508
+ ```
509
+
510
+ Measured on a real run: four of five calls lost that way, to the "issued_at" column on a code row and
511
+ "taken_at" on a past trip. If a field is a time, give it a plausible one rather than nothing.
512
+
513
+ **Write a value of the type the column actually holds.** A true or false field takes `True` or `False`,
514
+ never `1` or `0`. Some stores accept either and some reject the number outright, and the scenario then dies
515
+ in its own setup before the conversation starts: measured on a real run, three of five calls failed with
516
+ `column "phone_verified" is of type boolean but expression is of type smallint`, because the setup wrote
517
+ `1`. Records you read back may display as `1` and `0`; that is how they are shown, not what the column is.
518
+
519
+ `references/world-api.md` has the exact signatures, the `key=` caveat, how to handle either shape, and
520
+ a worked `ready_code`.
521
+
522
+ One exception. Where the contract says the target's store is hardcoded and process-local, with no
523
+ configuration seam, `setup_code` cannot alter target records, because the world and the live target
524
+ are separate copies. Use only records already in the frozen base, keep setup empty for them, and
525
+ settle outcomes from the captured calls. If coverage needs state the base lacks, report that the
526
+ target needs a seed or reset seam rather than writing a scenario that cannot run.
527
+
528
+ ## The solution is not optional
529
+
530
+ Every scenario carries what a correct agent would do. It is never run against the agent under test.
531
+ It exists to prove the scenario can be passed at all, and it is what gate 2 uses.
532
+
533
+ Work it out with `try_calls` before you submit: run the calls, pass your `setup_code` so you see the
534
+ world the agent will face, and confirm the sub-goals you name respond to the state they leave.
535
+
536
+ **A one-call solution is almost always wrong, and is refused.** The agent does not begin knowing who
537
+ it is dealing with or what is true of their account, so before the step that resolves the scenario it
538
+ has to find out: identify the person, read the record, check the state that decides the answer. Those
539
+ lookups belong in the solution, and the sub-goals have to name them.
540
+
541
+ Refusals and handoffs are where this goes wrong most often, because the terminal call looks so
542
+ obviously like the point. It is not. **Deciding** to refuse is the point, and a decision never reached
543
+ from evidence was never tested.
544
+
545
+ ## Realistic values
546
+
547
+ Placeholder data makes a paid run look like a demo, and several kinds are refused outright.
548
+
549
+ Recognisable stand-ins are refused outright, and `references/refusals.md` lists which. Two rules go
550
+ beyond what any check can see:
551
+
552
+ - **Keep every fact internally consistent.** The persona, the fixture, the records the setup creates
553
+ and the instruction must all describe the same person. A detail in the persona that does not match
554
+ the record the agent will find is a scenario that fails for its own reasons.
555
+ - **Vary the outcome as well as the wording.** Success, refusal, correction, ambiguity, retry, stale
556
+ state, unavailable dependency and recovery should not all share one happy-path fixture.
557
+
558
+ ## One coherent terminal outcome
559
+
560
+ Do not combine branches whose correct outcomes stop one another. A scenario that asks the agent to
561
+ hand off an out-of-scope request must not also require a transaction to finish afterwards. A scenario
562
+ that correctly refuses, escalates, cancels or ends the exchange must not carry a sub-goal for work
563
+ that only happens when it continues.
564
+
565
+ **An outcome the harness cannot observe is not a terminal outcome.** Where the agent can start
566
+ something whose completion happens elsewhere, test that it started it, and test the work that would
567
+ follow in a separate scenario.
568
+
569
+ Before keeping a scenario, read its instruction, solution and every named sub-goal as a single path.
570
+ If satisfying one sub-goal can correctly prevent another from being reached, split them. Never add an
571
+ unrelated sub-goal merely to make every scenario exercise a tool.
572
+
573
+ ## Two scenarios differ only if the right answer differs
574
+
575
+ Not if the wording differs. "The item is in stock" and "the item is out of stock" are two scenarios,
576
+ because the correct outcome differs. Two polite requests for the same thing are one scenario written
577
+ twice.
578
+
579
+ Changing who is asking, where they are going, or which option they pick does **not** make a second
580
+ scenario. The agent does the same things in the same order and the same checks decide the result; all
581
+ that changed is the noun. Ten of those look like coverage in a list and are one test.
582
+
583
+ Vary the person **within** a scenario you were already going to write, never to produce another one.
584
+ A suite where everybody is calm and cooperative tests one kind of person, so let temperament and
585
+ communication style differ across the suite. That is diversity inside the tests you have, not a source
586
+ of extra tests.
587
+
588
+ **A count you were given is a ceiling, not a quota.** If the agent's real branches run out at twelve,
589
+ submit twelve and say why. Padding buys rows that can never fail independently, and hides the branches
590
+ nobody wrote behind a suite that looks thorough. An even spread across every use case is a warning
591
+ sign, not a goal: real agents have use cases worth five scenarios and use cases worth one.
592
+
593
+ ## If the contract is wrong
594
+
595
+ You will sometimes find the contract does not match what the world does: a tool that accepts a value
596
+ it was not recorded as accepting, a rule that is not really a rule. Correct it with `amend_contract`,
597
+ `add_rule`, `drop_rule` or `fix_tool`, and say why. Every amendment is recorded on the contract.
598
+
599
+ Never work around a contract you believe is wrong. A scenario written to dodge a bad contract hides
600
+ the problem, and everything built afterwards inherits it.
601
+
602
+ ## Finishing
603
+
604
+ Say what the suite covers and what it does not, which sub-goals carry the most scenarios, and name
605
+ anything you could not test because the environment or the contract does not support it. Report the
606
+ honest number: a smaller suite that is entirely real is worth more than a padded one.
@@ -0,0 +1,28 @@
1
+ # Every refusal, its cause and its fix
2
+
3
+ Validation runs before the three gates when you submit a scenario. Every problem is reported at
4
+ once, so fix them together and submit again.
5
+
6
+ | What you are told | Why | Fix |
7
+ |---|---|---|
8
+ | `no name` / `no instruction` | Empty required field. | Supply it. |
9
+ | `persona has no details` | A persona was given with every field blank. | Fill it, or leave `persona` out entirely. |
10
+ | `persona is incomplete: ...` | A persona needs `name`, `personality`, `communication_style`, `initial_message`, `accent`, at least one language and at least one keyword. | Fill the named fields. |
11
+ | `persona <field> ... is not one the platform knows` | `personality`, `communication_style`, `accent` and `languages` must come from the offered values. A word of your own renders fine and then selects no behaviour. | Use an offered value. Anything else about the person goes in `metadata`. |
12
+ | `no sub_goals` | Nothing would grade the scenario. | Name the catalogue entries this scenario exercises. |
13
+ | `sub_goals not in the catalogue: ...` | A name that does not exist. The message lists what does. | Use an existing name, or `add_sub_goal` first. |
14
+ | `no solution` | Without the actions a correct agent would take, nothing can show the scenario is passable. | Work it out with `try_calls`. |
15
+ | `no fixture manifest` | The world has data and the scenario declared none. | Add `fixture` with `origin` and the facts the person relies on. |
16
+ | `fixture.origin must be seed, generated, or mixed` | Any other value. | Use one of the three. |
17
+ | `fixture.origin is 'generated' ... but setup_code is empty` | The fixture claims the scenario creates data while creating none. | Seed everything the fixture names, or declare `origin: seed` and use only records that already exist. |
18
+ | `the instruction gives the person ... to say back, and neither setup_code nor the world holds it` | The instruction hands over a code, reference or identifier that exists nowhere, so the conversation cannot succeed however well the agent behaves. | Seed that exact value in `setup_code`, or tell the person the value that is seeded. Naming it in `fixture` only declares it. |
19
+ | `setup_code only adjusts records that were already there ...` | The setup changes or drops rows it did not create, so the scenario shares its data with every other scenario touching those rows. | Create what the outcome turns on with `world.put`, or by driving the agent's own tool, then adjust that. An empty setup stays legal for the no-seam case. |
20
+ | `the reference solution is a single call ...` | Nothing had to be established before the outcome, so an agent that fires that call on arrival passes. | Show how the outcome is reached: the lookups the decision depends on, named as sub-goals too. |
21
+ | `the name contains the person's own name` | The name says who was on the other end rather than what broke. | Name it for the behaviour: `cancel_active_booking_with_fee`, not `dana_cancels_her_booking`. |
22
+ | `fixture uses predictable verification code(s)` | Sequential or repeated digits. | Generate an unremarkable value of the right shape. |
23
+ | `fixture contains placeholder demo data` | `test user`, `john doe`, `123 main street` and similar. | Use plausible real-world values. |
24
+ | `fixture uses placeholder payment-card ending(s)` | `4242`, `1234`, `0000` and similar, in the fixture or spoken in the instruction. | Use an unremarkable ending. |
25
+ | `fixture uses placeholder transaction identifier(s)` | Identifiers ending in a bare `1`, or obvious stand-ins. | Use values shaped like the agent's real ones. |
26
+ | `setup_code must define setup(world)` / `ready_code must define ready(world)` | Wrong entry point. | Define the function with that exact name. |
27
+ | `the prompt asks for ..., which this scenario does not supply` | The prompt has a slot nothing fills, and an unfilled slot reaches the person verbatim. | Add it to `variables`. |
28
+ | `<tool> requires <value> from this call, but the reference solution does not create it first` | A hard rule says a value must come from this conversation, and the solution supplies it from setup or `environment_arguments` instead. | Put the step that produces it earlier in the solution. |