agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1048 @@
1
+ """A scenario: a delta on the base environment, and what must hold afterwards.
2
+
3
+ The base is built once — the world, the simulator's prompt, the catalogue of sub-goals. A
4
+ scenario changes a few values in that world, fills the prompt's slots, and names which sub-goals
5
+ must hold. It is not a template with values slotted into it; the harness writes each one.
6
+
7
+ It also carries a **solution**: what a correct agent would do. That is not decoration. It is what
8
+ proves, before the scenario is ever used, that the scenario can be passed at all and that its
9
+ checks are not vacuous — the two gates in ``prove.py``. Terminal-bench keeps its tasks honest the
10
+ same way, and it needs no model to do it.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import ast
16
+ import hashlib
17
+ import json
18
+ import os
19
+ import re
20
+ from collections import Counter
21
+ from math import ceil
22
+ from typing import Any, ClassVar
23
+
24
+ from pydantic import BaseModel, Field, model_validator
25
+
26
+ from .catalogue import Catalogue
27
+ from .simulator import variables_in
28
+
29
+
30
+ # What a fixture's `origin` may say, and which of those claim the scenario creates data itself.
31
+ FIXTURE_ORIGINS = ("seed", "generated", "mixed")
32
+ ORIGINS_THAT_CREATE = ("generated", "mixed")
33
+
34
+ # For an outbound call, how much the person already knows about why they are being rung. The order
35
+ # is the axis: told to expect it, half remembers, no idea at all. Named once so the schema a writer
36
+ # is offered, the suite's spread rule and the caller's own briefing cannot drift apart.
37
+ CALLER_AWARENESS = ("expecting", "partial", "unaware")
38
+ LEAST_AWARE = "unaware"
39
+
40
+ # Who picked up an outbound call. Empty means a person; "voicemail" means nobody is on the line.
41
+ ANSWERED_BY = ("person", "voicemail")
42
+ VOICEMAIL = "voicemail"
43
+
44
+ # Which kind of mailbox answered: what the greeting says, and whether a tone follows it.
45
+ VOICEMAIL_STYLES = ("personal", "carrier", "operator", "full")
46
+ DEFAULT_VOICEMAIL_STYLE = "personal"
47
+
48
+ # The share of a suite a mailbox may occupy: a ceiling with no floor, since none is legitimate.
49
+ RARE_CONDITION_SHARE = 0.05
50
+
51
+ # The switch that removes mailboxes from a run altogether, for when they are not wanted at all
52
+ # rather than merely kept rare.
53
+ VOICEMAIL_SWITCH = "ALK_VOICEMAIL_SCENARIOS"
54
+ _OFF = ("0", "off", "false", "no")
55
+
56
+
57
+ def voicemail_enabled() -> bool:
58
+ """Whether this run may write or place a call a mailbox answers. On unless the switch says no."""
59
+ return os.environ.get(VOICEMAIL_SWITCH, "1").strip().lower() not in _OFF
60
+
61
+
62
+ class Step(BaseModel):
63
+ """One action in a reference solution."""
64
+
65
+ tool: str
66
+ arguments: dict[str, Any] = Field(default_factory=dict)
67
+ # Source-backed agents often add trusted session state between the model-facing function
68
+ # and the dependency API: internal identifiers, resolved lookups, priced results, and similar
69
+ # must never be exposed as arguments the model supposedly chose. A reference proof still
70
+ # has to drive the real dependency so its database effects can be checked, so it may carry
71
+ # that dependency payload separately. Agent runs never read this field.
72
+ environment_arguments: dict[str, Any] = Field(default_factory=dict)
73
+
74
+
75
+ class Persona(BaseModel):
76
+ """The simulated caller, in the same shape used by existing voice scenarios.
77
+
78
+ A persona controls how the caller pursues a scenario's task. The task itself remains on
79
+ ``Scenario.instruction`` so the harness can vary either one without conflating them.
80
+ """
81
+
82
+ name: str = ""
83
+ gender: str = ""
84
+ age_group: str = ""
85
+ occupation: str = ""
86
+ location: str = ""
87
+ personality: str = ""
88
+ communication_style: str = ""
89
+ # The first thing this person actually says. Voice agents often greet immediately; leaving
90
+ # this to the simulator model produced generic "Hello?" turns and avoidable silence races.
91
+ initial_message: str = ""
92
+ keywords: list[str] = Field(default_factory=list)
93
+ languages: list[str] = Field(default_factory=list)
94
+ accent: str = ""
95
+ multilingual: bool = False
96
+ metadata: dict[str, Any] = Field(default_factory=dict)
97
+ # Optional deterministic voice policy for transactional scenarios. It keeps
98
+ # caller facts realistic and varied while avoiding LLM role drift during a
99
+ # long tool-heavy phone flow.
100
+ scripted_caller: dict[str, Any] | None = None
101
+
102
+ def described(self) -> bool:
103
+ return bool(
104
+ self.name
105
+ or self.gender
106
+ or self.age_group
107
+ or self.occupation
108
+ or self.location
109
+ or self.personality
110
+ or self.communication_style
111
+ or self.keywords
112
+ or self.languages
113
+ or self.accent
114
+ or self.metadata
115
+ )
116
+
117
+ def missing_profile_fields(self) -> list[str]:
118
+ """The minimum needed for a scenario to exercise caller variation intentionally."""
119
+ missing = [
120
+ name
121
+ for name, value in (
122
+ ("name", self.name),
123
+ ("personality", self.personality),
124
+ ("communication_style", self.communication_style),
125
+ ("initial_message", self.initial_message),
126
+ ("accent", self.accent),
127
+ )
128
+ if not value.strip()
129
+ ]
130
+ if not self.languages:
131
+ missing.append("languages")
132
+ if not self.keywords:
133
+ missing.append("keywords")
134
+ return missing
135
+
136
+ def format_persona(self) -> str:
137
+ """A stable, human-readable profile the simulator can consistently embody."""
138
+ parts = []
139
+ identity = []
140
+ for label, value in (
141
+ ("Name", self.name),
142
+ ("Gender", self.gender),
143
+ ("Age Group", self.age_group),
144
+ ("Occupation", self.occupation),
145
+ ("Location", self.location),
146
+ ):
147
+ if value:
148
+ identity.append(f"- {label}: {value}")
149
+ if identity:
150
+ parts.append("# YOUR IDENTITY\n\n" + "\n".join(identity))
151
+
152
+ behavior = []
153
+ if self.personality:
154
+ behavior.append(f"- Personality: {self.personality}")
155
+ if self.communication_style:
156
+ behavior.append(f"- Communication Style: {self.communication_style}")
157
+ if self.keywords:
158
+ behavior.append("- Key Traits: " + ", ".join(self.keywords))
159
+ if behavior:
160
+ parts.append("# YOUR PERSONALITY & COMMUNICATION\n\n" + "\n".join(behavior))
161
+
162
+ speech = []
163
+ if self.languages:
164
+ speech.append("- Language(s): " + ", ".join(self.languages))
165
+ if self.accent:
166
+ speech.append(f"- Accent: {self.accent}")
167
+ if self.multilingual:
168
+ speech.append(
169
+ "- Switch languages naturally when the conversation calls for it."
170
+ )
171
+ if speech:
172
+ parts.append("# LANGUAGE & SPEECH PATTERNS\n\n" + "\n".join(speech))
173
+
174
+ if self.metadata:
175
+ characteristics = [
176
+ f"- {key.replace('_', ' ').title()}: {value}"
177
+ for key, value in self.metadata.items()
178
+ ]
179
+ parts.append(
180
+ "# ADDITIONAL CHARACTERISTICS\n\n" + "\n".join(characteristics)
181
+ )
182
+ return "\n".join(parts)
183
+
184
+
185
+ def _slug(name: str) -> str:
186
+ """An ASCII key for ``name``, safe to send as a header value.
187
+
188
+ Falls back to a digest rather than an empty string: an empty key would collapse every
189
+ scenario in a job onto one idempotency key on the receiving side.
190
+ """
191
+ cleaned = re.sub(r"[^a-z0-9]+", "-", (name or "").strip().lower()).strip("-")
192
+ return cleaned or "scenario-" + hashlib.sha256(name.encode()).hexdigest()[:12]
193
+
194
+
195
+ def _decided_by(name: str) -> bool:
196
+ """Whether this scenario is noisy, decided by its name so a rerun decides the same."""
197
+ return hashlib.sha256((name or "").encode()).digest()[0] % 2 == 0
198
+
199
+
200
+ class Scenario(BaseModel):
201
+ """One test: what changes, what is asked, what a correct agent does, what must hold."""
202
+
203
+ name: str
204
+ # How this scenario is identified on the wire. Derived from ``name``, which is already unique
205
+ # across a suite and already a slug because it is the folder name. It ships as a header, so
206
+ # anything outside ASCII is dropped and an empty result falls back to a digest.
207
+ scenario_key: str = ""
208
+ # Assigned by the platform when the scenario is pre-allocated. Never written here.
209
+ scenario_id: str = ""
210
+ use_case: str = ""
211
+ # What makes this row different from its siblings in the same use case. Coverage is counted
212
+ # on the pair, so a use case can carry many scenarios without any reading as a duplicate.
213
+ branch: str = ""
214
+ tests: str = ""
215
+
216
+ # What this scenario changes about the world after it is reset, as code: a file defining
217
+ # ``setup(world)``. Rows in a table were enough while every world was a database, and they
218
+ # are not enough now — a scenario may need a service to start returning errors, a file to be
219
+ # missing, a queue to be backed up. Code can express all of that; a table of rows cannot.
220
+ setup_code: str = ""
221
+
222
+ # Whether the world is actually ready for this scenario, as code: a file defining
223
+ # ``ready(world)`` that answers with nothing when the world holds what this scenario
224
+ # presumes, or a sentence saying what is missing.
225
+ #
226
+ # This is the precondition, and it is the difference between a real finding and a wasted
227
+ # run: a scenario about the last five chocolates is only a test of the agent if there really
228
+ # are five. Otherwise the agent fails for something we got wrong, and it looks like the
229
+ # agent's fault.
230
+ ready_code: str = ""
231
+
232
+ # The task. For a conversational agent it fills the simulator prompt's instruction slot; for
233
+ # a browser or coding agent it goes to the agent directly.
234
+ instruction: str = ""
235
+ # Who is making the request. This is deliberately separate from the task so a caller's
236
+ # communication needs do not get buried in an unstructured instruction.
237
+ persona: Persona | None = None
238
+ # Anything else that prompt asks for, by slot name.
239
+ variables: dict[str, str] = Field(default_factory=dict)
240
+ # A readable declaration of which data makes this scenario real. ``setup_code`` remains the
241
+ # executable delta; this is the index a person and the UI can inspect without reverse-
242
+ # engineering Python. Typical keys are origin (seed/generated/mixed), identity, credentials,
243
+ # location and account_state. It is intentionally open-ended across agent domains.
244
+ fixture: dict[str, Any] = Field(default_factory=dict)
245
+
246
+ # What a correct agent would do. Run by the gates, never by the agent under test.
247
+ solution: list[Step] = Field(default_factory=list)
248
+
249
+ # Which entries of the shared catalogue must hold. Named, not restated, so results roll up
250
+ # across the suite: the same sub-goal failing in seven of twelve scenarios is one sentence.
251
+ sub_goals: list[str] = Field(default_factory=list)
252
+
253
+ max_turns: int = 10
254
+
255
+ # Where this call is made from. A string names the place ("street", "vehicle", "retail"), and
256
+ # True asks for noise while leaving the place to the fixture. Left unset it is decided from
257
+ # the name, so a suite still covers both conditions but the same suite decides the same way
258
+ # twice; a coin flip here made a seeded run unreproducible.
259
+ background_noise: bool | str = ""
260
+
261
+ # Whether the agent placed this call or answered it. Voice only: a chat is always started by
262
+ # the person, so it stays inbound. An outbound caller has no opening request to make, which is
263
+ # a different test of the agent rather than the same one with a reworded greeting.
264
+ # Empty means defer to the run and then to the contract, which is where the agent's own
265
+ # direction was identified. Defaulting it to "inbound" here would be written into the saved
266
+ # document and silently outrank both of them.
267
+ call_direction: str = ""
268
+ # For an outbound call, how much this person already knows about why they are being rung:
269
+ # "expecting", "partial" or "unaware". Unset means unaware, the case the agent must work
270
+ # hardest for.
271
+ caller_awareness: str = ""
272
+ # Who answered. Outbound only: a mailbox cannot answer a call the person placed themselves.
273
+ answered_by: str = ""
274
+ # Which kind of mailbox answered. Only read where answered_by is "voicemail"; empty is personal.
275
+ voicemail_style: str = ""
276
+
277
+ # Slots the caller filled by the run rather than by the scenario. Listed so a template that
278
+ # uses one is not rejected as unfillable at write time.
279
+ RUNTIME_SLOTS: ClassVar[tuple[str, ...]] = ("channel", "situation")
280
+
281
+ @model_validator(mode="after")
282
+ def _identify(self) -> "Scenario":
283
+ if not self.scenario_key:
284
+ self.scenario_key = _slug(self.name)
285
+ if self.background_noise == "":
286
+ self.background_noise = _decided_by(self.name)
287
+ return self
288
+
289
+ def slots(self) -> dict[str, str]:
290
+ """Every value this scenario offers the simulator prompt."""
291
+ persona = {"persona": self.persona.format_persona()} if self.persona else {}
292
+ runtime = {name: "" for name in self.RUNTIME_SLOTS}
293
+ return {
294
+ "instruction": self.instruction,
295
+ **runtime,
296
+ **self.variables,
297
+ **persona,
298
+ }
299
+
300
+
301
+ def validate_scenario(
302
+ scenario: Scenario,
303
+ catalogue: Catalogue,
304
+ world_state: dict[str, list[dict[str, Any]]],
305
+ simulator_prompt: str = "",
306
+ *,
307
+ allow_empty_solution: bool = False,
308
+ ) -> list[str]:
309
+ """Problems that make a scenario unusable, found without running anything.
310
+
311
+ Whether it can actually be passed is a different question, and no amount of reading settles
312
+ it. That is what the gates are for.
313
+ """
314
+ problems: list[str] = []
315
+ if not scenario.name.strip():
316
+ problems.append("no name")
317
+ if not scenario.instruction.strip():
318
+ problems.append("no instruction: there is nothing for the run to be about")
319
+ if scenario.persona is not None and not scenario.persona.described():
320
+ problems.append("persona has no details")
321
+ elif scenario.persona is not None and (
322
+ missing := scenario.persona.missing_profile_fields()
323
+ ):
324
+ problems.append("persona is incomplete: " + ", ".join(missing))
325
+ elif scenario.persona is not None:
326
+ # A persona in words of its own renders fine and then does nothing: no behaviour guidance
327
+ # attaches, and the accent it names selects no voice.
328
+ from .persona_guides import unrecognised
329
+
330
+ problems.extend(unrecognised(scenario.persona.model_dump()))
331
+ if not scenario.sub_goals:
332
+ problems.append(
333
+ "no sub_goals: nothing would be graded. Name the entries of the catalogue this "
334
+ "scenario is meant to exercise"
335
+ )
336
+ if world_state and not scenario.fixture:
337
+ problems.append(
338
+ "no fixture manifest: declare the seed/generated/mixed data this scenario relies on"
339
+ )
340
+ elif scenario.fixture and str(
341
+ scenario.fixture.get("origin") or ""
342
+ ).lower() not in set(FIXTURE_ORIGINS):
343
+ problems.append(
344
+ "fixture.origin must be "
345
+ + ", ".join(FIXTURE_ORIGINS[:-1])
346
+ + f", or {FIXTURE_ORIGINS[-1]}"
347
+ )
348
+ elif (
349
+ scenario.fixture
350
+ and str(scenario.fixture.get("origin") or "").lower()
351
+ in set(ORIGINS_THAT_CREATE)
352
+ and not (scenario.setup_code or "").strip()
353
+ ):
354
+ # A fixture claiming data it never creates is the whole class of scenario that names a value
355
+ # the scenario reads as self-sufficient, the world has none of it, and the agent has nothing
356
+ # to answer with. Caught here because it is provable from the document alone.
357
+ problems.append(
358
+ f"fixture.origin is {scenario.fixture.get('origin')!r}, which claims this scenario "
359
+ "creates data, but setup_code is empty. Either seed everything the fixture names, or "
360
+ "declare origin 'seed' and use only records that already exist"
361
+ )
362
+
363
+ unknown = sorted(set(scenario.sub_goals) - catalogue.names())
364
+ if unknown:
365
+ problems.append(
366
+ f"sub_goals not in the catalogue: {', '.join(unknown)}. Use the shared names, or add "
367
+ f"them to the catalogue first. It has: {', '.join(sorted(catalogue.names())) or 'none'}"
368
+ )
369
+
370
+ # setup_code and ready_code are not read here. Whether they work is not a question reading
371
+ # them can answer, and running them is exactly what the first gate does.
372
+ if scenario.setup_code.strip() and "def setup(" not in scenario.setup_code:
373
+ problems.append("setup_code must define setup(world)")
374
+ if scenario.ready_code.strip() and "def ready(" not in scenario.ready_code:
375
+ problems.append("ready_code must define ready(world)")
376
+
377
+ if simulator_prompt:
378
+ unfilled = sorted(variables_in(simulator_prompt) - set(scenario.slots()))
379
+ if unfilled:
380
+ problems.append(
381
+ f"the simulator prompt asks for {', '.join(unfilled)}, which this scenario does "
382
+ "not supply. An unfilled slot reaches the caller verbatim"
383
+ )
384
+
385
+ if not scenario.solution and not allow_empty_solution:
386
+ problems.append(
387
+ "no solution: without the actions a correct agent would take, there is no way to "
388
+ "show this scenario can be passed at all"
389
+ )
390
+ problems.extend(fixture_problems(scenario))
391
+ problems.extend(answered_by_problems(scenario))
392
+ problems.extend(voicemail_style_problems(scenario))
393
+ problems.extend(voicemail_sub_goal_problems(scenario, catalogue))
394
+ problems.extend(_world_credential_problems(scenario, world_state))
395
+ problems.extend(self_sufficiency_problems(scenario))
396
+ problems.extend(alignment_problems(scenario, world_state))
397
+ problems.extend(hollow_scenario_problems(scenario))
398
+ problems.extend(naming_problems(scenario))
399
+ return problems
400
+
401
+
402
+ def answered_by_problems(scenario: Scenario) -> list[str]:
403
+ """Whether what answered could have: a mailbox only exists on a call the agent placed."""
404
+ chosen = str(scenario.answered_by or "").strip().lower()
405
+ if not chosen:
406
+ return []
407
+ if chosen not in set(ANSWERED_BY):
408
+ return [
409
+ "answered_by must be "
410
+ + ", ".join(ANSWERED_BY)
411
+ + f", not {scenario.answered_by!r}"
412
+ ]
413
+ if chosen != VOICEMAIL:
414
+ return []
415
+ if not voicemail_enabled():
416
+ return [
417
+ f"answered_by {VOICEMAIL!r} is turned off for this run, so write a scenario somebody "
418
+ "answers instead"
419
+ ]
420
+ if str(scenario.call_direction or "").strip().lower() != "outbound":
421
+ return [
422
+ "answered_by is 'voicemail', which only happens on a call the agent placed, so this "
423
+ "scenario must state call_direction 'outbound' itself. Left unset, the direction is "
424
+ "taken from the contract and a mailbox would be answering a call the person dialled"
425
+ ]
426
+ return []
427
+
428
+
429
+ def voicemail_sub_goal_problems(scenario: Scenario, catalogue: Catalogue) -> list[str]:
430
+ """Whether this mailbox scenario asks for something a mailbox call can produce.
431
+
432
+ A sub-goal needing a tool call cannot hold when nobody answers, so it would fail a correctly
433
+ handled mailbox. Read from the check rather than the name, since the check is what decides.
434
+ """
435
+ if str(scenario.answered_by or "").strip().lower() != VOICEMAIL:
436
+ return []
437
+ problems: list[str] = []
438
+ for name in scenario.sub_goals:
439
+ sub_goal = catalogue.named(name)
440
+ if sub_goal is None:
441
+ continue
442
+ check = " ".join(str(sub_goal.check or "").split())
443
+ if not check or "calls" not in check:
444
+ continue
445
+ # Any negative test over a filtered call list, which is the shape writers produce.
446
+ needs_a_call = (
447
+ "not any(" in check
448
+ or "not called" in check
449
+ or "calls == []" in check
450
+ or re.search(r"if not [A-Za-z_][A-Za-z0-9_]*\s*:", check) is not None
451
+ or re.search(r"len\([A-Za-z_][A-Za-z0-9_]*\) *== *0", check) is not None
452
+ )
453
+ if needs_a_call:
454
+ problems.append(
455
+ f"sub_goal {name!r} fails when a tool was not called, and on this scenario a mailbox "
456
+ "answers, so the agent never gets the turn that leads it to call anything. Ask for "
457
+ "what the agent can do with nobody on the line: that it recognised a machine, that "
458
+ "the message it left says who is calling and why, that it stopped instead of asking "
459
+ "questions. A sub-goal needing an answer marks a correctly handled mailbox as failed"
460
+ )
461
+ return problems
462
+
463
+
464
+ def voicemail_style_problems(scenario: Scenario) -> list[str]:
465
+ """Whether the named style exists, and whether a mailbox answered to play it at all."""
466
+ chosen = str(scenario.voicemail_style or "").strip().lower()
467
+ if not chosen:
468
+ return []
469
+ if chosen not in set(VOICEMAIL_STYLES):
470
+ return [
471
+ "voicemail_style must be "
472
+ + ", ".join(VOICEMAIL_STYLES)
473
+ + f", not {scenario.voicemail_style!r}"
474
+ ]
475
+ if str(scenario.answered_by or "").strip().lower() != VOICEMAIL:
476
+ return [
477
+ f"voicemail_style is {chosen!r} but answered_by is not 'voicemail', so no mailbox "
478
+ "answers and nothing plays it. State answered_by 'voicemail', or leave the style out"
479
+ ]
480
+ return []
481
+
482
+
483
+ def _world_credential_problems(
484
+ scenario: Scenario, world_state: dict[str, list[dict[str, Any]]]
485
+ ) -> list[str]:
486
+ """Reject caller credentials paired with the wrong world identity.
487
+
488
+ Hosted source authoring cannot execute a repository's runtime-owned tools until the sealed
489
+ bundle is provisioned. Static scenario validation must therefore catch identity-bound test
490
+ credentials that a deferred reference rehearsal cannot. The matching is deliberately
491
+ schema-shaped rather than application-shaped: any collection containing a phone-like
492
+ identity and an OTP/verification code participates.
493
+ """
494
+ credentials: dict[str, set[str]] = {}
495
+ for collection, rows in world_state.items():
496
+ if not any(token in collection.lower() for token in ("otp", "verification")):
497
+ continue
498
+ for row in rows:
499
+ phone = next(
500
+ (
501
+ str(value)
502
+ for key, value in row.items()
503
+ if "phone" in str(key).lower() and value not in (None, "")
504
+ ),
505
+ "",
506
+ )
507
+ code = next(
508
+ (
509
+ str(value).replace(" ", "")
510
+ for key, value in row.items()
511
+ if ("code" in str(key).lower() or "otp" in str(key).lower())
512
+ and re.fullmatch(r"\d{4,10}", str(value).replace(" ", ""))
513
+ ),
514
+ "",
515
+ )
516
+ if phone and code:
517
+ credentials.setdefault(phone, set()).add(code)
518
+
519
+ fixture = scenario.fixture or {}
520
+ phones: set[str] = set()
521
+ codes: set[str] = set()
522
+
523
+ def collect(value: Any, key: str = "") -> None:
524
+ if isinstance(value, dict):
525
+ for child, item in value.items():
526
+ collect(item, str(child))
527
+ elif isinstance(value, list):
528
+ for item in value:
529
+ collect(item, key)
530
+ elif "phone" in key.lower() and value not in (None, ""):
531
+ phones.add(str(value))
532
+ elif "otp" in key.lower() or key.lower() in {"code", "verification_code"}:
533
+ candidate = str(value).replace(" ", "")
534
+ if re.fullmatch(r"\d{4,10}", candidate):
535
+ codes.add(candidate)
536
+
537
+ collect(fixture)
538
+ if scenario.persona is not None:
539
+ collect(scenario.persona.metadata)
540
+ collect(scenario.persona.scripted_caller or {})
541
+ for step in scenario.solution:
542
+ if "verify" in step.tool.lower() or "otp" in step.tool.lower():
543
+ collect(step.arguments)
544
+
545
+ problems: list[str] = []
546
+ for phone in sorted(phones & credentials.keys()):
547
+ wrong = codes - credentials[phone]
548
+ if codes and wrong and not (codes & credentials[phone]):
549
+ problems.append(
550
+ "verification credential does not belong to the scenario caller "
551
+ f"{phone}; inspect the world and use that identity's code"
552
+ )
553
+ return problems
554
+
555
+
556
+ def contract_sequence_problems(
557
+ scenario: Scenario, hard_constraints: list[str]
558
+ ) -> list[str]:
559
+ """Catch reference solutions that hide required same-call state in a fixture.
560
+
561
+ A dependency can accept a pre-seeded identifier even when the public agent API cannot. For
562
+ a rule such as ``cancel_ride requires a booking_ref from this call``, require a producer
563
+ (``book_ride``) earlier in the same reference solution instead of allowing setup code or
564
+ environment-only arguments to make an impossible scenario look solvable.
565
+ """
566
+ problems: list[str] = []
567
+ names = [step.tool for step in scenario.solution]
568
+ pattern = re.compile(
569
+ r"\b(?P<consumer>[a-z][a-z0-9_]*)\b\s+requires\b.*?\b"
570
+ r"(?P<resource>[a-z][a-z0-9_]*(?:_id|_ref))\s+from this call\b",
571
+ re.IGNORECASE,
572
+ )
573
+ for constraint in hard_constraints:
574
+ found = pattern.search(constraint)
575
+ if found is None:
576
+ continue
577
+ consumer = found.group("consumer").lower()
578
+ lowered = [name.lower() for name in names]
579
+ if consumer not in lowered:
580
+ continue
581
+ resource = re.sub(r"_(?:id|ref)$", "", found.group("resource").lower())
582
+ stems = {resource, resource.removesuffix("ing")}
583
+ before = lowered[: lowered.index(consumer)]
584
+ produced = any(
585
+ any(stem and stem in tool for stem in stems)
586
+ and not tool.startswith(("get_", "list_", "find_", "cancel_"))
587
+ for tool in before
588
+ )
589
+ if not produced:
590
+ problems.append(
591
+ f"{consumer} requires {found.group('resource')} from this call, but the "
592
+ "reference solution does not create it first; do not hide it in setup or "
593
+ "environment_arguments"
594
+ )
595
+ return problems
596
+
597
+
598
+ _WEAK_CODES = {
599
+ "000000",
600
+ "111111",
601
+ "222222",
602
+ "333333",
603
+ "444444",
604
+ "555555",
605
+ "666666",
606
+ "777777",
607
+ "888888",
608
+ "999999",
609
+ "012345",
610
+ "123456",
611
+ "234567",
612
+ "345678",
613
+ "456789",
614
+ "987654",
615
+ "876543",
616
+ "765432",
617
+ "654321",
618
+ }
619
+
620
+
621
+ def _six_digit_values(scenario: Scenario) -> list[str]:
622
+ """Likely one-time codes declared by a scenario, without treating phone digits as OTPs."""
623
+ found: list[str] = []
624
+
625
+ def walk(value: Any, key: str = "") -> None:
626
+ if isinstance(value, dict):
627
+ for child, item in value.items():
628
+ walk(item, str(child))
629
+ elif isinstance(value, list):
630
+ for item in value:
631
+ walk(item, key)
632
+ elif "otp" in key.lower() or key.lower() in {"code", "verification_code"}:
633
+ found.extend(re.findall(r"(?<!\d)\d{6}(?!\d)", str(value)))
634
+
635
+ walk(scenario.fixture)
636
+ if scenario.persona:
637
+ walk(scenario.persona.metadata)
638
+ walk(scenario.persona.scripted_caller or {})
639
+ for step in scenario.solution:
640
+ walk(step.arguments)
641
+ walk(step.environment_arguments)
642
+ # Setup is code, so key-aware traversal is unavailable. Restrict matches to a nearby field
643
+ # name instead of collecting six digits from a phone number or an unrelated identifier.
644
+ found.extend(
645
+ match.group(1)
646
+ for match in re.finditer(
647
+ r"(?:otp|verification[_ ]?code|['\"]code['\"])[^\n]{0,80}?(?<!\d)(\d{6})(?!\d)",
648
+ scenario.setup_code,
649
+ flags=re.IGNORECASE,
650
+ )
651
+ )
652
+ return found
653
+
654
+
655
+ # A value the instruction hands the caller so they can say it back: a code, a reference, an account
656
+ # number, an id. Deliberately not named after any one domain, because the failure is the same
657
+ # whatever the agent does: the caller reads out something the agent then cannot find.
658
+ _QUOTED_VALUE = re.compile(
659
+ r"(?<![\w-])(?=[A-Za-z-]*\d)[A-Za-z0-9][A-Za-z0-9-]{3,}(?![\w-])"
660
+ )
661
+
662
+ # Values that look quotable but are never records the agent looks up.
663
+ _NOT_A_RECORD = re.compile(
664
+ r"^(?:\d{1,2}[:.]\d{2}|\d{1,4}(?:st|nd|rd|th)|20\d{2}|1?\d{1,2}[/-]\d{1,2}(?:[/-]\d{2,4})?)$",
665
+ re.IGNORECASE,
666
+ )
667
+
668
+
669
+ def _quotable_values(text: str) -> set[str]:
670
+ """Tokens in a piece of text that read as a value somebody would be asked to repeat."""
671
+ return {
672
+ token
673
+ for token in _QUOTED_VALUE.findall(text or "")
674
+ if not _NOT_A_RECORD.match(token)
675
+ }
676
+
677
+
678
+ # A value only has to be reachable if the caller is going to be asked for it. An address they are
679
+ # travelling to, or a price they are quoted, is the agent's to produce; a value they are told to say
680
+ # back is one the agent will check. Domain-neutral: the cue is the verb, not the kind of value.
681
+ _HANDED_OVER = re.compile(
682
+ r"(?:say|give|read|quote|provide|confirm|tell|repeat|use|enter|supply)\b[^.\n]{0,70}?"
683
+ r"(?<![\w-])((?=[A-Za-z-]*\d)[A-Za-z0-9][A-Za-z0-9-]{3,})(?![\w-])",
684
+ re.IGNORECASE,
685
+ )
686
+
687
+
688
+ def _handed_to_caller(text: str) -> set[str]:
689
+ """Values the instruction tells the caller to say back, which the agent will then check."""
690
+ return {
691
+ match.group(1)
692
+ for match in _HANDED_OVER.finditer(text or "")
693
+ if not _NOT_A_RECORD.match(match.group(1))
694
+ }
695
+
696
+
697
+ def naming_problems(scenario: Scenario) -> list[str]:
698
+ """Whether the name says what is tested, or only who the agent was dealing with.
699
+
700
+ The folder name is how a failure is read weeks later. A caller's name in it says the caller was
701
+ carrying the difference the test should have been carrying, which is the same mistake as planning
702
+ a second scenario because the person could be somebody else. Measured on an earlier suite: twelve
703
+ of thirty one were still named for the caller after the skill asked them not to be, which is why
704
+ this is checked rather than requested.
705
+ """
706
+ caller = str(getattr(scenario.persona, "name", "") or "").strip().lower()
707
+ if not caller:
708
+ return []
709
+ # Each part of the name, not the whole string: "marcus vance" is never a token of
710
+ # `refuse_expired_card_marcus`, so matching the full name lets every first-name suffix through.
711
+ parts = {part for part in caller.split() if len(part) > 2}
712
+ written = set(scenario.name.lower().replace("-", " ").replace("_", " ").split())
713
+ named_in = sorted(parts & written)
714
+ if not named_in:
715
+ return []
716
+ return [
717
+ f"the name contains the person's own name ({', '.join(named_in)}). Name it for the behaviour "
718
+ "under test, so a red result says which rule broke rather than who the agent was dealing "
719
+ "with, and so the suite sorts by what it covers rather than by who appeared in it"
720
+ ]
721
+
722
+
723
+ def hollow_scenario_problems(scenario: Scenario) -> list[str]:
724
+ """Whether the scenario tests reaching an outcome, or only the outcome itself.
725
+
726
+ A reference solution of one call, graded by one sub-goal naming that same call, is passed by an
727
+ agent that makes that call the moment it answers, having established nothing. Measured on a
728
+ suite of a hundred, eleven scenarios were a single `transfer_to_human` step graded by a single
729
+ `transferred_to_human` sub-goal, differing from each other only in the pretext, and every one of
730
+ them was passed by an agent that transfers every caller on arrival.
731
+
732
+ The bar is in the write skill and was not enough on its own, so it is checked here.
733
+ """
734
+ if len(scenario.solution) > 1:
735
+ return []
736
+ if not scenario.solution:
737
+ return []
738
+ return [
739
+ "the reference solution is a single call and there is nothing the agent has to establish "
740
+ "first, so an agent that makes that call on arrival passes without doing any of the work. "
741
+ "Either the solution shows how the outcome is reached, gathering what the decision depends "
742
+ "on before making it, or this is not a scenario"
743
+ ]
744
+
745
+
746
+ def alignment_problems(
747
+ scenario: Scenario, world_state: dict[str, list[dict[str, Any]]] | None = None
748
+ ) -> list[str]:
749
+ """Whether the values the caller is told are values the world actually holds.
750
+
751
+ The failure this exists for, seen across a whole suite: an instruction telling the caller a
752
+ verification code, a reference or an account number that the scenario never seeds and the world
753
+ never had. The call cannot succeed however well the agent behaves, and the result is reported as
754
+ a finding about the agent when it is a finding about the scenario.
755
+
756
+ Deliberately domain-neutral. A code, a booking reference, a policy number and an order id all
757
+ fail the same way, so the rule is about values rather than about any one kind of value: anything
758
+ the instruction hands the caller has to be somewhere the agent can reach, which means this
759
+ scenario's `setup_code` or the world it starts from. A fixture entry is not enough, because a
760
+ fixture describes what a scenario relies on and only `setup_code` changes what is there.
761
+ """
762
+ told = _handed_to_caller(scenario.instruction)
763
+ if not told:
764
+ return []
765
+ reachable = _quotable_values(scenario.setup_code or "")
766
+ for step in scenario.solution:
767
+ reachable |= _quotable_values(json.dumps(step.arguments, default=str))
768
+ reachable |= _quotable_values(
769
+ json.dumps(step.environment_arguments, default=str)
770
+ )
771
+ if world_state:
772
+ reachable |= _quotable_values(json.dumps(world_state, default=str)[:200000])
773
+ missing = sorted(told - reachable)
774
+ if not missing:
775
+ return []
776
+ return [
777
+ "the instruction gives the caller "
778
+ + ", ".join(missing)
779
+ + " to say back, and neither setup_code nor the world holds "
780
+ + ("them" if len(missing) > 1 else "it")
781
+ + ". Seed what the caller is told, or tell them what is seeded. Naming a value in fixture "
782
+ "only declares it: setup_code is what the world ends up holding"
783
+ ]
784
+
785
+
786
+ # What a setup does to the world, told apart by which call it makes. `put` adds a record and
787
+ # `call` drives a tool that produces one; `change` and `drop` only touch what was already there.
788
+ _CREATES_A_RECORD = re.compile(r"world\.(?:put|call)\s*\(")
789
+ _ONLY_TOUCHES_EXISTING = re.compile(r"world\.(?:change|drop)\s*\(")
790
+
791
+
792
+ def self_sufficiency_problems(scenario: Scenario) -> list[str]:
793
+ """Whether this scenario owns the records its outcome turns on, or borrows them.
794
+
795
+ A setup that only adjusts rows it did not create is building the test on state it does not
796
+ control: the row belongs to the frozen base, so a second scenario adjusting the same row is
797
+ testing the same record from two directions and neither describes a world it owns. Measured on
798
+ a fan-out suite of 86, sixty seven were one or two `world.change` calls against base rows, four
799
+ scenarios deep on the same rider, and the reused verification codes were the visible symptom of
800
+ it.
801
+
802
+ An empty setup stays legal. That is the documented case where the target's store is
803
+ process-local with no seam, so the scenario cannot alter it and says so by touching nothing.
804
+ """
805
+ body = (scenario.setup_code or "").strip()
806
+ if not body:
807
+ return []
808
+ if _CREATES_A_RECORD.search(body):
809
+ return []
810
+ if not _ONLY_TOUCHES_EXISTING.search(body):
811
+ return []
812
+ return [
813
+ "setup_code only adjusts records that were already there and creates none of its own, so "
814
+ "this scenario shares its data with every other scenario that touches the same records. "
815
+ "Create what the outcome turns on: its own person, its own record, its own code, with "
816
+ "world.put or by driving the agent's own tool. Shared reference data a whole world sits on "
817
+ "can be read as it is, but the thing being tested has to belong to this scenario"
818
+ ]
819
+
820
+
821
+ def _predictable(code: str) -> bool:
822
+ """Whether a one-time code is one nobody would be issued.
823
+
824
+ The hand-kept list of obvious ones caught `111111` and `123456` and let `000111` through, which then
825
+ turned up twice in a 200-scenario suite. Tested as a property instead: a code built from one or two
826
+ digits, or one that simply counts up or down, is a placeholder however it is arranged.
827
+ """
828
+ if not code.isdigit() or len(code) < 4:
829
+ return code in _WEAK_CODES
830
+ if len(set(code)) <= 2:
831
+ return True
832
+ steps = {ord(later) - ord(earlier) for earlier, later in zip(code, code[1:])}
833
+ if steps in ({1}, {-1}):
834
+ return True
835
+ return code in _WEAK_CODES
836
+
837
+
838
+ def fixture_problems(scenario: Scenario) -> list[str]:
839
+ """Reject demo-shaped data before a paid run makes it look like production traffic."""
840
+ problems: list[str] = []
841
+ codes = _six_digit_values(scenario)
842
+ weak = sorted({code for code in codes if _predictable(code)})
843
+ if weak:
844
+ problems.append(
845
+ "fixture uses predictable verification code(s): "
846
+ + ", ".join(weak)
847
+ + ". Generate a different non-sequential six-digit value for this scenario"
848
+ )
849
+ written = json.dumps(
850
+ {
851
+ "instruction": scenario.instruction,
852
+ "persona": scenario.persona.model_dump() if scenario.persona else {},
853
+ "fixture": scenario.fixture,
854
+ "setup": scenario.setup_code,
855
+ },
856
+ default=str,
857
+ ).lower()
858
+ clichés = [
859
+ value
860
+ for value in ("test user", "john doe", "jane doe", "123 main street")
861
+ if value in written
862
+ ]
863
+ if clichés:
864
+ problems.append("fixture contains placeholder demo data: " + ", ".join(clichés))
865
+ card_endings = sorted(
866
+ set(
867
+ re.findall(
868
+ r"(?:last4|card_last4|payment_last4)[^\n]{0,30}?[\"']?(0000|1111|1234|4242|4444)[\"']?",
869
+ written,
870
+ )
871
+ )
872
+ )
873
+ if card_endings:
874
+ problems.append(
875
+ "fixture uses placeholder payment-card ending(s): "
876
+ + ", ".join(card_endings)
877
+ )
878
+ spoken_card_endings = sorted(
879
+ set(
880
+ re.findall(
881
+ r"(?:ending(?:\s+in)?|last\s+four(?:\s+digits)?(?:\s+are)?)\D{0,12}"
882
+ r"(0000|1111|1234|4242|4444)",
883
+ written,
884
+ )
885
+ )
886
+ )
887
+ if spoken_card_endings:
888
+ problems.append(
889
+ "fixture/instruction uses placeholder payment-card ending(s): "
890
+ + ", ".join(spoken_card_endings)
891
+ )
892
+ demo_ids = sorted(
893
+ value
894
+ for value in ("ub12345678", "booking123", "booking_123", "test123")
895
+ if value in written
896
+ )
897
+ demo_ids.extend(
898
+ re.findall(r"\b(?:ub_[a-z]+_0*1|pay_[a-z]+(?:_[a-z]+)*0*1)\b", written)
899
+ )
900
+ demo_ids = sorted(set(demo_ids))
901
+ if demo_ids:
902
+ problems.append(
903
+ "fixture uses placeholder transaction identifier(s): " + ", ".join(demo_ids)
904
+ )
905
+ return problems
906
+
907
+
908
+ def rare_event_ceiling(suite_size: int) -> int:
909
+ """The most scenarios of this suite size that may carry a rare call condition, rounded up."""
910
+ return max(1, ceil(suite_size * RARE_CONDITION_SHARE))
911
+
912
+
913
+ def suite_diversity_problems(scenarios: list[Scenario]) -> list[str]:
914
+ """Whether a conversational suite represents meaningfully different people and data."""
915
+ if len(scenarios) < 4:
916
+ return []
917
+ problems: list[str] = []
918
+ personas = [one.persona for one in scenarios if one.persona]
919
+ names = [one.name.strip().lower() for one in personas if one and one.name.strip()]
920
+ unique_names = len(set(names))
921
+ required_names = min(len(scenarios), max(3, ceil(len(scenarios) * 0.9)))
922
+ if unique_names < required_names:
923
+ repeated = [name for name, count in Counter(names).items() if count > 2]
924
+ problems.append(
925
+ f"only {unique_names} distinct caller names across {len(scenarios)} scenarios; "
926
+ f"need at least {required_names}"
927
+ + (f". Overused: {', '.join(repeated)}" if repeated else "")
928
+ )
929
+ openings = [
930
+ one.initial_message.strip().lower()
931
+ for one in personas
932
+ if one and one.initial_message.strip()
933
+ ]
934
+ if len(set(openings)) != len(openings):
935
+ problems.append("caller opening messages repeat verbatim across scenarios")
936
+ locations = {
937
+ one.location.strip().lower() for one in personas if one and one.location.strip()
938
+ }
939
+ if len(scenarios) >= 8 and len(locations) < 3:
940
+ problems.append(
941
+ f"only {len(locations)} persona locations across {len(scenarios)} scenarios; need 3"
942
+ )
943
+ # An outbound suite that is all one awareness tests one opening repeatedly. Enforced rather than
944
+ # asked for: told to prefer `unaware`, writers made it the default and produced seven of eight,
945
+ # and told to cover more than one they had settled on `expecting` instead. Both leave two thirds
946
+ # of the opening untested.
947
+ outbound = [one for one in scenarios if one.call_direction == "outbound"]
948
+ if len(outbound) >= 3:
949
+ spread = Counter(one.caller_awareness or LEAST_AWARE for one in outbound)
950
+ if len(spread) < 2:
951
+ problems.append(
952
+ f"all {len(outbound)} outbound scenarios are caller_awareness "
953
+ f"{next(iter(spread))!r}; cover at least two of "
954
+ + ", ".join(CALLER_AWARENESS)
955
+ )
956
+ elif max(spread.values()) > ceil(len(outbound) * 0.7):
957
+ worst, count = spread.most_common(1)[0]
958
+ problems.append(
959
+ f"{count} of {len(outbound)} outbound scenarios are caller_awareness {worst!r}; "
960
+ "keep any one of "
961
+ + ", ".join(CALLER_AWARENESS)
962
+ + " under 70 percent of them"
963
+ )
964
+ if not spread.get(LEAST_AWARE):
965
+ problems.append(
966
+ f"no outbound scenario has caller_awareness {LEAST_AWARE!r}, the one that tests whether "
967
+ "the agent says who it is and why it called before asking for anything"
968
+ )
969
+ # A mailbox tests one narrow thing, so it is worth a few scenarios and never a theme.
970
+ mailboxes = [one for one in scenarios if one.answered_by == VOICEMAIL]
971
+ allowed = rare_event_ceiling(len(scenarios))
972
+ if len(mailboxes) > allowed:
973
+ problems.append(
974
+ f"{len(mailboxes)} of {len(scenarios)} scenarios are answered_by {VOICEMAIL!r}; keep "
975
+ f"them to at most {allowed} here"
976
+ )
977
+ # Two or more have to be different mailboxes. Three would need forty one scenarios at this share.
978
+ if len(mailboxes) >= 2:
979
+ greetings = {
980
+ " ".join(
981
+ (one.persona.initial_message if one.persona else "").lower().split()
982
+ )
983
+ for one in mailboxes
984
+ }
985
+ if len(greetings - {""}) < 2:
986
+ problems.append(
987
+ f"all {len(mailboxes)} voicemail scenarios use the same greeting; vary it, since a "
988
+ "named personal mailbox, a carrier mailbox with no name, a full mailbox and a long "
989
+ "greeting are four different tests of the agent"
990
+ )
991
+ # Style is the stronger axis: it also decides whether a tone follows the greeting.
992
+ styles = {
993
+ str(one.voicemail_style or DEFAULT_VOICEMAIL_STYLE).strip().lower()
994
+ for one in mailboxes
995
+ }
996
+ if len(styles) < 2:
997
+ problems.append(
998
+ f"all {len(mailboxes)} voicemail scenarios are voicemail_style "
999
+ f"{next(iter(styles))!r}; cover at least two of "
1000
+ + ", ".join(VOICEMAIL_STYLES)
1001
+ )
1002
+ # A code naturally appears several times inside one scenario (fixture, caller script,
1003
+ # reference verify call). Diversity is about reuse *between* callers, not repeated mention
1004
+ # of the same fact inside one test.
1005
+ codes = [
1006
+ code for scenario in scenarios for code in set(_six_digit_values(scenario))
1007
+ ]
1008
+ duplicated_codes = sorted(
1009
+ code for code, count in Counter(codes).items() if count > 1
1010
+ )
1011
+ if duplicated_codes:
1012
+ problems.append(
1013
+ "verification codes are reused across scenarios: "
1014
+ + ", ".join(duplicated_codes)
1015
+ )
1016
+ setups = [signature for one in scenarios if (signature := _setup_signature(one))]
1017
+ if len(set(setups)) != len(setups):
1018
+ problems.append("identical scenario setup data is reused more than once")
1019
+ return problems
1020
+
1021
+
1022
+ def _setup_signature(scenario: Scenario) -> str:
1023
+ """Comparable setup code, excluding the generated no-op function/documentation."""
1024
+ source = scenario.setup_code.strip()
1025
+ if not source:
1026
+ return ""
1027
+ try:
1028
+ tree = ast.parse(source)
1029
+ except SyntaxError:
1030
+ return " ".join(source.split())
1031
+ function = next(
1032
+ (node for node in tree.body if isinstance(node, ast.FunctionDef)), None
1033
+ )
1034
+ if function is None:
1035
+ return " ".join(source.split())
1036
+ meaningful = [
1037
+ node
1038
+ for node in function.body
1039
+ if not isinstance(node, ast.Pass)
1040
+ and not (
1041
+ isinstance(node, ast.Expr)
1042
+ and isinstance(node.value, ast.Constant)
1043
+ and isinstance(node.value.value, str)
1044
+ )
1045
+ ]
1046
+ return (
1047
+ "" if not meaningful else ast.dump(ast.Module(body=meaningful, type_ignores=[]))
1048
+ )