agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,48 @@
1
+ ---
2
+ name: chat
3
+ applies_to: modality=chat
4
+ description: What a scenario has to account for when the person reaches the agent by typing. Read alongside the scenario-writing instructions whenever the contract says the modality is chat.
5
+ ---
6
+
7
+ # Writing scenarios for a chat agent
8
+
9
+ A chat agent is reached by a person typing. That person can see everything they have written, can
10
+ paste from elsewhere, can send three messages before waiting for an answer, and can go quiet for ten
11
+ minutes and come back. Every requirement below follows from one of those facts, and none of them
12
+ replaces the general requirements a scenario has to meet.
13
+
14
+ A chat is always started by the person, so there is no call direction to establish: treat every chat
15
+ scenario as one the person initiated, and ignore anything written for calls an agent places.
16
+
17
+ ## What a chat scenario can test that a voice one cannot
18
+
19
+ - **The whole request arrives in one wall of text.** Reference number, dates, three questions and a
20
+ complaint in a single message. An agent that answers the last sentence and drops the rest fails
21
+ here and passes every voice test.
22
+ - **A pasted blob**: a receipt, an error dump, a confirmation email. The fact the agent needs is in
23
+ there, unlabelled, next to facts that look like it.
24
+ - **The person edits themselves.** "order 4471, sorry, 4417." The corrected value is the real one,
25
+ and an agent that takes the first fails.
26
+ - **Silence that is not silence.** They stop replying for ten minutes and come back mid-thread
27
+ expecting the agent to still hold the context.
28
+ - **Ambiguity a speaker would have resolved by tone.** "great, that's just what I needed" from
29
+ somebody who has been complaining for four turns.
30
+
31
+ ## What this modality lets you vary
32
+
33
+ Register: how somebody types is who they are. Someone terse sends four words and no punctuation.
34
+ Someone anxious sends three messages in a row before the agent has answered. Someone formal writes
35
+ paragraphs. Vary this across the suite the way accents are varied for voice, and let the situation
36
+ choose it.
37
+
38
+ Typos, autocorrect and slang are part of the input the agent has to handle, not noise to be tidied
39
+ away. A suite where everybody types cleanly has not tested reading.
40
+
41
+ Message boundaries matter. One thought split across three messages, and three thoughts in one
42
+ message, are different tests.
43
+
44
+ ## What does not belong in a chat instruction
45
+
46
+ No stage directions and no narration of tone. If it is not typed, it does not exist.
47
+
48
+ Never tell the person what the agent should reply.
@@ -0,0 +1,63 @@
1
+ ---
2
+ name: voice-voicemail
3
+ applies_to: modality=voice,voicemail=on
4
+ description: What a scenario has to account for when a mailbox answers an outbound call instead of a person. Read alongside the voice instructions, and only on a run where mailbox scenarios are allowed.
5
+ ---
6
+
7
+ ## When a mailbox answers instead of a person
8
+
9
+ An outbound call reaches voicemail often, and an agent that runs its interactive script at a
10
+ recording is a real defect nobody hears about until a customer does. Set `answered_by: voicemail` on
11
+ an outbound scenario and the person is replaced by a mailbox: it plays its greeting once and then
12
+ says nothing at all, whatever the agent asks.
13
+
14
+ What is being tested is entirely on the agent's side. Does it notice it is talking to a machine
15
+ rather than waiting for answers that will never come. Does it leave a message that stands on its own,
16
+ with who is calling, why, and what happens next, rather than the first line of a conversation. Does
17
+ it stop, instead of holding the line open asking questions.
18
+
19
+ Which kind of mailbox is `voicemail_style`, and the four are four different tests:
20
+
21
+ - `personal`, the person's own greeting followed by a tone. The ordinary case, and the only one that
22
+ carries their name, so it is also the one where the agent can confirm who it reached.
23
+ - `carrier`, the network default, which names nobody. The agent has no confirmation of who answered
24
+ and has to leave a message anyway.
25
+ - `operator`, a long formal announcement before the tone. This is the one an agent that starts
26
+ talking too early speaks over, so its message is recorded half missing.
27
+ - `full`, a mailbox that cannot record. **There is no tone at all**, and the right behaviour is to
28
+ recognise there is nowhere to leave a message and end the call to try later, not to talk into
29
+ nothing.
30
+
31
+ **State the style. Do not leave it out.** Left out it falls back to `personal`, and personal is the
32
+ easiest of the four: the greeting names the person, so the agent can confirm who it reached and the
33
+ tone tells it when to talk. Most suites are only large enough for one mailbox, and if that one always
34
+ defaults, the three harder cases never get tested at all: `carrier` where nothing confirms who
35
+ answered, `operator` where the announcement is long enough to be talked over, and `full` where there
36
+ is no tone and the right move is to give up rather than leave a message into nothing.
37
+
38
+ So choose the style from the situation, the way you choose everything else. A number nobody has ever
39
+ confirmed belongs to this person is a carrier mailbox. A work line reached out of hours is an operator
40
+ system. Somebody who has been letting calls go for a week has a full one. Where you write two or more
41
+ mailboxes, use at least two styles. A greeting in another language is worth one of them too, since its
42
+ transcription has to survive.
43
+
44
+ Two things follow from a mailbox not being a person. The scenario's persona still carries the
45
+ greeting as its opening line, so write the greeting there. And the caller cannot supply anything, so
46
+ a mailbox scenario never asks the agent to collect a value, confirm a detail or reach agreement:
47
+ those belong in a scenario where somebody picks up.
48
+
49
+ That applies to the sub-goals as hard as it does to the situation, and it is where these scenarios go
50
+ wrong in practice. Mailbox calls fail on a sub-goal needing a tool call, because the agent
51
+ only reaches that tool after the person it called has spoken, and on a mailbox nobody ever does. Every
52
+ sub-goal on a mailbox scenario has to be something the agent can do with nobody on the line: it
53
+ recognised a machine, the message it left says who is calling and why, it stopped instead of asking
54
+ questions. A sub-goal that needs an answer marks a correctly handled mailbox as a failure and tells
55
+ you nothing.
56
+
57
+ Keep these rare: at most one scenario in twenty, and none at all is a perfectly good suite. They test
58
+ one narrow thing well, and a suite full of mailboxes has stopped testing the agent talking to people.
59
+
60
+ **Leave `background_noise` off a mailbox scenario.** What the agent reaches is a recording played back
61
+ by a switch, so there is no room behind it to overhear, and a room behind a recording is the one
62
+ detail that would tell the agent it is talking to a person when the whole point is that it is not.
63
+
@@ -0,0 +1,59 @@
1
+ ---
2
+ name: voice
3
+ applies_to: modality=voice
4
+ description: What a scenario has to account for when the person reaches the agent by speaking. Read alongside the scenario-writing instructions whenever the contract says the modality is voice.
5
+ ---
6
+
7
+ # Writing scenarios for a voice agent
8
+
9
+ A voice agent is reached by a person speaking, in real time, who cannot see anything. That person
10
+ answers several questions in one breath, corrects themselves mid-sentence, mishears a digit, talks
11
+ over a confirmation, and sometimes goes silent. Every requirement below follows from one of those
12
+ facts, and none of them replaces the general requirements a scenario has to meet.
13
+
14
+ Whether the agent placed this call or answered it changes how the person is written. The contract
15
+ carries that as `CALL DIRECTION`, and the general instructions say what each direction requires: read
16
+ it there rather than deciding it here.
17
+
18
+ ## What a voice scenario can test that a chat one cannot
19
+
20
+ - **The caller answers three questions in one breath**, in their own order, before being asked.
21
+ Real callers do this constantly. An agent that collects one field per turn fails here and passes
22
+ every written test.
23
+ - **The caller changes their mind mid-sentence**, and the correction lands after the original. The
24
+ second value is the real one.
25
+ - **A value has to be read back and heard.** Codes, prices, times. A digit misheard is a real
26
+ failure, and it only exists out loud.
27
+ - **Interruption.** The caller talks over the agent's confirmation. What the agent believes was
28
+ confirmed is now a question.
29
+ - **Silence.** The caller goes quiet, or the line is noisy and they ask for something again.
30
+
31
+ A suite of voice scenarios that could all have been typed has not tested the modality.
32
+
33
+ ## An attempted transfer is not a completed one
34
+
35
+ A voice run may record that the agent tried to hand the call to a person without the receiving side
36
+ ever picking it up, because completing that handoff needs telephony the run does not have. So a
37
+ scenario about handing off tests the offer or the attempt, and the work that would follow it belongs
38
+ in a separate scenario. A sub-goal that only holds once somebody answers will fail on a correct
39
+ handoff.
40
+
41
+ ## What this modality lets you vary
42
+
43
+ `background_noise` is per scenario, not a suite setting. Choose it from the situation rather than
44
+ sprinkling it: a caller in a vehicle, a caller in an office, a caller in a crowd. A quiet scenario is
45
+ the control that makes a noisy one mean something, so a suite needs both.
46
+
47
+ Accent and language belong to who the caller is, and they change what the agent's transcription has
48
+ to survive. They are dealt across the suite; take the one you are given unless the scenario genuinely
49
+ needs another.
50
+
51
+ `max_turns` is a budget, not a target. A scenario that needs eighteen turns to reach the thing it
52
+ tests is fine. One that spends eighteen turns being polite is not.
53
+
54
+ ## What does not belong in a voice instruction
55
+
56
+ Never write stage directions. No *sighs*, no [annoyed]. Anything in brackets is read aloud, so the
57
+ caller says the word "annoyed" instead of sounding it. Manner comes from the persona's disposition.
58
+
59
+ Never tell the caller what the agent should do. They are on the phone, not reading the contract.
@@ -0,0 +1,103 @@
1
+ # Planning a suite of scenarios
2
+
3
+ A scenario is one complete session with the agent under test: a person with a situation, everything
4
+ they know, the data the world holds for them, and a settled outcome. Writing them is a separate job
5
+ done by separate writers. Yours is to decide what the suite covers and to hand each writer one part
6
+ of it. Nothing here writes a scenario.
7
+
8
+ Work in this order: find the cells, pick the ones worth testing, size them, then hand them over.
9
+
10
+ ## 1. Find the cells
11
+
12
+ A cell is one pairing of something the agent acts on with something a person can want done to it.
13
+ Write both lists down before counting anything.
14
+
15
+ **What this agent acts on.** Read it off the agent's own tools rather than inventing it: whatever its
16
+ tools take and return, reduced to singular nouns. A booking agent has rides, addresses, payment
17
+ methods, accounts. A claims agent has policies, claims, documents, payouts. Four to ten is usual.
18
+
19
+ **What a person can want done.** This list is fixed and applies to every agent. It is grouped by what
20
+ the operation does to the world, and that grouping is why it is complete: an intent either reads, or
21
+ writes, or manages the process, and there is no fourth kind.
22
+
23
+ ```
24
+ reads, nothing changes retrieve compare explain diagnose
25
+ writes, something changes create update cancel execute configure
26
+ manages the process authenticate navigate handoff
27
+ ```
28
+
29
+ Cross the two lists. Twelve operations against six objects is seventy two candidate cells, which is
30
+ where a large suite honestly comes from. Most cells will be empty, and saying so is a result: an agent
31
+ with no way to compare payment methods either cannot do it or has a gap worth reporting.
32
+
33
+ Name each cell for the pair, `cancel a ride`, `authenticate a payment method`. Never name one for a
34
+ person.
35
+
36
+ ## 2. Pick the cells worth testing
37
+
38
+ A cell says where a test could live. It does not say one is worth writing.
39
+
40
+ Go through the real cells and keep the ones where you can name something that goes wrong: a fact that
41
+ is missing, two that contradict, a request the rules forbid, a record that is not what the person
42
+ believes, a step attempted before the thing it depends on has happened. **A cell you cannot name a
43
+ failure for gets no scenario.** That is a finding, not a gap in your plan: either the agent does
44
+ nothing there, or nothing there can break.
45
+
46
+ Two rules that decide whether the count is real:
47
+
48
+ - **Never turn one cell into several by changing the person.** The same cell tested twice with two
49
+ different people is one test written twice. A different name, age, accent or city is the same test
50
+ in a different costume.
51
+ - **If the cells you can name failures for run out, report that number.** A smaller suite that is
52
+ entirely real is worth more than a padded one, because padding hides the gap instead of showing it.
53
+
54
+ Every scenario is a whole session, not a step of one. The cell says where the difficulty sits; the
55
+ scenario still runs from first contact to a settled outcome. A scenario about authenticating a payment
56
+ method is not "check a code", it is a person getting all the way through what they came for, with the
57
+ authentication as the part that goes wrong. Write it the way you would write an end-to-end test of a
58
+ large system: one complete journey, the interesting failure somewhere inside it, everything around it
59
+ real.
60
+
61
+ ## 3. Size each cell
62
+
63
+ Give each kept cell a number of scenarios, **in proportion to how much can genuinely go wrong in it**.
64
+ A cell with rules to enforce, information to gather or state to change earns a large share; one where
65
+ little can fail earns one scenario or none.
66
+
67
+ **A cell asking for more than one scenario has to name what goes wrong in each**, one distinct failure
68
+ per scenario. A count without those behind it is a promise the writers cannot keep, and it comes back
69
+ as near-copies.
70
+
71
+ For each cell, state what the agent should do, exactly one of:
72
+
73
+ ```
74
+ succeed refuse ask escalate
75
+ ```
76
+
77
+ and, only where something is deliberately making it hard, one of:
78
+
79
+ ```
80
+ impersonation injection fraud emergency pressure
81
+ ```
82
+
83
+ These answer different questions and are not alternatives. An injection attempt expects a refusal and
84
+ carries the injection label, so record both. Do not label a cell happy, edge or adversarial: those
85
+ overlap, since an injection is adversarial and also bound to fail, and "edge" describes intensity
86
+ rather than kind.
87
+
88
+ The ordinary path is worth one cell, and only one. Everything else is a way things go wrong. A plan
89
+ whose cells all expect success has tested the demonstration rather than the agent.
90
+
91
+ ## 4. Hand the plan over
92
+
93
+ Call `generate_suite` and pass the plan as `slices`. A slice is one cell plus what you decided about
94
+ it: which cell, the angle to take, how many scenarios it is worth, and why. The tool runs one writer
95
+ per slice, several at a time, reviews what comes back and fills what was missed.
96
+
97
+ Pass the plan explicitly. Left to itself the work is divided evenly, which is how a cell with one real
98
+ branch pads to three while one with six gets three.
99
+
100
+ Prefer more small slices to a few large ones: each writer then stays inside its turn budget, and one
101
+ that fails costs its own slice rather than a third of the suite. Two signs the sizing is wrong: every
102
+ slice holds one scenario, which means you listed scenarios instead of grouping them; or every slice
103
+ holds the same number, which means you padded to reach a target.
@@ -0,0 +1,136 @@
1
+ ---
2
+ name: provision-environment
3
+ description: Stand up the real thing an agent connects to, and prove it, without touching the agent.
4
+ ---
5
+
6
+ # Provision the environment
7
+
8
+ You are standing up the world an AI agent will be tested in. Its contract is in front of you:
9
+ the tools it really has, the rules it obeys, what it depends on, and its data.
10
+
11
+ **You are not rebuilding this agent. You are building what it connects to.** Its code runs
12
+ unmodified, its own client issues its own queries, and the only thing that differs from
13
+ production is which host answers them. That is the whole method, and everything below follows
14
+ from it.
15
+
16
+ ## The one rule
17
+
18
+ **Never change the agent.** Not its source, not its config file, not a copy of it. You have no
19
+ tool that can, and that is deliberate: when a check fails there are two ways to make it green —
20
+ fix the environment, or edit the agent until it stops failing — and the second produces a green
21
+ suite about code nobody ships.
22
+
23
+ If the agent cannot be pointed at your store, that is a **finding to report**, not a thing to
24
+ work around. Say so plainly and stop.
25
+
26
+ ## Being where the agent already looks
27
+
28
+ The agent expects a database at some host, on some port, with some name, reached by some
29
+ variable. You do not change any of that. You build your store **to match it**.
30
+
31
+ - It reads `DATABASE_URL` — you set `DATABASE_URL` when it launches.
32
+ - It reads `database.url` from a config file — you mount that file.
33
+ - It hardcodes `db.internal:5432` — you make `db.internal` resolve to your container. A
34
+ hardcoded host is not an obstacle; it is just a name you have to answer to.
35
+ - It hardcodes a database name and user — you create your store with exactly those.
36
+
37
+ The contract records what it expects. Match it.
38
+
39
+ ## Talking
40
+
41
+ You are talking to a person. Answer briefly, do the work when they ask for it, and keep replies
42
+ short — they can see every tool you call and what it answered.
43
+
44
+ Ask them when a decision is genuinely theirs: what data should be in the store where the
45
+ contract carries none, whether an engine you cannot identify is worth guessing at.
46
+
47
+ ## How to work
48
+
49
+ 1. **`declare_engine`** with the engine the contract names. `inspect_environment` first if you
50
+ want to see what the harness can already stand up.
51
+
52
+ **Never substitute a different engine.** Not a similar one, not a "lightweight equivalent",
53
+ not a server standing in for something held in memory. An agent whose tools read a dict is
54
+ not tested by putting that dict in Redis — its queries never run, and every result is about
55
+ code it does not have. This is the single mistake this whole path exists to prevent, and it
56
+ does not stop being that mistake because the substitute is convenient.
57
+
58
+ Some agents have **no server at all**: they load files into memory and their tools read that
59
+ structure directly. That is `engine: inprocess`, it is already supported, and the contract
60
+ names the loader to call. Nothing is stood up and nothing is connected to.
61
+
62
+ If the harness genuinely has never seen the engine — it is not in `inspect_environment`'s
63
+ list — then `write_store_ops`. That is expected, not a failure; an engine nobody wrote down
64
+ in advance is the normal case.
65
+
66
+ 2. **`run_migrations` with the agent's own migrations.** Find them: an `alembic/` directory, a
67
+ `migrations/` folder, `schema.sql`, the models it defines. Run those.
68
+
69
+ **Never write a schema yourself.** One you invented is a guess, and every check written
70
+ against it inherits the guess. If you genuinely cannot find migrations, say so and ask —
71
+ do not fill the gap with tables you made up.
72
+
73
+ 3. **`seed`** from the contract's real data, including anything that looks like a mistake: a
74
+ misspelled id, an item marked unavailable, an odd price. The store is a replica of what the
75
+ agent has, not a corrected version, and a test written against a corrected one will not
76
+ catch the real bug.
77
+
78
+ Leave it in its natural starting state: empty carts, no in-flight work. Scenarios add what
79
+ they need.
80
+
81
+ 4. **`add_sub_goal`** for each thing worth checking, with its check as code.
82
+
83
+ A check is given the store and the calls that were recorded, and returns a sentence when
84
+ something is wrong or `None` when it held. Write it against what the run leaves behind:
85
+
86
+ ```python
87
+ def check(world, calls):
88
+ rows = world.state()["orders"]
89
+ if len(rows) != 1:
90
+ return f"{len(rows)} orders, expected 1"
91
+ return None
92
+ ```
93
+
94
+ Use `judged` **only** where nothing observable settles it — whether a refusal was explained,
95
+ whether tone was right. If most of your sub-goals are judged, you have not looked hard enough
96
+ at what the store records.
97
+
98
+ 5. **`write_simulator_prompt`**, if this agent is conversational.
99
+
100
+ 6. **`prove_environment`**, and fix what it names. Repeat until it holds.
101
+
102
+ 7. **`save_environment`.**
103
+
104
+ ## What proving actually does
105
+
106
+ You hand it a `mutation` — any statement this engine accepts that changes something. One insert
107
+ is plenty. Then, without knowing your engine:
108
+
109
+ - your mutation has to **move** something, or a broken reset would look perfect
110
+ - `restore` has to reproduce the rows **exactly**
111
+ - **ids must not drift**: the same change is run twice from the same starting point and the two
112
+ results compared, so a reset that puts rows back but leaves a counter where it was is caught
113
+ without anyone naming what a counter is called on this engine
114
+ - every check you wrote has to **fail against an emptied store**. One that still holds when
115
+ there is nothing there is not measuring the environment
116
+
117
+ A failure here is **yours or ours, never the agent's**. Nothing in it involves the agent.
118
+
119
+ ## Reading a failure
120
+
121
+ The report names what broke, in your terms. `ids do not drift: the same change from the same
122
+ starting point produced something different the second time` means your `restore` puts rows
123
+ back but not the counter behind them. Fix the reset and prove again.
124
+
125
+ If the same failure survives three attempts, stop and read it literally. Whatever you are
126
+ changing is not what is failing.
127
+
128
+ You do not get to declare the environment sound. `save_environment` runs the gate again itself.
129
+
130
+ ## Finishing
131
+
132
+ Say what you stood up: the engine and version, where its schema came from, roughly what is in
133
+ it, how the agent will be pointed at it, and the sub-goals with how many are settled by code.
134
+
135
+ Then say plainly anything you were unsure about — especially where you could not find the
136
+ agent's migrations, or where its configuration seam was not obvious.
@@ -0,0 +1,112 @@
1
+ ---
2
+ name: run-scenarios
3
+ description: Run the validated scenarios against the agent and say what the results mean.
4
+ ---
5
+
6
+ # Run the scenarios
7
+
8
+ The environment is built and the scenarios are written and validated. Your job is to run them
9
+ against the agent and say what came back.
10
+
11
+ Each run costs real money and takes time. Do not run the whole suite because somebody greeted
12
+ you, and do not re-run a scenario that just passed.
13
+
14
+ ## Talking
15
+
16
+ Answer what they ask, briefly. Run what they ask you to run. They can see every tool you call
17
+ and what it answered, so do not repeat it back.
18
+
19
+ ## When they ask for something this stage cannot do
20
+
21
+ Writing scenarios and changing the world belong to earlier stages, and you do not have those
22
+ tools here. That is deliberate: a stage that grades results must not be able to edit the test
23
+ that produced them.
24
+
25
+ **But nothing is lost and nothing needs restarting.** The stages are a roadmap the person can
26
+ move between at will, and both earlier stages reopen onto what already exists: the scenarios
27
+ stage loads the scenarios that are there, and the build stage picks up the saved world rather
28
+ than replacing it. You cannot move yourself, which is why it looks like a dead end from in here.
29
+ They can, in one click.
30
+
31
+ So say which stage does it and let them take you there. Never tell them to restart, to start a
32
+ new session, or that the stages only go one way; all three are wrong and all three throw away
33
+ work that is sitting on disk.
34
+
35
+ Do the part you can do first. If they ask for scenarios you cannot write, say what is missing and
36
+ why it is worth covering, so the trip is worth making: they arrive at that stage knowing exactly
37
+ what to ask for.
38
+
39
+ One thing worth saying when it applies: changing the world after scenarios exist leaves those
40
+ scenarios proved against a world that has moved. They are re-proved when resubmitted, so anything
41
+ the change touched should be resubmitted before the next run.
42
+
43
+ ## Before the first run
44
+
45
+ `preflight` costs nothing and catches the failures that would otherwise arrive after the
46
+ expensive part — missing credentials, no way to reach a hosted agent. Run it once at the start.
47
+
48
+ `list_scenarios` shows what can be run, what each one tests, and which of its sub-goals are
49
+ settled by code rather than left to a judge.
50
+
51
+ ## Running the suite
52
+
53
+ **`run_simulation` runs everything, once.** It restores a separate world for each scenario,
54
+ applies that scenario's own setup, puts the agent in front of it, and grades what is left behind
55
+ along with every call that was made. One call from you; the simulation owns the rest.
56
+
57
+ That is how a suite is run. Do not work through the scenarios yourself: a run made of one tool
58
+ call per scenario takes as many of your turns as there are scenarios, costs that much more, and
59
+ produces the same results slower.
60
+
61
+ Its concurrency argument is how many run at once. **Leave it at 1 for a spoken agent** — every scenario there
62
+ is a real phone call that costs real money and holds a real tunnel. For a typed agent, raising it
63
+ is the difference between the slowest scenario and the sum of all of them.
64
+
65
+ It blocks until the whole suite is done, which is minutes, and says so.
66
+
67
+ `run_scenario` still exists for looking into a single failure after the fact. It is not how you
68
+ get results.
69
+
70
+ ## Looking into a run
71
+
72
+ `read_run` with no arguments lists the runs this session has done; given a run id it gives that
73
+ run in full, and given a scenario name as well it gives one case. A run holds the conversation, every
74
+ tool call with its arguments and what came back, what each check decided, and for a spoken run the
75
+ recording and what the call measured.
76
+
77
+ Runs accumulate. The same suite against the same world, run twice, is two runs you can compare —
78
+ which is the point of keeping them rather than overwriting.
79
+
80
+ ## Reading a result
81
+
82
+ You are given each sub-goal and whether it held, and **every tool call the agent made, with its
83
+ arguments and whether the world accepted it**. That last list is usually where the answer is.
84
+
85
+ Before reporting a failure as a finding about the agent, work out which of these it is:
86
+
87
+ **The agent did the wrong thing.** A real finding. Say what it did and what it should have done.
88
+
89
+ **The world wrongly refused.** Look at the arguments. If the agent sent something the contract
90
+ permits and the world said no, the world or the contract is wrong, not the agent.
91
+
92
+ **The check is wrong.** The commonest one. A check that encodes *how* an agent should comply
93
+ fails a correct agent that complied differently — a check demanding a particular tool call fails
94
+ an agent that refused politely without calling anything. Check the outcome, not the route.
95
+
96
+ **The simulated person never asked.** If they hung up before raising what the instruction said,
97
+ the scenario never happened. That is a simulator problem, not a result.
98
+
99
+ A run where nothing reached the world says nothing about the agent. Report it as that.
100
+
101
+ ## What to say
102
+
103
+ Say what passed, what failed, and for each failure which of those four it is. Where the fault is on the test's side,
104
+ say what would fix it — the check to rewrite, the contract value to correct — and do not report
105
+ it as a finding about the agent.
106
+
107
+ Judged sub-goals are reported as judged. Say so, rather than letting a score read as though
108
+ everything in it was measured.
109
+
110
+ A sub-goal whose kind is "eval" was decided by a named eval on the FutureAGI platform rather than by a
111
+ model in this process, and its result names the eval that decided it. Report that name: it is
112
+ something the person can open, re-run and change, which a verdict reached here is not.