agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,169 @@
1
+ """Stage one: read an agent and produce its contract.
2
+
3
+ The stage is the same whatever the agent is. What changes between a repository, a provider
4
+ connection and a pasted definition is where the truth lives, and that comes from the source.
5
+
6
+ It stays open after the first answer, because a contract is usually right on the second look and
7
+ not the first. Correcting it is the next thing said, not a re-run.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import os
14
+ from pathlib import Path
15
+ from typing import Any, Callable
16
+
17
+ from .config import artifact_dir, load_skill, read_only_session
18
+ from .contract import AgentContract
19
+ from .session import Stage
20
+ from .sources import AgentSource
21
+ from .tools import CONTRACT_SERVER, contract_tools
22
+
23
+ SKILL = "understand-agent"
24
+ PROVIDER_IMPORT_PROFILE_PATH_ENV = "ALK_PROVIDER_IMPORT_PROFILE_PATH"
25
+
26
+
27
+ def _provider_import_briefing() -> str:
28
+ """Load the sanitized external-provider definition prepared by the control process."""
29
+ configured = os.environ.get(PROVIDER_IMPORT_PROFILE_PATH_ENV, "").strip()
30
+ if not configured:
31
+ return ""
32
+ path = Path(configured)
33
+ try:
34
+ profile = json.loads(path.read_text(encoding="utf-8"))
35
+ except (OSError, ValueError):
36
+ return ""
37
+ return (
38
+ "\n\n## Imported provider target (authoritative, read-only)\n\n"
39
+ "The submitted repository implements this target's environment-facing webhooks. "
40
+ "The externally hosted assistant definition below is equally authoritative for its "
41
+ "conversation, prompt, model, voice, and tool schemas. Reconcile both sources. Do not "
42
+ "invent behavior or tool inputs that conflict with the provider definition.\n\n"
43
+ f"```json\n{json.dumps(profile, indent=2, sort_keys=True)}\n```"
44
+ )
45
+
46
+
47
+ def _eval_catalogue_briefing(available_evals: list[dict[str, Any]] | None) -> str:
48
+ """The eval catalogue grouped by modality, since the contract has not claimed one yet."""
49
+ grouped: dict[str, list[str]] = {}
50
+ for one in available_evals or []:
51
+ if not isinstance(one, dict):
52
+ continue
53
+ name = str(one.get("name") or "").strip()
54
+ if not name:
55
+ continue
56
+ keys = ", ".join(str(key) for key in one.get("required_keys") or [])
57
+ grouped.setdefault(
58
+ str(one.get("modality") or "any").strip().lower(), []
59
+ ).append(
60
+ "- {name}: {description}{keys}".format(
61
+ name=name,
62
+ description=str(one.get("description") or "").strip()[:240],
63
+ keys=f" [needs: {keys}]" if keys else "",
64
+ )
65
+ )
66
+ if not grouped:
67
+ return ""
68
+ sections = [
69
+ f"### Applies to {kind}\n\n" + "\n".join(sorted(lines))
70
+ for kind, lines in sorted(grouped.items())
71
+ ]
72
+ return (
73
+ "\n\n## Evals this platform can run on the finished calls\n\n"
74
+ "Record the ones this agent should be judged by in `chosen_evals`, by exact name. Each one "
75
+ "runs as a judge on every call of every scenario, so choose the few that tell you "
76
+ "something the scenarios' own deterministic checks cannot, and choose none rather than "
77
+ "padding.\n\n"
78
+ "Choose only from the section matching the `modality` you are about to record. The "
79
+ "platform refuses an eval belonging to another modality, so an eval for spoken calls "
80
+ "recorded against a chat agent costs the whole submission.\n\n"
81
+ "Two ways to choose wrongly, both common. An eval whose subject only exists in speech, "
82
+ "dead air, voicemail detection, voicemail handling, is meaningless for a chat agent and "
83
+ "worth having for a voice one. An eval named for a domain, misselling, advice authority, "
84
+ "lead qualification, claim intake, intake field accuracy, is only worth choosing when this "
85
+ "agent's own tools and constraints show it doing that work; choose it from the evidence in "
86
+ "front of you, never because its name sounds close to the agent's industry.\n\n"
87
+ + "\n\n".join(sections)
88
+ )
89
+
90
+
91
+ def open_stage(
92
+ source: AgentSource,
93
+ *,
94
+ out: Path | None = None,
95
+ ask: Callable[..., Any] | None = None,
96
+ max_turns: int = 70,
97
+ available_evals: list[dict[str, Any]] | None = None,
98
+ ) -> tuple[Stage, Path]:
99
+ """A live understand-the-agent stage, and where it will write."""
100
+ destination = out or artifact_dir(source.name)
101
+ spec = read_only_session(
102
+ system_prompt=(
103
+ f"{load_skill(SKILL)}\n\n## This agent\n\n{source.briefing()}"
104
+ # A source-free connect-only target already carries this definition in its
105
+ # ProviderSource briefing. Repository-backed provider imports still need the
106
+ # separately injected profile so the model can reconcile both sources of truth.
107
+ f"{_provider_import_briefing() if source.kind != 'provider' else ''}"
108
+ f"{_eval_catalogue_briefing(available_evals)}"
109
+ ),
110
+ cwd=source.workdir(),
111
+ servers={
112
+ **source.servers(),
113
+ CONTRACT_SERVER: contract_tools(
114
+ destination,
115
+ available_evals,
116
+ source_root=source.workdir() if source.kind == "repo" else None,
117
+ ),
118
+ },
119
+ extra_builtins=source.builtin_tools(),
120
+ max_turns=max_turns,
121
+ )
122
+ if ask is not None:
123
+ spec.permission_override = ask
124
+ return Stage(spec, name=SKILL), destination
125
+
126
+
127
+ def opening(source: AgentSource) -> str:
128
+ # The name is only a label for the artifact folder, and saying so matters: told to "read
129
+ # the agent named verify_fix", a model went hunting the whole workspace for something
130
+ # called verify_fix instead of reading the path it was given.
131
+ return (
132
+ "Read this agent and produce its contract. Where it lives is in your briefing; "
133
+ f"{source.name!r} is only the label its artifacts are filed under, not something to "
134
+ "search for.\n\n"
135
+ "Work through the tools, their exact argument names and types, the constrained argument "
136
+ "values, the rules it enforces, and its data. Ask me if the source genuinely does not "
137
+ "settle something that changes what gets built. Call submit_contract when you are done."
138
+ )
139
+
140
+
141
+ def load(destination: Path) -> AgentContract | None:
142
+ """The contract on disk, if the stage produced one."""
143
+ path = Path(destination) / "contract.json"
144
+ if not path.exists():
145
+ return None
146
+ return AgentContract.model_validate(json.loads(path.read_text(encoding="utf-8")))
147
+
148
+
149
+ async def understand(
150
+ source: AgentSource,
151
+ *,
152
+ out: Path | None = None,
153
+ follow_ups: list[str] | None = None,
154
+ on_event: Callable[..., Any] | None = None,
155
+ ask: Callable[..., Any] | None = None,
156
+ max_turns: int = 70,
157
+ ) -> AgentContract | None:
158
+ """Run the stage start to finish and return the contract.
159
+
160
+ ``follow_ups`` are corrections applied in the same session, the scripted equivalent of an
161
+ operator typing them. ``ask`` handles clarifying questions; without it the model records what
162
+ it could not resolve in ``open_questions`` instead of blocking.
163
+ """
164
+ stage, destination = open_stage(source, out=out, ask=ask, max_turns=max_turns)
165
+ async with stage:
166
+ await stage.say(opening(source), on_event=on_event)
167
+ for follow_up in follow_ups or []:
168
+ await stage.say(follow_up, on_event=on_event)
169
+ return load(destination)
@@ -0,0 +1,74 @@
1
+ """Which recorded mailbox greeting, if any, a voicemail scenario is heard through.
2
+
3
+ Clips must greet without naming anybody, so one recording fits any persona. The catalogue is a
4
+ local file and is not committed: absent, the mailbox speaks its greeting and the tone is generated.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import os
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ CATALOG = Path(__file__).with_name("data") / "voicemail" / "catalog.json"
15
+ # A run may point somewhere else, so a deployment can serve these from object storage.
16
+ CATALOG_ENV = "ALK_VOICEMAIL_CATALOG"
17
+
18
+
19
+ def _entries() -> list[dict[str, Any]]:
20
+ path = Path(os.environ.get(CATALOG_ENV, "").strip() or CATALOG)
21
+ try:
22
+ body = json.loads(path.read_text(encoding="utf-8"))
23
+ except Exception: # noqa: BLE001 - no catalogue is the ordinary case, never an error
24
+ return []
25
+ return [one for one in body if isinstance(one, dict)] if isinstance(body, list) else []
26
+
27
+
28
+ def _resolved(entry: dict[str, Any]) -> str:
29
+ """Where the audio actually is: a URL if the catalogue serves one, else a file beside us."""
30
+ url = str(entry.get("url") or "").strip()
31
+ if url:
32
+ return url
33
+ raw = str(entry.get("path") or "").strip()
34
+ if not raw:
35
+ return ""
36
+ here = Path(raw)
37
+ if here.is_file():
38
+ return str(here)
39
+ # Paths are relative to the repository root, so fall back to beside this module.
40
+ beside = CATALOG.parent / Path(str(entry.get("file_name") or here.name))
41
+ return str(beside) if beside.is_file() else ""
42
+
43
+
44
+ def clip_for(style: str, language: str = "") -> dict[str, Any] | None:
45
+ """A clip whose style and language match, or None where the catalogue has nothing to offer.
46
+
47
+ Language matters as much as style: a scenario whose mailbox greets in Hindi cannot be served an
48
+ English recording, and no clip is the right answer there because the session speaks the greeting
49
+ itself in the language the scenario asked for.
50
+
51
+ Deterministic rather than random: the first matching entry wins, so two runs of the same suite
52
+ are heard through the same mailbox and a difference between them is never the audio.
53
+ """
54
+ wanted = str(style or "").strip().lower()
55
+ if not wanted:
56
+ return None
57
+ spoken = (str(language or "").strip().lower() or "en").split("-")[0]
58
+ for entry in _entries():
59
+ if str(entry.get("style") or "").strip().lower() != wanted:
60
+ continue
61
+ if str(entry.get("language") or "en").strip().lower().split("-")[0] != spoken:
62
+ continue
63
+ source = _resolved(entry)
64
+ if not source:
65
+ continue
66
+ return {
67
+ "source": source,
68
+ # A clip that ends with its own tone must not be given a second one.
69
+ "has_tone": bool(entry.get("has_tone")),
70
+ "id": str(entry.get("id") or ""),
71
+ # The greeting must reach the transcript, or the call reads as the agent talking to nobody.
72
+ "transcript": str(entry.get("transcript") or "").strip(),
73
+ }
74
+ return None
@@ -0,0 +1,33 @@
1
+ """Generated worlds: a real data store behind an agent's tools.
2
+
3
+ The pieces here are the parts that must be exact, so that what gets generated per agent stays
4
+ small: the runtime a world executes on, the snapshot every scenario restores from, and the probe
5
+ suite that decides whether a world is usable at all.
6
+ """
7
+
8
+ from .kinds import WorldKind, register_kind, supported as supported_kinds
9
+ from .probe import EDGE, HAPPY, SEQUENCE, ProbeReport, ProbeResult, dirty_state, probe
10
+ from .runtime import Call, Db, GeneratedWorld, ToolError, WorldSpec
11
+ from .snapshot import apply_overlay, read_manifest, restore, save
12
+
13
+ __all__ = [
14
+ "Call",
15
+ "Db",
16
+ "EDGE",
17
+ "GeneratedWorld",
18
+ "HAPPY",
19
+ "ProbeReport",
20
+ "ProbeResult",
21
+ "WorldKind",
22
+ "dirty_state",
23
+ "register_kind",
24
+ "supported_kinds",
25
+ "SEQUENCE",
26
+ "ToolError",
27
+ "WorldSpec",
28
+ "apply_overlay",
29
+ "probe",
30
+ "read_manifest",
31
+ "restore",
32
+ "save",
33
+ ]
@@ -0,0 +1,68 @@
1
+ """What the hosted world handle raises when scenario code asks it for something it will not do.
2
+
3
+ Every one of these is scenario code at fault, never the world's contents and never the agent
4
+ under test: a `KeyError` from a missing table reads as a finding about data, and a bare
5
+ `StoreError` from the postgres driver reads as an infrastructure fault. Neither is right for
6
+ "you called `change` without saying which column `key` names," so the handle has its own
7
+ vocabulary, and folds it under one base class for whoever has to route "scenario code misused
8
+ the handle" to one outcome without naming all six.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+
14
+ class WorldError(RuntimeError):
15
+ """Scenario code asked the world handle for something it will not do."""
16
+
17
+
18
+ class WorldReadOnly(WorldError):
19
+ """`put`, `change`, `drop` or `call` reached the handle `ready()` or `check()` were given.
20
+
21
+ Those two only ever observe a run. A check that could write would be able to change the very
22
+ thing it is grading, and nothing downstream could tell the difference between a check that
23
+ found a problem and one that quietly fixed it.
24
+ """
25
+
26
+
27
+ class WorldReservedName(WorldError):
28
+ """Scenario code named the harness's own conformance canary.
29
+
30
+ That table exists to prove worlds are really isolated from each other, not to hold scenario
31
+ data, and it never appears in `state()` either.
32
+ """
33
+
34
+
35
+ class WorldQueryRejected(WorldError):
36
+ """`query()` was handed something that is not one plain read.
37
+
38
+ The database's own read-only transaction is what actually stops a write; this is the
39
+ friendlier rejection in front of it, so a statement that was never going to be allowed fails
40
+ on a message naming the reason rather than a lock error three layers down.
41
+ """
42
+
43
+
44
+ class WorldStateTooLarge(WorldError):
45
+ """`state()` reached a table whose row count, measured when the baseline was frozen, passed
46
+ the cap.
47
+
48
+ Measured once, at freeze time, by the provisioner — never recomputed here, so which tables
49
+ raise is fixed before a scenario ever runs and nothing a call does during one can move it.
50
+ """
51
+
52
+
53
+ class WorldUnavailable(WorldError):
54
+ """The handle cannot do this, given how the world in front of it is built — not what it holds.
55
+
56
+ A postgres world whose `public` schema has no tables, and `call()` — which raises
57
+ unconditionally until the `http_tool` shim's wire format is pinned somewhere in the contracts
58
+ — are both this: nothing went wrong, the capability was never there.
59
+ """
60
+
61
+
62
+ class WorldUsageError(WorldError):
63
+ """A `put`, `change` or `drop` cannot be carried out as asked.
64
+
65
+ Inserting into something that is not a table, or changing or dropping a record without
66
+ saying which column `key` names — hosted worlds cannot invent tables or guess a column, so
67
+ both are reported here rather than attempted.
68
+ """
@@ -0,0 +1,91 @@
1
+ """What a world is expected to look like afterwards, and whether it does.
2
+
3
+ Written once and used twice. The build stage declares a sequence and asserts the state it leaves
4
+ behind; a scenario declares the state a conversation should leave behind. Those are the same
5
+ question asked at two different scales, and if each had its own implementation they would drift
6
+ until a check that passes the gate fails the run for reasons that have nothing to do with the
7
+ agent.
8
+
9
+ The shape is ``{"table.count": 3, "table.column": "value"}``: how many records there are, and
10
+ whether a particular value is among them.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from typing import Any, Mapping
16
+
17
+ COUNT = "count"
18
+
19
+
20
+ def check_state(
21
+ state: Mapping[str, list[dict[str, Any]]], expected: Mapping[str, Any]
22
+ ) -> list[str]:
23
+ """Every expectation that does not hold, said in terms of what was found instead."""
24
+ failures: list[str] = []
25
+ for path, want in (expected or {}).items():
26
+ table, _, column = str(path).partition(".")
27
+ if table not in state:
28
+ failures.append(
29
+ f"{path}: no {table} in this world; it has "
30
+ f"{', '.join(sorted(state)) or 'nothing'}"
31
+ )
32
+ continue
33
+ rows = state[table]
34
+ if column in ("", COUNT):
35
+ if len(rows) != want:
36
+ failures.append(f"{path}: {len(rows)} rows, expected {want}")
37
+ continue
38
+ if rows and column not in rows[0]:
39
+ failures.append(
40
+ f"{path}: {table} has no {column}; its columns are "
41
+ f"{', '.join(sorted(rows[0]))}"
42
+ )
43
+ continue
44
+ present = {str(row.get(column)) for row in rows}
45
+ # A list means every one of these has to be somewhere, which is how an expectation about
46
+ # a basket of several items is naturally written. Compared as a single value it could
47
+ # never hold, and an expectation that cannot hold grades nothing while appearing to.
48
+ wanted = list(want) if isinstance(want, (list, tuple)) else [want]
49
+ absent = [value for value in wanted if str(value) not in present]
50
+ if absent:
51
+ found = ", ".join(sorted(present)[:6]) or "nothing"
52
+ failures.append(
53
+ f"{path}: no row has {column}="
54
+ + " or ".join(repr(value) for value in absent)
55
+ + f"; found {found}"
56
+ )
57
+ return failures
58
+
59
+
60
+ def unresolvable(
61
+ state: Mapping[str, list[dict[str, Any]]], expected: Mapping[str, Any]
62
+ ) -> list[str]:
63
+ """Expectations that name a table or column the world does not have.
64
+
65
+ Separate from whether they hold, because they are a different kind of wrong. An expectation
66
+ that fails is a finding about the agent; one that names a table nobody built is a finding
67
+ about the expectation, and letting it through means grading a run against a typo.
68
+ """
69
+ problems: list[str] = []
70
+ for path in expected or {}:
71
+ table, _, column = str(path).partition(".")
72
+ if table not in state:
73
+ # Indexing a particular row is the most common way to write an expectation this
74
+ # cannot carry, and saying only "no such table" sends the reader looking for a
75
+ # spelling mistake instead of at the shape.
76
+ indexed = "[" in table
77
+ problems.append(
78
+ f"{path}: no table called {table!r}"
79
+ + (
80
+ ". Expectations are about the whole table, not one row: use "
81
+ "'table.count' for how many, or 'table.column' for a value that has to "
82
+ "appear in some row."
83
+ if indexed
84
+ else ""
85
+ )
86
+ )
87
+ elif (
88
+ column not in ("", COUNT) and state[table] and column not in state[table][0]
89
+ ):
90
+ problems.append(f"{path}: {table} has no column {column!r}")
91
+ return problems