agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,587 @@
1
+ import re
2
+ from typing import Annotated, Literal, Optional
3
+
4
+ from pydantic import AnyHttpUrl, AnyUrl, BaseModel, Field, model_validator
5
+
6
+
7
+ # Always .fullmatch, never .match: "$" also matches immediately before a
8
+ # trailing newline, so .match alone would let "+14155551234\n" through.
9
+ _E164 = re.compile(r"^\+[1-9]\d{6,14}$")
10
+
11
+
12
+ class ProviderEvidenceConfig(BaseModel):
13
+ """Post-call provider-side evidence collection (Vapi/Retell).
14
+
15
+ After the phone leg completes, the SDK optionally queries the
16
+ provider's API to enrich the canonical report with their own
17
+ transcripts, recordings, tool calls, and latency. Read the
18
+ provider's call ID either from a LiveKit SIP participant attribute
19
+ (``participant_attribute``) or from a bounded polling window over
20
+ the provider's list-calls endpoint (``polling_window``).
21
+ """
22
+
23
+ provider: Literal["vapi", "retell"] = Field(
24
+ ...,
25
+ description="Provider whose evidence to collect after the call.",
26
+ )
27
+ call_id_source: Literal[
28
+ "participant_attribute", "polling_window", "originator_response"
29
+ ] = Field(
30
+ "participant_attribute",
31
+ description="How to key the provider's call ID.",
32
+ )
33
+ participant_attribute: Optional[str] = Field(
34
+ None,
35
+ description="LiveKit participant attribute name (e.g. sip.callID).",
36
+ )
37
+ polling_window_seconds: Optional[float] = Field(
38
+ None,
39
+ gt=0,
40
+ description="Window before + after case start for list-calls matching.",
41
+ )
42
+ poll_interval_seconds: float = Field(
43
+ 3.0,
44
+ gt=0,
45
+ description="Interval between provider status polls.",
46
+ )
47
+ poll_deadline_seconds: float = Field(
48
+ 60.0,
49
+ gt=0,
50
+ description="Total time to poll the provider before giving up.",
51
+ )
52
+
53
+ @model_validator(mode="after")
54
+ def _check_source(self) -> "ProviderEvidenceConfig":
55
+ if self.call_id_source == "participant_attribute":
56
+ if not self.participant_attribute or not self.participant_attribute.strip():
57
+ raise ValueError(
58
+ "participant_attribute source requires a non-empty attribute name"
59
+ )
60
+ elif self.call_id_source == "polling_window":
61
+ if not self.polling_window_seconds:
62
+ raise ValueError(
63
+ "polling_window source requires polling_window_seconds"
64
+ )
65
+ return self
66
+
67
+
68
+ class TelephonyTransport(BaseModel):
69
+ """Optional telephony transport for a LiveKit-backed target.
70
+
71
+ Default (or omitted): WebRTC — the SDK connects to the room over WS
72
+ and the target is dispatched as a registered agent, unchanged.
73
+
74
+ ``sip_outbound``: the SDK creates the per-case room and dials
75
+ ``sip_call_to`` through ``sip_trunk_id``; the target answers on the
76
+ phone. Requires ``sip_trunk_id`` and E.164 ``sip_call_to``.
77
+
78
+ ``sip_inbound``: the SDK does not dial; a dispatch rule routes an
79
+ incoming call into the per-case room. If ``dispatch_rule_name`` is
80
+ set the SDK verifies and reuses that rule; otherwise it provisions
81
+ a per-run rule and tears it down on cleanup.
82
+ """
83
+
84
+ kind: Literal[
85
+ "webrtc", "sip_outbound", "sip_inbound", "vapi_websocket", "retell_webcall"
86
+ ] = Field(
87
+ "webrtc",
88
+ description="Transport used to reach the target participant.",
89
+ )
90
+ sip_trunk_id: Optional[str] = Field(
91
+ None,
92
+ description="LiveKit outbound SIP trunk ID (sip_outbound only).",
93
+ )
94
+ sip_call_to: Optional[str] = Field(
95
+ None,
96
+ description="E.164 phone number to dial (sip_outbound only).",
97
+ )
98
+ sip_number: Optional[str] = Field(
99
+ None,
100
+ description="E.164 originating caller ID (sip_outbound only).",
101
+ )
102
+ participant_identity: Optional[str] = Field(
103
+ None,
104
+ description=(
105
+ "Template for the SIP participant identity. May contain {test_case_id}, "
106
+ "{run_id}, or {invocation_id}. The default includes invocation and case IDs."
107
+ ),
108
+ )
109
+ dispatch_rule_name: Optional[str] = Field(
110
+ None,
111
+ description="Dispatch rule that routes the inbound call (sip_inbound only).",
112
+ )
113
+ readiness_timeout_seconds: Optional[float] = Field(
114
+ None,
115
+ gt=0,
116
+ description="Seconds to wait for the inbound SIP participant to appear.",
117
+ )
118
+ answer_timeout_seconds: Optional[float] = Field(
119
+ None,
120
+ gt=0,
121
+ description="Seconds to wait for an outbound SIP call to be answered.",
122
+ )
123
+ inbound_call_originator: Literal["vapi", "retell"] | None = Field(
124
+ None,
125
+ description="Provider that originates an inbound SIP call after room readiness.",
126
+ )
127
+ originator_agent_id: Optional[str] = Field(
128
+ None,
129
+ description=(
130
+ "Provider agent/assistant ID to run for the originated call "
131
+ "(sip_inbound with the retell originator only)."
132
+ ),
133
+ )
134
+ originator_from_number: Optional[str] = Field(
135
+ None,
136
+ description=(
137
+ "E.164 number the originator dials from "
138
+ "(sip_inbound with the retell originator only)."
139
+ ),
140
+ )
141
+
142
+ @model_validator(mode="after")
143
+ def _check_kind_fields(self) -> "TelephonyTransport":
144
+ if self.kind == "sip_outbound":
145
+ if self.inbound_call_originator is not None:
146
+ raise ValueError("sip_outbound cannot set inbound_call_originator")
147
+ if (
148
+ self.originator_agent_id is not None
149
+ or self.originator_from_number is not None
150
+ ):
151
+ raise ValueError("sip_outbound cannot set originator fields")
152
+ if not self.sip_trunk_id or not self.sip_trunk_id.strip():
153
+ raise ValueError("sip_outbound requires sip_trunk_id")
154
+ if not self.sip_call_to or not _E164.fullmatch(self.sip_call_to):
155
+ raise ValueError(
156
+ "sip_outbound requires E.164 sip_call_to (e.g. +14155551234)"
157
+ )
158
+ if not self.sip_number or not _E164.fullmatch(self.sip_number):
159
+ raise ValueError(
160
+ "sip_outbound requires E.164 sip_number (e.g. +14155551234)"
161
+ )
162
+ elif self.kind == "sip_inbound":
163
+ if (
164
+ self.dispatch_rule_name is not None
165
+ and not self.dispatch_rule_name.strip()
166
+ ):
167
+ raise ValueError(
168
+ "sip_inbound dispatch_rule_name must be non-empty when set"
169
+ )
170
+ if (
171
+ self.originator_agent_id is not None
172
+ or self.originator_from_number is not None
173
+ ) and self.inbound_call_originator is None:
174
+ raise ValueError("originator fields require inbound_call_originator")
175
+ # C1: the two originator fields are Retell-only; a Vapi originator reads
176
+ # its config from env and would otherwise silently ignore them.
177
+ if (
178
+ self.inbound_call_originator is not None
179
+ and self.inbound_call_originator != "retell"
180
+ and (
181
+ self.originator_agent_id is not None
182
+ or self.originator_from_number is not None
183
+ )
184
+ ):
185
+ raise ValueError(
186
+ f"{self.inbound_call_originator}_originator_does_not_take_originator_fields"
187
+ )
188
+ # Emptiness is only meaningful once an originator is set (None stays forward-compatible).
189
+ if (
190
+ self.inbound_call_originator is not None
191
+ and self.originator_agent_id is not None
192
+ and not self.originator_agent_id.strip()
193
+ ):
194
+ raise ValueError("originator_agent_id must be non-empty when set")
195
+ if self.originator_from_number is not None and not _E164.fullmatch(
196
+ self.originator_from_number
197
+ ):
198
+ raise ValueError(
199
+ "originator_from_number must be E.164 (e.g. +14155551234)"
200
+ )
201
+ elif self.kind in {"webrtc", "vapi_websocket", "retell_webcall"}:
202
+ if any(
203
+ [
204
+ self.sip_trunk_id,
205
+ self.sip_call_to,
206
+ self.sip_number,
207
+ self.dispatch_rule_name,
208
+ self.inbound_call_originator,
209
+ self.originator_agent_id,
210
+ self.originator_from_number,
211
+ ]
212
+ ):
213
+ raise ValueError(f"{self.kind} transport cannot set SIP fields")
214
+ return self
215
+
216
+
217
+ class VapiTargetConfig(BaseModel):
218
+ """Non-secret configuration for a Vapi assistant under test."""
219
+
220
+ provider: Literal["vapi"] = "vapi"
221
+ assistant_id: str = Field(..., min_length=1)
222
+ api_base_url: AnyHttpUrl = Field(
223
+ "https://api.vapi.ai",
224
+ validate_default=True,
225
+ )
226
+ api_key_env: str = Field(
227
+ "VAPI_API_KEY",
228
+ pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
229
+ )
230
+
231
+
232
+ class RetellTargetConfig(BaseModel):
233
+ """Non-secret configuration for a Retell agent under test."""
234
+
235
+ provider: Literal["retell"] = "retell"
236
+ agent_id: str = Field(..., min_length=1)
237
+ api_url: AnyHttpUrl = Field(
238
+ "https://api.retellai.com/v2/create-web-call",
239
+ validate_default=True,
240
+ )
241
+ livekit_url: AnyUrl = Field(
242
+ "wss://retell-ai-4ihahnq7.livekit.cloud",
243
+ validate_default=True,
244
+ )
245
+ api_key_env: str = Field(
246
+ "RETELL_API_KEY",
247
+ pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
248
+ )
249
+
250
+ @model_validator(mode="after")
251
+ def _check_livekit_url(self) -> "RetellTargetConfig":
252
+ if self.livekit_url.scheme not in {"ws", "wss"}:
253
+ raise ValueError("retell_livekit_url_invalid: URL must use ws:// or wss://")
254
+ return self
255
+
256
+
257
+ VoiceProviderTarget = Annotated[
258
+ VapiTargetConfig | RetellTargetConfig,
259
+ Field(discriminator="provider"),
260
+ ]
261
+
262
+
263
+ class LiveKitSimulatorRuntime(BaseModel):
264
+ """FutureAGI-owned LiveKit runtime for the simulator and bridge."""
265
+
266
+ url: AnyUrl = Field(..., description="FutureAGI LiveKit WebSocket URL.")
267
+ room_name: str = Field(..., min_length=1)
268
+ room_mode: Literal["external", "managed"] = "managed"
269
+ room_name_verbatim: bool = Field(
270
+ False,
271
+ description=(
272
+ "When True, use room_name exactly as provided (no invocation/case "
273
+ "suffix). Required when routing to a pre-existing dispatch rule that "
274
+ "binds to a fixed room. A multi-persona run is allowed when cases run "
275
+ "one at a time (max_concurrency 1; always the case for SIP transports)."
276
+ ),
277
+ )
278
+ api_key_env: str = Field(
279
+ "LIVEKIT_API_KEY",
280
+ pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
281
+ )
282
+ api_secret_env: str = Field(
283
+ "LIVEKIT_API_SECRET",
284
+ pattern=r"^[A-Za-z_][A-Za-z0-9_]*$",
285
+ )
286
+
287
+ @model_validator(mode="after")
288
+ def _check_url(self) -> "LiveKitSimulatorRuntime":
289
+ if self.url.scheme not in {"ws", "wss"}:
290
+ raise ValueError("livekit_url_invalid: URL must use ws:// or wss://")
291
+ return self
292
+
293
+
294
+ class LLMConfig(BaseModel):
295
+ """Configuration for the simulator language model."""
296
+
297
+ provider: str = Field("openai", description="The LiveKit LLM provider.")
298
+ model: str = Field("gpt-4o", description="The language model to use.")
299
+ temperature: float = Field(
300
+ 0.7, ge=0.0, le=2.0, description="Controls randomness in the LLM's output."
301
+ )
302
+
303
+
304
+ class TTSConfig(BaseModel):
305
+ """Configuration for simulator text-to-speech."""
306
+
307
+ provider: str = Field("openai", description="The LiveKit TTS provider.")
308
+ model: str = Field("gpt-4o-mini-tts", description="The TTS model to use.")
309
+ voice: str = Field("alloy", description="The voice or voice ID to use.")
310
+ # Delivery, not content. Two callers reading the same words at the same pace is the tell that
311
+ # one generator wrote both. Cartesia documents 0.6 to 2.0 for sonic-3; None leaves the
312
+ # provider default so a provider without the control is unaffected.
313
+ speed: Optional[float] = Field(
314
+ None, description="Speech rate, where the provider supports one."
315
+ )
316
+ # Cartesia takes "<name>:<level>" strings and validates both halves. Left unset the provider
317
+ # default applies, so a provider without the control is unaffected.
318
+ emotion: Optional[list[str]] = Field(
319
+ None, description="Emotional colour, where the provider supports one."
320
+ )
321
+
322
+
323
+ class STTConfig(BaseModel):
324
+ """Configuration for simulator speech-to-text."""
325
+
326
+ provider: str = Field("openai", description="The LiveKit STT provider.")
327
+ model: str = Field(
328
+ "gpt-4o-mini-transcribe",
329
+ description="The STT model to use.",
330
+ )
331
+ language: Optional[str] = Field("en", description="The transcription language.")
332
+
333
+
334
+ class VADConfig(BaseModel):
335
+ """Configuration for Voice Activity Detection (VAD)."""
336
+
337
+ provider: str = Field(
338
+ "silero", description="The VAD provider to use. 'silero' is recommended."
339
+ )
340
+ min_silence_duration: float = Field(
341
+ 0.1,
342
+ description="Minimum duration of silence to consider as the end of a speech segment.",
343
+ )
344
+ speech_pad_ms: int = Field(
345
+ 200,
346
+ description="Additional padding in milliseconds to add to the end of a speech segment.",
347
+ )
348
+
349
+
350
+ class AgentDefinition(BaseModel):
351
+ """
352
+ The core configuration for a voice AI agent.
353
+ """
354
+
355
+ name: str = Field(..., description="A unique name for the target agent.")
356
+ description: Optional[str] = Field(
357
+ None,
358
+ description="A safe description of the target agent's purpose and capabilities.",
359
+ )
360
+ system_prompt: str = Field(
361
+ ...,
362
+ description="Current system prompt or instructions of the target agent.",
363
+ )
364
+ target: VoiceProviderTarget | None = Field(
365
+ None,
366
+ description="Non-secret provider configuration for a direct target agent.",
367
+ )
368
+ agent_name: Optional[str] = Field(
369
+ None,
370
+ description="Exact registered LiveKit target agent name used for managed dispatch.",
371
+ )
372
+ dispatch_metadata: Optional[dict] = Field(
373
+ None,
374
+ description=(
375
+ "Metadata sent to the target agent's LiveKit dispatch. Default None "
376
+ "sends EMPTY metadata: agents built from LiveKit templates treat any "
377
+ "job metadata as an outbound/no-greet call and stop publishing audio, "
378
+ "so injecting simulation context makes readiness time out. Set this "
379
+ "only for agents that are built to consume dispatch metadata."
380
+ ),
381
+ )
382
+ target_participant_identity: Optional[str] = Field(
383
+ None,
384
+ description="Exact target participant identity when it is known in advance.",
385
+ )
386
+ transport: Optional[TelephonyTransport] = Field(
387
+ None,
388
+ description="Transport used to reach the target agent.",
389
+ )
390
+ provider_evidence: Optional[ProviderEvidenceConfig] = Field(
391
+ None,
392
+ description=(
393
+ "Optional post-call provider evidence collection "
394
+ "(Vapi/Retell). None = SDK-observed evidence only."
395
+ ),
396
+ )
397
+ url: AnyUrl | None = Field(
398
+ None,
399
+ description=(
400
+ "Legacy FutureAGI LiveKit URL. Use LiveKitSimulatorRuntime for new "
401
+ "voice simulations."
402
+ ),
403
+ )
404
+ room_name: str | None = Field(
405
+ None,
406
+ description=(
407
+ "Legacy FutureAGI LiveKit room template. Use LiveKitSimulatorRuntime "
408
+ "for new voice simulations."
409
+ ),
410
+ )
411
+ room_mode: Literal["external", "managed"] = Field(
412
+ "external",
413
+ description=(
414
+ "Legacy FutureAGI LiveKit room lifecycle setting. Use "
415
+ "LiveKitSimulatorRuntime for new voice simulations."
416
+ ),
417
+ )
418
+
419
+ @model_validator(mode="after")
420
+ def _check_transport(self) -> "AgentDefinition":
421
+ if self.url is not None and self.url.scheme not in {"ws", "wss"}:
422
+ raise ValueError("livekit_url_invalid: URL must use ws:// or wss://")
423
+ transport = self.transport
424
+ if (
425
+ self.url is not None
426
+ and transport is not None
427
+ and transport.kind != "webrtc"
428
+ and self.room_mode != "managed"
429
+ ):
430
+ raise ValueError("managed_transport_requires_managed_room")
431
+ expected_target_provider = (
432
+ {
433
+ "vapi_websocket": "vapi",
434
+ "retell_webcall": "retell",
435
+ }.get(transport.kind)
436
+ if transport is not None
437
+ else None
438
+ )
439
+ if (
440
+ expected_target_provider is not None
441
+ and self.target is not None
442
+ and self.target.provider != expected_target_provider
443
+ ):
444
+ raise ValueError(
445
+ f"{transport.kind}_requires_{expected_target_provider}_target"
446
+ )
447
+ if self.target is not None:
448
+ expected_transport = {
449
+ "vapi": "vapi_websocket",
450
+ "retell": "retell_webcall",
451
+ }[self.target.provider]
452
+ if transport is None or transport.kind != expected_transport:
453
+ raise ValueError(
454
+ f"{self.target.provider}_target_requires_{expected_transport}"
455
+ )
456
+ evidence = self.provider_evidence
457
+ originator = (
458
+ transport.inbound_call_originator if transport is not None else None
459
+ )
460
+ if originator is not None:
461
+ # Templated on the originator name so Vapi's strings stay byte-identical.
462
+ if transport.kind != "sip_inbound":
463
+ raise ValueError(f"{originator}_originator_requires_sip_inbound")
464
+ if evidence is None or evidence.provider != originator:
465
+ raise ValueError(
466
+ f"{originator}_originator_requires_{originator}_evidence"
467
+ )
468
+ if evidence.call_id_source != "originator_response":
469
+ raise ValueError(
470
+ f"{originator}_originator_requires_originator_response"
471
+ )
472
+ if evidence is not None and transport is not None:
473
+ if evidence.provider == "retell" and transport.kind == "sip_outbound":
474
+ raise ValueError(
475
+ "retell_pstn_outbound_unsupported: Retell has no outbound "
476
+ "phone API; use sip_inbound or a different provider"
477
+ )
478
+ web_provider = {
479
+ "vapi_websocket": "vapi",
480
+ "retell_webcall": "retell",
481
+ }.get(transport.kind)
482
+ if web_provider is not None:
483
+ if evidence.provider != web_provider:
484
+ raise ValueError(
485
+ f"{transport.kind}_requires_{web_provider}_evidence"
486
+ )
487
+ if evidence.call_id_source != "originator_response":
488
+ raise ValueError(f"{transport.kind}_requires_originator_response")
489
+ return self
490
+
491
+ llm: LLMConfig = Field(default_factory=LLMConfig)
492
+ tts: TTSConfig = Field(default_factory=TTSConfig)
493
+ stt: STTConfig = Field(default_factory=STTConfig)
494
+ vad: VADConfig = Field(default_factory=VADConfig)
495
+ initial_message: str = Field(
496
+ "Hello! How can I help you today?",
497
+ description="The first message the agent speaks to start the conversation.",
498
+ )
499
+
500
+ class Config:
501
+ """Pydantic configuration."""
502
+
503
+ json_schema_extra = {
504
+ "example": {
505
+ "name": "vapi-support-agent",
506
+ "description": "Customer-support voice assistant.",
507
+ "system_prompt": "Copy the current target-agent prompt here.",
508
+ "target": {
509
+ "provider": "vapi",
510
+ "assistant_id": "assistant-id",
511
+ "api_key_env": "VAPI_API_KEY",
512
+ },
513
+ "transport": {"kind": "vapi_websocket"},
514
+ }
515
+ }
516
+
517
+
518
+ class SimulatorAgentDefinition(BaseModel):
519
+ """
520
+ Configuration for the simulated customer persona agent used by the TestRunner.
521
+
522
+ This is intentionally separate from the deployed AgentDefinition so tests can
523
+ run with lightweight/cheaper models and different voice/transcription settings.
524
+ """
525
+
526
+ name: Optional[str] = Field(
527
+ None, description="Optional label for the simulator agent"
528
+ )
529
+ instructions: Optional[str] = Field(
530
+ None,
531
+ description=(
532
+ "Optional policy appended to the scenario-derived simulator prompt. "
533
+ "It never replaces persona, situation, or outcome instructions."
534
+ ),
535
+ )
536
+
537
+ llm: LLMConfig = Field(
538
+ default_factory=lambda: LLMConfig(model="gpt-4o-mini", temperature=0.6)
539
+ )
540
+ tts: TTSConfig = Field(default_factory=TTSConfig)
541
+ stt: STTConfig = Field(default_factory=STTConfig)
542
+ vad: VADConfig = Field(default_factory=VADConfig)
543
+
544
+ allow_interruptions: Optional[bool] = Field(
545
+ None,
546
+ description="Whether the simulator agent allows interruptions during TTS.",
547
+ )
548
+ min_endpointing_delay: Optional[float] = Field(
549
+ None,
550
+ description="Minimum endpointing delay (s) to declare end of user turn.",
551
+ )
552
+ max_endpointing_delay: Optional[float] = Field(
553
+ None,
554
+ description="Maximum endpointing delay (s) to force end of user turn.",
555
+ )
556
+ use_tts_aligned_transcript: Optional[bool] = Field(
557
+ None,
558
+ description="Whether to use TTS-aligned transcript as transcription source.",
559
+ )
560
+
561
+ class Config:
562
+ json_schema_extra = {
563
+ "example": {
564
+ "name": "simulator-customer",
565
+ "instructions": "You are a concise customer. Ask clarifying questions and confirm resolution.",
566
+ "llm": {
567
+ "provider": "openai",
568
+ "model": "gpt-4o-mini",
569
+ "temperature": 0.6,
570
+ },
571
+ "tts": {
572
+ "provider": "openai",
573
+ "model": "gpt-4o-mini-tts",
574
+ "voice": "alloy",
575
+ },
576
+ "stt": {
577
+ "provider": "openai",
578
+ "model": "gpt-4o-mini-transcribe",
579
+ "language": "en",
580
+ },
581
+ "vad": {"provider": "silero"},
582
+ "allow_interruptions": True,
583
+ "min_endpointing_delay": 0.3,
584
+ "max_endpointing_delay": 4.0,
585
+ "use_tts_aligned_transcript": False,
586
+ }
587
+ }