agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,633 @@
1
+ """Retell target-agent adapter — capability declaration + Stage 6/8 seam,
2
+ plus ``RetellCallOriginator``, the outbound call originator the LiveKit
3
+ engine uses to place and clean up Retell phone calls.
4
+
5
+ Retell has no PSTN outbound API; ``VapiAgentEndpoint`` supports both
6
+ directions but this adapter refuses ``sip_outbound`` explicitly to
7
+ match the guard in ``AgentDefinition``.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import logging
13
+ import os
14
+ import re
15
+ import uuid
16
+ import warnings
17
+ from collections.abc import AsyncIterator
18
+ from dataclasses import dataclass
19
+ from datetime import datetime, timezone
20
+ from typing import Any, Literal
21
+
22
+ import httpx
23
+
24
+ try:
25
+ import retell
26
+ from retell import AsyncRetell, NoneType
27
+ except ImportError: # pragma: no cover - exercised via monkeypatch in tests
28
+ # retell-sdk is an optional dependency: an environment running only
29
+ # chat/Vapi jobs must still be able to import this module (and the rest
30
+ # of fi.simulate.endpoints) without it installed. Only
31
+ # RetellCallOriginator (the class that actually talks to the SDK) fails
32
+ # loudly, in its own __init__ below — RetellAgentEndpoint/RetellCall
33
+ # stay httpx-only and never touch these names.
34
+ retell = None # type: ignore[assignment]
35
+ AsyncRetell = None # type: ignore[assignment,misc]
36
+ NoneType = type(None)
37
+
38
+ from fi.simulate.realtime.events import RealtimeEvent
39
+ from fi.simulate.realtime.media import AudioFrame
40
+ from fi.simulate.runtime.capabilities import EndpointCapabilities
41
+
42
+ from .base import (
43
+ AgentEndpointManifest,
44
+ DiscoveryRequest,
45
+ DiscoverySnapshot,
46
+ EndpointHandle,
47
+ ReadinessResult,
48
+ ReconciliationResult,
49
+ )
50
+
51
+ logger = logging.getLogger(__name__)
52
+
53
+
54
+ def _select_call_rows(payload: Any) -> tuple[list[dict[str, Any]], bool]:
55
+ # Retell's list-calls response shape isn't pinned in docs; accept a bare
56
+ # list or a dict wrapping one under any of the observed key names. The
57
+ # bool distinguishes "recognised shape, zero rows" from "a shape this
58
+ # parser doesn't understand", so callers can tell an empty page apart
59
+ # from a payload they failed to read.
60
+ if isinstance(payload, list):
61
+ return [item for item in payload if isinstance(item, dict)], True
62
+ if isinstance(payload, dict):
63
+ for key in ("calls", "results", "items", "data"):
64
+ value = payload.get(key)
65
+ if isinstance(value, list):
66
+ return [item for item in value if isinstance(item, dict)], True
67
+ # Falls through here for a str (or any other) payload too: the SDK only
68
+ # raises when a JSON content-type response fails to parse as JSON at
69
+ # all, so a plain string body — including a proxy/WAF page served
70
+ # without a JSON content-type — reaches this function as `str` rather
71
+ # than the ValueError arm in reconcile_and_stop. recognised=False is
72
+ # still the right signal; it's the caller's job to remember
73
+ # that "unexpected payload" can mean either an unknown API shape or a
74
+ # gateway page, not only the former.
75
+ return [], False
76
+
77
+
78
+ def _dump_rows(raw_rows: Any) -> tuple[list[dict[str, Any]], int]:
79
+ # Convert an iterable of SDK row models to plain dicts so the existing
80
+ # dict-based row logic in reconcile_and_stop is untouched regardless of
81
+ # which response shape it came from (a bare-list body or a 5.64
82
+ # envelope's .items list). Returns (dumped_rows, skipped_count).
83
+ dumped: list[dict[str, Any]] = []
84
+ skipped = 0
85
+ # model_dump emits a pydantic UserWarning
86
+ # (PydanticSerializationUnexpectedValue) per row with a malformed field
87
+ # (e.g. a non-int start_timestamp); the value still comes through
88
+ # unchanged and is then type-checked by the fences in
89
+ # reconcile_and_stop, so this is noise, not a correctness signal. Scope
90
+ # the suppression tightly to this call only.
91
+ with warnings.catch_warnings():
92
+ warnings.simplefilter("ignore")
93
+ for row in raw_rows:
94
+ # A non-object array element (e.g. a malformed 200 JSON array
95
+ # containing a bare string/int/None) has no model_dump — the
96
+ # SDK's construct_type leaves it as the raw value instead of a
97
+ # typed row. Skip and count it rather than let AttributeError
98
+ # escape the swallow below, past every fence, with no log at
99
+ # all.
100
+ if not hasattr(row, "model_dump"):
101
+ skipped += 1
102
+ continue
103
+ dumped.append(row.model_dump(mode="json"))
104
+ return dumped, skipped
105
+
106
+
107
+ # A call id reaches an f-string URL untouched; restrict it to a plain token
108
+ # so it can never redirect a request at delete-call or any other path.
109
+ # '.' and ':' are included because they show up in real id formats and
110
+ # cannot create a traversal on their own; requiring an alphanumeric start
111
+ # and forbidding consecutive dots keeps a bare "." or ".." from matching.
112
+ _CALL_ID_PATTERN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_:-]*(?:\.[A-Za-z0-9_:-]+)*$")
113
+
114
+
115
+ def _validate_call_id(call_id: str) -> None:
116
+ if not isinstance(call_id, str) or not _CALL_ID_PATTERN.fullmatch(call_id):
117
+ raise ValueError("retell_call_id_invalid")
118
+
119
+
120
+ def _strip_or_none(value: Any) -> Any:
121
+ return value.strip() or None if isinstance(value, str) else value
122
+
123
+
124
+ @dataclass(frozen=True)
125
+ class RetellCall:
126
+ call_id: str
127
+ status: str | None
128
+
129
+
130
+ class RetellCallOriginator:
131
+ """Create an opt-in Retell call to the LiveKit inbound DID.
132
+
133
+ ``reconcile_and_stop`` exists because ``asyncio.wait_for`` cancels our
134
+ coroutine, not Retell's server-side dial — a slow response can leave a
135
+ live, billed call with no id in hand. Because ``from_number`` is the
136
+ customer's production Retell number, the guard is fenced hard: proven
137
+ filter vocabulary only, a client-side window and exact-destination
138
+ match, and at most one stop.
139
+ """
140
+
141
+ _base_url = "https://api.retellai.com"
142
+ _RECONCILE_TIMEOUT = httpx.Timeout(10.0, connect=10.0)
143
+ _STOPPABLE_STATUSES = {"registered", "ongoing"}
144
+ _TOLERATED_STOP_STATUSES = {200, 202, 204, 404, 422}
145
+ _LIST_CALLS_LIMIT = 50
146
+
147
+ def __init__(
148
+ self,
149
+ *,
150
+ api_key: str,
151
+ agent_id: str,
152
+ from_number: str,
153
+ destination: str,
154
+ client: httpx.AsyncClient | None = None,
155
+ ) -> None:
156
+ if retell is None:
157
+ raise RuntimeError(
158
+ "retell_sdk_missing: RetellCallOriginator needs "
159
+ "retell-sdk>=5.64,<6 (pip install 'retell-sdk>=5.64,<6')"
160
+ )
161
+ self._agent_id = agent_id
162
+ self._from_number = from_number
163
+ self._destination = destination
164
+ # Track ownership of the raw httpx client ourselves rather than via
165
+ # AsyncRetell.close() — that call closes whatever http_client it was
166
+ # given, owned or not, which would close a caller-injected client.
167
+ self._owns_client = client is None
168
+ self._http = client or httpx.AsyncClient()
169
+ self._client = AsyncRetell(
170
+ api_key=api_key,
171
+ base_url=os.environ.get("RETELL_API_BASE_URL", self._base_url),
172
+ timeout=httpx.Timeout(30.0, connect=10.0),
173
+ # Our contract already defines retry/tolerance behaviour (the
174
+ # tolerated-stop-status set, the reconcile guard); the SDK must
175
+ # not layer hidden retries on top of it.
176
+ max_retries=0,
177
+ http_client=self._http,
178
+ )
179
+
180
+ @classmethod
181
+ def from_env(cls, transport: Any = None) -> "RetellCallOriginator":
182
+ # Transport (the job's non-secret config) wins; env vars are the
183
+ # local-CLI fallback. Secrets and the leased DID are env-only.
184
+ agent_id = _strip_or_none(getattr(transport, "originator_agent_id", None)) or (
185
+ os.environ.get("RETELL_AGENT_ID", "").strip() or None
186
+ )
187
+ from_number = _strip_or_none(
188
+ getattr(transport, "originator_from_number", None)
189
+ ) or (os.environ.get("RETELL_FROM_NUMBER", "").strip() or None)
190
+ api_key = os.environ.get("RETELL_API_KEY", "").strip() or None
191
+ destination = os.environ.get("LIVEKIT_INBOUND_DID", "").strip() or None
192
+ values = {
193
+ "RETELL_API_KEY": api_key,
194
+ "RETELL_AGENT_ID": agent_id,
195
+ "RETELL_FROM_NUMBER": from_number,
196
+ "LIVEKIT_INBOUND_DID": destination,
197
+ }
198
+ missing = [name for name, value in values.items() if not value]
199
+ if missing:
200
+ raise ValueError(
201
+ "retell_originator_config_missing: " + ", ".join(sorted(missing))
202
+ )
203
+ return cls(
204
+ api_key=api_key,
205
+ agent_id=agent_id,
206
+ from_number=from_number,
207
+ destination=destination,
208
+ )
209
+
210
+ async def start(self) -> RetellCall:
211
+ # Non-2xx raises retell.APIStatusError (a subclass covers each HTTP
212
+ # status); we let it propagate, same failure surface as before.
213
+ response = await self._client.call.create_phone_call(
214
+ from_number=self._from_number,
215
+ to_number=self._destination,
216
+ override_agent_id=self._agent_id,
217
+ )
218
+ # A 2xx without a JSON content-type is passed through by the SDK as
219
+ # raw text (or NoneType for a 204), not the typed PhoneCallResponse
220
+ # model, so `.call_id` would raise AttributeError; getattr keeps the
221
+ # contracted ValueError the only failure mode here.
222
+ call_id = getattr(response, "call_id", None)
223
+ if not isinstance(call_id, str) or not call_id.strip():
224
+ raise ValueError("retell_call_response_missing_id")
225
+ status = response.call_status
226
+ return RetellCall(
227
+ call_id=call_id,
228
+ status=str(status) if status is not None else None,
229
+ )
230
+
231
+ async def stop(self, call_id: str, *, timeout: httpx.Timeout | None = None) -> None:
232
+ # Only stop-call may end a call; delete-call destroys the record and
233
+ # transcript. 5.8.0 has no typed stop-call method, so this uses the
234
+ # client's escape hatch to POST the path directly.
235
+ _validate_call_id(call_id)
236
+ # Mirror the SDK's own NoneType-cast pattern (call.py's delete():
237
+ # extra_headers={"Accept": "*/*"}) so this bodyless POST doesn't
238
+ # advertise a JSON body it never sends, nor demand JSON back from an
239
+ # endpoint that may answer 204. The escape-hatch post()
240
+ # takes a RequestOptions dict, not the typed extra_headers kwarg
241
+ # delete() has, so the header goes directly under "headers" —
242
+ # probed against the installed SDK: "extra_headers" here is silently
243
+ # ignored by FinalRequestOptions.construct, "headers" is not.
244
+ options: dict[str, Any] = {"headers": {"Accept": "*/*"}}
245
+ if timeout is not None:
246
+ options["timeout"] = timeout
247
+ try:
248
+ await self._client.post(
249
+ f"/v2/stop-call/{call_id}", cast_to=NoneType, options=options
250
+ )
251
+ except retell.APIStatusError as exc:
252
+ if exc.status_code not in self._TOLERATED_STOP_STATUSES:
253
+ raise
254
+
255
+ async def reconcile_and_stop(
256
+ self, *, started_after_ms: int, ended_before_ms: int
257
+ ) -> list[str]:
258
+ # Argument validation runs before any request so a wrong call site
259
+ # fails loudly instead of silently returning [].
260
+ for label, value in (
261
+ ("started_after_ms", started_after_ms),
262
+ ("ended_before_ms", ended_before_ms),
263
+ ):
264
+ if not isinstance(value, int) or isinstance(value, bool):
265
+ raise TypeError(
266
+ f"reconcile_and_stop requires int epoch ms for {label}, "
267
+ f"got {type(value).__name__}"
268
+ )
269
+
270
+ try:
271
+ # SDK-typed v3 FilterCriteria (retell/types/call_list_params.py,
272
+ # confirmed against the installed 5.64.0 wheel ~:605-745):
273
+ # AsyncCallResource.list posts to /v3/list-calls, and
274
+ # from_number/to_number are {"type": "string", "op": "eq",
275
+ # "value": <str>} filters, start_timestamp is a {"type": "range",
276
+ # "op": "bt", "value": [lower_ms, upper_ms]} filter — not the
277
+ # list-valued/{lower_threshold, upper_threshold} v2 shape a prior
278
+ # version of this comment described (PLAN D12, supersedes D10a).
279
+ # Confirmed on the wire via MockTransport against the installed
280
+ # SDK. The to_number filter is defense in depth only; every
281
+ # client-side fence below still applies.
282
+ response = await self._client.call.list(
283
+ filter_criteria={
284
+ "from_number": {
285
+ "type": "string",
286
+ "op": "eq",
287
+ "value": self._from_number,
288
+ },
289
+ "to_number": {
290
+ "type": "string",
291
+ "op": "eq",
292
+ "value": self._destination,
293
+ },
294
+ "start_timestamp": {
295
+ "type": "range",
296
+ "op": "bt",
297
+ "value": [started_after_ms, ended_before_ms],
298
+ },
299
+ },
300
+ limit=self._LIST_CALLS_LIMIT,
301
+ timeout=self._RECONCILE_TIMEOUT,
302
+ )
303
+ # 5.64+: call.list() returns retell.types.CallListResponse, a
304
+ # paginated envelope (items/has_more/pagination_key/total), not
305
+ # a bare list of rows (PLAN D12a). Response handling order:
306
+ # 1. dict — a test double, or any future/legacy body the SDK
307
+ # hands back unparsed. Checked first because dict.items is a
308
+ # bound method, not a key list — isinstance(response, dict)
309
+ # is the only reliable way to keep that from being confused
310
+ # with the envelope's `.items` attribute probed in branch 3.
311
+ # 2. bare list — a pre-5.64 SDK compatibility path (probed: the
312
+ # installed 5.64 SDK constructs one CallListResponse per
313
+ # top-level array element via extra="allow" absorption when
314
+ # the body itself is a JSON array). Kept for the legacy test
315
+ # and any older-SDK deploy.
316
+ # 3. else (the real 5.64 envelope) — `.items` is a list of
317
+ # typed Item pydantic models when the server sent one.
318
+ # When it's absent or not a list (an unrecognised envelope,
319
+ # or a legacy `{"calls": [...]}` body the SDK folded into an
320
+ # extra field instead of `.items` — probed), fall back to
321
+ # _select_call_rows on a full model_dump so its legacy
322
+ # dict-key fallbacks get one more pass at the envelope, and
323
+ # an unrecognised shape still logs a real payload_type
324
+ # (e.g. "CallListResponse") instead of "dict" below.
325
+ skipped = 0
326
+ if isinstance(response, dict):
327
+ payload: Any = response
328
+ rows, recognised = _select_call_rows(payload)
329
+ elif isinstance(response, list):
330
+ payload, skipped = _dump_rows(response)
331
+ rows, recognised = _select_call_rows(payload)
332
+ else:
333
+ items = getattr(response, "items", None)
334
+ if isinstance(items, list):
335
+ payload, skipped = _dump_rows(items)
336
+ rows, recognised = _select_call_rows(payload)
337
+ elif hasattr(response, "model_dump"):
338
+ rows, recognised = _select_call_rows(
339
+ response.model_dump(mode="json")
340
+ )
341
+ payload = response
342
+ else:
343
+ # Not actually a pydantic model — e.g. a scalar JSON
344
+ # body (a bare string or null) the SDK couldn't
345
+ # construct CallListResponse from at all, so it comes
346
+ # through as the raw str/None value. _select_call_rows
347
+ # already falls through such values to ([], False);
348
+ # payload stays this raw value so the unexpected_payload
349
+ # log below still reports its real type ("str",
350
+ # "NoneType"), same as before this round.
351
+ payload = response
352
+ rows, recognised = _select_call_rows(payload)
353
+ except (retell.APIError, httpx.HTTPError, ValueError) as exc:
354
+ # retell.APIError covers both APIStatusError (non-2xx) and
355
+ # APIConnectionError (transport failure) — the SDK's hierarchy
356
+ # is not httpx's, so APIConnectionError is NOT an
357
+ # httpx.HTTPError subclass (the test is the actual net
358
+ # for a transport failure). httpx.HTTPError is kept only as
359
+ # insurance for a raw-httpx code path that doesn't exist today —
360
+ # probed dead against the SDK, harmless to keep. ValueError
361
+ # covers a non-JSON 2xx body (e.g. a malformed-but-JSON-typed
362
+ # proxy/WAF page raising json.JSONDecodeError); the
363
+ # swallow-and-log promise applies to that too, not just
364
+ # transport-level errors.
365
+ logger.warning(
366
+ "retell_call_reconcile",
367
+ extra={"phase": "list_calls", "error": type(exc).__name__},
368
+ )
369
+ return []
370
+
371
+ if skipped:
372
+ # Ignore, informational: a non-object row is not a payload shape
373
+ # failure (recognised can still be True) and carries no
374
+ # customer data worth logging — count only, same as the other
375
+ # dropped-row counters below.
376
+ logger.warning(
377
+ "retell_reconcile_unexpected_row", extra={"skipped": skipped}
378
+ )
379
+
380
+ if not recognised:
381
+ # A JSON body that parsed fine but isn't a shape this parser
382
+ # understands (e.g. a bare string or null) must not be reported
383
+ # the same as a genuinely empty page — the two drive opposite
384
+ # calls on whether the guard is working.
385
+ logger.warning(
386
+ "retell_reconcile_unexpected_payload",
387
+ extra={"payload_type": type(payload).__name__},
388
+ )
389
+ return []
390
+
391
+ # 5.64's envelope carries an explicit ``has_more``; a full page is the
392
+ # pre-envelope heuristic. Either means the candidate set may be
393
+ # truncated; ordering isn't guaranteed so our own call could be off
394
+ # the page. Read from the envelope object, never from ``payload``.
395
+ # Checked BEFORE the empty-page return: an empty page the server says
396
+ # is truncated is page_full (guard may have missed its own call),
397
+ # never no_candidates (the "page genuinely empty" signal).
398
+ has_more = bool(getattr(response, "has_more", False))
399
+ if len(rows) == self._LIST_CALLS_LIMIT or has_more:
400
+ # Truncated/full page: ordering isn't guaranteed, so our own call
401
+ # may be off the page. A partial view of the customer's production
402
+ # account cannot establish ownership — fail closed, stop nothing.
403
+ logger.warning(
404
+ "retell_reconcile_page_full",
405
+ extra={"row_count": len(rows), "has_more": has_more},
406
+ )
407
+ return []
408
+
409
+ if not rows:
410
+ # The likely inert mode: a range filter on start_timestamp can't
411
+ # match a row that never got one, so a registered call with no
412
+ # timestamp yet is dropped server-side and never comes back.
413
+ logger.warning("retell_reconcile_no_candidates")
414
+ return []
415
+
416
+ candidates: list[dict[str, Any]] = []
417
+ dropped_no_start_timestamp = 0
418
+ dropped_out_of_window = 0
419
+ for row in rows:
420
+ start_ts = row.get("start_timestamp")
421
+ if not isinstance(start_ts, (int, float)) or isinstance(start_ts, bool):
422
+ dropped_no_start_timestamp += 1
423
+ continue
424
+ # Server-side `bt` filtering is unverified against a live key;
425
+ # re-check the window here so a widened query can never reach a
426
+ # call outside it.
427
+ if not (started_after_ms <= start_ts <= ended_before_ms):
428
+ dropped_out_of_window += 1
429
+ continue
430
+ candidates.append(row)
431
+ if dropped_no_start_timestamp:
432
+ logger.warning(
433
+ "retell_reconcile_no_start_timestamp",
434
+ extra={"dropped": dropped_no_start_timestamp},
435
+ )
436
+ if dropped_out_of_window:
437
+ logger.warning(
438
+ "retell_reconcile_out_of_window",
439
+ extra={"dropped": dropped_out_of_window},
440
+ )
441
+ if not candidates:
442
+ return []
443
+
444
+ # The server-side from_number filter (like to_number) is defense in
445
+ # depth only, not proven — the leased destination DID comes from a
446
+ # shared pool, so recheck from_number client-side too, the same way
447
+ # the window is rechecked above, rather than trusting the server
448
+ # filter to be the only thing scoping this query to our own line.
449
+ dropped_from_number_mismatch = 0
450
+ from_number_matches: list[dict[str, Any]] = []
451
+ for row in candidates:
452
+ if row.get("from_number") == self._from_number:
453
+ from_number_matches.append(row)
454
+ else:
455
+ dropped_from_number_mismatch += 1
456
+ if dropped_from_number_mismatch:
457
+ logger.warning(
458
+ "retell_reconcile_from_number_mismatch",
459
+ extra={"dropped": dropped_from_number_mismatch},
460
+ )
461
+ candidates = from_number_matches
462
+ if not candidates:
463
+ return []
464
+
465
+ # A non-string to_number (e.g. a malformed API row shaped as a list
466
+ # or dict) can never equal our destination string either, so it is
467
+ # folded into "no usable destination" rather than reaching a set
468
+ # comprehension, where it would raise instead of just not matching.
469
+ with_destination = [
470
+ row for row in candidates if isinstance(row.get("to_number"), str)
471
+ ]
472
+ if not with_destination:
473
+ # Structurally inert against this response shape — not proof no
474
+ # orphan existed.
475
+ logger.warning("retell_reconcile_no_destination_field")
476
+ return []
477
+
478
+ # Exact match only: a formatting difference (e.g. E.164 vs local)
479
+ # must fail closed rather than risk matching a stranger's row.
480
+ destination_matches = [
481
+ row for row in with_destination if row.get("to_number") == self._destination
482
+ ]
483
+ if not destination_matches:
484
+ # Never log the observed numbers themselves — they are the
485
+ # customer's own callees on their production line, not ours.
486
+ distinct_observed = len({row.get("to_number") for row in with_destination})
487
+ logger.warning(
488
+ "retell_reconcile_destination_mismatch",
489
+ extra={
490
+ "destination": self._destination,
491
+ "distinct_observed": distinct_observed,
492
+ "count": len(with_destination),
493
+ },
494
+ )
495
+ return []
496
+
497
+ stoppable = [
498
+ row
499
+ for row in destination_matches
500
+ if str(row.get("call_status") or "").lower() in self._STOPPABLE_STATUSES
501
+ ]
502
+ if not stoppable:
503
+ return []
504
+
505
+ if len(stoppable) > 1:
506
+ # More than one in-window stoppable row from our line to the leased
507
+ # DID. Source number + destination + window does not single out our
508
+ # own call — concurrent calls, a duplicate request, or a prior
509
+ # timed-out attempt could all sit here — and this is the customer's
510
+ # production account, so fail closed and stop nothing. Log the count
511
+ # only, never the third-party ids.
512
+ logger.warning(
513
+ "retell_reconcile_ambiguous",
514
+ extra={"stoppable": len(stoppable)},
515
+ )
516
+ return []
517
+
518
+ target = stoppable[0]
519
+ call_id = target.get("call_id")
520
+ has_id = isinstance(call_id, str)
521
+ if not has_id or not _CALL_ID_PATTERN.fullmatch(call_id):
522
+ # Silent [] here would look identical to "the guard worked and
523
+ # found nothing" — log which of the two shapes it was.
524
+ logger.warning(
525
+ "retell_reconcile_target_without_id",
526
+ extra={"has_id": has_id, "count": len(stoppable)},
527
+ )
528
+ return []
529
+
530
+ try:
531
+ await self.stop(call_id, timeout=self._RECONCILE_TIMEOUT)
532
+ except (retell.APIError, httpx.HTTPError) as exc:
533
+ # httpx.HTTPError is kept as insurance only — the SDK's own
534
+ # errors (APIStatusError, APIConnectionError) are never
535
+ # httpx.HTTPError subclasses (probed); the transport-error net
536
+ # is the test on the list_calls phase above.
537
+ logger.warning(
538
+ "retell_call_reconcile",
539
+ extra={"phase": "stop_call", "error": type(exc).__name__},
540
+ )
541
+ return []
542
+
543
+ return [call_id]
544
+
545
+ async def close(self) -> None:
546
+ # Never AsyncRetell.close() here — it closes self._http regardless
547
+ # of who created it, which would close a caller-injected client.
548
+ if self._owns_client:
549
+ await self._http.aclose()
550
+
551
+
552
+ class RetellAgentEndpoint:
553
+ def __init__(
554
+ self,
555
+ *,
556
+ name: str,
557
+ channel: Literal["sip_inbound", "web_call"] = "sip_inbound",
558
+ agent_id: str | None = None,
559
+ ) -> None:
560
+ if channel == "sip_outbound":
561
+ raise ValueError(
562
+ "retell_pstn_outbound_unsupported: Retell has no outbound API"
563
+ )
564
+ self.manifest = AgentEndpointManifest(
565
+ name=name,
566
+ provider="retell",
567
+ world_kinds=["voice"],
568
+ capabilities=EndpointCapabilities(
569
+ audio=True,
570
+ text=True,
571
+ streaming=True,
572
+ interruption=True,
573
+ dtmf=False,
574
+ transfer=False,
575
+ transcript_events=True,
576
+ tool_events=True,
577
+ usage_events=True,
578
+ internal_metrics=False,
579
+ recording=True,
580
+ web_rtc=channel == "web_call",
581
+ sip=channel == "sip_inbound",
582
+ ),
583
+ metadata={"channel": channel, "agent_id": agent_id}
584
+ if agent_id
585
+ else {"channel": channel},
586
+ )
587
+ self._channel = channel
588
+ self.capabilities = self.manifest.capabilities
589
+
590
+ async def discover(self, request: DiscoveryRequest) -> DiscoverySnapshot:
591
+ del request
592
+ return DiscoverySnapshot(capabilities=self.capabilities)
593
+
594
+ async def prepare(self, plan) -> EndpointHandle: # noqa: ANN001
595
+ return EndpointHandle(
596
+ handle_id=f"retell-{uuid.uuid4().hex[:12]}",
597
+ endpoint_name=self.manifest.name,
598
+ created_at=datetime.now(timezone.utc),
599
+ metadata={
600
+ "plan_id": getattr(plan, "plan_id", None),
601
+ "channel": self._channel,
602
+ },
603
+ )
604
+
605
+ async def wait_ready(self, handle: EndpointHandle) -> ReadinessResult:
606
+ del handle
607
+ raise NotImplementedError(
608
+ "Retell direct execution seam; live path uses LiveKit SIP inbound today"
609
+ )
610
+
611
+ async def send(
612
+ self, handle: EndpointHandle, event: RealtimeEvent | AudioFrame
613
+ ) -> None:
614
+ raise NotImplementedError("RetellAgentEndpoint.send is a Stage-8 seam")
615
+
616
+ async def receive(
617
+ self, handle: EndpointHandle
618
+ ) -> AsyncIterator[RealtimeEvent | AudioFrame]:
619
+ raise NotImplementedError("RetellAgentEndpoint.receive is a Stage-8 seam")
620
+ yield # type: ignore[unreachable]
621
+
622
+ async def stop(self, handle: EndpointHandle) -> None:
623
+ del handle
624
+
625
+ async def cleanup(self, handle: EndpointHandle) -> None:
626
+ del handle
627
+
628
+ async def reconcile(self, handle: EndpointHandle) -> ReconciliationResult:
629
+ del handle
630
+ return ReconciliationResult(reconciled=True)
631
+
632
+
633
+ __all__ = ["RetellAgentEndpoint", "RetellCall", "RetellCallOriginator"]