agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,517 @@
1
+ import asyncio
2
+ import os
3
+ import contextvars
4
+ import logging
5
+ import contextlib
6
+ from typing import Optional, Callable
7
+
8
+ from fi.simulate._logging import redacted_exc_info
9
+ from fi.simulate.agent.generic import wrap_agent
10
+ from fi.simulate.agent.wrapper import AgentWrapper, AgentInput, AgentResponse
11
+ from fi.simulate.simulation.models import TestReport
12
+ from fi.simulate.simulation.engines.base import BaseEngine
13
+ from fi.simulate.utils.routes import APIRoutes
14
+
15
+ # Context variable to track the current execution ID for future tool mocking
16
+ current_execution_id = contextvars.ContextVar("current_execution_id", default=None)
17
+
18
+ logger = logging.getLogger(__name__)
19
+
20
+ class CloudEngine(BaseEngine):
21
+ """
22
+ Execution engine that connects to the Future AGI backend to orchestrate simulations.
23
+ It acts as a bridge between the cloud-hosted simulator and the user's local agent.
24
+ """
25
+
26
+ def __init__(self, api_key: Optional[str] = None, secret_key: Optional[str] = None, api_url: Optional[str] = None, timeout: float = 120.0):
27
+ """
28
+ Args:
29
+ api_key: API key for authentication
30
+ secret_key: Secret key for authentication
31
+ api_url: Base URL of the backend API
32
+ timeout: Request timeout in seconds (default: 120s for LLM operations)
33
+ """
34
+ self.api_key = api_key or os.environ.get("FI_API_KEY")
35
+ self.secret_key = secret_key or os.environ.get("FI_SECRET_KEY")
36
+ self.api_url = api_url or os.environ.get("FI_BASE_URL") or "https://api.futureagi.com"
37
+ self.timeout = timeout
38
+
39
+ if not self.api_key or not self.secret_key:
40
+ logger.warning("FI_API_KEY or FI_SECRET_KEY not provided. CloudEngine will not function correctly.")
41
+
42
+ self.api = None
43
+ self.run_test_id = None
44
+ self.test_execution_id = None
45
+ self._using_simulator_attributes = None
46
+ try:
47
+ # Optional dependency: enables baggage propagation so user spans inherit simulator IDs
48
+ from fi_instrumentation import using_simulator_attributes # type: ignore
49
+ self._using_simulator_attributes = using_simulator_attributes
50
+ except Exception:
51
+ self._using_simulator_attributes = None
52
+
53
+ async def run(
54
+ self,
55
+ run_id: Optional[str] = None,
56
+ run_test_name: Optional[str] = None,
57
+ agent_callback: Optional[Callable | AgentWrapper] = None,
58
+ concurrency: int = 5,
59
+ **kwargs
60
+ ) -> TestReport:
61
+ """
62
+ Connects to the cloud run, receives user inputs, calls the agent_callback,
63
+ and sends responses back.
64
+ """
65
+ if not run_id and not run_test_name:
66
+ raise ValueError("CloudEngine requires either 'run_id' or 'run_test_name'.")
67
+
68
+ if not agent_callback:
69
+ raise ValueError("CloudEngine requires an 'agent_callback' (function or AgentWrapper).")
70
+
71
+ self.api = APIRoutes(self.api_key, self.secret_key, self.api_url, timeout=self.timeout)
72
+
73
+ # If run_test_name is provided, fetch the run_id first
74
+ if run_test_name and not run_id:
75
+ print(f"🔍 Fetching Run Test ID for name: {run_test_name}")
76
+ try:
77
+ name_resp = await self.api.get_run_test_id_by_name(run_test_name)
78
+ result = name_resp.get("result", {})
79
+ # Handle both camelCase and snake_case response formats
80
+ run_id = result.get("run_test_id") or result.get("runTestId")
81
+ if not run_id:
82
+ raise ValueError(f"Failed to get run_test_id for name '{run_test_name}'. Response: {name_resp}")
83
+ print(f"✓ Found Run Test ID: {run_id}")
84
+ except Exception as e:
85
+ logger.error(f"Failed to get run_test_id by name: {e}")
86
+ raise ValueError(f"Failed to get run_test_id for name '{run_test_name}': {e}")
87
+
88
+ wrapper = self._normalize_callback(agent_callback)
89
+ queue = asyncio.Queue()
90
+
91
+ # Store IDs for tracing attributes
92
+ self.run_test_id = run_id
93
+
94
+ print(f"Starting Simulation for Run ID: {run_id}")
95
+
96
+ try:
97
+ # 1. Start the Run (Create TestExecution)
98
+ start_resp = await self.api.start_test_execution(run_test_id=run_id)
99
+ result = start_resp.get("result", {})
100
+ # Handle both camelCase and snake_case response formats
101
+ test_execution_id = result.get("executionId") or result.get("execution_id")
102
+
103
+ if not test_execution_id:
104
+ raise ValueError(f"Failed to start test execution. Response: {start_resp}")
105
+
106
+ print(f"✓ Test Execution Started: {test_execution_id}")
107
+
108
+ # Store test execution ID for tracing
109
+ self.test_execution_id = test_execution_id
110
+
111
+ # 2. Start Producer and Consumers
112
+ producer_task = asyncio.create_task(
113
+ self._producer_loop(run_id, test_execution_id, queue)
114
+ )
115
+
116
+ consumers = [
117
+ asyncio.create_task(self._consumer_loop(queue, wrapper))
118
+ for _ in range(concurrency)
119
+ ]
120
+
121
+ # Wait for producer to finish fetching all batches
122
+ await producer_task
123
+
124
+ # Wait for queue to drain (all consumers process remaining items)
125
+ await queue.join()
126
+
127
+ # Cancel consumers
128
+ for c in consumers:
129
+ c.cancel()
130
+
131
+ print("✅ Cloud Simulation Completed.")
132
+
133
+ except Exception as exc:
134
+ logger.error(
135
+ "Cloud simulation failed",
136
+ exc_info=redacted_exc_info(exc),
137
+ extra={"exception_type": type(exc).__name__},
138
+ )
139
+ raise
140
+ finally:
141
+ if self.api:
142
+ await self.api.close()
143
+
144
+ # Return empty report for now as backend handles metrics
145
+ return TestReport(results=[])
146
+
147
+ async def _producer_loop(self, run_test_id: str, test_execution_id: str, queue: asyncio.Queue):
148
+ """
149
+ Polls the backend for batches of call execution IDs and puts them in the queue.
150
+ """
151
+ has_more = True
152
+
153
+ while has_more:
154
+ try:
155
+ print("🔄 Fetching batch of scenarios...")
156
+ resp = await self.api.fetch_execution_batch(test_execution_id)
157
+
158
+ result = resp.get("result", {})
159
+ # Handle both camelCase and snake_case response formats
160
+ call_ids = result.get("callExecutionIds") or result.get("call_execution_ids", [])
161
+ has_more = result.get("hasMore") if "hasMore" in result else result.get("has_more", False)
162
+
163
+ if not call_ids:
164
+ if has_more:
165
+ print("⚠️ Received empty batch but hasMore is true. Waiting...")
166
+ await asyncio.sleep(2)
167
+ continue
168
+ else:
169
+ break
170
+
171
+ print(f"📥 Received batch: {len(call_ids)} calls")
172
+ for cid in call_ids:
173
+ await queue.put(cid)
174
+
175
+ except Exception as e:
176
+ logger.error(f"Error fetching batch: {e}")
177
+ # Simple retry logic or break? For now, break to avoid infinite loop
178
+ break
179
+
180
+ def _simulator_baggage_context(self, call_execution_id: str):
181
+ """
182
+ Creates a context manager that sets simulator IDs into OTEL baggage (via fi_instrumentation),
183
+ so any user-agent spans created inside the block inherit these attributes.
184
+ """
185
+ if self._using_simulator_attributes is None:
186
+ return contextlib.nullcontext()
187
+
188
+ simulator_attributes = {
189
+ "is_simulator_trace": True,
190
+ "run_test_id": self.run_test_id,
191
+ "test_execution_id": self.test_execution_id,
192
+ "call_execution_id": call_execution_id,
193
+ }
194
+ # Remove None values to avoid serializing nulls
195
+ simulator_attributes = {k: v for k, v in simulator_attributes.items() if v is not None}
196
+
197
+ return self._using_simulator_attributes(simulator_attributes)
198
+
199
+ async def _consumer_loop(self, queue: asyncio.Queue, wrapper: AgentWrapper):
200
+ """
201
+ Worker that pulls execution IDs from the queue and runs the conversation.
202
+ """
203
+ while True:
204
+ try:
205
+ execution_id = await queue.get()
206
+ await self._handle_single_execution(execution_id, wrapper)
207
+ queue.task_done()
208
+ except asyncio.CancelledError:
209
+ break
210
+ except Exception as e:
211
+ error_msg = str(e) or f"{type(e).__name__}: {repr(e)}"
212
+ logger.error(f"Error in consumer: {error_msg}", exc_info=True)
213
+ print(f"❌ Consumer error: {error_msg}")
214
+ queue.task_done() # Mark done even if failed so join() works
215
+
216
+ async def _handle_single_execution(self, call_execution_id: str, wrapper: AgentWrapper):
217
+ """
218
+ Runs the conversation loop for a single call execution.
219
+ """
220
+ token = current_execution_id.set(call_execution_id)
221
+ try:
222
+ print(f"▶️ Processing Call: {call_execution_id}")
223
+ return await self._handle_single_execution_inner(call_execution_id, wrapper)
224
+ finally:
225
+ current_execution_id.reset(token)
226
+
227
+ async def _handle_single_execution_inner(self, call_execution_id: str, wrapper: AgentWrapper):
228
+ """
229
+ Inner implementation of a single call execution. Separated so we can optionally wrap
230
+ the entire conversation in a parent tracing span and other instrumentation.
231
+ """
232
+ try:
233
+
234
+ # Step 1: Initiate chat (POST with initiate_chat=True)
235
+ init_resp = await self.api.send_chat_message(
236
+ call_execution_id=call_execution_id,
237
+ initiate_chat=True
238
+ )
239
+ result = init_resp.get("result", {})
240
+
241
+ if not result:
242
+ logger.error(f"Failed to initiate chat for {call_execution_id}")
243
+ return
244
+
245
+ # Extract first message(s) from response
246
+ # Note: message_history is a list of ChatMessage objects (dicts)
247
+ message_history = result.get("message_history") or result.get("messageHistory", [])
248
+
249
+ if not message_history:
250
+ # Fallback to output_message if history is empty
251
+ output_msg = result.get("output_message") or result.get("outputMessage")
252
+ if output_msg:
253
+ # Ensure it's a list
254
+ if isinstance(output_msg, list):
255
+ message_history = output_msg
256
+ else:
257
+ message_history = [output_msg]
258
+
259
+ if not message_history:
260
+ logger.warning(f"No initial message received for {call_execution_id}")
261
+ return
262
+
263
+ # Build conversation history for SDK format
264
+ # Convert backend "assistant" → SDK "user" (simulator messages)
265
+ conversation_history = []
266
+ for msg in message_history:
267
+ backend_role = msg.get("role", "user")
268
+
269
+ # Filter out system and tool messages from backend (simulator artifacts)
270
+ if backend_role in ["system", "tool"]:
271
+ continue
272
+
273
+ # Filter out empty messages (often tool calls without output text yet)
274
+ content = msg.get("content", "")
275
+ if not content and backend_role == "assistant":
276
+ continue
277
+
278
+ # Backend sends simulator messages as "assistant", convert to "user" for SDK
279
+ sdk_role = "user" if backend_role == "assistant" else backend_role
280
+ conversation_history.append({
281
+ "role": sdk_role,
282
+ "content": content
283
+ })
284
+
285
+ # Step 2: Conversation loop
286
+ max_turns = 50 # Safety limit
287
+ turn_count = 0
288
+ agent_call_failed = False # Track if agent call failed
289
+
290
+ while turn_count < max_turns:
291
+ # Check if chat ended based on last response
292
+ chat_ended = result.get("chat_ended") or result.get("chatEnded", False)
293
+ if chat_ended:
294
+ break
295
+
296
+ # Get the last message (should be from simulator/user to reply to)
297
+ if not conversation_history:
298
+ break
299
+
300
+ last_msg = conversation_history[-1]
301
+
302
+ # Prepare AgentInput for user's wrapper
303
+ agent_input = AgentInput(
304
+ thread_id=call_execution_id,
305
+ messages=conversation_history,
306
+ new_message=last_msg,
307
+ execution_id=call_execution_id
308
+ )
309
+
310
+ # Call user's agent and measure latency
311
+ import time
312
+ start_time = time.time() # Fallback for latency calculation approximation
313
+ try:
314
+ # Propagate simulator IDs to any spans created by the user's agent instrumentation
315
+ with self._simulator_baggage_context(call_execution_id):
316
+ start_time = time.time() # Accurate start time for latency calculation
317
+ agent_response = await wrapper.call(agent_input)
318
+ except Exception as e:
319
+ error_msg = str(e) or f"{type(e).__name__}: {repr(e)}"
320
+ last_msg_content = agent_input.new_message.get('content', '') if agent_input.new_message else 'N/A'
321
+ logger.error(f"Agent call failed for {call_execution_id}: {error_msg}", exc_info=True)
322
+ print(f"❌ Agent call failed for {call_execution_id}: {error_msg}")
323
+ if last_msg_content != 'N/A':
324
+ print(f" Last message: {last_msg_content[:100]}...")
325
+ # Update call execution status in the backend
326
+ # If we have already completed some turns, mark as "completed" so evaluations can run
327
+ # on the partial data. Only mark as "failed" if we failed on the first turn.
328
+ status = "completed" if turn_count > 0 else "failed"
329
+ # Use generic error message to avoid leaking internal error details
330
+ generic_reason = "Error processing simulation"
331
+ try:
332
+ await self.api.update_call_execution_status(
333
+ call_execution_id,
334
+ status,
335
+ ended_reason=generic_reason
336
+ )
337
+ print(f" Status set to '{status}' (turn_count={turn_count})")
338
+ except Exception as status_error:
339
+ logger.warning(f"Failed to update call execution status for {call_execution_id}: {status_error}")
340
+ agent_call_failed = True
341
+ break
342
+ latency_ms = int((time.time() - start_time) * 1000) if start_time is not None else 0
343
+
344
+ # Normalize response and extract tool_calls and tool_responses
345
+ response_content = ""
346
+ tool_calls = None
347
+ tool_responses = None
348
+
349
+ if isinstance(agent_response, AgentResponse):
350
+ response_content = agent_response.content
351
+ tool_calls = agent_response.tool_calls
352
+ tool_responses = agent_response.tool_responses
353
+ # Back-compat: allow tool outputs to be passed via metadata["tool_outputs"]
354
+ # Expected shape: [{"call_id": "...", "output": ...}, ...]
355
+ if not tool_responses and agent_response.metadata:
356
+ tool_outputs = agent_response.metadata.get("tool_outputs")
357
+ if isinstance(tool_outputs, list) and tool_outputs:
358
+ import json
359
+ converted: list[dict] = []
360
+ for item in tool_outputs:
361
+ if not isinstance(item, dict):
362
+ continue
363
+ call_id = item.get("call_id") or item.get("tool_call_id")
364
+ output = item.get("output")
365
+ if call_id is None and output is None:
366
+ continue
367
+ converted.append(
368
+ {
369
+ "role": "tool",
370
+ "tool_call_id": call_id,
371
+ "content": output
372
+ if isinstance(output, str)
373
+ else json.dumps(output),
374
+ }
375
+ )
376
+ tool_responses = converted or None
377
+ else:
378
+ response_content = str(agent_response)
379
+
380
+ # Add agent response to history (with tool_calls if present)
381
+ assistant_msg = {
382
+ "role": "assistant",
383
+ "content": response_content
384
+ }
385
+ if tool_calls:
386
+ assistant_msg["tool_calls"] = tool_calls
387
+ conversation_history.append(assistant_msg)
388
+
389
+ # Add tool role messages (tool responses) after assistant message with tool_calls
390
+ if tool_responses:
391
+ for tool_response in tool_responses:
392
+ conversation_history.append(tool_response)
393
+
394
+ # Step 3: Send agent response to backend and get next message
395
+ # Send the assistant message with tool_calls and any tool responses
396
+ # SDK "assistant" (agent) → backend "user", SDK "tool" → backend "tool"
397
+ api_messages = []
398
+
399
+ # Add assistant message with tool_calls
400
+ assistant_api_msg = {
401
+ "role": "user", # Convert SDK "assistant" → backend "user"
402
+ "content": assistant_msg["content"]
403
+ }
404
+ if "tool_calls" in assistant_msg:
405
+ assistant_api_msg["tool_calls"] = assistant_msg["tool_calls"]
406
+ api_messages.append(assistant_api_msg)
407
+
408
+ # Add tool role messages if present
409
+ if tool_responses:
410
+ for tool_response in tool_responses:
411
+ api_messages.append({
412
+ "role": "tool", # Keep as "tool" for backend
413
+ "tool_call_id": tool_response.get("tool_call_id"),
414
+ "content": tool_response.get("content", "")
415
+ })
416
+
417
+ metrics = {"latency": latency_ms}
418
+
419
+ # Send
420
+ turn_resp = await self.api.send_chat_message(
421
+ call_execution_id=call_execution_id,
422
+ messages=api_messages,
423
+ metrics=metrics,
424
+ initiate_chat=False
425
+ )
426
+
427
+ result = turn_resp.get("result", {})
428
+ if not result:
429
+ logger.warning(f"No response from backend for {call_execution_id}")
430
+ break
431
+
432
+ # Update conversation history from backend response
433
+
434
+ new_history_data = result.get("message_history") or result.get("messageHistory", [])
435
+
436
+ if new_history_data:
437
+ # Convert backend "assistant" → SDK "user" (simulator messages)
438
+ conversation_history = []
439
+ for msg in new_history_data:
440
+ backend_role = msg.get("role", "user")
441
+
442
+ # Filter out system and tool messages
443
+ if backend_role in ["system", "tool"]:
444
+ continue
445
+
446
+ # Filter out empty messages
447
+ content = msg.get("content", "")
448
+ if not content and backend_role == "assistant":
449
+ continue
450
+
451
+ sdk_role = "user" if backend_role == "assistant" else backend_role
452
+ conversation_history.append({
453
+ "role": sdk_role,
454
+ "content": content
455
+ })
456
+ else:
457
+ # Fallback: append output_message if history missing
458
+ output_msgs = result.get("output_message") or result.get("outputMessage")
459
+ if output_msgs:
460
+ if isinstance(output_msgs, list):
461
+ for om in output_msgs:
462
+ backend_role = om.get("role", "user")
463
+ if backend_role in ["system", "tool"]:
464
+ continue
465
+
466
+ content = om.get("content", "")
467
+ if not content and backend_role == "assistant":
468
+ continue
469
+
470
+ sdk_role = "user" if backend_role == "assistant" else backend_role
471
+ conversation_history.append({
472
+ "role": sdk_role,
473
+ "content": content
474
+ })
475
+ else:
476
+ backend_role = output_msgs.get("role", "user")
477
+ if backend_role not in ["system", "tool"]:
478
+ content = output_msgs.get("content", "")
479
+ if content or backend_role != "assistant":
480
+ sdk_role = "user" if backend_role == "assistant" else backend_role
481
+ conversation_history.append({
482
+ "role": sdk_role,
483
+ "content": content
484
+ })
485
+
486
+ turn_count += 1
487
+
488
+ # Only print success if the call didn't fail
489
+ if not agent_call_failed:
490
+ print(f"✓ Call Finished: {call_execution_id} ({turn_count} turns)")
491
+
492
+ except Exception as e:
493
+ # Get detailed error message
494
+ error_msg = str(e)
495
+ if not error_msg:
496
+ error_msg = f"{type(e).__name__}: {repr(e)}"
497
+
498
+ # Log to both logger and console
499
+ logger.error(f"Call execution {call_execution_id} failed: {error_msg}", exc_info=True)
500
+ print(f"❌ Call execution {call_execution_id} failed: {error_msg}")
501
+
502
+ # Update call execution status to failed
503
+ try:
504
+ # Use "FAILED" (uppercase) to match Django model choices, and include error message as ended_reason
505
+ await self.api.update_call_execution_status(
506
+ call_execution_id,
507
+ "failed",
508
+ ended_reason=error_msg
509
+ )
510
+ except Exception as status_error:
511
+ # Don't let status update failure mask the original error
512
+ logger.warning(f"Failed to update call execution status for {call_execution_id}: {status_error}")
513
+ return None
514
+
515
+ def _normalize_callback(self, callback: Callable | AgentWrapper) -> AgentWrapper:
516
+ """Ensures we have a AgentWrapper instance."""
517
+ return wrap_agent(callback)