agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,99 @@
1
+ from typing import List, Dict, Union, Any, Optional, Literal
2
+ from pydantic import BaseModel, Field
3
+ from abc import ABC, abstractmethod
4
+
5
+
6
+ ArtifactType = Literal[
7
+ "text",
8
+ "image",
9
+ "audio",
10
+ "video",
11
+ "screenshot",
12
+ "browser_dom",
13
+ "file",
14
+ "json",
15
+ "trace",
16
+ ]
17
+
18
+
19
+ class SimulationArtifact(BaseModel):
20
+ """
21
+ Modality-neutral artifact carried through a simulation.
22
+
23
+ Use `uri` or `path` for large media, `data` for small inline payloads, and
24
+ `metadata` for framework-specific details like sample rate, viewport, page
25
+ URL, or image dimensions.
26
+ """
27
+
28
+ type: ArtifactType
29
+ uri: Optional[str] = None
30
+ path: Optional[str] = None
31
+ data: Optional[Any] = None
32
+ mime_type: Optional[str] = None
33
+ role: Optional[str] = None
34
+ metadata: Dict[str, Any] = Field(default_factory=dict)
35
+
36
+
37
+ class SimulationEvent(BaseModel):
38
+ """Normalized event for tools, memory, browser/CUA actions, voice states, and framework spans."""
39
+
40
+ type: str
41
+ name: Optional[str] = None
42
+ payload: Dict[str, Any] = Field(default_factory=dict)
43
+ timestamp_ms: Optional[int] = None
44
+ metadata: Dict[str, Any] = Field(default_factory=dict)
45
+
46
+
47
+ class AgentInput(BaseModel):
48
+ """
49
+ Input data passed to the user's agent wrapper during a simulation step.
50
+ """
51
+ thread_id: str
52
+ messages: List[Dict[str, Any]] # Full conversation history: [{"role": "user", "content": "..."}]
53
+ new_message: Optional[Dict[str, Any]] = None # The latest message to respond to
54
+
55
+ # Metadata for execution context (useful for logging/debugging)
56
+ execution_id: Optional[str] = None
57
+ turn_index: Optional[int] = None
58
+ scenario_name: Optional[str] = None
59
+ persona: Optional[Dict[str, Any]] = None
60
+ situation: Optional[str] = None
61
+ expected_outcome: Optional[str] = None
62
+ modality: Optional[str] = None
63
+ artifacts: List[SimulationArtifact] = Field(default_factory=list)
64
+ events: List[SimulationEvent] = Field(default_factory=list)
65
+ memory: Dict[str, Any] = Field(default_factory=dict)
66
+ tools: List[Dict[str, Any]] = Field(default_factory=list)
67
+ metadata: Dict[str, Any] = Field(default_factory=dict)
68
+
69
+ class AgentResponse(BaseModel):
70
+ """
71
+ Standardized response from the user's agent.
72
+ """
73
+ content: str
74
+ tool_calls: Optional[List[Dict[str, Any]]] = None
75
+ tool_responses: Optional[List[Dict[str, Any]]] = None # Tool role messages with results
76
+ artifacts: List[SimulationArtifact] = Field(default_factory=list)
77
+ events: List[SimulationEvent] = Field(default_factory=list)
78
+ memory_updates: Optional[Dict[str, Any]] = None
79
+ state: Optional[Dict[str, Any]] = None
80
+ metadata: Optional[Dict[str, Any]] = None
81
+
82
+ class AgentWrapper(ABC):
83
+ """
84
+ Base class for wrapping user agents to work with the simulation SDK.
85
+ Users should implement the `call` method.
86
+ """
87
+
88
+ @abstractmethod
89
+ async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
90
+ """
91
+ Process the input and return the agent's response.
92
+
93
+ Args:
94
+ input: The AgentInput object containing message history and context.
95
+
96
+ Returns:
97
+ A string (content only) or AgentResponse object.
98
+ """
99
+ pass
@@ -0,0 +1,18 @@
1
+ from fi.simulate.agent.wrappers.openai import OpenAIAgentWrapper
2
+ from fi.simulate.agent.wrappers.langchain import LangChainAgentWrapper
3
+ from fi.simulate.agent.wrappers.gemini import GeminiAgentWrapper
4
+ from fi.simulate.agent.wrappers.anthropic import AnthropicAgentWrapper
5
+ from fi.simulate.agent.wrappers.http import HTTPAgentWrapper
6
+ from fi.simulate.agent.wrappers.websocket import WebSocketAgentWrapper
7
+
8
+ OpenAICompatibleHTTPAgentWrapper = HTTPAgentWrapper
9
+
10
+ __all__ = [
11
+ "OpenAIAgentWrapper",
12
+ "LangChainAgentWrapper",
13
+ "GeminiAgentWrapper",
14
+ "AnthropicAgentWrapper",
15
+ "HTTPAgentWrapper",
16
+ "OpenAICompatibleHTTPAgentWrapper",
17
+ "WebSocketAgentWrapper",
18
+ ]
@@ -0,0 +1,62 @@
1
+ from typing import Any, Union
2
+ from fi.simulate.agent.wrapper import AgentWrapper, AgentInput, AgentResponse
3
+
4
+ class AnthropicAgentWrapper(AgentWrapper):
5
+ """
6
+ Wrapper for Anthropic (Claude) agents.
7
+ Automatically handles message conversion to Anthropic format.
8
+ """
9
+ def __init__(self, client: Any, model: str = "claude-sonnet-4-5-20250929", system_prompt: str = None, max_tokens: int = 1024):
10
+ """
11
+ Args:
12
+ client: The Anthropic client instance (AsyncAnthropic or Anthropic).
13
+ model: The model name to use.
14
+ system_prompt: Optional system instructions for the agent.
15
+ max_tokens: Maximum number of tokens to generate (default: 1024).
16
+ """
17
+ self.client = client
18
+ self.model = model
19
+ self.system_prompt = system_prompt
20
+ self.max_tokens = max_tokens
21
+
22
+ async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
23
+ # Convert internal message format to Anthropic format
24
+ # Anthropic messages API expects: [{"role": "user"|"assistant", "content": "..."}]
25
+ # It does NOT support "system" role in the messages list; system prompt is a top-level param.
26
+
27
+ messages = []
28
+ # Use configured system prompt by default
29
+ system_prompt = self.system_prompt
30
+
31
+ for msg in input.messages:
32
+ if msg["role"] == "system":
33
+ # If history has system message (unlikely due to filtering), it overrides?
34
+ # Or we ignore it to respect wrapper config?
35
+ # Let's check if it exists and use it if self.system_prompt is None
36
+ if system_prompt is None:
37
+ system_prompt = msg["content"]
38
+ else:
39
+ messages.append({
40
+ "role": msg["role"],
41
+ "content": msg["content"]
42
+ })
43
+
44
+ # Check for AsyncAnthropic vs Sync
45
+ # Heuristic: check for 'messages.create' and if client class name contains Async
46
+ is_async = type(self.client).__name__.startswith("Async")
47
+
48
+ kwargs = {
49
+ "model": self.model,
50
+ "max_tokens": self.max_tokens,
51
+ "messages": messages
52
+ }
53
+ if system_prompt:
54
+ kwargs["system"] = system_prompt
55
+
56
+ if is_async:
57
+ message = await self.client.messages.create(**kwargs)
58
+ else:
59
+ message = self.client.messages.create(**kwargs)
60
+
61
+ return message.content[0].text
62
+
@@ -0,0 +1,65 @@
1
+ from typing import Any, Union
2
+ from fi.simulate.agent.wrapper import AgentWrapper, AgentInput, AgentResponse
3
+
4
+ class GeminiAgentWrapper(AgentWrapper):
5
+ """
6
+ Wrapper for Google Gemini (Generative AI) agents.
7
+ Supports google-generativeai SDK.
8
+ """
9
+ def __init__(self, model: Any, system_prompt: str = None):
10
+ """
11
+ Args:
12
+ model: An instance of google.generativeai.GenerativeModel
13
+ system_prompt: Optional system instructions.
14
+ Note: Ideally configure system_instruction on the model itself.
15
+ If provided here, it will be prepended as a user message.
16
+ """
17
+ self.model = model
18
+ self.system_prompt = system_prompt
19
+
20
+ async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
21
+ # Convert internal messages to Gemini format (Content objects)
22
+ # Note: Gemini SDK manages chat history via ChatSession usually,
23
+ # but for stateless call we pass full history if supported,
24
+ # or we might need to reconstruct a chat session.
25
+
26
+ # Simple reconstruction of history for a chat session
27
+ history = []
28
+
29
+ if self.system_prompt:
30
+ # Prepend system prompt as a user message for context
31
+ history.append({"role": "user", "parts": [f"System Instruction: {self.system_prompt}"]})
32
+ # Add a dummy model acknowledgement to keep turns valid (User -> Model -> User)
33
+ history.append({"role": "model", "parts": ["Understood."]})
34
+
35
+ for msg in input.messages:
36
+ role = "user" if msg["role"] == "user" else "model"
37
+ content = msg["content"]
38
+
39
+ # Gemini typically expects history excluding the last message which is passed to send_message
40
+ history.append({"role": role, "parts": [content]})
41
+
42
+ if not history:
43
+ raise ValueError("No messages provided to Gemini wrapper")
44
+
45
+ # The last user message is the prompt
46
+ last_turn = history.pop()
47
+ if last_turn["role"] != "user":
48
+ # If the last message wasn't user, something is weird in the flow,
49
+ # but we can try to send empty or handle it.
50
+ # Ideally simulator sends User message last.
51
+ prompt = ""
52
+ else:
53
+ prompt = last_turn["parts"][0]
54
+
55
+ # Start a chat with the history
56
+ chat = self.model.start_chat(history=history)
57
+
58
+ # Check if async generation is supported (google-generativeai >= 0.3.0 has send_message_async)
59
+ if hasattr(chat, "send_message_async"):
60
+ response = await chat.send_message_async(prompt)
61
+ else:
62
+ # Fallback to sync
63
+ response = chat.send_message(prompt)
64
+
65
+ return response.text
@@ -0,0 +1,404 @@
1
+ from __future__ import annotations
2
+
3
+ import asyncio
4
+ import json
5
+ import os
6
+ import time
7
+ import urllib.error
8
+ import urllib.request
9
+ from typing import Any, Mapping, Optional, Sequence
10
+ from urllib.parse import urlparse
11
+
12
+ from fi.simulate.agent.wrapper import (
13
+ AgentInput,
14
+ AgentResponse,
15
+ SimulationArtifact,
16
+ SimulationEvent,
17
+ )
18
+ from fi.simulate.agent.wrapper import AgentWrapper
19
+
20
+
21
+ class HTTPAgentWrapper(AgentWrapper):
22
+ """HTTP/OpenAI-compatible target adapter for external agent simulation."""
23
+
24
+ def __init__(
25
+ self,
26
+ *,
27
+ endpoint: str,
28
+ protocol: str = "fi.alk",
29
+ model: Optional[str] = None,
30
+ api_key: Optional[str] = None,
31
+ api_key_env: Optional[str] = None,
32
+ headers: Optional[Mapping[str, str]] = None,
33
+ timeout: float = 30.0,
34
+ include_tools: bool = True,
35
+ system_prompt: Optional[str] = None,
36
+ metadata: Optional[Mapping[str, Any]] = None,
37
+ ) -> None:
38
+ if not endpoint:
39
+ raise ValueError("endpoint is required")
40
+ self.endpoint = endpoint
41
+ self.protocol = _normalize_protocol(protocol)
42
+ self.model = model
43
+ self.api_key = api_key
44
+ self.api_key_env = api_key_env
45
+ self.headers = {str(k): str(v) for k, v in dict(headers or {}).items()}
46
+ self.timeout = float(timeout)
47
+ self.include_tools = bool(include_tools)
48
+ self.system_prompt = system_prompt
49
+ self.metadata = dict(metadata or {})
50
+
51
+ async def call(self, input: AgentInput) -> AgentResponse:
52
+ started = time.time()
53
+ request_payload = self._request_payload(input)
54
+ headers = self._request_headers()
55
+ status_code = 0
56
+ response_payload: dict[str, Any] = {}
57
+ error: Optional[str] = None
58
+ try:
59
+ status_code, response_payload = await asyncio.to_thread(
60
+ self._post_json,
61
+ request_payload,
62
+ headers,
63
+ )
64
+ if status_code >= 400:
65
+ error = _response_error_text(response_payload) or (
66
+ f"HTTP target returned status {status_code}"
67
+ )
68
+ response = self._agent_response_from_payload(response_payload)
69
+ except Exception as exc:
70
+ error = str(exc)
71
+ response = AgentResponse(content=f"HTTP target failed: {exc}")
72
+
73
+ latency_ms = round((time.time() - started) * 1000, 4)
74
+ trace = {
75
+ "kind": "external_agent_http_trace",
76
+ "protocol": self.protocol,
77
+ "endpoint": _redacted_endpoint(self.endpoint),
78
+ "endpoint_host": urlparse(self.endpoint).netloc,
79
+ "model": self.model,
80
+ "status_code": status_code,
81
+ "latency_ms": latency_ms,
82
+ "request_message_count": len(input.messages),
83
+ "request_tool_count": len(input.tools) if self.include_tools else 0,
84
+ "response_tool_call_count": len(response.tool_calls or []),
85
+ "success": error is None and 200 <= status_code < 300,
86
+ "request_header_names": sorted(headers),
87
+ "auth": {
88
+ "mode": "bearer" if self._resolved_api_key() else "none",
89
+ "api_key_env": self.api_key_env,
90
+ "redacted": bool(self._resolved_api_key()),
91
+ },
92
+ "error": error,
93
+ **self.metadata,
94
+ }
95
+ response.events.append(
96
+ SimulationEvent(
97
+ type="external_agent",
98
+ name="external_agent_http_call",
99
+ payload=trace,
100
+ )
101
+ )
102
+ response.artifacts.append(
103
+ SimulationArtifact(
104
+ type="trace",
105
+ role="agent",
106
+ data=trace,
107
+ metadata={"kind": "external_agent_http_trace"},
108
+ )
109
+ )
110
+ state = dict(response.state or {})
111
+ state["external_agent"] = trace
112
+ state["external_agent_trace"] = trace
113
+ response.state = state
114
+ metadata = dict(response.metadata or {})
115
+ metadata["external_agent"] = trace
116
+ metadata["external_agent_trace"] = trace
117
+ response.metadata = metadata
118
+ return response
119
+
120
+ def _request_payload(self, input: AgentInput) -> dict[str, Any]:
121
+ messages = _messages_for_protocol(input.messages, self.protocol)
122
+ if self.system_prompt:
123
+ messages = [{"role": "system", "content": self.system_prompt}, *messages]
124
+ if self.protocol == "openai_chat":
125
+ payload: dict[str, Any] = {
126
+ "model": self.model or "agent-learning-target",
127
+ "messages": messages,
128
+ }
129
+ if self.include_tools and input.tools:
130
+ payload["tools"] = [_openai_tool_spec(tool) for tool in input.tools]
131
+ payload["tool_choice"] = "auto"
132
+ return payload
133
+ return {
134
+ "thread_id": input.thread_id,
135
+ "execution_id": input.execution_id,
136
+ "turn_index": input.turn_index,
137
+ "scenario_name": input.scenario_name,
138
+ "persona": input.persona,
139
+ "situation": input.situation,
140
+ "expected_outcome": input.expected_outcome,
141
+ "messages": messages,
142
+ "new_message": input.new_message,
143
+ "tools": list(input.tools) if self.include_tools else [],
144
+ "metadata": input.metadata,
145
+ }
146
+
147
+ def _request_headers(self) -> dict[str, str]:
148
+ headers = {"Content-Type": "application/json", **self.headers}
149
+ api_key = self._resolved_api_key()
150
+ if api_key and not any(key.lower() == "authorization" for key in headers):
151
+ headers["Authorization"] = f"Bearer {api_key}"
152
+ return headers
153
+
154
+ def _resolved_api_key(self) -> str:
155
+ if self.api_key not in (None, ""):
156
+ return str(self.api_key)
157
+ if self.api_key_env:
158
+ return os.environ.get(self.api_key_env, "")
159
+ return ""
160
+
161
+ def _post_json(
162
+ self,
163
+ payload: Mapping[str, Any],
164
+ headers: Mapping[str, str],
165
+ ) -> tuple[int, dict[str, Any]]:
166
+ body = json.dumps(payload, default=str).encode("utf-8")
167
+ request = urllib.request.Request(
168
+ self.endpoint,
169
+ data=body,
170
+ headers=dict(headers),
171
+ method="POST",
172
+ )
173
+ try:
174
+ with urllib.request.urlopen(request, timeout=self.timeout) as response:
175
+ status = int(getattr(response, "status", 200))
176
+ text = response.read().decode("utf-8")
177
+ except urllib.error.HTTPError as exc:
178
+ status = int(exc.code)
179
+ text = exc.read().decode("utf-8")
180
+ if not text:
181
+ return status, {}
182
+ try:
183
+ parsed = json.loads(text)
184
+ except json.JSONDecodeError as exc:
185
+ raise ValueError(f"HTTP target returned non-JSON response: {exc}") from exc
186
+ if not isinstance(parsed, dict):
187
+ raise ValueError("HTTP target response must be a JSON object")
188
+ return status, parsed
189
+
190
+ def _agent_response_from_payload(self, payload: Mapping[str, Any]) -> AgentResponse:
191
+ if self.protocol == "openai_chat":
192
+ message = _openai_message(payload)
193
+ return AgentResponse(
194
+ content=_content_text(message.get("content")),
195
+ tool_calls=_openai_tool_calls(message.get("tool_calls")),
196
+ metadata={
197
+ "finish_reason": _openai_finish_reason(payload),
198
+ "usage": dict(payload.get("usage") or {}),
199
+ },
200
+ )
201
+ return AgentResponse(
202
+ content=_content_text(payload.get("content") or payload.get("message")),
203
+ tool_calls=_tool_call_list(payload.get("tool_calls")),
204
+ tool_responses=_tool_response_list(payload.get("tool_responses")),
205
+ artifacts=_artifact_list(payload.get("artifacts")),
206
+ events=_event_list(payload.get("events")),
207
+ memory_updates=_optional_mapping(payload.get("memory_updates")),
208
+ state=_optional_mapping(payload.get("state")),
209
+ metadata=_optional_mapping(payload.get("metadata")),
210
+ )
211
+
212
+
213
+ def _normalize_protocol(value: str) -> str:
214
+ protocol = str(value or "fi.alk").lower().replace("-", "_")
215
+ aliases = {
216
+ "openai": "openai_chat",
217
+ "openai_compatible": "openai_chat",
218
+ "chat_completions": "openai_chat",
219
+ "agent_learning_http": "fi.alk",
220
+ "http": "fi.alk",
221
+ }
222
+ protocol = aliases.get(protocol, protocol)
223
+ if protocol not in {"fi.alk", "openai_chat"}:
224
+ raise ValueError("protocol must be one of: fi.alk, openai_chat")
225
+ return protocol
226
+
227
+
228
+ def _messages_for_protocol(
229
+ messages: Sequence[Mapping[str, Any]], protocol: str
230
+ ) -> list[dict[str, Any]]:
231
+ """Encode canonical ALK history for the selected external-agent protocol.
232
+
233
+ ALK keeps tool calls structured internally. OpenAI-compatible endpoints require that array
234
+ unchanged, while the FutureAGI callback ``AgentInput`` schema represents historical
235
+ ``tool_calls`` as a JSON string. Normalizing here keeps the conversation runner generic and
236
+ prevents a retry from accidentally executing a side-effecting tool twice.
237
+ """
238
+ normalized = [dict(message) for message in messages]
239
+ if protocol != "fi.alk":
240
+ return normalized
241
+ for message in normalized:
242
+ tool_calls = message.get("tool_calls")
243
+ if isinstance(tool_calls, Sequence) and not isinstance(
244
+ tool_calls, (str, bytes)
245
+ ):
246
+ message["tool_calls"] = json.dumps(
247
+ list(tool_calls), separators=(",", ":"), default=str
248
+ )
249
+ return normalized
250
+
251
+
252
+ def _openai_tool_spec(tool: Mapping[str, Any]) -> dict[str, Any]:
253
+ # Accept both the flat SDK tool shape ({name, description, parameters}) and
254
+ # the OpenAI-nested shape ({"type": "function", "function": {...}}). Without
255
+ # reading the nested ``function`` block, a nested spec loses its name and the
256
+ # model is handed a tool literally called "tool" — so it can never call the
257
+ # real tool and the environment's mock never matches.
258
+ fn = tool.get("function") if isinstance(tool.get("function"), Mapping) else {}
259
+ name = str(
260
+ tool.get("name")
261
+ or fn.get("name")
262
+ or tool.get("tool")
263
+ or tool.get("id")
264
+ or "tool"
265
+ )
266
+ parameters = tool.get("parameters")
267
+ if not isinstance(parameters, Mapping):
268
+ parameters = fn.get("parameters")
269
+ if not isinstance(parameters, Mapping):
270
+ parameters = {"type": "object", "properties": {}}
271
+ description = tool.get("description") or fn.get("description") or f"Tool {name}"
272
+ return {
273
+ "type": "function",
274
+ "function": {
275
+ "name": name,
276
+ "description": str(description),
277
+ "parameters": dict(parameters),
278
+ },
279
+ }
280
+
281
+
282
+ def _openai_message(payload: Mapping[str, Any]) -> dict[str, Any]:
283
+ choices = payload.get("choices")
284
+ if isinstance(choices, Sequence) and not isinstance(choices, (str, bytes)):
285
+ if choices:
286
+ choice = choices[0]
287
+ if isinstance(choice, Mapping):
288
+ message = choice.get("message")
289
+ if isinstance(message, Mapping):
290
+ return dict(message)
291
+ message = payload.get("message")
292
+ return dict(message) if isinstance(message, Mapping) else dict(payload)
293
+
294
+
295
+ def _openai_finish_reason(payload: Mapping[str, Any]) -> Optional[str]:
296
+ choices = payload.get("choices")
297
+ if isinstance(choices, Sequence) and not isinstance(choices, (str, bytes)):
298
+ if choices and isinstance(choices[0], Mapping):
299
+ value = choices[0].get("finish_reason")
300
+ return str(value) if value is not None else None
301
+ return None
302
+
303
+
304
+ def _openai_tool_calls(value: Any) -> list[dict[str, Any]]:
305
+ calls = _tool_call_list(value)
306
+ normalized: list[dict[str, Any]] = []
307
+ for index, call in enumerate(calls, start=1):
308
+ function = call.get("function")
309
+ if isinstance(function, Mapping):
310
+ name = function.get("name")
311
+ arguments = function.get("arguments", {})
312
+ else:
313
+ name = call.get("name") or call.get("tool")
314
+ arguments = call.get("arguments", call.get("args", {}))
315
+ normalized.append(
316
+ {
317
+ "id": str(call.get("id") or f"call_{index}"),
318
+ "type": str(call.get("type") or "function"),
319
+ "function": {
320
+ "name": str(name or ""),
321
+ "arguments": (
322
+ arguments
323
+ if isinstance(arguments, str)
324
+ else json.dumps(arguments or {}, default=str)
325
+ ),
326
+ },
327
+ }
328
+ )
329
+ return normalized
330
+
331
+
332
+ def _tool_call_list(value: Any) -> list[dict[str, Any]]:
333
+ if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
334
+ return []
335
+ return [dict(item) for item in value if isinstance(item, Mapping)]
336
+
337
+
338
+ def _tool_response_list(value: Any) -> list[dict[str, Any]]:
339
+ if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
340
+ return []
341
+ return [dict(item) for item in value if isinstance(item, Mapping)]
342
+
343
+
344
+ def _artifact_list(value: Any) -> list[SimulationArtifact]:
345
+ if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
346
+ return []
347
+ artifacts: list[SimulationArtifact] = []
348
+ for item in value:
349
+ if isinstance(item, SimulationArtifact):
350
+ artifacts.append(item)
351
+ elif isinstance(item, Mapping):
352
+ artifacts.append(SimulationArtifact(**dict(item)))
353
+ return artifacts
354
+
355
+
356
+ def _event_list(value: Any) -> list[SimulationEvent]:
357
+ if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
358
+ return []
359
+ events: list[SimulationEvent] = []
360
+ for item in value:
361
+ if isinstance(item, SimulationEvent):
362
+ events.append(item)
363
+ elif isinstance(item, Mapping):
364
+ events.append(SimulationEvent(**dict(item)))
365
+ return events
366
+
367
+
368
+ def _optional_mapping(value: Any) -> Optional[dict[str, Any]]:
369
+ return dict(value) if isinstance(value, Mapping) else None
370
+
371
+
372
+ def _content_text(value: Any) -> str:
373
+ if isinstance(value, str):
374
+ return value
375
+ if isinstance(value, Sequence) and not isinstance(value, (str, bytes)):
376
+ parts: list[str] = []
377
+ for item in value:
378
+ if isinstance(item, Mapping):
379
+ text = item.get("text") or item.get("content") or item.get("refusal")
380
+ if text not in (None, ""):
381
+ parts.append(str(text))
382
+ elif item not in (None, ""):
383
+ parts.append(str(item))
384
+ return "\n".join(parts)
385
+ return "" if value is None else str(value)
386
+
387
+
388
+ def _response_error_text(payload: Mapping[str, Any]) -> str:
389
+ error = payload.get("error")
390
+ if isinstance(error, Mapping):
391
+ return _content_text(error.get("message") or error.get("detail") or error)
392
+ if error not in (None, ""):
393
+ return _content_text(error)
394
+ for key in ("detail", "message", "status"):
395
+ if payload.get(key) not in (None, ""):
396
+ return _content_text(payload.get(key))
397
+ return ""
398
+
399
+
400
+ def _redacted_endpoint(endpoint: str) -> str:
401
+ parsed = urlparse(endpoint)
402
+ if not parsed.query:
403
+ return endpoint
404
+ return parsed._replace(query="<redacted>").geturl()