agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,340 @@
1
+ """Serving a real voice agent's tool calls from a generated world.
2
+
3
+ A hosted voice agent executes its tools by calling a webhook. So the whole integration is one
4
+ thing: stand up that webhook, and answer it from the world instead of from canned responses.
5
+
6
+ That single swap is what the environment was built for. The previous run's known issues were all
7
+ the same defect wearing different clothes:
8
+
9
+ - *"Mocked tools always succeed, including removing an item that was never added."*
10
+ - *"Mock responses do not vary by argument, so read-after-write flows are wrong."*
11
+ - *"World state does not change unless a scenario sets state_updates, which is often empty."*
12
+
13
+ A world that really holds rows and can really refuse answers all three, because the reply the
14
+ agent hears is produced by running the call rather than by looking it up.
15
+
16
+ Nothing here decides pass or fail. Grading reads the world afterwards and the calls this server
17
+ recorded, through the same sub-goal checks every other run uses.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ import logging
24
+ import os
25
+ import threading
26
+ from http.server import BaseHTTPRequestHandler, HTTPServer
27
+ from typing import Any, Mapping
28
+
29
+ from ..world.runtime import GeneratedWorld
30
+
31
+ logger = logging.getLogger(__name__)
32
+
33
+ VAPI_API = os.environ.get("VAPI_API_BASE_URL", "https://api.vapi.ai").rstrip("/")
34
+
35
+ # Vapi's edge rejects the default urllib User-Agent with a 403 that says nothing about why, while
36
+ # the identical request from curl succeeds. Sending one is the whole fix.
37
+ _AGENT = "alk-harness/0.1"
38
+
39
+
40
+ class WorldWebhook:
41
+ """The webhook a hosted agent calls, answered by a generated world.
42
+
43
+ One world at a time. ``bind`` swaps which world is live between scenarios, so the assistant
44
+ stays configured while every scenario still starts from its own restored copy.
45
+ """
46
+
47
+ def __init__(self, host: str | None = None, port: int | None = None) -> None:
48
+ # A container must bind 0.0.0.0 on a fixed port or nothing outside can
49
+ # be configured to call it; a laptop keeps loopback and an ephemeral port.
50
+ host = (
51
+ host
52
+ if host is not None
53
+ else os.environ.get("HARNESS_WEBHOOK_HOST", "127.0.0.1")
54
+ )
55
+ port = (
56
+ port
57
+ if port is not None
58
+ else int(os.environ.get("HARNESS_WEBHOOK_PORT", "0"))
59
+ )
60
+ self._world: GeneratedWorld | None = None
61
+ self._lock = threading.Lock()
62
+ try:
63
+ server = HTTPServer((host, port), _handler_for(self))
64
+ except OSError:
65
+ # A leftover server from an earlier run must not block this one; any free port works
66
+ # because the public URL is discovered after binding.
67
+ logger.warning("port %s busy, binding an ephemeral port instead", port)
68
+ server = HTTPServer((host, 0), _handler_for(self))
69
+ self._server = server
70
+ self.port = server.server_address[1]
71
+ self._thread = threading.Thread(target=server.serve_forever, daemon=True)
72
+
73
+ def start(self) -> "WorldWebhook":
74
+ self._thread.start()
75
+ logger.info("world webhook listening on port %s", self.port)
76
+ return self
77
+
78
+ def stop(self) -> None:
79
+ self._server.shutdown()
80
+ self._server.server_close()
81
+
82
+ def bind(self, world: GeneratedWorld) -> None:
83
+ """Make one world live. Its own call log is what grading reads afterwards."""
84
+ with self._lock:
85
+ self._world = world
86
+ world.reset()
87
+
88
+ @property
89
+ def calls(self) -> list[Any]:
90
+ with self._lock:
91
+ return list(self._world.calls) if self._world else []
92
+
93
+ def respond(
94
+ self,
95
+ name: str,
96
+ arguments: Mapping[str, Any],
97
+ *,
98
+ session_id: str = "harness",
99
+ ) -> str:
100
+ """Answer one tool call by running it.
101
+
102
+ A refusal is returned as the answer, not as an error: the agent has to hear "that item is
103
+ unavailable" and cope with it, which is the whole reason the world can say no. What it
104
+ must never hear is an acknowledgement for something that did not happen.
105
+ """
106
+ with self._lock:
107
+ world = self._world
108
+ if world is None:
109
+ return "the environment is not ready"
110
+
111
+ # Caller hydration is an adapter concern, not a contract tool. The local
112
+ # LiveKit worker normally gets this from its demo API before exposing any
113
+ # conversational tools. Resolve the same safe profile from the generated
114
+ # world without adding a call that scenario grading would mistake for an
115
+ # agent action.
116
+ if name == "lookup_rider_by_phone":
117
+ phone = str(arguments.get("phone") or "")
118
+ state = world.observe().state
119
+ user = next(
120
+ (
121
+ row
122
+ for row in state.get("users", [])
123
+ if str(row.get("phone")) == phone
124
+ ),
125
+ None,
126
+ )
127
+
128
+ def active_booking_ref() -> Any:
129
+ if user is None:
130
+ return None
131
+ active = [
132
+ row
133
+ for row in state.get("bookings", [])
134
+ if row.get("rider_id") == user.get("rider_id")
135
+ and str(row.get("status") or "").lower()
136
+ not in {"cancelled", "canceled", "completed"}
137
+ ]
138
+ active.sort(
139
+ key=lambda row: str(row.get("created_at") or ""), reverse=True
140
+ )
141
+ return active[0].get("booking_ref") if active else None
142
+
143
+ forward = getattr(world, "forward", None)
144
+ if callable(forward):
145
+ hydrated = forward(
146
+ name,
147
+ arguments,
148
+ record=False,
149
+ session_id=session_id,
150
+ )
151
+ if hydrated.ok:
152
+ result = hydrated.result
153
+ if isinstance(result, dict):
154
+ result = {**result, "booking_ref": active_booking_ref()}
155
+ return (
156
+ result
157
+ if isinstance(result, str)
158
+ else json.dumps(result, default=str)
159
+ )
160
+ return hydrated.error
161
+ if user is None:
162
+ return json.dumps({"rider_id": None, "phone": phone})
163
+ market = next(
164
+ (
165
+ row
166
+ for row in state.get("market_config", [])
167
+ if row.get("market") == user.get("default_market")
168
+ ),
169
+ {},
170
+ )
171
+ # A caller asking about or cancelling an existing ride does not know an internal
172
+ # booking reference. Real agent backends hydrate the active trip alongside ANI
173
+ # identity; expose the same seeded relationship from the generated world so the
174
+ # shipped agent can invoke its own reference-based tools without fixture leakage in
175
+ # the conversation.
176
+ return json.dumps(
177
+ {
178
+ **user,
179
+ "cash_supported_in_market": bool(market.get("cash_supported")),
180
+ "accessibility_needs": [],
181
+ "booking_ref": active_booking_ref(),
182
+ },
183
+ default=str,
184
+ )
185
+
186
+ # ``world.call`` distinguishes real dependency endpoints from actions executed inside
187
+ # the submitted worker and mirrored here only for evidence. Bypassing it through
188
+ # ``forward`` made a valid local action look like a missing HTTP endpoint and caused the
189
+ # agent to retry or abort after it had already updated its own state.
190
+ done = world.handle_tool_call({"name": name, "arguments": dict(arguments)})
191
+ if done is None:
192
+ return f"there is no tool called {name}"
193
+ return done.content or ("done" if done.success else "that could not be done")
194
+
195
+
196
+ def _handler_for(owner: "WorldWebhook"):
197
+ class Handler(BaseHTTPRequestHandler):
198
+ def log_message(self, *args: Any) -> None: # silence per-request stderr noise
199
+ return
200
+
201
+ def do_POST(self) -> None: # noqa: N802 - required name
202
+ length = int(self.headers.get("Content-Length") or 0)
203
+ raw = self.rfile.read(length) if length else b"{}"
204
+ try:
205
+ payload = json.loads(raw or b"{}")
206
+ except json.JSONDecodeError:
207
+ payload = {}
208
+
209
+ calls = tool_calls(payload)
210
+ session_id = self.headers.get("x-session-id") or "harness"
211
+ if not calls:
212
+ # An agent whose tools are its own HTTP API asks differently: the tool is the
213
+ # path and the body is the arguments, with the answer expected back plainly.
214
+ # Serving both shapes is what lets a world stand in for such an API without the
215
+ # agent being changed to suit us.
216
+ name = self.path.strip("/").split("?")[0]
217
+ if name:
218
+ answer = owner.respond(
219
+ name,
220
+ payload if isinstance(payload, dict) else {},
221
+ session_id=session_id,
222
+ )
223
+ plain = json.dumps(_as_body(answer)).encode()
224
+ self.send_response(200)
225
+ self.send_header("Content-Type", "application/json")
226
+ self.send_header("Content-Length", str(len(plain)))
227
+ self.end_headers()
228
+ self.wfile.write(plain)
229
+ return
230
+
231
+ results = [
232
+ {
233
+ "toolCallId": call_id,
234
+ "result": owner.respond(name, arguments, session_id=session_id),
235
+ }
236
+ for call_id, name, arguments in calls
237
+ ]
238
+ body = json.dumps({"results": results}).encode()
239
+ self.send_response(200)
240
+ self.send_header("Content-Type", "application/json")
241
+ self.send_header("Content-Length", str(len(body)))
242
+ self.end_headers()
243
+ self.wfile.write(body)
244
+
245
+ return Handler
246
+
247
+
248
+ def _as_body(answer: str) -> Any:
249
+ """A tool's answer as a JSON body, keeping structure when the handler produced any.
250
+
251
+ Handlers return text because that is what a spoken agent hears. An HTTP tool API expects an
252
+ object, so a JSON answer is passed through as itself and anything else is wrapped, rather
253
+ than a caller having to parse a string out of a string.
254
+ """
255
+ try:
256
+ parsed = json.loads(answer)
257
+ except (json.JSONDecodeError, TypeError):
258
+ return {"result": answer}
259
+ return parsed if isinstance(parsed, (dict, list)) else {"result": parsed}
260
+
261
+
262
+ def tool_calls(payload: Mapping[str, Any]) -> list[tuple[str, str, dict[str, Any]]]:
263
+ """Pull (id, name, arguments) out of a provider's tool-call webhook body."""
264
+ message = payload.get("message") or payload
265
+ raw = message.get("toolCalls") or message.get("toolCallList") or []
266
+ found: list[tuple[str, str, dict[str, Any]]] = []
267
+ for entry in raw if isinstance(raw, list) else []:
268
+ if not isinstance(entry, Mapping):
269
+ continue
270
+ function = entry.get("function") or {}
271
+ name = str(function.get("name") or entry.get("name") or "")
272
+ arguments = function.get("arguments") or entry.get("arguments") or {}
273
+ if isinstance(arguments, str):
274
+ try:
275
+ arguments = json.loads(arguments)
276
+ except json.JSONDecodeError:
277
+ arguments = {"_raw": arguments}
278
+ if name:
279
+ found.append((str(entry.get("id") or ""), name, dict(arguments)))
280
+ return found
281
+
282
+
283
+ def pointed_at(tools: list[dict[str, Any]], webhook_url: str) -> list[dict[str, Any]]:
284
+ """The agent's own tools, with only where they are answered changed.
285
+
286
+ The assistant under test already has its tools — the names, the arguments, the enums are the
287
+ agent's, defined by whoever built it. Redefining them here would mean testing an agent we
288
+ wrote rather than theirs, and any drift between the two would show up as a finding about
289
+ them. So nothing is rebuilt: the one thing that changes is the address the call goes to.
290
+ """
291
+ repointed: list[dict[str, Any]] = []
292
+ for tool in tools:
293
+ moved = json.loads(json.dumps(tool))
294
+ moved.setdefault("server", {})["url"] = f"{webhook_url.rstrip('/')}/tool"
295
+ repointed.append(moved)
296
+ return repointed
297
+
298
+
299
+ def fetch_assistant(assistant_id: str, api_key: str) -> dict[str, Any]:
300
+ """The assistant as it stands, so its own tools can be read rather than guessed."""
301
+ import urllib.request
302
+
303
+ request = urllib.request.Request(
304
+ f"{VAPI_API}/assistant/{assistant_id}",
305
+ headers={"Authorization": f"Bearer {api_key}", "User-Agent": _AGENT},
306
+ )
307
+ with urllib.request.urlopen(request, timeout=20) as answer:
308
+ return json.loads(answer.read())
309
+
310
+
311
+ def repoint_assistant(assistant_id: str, api_key: str, webhook_url: str) -> list[str]:
312
+ """Send the assistant's existing tool calls to our webhook. Returns the tools moved."""
313
+ import urllib.request
314
+
315
+ assistant = fetch_assistant(assistant_id, api_key)
316
+ tools = (assistant.get("model") or {}).get("tools") or []
317
+ if not tools:
318
+ raise RuntimeError(
319
+ f"assistant {assistant_id} has no tools, so there is nothing for the environment "
320
+ "to answer. It is the agent's own tools that get repointed, not ones we add."
321
+ )
322
+ model = json.loads(json.dumps(assistant.get("model") or {}))
323
+ model["tools"] = pointed_at(tools, webhook_url)
324
+
325
+ body = json.dumps({"model": model}).encode()
326
+ request = urllib.request.Request(
327
+ f"{VAPI_API}/assistant/{assistant_id}",
328
+ data=body,
329
+ method="PATCH",
330
+ headers={
331
+ "Authorization": f"Bearer {api_key}",
332
+ "Content-Type": "application/json",
333
+ "User-Agent": _AGENT,
334
+ },
335
+ )
336
+ with urllib.request.urlopen(request, timeout=20) as answer:
337
+ answer.read()
338
+ return [
339
+ str((one.get("function") or {}).get("name") or "") for one in model["tools"]
340
+ ]
@@ -0,0 +1,172 @@
1
+ """Runtime-provider boundary for ALK-owned test environments.
2
+
3
+ Providers decide *where* a sealed environment runs. They do not decide how an agent is
4
+ understood, how scenarios are written, or how results are graded. The local provider below
5
+ adapts the proven repository/Compose provisioner; the hosted sandbox implements the same port.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import asyncio
11
+ from enum import Enum
12
+ from pathlib import Path
13
+ from typing import Any, Protocol
14
+
15
+ from pydantic import BaseModel, Field, JsonValue
16
+
17
+ from .bundle import EnvironmentBundle
18
+
19
+
20
+ class RuntimeState(str, Enum):
21
+ PREPARING = "preparing"
22
+ READY = "ready"
23
+ UNHEALTHY = "unhealthy"
24
+ STOPPED = "stopped"
25
+
26
+
27
+ class RuntimeEndpoint(BaseModel):
28
+ capability: str
29
+ protocol: str
30
+ address: str
31
+ configuration_name: str | None = None
32
+ metadata: dict[str, JsonValue] = Field(default_factory=dict)
33
+
34
+
35
+ class EnvironmentRuntime(BaseModel):
36
+ runtime_id: str
37
+ provider: str
38
+ bundle_digest: str
39
+ state: RuntimeState
40
+ endpoints: dict[str, RuntimeEndpoint] = Field(default_factory=dict)
41
+ metadata: dict[str, JsonValue] = Field(default_factory=dict)
42
+
43
+
44
+ class RuntimeProvider(Protocol):
45
+ """Execution-location port implemented locally and by the hosted sandbox fleet."""
46
+
47
+ name: str
48
+
49
+ async def provision(
50
+ self,
51
+ bundle: EnvironmentBundle,
52
+ *,
53
+ source: Path,
54
+ work_directory: Path,
55
+ contract: Any | None = None,
56
+ ) -> EnvironmentRuntime: ...
57
+
58
+ async def reset(
59
+ self, runtime: EnvironmentRuntime, *, work_directory: Path
60
+ ) -> None: ...
61
+
62
+ async def healthy(
63
+ self, runtime: EnvironmentRuntime, *, work_directory: Path
64
+ ) -> bool: ...
65
+
66
+ async def close(
67
+ self, runtime: EnvironmentRuntime, *, work_directory: Path
68
+ ) -> None: ...
69
+
70
+
71
+ class LocalComposeRuntimeProvider:
72
+ """Run one repository environment as an isolated Docker Compose project.
73
+
74
+ The adapter delegates lifecycle mechanics to ``harness.provision``, which gives each run a
75
+ unique project, allocates ports, waits for declared health checks, fingerprints source reuse,
76
+ and removes volumes during cleanup. Only endpoint names and addresses cross this boundary;
77
+ resolved credentials remain process-local.
78
+ """
79
+
80
+ name = "local-compose"
81
+
82
+ async def provision(
83
+ self,
84
+ bundle: EnvironmentBundle,
85
+ *,
86
+ source: Path,
87
+ work_directory: Path,
88
+ contract: Any | None = None,
89
+ ) -> EnvironmentRuntime:
90
+ from .provision import provision
91
+
92
+ environment = await asyncio.to_thread(
93
+ provision, source, work_directory, contract
94
+ )
95
+ endpoints: dict[str, RuntimeEndpoint] = {}
96
+ overrides = dict(environment.overrides)
97
+ for capability, definition in bundle.capabilities.items():
98
+ address = ""
99
+ if definition.configuration_name:
100
+ address = overrides.get(definition.configuration_name, "")
101
+ if not address and len(overrides) == 1:
102
+ address = next(iter(overrides.values()))
103
+ discovered = next(
104
+ (
105
+ endpoint
106
+ for endpoint in environment.service_endpoints
107
+ if endpoint["service"] == definition.service
108
+ and endpoint["container_port"] == definition.container_port
109
+ ),
110
+ None,
111
+ )
112
+ if not address and discovered:
113
+ address = str(discovered["external_address"])
114
+ if address:
115
+ endpoints[capability] = RuntimeEndpoint(
116
+ capability=capability,
117
+ protocol=definition.protocol.value,
118
+ address=address,
119
+ configuration_name=definition.configuration_name,
120
+ metadata=(
121
+ {
122
+ "service": str(discovered["service"]),
123
+ "kind": str(discovered["kind"]),
124
+ }
125
+ if discovered
126
+ else {}
127
+ ),
128
+ )
129
+ return EnvironmentRuntime(
130
+ runtime_id=environment.project,
131
+ provider=self.name,
132
+ bundle_digest=bundle.digest,
133
+ state=RuntimeState.READY,
134
+ endpoints=endpoints,
135
+ metadata={
136
+ "services": environment.services,
137
+ "provision_seconds": environment.provision_seconds,
138
+ "managed": environment.managed,
139
+ },
140
+ )
141
+
142
+ async def reset(self, runtime: EnvironmentRuntime, *, work_directory: Path) -> None:
143
+ from .provision import reset
144
+
145
+ environment = await asyncio.to_thread(reset, work_directory)
146
+ runtime.state = (
147
+ RuntimeState.READY if environment.running else RuntimeState.UNHEALTHY
148
+ )
149
+
150
+ async def healthy(
151
+ self, runtime: EnvironmentRuntime, *, work_directory: Path
152
+ ) -> bool:
153
+ from .provision import healthy
154
+
155
+ is_healthy = await asyncio.to_thread(healthy, work_directory)
156
+ runtime.state = RuntimeState.READY if is_healthy else RuntimeState.UNHEALTHY
157
+ return is_healthy
158
+
159
+ async def close(self, runtime: EnvironmentRuntime, *, work_directory: Path) -> None:
160
+ from .provision import stop
161
+
162
+ await asyncio.to_thread(stop, work_directory)
163
+ runtime.state = RuntimeState.STOPPED
164
+
165
+
166
+ __all__ = [
167
+ "EnvironmentRuntime",
168
+ "LocalComposeRuntimeProvider",
169
+ "RuntimeEndpoint",
170
+ "RuntimeProvider",
171
+ "RuntimeState",
172
+ ]