agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,457 @@
1
+ """A harness backend that runs stages on Gemini through Vertex AI, on Google's ADK.
2
+
3
+ The Agent Development Kit is Google's counterpart to the Claude Agent SDK: it owns the agentic
4
+ loop, executes tools, and holds session history, the way this harness expects a backend to. We
5
+ adapt at the same seam as the Claude backend and nothing more: a ``ToolSpec`` becomes an ADK
6
+ tool through ADK's own extension point (``BaseTool`` with an explicit declaration), and ADK's
7
+ event stream is translated into the neutral reply vocabulary. The loop itself is not ours.
8
+
9
+ The Gemini 3.x models this exists for are served from the ``global`` endpoint only, which is
10
+ why ``ALK_VERTEX_LOCATION`` defaults to ``global`` rather than to a region. Regional Vertex
11
+ deployments of older models can point it elsewhere.
12
+
13
+ Read, Glob and Grep come from files.py when a stage grants them. AskUserQuestion is not
14
+ implemented here yet: unattended runs never use it, and an attended run on this backend simply
15
+ proceeds without the option, which is said out loud in the session rather than hidden.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ import logging
22
+ import os
23
+ import uuid
24
+ from datetime import date
25
+ from typing import Any, AsyncIterator
26
+
27
+ from .base import (
28
+ FILE_TOOLS,
29
+ Call,
30
+ ModelReply,
31
+ Say,
32
+ SessionOpened,
33
+ SessionSpec,
34
+ StageDone,
35
+ ToolReturned,
36
+ ToolSpec,
37
+ qualified,
38
+ )
39
+ from .files import file_tools
40
+
41
+ DEFAULT_MODEL = "gemini-3.7-flash"
42
+
43
+ _TERMINAL_SAVE_TOOLS = frozenset(
44
+ {
45
+ "mcp__world__save_world",
46
+ "mcp__provision__save_environment",
47
+ "mcp__scenarios__save_scenarios",
48
+ # Source-data review is another persisted authoring boundary. Its handler only
49
+ # succeeds after every scenario was reviewed and at least one executable invariant
50
+ # was declared. Letting ADK take another turn after that success can burn the entire
51
+ # call budget and turn a completed review into a spurious validation failure.
52
+ "mcp__source_data__finish_review",
53
+ }
54
+ )
55
+
56
+ # Vertex list pricing per 1M tokens: (input, output, the day this pair was last checked against
57
+ # the platform's litellm model table). An unknown or stale model reports no cost rather than a
58
+ # wrong one, and shows up in `unpriced_turns`.
59
+ PRICES_PER_MILLION = {
60
+ "gemini-3.8-flash": (0.75, 3.75, "2026-12-31"),
61
+ "gemini-3.7-flash": (0.75, 3.75, "2026-12-31"),
62
+ "gemini-3.6-flash": (0.75, 3.75, "2026-12-31"),
63
+ "gemini-3.5-transcribe-preview": (2.5, 12, "2026-12-31"),
64
+ "gemini-3.5-transcribe-live-preview": (3.5, 21, "2026-12-31"),
65
+ "gemini-3.5-flash-lite": (0.3, 2.5, "2026-12-31"),
66
+ "gemini-3.5-flash": (1.5, 9, "2026-12-31"),
67
+ "gemini-3.1-pro-preview-customtools": (2, 12, "2026-12-31"),
68
+ "gemini-3.1-pro-preview": (2, 12, "2026-12-31"),
69
+ "gemini-3.1-flash-lite-preview": (0.25, 1.5, "2026-12-31"),
70
+ "gemini-3.1-flash-lite-image": (0.25, 1.5, "2026-12-31"),
71
+ "gemini-3.1-flash-lite": (0.25, 1.5, "2026-12-31"),
72
+ "gemini-3.1-flash-image-preview": (0.5, 3, "2026-12-31"),
73
+ "gemini-3.1-flash-image": (0.5, 3, "2026-12-31"),
74
+ "gemini-3-pro-preview": (2, 12, "2026-12-31"),
75
+ "gemini-3-pro-image-preview": (2, 12, "2026-12-31"),
76
+ "gemini-3-pro-image": (2, 12, "2026-12-31"),
77
+ "gemini-3-flash-preview": (0.5, 3, "2026-12-31"),
78
+ }
79
+
80
+ _PYTHON_TYPES = {
81
+ str: "string",
82
+ int: "integer",
83
+ float: "number",
84
+ bool: "boolean",
85
+ list: "array",
86
+ dict: "object",
87
+ }
88
+
89
+ # The keys Vertex's Schema type accepts. JSON Schema carries more; anything else is dropped
90
+ # rather than passed through, because an unknown key fails declaration validation and takes
91
+ # the whole stage down before its first turn.
92
+ _SCHEMA_KEYS = (
93
+ "description",
94
+ "enum",
95
+ "format",
96
+ "items",
97
+ "maximum",
98
+ "maxItems",
99
+ "minimum",
100
+ "minItems",
101
+ "nullable",
102
+ "pattern",
103
+ "properties",
104
+ "required",
105
+ "type",
106
+ "anyOf",
107
+ )
108
+
109
+
110
+ def _gemini_schema(schema: Any) -> dict[str, Any]:
111
+ """A JSON Schema fragment as Vertex's Schema dialect.
112
+
113
+ The real difference is nullability: JSON Schema says ``"type": ["string", "null"]``,
114
+ Vertex says ``"type": "string", "nullable": true``. Everything Vertex does not know is
115
+ dropped, recursively, so a tool schema written for the loosest backend still declares.
116
+ """
117
+ if not isinstance(schema, dict):
118
+ return {"type": "string"}
119
+ cleaned: dict[str, Any] = {}
120
+ for key, value in schema.items():
121
+ if key not in _SCHEMA_KEYS:
122
+ continue
123
+ if key == "type" and isinstance(value, list):
124
+ bare = [entry for entry in value if entry != "null"]
125
+ cleaned["type"] = bare[0] if bare else "string"
126
+ if "null" in value:
127
+ cleaned["nullable"] = True
128
+ elif key == "properties" and isinstance(value, dict):
129
+ cleaned["properties"] = {
130
+ name: _gemini_schema(inner) for name, inner in value.items()
131
+ }
132
+ elif key == "items":
133
+ cleaned["items"] = _gemini_schema(value)
134
+ elif key == "anyOf" and isinstance(value, list):
135
+ cleaned["anyOf"] = [_gemini_schema(inner) for inner in value]
136
+ elif key == "enum" and isinstance(value, list):
137
+ # Vertex enums are strings; None inside one is JSON Schema's way of saying
138
+ # nullable, and everything else is stringified the way the model will echo it.
139
+ cleaned["enum"] = [str(entry) for entry in value if entry is not None]
140
+ if None in value:
141
+ cleaned["nullable"] = True
142
+ else:
143
+ cleaned[key] = value
144
+ # Vertex refuses an array that does not say what it holds. JSON Schema treats items as
145
+ # optional, and most such arrays here carry row-shaped dicts, so an open object is the
146
+ # faithful default.
147
+ if cleaned.get("type") == "array" and "items" not in cleaned:
148
+ cleaned["items"] = {"type": "object"}
149
+ return cleaned
150
+
151
+
152
+ def _json_schema(schema: Any) -> dict[str, Any]:
153
+ """The tool's schema as Vertex-safe schema, whichever shorthand it was declared in."""
154
+ if isinstance(schema, dict) and (
155
+ "properties" in schema or schema.get("type") == "object"
156
+ ):
157
+ return _gemini_schema(schema)
158
+ if isinstance(schema, dict):
159
+ return {
160
+ "type": "object",
161
+ "properties": {
162
+ name: (
163
+ {"type": "array", "items": {"type": "object"}}
164
+ if kind is list
165
+ else {"type": _PYTHON_TYPES.get(kind, "string")}
166
+ )
167
+ for name, kind in schema.items()
168
+ },
169
+ "required": list(schema),
170
+ }
171
+ return {"type": "object", "properties": {}}
172
+
173
+
174
+ def _project() -> str:
175
+ named = os.environ.get("GOOGLE_CLOUD_PROJECT")
176
+ if named:
177
+ return named
178
+ credentials = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS")
179
+ if credentials:
180
+ try:
181
+ with open(credentials, encoding="utf-8") as handle:
182
+ found = json.load(handle).get("project_id", "")
183
+ if found:
184
+ return found
185
+ except (OSError, ValueError):
186
+ pass
187
+ # Application Default Credentials already carry a project on a machine that has run
188
+ # ``gcloud auth application-default login`` or that runs on Google infrastructure. Asking
189
+ # for it again as an environment variable is a setting the operator does not need to know.
190
+ try:
191
+ import google.auth
192
+
193
+ _, discovered = google.auth.default()
194
+ if discovered:
195
+ return str(discovered)
196
+ except Exception: # noqa: BLE001 - fall through to the explicit instruction below
197
+ pass
198
+ raise RuntimeError(
199
+ "no GCP project named; run 'gcloud auth application-default login', or set "
200
+ "GOOGLE_CLOUD_PROJECT, or point GOOGLE_APPLICATION_CREDENTIALS at a service-account file"
201
+ )
202
+
203
+
204
+ def _location() -> str:
205
+ return os.environ.get("ALK_VERTEX_LOCATION", "global").strip() or "global"
206
+
207
+
208
+ def _flattened(result: Any) -> str:
209
+ content = result.get("content") if isinstance(result, dict) else None
210
+ if isinstance(content, list):
211
+ return "\n".join(
212
+ part.get("text", "") for part in content if isinstance(part, dict)
213
+ )
214
+ return content if isinstance(content, str) else str(result)
215
+
216
+
217
+ def _successful_terminal_save(name: str, response: Any) -> bool:
218
+ """Whether a tool response proves this authoring stage has persisted its final output.
219
+
220
+ Save tools deliberately reject incomplete work with ``is_error`` so the model can repair and
221
+ retry. Once one succeeds, another model turn can only rewrite already-valid output or burn the
222
+ stage budget; the persisted artifact is the stage's actual completion boundary.
223
+ """
224
+ return bool(
225
+ name in _TERMINAL_SAVE_TOOLS
226
+ and isinstance(response, dict)
227
+ and not response.get("is_error")
228
+ )
229
+
230
+
231
+ def _spec_tool(name: str, spec: ToolSpec) -> Any:
232
+ """A ToolSpec as an ADK tool, through ADK's own extension point.
233
+
234
+ ``BaseTool`` with an explicit ``_get_declaration`` is how ADK says a tool whose contract
235
+ is defined elsewhere should be wrapped; the handler runs unchanged and ADK owns calling
236
+ it, retrying the turn, and feeding the result back.
237
+ """
238
+ from google.adk.tools import BaseTool
239
+ from google.genai import types
240
+
241
+ class SpecTool(BaseTool):
242
+ def __init__(self) -> None:
243
+ super().__init__(name=name, description=spec.description)
244
+
245
+ def _get_declaration(self) -> Any:
246
+ return types.FunctionDeclaration(
247
+ name=name,
248
+ description=spec.description,
249
+ parameters=_json_schema(spec.input_schema),
250
+ )
251
+
252
+ async def run_async(self, *, args: dict[str, Any], tool_context: Any) -> Any:
253
+ return await spec.handler(args)
254
+
255
+ return SpecTool()
256
+
257
+
258
+ class VertexGeminiSession:
259
+ """One ADK-run conversation with Gemini; ADK holds the history across turns."""
260
+
261
+ def __init__(self, spec: SessionSpec, model: str) -> None:
262
+ self._spec = spec
263
+ self._model = model
264
+ self._runner: Any = None
265
+ self._pending: str | None = None
266
+ self.session_id = f"gemini-{uuid.uuid4().hex[:12]}"
267
+
268
+ def _tools(self) -> list[Any]:
269
+ # ASK_TOOL is deliberately absent: unattended runs never call it, and declaring a tool
270
+ # this backend cannot answer would cost the model a turn finding that out.
271
+ offered: list[Any] = []
272
+ wanted = {name for name in self._spec.builtins if name in FILE_TOOLS}
273
+ offered.extend(
274
+ _spec_tool(spec.name, spec)
275
+ for spec in file_tools(self._spec.cwd)
276
+ if spec.name in wanted
277
+ )
278
+ for server_name, server in self._spec.servers.items():
279
+ offered.extend(
280
+ _spec_tool(qualified(server_name, spec.name), spec)
281
+ for spec in server.tools
282
+ )
283
+ return offered
284
+
285
+ async def start(self) -> None:
286
+ from google.adk.agents import LlmAgent
287
+ from google.adk.runners import Runner
288
+ from google.adk.sessions import InMemorySessionService
289
+ from google.genai import types
290
+
291
+ # ADK builds its Vertex client from the environment, the same way the Claude backend
292
+ # passes provider env through its options.
293
+ os.environ["GOOGLE_GENAI_USE_VERTEXAI"] = "TRUE"
294
+ os.environ["GOOGLE_CLOUD_PROJECT"] = _project()
295
+ os.environ["GOOGLE_CLOUD_LOCATION"] = _location()
296
+ # static_instruction, not instruction: the skills are full of literal JSON braces,
297
+ # and ADK templates {placeholders} in `instruction` from session state. Static
298
+ # content is sent verbatim and is what ADK context-caches.
299
+ agent = LlmAgent(
300
+ name=self.session_id.replace("-", "_"),
301
+ model=self._model,
302
+ static_instruction=types.Content(
303
+ role="user", parts=[types.Part(text=self._spec.system_prompt)]
304
+ ),
305
+ tools=self._tools(),
306
+ )
307
+ sessions = InMemorySessionService()
308
+ await sessions.create_session(
309
+ app_name="alk-harness", user_id="stage", session_id=self.session_id
310
+ )
311
+ self._runner = Runner(
312
+ agent=agent, app_name="alk-harness", session_service=sessions
313
+ )
314
+
315
+ async def stop(self) -> None:
316
+ if self._runner is not None:
317
+ await self._runner.close()
318
+ self._runner = None
319
+
320
+ async def send(self, message: str) -> None:
321
+ if self._runner is None:
322
+ raise RuntimeError("session is not open")
323
+ self._pending = message
324
+
325
+ async def replies(self) -> AsyncIterator[Any]:
326
+ from google.adk.agents.run_config import RunConfig
327
+ from google.genai import types
328
+
329
+ if self._runner is None or self._pending is None:
330
+ raise RuntimeError("nothing to reply to; send a message first")
331
+ yield SessionOpened(session_id=self.session_id)
332
+ message = types.Content(role="user", parts=[types.Part(text=self._pending)])
333
+ self._pending = None
334
+ turns = 0
335
+ tokens_in = 0
336
+ tokens_out = 0
337
+ tokens_cached = 0
338
+ settled = False
339
+ terminal_save_succeeded = False
340
+ try:
341
+ async for event in self._runner.run_async(
342
+ user_id="stage",
343
+ session_id=self.session_id,
344
+ new_message=message,
345
+ run_config=RunConfig(max_llm_calls=max(self._spec.max_turns, 1)),
346
+ ):
347
+ usage = getattr(event, "usage_metadata", None)
348
+ if usage is not None:
349
+ tokens_in += usage.prompt_token_count or 0
350
+ tokens_out += usage.candidates_token_count or 0
351
+ tokens_cached += getattr(usage, "cached_content_token_count", 0) or 0
352
+ parts: list[Any] = []
353
+ returned: list[ToolReturned] = []
354
+ for part in (event.content.parts if event.content else []) or []:
355
+ if getattr(part, "text", None):
356
+ parts.append(Say(text=part.text))
357
+ if getattr(part, "function_call", None):
358
+ parts.append(
359
+ Call(
360
+ id=getattr(part.function_call, "id", None)
361
+ or f"call-{uuid.uuid4().hex[:8]}",
362
+ name=part.function_call.name or "",
363
+ arguments=dict(part.function_call.args or {}),
364
+ )
365
+ )
366
+ if getattr(part, "function_response", None):
367
+ response = part.function_response.response
368
+ response_name = part.function_response.name or ""
369
+ terminal_save_succeeded = terminal_save_succeeded or (
370
+ _successful_terminal_save(response_name, response)
371
+ )
372
+ returned.append(
373
+ ToolReturned(
374
+ id=getattr(part.function_response, "id", None) or "",
375
+ text=_flattened(response),
376
+ is_error=bool(
377
+ isinstance(response, dict)
378
+ and response.get("is_error")
379
+ ),
380
+ )
381
+ )
382
+ if parts:
383
+ turns += 1
384
+ yield ModelReply(parts=parts, model=self._model)
385
+ for outcome in returned:
386
+ yield outcome
387
+ if terminal_save_succeeded:
388
+ settled = True
389
+ break
390
+ if event.is_final_response():
391
+ settled = True
392
+ except Exception as exc:
393
+ yield StageDone(
394
+ outcome="failed",
395
+ turns=turns,
396
+ cost_usd=self._cost(tokens_in, tokens_out),
397
+ tokens_in=tokens_in,
398
+ tokens_out=tokens_out,
399
+ tokens_cached=tokens_cached,
400
+ session_id=self.session_id,
401
+ models={self._model},
402
+ is_error=True,
403
+ api_error_status=getattr(exc, "code", None),
404
+ errors=[str(exc)[:400]],
405
+ )
406
+ return
407
+ # A stream that ends without a final response ran out of its call budget, which is
408
+ # not the same as the model having finished. Reported as success it reads as a stage
409
+ # that did its work, and a half-written suite comes back green.
410
+ yield StageDone(
411
+ outcome="success" if settled else "max_turns",
412
+ is_error=not settled,
413
+ turns=turns,
414
+ cost_usd=self._cost(tokens_in, tokens_out),
415
+ tokens_in=tokens_in,
416
+ tokens_out=tokens_out,
417
+ tokens_cached=tokens_cached,
418
+ session_id=self.session_id,
419
+ models={self._model},
420
+ errors=(
421
+ []
422
+ if settled
423
+ else [
424
+ f"the stage spent its whole budget of {self._spec.max_turns} calls"
425
+ ]
426
+ ),
427
+ )
428
+
429
+ def _cost(self, tokens_in: int, tokens_out: int) -> float | None:
430
+ return priced(self._model, tokens_in, tokens_out)
431
+
432
+
433
+ logger = logging.getLogger(__name__)
434
+
435
+
436
+ def priced(model: str, tokens_in: int, tokens_out: int) -> float | None:
437
+ """What these tokens cost, or None where no price can be stood behind."""
438
+ prices = PRICES_PER_MILLION.get(model)
439
+ if prices is None:
440
+ return None
441
+ if len(prices) > 2 and date.today().isoformat() > str(prices[2]):
442
+ logger.warning(
443
+ "no current price for %s: the table's figures expired on %s", model, prices[2]
444
+ )
445
+ return None
446
+ return (tokens_in * prices[0] + tokens_out * prices[1]) / 1_000_000
447
+
448
+
449
+ class VertexGeminiBackend:
450
+ name = "vertex-gemini"
451
+ default_model = DEFAULT_MODEL
452
+
453
+ def can_drive(self, model: str) -> bool:
454
+ return (model or "").lower().startswith("gemini")
455
+
456
+ def create(self, spec: SessionSpec) -> VertexGeminiSession:
457
+ return VertexGeminiSession(spec, model=spec.model or self.default_model)
@@ -0,0 +1,95 @@
1
+ """Choose the caller-side ambient noise a scenario should be heard through.
2
+
3
+ A scenario that sets ``background_noise`` wants the agent to handle a caller phoning from somewhere
4
+ real: a car, a street, an office. The clip is chosen here and handed to the voice engine, which
5
+ mixes it under the simulated caller's audio.
6
+
7
+ Two sources, in order. A run may point ``ALK_BACKGROUND_NOISE_CATALOG`` at a JSON file of clips
8
+ (each with an ``environment`` tag and a ``url`` or ``path``); the catalog stays a local file so its
9
+ asset locations are never committed here. When no catalog matches, a LiveKit builtin clip is used,
10
+ which needs no external asset and always works.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import os
17
+ from pathlib import Path
18
+
19
+ # LiveKit ships these; they are the reliable default when no custom catalog is configured.
20
+ _BUILTIN_BY_ENVIRONMENT: dict[str, str] = {
21
+ "street": "CITY_AMBIENCE",
22
+ "transit": "CITY_AMBIENCE",
23
+ "vehicle": "CITY_AMBIENCE",
24
+ "outdoors": "FOREST_AMBIENCE",
25
+ "retail": "CROWDED_ROOM",
26
+ "office": "OFFICE_AMBIENCE",
27
+ "home": "OFFICE_AMBIENCE",
28
+ }
29
+ _DEFAULT_BUILTIN = "OFFICE_AMBIENCE"
30
+
31
+
32
+ def enabled() -> bool:
33
+ """Whether any scenario may be heard through background noise on this run.
34
+
35
+ **On unless ``ALK_BACKGROUND_NOISE`` turns it off.** A real caller is somewhere, and an agent
36
+ tested only against studio silence has not been tested against its callers, so noise is what a
37
+ run should fall into rather than something it has to ask for. It was opt-in and every deployment
38
+ forgot: a switch nobody sets is a feature nobody has.
39
+
40
+ The tradeoff is real and is why an opt-out exists. Continuous ambience under the caller competes
41
+ with endpoint detection, and calls carrying it end a little earlier and on fewer turns. Set
42
+ ``ALK_BACKGROUND_NOISE=0`` (or ``off``, ``false``, ``no``) for a run that needs a clean line.
43
+
44
+ Permission, not compulsion: a scenario whose own ``background_noise`` says none stays silent
45
+ either way.
46
+ """
47
+ return os.environ.get("ALK_BACKGROUND_NOISE", "1").strip().lower() not in (
48
+ "0",
49
+ "off",
50
+ "false",
51
+ "no",
52
+ )
53
+
54
+
55
+ def source_for(environment: str = "", seed: str = "") -> str:
56
+ """A background-noise source for a scenario.
57
+
58
+ Returns a ``url``/``path`` from the configured catalog when one matches the environment, else the
59
+ name of a LiveKit builtin clip. The choice is deterministic in ``seed`` so the same scenario
60
+ hears the same place across runs.
61
+ """
62
+ env = (environment or "").strip().lower()
63
+ catalog = os.environ.get("ALK_BACKGROUND_NOISE_CATALOG", "").strip()
64
+ if catalog and Path(catalog).is_file():
65
+ try:
66
+ entries = json.loads(Path(catalog).read_text(encoding="utf-8"))
67
+ except (OSError, ValueError):
68
+ entries = []
69
+ if isinstance(entries, list) and entries:
70
+ pool = [
71
+ entry
72
+ for entry in entries
73
+ if str(entry.get("environment", "")).strip().lower() == env
74
+ ] or entries
75
+ chosen = pool[sum(ord(character) for character in (seed or env or "x")) % len(pool)]
76
+ located = str(chosen.get("url") or chosen.get("path") or "").strip()
77
+ if located:
78
+ return located
79
+ return _BUILTIN_BY_ENVIRONMENT.get(env, _DEFAULT_BUILTIN)
80
+
81
+
82
+ def scenario_source(
83
+ background_noise, fixture, seed: str = ""
84
+ ) -> str:
85
+ """The noise source for one scenario, or "" when it should be heard in the clear.
86
+
87
+ The scenario names the place when it cares which one; otherwise the fixture says where the
88
+ caller is, and failing that any noise will do.
89
+ """
90
+ if not background_noise or not enabled():
91
+ return ""
92
+ environment = background_noise if isinstance(background_noise, str) else ""
93
+ if not environment and isinstance(fixture, dict):
94
+ environment = str(fixture.get("environment") or fixture.get("location") or "")
95
+ return source_for(environment, seed=seed)