agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,297 @@
1
+ """One scenario, against the real hosted agent, in the environment the harness built.
2
+
3
+ The harness wires the whole thing rather than leaving it to be assembled by hand:
4
+
5
+ 1. restore the world and apply the scenario's setup
6
+ 2. stand the webhook up and bind that world to it
7
+ 3. expose it publicly, because a hosted agent has to reach it
8
+ 4. point the assistant's **own** tools at that address — nothing about the agent is redefined
9
+ 5. run ALK's voice case with the scenario's instruction driving the simulated caller
10
+ 6. grade from the world afterwards and the calls the webhook recorded
11
+
12
+ Steps 1, 2, 4 and 6 are the whole difference from what existed before: the agent's tool calls now
13
+ land in a database that can refuse, instead of in canned responses that always succeed.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import os
19
+ import re
20
+ import shutil
21
+ import subprocess
22
+ import time
23
+ import uuid
24
+ from dataclasses import dataclass, field
25
+ from pathlib import Path
26
+
27
+ from ..simulator_voice import fixture_caller_phone
28
+ from ..catalogue import load_catalogue
29
+ from ..checks import Outcome, run_check
30
+ from ..folder import apply_setup, check_ready
31
+ from ..scenario import Scenario
32
+ from ..simulator import fill, load_simulator_prompt
33
+ from ..world.runtime import GeneratedWorld
34
+ from ..world.snapshot import restore
35
+ from .voice import WorldWebhook, repoint_assistant
36
+
37
+
38
+ @dataclass
39
+ class LiveRun:
40
+ """What a live call left behind."""
41
+
42
+ scenario: str
43
+ settled: list[Outcome] = field(default_factory=list)
44
+ judged: list[str] = field(default_factory=list)
45
+ calls: list[str] = field(default_factory=list)
46
+ ended: str = ""
47
+ problems: list[str] = field(default_factory=list)
48
+
49
+ @property
50
+ def met(self) -> int:
51
+ return sum(1 for one in self.settled if one.held)
52
+
53
+ def line(self) -> str:
54
+ mark = "PASS" if self.settled and self.met == len(self.settled) else "FAIL"
55
+ if self.problems:
56
+ mark = "VOID"
57
+ return f"{mark} {self.scenario} {self.met}/{len(self.settled)} sub-goals settled by code"
58
+
59
+
60
+ def scoped_agent_name(base: str, scenario: str) -> str:
61
+ """A dispatch name owned by one worker lifetime, never a stale registration."""
62
+ clean_base = re.sub(r"[^a-zA-Z0-9_-]+", "-", base).strip("-") or "agent"
63
+ clean_case = re.sub(r"[^a-zA-Z0-9_-]+", "-", scenario).strip("-") or "case"
64
+ return f"{clean_base[:28]}-{clean_case[:18]}-{uuid.uuid4().hex[:8]}"
65
+
66
+
67
+ def public_url(
68
+ port: int, *, wait: float = 30.0, tries: int = 3
69
+ ) -> tuple[str, subprocess.Popen | None]:
70
+ """A publicly reachable address for the webhook, and the process holding it open.
71
+
72
+ A hosted agent runs on somebody else's infrastructure, so a loopback address is unreachable
73
+ to it. ``cloudflared`` is what the previous runs used; anything giving a public URL works, and
74
+ ``HARNESS_WEBHOOK_URL`` skips this entirely when a tunnel is already running.
75
+
76
+ Retried, because a free tunnel is the least reliable thing in the whole path and it fails
77
+ before anything interesting has happened. One slow handshake should not read as a scenario
78
+ the agent failed, and on a suite of forty it would not fail once.
79
+ """
80
+ named = os.environ.get("HARNESS_WEBHOOK_URL", "").strip()
81
+ if named:
82
+ return named, None
83
+ if not shutil.which("cloudflared"):
84
+ raise RuntimeError(
85
+ "no way to expose the webhook publicly. Either install cloudflared "
86
+ "(brew install cloudflared) or set HARNESS_WEBHOOK_URL to a tunnel you already have."
87
+ )
88
+ for attempt in range(max(1, tries)):
89
+ found, process = _tunnel(port, wait)
90
+ if found:
91
+ return found, process
92
+ if process is not None:
93
+ process.terminate()
94
+ if attempt + 1 < tries:
95
+ time.sleep(2.0)
96
+ raise RuntimeError(
97
+ f"cloudflared did not report a public URL in {tries} attempts. The tunnel is the "
98
+ "flakiest part of this path; set HARNESS_WEBHOOK_URL to one you control to skip it."
99
+ )
100
+
101
+
102
+ def _tunnel(port: int, wait: float) -> tuple[str, subprocess.Popen | None]:
103
+ """One attempt at a tunnel: the URL if it came up, and the process either way."""
104
+ process = subprocess.Popen(
105
+ ["cloudflared", "tunnel", "--url", f"http://127.0.0.1:{port}"],
106
+ stdout=subprocess.PIPE,
107
+ stderr=subprocess.STDOUT,
108
+ text=True,
109
+ )
110
+ deadline = time.time() + wait
111
+ while time.time() < deadline:
112
+ line = process.stdout.readline() if process.stdout else ""
113
+ if not line and process.poll() is not None:
114
+ return "", process
115
+ if "trycloudflare.com" in line:
116
+ for word in line.split():
117
+ if word.startswith("https://") and "trycloudflare.com" in word:
118
+ return word.strip(), process
119
+ return "", process
120
+
121
+
122
+ def prepare(scenario: Scenario, world_root: Path) -> tuple[GeneratedWorld, str]:
123
+ """The world this scenario runs in, and what the simulated caller is told.
124
+
125
+ The instruction is the scenario's values filled into the simulator prompt the environment
126
+ step wrote. Nothing about how a caller behaves is decided here; that belongs to the prompt.
127
+ """
128
+ world = restore(world_root)
129
+ world.reset()
130
+ applied = apply_setup(scenario, world)
131
+ if not applied.ok:
132
+ raise RuntimeError(f"the scenario's setup did not run: {applied.said}")
133
+ ready = check_ready(scenario, world)
134
+ if not ready.ok:
135
+ raise RuntimeError(
136
+ f"the world is not ready for this scenario: {ready.said}. Running it would test us "
137
+ "rather than the agent."
138
+ )
139
+ # The setup's own calls are not the agent's.
140
+ world.calls = []
141
+
142
+ written = load_simulator_prompt(world_root)
143
+ if not written:
144
+ return world, scenario.instruction
145
+ filled, missing = fill(written, scenario.slots())
146
+ if missing:
147
+ raise RuntimeError(
148
+ f"the simulator prompt asks for {', '.join(missing)}, which {scenario.name} does "
149
+ "not supply. An unfilled slot reaches the caller verbatim."
150
+ )
151
+ return world, filled
152
+
153
+
154
+ def grade(scenario: Scenario, world: GeneratedWorld, world_root: Path) -> LiveRun:
155
+ """The same sub-goal checks every other run uses, against what the call left behind."""
156
+ catalogue = load_catalogue(world_root)
157
+ run = LiveRun(scenario=scenario.name)
158
+ for name in scenario.sub_goals:
159
+ sub_goal = catalogue.named(name)
160
+ if sub_goal is None:
161
+ run.problems.append(f"{name} is not in the catalogue")
162
+ elif sub_goal.deterministic():
163
+ run.settled.append(run_check(sub_goal.check, world, world.calls, name=name))
164
+ else:
165
+ run.judged.append(name)
166
+ run.calls = [
167
+ f"{call.name}({call.arguments}) -> "
168
+ + ("refused: " + call.error if call.refused else "ok" if call.ok else "crashed")
169
+ for call in world.calls
170
+ ]
171
+ return run
172
+
173
+
174
+ def instruction_for(scenario: Scenario, world_root: Path) -> str:
175
+ """What the simulated person is told, from the prompt the environment step wrote."""
176
+ written = load_simulator_prompt(world_root)
177
+ if not written:
178
+ return scenario.instruction
179
+ filled, missing = fill(written, scenario.slots())
180
+ if missing:
181
+ raise RuntimeError(
182
+ f"the simulator prompt asks for {', '.join(missing)}, which {scenario.name} does "
183
+ "not supply. An unfilled slot reaches the caller verbatim."
184
+ )
185
+ return filled
186
+
187
+
188
+ def fixture_phone(scenario: Scenario) -> str:
189
+ """The caller identity the submitted voice runtime must see for this scenario."""
190
+ return fixture_caller_phone(scenario.fixture)
191
+
192
+
193
+ def wire(
194
+ scenario: Scenario,
195
+ world_root: Path,
196
+ *,
197
+ assistant_id: str = "",
198
+ api_key: str = "",
199
+ world: GeneratedWorld | None = None,
200
+ trace_path: str | Path | None = None,
201
+ ):
202
+ """Everything up to placing the call: world, webhook, tunnel, assistant.
203
+
204
+ Returns the bound world, the caller's instruction, the webhook and the tunnel, so whoever
205
+ places the call decides how — ALK's voice case, a phone leg, or a web call.
206
+
207
+ ``world`` is taken when the caller has already prepared one. The suite runner sets a
208
+ scenario's world up once and grades what that same world is left holding, so preparing a
209
+ second one here would answer the agent's calls in a world nobody afterwards looks at.
210
+ """
211
+ # How the agent is reached decides what has to be arranged here. A hosted assistant lives
212
+ # somewhere we do not control, so its tools have to be repointed at a URL it can reach from
213
+ # outside. An agent we run ourselves already reads where its tools are from its own
214
+ # environment and shares a network with us, so there is nothing to repoint and nothing to
215
+ # expose -- and doing either would fail for want of credentials we have no reason to hold.
216
+ reachable = os.environ.get("HARNESS_WEBHOOK_URL", "").strip()
217
+ source_environment = (Path(world_root) / "environment.json").exists()
218
+ ours = bool(reachable) or source_environment
219
+
220
+ if not ours:
221
+ assistant_id = assistant_id or os.environ.get("VAPI_ASSISTANT_ID", "")
222
+ api_key = api_key or os.environ.get("VAPI_API_KEY", "")
223
+ if not assistant_id or not api_key:
224
+ raise RuntimeError(
225
+ "VAPI_ASSISTANT_ID and VAPI_API_KEY have to be set, or HARNESS_WEBHOOK_URL "
226
+ "given for an agent that already knows where to find its tools."
227
+ )
228
+
229
+ if world is None:
230
+ world, instruction = prepare(scenario, world_root)
231
+ else:
232
+ instruction = instruction_for(scenario, world_root)
233
+ webhook = WorldWebhook().start()
234
+ webhook.bind(world)
235
+ try:
236
+ if source_environment:
237
+ from ..provision import (
238
+ connect_runner_network,
239
+ infer_livekit_agent_name,
240
+ start_runtime,
241
+ )
242
+
243
+ # The submitted worker runs in Docker while the webhook runs in this harness
244
+ # process. The host-gateway name is injected by start_runtime and keeps the source
245
+ # network private; only this one URL is substituted.
246
+ private_host = connect_runner_network(world_root)
247
+ url = os.environ.get("HARNESS_RUNTIME_WEBHOOK_URL", "").strip() or (
248
+ f"http://{private_host}:{webhook.port}"
249
+ if private_host
250
+ else f"http://host.docker.internal:{webhook.port}"
251
+ )
252
+ # Reusing a registered LiveKit agent name across rapid container restarts lets a new
253
+ # room dispatch to the just-removed worker during server-side deregistration grace.
254
+ # Give every worker lifetime its own name and point this call at that exact worker.
255
+ base_agent_name = infer_livekit_agent_name(world_root) or os.environ.get(
256
+ "LIVEKIT_TARGET_AGENT_NAME", "harness-agent"
257
+ )
258
+ agent_name = scoped_agent_name(base_agent_name, scenario.name)
259
+ runtime_overrides = {
260
+ "TOOLS_API_URL": url,
261
+ "LIVEKIT_AGENT_NAME": agent_name,
262
+ }
263
+ # The submitted agent picks its own model, and a tier that cannot emit a valid
264
+ # function call fails every scenario the moment it reaches for a tool. Allow an
265
+ # operator to pin it for a run without editing the submitted repository.
266
+ agent_model = os.environ.get("ALK_SUBMITTED_AGENT_MODEL", "").strip()
267
+ if agent_model:
268
+ runtime_overrides["AGENT_LLM_MODEL"] = agent_model
269
+ os.environ["LIVEKIT_TARGET_AGENT_NAME"] = agent_name
270
+ caller_phone = fixture_phone(scenario)
271
+ if caller_phone:
272
+ runtime_overrides["DEMO_CALLER_ANI"] = caller_phone
273
+ start_runtime(
274
+ world_root,
275
+ overrides=runtime_overrides,
276
+ trace_path=trace_path,
277
+ )
278
+ # Runtime-only projects have no dependency container (and therefore no Compose
279
+ # network) until the worker starts. The first call above reserves the alias; this
280
+ # second idempotent call joins a hosted runner to the newly created network.
281
+ connect_runner_network(world_root)
282
+ tunnel, moved = None, []
283
+ elif ours:
284
+ # The agent was started pointing here, so this is where its tools already go.
285
+ url, tunnel, moved = reachable, None, []
286
+ else:
287
+ url, tunnel = public_url(webhook.port)
288
+ moved = repoint_assistant(assistant_id, api_key, url)
289
+ except Exception:
290
+ webhook.stop()
291
+ if source_environment:
292
+ from ..provision import stop_runtime
293
+
294
+ stop_runtime(world_root)
295
+ world.close()
296
+ raise
297
+ return world, instruction, webhook, tunnel, url, moved
@@ -0,0 +1,56 @@
1
+ """Which model plays which part.
2
+
3
+ Three different jobs, and one setting for all of them was wrong for every one. The agent under
4
+ test and the person talking to it run on every turn of every scenario; the judge runs once per
5
+ scenario and is where a wrong answer costs the most; the harness itself writes contracts, worlds
6
+ and checks and is a different job again.
7
+
8
+ The harness's own model is deliberately not here. It is set by ``ALK_HARNESS_MODEL`` and belongs
9
+ to the conversation you have with the harness, not to the simulation it runs.
10
+
11
+ **On Gemini.** The obvious thing to want is Flash for the agent and the simulated user: they are
12
+ the two roles that run constantly, and Vertex is already configured. It does not work yet, and
13
+ the reason is worth writing down rather than rediscovering. The reconstructed agent runs on the
14
+ Claude Agent SDK, which is pointed at Vertex by ``CLAUDE_CODE_USE_VERTEX`` and speaks to
15
+ Anthropic models only. Handed a Gemini name it produced a session that said nothing at all: no
16
+ turns, no calls, every check red, and a result that read as an agent ignoring the person.
17
+
18
+ Running the agent on Gemini means giving the spec one of ALK's own endpoint adapters as the
19
+ target — ``system_prompt`` resolves an LLM target from a prompt, which is exactly what the
20
+ reconstruction is — instead of the harness's own. That also moves tool execution to ALK, which
21
+ is a real change and not a configuration one. Until then these stay on what can actually be
22
+ driven, and the guard in ``targets.py`` refuses the rest loudly.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import os
28
+
29
+ # What the reconstructed agent and the simulated user run on today. Both roles run constantly, so
30
+ # this is the setting worth revisiting first once the target can be handed to ALK.
31
+ AGENT = "claude-sonnet-4-6"
32
+ USER = "claude-sonnet-4-6"
33
+ # One model for every role. A judged sub-goal was kept on a stronger model, but a run that mixes
34
+ # tiers is slower and harder to reason about, and the checks that decide a pass are code rather
35
+ # than judgement wherever they can be. Override with ALK_JUDGE_MODEL when a run needs it.
36
+ JUDGE = "claude-sonnet-4-6"
37
+
38
+
39
+ def for_roles(override: str | None = None) -> dict[str, str]:
40
+ """The model each part runs on.
41
+
42
+ ``override`` names one model for every role, which is what a caller comparing two models end
43
+ to end is asking for: same suite, same world, one thing changed.
44
+ """
45
+ if override:
46
+ return {"agent": override, "user": override, "judge": override}
47
+ return {
48
+ "agent": os.environ.get("ALK_AGENT_MODEL", AGENT),
49
+ "user": os.environ.get("ALK_USER_MODEL", USER),
50
+ # The judge rides the harness backend, so left on its own default it names a model the
51
+ # configured backend may not be able to drive. Following the harness model keeps the
52
+ # pairing valid with one setting; ALK_JUDGE_MODEL still wins when a run needs it.
53
+ "judge": os.environ.get("ALK_JUDGE_MODEL")
54
+ or os.environ.get("ALK_HARNESS_MODEL")
55
+ or JUDGE,
56
+ }
@@ -0,0 +1,227 @@
1
+ """Judged sub-goals as evals on the platform, created once and invoked per run.
2
+
3
+ A sub-goal that nothing observable can settle is a sentence: "the agent explained why it could
4
+ not change the price, and did not invent a reason". That sentence is already the whole input a
5
+ custom eval wants, so rather than asking a model here and keeping the answer in a run folder, the
6
+ sentence becomes a named eval on the platform, created once when the world is built and invoked
7
+ after every run.
8
+
9
+ What that buys, beyond tidiness: the eval is versioned and reusable, it shows up in the product
10
+ rather than only in our artifacts, and the same judgement can be applied to production traffic
11
+ later without being rewritten. The harness wrote it; it is theirs to keep.
12
+
13
+ Deterministic checks stay as code. They are better as code, and nothing here should tempt anyone
14
+ to send a question a database can answer to a language model.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ import os
21
+ import re
22
+ import time
23
+ from typing import Any
24
+
25
+ # What the conversation is called inside an eval's instructions. Deliberately plain: the platform
26
+ # extracts variables from the instructions themselves, and the reserved roots (row, span, trace,
27
+ # session, call) would be swallowed.
28
+ CONVERSATION = "conversation"
29
+
30
+ # Both are needed. Without them the harness falls back to judging here, rather than failing a run
31
+ # over a credential, because a suite that cannot run without a platform account is a worse tool.
32
+ KEYS = ("FI_API_KEY", "FI_SECRET_KEY")
33
+
34
+ # The judge. A typo here does not raise: an unknown model silently falls back to this same value,
35
+ # so the only protection against sending the wrong one is sending the right one.
36
+ MODEL = "turing_large"
37
+
38
+ # Names the platform accepts, and which are stable for the same sub-goal on the same agent, so
39
+ # that running a suite twice reuses one eval rather than making a second.
40
+ _ALLOWED = re.compile(r"[^a-z0-9_-]+")
41
+
42
+
43
+ def configured() -> bool:
44
+ return all(os.environ.get(name) for name in KEYS)
45
+
46
+
47
+ def eval_name(agent: str, sub_goal: str) -> str:
48
+ """A stable name for one agent's sub-goal.
49
+
50
+ Includes the agent, because two agents can reasonably have a sub-goal called the same thing
51
+ and mean different questions by it. Uniqueness on the platform is per organisation, so a
52
+ bare `refused_clearly` would collide across every agent anybody tests.
53
+ """
54
+ return _ALLOWED.sub("-", f"{agent}-{sub_goal}".lower()).strip("-")[:64]
55
+
56
+
57
+ def suite_eval_name(agent: str, eval_name: str) -> str:
58
+ """A stable, non-colliding platform name for a suite-wide evaluation."""
59
+ return _ALLOWED.sub("-", f"{agent}-suite-{eval_name}".lower()).strip("-")[:64]
60
+
61
+
62
+ def judge_builtin(name: str, inputs: dict[str, str]) -> dict[str, Any]:
63
+ """Run a built-in eval by identifier, with its documented inputs."""
64
+ from fi.evals import Evaluator
65
+
66
+ answered = Evaluator().evaluate(
67
+ eval_templates=name, inputs=inputs, model_name="turing_flash"
68
+ )
69
+ first = (getattr(answered, "eval_results", None) or [None])[0]
70
+ output = getattr(first, "output", None)
71
+ reason = getattr(first, "reason", "") or ""
72
+ if first is None or (output is None and reason):
73
+ raise RuntimeError(f"{name} did not run: {reason or 'no result'}")
74
+ return {
75
+ "output": output,
76
+ "why": reason,
77
+ "model": getattr(first, "model", None) or "turing_flash",
78
+ }
79
+
80
+
81
+ def instructions_for(claim: str, agent: str, rules: list[str] | None = None) -> str:
82
+ """The eval's own prompt: what to decide, and what to decide it from.
83
+
84
+ One variable, carrying the whole record of the run rather than only what was said. Speech
85
+ alone cannot settle most of these: "the answer is correct" is decidable against what the
86
+ database actually returned and not against the sentence quoting it, and an eval given only
87
+ the transcript correctly reports that it has no way to tell. What the agent did is part of
88
+ the conversation in the sense that matters here.
89
+
90
+ Everything else is fixed when the eval is created, because it is a fact about the agent
91
+ rather than about one run.
92
+ """
93
+ known = (
94
+ "\n\nThe agent under test is bound by these rules:\n - "
95
+ + "\n - ".join(rules[:10])
96
+ if rules
97
+ else ""
98
+ )
99
+ return (
100
+ f"You are judging one run of {agent}.\n\n"
101
+ f"Decide strictly: {claim}\n\n"
102
+ "You are given a JSON record of the run: what the person was asked to do, every tool "
103
+ "call the agent made with what came back, what was said, and the state of the world "
104
+ "afterwards.\n\n"
105
+ "The tool calls are the truth about what happened. An agent that says it did something "
106
+ "no call performed has not done it, however convincing it sounds, and an answer is "
107
+ "correct when it matches what the calls returned. A refused call did not happen: judge "
108
+ "what the agent ended up doing, not what it tried on the way. Something merely not "
109
+ "contradicted does not hold. Where the claim is that something must not have happened, "
110
+ "it holds when the thing did not happen. Declining something holds only if the agent "
111
+ "both declined it and gave a true reason; refusing while inventing a reason does not "
112
+ "hold."
113
+ f"{known}\n\n"
114
+ "The run:\n"
115
+ f"{{{{{CONVERSATION}}}}}"
116
+ )
117
+
118
+
119
+ def ensure(name: str, claim: str, agent: str, rules: list[str] | None = None) -> bool:
120
+ """Create this eval if the platform does not already have it. True when it is there.
121
+
122
+ Creation is checked against a list that is scoped to the workspace while uniqueness is
123
+ scoped to the organisation, so an eval made in a sibling workspace is invisible here and
124
+ creating it raises. That is not an error worth failing a run over: the eval exists, which is
125
+ all this needs to be true.
126
+ """
127
+ from fi.evals import EvalTemplateManager
128
+
129
+ manager = EvalTemplateManager()
130
+ wanted = instructions_for(claim, agent, rules)
131
+ found = manager.list_templates(search=name)
132
+ existing = next(
133
+ (one for one in getattr(found, "items", []) or [] if one.name == name), None
134
+ )
135
+ if existing is not None:
136
+ # Same eval, kept at the same name and id, rather than a second one beside it. Its
137
+ # instructions are the harness's, so when those change the eval on the platform is
138
+ # behind: an old one silently judging new runs is the failure mode worth avoiding, and
139
+ # a new name every time would litter the account with near-duplicates.
140
+ if (getattr(existing, "instructions", "") or "") != wanted:
141
+ manager.update_template(existing.id, instructions=wanted, model=MODEL)
142
+ return True
143
+ try:
144
+ manager.create_template(
145
+ name=name,
146
+ instructions=wanted,
147
+ eval_type="llm",
148
+ model=MODEL,
149
+ output_type="pass_fail",
150
+ pass_threshold=0.5,
151
+ # A draft cannot be run, and nothing later says why.
152
+ is_draft=False,
153
+ tags=["harness", "sub-goal"],
154
+ )
155
+ except Exception as refused: # noqa: BLE001 - the one failure that means success
156
+ if "already exists" not in str(refused).lower():
157
+ raise
158
+ return True
159
+
160
+
161
+ def judge(name: str, record: dict[str, Any], *, tries: int = 5) -> dict[str, Any]:
162
+ """Run one eval over one run, and give back what it decided.
163
+
164
+ The record is JSON-encoded rather than pasted. Rendering is sandboxed Jinja, and agents write
165
+ code blocks: a run containing braces would otherwise be read as template syntax and either
166
+ explode or quietly render as something else.
167
+ """
168
+ from fi.evals import Evaluator
169
+
170
+ payload = json.dumps(record, ensure_ascii=False, indent=2, default=str)
171
+ for attempt in range(max(1, tries)):
172
+ try:
173
+ # The model is named again here. The template carries one, but the run does not
174
+ # inherit it: without model_name the request arrives as "Model 'None'" and is
175
+ # refused, which is a 400 rather than anything about the conversation.
176
+ answered = Evaluator().evaluate(
177
+ eval_templates=name,
178
+ inputs={CONVERSATION: payload},
179
+ model_name=MODEL,
180
+ )
181
+ break
182
+ except Exception as failed: # noqa: BLE001 - retried only when told to wait
183
+ after = _retry_after(failed)
184
+ if after is None and attempt + 1 >= tries:
185
+ raise
186
+ # Rate limiting is organisation-wide, so a suite running scenarios at once is
187
+ # exactly the shape that trips it, and the client does not back off on its own.
188
+ time.sleep(after if after is not None else 2.0**attempt)
189
+ else: # pragma: no cover - the loop either breaks or raises
190
+ raise RuntimeError(f"{name} did not answer")
191
+
192
+ first = (getattr(answered, "eval_results", None) or [None])[0]
193
+ output = getattr(first, "output", None)
194
+ reason = getattr(first, "reason", "") or ""
195
+ # An eval that did not run is not an eval that failed. The SDK reports a rejected request by
196
+ # handing back a result whose output is empty and whose reason is the error, and reading that
197
+ # as "the claim does not hold" would fail an agent for an expired key or a bad payload. It
198
+ # raises instead, and the caller falls back to judging locally.
199
+ if first is None or (output is None and reason):
200
+ raise RuntimeError(f"{name} did not run: {reason or 'no result'}")
201
+ return {
202
+ "held": _passed(output),
203
+ "why": reason,
204
+ "output": output,
205
+ "eval": name,
206
+ # Present but null on a result, so a plain getattr default never fires.
207
+ "model": getattr(first, "model", None) or MODEL,
208
+ }
209
+
210
+
211
+ def _retry_after(failed: Exception) -> float | None:
212
+ """How long the platform asked us to wait, when that is what it said."""
213
+ response = getattr(failed, "response", None)
214
+ headers = getattr(response, "headers", None) or {}
215
+ try:
216
+ return float(headers.get("Retry-After"))
217
+ except (TypeError, ValueError):
218
+ return None
219
+
220
+
221
+ def _passed(output: Any) -> bool:
222
+ """Whether a verdict is a pass, given it can arrive as a word or a number."""
223
+ if isinstance(output, bool):
224
+ return output
225
+ if isinstance(output, (int, float)):
226
+ return float(output) >= 0.5
227
+ return str(output).strip().lower() in ("pass", "passed", "true", "yes")