agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,692 @@
1
+ """Reporting a run to the platform, so it appears where every other run does.
2
+
3
+ The platform already has somewhere to put this. Its simulate pages read `RunTest`,
4
+ `TestExecution` and `CallExecution`, and the ingestion API that the hosted runner posts to builds
5
+ exactly those. So a harness run is not shown by drawing it again somewhere else; it is shown by
6
+ walking the same API, and the pages that already exist render it unchanged.
7
+
8
+ provision ──► a RunTest for this session, once
9
+ start ──► a TestExecution, once per run, so running twice gives two runs
10
+ batch ──► a CallExecution per scenario
11
+ result ──► what the scenario did, one call at a time
12
+ recording ──► the audio, where a spoken run left any
13
+
14
+ What is deliberately *not* sent: interruption counts, talk ratio, latency, scores. The backend
15
+ derives those from the transcript it is given, and a second implementation here would drift from
16
+ the one the rest of the platform is measured by. This reports only what the run observed.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import hashlib
22
+ import json
23
+ import os
24
+ import re
25
+ import urllib.error
26
+ import urllib.parse
27
+ import urllib.request
28
+ from dataclasses import dataclass, field
29
+ from datetime import datetime, timezone
30
+ from pathlib import Path
31
+ from typing import Any
32
+
33
+ # Where runs are reported. Its own variables, because reporting and evaluating go to different
34
+ # places: the eval templates a run is scored against live on the hosted platform, while the runs
35
+ # themselves belong wherever the person is looking at them -- usually the backend beside this
36
+ # harness. Sharing FI_* for both means one of the two is always pointed at the wrong host.
37
+ # FI_* is the fallback, so a setup that genuinely uses one platform for both still works unchanged.
38
+ BASE_URL = ("HARNESS_PLATFORM_URL", "FI_BASE_URL")
39
+ API_KEY = ("HARNESS_PLATFORM_API_KEY", "FI_API_KEY")
40
+ SECRET_KEY = ("HARNESS_PLATFORM_SECRET_KEY", "FI_SECRET_KEY")
41
+ WORKSPACE_ID = "HARNESS_PLATFORM_WORKSPACE_ID"
42
+
43
+
44
+ def _setting(names: tuple[str, ...]) -> str:
45
+ """The first of these that is set, so the specific name wins over the shared one."""
46
+ for name in names:
47
+ found = os.environ.get(name, "").strip()
48
+ if found:
49
+ return found
50
+ return ""
51
+
52
+
53
+ INGESTION = "/simulate/api/alk-simulate"
54
+
55
+ # Django appends a slash and cannot redirect a POST while keeping its body, so every path here
56
+ # carries one already. Without it the call fails as a 500 that reads like a server fault.
57
+ TIMEOUT_SECONDS = 120.0
58
+
59
+
60
+ def _open(request: urllib.request.Request, timeout: float = TIMEOUT_SECONDS):
61
+ """Open an ingestion request without sending loopback traffic to a proxy.
62
+
63
+ Developer machines commonly export an HTTP(S) proxy for model/provider
64
+ traffic. urllib applies it to the local platform too unless NO_PROXY happens
65
+ to be configured, producing an unrelated proxy 400 with an empty body.
66
+ Remote platform URLs retain normal proxy behavior.
67
+ """
68
+ host = (urllib.parse.urlsplit(request.full_url).hostname or "").lower()
69
+ if host in {"127.0.0.1", "localhost", "::1"}:
70
+ return urllib.request.build_opener(urllib.request.ProxyHandler({})).open(
71
+ request, timeout=timeout
72
+ )
73
+ return urllib.request.urlopen(request, timeout=timeout)
74
+
75
+
76
+ class PlatformError(RuntimeError):
77
+ """The platform refused or could not be reached, with enough detail to act on."""
78
+
79
+
80
+ @dataclass
81
+ class Reported:
82
+ """Where a run ended up, so a caller can link to it."""
83
+
84
+ run_test_id: str = ""
85
+ test_execution_id: str = ""
86
+ calls: dict[str, str] = field(default_factory=dict)
87
+ problems: list[str] = field(default_factory=list)
88
+
89
+ @property
90
+ def url(self) -> str:
91
+ """Where this exact execution is on the platform."""
92
+ if not self.run_test_id:
93
+ return ""
94
+ if self.test_execution_id:
95
+ return (
96
+ f"/dashboard/simulate/test/{self.run_test_id}/{self.test_execution_id}"
97
+ )
98
+ return f"/dashboard/simulate/test/{self.run_test_id}/runs"
99
+
100
+
101
+ def display_run_name(agent: str, *, now: datetime | None = None) -> str:
102
+ """A readable, unique platform name without exposing a harness UUID."""
103
+ words = re.sub(r"[-_]+", " ", str(agent or "agent")).strip()
104
+ title = " ".join(
105
+ word if word.isupper() else word.capitalize() for word in words.split()
106
+ )
107
+ timestamp = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
108
+ return f"{title or 'Agent'} · {timestamp:%d %b %Y %H:%M UTC}"[:255]
109
+
110
+
111
+ def display_scenario_name(scenario: Any) -> str:
112
+ """Prefer the scenario's behavior over an internal slug or repeated run prefix."""
113
+ for value in (
114
+ getattr(scenario, "use_case", ""),
115
+ getattr(scenario, "tests", ""),
116
+ getattr(scenario, "name", ""),
117
+ ):
118
+ text = str(value or "").strip().rstrip(".")
119
+ if text:
120
+ if value == getattr(scenario, "name", ""):
121
+ text = re.sub(r"[-_]+", " ", text)
122
+ text = text[:1].upper() + text[1:]
123
+ return text[:255]
124
+ return "Scenario"
125
+
126
+
127
+ def configured() -> str:
128
+ """Why a run cannot be reported, or an empty string when it can."""
129
+ missing = [
130
+ names[0] for names in (BASE_URL, API_KEY, SECRET_KEY) if not _setting(names)
131
+ ]
132
+ if missing:
133
+ return f"{', '.join(missing)} not set, so this run stays local"
134
+ return ""
135
+
136
+
137
+ class Platform:
138
+ """The ingestion API, as the few calls a run actually makes."""
139
+
140
+ def __init__(self, base: str = "", key: str = "", secret: str = "") -> None:
141
+ self.base = (base or _setting(BASE_URL)).rstrip("/")
142
+ self.key = key or _setting(API_KEY)
143
+ self.secret = secret or _setting(SECRET_KEY)
144
+
145
+ def _headers(self) -> dict[str, str]:
146
+ headers = {
147
+ "X-Api-Key": self.key,
148
+ "X-Secret-Key": self.secret,
149
+ }
150
+ workspace_id = os.environ.get(WORKSPACE_ID, "").strip()
151
+ if workspace_id:
152
+ headers["X-Workspace-Id"] = workspace_id
153
+ return headers
154
+
155
+ def _call(
156
+ self, path: str, payload: dict[str, Any], method: str = "POST"
157
+ ) -> dict[str, Any]:
158
+ request = urllib.request.Request(
159
+ f"{self.base}{INGESTION}{path}",
160
+ data=json.dumps(payload).encode(),
161
+ headers={"Content-Type": "application/json", **self._headers()},
162
+ method=method,
163
+ )
164
+ try:
165
+ with _open(request) as answer:
166
+ body = json.loads(answer.read().decode() or "{}")
167
+ except urllib.error.HTTPError as refused:
168
+ detail = refused.read().decode(errors="replace")[:400]
169
+ raise PlatformError(
170
+ f"{method} {path} failed ({refused.code}): {detail}"
171
+ ) from refused
172
+ except Exception as unreachable: # noqa: BLE001 - reported, not handled
173
+ raise PlatformError(
174
+ f"{method} {path} could not be sent: {unreachable}"
175
+ ) from unreachable
176
+ # The platform wraps every answer; unwrap it here so callers read the payload itself.
177
+ return body.get("result", body) if isinstance(body, dict) else {}
178
+
179
+ def provision(
180
+ self, name: str, personas: list[dict[str, Any]], modality: str = "text"
181
+ ) -> dict[str, Any]:
182
+ agent_name = name.split(" · ", 1)[0].strip() or "ALK agent"
183
+ return self._call(
184
+ "/run-tests/provision/",
185
+ {
186
+ "name": name,
187
+ "agent_name": agent_name,
188
+ "personas": personas,
189
+ "modality": modality,
190
+ },
191
+ )
192
+
193
+ def start(
194
+ self,
195
+ run_test_id: str,
196
+ scenario_ids: list[str] | None = None,
197
+ *,
198
+ harness_job_id: str = "",
199
+ scenario_selectors: list[dict[str, str]] | None = None,
200
+ ) -> dict[str, Any]:
201
+ payload = {"scenario_ids": scenario_ids} if scenario_ids else {}
202
+ if harness_job_id:
203
+ payload["harness_job_id"] = harness_job_id
204
+ if scenario_selectors:
205
+ payload["scenario_selectors"] = scenario_selectors
206
+ return self._call(f"/run-tests/{run_test_id}/test-executions/", payload)
207
+
208
+ def batch(self, test_execution_id: str, count: int) -> dict[str, Any]:
209
+ return self._call(
210
+ f"/test-executions/{test_execution_id}/batch/", {"count": count}
211
+ )
212
+
213
+ def result(self, call_execution_id: str, payload: dict[str, Any]) -> dict[str, Any]:
214
+ return self._call(
215
+ f"/call-executions/{call_execution_id}/result/", payload, method="PATCH"
216
+ )
217
+
218
+ def ongoing(self, call_execution_id: str) -> dict[str, Any]:
219
+ """Mark one pre-allocated call as started using the established ingestion route."""
220
+ return self._call(
221
+ f"/call-executions/{call_execution_id}/status/",
222
+ {"status": "ongoing"},
223
+ method="PATCH",
224
+ )
225
+
226
+ def recording(self, call_execution_id: str, audio: Path) -> dict[str, Any]:
227
+ """Send one call's audio, as the multipart upload the endpoint expects.
228
+
229
+ Built by hand rather than with a library: this is the only multipart request the harness
230
+ makes, and a dependency for one boundary string is not worth carrying.
231
+ """
232
+ edge = "----harness" + os.urandom(8).hex()
233
+ content = audio.read_bytes()
234
+ digest = hashlib.sha256(content).hexdigest()
235
+ field = (
236
+ f"--{edge}\r\n"
237
+ f'Content-Disposition: form-data; name="file"; filename="{audio.name}"\r\n'
238
+ "Content-Type: audio/wav\r\n\r\n"
239
+ ).encode()
240
+ checksum = (
241
+ f"\r\n--{edge}\r\n"
242
+ 'Content-Disposition: form-data; name="sha256"\r\n\r\n'
243
+ f"{digest}"
244
+ ).encode()
245
+ tail = f"\r\n--{edge}--\r\n".encode()
246
+ body = field + content + checksum + tail
247
+ request = urllib.request.Request(
248
+ f"{self.base}{INGESTION}/call-executions/{call_execution_id}/recording/",
249
+ data=body,
250
+ headers={
251
+ "Content-Type": f"multipart/form-data; boundary={edge}",
252
+ **self._headers(),
253
+ },
254
+ )
255
+ try:
256
+ with _open(request) as answer:
257
+ return json.loads(answer.read().decode() or "{}")
258
+ except urllib.error.HTTPError as refused:
259
+ detail = refused.read().decode(errors="replace")[:300]
260
+ raise PlatformError(
261
+ f"recording upload failed ({refused.code}): {detail}"
262
+ ) from refused
263
+ except Exception as unreachable: # noqa: BLE001 - reported, not handled
264
+ raise PlatformError(
265
+ f"recording could not be sent: {unreachable}"
266
+ ) from unreachable
267
+
268
+
269
+ def persona_of(scenario: Any) -> dict[str, Any]:
270
+ """One scenario as the platform's persona record.
271
+
272
+ ``persona`` is carried whole so the simulator prompt's placeholder resolves against the same
273
+ person the scenario was written for, rather than a name reconstructed from it.
274
+ """
275
+ persona = getattr(scenario, "persona", None) or {}
276
+ if hasattr(persona, "model_dump"):
277
+ persona = persona.model_dump()
278
+ elif not isinstance(persona, dict):
279
+ persona = {}
280
+ return {
281
+ "name": str(persona.get("name") or getattr(scenario, "name", "") or "caller")[
282
+ :255
283
+ ],
284
+ # The scenario's own key, not its folder name: the key is ASCII-sanitised and falls back
285
+ # to a digest, which the name does not, and this value travels as an HTTP header.
286
+ "scenario_key": str(
287
+ getattr(scenario, "scenario_key", "") or getattr(scenario, "name", "") or ""
288
+ )[:255],
289
+ "scenario_name": display_scenario_name(scenario),
290
+ "role": str(persona.get("role") or persona.get("occupation") or "")[:255],
291
+ "situation": str(getattr(scenario, "instruction", "") or ""),
292
+ "outcome": str(getattr(scenario, "tests", "") or ""),
293
+ "persona": persona,
294
+ }
295
+
296
+
297
+ # What the harness calls a speaker, and what a transcript row is called on the platform. Anything
298
+ # unrecognised is the person, because the agent's turns are the ones we name.
299
+ SPEAKERS = {
300
+ "agent": "assistant",
301
+ "assistant": "assistant",
302
+ "bot": "assistant",
303
+ "system": "system",
304
+ "customer": "user",
305
+ "caller": "user",
306
+ "user": "user",
307
+ "tester": "user",
308
+ }
309
+
310
+
311
+ def segments_of(result: Any) -> list[dict[str, Any]]:
312
+ """A run's conversation as transcript rows, tool calls included.
313
+
314
+ No timings are invented. A typed run has none to give, and a made-up millisecond would be
315
+ indistinguishable from a measured one to everything downstream that averages them. A spoken
316
+ turn the runner timed carries those times through, because the platform derives duration,
317
+ silence, talk ratio and latency from them and can derive none of it from zeros.
318
+ """
319
+ rows: list[dict[str, Any]] = []
320
+ for turn in getattr(result, "exchanges", None) or []:
321
+ said = str(turn.get("text") or "").strip()
322
+ if not said:
323
+ continue
324
+ row = {
325
+ "speaker_role": SPEAKERS.get(str(turn.get("speaker", "")).lower(), "user"),
326
+ "content": said,
327
+ }
328
+ for when in ("start_time_ms", "end_time_ms"):
329
+ if turn.get(when) is not None:
330
+ row[when] = int(turn[when])
331
+ rows.append(row)
332
+ for call in getattr(result, "calls_detail", None) or []:
333
+ rows.append(
334
+ {
335
+ "speaker_role": "tool_calls",
336
+ "content": f"{call.get('name', '')}({json.dumps(call.get('arguments', {}), default=str)})",
337
+ }
338
+ )
339
+ outcome = call.get("error") or call.get("result") or ""
340
+ rows.append(
341
+ {
342
+ "speaker_role": "tool_call_result",
343
+ "content": ("refused: " if call.get("refused") else "")
344
+ + str(outcome)[:4000],
345
+ }
346
+ )
347
+ return rows
348
+
349
+
350
+ def evaluations_of(result: Any) -> list[dict[str, Any]]:
351
+ """Everything this run judged, as one list the platform can render per call.
352
+
353
+ Two kinds arrive from different places and mean different things, so both are named and
354
+ kept apart rather than averaged into a verdict. A sub-goal is deterministic: the world was
355
+ left in a state, or it was not. A metric is scored: the run placed it somewhere between
356
+ nothing and everything. Reporting only the first is what made a scored run look unjudged.
357
+ """
358
+ judged: list[dict[str, Any]] = []
359
+ for check in getattr(result, "checkpoints", None) or []:
360
+ decided_by = str(getattr(check, "by", "") or "")
361
+ # Platform-backed judgements carry ``<template name> (<model>)`` in
362
+ # ``by``. Keep the exact template name on the wire so ingestion can
363
+ # attach the already-computed result to that template/config instead of
364
+ # merely leaving a second, disconnected EvalTemplate in the library.
365
+ platform_template = decided_by.rsplit(" (", 1)[0] if decided_by else ""
366
+ evaluation = {
367
+ "name": getattr(check, "name", ""),
368
+ "kind": getattr(check, "kind", "") or "checkpoint",
369
+ "passed": bool(getattr(check, "passed", False)),
370
+ "reason": str(getattr(check, "detail", ""))[:2000],
371
+ "decided_by": decided_by[:2000],
372
+ "platform_template": platform_template[:2000],
373
+ }
374
+ if getattr(check, "grading_error", False):
375
+ evaluation["grading_error"] = True
376
+ judged.append(evaluation)
377
+ for metric in (getattr(result, "measured", None) or {}).get("metrics") or []:
378
+ if not metric.get("applicable", True):
379
+ continue
380
+ judged.append(
381
+ {
382
+ "name": str(metric.get("name", "")),
383
+ "kind": "metric",
384
+ "score": float(metric.get("score", 0.0) or 0.0),
385
+ "reason": str(metric.get("reason", ""))[:2000],
386
+ }
387
+ )
388
+ return [one for one in judged if one["name"]]
389
+
390
+
391
+ def conversation_seconds(result: Any) -> int:
392
+ """Return user-visible conversation time, excluding setup and retry waits."""
393
+ exchanges = getattr(result, "exchanges", None) or []
394
+ starts = [
395
+ float(exchange["start_time_ms"])
396
+ for exchange in exchanges
397
+ if isinstance(exchange, dict)
398
+ and isinstance(exchange.get("start_time_ms"), (int, float))
399
+ ]
400
+ ends = [
401
+ float(exchange["end_time_ms"])
402
+ for exchange in exchanges
403
+ if isinstance(exchange, dict)
404
+ and isinstance(exchange.get("end_time_ms"), (int, float))
405
+ ]
406
+ if starts and ends:
407
+ return max(0, int((max(ends) - min(starts)) / 1000))
408
+ if not exchanges and getattr(result, "problems", None):
409
+ return 0
410
+ return max(0, int(getattr(result, "seconds", 0) or 0))
411
+
412
+
413
+ def result_of(result: Any) -> dict[str, Any]:
414
+ """One scenario's outcome, in the shape the ingestion API takes.
415
+
416
+ Sub-goals travel in ``call_metadata`` rather than as free text: they are what this run
417
+ actually decided, and a page showing one goal per column needs them named and separate.
418
+ """
419
+ checkpoints = [
420
+ {
421
+ "name": getattr(check, "name", ""),
422
+ "kind": getattr(check, "kind", ""),
423
+ "passed": bool(getattr(check, "passed", False)),
424
+ "detail": str(getattr(check, "detail", ""))[:2000],
425
+ }
426
+ for check in getattr(result, "checkpoints", None) or []
427
+ ]
428
+ problems = list(getattr(result, "problems", None) or [])
429
+ grading_failures = list(getattr(result, "grading_failures", None) or [])
430
+ evaluations = evaluations_of(result)
431
+ evaluation_coverage = {
432
+ "expected": len(evaluations),
433
+ "executed": sum(
434
+ 1 for evaluation in evaluations if not evaluation.get("grading_error")
435
+ ),
436
+ "failed": sum(
437
+ 1 for evaluation in evaluations if evaluation.get("grading_error")
438
+ ),
439
+ "complete": not grading_failures,
440
+ }
441
+ payload: dict[str, Any] = {
442
+ # A scenario that never ran is not a scenario the agent failed, and the two must not
443
+ # arrive as the same status.
444
+ "status": "failed" if problems else "completed",
445
+ "duration_seconds": conversation_seconds(result),
446
+ "ended_reason": (getattr(result, "ended", "") or "")[:10000],
447
+ "call_summary": (getattr(result, "line", lambda: "")() or "")[:2000],
448
+ "transcript": segments_of(result),
449
+ "call_metadata": {
450
+ "harness_scenario": getattr(result, "scenario", ""),
451
+ "harness_passed": bool(getattr(result, "passed", False)),
452
+ "harness_met": int(getattr(result, "met", 0) or 0),
453
+ "harness_of": len(checkpoints),
454
+ "harness_checkpoints": checkpoints,
455
+ # Platform evaluations are backend-owned. Harness checks are direct
456
+ # execution evidence and stay namespaced in metadata rather than
457
+ # being sent through the removed SDK `evaluations` input field.
458
+ "harness_evaluations": evaluations,
459
+ "harness_eval_coverage": evaluation_coverage,
460
+ "harness_failure_classification": (
461
+ "grading_failure" if grading_failures else ""
462
+ ),
463
+ "harness_spent_usd": round(
464
+ float(getattr(result, "spent_usd", 0.0) or 0.0), 4
465
+ ),
466
+ },
467
+ }
468
+ if problems:
469
+ payload["error_message"] = "; ".join(problems)[:2000]
470
+ payload["result_digest"] = (
471
+ "sha256:"
472
+ + hashlib.sha256(
473
+ json.dumps(
474
+ payload, sort_keys=True, separators=(",", ":"), default=str
475
+ ).encode()
476
+ ).hexdigest()
477
+ )
478
+ return payload
479
+
480
+
481
+ def report(
482
+ results: list[Any],
483
+ scenarios: list[Any],
484
+ *,
485
+ name: str,
486
+ run_test_id: str = "",
487
+ modality: str = "text",
488
+ platform: Platform | None = None,
489
+ ) -> Reported:
490
+ """Report one suite run, and say where it landed.
491
+
492
+ ``run_test_id`` is reused when the session already has one, so a second run adds a second
493
+ execution to the same test rather than a second test with one run in it.
494
+
495
+ ``modality`` decides how the run is rendered: a spoken call reported as text lands in the
496
+ chat view, with no player and no audio, whatever actually happened on it.
497
+ """
498
+ api = platform or Platform()
499
+ reported, ids = begin(
500
+ scenarios,
501
+ name=name,
502
+ run_test_id=run_test_id,
503
+ modality=modality,
504
+ platform=api,
505
+ )
506
+
507
+ # Calls come back in the order the scenarios were attached, which is the order they were run
508
+ # in. Zip rather than assume equal length: a suite can be a subset of its own test.
509
+ for call_execution_id, result in zip(ids, results, strict=False):
510
+ send_result(reported, call_execution_id, result, platform=api)
511
+ if len(ids) < len(results):
512
+ reported.problems.append(
513
+ f"the platform allocated {len(ids)} calls for {len(results)} scenarios, "
514
+ "so the rest were not reported"
515
+ )
516
+ return reported
517
+
518
+
519
+ def begin(
520
+ scenarios: list[Any],
521
+ *,
522
+ name: str,
523
+ run_test_id: str = "",
524
+ modality: str = "text",
525
+ platform: Platform | None = None,
526
+ ) -> tuple[Reported, list[str]]:
527
+ """Create the platform rows before a suite starts, so the run is visible while it runs."""
528
+ api = platform or Platform()
529
+ reported = Reported(run_test_id=run_test_id)
530
+ provisioned_scenario_ids: list[str] = []
531
+ if not reported.run_test_id:
532
+ provisioned = api.provision(
533
+ name, [persona_of(one) for one in scenarios], modality=modality
534
+ )
535
+ reported.run_test_id = str(provisioned.get("run_test_id", ""))
536
+ # The provision endpoint returns IDs in the submitted persona order.
537
+ # Pass that order into execution creation; relying on a many-to-many
538
+ # queryset's database order can attach the right result to the wrong
539
+ # scenario row in the platform UI.
540
+ provisioned_scenario_ids = [
541
+ str(one) for one in provisioned.get("scenario_ids", [])
542
+ ]
543
+ if not reported.run_test_id:
544
+ raise PlatformError("the platform returned no run test to report against")
545
+
546
+ harness_job_id = os.getenv("ALK_HARNESS_JOB_ID", "").strip()
547
+ selectors = [
548
+ {
549
+ "scenario_key": str(getattr(one, "name", "") or "")[:255],
550
+ "persona_name": str(persona_of(one).get("name") or "")[:255],
551
+ }
552
+ for one in scenarios
553
+ ]
554
+ selector_kwargs = {"scenario_selectors": selectors} if selectors else {}
555
+ if harness_job_id:
556
+ started = api.start(
557
+ reported.run_test_id,
558
+ provisioned_scenario_ids,
559
+ harness_job_id=harness_job_id,
560
+ **selector_kwargs,
561
+ )
562
+ else:
563
+ # Keep the long-standing Platform-compatible call shape for local SDK and
564
+ # third-party implementations. The hosted ownership reference is additive
565
+ # and only exists inside a sandbox worker.
566
+ started = api.start(
567
+ reported.run_test_id,
568
+ provisioned_scenario_ids,
569
+ **selector_kwargs,
570
+ )
571
+ reported.test_execution_id = str(started.get("test_execution_id", ""))
572
+ if not reported.test_execution_id:
573
+ raise PlatformError("the platform returned no test execution for this run")
574
+ claimed = api.batch(reported.test_execution_id, max(1, len(scenarios)))
575
+ return reported, [str(one) for one in claimed.get("call_execution_ids", [])]
576
+
577
+
578
+ def send_result(
579
+ reported: Reported,
580
+ call_execution_id: str,
581
+ result: Any,
582
+ *,
583
+ platform: Platform | None = None,
584
+ ) -> None:
585
+ """Patch one pre-allocated platform row as soon as its scenario finishes."""
586
+ api = platform or Platform()
587
+ try:
588
+ api.result(call_execution_id, result_of(result))
589
+ reported.calls[getattr(result, "scenario", "")] = call_execution_id
590
+ audio = str(getattr(result, "recording", "") or "")
591
+ if audio and Path(audio).exists():
592
+ try:
593
+ api.recording(call_execution_id, Path(audio))
594
+ except PlatformError as refused:
595
+ reported.problems.append(f"recording not sent: {refused}")
596
+ except PlatformError as failed:
597
+ reported.problems.append(f"{getattr(result, 'scenario', '?')}: {failed}")
598
+
599
+
600
+ def mark_ongoing(
601
+ reported: Reported,
602
+ call_execution_id: str,
603
+ *,
604
+ platform: Platform | None = None,
605
+ ) -> None:
606
+ """Best-effort PENDING -> ONGOING transition when a scenario actually starts.
607
+
608
+ A status ping is presentation state, not result evidence. The backend applies it only to a
609
+ pending call, so a duplicate or late ping cannot overwrite a terminal result. Failure here
610
+ must not fail the call or enter result-reconciliation bookkeeping.
611
+ """
612
+ if not call_execution_id:
613
+ return
614
+ try:
615
+ (platform or Platform()).ongoing(call_execution_id)
616
+ except PlatformError:
617
+ pass
618
+
619
+
620
+ def deliver(
621
+ results: list[Any],
622
+ scenarios: list[Any],
623
+ destination: Path | None,
624
+ *,
625
+ modality: str = "text",
626
+ ) -> tuple[Reported | None, list[str]]:
627
+ """Report a finished run, and say what happened, without ever failing the run.
628
+
629
+ Shared by every way a suite can be started, because a run that only appears on the platform
630
+ when it was started from one particular button is worse than one that never appears: which
631
+ runs exist then depends on how they were launched, and nobody can tell that from the page.
632
+
633
+ The suite has finished and its results are on disk by the time this is called, so an
634
+ unreachable platform is worth saying out loud and not worth throwing a completed run away
635
+ over. Returns what was reported, if anything, and the lines to show whoever asked.
636
+ """
637
+ blocked = configured()
638
+ if blocked:
639
+ return None, [f"not reported to the platform: {blocked}"]
640
+ try:
641
+ reported = report(
642
+ results,
643
+ scenarios,
644
+ name=(destination.name if destination else "harness run"),
645
+ run_test_id=remembered(destination) if destination else "",
646
+ modality=modality,
647
+ )
648
+ except PlatformError as failed:
649
+ return None, [
650
+ f"the run finished, but reporting it to the platform failed: {failed}"
651
+ ]
652
+ if destination:
653
+ remember(destination, reported)
654
+ said = [f"partly reported: {problem}" for problem in reported.problems]
655
+ said.append(f"reported to the platform: {reported.url}")
656
+ return reported, said
657
+
658
+
659
+ def remember(destination: Path, reported: Reported) -> None:
660
+ """Keep where a session reports to, so its next run joins the same test."""
661
+ (Path(destination) / "platform.json").write_text(
662
+ json.dumps(
663
+ {
664
+ "run_test_id": reported.run_test_id,
665
+ "test_execution_id": reported.test_execution_id,
666
+ "url": reported.url,
667
+ },
668
+ indent=2,
669
+ ),
670
+ encoding="utf-8",
671
+ )
672
+
673
+
674
+ def reported_to(destination: Path | None) -> dict[str, str]:
675
+ """Where this session's runs have been reported, or nothing.
676
+
677
+ Read rather than held in memory, because a session outlives the process that reported it:
678
+ reopening one has to be able to find the run it already has.
679
+ """
680
+ kept = Path(destination) / "platform.json" if destination else None
681
+ if kept is None or not kept.exists():
682
+ return {}
683
+ try:
684
+ found = json.loads(kept.read_text(encoding="utf-8"))
685
+ except Exception: # noqa: BLE001 - a damaged file just means provisioning again
686
+ return {}
687
+ return found if isinstance(found, dict) else {}
688
+
689
+
690
+ def remembered(destination: Path) -> str:
691
+ """The run test this session already has, or an empty string."""
692
+ return str(reported_to(destination).get("run_test_id", ""))