agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,879 @@
1
+ """The scenario-source adapter: reads generated scenario documents out of the bundle (code-as-text,
2
+ on the on-disk layout `folder.py` documents) and turns them into the `Scenario`/`SubGoal` objects
3
+ `hosted_scheduler.py` actually drives.
4
+
5
+ Deliberately does NOT import `fi.alk.harness.folder` or `fi.alk.harness.scenario` for the model:
6
+ both exist at HEAD, but HEAD's `Scenario` carries no `scenario_key`/`scenario_id` (those are
7
+ pr63-only) and its default `extra="ignore"` would silently discard exactly the two fields the
8
+ scheduler needs off a `scenario.json` written in the newer shape. So this module reads
9
+ `scenario.json` as a plain dict and pulls fields out by key, mirroring the documented layout
10
+ instead of depending on either model -- see the report's design-decisions section for the
11
+ consequences of that choice (HEAD-model drift).
12
+
13
+ RESOLVED (p13-worker-r2, reports/p13-worker-r2.md CONTRACT NOTES): the `provision`/`begin` wire
14
+ shapes below follow the platform's actual, live route (futureagi/simulate/serializers/services/
15
+ views `hosted_harness.py`) rather than the Scenario Generation Contract text (PR #63), where
16
+ the two disagree -- a single `POST .../scenarios/` discriminated by a body-level `operation` field,
17
+ `begin` keyed on the full `scenario_keys` set, and a provision response KEYED by `scenario_key`
18
+ (never a position-ordered array). `register_with_platform` below is the seam that builds those
19
+ payloads and merges the platform-assigned `scenario_id`s back onto each scenario.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import asyncio
25
+ import json
26
+ import logging
27
+ from dataclasses import dataclass, field, replace
28
+ from pathlib import Path
29
+ from typing import TYPE_CHECKING, Any, Callable, Sequence
30
+
31
+ from . import outbound as ob
32
+ from .catalogue import CATALOGUE
33
+ from .job import FailureDomain
34
+
35
+ if TYPE_CHECKING:
36
+ from .hosted_entrypoint import ScenariosClient
37
+
38
+ # LAYOUT DECISION (contract-silent -- hosted-execution-seams.md v1.15 §2 never mentions scenario
39
+ # documents, and §7 assigns the on-disk layout to that contract, status "in review"). Scenario
40
+ # documents live at `<bundle_dir>/<SCENARIOS_DIRNAME>/<name>/...`, matching `folder.py`'s own
41
+ # `SCENARIOS` constant, so a write_folder destination of `<bundle_dir>` lands correctly with no
42
+ # translation. Kept as one module-level constant so a later contract can move it in one edit.
43
+ logger = logging.getLogger(__name__)
44
+
45
+ SCENARIOS_DIRNAME = "scenarios"
46
+
47
+ _CHECKS_DIRNAME = "checks"
48
+ _SCENARIO_JSON = "scenario.json"
49
+ _CATALOGUE_JSON = "sub_goals.json"
50
+ _SETUP_PY = "setup.py"
51
+ _READY_PY = "ready.py"
52
+
53
+ # R1-5: divergence (a) (compile once, at load) moves every scenario file's compile+exec into
54
+ # `asyncio.to_thread(load_scenarios, ...)`, OUTSIDE every phase budget the scheduler enforces
55
+ # (`SETUP_TIMEOUT_SECONDS`/`READY_TIMEOUT_SECONDS`/`CHECK_TIMEOUT_SECONDS` all apply downstream, to
56
+ # `_run_phase`). Pathological module-level code (`while True: pass`, a blocking socket read) would
57
+ # otherwise hang the load with zero terminal events, no timeout catching it, ever. One generous
58
+ # wall-clock budget here converts that hang into a typed terminal instead -- a worker thread cannot
59
+ # actually be killed, so this accepts a leaked thread over an unbounded one, on the reasoning that a
60
+ # terminal event today is strictly better than none ever.
61
+ _LOAD_TIMEOUT_SECONDS = 60.0
62
+
63
+
64
+ # Every scenario a job asked for is called. A job that wrote two hundred is asking for two hundred
65
+ # calls: that is the product, and calling fewer would quietly deliver a fraction of what somebody
66
+ # paid for. There is deliberately no setting here; a smaller run is a smaller `scenario_count`.
67
+ def sampled_for_calling(scenarios: Sequence[Any]) -> list[Any]:
68
+ """Which scenarios this job calls, which is all of them."""
69
+ return list(scenarios)
70
+
71
+
72
+ # The TEXT of a judged sub-goal's check is never persisted by `folder.py`'s `write_folder` (only
73
+ # `SubGoal.deterministic()` entries get a `checks/<name>.py` file) -- this fixed marker stands in
74
+ # for it so `SubGoal.judged` (mandatory; read by plain attribute access in `hosted_scheduler.py`)
75
+ # is never empty for a sub-goal this reader knows is judged. The round-trip loss this represents is
76
+ # recorded under CONTRACT QUESTIONS in the report.
77
+ _JUDGED_MARKER = "judged (reason not persisted by folder.py's on-disk layout)"
78
+
79
+
80
+ class ScenarioDocumentInvalid(RuntimeError):
81
+ """A scenario folder under `<bundle_dir>/scenarios/` is unreadable or malformed: missing
82
+ `scenario.json`, invalid JSON, a field of the wrong shape, or a `setup.py`/`ready.py`/
83
+ `checks/<goal>.py` that will not compile. Raised rather than skipped -- `folder.py`'s own
84
+ `read_all` swallows exactly this and continues, which is right for a human editing a suite by
85
+ hand and wrong for a hosted job, where a bad scenario silently vanishing from the run reads as
86
+ a suite that passed with fewer scenarios than it should have."""
87
+
88
+
89
+ def bundle_has_scenarios(bundle_dir: Path) -> bool:
90
+ """The LAYOUT DECISION's presence test: `<bundle_dir>/scenarios/` exists and at least one of
91
+ its subdirectories holds a `scenario.json`. Deliberately narrow -- an empty or missing
92
+ `scenarios/` must not flip the wiring decision away from the safe `NotWiredScenarioSource`
93
+ default. A bundle that means to carry scenarios but got the layout wrong fails loudly once
94
+ `load_scenarios` actually reads it, not silently by being treated as scenario-free here.
95
+
96
+ An unreadable `scenarios/` directory (permission denied, race with deletion, etc.) reports
97
+ False rather than raising `OSError` (R1-1): this call sits in `hosted_entrypoint.py`'s wiring
98
+ `if` BEFORE the `try`/`except` that maps `ScenarioDocumentInvalid` to a typed terminal, so an
99
+ escape here would kill the whole guest process with no terminal event at all. Falling back to
100
+ `NotWiredScenarioSource`'s existing typed failure is the safe direction -- the alternative of
101
+ raising here has nowhere typed to land.
102
+ """
103
+ root = bundle_dir / SCENARIOS_DIRNAME
104
+ if not root.is_dir():
105
+ return False
106
+ try:
107
+ children = list(root.iterdir())
108
+ except OSError:
109
+ return False
110
+ return any(
111
+ (child / _SCENARIO_JSON).is_file() for child in children if child.is_dir()
112
+ )
113
+
114
+
115
+ def _judged_placeholder_check(world: Any, calls: Any) -> None:
116
+ """The check for a judged (non-deterministic) sub-goal: always reports "held", via the same
117
+ convention a deterministic check uses. `SubGoalResult.judged` -- not this return value -- is
118
+ what has to tell downstream a real judge still has to run; see CONTRACT QUESTIONS in the
119
+ report for the gap that leaves (a judged-only scenario reports "passed" before any judge runs).
120
+ """
121
+ del world, calls
122
+ return None
123
+
124
+
125
+ def _compile_entry(
126
+ source: str, *, label: str, entry: str, allow_empty: bool = True
127
+ ) -> Callable[..., object]:
128
+ """One scenario code-text -> a bare callable that raises, compiled ONCE here rather than per
129
+ call. Mirrors `folder.py`'s `_run` in exactly two respects: `compile(source, name, "exec")`
130
+ into a fresh, empty-dict namespace with default builtins, and (when `allow_empty`)
131
+ empty/whitespace-only source is a no-op success. Deliberately diverges from `_run` in the two
132
+ respects the brief calls out:
133
+ (a) compiling here, at load, turns a syntax error into one typed terminal for the whole job
134
+ instead of a per-scenario fault discovered mid-run; (b) the compiled function is returned
135
+ BARE, never wrapped in `_run`'s `Outcome` -- `hosted_scheduler.py`'s `_run_phase` is what
136
+ classifies a raised exception into `setup_crashed`/`ready_broken`/`check_broken`/a timeout, and
137
+ an `Outcome` return here would swallow every one of those before `_run_phase` ever saw it.
138
+ `_run`'s complaint-sentence return convention (a non-None, non-True, non-empty-string value
139
+ means "did not hold") is left untranslated for the same reason: `hosted_scheduler.py`'s own
140
+ `_classify_ready`/`_classify_check` already own that classification on the return-VALUE side of
141
+ this boundary; only the raise-vs-return boundary belongs to this module.
142
+
143
+ `allow_empty=False` is for `check` entries only (R1-2): an EXISTING `checks/<name>.py` that is
144
+ empty or whitespace-only compiles to nothing, and handing back a no-op "held" callable -- the
145
+ right behavior for "no setup/ready code here" -- would silently turn "there is no check" into
146
+ "the check passed": a vacuous deterministic pass, which is exactly what `hosted_scheduler.py`
147
+ forbids one scenario level up ("scenario declared zero sub_goals"). Absence of the file
148
+ entirely is what means "judged" (see `_load_one`); an existing-but-empty file is malformed.
149
+ """
150
+ if not source.strip():
151
+ if allow_empty:
152
+ return lambda *args: None
153
+ raise ScenarioDocumentInvalid(f"{label} defines no {entry}()")
154
+ try:
155
+ code = compile(source, f"<{label}>", "exec")
156
+ except (SyntaxError, ValueError) as exc:
157
+ # SyntaxError is the common case; a NUL byte in the source raises ValueError on some
158
+ # interpreter versions (R1-1) rather than SyntaxError -- both are the same content defect.
159
+ raise ScenarioDocumentInvalid(f"{label} would not compile: {exc}") from exc
160
+ namespace: dict[str, Any] = {}
161
+ try:
162
+ exec(code, namespace) # noqa: S102 - scenario code is meant to be exec'd; see CONTRACT QUESTIONS
163
+ except (Exception, SystemExit, KeyboardInterrupt) as exc: # noqa: BLE001 - see R1-1
164
+ # A module-level `sys.exit()`/`raise SystemExit(...)` in the file itself is a malformed
165
+ # document, not a request to shut the guest process down -- `SystemExit`/`KeyboardInterrupt`
166
+ # are `BaseException`, not `Exception`, so a bare `except Exception` (the pre-R1-1 shape)
167
+ # let them straight through this boundary and out of `run_job` with zero terminal events,
168
+ # the guest exiting with whatever code the scenario file itself chose.
169
+ raise ScenarioDocumentInvalid(f"{label} would not compile: {exc}") from exc
170
+ function = namespace.get(entry)
171
+ if not callable(function):
172
+ raise ScenarioDocumentInvalid(f"{label} defines no {entry}()")
173
+ return function
174
+
175
+
176
+ @dataclass(frozen=True)
177
+ class _CompiledSubGoal:
178
+ """Satisfies `hosted_scheduler.SubGoal`: `name`/`judged` as plain attributes, `check` as a bare
179
+ callable. Stored as instance DATA rather than a `def check(self, world, calls)` method so
180
+ `goal.check(world, calls)` invokes the compiled function directly with exactly the two
181
+ positional arguments `_run_phase` passes -- a real method would prepend `self` as a third."""
182
+
183
+ name: str
184
+ judged: str
185
+ check: Callable[[Any, Any], object]
186
+ what: str = ""
187
+
188
+
189
+ @dataclass(frozen=True)
190
+ class _CompiledScenario:
191
+ """Satisfies `hosted_scheduler.Scenario`. `scenario_key`/`scenario_id` are carried VERBATIM
192
+ from the document, including an empty `scenario_id` -- synthesizing one here would hide that
193
+ pre-allocation has not actually run (see CONTRACT QUESTIONS: receipts carry `scenario_id ""`
194
+ until that seam is wired). `setup`/`ready` are likewise stored as data, for the same reason as
195
+ `_CompiledSubGoal.check` above."""
196
+
197
+ scenario_key: str
198
+ scenario_id: str
199
+ sub_goals: tuple[_CompiledSubGoal, ...]
200
+ setup: Callable[[Any], object]
201
+ ready: Callable[[Any], object]
202
+ requires_tool_evidence: bool = True
203
+ # Who this person is and what they came for, as the platform's own persona record. Read off the
204
+ # same document and sent at pre-allocation, so a call can be read on the platform without the
205
+ # scenario file beside it. Presentation only: nothing in the scheduler looks at it.
206
+ presented: dict[str, Any] = field(default_factory=dict)
207
+
208
+
209
+ def _read_text(path: Path, *, label: str) -> str:
210
+ """Missing is "" (mirrors `folder.py`'s own missing-setup/ready-is-empty convention); present
211
+ but unreadable (permission denied, a directory instead of a file) or present but not valid
212
+ UTF-8 is a typed `ScenarioDocumentInvalid`, never a raw `OSError`/`UnicodeDecodeError` escaping
213
+ this module (R1-1) -- both are equally "this scenario folder is malformed", the same
214
+ conclusion `_load_one`'s other reads already reach for a bad `scenario.json`.
215
+ """
216
+ if not path.exists():
217
+ return ""
218
+ try:
219
+ return path.read_text(encoding="utf-8")
220
+ except (OSError, UnicodeDecodeError) as exc:
221
+ raise ScenarioDocumentInvalid(
222
+ f"{label}: cannot read {path.name}: {exc}"
223
+ ) from exc
224
+
225
+
226
+ def _validate_subgoal_name(name: str, *, folder_name: str) -> None:
227
+ """R1-3: `sub_goals[]` entries are used verbatim to build `checks/<name>.py` -- a path
228
+ separator or a `..` segment lets a name escape `checks/` (and the sealed bundle) entirely: an
229
+ absolute name execs an arbitrary file never hashed into the bundle's `files[]` (bypassing the
230
+ §2e integrity seal), and a `../`-style traversal that resolves to nothing silently turns into a
231
+ JUDGED sub-goal (`check_path.is_file()` is False) instead of a typed failure. Rejecting
232
+ anything but a plain filename component closes both."""
233
+ if not name or "/" in name or "\\" in name or name in (".", ".."):
234
+ raise ScenarioDocumentInvalid(
235
+ f"{folder_name}: sub_goals name {name!r} is not a plain filename "
236
+ "(no path separators, no '..', no leading '/')"
237
+ )
238
+
239
+
240
+ def _presented(body: dict[str, Any], *, scenario_key: str) -> dict[str, Any]:
241
+ """The persona record the platform stores for a scenario, built by the platform module's own
242
+ helper rather than a second copy of it here, so both paths describe a person the same way.
243
+
244
+ Presentation only, and never load-bearing: anything missing from the document is simply absent
245
+ from what the platform shows, and a document this cannot read still runs.
246
+ """
247
+ from types import SimpleNamespace
248
+
249
+ from .platform import persona_of
250
+
251
+ # Derive a short display name for the scenario. The document may carry
252
+ # an explicit ``use_case``; when it does not, the scenario_key slug
253
+ # (e.g. ``refuse-booking-suspended-account``) is humanised so
254
+ # ``display_scenario_name`` never falls through to the long instruction
255
+ # text that ``name`` often contains in voice scenarios.
256
+ use_case = str(body.get("use_case") or "").strip()
257
+ if not use_case and scenario_key:
258
+ use_case = scenario_key.replace("-", " ").replace("_", " ")
259
+ use_case = use_case[:1].upper() + use_case[1:]
260
+
261
+ try:
262
+ return persona_of(
263
+ SimpleNamespace(
264
+ persona=body.get("persona") or {},
265
+ name=str(body.get("name") or ""),
266
+ scenario_key=scenario_key,
267
+ instruction=str(body.get("instruction") or ""),
268
+ tests=str(body.get("tests") or ""),
269
+ use_case=use_case,
270
+ )
271
+ )
272
+ except Exception: # noqa: BLE001 - a scenario never fails to run over how it is displayed
273
+ return {}
274
+
275
+
276
+ def _deterministic_names(bundle_dir: Path) -> set[str]:
277
+ """Sub-goals the catalogue settles in code, so a missing check file is a defect and not a judge.
278
+
279
+ Absence of `checks/<name>.py` is what marks a sub-goal judged, which is right when the catalogue
280
+ says nobody can settle it in code and wrong when the check simply never reached the folder: the
281
+ run then reports a judged verdict for something that was meant to be measured, and a judge asked
282
+ about state it cannot see tends to say yes.
283
+ """
284
+ path = bundle_dir / CATALOGUE
285
+ if not path.is_file():
286
+ return set()
287
+ try:
288
+ body = json.loads(path.read_text(encoding="utf-8"))
289
+ return {
290
+ str(one.get("name") or "")
291
+ for one in (body.get("sub_goals") or [])
292
+ if str(one.get("check") or "").strip()
293
+ } - {""}
294
+ except Exception: # noqa: BLE001 - an unreadable catalogue leaves the old behaviour
295
+ return set()
296
+
297
+
298
+ def _declared_tool_names(bundle_dir: Path) -> set[str]:
299
+ """Return the target tools declared by the authored contract.
300
+
301
+ The solution format intentionally permits semantic steps that are not executable tools. The
302
+ contract's tool inventory is therefore the only stable way to decide whether a runtime tool
303
+ trace is required. A malformed or absent inventory is treated as empty here; contract
304
+ validation owns reporting malformed contract content, while this reader must not invent tool
305
+ requirements that the target itself never declared.
306
+ """
307
+ path = bundle_dir / "contract.json"
308
+ if not path.is_file():
309
+ return set()
310
+ try:
311
+ body = json.loads(path.read_text(encoding="utf-8"))
312
+ tools = body.get("tools") if isinstance(body, dict) else None
313
+ if not isinstance(tools, list):
314
+ return set()
315
+ return {
316
+ str(tool.get("name") or "").strip()
317
+ for tool in tools
318
+ if isinstance(tool, dict)
319
+ } - {""}
320
+ except Exception: # noqa: BLE001 - contract validation reports the content defect
321
+ return set()
322
+
323
+
324
+ def _load_catalogue_claims(bundle_dir: Path) -> dict[str, dict[str, str]]:
325
+ """`sub_goals.json`'s `what`/`judged` text, which `folder.py` never writes into a scenario
326
+ folder. Without it a judged sub-goal reaches the platform as a name and nothing to decide.
327
+ """
328
+ path = bundle_dir / _CATALOGUE_JSON
329
+ if not path.is_file():
330
+ # Without this every sub-goal reaches the platform with no description, so a pass explains
331
+ # itself as "the check found nothing wrong" and a judged one arrives with nothing to
332
+ # decide. Said out loud because the symptom shows up two systems away from the cause.
333
+ logger.warning(
334
+ "no %s beside the scenarios in %s: sub-goals will carry no description or claim",
335
+ _CATALOGUE_JSON,
336
+ bundle_dir,
337
+ )
338
+ return {}
339
+ try:
340
+ raw = json.loads(path.read_text(encoding="utf-8"))
341
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError):
342
+ return {}
343
+ entries = raw.get("sub_goals") if isinstance(raw, dict) else None
344
+ if not isinstance(entries, list):
345
+ return {}
346
+ claims: dict[str, dict[str, str]] = {}
347
+ for entry in entries:
348
+ if not isinstance(entry, dict) or not isinstance(entry.get("name"), str):
349
+ continue
350
+ claims[entry["name"]] = {
351
+ "what": str(entry.get("what") or ""),
352
+ "judged": str(entry.get("judged") or ""),
353
+ }
354
+ return claims
355
+
356
+
357
+ def _load_one(
358
+ folder: Path,
359
+ *,
360
+ settled_in_code: set[str] | None = None,
361
+ declared_tools: set[str] | None = None,
362
+ ) -> _CompiledScenario:
363
+ """One scenario folder -> a `Scenario`-protocol object. Mirrors `folder.py`'s documented
364
+ layout (`scenario.json` + `setup.py` + `ready.py` + `checks/<goal>.py`) but reads
365
+ `scenario.json` itself as a plain dict rather than through `fi.alk.harness.scenario.Scenario`
366
+ -- see the module docstring. `folder.py`'s `read_folder` restores only `setup_code`/
367
+ `ready_code` from a folder; it does not read `checks/` at all, so every `checks/<goal>.py` for
368
+ each name in the document's `sub_goals` is read here, by this module, directly.
369
+ """
370
+ body_path = folder / _SCENARIO_JSON
371
+ try:
372
+ raw = body_path.read_text(encoding="utf-8")
373
+ except (OSError, UnicodeDecodeError) as exc:
374
+ # UnicodeDecodeError is a ValueError, not an OSError -- widened alongside it (R1-1) so a
375
+ # non-UTF-8 `scenario.json` is the same typed failure as an unreadable one, not an escape.
376
+ raise ScenarioDocumentInvalid(
377
+ f"{folder.name}: cannot read {_SCENARIO_JSON}: {exc}"
378
+ ) from exc
379
+ try:
380
+ body = json.loads(raw)
381
+ except json.JSONDecodeError as exc:
382
+ raise ScenarioDocumentInvalid(
383
+ f"{folder.name}: {_SCENARIO_JSON} is not valid JSON: {exc}"
384
+ ) from exc
385
+ if not isinstance(body, dict):
386
+ raise ScenarioDocumentInvalid(
387
+ f"{folder.name}: {_SCENARIO_JSON} is not a JSON object"
388
+ )
389
+
390
+ scenario_key = body.get("scenario_key", "")
391
+ if not isinstance(scenario_key, str):
392
+ raise ScenarioDocumentInvalid(f"{folder.name}: scenario_key is not a string")
393
+ scenario_id = body.get("scenario_id", "")
394
+ if not isinstance(scenario_id, str):
395
+ raise ScenarioDocumentInvalid(f"{folder.name}: scenario_id is not a string")
396
+ sub_goal_names = body.get("sub_goals", [])
397
+ if not isinstance(sub_goal_names, list) or not all(
398
+ isinstance(name, str) for name in sub_goal_names
399
+ ):
400
+ raise ScenarioDocumentInvalid(
401
+ f"{folder.name}: sub_goals is not a list of strings"
402
+ )
403
+ for name in sub_goal_names:
404
+ _validate_subgoal_name(name, folder_name=folder.name)
405
+
406
+ solution = body.get("solution", [])
407
+ if not isinstance(solution, list) or not all(
408
+ isinstance(step, dict) for step in solution
409
+ ):
410
+ raise ScenarioDocumentInvalid(
411
+ f"{folder.name}: solution is not a list of objects"
412
+ )
413
+ solution_tools = [step.get("tool") for step in solution]
414
+ if not all(isinstance(tool, str) and tool.strip() for tool in solution_tools):
415
+ raise ScenarioDocumentInvalid(
416
+ f"{folder.name}: every solution step must name a non-empty tool"
417
+ )
418
+ # A solution may contain semantic narration steps (for example ``listen`` and ``respond``)
419
+ # alongside real agent tool calls. The contract is the authoritative inventory of tools the
420
+ # target can actually emit, so evidence is required exactly when a solution references one of
421
+ # those declared tools. Inferring this from arbitrary step names creates false infrastructure
422
+ # failures for conversational agents and requires an ever-growing pseudo-tool denylist.
423
+ requires_tool_evidence = bool(set(solution_tools) & (declared_tools or set()))
424
+
425
+ setup_code = _read_text(folder / _SETUP_PY, label=folder.name)
426
+ ready_code = _read_text(folder / _READY_PY, label=folder.name)
427
+ setup = _compile_entry(
428
+ setup_code, label=f"{folder.name}/{_SETUP_PY}", entry="setup"
429
+ )
430
+ ready = _compile_entry(
431
+ ready_code, label=f"{folder.name}/{_READY_PY}", entry="ready"
432
+ )
433
+
434
+ sub_goals: list[_CompiledSubGoal] = []
435
+ for name in sub_goal_names:
436
+ check_path = folder / _CHECKS_DIRNAME / f"{name}.py"
437
+ if check_path.is_file():
438
+ check_code = _read_text(check_path, label=folder.name)
439
+ check = _compile_entry(
440
+ check_code,
441
+ label=f"{folder.name}/{_CHECKS_DIRNAME}/{name}.py",
442
+ entry="check",
443
+ allow_empty=False, # R1-2: an existing-but-empty check file is invalid, never a
444
+ # vacuous pass -- absence of the file is what means "judged".
445
+ )
446
+ judged = ""
447
+ elif name in (settled_in_code or set()):
448
+ # The catalogue settles this one in code and the file is not here, so the check was
449
+ # written and never materialised. Reporting it as judged is the expensive wrong answer:
450
+ # the scenario runs, the checkpoint reads as assessed, and nothing measured it.
451
+ raise ScenarioDocumentInvalid(
452
+ f"{folder.name}: sub-goal {name!r} is settled in code by the catalogue but "
453
+ f"{_CHECKS_DIRNAME}/{name}.py is missing, so nothing would measure it"
454
+ )
455
+ else:
456
+ # No `checks/<name>.py` -- per `write_folder`'s own `deterministic()` filter, this
457
+ # name is a JUDGED sub-goal.
458
+ judged = _JUDGED_MARKER
459
+ check = _judged_placeholder_check
460
+ sub_goals.append(_CompiledSubGoal(name=name, judged=judged, check=check))
461
+
462
+ return _CompiledScenario(
463
+ scenario_key=scenario_key,
464
+ scenario_id=scenario_id,
465
+ sub_goals=tuple(sub_goals),
466
+ requires_tool_evidence=requires_tool_evidence,
467
+ setup=setup,
468
+ ready=ready,
469
+ presented=_presented(body, scenario_key=scenario_key),
470
+ )
471
+
472
+
473
+ def _with_claims(
474
+ scenario: _CompiledScenario, claims: dict[str, dict[str, str]]
475
+ ) -> _CompiledScenario:
476
+ """Restore each sub-goal's real claim from the catalogue.
477
+
478
+ `_load_one` can only tell that a sub-goal is judged, never what it was meant to decide:
479
+ `folder.py` writes no file for one. Without this the platform judge gets a name and a
480
+ placeholder, which is not something a verdict can be reached from.
481
+
482
+ `what` is restored for CODED sub-goals too, not only judged ones. A check says nothing when it
483
+ holds, so `what` is the only thing a reader has to tell a real pass from one nobody wrote a
484
+ check for; withholding it left every passing sub-goal explaining itself as "the check found
485
+ nothing wrong". `judged` stays restricted to judged sub-goals, since a coded one has no claim
486
+ for a model to decide.
487
+ """
488
+ if not claims:
489
+ return scenario
490
+ restored = tuple(
491
+ replace(
492
+ goal,
493
+ judged=(claims[goal.name].get("judged") or goal.judged)
494
+ if goal.judged
495
+ else goal.judged,
496
+ what=claims[goal.name].get("what", "") or goal.what,
497
+ )
498
+ if goal.name in claims
499
+ else goal
500
+ for goal in scenario.sub_goals
501
+ )
502
+ return replace(scenario, sub_goals=restored)
503
+
504
+
505
+ def load_scenarios(bundle_dir: Path) -> list[_CompiledScenario]:
506
+ """Every scenario document under `<bundle_dir>/scenarios/`, compiled and wrapped, in the same
507
+ sorted-by-folder-name order `folder.py`'s `read_all` uses. Raises `ScenarioDocumentInvalid` on
508
+ the FIRST unreadable or malformed folder -- unlike `read_all`, which skips one and continues;
509
+ a hosted job has nobody watching a suite by hand to notice a scenario silently missing from the
510
+ count, so a folder this reader cannot use fails the whole job instead of shrinking it quietly.
511
+ """
512
+ root = bundle_dir / SCENARIOS_DIRNAME
513
+ if not root.is_dir():
514
+ raise ScenarioDocumentInvalid(f"{root} is not a directory")
515
+ try:
516
+ entries = sorted(root.iterdir())
517
+ except OSError as exc:
518
+ # An unreadable `scenarios/` directory is the same typed failure as any other malformed
519
+ # document (R1-1) -- this is inside `run_job`'s `try`/`except ScenarioDocumentInvalid`
520
+ # (unlike `bundle_has_scenarios`'s own guard above), so raising here is the safe direction.
521
+ raise ScenarioDocumentInvalid(
522
+ f"{root}: cannot list scenario folders: {exc}"
523
+ ) from exc
524
+ settled_in_code = _deterministic_names(bundle_dir)
525
+ declared_tools = _declared_tool_names(bundle_dir)
526
+ claims = _load_catalogue_claims(bundle_dir)
527
+ scenarios: list[_CompiledScenario] = []
528
+ for folder in entries:
529
+ if not folder.is_dir():
530
+ continue
531
+ scenarios.append(
532
+ _with_claims(
533
+ _load_one(
534
+ folder,
535
+ settled_in_code=settled_in_code,
536
+ declared_tools=declared_tools,
537
+ ),
538
+ claims,
539
+ )
540
+ )
541
+ if not scenarios:
542
+ raise ScenarioDocumentInvalid(f"{root} contains no scenario folders")
543
+ return scenarios
544
+
545
+
546
+ class BundleScenarioSource:
547
+ """The real `ScenarioSource`: reads and compiles the bundle's own scenario documents.
548
+ `hosted_entrypoint.run_job` wires this in only when the injected source is still the default
549
+ `NotWiredScenarioSource` AND the bundle actually carries a `scenarios/` directory (the LAYOUT
550
+ DECISION's presence test) -- an injected `ScenarioSource` (every test, every future caller)
551
+ always wins over this one.
552
+ """
553
+
554
+ async def build(
555
+ self,
556
+ job: Any,
557
+ bundle: Any,
558
+ scenarios_client: "ScenariosClient",
559
+ *,
560
+ pool: Any,
561
+ world_factory: Any,
562
+ bundle_dir: Path,
563
+ ) -> Sequence[_CompiledScenario]:
564
+ del bundle, pool, world_factory
565
+ # `Path.read_text`/`iterdir`/`compile` are all blocking filesystem+CPU work -- run off the
566
+ # event loop the same way `hosted_entrypoint.py` already does for `bundle_source.load` and
567
+ # `preflight_bundle`, rather than stalling every other in-flight scenario behind it.
568
+ try:
569
+ scenarios = await asyncio.wait_for(
570
+ asyncio.to_thread(load_scenarios, bundle_dir),
571
+ timeout=_LOAD_TIMEOUT_SECONDS,
572
+ )
573
+ except asyncio.TimeoutError as exc:
574
+ # R1-5: the underlying thread cannot actually be canceled/killed -- it is left running
575
+ # in the background. Converting the hang into a typed terminal here is still strictly
576
+ # better than the pre-fix behavior (no terminal event, ever): the job gets an honest,
577
+ # bounded FAILED verdict instead of hanging until the platform's own wall clock gives up.
578
+ raise ScenarioDocumentInvalid(
579
+ f"{bundle_dir / SCENARIOS_DIRNAME}: loading scenario documents exceeded "
580
+ f"{_LOAD_TIMEOUT_SECONDS:.0f}s"
581
+ ) from exc
582
+ # An empty `scenario_key` is carried VERBATIM off the document by design (module docstring
583
+ # -- never synthesized here), but it is also the one shape `hosted_entrypoint.py`'s own
584
+ # downstream `_validate_scenarios` would reject as a local, deterministic ENVIRONMENT-domain
585
+ # content defect -- checked HERE, before `register_with_platform` ever reaches the network,
586
+ # so that cheaper, existing local classification wins over a round trip that would only
587
+ # rediscover the same defect as a `platform_sync` failure instead (Azain's serializer
588
+ # rejects a blank `scenario_key` with its own 400 -- `scenario_key` is a plain
589
+ # non-`allow_blank` `CharField`, hosted_harness.py:169). `ScenarioDocumentInvalid` reuses
590
+ # `run_job`'s EXISTING `except ScenarioDocumentInvalid` clause (domain=environment) --
591
+ # nothing new to catch there.
592
+ if any(not scenario.scenario_key for scenario in scenarios):
593
+ raise ScenarioDocumentInvalid(
594
+ f"{bundle_dir / SCENARIOS_DIRNAME}: a scenario document has no non-empty "
595
+ "scenario_key"
596
+ )
597
+ # p13: pre-allocation, after load and before the scheduler ever sees a scenario (spine
598
+ # step 3.5) -- `register_with_platform` raises `ScenarioPreallocationError`/
599
+ # `ob.HostedFencedError`/`ob.HostedChannelFailedError`/`ob.HostedAttemptSupersededError` on
600
+ # any failure, all of which `hosted_entrypoint.run_job`'s existing call site around
601
+ # `scenario_source.build()` already maps to the typed `validating_scenarios`/`platform_sync`
602
+ # terminal (or the fenced exit) -- nothing new to catch here.
603
+ # Pre-allocate the WHOLE suite, then run a sample of it.
604
+ #
605
+ # The sample used to be taken first, so the platform was handed five personas for a suite of
606
+ # thirty and refused the job: "expected exactly 30 personas, got 5". Pre-allocation is sealed
607
+ # against the full set by design (`_begin_payload` sends every key and the platform 409s on a
608
+ # subset), so the suite is what gets registered and the sample is only what gets called. The
609
+ # rows that are not called stay unstarted, which is a truthful state rather than a broken job.
610
+ chosen_evals, agent_prompt, modality = _chosen_evals_and_prompt(bundle_dir)
611
+ run_name = _derive_run_name(job, bundle_dir)
612
+ agent_name = _derive_agent_name(job, bundle_dir)
613
+ registered = await register_with_platform(
614
+ scenarios_client,
615
+ scenarios,
616
+ run_name=run_name,
617
+ agent_name=agent_name,
618
+ chosen_evals=chosen_evals,
619
+ agent_prompt=agent_prompt,
620
+ modality=modality,
621
+ )
622
+ return sampled_for_calling(registered)
623
+
624
+
625
+ def _chosen_evals_and_prompt(bundle_dir: Path) -> tuple[list[str], str, str]:
626
+ """The contract's chosen evals, agent prompt and modality; read as plain JSON so it cannot fail a run."""
627
+ try:
628
+ body = json.loads((bundle_dir / "contract.json").read_text(encoding="utf-8"))
629
+ except Exception: # noqa: BLE001 - a run never fails over what it tells the platform about itself
630
+ return [], "", ""
631
+ if not isinstance(body, dict):
632
+ return [], "", ""
633
+ chosen = body.get("chosen_evals")
634
+ names = (
635
+ [str(one).strip() for one in chosen if str(one).strip()]
636
+ if isinstance(chosen, list)
637
+ else []
638
+ )
639
+ # Provisioning defaults to text, which would bind a voice run's evals to the transcript.
640
+ modality = str(body.get("modality") or "").strip().lower()
641
+ return (
642
+ names,
643
+ str(body.get("system_prompt_excerpt") or "").strip(),
644
+ modality if modality in ("voice", "text") else "",
645
+ )
646
+
647
+
648
+ def _preallocation_error(code: str, message: str) -> Exception:
649
+ """Builds a `hosted_entrypoint.ScenarioPreallocationError` for a guard failure below --
650
+ imported lazily (not at module level) because `hosted_entrypoint.py` imports THIS module at
651
+ its own top level (`BundleScenarioSource`/`ScenarioDocumentInvalid`/`bundle_has_scenarios`), so
652
+ a top-level import back would be a circular import. Reusing that exact exception class (rather
653
+ than inventing a new one) is what lets these guard failures land on `run_job`'s ALREADY-WIRED
654
+ `except (ScenarioSourceNotWired, ScenarioPreallocationError)` clause with no changes there.
655
+ """
656
+ from .hosted_entrypoint import ScenarioPreallocationError
657
+
658
+ return ScenarioPreallocationError(
659
+ ob.ChannelError(
660
+ ob.ChannelOutcome.PERMANENT_ITEM, FailureDomain.PLATFORM_SYNC, code, message
661
+ )
662
+ )
663
+
664
+
665
+ def _read_bundle_contract(bundle_dir: Path) -> dict[str, Any]:
666
+ """Best-effort read of the authored contract from the bundle directory."""
667
+ path = bundle_dir / "contract.json"
668
+ if not path.is_file():
669
+ return {}
670
+ try:
671
+ body = json.loads(path.read_text(encoding="utf-8"))
672
+ return body if isinstance(body, dict) else {}
673
+ except (OSError, ValueError):
674
+ return {}
675
+
676
+
677
+ def _derive_run_name(job: Any, bundle_dir: Path) -> str:
678
+ """Human-readable simulation run name from the contract or job source.
679
+
680
+ Prefer the authored contract's ``agent`` field (e.g.
681
+ ``uber_voice_agent`` -> ``Uber Voice Agent``); fall back to the last
682
+ path segment of ``source.repository`` (e.g.
683
+ ``future-agi/ride-voice-agent`` -> ``ride-voice-agent``).
684
+ """
685
+ contract = _read_bundle_contract(bundle_dir)
686
+ agent = str(contract.get("agent") or "").strip()
687
+ if agent:
688
+ return agent.replace("_", " ").replace("-", " ").title()[:200]
689
+
690
+ repo = getattr(getattr(job, "source", None), "repository", None) or ""
691
+ if "/" in repo:
692
+ return repo.rsplit("/", 1)[-1][:200]
693
+ if repo:
694
+ return repo[:200]
695
+ return "simulation"
696
+
697
+
698
+ def _derive_agent_name(job: Any, bundle_dir: Path) -> str:
699
+ """Agent name for the provision payload, from contract or source repo."""
700
+ contract = _read_bundle_contract(bundle_dir)
701
+ agent = str(contract.get("agent") or "").strip()
702
+ if agent:
703
+ return agent.replace("_", " ").replace("-", " ").title()[:200]
704
+
705
+ repo = getattr(getattr(job, "source", None), "repository", None) or ""
706
+ if "/" in repo:
707
+ return repo.rsplit("/", 1)[-1][:200]
708
+ if repo:
709
+ return repo[:200]
710
+ return "alk-agent"
711
+
712
+
713
+ def _provision_payload(
714
+ run_name: str,
715
+ scenarios: Sequence[_CompiledScenario],
716
+ chosen_evals: Sequence[str] = (),
717
+ agent_prompt: str = "",
718
+ modality: str = "",
719
+ *,
720
+ agent_name: str = "",
721
+ ) -> dict[str, Any]:
722
+ """`HarnessScenarioProvisionSerializer`/`HarnessProvisionPersonaSerializer`
723
+ (futureagi/simulate/serializers/hosted_harness.py): `operation`, `name` and `personas` are
724
+ required; each persona's `name`, `role`, `situation`, `outcome` and `persona` are optional and
725
+ are now supplied from the document (`_presented`). They were omitted while this module read
726
+ `scenario.json` for scheduler-facing fields only, and the cost was a platform that could show a
727
+ call but not who was on it or what they came for.
728
+ """
729
+ payload: dict[str, Any] = {
730
+ "operation": "provision",
731
+ "name": run_name,
732
+ # Everything the serializer accepts, where the document had it: name, role, situation,
733
+ # outcome and the persona itself. `scenario_key` is set last because it is the one field the
734
+ # platform matches on and it must be this scenario's, whatever the persona record says.
735
+ "personas": [
736
+ {**scenario.presented, "scenario_key": scenario.scenario_key}
737
+ for scenario in scenarios
738
+ ],
739
+ }
740
+ if agent_name:
741
+ payload["agent_name"] = agent_name
742
+ # Each omitted rather than sent empty, so an older platform is unaffected. Modality matters
743
+ # because provisioning defaults to text, which binds every eval to the transcript.
744
+ if chosen_evals:
745
+ payload["chosen_evals"] = list(chosen_evals)
746
+ if agent_prompt:
747
+ payload["agent_prompt"] = agent_prompt
748
+ if modality:
749
+ payload["modality"] = modality
750
+ return payload
751
+
752
+
753
+ def _begin_payload(
754
+ run_test_id: str, scenarios: Sequence[_CompiledScenario]
755
+ ) -> dict[str, Any]:
756
+ """`HarnessScenarioBeginSerializer` (futureagi/simulate/serializers/hosted_harness.py:193-198):
757
+ `scenario_keys` is `allow_empty=False` and REQUIRED, and `begin_scenarios`
758
+ (services/hosted_harness.py:323-329) 409s (`scenario_key_mismatch`) on anything but an EXACT
759
+ match against the full sealed set -- there is no "subset to run" semantics on the real
760
+ platform (that contract text describes an optional partial-subset `scenario_ids`; the
761
+ live route does not implement that -- CONTRACT NOTES). The full set is sent every time.
762
+ """
763
+ return {
764
+ "operation": "begin",
765
+ "run_test_id": run_test_id,
766
+ "scenario_keys": [scenario.scenario_key for scenario in scenarios],
767
+ }
768
+
769
+
770
+ def _scenario_ids_by_key(
771
+ submitted: Sequence[_CompiledScenario], raw_scenarios: Any
772
+ ) -> dict[str, str]:
773
+ """Matches the platform's KEYED provision response
774
+ (`{"scenarios": [{"scenario_key", "scenario_id"}, ...]}`,
775
+ futureagi/simulate/serializers/hosted_harness.py:251-260 +
776
+ services/hosted_harness.py:487-501's `_provision_response`) back onto `submitted` BY
777
+ `scenario_key` -- a dict lookup, never a positional zip. A positional zip (matching that contract's
778
+ documented `scenario_ids` array shape, not what the platform actually returns) would silently
779
+ mismatch scenario_id -> scenario the instant the response order differs from `submitted`'s
780
+ order, which nothing on the wire guarantees. Every check below raises rather than returning a
781
+ partial mapping -- "never partial assignment" per the brief: the caller only gets a mapping
782
+ once it is proven complete (every submitted key present, exactly once) and exact (no
783
+ unrecognized key).
784
+ """
785
+ if not isinstance(raw_scenarios, list):
786
+ raise _preallocation_error(
787
+ "scenarios_provision_response_invalid", "response 'scenarios' is not a list"
788
+ )
789
+ by_key: dict[str, str] = {}
790
+ for entry in raw_scenarios:
791
+ if not isinstance(entry, dict):
792
+ raise _preallocation_error(
793
+ "scenarios_provision_response_invalid",
794
+ "a 'scenarios' entry is not an object",
795
+ )
796
+ key = entry.get("scenario_key")
797
+ scenario_id = entry.get("scenario_id")
798
+ if not isinstance(key, str) or not key:
799
+ raise _preallocation_error(
800
+ "scenarios_provision_response_invalid",
801
+ "a 'scenarios' entry has no non-empty scenario_key",
802
+ )
803
+ if key in by_key:
804
+ raise _preallocation_error(
805
+ "scenario_registration_duplicate_key",
806
+ f"scenario_key {key!r} appears more than once in the provision response",
807
+ )
808
+ if not isinstance(scenario_id, str) or not scenario_id:
809
+ raise _preallocation_error(
810
+ "scenarios_provision_response_invalid",
811
+ f"scenario_key {key!r} has no non-empty scenario_id",
812
+ )
813
+ by_key[key] = scenario_id
814
+
815
+ submitted_keys = [scenario.scenario_key for scenario in submitted]
816
+ unknown = sorted(set(by_key) - set(submitted_keys))
817
+ if unknown:
818
+ raise _preallocation_error(
819
+ "scenario_registration_unknown_key",
820
+ f"provision response named scenario_key(s) never submitted: {unknown}",
821
+ )
822
+ missing = sorted(set(submitted_keys) - set(by_key))
823
+ if missing:
824
+ raise _preallocation_error(
825
+ "scenario_registration_missing",
826
+ f"provision response is missing scenario_key(s): {missing}",
827
+ )
828
+ return by_key
829
+
830
+
831
+ async def register_with_platform(
832
+ scenarios_client: "ScenariosClient",
833
+ scenarios: Sequence[_CompiledScenario],
834
+ *,
835
+ run_name: str,
836
+ agent_name: str = "",
837
+ chosen_evals: Sequence[str] = (),
838
+ agent_prompt: str = "",
839
+ modality: str = "",
840
+ ) -> Sequence[_CompiledScenario]:
841
+ """The scenario pre-allocation SEAM, now wired against the platform's real route (a single
842
+ `POST .../scenarios/`, discriminated by a body-level `operation` field -- see
843
+ `ScenariosClient`'s own docstring for the file:line evidence). `.provision()`/`.begin()` are
844
+ blocking network calls (same `ScenariosClient` the rest of `hosted_entrypoint.py` already
845
+ drives off the event loop via `asyncio.to_thread` -- matched here rather than diverging).
846
+
847
+ Sequence: provision (get platform-assigned ids, keyed by `scenario_key`) -> match ids back
848
+ onto `scenarios` with hard guards (`_scenario_ids_by_key`, raises before ANY assignment on any
849
+ mismatch) -> begin (seals execution against the FULL scenario_keys set; a begin failure means
850
+ NO scenario in this batch is returned with an id -- the whole call raises, same as a provision
851
+ failure) -> only then build and return the new scenario list with `scenario_id` filled in.
852
+ """
853
+ provision_result = await asyncio.to_thread(
854
+ scenarios_client.provision,
855
+ _provision_payload(
856
+ run_name,
857
+ scenarios,
858
+ chosen_evals,
859
+ agent_prompt,
860
+ modality,
861
+ agent_name=agent_name,
862
+ ),
863
+ )
864
+ run_test_id = provision_result.get("run_test_id")
865
+ if not isinstance(run_test_id, str) or not run_test_id:
866
+ raise _preallocation_error(
867
+ "scenarios_provision_response_invalid",
868
+ "provision response has no run_test_id",
869
+ )
870
+ id_by_key = _scenario_ids_by_key(scenarios, provision_result.get("scenarios"))
871
+
872
+ await asyncio.to_thread(
873
+ scenarios_client.begin, _begin_payload(run_test_id, scenarios)
874
+ )
875
+
876
+ return tuple(
877
+ replace(scenario, scenario_id=id_by_key[scenario.scenario_key])
878
+ for scenario in scenarios
879
+ )