agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/alk/harness/cli.py ADDED
@@ -0,0 +1,1354 @@
1
+ """Run a stage from a terminal.
2
+
3
+ This is one renderer over the stage loop, not the product. It prints events as lines and reads
4
+ follow-ups from stdin; a browser front end subscribes to the same events and draws them as a
5
+ transcript beside the artifact. Keeping the terminal a renderer rather than the interface is what
6
+ makes the second one cheap.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import asyncio
13
+ import json
14
+ import os
15
+ import sys
16
+ import time
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ from . import spend
21
+ from . import observability
22
+ from .build import open_stage as build_stage
23
+ from .build import opening as build_opening
24
+ from .build import require_buildable
25
+ from .chat import open_conversation
26
+ from .config import (
27
+ artifact_dir,
28
+ chosen_model,
29
+ credentials_hint,
30
+ permission_gate,
31
+ )
32
+ from .run.targets import supported as target_kinds
33
+ from .scenarios import load as load_written
34
+ from .scenarios import open_stage as scenario_stage
35
+ from .scenarios import opening as scenario_opening
36
+ from .session import TEXT, Event
37
+ from .sessions import Session, new_id, save as save_session
38
+ from .sources import resolve, supported
39
+ from .understand import load, open_stage, opening
40
+ from .world.snapshot import saved as world_saved
41
+
42
+
43
+ def _source_root(destination: Path, explicit: str = "") -> str:
44
+ """Recover the source path for commands resumed from a session folder."""
45
+ if explicit.strip():
46
+ return str(Path(explicit).expanduser().resolve())
47
+ metadata = destination / "session.json"
48
+ if metadata.exists():
49
+ try:
50
+ import json
51
+
52
+ return str(
53
+ json.loads(metadata.read_text(encoding="utf-8")).get("source") or ""
54
+ )
55
+ except (OSError, ValueError):
56
+ pass
57
+ return ""
58
+
59
+
60
+ def _render(event: Event) -> None:
61
+ line = event.line()
62
+ if event.kind == TEXT:
63
+ print(line, end="", flush=True)
64
+ else:
65
+ print(f"\n{line}", flush=True)
66
+
67
+
68
+ async def _prompt(question: str) -> str:
69
+ return (await asyncio.to_thread(input, question)).strip()
70
+
71
+
72
+ async def _ask_operator(_tool_name: str, payload: dict[str, Any], _context: Any) -> Any:
73
+ """Render the model's clarifying questions and return the operator's answers."""
74
+ from claude_agent_sdk.types import PermissionResultAllow
75
+
76
+ answers: dict[str, Any] = {}
77
+ for question in payload.get("questions", []):
78
+ print(f"\n\n {question.get('header', '?')}: {question.get('question', '')}")
79
+ options = question.get("options", []) or []
80
+ for index, option in enumerate(options, start=1):
81
+ print(
82
+ f" {index}. {option.get('label')} - {option.get('description', '')}"
83
+ )
84
+ raw = await _prompt(" > ")
85
+ chosen = raw
86
+ if raw.isdigit() and 1 <= int(raw) <= len(options):
87
+ chosen = options[int(raw) - 1].get("label", raw)
88
+ answers[question.get("question", "")] = chosen
89
+ print()
90
+ return PermissionResultAllow(
91
+ updated_input={"questions": payload.get("questions", []), "answers": answers}
92
+ )
93
+
94
+
95
+ def _guidance(args: argparse.Namespace) -> str:
96
+ instructions = [
97
+ str(item).strip()
98
+ for item in (getattr(args, "guidance", None) or [])
99
+ if str(item).strip()
100
+ ]
101
+ if not instructions:
102
+ return ""
103
+ return (
104
+ "\n\n## User adjustments (required completion criteria)\n\n"
105
+ "The user supplied the requirements below during this run. The final saved "
106
+ "artifact MUST directly represent every bullet; acknowledging a bullet or merely "
107
+ "changing the requested count is not enough. For scenario-stage adjustments, at "
108
+ "least one saved scenario must clearly test each requested behavior. Preserve all "
109
+ "unaffected validated work. Do not call save_scenarios until these requirements are "
110
+ "visible in the saved suite:\n- " + "\n- ".join(instructions)
111
+ )
112
+
113
+
114
+ def _scenario_adjustment_requirement(adjustment_id: str, instruction: str) -> str:
115
+ """Make a live scenario correction mechanically auditable after generation."""
116
+ return (
117
+ f"Adjustment {adjustment_id}: {instruction}\n"
118
+ "At least one scenario that directly tests this requirement MUST include "
119
+ f'fixture.adjustment_ids containing "{adjustment_id}". This marker is required '
120
+ "evidence that the saved suite reflects the instruction."
121
+ )
122
+
123
+
124
+ def _missing_scenario_adjustments(
125
+ written: list[Any], adjustment_ids: list[str]
126
+ ) -> list[str]:
127
+ covered: set[str] = set()
128
+ for scenario in written:
129
+ fixture = getattr(scenario, "fixture", None)
130
+ if not isinstance(fixture, dict):
131
+ continue
132
+ markers = fixture.get("adjustment_ids") or []
133
+ if isinstance(markers, str):
134
+ markers = [markers]
135
+ if isinstance(markers, list):
136
+ covered.update(str(marker) for marker in markers)
137
+ return [
138
+ adjustment_id
139
+ for adjustment_id in adjustment_ids
140
+ if adjustment_id not in covered
141
+ ]
142
+
143
+
144
+ async def _understand(args: argparse.Namespace) -> int:
145
+ if args.kind == "provider":
146
+ source = resolve(
147
+ args.kind,
148
+ name=args.name,
149
+ profile=getattr(args, "provider_profile", None) or {},
150
+ scratch=args.path,
151
+ )
152
+ else:
153
+ source = resolve(args.kind, name=args.name, root=args.path)
154
+ job = getattr(args, "job", None)
155
+ offered = (
156
+ (getattr(job, "metadata", None) or {}).get("available_evals") if job else None
157
+ )
158
+ stage, destination = open_stage(
159
+ source,
160
+ out=Path(args.out) if args.out else None,
161
+ # Unattended, there is nobody to answer, so the model records what it could not
162
+ # resolve in open_questions rather than blocking on a prompt nobody will see.
163
+ ask=permission_gate(_ask_operator) if args.interactive else None,
164
+ available_evals=offered if isinstance(offered, list) else None,
165
+ )
166
+
167
+ print(f"agent: {source.name} ({source.kind})")
168
+ print(f"model: {chosen_model()}")
169
+ print(f"out: {destination}\n")
170
+
171
+ await _converse(
172
+ stage,
173
+ opening(source) + _guidance(args),
174
+ interactive=args.interactive,
175
+ until=lambda: load(destination) is not None,
176
+ nudge=(
177
+ "Nothing was saved: you finished without calling submit_contract. Call it now "
178
+ "with the contract you worked out."
179
+ ),
180
+ )
181
+
182
+ contract = load(destination)
183
+ if contract is None:
184
+ print("\nNo contract was submitted.", file=sys.stderr)
185
+ return 1
186
+ print(
187
+ f"\ncontract: {len(contract.tools)} tools, "
188
+ f"{len(contract.hard_constraints)} rules, "
189
+ f"{len(contract.real_use_cases)} use cases, "
190
+ f"{len(contract.open_questions)} open questions"
191
+ )
192
+ print(f"spent: ${stage.spent_usd:.4f}")
193
+ return 0
194
+
195
+
196
+ async def _converse(
197
+ stage,
198
+ opening_message: str,
199
+ *,
200
+ interactive: bool,
201
+ until=None,
202
+ nudge: str = "",
203
+ ) -> None:
204
+ """Say the opening, then keep the stage open for corrections.
205
+
206
+ The same shape for every stage. A world is usually right on the second look, and the point
207
+ of holding the session open is that correcting it is the next thing said rather than a
208
+ rebuild from nothing.
209
+
210
+ ``until``/``nudge`` guard the unattended case. The commonest way an unattended stage fails
211
+ is finishing all the work and never calling the tool that saves it — the whole contract
212
+ written out as prose, submitted to nobody. One mechanical reminder costs a turn; rerunning
213
+ the stage costs everything it just did.
214
+ """
215
+ async with stage:
216
+ await stage.say(opening_message, on_event=_render)
217
+ if not interactive and until is not None and nudge and not until():
218
+ await stage.say(nudge, on_event=_render)
219
+ while interactive:
220
+ try:
221
+ said = await _prompt("\nyou ")
222
+ except (EOFError, KeyboardInterrupt):
223
+ break
224
+ if not said or said in {"q", "quit", "exit"}:
225
+ break
226
+ await stage.say(said, on_event=_render)
227
+
228
+
229
+ async def _build(args: argparse.Namespace) -> int:
230
+ destination = Path(args.out) if args.out else artifact_dir(args.name)
231
+ contract = load(destination)
232
+ if contract is None:
233
+ print(f"No contract at {destination}. Run `understand` first.", file=sys.stderr)
234
+ return 1
235
+
236
+ print(f"agent: {contract.agent} ({len(contract.tools)} tools)")
237
+ print(f"model: {chosen_model()}")
238
+ print(f"out: {destination}\n")
239
+
240
+ source_root = _source_root(destination, args.path or "")
241
+ external_runtime = bool(getattr(args, "external_runtime", False))
242
+ try:
243
+ require_buildable(
244
+ contract,
245
+ source_root,
246
+ external_runtime=external_runtime,
247
+ )
248
+ except RuntimeError as failed:
249
+ print(str(failed), file=sys.stderr)
250
+ return 1
251
+
252
+ from .provision import ProvisionError, provision_if_present
253
+
254
+ environment = None
255
+ if not bool(getattr(args, "skip_source_provision", False)):
256
+ try:
257
+ environment = await asyncio.to_thread(
258
+ provision_if_present, source_root, destination, contract
259
+ )
260
+ except ProvisionError as failed:
261
+ print(f"Cannot create the source environment: {failed}", file=sys.stderr)
262
+ return 1
263
+ if environment is not None:
264
+ print(f"environment: {environment.project} ({', '.join(environment.services)})")
265
+ for name, value in sorted(environment.overrides.items()):
266
+ print(f"override: {name}={value}")
267
+
268
+ stage, _ = build_stage(
269
+ contract,
270
+ out=destination,
271
+ ask=permission_gate(_ask_operator) if args.interactive else None,
272
+ source_root=source_root,
273
+ deferred_runtime=bool(getattr(args, "skip_source_provision", False)),
274
+ external_runtime=external_runtime,
275
+ )
276
+ deferred_runtime = bool(getattr(args, "skip_source_provision", False))
277
+ await _converse(
278
+ stage,
279
+ build_opening(
280
+ contract,
281
+ provisioned=environment is not None,
282
+ deferred_runtime=deferred_runtime,
283
+ external_runtime=external_runtime,
284
+ )
285
+ + _guidance(args),
286
+ interactive=args.interactive,
287
+ until=lambda: world_saved(destination),
288
+ nudge=(
289
+ "Nothing was saved: you finished without calling save_world. Call check_world, "
290
+ "fix what it names, then save_world."
291
+ ),
292
+ )
293
+
294
+ if not world_saved(destination):
295
+ print("\nNo world was saved.", file=sys.stderr)
296
+ return 1
297
+ # Seal the exact environment now that its generated world exists. Local and hosted
298
+ # execution consume this same internal bundle; the source repository is never a special
299
+ # runtime path after this boundary.
300
+ from .bundle import BundleError, export_session_bundle
301
+
302
+ try:
303
+ bundle_path, bundle = await asyncio.to_thread(
304
+ export_session_bundle,
305
+ source_root,
306
+ destination,
307
+ name=f"{contract.agent}-environment",
308
+ )
309
+ except BundleError as failed:
310
+ print(f"Cannot seal the environment bundle: {failed}", file=sys.stderr)
311
+ return 1
312
+ print(f"\nworld: {destination}")
313
+ print(f"bundle: {bundle_path} ({bundle.digest})")
314
+ print(f"spent: ${stage.spent_usd:.4f}")
315
+ return 0
316
+
317
+
318
+ async def _environment(args: argparse.Namespace) -> int:
319
+ """Provision or tear down the runtime shipped by the agent repository."""
320
+ from .provision import (
321
+ ProvisionedEnvironment,
322
+ ProvisionError,
323
+ provision,
324
+ reset,
325
+ stop,
326
+ )
327
+
328
+ destination = Path(args.out)
329
+ try:
330
+ if args.action == "down":
331
+ if not stop(destination):
332
+ print(f"No environment recorded at {destination}.", file=sys.stderr)
333
+ return 1
334
+ print(f"environment stopped: {destination}")
335
+ return 0
336
+ if args.action == "status":
337
+ environment = ProvisionedEnvironment.load(destination)
338
+ if environment is None:
339
+ print(f"No environment recorded at {destination}.", file=sys.stderr)
340
+ return 1
341
+ elif args.action == "reset":
342
+ environment = reset(destination)
343
+ else:
344
+ source_path = args.path
345
+ bundle_value = str(getattr(args, "bundle", "") or "")
346
+ if bundle_value:
347
+ from .bundle import BundleError, load_bundle
348
+ from .environment_plan import (
349
+ ENVIRONMENT_PLAN_FILE,
350
+ EnvironmentPlanError,
351
+ load_environment_plan,
352
+ )
353
+
354
+ bundle_root = Path(bundle_value).expanduser().resolve()
355
+ try:
356
+ bundle = load_bundle(bundle_root)
357
+ # New bundles carry the canonical decision record. Older sealed bundles
358
+ # remain rerunnable, but still receive full content verification.
359
+ if (bundle_root / ENVIRONMENT_PLAN_FILE).is_file():
360
+ load_environment_plan(bundle_root, bundle=bundle)
361
+ except (BundleError, EnvironmentPlanError) as failed:
362
+ print(f"Environment bundle failed: {failed}", file=sys.stderr)
363
+ return 1
364
+ bundled_source = bundle_root / "services" / "source"
365
+ if not bundled_source.is_dir():
366
+ print(
367
+ f"Environment bundle has no source snapshot: {bundled_source}",
368
+ file=sys.stderr,
369
+ )
370
+ return 1
371
+ source_path = str(bundled_source)
372
+ # Resuming a saved environment must use the same repository/runtime decision as the
373
+ # autonomous and hosted paths. In particular, a submitted Compose runtime may name
374
+ # a repository-local env file that is intentionally replaced by job-scoped values.
375
+ # Without the saved contract this command can incorrectly fall back to a Dockerfile
376
+ # and report that a previously valid Compose environment cannot be started.
377
+ environment = provision(source_path, destination, load(destination))
378
+ except ProvisionError as failed:
379
+ print(f"Environment failed: {failed}", file=sys.stderr)
380
+ return 1
381
+
382
+ print(f"environment: {environment.project}")
383
+ print(f"services: {', '.join(environment.services)}")
384
+ print(f"ready in: {environment.provision_seconds:.3f}s")
385
+ for name, value in sorted(environment.overrides.items()):
386
+ print(f"set: {name}={value}")
387
+ return 0
388
+
389
+
390
+ async def _scenarios(args: argparse.Namespace) -> int:
391
+ destination = Path(args.out) if args.out else artifact_dir(args.name)
392
+ contract = load(destination)
393
+ if contract is None:
394
+ print(f"No contract at {destination}. Run `understand` first.", file=sys.stderr)
395
+ return 1
396
+ if not world_saved(destination):
397
+ print(f"No world at {destination}. Run `build` first.", file=sys.stderr)
398
+ return 1
399
+
400
+ # With a suite already written, the target is what is there. Somebody who comes back to
401
+ # change one scenario is not asking for a different number of them.
402
+ existing = len(load_written(destination))
403
+ wanted = args.count or existing or 10
404
+
405
+ print(
406
+ f"agent: {contract.agent} "
407
+ + (f"({existing} scenarios, loaded)" if existing else f"(writing {wanted})")
408
+ )
409
+ print(f"model: {chosen_model()}")
410
+ print(f"out: {destination}\n")
411
+
412
+ stage, _ = scenario_stage(
413
+ contract,
414
+ out=destination,
415
+ wanted=wanted,
416
+ ask=permission_gate(_ask_operator) if args.interactive else None,
417
+ )
418
+ await _converse(
419
+ stage,
420
+ scenario_opening(contract, wanted, existing) + _guidance(args),
421
+ interactive=args.interactive,
422
+ until=lambda: bool(load_written(destination)),
423
+ nudge=(
424
+ "Nothing was saved: you finished without calling save_scenarios. Submit anything "
425
+ "still unsubmitted, then call save_scenarios."
426
+ ),
427
+ )
428
+
429
+ written = load_written(destination)
430
+ if not written:
431
+ print("\nNo scenarios were saved.", file=sys.stderr)
432
+ return 1
433
+ print(f"\nscenarios: {len(written)} in {destination / 'scenarios.json'}")
434
+ print(f"spent: ${stage.spent_usd:.4f}")
435
+ return 0
436
+
437
+
438
+ async def _live(args: argparse.Namespace) -> int:
439
+ """The run stage as a conversation: it decides what to run and reads what came back."""
440
+ from .run.stage import load as load_results
441
+ from .run.stage import open_stage as run_stage
442
+ from .run.stage import opening as run_opening
443
+
444
+ destination = Path(args.out) if args.out else artifact_dir(args.name)
445
+ contract = load(destination)
446
+ written = load_written(destination)
447
+ if contract is None or not written:
448
+ print(
449
+ f"Need a contract and scenarios at {destination}. Run `understand`, `build` and "
450
+ "`scenarios` first.",
451
+ file=sys.stderr,
452
+ )
453
+ return 1
454
+ source_root = _source_root(destination)
455
+ if source_root:
456
+ _load_connection_env(Path(source_root))
457
+
458
+ print(f"agent: {contract.agent} ({len(written)} scenarios)")
459
+ print(f"model: {chosen_model()}")
460
+ print(f"out: {destination}\n")
461
+
462
+ stage, _ = run_stage(
463
+ contract,
464
+ out=destination,
465
+ ask=permission_gate(_ask_operator) if args.interactive else None,
466
+ )
467
+ await _converse(
468
+ stage, run_opening(contract, destination), interactive=args.interactive
469
+ )
470
+
471
+ results = load_results(destination)
472
+ passed = sum(1 for record in results if record["passed"])
473
+ print(f"\nruns: {passed} of {len(results)} passed, in {destination / 'runs.json'}")
474
+ print(f"spent: ${stage.spent_usd:.4f}")
475
+ return 0
476
+
477
+
478
+ async def _run(args: argparse.Namespace) -> int:
479
+ from .run import run_suite
480
+ from .run.grade import summarise
481
+ from .world.snapshot import require_source_implementation
482
+
483
+ destination = Path(args.out) if args.out else artifact_dir(args.name)
484
+ contract = load(destination)
485
+ written = load_written(destination)
486
+ if contract is None or not written:
487
+ print(
488
+ f"Need a contract and scenarios at {destination}. Run `understand`, `build` "
489
+ "and `scenarios` first.",
490
+ file=sys.stderr,
491
+ )
492
+ return 1
493
+ try:
494
+ require_source_implementation(destination)
495
+ except (FileNotFoundError, RuntimeError) as failed:
496
+ print(str(failed), file=sys.stderr)
497
+ return 1
498
+
499
+ chosen = [s for s in written if s.name in args.only] if args.only else written
500
+ if not chosen:
501
+ print(f"No scenario matching {args.only}.", file=sys.stderr)
502
+ return 1
503
+
504
+ print(f"agent: {contract.agent} ({len(chosen)} scenarios, target {args.target})")
505
+ print(f"model: {chosen_model()}")
506
+ print(f"out: {destination}\n")
507
+
508
+ def overheard(exchange: Any) -> None:
509
+ if args.quiet:
510
+ return
511
+ print(f" {exchange.speaker:8} {exchange.text}", flush=True)
512
+
513
+ def show(result: Any) -> None:
514
+ # Just the verdict as it lands. The detail is in the summary at the end, and printing
515
+ # it in both places means every failure is read twice.
516
+ print(result.line(), flush=True)
517
+
518
+ results = await run_suite(
519
+ chosen,
520
+ contract,
521
+ destination,
522
+ target=args.target,
523
+ model=args.model,
524
+ on_result=show,
525
+ on_exchange=overheard,
526
+ )
527
+ print("\n" + summarise(results))
528
+ print(f"\nspent: ${sum(result.spent_usd for result in results):.4f}")
529
+ return 0 if all(result.passed for result in results) else 2
530
+
531
+
532
+ async def _simulate(args: argparse.Namespace) -> int:
533
+ """Run the suite through the modality/runtime inferred from the saved contract."""
534
+ from . import platform
535
+ from .run.simulation import simulate
536
+ from .world.snapshot import require_source_implementation
537
+
538
+ destination = Path(args.out) if args.out else artifact_dir(args.name)
539
+ contract = load(destination)
540
+ written = load_written(destination)
541
+ if contract is None or not written:
542
+ print(
543
+ f"Need a contract and scenarios at {destination}.",
544
+ file=sys.stderr,
545
+ )
546
+ return 1
547
+ try:
548
+ require_source_implementation(destination)
549
+ except (FileNotFoundError, RuntimeError) as failed:
550
+ print(str(failed), file=sys.stderr)
551
+ return 1
552
+ chosen = [s for s in written if s.name in args.only] if args.only else written
553
+ if not chosen:
554
+ print(f"No scenario matching {args.only}.", file=sys.stderr)
555
+ return 1
556
+
557
+ # A resumed simulation is a first-class execution path, not merely an internal stage of
558
+ # ``auto``. Load the submitted repository's connection settings here as well so restarting a
559
+ # completed build does not require the operator to rediscover and export its LiveKit/model
560
+ # credentials by hand. Existing worker/host values continue to win in _load_connection_env.
561
+ source_root = _source_root(destination)
562
+ if source_root:
563
+ _load_connection_env(Path(source_root))
564
+
565
+ reported = None
566
+ call_ids: dict[str, str] = {}
567
+ blocked = platform.configured()
568
+ if not blocked:
569
+ try:
570
+ reported, allocated = platform.begin(
571
+ chosen,
572
+ name=platform.display_run_name(args.name),
573
+ run_test_id=platform.remembered(destination),
574
+ modality=contract.modality or "text",
575
+ )
576
+ call_ids = {
577
+ scenario.name: call_id
578
+ for scenario, call_id in zip(chosen, allocated, strict=False)
579
+ }
580
+ # Persist the destination as soon as the platform execution exists. The list view
581
+ # can now show an in-progress run, and a process restart reuses the same RunTest.
582
+ platform.remember(destination, reported)
583
+ print(f"platform run: {reported.url}", flush=True)
584
+ except platform.PlatformError as failed:
585
+ print(f"platform reporting could not start: {failed}", file=sys.stderr)
586
+ else:
587
+ print(f"not reported to the platform: {blocked}", flush=True)
588
+
589
+ def show(result: Any) -> None:
590
+ print(result.line(), flush=True)
591
+ if reported is None:
592
+ return
593
+ call_id = call_ids.get(result.scenario)
594
+ if not call_id:
595
+ reported.problems.append(
596
+ f"the platform allocated no call for {result.scenario}"
597
+ )
598
+ return
599
+ platform.send_result(reported, call_id, result)
600
+
601
+ def show_started(scenario: Any) -> None:
602
+ if reported is None:
603
+ return
604
+ platform.mark_ongoing(reported, call_ids.get(scenario.name, ""))
605
+
606
+ summary = await simulate(
607
+ chosen,
608
+ contract,
609
+ destination,
610
+ destination=destination,
611
+ model=args.model,
612
+ on_case_start=show_started,
613
+ on_case_done=show,
614
+ )
615
+ print(
616
+ f"\n{summary['passed']}/{summary['scenarios']} scenarios passed "
617
+ f"in {summary['seconds']}s"
618
+ )
619
+ print(f"run: {destination / 'runs' / summary['run_id']}")
620
+ if reported is not None:
621
+ for problem in reported.problems:
622
+ print(f"platform reporting problem: {problem}", file=sys.stderr)
623
+ print(f"reported to the platform: {reported.url}")
624
+ # Exit 1 means the environment/call lane could not execute at least one scenario. Exit 2
625
+ # means every scenario ran and the submitted agent failed one or more checks. Hosted
626
+ # execution retries/classifies the former and preserves the latter as valid RL evidence.
627
+ if summary.get("unrunnable"):
628
+ return 1
629
+ return 0 if summary["passed"] == summary["scenarios"] else 2
630
+
631
+
632
+ def _load_connection_env(source: Path) -> list[str]:
633
+ """Fill missing connection variables from the submitted repository.
634
+
635
+ A dotenv file is data, not a shell program. Sourcing a customer's file executes command
636
+ substitutions and also breaks on perfectly valid unquoted values containing spaces. Values
637
+ already supplied by the workspace win: in particular, a worker's container-only credential
638
+ path must never replace the host's model-provider credential path.
639
+ """
640
+ loaded: list[str] = []
641
+ for candidate in (source / ".env.local", source / ".env"):
642
+ if not candidate.is_file():
643
+ continue
644
+ for raw in candidate.read_text(encoding="utf-8").splitlines():
645
+ line = raw.strip()
646
+ if not line or line.startswith("#") or "=" not in line:
647
+ continue
648
+ name, value = line.split("=", 1)
649
+ name = name.removeprefix("export ").strip()
650
+ if not name.replace("_", "").isalnum() or not name[0].isalpha():
651
+ continue
652
+ value = value.strip()
653
+ if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'":
654
+ value = value[1:-1]
655
+ if name not in os.environ:
656
+ os.environ[name] = value
657
+ loaded.append(name)
658
+ return loaded
659
+
660
+
661
+ def _new_adjustments(
662
+ path: Path | None, cursor: int
663
+ ) -> tuple[list[dict[str, Any]], int]:
664
+ if path is None or not path.is_file():
665
+ return [], cursor
666
+ records: list[dict[str, Any]] = []
667
+ lines = path.read_text(encoding="utf-8", errors="replace").splitlines()
668
+ for line in lines[cursor:]:
669
+ try:
670
+ value = json.loads(line)
671
+ except ValueError:
672
+ continue
673
+ if isinstance(value, dict):
674
+ records.append(value)
675
+ return records, len(lines)
676
+
677
+
678
+ def _write_adjustment_status(
679
+ inbox: Path | None,
680
+ adjustment_id: str,
681
+ status: str,
682
+ *,
683
+ applied_stage: str,
684
+ ) -> None:
685
+ if inbox is None:
686
+ return
687
+ status_path = inbox.with_name("adjustment-status.jsonl")
688
+ record = {
689
+ "adjustment_id": adjustment_id,
690
+ "status": status,
691
+ "applied_stage": applied_stage,
692
+ "updated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
693
+ }
694
+ with status_path.open("a", encoding="utf-8") as stream:
695
+ stream.write(json.dumps(record, separators=(",", ":")) + "\n")
696
+ stream.flush()
697
+
698
+
699
+ async def _auto(args: argparse.Namespace) -> int:
700
+ """Take one source connection through every stage without operator messages.
701
+
702
+ This is the product acceptance path. The individual stage commands remain useful while
703
+ developing the harness, but a submitted agent must not depend on somebody knowing their
704
+ order, nudging a model, or repairing an artifact between stages.
705
+ """
706
+ source = Path(args.path).expanduser().resolve()
707
+ loaded_connection = _load_connection_env(source)
708
+ name = (args.name or source.name).strip()
709
+ destination = (
710
+ Path(args.out).expanduser().resolve()
711
+ if args.out
712
+ else artifact_dir(new_id(name))
713
+ )
714
+ destination.mkdir(parents=True, exist_ok=False)
715
+ from fi.simulate.runtime.events import CanonicalEvent
716
+
717
+ from .events import BufferedEventSink, EventOutbox
718
+ from .job import (
719
+ AgentConnection,
720
+ ExecutionMode,
721
+ HarnessJob,
722
+ RepositorySource,
723
+ SourceKind,
724
+ )
725
+
726
+ job = getattr(args, "job", None) or HarnessJob(
727
+ job_id=destination.name,
728
+ run_id=destination.name,
729
+ execution=ExecutionMode.LOCAL,
730
+ source=RepositorySource(
731
+ kind=SourceKind.LOCAL_REPOSITORY, local_path=str(source)
732
+ ),
733
+ agent=AgentConnection(connector="auto"),
734
+ scenario_count=args.count,
735
+ metadata={"agent_name": name, "source_kind": args.kind},
736
+ )
737
+ (destination / "job.json").write_text(
738
+ job.model_dump_json(indent=2) + "\n", encoding="utf-8"
739
+ )
740
+ # Beside the authoring output, so the platform reads the running total while the sandbox lives.
741
+ spend.journal_to(destination / "cost.json")
742
+ events = BufferedEventSink(EventOutbox(destination.parent, destination.name))
743
+ event_sequence = 0
744
+
745
+ def emit(event_type: str, stage: str, **payload: Any) -> None:
746
+ nonlocal event_sequence
747
+ observability.stage_event(event_type, stage, payload)
748
+ events.write(
749
+ CanonicalEvent.create(
750
+ run_id=job.run_id,
751
+ test_case_id="harness",
752
+ event_type=event_type,
753
+ source="fi.alk.harness",
754
+ sequence=event_sequence,
755
+ payload={"stage": stage, **payload},
756
+ )
757
+ )
758
+ event_sequence += 1
759
+
760
+ now = time.time()
761
+ save_session(
762
+ Session(
763
+ id=destination.name,
764
+ path=destination,
765
+ agent=name,
766
+ source=str(source),
767
+ kind=args.kind,
768
+ created=now,
769
+ updated=now,
770
+ stage="understand",
771
+ title=name,
772
+ )
773
+ )
774
+
775
+ print("automatic acceptance run")
776
+ print(f"agent: {name}")
777
+ print(f"source: {source}")
778
+ print(f"out: {destination}")
779
+ print("operator input: disabled\n")
780
+ if loaded_connection:
781
+ print(
782
+ "connection: loaded missing variables from the submitted repository "
783
+ f"({len(loaded_connection)} names; values hidden)\n"
784
+ )
785
+
786
+ authoring_only = bool(getattr(args, "authoring_only", False))
787
+ stages = [
788
+ (
789
+ "understand",
790
+ _understand,
791
+ argparse.Namespace(
792
+ name=name,
793
+ path=str(source),
794
+ kind=args.kind,
795
+ out=str(destination),
796
+ interactive=False,
797
+ model=args.model,
798
+ guidance=[],
799
+ job=job,
800
+ provider_profile=getattr(args, "provider_profile", None),
801
+ ),
802
+ ),
803
+ (
804
+ "environment",
805
+ _build,
806
+ argparse.Namespace(
807
+ name=name,
808
+ path=str(source),
809
+ out=str(destination),
810
+ interactive=False,
811
+ guidance=[],
812
+ # Hosted V2 authoring resolves and provisions the submitted runtime inside the
813
+ # Daytona guest. The authoring worker still creates the same logical world and
814
+ # scenarios, but must not start customer Compose/Docker resources on the control
815
+ # plane worker merely to describe them.
816
+ skip_source_provision=authoring_only,
817
+ # A source-free provider connection deliberately keeps the provider's deployed
818
+ # HTTP tools in place. Their real execution is observed during calls; no local
819
+ # source entrypoint exists or is required.
820
+ external_runtime=authoring_only and args.kind == "provider",
821
+ ),
822
+ ),
823
+ (
824
+ "scenarios",
825
+ _scenarios,
826
+ argparse.Namespace(
827
+ name=name,
828
+ out=str(destination),
829
+ count=args.count,
830
+ interactive=False,
831
+ guidance=[],
832
+ ),
833
+ ),
834
+ ]
835
+ if not authoring_only:
836
+ stages.append(
837
+ (
838
+ "calls",
839
+ _simulate,
840
+ argparse.Namespace(
841
+ name=name,
842
+ out=str(destination),
843
+ only=None,
844
+ model=args.run_model,
845
+ ),
846
+ )
847
+ )
848
+ from .provision import ProvisionError, stop
849
+
850
+ cleanup_failed: ProvisionError | None = None
851
+ try:
852
+ stage_index = 0
853
+ adjustment_cursor = 0
854
+ applying: dict[str, list[str]] = {label: [] for label, *_ in stages}
855
+ scenario_adjustment_guidance: dict[str, str] = {}
856
+ adjustments_path = (
857
+ Path(args.adjustments_path)
858
+ if getattr(args, "adjustments_path", None)
859
+ else None
860
+ )
861
+ while stage_index < len(stages):
862
+ label, operation, stage_args = stages[stage_index]
863
+ print(f"\n=== {label} ===", flush=True)
864
+ emit("harness.stage.started", label)
865
+ status = await operation(stage_args)
866
+ # Hosted authoring is unattended, and a successful scenario-stage
867
+ # process is not sufficient evidence that it honoured the requested
868
+ # cardinality. Models can checkpoint a valid partial suite (for
869
+ # example 2/3); Bundle V2 must remain strict, so repair the producer
870
+ # output here while the authoring context is still available.
871
+ if label == "scenarios" and authoring_only and not status:
872
+ repair_attempt = 0
873
+ wanted = int(stage_args.count)
874
+ written_count = len(load_written(destination))
875
+ while written_count != wanted and repair_attempt < 2:
876
+ repair_attempt += 1
877
+ missing = wanted - written_count
878
+ # Count alone is the wrong instruction: asked only for a number, the
879
+ # stage pads with happy paths that satisfy cardinality and measure nothing.
880
+ stage_args.guidance = [
881
+ (
882
+ f"The hosted run requires exactly {wanted} scenarios, but "
883
+ f"only {written_count} are currently saved. "
884
+ + (
885
+ f"Add exactly {missing} distinct validated scenario(s) and "
886
+ "call save_scenarios. Preserve all existing scenarios. Each "
887
+ "one must meet the same bar as the rest of the suite: a "
888
+ "different branch of the agent's behaviour from every "
889
+ "scenario already saved, several steps deep, and failing "
890
+ "when the agent does the wrong thing. Do not pad with "
891
+ "variations of a scenario that already exists, and do not "
892
+ "add a happy path that an existing scenario already covers."
893
+ if missing > 0
894
+ else f"Remove exactly {-missing} excess scenario(s), preserve "
895
+ "the strongest coverage, and call save_scenarios. Drop the "
896
+ "ones that duplicate a branch another scenario already "
897
+ "exercises, not the ones that are hardest to pass."
898
+ )
899
+ )
900
+ ]
901
+ emit(
902
+ "harness.stage.repairing",
903
+ label,
904
+ attempt=repair_attempt,
905
+ expected_scenarios=wanted,
906
+ written_scenarios=written_count,
907
+ )
908
+ status = await operation(stage_args)
909
+ if status:
910
+ break
911
+ written_count = len(load_written(destination))
912
+ if not status and written_count != wanted:
913
+ emit(
914
+ "harness.stage.failed",
915
+ label,
916
+ status=1,
917
+ code="scenario_count_mismatch",
918
+ expected_scenarios=wanted,
919
+ written_scenarios=written_count,
920
+ )
921
+ print(
922
+ "\nautomatic run stopped: scenario generation saved "
923
+ f"{written_count}/{wanted} requested scenarios after "
924
+ f"{repair_attempt} repair attempts",
925
+ file=sys.stderr,
926
+ )
927
+ status = 1
928
+ if (
929
+ label == "scenarios"
930
+ and authoring_only
931
+ and not status
932
+ and applying[label]
933
+ ):
934
+ missing = _missing_scenario_adjustments(
935
+ load_written(destination), applying[label]
936
+ )
937
+ repair_attempt = 0
938
+ while missing and repair_attempt < 2:
939
+ repair_attempt += 1
940
+ stage_args.guidance = [
941
+ scenario_adjustment_guidance[adjustment_id]
942
+ + "\nNo saved scenario currently carries this marker. Correct the "
943
+ "suite, preserve unaffected scenarios, and call save_scenarios again."
944
+ for adjustment_id in missing
945
+ ]
946
+ emit(
947
+ "harness.stage.repairing",
948
+ label,
949
+ attempt=repair_attempt,
950
+ missing_adjustment_ids=missing,
951
+ )
952
+ status = await operation(stage_args)
953
+ if status:
954
+ break
955
+ missing = _missing_scenario_adjustments(
956
+ load_written(destination), applying[label]
957
+ )
958
+ if not status and missing:
959
+ emit(
960
+ "harness.stage.failed",
961
+ label,
962
+ status=1,
963
+ code="scenario_adjustment_not_reflected",
964
+ missing_adjustment_ids=missing,
965
+ )
966
+ print(
967
+ "\nautomatic run stopped: saved scenarios did not reflect "
968
+ f"adjustments {', '.join(missing)}",
969
+ file=sys.stderr,
970
+ )
971
+ status = 1
972
+ # A completed call suite returns 2 when the submitted agent fails one or more checks.
973
+ # That is a valid RL result. Earlier stages returning non-zero are harness failures.
974
+ if status and label != "calls":
975
+ emit("harness.stage.failed", label, status=status)
976
+ print(f"\nautomatic run stopped: {label} failed", file=sys.stderr)
977
+ return status
978
+ if label == "calls" and status not in (0, 2):
979
+ emit("harness.stage.failed", label, status=status)
980
+ return status
981
+ emit("harness.stage.completed", label, status=status)
982
+ for adjustment_id in applying[label]:
983
+ _write_adjustment_status(
984
+ adjustments_path,
985
+ adjustment_id,
986
+ "applied",
987
+ applied_stage=label,
988
+ )
989
+ emit(
990
+ "harness.adjustment.applied",
991
+ label,
992
+ adjustment_id=adjustment_id,
993
+ )
994
+ applying[label] = []
995
+ if hasattr(stage_args, "guidance"):
996
+ stage_args.guidance = []
997
+
998
+ incoming, adjustment_cursor = _new_adjustments(
999
+ adjustments_path, adjustment_cursor
1000
+ )
1001
+ rewind_to: int | None = None
1002
+ for adjustment in incoming:
1003
+ target = str(adjustment.get("target_stage") or label)
1004
+ target_index = next(
1005
+ (
1006
+ index
1007
+ for index, (stage_name, *_rest) in enumerate(stages)
1008
+ if stage_name == target
1009
+ ),
1010
+ min(stage_index, 2),
1011
+ )
1012
+ target_args = stages[target_index][2]
1013
+ instruction = str(adjustment.get("instruction") or "")
1014
+ adjustment_id = str(adjustment.get("adjustment_id") or "")
1015
+ if target == "scenarios" and adjustment_id:
1016
+ instruction = _scenario_adjustment_requirement(
1017
+ adjustment_id, instruction
1018
+ )
1019
+ scenario_adjustment_guidance[adjustment_id] = instruction
1020
+ target_args.guidance = [
1021
+ *getattr(target_args, "guidance", []),
1022
+ instruction,
1023
+ ]
1024
+ delta = adjustment.get("scenario_delta")
1025
+ if target == "scenarios" and isinstance(delta, int) and delta > 0:
1026
+ existing_count = len(load_written(destination))
1027
+ target_args.count = max(target_args.count, existing_count) + delta
1028
+ if adjustment_id:
1029
+ applying[target].append(adjustment_id)
1030
+ _write_adjustment_status(
1031
+ adjustments_path,
1032
+ adjustment_id,
1033
+ "applying",
1034
+ applied_stage=target,
1035
+ )
1036
+ emit(
1037
+ "harness.adjustment.applying",
1038
+ target,
1039
+ adjustment_id=adjustment_id,
1040
+ )
1041
+ if target_index <= stage_index:
1042
+ rewind_to = (
1043
+ target_index
1044
+ if rewind_to is None
1045
+ else min(rewind_to, target_index)
1046
+ )
1047
+ if rewind_to is not None:
1048
+ emit(
1049
+ "harness.pipeline.rewound",
1050
+ stages[rewind_to][0],
1051
+ from_stage=label,
1052
+ )
1053
+ stage_index = rewind_to
1054
+ else:
1055
+ stage_index += 1
1056
+ finally:
1057
+ # The source environment exists only for this run. This boundary covers normal stage
1058
+ # failures and exceptions raised by world/scenario construction. Cleanup happens before
1059
+ # sealing so environment.json's terminal state is part of the immutable manifest.
1060
+ emit("harness.stage.started", "cleaning_up")
1061
+ try:
1062
+ await asyncio.to_thread(stop, destination)
1063
+ except ProvisionError as exc:
1064
+ cleanup_failed = exc
1065
+ emit(
1066
+ "harness.stage.failed",
1067
+ "cleaning_up",
1068
+ status=1,
1069
+ code="environment_cleanup_failed",
1070
+ detail=str(exc),
1071
+ )
1072
+ print(f"\nautomatic run cleanup failed: {exc}", file=sys.stderr)
1073
+ else:
1074
+ emit("harness.stage.completed", "cleaning_up", status=0)
1075
+
1076
+ if cleanup_failed is not None:
1077
+ return 1
1078
+
1079
+ if authoring_only:
1080
+ emit("harness.authoring.completed", "scenarios")
1081
+ print(f"\nautomatic authoring complete: {destination}")
1082
+ return 0
1083
+
1084
+ from .artifacts import ArtifactIntegrityError, seal_artifacts
1085
+
1086
+ emit("harness.stage.started", "uploading_artifacts")
1087
+ try:
1088
+ manifest = seal_artifacts(
1089
+ destination,
1090
+ run_id=job.run_id,
1091
+ max_bytes=job.artifacts.max_artifact_bytes,
1092
+ expected_scenarios=len(load_written(destination)) or job.scenario_count,
1093
+ )
1094
+ except ArtifactIntegrityError as exc:
1095
+ emit(
1096
+ "harness.stage.failed",
1097
+ "uploading_artifacts",
1098
+ status=1,
1099
+ code="artifact_integrity_failed",
1100
+ detail=str(exc),
1101
+ )
1102
+ print(
1103
+ f"\nautomatic run stopped: artifact integrity failed: {exc}",
1104
+ file=sys.stderr,
1105
+ )
1106
+ return 1
1107
+ emit(
1108
+ "harness.stage.completed",
1109
+ "uploading_artifacts",
1110
+ status=0,
1111
+ artifact_manifest_digest=manifest.digest,
1112
+ artifact_count=len(manifest.files),
1113
+ artifact_bytes=manifest.total_bytes,
1114
+ )
1115
+ emit("harness.run.completed", "completed")
1116
+ print(f"\nautomatic run complete: {destination}")
1117
+ return 0
1118
+
1119
+
1120
+ async def _chat(args: argparse.Namespace) -> int:
1121
+ """One conversation for the whole thing: point at an agent and keep talking."""
1122
+ conversation = open_conversation(
1123
+ name=args.name or "",
1124
+ path=args.path or "",
1125
+ kind=args.kind,
1126
+ out=Path(args.out) if args.out else None,
1127
+ ask=permission_gate(_ask_operator),
1128
+ )
1129
+ print(f"model: {chosen_model()}")
1130
+ print(credentials_hint())
1131
+ print("\nSay what you want. Enter on its own moves to the next stage; 'q' ends.\n")
1132
+
1133
+ await conversation.start(on_event=_render)
1134
+ while True:
1135
+ try:
1136
+ said = await _prompt(f"\nyou ({conversation.stage_name}) ")
1137
+ except (EOFError, KeyboardInterrupt):
1138
+ break
1139
+ if said in {"q", "quit", "exit"}:
1140
+ break
1141
+ if not said:
1142
+ entered = await conversation.advance(on_event=_render)
1143
+ if entered is None:
1144
+ print(
1145
+ "\n [nothing to move on to yet; this stage has not produced its artifact]"
1146
+ )
1147
+ continue
1148
+ await conversation.say(said, on_event=_render)
1149
+ await conversation.close()
1150
+ print(f"\nspent: ${conversation.spent_usd:.4f}")
1151
+ return 0
1152
+
1153
+
1154
+ def build_parser() -> argparse.ArgumentParser:
1155
+ parser = argparse.ArgumentParser(prog="agent-harness", description=__doc__)
1156
+ # Talking to it is the way in, so that is what happens when you just start it.
1157
+ sub = parser.add_subparsers(dest="stage", required=False)
1158
+
1159
+ understand = sub.add_parser(
1160
+ "understand", help="read an agent and produce its contract"
1161
+ )
1162
+ understand.add_argument("--name", required=True, help="what to call this agent")
1163
+ understand.add_argument("--path", required=True, help="where the agent is")
1164
+ understand.add_argument(
1165
+ "--kind", default="repo", choices=supported(), help="how the agent is supplied"
1166
+ )
1167
+ understand.add_argument("--out", default=None, help="artifact directory")
1168
+ understand.add_argument(
1169
+ "--once",
1170
+ dest="interactive",
1171
+ action="store_false",
1172
+ help="run unattended instead of staying open for corrections",
1173
+ )
1174
+ understand.add_argument("--model", default=None, help=argparse.SUPPRESS)
1175
+ understand.set_defaults(run=_understand, interactive=True)
1176
+
1177
+ world = sub.add_parser("build", help="build the world from an agent's contract")
1178
+ world.add_argument("--name", required=True, help="which agent")
1179
+ world.add_argument("--out", default=None, help="artifact directory")
1180
+ world.add_argument(
1181
+ "--path",
1182
+ default=None,
1183
+ help="agent source path (normally recovered from the session automatically)",
1184
+ )
1185
+ world.add_argument(
1186
+ "--once",
1187
+ dest="interactive",
1188
+ action="store_false",
1189
+ help="run unattended instead of staying open for corrections",
1190
+ )
1191
+ world.set_defaults(run=_build, interactive=True)
1192
+
1193
+ environment = sub.add_parser(
1194
+ "environment", help="start, inspect, or stop the runtime shipped by an agent"
1195
+ )
1196
+ environment.add_argument("action", choices=("up", "status", "reset", "down"))
1197
+ environment.add_argument(
1198
+ "--path", default="", help="agent repository (required for up)"
1199
+ )
1200
+ environment.add_argument(
1201
+ "--bundle",
1202
+ default="",
1203
+ help="sealed environment bundle to verify and restart (preferred for reruns)",
1204
+ )
1205
+ environment.add_argument("--out", required=True, help="session artifact directory")
1206
+ environment.set_defaults(run=_environment)
1207
+
1208
+ scenarios = sub.add_parser(
1209
+ "scenarios", help="write the scenarios to test the agent with"
1210
+ )
1211
+ scenarios.add_argument("--name", required=True, help="which agent")
1212
+ scenarios.add_argument("--out", default=None, help="artifact directory")
1213
+ scenarios.add_argument(
1214
+ "--count",
1215
+ type=int,
1216
+ default=None,
1217
+ help="how many scenarios to write (defaults to however many already exist)",
1218
+ )
1219
+ scenarios.add_argument(
1220
+ "--once",
1221
+ dest="interactive",
1222
+ action="store_false",
1223
+ help="run unattended instead of staying open for corrections",
1224
+ )
1225
+ scenarios.add_argument(
1226
+ "--guidance",
1227
+ action="append",
1228
+ default=None,
1229
+ metavar="TEXT",
1230
+ help=(
1231
+ "a natural-language requirement the saved suite must satisfy; repeatable. "
1232
+ "Non-interactive extend/adjust runs pass these to steer the added scenarios "
1233
+ "while preserving existing validated work"
1234
+ ),
1235
+ )
1236
+ scenarios.set_defaults(run=_scenarios, interactive=True)
1237
+
1238
+ live = sub.add_parser(
1239
+ "live", help="run the scenarios against the real agent, as a conversation"
1240
+ )
1241
+ live.add_argument("--name", required=True, help="which agent")
1242
+ live.add_argument("--out", default=None, help="artifact directory")
1243
+ live.add_argument(
1244
+ "--once",
1245
+ dest="interactive",
1246
+ action="store_false",
1247
+ help="run unattended instead of staying open",
1248
+ )
1249
+ live.set_defaults(run=_live, interactive=True)
1250
+
1251
+ runs = sub.add_parser("run", help="run the scenarios and grade what happened")
1252
+ runs.add_argument("--name", required=True, help="which agent")
1253
+ runs.add_argument("--out", default=None, help="artifact directory")
1254
+ runs.add_argument(
1255
+ "--target",
1256
+ default="local",
1257
+ choices=target_kinds(),
1258
+ help="where the agent under test runs",
1259
+ )
1260
+ runs.add_argument(
1261
+ "--only", nargs="*", default=None, help="run only these scenarios, by name"
1262
+ )
1263
+ runs.add_argument("--model", default=None, help="model for the run")
1264
+ runs.add_argument(
1265
+ "--quiet",
1266
+ action="store_true",
1267
+ help="only the verdicts, without the conversations as they happen",
1268
+ )
1269
+ runs.set_defaults(run=_run)
1270
+
1271
+ simulation = sub.add_parser(
1272
+ "simulate",
1273
+ help="run through the agent modality and shipped runtime inferred from its contract",
1274
+ )
1275
+ simulation.add_argument("--name", required=True, help="which agent")
1276
+ simulation.add_argument("--out", default=None, help="artifact directory")
1277
+ simulation.add_argument(
1278
+ "--only", nargs="*", default=None, help="run only these scenarios, by name"
1279
+ )
1280
+ simulation.add_argument("--model", default=None, help=argparse.SUPPRESS)
1281
+ simulation.set_defaults(run=_simulate)
1282
+
1283
+ auto = sub.add_parser(
1284
+ "auto",
1285
+ help="from one agent source connection, build everything and run unattended",
1286
+ )
1287
+ auto.add_argument("--path", required=True, help="agent repository")
1288
+ auto.add_argument(
1289
+ "--name",
1290
+ default=None,
1291
+ help="agent name (defaults to the repository folder name)",
1292
+ )
1293
+ auto.add_argument(
1294
+ "--kind", default="repo", choices=supported(), help="how the agent is supplied"
1295
+ )
1296
+ auto.add_argument(
1297
+ "--out",
1298
+ default=None,
1299
+ help="fresh artifact directory (defaults to a unique session directory)",
1300
+ )
1301
+ auto.add_argument("--count", type=int, default=10, help="number of scenarios")
1302
+ auto.add_argument("--model", default=None, help=argparse.SUPPRESS)
1303
+ auto.add_argument("--run-model", default=None, help=argparse.SUPPRESS)
1304
+ auto.set_defaults(run=_auto)
1305
+
1306
+ author = sub.add_parser(
1307
+ "author",
1308
+ help=(
1309
+ "understand an agent, create its logical environment, and write scenarios "
1310
+ "without executing them"
1311
+ ),
1312
+ )
1313
+ author.add_argument("--path", required=True, help="agent repository")
1314
+ author.add_argument(
1315
+ "--name",
1316
+ default=None,
1317
+ help="agent name (defaults to the repository folder name)",
1318
+ )
1319
+ author.add_argument(
1320
+ "--kind", default="repo", choices=supported(), help="how the agent is supplied"
1321
+ )
1322
+ author.add_argument(
1323
+ "--out", required=True, help="fresh authoring artifact directory"
1324
+ )
1325
+ author.add_argument("--count", type=int, default=10, help="number of scenarios")
1326
+ author.add_argument("--model", default=chosen_model(), help=argparse.SUPPRESS)
1327
+ author.set_defaults(run=_auto, authoring_only=True, run_model=None)
1328
+
1329
+ chat = sub.add_parser(
1330
+ "chat",
1331
+ help="one conversation: understand, build the world, write the scenarios",
1332
+ )
1333
+ # Nothing is required. Which agent, where it lives and how many scenarios are all things
1334
+ # you say; naming one here is a shortcut back into work already in progress.
1335
+ chat.add_argument("--name", default=None, help=argparse.SUPPRESS)
1336
+ chat.add_argument("--path", default=None, help=argparse.SUPPRESS)
1337
+ chat.add_argument(
1338
+ "--kind", default="repo", choices=supported(), help=argparse.SUPPRESS
1339
+ )
1340
+ chat.add_argument("--out", default=None, help=argparse.SUPPRESS)
1341
+ chat.set_defaults(run=_chat)
1342
+ return parser
1343
+
1344
+
1345
+ def main(argv: list[str] | None = None) -> int:
1346
+ parser = build_parser()
1347
+ args = parser.parse_args(argv)
1348
+ if getattr(args, "run", None) is None:
1349
+ args = parser.parse_args([*(argv or []), "chat"])
1350
+ return asyncio.run(args.run(args))
1351
+
1352
+
1353
+ if __name__ == "__main__":
1354
+ raise SystemExit(main())