agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,304 @@
1
+ """One ALK harness executor used by the local CLI and hosted sandbox workers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import asyncio
7
+ from datetime import datetime, timezone
8
+ import json
9
+ import os
10
+ from pathlib import Path
11
+ import re
12
+ import subprocess
13
+ from typing import Callable, Protocol
14
+
15
+ from .job import (
16
+ FailureDomain,
17
+ HarnessFailure,
18
+ HarnessJob,
19
+ HarnessJobStatus,
20
+ HarnessStage,
21
+ SourceKind,
22
+ SourceVisibility,
23
+ )
24
+
25
+
26
+ class SourceAcquirer(Protocol):
27
+ """Materialize job source inside an executor-owned ephemeral workspace."""
28
+
29
+ async def acquire(self, job: HarnessJob, workspace: Path) -> Path: ...
30
+
31
+
32
+ class SourceAcquisitionError(RuntimeError):
33
+ pass
34
+
35
+
36
+ class GitHubSourceAcquirer:
37
+ """Clone one platform-authorized repository without exposing its token in argv or logs."""
38
+
39
+ _REPOSITORY = re.compile(r"^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$")
40
+
41
+ def __init__(self, installation_token: Callable[[str], str]) -> None:
42
+ self._installation_token = installation_token
43
+
44
+ async def acquire(self, job: HarnessJob, workspace: Path) -> Path:
45
+ source = job.source
46
+ repository = str(source.repository or "")
47
+ installation_id = str(source.installation_id or "")
48
+ if not self._REPOSITORY.fullmatch(repository):
49
+ raise SourceAcquisitionError("github_repository_invalid")
50
+ is_public = source.visibility is SourceVisibility.PUBLIC
51
+ if not is_public and not installation_id:
52
+ raise SourceAcquisitionError("github_installation_missing")
53
+ token = "" if is_public else self._installation_token(installation_id)
54
+ if not is_public and not token:
55
+ raise SourceAcquisitionError("github_installation_token_missing")
56
+ if source.ref and (
57
+ ".." in source.ref or not re.fullmatch(r"[A-Za-z0-9._/-]+", source.ref)
58
+ ):
59
+ raise SourceAcquisitionError("github_ref_invalid")
60
+ workspace = workspace.expanduser().resolve()
61
+ workspace.mkdir(parents=True, exist_ok=True)
62
+ destination = workspace / "repository"
63
+ if destination.exists():
64
+ raise SourceAcquisitionError(f"source_destination_not_empty: {destination}")
65
+
66
+ command = ["git", "clone", "--depth", "1"]
67
+ if source.ref:
68
+ command.extend(["--branch", source.ref])
69
+ command.extend([f"https://github.com/{repository}.git", str(destination)])
70
+ # Git reads the authorization header from its child environment. It never appears in
71
+ # the process command, exception, persisted job, or event stream.
72
+ environment = {**os.environ, "GIT_TERMINAL_PROMPT": "0"}
73
+ if token:
74
+ environment.update(
75
+ {
76
+ "GIT_CONFIG_COUNT": "1",
77
+ "GIT_CONFIG_KEY_0": "http.extraHeader",
78
+ "GIT_CONFIG_VALUE_0": f"Authorization: Bearer {token}",
79
+ }
80
+ )
81
+
82
+ def clone() -> None:
83
+ try:
84
+ completed = subprocess.run(
85
+ command,
86
+ env=environment,
87
+ cwd=workspace,
88
+ capture_output=True,
89
+ text=True,
90
+ timeout=300,
91
+ check=False,
92
+ )
93
+ except (OSError, subprocess.TimeoutExpired) as exc:
94
+ raise SourceAcquisitionError(
95
+ f"github_clone_unavailable: {type(exc).__name__}"
96
+ ) from exc
97
+ if completed.returncode:
98
+ detail = (
99
+ completed.stderr or completed.stdout or "clone failed"
100
+ ).strip()
101
+ # Git errors should not contain an env-only header, but redact defensively.
102
+ detail = detail.replace(token, "[REDACTED]")[:1000]
103
+ raise SourceAcquisitionError(f"github_clone_failed: {detail}")
104
+
105
+ if source.commit_sha:
106
+ verified = subprocess.run(
107
+ ["git", "rev-parse", "HEAD"],
108
+ cwd=destination,
109
+ capture_output=True,
110
+ text=True,
111
+ timeout=30,
112
+ check=False,
113
+ env={**os.environ, "GIT_TERMINAL_PROMPT": "0"},
114
+ )
115
+ actual = verified.stdout.strip().lower()
116
+ if verified.returncode or actual != source.commit_sha.lower():
117
+ raise SourceAcquisitionError("github_commit_mismatch")
118
+
119
+ await asyncio.to_thread(clone)
120
+ if not destination.is_dir():
121
+ raise SourceAcquisitionError("github_clone_missing_checkout")
122
+ return destination
123
+
124
+
125
+ class HarnessExecutor:
126
+ """Execute the autonomous pipeline without platform or scheduler dependencies.
127
+
128
+ A hosted worker first resolves a GitHub installation/archive/image through its
129
+ ``SourceAcquirer``. A local invocation already has a path. From that point onward both call
130
+ exactly the same ``auto`` pipeline and produce the same job, bundle, event, scenario, trace,
131
+ and result artifacts.
132
+ """
133
+
134
+ async def run(
135
+ self,
136
+ job: HarnessJob,
137
+ *,
138
+ source: Path,
139
+ output: Path,
140
+ model: str | None = None,
141
+ run_model: str | None = None,
142
+ adjustments_path: Path | None = None,
143
+ ) -> HarnessJobStatus:
144
+ from .cli import _auto
145
+
146
+ output = output.expanduser().resolve()
147
+ source = source.expanduser().resolve()
148
+ # A branch name is acquisition input, not immutable provenance. Resolve the checkout to
149
+ # its exact commit before job.json and the environment plan are written. The source
150
+ # content digest remains authoritative for uploads and non-Git sources.
151
+ if job.source.kind is SourceKind.GITHUB and not job.source.commit_sha:
152
+ completed = await asyncio.to_thread(
153
+ subprocess.run,
154
+ ["git", "rev-parse", "HEAD"],
155
+ cwd=source,
156
+ capture_output=True,
157
+ text=True,
158
+ timeout=30,
159
+ check=False,
160
+ env={**os.environ, "GIT_TERMINAL_PROMPT": "0"},
161
+ )
162
+ commit_sha = completed.stdout.strip().lower()
163
+ if completed.returncode or not re.fullmatch(r"[0-9a-f]{40}", commit_sha):
164
+ return HarnessJobStatus(
165
+ job_id=job.job_id,
166
+ run_id=job.run_id,
167
+ stage=HarnessStage.FAILED,
168
+ updated_at=datetime.now(timezone.utc),
169
+ detail="could not resolve the acquired GitHub revision",
170
+ failure=HarnessFailure(
171
+ domain=FailureDomain.INFRASTRUCTURE,
172
+ stage=HarnessStage.ACQUIRING_SOURCE,
173
+ code="github_commit_resolution_failed",
174
+ message="The runner could not resolve the acquired GitHub revision",
175
+ ),
176
+ total_scenarios=job.scenario_count,
177
+ )
178
+ job = job.model_copy(
179
+ update={
180
+ "source": job.source.model_copy(update={"commit_sha": commit_sha})
181
+ }
182
+ )
183
+ # Source acquisition has already materialized GitHub/archive/local inputs as a local
184
+ # checkout. The understanding registry describes how to inspect that materialized
185
+ # content (``repo``), while the immutable job retains its original source provenance.
186
+ # Passing transport kinds such as ``archive`` into source resolution makes a valid
187
+ # uploaded repository fail before its code is inspected.
188
+ understanding_kind = str(job.metadata.get("source_kind") or "repo")
189
+ if understanding_kind not in {"repo", "spec"}:
190
+ understanding_kind = "repo"
191
+ args = argparse.Namespace(
192
+ path=str(source),
193
+ name=str(job.metadata.get("agent_name") or source.name),
194
+ kind=understanding_kind,
195
+ out=str(output),
196
+ count=job.scenario_count,
197
+ model=model,
198
+ run_model=run_model,
199
+ job=job,
200
+ adjustments_path=str(adjustments_path) if adjustments_path else None,
201
+ )
202
+ status = await _auto(args)
203
+ failure = _failure_from_events(output) if status not in (0, 2) else None
204
+ return HarnessJobStatus(
205
+ job_id=job.job_id,
206
+ run_id=job.run_id,
207
+ stage=HarnessStage.COMPLETED if status in (0, 2) else HarnessStage.FAILED,
208
+ updated_at=datetime.now(timezone.utc),
209
+ detail=(
210
+ "agent checks failed"
211
+ if status == 2
212
+ else None
213
+ if status == 0
214
+ else f"exit {status}"
215
+ ),
216
+ failure=failure,
217
+ completed_scenarios=_scenario_count(output) if status in (0, 2) else 0,
218
+ total_scenarios=_scenario_count(output) or job.scenario_count,
219
+ )
220
+
221
+ async def acquire_and_run(
222
+ self,
223
+ job: HarnessJob,
224
+ *,
225
+ acquirer: SourceAcquirer,
226
+ workspace: Path,
227
+ output: Path,
228
+ model: str | None = None,
229
+ run_model: str | None = None,
230
+ ) -> HarnessJobStatus:
231
+ source = await acquirer.acquire(job, workspace)
232
+ return await self.run(
233
+ job,
234
+ source=source,
235
+ output=output,
236
+ model=model,
237
+ run_model=run_model,
238
+ )
239
+
240
+
241
+ def run_sync(job: HarnessJob, *, source: Path, output: Path) -> HarnessJobStatus:
242
+ """Small synchronous adapter for job consumers that do not own an event loop."""
243
+ return asyncio.run(HarnessExecutor().run(job, source=source, output=output))
244
+
245
+
246
+ def _scenario_count(output: Path) -> int:
247
+ try:
248
+ value = json.loads((output / "scenarios.json").read_text(encoding="utf-8"))
249
+ except (OSError, ValueError):
250
+ return 0
251
+ return len(value) if isinstance(value, list) else 0
252
+
253
+
254
+ def _failure_from_events(output: Path) -> HarnessFailure:
255
+ failed: dict = {}
256
+ path = output / "harness-events.jsonl"
257
+ if path.is_file():
258
+ for raw in path.read_text(encoding="utf-8", errors="replace").splitlines():
259
+ try:
260
+ event = json.loads(raw)
261
+ except ValueError:
262
+ continue
263
+ if event.get("type") == "harness.stage.failed":
264
+ failed = event.get("payload") or {}
265
+ label = str(failed.get("stage") or "running")
266
+ stage, domain = {
267
+ "understand": (HarnessStage.UNDERSTANDING_AGENT, FailureDomain.AGENT),
268
+ "environment": (
269
+ HarnessStage.VALIDATING_ENVIRONMENT,
270
+ FailureDomain.ENVIRONMENT,
271
+ ),
272
+ "scenarios": (
273
+ HarnessStage.VALIDATING_SCENARIOS,
274
+ FailureDomain.SIMULATOR,
275
+ ),
276
+ "calls": (HarnessStage.RUNNING, FailureDomain.CONNECTIVITY),
277
+ "cleaning_up": (
278
+ HarnessStage.CLEANING_UP,
279
+ FailureDomain.INFRASTRUCTURE,
280
+ ),
281
+ "uploading_artifacts": (
282
+ HarnessStage.UPLOADING_ARTIFACTS,
283
+ FailureDomain.ARTIFACT,
284
+ ),
285
+ }.get(label, (HarnessStage.RUNNING, FailureDomain.INFRASTRUCTURE))
286
+ return HarnessFailure(
287
+ domain=domain,
288
+ stage=stage,
289
+ code=str(failed.get("code") or f"{label}_failed"),
290
+ message=str(failed.get("detail") or f"Harness stage {label} failed"),
291
+ # A job replay can repeat real calls. Only a failing stage that explicitly proves
292
+ # it is safe may opt in to retry; deterministic agent/grading failures never do.
293
+ retryable=bool(failed.get("retryable", False)),
294
+ details={"status": failed.get("status", 1)},
295
+ )
296
+
297
+
298
+ __all__ = [
299
+ "GitHubSourceAcquirer",
300
+ "HarnessExecutor",
301
+ "SourceAcquirer",
302
+ "SourceAcquisitionError",
303
+ "run_sync",
304
+ ]
@@ -0,0 +1,234 @@
1
+ """A scenario as a folder of files, and running the code inside it.
2
+
3
+ A scenario used to be a row in one big JSON file, and its setup was a list of rows to insert.
4
+ That was enough while every world was a database. It stopped being enough the moment a world
5
+ could hold a service as well as a table: "the weather service starts returning errors" is not
6
+ expressible as rows, and neither is "the file is missing" or "the queue is backed up".
7
+
8
+ So a scenario owns a folder, and the parts that are logic are files:
9
+
10
+ scenarios/<name>/
11
+ scenario.json what it is: instruction, solution, which sub-goals
12
+ setup.py def setup(world) — the changes this scenario makes
13
+ ready.py def ready(world) — is the world ready for this scenario
14
+ checks/<goal>.py def check(world, calls) — one per deterministic sub-goal
15
+
16
+ The files are the artifact, not a rendering of one. Each is executable on its own, so a check
17
+ can be run by hand against what a run left behind and answer exactly what it answers inside the
18
+ harness. That is the whole point of them being files: something you can open, read and run is
19
+ something you can argue with.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import json
25
+ from dataclasses import dataclass
26
+ from pathlib import Path
27
+ from typing import Any
28
+
29
+ from .catalogue import Catalogue
30
+ from .scenario import Scenario
31
+ from .world.runtime import GeneratedWorld
32
+
33
+ SCENARIOS = "scenarios"
34
+ INDEX = "scenarios.json"
35
+
36
+ # Appended to every check file the harness writes. The model writes only ``check(world, calls)``;
37
+ # this is what makes that same file runnable by a person, so nobody has to keep two versions of
38
+ # one truth in step.
39
+ _RUNNABLE = """
40
+
41
+ if __name__ == "__main__":
42
+ # Run this check by hand against what a run left behind. The first argument is anything
43
+ # inside the saved world's folder, because not every world has a database to name:
44
+ # python <this file> <world folder>/manifest.json [calls.json]
45
+ import json as _json
46
+ import sys as _sys
47
+ from pathlib import Path as _Path
48
+
49
+ from fi.alk.harness.world.runtime import Call as _Call
50
+ from fi.alk.harness.world.snapshot import restore as _restore
51
+
52
+ _world = _restore(_Path(_sys.argv[1]).parent) if len(_sys.argv) > 1 else None
53
+ _calls = []
54
+ if len(_sys.argv) > 2:
55
+ _calls = [_Call(**_one) for _one in _json.loads(_Path(_sys.argv[2]).read_text())]
56
+ _said = check(_world, _calls)
57
+ _held = _said is None or _said is True or (isinstance(_said, str) and not _said.strip())
58
+ print("held" if _held else f"FAILED: {_said}")
59
+ raise SystemExit(0 if _held else 1)
60
+ """
61
+
62
+
63
+ @dataclass
64
+ class Outcome:
65
+ """What one piece of a scenario's own code did."""
66
+
67
+ ok: bool
68
+ said: str = ""
69
+ broken: bool = False
70
+
71
+
72
+ def _run(source: str, name: str, entry: str, *args: Any) -> Outcome:
73
+ """Execute one function out of a scenario's own code.
74
+
75
+ A file that will not compile, or that raises, is **broken** rather than failing: it is our
76
+ mistake, and scoring it as though the world were wrong would send somebody looking in the
77
+ wrong place.
78
+ """
79
+ if not source.strip():
80
+ return Outcome(True)
81
+ namespace: dict[str, Any] = {}
82
+ try:
83
+ exec(compile(source, f"<{name}>", "exec"), namespace)
84
+ except Exception as failed:
85
+ return Outcome(False, f"{name} would not compile: {failed}", broken=True)
86
+
87
+ function = namespace.get(entry)
88
+ if not callable(function):
89
+ return Outcome(False, f"{name} defines no {entry}()", broken=True)
90
+ try:
91
+ said = function(*args)
92
+ except Exception as failed:
93
+ return Outcome(
94
+ False, f"{name} raised {type(failed).__name__}: {failed}", broken=True
95
+ )
96
+ # The convention is that a complaint is a sentence, and anything else means it held. An empty
97
+ # string is the case worth naming: it reads as "no complaint" to whoever wrote it, and taking
98
+ # it as a failure produces a rejection with no reason attached, which cannot be acted on and
99
+ # sends the author hunting for a problem that is not there.
100
+ if said is None or said is True or (isinstance(said, str) and not said.strip()):
101
+ return Outcome(True)
102
+ if said is False:
103
+ # Bare False from ready() names no precondition, so a scenario that hits it cannot be
104
+ # told apart from one whose ready.py is simply wrong — that is our mistake, not a
105
+ # generation-time precondition failure, so ready() alone reports it broken.
106
+ return Outcome(
107
+ False,
108
+ f"{name} returned False without saying what is wrong. Return the sentence instead, "
109
+ "or None if it holds.",
110
+ broken=(entry == "ready"),
111
+ )
112
+ if entry == "ready" and not isinstance(said, str):
113
+ # Same reasoning as bare False, widened: ready() has no way to turn a non-string value
114
+ # into a precondition sentence, so any of them is our mistake rather than the world's.
115
+ return Outcome(
116
+ False,
117
+ f"{name} returned {type(said).__name__} {repr(said)[:200]}. Return the sentence "
118
+ "naming what is missing, or None if it holds.",
119
+ broken=True,
120
+ )
121
+ return Outcome(False, str(said))
122
+
123
+
124
+ def apply_setup(scenario: Scenario, world: GeneratedWorld) -> Outcome:
125
+ """Make this scenario's changes to the world."""
126
+ return _run(scenario.setup_code, f"{scenario.name}/setup.py", "setup", world)
127
+
128
+
129
+ def check_ready(scenario: Scenario, world: GeneratedWorld) -> Outcome:
130
+ """Whether the world now holds what this scenario presumes."""
131
+ return _run(scenario.ready_code, f"{scenario.name}/ready.py", "ready", world)
132
+
133
+
134
+ def folder_for(destination: Path, name: str) -> Path:
135
+ return Path(destination) / SCENARIOS / name
136
+
137
+
138
+ def write_folder(scenario: Scenario, catalogue: Catalogue, destination: Path) -> Path:
139
+ """Write one scenario out as its own folder of files."""
140
+ root = folder_for(destination, scenario.name)
141
+ (root / "checks").mkdir(parents=True, exist_ok=True)
142
+
143
+ body = scenario.model_dump()
144
+ # The code lives in its own files; keeping a second copy in the JSON would let the two drift
145
+ # and leave nobody able to say which one ran.
146
+ body.pop("setup_code", None)
147
+ body.pop("ready_code", None)
148
+ (root / "scenario.json").write_text(
149
+ json.dumps(body, indent=2, ensure_ascii=False), encoding="utf-8"
150
+ )
151
+
152
+ (root / "setup.py").write_text(
153
+ scenario.setup_code
154
+ or 'def setup(world):\n """This scenario runs on the base world unchanged."""\n',
155
+ encoding="utf-8",
156
+ )
157
+ (root / "ready.py").write_text(
158
+ scenario.ready_code
159
+ or 'def ready(world):\n """Nothing beyond the base world is presumed."""\n',
160
+ encoding="utf-8",
161
+ )
162
+
163
+ for name in scenario.sub_goals:
164
+ sub_goal = catalogue.named(name)
165
+ if sub_goal is None or not sub_goal.deterministic():
166
+ continue
167
+ (root / "checks" / f"{name}.py").write_text(
168
+ sub_goal.check.rstrip() + "\n" + _RUNNABLE, encoding="utf-8"
169
+ )
170
+ return root
171
+
172
+
173
+ def read_folder(destination: Path, name: str) -> Scenario | None:
174
+ """One scenario, reassembled from its folder."""
175
+ root = folder_for(destination, name)
176
+ body = root / "scenario.json"
177
+ if not body.exists():
178
+ return None
179
+ payload = json.loads(body.read_text(encoding="utf-8"))
180
+ for field, filename in (("setup_code", "setup.py"), ("ready_code", "ready.py")):
181
+ path = root / filename
182
+ payload[field] = path.read_text(encoding="utf-8") if path.exists() else ""
183
+ return Scenario.model_validate(payload)
184
+
185
+
186
+ def write_index(scenarios: list[Scenario], destination: Path) -> Path:
187
+ """The whole suite at a glance, over the folders.
188
+
189
+ Regenerated from the folders rather than maintained alongside them, so it can never disagree
190
+ with what is actually on disk.
191
+ """
192
+ destination = Path(destination)
193
+ destination.mkdir(parents=True, exist_ok=True)
194
+ path = destination / INDEX
195
+ path.write_text(
196
+ json.dumps(
197
+ [
198
+ {
199
+ "name": one.name,
200
+ "use_case": one.use_case,
201
+ "tests": one.tests,
202
+ "instruction": one.instruction,
203
+ "sub_goals": one.sub_goals,
204
+ "steps": len(one.solution),
205
+ "folder": f"{SCENARIOS}/{one.name}",
206
+ }
207
+ for one in scenarios
208
+ ],
209
+ indent=2,
210
+ ensure_ascii=False,
211
+ ),
212
+ encoding="utf-8",
213
+ )
214
+ return path
215
+
216
+
217
+ def read_all(destination: Path) -> list[Scenario]:
218
+ """Every scenario on disk, read from the folders."""
219
+ root = Path(destination) / SCENARIOS
220
+ if not root.exists():
221
+ return []
222
+ found: list[Scenario] = []
223
+ for folder in sorted(root.iterdir()):
224
+ if not folder.is_dir():
225
+ continue
226
+ try:
227
+ scenario = read_folder(destination, folder.name)
228
+ except Exception:
229
+ # A folder we cannot read is skipped rather than crashing the stage: the rest of the
230
+ # suite is still usable, and the gap shows up as a missing scenario.
231
+ continue
232
+ if scenario is not None:
233
+ found.append(scenario)
234
+ return found