agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,2011 @@
1
+ """Local implementation of the hosted ALK sandbox boundary.
2
+
3
+ The platform talks to this HTTP contract in development. Production can replace the
4
+ implementation with a Kubernetes or microVM service without changing platform job APIs.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import asyncio
10
+ from datetime import datetime, timezone
11
+ import hashlib
12
+ import json
13
+ import os
14
+ from pathlib import Path
15
+ from pathlib import PurePosixPath
16
+ import re
17
+ import shutil
18
+ import signal
19
+ import sys
20
+ import time
21
+ from typing import Any
22
+ import uuid
23
+
24
+ from fastapi import (
25
+ Depends,
26
+ FastAPI,
27
+ File,
28
+ Form,
29
+ Header,
30
+ HTTPException,
31
+ UploadFile,
32
+ status,
33
+ )
34
+ from pydantic import BaseModel, Field, SecretStr, model_validator
35
+
36
+ from fi.simulate.runtime.spec import SecretRef
37
+
38
+ from .credentials import (
39
+ CredentialManifest,
40
+ CredentialRequirement,
41
+ RequirementKind,
42
+ RequirementStatus,
43
+ discover_credentials,
44
+ )
45
+ from .executor import GitHubSourceAcquirer, SourceAcquisitionError
46
+ from .github import parse_github_location
47
+ from .job import (
48
+ AgentConnection,
49
+ ExecutionMode,
50
+ HarnessArtifactPolicy,
51
+ HarnessJob,
52
+ HarnessJobStatus,
53
+ HarnessStage,
54
+ RepositorySource,
55
+ SourceKind,
56
+ SourceVisibility,
57
+ )
58
+ from .packaging import PackagingManifest, inspect_packaging
59
+ from .provision import ProvisionError, source_fingerprint, stop
60
+ from .secrets import resolve_worker_secrets, worker_environment
61
+
62
+
63
+ _CONTROLLER_TOKEN = uuid.uuid4().hex
64
+
65
+
66
+ class LocalSandboxRequest(BaseModel):
67
+ source_path: str | None = None
68
+ source_id: str | None = None
69
+ github_repository: str | None = None
70
+ github_ref: str | None = None
71
+ github_commit_sha: str | None = None
72
+ github_visibility: SourceVisibility = SourceVisibility.PUBLIC
73
+ github_installation_id: str | None = None
74
+ scenario_count: int = Field(default=10, ge=1, le=200)
75
+ seed: int | None = None
76
+ agent_name: str | None = None
77
+ connector: str = "auto"
78
+ connector_config: dict[str, Any] = Field(default_factory=dict)
79
+ secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
80
+ environment_values: dict[str, SecretStr] = Field(default_factory=dict)
81
+ platform_run_id: str | None = None
82
+ metadata: dict[str, Any] = Field(default_factory=dict)
83
+
84
+ @model_validator(mode="after")
85
+ def _one_source(self) -> "LocalSandboxRequest":
86
+ if (
87
+ sum(
88
+ bool(value)
89
+ for value in (self.source_path, self.source_id, self.github_repository)
90
+ )
91
+ != 1
92
+ ):
93
+ raise ValueError("exactly_one_source_required")
94
+ _validate_environment_values(self.environment_values, self.secret_refs)
95
+ return self
96
+
97
+
98
+ class SandboxPreflightRequest(BaseModel):
99
+ source_path: str | None = None
100
+ source_id: str | None = None
101
+ github_repository: str | None = None
102
+ github_visibility: SourceVisibility = SourceVisibility.PUBLIC
103
+ github_installation_id: str | None = None
104
+ connector_config: dict[str, Any] = Field(default_factory=dict)
105
+ secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
106
+ environment_values: dict[str, SecretStr] = Field(default_factory=dict)
107
+
108
+ @model_validator(mode="after")
109
+ def _one_source(self) -> "SandboxPreflightRequest":
110
+ if (
111
+ sum(
112
+ bool(value)
113
+ for value in (self.source_path, self.source_id, self.github_repository)
114
+ )
115
+ != 1
116
+ ):
117
+ raise ValueError("exactly_one_source_required")
118
+ _validate_environment_values(self.environment_values, self.secret_refs)
119
+ return self
120
+
121
+
122
+ class SandboxRerunRequest(BaseModel):
123
+ """Fresh credentials for replaying an already-built harness session.
124
+
125
+ The saved contract, sealed environment bundle and scenarios are reused. Secret values are
126
+ deliberately supplied again (or resolved from fresh references); they are never recovered
127
+ from the completed job's persisted payload.
128
+ """
129
+
130
+ secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
131
+ environment_values: dict[str, SecretStr] = Field(default_factory=dict)
132
+ only: list[str] = Field(default_factory=list)
133
+
134
+ @model_validator(mode="after")
135
+ def _valid_environment(self) -> "SandboxRerunRequest":
136
+ _validate_environment_values(self.environment_values, self.secret_refs)
137
+ return self
138
+
139
+
140
+ class SandboxPreflightResponse(BaseModel):
141
+ source_kind: SourceKind
142
+ source_label: str
143
+ ready_to_submit: bool
144
+ checkout_required: bool = False
145
+ credentials: CredentialManifest
146
+ packaging: PackagingManifest | None = None
147
+ notes: list[str] = Field(default_factory=list)
148
+
149
+
150
+ class SandboxJobResponse(BaseModel):
151
+ job: HarnessJob
152
+ status: HarnessJobStatus
153
+ events: list[dict[str, Any]] = Field(default_factory=list)
154
+ stage_outputs: list[dict[str, Any]] = Field(default_factory=list)
155
+ artifact_path: str | None = None
156
+ credentials: CredentialManifest | None = None
157
+ stage_outputs: list[dict[str, Any]] = Field(default_factory=list)
158
+ adjustments: list[dict[str, Any]] = Field(default_factory=list)
159
+
160
+
161
+ class SandboxAdjustmentRequest(BaseModel):
162
+ instruction: str = Field(min_length=1, max_length=2_000)
163
+ client_request_id: str | None = Field(default=None, max_length=128)
164
+
165
+
166
+ class UploadedSourceResponse(BaseModel):
167
+ source_id: str
168
+ name: str
169
+ file_count: int
170
+ total_bytes: int
171
+
172
+
173
+ class UploadedSecretFileResponse(BaseModel):
174
+ """Opaque handle for a file that is materialized only at the worker boundary."""
175
+
176
+ environment_name: str
177
+ secret_ref: SecretRef
178
+ size: int
179
+
180
+
181
+ class LocalSandbox:
182
+ def __init__(
183
+ self,
184
+ root: Path,
185
+ *,
186
+ max_concurrency: int = 2,
187
+ upload_root: Path | None = None,
188
+ secret_file_root: Path | None = None,
189
+ ) -> None:
190
+ self.root = root.expanduser().resolve()
191
+ self.jobs_root = self.root / "jobs"
192
+ self.artifacts_root = self.root / "artifacts"
193
+ self.uploads_root = (
194
+ (
195
+ upload_root
196
+ or Path(
197
+ os.getenv("ALK_SANDBOX_UPLOAD_ROOT", str(self.root / "uploads"))
198
+ )
199
+ )
200
+ .expanduser()
201
+ .resolve()
202
+ )
203
+ self.jobs_root.mkdir(parents=True, exist_ok=True)
204
+ self.artifacts_root.mkdir(parents=True, exist_ok=True)
205
+ self.uploads_root.mkdir(parents=True, exist_ok=True)
206
+ self.secret_files_root = (
207
+ (
208
+ secret_file_root
209
+ or Path(
210
+ os.getenv(
211
+ "ALK_SANDBOX_SECRET_FILE_ROOT",
212
+ str(self.root / "secret-files"),
213
+ )
214
+ )
215
+ )
216
+ .expanduser()
217
+ .resolve()
218
+ )
219
+ # A sandbox-service restart orphans every in-flight worker. Secret files therefore have
220
+ # no legitimate consumer after restart and must fail closed instead of lingering on disk.
221
+ shutil.rmtree(self.secret_files_root, ignore_errors=True)
222
+ (self.secret_files_root / "pending").mkdir(parents=True, mode=0o700)
223
+ (self.secret_files_root / "jobs").mkdir(parents=True, mode=0o700)
224
+ self._semaphore = asyncio.Semaphore(max(1, max_concurrency))
225
+ self._processes: dict[str, asyncio.subprocess.Process] = {}
226
+ self._tasks: dict[str, asyncio.Task[None]] = {}
227
+ # Values supplied through the platform's .env flow live only for the lifetime of the
228
+ # job. Persisted job/state artifacts contain the opaque mounted references created below.
229
+ self._ephemeral_secrets: dict[str, dict[str, str]] = {}
230
+ self._ephemeral_secret_file_names: dict[str, set[str]] = {}
231
+ self._recover_orphans()
232
+
233
+ async def upload_secret_file(
234
+ self, uploaded: UploadFile, environment_name: str
235
+ ) -> UploadedSecretFileResponse:
236
+ """Stage one credential file and return a bearer-style opaque reference.
237
+
238
+ Contents never enter a request JSON, job, bundle, event, log or artifact. The upload is
239
+ claimed by exactly one job and removed at that job's terminal boundary.
240
+ """
241
+ name = str(environment_name).strip()
242
+ if not _ENVIRONMENT_NAME.fullmatch(name):
243
+ raise HTTPException(status_code=400, detail="environment_name is invalid")
244
+ if name in _RUNNER_RESERVED_ENVIRONMENT or name.startswith("ALK_"):
245
+ raise HTTPException(status_code=400, detail="environment_name is reserved")
246
+ self._purge_stale_secret_files()
247
+ identifier = str(uuid.uuid4())
248
+ name_digest = hashlib.sha256(name.encode()).hexdigest()[:12].upper()
249
+ internal_key = (
250
+ f"ALK_SECRET_FILE_{identifier.replace('-', '').upper()}_{name_digest}"
251
+ )
252
+ destination = self.secret_files_root / "pending" / identifier
253
+ total = 0
254
+ try:
255
+ with destination.open("xb") as handle:
256
+ while chunk := await uploaded.read(1024 * 1024):
257
+ total += len(chunk)
258
+ if total > 5 * 1024 * 1024:
259
+ raise HTTPException(
260
+ status_code=413,
261
+ detail="credential file may not exceed 5 MiB",
262
+ )
263
+ handle.write(chunk)
264
+ destination.chmod(0o600)
265
+ if total == 0:
266
+ raise HTTPException(status_code=400, detail="credential file is empty")
267
+ if name == "GOOGLE_APPLICATION_CREDENTIALS":
268
+ try:
269
+ document = json.loads(destination.read_text(encoding="utf-8"))
270
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc:
271
+ raise HTTPException(
272
+ status_code=400,
273
+ detail="Google application credentials must be a valid JSON object",
274
+ ) from exc
275
+ if not isinstance(document, dict) or not document:
276
+ raise HTTPException(
277
+ status_code=400,
278
+ detail="Google application credentials must be a valid JSON object",
279
+ )
280
+ except Exception:
281
+ destination.unlink(missing_ok=True)
282
+ raise
283
+ return UploadedSecretFileResponse(
284
+ environment_name=name,
285
+ secret_ref=SecretRef(
286
+ manager="mounted",
287
+ key=internal_key,
288
+ purpose=f"job-scoped credential file for {name}",
289
+ ),
290
+ size=total,
291
+ )
292
+
293
+ def _claim_secret_files(
294
+ self, job_id: str, references: dict[str, SecretRef]
295
+ ) -> tuple[dict[str, str], set[str]]:
296
+ """Atomically move provider uploads into one job-private directory."""
297
+ self._purge_stale_secret_files()
298
+ candidates: list[tuple[str, str, Path]] = []
299
+ for environment_name, reference in references.items():
300
+ if reference.manager != "mounted" or not reference.key.startswith(
301
+ "ALK_SECRET_FILE_"
302
+ ):
303
+ continue
304
+ suffix = reference.key.removeprefix("ALK_SECRET_FILE_")
305
+ match = re.fullmatch(r"([0-9A-F]{32})_([0-9A-F]{12})", suffix)
306
+ expected_digest = (
307
+ hashlib.sha256(environment_name.encode()).hexdigest()[:12].upper()
308
+ )
309
+ if match is None or match.group(2) != expected_digest:
310
+ raise HTTPException(
311
+ status_code=400, detail="secret_file_reference_invalid"
312
+ )
313
+ identifier = str(uuid.UUID(hex=match.group(1).lower()))
314
+ source = self.secret_files_root / "pending" / identifier
315
+ if not source.is_file():
316
+ raise HTTPException(
317
+ status_code=400,
318
+ detail="secret_file_reference_unavailable_or_consumed",
319
+ )
320
+ candidates.append((environment_name, reference.key, source))
321
+ if not candidates:
322
+ return {}, set()
323
+ job_root = self.secret_files_root / "jobs" / job_id
324
+ job_root.mkdir(mode=0o700)
325
+ claimed: dict[str, str] = {}
326
+ names: set[str] = set()
327
+ try:
328
+ for environment_name, internal_key, source in candidates:
329
+ target = job_root / environment_name
330
+ source.replace(target)
331
+ target.chmod(0o400)
332
+ claimed[internal_key] = str(target)
333
+ names.add(environment_name)
334
+ except Exception:
335
+ shutil.rmtree(job_root, ignore_errors=True)
336
+ raise
337
+ return claimed, names
338
+
339
+ def _delete_job_secret_files(self, job_id: str) -> None:
340
+ shutil.rmtree(self.secret_files_root / "jobs" / job_id, ignore_errors=True)
341
+
342
+ def _purge_stale_secret_files(self) -> None:
343
+ ttl = max(60, int(os.getenv("ALK_SECRET_FILE_TTL_SECONDS", "900")))
344
+ cutoff = time.time() - ttl
345
+ for path in (self.secret_files_root / "pending").iterdir():
346
+ try:
347
+ if path.is_file() and path.stat().st_mtime < cutoff:
348
+ path.unlink()
349
+ except FileNotFoundError:
350
+ continue
351
+
352
+ async def upload_source(
353
+ self, files: list[UploadFile], paths: list[str], name: str
354
+ ) -> UploadedSourceResponse:
355
+ if not files or len(files) != len(paths):
356
+ raise HTTPException(
357
+ status_code=400, detail="one relative path is required per file"
358
+ )
359
+ if len(files) > 5_000:
360
+ raise HTTPException(
361
+ status_code=413, detail="source may contain at most 5000 files"
362
+ )
363
+ source_id = str(uuid.uuid4())
364
+ staging = self.uploads_root / f".{source_id}.uploading"
365
+ destination = self.uploads_root / source_id
366
+ staging.mkdir(mode=0o700)
367
+ total = 0
368
+ seen: set[str] = set()
369
+ try:
370
+ for uploaded, raw_path in zip(files, paths, strict=True):
371
+ relative = _safe_uploaded_path(raw_path)
372
+ key = relative.as_posix()
373
+ if key in seen:
374
+ raise HTTPException(
375
+ status_code=400, detail=f"duplicate source path: {key}"
376
+ )
377
+ seen.add(key)
378
+ target = staging.joinpath(*relative.parts)
379
+ target.parent.mkdir(parents=True, exist_ok=True)
380
+ file_size = 0
381
+ starts_with_shebang = False
382
+ with target.open("xb") as handle:
383
+ while chunk := await uploaded.read(1024 * 1024):
384
+ if file_size == 0:
385
+ starts_with_shebang = chunk.startswith(b"#!")
386
+ file_size += len(chunk)
387
+ total += len(chunk)
388
+ if file_size > 50 * 1024 * 1024 or total > 200 * 1024 * 1024:
389
+ raise HTTPException(
390
+ status_code=413,
391
+ detail="source exceeds the 50 MiB file or 200 MiB bundle limit",
392
+ )
393
+ handle.write(chunk)
394
+ target.chmod(0o755 if starts_with_shebang else 0o644)
395
+ manifest = {
396
+ "source_id": source_id,
397
+ "name": (name or "uploaded-agent")[:255],
398
+ "file_count": len(files),
399
+ "total_bytes": total,
400
+ "created_at": _now(),
401
+ }
402
+ staging.replace(destination)
403
+ _write_json(
404
+ self.uploads_root / f"{source_id}.json",
405
+ manifest,
406
+ )
407
+ except Exception:
408
+ shutil.rmtree(staging, ignore_errors=True)
409
+ shutil.rmtree(destination, ignore_errors=True)
410
+ (self.uploads_root / f"{source_id}.json").unlink(missing_ok=True)
411
+ raise
412
+ return UploadedSourceResponse(
413
+ source_id=source_id,
414
+ name=(name or "uploaded-agent")[:255],
415
+ file_count=len(files),
416
+ total_bytes=total,
417
+ )
418
+
419
+ def uploaded_source(self, source_id: str) -> Path:
420
+ if not re.fullmatch(r"[0-9a-f-]{36}", source_id):
421
+ raise HTTPException(status_code=400, detail="source_id is invalid")
422
+ source = (self.uploads_root / source_id).resolve()
423
+ if source.parent != self.uploads_root or not source.is_dir():
424
+ raise HTTPException(status_code=404, detail="uploaded source was not found")
425
+ if not (self.uploads_root / f"{source_id}.json").is_file():
426
+ raise HTTPException(status_code=400, detail="uploaded source is incomplete")
427
+ return source
428
+
429
+ def _recover_orphans(self) -> None:
430
+ for path in self.jobs_root.glob("*/state.json"):
431
+ state = _read_json(path)
432
+ if state.get("stage") not in _TERMINAL_STAGES:
433
+ controller_pid = int(state.get("controller_pid", 0) or 0)
434
+ controller_identity = str(state.get("controller_identity") or "")
435
+ controller_token = str(state.get("controller_token") or "")
436
+ if (
437
+ controller_pid == os.getpid()
438
+ and controller_token == _CONTROLLER_TOKEN
439
+ ) or (
440
+ controller_pid > 0
441
+ and controller_identity
442
+ and _process_identity(controller_pid) == controller_identity
443
+ ):
444
+ # Another app object or health process may inspect the same provider root.
445
+ # It must not declare a job orphaned while the owning controller still exists.
446
+ continue
447
+ state.update(
448
+ stage=HarnessStage.FAILED.value,
449
+ detail="sandbox service restarted while the job was running",
450
+ updated_at=_now(),
451
+ )
452
+ _write_json(path, state)
453
+
454
+ def submit(self, request: LocalSandboxRequest) -> SandboxJobResponse:
455
+ source = (
456
+ _allowed_source(request.source_path)
457
+ if request.source_path
458
+ else self.uploaded_source(request.source_id)
459
+ if request.source_id
460
+ else None
461
+ )
462
+ identifier = str(uuid.uuid4())
463
+ run_id = f"harness-{identifier}"
464
+ github_location = (
465
+ _github_location(request.github_repository)
466
+ if request.github_repository
467
+ else None
468
+ )
469
+ github_repository = github_location.repository if github_location else None
470
+ mounted_refs: dict[str, SecretRef] = {}
471
+ mounted_values: dict[str, str] = {}
472
+ for index, (name, value) in enumerate(request.environment_values.items()):
473
+ internal_key = f"ALK_JOB_{identifier.replace('-', '').upper()}_{index}"
474
+ mounted_refs[name] = SecretRef(
475
+ manager="mounted",
476
+ key=internal_key,
477
+ purpose=f"job-scoped environment value for {name}",
478
+ )
479
+ mounted_values[internal_key] = value.get_secret_value()
480
+ claimed_files, secret_file_names = self._claim_secret_files(
481
+ identifier, request.secret_refs
482
+ )
483
+ mounted_values.update(claimed_files)
484
+ try:
485
+ source_spec = (
486
+ RepositorySource(
487
+ kind=SourceKind.ARCHIVE,
488
+ archive_artifact_id=request.source_id,
489
+ )
490
+ if request.source_id
491
+ else RepositorySource(
492
+ kind=SourceKind.LOCAL_REPOSITORY, local_path=str(source)
493
+ )
494
+ if source
495
+ else RepositorySource(
496
+ kind=SourceKind.GITHUB,
497
+ repository=github_repository,
498
+ ref=(github_location.ref if github_location else None)
499
+ or request.github_ref,
500
+ commit_sha=request.github_commit_sha,
501
+ visibility=request.github_visibility,
502
+ installation_id=request.github_installation_id,
503
+ )
504
+ )
505
+ job = HarnessJob(
506
+ job_id=identifier,
507
+ run_id=run_id,
508
+ execution=(
509
+ ExecutionMode.HOSTED
510
+ if request.source_id or request.github_repository
511
+ else ExecutionMode.LOCAL
512
+ ),
513
+ source=source_spec,
514
+ agent=AgentConnection(
515
+ connector=request.connector,
516
+ config=request.connector_config,
517
+ secret_refs={**request.secret_refs, **mounted_refs},
518
+ ),
519
+ scenario_count=request.scenario_count,
520
+ seed=request.seed,
521
+ artifacts=HarnessArtifactPolicy(
522
+ level="full", allow_bundle_download=True
523
+ ),
524
+ platform_run_id=request.platform_run_id,
525
+ metadata={
526
+ **request.metadata,
527
+ "agent_name": request.agent_name
528
+ or (
529
+ _uploaded_source_name(source)
530
+ if request.source_id and source
531
+ else source.name
532
+ if source
533
+ else str(github_repository).split("/")[-1]
534
+ ),
535
+ "source_kind": source_spec.kind.value,
536
+ # Both dotenv values and credential-file paths are runtime configuration. Only
537
+ # their names are persisted; values and provider-local paths remain ephemeral.
538
+ "environment_value_names": sorted(
539
+ {*mounted_refs, *secret_file_names}
540
+ ),
541
+ "secret_file_names": sorted(secret_file_names),
542
+ },
543
+ )
544
+ directory = self.jobs_root / identifier
545
+ directory.mkdir()
546
+ _write_json(directory / "job.json", job.model_dump(mode="json"))
547
+ _write_json(
548
+ directory / "state.json",
549
+ {
550
+ "job_id": identifier,
551
+ "run_id": run_id,
552
+ "stage": HarnessStage.QUEUED.value,
553
+ "updated_at": _now(),
554
+ "detail": "waiting for a local sandbox slot",
555
+ "completed_scenarios": 0,
556
+ "total_scenarios": request.scenario_count,
557
+ "attempt": 1,
558
+ "controller_pid": os.getpid(),
559
+ "controller_identity": _process_identity(os.getpid()),
560
+ "controller_token": _CONTROLLER_TOKEN,
561
+ },
562
+ )
563
+ except Exception:
564
+ self._delete_job_secret_files(identifier)
565
+ shutil.rmtree(self.jobs_root / identifier, ignore_errors=True)
566
+ raise
567
+ if mounted_values:
568
+ self._ephemeral_secrets[identifier] = mounted_values
569
+ if secret_file_names:
570
+ self._ephemeral_secret_file_names[identifier] = secret_file_names
571
+ self._tasks[identifier] = asyncio.create_task(self._execute(job, source))
572
+ return self.get(identifier)
573
+
574
+ def rerun(self, job_id: str, request: SandboxRerunRequest) -> SandboxJobResponse:
575
+ response = self.get(job_id)
576
+ if not response.status.stage.terminal:
577
+ raise HTTPException(
578
+ status_code=409, detail="harness job is already running"
579
+ )
580
+ output = self.artifacts_root / response.job.run_id
581
+ required = (
582
+ output / "contract.json",
583
+ output / "scenarios.json",
584
+ output / "environment-bundle" / "manifest.json",
585
+ output / "platform.json",
586
+ )
587
+ missing = [path.name for path in required if not path.is_file()]
588
+ if missing:
589
+ raise HTTPException(
590
+ status_code=409,
591
+ detail=(
592
+ "saved harness session cannot be rerun; missing "
593
+ + ", ".join(missing)
594
+ ),
595
+ )
596
+
597
+ runtime_configuration = {
598
+ name: value.get_secret_value()
599
+ for name, value in request.environment_values.items()
600
+ }
601
+ claimed_files, secret_file_names = self._claim_secret_files(
602
+ job_id, request.secret_refs
603
+ )
604
+ try:
605
+ resolved = resolve_worker_secrets(
606
+ request.secret_refs,
607
+ environment={**os.environ, **claimed_files},
608
+ )
609
+ except Exception:
610
+ self._delete_job_secret_files(job_id)
611
+ raise
612
+ runtime_configuration.update(resolved)
613
+ self._ephemeral_secrets[job_id] = dict(runtime_configuration)
614
+ self._ephemeral_secret_file_names[job_id] = secret_file_names
615
+
616
+ state_path = self.jobs_root / job_id / "state.json"
617
+ state = _read_json(state_path)
618
+ operation_started_at = _now()
619
+ state.update(
620
+ stage=HarnessStage.QUEUED.value,
621
+ detail="waiting to restart the saved environment",
622
+ failure=None,
623
+ completed_scenarios=0,
624
+ total_scenarios=(len(request.only) or response.job.scenario_count),
625
+ operation="rerun",
626
+ operation_started_at=operation_started_at,
627
+ attempt=int(state.get("attempt", 0) or 0) + 1,
628
+ updated_at=operation_started_at,
629
+ )
630
+ _write_json(state_path, state)
631
+ self._tasks[job_id] = asyncio.create_task(
632
+ self._execute_rerun(response.job, request.only)
633
+ )
634
+ return self.get(job_id)
635
+
636
+ async def _execute_rerun(self, job: HarnessJob, only: list[str]) -> None:
637
+ """Run calls from immutable saved artifacts, with a fresh environment lifecycle."""
638
+
639
+ directory = self.jobs_root / job.job_id
640
+ state_path = directory / "state.json"
641
+ output = self.artifacts_root / job.run_id
642
+ # The local runner controls a host Docker daemon. Paths in Compose bind mounts are
643
+ # therefore resolved by that daemon, not inside this service container. Artifacts live
644
+ # on the runner's private volume, so replaying a bundle directly from ``output`` makes
645
+ # repository seed/config files invisible to Docker. Materialize a disposable copy in
646
+ # the provider's Docker-visible upload workspace. Hosted providers must offer the same
647
+ # workspace guarantee even when their implementation is not a nested Docker daemon.
648
+ replay_root = self.uploads_root / ".reruns" / job.job_id
649
+ async with self._semaphore:
650
+ state = _read_json(state_path)
651
+ if state.get("stage") == HarnessStage.CANCELED.value:
652
+ return
653
+ state.update(
654
+ stage=HarnessStage.RUNNING.value,
655
+ detail="restarting the saved environment and running scenarios",
656
+ updated_at=_now(),
657
+ )
658
+ _write_json(state_path, state)
659
+ log_handle = (directory / "worker.log").open("ab")
660
+ try:
661
+ shutil.rmtree(replay_root, ignore_errors=True)
662
+ replay_root.parent.mkdir(parents=True, exist_ok=True)
663
+ shutil.copytree(output / "environment-bundle", replay_root)
664
+ runtime_configuration = self._ephemeral_secrets.get(job.job_id, {})
665
+ child_environment = worker_environment(
666
+ {}, runtime_configuration=runtime_configuration
667
+ )
668
+ child_environment["ALK_RUNTIME_CONFIGURATION_NAMES"] = ",".join(
669
+ sorted(runtime_configuration)
670
+ )
671
+ child_environment["ALK_RUNTIME_SECRET_FILE_NAMES"] = ",".join(
672
+ sorted(self._ephemeral_secret_file_names.get(job.job_id, set()))
673
+ )
674
+ child_environment["ALK_HOSTED_EXECUTION"] = (
675
+ "1" if job.execution is ExecutionMode.HOSTED else "0"
676
+ )
677
+ child_environment["ALK_HARNESS_JOB_ID"] = job.job_id
678
+
679
+ async def run_stage(command: list[str]) -> int:
680
+ process = await asyncio.create_subprocess_exec(
681
+ *command,
682
+ stdout=log_handle,
683
+ stderr=asyncio.subprocess.STDOUT,
684
+ start_new_session=True,
685
+ env=child_environment,
686
+ )
687
+ self._processes[job.job_id] = process
688
+ return await process.wait()
689
+
690
+ environment_up = [
691
+ sys.executable,
692
+ "-m",
693
+ "fi.alk.harness.cli",
694
+ "environment",
695
+ "up",
696
+ "--bundle",
697
+ str(replay_root),
698
+ "--out",
699
+ str(output),
700
+ ]
701
+ simulation = [
702
+ sys.executable,
703
+ "-m",
704
+ "fi.alk.harness.cli",
705
+ "simulate",
706
+ "--name",
707
+ str(job.metadata.get("agent_name") or job.run_id),
708
+ "--out",
709
+ str(output),
710
+ ]
711
+ if only:
712
+ simulation.extend(["--only", *only])
713
+ environment_down = [
714
+ sys.executable,
715
+ "-m",
716
+ "fi.alk.harness.cli",
717
+ "environment",
718
+ "down",
719
+ "--out",
720
+ str(output),
721
+ ]
722
+
723
+ up_code = await run_stage(environment_up)
724
+ simulation_code = 1
725
+ cleanup_code = 0
726
+ if up_code == 0:
727
+ try:
728
+ simulation_code = await run_stage(simulation)
729
+ finally:
730
+ cleanup_code = await run_stage(environment_down)
731
+
732
+ # Exit 2 is a completed suite whose submitted agent failed checks.
733
+ successful_execution = (
734
+ up_code == 0 and simulation_code in (0, 2) and cleanup_code == 0
735
+ )
736
+ failed_stage = (
737
+ "environment_up"
738
+ if up_code != 0
739
+ else "environment_down"
740
+ if cleanup_code != 0
741
+ else "simulation"
742
+ )
743
+ failure_stage = (
744
+ HarnessStage.BUILDING_ENVIRONMENT.value
745
+ if up_code != 0
746
+ else HarnessStage.CLEANING_UP.value
747
+ if cleanup_code != 0
748
+ else HarnessStage.RUNNING.value
749
+ )
750
+ return_code = (
751
+ up_code
752
+ if up_code != 0
753
+ else cleanup_code
754
+ if cleanup_code != 0
755
+ else simulation_code
756
+ )
757
+ state = _read_json(state_path)
758
+ if state.get("stage") != HarnessStage.CANCELED.value:
759
+ state.update(
760
+ stage=(
761
+ HarnessStage.COMPLETED.value
762
+ if successful_execution
763
+ else HarnessStage.FAILED.value
764
+ ),
765
+ detail=(
766
+ "saved harness session rerun completed"
767
+ if successful_execution
768
+ else f"saved harness session rerun exited {return_code}"
769
+ ),
770
+ completed_scenarios=(len(only) if only else job.scenario_count)
771
+ if successful_execution
772
+ else 0,
773
+ failure=None
774
+ if successful_execution
775
+ else {
776
+ "domain": "environment",
777
+ "stage": failure_stage,
778
+ "code": "saved_session_rerun_failed",
779
+ "message": f"{failed_stage} exited {return_code}",
780
+ "retryable": False,
781
+ },
782
+ updated_at=_now(),
783
+ )
784
+ _write_json(state_path, state)
785
+ except Exception as exc:
786
+ state = _read_json(state_path)
787
+ state.update(
788
+ stage=HarnessStage.FAILED.value,
789
+ detail=f"{type(exc).__name__}: {exc}",
790
+ failure={
791
+ "domain": "infrastructure",
792
+ "stage": HarnessStage.RUNNING.value,
793
+ "code": type(exc).__name__,
794
+ "message": str(exc),
795
+ "retryable": False,
796
+ },
797
+ updated_at=_now(),
798
+ )
799
+ _write_json(state_path, state)
800
+ finally:
801
+ log_handle.close()
802
+ self._processes.pop(job.job_id, None)
803
+ self._tasks.pop(job.job_id, None)
804
+ self._ephemeral_secrets.pop(job.job_id, None)
805
+ self._ephemeral_secret_file_names.pop(job.job_id, None)
806
+ self._delete_job_secret_files(job.job_id)
807
+ shutil.rmtree(replay_root, ignore_errors=True)
808
+
809
+ def preflight(self, request: SandboxPreflightRequest) -> SandboxPreflightResponse:
810
+ if request.source_path or request.source_id:
811
+ source = (
812
+ _allowed_source(request.source_path)
813
+ if request.source_path
814
+ else self.uploaded_source(request.source_id or "")
815
+ )
816
+ packaging = inspect_packaging(
817
+ source,
818
+ external_environment=bool(request.environment_values),
819
+ )
820
+ manifest = discover_credentials(
821
+ source,
822
+ secret_refs=request.secret_refs,
823
+ provided_environment={
824
+ **request.connector_config,
825
+ **{
826
+ name: value.get_secret_value()
827
+ for name, value in request.environment_values.items()
828
+ },
829
+ },
830
+ scan_paths=_credential_scan_paths(packaging),
831
+ )
832
+ return SandboxPreflightResponse(
833
+ source_kind=(
834
+ SourceKind.ARCHIVE
835
+ if request.source_id
836
+ else SourceKind.LOCAL_REPOSITORY
837
+ ),
838
+ source_label=(
839
+ _uploaded_source_name(source) if request.source_id else str(source)
840
+ ),
841
+ ready_to_submit=(
842
+ manifest.ready
843
+ and packaging.ready
844
+ and (packaging.agent_runtime_packaged or not packaging.candidates)
845
+ ),
846
+ credentials=manifest,
847
+ packaging=packaging,
848
+ )
849
+ location = _github_location(request.github_repository or "")
850
+ repository = location.repository
851
+ requirement = CredentialRequirement(
852
+ id="github_installation",
853
+ environment_name="GITHUB_INSTALLATION",
854
+ provider="github",
855
+ purpose="read private repository source",
856
+ kind=RequirementKind.SECRET,
857
+ required=request.github_visibility is SourceVisibility.PRIVATE,
858
+ status=(
859
+ RequirementStatus.CONFIGURED
860
+ if request.github_installation_id
861
+ else RequirementStatus.MISSING
862
+ if request.github_visibility is SourceVisibility.PRIVATE
863
+ else RequirementStatus.OPTIONAL
864
+ ),
865
+ accepted_secret_types=["github_app_installation"],
866
+ )
867
+ manifest = CredentialManifest(
868
+ source_digest="0" * 64,
869
+ detected_connectors=[],
870
+ requirements=[requirement],
871
+ scanned_files=0,
872
+ )
873
+ return SandboxPreflightResponse(
874
+ source_kind=SourceKind.GITHUB,
875
+ source_label=repository,
876
+ ready_to_submit=manifest.ready,
877
+ checkout_required=True,
878
+ credentials=manifest,
879
+ notes=[
880
+ "Agent credential discovery continues inside the sandbox after checkout."
881
+ ],
882
+ )
883
+
884
+ async def _execute(self, job: HarnessJob, source: Path | None) -> None:
885
+ directory = self.jobs_root / job.job_id
886
+ state_path = directory / "state.json"
887
+ output = self.artifacts_root / job.run_id
888
+ async with self._semaphore:
889
+ state = _read_json(state_path)
890
+ if state.get("stage") == HarnessStage.CANCELED.value:
891
+ return
892
+ state.update(
893
+ stage=HarnessStage.ACQUIRING_SOURCE.value,
894
+ detail="acquiring source inside the ALK runner",
895
+ updated_at=_now(),
896
+ )
897
+ _write_json(state_path, state)
898
+ log_handle = (directory / "worker.log").open("ab")
899
+ try:
900
+ if source is None:
901
+ workspace = directory / "workspace"
902
+ for attempt in range(1, job.retry.max_infrastructure_attempts + 1):
903
+ state.update(
904
+ attempt=attempt,
905
+ detail=f"cloning public GitHub source (attempt {attempt})",
906
+ updated_at=_now(),
907
+ )
908
+ _write_json(state_path, state)
909
+ try:
910
+ source = await GitHubSourceAcquirer(
911
+ _github_installation_token
912
+ ).acquire(job, workspace)
913
+ break
914
+ except SourceAcquisitionError:
915
+ checkout = workspace / "repository"
916
+ if checkout.exists():
917
+ shutil.rmtree(checkout)
918
+ if attempt >= job.retry.max_infrastructure_attempts:
919
+ raise
920
+ delay = min(
921
+ job.retry.max_backoff_seconds,
922
+ job.retry.initial_backoff_seconds
923
+ * (2 ** (attempt - 1)),
924
+ )
925
+ await asyncio.sleep(delay)
926
+ if source is None:
927
+ raise SourceAcquisitionError("github_checkout_missing")
928
+ packaging_manifest = inspect_packaging(
929
+ source,
930
+ external_environment=bool(
931
+ job.metadata.get("environment_value_names", [])
932
+ ),
933
+ )
934
+ credential_manifest = discover_credentials(
935
+ source,
936
+ secret_refs=job.agent.secret_refs,
937
+ provided_environment=job.agent.config,
938
+ scan_paths=_credential_scan_paths(packaging_manifest),
939
+ )
940
+ _write_json(
941
+ directory / "packaging.json",
942
+ packaging_manifest.model_dump(mode="json"),
943
+ )
944
+ if not packaging_manifest.ready:
945
+ detail = (
946
+ "; ".join(packaging_manifest.notes)
947
+ or "packaging is not runnable"
948
+ )
949
+ state.update(
950
+ stage=HarnessStage.FAILED.value,
951
+ detail=f"repository packaging preflight failed: {detail}",
952
+ failure={
953
+ "domain": "environment",
954
+ "stage": HarnessStage.ACQUIRING_SOURCE.value,
955
+ "code": "packaging_preflight_failed",
956
+ "message": detail,
957
+ "retryable": False,
958
+ },
959
+ updated_at=_now(),
960
+ )
961
+ _write_json(state_path, state)
962
+ return
963
+ _write_json(
964
+ directory / "credentials.json",
965
+ credential_manifest.model_dump(mode="json"),
966
+ )
967
+ if not credential_manifest.ready:
968
+ missing = [
969
+ item.environment_name
970
+ for item in credential_manifest.missing_required
971
+ ]
972
+ missing.extend(
973
+ f"one of {choice.options}"
974
+ for choice in credential_manifest.credential_choices
975
+ if not choice.satisfied
976
+ )
977
+ names = ", ".join(missing)
978
+ state.update(
979
+ stage=HarnessStage.FAILED.value,
980
+ detail=f"missing required credentials: {names}",
981
+ failure={
982
+ "domain": "connectivity",
983
+ "stage": HarnessStage.ACQUIRING_SOURCE.value,
984
+ "code": "credentials_missing",
985
+ "message": f"Configure secret references for: {names}",
986
+ "retryable": False,
987
+ },
988
+ updated_at=_now(),
989
+ )
990
+ _write_json(state_path, state)
991
+ return
992
+ resolved_secrets = resolve_worker_secrets(
993
+ job.agent.secret_refs,
994
+ environment={
995
+ **os.environ,
996
+ **self._ephemeral_secrets.get(job.job_id, {}),
997
+ },
998
+ )
999
+ submitted_source_digest = source_fingerprint(source)
1000
+ result: dict[str, Any] = {}
1001
+ return_code = 1
1002
+ for worker_attempt in range(
1003
+ 1, job.retry.max_infrastructure_attempts + 1
1004
+ ):
1005
+ state.update(
1006
+ attempt=worker_attempt,
1007
+ detail=f"running isolated harness worker (attempt {worker_attempt})",
1008
+ updated_at=_now(),
1009
+ )
1010
+ _write_json(state_path, state)
1011
+ if worker_attempt > 1 and output.exists():
1012
+ archived = directory / f"failed-attempt-{worker_attempt - 1}"
1013
+ if archived.exists():
1014
+ shutil.rmtree(archived)
1015
+ output.replace(archived)
1016
+ runtime_names = {
1017
+ str(name)
1018
+ for name in job.metadata.get("environment_value_names", [])
1019
+ }
1020
+ runtime_configuration = {
1021
+ name: value
1022
+ for name, value in resolved_secrets.items()
1023
+ if name in runtime_names
1024
+ }
1025
+ controller_configuration = {
1026
+ **_configuration_environment(job.agent.config),
1027
+ **{
1028
+ name: value
1029
+ for name, value in resolved_secrets.items()
1030
+ if name not in runtime_names
1031
+ },
1032
+ }
1033
+ child_environment = worker_environment(
1034
+ controller_configuration,
1035
+ runtime_configuration=runtime_configuration,
1036
+ )
1037
+ # The provisioner must explicitly pass customer-provided values into
1038
+ # Dockerfile/generated runtime containers. Persist names only; values stay
1039
+ # in this child process environment and never enter the job or bundle.
1040
+ child_environment["ALK_RUNTIME_CONFIGURATION_NAMES"] = ",".join(
1041
+ sorted(runtime_configuration)
1042
+ )
1043
+ child_environment["ALK_RUNTIME_SECRET_FILE_NAMES"] = ",".join(
1044
+ sorted(self._ephemeral_secret_file_names.get(job.job_id, set()))
1045
+ )
1046
+ child_environment["ALK_HOSTED_EXECUTION"] = (
1047
+ "1" if job.execution is ExecutionMode.HOSTED else "0"
1048
+ )
1049
+ child_environment["ALK_HARNESS_JOB_ID"] = job.job_id
1050
+ process = await asyncio.create_subprocess_exec(
1051
+ sys.executable,
1052
+ "-m",
1053
+ "fi.alk.harness.sandbox_worker",
1054
+ str(directory / "job.json"),
1055
+ "--source",
1056
+ str(source),
1057
+ "--output",
1058
+ str(output),
1059
+ "--status",
1060
+ str(directory / "result.json"),
1061
+ "--adjustments",
1062
+ str(directory / "adjustments.jsonl"),
1063
+ stdout=log_handle,
1064
+ stderr=asyncio.subprocess.STDOUT,
1065
+ start_new_session=True,
1066
+ env=child_environment,
1067
+ )
1068
+ self._processes[job.job_id] = process
1069
+ return_code = await process.wait()
1070
+ result = _read_json(directory / "result.json")
1071
+ if source_fingerprint(source) != submitted_source_digest:
1072
+ return_code = 1
1073
+ result = {
1074
+ "stage": HarnessStage.FAILED.value,
1075
+ "detail": "submitted source changed during isolated execution",
1076
+ "failure": {
1077
+ "domain": "infrastructure",
1078
+ "stage": HarnessStage.CLEANING_UP.value,
1079
+ "code": "source_mutation_detected",
1080
+ "message": (
1081
+ "The runner detected writes to the read-only source tree"
1082
+ ),
1083
+ "retryable": False,
1084
+ },
1085
+ }
1086
+ failure = result.get("failure") or {}
1087
+ retryable = _worker_failure_retryable(
1088
+ return_code,
1089
+ failure,
1090
+ job.retry.retryable_domains,
1091
+ )
1092
+ if (
1093
+ not retryable
1094
+ or worker_attempt >= job.retry.max_infrastructure_attempts
1095
+ ):
1096
+ break
1097
+ delay = min(
1098
+ job.retry.max_backoff_seconds,
1099
+ job.retry.initial_backoff_seconds * (2 ** (worker_attempt - 1)),
1100
+ )
1101
+ state.update(
1102
+ detail=(
1103
+ f"retrying {failure.get('domain', 'infrastructure')} "
1104
+ f"failure in {delay:g}s"
1105
+ ),
1106
+ updated_at=_now(),
1107
+ )
1108
+ _write_json(state_path, state)
1109
+ await asyncio.sleep(delay)
1110
+ state = _read_json(state_path)
1111
+ if state.get("stage") != HarnessStage.CANCELED.value:
1112
+ state.update(
1113
+ stage=result.get(
1114
+ "stage",
1115
+ HarnessStage.COMPLETED.value
1116
+ if return_code == 0
1117
+ else HarnessStage.FAILED.value,
1118
+ ),
1119
+ detail=result.get("detail")
1120
+ or (
1121
+ None if return_code == 0 else f"worker exited {return_code}"
1122
+ ),
1123
+ completed_scenarios=result.get("completed_scenarios", 0),
1124
+ failure=result.get("failure"),
1125
+ attempt=state.get("attempt", 1),
1126
+ updated_at=_now(),
1127
+ )
1128
+ _write_json(state_path, state)
1129
+ except Exception as exc:
1130
+ state = _read_json(state_path)
1131
+ domain = (
1132
+ "connectivity"
1133
+ if isinstance(exc, SourceAcquisitionError)
1134
+ else "infrastructure"
1135
+ )
1136
+ state.update(
1137
+ stage=HarnessStage.FAILED.value,
1138
+ detail=f"{type(exc).__name__}: {exc}",
1139
+ failure={
1140
+ "domain": domain,
1141
+ "stage": HarnessStage.ACQUIRING_SOURCE.value,
1142
+ "code": type(exc).__name__,
1143
+ "message": str(exc),
1144
+ "retryable": isinstance(exc, SourceAcquisitionError),
1145
+ },
1146
+ updated_at=_now(),
1147
+ )
1148
+ _write_json(state_path, state)
1149
+ finally:
1150
+ log_handle.close()
1151
+ self._processes.pop(job.job_id, None)
1152
+ self._tasks.pop(job.job_id, None)
1153
+ self._ephemeral_secrets.pop(job.job_id, None)
1154
+ self._ephemeral_secret_file_names.pop(job.job_id, None)
1155
+ self._delete_job_secret_files(job.job_id)
1156
+
1157
+ def get(self, job_id: str) -> SandboxJobResponse:
1158
+ directory = self.jobs_root / job_id
1159
+ if not directory.is_dir():
1160
+ raise KeyError(job_id)
1161
+ job = HarnessJob.model_validate(_read_json(directory / "job.json"))
1162
+ raw_state = _read_json(directory / "state.json")
1163
+ events = _events(self.artifacts_root / job.run_id / "harness-events.jsonl")
1164
+ event_stage = _stage_from_events(events)
1165
+ if (
1166
+ raw_state.get("operation") == "rerun"
1167
+ and raw_state.get("stage") not in _TERMINAL_STAGES
1168
+ ):
1169
+ # Historic event streams end in ``completed``. During a rerun the current state is
1170
+ # authoritative until the new invocation emits/commits its terminal result.
1171
+ event_stage = None
1172
+ stage = event_stage or raw_state.get("stage", "queued")
1173
+ updated_at = raw_state.get("updated_at", _now())
1174
+ detail = raw_state.get("detail")
1175
+ if event_stage and events:
1176
+ # While the worker is alive, state.json is intentionally written only
1177
+ # at process boundaries. The canonical event stream is the live
1178
+ # source of truth, so expose its timestamp and stage instead of making
1179
+ # a healthy long-running job look stale in the platform UI.
1180
+ updated_at = events[-1].get("wall_time") or updated_at
1181
+ detail = {
1182
+ HarnessStage.UNDERSTANDING_AGENT.value: "understanding agent source",
1183
+ HarnessStage.GENERATING_ENVIRONMENT.value: "provisioning and validating environment",
1184
+ HarnessStage.GENERATING_SCENARIOS.value: "generating and validating scenarios",
1185
+ HarnessStage.RUNNING.value: "running scenarios",
1186
+ }.get(event_stage, detail)
1187
+ if raw_state.get("stage") in _TERMINAL_STAGES:
1188
+ stage = raw_state["stage"]
1189
+ updated_at = raw_state.get("updated_at", updated_at)
1190
+ detail = raw_state.get("detail")
1191
+ scenario_index = _json_artifact(
1192
+ self.artifacts_root / job.run_id / "scenarios.json"
1193
+ )
1194
+ discovered_scenarios = (
1195
+ len(scenario_index) if isinstance(scenario_index, list) else 0
1196
+ )
1197
+ completed_scenarios = int(raw_state.get("completed_scenarios", 0) or 0)
1198
+ if stage == HarnessStage.RUNNING.value:
1199
+ # The worker writes state.json only at process boundaries, while each scenario result
1200
+ # is committed as soon as that scenario finishes. Derive live progress from the newest
1201
+ # campaign directory so a multi-call run does not remain at 0/N until finalization.
1202
+ # Only immediate scenario result files count; nested SDK/debug artifacts are ignored.
1203
+ runs_root = self.artifacts_root / job.run_id / "runs"
1204
+ campaigns = [path for path in runs_root.glob("run-*") if path.is_dir()]
1205
+ operation_started_at = str(
1206
+ raw_state.get("operation_started_at") or ""
1207
+ ).strip()
1208
+ if raw_state.get("operation") == "rerun" and operation_started_at:
1209
+ try:
1210
+ started_ns = int(
1211
+ datetime.fromisoformat(operation_started_at).timestamp()
1212
+ * 1_000_000_000
1213
+ )
1214
+ campaigns = [
1215
+ path
1216
+ for path in campaigns
1217
+ if path.stat().st_mtime_ns >= started_ns
1218
+ ]
1219
+ except ValueError:
1220
+ campaigns = []
1221
+ if campaigns:
1222
+ latest = max(campaigns, key=lambda path: path.stat().st_mtime_ns)
1223
+ committed = sum(
1224
+ 1
1225
+ for scenario in latest.iterdir()
1226
+ if scenario.is_dir() and (scenario / "result.json").is_file()
1227
+ )
1228
+ completed_scenarios = min(
1229
+ raw_state.get("total_scenarios", job.scenario_count),
1230
+ max(completed_scenarios, committed),
1231
+ )
1232
+ failure = raw_state.get("failure")
1233
+ if isinstance(failure, dict):
1234
+ legacy_stage = str(failure.get("stage") or "")
1235
+ normalized_stage = {
1236
+ "environment_up": HarnessStage.BUILDING_ENVIRONMENT.value,
1237
+ "environment_down": HarnessStage.CLEANING_UP.value,
1238
+ "simulation": HarnessStage.RUNNING.value,
1239
+ }.get(legacy_stage)
1240
+ if normalized_stage:
1241
+ failure = {**failure, "stage": normalized_stage}
1242
+ status_value = HarnessJobStatus(
1243
+ job_id=job.job_id,
1244
+ run_id=job.run_id,
1245
+ stage=stage,
1246
+ updated_at=updated_at,
1247
+ detail=detail,
1248
+ failure=failure,
1249
+ completed_scenarios=completed_scenarios,
1250
+ total_scenarios=max(
1251
+ raw_state.get("total_scenarios", job.scenario_count),
1252
+ discovered_scenarios,
1253
+ ),
1254
+ attempt=raw_state.get("attempt", 1),
1255
+ )
1256
+ return SandboxJobResponse(
1257
+ job=job,
1258
+ status=status_value,
1259
+ events=events,
1260
+ stage_outputs=_stage_outputs(self.artifacts_root / job.run_id),
1261
+ artifact_path=str(self.artifacts_root / job.run_id),
1262
+ credentials=(
1263
+ CredentialManifest.model_validate(
1264
+ _read_json(directory / "credentials.json")
1265
+ )
1266
+ if (directory / "credentials.json").is_file()
1267
+ else None
1268
+ ),
1269
+ adjustments=_adjustments(directory),
1270
+ )
1271
+
1272
+ def list(self) -> list[SandboxJobResponse]:
1273
+ responses = []
1274
+ for directory in sorted(
1275
+ self.jobs_root.iterdir(),
1276
+ key=lambda path: path.stat().st_mtime,
1277
+ reverse=True,
1278
+ ):
1279
+ if directory.is_dir():
1280
+ responses.append(self.get(directory.name))
1281
+ return responses
1282
+
1283
+ async def cancel(self, job_id: str) -> SandboxJobResponse:
1284
+ response = self.get(job_id)
1285
+ if response.status.stage.terminal:
1286
+ return response
1287
+ state_path = self.jobs_root / job_id / "state.json"
1288
+ state = _read_json(state_path)
1289
+ state.update(
1290
+ stage=HarnessStage.CANCELED.value,
1291
+ detail="canceled by user",
1292
+ updated_at=_now(),
1293
+ )
1294
+ _write_json(state_path, state)
1295
+ process = self._processes.get(job_id)
1296
+ if process and process.returncode is None:
1297
+ try:
1298
+ os.killpg(process.pid, signal.SIGTERM)
1299
+ except ProcessLookupError:
1300
+ pass
1301
+ try:
1302
+ await asyncio.wait_for(process.wait(), timeout=10)
1303
+ except TimeoutError:
1304
+ try:
1305
+ os.killpg(process.pid, signal.SIGKILL)
1306
+ except ProcessLookupError:
1307
+ pass
1308
+ task = self._tasks.get(job_id)
1309
+ if task and not task.done():
1310
+ task.cancel()
1311
+ # SIGTERM stops the worker before its async ``finally`` can reliably run. Cancellation
1312
+ # must therefore own the same environment boundary explicitly; otherwise every timed-out
1313
+ # hosted job leaves its database, network and volumes behind.
1314
+ output = self.artifacts_root / response.job.run_id
1315
+ try:
1316
+ await asyncio.to_thread(stop, output)
1317
+ except ProvisionError as exc:
1318
+ state.update(
1319
+ detail=f"canceled; isolated environment cleanup failed: {exc}",
1320
+ updated_at=_now(),
1321
+ )
1322
+ _write_json(state_path, state)
1323
+ self._ephemeral_secrets.pop(job_id, None)
1324
+ self._ephemeral_secret_file_names.pop(job_id, None)
1325
+ self._delete_job_secret_files(job_id)
1326
+ return self.get(job_id)
1327
+
1328
+ def adjust(
1329
+ self, job_id: str, request: SandboxAdjustmentRequest
1330
+ ) -> SandboxJobResponse:
1331
+ response = self.get(job_id)
1332
+ if response.status.stage.terminal:
1333
+ raise HTTPException(
1334
+ status_code=409, detail="a completed run cannot be adjusted"
1335
+ )
1336
+ if response.status.stage in {
1337
+ HarnessStage.CLEANING_UP,
1338
+ HarnessStage.UPLOADING_ARTIFACTS,
1339
+ }:
1340
+ raise HTTPException(
1341
+ status_code=409,
1342
+ detail="the run is already finalizing; start a follow-up run to change it",
1343
+ )
1344
+ instruction = request.instruction.strip()
1345
+ record = {
1346
+ "adjustment_id": str(uuid.uuid4()),
1347
+ "client_request_id": request.client_request_id,
1348
+ "instruction": instruction,
1349
+ "target_stage": _adjustment_stage(instruction, response.status.stage.value),
1350
+ "scenario_delta": _scenario_delta(instruction),
1351
+ "status": "pending",
1352
+ "created_at": _now(),
1353
+ }
1354
+ path = self.jobs_root / job_id / "adjustments.jsonl"
1355
+ with path.open("a", encoding="utf-8") as stream:
1356
+ stream.write(json.dumps(record, separators=(",", ":")) + "\n")
1357
+ stream.flush()
1358
+ return self.get(job_id)
1359
+
1360
+
1361
+ def _configuration_environment(config: dict[str, Any]) -> dict[str, str]:
1362
+ """Convert explicit, non-secret connector configuration into child env vars."""
1363
+ result: dict[str, str] = {}
1364
+ for name, value in config.items():
1365
+ if not re.fullmatch(r"[A-Z][A-Z0-9_]{2,}", str(name)):
1366
+ continue
1367
+ if isinstance(value, bool):
1368
+ result[str(name)] = "true" if value else "false"
1369
+ elif isinstance(value, (str, int, float)):
1370
+ result[str(name)] = str(value)
1371
+ return result
1372
+
1373
+
1374
+ _ENVIRONMENT_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
1375
+ _RUNNER_RESERVED_ENVIRONMENT = {
1376
+ "DOCKER_HOST",
1377
+ "FI_API_KEY",
1378
+ "FI_BASE_URL",
1379
+ "FI_SECRET_KEY",
1380
+ "HARNESS_PLATFORM_API_KEY",
1381
+ "HARNESS_PLATFORM_SECRET_KEY",
1382
+ "HARNESS_PLATFORM_URL",
1383
+ "HARNESS_WEBHOOK_HOST",
1384
+ "HARNESS_WEBHOOK_PORT",
1385
+ "HOME",
1386
+ "PATH",
1387
+ "PYTHONPATH",
1388
+ }
1389
+
1390
+
1391
+ def _validate_environment_values(
1392
+ values: dict[str, SecretStr], references: dict[str, SecretRef]
1393
+ ) -> None:
1394
+ if len(values) > 256:
1395
+ raise ValueError("environment_values_limit_exceeded")
1396
+ if set(values) & set(references):
1397
+ raise ValueError("environment_value_conflicts_with_secret_reference")
1398
+ total = 0
1399
+ for name, secret in values.items():
1400
+ if not _ENVIRONMENT_NAME.fullmatch(name):
1401
+ raise ValueError(f"environment_name_invalid: {name}")
1402
+ if name in _RUNNER_RESERVED_ENVIRONMENT or name.startswith("ALK_"):
1403
+ raise ValueError(f"environment_name_reserved: {name}")
1404
+ value = secret.get_secret_value()
1405
+ if "\x00" in value:
1406
+ raise ValueError(f"environment_value_contains_nul: {name}")
1407
+ total += len(name.encode()) + len(value.encode())
1408
+ if total > 262_144:
1409
+ raise ValueError("environment_values_size_exceeded")
1410
+
1411
+
1412
+ def _credential_scan_paths(packaging: PackagingManifest) -> list[str] | None:
1413
+ """Scope preflight credentials to the Compose runtime that will actually run."""
1414
+ if packaging.selected_kind is None or packaging.selected_kind.value != "compose":
1415
+ return None
1416
+ selected = next(
1417
+ (
1418
+ candidate
1419
+ for candidate in packaging.candidates
1420
+ if candidate.path == packaging.selected_path
1421
+ ),
1422
+ None,
1423
+ )
1424
+ if selected is None or packaging.selected_path is None:
1425
+ return None
1426
+ return [packaging.selected_path, *selected.runtime_source_roots]
1427
+
1428
+
1429
+ def _worker_failure_retryable(
1430
+ return_code: int, failure: dict[str, Any], retryable_domains: list[str]
1431
+ ) -> bool:
1432
+ """Retry infrastructure transport, never agent behavior or grading outcomes."""
1433
+ return (
1434
+ return_code != 0
1435
+ and failure.get("retryable") is True
1436
+ and failure.get("domain") in retryable_domains
1437
+ )
1438
+
1439
+
1440
+ def _allowed_source(raw: str) -> Path:
1441
+ source_text = raw
1442
+ for mapping in os.getenv("ALK_SANDBOX_PATH_MAP", "").split(os.pathsep):
1443
+ if "=" not in mapping:
1444
+ continue
1445
+ external, internal = mapping.split("=", 1)
1446
+ if source_text == external or source_text.startswith(
1447
+ external.rstrip("/") + "/"
1448
+ ):
1449
+ source_text = (
1450
+ internal.rstrip("/") + source_text[len(external.rstrip("/")) :]
1451
+ )
1452
+ break
1453
+ source = Path(source_text).expanduser().resolve()
1454
+ if not source.is_dir():
1455
+ raise HTTPException(
1456
+ status_code=400, detail="source_path must be an existing directory"
1457
+ )
1458
+ configured = os.getenv("ALK_SANDBOX_SOURCE_ROOTS")
1459
+ roots = [
1460
+ Path(item).expanduser().resolve()
1461
+ for item in (
1462
+ configured.split(os.pathsep) if configured else [str(Path.cwd().parent)]
1463
+ )
1464
+ if item
1465
+ ]
1466
+ if not any(source == root or root in source.parents for root in roots):
1467
+ raise HTTPException(
1468
+ status_code=403, detail="source_path is outside allowed roots"
1469
+ )
1470
+ return source
1471
+
1472
+
1473
+ def _safe_uploaded_path(raw: str) -> PurePosixPath:
1474
+ normalized = str(raw).replace("\\", "/").strip("/")
1475
+ relative = PurePosixPath(normalized)
1476
+ if (
1477
+ not normalized
1478
+ or len(normalized) > 1024
1479
+ or relative.is_absolute()
1480
+ or any(part in {"", ".", ".."} for part in relative.parts)
1481
+ or len(relative.parts) > 64
1482
+ ):
1483
+ raise HTTPException(status_code=400, detail=f"unsafe source path: {raw[:200]}")
1484
+ leaf = relative.name.lower()
1485
+ safe_environment_templates = {".env.example", ".env.sample", ".env.template"}
1486
+ if (
1487
+ leaf == ".env" or leaf.startswith(".env.")
1488
+ ) and leaf not in safe_environment_templates:
1489
+ raise HTTPException(
1490
+ status_code=400,
1491
+ detail=f"{relative.as_posix()} may contain secrets; upload it through the .env control",
1492
+ )
1493
+ return relative
1494
+
1495
+
1496
+ def _uploaded_source_name(source: Path) -> str:
1497
+ manifest = _read_json(source.parent / f"{source.name}.json")
1498
+ return str(manifest.get("name") or "uploaded-agent")
1499
+
1500
+
1501
+ def _github_location(raw: str):
1502
+ try:
1503
+ return parse_github_location(raw)
1504
+ except ValueError as exc:
1505
+ raise HTTPException(status_code=400, detail=str(exc)) from exc
1506
+
1507
+
1508
+ def _github_installation_token(installation_id: str) -> str:
1509
+ """Local broker adapter; production exchanges the installation via its vault."""
1510
+ safe_id = re.sub(r"[^A-Za-z0-9_]", "_", installation_id)
1511
+ token = os.getenv(f"GITHUB_INSTALLATION_{safe_id}_TOKEN")
1512
+ if not token:
1513
+ raise SourceAcquisitionError(
1514
+ "github_installation_token_unavailable; authorize the GitHub App installation"
1515
+ )
1516
+ return token
1517
+
1518
+
1519
+ def _events(path: Path) -> list[dict[str, Any]]:
1520
+ if not path.exists():
1521
+ return []
1522
+ result = []
1523
+ for line in path.read_text(encoding="utf-8").splitlines():
1524
+ try:
1525
+ result.append(json.loads(line))
1526
+ except ValueError:
1527
+ continue
1528
+ return result
1529
+
1530
+
1531
+ def _json_artifact(path: Path, *, max_bytes: int = 1_000_000) -> Any | None:
1532
+ """Read a bounded, generated JSON artifact for control-plane presentation."""
1533
+ try:
1534
+ if not path.is_file() or path.stat().st_size > max_bytes:
1535
+ return None
1536
+ return json.loads(path.read_text(encoding="utf-8"))
1537
+ except (OSError, ValueError):
1538
+ return None
1539
+
1540
+
1541
+ def _stage_outputs(root: Path) -> list[dict[str, Any]]:
1542
+ outputs: list[dict[str, Any]] = []
1543
+ contract = _json_artifact(root / "contract.json")
1544
+ if isinstance(contract, dict):
1545
+ outputs.append(
1546
+ {
1547
+ "id": "contract",
1548
+ "stage": HarnessStage.UNDERSTANDING_AGENT.value,
1549
+ "title": "Agent contract",
1550
+ "summary": (
1551
+ f"{len(contract.get('tools') or [])} tools, "
1552
+ f"{len(contract.get('hard_constraints') or [])} constraints and "
1553
+ f"{len(contract.get('real_use_cases') or [])} use cases"
1554
+ ),
1555
+ "kind": "contract",
1556
+ "data": _presentation_value(contract),
1557
+ "updated_at": datetime.fromtimestamp(
1558
+ (root / "contract.json").stat().st_mtime, timezone.utc
1559
+ ).isoformat(),
1560
+ }
1561
+ )
1562
+ environment = _json_artifact(root / "environment.json")
1563
+ if isinstance(environment, dict):
1564
+ visible = _presentation_value(
1565
+ {
1566
+ key: value
1567
+ for key, value in environment.items()
1568
+ if key
1569
+ not in {
1570
+ "source",
1571
+ "compose_file",
1572
+ "compose_override_file",
1573
+ "internal_overrides",
1574
+ "runtime_trace_path",
1575
+ "source_fingerprint",
1576
+ }
1577
+ }
1578
+ )
1579
+ outputs.append(
1580
+ {
1581
+ "id": "environment",
1582
+ "stage": HarnessStage.GENERATING_ENVIRONMENT.value,
1583
+ "title": "Execution environment",
1584
+ "summary": (
1585
+ f"{len(environment.get('services') or [])} services ready"
1586
+ + (
1587
+ f" in {environment.get('provision_seconds')}s"
1588
+ if environment.get("provision_seconds") is not None
1589
+ else ""
1590
+ )
1591
+ ),
1592
+ "kind": "environment",
1593
+ "data": visible,
1594
+ "updated_at": datetime.fromtimestamp(
1595
+ (root / "environment.json").stat().st_mtime, timezone.utc
1596
+ ).isoformat(),
1597
+ }
1598
+ )
1599
+ scenarios = _json_artifact(root / "scenarios.json")
1600
+ if isinstance(scenarios, list):
1601
+ outputs.append(
1602
+ {
1603
+ "id": "scenarios",
1604
+ "stage": HarnessStage.GENERATING_SCENARIOS.value,
1605
+ "title": "Generated scenarios",
1606
+ "summary": f"{len(scenarios)} grounded scenarios",
1607
+ "kind": "scenarios",
1608
+ "data": scenarios,
1609
+ "updated_at": datetime.fromtimestamp(
1610
+ (root / "scenarios.json").stat().st_mtime, timezone.utc
1611
+ ).isoformat(),
1612
+ }
1613
+ )
1614
+ results = _json_artifact(root / "results.json")
1615
+ if isinstance(results, dict):
1616
+ outputs.append(
1617
+ {
1618
+ "id": "results",
1619
+ "stage": HarnessStage.GRADING.value,
1620
+ "title": "Run results",
1621
+ "summary": "Scenario execution and grading evidence",
1622
+ "kind": "results",
1623
+ "data": results,
1624
+ "updated_at": datetime.fromtimestamp(
1625
+ (root / "results.json").stat().st_mtime, timezone.utc
1626
+ ).isoformat(),
1627
+ }
1628
+ )
1629
+ return outputs
1630
+
1631
+
1632
+ def _presentation_value(value: Any, key: str = "") -> Any:
1633
+ lowered = key.lower()
1634
+ if any(marker in lowered for marker in ("password", "secret", "token", "api_key")):
1635
+ return "[redacted]"
1636
+ if isinstance(value, dict):
1637
+ return {
1638
+ str(item_key): _presentation_value(item, str(item_key))
1639
+ for item_key, item in value.items()
1640
+ }
1641
+ if isinstance(value, list):
1642
+ return [_presentation_value(item, key) for item in value]
1643
+ if isinstance(value, str) and "://" in value and "@" in value:
1644
+ scheme, remainder = value.split("://", 1)
1645
+ # Prose can contain an email before a later URL. Only redact when the URI
1646
+ # remainder itself contains userinfo.
1647
+ if "@" in remainder:
1648
+ return f"{scheme}://[redacted]@{remainder.split('@', 1)[1]}"
1649
+ return value
1650
+
1651
+
1652
+ def _read_jsonl(path: Path) -> list[dict[str, Any]]:
1653
+ if not path.is_file():
1654
+ return []
1655
+ records: list[dict[str, Any]] = []
1656
+ for line in path.read_text(encoding="utf-8").splitlines():
1657
+ try:
1658
+ value = json.loads(line)
1659
+ except ValueError:
1660
+ continue
1661
+ if isinstance(value, dict):
1662
+ records.append(value)
1663
+ return records
1664
+
1665
+
1666
+ def _adjustments(directory: Path) -> list[dict[str, Any]]:
1667
+ requested = _read_jsonl(directory / "adjustments.jsonl")
1668
+ acknowledgements = {
1669
+ item.get("adjustment_id"): item
1670
+ for item in _read_jsonl(directory / "adjustment-status.jsonl")
1671
+ }
1672
+ return [
1673
+ {**item, **acknowledgements.get(item.get("adjustment_id"), {})}
1674
+ for item in requested
1675
+ ]
1676
+
1677
+
1678
+ def _scenario_delta(instruction: str) -> int | None:
1679
+ match = re.search(
1680
+ r"\b(?:add|create|generate|write)\s+(\d{1,3})\s+(?:more\s+)?scenarios?\b",
1681
+ instruction,
1682
+ re.IGNORECASE,
1683
+ )
1684
+ if not match:
1685
+ return None
1686
+ return min(100, int(match.group(1)))
1687
+
1688
+
1689
+ def _adjustment_stage(instruction: str, current_stage: str) -> str:
1690
+ lowered = instruction.lower()
1691
+ if any(word in lowered for word in ("scenario", "persona", "test case")):
1692
+ return "scenarios"
1693
+ if any(
1694
+ word in lowered
1695
+ for word in ("environment", "database", "service", "seed", "test data")
1696
+ ):
1697
+ return "environment"
1698
+ if any(word in lowered for word in ("contract", "tool", "capability")):
1699
+ return "understand"
1700
+ return {
1701
+ HarnessStage.UNDERSTANDING_AGENT.value: "understand",
1702
+ HarnessStage.GENERATING_ENVIRONMENT.value: "environment",
1703
+ HarnessStage.BUILDING_ENVIRONMENT.value: "environment",
1704
+ HarnessStage.VALIDATING_ENVIRONMENT.value: "environment",
1705
+ HarnessStage.GENERATING_SCENARIOS.value: "scenarios",
1706
+ HarnessStage.VALIDATING_SCENARIOS.value: "scenarios",
1707
+ }.get(current_stage, "scenarios")
1708
+
1709
+
1710
+ def _stage_from_events(events: list[dict[str, Any]]) -> str | None:
1711
+ for event in reversed(events):
1712
+ stage = event.get("payload", {}).get("stage")
1713
+ if stage:
1714
+ aliases = {
1715
+ "understand": HarnessStage.UNDERSTANDING_AGENT.value,
1716
+ "environment": HarnessStage.GENERATING_ENVIRONMENT.value,
1717
+ "scenarios": HarnessStage.GENERATING_SCENARIOS.value,
1718
+ "run": HarnessStage.RUNNING.value,
1719
+ "calls": HarnessStage.RUNNING.value,
1720
+ }
1721
+ return aliases.get(
1722
+ stage, stage if stage in HarnessStage._value2member_map_ else None
1723
+ )
1724
+ return None
1725
+
1726
+
1727
+ def _process_identity(pid: int) -> str:
1728
+ """Return a PID-reuse-safe Linux process identity for orphan ownership."""
1729
+ try:
1730
+ # Field 22 is process start time in clock ticks since boot. ``comm`` may contain spaces,
1731
+ # so split only after the final closing parenthesis instead of indexing the whole line.
1732
+ stat = Path(f"/proc/{pid}/stat").read_text(encoding="utf-8")
1733
+ fields = stat.rsplit(")", 1)[1].strip().split()
1734
+ start_ticks = fields[19]
1735
+ boot_id = (
1736
+ Path("/proc/sys/kernel/random/boot_id").read_text(encoding="utf-8").strip()
1737
+ )
1738
+ return f"{boot_id}:{pid}:{start_ticks}"
1739
+ except (OSError, IndexError, ValueError):
1740
+ return ""
1741
+
1742
+
1743
+ def _read_json(path: Path) -> dict[str, Any]:
1744
+ try:
1745
+ return json.loads(path.read_text(encoding="utf-8"))
1746
+ except (OSError, ValueError):
1747
+ return {}
1748
+
1749
+
1750
+ def _read_json_value(path: Path) -> Any:
1751
+ try:
1752
+ return json.loads(path.read_text(encoding="utf-8"))
1753
+ except (OSError, ValueError):
1754
+ return None
1755
+
1756
+
1757
+ def _stage_outputs(run_root: Path) -> list[dict[str, Any]]:
1758
+ """Return bounded, secret-safe snapshots for the platform's live run UI."""
1759
+ outputs: list[dict[str, Any]] = []
1760
+ contract = _read_json_value(run_root / "contract.json")
1761
+ if isinstance(contract, dict):
1762
+ tools = [
1763
+ {"name": str(tool.get("name") or "")}
1764
+ for tool in contract.get("tools", [])
1765
+ if isinstance(tool, dict) and tool.get("name")
1766
+ ]
1767
+ data = {
1768
+ "one_liner": str(contract.get("one_liner") or ""),
1769
+ "modality": str(contract.get("modality") or "unknown"),
1770
+ "runtime": contract.get("runtime") or {},
1771
+ "tools": tools,
1772
+ "hard_constraints": [
1773
+ str(item) for item in contract.get("hard_constraints", [])
1774
+ ],
1775
+ }
1776
+ outputs.append(
1777
+ {
1778
+ "id": "contract",
1779
+ "kind": "contract",
1780
+ "title": "Agent contract",
1781
+ "summary": f"{len(tools)} tools · {data['modality']} modality",
1782
+ "data": data,
1783
+ }
1784
+ )
1785
+
1786
+ environment = _read_json_value(run_root / "environment.json")
1787
+ if isinstance(environment, dict):
1788
+ services = [str(item) for item in environment.get("services", [])]
1789
+ # Endpoint values are useful operational feedback, but credentials embedded in a URI
1790
+ # must never reach the platform response.
1791
+ overrides = {
1792
+ str(name): re.sub(r"(://)[^/@]+@", r"\1***@", str(value))
1793
+ for name, value in dict(environment.get("overrides") or {}).items()
1794
+ }
1795
+ outputs.append(
1796
+ {
1797
+ "id": "environment",
1798
+ "kind": "environment",
1799
+ "title": "Execution environment",
1800
+ "summary": f"{len(services)} services ready",
1801
+ "data": {
1802
+ "services": services,
1803
+ "project": str(environment.get("project") or ""),
1804
+ "managed": bool(environment.get("managed")),
1805
+ "overrides": overrides,
1806
+ },
1807
+ }
1808
+ )
1809
+
1810
+ scenarios = _read_json_value(run_root / "scenarios.json")
1811
+ if isinstance(scenarios, list):
1812
+ data = [
1813
+ {
1814
+ "name": str(item.get("name") or "scenario"),
1815
+ "instruction": str(item.get("instruction") or ""),
1816
+ "use_case": str(item.get("use_case") or ""),
1817
+ }
1818
+ for item in scenarios
1819
+ if isinstance(item, dict)
1820
+ ]
1821
+ outputs.append(
1822
+ {
1823
+ "id": "scenarios",
1824
+ "kind": "scenarios",
1825
+ "title": "Generated scenarios",
1826
+ "summary": f"{len(data)} grounded scenarios",
1827
+ "data": data,
1828
+ }
1829
+ )
1830
+
1831
+ platform = _read_json_value(run_root / "platform.json")
1832
+ if isinstance(platform, dict):
1833
+ run_test_id = str(platform.get("run_test_id") or "").strip()
1834
+ execution_id = str(platform.get("test_execution_id") or "").strip()
1835
+ if run_test_id:
1836
+ url = (
1837
+ f"/dashboard/simulate/test/{run_test_id}/{execution_id}"
1838
+ if execution_id
1839
+ else f"/dashboard/simulate/test/{run_test_id}/runs"
1840
+ )
1841
+ outputs.append(
1842
+ {
1843
+ "id": "simulation",
1844
+ "kind": "simulation",
1845
+ "title": "Simulation results",
1846
+ "summary": "Open this run in the simulation view",
1847
+ "data": {
1848
+ "url": url,
1849
+ "run_test_id": run_test_id,
1850
+ "test_execution_id": execution_id,
1851
+ },
1852
+ }
1853
+ )
1854
+ return outputs
1855
+
1856
+
1857
+ def _write_json(path: Path, value: dict[str, Any]) -> None:
1858
+ temporary = path.with_suffix(".tmp")
1859
+ temporary.write_text(
1860
+ json.dumps(value, indent=2, default=str) + "\n", encoding="utf-8"
1861
+ )
1862
+ temporary.replace(path)
1863
+
1864
+
1865
+ def _now() -> str:
1866
+ return datetime.now(timezone.utc).isoformat()
1867
+
1868
+
1869
+ _TERMINAL_STAGES = {
1870
+ HarnessStage.COMPLETED.value,
1871
+ HarnessStage.FAILED.value,
1872
+ HarnessStage.CANCELED.value,
1873
+ }
1874
+
1875
+
1876
+ def _authorize(authorization: str | None = Header(default=None)) -> None:
1877
+ expected = os.getenv("ALK_SANDBOX_TOKEN")
1878
+ if expected and authorization != f"Bearer {expected}":
1879
+ raise HTTPException(
1880
+ status_code=status.HTTP_401_UNAUTHORIZED, detail="invalid token"
1881
+ )
1882
+
1883
+
1884
+ def create_app(root: Path | None = None) -> FastAPI:
1885
+ app = FastAPI(title="ALK local sandbox", version="1")
1886
+ sandbox = LocalSandbox(
1887
+ root or Path(os.getenv("ALK_SANDBOX_ROOT", "./artifacts/sandbox")),
1888
+ max_concurrency=int(os.getenv("ALK_SANDBOX_MAX_CONCURRENCY", "2")),
1889
+ )
1890
+
1891
+ @app.get("/health")
1892
+ async def health() -> dict[str, str]:
1893
+ return {"status": "ok", "provider": "local-process"}
1894
+
1895
+ @app.post(
1896
+ "/v1/sources",
1897
+ response_model=UploadedSourceResponse,
1898
+ dependencies=[Depends(_authorize)],
1899
+ status_code=status.HTTP_201_CREATED,
1900
+ )
1901
+ async def upload_source(
1902
+ files: list[UploadFile] = File(...),
1903
+ paths: list[str] = Form(...),
1904
+ name: str = Form(default="uploaded-agent"),
1905
+ ) -> UploadedSourceResponse:
1906
+ return await sandbox.upload_source(files, paths, name)
1907
+
1908
+ @app.post(
1909
+ "/v1/secret-files",
1910
+ response_model=UploadedSecretFileResponse,
1911
+ dependencies=[Depends(_authorize)],
1912
+ status_code=status.HTTP_201_CREATED,
1913
+ )
1914
+ async def upload_secret_file(
1915
+ file: UploadFile = File(...),
1916
+ environment_name: str = Form(...),
1917
+ ) -> UploadedSecretFileResponse:
1918
+ return await sandbox.upload_secret_file(file, environment_name)
1919
+
1920
+ @app.post(
1921
+ "/v1/preflight",
1922
+ response_model=SandboxPreflightResponse,
1923
+ dependencies=[Depends(_authorize)],
1924
+ )
1925
+ async def preflight(
1926
+ request: SandboxPreflightRequest,
1927
+ ) -> SandboxPreflightResponse:
1928
+ return sandbox.preflight(request)
1929
+
1930
+ @app.post(
1931
+ "/v1/jobs",
1932
+ response_model=SandboxJobResponse,
1933
+ dependencies=[Depends(_authorize)],
1934
+ )
1935
+ async def submit(request: LocalSandboxRequest) -> SandboxJobResponse:
1936
+ return sandbox.submit(request)
1937
+
1938
+ @app.get(
1939
+ "/v1/jobs",
1940
+ response_model=list[SandboxJobResponse],
1941
+ dependencies=[Depends(_authorize)],
1942
+ )
1943
+ async def list_jobs() -> list[SandboxJobResponse]:
1944
+ return sandbox.list()
1945
+
1946
+ @app.get(
1947
+ "/v1/jobs/{job_id}",
1948
+ response_model=SandboxJobResponse,
1949
+ dependencies=[Depends(_authorize)],
1950
+ )
1951
+ async def get_job(job_id: str) -> SandboxJobResponse:
1952
+ try:
1953
+ return sandbox.get(job_id)
1954
+ except KeyError as exc:
1955
+ raise HTTPException(status_code=404, detail="job not found") from exc
1956
+
1957
+ @app.post(
1958
+ "/v1/jobs/{job_id}/rerun",
1959
+ response_model=SandboxJobResponse,
1960
+ dependencies=[Depends(_authorize)],
1961
+ )
1962
+ async def rerun_job(
1963
+ job_id: str, request: SandboxRerunRequest
1964
+ ) -> SandboxJobResponse:
1965
+ try:
1966
+ return sandbox.rerun(job_id, request)
1967
+ except KeyError as exc:
1968
+ raise HTTPException(status_code=404, detail="job not found") from exc
1969
+
1970
+ @app.post(
1971
+ "/v1/jobs/{job_id}/cancel",
1972
+ response_model=SandboxJobResponse,
1973
+ dependencies=[Depends(_authorize)],
1974
+ )
1975
+ async def cancel_job(job_id: str) -> SandboxJobResponse:
1976
+ try:
1977
+ return await sandbox.cancel(job_id)
1978
+ except KeyError as exc:
1979
+ raise HTTPException(status_code=404, detail="job not found") from exc
1980
+
1981
+ @app.post(
1982
+ "/v1/jobs/{job_id}/adjust",
1983
+ response_model=SandboxJobResponse,
1984
+ dependencies=[Depends(_authorize)],
1985
+ )
1986
+ async def adjust_job(
1987
+ job_id: str, request: SandboxAdjustmentRequest
1988
+ ) -> SandboxJobResponse:
1989
+ try:
1990
+ return sandbox.adjust(job_id, request)
1991
+ except KeyError as exc:
1992
+ raise HTTPException(status_code=404, detail="job not found") from exc
1993
+
1994
+ return app
1995
+
1996
+
1997
+ app = create_app()
1998
+
1999
+
2000
+ def main() -> None:
2001
+ import uvicorn
2002
+
2003
+ uvicorn.run(
2004
+ "fi.alk.harness.sandbox_server:app",
2005
+ host=os.getenv("ALK_SANDBOX_HOST", "127.0.0.1"),
2006
+ port=int(os.getenv("ALK_SANDBOX_PORT", "8788")),
2007
+ )
2008
+
2009
+
2010
+ if __name__ == "__main__":
2011
+ main()