agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,189 @@
1
+ """Run the established ALK authoring stages for one hosted job.
2
+
3
+ This is intentionally a thin process boundary over :func:`fi.alk.harness.cli._auto`.
4
+ Contract creation, logical environment creation, and scenario generation therefore remain the
5
+ same implementation used by the local SDK and sandbox flows. Daytona consumes the frozen output
6
+ afterward; this command never executes scenarios itself.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import asyncio
13
+ import json
14
+ import os
15
+ from pathlib import Path
16
+ import tempfile
17
+
18
+ from .cli import _auto
19
+ from .job import HarnessJob, ProviderExecutionMode, SourceKind
20
+ from .provider_import import inspect_provider_target
21
+ from .scenarios import load as load_written
22
+ from .understand import PROVIDER_IMPORT_PROFILE_PATH_ENV
23
+
24
+
25
+ def _persist_authored_scenario_count(
26
+ job_path: Path, job: HarnessJob, output: Path
27
+ ) -> None:
28
+ """Keep the frozen job in sync with adjustments applied during authoring.
29
+
30
+ The control plane may increase ``scenario_count`` while this process is already
31
+ running. ``_auto`` sees that adjustment and writes the larger validated suite,
32
+ but the following Bundle V2 process reloads this on-disk job document. Without
33
+ reconciling it here the bundler copies the original number of scenarios and the
34
+ platform correctly rejects preallocation because its expected count is newer.
35
+ """
36
+ authored_count = len(load_written(output))
37
+ if authored_count <= 0 or authored_count == job.scenario_count:
38
+ return
39
+ updated = job.model_copy(update={"scenario_count": authored_count})
40
+ temporary = job_path.with_suffix(f"{job_path.suffix}.tmp")
41
+ temporary.write_text(
42
+ json.dumps(updated.model_dump(mode="json"), indent=2, sort_keys=True) + "\n",
43
+ encoding="utf-8",
44
+ )
45
+ temporary.replace(job_path)
46
+
47
+
48
+ def _load_provider_import_profile(
49
+ job: HarnessJob,
50
+ secrets_path: Path | None,
51
+ profile_cache_path: Path | None = None,
52
+ ) -> dict[str, object] | None:
53
+ inspect_connect_only_provider = (
54
+ job.agent.mode is ProviderExecutionMode.CONNECT_ONLY
55
+ and job.agent.connector.strip().lower() in {"vapi", "retell", "retell_chat"}
56
+ )
57
+ if (
58
+ job.agent.mode is not ProviderExecutionMode.PROVIDER_IMPORT
59
+ and not inspect_connect_only_provider
60
+ ):
61
+ return None
62
+ if profile_cache_path is not None and profile_cache_path.is_file():
63
+ cached = json.loads(profile_cache_path.read_text(encoding="utf-8"))
64
+ if not isinstance(cached, dict):
65
+ raise RuntimeError("provider_import_authoring_profile_invalid")
66
+ return cached
67
+ if secrets_path is None:
68
+ raise RuntimeError("provider_import_authoring_secrets_missing")
69
+ try:
70
+ values = json.loads(secrets_path.read_text(encoding="utf-8"))
71
+ finally:
72
+ # The provider credential is needed only for this read-only inspection. Remove the file
73
+ # before any model session or source/environment process starts.
74
+ secrets_path.unlink(missing_ok=True)
75
+ if not isinstance(values, dict):
76
+ raise RuntimeError("provider_import_authoring_secrets_invalid")
77
+ connector = job.agent.connector.strip().lower()
78
+ provider = "retell" if connector == "retell_chat" else connector
79
+ secret_name = "VAPI_API_KEY" if provider == "vapi" else "RETELL_API_KEY"
80
+ target_key = "assistant_id" if provider == "vapi" else "agent_id"
81
+ profile = inspect_provider_target(
82
+ provider,
83
+ source_target_id=str(job.agent.config.get(target_key) or ""),
84
+ api_key=str(values.get(secret_name) or ""),
85
+ api_base_url=str(job.agent.config.get("provider_api_base_url") or "") or None,
86
+ target_modality="chat" if connector == "retell_chat" else "voice",
87
+ )
88
+ if profile_cache_path is not None:
89
+ profile_cache_path.write_text(
90
+ json.dumps(profile, indent=2, sort_keys=True) + "\n", encoding="utf-8"
91
+ )
92
+ return profile
93
+
94
+
95
+ def main(argv: list[str] | None = None, *, validate_runtime: bool = False) -> int:
96
+ parser = argparse.ArgumentParser()
97
+ parser.add_argument("job", type=Path)
98
+ parser.add_argument("--source", type=Path, required=True)
99
+ parser.add_argument("--output", type=Path, required=True)
100
+ parser.add_argument(
101
+ "--adjustments",
102
+ type=Path,
103
+ help="JSONL inbox for user corrections applied at safe stage boundaries",
104
+ )
105
+ parser.add_argument(
106
+ "--target-secrets",
107
+ type=Path,
108
+ help="One-shot control-process secrets used to inspect an imported provider target",
109
+ )
110
+ parser.add_argument(
111
+ "--provider-profile-cache",
112
+ type=Path,
113
+ help="Control-owned sanitized profile reused across authoring retries",
114
+ )
115
+ args = parser.parse_args(argv)
116
+
117
+ job = HarnessJob.model_validate(json.loads(args.job.read_text(encoding="utf-8")))
118
+ profile = _load_provider_import_profile(
119
+ job, args.target_secrets, args.provider_profile_cache
120
+ )
121
+ # Transport kinds such as ``archive`` and ``github`` describe how the platform acquired the
122
+ # source. Once extracted, they are repositories. A source-free connect-only provider is the
123
+ # exception: its fetched definition is the source of truth and must never be represented by
124
+ # the intentionally empty /work/source directory.
125
+ source_free_provider = (
126
+ job.source.kind is SourceKind.PROVIDER
127
+ and job.agent.mode is ProviderExecutionMode.CONNECT_ONLY
128
+ and profile is not None
129
+ )
130
+ source_kind = (
131
+ "provider"
132
+ if source_free_provider
133
+ else str(job.metadata.get("source_kind") or "repo")
134
+ )
135
+ if source_kind not in {"repo", "spec", "provider"}:
136
+ source_kind = "repo"
137
+ namespace = argparse.Namespace(
138
+ path=str(args.source.resolve()),
139
+ name=str(job.metadata.get("agent_name") or args.source.name),
140
+ kind=source_kind,
141
+ out=str(args.output.resolve()),
142
+ count=job.scenario_count,
143
+ model=None,
144
+ run_model=None,
145
+ job=job,
146
+ adjustments_path=str(args.adjustments) if args.adjustments else None,
147
+ authoring_only=True,
148
+ provider_profile=profile if source_free_provider else None,
149
+ )
150
+ previous_profile_path = os.environ.get(PROVIDER_IMPORT_PROFILE_PATH_ENV)
151
+ with tempfile.TemporaryDirectory(prefix="alk-provider-profile-") as temporary:
152
+ if profile is not None:
153
+ profile_path = Path(temporary) / "provider-import-profile.json"
154
+ profile_path.write_text(
155
+ json.dumps(profile, indent=2, sort_keys=True) + "\n", encoding="utf-8"
156
+ )
157
+ os.environ[PROVIDER_IMPORT_PROFILE_PATH_ENV] = str(profile_path)
158
+ try:
159
+ status = asyncio.run(_auto(namespace))
160
+ if status == 0 and validate_runtime:
161
+ from .authoring_runtime_validation import validate_and_repair
162
+
163
+ _persist_authored_scenario_count(args.job, job, args.output.resolve())
164
+ runtime_job = HarnessJob.model_validate_json(args.job.read_text())
165
+ if profile is not None:
166
+ (args.output / "provider-import-profile.json").write_text(
167
+ json.dumps(profile) + "\n", encoding="utf-8"
168
+ )
169
+ asyncio.run(
170
+ validate_and_repair(
171
+ runtime_job, args.source.resolve(), args.output.resolve()
172
+ )
173
+ )
174
+ finally:
175
+ if previous_profile_path is None:
176
+ os.environ.pop(PROVIDER_IMPORT_PROFILE_PATH_ENV, None)
177
+ else:
178
+ os.environ[PROVIDER_IMPORT_PROFILE_PATH_ENV] = previous_profile_path
179
+ if status == 0:
180
+ _persist_authored_scenario_count(args.job, job, args.output.resolve())
181
+ if profile is not None:
182
+ (args.output.resolve() / "provider-import-profile.json").write_text(
183
+ json.dumps(profile, indent=2, sort_keys=True) + "\n", encoding="utf-8"
184
+ )
185
+ return status
186
+
187
+
188
+ if __name__ == "__main__":
189
+ raise SystemExit(main())
@@ -0,0 +1,267 @@
1
+ """Validate generated setup against the actual hosted runtime before accepting authoring.
2
+
3
+ This is setup proof, not a claim that a reference tool trajectory executed. Actual
4
+ agent tool execution remains evidence collected during the calls.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import asyncio
11
+ import json
12
+ import random
13
+ import tempfile
14
+ from concurrent.futures import ThreadPoolExecutor
15
+ from pathlib import Path
16
+
17
+
18
+ class RuntimeValidationError(RuntimeError):
19
+ def __init__(self, phase: str, detail: str):
20
+ self.phase = phase
21
+ super().__init__(detail)
22
+
23
+
24
+ async def validate_once(
25
+ job,
26
+ source: Path,
27
+ authoring: Path,
28
+ *,
29
+ secrets_path=Path("/run/futureagi/secrets.json"),
30
+ ) -> int:
31
+ from . import outbound
32
+ from .bundle_author_v2 import author_bundle_v2
33
+ from .hosted_entrypoint import (
34
+ ProcessWorldFactory,
35
+ _resolve_hosted_public_url,
36
+ job_secret_purposes,
37
+ )
38
+ from .hosted_scheduler import _classify_ready, _run_phase
39
+ from .job import ProviderExecutionMode, SourceKind
40
+ from .process_preflight import preflight_bundle
41
+ from .process_runtime import ProcessRuntimeProvider
42
+ from .scenario_source import load_scenarios
43
+ from .source_data_invariants import author_invariants, check_invariants
44
+
45
+ external_provider = (
46
+ getattr(getattr(job, "source", None), "kind", None) is SourceKind.PROVIDER
47
+ and getattr(getattr(job, "agent", None), "mode", None)
48
+ is ProviderExecutionMode.CONNECT_ONLY
49
+ )
50
+
51
+ # The real execution consumes its credential file. Validation gets a private copy,
52
+ # with the same purpose map, so it cannot destroy the execution handoff.
53
+ with tempfile.TemporaryDirectory(
54
+ prefix="runtime-validation-", dir=authoring.parent
55
+ ) as root:
56
+ work = Path(root)
57
+ secrets = work / "secrets.json"
58
+ secrets.write_bytes(secrets_path.read_bytes())
59
+ secrets.chmod(0o600)
60
+ secret_values = tuple(
61
+ str(value) for value in json.loads(secrets.read_text()).values()
62
+ )
63
+ capabilities = outbound.load_capabilities(unlink=False)
64
+ transport = outbound.RequestsTransport()
65
+ provider = ProcessRuntimeProvider(
66
+ secrets_path=secrets,
67
+ secret_purpose_map=job_secret_purposes(job),
68
+ user_resolver=lambda _name: None,
69
+ require_declared_user=False,
70
+ public_url_resolver=lambda port, ttl: _resolve_hosted_public_url(
71
+ capabilities, transport, port=port, expires_in_seconds=ttl
72
+ ),
73
+ provider_attempt_id=capabilities.attempt_id,
74
+ provider_expires_at=capabilities.expires_at,
75
+ )
76
+ executor = ThreadPoolExecutor(
77
+ max_workers=1, thread_name_prefix="runtime-validation"
78
+ )
79
+ phase = "environment"
80
+ try:
81
+ bundle = work / "bundle"
82
+ manifest = await asyncio.to_thread(
83
+ author_bundle_v2,
84
+ source=source,
85
+ job=job,
86
+ authoring=authoring,
87
+ output=bundle,
88
+ )
89
+ preflight_bundle(
90
+ bundle, manifest, parallelism=1, secret_refs=job_secret_purposes(job)
91
+ )
92
+ runtimes = await provider.provision(
93
+ manifest,
94
+ source=source,
95
+ bundle_dir=bundle,
96
+ work_directory=work,
97
+ instances=1,
98
+ )
99
+ factory = ProcessWorldFactory(work)
100
+ runtime = runtimes[0]
101
+ phase = "scenarios"
102
+ scenarios = await asyncio.to_thread(load_scenarios, bundle)
103
+ if len(scenarios) != job.scenario_count:
104
+ raise RuntimeValidationError(
105
+ phase, "Runtime scenario count differs from the requested count"
106
+ )
107
+
108
+ async def check_setups(invariants):
109
+ failures = []
110
+ for scenario in scenarios:
111
+ try:
112
+ await check_setup(scenario, invariants)
113
+ except RuntimeValidationError as exc:
114
+ failures.append(str(exc))
115
+ if failures:
116
+ raise RuntimeValidationError(phase, "\n".join(failures))
117
+
118
+ async def check_setup(scenario, invariants):
119
+ await provider.reset(runtime, work_directory=work)
120
+ world = await factory.create(runtime, rng=random.Random(job.seed or 0))
121
+ for name, fn, target, timeout in (
122
+ ("setup", scenario.setup, world, 30.0),
123
+ ("ready", scenario.ready, world.read_only(), 15.0),
124
+ ):
125
+ result = await _run_phase(
126
+ fn, target, timeout=timeout, phase=name, executor=executor
127
+ )
128
+ if result.failure:
129
+ raise RuntimeValidationError(
130
+ phase, f"{scenario.scenario_key}: {name}: {result.failure}"
131
+ )
132
+ if name == "setup":
133
+ try:
134
+ await check_invariants(
135
+ world.read_only(),
136
+ invariants,
137
+ scenario_key=scenario.scenario_key,
138
+ )
139
+ except Exception as exc:
140
+ raise RuntimeValidationError(
141
+ phase, f"{scenario.scenario_key}: source data: {exc}"
142
+ ) from exc
143
+ if name == "ready":
144
+ verdict = _classify_ready(result.value)
145
+ if verdict.broken or not verdict.held:
146
+ raise RuntimeValidationError(
147
+ phase,
148
+ f"{scenario.scenario_key}: ready precondition did not hold",
149
+ )
150
+
151
+ # Collect all executable setup errors before spending a model review or
152
+ # a repair attempt. Each scenario still gets an independent clean world.
153
+ await check_setups([])
154
+ if external_provider:
155
+ # A connect-only provider owns its state and executes its tools outside
156
+ # this sandbox. There is no harness-owned source database to probe or
157
+ # seed, so source-data invariant review would invent a local environment.
158
+ print(
159
+ "runtime validation: external provider black-box mode; "
160
+ "skipping local source-data invariant review",
161
+ flush=True,
162
+ )
163
+ return len(scenarios)
164
+ phase = "environment"
165
+ await provider.reset(runtime, work_directory=work)
166
+ baseline = await factory.create(runtime, rng=random.Random(job.seed or 0))
167
+ print("runtime validation: reviewing source data invariants", flush=True)
168
+ invariants = await author_invariants(
169
+ source, authoring, baseline.read_only(), endpoints=runtime.endpoints
170
+ )
171
+ # Review probes may have effects; none belongs in the test baseline.
172
+ await provider.reset(runtime, work_directory=work)
173
+ baseline = await factory.create(runtime, rng=random.Random(job.seed or 0))
174
+ await check_invariants(baseline.read_only(), invariants)
175
+ phase = "scenarios"
176
+ if invariants:
177
+ await check_setups(invariants)
178
+ return len(scenarios)
179
+ except RuntimeValidationError as exc:
180
+ raise RuntimeValidationError(
181
+ exc.phase,
182
+ outbound.redact_outbound_text(
183
+ str(exc), extra_secret_values=secret_values
184
+ ),
185
+ ) from None
186
+ except Exception as exc:
187
+ if "CERTIFICATE_VERIFY_FAILED" in str(exc):
188
+ # Generated data cannot repair the infrastructure trust store.
189
+ phase = "infrastructure"
190
+ raise RuntimeValidationError(
191
+ phase,
192
+ outbound.redact_outbound_text(
193
+ f"{type(exc).__name__}: {exc}", extra_secret_values=secret_values
194
+ ),
195
+ ) from None
196
+ finally:
197
+ executor.shutdown(wait=False, cancel_futures=True)
198
+ await provider.close(work_directory=work)
199
+
200
+
201
+ async def validate_and_repair(
202
+ job, source: Path, authoring: Path, *, validate=validate_once, repair=None
203
+ ) -> None:
204
+ """Two repairs per phase; environment repairs cannot exhaust setup's budget."""
205
+ if repair is None:
206
+
207
+ async def repair(phase, guidance):
208
+ from .cli import _build, _scenarios
209
+
210
+ if phase == "environment":
211
+ return await _build(
212
+ argparse.Namespace(
213
+ name=source.name,
214
+ path=str(source),
215
+ out=str(authoring),
216
+ interactive=False,
217
+ guidance=[guidance],
218
+ skip_source_provision=True,
219
+ external_runtime=(
220
+ job.source.kind.value == "provider"
221
+ and getattr(job.agent.mode, "value", None) == "connect_only"
222
+ ),
223
+ )
224
+ )
225
+ return await _scenarios(
226
+ argparse.Namespace(
227
+ name=source.name,
228
+ out=str(authoring),
229
+ count=job.scenario_count,
230
+ interactive=False,
231
+ guidance=[guidance],
232
+ )
233
+ )
234
+
235
+ repairs = {"environment": 0, "scenarios": 0}
236
+ for attempt in range(5):
237
+ print(f"runtime validation: attempt {attempt + 1}/5", flush=True)
238
+ try:
239
+ count = await validate(job, source, authoring)
240
+ except RuntimeValidationError as exc:
241
+ print(f"runtime validation: {exc.phase}: {exc}", flush=True)
242
+ if exc.phase not in repairs or repairs[exc.phase] >= 2:
243
+ raise
244
+ repairs[exc.phase] += 1
245
+ guidance = (
246
+ "Actual hosted runtime validation failed. Repair the generated environment/data "
247
+ "or scenario setup using the submitted source as authority. Do not modify source, "
248
+ "disable database constraints, weaken checks or ready conditions, drop scenarios, "
249
+ "or replace tools with invented implementations. Preserve scenario count and intent. "
250
+ f"Diagnostic: {str(exc)[:4000]}"
251
+ )
252
+ if await repair(exc.phase, guidance):
253
+ raise
254
+ else:
255
+ (authoring / "runtime-validation.json").write_text(
256
+ json.dumps(
257
+ {
258
+ "status": "passed",
259
+ "attempts": attempt + 1,
260
+ "setup_ready_scenarios": count,
261
+ "reference_tools_proven": False,
262
+ },
263
+ indent=2,
264
+ )
265
+ + "\n"
266
+ )
267
+ return
@@ -0,0 +1,43 @@
1
+ # Harness backends
2
+
3
+ A stage of this harness is a conversation: a system prompt, tools, a turn budget, and a loop
4
+ that feeds tool results back until the model stops. A backend is whoever runs that loop.
5
+
6
+ ```
7
+ ALK_HARNESS=claude # default; Claude Code loop, exactly the pre-seam behaviour
8
+ ALK_HARNESS=vertex-gemini # Google's ADK against Vertex (location=global)
9
+ ALK_HARNESS_MODEL=... # optional; unset means the backend's own default
10
+ ALK_VERTEX_LOCATION=... # vertex-gemini only; defaults to global (Gemini 3.x lives there)
11
+ ```
12
+
13
+ ## The seam
14
+
15
+ - `base.py` is the whole contract. `SessionSpec` is what a stage asks for; the reply
16
+ vocabulary (`SessionOpened`, `ModelReply`, `ToolReturned`, `StageDone`) is what a session
17
+ emits; `HarnessBackend` / `HarnessSession` are the two protocols a backend implements.
18
+ - Tools are declared once, neutrally (`tool`, `tool_server`), and each backend adapts them:
19
+ the Claude backend builds an in-process MCP server, the Gemini backend builds function
20
+ declarations. Gating follows the same split: hook-based on Claude, structural on Gemini
21
+ (an ungranted tool is never declared).
22
+ - `Stage` (in `session.py`) drives any backend and renders the replies into events. It never
23
+ names a vendor.
24
+
25
+ ## Adding a backend
26
+
27
+ Write a module with a class exposing `name`, `default_model`, `can_drive(model)` and
28
+ `create(spec) -> HarnessSession`, where the session yields the neutral replies and ends every
29
+ exchange with a `StageDone`. Register it in `__init__.py` (or call `register` from anywhere).
30
+ Nothing else in the harness changes: every stage, gate, and artifact works as-is. That is the
31
+ slot a Bedrock, Azure, or Gemini-CLI backend drops into.
32
+
33
+ Backends load lazily, so one backend's SDK is never imported because a different one ran.
34
+
35
+ ## What the Gemini backend supplies itself
36
+
37
+ The ADK owns the loop, tool execution, and session history; the backend only adapts a
38
+ ``ToolSpec`` through ADK's ``BaseTool`` extension point and translates its event stream into
39
+ the neutral replies. Claude Code ships Read/Glob/Grep and an operator-question tool; ADK has
40
+ no coding-CLI file tools, so `files.py` implements the read-only file tools once for any
41
+ backend that needs them. AskUserQuestion is deliberately not declared on the Gemini backend
42
+ yet: unattended runs never call it, and declaring a tool the backend cannot answer would cost
43
+ the model a turn finding that out.
@@ -0,0 +1,122 @@
1
+ """Harness backends, selected by name.
2
+
3
+ ``ALK_HARNESS`` picks the backend the way ``ALK_HARNESS_MODEL`` already picks the model. With
4
+ nothing set the choice is ``vertex-gemini``, so a machine holding only Google credentials runs
5
+ without being told to; ``ALK_HARNESS=claude`` selects the Claude Code loop instead.
6
+
7
+ Backends load lazily: choosing one never imports the other's SDK, so a deployment installs only
8
+ the provider it uses. A new backend is a module implementing ``HarnessBackend`` plus one
9
+ ``register`` call, from anywhere; nothing else in the harness changes.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import os
15
+ from typing import Callable
16
+
17
+ from .base import (
18
+ ASK_TOOL,
19
+ FILE_TOOLS,
20
+ KNOWN_BUILTINS,
21
+ Call,
22
+ HarnessBackend,
23
+ HarnessSession,
24
+ ModelReply,
25
+ Say,
26
+ SessionOpened,
27
+ SessionSpec,
28
+ StageDone,
29
+ ToolReturned,
30
+ ToolServer,
31
+ ToolSpec,
32
+ qualified,
33
+ tool,
34
+ tool_server,
35
+ )
36
+
37
+ __all__ = [
38
+ "ASK_TOOL",
39
+ "FILE_TOOLS",
40
+ "KNOWN_BUILTINS",
41
+ "Call",
42
+ "HarnessBackend",
43
+ "HarnessSession",
44
+ "ModelReply",
45
+ "Say",
46
+ "SessionOpened",
47
+ "SessionSpec",
48
+ "StageDone",
49
+ "ToolReturned",
50
+ "ToolServer",
51
+ "ToolSpec",
52
+ "qualified",
53
+ "tool",
54
+ "tool_server",
55
+ "register",
56
+ "resolve",
57
+ "backend_names",
58
+ ]
59
+
60
+ DEFAULT_BACKEND = "vertex-gemini"
61
+
62
+ _LOADERS: dict[str, Callable[[], HarnessBackend]] = {}
63
+ _ALIASES = {
64
+ "gemini": "vertex-gemini",
65
+ "vertex_gemini": "vertex-gemini",
66
+ "vertexai-gemini": "vertex-gemini",
67
+ "claude-code": "claude",
68
+ }
69
+ _LIVE: dict[str, HarnessBackend] = {}
70
+
71
+
72
+ def register(name: str, loader: Callable[[], HarnessBackend]) -> None:
73
+ """Make a backend selectable by name. Loader runs on first use, not at registration."""
74
+ _LOADERS[name] = loader
75
+
76
+
77
+ def _load_claude() -> HarnessBackend:
78
+ from .claude import ClaudeBackend
79
+
80
+ return ClaudeBackend()
81
+
82
+
83
+ def _load_vertex_gemini() -> HarnessBackend:
84
+ from .vertex_gemini import VertexGeminiBackend
85
+
86
+ return VertexGeminiBackend()
87
+
88
+
89
+ register("claude", _load_claude)
90
+ register("vertex-gemini", _load_vertex_gemini)
91
+
92
+
93
+ def backend_names() -> list[str]:
94
+ return sorted(_LOADERS)
95
+
96
+
97
+ def resolve(name: str | None = None) -> HarnessBackend:
98
+ """The backend a run will use: the one named, or ALK_HARNESS, or the default.
99
+
100
+ An unknown name is a loud error naming what exists. Falling back silently would run a whole
101
+ suite on the wrong harness, which is only discovered from the bill.
102
+ """
103
+ asked = (name or os.environ.get("ALK_HARNESS") or DEFAULT_BACKEND).strip().lower()
104
+ asked = _ALIASES.get(asked, asked)
105
+ if asked not in _LOADERS:
106
+ raise ValueError(
107
+ f"no harness backend named {asked!r}; installed backends: "
108
+ f"{', '.join(backend_names())}"
109
+ )
110
+ if asked not in _LIVE:
111
+ _LIVE[asked] = _LOADERS[asked]()
112
+ backend = _LIVE[asked]
113
+ # A named model that this backend cannot reach is a configuration mistake, and it is only
114
+ # visible here. Left to run, the provider rejects the model mid-stage and the failure reads
115
+ # as the harness having nothing to say rather than as the wrong pairing.
116
+ wanted = os.environ.get("ALK_HARNESS_MODEL", "").strip()
117
+ if wanted and not backend.can_drive(wanted):
118
+ raise ValueError(
119
+ f"harness backend {backend.name!r} cannot drive model {wanted!r}; "
120
+ f"its default is {backend.default_model!r}"
121
+ )
122
+ return backend