agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/alk/harness/job.py ADDED
@@ -0,0 +1,426 @@
1
+ """Versioned job and runtime contracts for the complete ALK harness workflow.
2
+
3
+ These contracts are control-plane neutral. A local CLI creates the same ``HarnessJob`` as the
4
+ platform hosted-job API. The platform may enqueue it, but only an ALK executor interprets its
5
+ stages.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from datetime import datetime
11
+ from enum import Enum
12
+ from typing import Protocol
13
+
14
+ from pydantic import BaseModel, Field, JsonValue, model_validator
15
+
16
+ from fi.simulate.runtime.spec import RuntimeIsolation, RuntimeRequirements, SecretRef
17
+
18
+ from .github import parse_github_location
19
+
20
+ HARNESS_JOB_SCHEMA_VERSION = "futureagi.harness-job.v1"
21
+ MAX_HOSTED_SCENARIO_COUNT = 200
22
+
23
+
24
+ class ExecutionMode(str, Enum):
25
+ LOCAL = "local"
26
+ HOSTED = "hosted"
27
+
28
+
29
+ class SourceKind(str, Enum):
30
+ LOCAL_REPOSITORY = "local_repository"
31
+ GITHUB = "github"
32
+ ARCHIVE = "archive"
33
+ IMAGE = "image"
34
+ REMOTE = "remote"
35
+ PROVIDER = "provider"
36
+
37
+
38
+ class SourceVisibility(str, Enum):
39
+ PRIVATE = "private"
40
+ PUBLIC = "public"
41
+
42
+
43
+ class RepositorySource(BaseModel):
44
+ kind: SourceKind
45
+ local_path: str | None = None
46
+ installation_id: str | None = None
47
+ repository: str | None = None
48
+ ref: str | None = None
49
+ commit_sha: str | None = None
50
+ visibility: SourceVisibility = SourceVisibility.PRIVATE
51
+ archive_artifact_id: str | None = None
52
+ image: str | None = None
53
+ endpoint: str | None = None
54
+
55
+ @model_validator(mode="after")
56
+ def _required_locator(self) -> "RepositorySource":
57
+ # A connected provider agent is itself the source of truth. Its ID lives
58
+ # in AgentConnection.config and its definition is fetched with the
59
+ # run-scoped provider credential during authoring, so no repository
60
+ # locator is required.
61
+ if self.kind is SourceKind.PROVIDER:
62
+ return self
63
+ if self.kind is SourceKind.GITHUB and self.repository:
64
+ location = parse_github_location(self.repository)
65
+ if self.ref and location.ref and self.ref != location.ref:
66
+ raise ValueError("github_ref_conflicts_with_repository_url")
67
+ object.__setattr__(self, "repository", location.repository)
68
+ object.__setattr__(self, "ref", location.ref or self.ref)
69
+ required = {
70
+ SourceKind.LOCAL_REPOSITORY: self.local_path,
71
+ SourceKind.GITHUB: self.repository
72
+ and (self.visibility is SourceVisibility.PUBLIC or self.installation_id),
73
+ SourceKind.ARCHIVE: self.archive_artifact_id,
74
+ SourceKind.IMAGE: self.image,
75
+ SourceKind.REMOTE: self.endpoint,
76
+ }[self.kind]
77
+ if not required:
78
+ raise ValueError(f"source_locator_missing: {self.kind.value}")
79
+ if self.kind is SourceKind.GITHUB and self.commit_sha:
80
+ normalized = self.commit_sha.lower()
81
+ if len(normalized) != 40 or any(
82
+ char not in "0123456789abcdef" for char in normalized
83
+ ):
84
+ raise ValueError("github_commit_sha_invalid")
85
+ return self
86
+
87
+
88
+ class AgentConnection(BaseModel):
89
+ connector: str
90
+ mode: "ProviderExecutionMode | None" = None
91
+ config: dict[str, JsonValue] = Field(default_factory=dict)
92
+ secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
93
+
94
+ @model_validator(mode="after")
95
+ def _provider_mode_is_explicit_and_safe(self) -> "AgentConnection":
96
+ connector = self.connector.strip().lower()
97
+ provider_connector = "retell" if connector == "retell_chat" else connector
98
+ if self.mode is ProviderExecutionMode.ENVIRONMENT_BACKED:
99
+ if connector == "retell_chat":
100
+ raise ValueError("retell_chat_environment_backed_not_supported")
101
+ if provider_connector not in {"vapi", "retell"}:
102
+ raise ValueError("environment_backed_requires_vapi_or_retell")
103
+ manifest = str(self.config.get("lifecycle_manifest") or "alk.yaml")
104
+ if manifest.startswith("/") or ".." in manifest.split("/"):
105
+ raise ValueError("provider_lifecycle_manifest_must_be_source_relative")
106
+ forbidden = {"assistant_id", "agent_id"} & set(self.config)
107
+ if forbidden:
108
+ raise ValueError(
109
+ "environment_backed_target_is_provision_output: "
110
+ + ", ".join(sorted(forbidden))
111
+ )
112
+ elif self.mode is ProviderExecutionMode.PROVIDER_IMPORT:
113
+ if provider_connector not in {"vapi", "retell"}:
114
+ raise ValueError("provider_import_requires_vapi_or_retell")
115
+ target_key = {"vapi": "assistant_id", "retell": "agent_id"}[
116
+ provider_connector
117
+ ]
118
+ if not str(self.config.get(target_key) or "").strip():
119
+ raise ValueError(f"provider_import_requires_{target_key}")
120
+ for path_key in ("event_path", "tool_path"):
121
+ value = str(self.config.get(path_key) or "")
122
+ if value and (
123
+ not value.startswith("/")
124
+ or value.startswith("//")
125
+ or ".." in value.split("/")
126
+ ):
127
+ raise ValueError(f"provider_import_{path_key}_invalid")
128
+ elif self.mode is ProviderExecutionMode.CONNECT_ONLY:
129
+ target_key = {"vapi": "assistant_id", "retell": "agent_id"}.get(
130
+ provider_connector
131
+ )
132
+ if target_key and not str(self.config.get(target_key) or "").strip():
133
+ raise ValueError(f"connect_only_requires_{target_key}")
134
+ elif self.mode is not None and provider_connector not in {"vapi", "retell"}:
135
+ raise ValueError("provider_mode_only_supported_for_vapi_or_retell")
136
+ return self
137
+
138
+
139
+ class ProviderExecutionMode(str, Enum):
140
+ """How a provider-hosted target enters a harness run.
141
+
142
+ ``None`` on :class:`AgentConnection` preserves existing LiveKit/HTTP jobs and legacy
143
+ connector-only payloads. New Vapi/Retell submissions must choose one of these modes.
144
+ """
145
+
146
+ CONNECT_ONLY = "connect_only"
147
+ ENVIRONMENT_BACKED = "environment_backed"
148
+ PROVIDER_IMPORT = "provider_import"
149
+
150
+
151
+ class ArtifactLevel(str, Enum):
152
+ METADATA_ONLY = "metadata-only"
153
+ TRACES = "traces"
154
+ TRACES_AND_RECORDINGS = "traces-and-recordings"
155
+ FULL = "full"
156
+ LOCAL_ONLY = "local-only"
157
+
158
+
159
+ class HarnessArtifactPolicy(BaseModel):
160
+ level: ArtifactLevel = ArtifactLevel.TRACES
161
+ retention_days: int | None = Field(default=30, ge=1, le=3650)
162
+ allow_bundle_download: bool = False
163
+ max_artifact_bytes: int = Field(default=1_073_741_824, ge=0)
164
+
165
+
166
+ class SandboxSecurityPolicy(BaseModel):
167
+ """Security invariants a provider must satisfy before executing customer code."""
168
+
169
+ untrusted_source: bool = True
170
+ read_only_source: bool = True
171
+ allow_privileged: bool = False
172
+ allow_host_runtime_control: bool = False
173
+ allowed_egress_domains: list[str] = Field(default_factory=list)
174
+
175
+
176
+ class HarnessRetryPolicy(BaseModel):
177
+ """Conservative job retry policy; agent behavior is intentionally absent."""
178
+
179
+ max_infrastructure_attempts: int = Field(default=2, ge=1, le=5)
180
+ initial_backoff_seconds: float = Field(default=1.0, ge=0, le=60)
181
+ max_backoff_seconds: float = Field(default=15.0, ge=0, le=300)
182
+ retryable_domains: list[str] = Field(
183
+ default_factory=lambda: ["infrastructure", "connectivity"]
184
+ )
185
+
186
+ @model_validator(mode="after")
187
+ def _valid_retry_policy(self) -> HarnessRetryPolicy:
188
+ allowed = {"infrastructure", "connectivity", "platform_sync"}
189
+ unsupported = set(self.retryable_domains) - allowed
190
+ if unsupported:
191
+ raise ValueError("retry_domain_unsafe: " + ", ".join(sorted(unsupported)))
192
+ if self.max_backoff_seconds < self.initial_backoff_seconds:
193
+ raise ValueError("retry_backoff_invalid")
194
+ return self
195
+
196
+
197
+ class HarnessJob(BaseModel):
198
+ schema_version: str = HARNESS_JOB_SCHEMA_VERSION
199
+ job_id: str
200
+ run_id: str
201
+ execution: ExecutionMode
202
+ source: RepositorySource
203
+ agent: AgentConnection
204
+ scenario_count: int = Field(default=10, ge=1, le=1000)
205
+ seed: int | None = None
206
+ runtime: RuntimeRequirements = Field(default_factory=RuntimeRequirements)
207
+ security: SandboxSecurityPolicy = Field(default_factory=SandboxSecurityPolicy)
208
+ retry: HarnessRetryPolicy = Field(default_factory=HarnessRetryPolicy)
209
+ artifacts: HarnessArtifactPolicy = Field(default_factory=HarnessArtifactPolicy)
210
+ platform_run_id: str | None = None
211
+ metadata: dict[str, JsonValue] = Field(default_factory=dict)
212
+
213
+ @model_validator(mode="after")
214
+ def _validate_job(self) -> HarnessJob:
215
+ if self.schema_version != HARNESS_JOB_SCHEMA_VERSION:
216
+ raise ValueError(f"harness_job_version_unsupported: {self.schema_version}")
217
+ if (
218
+ self.execution is ExecutionMode.LOCAL
219
+ and self.source.kind is SourceKind.GITHUB
220
+ and (
221
+ self.source.visibility is not SourceVisibility.PUBLIC
222
+ or self.source.installation_id
223
+ )
224
+ ):
225
+ # A local runner may clone public source, but it must never receive a platform
226
+ # GitHub installation credential. Private source is acquired by a hosted provider.
227
+ raise ValueError("local_job_cannot_use_private_platform_github_source")
228
+ if (
229
+ self.execution is ExecutionMode.HOSTED
230
+ and self.source.kind is SourceKind.LOCAL_REPOSITORY
231
+ ):
232
+ raise ValueError("hosted_job_cannot_use_local_path")
233
+ if self.execution is ExecutionMode.HOSTED:
234
+ if self.source.kind is SourceKind.IMAGE:
235
+ raise ValueError("image_source_not_hosted")
236
+ if self.source.kind is SourceKind.GITHUB and not self.source.commit_sha:
237
+ raise ValueError("github_commit_sha_required")
238
+ if self.scenario_count > MAX_HOSTED_SCENARIO_COUNT:
239
+ raise ValueError("hosted_scenario_count_out_of_range")
240
+ if self.runtime.isolation is not RuntimeIsolation.DEDICATED_VM:
241
+ raise ValueError("hosted_isolation_must_be_dedicated_vm")
242
+ if self.runtime.parallelism > self.runtime.cpu_units:
243
+ raise ValueError("hosted_parallelism_exceeds_cpu")
244
+ if self.artifacts.level is ArtifactLevel.LOCAL_ONLY:
245
+ raise ValueError("local_only_not_hosted")
246
+ for alias, reference in self.agent.secret_refs.items():
247
+ expected_manager = (
248
+ "platform-config"
249
+ if reference.purpose == "simulator_provider"
250
+ else "platform-vault"
251
+ )
252
+ if reference.manager != expected_manager:
253
+ raise ValueError(f"hosted_secret_manager_unsupported: {alias}")
254
+ if reference.purpose not in {
255
+ "target_provider",
256
+ "simulator_provider",
257
+ }:
258
+ raise ValueError(f"hosted_secret_purpose_invalid: {alias}")
259
+ if self.security.allow_privileged:
260
+ raise ValueError("hosted_privileged_execution_forbidden")
261
+ if self.security.allow_host_runtime_control:
262
+ raise ValueError("hosted_runtime_control_forbidden")
263
+ if not self.security.read_only_source:
264
+ raise ValueError("hosted_source_must_be_read_only")
265
+ _reject_secret_fields(self.model_dump(exclude={"agent": {"secret_refs"}}))
266
+ return self
267
+
268
+
269
+ class HarnessStage(str, Enum):
270
+ QUEUED = "queued"
271
+ ACQUIRING_SOURCE = "acquiring_source"
272
+ UNDERSTANDING_AGENT = "understanding_agent"
273
+ GENERATING_ENVIRONMENT = "generating_environment"
274
+ BUILDING_ENVIRONMENT = "building_environment"
275
+ VALIDATING_ENVIRONMENT = "validating_environment"
276
+ GENERATING_DATA = "generating_data"
277
+ GENERATING_SCENARIOS = "generating_scenarios"
278
+ VALIDATING_SCENARIOS = "validating_scenarios"
279
+ CONNECTING_AGENT = "connecting_agent"
280
+ RUNNING = "running"
281
+ GRADING = "grading"
282
+ UPLOADING_ARTIFACTS = "uploading_artifacts"
283
+ CLEANING_UP = "cleaning_up"
284
+ COMPLETED = "completed"
285
+ FAILED = "failed"
286
+ CANCELED = "canceled"
287
+
288
+ @property
289
+ def terminal(self) -> bool:
290
+ return self in {self.COMPLETED, self.FAILED, self.CANCELED}
291
+
292
+
293
+ class FailureDomain(str, Enum):
294
+ AGENT = "agent"
295
+ SIMULATOR = "simulator"
296
+ ENVIRONMENT = "environment"
297
+ CONNECTIVITY = "connectivity"
298
+ INFRASTRUCTURE = "infrastructure"
299
+ GRADING = "grading"
300
+ ARTIFACT = "artifact"
301
+ PLATFORM_SYNC = "platform_sync"
302
+
303
+
304
+ class FailureOwner(str, Enum):
305
+ CUSTOMER_AGENT = "customer_agent"
306
+ FUTUREAGI = "futureagi"
307
+
308
+
309
+ class CustomerFailure(BaseModel):
310
+ """Small, ownership-correct failure surface safe for customer-facing APIs."""
311
+
312
+ category: str
313
+ owner: FailureOwner
314
+ code: str
315
+ message: str
316
+ retryable: bool = False
317
+
318
+
319
+ class HarnessFailure(BaseModel):
320
+ domain: FailureDomain
321
+ stage: HarnessStage
322
+ code: str
323
+ message: str
324
+ retryable: bool = False
325
+ details: dict[str, JsonValue] = Field(default_factory=dict)
326
+
327
+ @property
328
+ def owner(self) -> FailureOwner:
329
+ return (
330
+ FailureOwner.CUSTOMER_AGENT
331
+ if self.domain is FailureDomain.AGENT
332
+ else FailureOwner.FUTUREAGI
333
+ )
334
+
335
+ def for_customer(self) -> CustomerFailure:
336
+ """Do not present a harness subsystem failure as a customer-agent failure."""
337
+ if self.owner is FailureOwner.CUSTOMER_AGENT:
338
+ return CustomerFailure(
339
+ category="agent_failure",
340
+ owner=self.owner,
341
+ code=self.code,
342
+ message=self.message,
343
+ retryable=False,
344
+ )
345
+ return CustomerFailure(
346
+ category="system_failure",
347
+ owner=self.owner,
348
+ code=self.code,
349
+ message=(
350
+ "FutureAGI could not complete this run. The failure has been recorded "
351
+ "for diagnosis; the submitted agent has not been marked as failed."
352
+ ),
353
+ retryable=self.retryable,
354
+ )
355
+
356
+
357
+ class HarnessJobStatus(BaseModel):
358
+ job_id: str
359
+ run_id: str
360
+ stage: HarnessStage
361
+ updated_at: datetime
362
+ detail: str | None = None
363
+ failure: HarnessFailure | None = None
364
+ environment_digest: str | None = None
365
+ scenario_set_digest: str | None = None
366
+ completed_scenarios: int = Field(default=0, ge=0)
367
+ total_scenarios: int = Field(default=0, ge=0)
368
+ attempt: int = Field(default=1, ge=1)
369
+
370
+
371
+ class HostedHarnessPort(Protocol):
372
+ """Scheduler-neutral interface implemented by the hosted ALK sandbox fleet."""
373
+
374
+ async def submit(self, job: HarnessJob) -> HarnessJobStatus: ...
375
+
376
+ async def status(self, job_id: str) -> HarnessJobStatus: ...
377
+
378
+ async def cancel(self, job_id: str) -> None: ...
379
+
380
+
381
+ _SECRET_NAMES = {
382
+ "api_key",
383
+ "api_secret",
384
+ "authorization",
385
+ "credential",
386
+ "credentials",
387
+ "password",
388
+ "private_key",
389
+ "secret",
390
+ "token",
391
+ }
392
+
393
+
394
+ def _reject_secret_fields(value: object, path: tuple[str, ...] = ()) -> None:
395
+ if isinstance(value, dict):
396
+ for key, item in value.items():
397
+ normalized = str(key).lower().replace("-", "_")
398
+ current = (*path, str(key))
399
+ if normalized in _SECRET_NAMES and item not in (None, "", {}, []):
400
+ raise ValueError("resolved_secret_forbidden: " + ".".join(current))
401
+ _reject_secret_fields(item, current)
402
+ elif isinstance(value, list):
403
+ for index, item in enumerate(value):
404
+ _reject_secret_fields(item, (*path, str(index)))
405
+
406
+
407
+ __all__ = [
408
+ "HARNESS_JOB_SCHEMA_VERSION",
409
+ "AgentConnection",
410
+ "ArtifactLevel",
411
+ "ExecutionMode",
412
+ "CustomerFailure",
413
+ "FailureDomain",
414
+ "FailureOwner",
415
+ "HarnessArtifactPolicy",
416
+ "HarnessFailure",
417
+ "HarnessJob",
418
+ "HarnessJobStatus",
419
+ "HarnessRetryPolicy",
420
+ "HarnessStage",
421
+ "HostedHarnessPort",
422
+ "RepositorySource",
423
+ "SandboxSecurityPolicy",
424
+ "SourceKind",
425
+ "SourceVisibility",
426
+ ]
@@ -0,0 +1,184 @@
1
+ """Deciding a judged sub-goal against the world the session left behind.
2
+
3
+ A judged sub-goal is one nothing observable settles, so a model reaches the verdict. Until now
4
+ nothing did: judged sub-goals carried a placeholder check returning None, which the scheduler read
5
+ as "held", so every one passed before anything looked. This runs in the sandbox at the end of the
6
+ session, while the world is still alive, so the judge reads real state rather than a snapshot.
7
+
8
+ Nothing here is modality-specific. It reasons over the world's tables, the actions the agent
9
+ took and what was said, which a voice call, a typed conversation and a browser the agent drives all
10
+ leave behind in the same shape, so the wording stays neutral rather than naming a call.
11
+
12
+ A judge that cannot decide returns held None: the unjudged path the platform already skips, because
13
+ a judge that failed to run is not evidence against the agent.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import json
19
+ import os
20
+ from typing import Any, Sequence
21
+
22
+ from .backends import SessionSpec, tool, tool_server
23
+ from .config import chosen_model
24
+ from .session import Stage
25
+ from .tools import schema
26
+
27
+ JUDGE_MODEL_ALIAS = "ALK_JUDGE_MODEL"
28
+ _LIMIT = 600
29
+ _TRANSCRIPT_LIMIT = 24000
30
+
31
+ _INSTRUCTIONS = """
32
+ You decide one claim about a session that already happened, against the world it left behind.
33
+
34
+ Look before you answer: read the actions you were given, inspect the tables the claim touches, query
35
+ for a specific row when the claim is about one, and read the transcript when the claim is about what
36
+ was said. The world is the run's final state. Its SQL dialect is PostgreSQL, not SQLite. Use
37
+ inspect_world to discover tables; do not query sqlite_master or make up a schema. A claim about
38
+ wording or about what the agent told the caller is settled by read_transcript: do not call it
39
+ undecided before reading it.
40
+
41
+ Then call decide, once. `passed` true when the claim holds, false when it does not, and an
42
+ explanation citing what you saw: a value, a row, an action and its arguments. An explanation that
43
+ only restates the claim is not a verdict. If you genuinely cannot tell, pass undecided true rather
44
+ than guess. Judge only the claim you were given.
45
+ """.strip()
46
+
47
+
48
+ def judge_model() -> str:
49
+ """The judge's model. Set ALK_JUDGE_MODEL to run it on something other than the harness model."""
50
+ return os.environ.get(JUDGE_MODEL_ALIAS, "").strip() or chosen_model()
51
+
52
+
53
+ def _short(value: object) -> object:
54
+ if isinstance(value, str) and len(value) > _LIMIT:
55
+ return value[:_LIMIT] + f"... [{len(value)} chars]"
56
+ if isinstance(value, dict):
57
+ return {key: _short(item) for key, item in value.items()}
58
+ if isinstance(value, list):
59
+ return [_short(item) for item in value[:20]]
60
+ return value
61
+
62
+
63
+ def _dump(value: object) -> str:
64
+ return json.dumps(_short(value), default=str, indent=1)
65
+
66
+
67
+ async def judge(
68
+ goal: Any, world: Any, calls: Sequence[Any], *, messages: Sequence[Any] = ()
69
+ ) -> tuple[bool | None, str]:
70
+ """Decide one judged sub-goal: (passed, explanation). Never raises."""
71
+ verdict: dict[str, tuple[bool | None, str]] = {}
72
+
73
+ @tool(
74
+ "inspect_world",
75
+ "The world's tables: without a name, every table and its row count; with one, its rows.",
76
+ schema({"table": str}, []),
77
+ )
78
+ async def inspect_world(args: dict[str, Any]) -> dict[str, Any]:
79
+ table = str(args.get("table") or "").strip()
80
+ state = world.state(table or None)
81
+ if table:
82
+ return _say(_dump(state))
83
+ return _say(_dump({name: len(rows or []) for name, rows in state.items()}))
84
+
85
+ @tool(
86
+ "query_world",
87
+ "Run one read-only PostgreSQL query for a specific row. Use inspect_world for schema discovery.",
88
+ schema({"sql": str}, ["sql"]),
89
+ )
90
+ async def query_world(args: dict[str, Any]) -> dict[str, Any]:
91
+ sql = str(args.get("sql") or "")
92
+ try:
93
+ return _say(_dump(world.query(sql)))
94
+ except Exception as exc: # noqa: BLE001 - bad model SQL is feedback, not a stage crash
95
+ return _say(
96
+ _dump(
97
+ {
98
+ "error": f"{type(exc).__name__}: {exc}",
99
+ "dialect": "postgresql",
100
+ "recovery": "Use inspect_world to discover real tables, then retry once.",
101
+ }
102
+ ),
103
+ error=True,
104
+ )
105
+
106
+ @tool(
107
+ "read_transcript",
108
+ "What was said, in order. `user` is the caller, `assistant` is the agent being judged.",
109
+ schema({}, []),
110
+ )
111
+ async def read_transcript(args: dict[str, Any]) -> dict[str, Any]:
112
+ if not messages:
113
+ return _say(
114
+ "no transcript was recorded for this session, so nothing here can settle a claim "
115
+ "about what was said",
116
+ error=True,
117
+ )
118
+ return _say(_transcript(messages))
119
+
120
+ @tool(
121
+ "decide",
122
+ "Commit to the verdict, once, after looking.",
123
+ schema({"passed": bool, "explanation": str, "undecided": bool}, ["explanation"]),
124
+ )
125
+ async def decide(args: dict[str, Any]) -> dict[str, Any]:
126
+ explanation = str(args.get("explanation") or "").strip()
127
+ if not explanation:
128
+ return _say("a verdict needs an explanation naming what you saw", error=True)
129
+ held = None if bool(args.get("undecided")) else bool(args.get("passed"))
130
+ verdict["it"] = (held, explanation)
131
+ return _say("recorded")
132
+
133
+ spec = SessionSpec(
134
+ system_prompt=_INSTRUCTIONS,
135
+ servers={
136
+ "world": tool_server(
137
+ name="world",
138
+ tools=[inspect_world, query_world, read_transcript, decide],
139
+ )
140
+ },
141
+ max_turns=12,
142
+ model=judge_model(),
143
+ )
144
+ prompt = (
145
+ f"Claim {getattr(goal, 'name', '')!r}.\n"
146
+ f"What it means: {getattr(goal, 'what', '') or '(none written)'}\n"
147
+ f"Why a model must decide it: {getattr(goal, 'judged', '')}\n\n"
148
+ f"Actions the agent took:\n{_dump([_call(c) for c in calls])}\n\n"
149
+ "Inspect the world and the transcript as needed, then call decide."
150
+ )
151
+ try:
152
+ async with Stage(spec, name="judge-sub-goals") as stage:
153
+ await stage.say(prompt)
154
+ except Exception as exc: # noqa: BLE001 - a judge that could not run is not a failed agent
155
+ return None, f"the judge could not run: {type(exc).__name__}: {exc}"
156
+ return verdict.get("it", (None, "the judge finished without a verdict"))
157
+
158
+
159
+ def _transcript(messages: Sequence[Any]) -> str:
160
+ """The turns as spoken, oldest first. A long call keeps its tail, where a readback would be."""
161
+ body = "\n".join(
162
+ f"{m.get('role') or 'unknown'}: {str(m.get('content') or '').strip()}"
163
+ for m in messages
164
+ if isinstance(m, dict)
165
+ )
166
+ return body if len(body) <= _TRANSCRIPT_LIMIT else "...\n" + body[-_TRANSCRIPT_LIMIT:]
167
+
168
+
169
+ def _call(call: Any) -> dict[str, Any]:
170
+ return {
171
+ "name": getattr(call, "name", ""),
172
+ "arguments": getattr(call, "arguments", None) or {},
173
+ "result": getattr(call, "result", None),
174
+ "ok": bool(getattr(call, "ok", False)),
175
+ "refused": bool(getattr(call, "refused", False)),
176
+ "error": str(getattr(call, "error", "") or ""),
177
+ }
178
+
179
+
180
+ def _say(text: str, *, error: bool = False) -> dict[str, Any]:
181
+ body: dict[str, Any] = {"content": [{"type": "text", "text": text}]}
182
+ if error:
183
+ body["is_error"] = True
184
+ return body
@@ -0,0 +1,50 @@
1
+ """Static LiveKit worker metadata recovered from untrusted source code."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Collection
7
+ from pathlib import Path
8
+
9
+
10
+ _ENV_AGENT_NAME = re.compile(
11
+ r"agent_name\s*=\s*os\.(?:environ\.get|getenv)\(\s*"
12
+ r"[\"']LIVEKIT_AGENT_NAME[\"']\s*,\s*[\"'](?P<name>[^\"']+)[\"']"
13
+ )
14
+ _STATIC_RTC_SESSION_AGENT_NAME = re.compile(
15
+ r"@(?:[A-Za-z_][A-Za-z0-9_]*\.)*rtc_session\s*\("
16
+ r"(?:(?!\)\s*(?:\r?\n|$)).)*?"
17
+ r"agent_name\s*=\s*[\"'](?P<name>[^\"']+)[\"']",
18
+ re.DOTALL,
19
+ )
20
+
21
+
22
+ def infer_livekit_agent_name_from_source(
23
+ source_root: Path, *, ignored_parts: Collection[str] = ()
24
+ ) -> str:
25
+ """Return a literal LiveKit dispatch name declared by common SDK forms.
26
+
27
+ Dynamic expressions remain unresolved deliberately: guessing could route a call to the wrong
28
+ worker on a shared LiveKit project.
29
+ """
30
+
31
+ try:
32
+ if not source_root.is_dir():
33
+ return ""
34
+ for path in source_root.rglob("*.py"):
35
+ relative = path.relative_to(source_root)
36
+ if any(part in ignored_parts for part in relative.parts):
37
+ continue
38
+ try:
39
+ source = path.read_text(encoding="utf-8")
40
+ except (OSError, UnicodeDecodeError):
41
+ continue
42
+ for pattern in (_ENV_AGENT_NAME, _STATIC_RTC_SESSION_AGENT_NAME):
43
+ if match := pattern.search(source):
44
+ return match.group("name").strip()
45
+ except OSError:
46
+ pass
47
+ return ""
48
+
49
+
50
+ __all__ = ["infer_livekit_agent_name_from_source"]