agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,91 @@
1
+ """Stage four: run the scenarios against the real agent, and say what came back.
2
+
3
+ The last stage that was a command rather than a conversation. Nothing about it needed to be:
4
+ wiring the world to the assistant and running the checks is already code, and the part worth
5
+ having judgement on is which scenario to run and what a failure actually means.
6
+
7
+ That second part is why this is a stage at all. A failing check has four possible causes and only
8
+ one of them is a finding about the agent — the others are a wrong check, a wrong contract, or a
9
+ simulated caller that never asked for the thing. Deciding which is reading, not arithmetic.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from pathlib import Path
15
+ from typing import Any, Callable
16
+
17
+ from ..backends import SessionSpec
18
+ from ..config import artifact_dir, chosen_model, load_skill
19
+ from ..contract import AgentContract
20
+ from ..scenario_tools import load_scenarios
21
+ from ..session import Stage
22
+ from .tools import (
23
+ RUN_SERVER,
24
+ load_results,
25
+ missing_prerequisites,
26
+ run_tools,
27
+ )
28
+
29
+ SKILL = "run-scenarios"
30
+
31
+
32
+ def open_stage(
33
+ contract: AgentContract,
34
+ *,
35
+ out: Path | None = None,
36
+ ask: Callable[..., Any] | None = None,
37
+ max_turns: int = 40,
38
+ ) -> tuple[Stage, Path]:
39
+ """A live run-the-scenarios stage, and where it will write its results."""
40
+ destination = out or artifact_dir(contract.agent)
41
+ server = run_tools(destination, destination, contract=contract)
42
+ spec = SessionSpec(
43
+ system_prompt=(f"{load_skill(SKILL)}\n\n## This agent\n\n{contract.brief()}"),
44
+ servers={RUN_SERVER: server},
45
+ builtins=("AskUserQuestion",),
46
+ cwd=str(destination.parent if destination.parent.exists() else Path.cwd()),
47
+ max_turns=max_turns,
48
+ model=chosen_model(),
49
+ ask=ask,
50
+ )
51
+ return Stage(spec, name=SKILL), destination
52
+
53
+
54
+ def opening(contract: AgentContract, destination: Path) -> str:
55
+ """What to tell the stage when it opens.
56
+
57
+ Deliberately does not tell it to run everything. Each call costs money and takes minutes, and
58
+ a stage that opens by spending the whole suite gives nobody a chance to say which one they
59
+ cared about.
60
+ """
61
+ written = load_scenarios(destination)
62
+ already = load_results(destination)
63
+ blocked = (
64
+ missing_prerequisites(destination, contract)
65
+ if contract.modality == "voice"
66
+ else []
67
+ )
68
+ if blocked:
69
+ return (
70
+ f"There are {len(written)} scenarios for {contract.agent!r}, but a live call cannot "
71
+ "be placed yet:\n - "
72
+ + "\n - ".join(blocked)
73
+ + "\n\nSay this plainly and stop."
74
+ )
75
+ if already:
76
+ passed = sum(1 for record in already if record["passed"])
77
+ return (
78
+ f"{len(already)} of {len(written)} scenarios for {contract.agent!r} have been run, "
79
+ f"{passed} passing. Say where things stand with read_results, then ask which to run."
80
+ )
81
+ return (
82
+ f"{len(written)} scenarios are ready for {contract.agent!r} and none has been run.\n\n"
83
+ "Run preflight, then list_scenarios, then say which ones you would run first and why. "
84
+ "Do not start running them until you are asked to — each call takes minutes and costs "
85
+ "real money."
86
+ )
87
+
88
+
89
+ def load(destination: Path) -> list[dict[str, Any]]:
90
+ """What has been run for this agent, if anything has."""
91
+ return load_results(Path(destination))
@@ -0,0 +1,508 @@
1
+ """What is being tested, and how the harness talks to it.
2
+
3
+ The rest of the run does not care what the agent under test is. It says something and gets a
4
+ reply back, and whatever tool calls happened in between landed in the world. That is the entire
5
+ interface, and keeping it that narrow is what lets the same scenarios, the same world and the
6
+ same grading run against an agent hosted anywhere.
7
+
8
+ Two things are supplied per target: how to say something to it, and how its tool calls reach the
9
+ world. ``LocalAgent`` is only for contract-only specs with no submitted implementation.
10
+ ``RepositoryChatTarget`` starts the submitted runtime and reaches its existing HTTP/WebSocket
11
+ interface. A hosted target uses the same narrow protocol with the transport swapped: its tool
12
+ calls arrive over a webhook or in a turn response, and the same world answers them. The world,
13
+ scenarios and grading do not change.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import asyncio
19
+ import json
20
+ from pathlib import Path
21
+ import socket
22
+ import time
23
+ from typing import Any, Callable, Protocol, runtime_checkable
24
+ from urllib.parse import urljoin
25
+
26
+ from ..backends import SessionSpec, resolve as resolve_backend, tool, tool_server
27
+
28
+ from ..config import chosen_model
29
+ from ..contract import AgentContract
30
+ from ..session import Stage
31
+ from ..world.runtime import GeneratedWorld
32
+
33
+ AGENT_SERVER = "agent"
34
+
35
+ _TYPES: dict[str, type] = {
36
+ "str": str,
37
+ "string": str,
38
+ "int": int,
39
+ "integer": int,
40
+ "float": float,
41
+ "number": float,
42
+ "bool": bool,
43
+ "boolean": bool,
44
+ "list": list,
45
+ "dict": dict,
46
+ }
47
+
48
+
49
+ def _python_type(declared: str) -> type:
50
+ """The type a tool's argument is declared with, as something a schema can carry."""
51
+ lowered = (declared or "").strip().lower()
52
+ if lowered.startswith(("list", "sequence", "array")):
53
+ return list
54
+ if lowered.startswith(("dict", "mapping", "object")):
55
+ return dict
56
+ return _TYPES.get(lowered, str)
57
+
58
+
59
+ def describe(spec: Any, contract: AgentContract) -> str:
60
+ """What the agent is told a tool takes, including the values it accepts.
61
+
62
+ The values matter more than they look. An agent whose real schema enumerates its menu knows
63
+ that a Big Mac combo is ``big_mac_combo``; the same agent without them guesses, gets refused,
64
+ and reads as broken when what is broken is the harness that withheld them. Anything the
65
+ contract recorded as permitted, the agent under test is told.
66
+ """
67
+ parts = [spec.description or f"{spec.name} for {contract.agent}"]
68
+ for arg in spec.args:
69
+ values = spec.arg_values.get(arg)
70
+ if isinstance(values, (list, tuple)) and values:
71
+ rendered = ", ".join(str(value) for value in values)
72
+ parts.append(f" {arg} accepts: {rendered}")
73
+ elif arg in spec.arg_types:
74
+ parts.append(f" {arg}: {spec.arg_types[arg]}")
75
+ return "\n".join(parts)
76
+
77
+
78
+ def agent_tools(contract: AgentContract, world: GeneratedWorld) -> Any:
79
+ """The agent's own tools, wired to the world so a call really happens.
80
+
81
+ Every call goes through ``world.call``, so a refusal comes back as a refusal the agent can
82
+ read and recover from, rather than as a success it will happily build on.
83
+ """
84
+
85
+ def bind(spec: Any) -> Any:
86
+ schema = {
87
+ arg: _python_type(spec.arg_types.get(arg, "str")) for arg in spec.args
88
+ }
89
+
90
+ @tool(spec.name, describe(spec, contract), schema)
91
+ async def call_tool(
92
+ args: dict[str, Any], _name: str = spec.name
93
+ ) -> dict[str, Any]:
94
+ # Through handle_tool_call, not straight to world.call. That method is the interface
95
+ # ALK's own runners drive an environment by, so going around it would leave the
96
+ # claim that a generated world plugs into them untested — and free to drift.
97
+ done = world.handle_tool_call({"name": _name, "arguments": args})
98
+ if done is None:
99
+ return {
100
+ "content": [{"type": "text", "text": f"no such tool {_name}"}],
101
+ "is_error": True,
102
+ }
103
+ return {
104
+ "content": [{"type": "text", "text": done.content or ""}],
105
+ **({} if done.success else {"is_error": True}),
106
+ }
107
+
108
+ return call_tool
109
+
110
+ return tool_server(
111
+ name=AGENT_SERVER,
112
+ version="0.1.0",
113
+ tools=[bind(spec) for spec in contract.tools],
114
+ )
115
+
116
+
117
+ def agent_prompt(contract: AgentContract) -> str:
118
+ """The agent under test, as its contract describes it.
119
+
120
+ Only what the contract records, because anything added here is a difference between the agent
121
+ being graded and the agent that exists.
122
+ """
123
+ parts = [
124
+ f"You are {contract.agent}: {contract.one_liner}".strip(),
125
+ contract.system_prompt_excerpt.strip(),
126
+ ]
127
+ if contract.hard_constraints:
128
+ parts.append(
129
+ "Rules you must follow:\n - " + "\n - ".join(contract.hard_constraints)
130
+ )
131
+ if contract.modality == "voice":
132
+ parts.append(
133
+ "You are speaking out loud. Keep replies to what a person would actually say: "
134
+ "short, no lists, no markdown."
135
+ )
136
+ parts.append(
137
+ "Use your tools to do anything real. Never tell the customer something is done unless a "
138
+ "tool confirmed it, and if a tool refuses, say so plainly and offer what is possible."
139
+ )
140
+ return "\n\n".join(part for part in parts if part)
141
+
142
+
143
+ @runtime_checkable
144
+ class Target(Protocol):
145
+ """An agent under test, reachable by saying something to it."""
146
+
147
+ key: str
148
+
149
+ async def open(self) -> None: ...
150
+ async def say(self, utterance: str) -> str: ...
151
+ async def close(self) -> None: ...
152
+ @property
153
+ def spent_usd(self) -> float: ...
154
+
155
+
156
+ def _drivable(model: str | None) -> None:
157
+ """Refuse a model the selected backend cannot actually run, before a suite is graded on it.
158
+
159
+ Handed a model it cannot reach, a backend does not fail: it produces a session that answers
160
+ nothing, which arrives as a scenario with no turns and no calls and every check red. That
161
+ reads exactly like an agent that ignored the person, and the whole suite is wrong in a way
162
+ nobody would think to question.
163
+ """
164
+ named = (model or "").strip().lower()
165
+ if not named:
166
+ return
167
+ backend = resolve_backend()
168
+ if backend.can_drive(named):
169
+ return
170
+ raise RuntimeError(
171
+ f"this target cannot run {model!r}. The selected harness backend "
172
+ f"({backend.name}) does not drive that model. Pick a model that backend serves, "
173
+ "select the backend that serves it through ALK_HARNESS, or point the spec's target "
174
+ "at one of ALK's own endpoint adapters rather than at this one."
175
+ )
176
+
177
+
178
+ class LocalAgent:
179
+ """The agent run here, from its contract, with its tools bound to the world."""
180
+
181
+ key = "local"
182
+
183
+ def __init__(
184
+ self,
185
+ contract: AgentContract,
186
+ world: GeneratedWorld,
187
+ *,
188
+ model: str | None = None,
189
+ max_turns: int = 12,
190
+ ) -> None:
191
+ self.contract = contract
192
+ self.world = world
193
+ if contract.runtime or contract.tool_entrypoints or contract.implementation:
194
+ raise RuntimeError(
195
+ "the local contract target is disabled for repository-backed agents because it "
196
+ "reconstructs the agent from its prompt. Register a target that starts the "
197
+ "agent's shipped runtime and applies the provisioned endpoint overrides."
198
+ )
199
+ _drivable(model)
200
+ # The agent under test gets its own tools and nothing else. A target that can reach a
201
+ # file or a shell is not the agent anybody deployed.
202
+ spec = SessionSpec(
203
+ system_prompt=agent_prompt(contract),
204
+ servers={AGENT_SERVER: agent_tools(contract, world)},
205
+ max_turns=max_turns,
206
+ model=chosen_model(model),
207
+ )
208
+ self._stage = Stage(spec, name="target")
209
+
210
+ async def open(self) -> None:
211
+ await self._stage.__aenter__()
212
+
213
+ async def say(self, utterance: str) -> str:
214
+ turn = await self._stage.say(utterance)
215
+ return turn.text.strip()
216
+
217
+ async def close(self) -> None:
218
+ await self._stage.__aexit__(None, None, None)
219
+
220
+ @property
221
+ def spent_usd(self) -> float:
222
+ return self._stage.spent_usd
223
+
224
+
225
+ class RepositoryChatTarget:
226
+ """The submitted chat runtime, reached over its existing turn ingress.
227
+
228
+ Lifecycle mirrors the repository-backed voice path: bind the scenario's generated world,
229
+ start the isolated submitted runtime with only endpoint substitutions, wait for its real
230
+ ingress, converse, then remove only that runtime. The source prompt and tools are never
231
+ reconstructed in this process.
232
+ """
233
+
234
+ key = "repository"
235
+
236
+ def __init__(
237
+ self,
238
+ contract: AgentContract,
239
+ world: GeneratedWorld,
240
+ *,
241
+ world_root: str | Path,
242
+ trace_path: str | Path | None = None,
243
+ scenario_name: str = "scenario",
244
+ ) -> None:
245
+ runtime = contract.runtime
246
+ interface = runtime.interface if runtime is not None else None
247
+ if interface is None:
248
+ raise RuntimeError(
249
+ "the submitted chat runtime has no recorded conversational interface. "
250
+ "Record its existing HTTP port, path and protocol during understanding; the "
251
+ "harness will not reconstruct the repository agent from its prompt."
252
+ )
253
+ if interface.kind not in {"http", "websocket"}:
254
+ raise RuntimeError(
255
+ f"submitted chat interface {interface.kind!r} is not wired for hosted repository "
256
+ "execution yet; supported now: HTTP and WebSocket"
257
+ )
258
+ if interface.port is None:
259
+ raise RuntimeError("submitted HTTP chat interface has no container port")
260
+ self.contract = contract
261
+ self.world = world
262
+ self.world_root = Path(world_root)
263
+ self.trace_path = Path(trace_path) if trace_path is not None else None
264
+ self.scenario_name = scenario_name
265
+ self.interface = interface
266
+ self._webhook: Any | None = None
267
+ self._wrapper: Any | None = None
268
+ self._messages: list[dict[str, Any]] = []
269
+ self._turn = 0
270
+
271
+ async def open(self) -> None:
272
+ from ..provision import (
273
+ connect_runner_network,
274
+ runtime_endpoint,
275
+ start_runtime,
276
+ )
277
+ from .voice import WorldWebhook
278
+
279
+ webhook = WorldWebhook().start()
280
+ webhook.bind(self.world)
281
+ self._webhook = webhook
282
+ try:
283
+ private_host = await asyncio.to_thread(
284
+ connect_runner_network, self.world_root
285
+ )
286
+ tool_url = (
287
+ f"http://{private_host}:{webhook.port}"
288
+ if private_host
289
+ else f"http://host.docker.internal:{webhook.port}"
290
+ )
291
+ await asyncio.to_thread(
292
+ start_runtime,
293
+ self.world_root,
294
+ overrides={"TOOLS_API_URL": tool_url},
295
+ trace_path=self.trace_path,
296
+ publish_ports=[self.interface.port],
297
+ stable_seconds=0.5,
298
+ )
299
+ # A runtime-only Compose project creates its network at start. This second call is
300
+ # the same idempotent attach used by voice and makes its private container address
301
+ # reachable from a hosted runner container.
302
+ await asyncio.to_thread(connect_runner_network, self.world_root)
303
+ scheme = "ws" if self.interface.kind == "websocket" else "http"
304
+ base = await asyncio.to_thread(
305
+ runtime_endpoint,
306
+ self.world_root,
307
+ self.interface.port,
308
+ scheme=scheme,
309
+ )
310
+ await asyncio.to_thread(self._wait_ready, base)
311
+ if self.interface.kind == "websocket":
312
+ from fi.simulate.agent.wrappers.websocket import WebSocketAgentWrapper
313
+
314
+ wrapper = WebSocketAgentWrapper
315
+ else:
316
+ from fi.simulate.agent.wrappers.http import HTTPAgentWrapper
317
+
318
+ wrapper = HTTPAgentWrapper
319
+ self._wrapper = wrapper(
320
+ endpoint=urljoin(
321
+ base.rstrip("/") + "/", self.interface.path.lstrip("/")
322
+ ),
323
+ protocol=self.interface.protocol,
324
+ include_tools=self.interface.include_tools,
325
+ timeout=30.0,
326
+ metadata={
327
+ "target": "submitted_repository_runtime",
328
+ "scenario": self.scenario_name,
329
+ },
330
+ )
331
+ except Exception:
332
+ await self.close()
333
+ raise
334
+
335
+ def _wait_ready(self, base: str) -> None:
336
+ from urllib import error as urllib_error
337
+ from urllib import request as urllib_request
338
+ from urllib.parse import urlsplit
339
+
340
+ deadline = time.monotonic() + 60.0
341
+ health = self.interface.health_path
342
+ last = "not reachable"
343
+ while time.monotonic() < deadline:
344
+ try:
345
+ if health and self.interface.kind == "http":
346
+ url = urljoin(base.rstrip("/") + "/", health.lstrip("/"))
347
+ with urllib_request.urlopen(url, timeout=2) as response:
348
+ if int(getattr(response, "status", 200)) < 500:
349
+ return
350
+ else:
351
+ parsed = urlsplit(base)
352
+ with socket.create_connection(
353
+ (str(parsed.hostname), int(parsed.port or 80)), timeout=2
354
+ ):
355
+ return
356
+ except (OSError, urllib_error.URLError) as exc:
357
+ last = f"{type(exc).__name__}: {exc}"
358
+ time.sleep(0.25)
359
+ raise RuntimeError(
360
+ f"submitted chat runtime did not become ready on port {self.interface.port}: {last}"
361
+ )
362
+
363
+ async def say(self, utterance: str) -> str:
364
+ if self._wrapper is None:
365
+ raise RuntimeError("submitted chat runtime is not open")
366
+ from fi.simulate.agent.wrapper import AgentInput
367
+
368
+ self._messages.append({"role": "user", "content": utterance})
369
+ for _step in range(8):
370
+ request = AgentInput(
371
+ thread_id=self.scenario_name,
372
+ execution_id=self.scenario_name,
373
+ turn_index=self._turn,
374
+ scenario_name=self.scenario_name,
375
+ modality="text",
376
+ messages=list(self._messages),
377
+ new_message=dict(self._messages[-1]),
378
+ tools=self._tools() if self.interface.include_tools else [],
379
+ )
380
+ response = await self._wrapper.call(request)
381
+ trace = dict((response.metadata or {}).get("external_agent") or {})
382
+ if trace and not trace.get("success", False):
383
+ raise RuntimeError(
384
+ str(trace.get("error") or "submitted chat endpoint request failed")
385
+ )
386
+ calls = list(response.tool_calls or [])
387
+ if calls:
388
+ self._messages.append(
389
+ {
390
+ "role": "assistant",
391
+ "content": response.content or "",
392
+ "tool_calls": calls,
393
+ }
394
+ )
395
+ for index, call in enumerate(calls, start=1):
396
+ name, arguments, call_id = self._tool_call(call, index)
397
+ result = self.world.handle_tool_call(
398
+ {"id": call_id, "name": name, "arguments": arguments}
399
+ )
400
+ self._messages.append(
401
+ {
402
+ "role": "tool",
403
+ "tool_call_id": call_id,
404
+ "name": name,
405
+ "content": (
406
+ result.content
407
+ if result is not None
408
+ else f"there is no tool called {name}"
409
+ ),
410
+ }
411
+ )
412
+ continue
413
+ said = response.content.strip()
414
+ self._messages.append({"role": "assistant", "content": said})
415
+ self._turn += 1
416
+ return said
417
+ raise RuntimeError(
418
+ "submitted chat agent exceeded 8 tool continuations in one turn"
419
+ )
420
+
421
+ @staticmethod
422
+ def _tool_call(call: dict[str, Any], index: int) -> tuple[str, dict[str, Any], str]:
423
+ function = (
424
+ call.get("function") if isinstance(call.get("function"), dict) else {}
425
+ )
426
+ name = str(call.get("name") or function.get("name") or "")
427
+ raw = call.get("arguments", function.get("arguments", {}))
428
+ if isinstance(raw, str):
429
+ try:
430
+ parsed = json.loads(raw)
431
+ except ValueError:
432
+ parsed = {"_raw": raw}
433
+ else:
434
+ parsed = dict(raw or {}) if isinstance(raw, dict) else {}
435
+ return name, parsed, str(call.get("id") or f"call_{index}")
436
+
437
+ def _tools(self) -> list[dict[str, Any]]:
438
+ tools: list[dict[str, Any]] = []
439
+ for spec in self.contract.tools:
440
+ properties = {
441
+ arg: {"type": self._json_type(spec.arg_types.get(arg, "string"))}
442
+ for arg in spec.args
443
+ }
444
+ tools.append(
445
+ {
446
+ "name": spec.name,
447
+ "description": spec.description,
448
+ "parameters": {
449
+ "type": "object",
450
+ "properties": properties,
451
+ "required": list(spec.args),
452
+ },
453
+ }
454
+ )
455
+ return tools
456
+
457
+ @staticmethod
458
+ def _json_type(declared: str) -> str:
459
+ normalized = str(declared or "").lower()
460
+ if any(mark in normalized for mark in ("int", "float", "number")):
461
+ return "number"
462
+ if "bool" in normalized:
463
+ return "boolean"
464
+ if any(mark in normalized for mark in ("list", "array", "sequence")):
465
+ return "array"
466
+ if any(mark in normalized for mark in ("dict", "map", "object")):
467
+ return "object"
468
+ return "string"
469
+
470
+ async def close(self) -> None:
471
+ from ..provision import stop_runtime
472
+
473
+ try:
474
+ await asyncio.to_thread(stop_runtime, self.world_root)
475
+ finally:
476
+ if self._webhook is not None:
477
+ self._webhook.stop()
478
+ self._webhook = None
479
+ self._wrapper = None
480
+
481
+ @property
482
+ def spent_usd(self) -> float:
483
+ # The submitted target owns its model/provider accounting. Provider evidence may add it
484
+ # later; the harness must not fabricate a cost from HTTP traffic.
485
+ return 0.0
486
+
487
+
488
+ _REGISTRY: dict[str, Callable[..., Target]] = {
489
+ LocalAgent.key: LocalAgent,
490
+ RepositoryChatTarget.key: RepositoryChatTarget,
491
+ }
492
+
493
+
494
+ def register_target(key: str, factory: Callable[..., Target]) -> None:
495
+ """Add a way of reaching an agent. A hosted runtime is a class and this line."""
496
+ _REGISTRY[key] = factory
497
+
498
+
499
+ def resolve(key: str) -> Callable[..., Target]:
500
+ if key not in _REGISTRY:
501
+ raise NotImplementedError(
502
+ f"no target {key!r}; registered targets are {', '.join(sorted(_REGISTRY))}"
503
+ )
504
+ return _REGISTRY[key]
505
+
506
+
507
+ def supported() -> tuple[str, ...]:
508
+ return tuple(sorted(_REGISTRY))