agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,417 @@
1
+ # The harness
2
+
3
+ Point it at an agent. It reads the agent, builds a real database its tools run against, writes
4
+ test scenarios, runs them as conversations, and tells you what held and what did not.
5
+
6
+ Nothing here is written for a particular agent. Every stage takes the contract and the world as
7
+ input, so a different agent is the same commands with a different name.
8
+
9
+ The normal product path is autonomous—no operator messages or stage-by-stage nudges:
10
+
11
+ ```bash
12
+ agent-learn harness auto \
13
+ --path /absolute/path/to/private-agent \
14
+ --count 10
15
+ ```
16
+
17
+ This creates one session containing `job.json`, a sealed `environment-bundle/`, contract,
18
+ generated world/data, validated scenarios, canonical progress events, calls, evidence and run
19
+ artifacts. Agent check failures still complete the run and remain visible as RL evidence.
20
+
21
+ The same `HarnessJob` and `HarnessExecutor` run in Future AGI's isolated hosted sandbox. The
22
+ platform creates jobs and stores their events/artifacts; it does not execute harness stages. See
23
+ [ARCHITECTURE.md](ARCHITECTURE.md) for boundaries, isolation and extension points.
24
+
25
+ Repository-backed chat agents now follow the same runtime lifecycle as voice agents when their
26
+ submitted service exposes an HTTP (ALK or OpenAI-compatible) or JSON WebSocket ingress. ALK starts
27
+ the real service per scenario, connects the simulated user, routes declared environment endpoints,
28
+ records tool/state evidence and tears the service down. It does not reconstruct a repository agent
29
+ from its prompt when no conversational ingress exists.
30
+
31
+ ---
32
+
33
+ # Part 1 — Setting up, from nothing
34
+
35
+ If you have never run this before, do these five steps in order. They take about ten minutes,
36
+ most of which is waiting for the install.
37
+
38
+ ## Before you start
39
+
40
+ You need four things on your machine:
41
+
42
+ | What | Check it with | If missing |
43
+ |---|---|---|
44
+ | Python 3.10 or newer | `python3 --version` | install from python.org, or `brew install python` |
45
+ | `uv` (the package manager this repo uses) | `uv --version` | `brew install uv` |
46
+ | The `claude` command | `claude --version` | `npm install -g @anthropic-ai/claude-code` |
47
+ | A Google Cloud service-account key file (`.json`) for Vertex AI | you were given one, or ask | ask whoever set up your GCP access |
48
+
49
+ The `claude` command matters: the harness talks to the model through the Claude Agent SDK, and
50
+ that SDK runs the `claude` binary under the hood. If it is not installed, every stage fails
51
+ immediately with a connection error.
52
+
53
+ ## Step 1. Get the repo, and work from its root
54
+
55
+ Every command in this document is run from the **root of the repo**, not from this folder:
56
+
57
+ ```bash
58
+ git clone https://github.com/future-agi/agent-learning-kit
59
+ cd agent-learning-kit
60
+ git checkout feat/environment-generation # until this branch is merged
61
+ ```
62
+
63
+ Wherever you cloned it, that directory is the one containing `pyproject.toml`. Check you are in
64
+ the right place:
65
+
66
+ ```bash
67
+ ls pyproject.toml # should print: pyproject.toml
68
+ ```
69
+
70
+ If that errors, you are in the wrong directory. Do not continue until it works.
71
+
72
+ ## Step 2 — Install the dependencies
73
+
74
+ ```bash
75
+ uv sync --extra livekit --group dev
76
+ ```
77
+
78
+ This reads `pyproject.toml`, downloads everything, and creates a folder called `.venv` in the
79
+ repo root. That folder is the "virtual environment": a private copy of Python with this
80
+ project's packages in it, so they do not collide with anything else on your machine.
81
+
82
+ The `--extra livekit` matters even though the harness never makes a voice call. The harness
83
+ builds on `fi.simulate.environment`, and importing anything from `fi.simulate` runs that
84
+ package's `__init__`, which pulls in its LiveKit scenario generator. Plain `uv sync` leaves that
85
+ out and every command dies with `No module named 'livekit'`.
86
+
87
+ It takes a few minutes the first time. You only do this once.
88
+
89
+ ## Step 3 — Use the virtual environment
90
+
91
+ Two ways. **Pick one and stick with it.**
92
+
93
+ **Option A — no activation (what this document uses).** Call the Python inside `.venv` directly:
94
+
95
+ ```bash
96
+ .venv/bin/python -m fi.alk.harness
97
+ ```
98
+
99
+ Nothing to remember, nothing to undo, works in a fresh terminal every time. Every command below
100
+ is written this way.
101
+
102
+ **Option B — activate it.** If you prefer typing plain `python`:
103
+
104
+ ```bash
105
+ source .venv/bin/activate # your prompt now shows (agent-learning-kit)
106
+ python -m fi.alk.harness # plain "python" now means the one in .venv
107
+ deactivate # when you are done
108
+ ```
109
+
110
+ Activation only lasts for that terminal window. Open a new tab and you must activate again. If a
111
+ command ever fails with `No module named fi`, you almost certainly forgot.
112
+
113
+ ## Step 4 — Credentials
114
+
115
+ The harness reaches the model through Vertex AI, which needs your Google Cloud service-account
116
+ key. Nothing is hardcoded and no key is ever read from source.
117
+
118
+ Create a local env file from the template that ships with the repo:
119
+
120
+ ```bash
121
+ cp oss/simulation-acceptance/.env.example .env.acceptance
122
+ ```
123
+
124
+ Open `.env.acceptance` in an editor and fill in two lines:
125
+
126
+ ```bash
127
+ GOOGLE_APPLICATION_CREDENTIALS=/absolute/path/to/your-service-account.json
128
+ GOOGLE_CLOUD_PROJECT=your-gcp-project-id
129
+ ```
130
+
131
+ `.env.acceptance` is git-ignored. It holds a path to a private key: **never commit it, never
132
+ paste its contents into Slack or a PR.**
133
+
134
+ Now load it into your terminal, and pick a model:
135
+
136
+ ```bash
137
+ set -a; . ./.env.acceptance; set +a
138
+ export CLOUD_ML_REGION=global
139
+ export ALK_HARNESS_MODEL=claude-sonnet-4-6
140
+ ```
141
+
142
+ - `set -a; . ./file; set +a` means "read this file and export everything in it". The leading
143
+ `. ` (dot space) is what runs it in your *current* shell, so the variables stick around.
144
+ - `ALK_HARNESS_MODEL` picks the model. **Use `claude-sonnet-4-6` or better.** Haiku is cheaper
145
+ but has twice misread an agent's modality, and modality decides how every later test is run.
146
+
147
+ These last only for the current terminal window. Every new terminal, run these three lines again.
148
+
149
+ ## Step 5 — Check it works
150
+
151
+ ```bash
152
+ .venv/bin/python -m pytest tests/test_harness.py -q
153
+ ```
154
+
155
+ These are offline tests: no model calls, no credentials, no network. If they pass, your
156
+ install is fine. If they fail, the problem is Step 2, not your credentials.
157
+
158
+ Then check the credentials separately, with the cheapest thing that talks to the model:
159
+
160
+ ```bash
161
+ .venv/bin/python -m fi.alk.harness
162
+ ```
163
+
164
+ Say hello. If it answers, the credentials work; type `q` to leave before it spends anything
165
+ real.
166
+
167
+ ---
168
+
169
+ # Part 2 — Using it
170
+
171
+ ## The short version
172
+
173
+ ```bash
174
+ cd path/to/agent-learning-kit
175
+ set -a; . ./.env.acceptance; set +a
176
+ export CLOUD_ML_REGION=global ALK_HARNESS_MODEL=claude-sonnet-4-6
177
+
178
+ .venv/bin/python harness-ui/server.py # a web page, on :8777
179
+ .venv/bin/python -m fi.alk.harness # the same thing in the terminal
180
+ ```
181
+
182
+ Either one is the whole interface. Both open with "which agent would you like to test, and where
183
+ is it?", and everything after that is a conversation. It finds the agent, reads it, builds the
184
+ world, writes the scenarios, and runs them, moving on as each stage produces its artifact.
185
+
186
+ **The page is the one to start with**: it shows what each stage produced while you talk, and it
187
+ is the same harness underneath. There is nothing separate to build or serve; see
188
+ `harness-ui/README.md`.
189
+
190
+ One message is enough to begin:
191
+
192
+ ```
193
+ i want to test my voice ordering agent. the code is at /absolute/path/to/the/agent
194
+ ```
195
+
196
+ In the terminal version: type what you want and press enter, press enter on an **empty** line to
197
+ move to the next stage, and type `q` to leave.
198
+
199
+ ## Where things are written
200
+
201
+ One conversation, one folder. Everything about testing one agent lives together, so closing the
202
+ page, restarting the server or coming back tomorrow all resume by reading the folder.
203
+
204
+ ```
205
+ artifacts/sessions/<id>/
206
+ session.json which agent, where its source is, when it started
207
+ chat.jsonl the conversation itself
208
+ contract.json what the agent verifiably is
209
+ world.sqlite the world, with handlers/, simulator_prompt.md, sub_goals.json
210
+ scenarios/<name>/ one folder per scenario
211
+ runs.json what happened when they ran
212
+ ```
213
+
214
+ The id is readable and unique (`drive-thru-aaea25`), so two attempts at the same agent are two
215
+ sessions rather than one overwriting the other. To start from nothing:
216
+ `rm -rf artifacts/sessions/* artifacts/.open-session`.
217
+
218
+ ## The same stages, one at a time
219
+
220
+ Useful when you want to redo one thing without walking the whole conversation. Each of these
221
+ stays open for corrections until you type `q`; add `--once` to run it unattended and exit.
222
+
223
+ ```bash
224
+ # read an agent's source and write down what it verifiably is
225
+ .venv/bin/python -m fi.alk.harness understand --name my_agent --path ../my-agent-repo
226
+
227
+ # build the environment: the world, the simulator prompt, the sub-goal catalogue
228
+ .venv/bin/python -m fi.alk.harness build --name my_agent
229
+
230
+ # write the test scenarios, each proved before it is kept
231
+ .venv/bin/python -m fi.alk.harness scenarios --name my_agent --count 10
232
+
233
+ # run them against the world here, and grade
234
+ .venv/bin/python -m fi.alk.harness run --name my_agent
235
+
236
+ # or run them against the real hosted agent, as a conversation
237
+ .venv/bin/python -m fi.alk.harness live --name my_agent
238
+ ```
239
+
240
+ `--name` is just a label for the folder your artifacts go in. `--path` is where the agent's code
241
+ lives — a path to another repo on your disk.
242
+
243
+ Useful extras:
244
+
245
+ - `run --only <name> [<name> ...]` runs a single scenario instead of all of them
246
+ - `run --quiet` hides the conversation and prints only verdicts
247
+ - `scenarios` without `--count` uses however many already exist, because coming back to change
248
+ one is not a request for a different number of them
249
+
250
+ ## What each stage does
251
+
252
+ **understand** reads the agent's source and produces `contract.json`: its tools, the exact
253
+ argument names and permitted values, its hard rules, its real data. Everything downstream is
254
+ confined to this, which is what stops later stages inventing tools or menu items. Anything
255
+ changed later goes through an amendment tool and is recorded with its reason, so what came from
256
+ the agent and what came from us stay distinguishable.
257
+
258
+ **build** produces everything common to every test of this agent:
259
+
260
+ - **the world** — a real database behind the agent's tools, with one handler per tool that can
261
+ genuinely refuse: a nonexistent id, an unavailable item, an argument outside what the tool
262
+ accepts. A refusal is the world working; a crash is a defect, and the two are never confused.
263
+ - **the simulator prompt** — for a conversational agent, the person on the other side, written
264
+ once with `{{ slot }}` variables each scenario fills.
265
+ - **the sub-goal catalogue** — the named things this agent can be checked on, each carrying its
266
+ check **as code** wherever the answer is observable, and marked judged only where nothing is.
267
+
268
+ It is exercised before it can be saved — every tool probed with a valid call, a bogus id and a
269
+ missing argument, plus declared sequences where state must carry across calls — and `save_world`
270
+ refuses a world that fails, has no sequences, no sub-goals, only judged sub-goals, no simulator
271
+ prompt for a conversational agent, or rows left over from its own testing.
272
+
273
+ **scenarios** writes each test as a change on that base. Each one owns a folder, and the code in
274
+ it is code, not strings inside a JSON file:
275
+
276
+ ```
277
+ scenarios/<name>/
278
+ scenario.json the instruction, the reference solution, which sub-goals it names
279
+ setup.py def setup(world) what this scenario changes first
280
+ ready.py def ready(world) is the world ready for it
281
+ checks/<goal>.py def check(world, calls) one per deterministic sub-goal
282
+ ```
283
+
284
+ `setup` is code rather than a list of rows because "not necessarily the database alone" cannot be
285
+ written as rows. The check files genuinely run on their own:
286
+
287
+ ```bash
288
+ python scenarios/<name>/checks/<goal>.py path/to/world.sqlite # prints held, or FAILED: ...
289
+ ```
290
+
291
+ Before a scenario is kept it is **proved** by three gates, all pure code, no model involved:
292
+
293
+ 1. **ready**: reset → `setup` → `ready`. The world must hold what the scenario presumes. A
294
+ scenario about the last five items is only a test of the agent if there really are five;
295
+ otherwise the agent fails for something we got wrong and it reads as the agent's fault.
296
+ 2. **solvable**: then run the reference solution and the checks. They must **pass**, or either
297
+ the scenario cannot be passed or a check is wrong.
298
+ 3. **not vacuous**: then reset, set up again, run **nothing**, and run the checks. They must
299
+ **fail**. A check that passes while the agent does nothing grades nothing while reporting a
300
+ result.
301
+
302
+ Only a scenario clearing all three is kept. The reference solution is kept with it, and is never
303
+ run against the agent under test.
304
+
305
+ **run** gives each scenario its own restored copy of the world and grades from what is left
306
+ behind: the state of the world plus every tool call with its arguments. `run` converses with the
307
+ agent locally, rebuilt from its contract. `live` is the same grading against the **real hosted
308
+ agent**: the webhook its own tools call is answered by the world, so a call for something that
309
+ is not there is refused rather than mocked into success.
310
+
311
+ ## How it grades
312
+
313
+ Deterministic by default, a judge only as the fallback.
314
+
315
+ Every sub-goal with a check in code is settled by running that check against two things the run
316
+ left behind: the world afterwards, and the recorded tool calls with their arguments — so "booked
317
+ 10 PM when 11 PM was asked" is caught without any judgement. Sub-goals marked judged are handed
318
+ to a model with three kinds of evidence: what was said, what the agent actually did, and the
319
+ state afterwards. An unanswered claim counts as failed, never as passed, and judged results are
320
+ always reported as judged rather than blended into the code-settled score.
321
+
322
+ ```
323
+ PASS quantity_and_unavailable 3/3 sub-goals settled by code
324
+ [x] quantity_honored
325
+ [x] unavailable_drink_refused
326
+ [x] regular_item_placed_correctly
327
+ [?] no_unrequested_items — judged, not settled by code
328
+
329
+ what the agent actually did:
330
+ order_regular_item({'item_id': 'hamburger'}) -> ok
331
+ order_regular_item({'item_id': 'hamburger'}) -> ok
332
+ ```
333
+
334
+ A run where the world crashed is `VOID`, not `FAIL` — that says nothing about the agent. A check
335
+ that raises is a **broken check**, reported as ours, never scored against the agent.
336
+
337
+ ## What it refuses to do
338
+
339
+ These are the parts worth understanding, because they are what make a result mean something.
340
+
341
+ - A world that fails its own probes will not save; nor will one with no sequences, no sub-goals,
342
+ only judged sub-goals, or rows left over from building it.
343
+ - A scenario is not kept until the world is ready for it, its own solution passes its own checks,
344
+ and those checks fail when nothing is done. Missing preconditions, unsolvable scenarios and
345
+ vacuous checks all die here, at write time.
346
+ - A scenario naming a sub-goal nobody defined, or a table nobody built, is rejected and told
347
+ what does exist.
348
+ - A suite where no sub-goal is shared between scenarios will not save, because nothing would
349
+ roll up across it.
350
+ - Changing the contract is allowed but never silent: every widening, added rule or corrected
351
+ tool is recorded with its reason in `amendments[]`.
352
+
353
+ If a stage tells you it will not do something, that is the design, not a bug to route around.
354
+
355
+ ## What a full pass costs, and how long it takes
356
+
357
+ Measured on Sonnet, on a five-tool voice agent, all three stages in one conversation:
358
+
359
+ | Stage | Turns | Time | Cost |
360
+ |---|---|---|---|
361
+ | reading the agent | 6 | under a minute | ~$0.55 |
362
+ | building the environment | 33 | ~10 minutes | ~$1.40 |
363
+ | five proved scenarios | 22 | ~5 minutes | ~$0.92 |
364
+
365
+ About **$3.30 and twenty minutes** end to end. Building the environment is the long stage, and
366
+ **the Environment tab stays empty until it finishes**: the world is held in memory until
367
+ `save_world` writes it. Watch the chat for progress instead. Grading a local run afterwards is a
368
+ few cents per scenario.
369
+
370
+ ## When something goes wrong
371
+
372
+ | What you see | What it means |
373
+ |---|---|
374
+ | `No module named fi` | Wrong directory, or you are using system `python` instead of `.venv/bin/python` |
375
+ | `command not found: uv` | `brew install uv` |
376
+ | `No module named 'livekit'` | You ran plain `uv sync`. Run `uv sync --extra livekit --group dev` |
377
+ | `No module named 'fastapi'` | Same cause. The UI's dependencies come in with `--group dev` (or `--extra harness-ui`) |
378
+ | Fails instantly on any model call | The `claude` command is not installed, or your env vars are not loaded in this terminal |
379
+ | `Could not load the default credentials` | `GOOGLE_APPLICATION_CREDENTIALS` is unset or points at a file that is not there |
380
+ | `nobody has said which agent this is about yet` | Say where the agent's code lives, with an absolute path |
381
+ | `No contract at ...` | Read the agent first |
382
+ | `No world at ...` | Build the environment first |
383
+ | The page shows empty tabs | Look at which session is open. A build in progress has not written its world yet |
384
+ | A stage does nothing and exits | It ran out of turns. Look at the last few lines: it usually says what it was stuck on |
385
+ | A change to the harness seems to have no effect | Restart the server. A long-lived process does not reload code or skills |
386
+ | `lsof -ti:8777` says the server is up after you stopped it | That matches a browser's leftover sockets. Use `lsof -nP -iTCP:8777 -sTCP:LISTEN` |
387
+
388
+ Everything a stage did is printed as it happens, and every run is kept in
389
+ `artifacts/sessions/<id>/runs.json`, including the transcript and every tool call.
390
+
391
+ ---
392
+
393
+ # Part 3 — For developers
394
+
395
+ ## Adding to it
396
+
397
+ - A new **agent** is nothing: the same stages read its contract.
398
+ - A new **kind of world** is a class and a registration in `world/kinds.py`. Browser is registered
399
+ and stubbed; sqlite is the one built out.
400
+ - A new **place the agent runs** is a class and a registration in `run/targets.py`. `local` runs
401
+ the agent here from its contract; the live voice path answers a hosted assistant's webhook from
402
+ the same `world.handle_tool_call`, so the world, the scenarios and the grading do not change.
403
+ - A change to **how a stage works** is an edit to its `skills/<stage>/SKILL.md`. The markdown is
404
+ the method; code holds only what must be exact.
405
+
406
+ ## Not done yet
407
+
408
+ - Browser worlds are registered but not built.
409
+ - Snapshots are local files, not object storage.
410
+ - Judged sub-goals on the live path are reported as judged, not yet sent to a judge.
411
+ - Nothing reports which of the contract's use cases have no scenario.
412
+
413
+ ## Tests
414
+
415
+ ```bash
416
+ .venv/bin/python -m pytest tests/test_harness.py -q # offline, no credentials needed
417
+ ```
@@ -0,0 +1,77 @@
1
+ """The harness: an agent that builds test environments for other agents.
2
+
3
+ It reads an agent, works out what it verifiably is, builds a world its tools can run against,
4
+ generates scenarios, runs them, and reads the results back. Each of those is a stage, each stage
5
+ is its own session, and stages hand work to each other as artifacts on disk.
6
+
7
+ The split that matters: the model does judgement, and code decides outcomes. Reading unfamiliar
8
+ source, designing a schema, and choosing what is worth testing are judgement. Executing a tool
9
+ call and grading a run are not, and are never delegated to a model.
10
+
11
+ Stages are described in files under ``skills/``, so the method is editable without touching
12
+ code, and where an agent comes from is a registered source, so a new kind of agent is a class
13
+ rather than a new code path.
14
+ """
15
+
16
+ from .chat import Conversation, open_conversation
17
+ from .bundle import EnvironmentBundle, load_bundle, seal_bundle
18
+ from .environment_plan import EnvironmentPlan, load_environment_plan
19
+ from .config import (
20
+ DEFAULT_MODEL,
21
+ artifact_dir,
22
+ load_skill,
23
+ provider_env,
24
+ read_only_session,
25
+ )
26
+ from .contract import AgentContract, Runtime, RuntimeInterface, ToolSpec, validate_contract
27
+ from .job import ExecutionMode, HarnessJob, HarnessStage
28
+ from .scenario import Scenario, validate_scenario
29
+ from .session import Stage, Turn
30
+ from .sources import (
31
+ AgentSource,
32
+ GitHubSource,
33
+ ProviderSource,
34
+ RepoSource,
35
+ SpecSource,
36
+ register_source,
37
+ resolve,
38
+ supported,
39
+ )
40
+ from .understand import open_stage, understand
41
+
42
+ __all__ = [
43
+ "AgentContract",
44
+ "AgentSource",
45
+ "Conversation",
46
+ "DEFAULT_MODEL",
47
+ "EnvironmentBundle",
48
+ "EnvironmentPlan",
49
+ "ExecutionMode",
50
+ "GitHubSource",
51
+ "HarnessJob",
52
+ "HarnessStage",
53
+ "ProviderSource",
54
+ "RepoSource",
55
+ "Runtime",
56
+ "RuntimeInterface",
57
+ "Scenario",
58
+ "SpecSource",
59
+ "Stage",
60
+ "ToolSpec",
61
+ "Turn",
62
+ "artifact_dir",
63
+ "load_skill",
64
+ "load_bundle",
65
+ "load_environment_plan",
66
+ "open_conversation",
67
+ "open_stage",
68
+ "provider_env",
69
+ "read_only_session",
70
+ "register_source",
71
+ "resolve",
72
+ "seal_bundle",
73
+ "supported",
74
+ "understand",
75
+ "validate_contract",
76
+ "validate_scenario",
77
+ ]
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())