agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,241 @@
1
+ """What a harness backend is, said without naming any vendor.
2
+
3
+ A stage of this harness is a conversation: a system prompt, a set of tools the model may call,
4
+ a turn budget, and a loop that feeds tool results back until the model stops. Today that loop is
5
+ Claude Code's; tomorrow it may be Gemini's, Bedrock's, or a harness of our own. The stages do
6
+ not care, so nothing they say may mention a vendor.
7
+
8
+ This module is that neutrality, in four pieces:
9
+
10
+ - ``ToolSpec`` / ``ToolServer``: a tool as the harness defines one, with the async handler that
11
+ executes it. Backends adapt these to whatever their loop natively speaks.
12
+ - ``SessionSpec``: everything a stage asks of a session. This is the real contract the ten
13
+ construction sites were already expressing through a vendor options class.
14
+ - The reply vocabulary (``SessionOpened``, ``ModelReply``, ``ToolReturned``, ``StageDone``):
15
+ what a running session emits, which ``Stage`` renders into events. A backend translates its
16
+ provider's stream into these and nothing else leaks through.
17
+ - ``HarnessBackend`` / ``HarnessSession``: the two protocols a new backend implements. A backend
18
+ with its own loop supplies its own session type; the harness never sees inside it.
19
+
20
+ Nothing here imports a provider SDK, so a deployment that uses one backend does not need the
21
+ other's dependencies installed.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from dataclasses import dataclass, field
27
+ from typing import Any, AsyncIterator, Awaitable, Callable, Protocol, runtime_checkable
28
+
29
+ ToolHandler = Callable[[dict[str, Any]], Awaitable[dict[str, Any]]]
30
+
31
+
32
+ def qualified(server: str, tool_name: str) -> str:
33
+ """The fully qualified name a session grants and a model calls.
34
+
35
+ The ``mcp__{server}__{tool}`` convention comes from the first backend, but the skills and
36
+ gates all speak it, so every backend keeps it. Renaming per backend would mean rewriting
37
+ every prompt that names a tool.
38
+ """
39
+ return f"mcp__{server}__{tool_name}"
40
+
41
+
42
+ @dataclass
43
+ class ToolSpec:
44
+ """One tool: its contract for the model, and the code that executes it.
45
+
46
+ ``input_schema`` is either a JSON Schema dict or the shorthand ``{"arg": str}`` mapping the
47
+ tool decorator accepts. ``handler`` receives the arguments dict and returns
48
+ ``{"content": [{"type": "text", "text": ...}], "is_error"?: bool}``, the shape every
49
+ existing tool already returns.
50
+ """
51
+
52
+ name: str
53
+ description: str
54
+ input_schema: Any
55
+ handler: ToolHandler
56
+
57
+
58
+ @dataclass
59
+ class ToolServer:
60
+ """A named group of tools granted to a session together."""
61
+
62
+ name: str
63
+ version: str = "0.1.0"
64
+ tools: list[ToolSpec] = field(default_factory=list)
65
+
66
+
67
+ def tool(
68
+ name: str, description: str, input_schema: Any
69
+ ) -> Callable[[ToolHandler], ToolSpec]:
70
+ """Declare a tool. Same signature the stages have always used, no vendor behind it."""
71
+
72
+ def decorator(handler: ToolHandler) -> ToolSpec:
73
+ return ToolSpec(
74
+ name=name,
75
+ description=description,
76
+ input_schema=input_schema,
77
+ handler=handler,
78
+ )
79
+
80
+ return decorator
81
+
82
+
83
+ def tool_server(
84
+ name: str, version: str = "0.1.0", tools: list[ToolSpec] | None = None
85
+ ) -> ToolServer:
86
+ """Group tools under a server name, as the stages have always done."""
87
+ return ToolServer(name=name, version=version, tools=list(tools or []))
88
+
89
+
90
+ # Host tools a backend may be asked to supply itself. Claude Code ships these; a backend without
91
+ # a host CLI implements them from files.py. Anything else asked for as a builtin is refused at
92
+ # session build time rather than silently dropped.
93
+ FILE_TOOLS = ("Read", "Glob", "Grep")
94
+ ASK_TOOL = "AskUserQuestion"
95
+ KNOWN_BUILTINS = (*FILE_TOOLS, ASK_TOOL)
96
+
97
+
98
+ @dataclass
99
+ class SessionSpec:
100
+ """Everything a stage asks of a session, with no vendor vocabulary in it.
101
+
102
+ ``builtins`` are host tools by bare name (``Read``, ``Glob``, ``Grep``,
103
+ ``AskUserQuestion``); ``servers`` are the harness's own tools. ``ask`` is the operator
104
+ callback consulted when the model asks a question; None means the run is unattended.
105
+ ``gated`` selects the deny-by-default permission regime every tool-bearing stage runs
106
+ under; the one stage that runs bare (the simulated customer, which has no tools) turns it
107
+ off to keep its behaviour byte-identical.
108
+ ``thinking`` opts into the harness's thinking policy (config.thinking_config); stages that
109
+ never set one keep their backend's default.
110
+ """
111
+
112
+ system_prompt: str
113
+ servers: dict[str, ToolServer] = field(default_factory=dict)
114
+ builtins: tuple[str, ...] = ()
115
+ cwd: str | None = None
116
+ max_turns: int = 40
117
+ model: str = ""
118
+ ask: Any = None
119
+ gated: bool = True
120
+ thinking: bool = False
121
+ # A ready-made permission callable that replaces the backend's own gate wholesale. One
122
+ # stage (understand, interactive) passes its gate in fully built; backends without a
123
+ # permission callback concept ignore it, which is safe because their gating is structural.
124
+ permission_override: Any = None
125
+ # How long this stage may go without emitting anything before it is treated as hung. Zero
126
+ # takes the harness default. A stage whose work happens inside one long tool call needs its
127
+ # own bound: it is working the whole time and has nothing to say while it does, so the
128
+ # default reads honest work as a hang and kills it.
129
+ idle_timeout_seconds: float = 0.0
130
+
131
+ def granted(self) -> list[str]:
132
+ """Every tool name this session may call, qualified the way the model calls it."""
133
+ names = [*self.builtins]
134
+ for server_name, server in self.servers.items():
135
+ names.extend(qualified(server_name, spec.name) for spec in server.tools)
136
+ return names
137
+
138
+ def grant(self, server_name: str, server: ToolServer) -> None:
139
+ """Add a tool server before the session opens."""
140
+ self.servers[server_name] = server
141
+
142
+
143
+ # -- what a running session emits --------------------------------------------------------------
144
+
145
+
146
+ @dataclass
147
+ class SessionOpened:
148
+ """The session exists and has an identity, if the backend assigns one."""
149
+
150
+ session_id: str | None = None
151
+
152
+
153
+ @dataclass
154
+ class Say:
155
+ """The model said something."""
156
+
157
+ text: str
158
+
159
+
160
+ @dataclass
161
+ class Call:
162
+ """The model called a tool."""
163
+
164
+ id: str
165
+ name: str
166
+ arguments: dict[str, Any] = field(default_factory=dict)
167
+
168
+
169
+ @dataclass
170
+ class ModelReply:
171
+ """One assistant message: text and tool calls, in the order they were produced."""
172
+
173
+ parts: list[Any] = field(default_factory=list)
174
+ model: str = ""
175
+
176
+
177
+ @dataclass
178
+ class ToolReturned:
179
+ """What a tool call produced, flattened to text."""
180
+
181
+ id: str
182
+ text: str
183
+ is_error: bool = False
184
+
185
+
186
+ @dataclass
187
+ class StageDone:
188
+ """The exchange is over. The raw facts; Stage turns them into words."""
189
+
190
+ outcome: str = "success"
191
+ turns: int = 0
192
+ cost_usd: float | None = None
193
+ # The units behind the price, so a bill can be checked rather than trusted.
194
+ tokens_in: int = 0
195
+ tokens_out: int = 0
196
+ # Input the provider served from its own cache. Included in tokens_in, and billed well below
197
+ # fresh input, so a ledger that does not carry it overstates a rerun. Carried rather than
198
+ # discounted here: inventing a cache rate would be a guess presented as a price.
199
+ tokens_cached: int = 0
200
+ session_id: str | None = None
201
+ models: set[str] = field(default_factory=set)
202
+ is_error: bool = False
203
+ api_error_status: Any = None
204
+ errors: list[Any] = field(default_factory=list)
205
+
206
+
207
+ # -- what a backend implements -----------------------------------------------------------------
208
+
209
+
210
+ @runtime_checkable
211
+ class HarnessSession(Protocol):
212
+ """One open session. Backends with their own loop implement this around it."""
213
+
214
+ async def start(self) -> None: ...
215
+
216
+ async def stop(self) -> None: ...
217
+
218
+ async def send(self, message: str) -> None: ...
219
+
220
+ def replies(self) -> AsyncIterator[Any]:
221
+ """Everything the session emits for the message just sent, ending with StageDone."""
222
+ ...
223
+
224
+
225
+ @runtime_checkable
226
+ class HarnessBackend(Protocol):
227
+ """A way of running stages. Selected by name through the registry."""
228
+
229
+ name: str
230
+ default_model: str
231
+
232
+ def create(self, spec: SessionSpec) -> HarnessSession: ...
233
+
234
+ def can_drive(self, model: str) -> bool:
235
+ """Whether this backend can actually run the named model.
236
+
237
+ A backend handed a model it cannot reach must refuse loudly here. Left unchecked it
238
+ produces a session that answers nothing, which downstream reads as an agent that ignored
239
+ the person rather than as a configuration mistake.
240
+ """
241
+ ...
@@ -0,0 +1,211 @@
1
+ """The Claude Code backend: the loop this harness grew up on.
2
+
3
+ Adapts the neutral ``SessionSpec`` to ``ClaudeAgentOptions`` exactly the way the stages built
4
+ them before the seam existed: same gate hooks, same permission callback, same provider env,
5
+ same disallowed list. With ``ALK_HARNESS`` unset this backend runs, so nothing here may drift
6
+ from what the stages did on their own.
7
+
8
+ Claude Code supplies Read/Glob/Grep and AskUserQuestion itself, so builtins are granted by
9
+ name rather than implemented here.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from typing import Any, AsyncIterator
15
+
16
+ from claude_agent_sdk import (
17
+ AssistantMessage,
18
+ ClaudeAgentOptions,
19
+ ClaudeSDKClient,
20
+ ResultMessage,
21
+ SdkMcpTool,
22
+ SystemMessage,
23
+ TextBlock,
24
+ ToolResultBlock,
25
+ ToolUseBlock,
26
+ create_sdk_mcp_server,
27
+ )
28
+
29
+ from .base import (
30
+ Call,
31
+ ModelReply,
32
+ Say,
33
+ SessionOpened,
34
+ SessionSpec,
35
+ StageDone,
36
+ ToolReturned,
37
+ ToolServer,
38
+ qualified,
39
+ )
40
+
41
+ DEFAULT_MODEL = "claude-sonnet-4-6"
42
+
43
+
44
+ def _sdk_server(server: ToolServer) -> Any:
45
+ """A ToolServer as the in-process MCP server the SDK routes calls to."""
46
+ return create_sdk_mcp_server(
47
+ name=server.name,
48
+ version=server.version,
49
+ tools=[
50
+ SdkMcpTool(
51
+ name=spec.name,
52
+ description=spec.description,
53
+ input_schema=spec.input_schema,
54
+ handler=spec.handler,
55
+ )
56
+ for spec in server.tools
57
+ ],
58
+ )
59
+
60
+
61
+ def _flattened(content: Any) -> str:
62
+ """A tool result's content as one string, however the SDK packaged it."""
63
+ if isinstance(content, list):
64
+ return "\n".join(
65
+ part.get("text", "") for part in content if isinstance(part, dict)
66
+ )
67
+ return content if isinstance(content, str) else str(content)
68
+
69
+
70
+ class ClaudeSession:
71
+ """One Claude Code session, translated to the neutral reply vocabulary."""
72
+
73
+ def __init__(self, options: ClaudeAgentOptions) -> None:
74
+ self._options = options
75
+ self._client: ClaudeSDKClient | None = None
76
+
77
+ async def start(self) -> None:
78
+ self._client = ClaudeSDKClient(options=self._options)
79
+ await self._client.connect()
80
+
81
+ async def stop(self) -> None:
82
+ if self._client is not None:
83
+ await self._client.disconnect()
84
+ self._client = None
85
+
86
+ async def send(self, message: str) -> None:
87
+ if self._client is None:
88
+ raise RuntimeError("session is not open")
89
+ await self._client.query(message)
90
+
91
+ async def replies(self) -> AsyncIterator[Any]:
92
+ if self._client is None:
93
+ raise RuntimeError("session is not open")
94
+ async for received in self._client.receive_response():
95
+ for reply in self._translate(received):
96
+ yield reply
97
+
98
+ def _translate(self, received: Any) -> list[Any]:
99
+ if isinstance(received, SystemMessage):
100
+ data = received.data if isinstance(received.data, dict) else {}
101
+ return [SessionOpened(session_id=data.get("session_id"))]
102
+ if isinstance(received, AssistantMessage):
103
+ parts: list[Any] = []
104
+ for block in received.content:
105
+ if isinstance(block, TextBlock):
106
+ parts.append(Say(text=block.text))
107
+ elif isinstance(block, ToolUseBlock):
108
+ parts.append(
109
+ Call(id=block.id, name=block.name, arguments=block.input)
110
+ )
111
+ return [ModelReply(parts=parts, model=getattr(received, "model", "") or "")]
112
+ if isinstance(received, ResultMessage):
113
+ # subtype alone is not the outcome. A call that failed upstream still arrives with
114
+ # subtype "success", so the error facts ride along and Stage decides what failed.
115
+ return [
116
+ StageDone(
117
+ outcome=received.subtype,
118
+ turns=received.num_turns,
119
+ cost_usd=received.total_cost_usd,
120
+ **_tokens(getattr(received, "model_usage", None)),
121
+ session_id=received.session_id,
122
+ models=set(getattr(received, "model_usage", None) or {}),
123
+ is_error=bool(getattr(received, "is_error", False)),
124
+ api_error_status=getattr(received, "api_error_status", None),
125
+ errors=list(getattr(received, "errors", None) or []),
126
+ )
127
+ ]
128
+ blocks = getattr(received, "content", None)
129
+ if isinstance(blocks, list):
130
+ returned = []
131
+ for block in blocks:
132
+ if isinstance(block, ToolResultBlock):
133
+ returned.append(
134
+ ToolReturned(
135
+ id=block.tool_use_id,
136
+ text=_flattened(block.content),
137
+ is_error=bool(getattr(block, "is_error", False)),
138
+ )
139
+ )
140
+ return returned
141
+ return []
142
+
143
+
144
+ def _tokens(model_usage: Any) -> dict[str, int]:
145
+ """Input and output tokens across every model a stage used, for the ledger to audit against.
146
+
147
+ Read defensively: this is the SDK's shape, not ours, and a stage must not fail over accounting.
148
+ """
149
+ read = 0
150
+ written = 0
151
+ for usage in (model_usage or {}).values():
152
+ if isinstance(usage, dict):
153
+ read += int(usage.get("inputTokens") or usage.get("input_tokens") or 0)
154
+ written += int(usage.get("outputTokens") or usage.get("output_tokens") or 0)
155
+ else:
156
+ read += int(getattr(usage, "input_tokens", 0) or 0)
157
+ written += int(getattr(usage, "output_tokens", 0) or 0)
158
+ return {"tokens_in": read, "tokens_out": written}
159
+
160
+
161
+ class ClaudeBackend:
162
+ name = "claude"
163
+ default_model = DEFAULT_MODEL
164
+
165
+ def can_drive(self, model: str) -> bool:
166
+ return "claude" in (model or "").lower()
167
+
168
+ def create(self, spec: SessionSpec) -> ClaudeSession:
169
+ from ..config import (
170
+ UNWANTED,
171
+ gate_hooks,
172
+ permission_gate,
173
+ provider_env,
174
+ thinking_config,
175
+ )
176
+
177
+ allowed = [
178
+ *spec.builtins,
179
+ *(
180
+ qualified(server_name, tool_spec.name)
181
+ for server_name, server in spec.servers.items()
182
+ for tool_spec in server.tools
183
+ ),
184
+ ]
185
+ options = ClaudeAgentOptions(
186
+ system_prompt=spec.system_prompt,
187
+ allowed_tools=allowed,
188
+ mcp_servers={
189
+ server_name: _sdk_server(server)
190
+ for server_name, server in spec.servers.items()
191
+ },
192
+ setting_sources=[],
193
+ max_turns=spec.max_turns,
194
+ model=spec.model,
195
+ env=provider_env(spec.model),
196
+ )
197
+ if spec.cwd is not None:
198
+ options.cwd = spec.cwd
199
+ if spec.gated:
200
+ # Not acceptEdits: that auto-approves Edit and Write before the permission callback
201
+ # is consulted, so a stage could rewrite an artifact by hand and skip the tool whose
202
+ # whole job is to validate that change.
203
+ options.permission_mode = "default"
204
+ options.disallowed_tools = list(UNWANTED)
205
+ options.hooks = gate_hooks(allowed)
206
+ options.can_use_tool = spec.permission_override or permission_gate(
207
+ spec.ask, allowed
208
+ )
209
+ if spec.thinking:
210
+ options.thinking = thinking_config()
211
+ return ClaudeSession(options)
@@ -0,0 +1,182 @@
1
+ """Read-only file tools for backends that have no host CLI behind them.
2
+
3
+ Claude Code brings its own Read, Glob and Grep. Any other loop that is granted those names gets
4
+ these: same names, same core arguments, same read-only stance. They exist so a stage's skill
5
+ text ("read the repository, never write to it") means the same thing on every backend.
6
+
7
+ Write access is not implemented on purpose. The harness's own artifacts go through its tools,
8
+ which validate them; the agent under test is somebody's real repository. A backend that wants
9
+ write tools is asking to skip the gates, and the answer is no.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import fnmatch
15
+ import re
16
+ from pathlib import Path
17
+
18
+ from .base import ToolSpec
19
+
20
+ MAX_READ_LINES = 2000
21
+ MAX_LINE_CHARS = 2000
22
+ MAX_MATCHES = 200
23
+ _SKIP_DIRS = {".git", ".venv", "node_modules", "__pycache__", ".pytest_cache"}
24
+
25
+
26
+ def _ok(text: str) -> dict:
27
+ return {"content": [{"type": "text", "text": text}]}
28
+
29
+
30
+ def _error(text: str) -> dict:
31
+ return {"content": [{"type": "text", "text": text}], "is_error": True}
32
+
33
+
34
+ def _resolved(raw: str, cwd: Path) -> Path:
35
+ path = Path(raw).expanduser()
36
+ return path if path.is_absolute() else cwd / path
37
+
38
+
39
+ def _walk(root: Path):
40
+ """Every file under root, skipping the trees nobody means when they say 'the repo'."""
41
+ stack = [root]
42
+ while stack:
43
+ folder = stack.pop()
44
+ try:
45
+ entries = sorted(folder.iterdir())
46
+ except OSError:
47
+ continue
48
+ for entry in entries:
49
+ if entry.is_dir():
50
+ if entry.name not in _SKIP_DIRS:
51
+ stack.append(entry)
52
+ else:
53
+ yield entry
54
+
55
+
56
+ def file_tools(cwd: str | None) -> list[ToolSpec]:
57
+ """Read, Glob and Grep rooted at the session's working directory."""
58
+ base = Path(cwd) if cwd else Path.cwd()
59
+
60
+ async def read(args: dict) -> dict:
61
+ raw = str(args.get("file_path") or args.get("path") or "")
62
+ if not raw:
63
+ return _error("file_path is required")
64
+ path = _resolved(raw, base)
65
+ if not path.is_file():
66
+ return _error(f"{path} is not a file that exists")
67
+ try:
68
+ lines = path.read_text(encoding="utf-8", errors="replace").splitlines()
69
+ except OSError as exc:
70
+ return _error(f"could not read {path}: {exc}")
71
+ offset = max(int(args.get("offset") or 1), 1)
72
+ limit = min(int(args.get("limit") or MAX_READ_LINES), MAX_READ_LINES)
73
+ window = lines[offset - 1 : offset - 1 + limit]
74
+ numbered = "\n".join(
75
+ f"{offset + index}\t{line[:MAX_LINE_CHARS]}"
76
+ for index, line in enumerate(window)
77
+ )
78
+ remaining = len(lines) - (offset - 1 + len(window))
79
+ if remaining > 0:
80
+ numbered += f"\n... ({remaining} more lines; call again with offset={offset + len(window)})"
81
+ return _ok(numbered or "(empty file)")
82
+
83
+ async def glob(args: dict) -> dict:
84
+ pattern = str(args.get("pattern") or "")
85
+ if not pattern:
86
+ return _error("pattern is required")
87
+ root = _resolved(str(args.get("path") or "."), base)
88
+ if not root.is_dir():
89
+ return _error(f"{root} is not a directory that exists")
90
+ matches = []
91
+ for entry in _walk(root):
92
+ relative = str(entry.relative_to(root))
93
+ if fnmatch.fnmatch(relative, pattern) or fnmatch.fnmatch(
94
+ entry.name, pattern
95
+ ):
96
+ matches.append(str(entry))
97
+ if len(matches) >= MAX_MATCHES:
98
+ break
99
+ return _ok("\n".join(matches) or f"no files match {pattern!r} under {root}")
100
+
101
+ async def grep(args: dict) -> dict:
102
+ pattern = str(args.get("pattern") or "")
103
+ if not pattern:
104
+ return _error("pattern is required")
105
+ try:
106
+ expression = re.compile(pattern)
107
+ except re.error as exc:
108
+ return _error(f"invalid regular expression: {exc}")
109
+ root = _resolved(str(args.get("path") or "."), base)
110
+ wanted = str(args.get("glob") or "")
111
+ hits: list[str] = []
112
+ targets = [root] if root.is_file() else list(_walk(root)) if root.is_dir() else []
113
+ if not targets:
114
+ return _error(f"{root} does not exist")
115
+ for entry in targets:
116
+ if wanted and not fnmatch.fnmatch(entry.name, wanted):
117
+ continue
118
+ try:
119
+ text = entry.read_text(encoding="utf-8", errors="replace")
120
+ except OSError:
121
+ continue
122
+ for number, line in enumerate(text.splitlines(), 1):
123
+ if expression.search(line):
124
+ hits.append(f"{entry}:{number}:{line[:400]}")
125
+ if len(hits) >= MAX_MATCHES:
126
+ break
127
+ if len(hits) >= MAX_MATCHES:
128
+ break
129
+ return _ok("\n".join(hits) or f"no lines match {pattern!r}")
130
+
131
+ return [
132
+ ToolSpec(
133
+ name="Read",
134
+ description=(
135
+ "Read a file. Arguments: file_path (absolute or relative to the working "
136
+ "directory), optional offset (1-based first line) and limit (line count)."
137
+ ),
138
+ input_schema={
139
+ "type": "object",
140
+ "properties": {
141
+ "file_path": {"type": "string"},
142
+ "offset": {"type": "integer"},
143
+ "limit": {"type": "integer"},
144
+ },
145
+ "required": ["file_path"],
146
+ },
147
+ handler=read,
148
+ ),
149
+ ToolSpec(
150
+ name="Glob",
151
+ description=(
152
+ "Find files by name pattern, e.g. **/*.py. Arguments: pattern, optional path "
153
+ "to search under."
154
+ ),
155
+ input_schema={
156
+ "type": "object",
157
+ "properties": {
158
+ "pattern": {"type": "string"},
159
+ "path": {"type": "string"},
160
+ },
161
+ "required": ["pattern"],
162
+ },
163
+ handler=glob,
164
+ ),
165
+ ToolSpec(
166
+ name="Grep",
167
+ description=(
168
+ "Search file contents with a regular expression. Arguments: pattern, optional "
169
+ "path (file or directory) and glob to filter file names."
170
+ ),
171
+ input_schema={
172
+ "type": "object",
173
+ "properties": {
174
+ "pattern": {"type": "string"},
175
+ "path": {"type": "string"},
176
+ "glob": {"type": "string"},
177
+ },
178
+ "required": ["pattern"],
179
+ },
180
+ handler=grep,
181
+ ),
182
+ ]