agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,506 @@
1
+ """Hosted HTTP chat calls against an already-provisioned Bundle V2 process world."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import re
8
+ import time
9
+ from datetime import datetime, timezone
10
+ from pathlib import Path
11
+ from typing import Any, Mapping, Sequence
12
+ from urllib.parse import urljoin
13
+
14
+ from fi.simulate.agent.wrapper import AgentInput
15
+ from fi.simulate.agent.wrappers.http import HTTPAgentWrapper
16
+
17
+ from .call_runner import ArtifactUploader, CallRunnerContext
18
+ from .contract import AgentContract
19
+ from .hosted_scheduler import CallAborted, CallOutcome, Scenario, World
20
+ from .outbound import ArtifactKind, format_rfc3339_millis
21
+ from .process_runtime import EnvironmentRuntime
22
+ from .run.conversation import TargetConversationEnded, Transcript, converse
23
+ from .scenario import Scenario as ConversationScenario
24
+ from .world.runtime import Call, GeneratedWorld
25
+ from .world.stores.postgres import AttachedPostgresStore
26
+
27
+
28
+ DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS = 120.0
29
+
30
+
31
+ def _chat_target_timeout_seconds() -> float:
32
+ """Return the per-turn target deadline without accepting unusable values."""
33
+ raw = os.getenv("ALK_CHAT_TARGET_TIMEOUT_SECONDS", "").strip()
34
+ if not raw:
35
+ return DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS
36
+ try:
37
+ configured = float(raw)
38
+ except ValueError:
39
+ return DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS
40
+ return (
41
+ configured
42
+ if 1.0 <= configured <= 600.0
43
+ else DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS
44
+ )
45
+
46
+
47
+ def _duration_ms(started: datetime, ended: datetime) -> int:
48
+ return max(0, round((ended - started).total_seconds() * 1000))
49
+
50
+
51
+ def _scenario_document(bundle_dir: Path, key: str) -> dict[str, Any]:
52
+ for path in sorted((bundle_dir / "scenarios").glob("*/scenario.json")):
53
+ try:
54
+ body = json.loads(path.read_text(encoding="utf-8"))
55
+ except (OSError, ValueError):
56
+ continue
57
+ if isinstance(body, dict) and body.get("scenario_key") == key:
58
+ return body
59
+ raise CallAborted(f"chat_scenario_document_unavailable: scenario_key={key!r}")
60
+
61
+
62
+ def _qmark_to_postgres(statement: str) -> str:
63
+ """Translate generated SQLite-style positional placeholders, never SQL structure."""
64
+ return re.sub(r"\?", "%s", statement)
65
+
66
+
67
+ class _HostedToolStore:
68
+ """Generated handler ``Db`` adapter over the leased world's attached Postgres database."""
69
+
70
+ key = "hosted-postgres"
71
+
72
+ def __init__(self, dsn: str) -> None:
73
+ self._store = AttachedPostgresStore(dsn)
74
+
75
+ def start(self) -> None:
76
+ self._store.start()
77
+
78
+ def stop(self) -> None:
79
+ return
80
+
81
+ def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
82
+ return self._store.query(_qmark_to_postgres(sql), params)
83
+
84
+ def execute(self, sql: str, params: Sequence[Any] = ()) -> int:
85
+ return self._store.execute(_qmark_to_postgres(sql), params)
86
+
87
+ def state(
88
+ self, only: Sequence[str] | None = None
89
+ ) -> dict[str, list[dict[str, Any]]]:
90
+ return self._store.state(only=only)
91
+
92
+ def collections(self) -> list[str]:
93
+ return list(self._store.state())
94
+
95
+ def records(self, collection: str) -> list[dict[str, Any]]:
96
+ return self._store.table(collection)
97
+
98
+ def add(self, collection: str, record: Mapping[str, Any]) -> dict[str, Any]:
99
+ return self._store.add(collection, dict(record))
100
+
101
+
102
+ def _tool_world(
103
+ bundle_dir: Path,
104
+ contract: AgentContract,
105
+ runtime: EnvironmentRuntime,
106
+ source_directory: Path | None = None,
107
+ ) -> GeneratedWorld:
108
+ endpoint = runtime.endpoints.get("world_db")
109
+ if endpoint is None or endpoint.protocol != "postgres":
110
+ raise CallAborted(
111
+ "chat_world_unavailable: world_db postgres endpoint is absent"
112
+ )
113
+ world = GeneratedWorld(store=_HostedToolStore(endpoint.address))
114
+ world.tools = [tool.model_dump(mode="json") for tool in contract.tools]
115
+ world.handlers = {}
116
+ for tool in contract.tools:
117
+ path = bundle_dir / "handlers" / f"{tool.name}.py"
118
+ if path.is_file():
119
+ world.handlers[tool.name] = path.read_text(encoding="utf-8")
120
+ world.refusal_signature = contract.refusal_signature
121
+ if source_directory is not None:
122
+ world.reach(str(source_directory))
123
+ return world
124
+
125
+
126
+ def _tools(contract: AgentContract) -> list[dict[str, Any]]:
127
+ values: list[dict[str, Any]] = []
128
+ for spec in contract.tools:
129
+ properties = {
130
+ argument: {"type": _json_type(spec.arg_types.get(argument, "string"))}
131
+ for argument in spec.args
132
+ }
133
+ values.append(
134
+ {
135
+ "name": spec.name,
136
+ "description": spec.description,
137
+ "parameters": {
138
+ "type": "object",
139
+ "properties": properties,
140
+ "required": list(spec.args),
141
+ },
142
+ }
143
+ )
144
+ return values
145
+
146
+
147
+ def _json_type(declared: str) -> str:
148
+ normalized = str(declared or "").lower()
149
+ if any(mark in normalized for mark in ("int", "float", "number")):
150
+ return "number"
151
+ if "bool" in normalized:
152
+ return "boolean"
153
+ if any(mark in normalized for mark in ("list", "array", "sequence")):
154
+ return "array"
155
+ if any(mark in normalized for mark in ("dict", "map", "object")):
156
+ return "object"
157
+ return "string"
158
+
159
+
160
+ def _tool_call(call: dict[str, Any], index: int) -> tuple[str, dict[str, Any], str]:
161
+ function = call.get("function") if isinstance(call.get("function"), dict) else {}
162
+ name = str(call.get("name") or function.get("name") or "")
163
+ raw = call.get("arguments", function.get("arguments", {}))
164
+ if isinstance(raw, str):
165
+ try:
166
+ arguments = json.loads(raw)
167
+ except ValueError:
168
+ arguments = {"_raw": raw}
169
+ else:
170
+ arguments = dict(raw or {}) if isinstance(raw, dict) else {}
171
+ return name, arguments, str(call.get("id") or f"call_{index}")
172
+
173
+
174
+ def _tool_response_result(response: Mapping[str, Any]) -> Any:
175
+ """Return the callback's real tool result without inventing a second execution."""
176
+ value = response.get("result", response.get("content"))
177
+ if isinstance(value, str):
178
+ try:
179
+ return json.loads(value)
180
+ except ValueError:
181
+ return value
182
+ return value
183
+
184
+
185
+ def _record_completed_tool_call(
186
+ world: GeneratedWorld,
187
+ *,
188
+ name: str,
189
+ arguments: Mapping[str, Any],
190
+ response: Mapping[str, Any],
191
+ ) -> None:
192
+ """Record a tool the submitted callback already executed.
193
+
194
+ Callback-backed agents can return the request and its completed response together. Replaying
195
+ that request through the generated world both risks repeating a side effect and incorrectly
196
+ turns a real success into ``no such tool`` when no mock handler was authored. The callback's
197
+ response is the authoritative execution evidence at this seam.
198
+ """
199
+ error_value = response.get("error")
200
+ error = str(error_value) if error_value not in (None, "") else ""
201
+ refused = bool(response.get("refused", False))
202
+ declared_success = response.get("success", response.get("ok"))
203
+ ok = (
204
+ bool(declared_success)
205
+ if declared_success is not None
206
+ else not error and not refused
207
+ )
208
+ world.calls.append(
209
+ Call(
210
+ name=name,
211
+ arguments=dict(arguments),
212
+ result=_tool_response_result(response),
213
+ ok=ok,
214
+ refused=refused,
215
+ error=error,
216
+ at=time.time(),
217
+ )
218
+ )
219
+
220
+
221
+ class _HostedChatTarget:
222
+ """The already-running Bundle V2 chat process, exposed as the normal conversation target.
223
+
224
+ ``converse`` owns the simulated customer's turns. This target owns only the submitted
225
+ agent's side of the exchange and response-carried tool evidence. Keeping that split identical
226
+ to the local repository target prevents the hosted lane from silently becoming a one-message
227
+ smoke test again.
228
+ """
229
+
230
+ key = "hosted_repository"
231
+
232
+ def __init__(
233
+ self,
234
+ *,
235
+ wrapper: Any,
236
+ contract: AgentContract,
237
+ world: GeneratedWorld,
238
+ scenario_key: str,
239
+ scenario_id: str,
240
+ ) -> None:
241
+ self._wrapper = wrapper
242
+ self._contract = contract
243
+ self.world = world
244
+ self._scenario_key = scenario_key
245
+ self._scenario_id = scenario_id
246
+ self._messages: list[dict[str, Any]] = []
247
+ self._turn = 0
248
+
249
+ async def open(self) -> None:
250
+ return
251
+
252
+ async def say(self, utterance: str) -> str:
253
+ self._messages.append({"role": "user", "content": utterance})
254
+ for continuation in range(8):
255
+ response = await self._wrapper.call(
256
+ AgentInput(
257
+ thread_id=self._scenario_key,
258
+ execution_id=self._scenario_id,
259
+ turn_index=self._turn,
260
+ scenario_name=self._scenario_key,
261
+ modality="text",
262
+ messages=list(self._messages),
263
+ new_message=dict(self._messages[-1]),
264
+ tools=_tools(self._contract)
265
+ if self._contract.runtime
266
+ and self._contract.runtime.interface
267
+ and self._contract.runtime.interface.include_tools
268
+ else [],
269
+ )
270
+ )
271
+ trace = dict((response.metadata or {}).get("external_agent") or {})
272
+ if trace and not trace.get("success", False):
273
+ raise RuntimeError(
274
+ str(trace.get("error") or "submitted endpoint request failed")
275
+ )
276
+ conversation_ended = bool(
277
+ (response.metadata or {}).get("conversation_ended")
278
+ )
279
+ returned = list(response.tool_calls or [])
280
+ if not returned:
281
+ answer = response.content.strip()
282
+ if conversation_ended:
283
+ raise TargetConversationEnded(answer)
284
+ self._messages.append({"role": "assistant", "content": answer})
285
+ self._turn += 1
286
+ return answer
287
+
288
+ self._messages.append(
289
+ {
290
+ "role": "assistant",
291
+ "content": response.content or "",
292
+ "tool_calls": returned,
293
+ }
294
+ )
295
+ response_by_id = {
296
+ str(item.get("tool_call_id") or item.get("id") or ""): item
297
+ for item in response.tool_responses or []
298
+ if isinstance(item, Mapping)
299
+ and (item.get("tool_call_id") or item.get("id"))
300
+ }
301
+ returned_ids: set[str] = set()
302
+ for index, call in enumerate(returned, start=1):
303
+ name, arguments, call_id = _tool_call(call, index)
304
+ returned_ids.add(call_id)
305
+ provided = response_by_id.get(call_id)
306
+ if provided is not None:
307
+ _record_completed_tool_call(
308
+ self.world,
309
+ name=name,
310
+ arguments=arguments,
311
+ response=provided,
312
+ )
313
+ result_content = provided.get("content", provided.get("result"))
314
+ if not isinstance(result_content, str):
315
+ result_content = json.dumps(result_content, default=str)
316
+ else:
317
+ result = self.world.handle_tool_call(
318
+ {"id": call_id, "name": name, "arguments": arguments}
319
+ )
320
+ result_content = (
321
+ result.content if result is not None else f"no such tool {name}"
322
+ )
323
+ self._messages.append(
324
+ {
325
+ "role": "tool",
326
+ "tool_call_id": call_id,
327
+ "name": name,
328
+ "content": result_content,
329
+ }
330
+ )
331
+
332
+ # A callback-backed repository has already run its own real tools. It returns their
333
+ # responses beside the final text; replaying the calls above records deterministic
334
+ # evidence in the generated world, but asking the callback a second time would run the
335
+ # tools twice. HTTP agents that only return requests continue normally with the
336
+ # generated-world responses appended above.
337
+ provided_ids = set(response_by_id)
338
+ if returned_ids and returned_ids.issubset(provided_ids):
339
+ answer = response.content.strip()
340
+ if conversation_ended:
341
+ raise TargetConversationEnded(answer)
342
+ self._messages.append({"role": "assistant", "content": answer})
343
+ self._turn += 1
344
+ return answer
345
+ raise RuntimeError(
346
+ "submitted chat agent exceeded 8 tool continuations in one turn"
347
+ )
348
+
349
+ async def close(self) -> None:
350
+ return
351
+
352
+ @property
353
+ def spent_usd(self) -> float:
354
+ # Provider-side target cost is not observable at this transport seam.
355
+ return 0.0
356
+
357
+
358
+ def _conversation_scenario(document: dict[str, Any]) -> ConversationScenario:
359
+ try:
360
+ normalized = dict(document)
361
+ normalized.setdefault(
362
+ "name", str(normalized.get("scenario_key") or "hosted-chat-scenario")
363
+ )
364
+ return ConversationScenario.model_validate(normalized)
365
+ except Exception as exc: # noqa: BLE001 - normalize malformed bundle content at the call seam
366
+ raise CallAborted(f"chat_scenario_invalid: {exc}") from exc
367
+
368
+
369
+ async def _drive_conversation(
370
+ target: _HostedChatTarget,
371
+ scenario: ConversationScenario,
372
+ contract: AgentContract,
373
+ bundle_dir: Path,
374
+ ) -> Transcript:
375
+ return await converse(
376
+ target,
377
+ scenario,
378
+ contract,
379
+ world_root=bundle_dir,
380
+ )
381
+
382
+
383
+ class HostedChatCallRunner:
384
+ """Drive a repository chat ingress inside its leased world."""
385
+
386
+ def __init__(self, adapter: ArtifactUploader, context: CallRunnerContext) -> None:
387
+ self._adapter = adapter
388
+ self._context = context
389
+ contract_path = context.bundle_dir / "contract.json"
390
+ if not contract_path.is_file():
391
+ self._contract: AgentContract | None = None
392
+ else:
393
+ self._contract = AgentContract.model_validate_json(
394
+ contract_path.read_text(encoding="utf-8")
395
+ )
396
+
397
+ async def run(
398
+ self,
399
+ scenario: Scenario,
400
+ runtime: EnvironmentRuntime,
401
+ *,
402
+ world: World | None = None,
403
+ ) -> CallOutcome:
404
+ del (
405
+ world
406
+ ) # Handler execution uses the same leased world's endpoint from runtime.
407
+ if self._contract is None:
408
+ raise CallAborted(
409
+ "chat_contract_unavailable: bundle/contract.json is absent"
410
+ )
411
+ interface = self._contract.runtime.interface if self._contract.runtime else None
412
+ if interface is None or interface.kind not in {"http", "callable"}:
413
+ raise CallAborted(
414
+ "chat_interface_unsupported: an HTTP or callable runtime interface is required"
415
+ )
416
+ endpoint = runtime.endpoints.get("target_http")
417
+ if endpoint is None:
418
+ raise CallAborted(
419
+ "chat_capability_unavailable: target_http endpoint is absent"
420
+ )
421
+
422
+ document = _scenario_document(self._context.bundle_dir, scenario.scenario_key)
423
+ conversation_scenario = _conversation_scenario(document)
424
+ if not conversation_scenario.instruction.strip():
425
+ raise CallAborted("chat_scenario_invalid: instruction is empty")
426
+ target_world = _tool_world(
427
+ self._context.bundle_dir,
428
+ self._contract,
429
+ runtime,
430
+ self._context.source_directory,
431
+ )
432
+ adapter_path = "/invoke" if interface.kind == "callable" else interface.path
433
+ adapter_protocol = (
434
+ "fi.alk" if interface.kind == "callable" else interface.protocol
435
+ )
436
+ wrapper = HTTPAgentWrapper(
437
+ endpoint=urljoin(
438
+ endpoint.address.rstrip("/") + "/", adapter_path.lstrip("/")
439
+ ),
440
+ protocol=adapter_protocol,
441
+ include_tools=interface.include_tools,
442
+ timeout=_chat_target_timeout_seconds(),
443
+ metadata={
444
+ "target": "hosted_repository_runtime",
445
+ "scenario": scenario.scenario_key,
446
+ },
447
+ )
448
+ started = datetime.now(timezone.utc)
449
+ try:
450
+ transcript = await _drive_conversation(
451
+ _HostedChatTarget(
452
+ wrapper=wrapper,
453
+ contract=self._contract,
454
+ world=target_world,
455
+ scenario_key=scenario.scenario_key,
456
+ scenario_id=scenario.scenario_id,
457
+ ),
458
+ conversation_scenario,
459
+ self._contract,
460
+ self._context.bundle_dir,
461
+ )
462
+ except CallAborted:
463
+ raise
464
+ except Exception as exc: # noqa: BLE001 - convert target transport failures to call faults
465
+ raise CallAborted(
466
+ f"chat_target_failed: {type(exc).__name__}: {exc}"
467
+ ) from exc
468
+
469
+ ended = datetime.now(timezone.utc)
470
+ transcript_id = await self._adapter.upload_artifact(
471
+ transcript.artifact(),
472
+ kind=ArtifactKind.TRANSCRIPT,
473
+ scenario_key=scenario.scenario_key,
474
+ )
475
+ calls = tuple(transcript.calls)
476
+ if calls:
477
+ tool_trace = "\n".join(
478
+ json.dumps(
479
+ {
480
+ "name": call.name,
481
+ "arguments": call.arguments,
482
+ "result": call.result,
483
+ "ok": call.ok,
484
+ "error": call.error,
485
+ "refused": call.refused,
486
+ "at": call.at,
487
+ },
488
+ sort_keys=True,
489
+ default=str,
490
+ )
491
+ for call in calls
492
+ ).encode("utf-8")
493
+ await self._adapter.upload_artifact(
494
+ tool_trace,
495
+ kind=ArtifactKind.TOOL_TRACE,
496
+ scenario_key=scenario.scenario_key,
497
+ )
498
+ return CallOutcome(
499
+ calls=calls,
500
+ turns=len(transcript.exchanges),
501
+ started_at=format_rfc3339_millis(started),
502
+ ended_at=format_rfc3339_millis(ended),
503
+ duration_ms=_duration_ms(started, ended),
504
+ transcript_artifact=transcript_id,
505
+ messages=tuple(transcript.canonical_messages()),
506
+ )
@@ -0,0 +1,136 @@
1
+ """Running a check the harness wrote, and deciding what its answer means.
2
+
3
+ A check is Python because an environment can be a database, a filesystem or a page, and any
4
+ little assertion language invented here would fit only the first. It is given the two things a
5
+ run leaves behind and returns a sentence when something is wrong:
6
+
7
+ def check(world, calls):
8
+ rows = world.state()["orders"]
9
+ if len(rows) != 1:
10
+ return f"{len(rows)} orders, expected 1"
11
+ if not any(c.name == "order_combo_meal" for c in calls):
12
+ return "the combo was never ordered"
13
+ return None
14
+
15
+ ``world`` is the environment afterwards. ``calls`` is every tool call that was made, each with
16
+ its arguments and whether it succeeded — so a check can insist not only that a call happened but
17
+ that it happened with the right arguments, which is the difference between booking 11 PM and
18
+ booking 10 PM.
19
+
20
+ A check that raises is a broken check, not a failed one, and is reported that way. Confusing the
21
+ two would let a typo read as a finding about the agent.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from dataclasses import dataclass
27
+ from typing import Any, Sequence
28
+
29
+ from .world.runtime import Call, GeneratedWorld
30
+
31
+
32
+ @dataclass
33
+ class Outcome:
34
+ """What one check said."""
35
+
36
+ name: str
37
+ held: bool
38
+ said: str = ""
39
+ broken: bool = False
40
+
41
+ def line(self) -> str:
42
+ mark = "!" if self.broken else ("x" if self.held else " ")
43
+ return f" [{mark}] {self.name}" + (f" — {self.said}" if self.said else "")
44
+
45
+
46
+ def run_check(
47
+ source: str, world: GeneratedWorld, calls: Sequence[Call], *, name: str = "check"
48
+ ) -> Outcome:
49
+ """Execute one check against what the run left behind."""
50
+ namespace: dict[str, Any] = {}
51
+ try:
52
+ exec(compile(source, f"<check:{name}>", "exec"), namespace)
53
+ except Exception as failed:
54
+ return Outcome(
55
+ name, False, f"the check would not compile: {failed}", broken=True
56
+ )
57
+
58
+ checker = namespace.get("check")
59
+ if not callable(checker):
60
+ return Outcome(
61
+ name, False, "the check defines no check(world, calls)", broken=True
62
+ )
63
+
64
+ try:
65
+ said = checker(world, list(calls))
66
+ except Exception as failed:
67
+ # The check is at fault, not the agent. A KeyError in an assertion is our bug, and
68
+ # scoring it against the agent is how a harness invents findings.
69
+ return Outcome(
70
+ name,
71
+ False,
72
+ f"the check raised {type(failed).__name__}: {str(failed)[:200]}",
73
+ broken=True,
74
+ )
75
+
76
+ if said is None or said is True or (isinstance(said, str) and not said.strip()):
77
+ return Outcome(name, True)
78
+ if said is False:
79
+ return Outcome(name, False, "False")
80
+ if not isinstance(said, str):
81
+ return Outcome(
82
+ name,
83
+ False,
84
+ f"the check returned {type(said).__name__} {repr(said)[:200]}; a check returns a "
85
+ "sentence naming what is wrong, or None when it held.",
86
+ broken=True,
87
+ )
88
+ return Outcome(name, False, said)
89
+
90
+
91
+ def all_held(outcomes: Sequence[Outcome]) -> bool:
92
+ return all(one.held for one in outcomes) and not any(one.broken for one in outcomes)
93
+
94
+
95
+ def broken(outcomes: Sequence[Outcome]) -> list[Outcome]:
96
+ return [one for one in outcomes if one.broken]
97
+
98
+
99
+ def run_world_check(
100
+ source: str, world: GeneratedWorld, *, name: str = "check"
101
+ ) -> Outcome:
102
+ """Execute one check about the world itself, rather than about a run.
103
+
104
+ A world check asks whether the environment is usable at all, so it is written ``check(world)``
105
+ and there are no calls to give it. Both arities are accepted, because the difference is not
106
+ worth a rejection: a check written ``check(world, calls)`` out of habit is answering the same
107
+ question, and gets an empty list.
108
+ """
109
+ import inspect
110
+
111
+ namespace: dict[str, Any] = {}
112
+ try:
113
+ exec(compile(source, f"<world-check:{name}>", "exec"), namespace)
114
+ except Exception as failed:
115
+ return Outcome(
116
+ name, False, f"the check would not compile: {failed}", broken=True
117
+ )
118
+
119
+ checker = namespace.get("check")
120
+ if not callable(checker):
121
+ return Outcome(name, False, "the check defines no check(world)", broken=True)
122
+
123
+ try:
124
+ wants = len(inspect.signature(checker).parameters)
125
+ except (TypeError, ValueError):
126
+ wants = 1
127
+ try:
128
+ said = checker(world) if wants < 2 else checker(world, [])
129
+ except Exception as failed:
130
+ return Outcome(
131
+ name,
132
+ False,
133
+ f"the check raised {type(failed).__name__}: {str(failed)[:200]}",
134
+ broken=True,
135
+ )
136
+ return Outcome(name, said is None, "" if said is None else str(said))