agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,481 @@
1
+ """Postgres, as the worked example of what an engine has to supply.
2
+
3
+ This is not "the database the harness supports". It is the reference: when the build stage
4
+ finds an agent on ClickHouse or MySQL or DuckDB, what it writes is a class this shape, and
5
+ what it has to work out is only what is in this file below ``boot_env`` -- how to reach the
6
+ engine, how to read what it holds, and how to put that back. Starting a container, finding a
7
+ free port, waiting for the thing to genuinely answer and not leaking it afterwards are all in
8
+ ``ContainerStore`` and are never rewritten.
9
+
10
+ Nothing here knows what the agent's tools do. The agent keeps its own client, its own SQL and
11
+ its own migrations; the only thing that changed is the host on the far end of its DSN. The
12
+ schema is not invented either -- the build stage runs the agent's own migrations through
13
+ ``apply``, so the tables are the agent's tables, spelled the way the agent spells them. A
14
+ schema we wrote ourselves would be a guess, and every check written against it would inherit
15
+ the guess.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ import os
22
+ from collections.abc import Sequence
23
+ from pathlib import Path
24
+ from typing import Any
25
+
26
+ from ..errors import WorldQueryRejected
27
+ from . import Held, Snapshot, StoreError
28
+ from .container import ContainerStore, docker
29
+
30
+ SCHEMA = "schema.sql"
31
+
32
+
33
+ def _psycopg() -> Any:
34
+ try:
35
+ import psycopg
36
+ except ImportError as exc: # pragma: no cover - depends on the install
37
+ raise StoreError(
38
+ "psycopg is not installed, so a Postgres store cannot be read. Install it with "
39
+ "`uv sync --extra harness-stores`."
40
+ ) from exc
41
+ return psycopg
42
+
43
+
44
+ class PostgresStore(ContainerStore):
45
+ """A Postgres container the agent under test is pointed at."""
46
+
47
+ engine = "postgres"
48
+ image = "postgres:16"
49
+ container_port = 5432
50
+ boot_env = {
51
+ "POSTGRES_USER": "{user}",
52
+ "POSTGRES_PASSWORD": "{password}",
53
+ "POSTGRES_DB": "{database}",
54
+ }
55
+
56
+ # -- how to reach it -------------------------------------------------------------
57
+
58
+ def dsn(self) -> str:
59
+ external = os.environ.get("ALK_POSTGRES_DSN", "").strip()
60
+ if external:
61
+ return external
62
+ host, port = self.address()
63
+ return f"postgresql://{self.user}:{self.password}@{host}:{port}/{self.database}"
64
+
65
+ def probe(self) -> None:
66
+ """Really connect. A running container is not yet a database that listens."""
67
+ with _psycopg().connect(self.dsn(), connect_timeout=3) as connection:
68
+ connection.execute("SELECT 1")
69
+
70
+ def _connect(self) -> Any:
71
+ """A short-lived autocommit connection.
72
+
73
+ Deliberately not pooled and never held open. An idle transaction of ours would block
74
+ the ``TRUNCATE`` in ``restore``, and a reset that hangs on the harness's own connection
75
+ is a very expensive thing to debug.
76
+ """
77
+ return _psycopg().connect(self.dsn(), autocommit=True)
78
+
79
+ # -- how to read what it holds ---------------------------------------------------
80
+
81
+ def apply(self, script: str) -> None:
82
+ """Run whatever was handed in: the agent's migrations, or its seed."""
83
+ if not script.strip():
84
+ return
85
+ with self._connect() as connection:
86
+ connection.execute(script)
87
+ # Remembered because the snapshot holds rows, not DDL. A restore into a fresh
88
+ # container finds no tables, and restoring rows into a schema that is not there
89
+ # quietly restores nothing.
90
+ self.applied.append(script)
91
+
92
+ def execute(self, statement: str, params: Sequence[Any] = ()) -> int:
93
+ """Run one mutation and report the rows it actually changed."""
94
+ with self._connect() as connection:
95
+ cursor = connection.execute(statement, tuple(params))
96
+ return max(0, int(cursor.rowcount or 0))
97
+
98
+ def query(self, statement: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
99
+ """Run one read statement, on a connection Postgres itself will not let write.
100
+
101
+ Autocommit, with the session's own default flipped to read-only rather than one shared
102
+ ``SET TRANSACTION READ ONLY`` transaction: every statement becomes its own implicit
103
+ read-only transaction, so nothing here is ever held open, and ``reset``'s drop of the
104
+ database can always proceed regardless of what a caller just read.
105
+
106
+ An empty ``params`` tuple is passed through as ``None`` rather than as itself: psycopg
107
+ scans for placeholders whenever it is handed anything other than ``None``, and an
108
+ ordinary ``LIKE '%turkey%'`` with nothing to bind then reads its own ``%t`` as an
109
+ unmatched one and raises before the statement ever reaches Postgres.
110
+ """
111
+ with _psycopg().connect(
112
+ self.dsn(), autocommit=True, options="-c default_transaction_read_only=on"
113
+ ) as connection:
114
+ cursor = connection.execute(statement, tuple(params) if params else None)
115
+ columns = [description[0] for description in cursor.description or []]
116
+ seen: set[str] = set()
117
+ for column in columns:
118
+ if column in seen:
119
+ raise WorldQueryRejected(
120
+ f"query() returned more than one column named {column!r}; alias one "
121
+ "of them so a row does not silently lose one under the other."
122
+ )
123
+ seen.add(column)
124
+ return [dict(zip(columns, row, strict=True)) for row in cursor.fetchall()]
125
+
126
+ def _tables(self, connection: Any) -> list[str]:
127
+ rows = connection.execute(
128
+ "SELECT tablename FROM pg_tables WHERE schemaname = 'public' ORDER BY tablename"
129
+ ).fetchall()
130
+ return [row[0] for row in rows]
131
+
132
+ def _primary_key(self, connection: Any, table: str) -> list[str]:
133
+ """The primary key columns, used only to read rows back in a stable order.
134
+
135
+ Joined through ``pg_class``/``pg_namespace`` rather than a ``%s::regclass`` cast over an
136
+ f-string, so a table name is only ever a bound value — an embedded ``"`` (a table
137
+ created as ``CREATE TABLE "we""ird" (...)``) is just a character in that value instead
138
+ of something a regclass cast has to parse.
139
+ """
140
+ rows = connection.execute(
141
+ """
142
+ SELECT a.attname
143
+ FROM pg_index i
144
+ JOIN pg_attribute a ON a.attrelid = i.indrelid AND a.attnum = ANY(i.indkey)
145
+ JOIN pg_class c ON c.oid = i.indrelid
146
+ JOIN pg_namespace n ON n.oid = c.relnamespace
147
+ WHERE n.nspname = 'public' AND c.relname = %s AND i.indisprimary
148
+ ORDER BY array_position(i.indkey, a.attnum)
149
+ """,
150
+ (table,),
151
+ ).fetchall()
152
+ return [row[0] for row in rows]
153
+
154
+ def _column_types(self, connection: Any, table: str) -> dict[str, str]:
155
+ """Return declared column types so writes use the source schema's representation.
156
+
157
+ New managed clusters are always UTF-8, but an externally attached or older SQL_ASCII
158
+ database can return even information-schema text as ``bytes``. Decode the catalogue
159
+ defensively: silently missing every lookup would disable all type adaptation and turn one
160
+ environment detail into a sequence of misleading per-column database failures.
161
+ """
162
+ rows = connection.execute(
163
+ """
164
+ SELECT column_name, data_type
165
+ FROM information_schema.columns
166
+ WHERE table_schema = 'public' AND table_name = %s
167
+ """,
168
+ (table,),
169
+ ).fetchall()
170
+
171
+ def text(value: Any) -> str:
172
+ return value.decode("utf-8") if isinstance(value, bytes) else str(value)
173
+
174
+ return {text(row[0]): text(row[1]) for row in rows}
175
+
176
+ def _select_ordered(self, connection: Any, table: str) -> list[dict[str, Any]]:
177
+ """Every row of one table, ordered by its primary key where it has one.
178
+
179
+ Without that order the same data comes back in whatever sequence the heap happens to
180
+ hold it, and a check comparing the first row is reading a coin toss rather than the
181
+ agent's behaviour. Built with ``sql.Identifier`` rather than an f-string because
182
+ ``table`` is a name the harness only just read out of the catalogue, not a literal it
183
+ wrote itself.
184
+ """
185
+ sql = _psycopg().sql
186
+ key = self._primary_key(connection, table)
187
+ statement = sql.SQL("SELECT * FROM {}").format(sql.Identifier(table))
188
+ if key:
189
+ statement += sql.SQL(" ORDER BY ") + sql.SQL(", ").join(
190
+ sql.Identifier(column) for column in key
191
+ )
192
+ cursor = connection.execute(statement)
193
+ columns = [description[0] for description in cursor.description or []]
194
+ return [dict(zip(columns, row, strict=True)) for row in cursor.fetchall()]
195
+
196
+ def state(
197
+ self, only: Sequence[str] | None = None
198
+ ) -> dict[str, list[dict[str, Any]]]:
199
+ """Every table and its rows, in the shape the checks already expect.
200
+
201
+ ``only`` narrows the read to the named tables, still inside the one connection — a
202
+ caller that already knows it wants a subset (``HostedWorld`` excluding over-cap tables)
203
+ never pays to read and discard rows for the ones it does not. ``None`` reads every table
204
+ in ``public``, exactly as before.
205
+ """
206
+ with self._connect() as connection:
207
+ tables = self._tables(connection) if only is None else list(only)
208
+ return {table: self._select_ordered(connection, table) for table in tables}
209
+
210
+ def table(self, name: str) -> list[dict[str, Any]]:
211
+ """One table's rows, ordered by primary key where it has one.
212
+
213
+ The read-side counterpart to ``state()``'s per-table loop, for a caller that wants only
214
+ one of them: still a single connection, so reading one table never costs a second round
215
+ trip just to learn how to order it.
216
+ """
217
+ with self._connect() as connection:
218
+ return self._select_ordered(connection, name)
219
+
220
+ # -- how to put it back ----------------------------------------------------------
221
+
222
+ def freeze(self) -> Snapshot:
223
+ """Rows and sequence counters, which together are the whole mutable state."""
224
+ with self._connect() as connection:
225
+ counters = {
226
+ row[0]: row[1]
227
+ for row in connection.execute(
228
+ "SELECT sequencename, last_value FROM pg_sequences "
229
+ "WHERE schemaname = 'public'"
230
+ ).fetchall()
231
+ if row[1] is not None
232
+ }
233
+ return Snapshot(rows=self.state(), counters=counters)
234
+
235
+ def restore(self, snapshot: Snapshot) -> None:
236
+ """Put the data back exactly as the snapshot found it.
237
+
238
+ Foreign keys are suspended for the duration rather than the rows being sorted into
239
+ dependency order: the snapshot was taken from a consistent database, so what goes back
240
+ is consistent by construction, and ordering it would be solving a problem we do not
241
+ have. Counters are set last, so the next scenario's first insert gets the id the first
242
+ scenario's did.
243
+ """
244
+ with self._connect() as connection:
245
+ tables = self._tables(connection)
246
+ if not tables:
247
+ return
248
+ listed = ", ".join(f'"{table}"' for table in tables)
249
+ # One statement, so Postgres resolves the dependency order between them itself.
250
+ connection.execute(f"TRUNCATE TABLE {listed} RESTART IDENTITY CASCADE")
251
+
252
+ connection.execute("SET session_replication_role = replica")
253
+ try:
254
+ for table, rows in snapshot.rows.items():
255
+ if not rows or table not in tables:
256
+ continue
257
+ columns = list(rows[0])
258
+ types = self._column_types(connection, table)
259
+ quoted = ", ".join(f'"{column}"' for column in columns)
260
+ placeholders = ", ".join(["%s"] * len(columns))
261
+ statement = (
262
+ f'INSERT INTO "{table}" ({quoted}) VALUES ({placeholders})'
263
+ )
264
+ with connection.cursor() as cursor:
265
+ cursor.executemany(
266
+ statement,
267
+ [
268
+ tuple(
269
+ _adapt(row.get(column), types.get(column, ""))
270
+ for column in columns
271
+ )
272
+ for row in rows
273
+ ],
274
+ )
275
+ finally:
276
+ connection.execute("SET session_replication_role = DEFAULT")
277
+
278
+ for sequence, value in snapshot.counters.items():
279
+ connection.execute(
280
+ "SELECT setval(%s, %s, true)", (f'public."{sequence}"', value)
281
+ )
282
+
283
+ def save_to(self, path: str | Path) -> None:
284
+ """Save both the records and the DDL a fresh Postgres store needs."""
285
+ Held.save_to(self, path)
286
+ root = Path(path)
287
+ schema = docker(
288
+ "exec",
289
+ self.container,
290
+ "pg_dump",
291
+ "--schema-only",
292
+ "--no-owner",
293
+ "--no-privileges",
294
+ "--username",
295
+ self.user,
296
+ "--dbname",
297
+ self.database,
298
+ )
299
+ # pg_dump can emit psql-only safety commands. The snapshot is replayed through psycopg,
300
+ # so keep SQL and discard client meta-commands.
301
+ schema = "\n".join(
302
+ line for line in schema.splitlines() if not line.startswith("\\")
303
+ )
304
+ (root / SCHEMA).write_text(schema, encoding="utf-8")
305
+
306
+ def load_from(self, path: str | Path) -> None:
307
+ root = Path(path)
308
+ schema = root / SCHEMA
309
+ if not schema.exists():
310
+ raise StoreError(f"no saved Postgres schema at {schema}")
311
+ self.apply(schema.read_text(encoding="utf-8"))
312
+ Held.load_from(self, root)
313
+
314
+ # -- what a scenario changes -----------------------------------------------------
315
+
316
+ def add(self, collection: str, record: Any) -> dict[str, Any]:
317
+ """Insert one record and hand back exactly what Postgres stored.
318
+
319
+ ``RETURNING *`` rather than a second read: a caller after the row's generated key (an
320
+ identity column, a default, a trigger) would otherwise have to guess which column that
321
+ is, and a table with no natural way to re-select the row it just inserted could not be
322
+ read back at all.
323
+ """
324
+ columns = list(record)
325
+ sql = _psycopg().sql
326
+ statement = sql.SQL("INSERT INTO {} ({}) VALUES ({}) RETURNING *").format(
327
+ sql.Identifier(collection),
328
+ sql.SQL(", ").join(sql.Identifier(column) for column in columns),
329
+ sql.SQL(", ").join(sql.SQL("%s") for _ in columns),
330
+ )
331
+ with self._connect() as connection:
332
+ types = self._column_types(connection, collection)
333
+ cursor = connection.execute(
334
+ statement,
335
+ tuple(
336
+ _adapt(record[column], types.get(column, "")) for column in columns
337
+ ),
338
+ )
339
+ stored = cursor.fetchone()
340
+ if stored is None:
341
+ raise StoreError(
342
+ f"INSERT INTO {collection!r} ... RETURNING * came back with no row; a rule "
343
+ "or a BEFORE INSERT trigger the agent's own migrations declared can turn an "
344
+ "insert into a no-op, and put() cannot report a record that was never "
345
+ "written."
346
+ )
347
+ out = [description[0] for description in cursor.description or []]
348
+ return dict(zip(out, stored, strict=True))
349
+
350
+ def amend(self, collection: str, key: str, changes: Any, *, by: str = "") -> int:
351
+ if not by:
352
+ raise StoreError(
353
+ f"{collection} is a table, so changing a record needs the column it is keyed on"
354
+ )
355
+ sql = _psycopg().sql
356
+ statement = sql.SQL("UPDATE {} SET {} WHERE {} = %s").format(
357
+ sql.Identifier(collection),
358
+ sql.SQL(", ").join(
359
+ sql.SQL("{} = %s").format(sql.Identifier(column)) for column in changes
360
+ ),
361
+ sql.Identifier(by),
362
+ )
363
+ with self._connect() as connection:
364
+ types = self._column_types(connection, collection)
365
+ cursor = connection.execute(
366
+ statement,
367
+ (
368
+ *(
369
+ _adapt(value, types.get(column, ""))
370
+ for column, value in changes.items()
371
+ ),
372
+ key,
373
+ ),
374
+ )
375
+ return cursor.rowcount
376
+
377
+ def remove(self, collection: str, key: str = "", *, by: str = "") -> int:
378
+ if key and not by:
379
+ raise StoreError(
380
+ f"{collection} is a table, so removing one record needs the column it is keyed on"
381
+ )
382
+ sql = _psycopg().sql
383
+ statement = sql.SQL("DELETE FROM {}").format(sql.Identifier(collection))
384
+ if key:
385
+ statement += sql.SQL(" WHERE {} = %s").format(sql.Identifier(by))
386
+ with self._connect() as connection:
387
+ cursor = connection.execute(statement, (key,) if key else ())
388
+ return cursor.rowcount
389
+
390
+
391
+ class AttachedPostgresStore(PostgresStore):
392
+ """A Postgres store already started by the submitted repository's Compose project.
393
+
394
+ The record and snapshot operations are exactly the same as ``PostgresStore``. Only ownership
395
+ differs: the harness may inspect and reset this database, but the Compose provisioner owns
396
+ its process and lifecycle, so closing one scenario must never remove the shared container.
397
+ """
398
+
399
+ def __init__(self, dsn: str) -> None:
400
+ self._external_dsn = dsn
401
+ self.applied: list[str] = []
402
+ self._started = True
403
+
404
+ def dsn(self) -> str:
405
+ return self._external_dsn
406
+
407
+ def start(self) -> None:
408
+ self.probe()
409
+
410
+ def stop(self) -> None:
411
+ return
412
+
413
+ def save_to(self, path: str | Path) -> None:
414
+ # Compose owns the schema and reruns the repository's migrations/initialisers whenever
415
+ # the project is recreated. The harness snapshot therefore owns only mutable rows and
416
+ # counters, avoiding a second generated schema that can drift from the submitted code.
417
+ from . import Held
418
+
419
+ Held.save_to(self, path)
420
+
421
+ def load_from(self, path: str | Path) -> None:
422
+ from . import Held
423
+
424
+ Held.load_from(self, path)
425
+
426
+
427
+ def _adapt(value: Any, data_type: str = "") -> Any:
428
+ """Hand back a value in the form psycopg will write.
429
+
430
+ A list in a JSON column must be wrapped, while a list in an ARRAY column must remain a list
431
+ so psycopg emits a native Postgres array. The authored setup crosses JSON boundaries before
432
+ it reaches a store, and some producers consequently leave a structured array as the JSON
433
+ string ``"[]"``. PostgreSQL interprets a bound string as its own array-literal syntax, where
434
+ square brackets mean dimensions rather than values, and rejects it. Decode only an actual
435
+ JSON list for an ARRAY column; arbitrary strings remain arbitrary strings and PostgreSQL can
436
+ enforce the declared element type.
437
+
438
+ Scenario setup may likewise represent a boolean as ``0``/``1`` even though PostgreSQL
439
+ deliberately does not implicitly cast a bound smallint to boolean. Normalize only the
440
+ finite, unambiguous boolean vocabulary; using ``bool(value)`` here would silently turn values
441
+ such as ``2`` or ``"disabled"`` into true.
442
+ """
443
+ normalized_type = data_type.strip().lower()
444
+ if value is not None and normalized_type == "boolean":
445
+ if isinstance(value, bool):
446
+ return value
447
+ if isinstance(value, int) and value in (0, 1):
448
+ return bool(value)
449
+ if isinstance(value, str):
450
+ normalized_value = value.strip().lower()
451
+ if normalized_value in {"true", "t", "1"}:
452
+ return True
453
+ if normalized_value in {"false", "f", "0"}:
454
+ return False
455
+ raise StoreError(
456
+ "a PostgreSQL boolean column received "
457
+ f"{value!r}; expected true/false or the equivalent 1/0"
458
+ )
459
+ if value is not None and normalized_type == "array":
460
+ if isinstance(value, list):
461
+ return value
462
+ if isinstance(value, tuple):
463
+ return list(value)
464
+ if isinstance(value, str) and value.strip().startswith("["):
465
+ try:
466
+ decoded = json.loads(value)
467
+ except json.JSONDecodeError as exc:
468
+ raise StoreError(
469
+ "a PostgreSQL array column received malformed JSON array text "
470
+ f"{value!r}"
471
+ ) from exc
472
+ if not isinstance(decoded, list): # pragma: no cover - guarded by '['
473
+ raise StoreError(
474
+ f"a PostgreSQL array column received {value!r}; expected a JSON list"
475
+ )
476
+ return decoded
477
+ if value is not None and normalized_type in ("json", "jsonb"):
478
+ from psycopg.types.json import Jsonb
479
+
480
+ return Jsonb(value)
481
+ return value
@@ -0,0 +1,202 @@
1
+ """Proving a store, without knowing which engine it is.
2
+
3
+ The build stage writes the engine-specific half: which image, how to read what it holds, how
4
+ to put it back. That half is written per agent, by a model, against an engine nobody vetted in
5
+ advance -- so the only thing standing between a subtly wrong reset and a suite of results that
6
+ mean nothing is this file.
7
+
8
+ Everything here is pure code and engine-independent. It never issues a query of its own,
9
+ because it cannot know the dialect; the one piece of engine-specific material it needs is a
10
+ ``mutation`` -- any statement that changes something -- and even that is checked before it is
11
+ trusted, since a mutation that does nothing would make a broken restore look perfect.
12
+
13
+ The sharp one is ``ids do not drift``. Rows going back is easy and most wrong restores manage
14
+ it; what they miss is the counter behind the rows, so the next scenario's first insert gets an
15
+ id continuing from the last one. Rather than ask what a counter is called on this engine --
16
+ which is exactly the kind of thing we cannot know -- the same mutation is run twice from the
17
+ same starting point and the two results are compared. Any drift, in anything, shows up as a
18
+ difference.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from typing import Any, Callable
24
+
25
+ from ..probe import ProbeReport, ProbeResult
26
+ from . import Snapshot, Store
27
+
28
+ STORE = "store"
29
+ BITES = "bites"
30
+
31
+ # A check over a proven store: a sentence when something is wrong, None when it held.
32
+ Check = Callable[[Store], "str | None"]
33
+
34
+ # Whether the checks themselves can fail is asked elsewhere, by ``world/mutate.py``: it damages
35
+ # the whole world rather than only emptying the store, silences every tool as well, and runs each
36
+ # kind of damage against its own restored copy. Two gates asking the same question in different
37
+ # words is how one of them quietly stops being run, so there is deliberately only the one.
38
+
39
+
40
+ def _result(
41
+ name: str, passed: bool, detail: str = "", kind: str = STORE
42
+ ) -> ProbeResult:
43
+ return ProbeResult(name=name, kind=kind, passed=passed, detail=detail)
44
+
45
+
46
+ def prove_store(store: Store, mutation: str) -> ProbeReport:
47
+ """Run a store through what it has to survive before any scenario is written against it.
48
+
49
+ ``mutation`` is anything the engine accepts that changes what it holds -- one insert is
50
+ plenty. It comes from the build stage because it is the one part of this that has to be
51
+ written in the engine's own language.
52
+
53
+ A failure here is ours, never the agent's. Nothing in this function involves the agent, so
54
+ a report with anything red means the environment is not yet a thing worth measuring against.
55
+ """
56
+ report = ProbeReport()
57
+
58
+ try:
59
+ baseline = store.freeze()
60
+ except Exception as exc: # noqa: BLE001 - a store that cannot be frozen fails here
61
+ report.results.append(_result("can be frozen", False, f"freeze raised: {exc}"))
62
+ return report
63
+ report.results.append(_result("can be frozen", True))
64
+
65
+ # Migrations that did not run leave a store with nothing in it, and every check written
66
+ # afterwards would pass or fail for reasons that have nothing to do with the agent.
67
+ if not baseline.rows:
68
+ report.results.append(
69
+ _result(
70
+ "holds a schema",
71
+ False,
72
+ "the store has no tables at all, so its migrations did not run",
73
+ )
74
+ )
75
+ return report
76
+ report.results.append(
77
+ _result("holds a schema", True, f"{len(baseline.rows)} tables")
78
+ )
79
+
80
+ seeded = sum(len(rows) for rows in baseline.rows.values())
81
+ report.results.append(
82
+ _result(
83
+ "holds a seed",
84
+ seeded > 0,
85
+ f"{seeded} rows"
86
+ if seeded
87
+ else "every table is empty, so nothing can be presumed",
88
+ )
89
+ )
90
+
91
+ # -- the mutation has to be worth something before it can prove anything ------------
92
+ try:
93
+ store.apply(mutation)
94
+ except Exception as exc: # noqa: BLE001 - the caller's statement, reported as given
95
+ report.results.append(_result("the mutation runs", False, f"{exc}"))
96
+ return report
97
+ report.results.append(_result("the mutation runs", True))
98
+
99
+ mutated = store.state()
100
+ if mutated == baseline.rows:
101
+ report.results.append(
102
+ _result(
103
+ "the mutation moves it",
104
+ False,
105
+ "the store is unchanged after it, so it cannot prove a restore works",
106
+ )
107
+ )
108
+ return report
109
+ report.results.append(_result("the mutation moves it", True))
110
+
111
+ # -- putting it back has to be exact -------------------------------------------------
112
+ try:
113
+ store.restore(baseline)
114
+ except Exception as exc: # noqa: BLE001
115
+ report.results.append(_result("restore runs", False, f"restore raised: {exc}"))
116
+ return report
117
+ report.results.append(_result("restore runs", True))
118
+
119
+ back = store.state()
120
+ report.results.append(
121
+ _result(
122
+ "restore is exact",
123
+ back == baseline.rows,
124
+ "" if back == baseline.rows else _difference(baseline.rows, back),
125
+ )
126
+ )
127
+
128
+ # -- and it has to put back what is behind the rows, not only the rows ---------------
129
+ try:
130
+ store.apply(mutation)
131
+ again = store.state()
132
+ except Exception as exc: # noqa: BLE001
133
+ report.results.append(_result("ids do not drift", False, f"{exc}"))
134
+ return report
135
+
136
+ report.results.append(
137
+ _result(
138
+ "ids do not drift",
139
+ again == mutated,
140
+ ""
141
+ if again == mutated
142
+ else (
143
+ "the same change from the same starting point produced something different "
144
+ "the second time, so the restore left a counter where it was: "
145
+ + _difference(mutated, again)
146
+ ),
147
+ )
148
+ )
149
+
150
+ store.restore(baseline)
151
+ report.results.append(_result("restore repeats", store.state() == baseline.rows))
152
+ return report
153
+
154
+
155
+ def prove_checks_bite(
156
+ store: Store, checks: dict[str, Check], baseline: Snapshot | None = None
157
+ ) -> ProbeReport:
158
+ """Empty a proven store and reject any check that still reports success."""
159
+ baseline = baseline or store.freeze()
160
+ report = ProbeReport()
161
+ store.restore(Snapshot())
162
+ try:
163
+ for name, check in checks.items():
164
+ try:
165
+ complaint = check(store)
166
+ except Exception as exc: # noqa: BLE001 - raising still notices the damage
167
+ complaint = f"raised {type(exc).__name__}: {exc}"
168
+ report.results.append(
169
+ _result(
170
+ name,
171
+ complaint is not None,
172
+ ""
173
+ if complaint is not None
174
+ else "held against an empty store, so it is not checking the environment",
175
+ kind=BITES,
176
+ )
177
+ )
178
+ finally:
179
+ store.restore(baseline)
180
+ return report
181
+
182
+
183
+ def _difference(expected: dict[str, Any], found: dict[str, Any]) -> str:
184
+ """The first place two states disagree, said plainly.
185
+
186
+ Whole-state diffs are unreadable at any real size, and the first disagreement is almost
187
+ always the whole story.
188
+ """
189
+ for table in sorted(set(expected) | set(found)):
190
+ before, after = expected.get(table), found.get(table)
191
+ if before == after:
192
+ continue
193
+ if before is None:
194
+ return f"{table} appeared"
195
+ if after is None:
196
+ return f"{table} disappeared"
197
+ if len(before) != len(after):
198
+ return f"{table}: {len(before)} rows expected, {len(after)} found"
199
+ for index, (one, two) in enumerate(zip(before, after)):
200
+ if one != two:
201
+ return f"{table} row {index}: expected {one}, found {two}"
202
+ return "no difference found, which should not happen"