agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,538 @@
1
+ """The hosted world handle: the shipped world vocabulary, backed by one world's own postgres.
2
+
3
+ `GeneratedWorld` is a database an agent's own generated handlers reach through `Db`. A hosted
4
+ world has no handlers to generate — the tables are whatever the agent's own migrations made, and
5
+ what a scenario needs is the same six-verb surface (`state`, `put`, `change`, `drop`, `call`,
6
+ `query`) built directly on the store, with nothing to adopt or reimplement per agent. Everything
7
+ this module refuses, it refuses before the database sees it, so a scenario's mistake reads as a
8
+ message naming what it did wrong rather than a `KeyError` or a driver traceback three layers down.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import random
14
+ import re
15
+ from collections.abc import Mapping, Sequence
16
+ from typing import Any
17
+
18
+ from .errors import (
19
+ WorldQueryRejected,
20
+ WorldReadOnly,
21
+ WorldReservedName,
22
+ WorldStateTooLarge,
23
+ WorldUnavailable,
24
+ WorldUsageError,
25
+ )
26
+ from .runtime import Call
27
+ from .stores.postgres import PostgresStore
28
+
29
+ # The harness's own isolation canary. It exists to prove worlds are really separate from each
30
+ # other, never to hold scenario data, so scenario code is never allowed to see or touch it.
31
+ CONFORMANCE_TABLE = "_alk_conformance"
32
+
33
+ # Measured once, at baseline freeze, by the provisioner — never recomputed here. A table over
34
+ # this stays over it for the whole run; nothing a scenario does can move which tables raise.
35
+ STATE_ROW_CAP = 5000
36
+
37
+ # The only statement shapes `query()` accepts. Anything else is refused before it reaches the
38
+ # database's own read-only transaction, so a statement that was never going to be allowed fails
39
+ # on a message naming why rather than a lock error from three layers down.
40
+ _READ_KEYWORDS = {"select", "with", "values"}
41
+
42
+ # The read-only view's fallback answer is restricted to exactly this vocabulary, so a genuinely
43
+ # unknown attribute — a capability probe, a dunder, a typo — still reads as a plain
44
+ # `AttributeError` instead of masquerading as a write refusal.
45
+ _WRITE_VERBS = frozenset({"put", "change", "drop", "call"})
46
+
47
+
48
+ class HostedWorld:
49
+ """The `World` surface, backed by this scenario's own logical postgres database.
50
+
51
+ One handle per scenario, over one short-lived autocommit connection per operation — nothing
52
+ held open, which is what lets `reset` drop the database out from under a discarded world.
53
+ `world_index` and `rng` are plain data: the index is for diagnostics only, and the generator
54
+ is the only sanctioned source of randomness scenario code may use.
55
+ """
56
+
57
+ def __init__(
58
+ self,
59
+ store: PostgresStore,
60
+ world_index: int,
61
+ rng: random.Random,
62
+ baseline_row_counts: Mapping[str, int],
63
+ ) -> None:
64
+ """Wrap `store` as the `World` surface for one scenario.
65
+
66
+ `baseline_row_counts` is keyed by the bare `pg_tables.tablename` value — no schema
67
+ prefix, no quoting — for every table `public` held when the baseline was frozen. A
68
+ visible table missing from this map has no measured cap to enforce, so the coverage
69
+ check below fails construction outright rather than waiting for a scenario's first
70
+ access to discover a provisioning gap through its own retry.
71
+ """
72
+ self._store = store
73
+ self.world_index = world_index
74
+ self.rng = rng
75
+ # Row counts as they stood when the baseline was frozen, keyed by table. Never
76
+ # re-measured: a live count would make the cap depend on what a scenario already wrote,
77
+ # and the whole point is that it is decided before any scenario runs.
78
+ self._baseline_row_counts = dict(baseline_row_counts)
79
+ self._require_baseline_coverage(self._visible_tables())
80
+
81
+ # -- reading ------------------------------------------------------------------------------
82
+
83
+ def state(self, table: str | None = None) -> dict[str, list[dict[str, Any]]]:
84
+ """A snapshot of the public schema, or of one table in it.
85
+
86
+ Bare `state()` leaves an over-cap table out of the snapshot rather than raising through
87
+ it — one seeded audit table must not make the primary read verb inert for the whole
88
+ run. A table nothing measured at baseline freeze gets the same treatment: the only way
89
+ one exists is a table the agent under test created since construction, and nothing it
90
+ does during a call may decide whether bare `state()` raises. Naming either kind of table
91
+ explicitly (`state("big_table")`) still raises: the exclusion is a property of the
92
+ snapshot, not a way to read the table around its own cap. If every visible table is
93
+ over-cap or unmeasured, the exclusion would leave the snapshot `{}` — the one thing
94
+ state() must never return, since an empty snapshot reads as an observation and makes a
95
+ negative check pass on a world nobody actually looked at — so that case raises too,
96
+ naming the tables it would have excluded. The exclusion itself happens at the read: the
97
+ store is asked for only the included tables, not for every table with the excluded ones
98
+ thrown away afterward, so one huge seeded table can no longer make every bare `state()`
99
+ pay to materialise and discard rows nobody asked to see.
100
+ """
101
+ self._reject_reserved(table)
102
+ if table is not None:
103
+ visible = self._visible_tables()
104
+ self._require_nonempty_schema(visible)
105
+ if table not in visible:
106
+ raise WorldUsageError(
107
+ f"{table!r} is not a table in this world; it holds {sorted(visible)}."
108
+ )
109
+ count = self._row_count(table)
110
+ if count > STATE_ROW_CAP:
111
+ raise WorldStateTooLarge(
112
+ f"{table!r} held {count} rows when the baseline "
113
+ f"was frozen, over the {STATE_ROW_CAP}-row cap; state() will not read it "
114
+ "back."
115
+ )
116
+ return {table: self._store.table(table)}
117
+
118
+ names = self._visible_tables()
119
+ self._require_nonempty_schema(names)
120
+ # A table nothing measured at freeze is treated the same as an over-cap one here, not
121
+ # routed through `_row_count`'s typed refusal — that refusal is for a scenario naming a
122
+ # table explicitly; the bare snapshot must not go unavailable over a table the agent
123
+ # itself created since construction (the only actor besides the provisioner that can).
124
+ included = [
125
+ name
126
+ for name in names
127
+ if name in self._baseline_row_counts
128
+ and self._baseline_row_counts[name] <= STATE_ROW_CAP
129
+ ]
130
+ if not included:
131
+ raise WorldStateTooLarge(
132
+ f"every table this world holds — {sorted(names)} — is over the "
133
+ f"{STATE_ROW_CAP}-row cap or was never measured at baseline freeze; state() "
134
+ "will not return {} in their place."
135
+ )
136
+ # One connection, asked for only the included tables — a bare state() used to open a
137
+ # fresh connection per table (and a second one just to look up its primary key) to read
138
+ # every table including the over-cap ones, then throw the over-cap rows away here.
139
+ return self._store.state(only=included)
140
+
141
+ def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
142
+ """The read escape hatch: one statement, on a transaction that cannot write.
143
+
144
+ The database's own read-only transaction is the actual guard; this token check only
145
+ makes the common mistake — a stray write, a second statement — fail with a reason
146
+ attached instead of a lock error from underneath.
147
+ """
148
+ _reject_unless_read(sql)
149
+ return self._store.query(sql, tuple(params))
150
+
151
+ # -- writing ------------------------------------------------------------------------------
152
+
153
+ def put(self, collection: str, record: Mapping[str, Any], *, key: str = "") -> dict[str, Any]:
154
+ """Insert one record; return exactly what the table stored, generated key included.
155
+
156
+ `key` exists only to keep this signature a superset of `GeneratedWorld.put`; a hosted
157
+ table already knows its own key — the column its own migrations gave it. Scenario
158
+ authoring has emitted both ``key=<primary-key column>`` and
159
+ ``key=<primary-key value>`` for table-backed worlds, while ``GeneratedWorld`` harmlessly
160
+ ignores either hint. Accept both redundant spellings only when the table has one primary
161
+ key and the record contains it; the value spelling must exactly equal the record's key.
162
+ Continue rejecting arbitrary values and non-primary columns so a mapping-style setup
163
+ cannot silently acquire different semantics after moving to hosted Postgres.
164
+ """
165
+ self._reject_reserved(collection)
166
+ if collection not in self._visible_tables():
167
+ raise WorldUsageError(
168
+ f"{collection!r} is not a table in this world; hosted worlds cannot invent "
169
+ "one, so put() only reaches what the agent's own migrations made."
170
+ )
171
+ if key:
172
+ primary_key = self._primary_key_order(collection)
173
+ primary_key_column = primary_key[0] if len(primary_key) == 1 else ""
174
+ names_primary_key = key == primary_key_column and key in record
175
+ matches_primary_key_value = (
176
+ bool(primary_key_column)
177
+ and primary_key_column in record
178
+ and key == str(record[primary_key_column])
179
+ )
180
+ if not (names_primary_key or matches_primary_key_value):
181
+ raise WorldUsageError(
182
+ "a hosted table's key is the table's own; key= is accepted only when it "
183
+ "names the table's single-column primary key or exactly matches that key's "
184
+ "value in the record."
185
+ )
186
+ return self._store.add(collection, dict(record))
187
+
188
+ def change(
189
+ self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
190
+ ) -> int:
191
+ """Update matching records; return how many changed."""
192
+ self._reject_reserved(collection)
193
+ if collection not in self._visible_tables():
194
+ raise WorldUsageError(
195
+ f"{collection!r} is not a table in this world; hosted worlds cannot invent "
196
+ "one, so change() only reaches what the agent's own migrations made."
197
+ )
198
+ by = by or self._resolve_by(collection)
199
+ if by not in self._table_columns(collection):
200
+ raise WorldUsageError(
201
+ f"change({collection!r}, {key!r}, ...) was given by={by!r}, which is not a "
202
+ f"column of {collection!r}."
203
+ )
204
+ return self._store.amend(collection, key, dict(changes), by=by)
205
+
206
+ def drop(self, collection: str, key: str = "", *, by: str = "") -> int:
207
+ """Delete matching records, or every record when `key` is empty; return the count."""
208
+ self._reject_reserved(collection)
209
+ if collection not in self._visible_tables():
210
+ raise WorldUsageError(
211
+ f"{collection!r} is not a table in this world; hosted worlds cannot invent "
212
+ "one, so drop() only reaches what the agent's own migrations made."
213
+ )
214
+ if key:
215
+ by = by or self._resolve_by(collection)
216
+ if by not in self._table_columns(collection):
217
+ raise WorldUsageError(
218
+ f"drop({collection!r}, {key!r}) was given by={by!r}, which is not a "
219
+ f"column of {collection!r}."
220
+ )
221
+ return self._store.remove(collection, key, by=by)
222
+
223
+ def call(self, name: str, arguments: Mapping[str, Any] | None = None) -> Call:
224
+ """Play one of the agent's own tools against this world.
225
+
226
+ Not implemented: the `http_tool` evidence seam's wire format is not pinned anywhere in
227
+ the contracts yet, and guessing at a shape here would ship one nobody agreed to and
228
+ scenario code would end up depending on. Raising unconditionally is the honest answer
229
+ until the evidence layer pins it — run-time scenario code should not need this anyway,
230
+ since `setup` already runs at proof time and the data verbs cover the rest.
231
+ """
232
+ raise WorldUnavailable(
233
+ "the http_tool shim wire format is not yet pinned by the contracts — report, "
234
+ "don't guess."
235
+ )
236
+
237
+ def read_only(self) -> "ReadOnlyWorld":
238
+ """The view `ready` and `check` run against: every write verb refuses outright.
239
+
240
+ A check able to write could not be told apart from one that quietly repaired what it was
241
+ supposed to be grading, so this is what stands between those two functions and the real
242
+ handle.
243
+ """
244
+ return ReadOnlyWorld(self)
245
+
246
+ # -- internal -----------------------------------------------------------------------------
247
+
248
+ def _reject_reserved(self, collection: str | None) -> None:
249
+ if collection == CONFORMANCE_TABLE:
250
+ raise WorldReservedName(
251
+ f"{collection!r} is the harness's own conformance canary, not scenario data; "
252
+ "it never appears to scenario code."
253
+ )
254
+
255
+ def _require_nonempty_schema(self, visible: list[str]) -> None:
256
+ if not visible:
257
+ raise WorldUnavailable(
258
+ "this world's public schema holds no tables to observe; a postgres world with "
259
+ "nothing in it is not a world state() can honestly report on."
260
+ )
261
+
262
+ def _require_baseline_coverage(self, names: list[str]) -> None:
263
+ """Refuse rather than assume a table the baseline never measured is under the cap.
264
+
265
+ A missing entry used to default to a row count of 0, which would let a table nobody
266
+ measured at freeze read back as though it were known to be small; failing loud here is
267
+ the whole point of deciding the cap before any scenario runs instead of guessing at it.
268
+ Called once, from `__init__`: the table set and the baseline are both fixed for the
269
+ life of the handle, so checking again on every later access would only repeat a answer
270
+ construction already gave.
271
+ """
272
+ missing = [name for name in names if name not in self._baseline_row_counts]
273
+ if missing:
274
+ raise WorldUnavailable(
275
+ f"the baseline row counts this world was built with never measured "
276
+ f"{sorted(missing)}; state()'s cap cannot be decided for a table nobody "
277
+ "measured at freeze."
278
+ )
279
+
280
+ def _row_count(self, table: str) -> int:
281
+ """This table's baseline row count, or the same typed refusal `_require_baseline_coverage`
282
+ gives a construction-time gap — for a table that only shows up afterward, so it never
283
+ reaches the caller as a bare `KeyError`.
284
+ """
285
+ try:
286
+ return self._baseline_row_counts[table]
287
+ except KeyError:
288
+ raise WorldUnavailable(
289
+ f"the baseline row counts this world was built with never measured "
290
+ f"{table!r}; state()'s cap cannot be decided for a table nobody measured at "
291
+ "freeze."
292
+ ) from None
293
+
294
+ def _visible_tables(self) -> list[str]:
295
+ rows = self._store.query(
296
+ "SELECT tablename FROM pg_tables WHERE schemaname = 'public' ORDER BY tablename"
297
+ )
298
+ return [row["tablename"] for row in rows if row["tablename"] != CONFORMANCE_TABLE]
299
+
300
+ def _table_columns(self, table: str) -> set[str]:
301
+ rows = self._store.query(
302
+ "SELECT column_name FROM information_schema.columns "
303
+ "WHERE table_schema = 'public' AND table_name = %s",
304
+ (table,),
305
+ )
306
+ return {row["column_name"] for row in rows}
307
+
308
+ def _primary_key_order(self, table: str) -> list[str]:
309
+ """This table's primary key columns, in index order.
310
+
311
+ Joined through `pg_class`/`pg_namespace` rather than a `%s::regclass` cast over an
312
+ f-string, so a table name is only ever a bound value — an embedded `"` (a table created
313
+ as `CREATE TABLE "we""ird" (...)`) is just a character in that value instead of
314
+ something a regclass cast has to parse.
315
+ """
316
+ rows = self._store.query(
317
+ """
318
+ SELECT a.attname
319
+ FROM pg_index i
320
+ JOIN pg_attribute a ON a.attrelid = i.indrelid AND a.attnum = ANY(i.indkey)
321
+ JOIN pg_class c ON c.oid = i.indrelid
322
+ JOIN pg_namespace n ON n.oid = c.relnamespace
323
+ WHERE n.nspname = 'public' AND c.relname = %s AND i.indisprimary
324
+ ORDER BY array_position(i.indkey, a.attnum)
325
+ """,
326
+ (table,),
327
+ )
328
+ return [row["attname"] for row in rows]
329
+
330
+ def _resolve_by(self, table: str) -> str:
331
+ """The column to key `change`/`drop` on when scenario code did not name one.
332
+
333
+ A single-column primary key is the only case unambiguous enough to guess; no primary
334
+ key at all, or a composite one, means the store genuinely cannot tell which column
335
+ `key` names, so the scenario has to say.
336
+ """
337
+ columns = self._primary_key_order(table)
338
+ if len(columns) == 1:
339
+ return columns[0]
340
+ raise WorldUsageError(
341
+ f"change/drop on {table!r} needs by=<column>; it has no single-column primary "
342
+ "key to default to."
343
+ )
344
+
345
+
346
+ class ReadOnlyWorld:
347
+ """A `World` whose write verbs are refused before they reach `HostedWorld` at all."""
348
+
349
+ def __init__(self, world: HostedWorld) -> None:
350
+ # Name-mangled so code outside this class reaching for `._world` cannot casually
351
+ # recover the writable handle a "read-only" view exists to stand in front of.
352
+ self.__world = world
353
+ self.world_index = world.world_index
354
+ self.rng = world.rng
355
+
356
+ def state(self, table: str | None = None) -> dict[str, list[dict[str, Any]]]:
357
+ return self.__world.state(table)
358
+
359
+ def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
360
+ return self.__world.query(sql, params)
361
+
362
+ def put(self, collection: str, record: Mapping[str, Any], *, key: str = "") -> dict[str, Any]:
363
+ raise WorldReadOnly(
364
+ f"put({collection!r}, ...) reached a read-only handle; ready() and check() only "
365
+ "ever observe a run."
366
+ )
367
+
368
+ def change(
369
+ self, collection: str, key: str, changes: Mapping[str, Any], *, by: str = ""
370
+ ) -> int:
371
+ raise WorldReadOnly(
372
+ f"change({collection!r}, {key!r}, ...) reached a read-only handle; ready() and "
373
+ "check() only ever observe a run."
374
+ )
375
+
376
+ def drop(self, collection: str, key: str = "", *, by: str = "") -> int:
377
+ raise WorldReadOnly(
378
+ f"drop({collection!r}, {key!r}) reached a read-only handle; ready() and check() "
379
+ "only ever observe a run."
380
+ )
381
+
382
+ def call(self, name: str, arguments: Mapping[str, Any] | None = None) -> Call:
383
+ raise WorldReadOnly(
384
+ f"call({name!r}, ...) reached a read-only handle; ready() and check() only ever "
385
+ "observe a run."
386
+ )
387
+
388
+ def __getattr__(self, name: str) -> Any:
389
+ """A write verb `HostedWorld` grows later, with no override here yet, still reads as
390
+ `WorldReadOnly` rather than a plain `AttributeError` indistinguishable from a typo.
391
+
392
+ Restricted to that known vocabulary and never to dunders: this repo's own runner code
393
+ reaches for `hasattr(world, "forward")` and `getattr(world, "runtime_tools", set())` on
394
+ world objects, and both only work through the ordinary `AttributeError` those tools
395
+ expect from a name that is simply not there, not from a refusal that happens to look
396
+ like one.
397
+ """
398
+ if name.startswith("__") and name.endswith("__"):
399
+ raise AttributeError(name)
400
+ if name in _WRITE_VERBS:
401
+ raise WorldReadOnly(
402
+ f"{name!r} reached a read-only handle; ready() and check() only ever observe "
403
+ "a run."
404
+ )
405
+ raise AttributeError(name)
406
+
407
+
408
+ def _is_word_char(char: str) -> bool:
409
+ return char.isalnum() or char == "_"
410
+
411
+
412
+ _DOLLAR_TAG = re.compile(r"\$([A-Za-z_][A-Za-z0-9_]*)?\$")
413
+
414
+
415
+ def _dollar_quote_end(sql: str, start: int) -> int | None:
416
+ """The index just past the matching closing `$tag$`, if `sql[start:]` opens one.
417
+
418
+ `None` if `start` is not a dollar-quote opener at all — a `$1` parameter placeholder, or a
419
+ bare `$`, never matches the tag grammar and is left for the caller to treat as an ordinary
420
+ character. An opener with no matching close consumes to the end of the string, the same
421
+ fate an unterminated `'`/`"` span already gets below.
422
+ """
423
+ opener = _DOLLAR_TAG.match(sql, start)
424
+ if opener is None:
425
+ return None
426
+ delimiter = opener.group(0)
427
+ end = sql.find(delimiter, opener.end())
428
+ return len(sql) if end == -1 else end + len(delimiter)
429
+
430
+
431
+ def _blank(sql: str, quote_chars: str) -> str:
432
+ """`sql` with every comment blanked, plus any quoted span opened by a character in
433
+ `quote_chars`, plus every dollar-quoted `$tag$...$tag$` span.
434
+
435
+ Two callers need two different blindnesses: the shape check does not care what a string
436
+ literal's characters are, so it blanks both quote styles; the reserved-name check must not
437
+ let a quoted identifier hide inside a span it no longer looks at, so it blanks only string
438
+ literals and leaves double-quoted identifiers as text. Dollar-quoting is blanked for both
439
+ callers regardless of `quote_chars` — it is never an identifier, only ever a literal, and a
440
+ `'` or `;` sitting inside one is exactly what both callers must not see as SQL.
441
+
442
+ A `'` immediately after a bare `E`/`e` opens an escape string, where a `\\` escapes whatever
443
+ follows it the same way doubling the quote does. Missing that let a `\\'` inside one close
444
+ the literal early: the real closing quote right after it then read as opening a fresh span,
445
+ and everything up to the next quote — semicolon, second statement and all — vanished into
446
+ it as though it were still part of the string.
447
+ """
448
+ out: list[str] = []
449
+ i, n = 0, len(sql)
450
+ while i < n:
451
+ if sql.startswith("--", i):
452
+ end = sql.find("\n", i)
453
+ i = n if end == -1 else end + 1
454
+ continue
455
+ if sql.startswith("/*", i):
456
+ end = sql.find("*/", i + 2)
457
+ i = n if end == -1 else end + 2
458
+ continue
459
+ char = sql[i]
460
+ if char == "$":
461
+ dollar_end = _dollar_quote_end(sql, i)
462
+ if dollar_end is not None:
463
+ i = dollar_end
464
+ out.append(" ")
465
+ continue
466
+ if char in quote_chars:
467
+ escapes = (
468
+ char == "'"
469
+ and i > 0
470
+ and sql[i - 1] in "Ee"
471
+ and (i == 1 or not _is_word_char(sql[i - 2]))
472
+ )
473
+ i += 1
474
+ while i < n:
475
+ if escapes and sql[i] == "\\":
476
+ i += 2
477
+ continue
478
+ if sql[i] == char:
479
+ if sql[i : i + 2] == char * 2: # an escaped quote inside the literal
480
+ i += 2
481
+ continue
482
+ i += 1
483
+ break
484
+ i += 1
485
+ out.append(" ")
486
+ continue
487
+ out.append(char)
488
+ i += 1
489
+ return "".join(out)
490
+
491
+
492
+ def _sql_skeleton(sql: str) -> str:
493
+ """`sql` with every comment and string literal blanked out to a single space.
494
+
495
+ The token check only needs to know what kind of statement this is and how many of them
496
+ there are; without this a semicolon or the word FOR inside a quoted value would count as
497
+ SQL, and the check would end up rejecting a value instead of a statement.
498
+ """
499
+ return _blank(sql, "'\"")
500
+
501
+
502
+ def _names_the_reserved_table(sql: str) -> bool:
503
+ """Whether `_alk_conformance` appears anywhere as an identifier, quoted or not.
504
+
505
+ Blanked for string literals only, deliberately not double-quoted identifiers — an
506
+ identifier is exactly where this name could hide from a check that blanked those away too.
507
+ The boundary either side of the name is `\\w`, not a quote: a `"` sitting right against it
508
+ is exactly the character that must not shield it, or `"_alk_conformance"` would read as
509
+ hidden the same way a comment or a literal already is.
510
+ """
511
+ identifiers = _blank(sql, "'")
512
+ pattern = rf"(?<!\w){re.escape(CONFORMANCE_TABLE)}(?!\w)"
513
+ return re.search(pattern, identifiers, re.IGNORECASE) is not None
514
+
515
+
516
+ def _reject_unless_read(sql: str) -> None:
517
+ body = _sql_skeleton(sql).strip()
518
+ if not body:
519
+ raise WorldQueryRejected("query() was given nothing to run.")
520
+ unterminated = body[:-1].strip() if body.endswith(";") else body
521
+ if ";" in unterminated:
522
+ raise WorldQueryRejected("query() runs one statement; this text holds more than one.")
523
+ leading = re.match(r"[A-Za-z_]+", unterminated)
524
+ word = leading.group(0).lower() if leading else ""
525
+ if word not in _READ_KEYWORDS:
526
+ raise WorldQueryRejected(
527
+ f"query() only reads: it takes SELECT, WITH or VALUES, not {word or sql[:20]!r}."
528
+ )
529
+ if re.search(r"\bfor\s+(update|share)\b", unterminated, re.IGNORECASE):
530
+ raise WorldQueryRejected(
531
+ "query() runs on a read-only transaction; FOR UPDATE/FOR SHARE lock rows for a "
532
+ "write that can never follow."
533
+ )
534
+ if _names_the_reserved_table(sql):
535
+ raise WorldQueryRejected(
536
+ f"query() refuses to name {CONFORMANCE_TABLE!r}; it is the harness's own "
537
+ "conformance canary, not scenario data."
538
+ )