agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1018 @@
1
+ """The tools the harness offers a session, and the gates behind them.
2
+
3
+ The model does judgement; these do the parts that must be exact. Validation lives inside the
4
+ tool rather than after the session, so a problem is returned into the conversation and fixed on
5
+ the next turn instead of surfacing once the session is already over.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ from .backends import qualified as qualified # noqa: F401 (re-export; callers import it here)
15
+ from .backends import tool, tool_server
16
+
17
+ from .contract import CALL_DIRECTIONS, MODALITIES, AgentContract, validate_contract
18
+
19
+ CONTRACT_SERVER = "contract"
20
+
21
+
22
+ def _ok(text: str) -> dict[str, Any]:
23
+ return {"content": [{"type": "text", "text": text}]}
24
+
25
+
26
+ # validate_contract returns short codes: they are stable, testable, and the same string every
27
+ # time. What a code means is a separate question, and answering it here keeps the codes exact
28
+ # while the message the model reads says what to actually do.
29
+ _GUIDANCE = {
30
+ "empty:agent": "the `agent` field is empty. A short lower-case name; it is only the "
31
+ "artifact folder's label",
32
+ "no-use-cases": "the `real_use_cases` field is empty — note the name, it is not "
33
+ "`use_cases`. List the concrete situations this agent handles, from its tools and data",
34
+ "duplicate-tool-names": "the same tool is listed twice; keep one entry per tool",
35
+ "types-for-unknown-args": "arg_types names an argument that is not in args. The names must "
36
+ "match the source exactly",
37
+ }
38
+
39
+
40
+ def _advice(code: str) -> str:
41
+ for key, said in _GUIDANCE.items():
42
+ if code.startswith(key) or key in code:
43
+ return f"{code} — {said}"
44
+ return code
45
+
46
+
47
+ def _problems(problems: list[str], arrived: list[str] | None = None) -> dict[str, Any]:
48
+ """Every problem at once, each with what to do about it.
49
+
50
+ All of them together, never one at a time: a gate that reveals the next problem only after
51
+ the last is fixed costs a full turn per problem and reads as though the rules are being
52
+ invented as it goes.
53
+
54
+ When the fields arrived under names this does not recognise, it says which names it got.
55
+ Without that the answer is "agent is empty, there are no tools" about a submission that
56
+ contained both, and the only way out is guessing at the packaging.
57
+ """
58
+ said = "Not accepted. Fix all of these and call submit_contract again:\n - " + (
59
+ "\n - ".join(_advice(problem) for problem in problems)
60
+ )
61
+ unrecognised = arrived is not None and not any(
62
+ key in arrived for key in ("agent", "tools", "real_use_cases")
63
+ )
64
+ if unrecognised:
65
+ said += (
66
+ f"\n\nWhat arrived was: {', '.join(arrived) or '(nothing)'}. None of those are "
67
+ "contract fields, so the fields were probably nested inside something or sent as "
68
+ "one JSON string. Send them as the tool's own top-level arguments — agent, tools, "
69
+ "real_use_cases and the rest — not wrapped in an outer object."
70
+ )
71
+ return {
72
+ "content": [{"type": "text", "text": said}],
73
+ "is_error": True,
74
+ }
75
+
76
+
77
+ _CONTRACT_KEYS = ("agent", "tools", "real_use_cases", "one_liner", "hard_constraints")
78
+
79
+
80
+ def _looks_like_a_contract(value: Any) -> bool:
81
+ return isinstance(value, dict) and any(key in value for key in _CONTRACT_KEYS)
82
+
83
+
84
+ def unwrapped(payload: dict[str, Any]) -> dict[str, Any]:
85
+ """The contract itself, however it was packaged.
86
+
87
+ A contract is a nested thing being described, so it arrives wrapped — ``{"contract": {...}}``
88
+ — or stringified, as JSON in a single argument, often enough to matter. In both the fields
89
+ are present and correct and only the packaging is wrong. Rejecting that teaches nothing
90
+ about the agent and costs a full turn, so it is unpacked; only an object that actually looks
91
+ like a contract is unwrapped, so a real field that happens to hold a dict is never mistaken
92
+ for an envelope.
93
+ """
94
+ if not isinstance(payload, dict):
95
+ payload = {}
96
+ if any(key in payload for key in ("agent", "tools", "real_use_cases")):
97
+ return payload
98
+ for value in payload.values():
99
+ if _looks_like_a_contract(value):
100
+ return value
101
+ if isinstance(value, str):
102
+ text = value.strip()
103
+ if text.startswith("```"):
104
+ # Fenced JSON: the model wrote it as it would in a message.
105
+ text = text.strip("`").removeprefix("json").strip()
106
+ if not text.startswith("{"):
107
+ continue
108
+ try:
109
+ parsed = json.loads(text)
110
+ except json.JSONDecodeError:
111
+ continue
112
+ if _looks_like_a_contract(parsed):
113
+ return parsed
114
+ for inner in parsed.values() if isinstance(parsed, dict) else []:
115
+ if _looks_like_a_contract(inner):
116
+ return inner
117
+ return payload
118
+
119
+
120
+ def accept_contract(
121
+ payload: dict[str, Any], destination: Path, *, source_root: Path | None = None
122
+ ) -> dict[str, Any]:
123
+ """The gate itself: validate, and write only if it passes.
124
+
125
+ A plain function rather than only a tool body, so the rule that decides whether a contract
126
+ is usable can be exercised and reasoned about without standing up a session.
127
+ """
128
+ arrived = sorted(payload) if isinstance(payload, dict) else [type(payload).__name__]
129
+ payload = _without_nulls(unwrapped(payload))
130
+ try:
131
+ contract = AgentContract.model_validate(payload)
132
+ except Exception as invalid:
133
+ return _problems([f"schema:{invalid}"[:600]], arrived)
134
+
135
+ problems = validate_contract(contract)
136
+ if source_root is not None:
137
+ from .source_tool_evidence import tool_evidence_problems
138
+
139
+ problems.extend(tool_evidence_problems(contract, source_root))
140
+ if problems:
141
+ return _problems(problems, arrived)
142
+
143
+ destination.mkdir(parents=True, exist_ok=True)
144
+ path = destination / "contract.json"
145
+ path.write_text(
146
+ json.dumps(contract.model_dump(), indent=2, ensure_ascii=False),
147
+ encoding="utf-8",
148
+ )
149
+ return _ok(
150
+ f"Accepted and saved to {path}.\n"
151
+ f"{len(contract.tools)} tools: {', '.join(sorted(contract.tool_names()))}\n"
152
+ f"{len(contract.hard_constraints)} rules, "
153
+ f"{len(contract.real_use_cases)} use cases, "
154
+ f"{len(contract.open_questions)} open questions."
155
+ )
156
+
157
+
158
+ def _without_nulls(value: Any) -> Any:
159
+ """Treat JSON null like an omitted optional field, including in nested records."""
160
+ if isinstance(value, dict):
161
+ return {
162
+ key: _without_nulls(item) for key, item in value.items() if item is not None
163
+ }
164
+ if isinstance(value, list):
165
+ return [_without_nulls(item) for item in value if item is not None]
166
+ return value
167
+
168
+
169
+ # Each selected eval is one judge call per call in the suite.
170
+ MOST_CHOSEN_EVALS = 6
171
+
172
+
173
+ def contract_tools(
174
+ destination: Path,
175
+ available_evals: list[dict[str, Any]] | None = None,
176
+ *,
177
+ source_root: Path | None = None,
178
+ ) -> Any:
179
+ """A server exposing ``submit_contract``, writing to ``destination`` on acceptance.
180
+
181
+ ``available_evals`` is the platform's own eval catalogue when the job carried one. Absent, the
182
+ contract simply chooses none, which is what every job did before the catalogue existed.
183
+ """
184
+ # Name to modality, empty or "any" for all. Refused here too, where it is still free to fix.
185
+ offered_modality = {
186
+ str(one.get("name") or "").strip(): str(one.get("modality") or "")
187
+ .strip()
188
+ .lower()
189
+ for one in (available_evals or [])
190
+ if isinstance(one, dict) and str(one.get("name") or "").strip()
191
+ }
192
+ offered = set(offered_modality)
193
+ # Each of these is a nudge, not a wall: the first submission missing something that is
194
+ # nearly always there gets sent back with directions, and a second submission is accepted.
195
+ # A gate with no way through would permanently block the rare agent that genuinely lacks it,
196
+ # and this stage cannot tell those two apart from the outside.
197
+ nudged: set[str] = set()
198
+
199
+ @tool(
200
+ "submit_contract",
201
+ "Submit the agent's testing contract: everything verifiably true about this agent, as "
202
+ "one flat object. Every field is described in the schema; fill in what the source "
203
+ "supports and leave the rest out.\n\n"
204
+ "It is validated when you call it. If anything is wrong you get the whole list back at "
205
+ "once, in terms of what to fix, and you submit again.",
206
+ # Nothing required, and that is deliberate. This layer runs before the tool body, so
207
+ # anything it rejects never reaches the code that could have understood it — a contract
208
+ # sent inside a wrapper is complete and correct, and is unwrapped a few lines below, but
209
+ # only if it gets there. accept_contract is the single gate; it reports every problem at
210
+ # once and says what to do about each.
211
+ #
212
+ # The descriptions are the point of this block. The schema is shown to the model before
213
+ # it calls anything, so what is written here is the difference between a correct first
214
+ # call and a sequence of rejected guesses.
215
+ schema(
216
+ {
217
+ "agent": {
218
+ "type": "string",
219
+ "description": "Short lower-case identifier, no spaces. Only a label for "
220
+ "the artifact folder.",
221
+ },
222
+ "one_liner": {
223
+ "type": "string",
224
+ "description": "One sentence: what this agent is for.",
225
+ },
226
+ "modality": {
227
+ "type": "string",
228
+ "enum": list(MODALITIES),
229
+ "description": "How a person reaches it, read from its runtime. A voice "
230
+ "session (LiveKit, telephony, TTS/STT) is voice; a text interface is chat; "
231
+ "a browser-driving agent is browser. This decides how it is later run.",
232
+ },
233
+ "call_direction": {
234
+ "type": "string",
235
+ "enum": list(CALL_DIRECTIONS),
236
+ "description": "Voice only, and read from the agent's own instructions rather "
237
+ "than guessed. Outbound if it places the call and the person is not expecting "
238
+ 'it ("you placed this call", "this is us calling about"); inbound if '
239
+ 'people dial in to it ("callers dial in", "thanks for calling"). '
240
+ "Leave unset for chat, which a person always starts. This decides how the "
241
+ "simulated person is briefed: someone who did not dial has no opening request "
242
+ "to make.",
243
+ },
244
+ "conversational": {
245
+ "type": "boolean",
246
+ "description": "True if a person talks with it turn by turn. False for an "
247
+ "agent given one task and left to it.",
248
+ },
249
+ "system_prompt_excerpt": {
250
+ "type": "string",
251
+ "description": "The agent's own instructions, quoted. Often lives away from "
252
+ "the main agent file.",
253
+ },
254
+ "hard_constraints": {
255
+ "type": "array",
256
+ "items": {"type": "string"},
257
+ "description": "Rules it must obey, in the source's own words. The agent "
258
+ "under test is told these and graded against them.",
259
+ },
260
+ "tools": {
261
+ "type": "array",
262
+ "description": "Every tool the agent really has. Everything downstream is "
263
+ "built from these, so a tool without its arguments cannot be tested.",
264
+ "items": {
265
+ "type": "object",
266
+ "properties": {
267
+ "name": {
268
+ "type": "string",
269
+ "description": "The exact callable name the model emits.",
270
+ },
271
+ "args": {
272
+ "type": "array",
273
+ "items": {"type": "string"},
274
+ "description": "Exact parameter names, in order.",
275
+ },
276
+ "arg_types": {
277
+ "type": "object",
278
+ "description": "Declared type per argument where the source "
279
+ 'states one: {"recipient_ids": "list[str]"}.',
280
+ },
281
+ "arg_values": {
282
+ "type": "object",
283
+ "description": "Real permitted values per argument where it is "
284
+ "constrained to a set, an enum or a lookup: "
285
+ '{"priority": ["low", "normal", "urgent"]}.',
286
+ },
287
+ "description": {"type": "string"},
288
+ "requires": {
289
+ "type": "array",
290
+ "items": {"type": "string"},
291
+ "description": "Tools that must have run before this one stops "
292
+ "refusing, read from a guard in its own body. Only state the agent "
293
+ "builds during the conversation counts: a quote taken, an option "
294
+ "selected. Identity established when the call opens does not, "
295
+ "since it holds before any tool runs. Empty means callable first "
296
+ "thing, and naming a tool that is not gated makes every future "
297
+ "test of it replay a preamble it never needed.",
298
+ },
299
+ },
300
+ # Nothing required: a tool genuinely taking no arguments is ordinary,
301
+ # and requiring args here rejects the whole contract because of one.
302
+ # That every tool has none is the real defect, and validate_contract
303
+ # is where it is caught, with an explanation.
304
+ },
305
+ },
306
+ "data_schema": {
307
+ "type": "object",
308
+ "description": "The shape of the records the agent works on: which fields "
309
+ "each kind of record has.",
310
+ },
311
+ "base_environment": {
312
+ "type": "object",
313
+ "description": "Its real starting data, reproduced exactly — including "
314
+ "anything that looks like a mistake. The world is a replica, not a "
315
+ "corrected version.",
316
+ },
317
+ "runtime_dependencies": {
318
+ "type": "array",
319
+ "description": "External runtime connections such as RTC transport and model "
320
+ "inference providers. These use the supplied configuration/credentials; "
321
+ "they are not business-world stores or services to recreate. Never put "
322
+ "business databases, files, queues or tools backends here.",
323
+ "items": {
324
+ "type": "object",
325
+ "properties": {
326
+ "name": {"type": "string"},
327
+ "kind": {
328
+ "type": "string",
329
+ "description": "transport or inference",
330
+ },
331
+ "what": {"type": "string"},
332
+ "engine": {"type": "string"},
333
+ "reached": {
334
+ "type": "object",
335
+ "properties": {
336
+ "dsn_env": {
337
+ "type": "string",
338
+ "description": "Environment variable NAME, never its secret value",
339
+ },
340
+ },
341
+ },
342
+ },
343
+ "required": ["name"],
344
+ },
345
+ },
346
+ "dependencies": {
347
+ "type": "array",
348
+ "description": "Business-world dependencies only. Put RTC transport and model "
349
+ "provider connections in runtime_dependencies instead. Everything this agent reaches for that has to exist before "
350
+ "it can work, so the next stage knows what to build. A datastore, a service "
351
+ "it calls over HTTP, a file it reads, a queue it publishes to. The world is "
352
+ "a sandbox and nothing reaches outside it, so each of these is built inside "
353
+ "it — the agent's call goes to something real that happens to be ours.",
354
+ "items": {
355
+ "type": "object",
356
+ "properties": {
357
+ "name": {"type": "string"},
358
+ "kind": {
359
+ "type": "string",
360
+ "description": "datastore, service, file, queue, or whatever "
361
+ "this actually is.",
362
+ },
363
+ "what": {
364
+ "type": "string",
365
+ "description": "What it holds or answers, and what the agent "
366
+ "needs from it.",
367
+ },
368
+ "used_by": {
369
+ "type": "array",
370
+ "items": {"type": "string"},
371
+ "description": "The tools that cannot work without it.",
372
+ },
373
+ "engine": {
374
+ "type": "string",
375
+ "description": "The exact database/service engine the agent uses.",
376
+ },
377
+ "version": {
378
+ "type": "string",
379
+ "description": "The engine version where the source pins one.",
380
+ },
381
+ "reached": {
382
+ "type": "object",
383
+ "description": "The agent's existing connection seam. Record "
384
+ "where secrets come from, never their values.",
385
+ "properties": {
386
+ "dsn_env": {"type": "string"},
387
+ "config_key": {"type": "string"},
388
+ "host": {"type": "string"},
389
+ "port": {"type": "integer"},
390
+ "database": {"type": "string"},
391
+ "user": {"type": "string"},
392
+ "password_from": {"type": "string"},
393
+ "loader_module": {"type": "string"},
394
+ "loader_function": {"type": "string"},
395
+ },
396
+ },
397
+ },
398
+ },
399
+ },
400
+ "real_use_cases": {
401
+ "type": "array",
402
+ "items": {"type": "string"},
403
+ "description": "What this agent is for, one plain sentence each. These are "
404
+ "capabilities, not test cases: 'cancel an order that has not shipped', not "
405
+ "a narrated situation with a customer, a name and an outcome. Scenarios are "
406
+ "written later, from these.",
407
+ },
408
+ "notes": {
409
+ "type": "string",
410
+ "description": "Free-form, yours. Anything else worth carrying forward: "
411
+ "quirks, traps, a plausible name that does not exist, an id that looks like "
412
+ "a typo but is real. Shown verbatim to every later stage.",
413
+ },
414
+ "open_questions": {
415
+ "type": "array",
416
+ "items": {"type": "string"},
417
+ "description": "What the source did not settle and you could not ask about.",
418
+ },
419
+ "chosen_evals": {
420
+ "type": "array",
421
+ "items": {"type": "string"},
422
+ "description": "Names of platform evals this agent should be judged by, "
423
+ f"at most {MOST_CHOSEN_EVALS}, taken only from the catalogue in your "
424
+ "briefing. Choose the ones that judge how the agent conducted the "
425
+ "conversation, which its scenarios' own checks cannot see, and only from "
426
+ "the section matching the modality you record here. Leave it out where the "
427
+ "catalogue offers nothing this agent can be judged by; an eval whose inputs "
428
+ "this agent never produces scores it against nothing, and an eval for spoken "
429
+ "calls means nothing for a chat agent.",
430
+ },
431
+ "implementation": {
432
+ "type": "string",
433
+ "enum": ["present", "absent", "partial"],
434
+ "description": "Whether the agent ships working code for its tools, as "
435
+ "opposed to only declaring them. The environment runs the agent's own code "
436
+ "wherever it exists, so this decides whether anything gets written for it.",
437
+ },
438
+ "tool_entrypoints": {
439
+ "type": "array",
440
+ "description": "How to reach the agent's own implementation of each tool. "
441
+ "One entry per tool that has code. Without this the environment has to write "
442
+ "a replacement, which tests our reading of the agent instead of the agent.",
443
+ "items": {
444
+ "type": "object",
445
+ "properties": {
446
+ "tool": {
447
+ "type": "string",
448
+ "description": "The tool name, exactly as in `tools`.",
449
+ },
450
+ "mode": {
451
+ "type": "string",
452
+ "enum": [
453
+ "import",
454
+ "construct",
455
+ "service",
456
+ "unreachable",
457
+ ],
458
+ "description": "import: a module-level function or a method on a "
459
+ "class, reachable directly. construct: it hangs off an object "
460
+ "that has to be built first. service: its effect is implemented "
461
+ "by a shipped HTTP service. unreachable: no runnable seam exists, "
462
+ "which blocks environment creation rather than generating a "
463
+ "replacement.",
464
+ },
465
+ "module": {
466
+ "type": "string",
467
+ "description": "Importable path as the agent's own code would "
468
+ "write it, e.g. package.module.file. Not a filesystem path.",
469
+ },
470
+ "callable": {
471
+ "type": "string",
472
+ "description": "What to call inside that module. May be dotted "
473
+ "to reach a method on a class, e.g. TheClass.the_method.",
474
+ },
475
+ "factory": {
476
+ "type": "string",
477
+ "description": "For construct: the expression that builds the "
478
+ "object, including whatever it needs to be constructed with.",
479
+ },
480
+ "first_arg": {
481
+ "type": "string",
482
+ "description": "If the callable takes the agent's own state as "
483
+ "its first argument, its name. Empty when the callable opens its "
484
+ "own connection instead.",
485
+ },
486
+ "service": {
487
+ "type": "string",
488
+ "description": "For service mode, the Compose service or source "
489
+ "dependency that answers this tool.",
490
+ },
491
+ "endpoint": {
492
+ "type": "string",
493
+ "description": "The dependency POST path invoked by this tool, "
494
+ "without a leading slash. Record it even for import/construct "
495
+ "tools when their implementation calls a service, and especially "
496
+ "when it differs from the model-facing name (for example "
497
+ "check_status calling get_status). Leave empty only for a fully "
498
+ "local tool.",
499
+ },
500
+ "method": {
501
+ "type": "string",
502
+ "enum": ["POST"],
503
+ "description": "HTTP method used by the submitted implementation.",
504
+ },
505
+ "notes": {
506
+ "type": "string",
507
+ "description": "Anything about reaching it that the fields above "
508
+ "do not carry, especially why a tool cannot be reached.",
509
+ },
510
+ },
511
+ },
512
+ },
513
+ "refusal_signature": {
514
+ "type": "string",
515
+ "description": "How this agent's own code says no in a value it returns "
516
+ "rather than by raising, described so it can be recognised, e.g. a string "
517
+ "beginning with a particular marker. Production code often reports failure "
518
+ "this way, and without this a refusal is recorded as a success, which hides "
519
+ "the behaviour most worth testing.",
520
+ },
521
+ "data_store": {
522
+ "type": "object",
523
+ "description": "What the agent's tools read and write, and how to point them "
524
+ "at a different one.",
525
+ "properties": {
526
+ "kind": {
527
+ "type": "string",
528
+ "description": "postgres, clickhouse, mysql, sqlite, in_process for "
529
+ "state held in memory, or none.",
530
+ },
531
+ "configured_by": {
532
+ "type": "string",
533
+ "description": "How the code chooses its connection: the environment "
534
+ "variable it reads, the config file, or the constructor argument. "
535
+ "This is what makes substituting a store possible without editing "
536
+ "the agent, so say if it is hardcoded.",
537
+ },
538
+ "schema_from": {
539
+ "type": "string",
540
+ "description": "Where the schema comes from: its migrations, a DDL "
541
+ "file, its ORM models.",
542
+ },
543
+ "loaded_by": {
544
+ "type": "string",
545
+ "description": "The agent's own loader, if it has one that builds "
546
+ "its starting data, as module and callable.",
547
+ },
548
+ "loader_module": {
549
+ "type": "string",
550
+ "description": "The module that loader is imported from, so it can "
551
+ "be called rather than reimplemented.",
552
+ },
553
+ "version": {
554
+ "type": "string",
555
+ "description": "The engine version, where the agent pins one.",
556
+ },
557
+ "config_key": {
558
+ "type": "string",
559
+ "description": "Where a config file holds the connection instead, as "
560
+ "a dotted path such as database.url.",
561
+ },
562
+ "host": {
563
+ "type": "string",
564
+ "description": "The host the agent expects. Record it even when it "
565
+ "is hardcoded: a hardcoded name is not a dead end, it is a name our "
566
+ "store can answer to.",
567
+ },
568
+ "port": {
569
+ "type": "integer",
570
+ "description": "The port it expects.",
571
+ },
572
+ "database": {
573
+ "type": "string",
574
+ "description": "The database name it expects. Ours is created with "
575
+ "exactly this name rather than the agent being changed.",
576
+ },
577
+ "user": {
578
+ "type": "string",
579
+ "description": "The user it connects as.",
580
+ },
581
+ "password_from": {
582
+ "type": "string",
583
+ "description": "Where the password comes from, never the password "
584
+ "itself. A contract is written to disk and read by people, so a "
585
+ "secret in it outlives the run that needed it.",
586
+ },
587
+ },
588
+ },
589
+ "runtime": {
590
+ "type": "object",
591
+ "description": "What it takes to run the agent's code.",
592
+ "properties": {
593
+ "language": {"type": "string"},
594
+ "version": {"type": "string"},
595
+ "install": {
596
+ "type": "string",
597
+ "description": "Its own install command, e.g. from its lockfile or "
598
+ "requirements. Recorded as evidence; generated packaging derives "
599
+ "the executable install from the submitted manifest/lockfile.",
600
+ },
601
+ "extras": {
602
+ "type": "array",
603
+ "items": {"type": "string"},
604
+ "description": "Optional dependency groups declared by the source "
605
+ "manifest that this runtime needs, such as voice or qdrant. Never "
606
+ "invent a group that the manifest does not declare.",
607
+ },
608
+ "workdir": {
609
+ "type": "string",
610
+ "description": "Where in the source imports resolve from, if not the "
611
+ "root.",
612
+ },
613
+ "compose_file": {
614
+ "type": "string",
615
+ "description": "Exact path to the submitted Compose file to run "
616
+ "when the repository contains more than one. Never invent one.",
617
+ },
618
+ "dockerfile": {
619
+ "type": "string",
620
+ "description": "Path to its own Dockerfile, if it has one. Theirs is "
621
+ "used in preference to anything written for it.",
622
+ },
623
+ "command": {
624
+ "type": "array",
625
+ "items": {"type": "string"},
626
+ "description": "Exact argv for the submitted process when the "
627
+ "repository has no container entrypoint. Leave empty only when one "
628
+ "conventional entrypoint is unambiguous in source.",
629
+ },
630
+ "context_excludes": {
631
+ "type": "array",
632
+ "items": {"type": "string"},
633
+ "description": "Repository-relative large generated outputs or "
634
+ "documentation to omit from a generated build context. Never omit "
635
+ "runtime source, dependency manifests, seed data, or tool data.",
636
+ },
637
+ "platform": {
638
+ "type": "string",
639
+ "description": "Container target declared by the repository, such "
640
+ "as linux/amd64. Preserve it exactly; never infer one from the "
641
+ "runner machine.",
642
+ },
643
+ "interface": {
644
+ "type": "object",
645
+ "description": "For a chat agent, the existing ingress exposed by "
646
+ "the submitted runtime. Record only an endpoint the repository "
647
+ "actually implements; never invent a gateway or route.",
648
+ "properties": {
649
+ "kind": {
650
+ "type": "string",
651
+ "enum": ["http", "websocket", "callable"],
652
+ "description": "How the simulator reaches the running "
653
+ "agent. HTTP is currently the hosted repository path.",
654
+ },
655
+ "protocol": {
656
+ "type": "string",
657
+ "enum": ["fi.alk", "openai_chat"],
658
+ "description": "The submitted endpoint's request/response "
659
+ "envelope. openai_chat means Chat Completions-compatible.",
660
+ },
661
+ "port": {
662
+ "type": "integer",
663
+ "description": "Container port the submitted chat service "
664
+ "listens on.",
665
+ },
666
+ "path": {
667
+ "type": "string",
668
+ "description": "Exact POST path for one conversational turn.",
669
+ },
670
+ "health_path": {
671
+ "type": "string",
672
+ "description": "Optional existing GET readiness path.",
673
+ },
674
+ "include_tools": {
675
+ "type": "boolean",
676
+ "description": "Whether the endpoint accepts tool schemas "
677
+ "with each request.",
678
+ },
679
+ },
680
+ },
681
+ },
682
+ },
683
+ },
684
+ [],
685
+ ),
686
+ )
687
+ async def submit_contract(args: dict[str, Any]) -> dict[str, Any]:
688
+ payload = unwrapped(args)
689
+ data_store = payload.get("data_store")
690
+ if not isinstance(data_store, dict):
691
+ data_store = {}
692
+ source_has_seed_loader = bool(
693
+ str(data_store.get("loaded_by") or "").strip()
694
+ or str(data_store.get("loader_module") or "").strip()
695
+ )
696
+ data_schema = payload.get("data_schema")
697
+ if not isinstance(data_schema, dict):
698
+ data_schema = {}
699
+ base_environment = payload.get("base_environment")
700
+ if not isinstance(base_environment, dict):
701
+ base_environment = {}
702
+ schema_collections = {
703
+ str(name).strip().lower().replace("-", "_").replace(" ", "_")
704
+ for name in data_schema
705
+ }
706
+ starting_collections = {
707
+ str(name).strip().lower().replace("-", "_").replace(" ", "_")
708
+ for name in base_environment
709
+ }
710
+ collection_overlap = schema_collections & starting_collections
711
+ store_kind = str(data_store.get("kind") or "").strip().lower()
712
+ structured_seed_store = any(
713
+ marker in store_kind
714
+ for marker in ("sqlite", "postgres", "mysql", "mariadb", "clickhouse")
715
+ )
716
+ badly_aligned_seed = (
717
+ source_has_seed_loader
718
+ and structured_seed_store
719
+ and len(schema_collections) > 1
720
+ and bool(starting_collections)
721
+ and len(collection_overlap) * 2 < len(starting_collections)
722
+ )
723
+ if badly_aligned_seed:
724
+ return _problems(
725
+ [
726
+ "base_environment does not use enough of the collections declared by "
727
+ "data_schema. Record source fixture rows under their exact collection/table "
728
+ "names (for example hotel_rooms, hotel_bookings, "
729
+ "restaurant_reservations), preserving real IDs and values. Semantic "
730
+ "summaries such as rooms, notable_bookings, pricing or restaurant_slots "
731
+ "cannot seed or verify a structured submitted store. This is a correctness "
732
+ "requirement for a source-loaded structured store; submit a corrected "
733
+ "contract rather than the same summary again."
734
+ ]
735
+ )
736
+
737
+ def rows_for(collection: str) -> list[dict[str, Any]]:
738
+ value = base_environment.get(collection, [])
739
+ if isinstance(value, list):
740
+ return [row for row in value if isinstance(row, dict)]
741
+ if isinstance(value, dict):
742
+ return [value]
743
+ return []
744
+
745
+ missing_fk_parents: list[str] = []
746
+ if source_has_seed_loader and structured_seed_store:
747
+ for collection, fields in data_schema.items():
748
+ if not isinstance(fields, dict):
749
+ continue
750
+ for field, specification in fields.items():
751
+ words = (
752
+ str(specification).replace("(", " ").replace(")", " ").split()
753
+ )
754
+ try:
755
+ marker = next(
756
+ index
757
+ for index, word in enumerate(words)
758
+ if word.upper() == "FK"
759
+ )
760
+ except StopIteration:
761
+ continue
762
+ if marker + 1 >= len(words):
763
+ continue
764
+ target_reference = words[marker + 1].strip("`'\".,:;")
765
+ if "." in target_reference:
766
+ target, target_key = target_reference.rsplit(".", 1)
767
+ else:
768
+ target = target_reference
769
+ inferred_key = str(field).rsplit("_", 1)[-1]
770
+ target_fields = data_schema.get(target)
771
+ if (
772
+ isinstance(target_fields, dict)
773
+ and inferred_key in target_fields
774
+ ):
775
+ target_key = inferred_key
776
+ elif (
777
+ isinstance(target_fields, dict)
778
+ and str(field) in target_fields
779
+ ):
780
+ # A same-named key such as payment_methods.rider_id ->
781
+ # users.rider_id must not be shortened blindly to users.id.
782
+ target_key = str(field)
783
+ else:
784
+ target_key = inferred_key
785
+ target_values = {
786
+ row.get(target_key)
787
+ for row in rows_for(target)
788
+ if row.get(target_key) not in (None, "")
789
+ }
790
+ for row in rows_for(str(collection)):
791
+ value = row.get(field)
792
+ if value not in (None, "") and value not in target_values:
793
+ missing_fk_parents.append(
794
+ f"{collection}.{field}={value!r} -> {target}.{target_key}"
795
+ )
796
+ if missing_fk_parents:
797
+ examples = ", ".join(sorted(set(missing_fk_parents))[:8])
798
+ return _problems(
799
+ [
800
+ "base_environment is not referentially closed: "
801
+ + examples
802
+ + ". Include the exact submitted parent rows for every represented FK. "
803
+ "Downstream environment generation must never invent a stub row that the "
804
+ "target's source loader does not contain."
805
+ ]
806
+ )
807
+
808
+ # Refused rather than dropped: dropping leaves the run judged by fewer evals than claimed.
809
+ asked_evals = payload.get("chosen_evals")
810
+ if isinstance(asked_evals, str):
811
+ asked_evals = [asked_evals]
812
+ if isinstance(asked_evals, list):
813
+ wanted: list[str] = []
814
+ for one in asked_evals:
815
+ name = str(one).strip()
816
+ if name and name not in wanted:
817
+ wanted.append(name)
818
+ unknown = [name for name in wanted if offered and name not in offered]
819
+ if unknown:
820
+ return _problems(
821
+ [
822
+ "chosen_evals names evals the platform did not offer: "
823
+ + ", ".join(unknown)
824
+ + ". Choose only from the catalogue in your briefing, by exact name."
825
+ ]
826
+ )
827
+ if not offered and wanted:
828
+ return _problems(
829
+ [
830
+ "chosen_evals was given but this job carries no eval catalogue, so there "
831
+ "is nothing to choose from. Leave it out."
832
+ ]
833
+ )
834
+ claimed = str(payload.get("modality") or "").strip().lower()
835
+ mismatched = [
836
+ f"{name} (applies to {offered_modality[name]})"
837
+ for name in wanted
838
+ if offered_modality.get(name)
839
+ and offered_modality[name] not in ("any", claimed)
840
+ ]
841
+ if mismatched:
842
+ return _problems(
843
+ [
844
+ f"chosen_evals names evals belonging to another modality than {claimed!r}: "
845
+ + ", ".join(mismatched)
846
+ + ". The platform refuses those, so choose only from the section of the "
847
+ "catalogue matching the modality you recorded."
848
+ ]
849
+ )
850
+ if len(wanted) > MOST_CHOSEN_EVALS:
851
+ return _problems(
852
+ [
853
+ f"chosen_evals has {len(wanted)} evals and at most {MOST_CHOSEN_EVALS} may "
854
+ "be chosen. Keep the ones that judge conduct this agent's own checks "
855
+ "cannot see, and drop the rest."
856
+ ]
857
+ )
858
+ payload["chosen_evals"] = wanted
859
+
860
+ thin = [
861
+ (
862
+ "prompt",
863
+ bool(payload.get("conversational", True))
864
+ and not payload.get("hard_constraints")
865
+ and not str(payload.get("system_prompt_excerpt") or "").strip(),
866
+ "no hard_constraints and no system_prompt_excerpt, for a conversational agent. "
867
+ "Its prompt usually exists and often lives away from the main agent file — "
868
+ "search the whole source for a long instructions string before deciding there "
869
+ "is none.",
870
+ ),
871
+ (
872
+ "data",
873
+ bool(payload.get("tools"))
874
+ and not payload.get("data_schema")
875
+ and not payload.get("base_environment"),
876
+ "no data_schema and no base_environment, for an agent that has tools. The world "
877
+ "every test runs against is built from exactly these two, so without them the "
878
+ "next stage has no schema to create and no rows to seed, and every tool call it "
879
+ "makes will refuse. Record the shape of each kind of record the tools read or "
880
+ "write, and enough real rows to reach every branch those tools have — a "
881
+ "representative sample for a large dataset, the whole thing for a small one.",
882
+ ),
883
+ (
884
+ "starting-data",
885
+ bool(payload.get("tools"))
886
+ and bool(payload.get("data_schema"))
887
+ and not payload.get("base_environment")
888
+ and source_has_seed_loader,
889
+ "data_schema is present but base_environment is empty even though data_store "
890
+ "records a source seed loader. Read that fixture and record its real starting "
891
+ "identifiers and values: the whole dataset when small, or a representative "
892
+ "sample that includes every branch likely to be tested. Downstream scenarios "
893
+ "must use those exact records; a plausible invented identifier creates a world "
894
+ "the submitted agent does not have.",
895
+ ),
896
+ (
897
+ "starting-data-collections",
898
+ source_has_seed_loader
899
+ and len(schema_collections) > 1
900
+ and bool(starting_collections)
901
+ and len(collection_overlap) < min(2, len(schema_collections)),
902
+ "base_environment does not use the collections declared by data_schema. Record "
903
+ "the source fixture rows under their exact collection/table names (for example "
904
+ "hotel_rooms, hotel_bookings, restaurant_reservations), preserving real IDs and "
905
+ "values. Semantic summaries such as rooms, notable_bookings or restaurant_slots "
906
+ "cannot seed or verify the submitted store.",
907
+ ),
908
+ ]
909
+ # All of them together, and each only once. Nudging in sequence would cost a turn per
910
+ # nudge and read as though the requirements were being invented one at a time.
911
+ say = [said for key, when, said in thin if when and key not in nudged]
912
+ nudged.update(key for key, when, _ in thin if when)
913
+ if say:
914
+ return _problems(
915
+ say + ["If any of these genuinely does not apply, submit again as is."]
916
+ )
917
+ return accept_contract(payload, destination, source_root=source_root)
918
+
919
+ return tool_server(name=CONTRACT_SERVER, version="0.1.0", tools=[submit_contract])
920
+
921
+
922
+ _JSON_TYPES = {
923
+ str: "string",
924
+ int: "integer",
925
+ float: "number",
926
+ bool: "boolean",
927
+ list: "array",
928
+ dict: "object",
929
+ }
930
+
931
+
932
+ def schema(properties: dict[str, Any], required: list[str]) -> dict[str, Any]:
933
+ """A tool's inputs, described well enough to be filled in correctly the first time.
934
+
935
+ Two things this exists for.
936
+
937
+ **Required means required.** Handing the decorator a plain ``{name: type}`` mapping marks
938
+ every parameter mandatory, so a tool with an optional field refuses any call that leaves it
939
+ out — "Input validation error: 'seed' is a required property" — for a field the tool itself
940
+ treats as optional.
941
+
942
+ **A schema is documentation, not just validation.** It is shown to the model before it calls
943
+ anything, so a property carrying only ``{"type": "array"}`` says nothing about what belongs
944
+ in it, and the model discovers the shape by being rejected. That is a full turn per guess and
945
+ it is avoidable: pass a full JSON-schema fragment instead of a bare type wherever the shape
946
+ is not obvious from the name, and it is right on the first call.
947
+
948
+ schema({"name": str,
949
+ "size": {"type": "string", "enum": ["S", "M", "L"]}}, ["name"])
950
+ """
951
+ wanted = list(required)
952
+ return {
953
+ "type": "object",
954
+ "properties": {
955
+ name: _schema_fragment(dict(kind), optional=name not in wanted)
956
+ if isinstance(kind, dict)
957
+ else _typed(_JSON_TYPES.get(kind, "string"), optional=name not in wanted)
958
+ for name, kind in properties.items()
959
+ },
960
+ "required": wanted,
961
+ }
962
+
963
+
964
+ def _schema_fragment(fragment: dict[str, Any], *, optional: bool) -> dict[str, Any]:
965
+ """Make every non-required object property nullable before the SDK validates it.
966
+
967
+ Tool input validation happens before our handler. Models correctly use null for optional
968
+ nested fields, so allowing null at only the top level still rejects useful payloads such as
969
+ ``{"tool_entrypoints": [{"module": null}]}`` before normalization can omit the value.
970
+ """
971
+ result = dict(fragment)
972
+ if result.get("type") == "object" and isinstance(result.get("properties"), dict):
973
+ required = set(result.get("required") or [])
974
+ result["properties"] = {
975
+ name: _schema_fragment(dict(value), optional=name not in required)
976
+ for name, value in result["properties"].items()
977
+ }
978
+ elif result.get("type") == "array" and isinstance(result.get("items"), dict):
979
+ result["items"] = _schema_fragment(dict(result["items"]), optional=False)
980
+ if optional:
981
+ kind = result.get("type")
982
+ if isinstance(kind, str):
983
+ result["type"] = [kind, "null"]
984
+ if isinstance(result.get("enum"), list) and None not in result["enum"]:
985
+ result["enum"] = [*result["enum"], None]
986
+ return result
987
+
988
+
989
+ def _typed(kind: str, *, optional: bool) -> dict[str, Any]:
990
+ """One property's type, letting an optional field be null.
991
+
992
+ Filling a field that does not apply with null is what a model does, and it is not wrong: the
993
+ alternative is inventing a value. Rejecting it costs a whole turn, and the rejection does not
994
+ even say which field was at fault: "None is not of type 'string'" is the entire message.
995
+ """
996
+ return {"type": [kind, "null"]} if optional else {"type": kind}
997
+
998
+
999
+ def brief(value: Any, limit: int = 1800) -> str:
1000
+ """What a call returned, shortened only when it has to be.
1001
+
1002
+ Generous, and explicit when it cuts. A record from a real agent's data is long, and a reply
1003
+ trimmed silently in the middle of it reads as though the field being looked for is absent:
1004
+ the answer is then six more calls working around something that was there all along.
1005
+
1006
+ Shared, because every stage that shows a caller what a tool answered has the same problem and
1007
+ they were not agreeing about it: one showed 1800 characters and said when it cut, the other
1008
+ showed 200 and said nothing, so the stage that most needs to read a record was the one that
1009
+ could not.
1010
+ """
1011
+ rendered = value if isinstance(value, str) else json.dumps(value, default=str)
1012
+ if len(rendered) <= limit:
1013
+ return rendered
1014
+ return (
1015
+ rendered[:limit]
1016
+ + f"\n... cut here, {len(rendered) - limit} more characters. Ask for one record rather "
1017
+ "than many if you need the whole of it."
1018
+ )