agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,718 @@
1
+ """The agent contract: what the agent verifiably is, read from its own source.
2
+
3
+ Everything downstream is confined to this. A world may only implement tools listed here, a
4
+ scenario may only reference values grounded in here, and a checkpoint may only assert against
5
+ what is here. It is the anti-hallucination device for every later stage.
6
+
7
+ The harness produces it by reading the agent's code and calling ``submit_contract``. Validation
8
+ runs inside that tool, so problems are returned into the conversation and the model tries again
9
+ rather than a bad contract reaching disk.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import shlex
16
+ from typing import Any
17
+
18
+ from pydantic import BaseModel, Field, field_validator, model_validator
19
+
20
+ # How a person reaches an agent. This decides how it is later run — voice goes out as a live
21
+ # call, everything else runs locally — so it is defined once and referenced, never retyped.
22
+ MODALITIES = ("voice", "chat", "browser")
23
+ # Voice only, and only two: either the agent placed the call or it answered one.
24
+ CALL_DIRECTIONS = ("inbound", "outbound")
25
+
26
+ _STRING_FIELDS = (
27
+ "agent",
28
+ "one_liner",
29
+ "modality",
30
+ "call_direction",
31
+ "system_prompt_excerpt",
32
+ "notes",
33
+ )
34
+ _LIST_FIELDS = (
35
+ "hard_constraints",
36
+ "real_use_cases",
37
+ "amendments",
38
+ "chosen_evals",
39
+ )
40
+ _DICT_FIELDS = ("data_schema", "base_environment")
41
+
42
+ # What each field gets called when it is not called what we call it. Every one of these was
43
+ # written by a model that had read the schema and still reached for the more obvious word.
44
+ _ALIASES = {
45
+ "real_use_cases": ("use_cases", "usecases", "scenarios", "capabilities"),
46
+ "hard_constraints": ("constraints", "rules", "policies", "policy", "guardrails"),
47
+ "system_prompt_excerpt": ("system_prompt", "prompt", "instructions"),
48
+ "base_environment": ("data", "seed_data", "starting_data", "records"),
49
+ "data_schema": ("schema", "record_schema", "data_shape"),
50
+ "agent": ("name", "agent_name"),
51
+ "one_liner": ("summary", "description"),
52
+ "notes": ("observations", "remarks"),
53
+ }
54
+
55
+
56
+ class ToolSpec(BaseModel):
57
+ """One tool the agent really has.
58
+
59
+ ``args`` is the load-bearing field: the world's handlers, the probes and every scenario are
60
+ built from these exact names. It is also the one most often written under another name —
61
+ ``parameters``, ``arguments``, ``params`` — or left out while ``arg_types`` names every
62
+ argument anyway. All of those are the same information, so they are accepted and normalised
63
+ rather than rejected, because a contract bounced for a synonym costs a full turn and teaches
64
+ nothing about the agent.
65
+ """
66
+
67
+ @model_validator(mode="before")
68
+ @classmethod
69
+ def _normalize_args(cls, payload: Any) -> Any:
70
+ if not isinstance(payload, dict):
71
+ return payload
72
+ if not payload.get("args"):
73
+ for alias in ("parameters", "arguments", "params", "arg_names"):
74
+ value = payload.get(alias)
75
+ if isinstance(value, list) and value:
76
+ payload["args"] = value
77
+ break
78
+ # Some writers give {name: type} where a list was asked for. The keys are the
79
+ # argument names, which is exactly what was wanted.
80
+ if isinstance(value, dict) and value:
81
+ payload["args"] = list(value)
82
+ payload.setdefault(
83
+ "arg_types", {k: str(v) for k, v in value.items()}
84
+ )
85
+ break
86
+ if not payload.get("args"):
87
+ # Nothing named the arguments directly, but a per-argument map still names them.
88
+ for source in ("arg_types", "arg_values"):
89
+ mapping = payload.get(source)
90
+ if isinstance(mapping, dict) and mapping:
91
+ payload["args"] = list(mapping)
92
+ break
93
+ if isinstance(payload.get("args"), str):
94
+ payload["args"] = [payload["args"]]
95
+ if isinstance(payload.get("args"), list):
96
+ payload["args"] = [str(one) for one in payload["args"]]
97
+ return payload
98
+
99
+ name: str
100
+ args: list[str] = Field(default_factory=list)
101
+ arg_types: dict[str, str] = Field(default_factory=dict)
102
+ arg_values: dict[str, Any] = Field(default_factory=dict)
103
+ description: str = ""
104
+ # Tools that must have run before this one stops refusing, and only for state the agent builds
105
+ # during the conversation. Empty means callable first thing. Marking a tool gated when it is not
106
+ # costs every future test of it a preamble it never needed.
107
+ requires: list[str] = Field(default_factory=list)
108
+
109
+
110
+ class ToolEntry(BaseModel):
111
+ """How to reach the agent's own implementation of one tool.
112
+
113
+ Recorded rather than assumed, because there is no shape every agent shares. A benchmark
114
+ writes static methods on a class; a framework agent writes closures inside ``__init__`` that
115
+ cannot be imported at all. What the environment does about a tool is decided from ``mode``,
116
+ so a tool nobody can reach is visible here rather than quietly reimplemented.
117
+ """
118
+
119
+ tool: str
120
+ # import: a module-level callable. construct: a method needing an instance built first.
121
+ # service: reachable over HTTP. unreachable: no runnable seam was found and building stops.
122
+ # The harness never generates agent behavior.
123
+ mode: str = "unreachable"
124
+ module: str = ""
125
+ callable: str = ""
126
+ # An expression that builds the object a `construct` tool hangs off.
127
+ factory: str = ""
128
+ # What the agent's own state is passed as, where a tool takes it as an argument.
129
+ first_arg: str = ""
130
+ # For a service-backed tool, the submitted service and HTTP path its implementation calls.
131
+ # The path is recorded separately from the semantic tool name because production APIs often
132
+ # use a different route name (for example check_status -> /get_status).
133
+ service: str = ""
134
+ endpoint: str = ""
135
+ method: str = "POST"
136
+ notes: str = ""
137
+
138
+
139
+ class DataStore(BaseModel):
140
+ """What the agent's tools read and write, and how to be there instead of it.
141
+
142
+ Nothing recorded here is a change to the agent. It is what the agent **already expects**,
143
+ written down so the environment can be built to match: the same host, the same port, the same
144
+ database, the same user. Where it reads a value from configuration we set that configuration;
145
+ where it hardcodes one we shape our own store to it, which is why a hardcoded value is worth
146
+ recording rather than treated as a dead end.
147
+
148
+ That inversion is the point. The alternative, editing the agent until it points at us, means
149
+ testing something other than what ships.
150
+ """
151
+
152
+ # Read off the agent, never chosen for it. Postgres and ClickHouse disagree about dialect,
153
+ # types and what a transaction even means, so an agent tested against the wrong one is graded
154
+ # on queries it never runs. Free text because the next agent will be on an engine nobody has
155
+ # written down yet.
156
+ kind: str = ""
157
+ version: str = ""
158
+
159
+ # The easiest seam, and the one most agents have: one variable or config key holding the whole
160
+ # connection string. Set it at launch and nothing else matters.
161
+ configured_by: str = ""
162
+ config_key: str = ""
163
+
164
+ # What the agent expects to find, whether it reads these from config or has them written into
165
+ # its source. A hardcoded host is not an obstacle: a network alias makes that name resolve to
166
+ # our container, and the agent connects to us believing nothing changed.
167
+ host: str = ""
168
+ port: int | None = None
169
+ database: str = ""
170
+ user: str = ""
171
+ # Deliberately never the password itself. A contract is written to disk and read by people, so
172
+ # a secret in it outlives the run that needed it. What is recorded is where the value comes
173
+ # from; if it is genuinely needed it is read at build time and not persisted.
174
+ password_from: str = ""
175
+
176
+ # An agent that holds its data in memory is reached by calling the function that loads it, not
177
+ # by connecting to anything. Recorded so the environment can call the agent's own loader
178
+ # rather than reading its files and rebuilding the structure itself, which would be a second
179
+ # implementation of the one thing this path exists to stop reimplementing.
180
+ schema_from: str = ""
181
+ loaded_by: str = ""
182
+ loader_module: str = ""
183
+
184
+ def has_seam(self) -> bool:
185
+ """Whether there is any way to point this agent at our store.
186
+
187
+ An agent with no seam at all is a finding, not a thing to work around: it cannot be tested
188
+ without one, and saying so is more useful than editing it until it can.
189
+ """
190
+ return bool(
191
+ self.configured_by
192
+ or self.config_key
193
+ or self.host
194
+ or self.port
195
+ or self.database
196
+ or self.loader_module
197
+ or self.loaded_by
198
+ )
199
+
200
+
201
+ class Reached(BaseModel):
202
+ """How an existing agent reaches a dependency, without storing its secret values."""
203
+
204
+ dsn_env: str = ""
205
+ config_key: str = ""
206
+ host: str = ""
207
+ port: int | None = None
208
+ database: str = ""
209
+ user: str = ""
210
+ password_from: str = ""
211
+ loader_module: str = ""
212
+ loader_function: str = ""
213
+
214
+ def has_seam(self) -> bool:
215
+ return bool(
216
+ self.dsn_env
217
+ or self.config_key
218
+ or self.host
219
+ or self.port
220
+ or self.database
221
+ or self.loader_module
222
+ )
223
+
224
+
225
+ class RuntimeInterface(BaseModel):
226
+ """The submitted runtime's existing conversational ingress.
227
+
228
+ This is connection metadata, not generated agent behavior. The harness may publish the
229
+ declared container port and translate its request/response envelope, but it never adds an
230
+ endpoint the repository does not already implement.
231
+ """
232
+
233
+ kind: str = ""
234
+ protocol: str = "fi.alk"
235
+ port: int | None = Field(default=None, ge=1, le=65535)
236
+ path: str = ""
237
+ health_path: str = ""
238
+ include_tools: bool = True
239
+
240
+ @field_validator("kind")
241
+ @classmethod
242
+ def _known_kind(cls, value: str) -> str:
243
+ normalized = str(value or "").strip().lower().replace("-", "_")
244
+ aliases = {"openai": "http", "openai_compatible": "http"}
245
+ return aliases.get(normalized, normalized)
246
+
247
+ @field_validator("protocol")
248
+ @classmethod
249
+ def _known_protocol(cls, value: str) -> str:
250
+ normalized = str(value or "fi.alk").strip().lower().replace("-", "_")
251
+ aliases = {
252
+ "openai": "openai_chat",
253
+ "openai_compatible": "openai_chat",
254
+ "chat_completions": "openai_chat",
255
+ "http": "fi.alk",
256
+ }
257
+ return aliases.get(normalized, normalized)
258
+
259
+ @field_validator("path", "health_path")
260
+ @classmethod
261
+ def _absolute_http_path(cls, value: str) -> str:
262
+ path = str(value or "").strip()
263
+ if path and not path.startswith("/"):
264
+ path = "/" + path
265
+ return path
266
+
267
+ @model_validator(mode="after")
268
+ def _complete(self) -> "RuntimeInterface":
269
+ if self.kind in {"http", "websocket"}:
270
+ if self.port is None:
271
+ raise ValueError(f"runtime_{self.kind}_interface_requires_port")
272
+ if not self.path:
273
+ raise ValueError(f"runtime_{self.kind}_interface_requires_path")
274
+ if self.kind == "http":
275
+ if self.protocol not in {"fi.alk", "openai_chat"}:
276
+ raise ValueError(
277
+ "runtime_http_protocol_unsupported: expected fi.alk or openai_chat"
278
+ )
279
+ if self.kind == "websocket" and self.protocol != "fi.alk":
280
+ raise ValueError("runtime_websocket_protocol_unsupported: expected fi.alk")
281
+ return self
282
+
283
+
284
+ class Runtime(BaseModel):
285
+ """What it takes to run the agent's code."""
286
+
287
+ # Empty means detect from the submitted dependency manifest. Defaulting this to Python makes
288
+ # an otherwise unambiguous Node repository fail the generated-runtime admission path.
289
+ language: str = ""
290
+ version: str = ""
291
+ install: str = ""
292
+ # Optional dependency groups declared by the repository itself (for example ``voice`` in
293
+ # pyproject.toml). Generated packaging validates these names against the manifest.
294
+ extras: list[str] = Field(default_factory=list)
295
+ workdir: str = ""
296
+ # Select one submitted Compose file when a repository contains multiple runnable stacks.
297
+ compose_file: str = ""
298
+ dockerfile: str = ""
299
+ # For repositories without container metadata, this is an argv vector for the submitted
300
+ # process. It is optional when one conventional entrypoint can be proven from source.
301
+ command: list[str] = Field(default_factory=list)
302
+ # Repository-relative generated-build exclusions selected during understanding. This is for
303
+ # large checked-in outputs or documentation, never for dependency manifests or source code.
304
+ context_excludes: list[str] = Field(default_factory=list)
305
+ # Preserve a repository-declared target architecture (for example linux/amd64 on an ARM
306
+ # runner). This is execution metadata, not a change to the submitted application.
307
+ platform: str = ""
308
+
309
+ # How a turn-based simulator reaches the submitted process after it starts. Voice has a
310
+ # standard rendezvous (LiveKit dispatch); chat repositories do not. Recording this seam is
311
+ # what lets the harness start the real runtime instead of reconstructing the agent from its
312
+ # prompt. Empty is valid for voice/browser agents and for contracts that are not backed by a
313
+ # repository.
314
+ interface: RuntimeInterface | None = None
315
+
316
+ @field_validator("command", mode="before")
317
+ @classmethod
318
+ def _normalize_command(cls, value: Any) -> Any:
319
+ if isinstance(value, str):
320
+ return shlex.split(value)
321
+ return value
322
+
323
+ @field_validator("extras", mode="before")
324
+ @classmethod
325
+ def _normalize_extras(cls, value: Any) -> Any:
326
+ if isinstance(value, str):
327
+ return [item.strip() for item in value.split(",") if item.strip()]
328
+ return value
329
+
330
+
331
+ class Dependency(BaseModel):
332
+ """Something the agent reaches for that has to exist before it can work.
333
+
334
+ This is what tells the environment stage there is a service to stand up, rather than leaving
335
+ it to notice halfway through that a tool has nothing to answer it. The world is a sandbox:
336
+ whatever is named here gets built inside it, so the agent's call goes to something real that
337
+ happens to be ours.
338
+ """
339
+
340
+ name: str
341
+ # datastore, service, file, queue — whatever kind of thing this is. Left open rather than
342
+ # enumerated, because the next agent will need a kind nobody has thought of yet.
343
+ kind: str = ""
344
+ what: str = ""
345
+ # The tools that cannot work without it. An unreferenced dependency is usually a mistake.
346
+ used_by: list[str] = Field(default_factory=list)
347
+ engine: str = ""
348
+ version: str = ""
349
+ reached: Reached = Field(default_factory=Reached)
350
+
351
+ def provisionable(self) -> bool:
352
+ return bool(self.engine) and self.reached.has_seam()
353
+
354
+
355
+ def _reached(one: Dependency) -> str:
356
+ if not one.engine and not one.reached.has_seam():
357
+ return ""
358
+ said: list[str] = []
359
+ if one.engine:
360
+ said.append(f"stand up {one.engine}{' ' + one.version if one.version else ''}")
361
+ where = one.reached
362
+ if where.loader_module or where.loader_function:
363
+ said.append(
364
+ "call the agent's own "
365
+ f"{where.loader_module or 'MODULE NOT RECORDED'}."
366
+ f"{where.loader_function or 'load_data'} for it; nothing is connected to and no "
367
+ "server is involved"
368
+ )
369
+ elif where.dsn_env:
370
+ said.append(f"point it there with ${where.dsn_env}")
371
+ elif where.config_key:
372
+ said.append(f"point it there with {where.config_key} in its config")
373
+ expected = [
374
+ f"{label} {value}"
375
+ for label, value in (
376
+ ("host", where.host),
377
+ ("port", where.port),
378
+ ("database", where.database),
379
+ ("user", where.user),
380
+ )
381
+ if value
382
+ ]
383
+ if expected:
384
+ said.append("build it to match " + ", ".join(expected))
385
+ if one.engine and not where.has_seam():
386
+ said.append("NO CONFIGURATION SEAM RECORDED")
387
+ return "; ".join(said)
388
+
389
+
390
+ class AgentContract(BaseModel):
391
+ """What the agent verifiably is. Nothing downstream may contradict this."""
392
+
393
+ @model_validator(mode="before")
394
+ @classmethod
395
+ def _normalize_shapes(cls, payload: Any) -> Any:
396
+ """Model JSON varies in benign ways: a list where prose was asked, a bare string where a
397
+ list was, a field under the obvious name rather than ours. Normalize instead of
398
+ rejecting, because none of that is a grounding error and rejecting it burns turns on
399
+ something that does not matter."""
400
+ if not isinstance(payload, dict):
401
+ return payload
402
+ # The name we chose is not always the obvious one. `real_use_cases` in particular gets
403
+ # written as `use_cases`, and the answer it then gets — "no-use-cases" — reads as
404
+ # missing rather than misnamed, so the same submission comes back again and again with
405
+ # the shape changed and the name untouched.
406
+ for ours, others in _ALIASES.items():
407
+ if payload.get(ours):
408
+ continue
409
+ for other in others:
410
+ if payload.get(other):
411
+ payload[ours] = payload[other]
412
+ break
413
+ for key in _STRING_FIELDS:
414
+ value = payload.get(key)
415
+ if isinstance(value, list):
416
+ payload[key] = "\n".join(str(item) for item in value)
417
+ elif value is not None and not isinstance(value, str):
418
+ payload[key] = str(value)
419
+ for key in _LIST_FIELDS:
420
+ value = payload.get(key)
421
+ if isinstance(value, str):
422
+ payload[key] = [value]
423
+ elif isinstance(value, list):
424
+ payload[key] = [
425
+ str(item) if not isinstance(item, str) else item for item in value
426
+ ]
427
+ for key in _DICT_FIELDS:
428
+ value = payload.get(key)
429
+ if value is not None and not isinstance(value, dict):
430
+ payload[key] = {"value": value}
431
+ return payload
432
+
433
+ # Defaulted rather than mandatory so a submission that forgets it reaches validate_contract,
434
+ # which says what to do about it, instead of dying in the schema layer with a type error.
435
+ agent: str = ""
436
+ one_liner: str = ""
437
+ modality: str = "chat"
438
+ # Voice only: whether this agent places calls or answers them. Chat is always started by the
439
+ # person, so it stays inbound. It changes how the simulated person is briefed, not who speaks
440
+ # first: an outbound agent still greets, it just has to say who it is and why it called.
441
+ call_direction: str = "inbound"
442
+ conversational: bool = True
443
+ system_prompt_excerpt: str = ""
444
+ hard_constraints: list[str] = Field(default_factory=list)
445
+ tools: list[ToolSpec] = Field(default_factory=list)
446
+ data_schema: dict[str, Any] = Field(default_factory=dict)
447
+ base_environment: dict[str, Any] = Field(default_factory=dict)
448
+ # What the environment stage has to build before any tool can be answered.
449
+ dependencies: list[Dependency] = Field(default_factory=list)
450
+ # Transport/model connections are configuration, not business services to reconstruct.
451
+ runtime_dependencies: list[Dependency] = Field(default_factory=list)
452
+ # Whether the agent ships code for its tools: present, absent, or partial. Missing code is a
453
+ # build blocker: the harness never supplies replacement agent behavior.
454
+ implementation: str = ""
455
+ tool_entrypoints: list[ToolEntry] = Field(default_factory=list)
456
+ # How this agent's tools say no in a value they return, rather than by raising. Without it a
457
+ # refusal cannot be told from a success once the agent's own code is answering the call.
458
+ refusal_signature: str = ""
459
+ data_store: DataStore | None = None
460
+ runtime: Runtime | None = None
461
+ real_use_cases: list[str] = Field(default_factory=list)
462
+ # Free-form. The fields above are the fixed core because code consumes them; this is where
463
+ # the reader records whatever else about *this* agent is worth carrying forward — quirks,
464
+ # traps, names that look real but are not — in whatever form fits. It is shown verbatim to
465
+ # every later stage.
466
+ notes: str = ""
467
+ open_questions: list[str] = Field(default_factory=list)
468
+ # Names only; the platform owns the catalogue. Empty means judge by the scenarios' checks alone.
469
+ chosen_evals: list[str] = Field(default_factory=list)
470
+ # Anything in here was not read from the agent's source. The contract is meant to be what
471
+ # the agent verifiably is, so when the harness widens it the difference is recorded rather
472
+ # than blended in, and whoever reads it later can tell the two apart.
473
+ amendments: list[str] = Field(default_factory=list)
474
+
475
+ def tool_names(self) -> set[str]:
476
+ return {tool.name for tool in self.tools}
477
+
478
+ def brief(self, *, full_schema: bool = True, with_data: bool = False) -> str:
479
+ """The grounding block handed to the model on every downstream call.
480
+
481
+ ``with_data`` includes the agent's real starting records rather than only their shape.
482
+ A stage that writes scenarios needs to know a menu exists; a stage that builds the world
483
+ has to reproduce it row for row, and a shape without records is not enough to do that.
484
+ """
485
+ lines: list[str] = []
486
+ for tool in self.tools:
487
+ signature = ", ".join(
488
+ f"{arg}: {tool.arg_types[arg]}" if arg in tool.arg_types else arg
489
+ for arg in tool.args
490
+ )
491
+ values = (
492
+ f" [values: {json.dumps(tool.arg_values)[:300]}]"
493
+ if tool.arg_values
494
+ else ""
495
+ )
496
+ # Preconditions belong on the tool line or they are not read. A writer that cannot see
497
+ # what a tool refuses until another has run replays the agent's whole flow to reach it.
498
+ needs = f" [after: {', '.join(tool.requires)}]" if tool.requires else ""
499
+ lines.append(
500
+ f" - {tool.name}({signature}){values}{needs} : {tool.description[:140]}"
501
+ )
502
+ parts = [
503
+ f"AGENT: {self.agent} - {self.one_liner}",
504
+ f"MODALITY: {self.modality}",
505
+ "REAL TOOLS (use ONLY these, with these exact arg names and types):\n"
506
+ + ("\n".join(lines) or " (none)"),
507
+ ]
508
+ # Voice only, and stated plainly: a scenario written as though the person dialled in tests
509
+ # nothing when the agent is the one placing the call.
510
+ if self.modality == "voice" and self.call_direction:
511
+ parts.insert(
512
+ 2,
513
+ f"CALL DIRECTION: {self.call_direction} - "
514
+ + (
515
+ "this agent places the call, so the person did not dial and has no request "
516
+ "to open with"
517
+ if self.call_direction == "outbound"
518
+ else "people dial in to this agent"
519
+ ),
520
+ )
521
+ if self.hard_constraints:
522
+ parts.append(
523
+ "HARD CONSTRAINTS the agent MUST follow (nothing may contradict these):\n - "
524
+ + "\n - ".join(self.hard_constraints[:14])
525
+ )
526
+ if self.data_schema and full_schema:
527
+ parts.append(
528
+ "DATA SHAPE (the fields each record has):\n"
529
+ + json.dumps(self.data_schema)[: 24000 if with_data else 2400]
530
+ )
531
+ if self.base_environment and with_data:
532
+ parts.append(
533
+ "THE AGENT'S REAL STARTING DATA. Reproduce this exactly, including anything\n"
534
+ "that looks like a mistake: a misspelled id, an item marked unavailable, an odd\n"
535
+ "price. The world is a replica of what the agent has, not a corrected version,\n"
536
+ "and a test written against a corrected world will not catch the real bug.\n"
537
+ + json.dumps(self.base_environment, ensure_ascii=False)
538
+ )
539
+ if self.dependencies:
540
+ parts.append(
541
+ "WHAT THIS AGENT DEPENDS ON (the environment has to provide each of these):\n - "
542
+ + "\n - ".join(
543
+ f"{one.name} ({one.kind or 'unspecified'}): {one.what}"
544
+ + (f" — used by {', '.join(one.used_by)}" if one.used_by else "")
545
+ + (f" — {_reached(one)}" if _reached(one) else "")
546
+ for one in self.dependencies
547
+ )
548
+ + "\nThe agent's code is never edited; the environment must match its existing seam."
549
+ )
550
+ if self.real_use_cases:
551
+ # Every one of them. A scenario writer covers what it is shown, so a truncated
552
+ # list silently caps coverage at the cut rather than at the agent's surface.
553
+ parts.append(
554
+ "REAL USE CASES (what this agent is actually for):\n - "
555
+ + "\n - ".join(self.real_use_cases)
556
+ )
557
+ if self.tool_entrypoints:
558
+ parts.append(
559
+ "THE AGENT'S OWN TOOL CODE. Run these rather than writing replacements:\n - "
560
+ + "\n - ".join(
561
+ f"{one.tool}: {one.mode}"
562
+ + (f" {one.module}.{one.callable}" if one.module else "")
563
+ + (f", state passed as {one.first_arg}" if one.first_arg else "")
564
+ + (f", build with {one.factory}" if one.factory else "")
565
+ for one in self.tool_entrypoints
566
+ )
567
+ )
568
+ if self.runtime_dependencies:
569
+ parts.append(
570
+ "RUNTIME CONNECTIONS (not business-world state; do not rebuild):\n"
571
+ + "\n".join(
572
+ f" {one.name}: {one.what} {_reached(one)}"
573
+ for one in self.runtime_dependencies
574
+ )
575
+ )
576
+ if self.refusal_signature:
577
+ parts.append(
578
+ "HOW THIS AGENT REFUSES, in a value rather than by raising:\n "
579
+ f"{self.refusal_signature}"
580
+ )
581
+ if self.data_store:
582
+ store = self.data_store
583
+ parts.append(
584
+ "ITS DATA STORE:\n"
585
+ f" kind: {store.kind or 'unspecified'}\n"
586
+ f" connection comes from: {store.configured_by or 'unknown'}\n"
587
+ f" schema from: {store.schema_from or 'unknown'}\n"
588
+ f" its own loader: {store.loaded_by or 'none'}"
589
+ )
590
+ if self.runtime:
591
+ run = self.runtime
592
+ parts.append(
593
+ "RUNNING ITS CODE:\n"
594
+ f" {run.language or 'language unspecified'} {run.version}, "
595
+ f"install with {run.install or 'unknown'}"
596
+ + (f", imports resolve from {run.workdir}" if run.workdir else "")
597
+ + (
598
+ f", its own Compose file at {run.compose_file}"
599
+ if run.compose_file
600
+ else ""
601
+ )
602
+ + (
603
+ f", its own Dockerfile at {run.dockerfile}"
604
+ if run.dockerfile
605
+ else ""
606
+ )
607
+ + (
608
+ f", reached over {run.interface.kind} {run.interface.protocol} on "
609
+ f"port {run.interface.port}{run.interface.path}"
610
+ if run.interface
611
+ else ""
612
+ )
613
+ )
614
+ if self.notes:
615
+ parts.append(f"NOTES from reading the agent:\n{self.notes[:1500]}")
616
+ return "\n\n".join(parts)
617
+
618
+ def entry_for(self, tool: str) -> ToolEntry | None:
619
+ for one in self.tool_entrypoints:
620
+ # Coerced rather than assumed. Assigning this field directly bypasses validation, so
621
+ # an entry can arrive as a plain mapping, and reading it as an object would raise
622
+ # somewhere far from the assignment.
623
+ found = one if isinstance(one, ToolEntry) else ToolEntry(**dict(one))
624
+ if found.tool == tool:
625
+ return found
626
+ return None
627
+
628
+ def adoptable(self, tool: str) -> bool:
629
+ """Whether this tool has code of its own that should be run instead of replaced."""
630
+ found = self.entry_for(tool)
631
+ return bool(found and found.mode in ("import", "construct", "service"))
632
+
633
+
634
+ def is_data_free_conversation(contract: AgentContract) -> bool:
635
+ """Whether the contract claims conversation without custom tools or business state.
636
+
637
+ Callers must also inspect the actual world; this claim alone is not a runtime exemption.
638
+ """
639
+ store = (
640
+ contract.data_store.model_dump(exclude_defaults=True)
641
+ if contract.data_store
642
+ else {}
643
+ )
644
+ if store.get("kind") in {"", "none", "in_process"}:
645
+ store.pop("kind", None)
646
+ return bool(
647
+ contract.conversational
648
+ and not (
649
+ contract.tools
650
+ or contract.tool_entrypoints
651
+ or contract.dependencies
652
+ or contract.data_schema
653
+ or contract.base_environment
654
+ or store
655
+ )
656
+ )
657
+
658
+
659
+ def validate_contract(contract: AgentContract) -> list[str]:
660
+ """Structural problems that make a contract unusable downstream.
661
+
662
+ Deliberately narrow. This cannot tell whether the model read the agent correctly, only
663
+ whether the result is shaped well enough to build a world from. Semantic grounding is the
664
+ operator's job, which is why the harness surfaces the contract for review.
665
+ """
666
+ problems: list[str] = []
667
+ if not contract.agent.strip():
668
+ problems.append("empty:agent")
669
+ for dependency in contract.dependencies:
670
+ if dependency.reached.dsn_env in {"LIVEKIT_URL", "LIVEKIT_INFERENCE_URL"}:
671
+ problems.append(
672
+ f"dependency[{dependency.name}]:runtime-connection-in-world — "
673
+ "LiveKit RTC/Inference is a runtime connection, not business-world data. "
674
+ "Move it to runtime_dependencies. Do not generate tables or tool sequences for it."
675
+ )
676
+ for dependency in contract.runtime_dependencies:
677
+ if (
678
+ dependency.kind.lower() in {"datastore", "database", "file", "queue"}
679
+ or dependency.reached.database
680
+ ):
681
+ problems.append(
682
+ f"dependency[{dependency.name}]:business-data-in-runtime — "
683
+ "Business data belongs in dependencies, not runtime_dependencies."
684
+ )
685
+ for index, tool in enumerate(contract.tools):
686
+ if not tool.name.strip():
687
+ problems.append(f"tool[{index}]:no-name")
688
+ continue
689
+ unknown = sorted(set(tool.arg_types) - set(tool.args))
690
+ if unknown:
691
+ problems.append(
692
+ f"tool[{tool.name}]:types-for-unknown-args:{','.join(unknown)}"
693
+ )
694
+ # Conversational agents may expose no tools. Zero-argument tools are also legitimate;
695
+ # cardinality alone is not evidence that authoring omitted something.
696
+ if not contract.real_use_cases:
697
+ problems.append("no-use-cases")
698
+ # An `import`/`construct` entry without both halves is unusable: bundling compiles a binding
699
+ # from exactly these two fields. Caught here so the model is told while it can still fix the
700
+ # entry, rather than the run dying much later in `_compile_source_tool_handlers` after the
701
+ # runtime-validation attempts have been spent.
702
+ for entry in contract.tool_entrypoints:
703
+ if entry.mode not in {"import", "construct"}:
704
+ continue
705
+ if entry.module.strip() and entry.callable.strip():
706
+ continue
707
+ problems.append(
708
+ f"tool_entrypoint[{entry.tool}]:{entry.mode}-needs-module-and-callable — "
709
+ "give the importable module path and the callable name, or record a mode that "
710
+ "matches what the repository actually exposes."
711
+ )
712
+ # Iterate the tools, not tool_names(): that returns a set, so duplicates collapse before
713
+ # they can be counted and the check silently never fires.
714
+ names = [tool.name for tool in contract.tools if tool.name.strip()]
715
+ duplicates = sorted({name for name in names if names.count(name) > 1})
716
+ if duplicates:
717
+ problems.append(f"duplicate-tool-names:{','.join(duplicates)}")
718
+ return problems