agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,312 @@
1
+ """Changing the contract after the fact, and being honest about having done it.
2
+
3
+ The contract is what the agent verifiably is, read from its own source. That makes it the thing
4
+ everything downstream is confined to, and it is why the harness cannot invent a tool or a value.
5
+
6
+ But it is not permanent. Two situations genuinely require changing it, and they are different:
7
+
8
+ - **It was read wrong.** Stage one missed a value the agent really accepts. Correcting that is
9
+ restoring the truth, and the correction should come from the source.
10
+ - **The agent is being changed.** Somebody adds an item to the world because the real menu is
11
+ gaining one. The world and the action space have to move together: an item the world holds but
12
+ the agent cannot name is dead data, and a scenario about it can only fail.
13
+
14
+ Either way the amendment is recorded on the contract itself rather than blended into what was
15
+ read, so that a month later it is still possible to tell what came from the agent and what came
16
+ from us. That distinction is the whole value of the contract; quietly widening it would make it
17
+ the same kind of guess it exists to prevent.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ from pathlib import Path
24
+
25
+ from .contract import MODALITIES, AgentContract, validate_contract
26
+
27
+ CONTRACT = "contract.json"
28
+
29
+
30
+ def widen(
31
+ contract: AgentContract,
32
+ destination: Path,
33
+ *,
34
+ tool_name: str,
35
+ argument: str,
36
+ values: list[str],
37
+ why: str,
38
+ ) -> tuple[bool, str]:
39
+ """Let a tool's argument accept values it did not before.
40
+
41
+ Amends the contract the stage is holding, then persists it. Loading a second copy from disk
42
+ and writing that back would leave the stage still working from the old one, so the world it
43
+ goes on to check would be checked against an action space that no longer matches.
44
+
45
+ Returns whether it was amended, and what happened.
46
+ """
47
+ spec = next((tool for tool in contract.tools if tool.name == tool_name), None)
48
+ if spec is None:
49
+ return False, (
50
+ f"{tool_name!r} is not a tool this agent has. It has: "
51
+ f"{', '.join(sorted(contract.tool_names()))}"
52
+ )
53
+ if argument not in spec.args:
54
+ return False, (
55
+ f"{tool_name} takes no argument called {argument!r}. It takes: "
56
+ f"{', '.join(spec.args) or 'nothing'}"
57
+ )
58
+ if not why.strip():
59
+ return (
60
+ False,
61
+ "say why: an unexplained amendment is indistinguishable from a guess",
62
+ )
63
+
64
+ existing = spec.arg_values.get(argument)
65
+ current = list(existing) if isinstance(existing, (list, tuple)) else []
66
+ fresh = [value for value in values if value and value not in current]
67
+ if not fresh:
68
+ return False, f"{argument} already accepts {', '.join(values) or 'nothing new'}"
69
+
70
+ spec.arg_values[argument] = [*current, *fresh]
71
+ contract.amendments.append(
72
+ f"{tool_name}.{argument} widened by {', '.join(fresh)}: {why.strip()}"
73
+ )
74
+
75
+ problems = validate_contract(contract)
76
+ if problems:
77
+ spec.arg_values[argument] = current
78
+ contract.amendments.pop()
79
+ return False, "the amended contract would not be valid: " + "; ".join(problems)
80
+
81
+ destination = Path(destination)
82
+ destination.mkdir(parents=True, exist_ok=True)
83
+ (destination / CONTRACT).write_text(
84
+ json.dumps(contract.model_dump(), indent=2, ensure_ascii=False),
85
+ encoding="utf-8",
86
+ )
87
+ return True, (
88
+ f"{tool_name}.{argument} now accepts {', '.join(fresh)}. "
89
+ f"{len(contract.amendments)} amendment(s) recorded on the contract."
90
+ )
91
+
92
+
93
+ def add_rule(
94
+ contract: AgentContract, destination: Path, *, rule: str, why: str
95
+ ) -> tuple[bool, str]:
96
+ """Give the agent a rule its source did not state.
97
+
98
+ A hard constraint is not decoration: the agent under test is told it, and the judge grades
99
+ against it. So this is a real change to what is being tested, and like a widened argument it
100
+ is recorded rather than blended into what was read from the source.
101
+ """
102
+ rule = rule.strip()
103
+ if not rule:
104
+ return False, "no rule given"
105
+ if not why.strip():
106
+ return False, "say why: an unexplained rule is indistinguishable from a guess"
107
+ if any(rule.lower() == existing.lower() for existing in contract.hard_constraints):
108
+ return False, f"the agent already has that rule: {rule}"
109
+
110
+ contract.hard_constraints.append(rule)
111
+ contract.amendments.append(f"rule added — {rule}: {why.strip()}")
112
+ problems = validate_contract(contract)
113
+ if problems:
114
+ contract.hard_constraints.pop()
115
+ contract.amendments.pop()
116
+ return False, "the amended contract would not be valid: " + "; ".join(problems)
117
+
118
+ destination = Path(destination)
119
+ destination.mkdir(parents=True, exist_ok=True)
120
+ (destination / CONTRACT).write_text(
121
+ json.dumps(contract.model_dump(), indent=2, ensure_ascii=False),
122
+ encoding="utf-8",
123
+ )
124
+ return True, (
125
+ f"added. The agent now has {len(contract.hard_constraints)} rules, and this one is "
126
+ "graded from here on."
127
+ )
128
+
129
+
130
+ def set_modality(
131
+ contract: AgentContract, destination: Path, *, modality: str, why: str
132
+ ) -> tuple[bool, str]:
133
+ """Correct how a person actually reaches this agent.
134
+
135
+ Worth its own amendment because modality is the one field that reroutes everything: it picks
136
+ the world, the simulator and the transport, so a wrong value does not degrade a run, it runs
137
+ a different test. And it is the field the source is least able to settle. An agent's code
138
+ reads the same whether it is answering a chat window or a phone call, so a reader with no
139
+ other evidence concludes whatever the repository looks like, which for a text benchmark is
140
+ text, even when the person asking said they had deployed it to a phone number.
141
+
142
+ Where the agent is deployed is a fact about somebody's setup rather than about the source, so
143
+ when the two disagree the person is right and the source is describing a different runtime of
144
+ the same agent. Recorded like every other amendment, because it is still a change to what was
145
+ read.
146
+ """
147
+ named = (modality or "").strip().lower()
148
+ if named not in MODALITIES:
149
+ return (
150
+ False,
151
+ f"{named!r} is not a modality. It is one of: {', '.join(MODALITIES)}",
152
+ )
153
+ if not why.strip():
154
+ return False, "say why: modality decides how every scenario is run"
155
+ if named == contract.modality:
156
+ return False, f"the contract already says {named}"
157
+
158
+ was = contract.modality
159
+ contract.modality = named
160
+ contract.amendments.append(f"modality {was} -> {named}: {why.strip()}")
161
+ problems = validate_contract(contract)
162
+ if problems:
163
+ contract.modality = was
164
+ contract.amendments.pop()
165
+ return False, "the amended contract would not be valid: " + "; ".join(problems)
166
+
167
+ destination = Path(destination)
168
+ destination.mkdir(parents=True, exist_ok=True)
169
+ (destination / CONTRACT).write_text(
170
+ json.dumps(contract.model_dump(), indent=2, ensure_ascii=False),
171
+ encoding="utf-8",
172
+ )
173
+ reached = {
174
+ "voice": "a call placed to the agent where it is hosted, its own tools answered over a "
175
+ "webhook",
176
+ "chat": "a typed conversation, the agent reconstructed here from its contract",
177
+ "browser": "a browser the agent drives",
178
+ }[named]
179
+ return True, f"modality is now {named}. Every scenario will be run as {reached}."
180
+
181
+
182
+ def drop_rule(
183
+ contract: AgentContract, destination: Path, *, rule: str, why: str
184
+ ) -> tuple[bool, str]:
185
+ """Take away a rule the agent does not really have.
186
+
187
+ Stage one can misread a comment as a constraint, and a rule nobody has is worse than a
188
+ missing one: the agent under test is told to obey it and the judge fails it for not doing
189
+ something it was never supposed to do.
190
+ """
191
+ if not why.strip():
192
+ return False, "say why: removing a rule changes what is being graded"
193
+ match = next(
194
+ (
195
+ existing
196
+ for existing in contract.hard_constraints
197
+ if existing.lower() == rule.strip().lower()
198
+ ),
199
+ None,
200
+ ) or next(
201
+ (
202
+ existing
203
+ for existing in contract.hard_constraints
204
+ if rule.strip().lower() in existing.lower()
205
+ ),
206
+ None,
207
+ )
208
+ if match is None:
209
+ return False, (
210
+ "no rule like that. It has:\n - "
211
+ + "\n - ".join(contract.hard_constraints)
212
+ )
213
+ contract.hard_constraints.remove(match)
214
+ contract.amendments.append(f"rule removed — {match}: {why.strip()}")
215
+ _persist(contract, destination)
216
+ return True, f"removed. {len(contract.hard_constraints)} rules left"
217
+
218
+
219
+ def fix_tool(
220
+ contract: AgentContract,
221
+ destination: Path,
222
+ *,
223
+ tool_name: str,
224
+ why: str,
225
+ args: list[str] | None = None,
226
+ arg_types: dict[str, str] | None = None,
227
+ description: str = "",
228
+ remove: bool = False,
229
+ ) -> tuple[bool, str]:
230
+ """Correct a tool that was read wrong, or take away one the agent does not have.
231
+
232
+ The most damaging thing stage one can get wrong. Every argument name flows into the world's
233
+ handlers, the probes and the scenarios, so a tool recorded with the wrong argument produces
234
+ a world that refuses everything and a suite that blames the agent for it.
235
+ """
236
+ if not why.strip():
237
+ return False, "say why: this changes what everything downstream is built from"
238
+ spec = next((tool for tool in contract.tools if tool.name == tool_name), None)
239
+ if spec is None:
240
+ return False, (
241
+ f"{tool_name!r} is not a tool this agent has. It has: "
242
+ f"{', '.join(sorted(contract.tool_names()))}"
243
+ )
244
+
245
+ if remove:
246
+ contract.tools.remove(spec)
247
+ contract.amendments.append(f"tool removed — {tool_name}: {why.strip()}")
248
+ problems = validate_contract(contract)
249
+ if problems:
250
+ contract.tools.append(spec)
251
+ contract.amendments.pop()
252
+ return False, "cannot remove it: " + "; ".join(problems)
253
+ _persist(contract, destination)
254
+ return True, f"{tool_name} removed. {len(contract.tools)} tools left"
255
+
256
+ changed = []
257
+ if args is not None:
258
+ # Values recorded against an argument that no longer exists would silently be lost, so
259
+ # they are carried across by name and anything orphaned is said out loud.
260
+ orphaned = sorted(set(spec.arg_values) - set(args))
261
+ spec.args = list(args)
262
+ spec.arg_types = {k: v for k, v in spec.arg_types.items() if k in spec.args}
263
+ spec.arg_values = {k: v for k, v in spec.arg_values.items() if k in spec.args}
264
+ changed.append(f"arguments are now {', '.join(args)}")
265
+ if orphaned:
266
+ changed.append(f"dropped values recorded for {', '.join(orphaned)}")
267
+ if arg_types:
268
+ unknown = sorted(set(arg_types) - set(spec.args))
269
+ if unknown:
270
+ return False, f"{tool_name} takes no argument called {', '.join(unknown)}"
271
+ spec.arg_types.update(arg_types)
272
+ changed.append("types updated")
273
+ if description:
274
+ spec.description = description
275
+ changed.append("description updated")
276
+ if not changed:
277
+ return False, "nothing to change: give args, arg_types, description, or remove"
278
+
279
+ contract.amendments.append(f"tool corrected — {tool_name}: {why.strip()}")
280
+ problems = validate_contract(contract)
281
+ if problems:
282
+ return False, "the amended contract would not be valid: " + "; ".join(problems)
283
+ _persist(contract, destination)
284
+ return True, f"{tool_name}: {', '.join(changed)}"
285
+
286
+
287
+ def _persist(contract: AgentContract, destination: Path) -> None:
288
+ destination = Path(destination)
289
+ destination.mkdir(parents=True, exist_ok=True)
290
+ (destination / CONTRACT).write_text(
291
+ json.dumps(contract.model_dump(), indent=2, ensure_ascii=False),
292
+ encoding="utf-8",
293
+ )
294
+
295
+
296
+ def not_offered(contract: AgentContract, candidates: dict[str, set[str]]) -> list[str]:
297
+ """For each argument, the candidate values the contract does not let the agent send.
298
+
299
+ ``candidates`` maps an argument name to identifiers found in the world that plausibly belong
300
+ to it. Kept as an argument rather than inferred here, because which column feeds which
301
+ argument is knowledge about one agent, not something a schema states.
302
+ """
303
+ missing: list[str] = []
304
+ for tool in contract.tools:
305
+ for argument, values in (tool.arg_values or {}).items():
306
+ if argument not in candidates or not isinstance(values, (list, tuple)):
307
+ continue
308
+ permitted = {str(value) for value in values}
309
+ absent = sorted(candidates[argument] - permitted)
310
+ if absent:
311
+ missing.append(f"{tool.name}.{argument}: {', '.join(absent)}")
312
+ return missing
@@ -0,0 +1,319 @@
1
+ """Content-addressed artifact manifests and call-evidence integrity checks."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import mimetypes
8
+ import os
9
+ import re
10
+ from pathlib import Path, PurePosixPath
11
+ from pydantic import BaseModel, Field, model_validator
12
+
13
+ ARTIFACT_MANIFEST_VERSION = "futureagi.harness-artifacts.v1"
14
+ ARTIFACT_MANIFEST_NAME = "artifact-manifest.json"
15
+
16
+
17
+ class ArtifactIntegrityError(RuntimeError):
18
+ pass
19
+
20
+
21
+ class ArtifactRecord(BaseModel):
22
+ path: str
23
+ sha256: str
24
+ size: int = Field(ge=0)
25
+ kind: str
26
+ media_type: str
27
+ scenario: str | None = None
28
+
29
+ @model_validator(mode="after")
30
+ def _safe_path(self) -> "ArtifactRecord":
31
+ path = PurePosixPath(self.path)
32
+ if path.is_absolute() or ".." in path.parts or not self.path:
33
+ raise ValueError(f"artifact_path_unsafe: {self.path}")
34
+ if not re.fullmatch(r"[0-9a-f]{64}", self.sha256):
35
+ raise ValueError(f"artifact_digest_invalid: {self.path}")
36
+ return self
37
+
38
+
39
+ class ScenarioEvidence(BaseModel):
40
+ scenario: str
41
+ result_path: str
42
+ canonical_status: str
43
+ transcript_path: str | None = None
44
+ recording_paths: list[str] = Field(default_factory=list)
45
+ tool_trace_path: str | None = None
46
+
47
+
48
+ class ArtifactManifest(BaseModel):
49
+ schema_version: str = ARTIFACT_MANIFEST_VERSION
50
+ run_id: str
51
+ digest: str
52
+ files: list[ArtifactRecord] = Field(default_factory=list)
53
+ scenarios: list[ScenarioEvidence] = Field(default_factory=list)
54
+ total_bytes: int = Field(ge=0)
55
+ complete: bool = True
56
+
57
+ @model_validator(mode="after")
58
+ def _valid(self) -> "ArtifactManifest":
59
+ if self.schema_version != ARTIFACT_MANIFEST_VERSION:
60
+ raise ValueError("artifact_manifest_version_unsupported")
61
+ if not re.fullmatch(r"sha256:[0-9a-f]{64}", self.digest):
62
+ raise ValueError("artifact_manifest_digest_invalid")
63
+ paths = [item.path for item in self.files]
64
+ if len(paths) != len(set(paths)):
65
+ raise ValueError("artifact_manifest_duplicate_path")
66
+ return self
67
+
68
+
69
+ _SECRET_CONTENT = (
70
+ re.compile(rb"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----"),
71
+ re.compile(rb"\b(?:ghp|github_pat)_[A-Za-z0-9_]{20,}\b"),
72
+ re.compile(rb"\bAKIA[0-9A-Z]{16}\b"),
73
+ re.compile(rb"\bsk-[A-Za-z0-9_-]{20,}\b"),
74
+ )
75
+ _SECRET_FILES = {".env", ".env.local", ".npmrc", ".pypirc", "id_rsa"}
76
+ _TERMINAL_STATUSES = {
77
+ "completed",
78
+ "failed",
79
+ "timed_out",
80
+ "cancelled",
81
+ "canceled",
82
+ # Chat conversations use domain-specific endings. All three mean execution
83
+ # stopped and produced durable evidence; only ``finished`` implies that the
84
+ # simulated user achieved its conversational goal. Artifact sealing must
85
+ # preserve the other two as findings rather than turn them into an
86
+ # infrastructure failure.
87
+ "finished",
88
+ "gave-up",
89
+ "ran-out-of-turns",
90
+ }
91
+
92
+
93
+ def seal_artifacts(
94
+ root: str | Path,
95
+ *,
96
+ run_id: str,
97
+ max_bytes: int = 1_073_741_824,
98
+ expected_scenarios: int | None = None,
99
+ ) -> ArtifactManifest:
100
+ """Validate evidence, hash every retained file, and atomically seal the directory."""
101
+ root = Path(root).expanduser().resolve()
102
+ if not root.is_dir():
103
+ raise ArtifactIntegrityError(f"artifact_root_missing: {root}")
104
+ records: list[ArtifactRecord] = []
105
+ total = 0
106
+ for path in sorted(root.rglob("*")):
107
+ # The outbox remains append-only until the platform acknowledges the terminal event.
108
+ # It is transport state, not immutable run evidence, and is therefore not sealed.
109
+ if (
110
+ path.name in {ARTIFACT_MANIFEST_NAME, "harness-events.jsonl"}
111
+ or path.is_dir()
112
+ ):
113
+ continue
114
+ if path.is_symlink():
115
+ raise ArtifactIntegrityError(
116
+ f"artifact_symlink_forbidden: {path.relative_to(root)}"
117
+ )
118
+ relative = path.relative_to(root).as_posix()
119
+ if path.name in _SECRET_FILES or path.suffix.lower() in {
120
+ ".pem",
121
+ ".key",
122
+ ".p12",
123
+ ".pfx",
124
+ }:
125
+ raise ArtifactIntegrityError(f"artifact_secret_file_forbidden: {relative}")
126
+ digest, size = _hash_file(path, relative)
127
+ total += size
128
+ if total > max_bytes:
129
+ raise ArtifactIntegrityError(
130
+ f"artifact_size_limit_exceeded: {total} > {max_bytes}"
131
+ )
132
+ records.append(
133
+ ArtifactRecord(
134
+ path=relative,
135
+ sha256=digest,
136
+ size=size,
137
+ kind=_kind(path),
138
+ media_type=mimetypes.guess_type(path.name)[0]
139
+ or "application/octet-stream",
140
+ scenario=_scenario(relative),
141
+ )
142
+ )
143
+
144
+ scenarios = _scenario_evidence(root)
145
+ if expected_scenarios is not None and len(scenarios) != expected_scenarios:
146
+ raise ArtifactIntegrityError(
147
+ "artifact_scenario_count_mismatch: "
148
+ f"expected {expected_scenarios}, found {len(scenarios)}"
149
+ )
150
+ core = {
151
+ "schema_version": ARTIFACT_MANIFEST_VERSION,
152
+ "run_id": run_id,
153
+ "files": [record.model_dump(mode="json") for record in records],
154
+ "scenarios": [item.model_dump(mode="json") for item in scenarios],
155
+ "total_bytes": total,
156
+ "complete": True,
157
+ }
158
+ digest = (
159
+ "sha256:"
160
+ + hashlib.sha256(
161
+ json.dumps(core, sort_keys=True, separators=(",", ":")).encode()
162
+ ).hexdigest()
163
+ )
164
+ manifest = ArtifactManifest(**core, digest=digest)
165
+ target = root / ARTIFACT_MANIFEST_NAME
166
+ temporary = root / f".{ARTIFACT_MANIFEST_NAME}.tmp-{os.getpid()}"
167
+ temporary.write_text(manifest.model_dump_json(indent=2) + "\n", encoding="utf-8")
168
+ temporary.replace(target)
169
+ return manifest
170
+
171
+
172
+ def load_artifact_manifest(
173
+ root: str | Path, *, verify: bool = True
174
+ ) -> ArtifactManifest:
175
+ root = Path(root).expanduser().resolve()
176
+ path = root / ARTIFACT_MANIFEST_NAME
177
+ try:
178
+ manifest = ArtifactManifest.model_validate_json(
179
+ path.read_text(encoding="utf-8")
180
+ )
181
+ except (OSError, ValueError) as exc:
182
+ raise ArtifactIntegrityError(f"artifact_manifest_invalid: {exc}") from exc
183
+ if verify:
184
+ for record in manifest.files:
185
+ candidate = (root / record.path).resolve()
186
+ if root not in candidate.parents or not candidate.is_file():
187
+ raise ArtifactIntegrityError(f"artifact_missing: {record.path}")
188
+ digest, size = _hash_file(candidate, record.path)
189
+ if digest != record.sha256 or size != record.size:
190
+ raise ArtifactIntegrityError(f"artifact_changed: {record.path}")
191
+ expected = (
192
+ "sha256:"
193
+ + hashlib.sha256(
194
+ json.dumps(
195
+ manifest.model_dump(mode="json", exclude={"digest"}),
196
+ sort_keys=True,
197
+ separators=(",", ":"),
198
+ ).encode()
199
+ ).hexdigest()
200
+ )
201
+ if expected != manifest.digest:
202
+ raise ArtifactIntegrityError("artifact_manifest_digest_mismatch")
203
+ return manifest
204
+
205
+
206
+ def _scenario_evidence(root: Path) -> list[ScenarioEvidence]:
207
+ evidence: list[ScenarioEvidence] = []
208
+ for result_path in sorted((root / "runs").glob("*/*/result.json")):
209
+ try:
210
+ result = json.loads(result_path.read_text(encoding="utf-8"))
211
+ except (OSError, ValueError) as exc:
212
+ raise ArtifactIntegrityError(
213
+ f"result_invalid: {result_path.relative_to(root)}"
214
+ ) from exc
215
+ relative = result_path.relative_to(root).as_posix()
216
+ scenario = result_path.parent.name
217
+ status = str(
218
+ result.get("ended") or (result.get("measured") or {}).get("status") or ""
219
+ ).lower()
220
+ if status not in _TERMINAL_STATUSES:
221
+ raise ArtifactIntegrityError(
222
+ f"result_not_terminal: {scenario}: {status or 'missing'}"
223
+ )
224
+ transcript_path = result_path.parent / "transcript.txt"
225
+ transcript = str(result.get("transcript") or "").strip()
226
+ if not transcript and (
227
+ not transcript_path.is_file()
228
+ or not transcript_path.read_text(encoding="utf-8").strip()
229
+ ):
230
+ raise ArtifactIntegrityError(f"transcript_missing: {scenario}")
231
+
232
+ recordings: list[str] = []
233
+ for track in result.get("tracks") or []:
234
+ raw = str(track.get("path") or "") if isinstance(track, dict) else ""
235
+ if not raw:
236
+ continue
237
+ candidate = Path(raw)
238
+ if not candidate.is_absolute():
239
+ candidate = result_path.parent / candidate
240
+ candidate = candidate.resolve()
241
+ if (
242
+ root not in candidate.parents
243
+ or not candidate.is_file()
244
+ or candidate.stat().st_size == 0
245
+ ):
246
+ raise ArtifactIntegrityError(f"recording_missing: {scenario}: {raw}")
247
+ recordings.append(candidate.relative_to(root).as_posix())
248
+ if (
249
+ result.get("recording") or (result.get("measured") or {}).get("room")
250
+ ) and not recordings:
251
+ raise ArtifactIntegrityError(f"recording_evidence_missing: {scenario}")
252
+
253
+ tool_trace = result_path.parent / "agent-tool-calls.jsonl"
254
+ evidence.append(
255
+ ScenarioEvidence(
256
+ scenario=scenario,
257
+ result_path=relative,
258
+ canonical_status=status,
259
+ transcript_path=(
260
+ transcript_path.relative_to(root).as_posix()
261
+ if transcript_path.is_file()
262
+ else None
263
+ ),
264
+ recording_paths=sorted(recordings),
265
+ tool_trace_path=(
266
+ tool_trace.relative_to(root).as_posix()
267
+ if tool_trace.is_file()
268
+ else None
269
+ ),
270
+ )
271
+ )
272
+ return evidence
273
+
274
+
275
+ def _hash_file(path: Path, relative: str) -> tuple[str, int]:
276
+ digest = hashlib.sha256()
277
+ size = 0
278
+ with path.open("rb") as stream:
279
+ while chunk := stream.read(1024 * 1024):
280
+ size += len(chunk)
281
+ digest.update(chunk)
282
+ if any(pattern.search(chunk) for pattern in _SECRET_CONTENT):
283
+ raise ArtifactIntegrityError(
284
+ f"artifact_secret_material_detected: {relative}"
285
+ )
286
+ return digest.hexdigest(), size
287
+
288
+
289
+ def _kind(path: Path) -> str:
290
+ if path.name == "result.json":
291
+ return "result"
292
+ if path.name == "transcript.txt":
293
+ return "transcript"
294
+ if path.name == "agent-tool-calls.jsonl":
295
+ return "tool_trace"
296
+ if path.suffix.lower() in {".wav", ".mp3", ".ogg", ".webm", ".m4a"}:
297
+ return "recording"
298
+ if path.name == "harness-events.jsonl":
299
+ return "event_stream"
300
+ return "artifact"
301
+
302
+
303
+ def _scenario(relative: str) -> str | None:
304
+ parts = PurePosixPath(relative).parts
305
+ if len(parts) >= 4 and parts[0] == "runs":
306
+ return parts[2]
307
+ return None
308
+
309
+
310
+ __all__ = [
311
+ "ARTIFACT_MANIFEST_NAME",
312
+ "ARTIFACT_MANIFEST_VERSION",
313
+ "ArtifactIntegrityError",
314
+ "ArtifactManifest",
315
+ "ArtifactRecord",
316
+ "ScenarioEvidence",
317
+ "load_artifact_manifest",
318
+ "seal_artifacts",
319
+ ]