agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,894 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import random
5
+ from typing import Any, Callable, Iterable, List, Mapping, Optional, Sequence
6
+
7
+ from ..base.base_optimizer import BaseOptimizer
8
+ from ..components import ComponentDiagnosis
9
+ from ..mutations import (
10
+ AgentMutationBundle,
11
+ AgentMutationLibrary,
12
+ dump_mutation_bundle,
13
+ resolve_agent_mutation_library,
14
+ )
15
+ from ..targets import AgentCandidate, CandidateEvaluation, OptimizationTarget
16
+ from ..types import EvaluationResult, IterationHistory, OptimizationResult
17
+ from .agent import (
18
+ _diagnose_candidate_evaluation,
19
+ _dump_model,
20
+ _history_from_candidate,
21
+ _normalize_candidate_evaluation,
22
+ _normalize_diagnoses,
23
+ _target_for_diagnoses,
24
+ )
25
+
26
+ logger = logging.getLogger(__name__)
27
+
28
+
29
+ DEFAULT_LAYER_PATH_BIAS: Mapping[str, Sequence[str]] = {
30
+ "prompt": ("prompt", "instructions", "system"),
31
+ "policy": ("policy", "guardrail", "security", "safety"),
32
+ "tools": ("tools", "tool", "function"),
33
+ "memory": ("memory", "session", "checkpoint"),
34
+ "router": ("router", "routing", "model"),
35
+ "retrieval": ("retrieval", "retriever", "rag", "knowledge"),
36
+ "retriever": ("retrieval", "retriever", "rag", "knowledge"),
37
+ "model": ("model", "router"),
38
+ "voice": ("voice", "audio", "vad", "stt", "tts"),
39
+ "browser": ("browser", "cua", "selectors", "screenshot"),
40
+ "cua": ("browser", "cua", "action"),
41
+ "multi_agent": ("multi_agent", "handoff", "review", "reconciliation"),
42
+ "orchestration": ("orchestration", "graph", "workflow"),
43
+ "streaming": ("streaming", "chunk", "interruption"),
44
+ "world": ("world", "state", "contract"),
45
+ "framework": ("framework", "trace", "adapter"),
46
+ "security": ("security", "policy", "trust"),
47
+ }
48
+
49
+
50
+ class AgentEvolutionOptimizer(BaseOptimizer):
51
+ """
52
+ Optimizes agent configs with deterministic evolutionary mutation.
53
+
54
+ Mutation is domain-aware: config paths tied to active target layers or
55
+ component diagnoses receive higher mutation probability than unrelated
56
+ paths. This is useful when interacting config changes must be discovered
57
+ without exhaustively enumerating the whole search space.
58
+ """
59
+
60
+ def __init__(
61
+ self,
62
+ target: Optional[OptimizationTarget] = None,
63
+ *,
64
+ evaluate_candidate: Optional[
65
+ Callable[[AgentCandidate], CandidateEvaluation | EvaluationResult | float]
66
+ ] = None,
67
+ simulation_evaluator: Any = None,
68
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
69
+ population_size: int = 12,
70
+ generations: int = 4,
71
+ elite_count: int = 2,
72
+ mutation_rate: float = 0.65,
73
+ crossover_rate: float = 0.75,
74
+ max_mutations_per_candidate: int = 2,
75
+ tournament_size: int = 3,
76
+ selection: str = "tournament", # "tournament" (legacy) | "elo" — explicit opt-in
77
+ eval_budget: Optional[int] = None, # declared budget; None = unbounded (legacy)
78
+ elo_k_factor: float = 32.0,
79
+ elo_initial_rating: float = 1500.0, # ARCH Decision 6
80
+ seed: int = 42,
81
+ include_seed: bool = True,
82
+ auto_diagnose: bool = True,
83
+ diagnostic_score_threshold: float = 0.85,
84
+ target_score: Optional[float] = None,
85
+ layer_path_bias: Optional[Mapping[str, Sequence[str]]] = None,
86
+ mutation_library: Optional[
87
+ AgentMutationLibrary | Iterable[AgentMutationBundle] | bool
88
+ ] = True,
89
+ max_library_candidates: int = 8,
90
+ ) -> None:
91
+ _validate_evolution_params(
92
+ population_size=population_size,
93
+ generations=generations,
94
+ elite_count=elite_count,
95
+ mutation_rate=mutation_rate,
96
+ crossover_rate=crossover_rate,
97
+ max_mutations_per_candidate=max_mutations_per_candidate,
98
+ tournament_size=tournament_size,
99
+ )
100
+ _validate_selection_params(
101
+ selection=selection,
102
+ eval_budget=eval_budget,
103
+ elo_k_factor=elo_k_factor,
104
+ elo_initial_rating=elo_initial_rating,
105
+ )
106
+ self.target = target
107
+ self.evaluate_candidate = evaluate_candidate
108
+ self.simulation_evaluator = simulation_evaluator
109
+ self.diagnoses = _normalize_diagnoses(diagnoses)
110
+ self.population_size = population_size
111
+ self.generations = generations
112
+ self.elite_count = elite_count
113
+ self.mutation_rate = mutation_rate
114
+ self.crossover_rate = crossover_rate
115
+ self.max_mutations_per_candidate = max_mutations_per_candidate
116
+ self.tournament_size = tournament_size
117
+ self.selection = selection
118
+ self.eval_budget = eval_budget
119
+ self.elo_k_factor = elo_k_factor
120
+ self.elo_initial_rating = elo_initial_rating
121
+ self.seed = seed
122
+ self.include_seed = include_seed
123
+ self.auto_diagnose = auto_diagnose
124
+ self.diagnostic_score_threshold = diagnostic_score_threshold
125
+ self.target_score = target_score
126
+ self.layer_path_bias = _merged_layer_path_bias(layer_path_bias)
127
+ self.mutation_library = mutation_library
128
+ self.max_library_candidates = max_library_candidates
129
+ super().__init__()
130
+
131
+ def optimize(
132
+ self,
133
+ evaluator: Any = None,
134
+ data_mapper: Any = None,
135
+ dataset: Optional[List[dict[str, Any]]] = None,
136
+ metric: Optional[Callable] = None,
137
+ *,
138
+ target: Optional[OptimizationTarget] = None,
139
+ evaluate_candidate: Optional[
140
+ Callable[[AgentCandidate], CandidateEvaluation | EvaluationResult | float]
141
+ ] = None,
142
+ simulation_evaluator: Any = None,
143
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
144
+ population_size: Optional[int] = None,
145
+ generations: Optional[int] = None,
146
+ elite_count: Optional[int] = None,
147
+ mutation_rate: Optional[float] = None,
148
+ crossover_rate: Optional[float] = None,
149
+ max_mutations_per_candidate: Optional[int] = None,
150
+ tournament_size: Optional[int] = None,
151
+ selection: Optional[str] = None,
152
+ eval_budget: Optional[int] = None,
153
+ elo_k_factor: Optional[float] = None,
154
+ elo_initial_rating: Optional[float] = None,
155
+ seed: Optional[int] = None,
156
+ include_seed: Optional[bool] = None,
157
+ auto_diagnose: Optional[bool] = None,
158
+ diagnostic_score_threshold: Optional[float] = None,
159
+ target_score: Optional[float] = None,
160
+ layer_path_bias: Optional[Mapping[str, Sequence[str]]] = None,
161
+ mutation_library: Optional[
162
+ AgentMutationLibrary | Iterable[AgentMutationBundle] | bool
163
+ ] = None,
164
+ max_library_candidates: Optional[int] = None,
165
+ **kwargs: Any,
166
+ ) -> OptimizationResult:
167
+ active_target = target or self.target
168
+ if active_target is None:
169
+ raise ValueError("AgentEvolutionOptimizer requires a target.")
170
+
171
+ active_evaluator = (
172
+ evaluate_candidate
173
+ or self.evaluate_candidate
174
+ or getattr(simulation_evaluator, "evaluate_candidate", None)
175
+ or getattr(self.simulation_evaluator, "evaluate_candidate", None)
176
+ )
177
+ if active_evaluator is None:
178
+ raise ValueError(
179
+ "AgentEvolutionOptimizer requires evaluate_candidate or simulation_evaluator."
180
+ )
181
+
182
+ active_population_size = (
183
+ self.population_size if population_size is None else population_size
184
+ )
185
+ active_generations = self.generations if generations is None else generations
186
+ active_elite_count = self.elite_count if elite_count is None else elite_count
187
+ active_mutation_rate = self.mutation_rate if mutation_rate is None else mutation_rate
188
+ active_crossover_rate = (
189
+ self.crossover_rate if crossover_rate is None else crossover_rate
190
+ )
191
+ active_max_mutations = (
192
+ self.max_mutations_per_candidate
193
+ if max_mutations_per_candidate is None
194
+ else max_mutations_per_candidate
195
+ )
196
+ active_tournament_size = (
197
+ self.tournament_size if tournament_size is None else tournament_size
198
+ )
199
+ active_selection = self.selection if selection is None else selection
200
+ active_eval_budget = self.eval_budget if eval_budget is None else eval_budget
201
+ active_elo_k_factor = (
202
+ self.elo_k_factor if elo_k_factor is None else elo_k_factor
203
+ )
204
+ active_elo_initial_rating = (
205
+ self.elo_initial_rating
206
+ if elo_initial_rating is None
207
+ else elo_initial_rating
208
+ )
209
+ active_seed = self.seed if seed is None else seed
210
+ use_include_seed = self.include_seed if include_seed is None else include_seed
211
+ use_auto_diagnose = self.auto_diagnose if auto_diagnose is None else auto_diagnose
212
+ active_diagnostic_threshold = (
213
+ self.diagnostic_score_threshold
214
+ if diagnostic_score_threshold is None
215
+ else diagnostic_score_threshold
216
+ )
217
+ active_target_score = self.target_score if target_score is None else target_score
218
+ active_layer_path_bias = (
219
+ self.layer_path_bias
220
+ if layer_path_bias is None
221
+ else _merged_layer_path_bias(layer_path_bias)
222
+ )
223
+ active_mutation_library = resolve_agent_mutation_library(
224
+ self.mutation_library if mutation_library is None else mutation_library
225
+ )
226
+ active_max_library_candidates = (
227
+ self.max_library_candidates
228
+ if max_library_candidates is None
229
+ else max_library_candidates
230
+ )
231
+ _validate_evolution_params(
232
+ population_size=active_population_size,
233
+ generations=active_generations,
234
+ elite_count=active_elite_count,
235
+ mutation_rate=active_mutation_rate,
236
+ crossover_rate=active_crossover_rate,
237
+ max_mutations_per_candidate=active_max_mutations,
238
+ tournament_size=active_tournament_size,
239
+ )
240
+ _validate_selection_params(
241
+ selection=active_selection,
242
+ eval_budget=active_eval_budget,
243
+ elo_k_factor=active_elo_k_factor,
244
+ elo_initial_rating=active_elo_initial_rating,
245
+ )
246
+ if active_max_library_candidates < 0:
247
+ raise ValueError("max_library_candidates must be non-negative.")
248
+
249
+ active_diagnoses = _normalize_diagnoses(diagnoses)
250
+ if diagnoses is None:
251
+ active_diagnoses = list(self.diagnoses)
252
+
253
+ rng = random.Random(active_seed)
254
+ original_target = active_target
255
+ seed_candidate = original_target.seed_candidate()
256
+ evaluated: dict[str, CandidateEvaluation] = {}
257
+ history: List[IterationHistory] = []
258
+ generation_summaries: List[dict[str, Any]] = []
259
+
260
+ if use_auto_diagnose and not active_diagnoses and use_include_seed:
261
+ seed_evaluation = self._evaluate(
262
+ seed_candidate,
263
+ active_evaluator,
264
+ evaluated,
265
+ history,
266
+ generation=0,
267
+ role="seed",
268
+ )
269
+ active_diagnoses = _diagnose_candidate_evaluation(
270
+ seed_evaluation,
271
+ failing_threshold=active_diagnostic_threshold,
272
+ )
273
+
274
+ active_target = _target_for_diagnoses(active_target, active_diagnoses)
275
+ search_paths = [
276
+ path
277
+ for path in active_target.search_space
278
+ if active_target.search_space[path]
279
+ ]
280
+ library_bundles: List[AgentMutationBundle] = []
281
+ if active_mutation_library is not None and active_max_library_candidates:
282
+ library_bundles = active_mutation_library.propose(
283
+ original_target,
284
+ diagnoses=active_diagnoses,
285
+ search_paths=search_paths,
286
+ max_bundles=active_max_library_candidates,
287
+ )
288
+ for bundle in library_bundles:
289
+ for path in bundle.patch:
290
+ if path not in active_target.search_space and path in original_target.search_space:
291
+ active_target.search_space[path] = original_target.search_space[path]
292
+ if path not in search_paths and path in active_target.search_space:
293
+ search_paths.append(path)
294
+ if not search_paths:
295
+ raise ValueError("AgentEvolutionOptimizer target search space cannot be empty.")
296
+
297
+ path_weights = _mutation_path_weights(
298
+ search_paths,
299
+ target=active_target,
300
+ diagnoses=active_diagnoses,
301
+ layer_path_bias=active_layer_path_bias,
302
+ )
303
+ population = _initial_population(
304
+ seed_candidate=seed_candidate,
305
+ search_space=active_target.search_space,
306
+ search_paths=search_paths,
307
+ population_size=active_population_size,
308
+ include_seed=use_include_seed,
309
+ rng=rng,
310
+ path_weights=path_weights,
311
+ library_bundles=library_bundles,
312
+ )
313
+
314
+ best: Optional[CandidateEvaluation] = None
315
+ budget_exhausted = False
316
+ for generation in range(active_generations + 1):
317
+ generation_evaluations: List[CandidateEvaluation] = []
318
+ for candidate in population:
319
+ if (
320
+ active_eval_budget is not None
321
+ and candidate.id not in evaluated
322
+ and len(evaluated) >= active_eval_budget
323
+ ):
324
+ budget_exhausted = True
325
+ break
326
+ evaluation = self._evaluate(
327
+ candidate,
328
+ active_evaluator,
329
+ evaluated,
330
+ history,
331
+ generation=generation,
332
+ role=candidate.metadata.get("evolution_role", "population"),
333
+ )
334
+ generation_evaluations.append(evaluation)
335
+ if best is None or evaluation.score > best.score:
336
+ best = evaluation
337
+
338
+ if generation_evaluations:
339
+ generation_evaluations = sorted(
340
+ generation_evaluations,
341
+ key=lambda item: (
342
+ item.score,
343
+ -len(item.candidate.patch),
344
+ item.candidate.id,
345
+ ),
346
+ reverse=True,
347
+ )
348
+ generation_summaries.append(
349
+ {
350
+ "generation": generation,
351
+ "population": len(population),
352
+ "best_score": generation_evaluations[0].score,
353
+ "best_candidate_id": generation_evaluations[0].candidate.id,
354
+ }
355
+ )
356
+ if budget_exhausted:
357
+ break
358
+ if (
359
+ active_target_score is not None
360
+ and best is not None
361
+ and best.score >= active_target_score
362
+ ):
363
+ break
364
+ if generation >= active_generations:
365
+ break
366
+
367
+ elo_rankings: Optional[List[tuple[CandidateEvaluation, float]]] = None
368
+ if active_selection == "elo":
369
+ # Deterministic round-robin Elo over already-evaluated candidates
370
+ # (RoboPhD discipline): selection pressure changes under a fixed
371
+ # budget; no extra rollouts, no LLM ranking.
372
+ elo_rankings = _elo_tournament_ranking(
373
+ generation_evaluations,
374
+ k_factor=active_elo_k_factor,
375
+ initial_rating=active_elo_initial_rating,
376
+ rng=random.Random(active_seed * 1000003 + generation + 1),
377
+ )
378
+
379
+ elites = [item.candidate for item in generation_evaluations[:active_elite_count]]
380
+ population = _next_population(
381
+ seed_candidate=seed_candidate,
382
+ current_evaluations=generation_evaluations,
383
+ elites=elites,
384
+ search_space=active_target.search_space,
385
+ search_paths=search_paths,
386
+ population_size=active_population_size,
387
+ mutation_rate=active_mutation_rate,
388
+ crossover_rate=active_crossover_rate,
389
+ max_mutations_per_candidate=active_max_mutations,
390
+ tournament_size=active_tournament_size,
391
+ rng=rng,
392
+ path_weights=path_weights,
393
+ generation=generation + 1,
394
+ selection=active_selection,
395
+ elo_rankings=elo_rankings,
396
+ )
397
+
398
+ if best is None:
399
+ raise RuntimeError("AgentEvolutionOptimizer did not evaluate any candidates.")
400
+
401
+ final_elo_rankings: Optional[List[tuple[CandidateEvaluation, float]]] = None
402
+ if active_selection == "elo" and evaluated:
403
+ final_elo_rankings = _elo_tournament_ranking(
404
+ list(evaluated.values()),
405
+ k_factor=active_elo_k_factor,
406
+ initial_rating=active_elo_initial_rating,
407
+ rng=random.Random(active_seed * 1000003),
408
+ )
409
+ # Final-winner selection uses the Elo order instead of raw
410
+ # best-score order (explicit opt-in mode only).
411
+ best = final_elo_rankings[0][0]
412
+
413
+ metadata = {
414
+ "optimizer": "AgentEvolutionOptimizer",
415
+ "strategy": "domain_aware_evolution",
416
+ "target_name": best.candidate.target_name,
417
+ "best_candidate_id": best.candidate.id,
418
+ "search_paths": list(search_paths),
419
+ "population_size": active_population_size,
420
+ "generations": active_generations,
421
+ "elite_count": active_elite_count,
422
+ "mutation_rate": active_mutation_rate,
423
+ "crossover_rate": active_crossover_rate,
424
+ "max_mutations_per_candidate": active_max_mutations,
425
+ "tournament_size": active_tournament_size,
426
+ "selection": active_selection,
427
+ "eval_budget": active_eval_budget,
428
+ "evaluations_used": len(evaluated),
429
+ "seed": active_seed,
430
+ "mutation_path_weights": path_weights,
431
+ "path_weights": path_weights,
432
+ "mutation_library": getattr(active_mutation_library, "name", None)
433
+ if active_mutation_library is not None
434
+ else None,
435
+ "mutation_library_bundles": [
436
+ dump_mutation_bundle(bundle) for bundle in library_bundles
437
+ ],
438
+ "generation_summaries": generation_summaries,
439
+ "evaluated_candidates": len(evaluated),
440
+ }
441
+ if final_elo_rankings is not None:
442
+ metadata["elo_ratings"] = {
443
+ evaluation.candidate.id: rating
444
+ for evaluation, rating in final_elo_rankings
445
+ }
446
+ metadata["elo_k_factor"] = active_elo_k_factor
447
+ metadata["elo_initial_rating"] = active_elo_initial_rating
448
+ if active_diagnoses:
449
+ metadata["diagnostics"] = [_dump_model(item) for item in active_diagnoses]
450
+ metadata["auto_diagnosed"] = use_auto_diagnose
451
+
452
+ return OptimizationResult(
453
+ best_generator=best.candidate,
454
+ best_candidate=best.candidate,
455
+ history=history,
456
+ final_score=best.score,
457
+ total_iterations=len(history),
458
+ total_evaluations=len(history),
459
+ metadata=metadata,
460
+ early_stopped=budget_exhausted,
461
+ stop_reason="eval_budget_exhausted" if budget_exhausted else None,
462
+ )
463
+
464
+ def _evaluate(
465
+ self,
466
+ candidate: AgentCandidate,
467
+ evaluator: Callable[[AgentCandidate], CandidateEvaluation | EvaluationResult | float],
468
+ evaluated: dict[str, CandidateEvaluation],
469
+ history: List[IterationHistory],
470
+ *,
471
+ generation: int,
472
+ role: str,
473
+ ) -> CandidateEvaluation:
474
+ if candidate.id in evaluated:
475
+ return evaluated[candidate.id]
476
+ value = evaluator(candidate)
477
+ evaluation = _normalize_candidate_evaluation(value, candidate)
478
+ evaluation.metadata = {
479
+ **candidate.metadata,
480
+ **evaluation.metadata,
481
+ "optimizer": "AgentEvolutionOptimizer",
482
+ "evolution_generation": generation,
483
+ "evolution_role": role,
484
+ }
485
+ evaluated[candidate.id] = evaluation
486
+ history.append(_history_from_candidate(evaluation))
487
+ logger.info(
488
+ "Evaluated evolution candidate %s score=%.4f generation=%s",
489
+ candidate.id,
490
+ evaluation.score,
491
+ generation,
492
+ )
493
+ return evaluation
494
+
495
+
496
+ def _validate_evolution_params(
497
+ *,
498
+ population_size: int,
499
+ generations: int,
500
+ elite_count: int,
501
+ mutation_rate: float,
502
+ crossover_rate: float,
503
+ max_mutations_per_candidate: int,
504
+ tournament_size: int,
505
+ ) -> None:
506
+ if population_size < 2:
507
+ raise ValueError("population_size must be at least 2.")
508
+ if generations < 0:
509
+ raise ValueError("generations must be non-negative.")
510
+ if elite_count < 1 or elite_count >= population_size:
511
+ raise ValueError("elite_count must be at least 1 and less than population_size.")
512
+ if not 0 <= mutation_rate <= 1:
513
+ raise ValueError("mutation_rate must be between 0 and 1.")
514
+ if not 0 <= crossover_rate <= 1:
515
+ raise ValueError("crossover_rate must be between 0 and 1.")
516
+ if max_mutations_per_candidate < 1:
517
+ raise ValueError("max_mutations_per_candidate must be at least 1.")
518
+ if tournament_size < 1:
519
+ raise ValueError("tournament_size must be at least 1.")
520
+
521
+
522
+ def _merged_layer_path_bias(
523
+ overrides: Optional[Mapping[str, Sequence[str]]],
524
+ ) -> dict[str, tuple[str, ...]]:
525
+ merged = {
526
+ layer: tuple(paths)
527
+ for layer, paths in DEFAULT_LAYER_PATH_BIAS.items()
528
+ }
529
+ for layer, paths in dict(overrides or {}).items():
530
+ merged[layer] = tuple(paths)
531
+ return merged
532
+
533
+
534
+ def _mutation_path_weights(
535
+ search_paths: Sequence[str],
536
+ *,
537
+ target: OptimizationTarget,
538
+ diagnoses: Sequence[ComponentDiagnosis],
539
+ layer_path_bias: Mapping[str, Sequence[str]],
540
+ ) -> dict[str, float]:
541
+ weights: dict[str, float] = {}
542
+ for path in search_paths:
543
+ weight = 1.0
544
+ for layer in target.layers:
545
+ for prefix in layer_path_bias.get(layer, ()):
546
+ if path == prefix or path.startswith(f"{prefix}."):
547
+ weight += 2.0
548
+ for diagnosis in diagnoses:
549
+ if path == diagnosis.component or path.startswith(f"{diagnosis.component}."):
550
+ weight += 3.0 * diagnosis.confidence
551
+ for suggested_path in diagnosis.suggested_paths:
552
+ if path == suggested_path or path.startswith(f"{suggested_path}."):
553
+ weight += 4.0 * diagnosis.confidence
554
+ weights[path] = round(weight, 4)
555
+ return weights
556
+
557
+
558
+ def _initial_population(
559
+ *,
560
+ seed_candidate: AgentCandidate,
561
+ search_space: Mapping[str, List[Any]],
562
+ search_paths: Sequence[str],
563
+ population_size: int,
564
+ include_seed: bool,
565
+ rng: random.Random,
566
+ path_weights: Mapping[str, float],
567
+ library_bundles: Sequence[AgentMutationBundle],
568
+ ) -> List[AgentCandidate]:
569
+ population: List[AgentCandidate] = []
570
+ seen: set[str] = set()
571
+ if include_seed:
572
+ _append_candidate(population, seen, seed_candidate)
573
+
574
+ for bundle in library_bundles:
575
+ candidate = seed_candidate.with_patch(
576
+ dict(bundle.patch),
577
+ metadata={
578
+ "kind": "evolution_library",
579
+ "evolution_role": "library",
580
+ "mutation_bundle": bundle.name,
581
+ "mutation_framework": bundle.framework,
582
+ "mutation_component": bundle.component,
583
+ "mutation_reason": bundle.reason,
584
+ "mutation_tags": list(bundle.tags),
585
+ },
586
+ )
587
+ _append_candidate(population, seen, candidate)
588
+ if len(population) >= population_size:
589
+ return population
590
+
591
+ weighted_paths = sorted(
592
+ search_paths,
593
+ key=lambda path: (path_weights.get(path, 1.0), path),
594
+ reverse=True,
595
+ )
596
+ for path in weighted_paths:
597
+ for value in search_space[path]:
598
+ if seed_candidate.get_path(path) == value:
599
+ continue
600
+ candidate = seed_candidate.with_patch(
601
+ {path: value},
602
+ metadata={
603
+ "kind": "evolution_initial",
604
+ "evolution_role": "mutant",
605
+ },
606
+ )
607
+ _append_candidate(population, seen, candidate)
608
+ if len(population) >= population_size:
609
+ return population
610
+
611
+ attempts = 0
612
+ while len(population) < population_size and attempts < population_size * 20:
613
+ attempts += 1
614
+ patch = _mutated_patch(
615
+ {},
616
+ seed_candidate=seed_candidate,
617
+ search_space=search_space,
618
+ search_paths=search_paths,
619
+ rng=rng,
620
+ path_weights=path_weights,
621
+ max_mutations=2,
622
+ )
623
+ candidate = seed_candidate.with_patch(
624
+ patch,
625
+ metadata={
626
+ "kind": "evolution_initial",
627
+ "evolution_role": "mutant",
628
+ },
629
+ )
630
+ _append_candidate(population, seen, candidate)
631
+ return population
632
+
633
+
634
+ def _next_population(
635
+ *,
636
+ seed_candidate: AgentCandidate,
637
+ current_evaluations: Sequence[CandidateEvaluation],
638
+ elites: Sequence[AgentCandidate],
639
+ search_space: Mapping[str, List[Any]],
640
+ search_paths: Sequence[str],
641
+ population_size: int,
642
+ mutation_rate: float,
643
+ crossover_rate: float,
644
+ max_mutations_per_candidate: int,
645
+ tournament_size: int,
646
+ rng: random.Random,
647
+ path_weights: Mapping[str, float],
648
+ generation: int,
649
+ selection: str = "tournament",
650
+ elo_rankings: Optional[Sequence[tuple[CandidateEvaluation, float]]] = None,
651
+ ) -> List[AgentCandidate]:
652
+ population: List[AgentCandidate] = []
653
+ seen: set[str] = set()
654
+ for elite in elites:
655
+ _append_candidate(population, seen, elite)
656
+
657
+ use_elo = selection == "elo" and bool(elo_rankings)
658
+
659
+ attempts = 0
660
+ while len(population) < population_size and attempts < population_size * 30:
661
+ attempts += 1
662
+ if use_elo:
663
+ parent = _elo_weighted_select(elo_rankings, rng)
664
+ else:
665
+ parent = _tournament_select(current_evaluations, tournament_size, rng)
666
+ patch = dict(parent.candidate.patch)
667
+ role = "mutant"
668
+ parent_ids = [parent.candidate.id]
669
+ if rng.random() < crossover_rate and len(current_evaluations) > 1:
670
+ if use_elo:
671
+ other = _elo_weighted_select(elo_rankings, rng)
672
+ else:
673
+ other = _tournament_select(current_evaluations, tournament_size, rng)
674
+ patch = _crossover_patch(
675
+ patch,
676
+ dict(other.candidate.patch),
677
+ rng,
678
+ )
679
+ parent_ids.append(other.candidate.id)
680
+ role = "crossover"
681
+ if rng.random() < mutation_rate or not patch:
682
+ patch = _mutated_patch(
683
+ patch,
684
+ seed_candidate=seed_candidate,
685
+ search_space=search_space,
686
+ search_paths=search_paths,
687
+ rng=rng,
688
+ path_weights=path_weights,
689
+ max_mutations=max_mutations_per_candidate,
690
+ )
691
+ role = "mutant" if role != "crossover" else "crossover_mutant"
692
+ candidate = seed_candidate.with_patch(
693
+ patch,
694
+ metadata={
695
+ "kind": "evolution_candidate",
696
+ "evolution_role": role,
697
+ "evolution_generation": generation,
698
+ "evolution_parent_ids": parent_ids,
699
+ },
700
+ )
701
+ _append_candidate(population, seen, candidate)
702
+
703
+ if len(population) < population_size:
704
+ for path in search_paths:
705
+ for value in search_space[path]:
706
+ if len(population) >= population_size:
707
+ return population
708
+ patch = {path: value}
709
+ candidate = seed_candidate.with_patch(
710
+ patch,
711
+ metadata={
712
+ "kind": "evolution_backfill",
713
+ "evolution_role": "backfill",
714
+ "evolution_generation": generation,
715
+ },
716
+ )
717
+ _append_candidate(population, seen, candidate)
718
+ return population
719
+
720
+
721
+ def _append_candidate(
722
+ population: List[AgentCandidate],
723
+ seen: set[str],
724
+ candidate: AgentCandidate,
725
+ ) -> None:
726
+ if candidate.id in seen:
727
+ return
728
+ seen.add(candidate.id)
729
+ population.append(candidate)
730
+
731
+
732
+ def _tournament_select(
733
+ evaluations: Sequence[CandidateEvaluation],
734
+ tournament_size: int,
735
+ rng: random.Random,
736
+ ) -> CandidateEvaluation:
737
+ sample_size = min(tournament_size, len(evaluations))
738
+ sample = rng.sample(list(evaluations), sample_size)
739
+ return max(
740
+ sample,
741
+ key=lambda item: (
742
+ item.score,
743
+ -len(item.candidate.patch),
744
+ item.candidate.id,
745
+ ),
746
+ )
747
+
748
+
749
+ def _validate_selection_params(
750
+ *,
751
+ selection: str,
752
+ eval_budget: Optional[int],
753
+ elo_k_factor: float,
754
+ elo_initial_rating: float,
755
+ ) -> None:
756
+ if selection not in {"tournament", "elo"}:
757
+ raise ValueError("selection must be 'tournament' or 'elo'.")
758
+ if eval_budget is not None and eval_budget < 1:
759
+ raise ValueError("eval_budget must be at least 1 when declared.")
760
+ if elo_k_factor <= 0:
761
+ raise ValueError("elo_k_factor must be positive.")
762
+ if elo_initial_rating <= 0:
763
+ raise ValueError("elo_initial_rating must be positive.")
764
+
765
+
766
+ def _elo_tournament_ranking(
767
+ evaluations: Sequence[CandidateEvaluation],
768
+ *,
769
+ k_factor: float,
770
+ initial_rating: float,
771
+ rng: random.Random,
772
+ ) -> List[tuple[CandidateEvaluation, float]]:
773
+ """Deterministic round-robin Elo over already-evaluated candidates.
774
+
775
+ Pairings are seeded-shuffled once; each pair plays one 'match' decided by
776
+ the existing scalar scores (win/draw/loss); ratings update with fixed K.
777
+ Returns (evaluation, rating) sorted by rating desc, candidate.id asc.
778
+ The ranking consumes scores the eval suite already produced — it changes
779
+ selection pressure under a fixed budget, it never adds rollouts and never
780
+ asks any LLM to rank (external-verification rule).
781
+ """
782
+
783
+ unique: dict[str, CandidateEvaluation] = {}
784
+ for evaluation in evaluations:
785
+ unique.setdefault(evaluation.candidate.id, evaluation)
786
+ entries = sorted(unique.values(), key=lambda item: item.candidate.id)
787
+ ratings = {entry.candidate.id: float(initial_rating) for entry in entries}
788
+ pairs = [
789
+ (left_index, right_index)
790
+ for left_index in range(len(entries))
791
+ for right_index in range(left_index + 1, len(entries))
792
+ ]
793
+ rng.shuffle(pairs)
794
+ for left_index, right_index in pairs:
795
+ left = entries[left_index]
796
+ right = entries[right_index]
797
+ left_rating = ratings[left.candidate.id]
798
+ right_rating = ratings[right.candidate.id]
799
+ expected_left = 1.0 / (1.0 + 10 ** ((right_rating - left_rating) / 400.0))
800
+ if left.score > right.score:
801
+ actual_left = 1.0
802
+ elif left.score < right.score:
803
+ actual_left = 0.0
804
+ else:
805
+ actual_left = 0.5
806
+ delta = k_factor * (actual_left - expected_left)
807
+ ratings[left.candidate.id] = left_rating + delta
808
+ ratings[right.candidate.id] = right_rating - delta
809
+ ranked = sorted(
810
+ entries,
811
+ key=lambda item: (-ratings[item.candidate.id], item.candidate.id),
812
+ )
813
+ return [(entry, round(ratings[entry.candidate.id], 4)) for entry in ranked]
814
+
815
+
816
+ def _elo_weighted_select(
817
+ rankings: Sequence[tuple[CandidateEvaluation, float]],
818
+ rng: random.Random,
819
+ ) -> CandidateEvaluation:
820
+ total = sum(max(1.0, rating) for _, rating in rankings)
821
+ threshold = rng.random() * total if total > 0 else 0.0
822
+ running = 0.0
823
+ for evaluation, rating in rankings:
824
+ running += max(1.0, rating)
825
+ if running >= threshold:
826
+ return evaluation
827
+ return rankings[-1][0]
828
+
829
+
830
+ def _crossover_patch(
831
+ left: Mapping[str, Any],
832
+ right: Mapping[str, Any],
833
+ rng: random.Random,
834
+ ) -> dict[str, Any]:
835
+ patch: dict[str, Any] = {}
836
+ for path in sorted(set(left) | set(right)):
837
+ if path in left and path in right:
838
+ patch[path] = left[path] if rng.random() < 0.5 else right[path]
839
+ elif path in left:
840
+ patch[path] = left[path]
841
+ else:
842
+ patch[path] = right[path]
843
+ return patch
844
+
845
+
846
+ def _mutated_patch(
847
+ patch: Mapping[str, Any],
848
+ *,
849
+ seed_candidate: AgentCandidate,
850
+ search_space: Mapping[str, List[Any]],
851
+ search_paths: Sequence[str],
852
+ rng: random.Random,
853
+ path_weights: Mapping[str, float],
854
+ max_mutations: int,
855
+ ) -> dict[str, Any]:
856
+ mutated = dict(patch)
857
+ mutation_count = rng.randint(1, min(max_mutations, len(search_paths)))
858
+ for path in _weighted_sample_paths(search_paths, path_weights, mutation_count, rng):
859
+ values = [
860
+ value
861
+ for value in search_space[path]
862
+ if value != mutated.get(path, seed_candidate.get_path(path))
863
+ ]
864
+ if not values:
865
+ continue
866
+ value = rng.choice(values)
867
+ if value == seed_candidate.get_path(path):
868
+ mutated.pop(path, None)
869
+ else:
870
+ mutated[path] = value
871
+ return mutated
872
+
873
+
874
+ def _weighted_sample_paths(
875
+ search_paths: Sequence[str],
876
+ weights: Mapping[str, float],
877
+ count: int,
878
+ rng: random.Random,
879
+ ) -> List[str]:
880
+ remaining = list(search_paths)
881
+ selected: List[str] = []
882
+ for _ in range(min(count, len(remaining))):
883
+ total = sum(max(0.0, weights.get(path, 1.0)) for path in remaining)
884
+ threshold = rng.random() * total if total > 0 else 0.0
885
+ running = 0.0
886
+ chosen = remaining[-1]
887
+ for path in remaining:
888
+ running += max(0.0, weights.get(path, 1.0))
889
+ if running >= threshold:
890
+ chosen = path
891
+ break
892
+ selected.append(chosen)
893
+ remaining.remove(chosen)
894
+ return selected