agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1863 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import re
5
+ from typing import Any, Callable, Iterable, List, Mapping, Optional, Sequence
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+ from ..base.base_optimizer import BaseOptimizer
10
+ from ..components import (
11
+ ComponentDiagnosis,
12
+ diagnose_agent_report_evaluation,
13
+ diagnose_text,
14
+ relevant_search_paths,
15
+ )
16
+ from ..deployment import (
17
+ AgentDeploymentExport,
18
+ AgentPromotionCheck,
19
+ AgentRollbackDecision,
20
+ check_agent_deployment_rollback,
21
+ export_agent_deployment,
22
+ )
23
+ from ..targets import AgentCandidate, CandidateEvaluation, OptimizationTarget
24
+ from ..types import EvaluationResult, OptimizationResult
25
+ from .agent import AgentOptimizer, _dedupe_diagnoses, _normalize_diagnoses
26
+ from .agent_bandit import AgentBanditOptimizer
27
+ from .agent_curriculum import AgentCurriculumOptimizer
28
+ from .agent_evolution import AgentEvolutionOptimizer
29
+ from .agent_pareto import AgentParetoOptimizer
30
+ from .agent_social_memory import AgentSocialMemoryOptimizer
31
+ from .agent_tpe import AgentTPEOptimizer
32
+ from .council import CouncilAgentOptimizer, SocietyAgentOptimizer
33
+
34
+
35
+ FEEDBACK_SCHEMA_VERSION = "agent-opt.feedback.v1"
36
+ MULTI_INTERACTION_SCHEMA_VERSION = "agent-opt.multi-interaction.v1"
37
+ DEFAULT_MULTI_INTERACTION_BACKENDS = (
38
+ "curriculum",
39
+ "council",
40
+ "society",
41
+ "social_memory",
42
+ "evolution",
43
+ "pareto",
44
+ "tpe",
45
+ "bandit",
46
+ "agent",
47
+ )
48
+ MULTI_INTERACTION_BACKEND_PROFILES: dict[str, dict[str, Any]] = {
49
+ "society": {
50
+ "allocation_kind": "role_graph_society_search",
51
+ "roles": (
52
+ "sutradhara",
53
+ "smriti",
54
+ "arjuna",
55
+ "hanuman",
56
+ "vidura",
57
+ "krishna",
58
+ "sangha",
59
+ "dharma_steward",
60
+ ),
61
+ "role_archetypes": (
62
+ "orchestrator",
63
+ "working_memory",
64
+ "focused_action",
65
+ "bridge_builder",
66
+ "prudent_critic",
67
+ "charioteer_counsel",
68
+ "collective_synthesis",
69
+ "minimal_process_guardian",
70
+ ),
71
+ "path_prefixes": (
72
+ "multi_agent",
73
+ "memory",
74
+ "policy",
75
+ "security",
76
+ "orchestration",
77
+ "framework",
78
+ ),
79
+ "role_path_prefixes": {
80
+ "sutradhara": ("multi_agent", "orchestration", "framework"),
81
+ "smriti": (
82
+ "memory",
83
+ "framework.memory",
84
+ "framework.checkpoints",
85
+ "framework.sessions",
86
+ ),
87
+ "arjuna": ("tools", "action", "policy"),
88
+ "hanuman": ("multi_agent", "framework", "orchestration"),
89
+ "vidura": ("policy", "security", "adversarial"),
90
+ "krishna": ("multi_agent", "memory", "policy"),
91
+ "sangha": (),
92
+ "dharma_steward": ("policy", "security", "reliability"),
93
+ },
94
+ },
95
+ "council": {
96
+ "allocation_kind": "council_deliberation",
97
+ "roles": ("explorer", "critic", "synthesizer", "steward"),
98
+ "role_archetypes": (
99
+ "exploration",
100
+ "critique",
101
+ "synthesis",
102
+ "process_guardian",
103
+ ),
104
+ "path_prefixes": ("multi_agent", "memory", "policy", "tools", "framework"),
105
+ "role_path_prefixes": {
106
+ "explorer": (),
107
+ "critic": ("policy", "security", "adversarial"),
108
+ "synthesizer": (),
109
+ "steward": ("policy", "reliability", "framework"),
110
+ },
111
+ },
112
+ "social_memory": {
113
+ "allocation_kind": "social_memory_credit_ledger",
114
+ "roles": ("smriti", "arjuna", "vidura", "sangha", "dharma_steward"),
115
+ "role_archetypes": (
116
+ "working_memory",
117
+ "focused_action",
118
+ "prudent_critic",
119
+ "collective_synthesis",
120
+ "minimal_process_guardian",
121
+ ),
122
+ "path_prefixes": ("memory", "multi_agent", "policy", "framework"),
123
+ "role_path_prefixes": {
124
+ "smriti": (
125
+ "memory",
126
+ "framework.memory",
127
+ "framework.checkpoints",
128
+ "framework.sessions",
129
+ ),
130
+ "arjuna": ("multi_agent", "tools", "action"),
131
+ "vidura": ("policy", "security", "adversarial"),
132
+ "sangha": (),
133
+ "dharma_steward": ("policy", "reliability", "framework"),
134
+ },
135
+ },
136
+ "curriculum": {
137
+ "allocation_kind": "deliberate_practice_curriculum",
138
+ "roles": ("teacher", "student", "coach"),
139
+ "role_archetypes": (
140
+ "staged_practice",
141
+ "metric_drill",
142
+ "remediation_coach",
143
+ ),
144
+ "path_prefixes": ("objective", "planner", "memory", "policy", "framework"),
145
+ "role_path_prefixes": {
146
+ "teacher": ("objective", "evaluation", "framework"),
147
+ "student": (),
148
+ "coach": ("memory", "policy", "planner"),
149
+ },
150
+ },
151
+ "evolution": {
152
+ "allocation_kind": "evolutionary_exploration",
153
+ "roles": ("population_explorer", "mutation_stressor", "fitness_selector"),
154
+ "role_archetypes": ("variation", "stress", "selection"),
155
+ "path_prefixes": (),
156
+ "role_path_prefixes": {
157
+ "population_explorer": (),
158
+ "mutation_stressor": ("security", "policy", "tools", "framework"),
159
+ "fitness_selector": (),
160
+ },
161
+ },
162
+ "pareto": {
163
+ "allocation_kind": "pareto_tradeoff_search",
164
+ "roles": ("tradeoff_arbiter", "frontier_keeper"),
165
+ "role_archetypes": ("multi_objective_balance", "frontier_selection"),
166
+ "path_prefixes": (),
167
+ "role_path_prefixes": {
168
+ "tradeoff_arbiter": (),
169
+ "frontier_keeper": (),
170
+ },
171
+ },
172
+ "tpe": {
173
+ "allocation_kind": "tpe_prior_sampling",
174
+ "roles": ("prior_sampler", "density_estimator"),
175
+ "role_archetypes": ("probabilistic_prior", "expected_improvement"),
176
+ "path_prefixes": (),
177
+ "role_path_prefixes": {
178
+ "prior_sampler": (),
179
+ "density_estimator": (),
180
+ },
181
+ },
182
+ "bandit": {
183
+ "allocation_kind": "bandit_budget_allocation",
184
+ "roles": ("allocation_arbiter", "exploit_explore_allocator"),
185
+ "role_archetypes": ("budget_allocator", "adaptive_selection"),
186
+ "path_prefixes": (),
187
+ "role_path_prefixes": {
188
+ "allocation_arbiter": (),
189
+ "exploit_explore_allocator": (),
190
+ },
191
+ },
192
+ "agent": {
193
+ "allocation_kind": "deterministic_candidate_search",
194
+ "roles": ("deterministic_engineer",),
195
+ "role_archetypes": ("metric_patch_search",),
196
+ "path_prefixes": (),
197
+ "role_path_prefixes": {"deterministic_engineer": ()},
198
+ },
199
+ }
200
+ DeploymentLike = (
201
+ AgentPromotionCheck
202
+ | AgentDeploymentExport
203
+ | OptimizationResult
204
+ | AgentCandidate
205
+ | Mapping[str, Any]
206
+ )
207
+ CandidateScorer = Callable[
208
+ [AgentCandidate],
209
+ CandidateEvaluation | EvaluationResult | float,
210
+ ]
211
+
212
+
213
+ class AgentFeedbackCase(BaseModel):
214
+ """One production or replayed feedback observation used for re-optimization."""
215
+
216
+ index: int
217
+ source: str = "rollback_observation"
218
+ candidate_id: Optional[str] = None
219
+ score: float
220
+ passed: bool
221
+ failures: list[str] = Field(default_factory=list)
222
+ metrics: dict[str, float] = Field(default_factory=dict)
223
+ metadata: dict[str, Any] = Field(default_factory=dict)
224
+
225
+
226
+ class AgentFeedbackOptimizationResult(BaseModel):
227
+ """Audit record for a live-feedback-triggered optimization round."""
228
+
229
+ schema_version: str = FEEDBACK_SCHEMA_VERSION
230
+ optimizer: str
231
+ feedback_source: str
232
+ rollback_decision: AgentRollbackDecision
233
+ feedback_cases: list[AgentFeedbackCase] = Field(default_factory=list)
234
+ diagnoses: list[ComponentDiagnosis] = Field(default_factory=list)
235
+ search_paths: list[str] = Field(default_factory=list)
236
+ reoptimization_result: OptimizationResult
237
+ baseline_score: Optional[float] = None
238
+ feedback_score: Optional[float] = None
239
+ final_score: float
240
+ baseline_delta: Optional[float] = None
241
+ feedback_delta: Optional[float] = None
242
+ improved: bool
243
+ metadata: dict[str, Any] = Field(default_factory=dict)
244
+
245
+ def to_manifest(self) -> dict[str, Any]:
246
+ return self.model_dump()
247
+
248
+ def to_json(self, *, indent: int = 2) -> str:
249
+ return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
250
+
251
+
252
+ class AgentMultiInteractionBackendPlan(BaseModel):
253
+ """One deterministic backend allocation in a multi-interaction round."""
254
+
255
+ optimizer: str
256
+ rank: int
257
+ weight: float
258
+ reason: str
259
+ kwargs: dict[str, Any] = Field(default_factory=dict)
260
+
261
+
262
+ class AgentMultiInteractionBackendRun(BaseModel):
263
+ """Result from running one allocated optimizer backend."""
264
+
265
+ optimizer: str
266
+ rank: int
267
+ status: str
268
+ final_score: Optional[float] = None
269
+ improved: bool = False
270
+ total_evaluations: int = 0
271
+ failure: Optional[str] = None
272
+ result: Optional[AgentFeedbackOptimizationResult] = None
273
+ metadata: dict[str, Any] = Field(default_factory=dict)
274
+
275
+
276
+ class AgentMultiInteractionBackendLineage(BaseModel):
277
+ """Candidate contribution summary for one backend in a portfolio run."""
278
+
279
+ optimizer: str
280
+ rank: int
281
+ allocation_weight: float = 0.0
282
+ allocation_reason: str = ""
283
+ status: str
284
+ final_score: Optional[float] = None
285
+ improved: bool = False
286
+ total_evaluations: int = 0
287
+ candidate_id: Optional[str] = None
288
+ parent_candidate_id: Optional[str] = None
289
+ candidate_patch: dict[str, Any] = Field(default_factory=dict)
290
+ patch_paths: list[str] = Field(default_factory=list)
291
+ unique_candidate_patch: dict[str, Any] = Field(default_factory=dict)
292
+ unique_patch_paths: list[str] = Field(default_factory=list)
293
+ shared_candidate_patch: dict[str, Any] = Field(default_factory=dict)
294
+ shared_patch_paths: list[str] = Field(default_factory=list)
295
+ equivalent_backends: list[str] = Field(default_factory=list)
296
+ equivalent_backend_count: int = 0
297
+ selection_relation: str = "unclassified"
298
+ metadata: dict[str, Any] = Field(default_factory=dict)
299
+
300
+
301
+ class AgentMultiInteractionAblationReport(BaseModel):
302
+ """Leave-one-backend-out summary for the selected portfolio result."""
303
+
304
+ selected_optimizer: str
305
+ selected_candidate_id: Optional[str] = None
306
+ selected_patch: dict[str, Any] = Field(default_factory=dict)
307
+ selected_patch_paths: list[str] = Field(default_factory=list)
308
+ final_score: float
309
+ best_without_selected_optimizer: Optional[str] = None
310
+ best_without_selected_score: Optional[float] = None
311
+ score_delta_without_selected: Optional[float] = None
312
+ selected_backend_required: bool
313
+ dependency: str
314
+ dependency_reason: str
315
+ consensus_backends: list[str] = Field(default_factory=list)
316
+ consensus_backend_count: int = 0
317
+ shared_selected_patch_paths: list[str] = Field(default_factory=list)
318
+ unique_selected_patch_paths: list[str] = Field(default_factory=list)
319
+ selected_patch_support: dict[str, list[str]] = Field(default_factory=dict)
320
+ backend_scoreboard: list[dict[str, Any]] = Field(default_factory=list)
321
+
322
+
323
+ class AgentMultiInteractionOptimizationResult(BaseModel):
324
+ """Audit record for automatic multi-backend agent re-optimization."""
325
+
326
+ schema_version: str = MULTI_INTERACTION_SCHEMA_VERSION
327
+ selected_optimizer: str
328
+ feedback_source: str
329
+ rollback_decision: AgentRollbackDecision
330
+ feedback_cases: list[AgentFeedbackCase] = Field(default_factory=list)
331
+ diagnoses: list[ComponentDiagnosis] = Field(default_factory=list)
332
+ search_paths: list[str] = Field(default_factory=list)
333
+ backend_plan: list[AgentMultiInteractionBackendPlan] = Field(default_factory=list)
334
+ backend_runs: list[AgentMultiInteractionBackendRun] = Field(default_factory=list)
335
+ backend_lineage: list[AgentMultiInteractionBackendLineage] = Field(default_factory=list)
336
+ ablation_report: AgentMultiInteractionAblationReport
337
+ best_result: AgentFeedbackOptimizationResult
338
+ final_score: float
339
+ improved: bool
340
+ metadata: dict[str, Any] = Field(default_factory=dict)
341
+
342
+ def to_manifest(self) -> dict[str, Any]:
343
+ return self.model_dump()
344
+
345
+ def to_json(self, *, indent: int = 2) -> str:
346
+ return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
347
+
348
+
349
+ class AgentFeedbackOptimizer(BaseOptimizer):
350
+ """
351
+ Re-optimize an agent from live trace/evaluation feedback.
352
+
353
+ The optimizer first turns post-deployment rollback evidence into component
354
+ diagnoses and search paths, then delegates the actual search to one of the
355
+ existing agent optimizers (`society`, `social_memory`, `curriculum`,
356
+ `council`, `evolution`, `tpe`, `pareto`, `bandit`, or deterministic
357
+ `agent`).
358
+ """
359
+
360
+ def __init__(
361
+ self,
362
+ target: Optional[OptimizationTarget] = None,
363
+ *,
364
+ deployment: Optional[DeploymentLike] = None,
365
+ rollback_decision: Optional[AgentRollbackDecision] = None,
366
+ live_evaluations: Optional[Sequence[Any]] = None,
367
+ evaluate_candidate: Optional[CandidateScorer] = None,
368
+ simulation_evaluator: Any = None,
369
+ optimizer: str = "society",
370
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
371
+ diagnostic_score_threshold: float = 0.85,
372
+ optimizer_kwargs: Optional[Mapping[str, Any]] = None,
373
+ rollback_kwargs: Optional[Mapping[str, Any]] = None,
374
+ metadata: Optional[Mapping[str, Any]] = None,
375
+ ) -> None:
376
+ self.target = target
377
+ self.deployment = deployment
378
+ self.rollback_decision = rollback_decision
379
+ self.live_evaluations = (
380
+ list(live_evaluations) if live_evaluations is not None else None
381
+ )
382
+ self.evaluate_candidate = evaluate_candidate
383
+ self.simulation_evaluator = simulation_evaluator
384
+ self.optimizer = optimizer
385
+ self.diagnoses = _normalize_diagnoses(diagnoses)
386
+ self.diagnostic_score_threshold = diagnostic_score_threshold
387
+ self.optimizer_kwargs = dict(optimizer_kwargs or {})
388
+ self.rollback_kwargs = dict(rollback_kwargs or {})
389
+ self.metadata = dict(metadata or {})
390
+ super().__init__()
391
+
392
+ def optimize(
393
+ self,
394
+ evaluator: Any = None,
395
+ data_mapper: Any = None,
396
+ dataset: Optional[List[dict[str, Any]]] = None,
397
+ metric: Optional[Callable] = None,
398
+ *,
399
+ target: Optional[OptimizationTarget] = None,
400
+ deployment: Optional[DeploymentLike] = None,
401
+ rollback_decision: Optional[AgentRollbackDecision] = None,
402
+ live_evaluations: Optional[Sequence[Any]] = None,
403
+ evaluate_candidate: Optional[CandidateScorer] = None,
404
+ simulation_evaluator: Any = None,
405
+ optimizer: Optional[str] = None,
406
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
407
+ diagnostic_score_threshold: Optional[float] = None,
408
+ optimizer_kwargs: Optional[Mapping[str, Any]] = None,
409
+ rollback_kwargs: Optional[Mapping[str, Any]] = None,
410
+ metadata: Optional[Mapping[str, Any]] = None,
411
+ **backend_kwargs: Any,
412
+ ) -> AgentFeedbackOptimizationResult:
413
+ active_target = target or self.target
414
+ if active_target is None:
415
+ raise ValueError("AgentFeedbackOptimizer requires a target.")
416
+
417
+ active_evaluator = evaluate_candidate or self.evaluate_candidate
418
+ active_simulation = simulation_evaluator or self.simulation_evaluator
419
+ if (
420
+ active_evaluator is None
421
+ and getattr(active_simulation, "evaluate_candidate", None) is None
422
+ ):
423
+ raise ValueError(
424
+ "AgentFeedbackOptimizer requires evaluate_candidate or simulation_evaluator."
425
+ )
426
+
427
+ active_diagnostic_threshold = (
428
+ self.diagnostic_score_threshold
429
+ if diagnostic_score_threshold is None
430
+ else diagnostic_score_threshold
431
+ )
432
+ explicit_diagnoses = _normalize_diagnoses(diagnoses)
433
+ if diagnoses is None:
434
+ explicit_diagnoses = list(self.diagnoses)
435
+
436
+ active_rollback_decision = rollback_decision or self.rollback_decision
437
+ active_live_evaluations = (
438
+ list(live_evaluations)
439
+ if live_evaluations is not None
440
+ else self.live_evaluations
441
+ )
442
+ active_deployment = deployment or self.deployment
443
+ active_deployment, auto_seed_deployment = _auto_seed_deployment_for_replay(
444
+ target=active_target,
445
+ deployment=active_deployment,
446
+ rollback_decision=active_rollback_decision,
447
+ live_evaluations=active_live_evaluations,
448
+ simulation_evaluator=active_simulation,
449
+ metadata={**self.metadata, **dict(metadata or {})},
450
+ )
451
+ decision, feedback_source = _resolve_rollback_decision(
452
+ rollback_decision=active_rollback_decision,
453
+ deployment=active_deployment,
454
+ live_evaluations=active_live_evaluations,
455
+ simulation_evaluator=active_simulation,
456
+ rollback_kwargs={
457
+ **self.rollback_kwargs,
458
+ **dict(rollback_kwargs or {}),
459
+ },
460
+ )
461
+ feedback_cases = _feedback_cases_from_rollback(decision)
462
+ feedback_diagnoses = _diagnose_feedback_cases(
463
+ feedback_cases,
464
+ target=active_target,
465
+ failing_threshold=active_diagnostic_threshold,
466
+ )
467
+ active_diagnoses = _dedupe_diagnoses([*explicit_diagnoses, *feedback_diagnoses])
468
+ search_paths = _search_paths_for_feedback(active_target, active_diagnoses)
469
+
470
+ backend_name = optimizer or self.optimizer
471
+ resolved_optimizer = _resolve_feedback_optimizer(backend_name)
472
+ combined_backend_kwargs = {
473
+ **self.optimizer_kwargs,
474
+ **dict(optimizer_kwargs or {}),
475
+ **backend_kwargs,
476
+ }
477
+ backend = resolved_optimizer(
478
+ target=active_target,
479
+ evaluate_candidate=active_evaluator,
480
+ simulation_evaluator=active_simulation,
481
+ diagnoses=active_diagnoses,
482
+ diagnostic_score_threshold=active_diagnostic_threshold,
483
+ **combined_backend_kwargs,
484
+ )
485
+ reoptimization = backend.optimize()
486
+ baseline_score = decision.baseline_score
487
+ feedback_score = decision.latest_score
488
+ baseline_delta = (
489
+ reoptimization.final_score - baseline_score
490
+ if baseline_score is not None
491
+ else None
492
+ )
493
+ feedback_delta = (
494
+ reoptimization.final_score - feedback_score
495
+ if feedback_score is not None
496
+ else None
497
+ )
498
+ improved = (
499
+ reoptimization.final_score >= decision.min_score
500
+ and (feedback_delta is None or feedback_delta > 0)
501
+ )
502
+ result_metadata = {
503
+ **self.metadata,
504
+ **dict(metadata or {}),
505
+ "rollback_required": decision.rollback_required,
506
+ "failure_count": decision.failure_count,
507
+ "consecutive_failure_count": decision.consecutive_failure_count,
508
+ "auto_seed_deployment": auto_seed_deployment,
509
+ "backend_optimizer": reoptimization.metadata.get("optimizer"),
510
+ }
511
+ return AgentFeedbackOptimizationResult(
512
+ optimizer=_normalize_optimizer_name(backend_name),
513
+ feedback_source=feedback_source,
514
+ rollback_decision=decision,
515
+ feedback_cases=feedback_cases,
516
+ diagnoses=active_diagnoses,
517
+ search_paths=search_paths,
518
+ reoptimization_result=reoptimization,
519
+ baseline_score=baseline_score,
520
+ feedback_score=feedback_score,
521
+ final_score=reoptimization.final_score,
522
+ baseline_delta=baseline_delta,
523
+ feedback_delta=feedback_delta,
524
+ improved=improved,
525
+ metadata=result_metadata,
526
+ )
527
+
528
+
529
+ class AgentMultiInteractionOptimizer(BaseOptimizer):
530
+ """
531
+ Diagnose feedback, allocate deterministic optimizer backends, and select the best.
532
+
533
+ This is the Future AGI-native portfolio layer above `AgentFeedbackOptimizer`:
534
+ every backend receives the same rollback/replay evidence and metric-derived
535
+ diagnoses, while the allocator chooses backend priority from feedback
536
+ metrics, target layers, and search-space shape. Social/psychological
537
+ inspiration stays metadata-only; candidate acceptance is numeric.
538
+ """
539
+
540
+ def __init__(
541
+ self,
542
+ target: Optional[OptimizationTarget] = None,
543
+ *,
544
+ deployment: Optional[DeploymentLike] = None,
545
+ rollback_decision: Optional[AgentRollbackDecision] = None,
546
+ live_evaluations: Optional[Sequence[Any]] = None,
547
+ evaluate_candidate: Optional[CandidateScorer] = None,
548
+ simulation_evaluator: Any = None,
549
+ optimizer_pool: Optional[Sequence[str]] = None,
550
+ max_backends: Optional[int] = None,
551
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
552
+ diagnostic_score_threshold: float = 0.85,
553
+ optimizer_kwargs: Optional[Mapping[str, Any]] = None,
554
+ optimizer_kwargs_by_backend: Optional[Mapping[str, Mapping[str, Any]]] = None,
555
+ rollback_kwargs: Optional[Mapping[str, Any]] = None,
556
+ metadata: Optional[Mapping[str, Any]] = None,
557
+ ) -> None:
558
+ self.target = target
559
+ self.deployment = deployment
560
+ self.rollback_decision = rollback_decision
561
+ self.live_evaluations = (
562
+ list(live_evaluations) if live_evaluations is not None else None
563
+ )
564
+ self.evaluate_candidate = evaluate_candidate
565
+ self.simulation_evaluator = simulation_evaluator
566
+ self.optimizer_pool = list(optimizer_pool) if optimizer_pool is not None else None
567
+ self.max_backends = max_backends
568
+ self.diagnoses = _normalize_diagnoses(diagnoses)
569
+ self.diagnostic_score_threshold = diagnostic_score_threshold
570
+ self.optimizer_kwargs = dict(optimizer_kwargs or {})
571
+ self.optimizer_kwargs_by_backend = {
572
+ _normalize_optimizer_name(key): dict(value)
573
+ for key, value in dict(optimizer_kwargs_by_backend or {}).items()
574
+ }
575
+ self.rollback_kwargs = dict(rollback_kwargs or {})
576
+ self.metadata = dict(metadata or {})
577
+ super().__init__()
578
+
579
+ def optimize(
580
+ self,
581
+ evaluator: Any = None,
582
+ data_mapper: Any = None,
583
+ dataset: Optional[List[dict[str, Any]]] = None,
584
+ metric: Optional[Callable] = None,
585
+ *,
586
+ target: Optional[OptimizationTarget] = None,
587
+ deployment: Optional[DeploymentLike] = None,
588
+ rollback_decision: Optional[AgentRollbackDecision] = None,
589
+ live_evaluations: Optional[Sequence[Any]] = None,
590
+ evaluate_candidate: Optional[CandidateScorer] = None,
591
+ simulation_evaluator: Any = None,
592
+ optimizer_pool: Optional[Sequence[str]] = None,
593
+ max_backends: Optional[int] = None,
594
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
595
+ diagnostic_score_threshold: Optional[float] = None,
596
+ optimizer_kwargs: Optional[Mapping[str, Any]] = None,
597
+ optimizer_kwargs_by_backend: Optional[Mapping[str, Mapping[str, Any]]] = None,
598
+ rollback_kwargs: Optional[Mapping[str, Any]] = None,
599
+ metadata: Optional[Mapping[str, Any]] = None,
600
+ **backend_kwargs: Any,
601
+ ) -> AgentMultiInteractionOptimizationResult:
602
+ active_target = target or self.target
603
+ if active_target is None:
604
+ raise ValueError("AgentMultiInteractionOptimizer requires a target.")
605
+
606
+ active_evaluator = evaluate_candidate or self.evaluate_candidate or evaluator
607
+ active_simulation = simulation_evaluator or self.simulation_evaluator
608
+ if (
609
+ active_evaluator is None
610
+ and getattr(active_simulation, "evaluate_candidate", None) is None
611
+ ):
612
+ raise ValueError(
613
+ "AgentMultiInteractionOptimizer requires evaluate_candidate or simulation_evaluator."
614
+ )
615
+
616
+ active_diagnostic_threshold = (
617
+ self.diagnostic_score_threshold
618
+ if diagnostic_score_threshold is None
619
+ else diagnostic_score_threshold
620
+ )
621
+ explicit_diagnoses = _normalize_diagnoses(diagnoses)
622
+ if diagnoses is None:
623
+ explicit_diagnoses = list(self.diagnoses)
624
+
625
+ active_rollback_decision = rollback_decision or self.rollback_decision
626
+ active_live_evaluations = (
627
+ list(live_evaluations)
628
+ if live_evaluations is not None
629
+ else self.live_evaluations
630
+ )
631
+ active_deployment = deployment or self.deployment
632
+ active_deployment, auto_seed_deployment = _auto_seed_deployment_for_replay(
633
+ target=active_target,
634
+ deployment=active_deployment,
635
+ rollback_decision=active_rollback_decision,
636
+ live_evaluations=active_live_evaluations,
637
+ simulation_evaluator=active_simulation,
638
+ metadata={**self.metadata, **dict(metadata or {})},
639
+ )
640
+ decision, feedback_source = _resolve_rollback_decision(
641
+ rollback_decision=active_rollback_decision,
642
+ deployment=active_deployment,
643
+ live_evaluations=active_live_evaluations,
644
+ simulation_evaluator=active_simulation,
645
+ rollback_kwargs={
646
+ **self.rollback_kwargs,
647
+ **dict(rollback_kwargs or {}),
648
+ },
649
+ )
650
+ feedback_cases = _feedback_cases_from_rollback(decision)
651
+ feedback_diagnoses = _diagnose_feedback_cases(
652
+ feedback_cases,
653
+ target=active_target,
654
+ failing_threshold=active_diagnostic_threshold,
655
+ )
656
+ active_diagnoses = _dedupe_diagnoses([*explicit_diagnoses, *feedback_diagnoses])
657
+ search_paths = _search_paths_for_feedback(active_target, active_diagnoses)
658
+
659
+ base_optimizer_kwargs = {
660
+ **self.optimizer_kwargs,
661
+ **dict(optimizer_kwargs or {}),
662
+ **backend_kwargs,
663
+ }
664
+ per_backend_kwargs = dict(self.optimizer_kwargs_by_backend)
665
+ for key, value in dict(optimizer_kwargs_by_backend or {}).items():
666
+ per_backend_kwargs[_normalize_optimizer_name(key)] = dict(value)
667
+
668
+ plan = _multi_interaction_backend_plan(
669
+ target=active_target,
670
+ feedback_cases=feedback_cases,
671
+ diagnoses=active_diagnoses,
672
+ search_paths=search_paths,
673
+ optimizer_pool=optimizer_pool or self.optimizer_pool,
674
+ max_backends=self.max_backends if max_backends is None else max_backends,
675
+ optimizer_kwargs=base_optimizer_kwargs,
676
+ optimizer_kwargs_by_backend=per_backend_kwargs,
677
+ )
678
+ if not plan:
679
+ raise ValueError("AgentMultiInteractionOptimizer backend plan cannot be empty.")
680
+
681
+ runs: list[AgentMultiInteractionBackendRun] = []
682
+ for allocation in plan:
683
+ try:
684
+ result = AgentFeedbackOptimizer(
685
+ target=active_target,
686
+ rollback_decision=decision,
687
+ evaluate_candidate=active_evaluator,
688
+ simulation_evaluator=active_simulation,
689
+ optimizer=allocation.optimizer,
690
+ diagnoses=active_diagnoses,
691
+ diagnostic_score_threshold=active_diagnostic_threshold,
692
+ optimizer_kwargs=allocation.kwargs,
693
+ metadata={
694
+ "multi_interaction_optimizer": True,
695
+ "backend_rank": allocation.rank,
696
+ "backend_weight": allocation.weight,
697
+ "backend_reason": allocation.reason,
698
+ },
699
+ ).optimize()
700
+ runs.append(
701
+ AgentMultiInteractionBackendRun(
702
+ optimizer=allocation.optimizer,
703
+ rank=allocation.rank,
704
+ status="completed",
705
+ final_score=result.final_score,
706
+ improved=result.improved,
707
+ total_evaluations=result.reoptimization_result.total_evaluations,
708
+ result=result,
709
+ metadata={
710
+ "backend_optimizer": result.metadata.get("backend_optimizer"),
711
+ },
712
+ )
713
+ )
714
+ except Exception as exc:
715
+ runs.append(
716
+ AgentMultiInteractionBackendRun(
717
+ optimizer=allocation.optimizer,
718
+ rank=allocation.rank,
719
+ status="failed",
720
+ failure=str(exc),
721
+ )
722
+ )
723
+
724
+ successful_runs = [run for run in runs if run.result is not None]
725
+ if not successful_runs:
726
+ failures = "; ".join(
727
+ f"{run.optimizer}: {run.failure}" for run in runs if run.failure
728
+ )
729
+ raise RuntimeError(
730
+ "AgentMultiInteractionOptimizer did not complete any backend"
731
+ + (f": {failures}" if failures else ".")
732
+ )
733
+ best_run = max(
734
+ successful_runs,
735
+ key=lambda run: (
736
+ run.final_score if run.final_score is not None else float("-inf"),
737
+ 1 if run.improved else 0,
738
+ -run.rank,
739
+ -run.total_evaluations,
740
+ ),
741
+ )
742
+ assert best_run.result is not None
743
+ backend_lineage = _multi_interaction_backend_lineage(
744
+ target=active_target,
745
+ plan=plan,
746
+ runs=runs,
747
+ selected_run=best_run,
748
+ )
749
+ ablation_report = _multi_interaction_ablation_report(
750
+ lineage=backend_lineage,
751
+ selected_run=best_run,
752
+ )
753
+ allocation_metadata = _multi_interaction_allocation_metadata(
754
+ target=active_target,
755
+ plan=plan,
756
+ feedback_cases=feedback_cases,
757
+ diagnoses=active_diagnoses,
758
+ search_paths=search_paths,
759
+ )
760
+ result_metadata = {
761
+ **self.metadata,
762
+ **dict(metadata or {}),
763
+ "allocator": "metric_diagnosis_backend_portfolio",
764
+ **allocation_metadata,
765
+ "auto_seed_deployment": auto_seed_deployment,
766
+ "backend_count": len(plan),
767
+ "completed_backend_count": len(successful_runs),
768
+ "failed_backend_count": len(runs) - len(successful_runs),
769
+ "optimizer_pool": [allocation.optimizer for allocation in plan],
770
+ "selection_rule": "highest_final_score_then_improved_then_rank",
771
+ "ablation_dependency": ablation_report.dependency,
772
+ "selected_backend_required": ablation_report.selected_backend_required,
773
+ "consensus_backend_count": ablation_report.consensus_backend_count,
774
+ "selected_patch_paths": list(ablation_report.selected_patch_paths),
775
+ "strategy_inspiration": (
776
+ "diagnostic triage, deliberate practice, council synthesis, "
777
+ "social memory, evolutionary exploration, Pareto tradeoff, "
778
+ "TPE sampling, bandit allocation, human team roles, and "
779
+ "Hindu-mythology-inspired society labels; labels are metadata only"
780
+ ),
781
+ }
782
+ return AgentMultiInteractionOptimizationResult(
783
+ selected_optimizer=best_run.optimizer,
784
+ feedback_source=feedback_source,
785
+ rollback_decision=decision,
786
+ feedback_cases=feedback_cases,
787
+ diagnoses=active_diagnoses,
788
+ search_paths=search_paths,
789
+ backend_plan=plan,
790
+ backend_runs=runs,
791
+ backend_lineage=backend_lineage,
792
+ ablation_report=ablation_report,
793
+ best_result=best_run.result,
794
+ final_score=best_run.result.final_score,
795
+ improved=best_run.result.improved,
796
+ metadata=result_metadata,
797
+ )
798
+
799
+
800
+ def _multi_interaction_backend_plan(
801
+ *,
802
+ target: OptimizationTarget,
803
+ feedback_cases: Sequence[AgentFeedbackCase],
804
+ diagnoses: Sequence[ComponentDiagnosis],
805
+ search_paths: Sequence[str],
806
+ optimizer_pool: Optional[Sequence[str]],
807
+ max_backends: Optional[int],
808
+ optimizer_kwargs: Mapping[str, Any],
809
+ optimizer_kwargs_by_backend: Mapping[str, Mapping[str, Any]],
810
+ ) -> list[AgentMultiInteractionBackendPlan]:
811
+ if max_backends is not None and max_backends < 1:
812
+ raise ValueError("max_backends must be at least 1.")
813
+
814
+ metric_names = _failed_feedback_metric_names(feedback_cases) or _feedback_metric_names(
815
+ feedback_cases
816
+ )
817
+ normalized_pool = _dedupe_optimizer_pool(optimizer_pool or DEFAULT_MULTI_INTERACTION_BACKENDS)
818
+ scored: list[tuple[float, int, str, str, dict[str, Any]]] = []
819
+ default_order = {
820
+ name: index for index, name in enumerate(DEFAULT_MULTI_INTERACTION_BACKENDS)
821
+ }
822
+ for optimizer_name in normalized_pool:
823
+ _resolve_feedback_optimizer(optimizer_name)
824
+ backend_kwargs = _backend_kwargs_for_multi_interaction(
825
+ optimizer_name,
826
+ target=target,
827
+ metric_names=metric_names,
828
+ optimizer_kwargs=optimizer_kwargs,
829
+ optimizer_kwargs_by_backend=optimizer_kwargs_by_backend,
830
+ )
831
+ if optimizer_name == "pareto" and not backend_kwargs.get("objective_names"):
832
+ continue
833
+ weight, reason = _backend_allocation_weight(
834
+ optimizer_name,
835
+ target=target,
836
+ feedback_cases=feedback_cases,
837
+ diagnoses=diagnoses,
838
+ search_paths=search_paths,
839
+ metric_names=metric_names,
840
+ )
841
+ scored.append(
842
+ (
843
+ weight,
844
+ -default_order.get(optimizer_name, len(DEFAULT_MULTI_INTERACTION_BACKENDS)),
845
+ optimizer_name,
846
+ reason,
847
+ backend_kwargs,
848
+ )
849
+ )
850
+
851
+ scored.sort(key=lambda item: (item[0], item[1], item[2]), reverse=True)
852
+ if max_backends is not None:
853
+ scored = scored[:max_backends]
854
+ return [
855
+ AgentMultiInteractionBackendPlan(
856
+ optimizer=optimizer_name,
857
+ rank=index,
858
+ weight=round(weight, 4),
859
+ reason=reason,
860
+ kwargs=backend_kwargs,
861
+ )
862
+ for index, (weight, _, optimizer_name, reason, backend_kwargs) in enumerate(
863
+ scored,
864
+ start=1,
865
+ )
866
+ ]
867
+
868
+
869
+ def _auto_seed_deployment_for_replay(
870
+ *,
871
+ target: OptimizationTarget,
872
+ deployment: Optional[DeploymentLike],
873
+ rollback_decision: Optional[AgentRollbackDecision],
874
+ live_evaluations: Optional[Sequence[Any]],
875
+ simulation_evaluator: Any,
876
+ metadata: Mapping[str, Any],
877
+ ) -> tuple[Optional[DeploymentLike], bool]:
878
+ if deployment is not None or rollback_decision is not None:
879
+ return deployment, False
880
+ if live_evaluations is not None:
881
+ return deployment, False
882
+ if getattr(simulation_evaluator, "evaluate_candidate", None) is None:
883
+ return deployment, False
884
+
885
+ seed = target.seed_candidate()
886
+ return (
887
+ export_agent_deployment(
888
+ seed,
889
+ framework="auto",
890
+ metadata={
891
+ **dict(metadata),
892
+ "auto_seed_deployment": True,
893
+ "auto_seed_deployment_source": "simulation_replay",
894
+ },
895
+ ),
896
+ True,
897
+ )
898
+
899
+
900
+ def _multi_interaction_allocation_metadata(
901
+ *,
902
+ target: OptimizationTarget,
903
+ plan: Sequence[AgentMultiInteractionBackendPlan],
904
+ feedback_cases: Sequence[AgentFeedbackCase],
905
+ diagnoses: Sequence[ComponentDiagnosis],
906
+ search_paths: Sequence[str],
907
+ ) -> dict[str, Any]:
908
+ metric_coverage = _diagnostic_metric_coverage(
909
+ diagnoses,
910
+ metric_names=_failed_feedback_metric_names(feedback_cases)
911
+ or _feedback_metric_names(feedback_cases),
912
+ )
913
+ active_paths = list(search_paths or target.search_space)
914
+ ledger: list[dict[str, Any]] = []
915
+ role_coverage: dict[str, int] = {}
916
+ archetype_coverage: dict[str, int] = {}
917
+
918
+ for allocation in plan:
919
+ profile = _multi_interaction_backend_profile(allocation.optimizer)
920
+ path_focus = _allocation_profile_path_focus(profile, active_paths)
921
+ role_path_focus = _allocation_role_path_focus(profile, path_focus)
922
+ diagnosis_focus = _allocation_diagnosis_focus(
923
+ profile=profile,
924
+ diagnoses=diagnoses,
925
+ active_paths=active_paths,
926
+ path_focus=path_focus,
927
+ )
928
+ for role in profile["roles"]:
929
+ role_coverage[role] = role_coverage.get(role, 0) + 1
930
+ for archetype in profile["role_archetypes"]:
931
+ archetype_coverage[archetype] = archetype_coverage.get(archetype, 0) + 1
932
+
933
+ focused_metrics = _diagnosis_focus_metric_coverage(diagnosis_focus)
934
+ ledger.append(
935
+ {
936
+ "optimizer": allocation.optimizer,
937
+ "rank": allocation.rank,
938
+ "weight": allocation.weight,
939
+ "reason": allocation.reason,
940
+ "allocation_kind": profile["allocation_kind"],
941
+ "roles": list(profile["roles"]),
942
+ "role_archetypes": list(profile["role_archetypes"]),
943
+ "path_focus": path_focus,
944
+ "role_path_focus": role_path_focus,
945
+ "diagnostic_components": _diagnosis_focus_values(
946
+ diagnosis_focus,
947
+ "component",
948
+ ),
949
+ "diagnostic_failure_modes": _diagnosis_focus_values(
950
+ diagnosis_focus,
951
+ "failure_mode",
952
+ ),
953
+ "diagnostic_metrics": focused_metrics or metric_coverage,
954
+ "diagnosis_focus": diagnosis_focus,
955
+ }
956
+ )
957
+
958
+ path_coverage = _ordered_patch_paths_for_keys(
959
+ _flatten_ledger_path_focus(ledger),
960
+ list(target.search_space),
961
+ )
962
+ return {
963
+ "allocation_algorithm": "deterministic_metric_diagnosis_society_agent_anchor_allocator",
964
+ "allocation_inspiration": (
965
+ "Human-team and society-role labels guide audit metadata only; "
966
+ "candidate acceptance remains metric-based."
967
+ ),
968
+ "deterministic_agent_anchor": any(
969
+ allocation.optimizer == "agent"
970
+ and "focused deterministic diagnosis search" in allocation.reason
971
+ for allocation in plan
972
+ ),
973
+ "society_allocation_ledger": ledger,
974
+ "allocation_role_coverage": dict(sorted(role_coverage.items())),
975
+ "allocation_archetype_coverage": dict(sorted(archetype_coverage.items())),
976
+ "allocation_metric_coverage": metric_coverage,
977
+ "allocation_search_path_coverage": path_coverage,
978
+ "allocation_diagnosis_coverage": _diagnosis_coverage_keys(diagnoses),
979
+ }
980
+
981
+
982
+ def _multi_interaction_backend_profile(optimizer_name: str) -> dict[str, Any]:
983
+ profile = MULTI_INTERACTION_BACKEND_PROFILES.get(optimizer_name)
984
+ if profile is not None:
985
+ return profile
986
+ return {
987
+ "allocation_kind": "custom_backend_search",
988
+ "roles": (optimizer_name,),
989
+ "role_archetypes": ("custom_optimizer",),
990
+ "path_prefixes": (),
991
+ "role_path_prefixes": {optimizer_name: ()},
992
+ }
993
+
994
+
995
+ def _allocation_profile_path_focus(
996
+ profile: Mapping[str, Any],
997
+ active_paths: Sequence[str],
998
+ ) -> list[str]:
999
+ path_focus = _path_prefix_focus(active_paths, profile.get("path_prefixes", ()))
1000
+ return path_focus or list(dict.fromkeys(active_paths))
1001
+
1002
+
1003
+ def _allocation_role_path_focus(
1004
+ profile: Mapping[str, Any],
1005
+ path_focus: Sequence[str],
1006
+ ) -> dict[str, list[str]]:
1007
+ role_path_prefixes = dict(profile.get("role_path_prefixes", {}))
1008
+ role_focus: dict[str, list[str]] = {}
1009
+ for role in profile.get("roles", ()):
1010
+ prefixes = role_path_prefixes.get(role, ())
1011
+ focused = _path_prefix_focus(path_focus, prefixes)
1012
+ role_focus[str(role)] = focused or list(path_focus)
1013
+ return role_focus
1014
+
1015
+
1016
+ def _allocation_diagnosis_focus(
1017
+ *,
1018
+ profile: Mapping[str, Any],
1019
+ diagnoses: Sequence[ComponentDiagnosis],
1020
+ active_paths: Sequence[str],
1021
+ path_focus: Sequence[str],
1022
+ ) -> list[dict[str, Any]]:
1023
+ path_focus_set = set(path_focus)
1024
+ profile_prefixes = tuple(str(prefix) for prefix in profile.get("path_prefixes", ()))
1025
+ rows: list[dict[str, Any]] = []
1026
+ for diagnosis in diagnoses:
1027
+ diagnosis_paths = _diagnosis_search_path_focus(diagnosis, active_paths)
1028
+ if (
1029
+ profile_prefixes
1030
+ and diagnosis_paths
1031
+ and path_focus_set
1032
+ and not path_focus_set.intersection(diagnosis_paths)
1033
+ ):
1034
+ continue
1035
+ metrics = _diagnosis_metric_names(diagnosis)
1036
+ row: dict[str, Any] = {
1037
+ "component": diagnosis.component,
1038
+ "failure_mode": diagnosis.failure_mode,
1039
+ "confidence": round(float(diagnosis.confidence), 4),
1040
+ }
1041
+ if metrics:
1042
+ row["metrics"] = metrics
1043
+ if diagnosis_paths:
1044
+ row["paths"] = diagnosis_paths
1045
+ if diagnosis.patch_strategy:
1046
+ row["patch_strategy"] = diagnosis.patch_strategy
1047
+ if diagnosis.evidence:
1048
+ row["evidence"] = diagnosis.evidence
1049
+ rows.append(row)
1050
+ return rows
1051
+
1052
+
1053
+ def _diagnosis_search_path_focus(
1054
+ diagnosis: ComponentDiagnosis,
1055
+ active_paths: Sequence[str],
1056
+ ) -> list[str]:
1057
+ prefixes = [str(path) for path in diagnosis.suggested_paths]
1058
+ prefixes.append(str(diagnosis.component))
1059
+ return _path_prefix_focus(active_paths, prefixes)
1060
+
1061
+
1062
+ def _path_prefix_focus(
1063
+ paths: Sequence[str],
1064
+ prefixes: Sequence[Any],
1065
+ ) -> list[str]:
1066
+ unique_paths = list(dict.fromkeys(str(path) for path in paths))
1067
+ unique_prefixes = [str(prefix) for prefix in prefixes if str(prefix)]
1068
+ if not unique_prefixes:
1069
+ return unique_paths
1070
+ return [
1071
+ path
1072
+ for path in unique_paths
1073
+ if any(
1074
+ path == prefix or path.startswith(f"{prefix}.")
1075
+ for prefix in unique_prefixes
1076
+ )
1077
+ ]
1078
+
1079
+
1080
+ def _diagnostic_metric_coverage(
1081
+ diagnoses: Sequence[ComponentDiagnosis],
1082
+ *,
1083
+ metric_names: Sequence[str],
1084
+ ) -> list[str]:
1085
+ metrics = {str(metric) for metric in metric_names}
1086
+ for diagnosis in diagnoses:
1087
+ metrics.update(_diagnosis_metric_names(diagnosis))
1088
+ return sorted(metrics)
1089
+
1090
+
1091
+ def _diagnosis_metric_names(diagnosis: ComponentDiagnosis) -> list[str]:
1092
+ metrics: set[str] = set()
1093
+ metadata = dict(diagnosis.metadata or {})
1094
+ for key in ("metric", "metric_name", "name"):
1095
+ value = metadata.get(key)
1096
+ if value:
1097
+ metrics.add(str(value))
1098
+ for key in ("metric_result", "finding"):
1099
+ value = metadata.get(key)
1100
+ if isinstance(value, Mapping):
1101
+ for nested_key in ("metric", "metric_name", "name"):
1102
+ nested_value = value.get(nested_key)
1103
+ if nested_value:
1104
+ metrics.add(str(nested_value))
1105
+ return sorted(metrics)
1106
+
1107
+
1108
+ def _diagnosis_focus_metric_coverage(
1109
+ diagnosis_focus: Sequence[Mapping[str, Any]],
1110
+ ) -> list[str]:
1111
+ metrics: set[str] = set()
1112
+ for row in diagnosis_focus:
1113
+ metrics.update(str(metric) for metric in row.get("metrics", ()))
1114
+ return sorted(metrics)
1115
+
1116
+
1117
+ def _diagnosis_focus_values(
1118
+ diagnosis_focus: Sequence[Mapping[str, Any]],
1119
+ key: str,
1120
+ ) -> list[str]:
1121
+ return sorted({str(row[key]) for row in diagnosis_focus if key in row})
1122
+
1123
+
1124
+ def _flatten_ledger_path_focus(ledger: Sequence[Mapping[str, Any]]) -> list[str]:
1125
+ paths: list[str] = []
1126
+ for entry in ledger:
1127
+ paths.extend(str(path) for path in entry.get("path_focus", ()))
1128
+ for role_paths in dict(entry.get("role_path_focus", {})).values():
1129
+ paths.extend(str(path) for path in role_paths)
1130
+ return list(dict.fromkeys(paths))
1131
+
1132
+
1133
+ def _diagnosis_coverage_keys(diagnoses: Sequence[ComponentDiagnosis]) -> list[str]:
1134
+ return sorted(
1135
+ {
1136
+ f"{diagnosis.component}:{diagnosis.failure_mode}"
1137
+ for diagnosis in diagnoses
1138
+ }
1139
+ )
1140
+
1141
+
1142
+ def _multi_interaction_backend_lineage(
1143
+ *,
1144
+ target: OptimizationTarget,
1145
+ plan: Sequence[AgentMultiInteractionBackendPlan],
1146
+ runs: Sequence[AgentMultiInteractionBackendRun],
1147
+ selected_run: AgentMultiInteractionBackendRun,
1148
+ ) -> list[AgentMultiInteractionBackendLineage]:
1149
+ plan_by_optimizer = {allocation.optimizer: allocation for allocation in plan}
1150
+ selected_patch_signature = _patch_signature(
1151
+ _backend_run_candidate_patch(selected_run, target)
1152
+ )
1153
+ rows: list[AgentMultiInteractionBackendLineage] = []
1154
+ for run in runs:
1155
+ allocation = plan_by_optimizer.get(run.optimizer)
1156
+ reoptimization_result = (
1157
+ run.result.reoptimization_result if run.result is not None else None
1158
+ )
1159
+ candidate = (
1160
+ getattr(reoptimization_result, "best_candidate", None)
1161
+ if reoptimization_result is not None
1162
+ else None
1163
+ )
1164
+ candidate_patch = _candidate_contribution_patch(candidate, target)
1165
+ rows.append(
1166
+ AgentMultiInteractionBackendLineage(
1167
+ optimizer=run.optimizer,
1168
+ rank=run.rank,
1169
+ allocation_weight=allocation.weight if allocation else 0.0,
1170
+ allocation_reason=allocation.reason if allocation else "",
1171
+ status=run.status,
1172
+ final_score=run.final_score,
1173
+ improved=run.improved,
1174
+ total_evaluations=run.total_evaluations,
1175
+ candidate_id=getattr(candidate, "id", None),
1176
+ parent_candidate_id=getattr(candidate, "parent_id", None),
1177
+ candidate_patch=candidate_patch,
1178
+ patch_paths=_ordered_patch_paths(target, candidate_patch),
1179
+ metadata={
1180
+ "backend_strategy": (
1181
+ reoptimization_result.metadata.get("strategy")
1182
+ if reoptimization_result is not None
1183
+ else None
1184
+ ),
1185
+ "backend_optimizer": (
1186
+ run.result.metadata.get("backend_optimizer")
1187
+ if run.result is not None
1188
+ else None
1189
+ ),
1190
+ },
1191
+ )
1192
+ )
1193
+
1194
+ completed_rows = [row for row in rows if row.status == "completed"]
1195
+ patch_backends: dict[str, list[str]] = {}
1196
+ patch_value_backends: dict[tuple[str, str], list[str]] = {}
1197
+ for row in completed_rows:
1198
+ patch_backends.setdefault(_patch_signature(row.candidate_patch), []).append(
1199
+ row.optimizer
1200
+ )
1201
+ for path, value in row.candidate_patch.items():
1202
+ patch_value_backends.setdefault(
1203
+ (path, _value_signature(value)),
1204
+ [],
1205
+ ).append(row.optimizer)
1206
+
1207
+ selected_patch = _backend_run_candidate_patch(selected_run, target)
1208
+ for row in rows:
1209
+ if row.status != "completed":
1210
+ row.selection_relation = "failed"
1211
+ continue
1212
+
1213
+ patch_signature = _patch_signature(row.candidate_patch)
1214
+ equivalent_backends = patch_backends.get(patch_signature, [])
1215
+ unique_patch: dict[str, Any] = {}
1216
+ shared_patch: dict[str, Any] = {}
1217
+ for path, value in row.candidate_patch.items():
1218
+ supporters = patch_value_backends.get((path, _value_signature(value)), [])
1219
+ if len(supporters) == 1:
1220
+ unique_patch[path] = value
1221
+ else:
1222
+ shared_patch[path] = value
1223
+
1224
+ row.equivalent_backends = list(equivalent_backends)
1225
+ row.equivalent_backend_count = len(equivalent_backends)
1226
+ row.unique_candidate_patch = unique_patch
1227
+ row.unique_patch_paths = _ordered_patch_paths(target, unique_patch)
1228
+ row.shared_candidate_patch = shared_patch
1229
+ row.shared_patch_paths = _ordered_patch_paths(target, shared_patch)
1230
+ if row.optimizer == selected_run.optimizer:
1231
+ row.selection_relation = "selected"
1232
+ elif patch_signature == selected_patch_signature:
1233
+ row.selection_relation = "consensus_peer"
1234
+ elif _patches_share_values(row.candidate_patch, selected_patch):
1235
+ row.selection_relation = "partial_support"
1236
+ else:
1237
+ row.selection_relation = "divergent"
1238
+
1239
+ return rows
1240
+
1241
+
1242
+ def _multi_interaction_ablation_report(
1243
+ *,
1244
+ lineage: Sequence[AgentMultiInteractionBackendLineage],
1245
+ selected_run: AgentMultiInteractionBackendRun,
1246
+ ) -> AgentMultiInteractionAblationReport:
1247
+ selected_lineage = next(
1248
+ (row for row in lineage if row.optimizer == selected_run.optimizer),
1249
+ None,
1250
+ )
1251
+ selected_patch = selected_lineage.candidate_patch if selected_lineage else {}
1252
+ selected_signature = _patch_signature(selected_patch)
1253
+ final_score = float(selected_run.final_score or 0.0)
1254
+ completed = [row for row in lineage if row.status == "completed"]
1255
+ peers = [row for row in completed if row.optimizer != selected_run.optimizer]
1256
+ best_without_selected = (
1257
+ max(peers, key=_lineage_selection_key) if peers else None
1258
+ )
1259
+ score_delta_without_selected: Optional[float] = None
1260
+ if best_without_selected and best_without_selected.final_score is not None:
1261
+ score_delta_without_selected = round(
1262
+ final_score - best_without_selected.final_score,
1263
+ 8,
1264
+ )
1265
+
1266
+ score_tolerance = 1e-9
1267
+ consensus_backends = [
1268
+ row.optimizer
1269
+ for row in completed
1270
+ if _patch_signature(row.candidate_patch) == selected_signature
1271
+ and row.final_score is not None
1272
+ and abs(row.final_score - final_score) <= score_tolerance
1273
+ ]
1274
+ peer_reproduced_selected = any(
1275
+ optimizer != selected_run.optimizer for optimizer in consensus_backends
1276
+ )
1277
+ selected_backend_required = not peer_reproduced_selected
1278
+
1279
+ selected_patch_support: dict[str, list[str]] = {}
1280
+ for path, value in selected_patch.items():
1281
+ selected_patch_support[path] = [
1282
+ row.optimizer
1283
+ for row in completed
1284
+ if path in row.candidate_patch
1285
+ and _value_signature(row.candidate_patch[path]) == _value_signature(value)
1286
+ ]
1287
+ shared_selected_patch_paths = [
1288
+ path for path, supporters in selected_patch_support.items() if len(supporters) > 1
1289
+ ]
1290
+ unique_selected_patch_paths = [
1291
+ path for path, supporters in selected_patch_support.items() if len(supporters) == 1
1292
+ ]
1293
+
1294
+ if best_without_selected is None:
1295
+ dependency = "single_backend_only"
1296
+ dependency_reason = "No other backend completed, so no leave-one-out comparison exists."
1297
+ elif peer_reproduced_selected:
1298
+ dependency = "backend_consensus"
1299
+ dependency_reason = (
1300
+ "At least one other backend reproduced the selected patch at the same score."
1301
+ )
1302
+ elif (
1303
+ best_without_selected.final_score is not None
1304
+ and abs(final_score - best_without_selected.final_score) <= score_tolerance
1305
+ ):
1306
+ dependency = "score_tie_different_patch"
1307
+ dependency_reason = (
1308
+ "Removing the selected backend preserves the score, but the best peer "
1309
+ "uses a different patch."
1310
+ )
1311
+ else:
1312
+ dependency = "selected_backend_dependent"
1313
+ dependency_reason = (
1314
+ "Removing the selected backend lowers the best observed portfolio score."
1315
+ )
1316
+
1317
+ return AgentMultiInteractionAblationReport(
1318
+ selected_optimizer=selected_run.optimizer,
1319
+ selected_candidate_id=selected_lineage.candidate_id if selected_lineage else None,
1320
+ selected_patch=selected_patch,
1321
+ selected_patch_paths=(
1322
+ list(selected_lineage.patch_paths) if selected_lineage else []
1323
+ ),
1324
+ final_score=final_score,
1325
+ best_without_selected_optimizer=(
1326
+ best_without_selected.optimizer if best_without_selected else None
1327
+ ),
1328
+ best_without_selected_score=(
1329
+ best_without_selected.final_score if best_without_selected else None
1330
+ ),
1331
+ score_delta_without_selected=score_delta_without_selected,
1332
+ selected_backend_required=selected_backend_required,
1333
+ dependency=dependency,
1334
+ dependency_reason=dependency_reason,
1335
+ consensus_backends=consensus_backends,
1336
+ consensus_backend_count=len(consensus_backends),
1337
+ shared_selected_patch_paths=_ordered_patch_paths_for_keys(
1338
+ shared_selected_patch_paths,
1339
+ list(selected_patch),
1340
+ ),
1341
+ unique_selected_patch_paths=_ordered_patch_paths_for_keys(
1342
+ unique_selected_patch_paths,
1343
+ list(selected_patch),
1344
+ ),
1345
+ selected_patch_support={
1346
+ path: selected_patch_support[path]
1347
+ for path in _ordered_patch_paths_for_keys(
1348
+ selected_patch_support,
1349
+ list(selected_patch),
1350
+ )
1351
+ },
1352
+ backend_scoreboard=[
1353
+ {
1354
+ "optimizer": row.optimizer,
1355
+ "rank": row.rank,
1356
+ "status": row.status,
1357
+ "final_score": row.final_score,
1358
+ "improved": row.improved,
1359
+ "candidate_id": row.candidate_id,
1360
+ "patch_paths": list(row.patch_paths),
1361
+ "selection_relation": row.selection_relation,
1362
+ }
1363
+ for row in sorted(completed, key=_lineage_selection_key, reverse=True)
1364
+ ],
1365
+ )
1366
+
1367
+
1368
+ def _backend_run_candidate_patch(
1369
+ run: AgentMultiInteractionBackendRun,
1370
+ target: OptimizationTarget,
1371
+ ) -> dict[str, Any]:
1372
+ if run.result is None:
1373
+ return {}
1374
+ return _candidate_contribution_patch(
1375
+ run.result.reoptimization_result.best_candidate,
1376
+ target,
1377
+ )
1378
+
1379
+
1380
+ def _candidate_contribution_patch(
1381
+ candidate: Any,
1382
+ target: OptimizationTarget,
1383
+ ) -> dict[str, Any]:
1384
+ if candidate is None:
1385
+ return {}
1386
+ base_candidate = target.seed_candidate()
1387
+ changed = {
1388
+ path: candidate.get_path(path)
1389
+ for path in target.search_space
1390
+ if candidate.get_path(path) != base_candidate.get_path(path)
1391
+ }
1392
+ if changed:
1393
+ return changed
1394
+ raw_patch = getattr(candidate, "patch", None)
1395
+ return dict(raw_patch or {})
1396
+
1397
+
1398
+ def _patch_signature(patch: Mapping[str, Any]) -> str:
1399
+ return json.dumps(dict(patch), sort_keys=True, default=str, separators=(",", ":"))
1400
+
1401
+
1402
+ def _value_signature(value: Any) -> str:
1403
+ return json.dumps(value, sort_keys=True, default=str, separators=(",", ":"))
1404
+
1405
+
1406
+ def _patches_share_values(
1407
+ first: Mapping[str, Any],
1408
+ second: Mapping[str, Any],
1409
+ ) -> bool:
1410
+ return any(
1411
+ path in second and _value_signature(value) == _value_signature(second[path])
1412
+ for path, value in first.items()
1413
+ )
1414
+
1415
+
1416
+ def _ordered_patch_paths(
1417
+ target: OptimizationTarget,
1418
+ patch: Mapping[str, Any],
1419
+ ) -> list[str]:
1420
+ ordered = [path for path in target.search_space if path in patch]
1421
+ ordered.extend(path for path in patch if path not in target.search_space)
1422
+ return ordered
1423
+
1424
+
1425
+ def _ordered_patch_paths_for_keys(
1426
+ paths: Iterable[str],
1427
+ order: Sequence[str],
1428
+ ) -> list[str]:
1429
+ path_set = set(paths)
1430
+ ordered = [path for path in order if path in path_set]
1431
+ ordered.extend(sorted(path for path in path_set if path not in order))
1432
+ return ordered
1433
+
1434
+
1435
+ def _lineage_selection_key(
1436
+ row: AgentMultiInteractionBackendLineage,
1437
+ ) -> tuple[float, int, int, int]:
1438
+ return (
1439
+ row.final_score if row.final_score is not None else float("-inf"),
1440
+ 1 if row.improved else 0,
1441
+ -row.rank,
1442
+ -row.total_evaluations,
1443
+ )
1444
+
1445
+
1446
+ def _dedupe_optimizer_pool(pool: Sequence[str]) -> list[str]:
1447
+ seen: set[str] = set()
1448
+ names: list[str] = []
1449
+ for item in pool:
1450
+ normalized = _normalize_optimizer_name(str(item))
1451
+ if normalized in seen:
1452
+ continue
1453
+ seen.add(normalized)
1454
+ names.append(normalized)
1455
+ return names
1456
+
1457
+
1458
+ def _backend_kwargs_for_multi_interaction(
1459
+ optimizer_name: str,
1460
+ *,
1461
+ target: OptimizationTarget,
1462
+ metric_names: Sequence[str],
1463
+ optimizer_kwargs: Mapping[str, Any],
1464
+ optimizer_kwargs_by_backend: Mapping[str, Mapping[str, Any]],
1465
+ ) -> dict[str, Any]:
1466
+ defaults: dict[str, Any] = {}
1467
+ target_score = float(optimizer_kwargs.get("target_score", 0.99))
1468
+ if optimizer_name == "agent":
1469
+ defaults.update({"max_candidates": 16})
1470
+ elif optimizer_name in {"council", "society"}:
1471
+ defaults.update(
1472
+ {
1473
+ "max_rounds": 2,
1474
+ "beam_width": 4,
1475
+ "max_proposals_per_round": 16,
1476
+ "target_score": target_score,
1477
+ }
1478
+ )
1479
+ elif optimizer_name == "social_memory":
1480
+ defaults.update(
1481
+ {
1482
+ "max_rounds": 2,
1483
+ "beam_width": 4,
1484
+ "max_proposals_per_round": 16,
1485
+ "target_score": target_score,
1486
+ }
1487
+ )
1488
+ elif optimizer_name == "curriculum":
1489
+ defaults.update({"max_candidates_per_stage": 8, "target_score": target_score})
1490
+ elif optimizer_name == "evolution":
1491
+ defaults.update(
1492
+ {
1493
+ "population_size": min(10, max(4, len(target.search_space) * 2)),
1494
+ "generations": 2,
1495
+ "elite_count": 2,
1496
+ "seed": 42,
1497
+ "target_score": target_score,
1498
+ }
1499
+ )
1500
+ elif optimizer_name == "tpe":
1501
+ defaults.update({"n_trials": 8, "seed": 42, "target_score": target_score})
1502
+ elif optimizer_name == "pareto":
1503
+ defaults.update({"n_trials": 8, "seed": 42, "target_score": target_score})
1504
+ if metric_names:
1505
+ defaults["objective_names"] = list(metric_names[:4])
1506
+ elif optimizer_name == "bandit":
1507
+ defaults.update(
1508
+ {
1509
+ "max_candidates": 8,
1510
+ "total_budget": 12,
1511
+ "selection": "best",
1512
+ "target_score": target_score,
1513
+ }
1514
+ )
1515
+ shared_keys = {"include_seed", "auto_diagnose", "diagnostic_score_threshold"}
1516
+ shared_kwargs = {
1517
+ key: value
1518
+ for key, value in dict(optimizer_kwargs).items()
1519
+ if key in shared_keys
1520
+ }
1521
+ if optimizer_name != "agent" and "target_score" in optimizer_kwargs:
1522
+ shared_kwargs["target_score"] = optimizer_kwargs["target_score"]
1523
+ combined = {
1524
+ **defaults,
1525
+ **shared_kwargs,
1526
+ **dict(optimizer_kwargs_by_backend.get(optimizer_name, {})),
1527
+ }
1528
+ return combined
1529
+
1530
+
1531
+ def _backend_allocation_weight(
1532
+ optimizer_name: str,
1533
+ *,
1534
+ target: OptimizationTarget,
1535
+ feedback_cases: Sequence[AgentFeedbackCase],
1536
+ diagnoses: Sequence[ComponentDiagnosis],
1537
+ search_paths: Sequence[str],
1538
+ metric_names: Sequence[str],
1539
+ ) -> tuple[float, str]:
1540
+ layers = set(target.layers)
1541
+ text = " ".join(
1542
+ [
1543
+ " ".join(layers),
1544
+ " ".join(search_paths),
1545
+ " ".join(metric_names),
1546
+ " ".join(diagnosis.component for diagnosis in diagnoses),
1547
+ " ".join(diagnosis.failure_mode for diagnosis in diagnoses),
1548
+ ]
1549
+ ).lower()
1550
+ path_count = len(search_paths) if search_paths else len(target.search_space)
1551
+ metric_count = len(metric_names)
1552
+ failed_count = sum(1 for case in feedback_cases if not case.passed)
1553
+ candidate_space_size = _target_search_space_cardinality(target)
1554
+ architecture_config_signal = _architecture_config_signal(text)
1555
+
1556
+ weights = {
1557
+ "agent": 0.25,
1558
+ "curriculum": 0.55,
1559
+ "council": 0.6,
1560
+ "society": 0.65,
1561
+ "social_memory": 0.55,
1562
+ "evolution": 0.5,
1563
+ "pareto": 0.45,
1564
+ "tpe": 0.4,
1565
+ "bandit": 0.4,
1566
+ }
1567
+ reasons: list[str] = []
1568
+ weight = weights.get(optimizer_name, 0.1)
1569
+ if failed_count:
1570
+ weight += 0.1
1571
+ reasons.append(f"{failed_count} failing feedback case(s)")
1572
+ if optimizer_name == "agent":
1573
+ if 0 < path_count <= 3:
1574
+ weight += 0.45
1575
+ reasons.append("focused deterministic diagnosis search")
1576
+ if target.search_space and candidate_space_size <= 32:
1577
+ weight += 0.25
1578
+ reasons.append(f"exact categorical search space size {candidate_space_size}")
1579
+ if architecture_config_signal:
1580
+ weight += 0.35
1581
+ reasons.append("architecture/config signal")
1582
+ if path_count > 1:
1583
+ if optimizer_name in {"council", "society", "evolution"}:
1584
+ weight += 0.3
1585
+ if optimizer_name in {"curriculum", "social_memory"}:
1586
+ weight += 0.15
1587
+ reasons.append(f"{path_count} diagnosed search paths")
1588
+ if metric_count > 1:
1589
+ if optimizer_name in {"curriculum", "pareto"}:
1590
+ weight += 0.35
1591
+ if optimizer_name in {"society", "council", "social_memory"}:
1592
+ weight += 0.15
1593
+ reasons.append(f"{metric_count} failed metrics")
1594
+ if any(token in text for token in ("multi_agent", "handoff", "coordination", "review")):
1595
+ if optimizer_name in {"society", "council"}:
1596
+ weight += 0.45
1597
+ if optimizer_name == "social_memory":
1598
+ weight += 0.15
1599
+ reasons.append("multi-agent coordination signal")
1600
+ if "memory" in text or "cross_trial" in text:
1601
+ if optimizer_name == "social_memory":
1602
+ weight += 0.45
1603
+ if optimizer_name in {"society", "council"}:
1604
+ weight += 0.1
1605
+ reasons.append("memory/history signal")
1606
+ if "policy" in text or "security" in text:
1607
+ if optimizer_name in {"agent", "evolution", "bandit"}:
1608
+ weight += 0.15
1609
+ reasons.append("policy/security signal")
1610
+ if len(target.search_space) >= 6:
1611
+ if optimizer_name in {"tpe", "evolution"}:
1612
+ weight += 0.3
1613
+ if optimizer_name == "bandit":
1614
+ weight += 0.15
1615
+ reasons.append("larger categorical search space")
1616
+ if len(feedback_cases) > 1:
1617
+ if optimizer_name in {"bandit", "social_memory", "curriculum"}:
1618
+ weight += 0.2
1619
+ reasons.append("multi-observation replay window")
1620
+ if not reasons:
1621
+ reasons.append("deterministic fallback allocation")
1622
+ return weight, "; ".join(dict.fromkeys(reasons))
1623
+
1624
+
1625
+ def _target_search_space_cardinality(target: OptimizationTarget) -> int:
1626
+ total = 1
1627
+ for values in target.search_space.values():
1628
+ if isinstance(values, (list, tuple, set)):
1629
+ total *= max(1, len(values))
1630
+ else:
1631
+ total *= 1
1632
+ if total > 1_000_000:
1633
+ return total
1634
+ return total
1635
+
1636
+
1637
+ def _architecture_config_signal(text: str) -> bool:
1638
+ return any(
1639
+ token in text
1640
+ for token in (
1641
+ "architecture",
1642
+ "config",
1643
+ "framework",
1644
+ "adapter",
1645
+ "trace",
1646
+ "event_stream",
1647
+ "streaming",
1648
+ "orchestration",
1649
+ "workflow",
1650
+ "runtime",
1651
+ "instrumentation",
1652
+ "otel",
1653
+ "langchain",
1654
+ "langgraph",
1655
+ "openai_agents",
1656
+ "pipecat",
1657
+ "livekit",
1658
+ )
1659
+ )
1660
+
1661
+
1662
+ def _feedback_metric_names(feedback_cases: Sequence[AgentFeedbackCase]) -> list[str]:
1663
+ names: set[str] = set()
1664
+ for case in feedback_cases:
1665
+ names.update(str(key) for key in case.metrics.keys())
1666
+ for failure in case.failures:
1667
+ names.update(_METRIC_NAME_RE.findall(failure))
1668
+ return sorted(names)
1669
+
1670
+
1671
+ def _failed_feedback_metric_names(feedback_cases: Sequence[AgentFeedbackCase]) -> list[str]:
1672
+ names: set[str] = set()
1673
+ for case in feedback_cases:
1674
+ for failure in case.failures:
1675
+ names.update(_METRIC_NAME_RE.findall(failure))
1676
+ return sorted(names)
1677
+
1678
+
1679
+ def _resolve_rollback_decision(
1680
+ *,
1681
+ rollback_decision: Optional[AgentRollbackDecision],
1682
+ deployment: Optional[DeploymentLike],
1683
+ live_evaluations: Optional[Sequence[Any]],
1684
+ simulation_evaluator: Any,
1685
+ rollback_kwargs: Mapping[str, Any],
1686
+ ) -> tuple[AgentRollbackDecision, str]:
1687
+ if rollback_decision is not None:
1688
+ return rollback_decision, "rollback_decision"
1689
+ if deployment is None:
1690
+ raise ValueError(
1691
+ "AgentFeedbackOptimizer requires deployment or rollback_decision."
1692
+ )
1693
+ decision = check_agent_deployment_rollback(
1694
+ deployment,
1695
+ live_evaluations=(
1696
+ list(live_evaluations) if live_evaluations is not None else None
1697
+ ),
1698
+ simulation_evaluator=simulation_evaluator,
1699
+ **dict(rollback_kwargs),
1700
+ )
1701
+ source = "live_evaluations" if live_evaluations is not None else "simulation_replay"
1702
+ return decision, source
1703
+
1704
+
1705
+ def _feedback_cases_from_rollback(
1706
+ decision: AgentRollbackDecision,
1707
+ ) -> list[AgentFeedbackCase]:
1708
+ return [
1709
+ AgentFeedbackCase(
1710
+ index=observation.index,
1711
+ candidate_id=observation.candidate_id,
1712
+ score=observation.score,
1713
+ passed=observation.passed,
1714
+ failures=list(observation.failures),
1715
+ metrics=dict(observation.metrics),
1716
+ metadata={
1717
+ **dict(observation.metadata),
1718
+ "rollback_required": decision.rollback_required,
1719
+ },
1720
+ )
1721
+ for observation in decision.observations
1722
+ ]
1723
+
1724
+
1725
+ def _diagnose_feedback_cases(
1726
+ feedback_cases: Sequence[AgentFeedbackCase],
1727
+ *,
1728
+ target: OptimizationTarget,
1729
+ failing_threshold: float,
1730
+ ) -> list[ComponentDiagnosis]:
1731
+ diagnostics: list[ComponentDiagnosis] = []
1732
+ failed_cases = [case for case in feedback_cases if not case.passed]
1733
+ for case in failed_cases:
1734
+ diagnostics.extend(
1735
+ diagnose_agent_report_evaluation(
1736
+ _agent_report_from_feedback_case(case),
1737
+ failing_threshold=failing_threshold,
1738
+ confidence=0.9,
1739
+ )
1740
+ )
1741
+ for failure in case.failures:
1742
+ diagnostics.extend(diagnose_text(failure, confidence=0.75))
1743
+
1744
+ if not diagnostics and failed_cases:
1745
+ diagnostics.append(
1746
+ ComponentDiagnosis(
1747
+ component="custom",
1748
+ failure_mode="unknown",
1749
+ confidence=0.5,
1750
+ evidence="Live feedback score regression without metric-specific diagnosis.",
1751
+ suggested_paths=list(target.search_space),
1752
+ metadata={"failed_feedback_cases": len(failed_cases)},
1753
+ )
1754
+ )
1755
+ return _dedupe_diagnoses(diagnostics)
1756
+
1757
+
1758
+ def _agent_report_from_feedback_case(case: AgentFeedbackCase) -> dict[str, Any]:
1759
+ metrics = dict(case.metrics)
1760
+ for failure in case.failures:
1761
+ for metric_name in _METRIC_NAME_RE.findall(failure):
1762
+ metrics.setdefault(metric_name, 0.0)
1763
+ return {
1764
+ "summary": {"metric_averages": metrics},
1765
+ "cases": [
1766
+ {
1767
+ "id": f"feedback-{case.index}",
1768
+ "metrics": [
1769
+ {
1770
+ "name": name,
1771
+ "score": score,
1772
+ "reason": "; ".join(case.failures),
1773
+ }
1774
+ for name, score in metrics.items()
1775
+ ],
1776
+ "findings": [
1777
+ {
1778
+ "metric": name,
1779
+ "score": score,
1780
+ "evidence": "; ".join(case.failures),
1781
+ }
1782
+ for name, score in metrics.items()
1783
+ ],
1784
+ }
1785
+ ],
1786
+ }
1787
+
1788
+
1789
+ _METRIC_NAME_RE = re.compile(r"metric '([^']+)'")
1790
+
1791
+
1792
+ def _search_paths_for_feedback(
1793
+ target: OptimizationTarget,
1794
+ diagnoses: Sequence[ComponentDiagnosis],
1795
+ ) -> list[str]:
1796
+ allowed_paths = relevant_search_paths(target.search_space, diagnoses)
1797
+ return [path for path in target.search_space if path in allowed_paths]
1798
+
1799
+
1800
+ def _resolve_feedback_optimizer(name: str) -> type:
1801
+ normalized = _normalize_optimizer_name(name)
1802
+ optimizers = {
1803
+ "agent": AgentOptimizer,
1804
+ "deterministic": AgentOptimizer,
1805
+ "council": CouncilAgentOptimizer,
1806
+ "society": SocietyAgentOptimizer,
1807
+ "social_memory": AgentSocialMemoryOptimizer,
1808
+ "curriculum": AgentCurriculumOptimizer,
1809
+ "evolution": AgentEvolutionOptimizer,
1810
+ "tpe": AgentTPEOptimizer,
1811
+ "pareto": AgentParetoOptimizer,
1812
+ "bandit": AgentBanditOptimizer,
1813
+ }
1814
+ if normalized not in optimizers:
1815
+ raise ValueError(
1816
+ "optimizer must be one of: agent, deterministic, council, society, "
1817
+ "social_memory, curriculum, evolution, tpe, pareto, or bandit."
1818
+ )
1819
+ return optimizers[normalized]
1820
+
1821
+
1822
+ def _normalize_optimizer_name(name: str) -> str:
1823
+ normalized = name.strip().lower().replace("-", "_")
1824
+ aliases = {
1825
+ "agentoptimizer": "agent",
1826
+ "agent_optimizer": "agent",
1827
+ "deterministic_agent_optimizer": "deterministic",
1828
+ "councilagentoptimizer": "council",
1829
+ "council_agent_optimizer": "council",
1830
+ "societyagentoptimizer": "society",
1831
+ "society_agent_optimizer": "society",
1832
+ "agentsocialmemoryoptimizer": "social_memory",
1833
+ "agent_social_memory_optimizer": "social_memory",
1834
+ "socialmemory": "social_memory",
1835
+ "social_memory_optimizer": "social_memory",
1836
+ "futureagi_social_memory": "social_memory",
1837
+ "futureagi_social_memory_optimizer": "social_memory",
1838
+ "agentcurriculumoptimizer": "curriculum",
1839
+ "agent_curriculum_optimizer": "curriculum",
1840
+ "curriculum_optimizer": "curriculum",
1841
+ "deliberate_practice": "curriculum",
1842
+ "deliberate_practice_curriculum": "curriculum",
1843
+ "agentevolutionoptimizer": "evolution",
1844
+ "agent_evolution_optimizer": "evolution",
1845
+ "agenttpeoptimizer": "tpe",
1846
+ "agent_tpe_optimizer": "tpe",
1847
+ "agentparetooptimizer": "pareto",
1848
+ "agent_pareto_optimizer": "pareto",
1849
+ "agentbanditoptimizer": "bandit",
1850
+ "agent_bandit_optimizer": "bandit",
1851
+ "agentmultiinteractionoptimizer": "multi_interaction",
1852
+ "agent_multi_interaction_optimizer": "multi_interaction",
1853
+ "multiinteraction": "multi_interaction",
1854
+ "multi_interaction_optimizer": "multi_interaction",
1855
+ "portfolio": "multi_interaction",
1856
+ "portfolio_optimizer": "multi_interaction",
1857
+ "auto": "multi_interaction",
1858
+ "auto_backend": "multi_interaction",
1859
+ }
1860
+ return aliases.get(
1861
+ normalized,
1862
+ normalized.replace("agent_", "").replace("_optimizer", "").replace("optimizer", ""),
1863
+ )