agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,1113 @@
1
+ from __future__ import annotations
2
+
3
+ import copy
4
+ import json
5
+ import logging
6
+ from dataclasses import dataclass, field
7
+ from typing import Any, Callable, Iterable, List, Mapping, Optional, Sequence
8
+
9
+ from ..base.base_optimizer import BaseOptimizer
10
+ from ..components import ComponentDiagnosis, relevant_search_paths
11
+ from ..observability import AgentObservabilityRecord, AgentObservabilityWindow
12
+ from ..targets import AgentCandidate, CandidateEvaluation, OptimizationTarget
13
+ from ..types import EvaluationResult, IterationHistory, OptimizationResult
14
+ from .agent import (
15
+ _dedupe_diagnoses,
16
+ _diagnose_candidate_evaluation,
17
+ _dump_model,
18
+ _history_from_candidate,
19
+ _normalize_candidate_evaluation,
20
+ _normalize_diagnoses,
21
+ )
22
+
23
+ logger = logging.getLogger(__name__)
24
+
25
+
26
+ CandidateScorer = Callable[
27
+ [AgentCandidate],
28
+ CandidateEvaluation | EvaluationResult | float,
29
+ ]
30
+
31
+
32
+ @dataclass
33
+ class _PatchCredit:
34
+ path: str
35
+ value: Any
36
+ observations: int = 0
37
+ total_delta: float = 0.0
38
+ total_score: float = 0.0
39
+ best_score: float = float("-inf")
40
+ passed: int = 0
41
+ failed: int = 0
42
+ sources: set[str] = field(default_factory=set)
43
+
44
+ @property
45
+ def mean_delta(self) -> float:
46
+ if self.observations == 0:
47
+ return 0.0
48
+ return self.total_delta / self.observations
49
+
50
+ @property
51
+ def mean_score(self) -> float:
52
+ if self.observations == 0:
53
+ return 0.0
54
+ return self.total_score / self.observations
55
+
56
+
57
+ @dataclass(frozen=True)
58
+ class _MemoryProposal:
59
+ patch: dict[str, Any]
60
+ role: str
61
+ reason: str
62
+ parent_ids: tuple[str, ...] = ()
63
+ metadata: Mapping[str, Any] = field(default_factory=dict)
64
+
65
+
66
+ class AgentSocialMemoryOptimizer(BaseOptimizer):
67
+ """
68
+ Multi-round agent optimizer with metric-bound social memory.
69
+
70
+ Each evaluated patch updates a deterministic credit ledger. Later rounds
71
+ combine high-credit choices, critique promising candidates with one more
72
+ change, and remove weak changes. Role and archetype labels are metadata
73
+ only; the evaluator's numeric scores decide every candidate.
74
+ """
75
+
76
+ def __init__(
77
+ self,
78
+ target: Optional[OptimizationTarget] = None,
79
+ *,
80
+ evaluate_candidate: Optional[CandidateScorer] = None,
81
+ simulation_evaluator: Any = None,
82
+ experiment_history: Optional[AgentObservabilityWindow] = None,
83
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
84
+ max_rounds: int = 4,
85
+ beam_width: int = 4,
86
+ max_proposals_per_round: int = 16,
87
+ target_score: float = 1.0,
88
+ include_seed: bool = True,
89
+ auto_diagnose: bool = True,
90
+ diagnostic_score_threshold: float = 0.85,
91
+ ) -> None:
92
+ if max_rounds < 1:
93
+ raise ValueError("max_rounds must be at least 1.")
94
+ if beam_width < 1:
95
+ raise ValueError("beam_width must be at least 1.")
96
+ if max_proposals_per_round < 1:
97
+ raise ValueError("max_proposals_per_round must be at least 1.")
98
+
99
+ self.target = target
100
+ self.evaluate_candidate = evaluate_candidate
101
+ self.simulation_evaluator = simulation_evaluator
102
+ self.experiment_history = experiment_history
103
+ self.diagnoses = _normalize_diagnoses(diagnoses)
104
+ self.max_rounds = max_rounds
105
+ self.beam_width = beam_width
106
+ self.max_proposals_per_round = max_proposals_per_round
107
+ self.target_score = target_score
108
+ self.include_seed = include_seed
109
+ self.auto_diagnose = auto_diagnose
110
+ self.diagnostic_score_threshold = diagnostic_score_threshold
111
+ super().__init__()
112
+
113
+ def optimize(
114
+ self,
115
+ evaluator: Any = None,
116
+ data_mapper: Any = None,
117
+ dataset: Optional[List[dict[str, Any]]] = None,
118
+ metric: Optional[Callable] = None,
119
+ *,
120
+ target: Optional[OptimizationTarget] = None,
121
+ evaluate_candidate: Optional[CandidateScorer] = None,
122
+ simulation_evaluator: Any = None,
123
+ experiment_history: Optional[AgentObservabilityWindow] = None,
124
+ diagnoses: Optional[Iterable[ComponentDiagnosis | dict[str, Any]]] = None,
125
+ max_rounds: Optional[int] = None,
126
+ beam_width: Optional[int] = None,
127
+ max_proposals_per_round: Optional[int] = None,
128
+ target_score: Optional[float] = None,
129
+ include_seed: Optional[bool] = None,
130
+ auto_diagnose: Optional[bool] = None,
131
+ diagnostic_score_threshold: Optional[float] = None,
132
+ **kwargs: Any,
133
+ ) -> OptimizationResult:
134
+ active_target = target or self.target
135
+ if active_target is None:
136
+ raise ValueError("AgentSocialMemoryOptimizer requires a target.")
137
+
138
+ active_evaluator = (
139
+ evaluate_candidate
140
+ or self.evaluate_candidate
141
+ or getattr(simulation_evaluator, "evaluate_candidate", None)
142
+ or getattr(self.simulation_evaluator, "evaluate_candidate", None)
143
+ or evaluator
144
+ )
145
+ if active_evaluator is None:
146
+ raise ValueError(
147
+ "AgentSocialMemoryOptimizer requires evaluate_candidate or simulation_evaluator."
148
+ )
149
+
150
+ active_max_rounds = self.max_rounds if max_rounds is None else max_rounds
151
+ active_beam_width = self.beam_width if beam_width is None else beam_width
152
+ active_max_proposals = (
153
+ self.max_proposals_per_round
154
+ if max_proposals_per_round is None
155
+ else max_proposals_per_round
156
+ )
157
+ if active_max_rounds < 1:
158
+ raise ValueError("max_rounds must be at least 1.")
159
+ if active_beam_width < 1:
160
+ raise ValueError("beam_width must be at least 1.")
161
+ if active_max_proposals < 1:
162
+ raise ValueError("max_proposals_per_round must be at least 1.")
163
+
164
+ active_target_score = (
165
+ self.target_score if target_score is None else target_score
166
+ )
167
+ use_include_seed = self.include_seed if include_seed is None else include_seed
168
+ use_auto_diagnose = self.auto_diagnose if auto_diagnose is None else auto_diagnose
169
+ active_diagnostic_threshold = (
170
+ self.diagnostic_score_threshold
171
+ if diagnostic_score_threshold is None
172
+ else diagnostic_score_threshold
173
+ )
174
+ active_history = experiment_history or self.experiment_history
175
+ active_diagnoses = _normalize_diagnoses(diagnoses)
176
+ if diagnoses is None:
177
+ active_diagnoses = list(self.diagnoses)
178
+
179
+ seed_candidate = active_target.seed_candidate()
180
+ evaluated: dict[str, CandidateEvaluation] = {}
181
+ history: List[IterationHistory] = []
182
+ role_counts: dict[str, int] = {}
183
+ round_summaries: list[dict[str, Any]] = []
184
+ proposal_audit: list[dict[str, Any]] = []
185
+ ledger: dict[str, _PatchCredit] = {}
186
+ best: CandidateEvaluation | None = None
187
+
188
+ if use_include_seed:
189
+ seed_evaluation = self._evaluate(
190
+ seed_candidate,
191
+ active_evaluator,
192
+ evaluated,
193
+ history,
194
+ role_counts,
195
+ role="seed",
196
+ round_number=0,
197
+ reason="evaluate_deployed_seed",
198
+ metadata={},
199
+ )
200
+ best = seed_evaluation
201
+ if use_auto_diagnose and not active_diagnoses:
202
+ active_diagnoses = _diagnose_candidate_evaluation(
203
+ seed_evaluation,
204
+ failing_threshold=active_diagnostic_threshold,
205
+ )
206
+
207
+ search_paths = _ordered_search_paths(active_target, active_diagnoses)
208
+ search_paths = _merge_history_search_paths(
209
+ search_paths,
210
+ active_history,
211
+ target=active_target,
212
+ seed_candidate=seed_candidate,
213
+ )
214
+ if not search_paths:
215
+ raise ValueError(
216
+ "AgentSocialMemoryOptimizer target search space cannot be empty."
217
+ )
218
+
219
+ baseline_score = (
220
+ best.score
221
+ if best is not None
222
+ else _history_baseline_score(active_history)
223
+ )
224
+ historical_prior_count = _seed_credit_from_history(
225
+ active_history,
226
+ target=active_target,
227
+ seed_candidate=seed_candidate,
228
+ search_paths=search_paths,
229
+ baseline_score=baseline_score,
230
+ ledger=ledger,
231
+ )
232
+ prior_proposals = _prior_proposals_from_history(
233
+ active_history,
234
+ target=active_target,
235
+ seed_candidate=seed_candidate,
236
+ search_paths=search_paths,
237
+ )
238
+
239
+ for round_number in range(1, active_max_rounds + 1):
240
+ proposals = _build_social_memory_proposals(
241
+ seed_candidate=seed_candidate,
242
+ evaluations=list(evaluated.values()),
243
+ search_space=active_target.search_space,
244
+ search_paths=search_paths,
245
+ diagnoses=active_diagnoses,
246
+ ledger=ledger,
247
+ prior_proposals=prior_proposals if round_number == 1 else (),
248
+ beam_width=active_beam_width,
249
+ max_proposals=active_max_proposals,
250
+ round_number=round_number,
251
+ )
252
+ proposals = [
253
+ proposal
254
+ for proposal in proposals
255
+ if _candidate_id_for_patch(seed_candidate, proposal.patch)
256
+ not in evaluated
257
+ ]
258
+ logger.info(
259
+ "Social memory round %s evaluating %s proposal(s)",
260
+ round_number,
261
+ len(proposals),
262
+ )
263
+
264
+ round_best = best
265
+ round_evaluated = 0
266
+ for proposal in proposals:
267
+ candidate = seed_candidate.with_patch(
268
+ proposal.patch,
269
+ metadata={
270
+ "kind": "social_memory_proposal",
271
+ "optimizer": "AgentSocialMemoryOptimizer",
272
+ "proposal_role": proposal.role,
273
+ "proposal_reason": proposal.reason,
274
+ "proposal_round": round_number,
275
+ "proposal_parent_ids": list(proposal.parent_ids),
276
+ "proposal_metadata": dict(proposal.metadata),
277
+ },
278
+ )
279
+ evaluation = self._evaluate(
280
+ candidate,
281
+ active_evaluator,
282
+ evaluated,
283
+ history,
284
+ role_counts,
285
+ role=proposal.role,
286
+ round_number=round_number,
287
+ reason=proposal.reason,
288
+ metadata=dict(proposal.metadata),
289
+ )
290
+ _record_credit_from_evaluation(
291
+ evaluation,
292
+ baseline_score=baseline_score,
293
+ ledger=ledger,
294
+ source=proposal.role,
295
+ )
296
+ proposal_audit.append(
297
+ {
298
+ "round": round_number,
299
+ "role": proposal.role,
300
+ "candidate_id": candidate.id,
301
+ "patch": copy.deepcopy(candidate.patch),
302
+ "score": evaluation.score,
303
+ "reason": proposal.reason,
304
+ }
305
+ )
306
+ round_evaluated += 1
307
+ if round_best is None or evaluation.score > round_best.score:
308
+ round_best = evaluation
309
+ if best is None or evaluation.score > best.score:
310
+ best = evaluation
311
+ logger.info(
312
+ "New best social-memory candidate %s score=%.4f",
313
+ candidate.id,
314
+ evaluation.score,
315
+ )
316
+ if best.score >= active_target_score:
317
+ break
318
+
319
+ if round_best is not None and use_auto_diagnose:
320
+ round_diagnoses = _diagnose_candidate_evaluation(
321
+ round_best,
322
+ failing_threshold=active_diagnostic_threshold,
323
+ )
324
+ if round_diagnoses:
325
+ active_diagnoses = _dedupe_diagnoses(
326
+ [*active_diagnoses, *round_diagnoses]
327
+ )
328
+ search_paths = _ordered_search_paths(active_target, active_diagnoses)
329
+ search_paths = _merge_history_search_paths(
330
+ search_paths,
331
+ active_history,
332
+ target=active_target,
333
+ seed_candidate=seed_candidate,
334
+ )
335
+
336
+ round_summaries.append(
337
+ {
338
+ "round": round_number,
339
+ "proposals": len(proposals),
340
+ "evaluated": round_evaluated,
341
+ "best_score": best.score if best is not None else None,
342
+ "search_paths": list(search_paths),
343
+ "ledger_size": len(ledger),
344
+ }
345
+ )
346
+ if best is not None and best.score >= active_target_score:
347
+ break
348
+
349
+ if best is None:
350
+ raise ValueError("AgentSocialMemoryOptimizer did not evaluate any candidates.")
351
+
352
+ metadata = {
353
+ "optimizer": "AgentSocialMemoryOptimizer",
354
+ "strategy": "futureagi_social_memory",
355
+ "strategy_inspiration": (
356
+ "social credit assignment, working memory, critique, synthesis, "
357
+ "and stewardship; names are metadata only"
358
+ ),
359
+ "roles": ["smriti", "arjuna", "vidura", "sangha", "dharma_steward"],
360
+ "target_name": best.candidate.target_name,
361
+ "best_candidate_id": best.candidate.id,
362
+ "search_paths": list(search_paths),
363
+ "rounds": round_summaries,
364
+ "beam_width": active_beam_width,
365
+ "max_proposals_per_round": active_max_proposals,
366
+ "role_evaluations": role_counts,
367
+ "historical_prior_count": historical_prior_count,
368
+ "history_source": active_history.source if active_history else None,
369
+ "history_record_count": len(active_history.records) if active_history else 0,
370
+ "credit_ledger": _ledger_summary(ledger),
371
+ "proposal_audit": proposal_audit,
372
+ }
373
+ if active_diagnoses:
374
+ metadata["diagnostics"] = [_dump_model(item) for item in active_diagnoses]
375
+ metadata["auto_diagnosed"] = use_auto_diagnose
376
+
377
+ return OptimizationResult(
378
+ best_generator=best.candidate,
379
+ best_candidate=best.candidate,
380
+ history=history,
381
+ final_score=best.score,
382
+ total_iterations=len(history),
383
+ total_evaluations=len(history),
384
+ metadata=metadata,
385
+ )
386
+
387
+ def _evaluate(
388
+ self,
389
+ candidate: AgentCandidate,
390
+ evaluator: CandidateScorer,
391
+ evaluated: dict[str, CandidateEvaluation],
392
+ history: List[IterationHistory],
393
+ role_counts: dict[str, int],
394
+ *,
395
+ role: str,
396
+ round_number: int,
397
+ reason: str,
398
+ metadata: Mapping[str, Any],
399
+ ) -> CandidateEvaluation:
400
+ if candidate.id in evaluated:
401
+ return evaluated[candidate.id]
402
+
403
+ value = evaluator(candidate)
404
+ evaluation = _normalize_candidate_evaluation(value, candidate)
405
+ evaluation.metadata = {
406
+ **candidate.metadata,
407
+ **evaluation.metadata,
408
+ "optimizer": "AgentSocialMemoryOptimizer",
409
+ "proposal_role": role,
410
+ "proposal_round": round_number,
411
+ "proposal_reason": reason,
412
+ "proposal_metadata": dict(metadata),
413
+ }
414
+ evaluated[candidate.id] = evaluation
415
+ history.append(_history_from_candidate(evaluation))
416
+ role_counts[role] = role_counts.get(role, 0) + 1
417
+ return evaluation
418
+
419
+
420
+ def _ordered_search_paths(
421
+ target: OptimizationTarget,
422
+ diagnoses: Sequence[ComponentDiagnosis],
423
+ ) -> List[str]:
424
+ allowed_paths = relevant_search_paths(target.search_space, diagnoses)
425
+ return [path for path in target.search_space if path in allowed_paths]
426
+
427
+
428
+ def _merge_history_search_paths(
429
+ search_paths: Sequence[str],
430
+ history: Optional[AgentObservabilityWindow],
431
+ *,
432
+ target: OptimizationTarget,
433
+ seed_candidate: AgentCandidate,
434
+ ) -> list[str]:
435
+ merged = list(search_paths)
436
+ if history is None:
437
+ return merged
438
+ all_paths = list(target.search_space)
439
+ for record in history.records:
440
+ patch = _patch_from_record(
441
+ record,
442
+ target=target,
443
+ seed_candidate=seed_candidate,
444
+ search_paths=all_paths,
445
+ )
446
+ for path in all_paths:
447
+ if path in patch and path not in merged:
448
+ merged.append(path)
449
+ return merged
450
+
451
+
452
+ def _build_social_memory_proposals(
453
+ *,
454
+ seed_candidate: AgentCandidate,
455
+ evaluations: Sequence[CandidateEvaluation],
456
+ search_space: Mapping[str, List[Any]],
457
+ search_paths: Sequence[str],
458
+ diagnoses: Sequence[ComponentDiagnosis],
459
+ ledger: Mapping[str, _PatchCredit],
460
+ prior_proposals: Sequence[_MemoryProposal],
461
+ beam_width: int,
462
+ max_proposals: int,
463
+ round_number: int,
464
+ ) -> list[_MemoryProposal]:
465
+ proposals: list[_MemoryProposal] = []
466
+ seen: set[str] = set()
467
+ ranked = sorted(
468
+ evaluations,
469
+ key=lambda item: (item.score, -len(item.candidate.patch), item.candidate.id),
470
+ reverse=True,
471
+ )
472
+ changed_ranked = [item for item in ranked if item.candidate.patch]
473
+
474
+ for proposal in prior_proposals:
475
+ _append_proposal(proposals, seen, proposal, max_proposals)
476
+
477
+ if round_number > 1:
478
+ for proposal in _ledger_synthesis_proposals(
479
+ ledger,
480
+ search_paths=search_paths,
481
+ evaluations=changed_ranked,
482
+ ):
483
+ _append_proposal(proposals, seen, proposal, max_proposals)
484
+
485
+ for evaluation in changed_ranked[:beam_width]:
486
+ for proposal in _critic_proposals(
487
+ evaluation,
488
+ search_space=search_space,
489
+ search_paths=search_paths,
490
+ ledger=ledger,
491
+ ):
492
+ _append_proposal(proposals, seen, proposal, max_proposals)
493
+
494
+ for evaluation in changed_ranked[:beam_width]:
495
+ for proposal in _steward_proposals(evaluation, ledger=ledger):
496
+ _append_proposal(proposals, seen, proposal, max_proposals)
497
+
498
+ for proposal in _specialist_proposals(
499
+ seed_candidate,
500
+ search_space=search_space,
501
+ search_paths=search_paths,
502
+ diagnoses=diagnoses,
503
+ ):
504
+ _append_proposal(proposals, seen, proposal, max_proposals)
505
+
506
+ for proposal in _explorer_proposals(
507
+ seed_candidate,
508
+ search_space=search_space,
509
+ search_paths=search_paths,
510
+ ledger=ledger,
511
+ ):
512
+ _append_proposal(proposals, seen, proposal, max_proposals)
513
+
514
+ if round_number > 1:
515
+ for proposal in _adversary_proposals(
516
+ seed_candidate,
517
+ search_space=search_space,
518
+ search_paths=search_paths,
519
+ ):
520
+ _append_proposal(proposals, seen, proposal, max_proposals)
521
+
522
+ return proposals
523
+
524
+
525
+ def _specialist_proposals(
526
+ seed_candidate: AgentCandidate,
527
+ *,
528
+ search_space: Mapping[str, List[Any]],
529
+ search_paths: Sequence[str],
530
+ diagnoses: Sequence[ComponentDiagnosis],
531
+ ) -> Iterable[_MemoryProposal]:
532
+ for group_key, paths in _path_groups(search_paths, diagnoses).items():
533
+ patch: dict[str, Any] = {}
534
+ for path in paths:
535
+ value = _first_non_seed_value(seed_candidate, search_space, path)
536
+ if value is not _NO_VALUE:
537
+ patch[path] = value
538
+ if patch:
539
+ yield _MemoryProposal(
540
+ patch=patch,
541
+ role="smriti",
542
+ parent_ids=(seed_candidate.id,),
543
+ reason=f"apply_diagnosed_memory_bundle:{group_key}",
544
+ metadata={
545
+ "role_archetype": "working_memory",
546
+ "role_kind": "specialist",
547
+ },
548
+ )
549
+
550
+
551
+ def _explorer_proposals(
552
+ seed_candidate: AgentCandidate,
553
+ *,
554
+ search_space: Mapping[str, List[Any]],
555
+ search_paths: Sequence[str],
556
+ ledger: Mapping[str, _PatchCredit],
557
+ ) -> Iterable[_MemoryProposal]:
558
+ tested = {_credit_key(credit.path, credit.value) for credit in ledger.values()}
559
+ for path in search_paths:
560
+ for value in search_space.get(path, []):
561
+ if seed_candidate.get_path(path) == value:
562
+ continue
563
+ yield _MemoryProposal(
564
+ patch={path: value},
565
+ role="arjuna",
566
+ parent_ids=(seed_candidate.id,),
567
+ reason="isolate_single_path_effect",
568
+ metadata={
569
+ "role_archetype": "focused_action",
570
+ "role_kind": "explorer",
571
+ "previously_tested": _credit_key(path, value) in tested,
572
+ },
573
+ )
574
+
575
+
576
+ def _critic_proposals(
577
+ evaluation: CandidateEvaluation,
578
+ *,
579
+ search_space: Mapping[str, List[Any]],
580
+ search_paths: Sequence[str],
581
+ ledger: Mapping[str, _PatchCredit],
582
+ ) -> Iterable[_MemoryProposal]:
583
+ source_patch = dict(evaluation.candidate.patch)
584
+ for path in search_paths:
585
+ if path in source_patch:
586
+ continue
587
+ value = _best_credit_value(path, ledger)
588
+ if value is _NO_VALUE:
589
+ value = _first_non_candidate_value(evaluation.candidate, search_space, path)
590
+ if value is _NO_VALUE:
591
+ continue
592
+ yield _MemoryProposal(
593
+ patch={**source_patch, path: value},
594
+ role="vidura",
595
+ parent_ids=(evaluation.candidate.id,),
596
+ reason="critique_best_candidate_with_next_memory",
597
+ metadata={
598
+ "role_archetype": "prudent_critic",
599
+ "role_kind": "critic",
600
+ },
601
+ )
602
+
603
+
604
+ def _ledger_synthesis_proposals(
605
+ ledger: Mapping[str, _PatchCredit],
606
+ *,
607
+ search_paths: Sequence[str],
608
+ evaluations: Sequence[CandidateEvaluation],
609
+ ) -> Iterable[_MemoryProposal]:
610
+ patch = _top_credit_patch(ledger, search_paths=search_paths)
611
+ parent_ids: list[str] = []
612
+ for evaluation in evaluations:
613
+ if set(evaluation.candidate.patch) & set(patch):
614
+ parent_ids.append(evaluation.candidate.id)
615
+ if patch:
616
+ yield _MemoryProposal(
617
+ patch=patch,
618
+ role="sangha",
619
+ parent_ids=tuple(dict.fromkeys(parent_ids)),
620
+ reason="combine_high_credit_path_memories",
621
+ metadata={
622
+ "role_archetype": "collective_synthesis",
623
+ "role_kind": "synthesizer",
624
+ },
625
+ )
626
+
627
+
628
+ def _steward_proposals(
629
+ evaluation: CandidateEvaluation,
630
+ *,
631
+ ledger: Mapping[str, _PatchCredit],
632
+ ) -> Iterable[_MemoryProposal]:
633
+ source_patch = dict(evaluation.candidate.patch)
634
+ if len(source_patch) < 2:
635
+ return
636
+ ranked_paths = sorted(
637
+ source_patch,
638
+ key=lambda path: (
639
+ _credit_for_value(path, source_patch[path], ledger).mean_delta
640
+ if _credit_for_value(path, source_patch[path], ledger)
641
+ else 0.0,
642
+ path,
643
+ ),
644
+ )
645
+ for path in ranked_paths:
646
+ patch = {
647
+ key: value
648
+ for key, value in source_patch.items()
649
+ if key != path
650
+ }
651
+ yield _MemoryProposal(
652
+ patch=patch,
653
+ role="dharma_steward",
654
+ parent_ids=(evaluation.candidate.id,),
655
+ reason="remove_low_credit_change_to_check_minimality",
656
+ metadata={
657
+ "role_archetype": "minimal_process_guardian",
658
+ "role_kind": "steward",
659
+ "removed_path": path,
660
+ },
661
+ )
662
+
663
+
664
+ def _adversary_proposals(
665
+ seed_candidate: AgentCandidate,
666
+ *,
667
+ search_space: Mapping[str, List[Any]],
668
+ search_paths: Sequence[str],
669
+ ) -> Iterable[_MemoryProposal]:
670
+ patch: dict[str, Any] = {}
671
+ for path in search_paths:
672
+ value = _last_non_seed_value(seed_candidate, search_space, path)
673
+ if value is not _NO_VALUE:
674
+ patch[path] = value
675
+ if len(patch) >= 3:
676
+ break
677
+ if patch:
678
+ yield _MemoryProposal(
679
+ patch=patch,
680
+ role="vidura",
681
+ parent_ids=(seed_candidate.id,),
682
+ reason="stress_boundary_combination",
683
+ metadata={
684
+ "role_archetype": "prudent_critic",
685
+ "role_kind": "adversary",
686
+ },
687
+ )
688
+
689
+
690
+ def _seed_credit_from_history(
691
+ history: Optional[AgentObservabilityWindow],
692
+ *,
693
+ target: OptimizationTarget,
694
+ seed_candidate: AgentCandidate,
695
+ search_paths: Sequence[str],
696
+ baseline_score: float,
697
+ ledger: dict[str, _PatchCredit],
698
+ ) -> int:
699
+ if history is None:
700
+ return 0
701
+ count = 0
702
+ for record in history.records:
703
+ patch = _patch_from_record(
704
+ record,
705
+ target=target,
706
+ seed_candidate=seed_candidate,
707
+ search_paths=search_paths,
708
+ )
709
+ if not patch:
710
+ continue
711
+ count += 1
712
+ delta = record.score - baseline_score
713
+ for path, value in patch.items():
714
+ _record_credit(
715
+ ledger,
716
+ path=path,
717
+ value=value,
718
+ score=record.score,
719
+ delta=delta,
720
+ passed=record.passed,
721
+ source="futureagi_history",
722
+ )
723
+ return count
724
+
725
+
726
+ def _prior_proposals_from_history(
727
+ history: Optional[AgentObservabilityWindow],
728
+ *,
729
+ target: OptimizationTarget,
730
+ seed_candidate: AgentCandidate,
731
+ search_paths: Sequence[str],
732
+ ) -> list[_MemoryProposal]:
733
+ if history is None:
734
+ return []
735
+ proposals: list[_MemoryProposal] = []
736
+ seen: set[str] = set()
737
+ ranked_records = sorted(
738
+ history.records,
739
+ key=lambda item: (item.passed, item.score, item.run_id or "", item.index),
740
+ reverse=True,
741
+ )
742
+ for record in ranked_records:
743
+ patch = _patch_from_record(
744
+ record,
745
+ target=target,
746
+ seed_candidate=seed_candidate,
747
+ search_paths=search_paths,
748
+ )
749
+ if not patch:
750
+ continue
751
+ proposal = _MemoryProposal(
752
+ patch=patch,
753
+ role="smriti",
754
+ parent_ids=tuple(filter(None, [record.candidate_id, record.run_id])),
755
+ reason="replay_high_signal_futureagi_history_patch",
756
+ metadata={
757
+ "role_archetype": "working_memory",
758
+ "role_kind": "futureagi_prior",
759
+ "futureagi_record_index": record.index,
760
+ "futureagi_run_id": record.run_id,
761
+ "futureagi_record_score": record.score,
762
+ "futureagi_record_passed": record.passed,
763
+ },
764
+ )
765
+ _append_proposal(proposals, seen, proposal, max_proposals=64)
766
+ return proposals
767
+
768
+
769
+ def _patch_from_record(
770
+ record: AgentObservabilityRecord,
771
+ *,
772
+ target: OptimizationTarget,
773
+ seed_candidate: AgentCandidate,
774
+ search_paths: Sequence[str],
775
+ ) -> dict[str, Any]:
776
+ allowed = set(search_paths)
777
+ for payload in _record_payloads(record):
778
+ patch = _explicit_patch_from_payload(payload, target=target, allowed=allowed)
779
+ if patch:
780
+ return patch
781
+ config = _candidate_config_from_payload(payload)
782
+ if config:
783
+ return _patch_from_config(
784
+ config,
785
+ target=target,
786
+ seed_candidate=seed_candidate,
787
+ allowed=allowed,
788
+ )
789
+ return {}
790
+
791
+
792
+ def _record_payloads(record: AgentObservabilityRecord) -> Iterable[Mapping[str, Any]]:
793
+ yield record.metadata
794
+ yield record.raw
795
+ for payload in (record.metadata, record.raw):
796
+ for key in (
797
+ "metadata",
798
+ "raw_variant",
799
+ "row_values",
800
+ "raw_row",
801
+ "candidate",
802
+ "variant",
803
+ "outputs",
804
+ ):
805
+ value = payload.get(key)
806
+ if isinstance(value, Mapping):
807
+ yield value
808
+
809
+
810
+ def _explicit_patch_from_payload(
811
+ payload: Mapping[str, Any],
812
+ *,
813
+ target: OptimizationTarget,
814
+ allowed: set[str],
815
+ ) -> dict[str, Any]:
816
+ for key in (
817
+ "candidate_patch",
818
+ "config_patch",
819
+ "patch",
820
+ "agent_patch",
821
+ "optimized_patch",
822
+ ):
823
+ value = payload.get(key)
824
+ if isinstance(value, Mapping):
825
+ patch = {
826
+ str(path): copy.deepcopy(patch_value)
827
+ for path, patch_value in value.items()
828
+ if str(path) in allowed
829
+ and _value_allowed(str(path), patch_value, target.search_space)
830
+ }
831
+ if patch:
832
+ return patch
833
+ return {}
834
+
835
+
836
+ def _candidate_config_from_payload(payload: Mapping[str, Any]) -> dict[str, Any]:
837
+ for key in (
838
+ "candidate_config",
839
+ "config",
840
+ "agent_config",
841
+ "optimized_config",
842
+ "workflow_config",
843
+ ):
844
+ value = payload.get(key)
845
+ if isinstance(value, Mapping):
846
+ return copy.deepcopy(dict(value))
847
+ return {}
848
+
849
+
850
+ def _patch_from_config(
851
+ config: Mapping[str, Any],
852
+ *,
853
+ target: OptimizationTarget,
854
+ seed_candidate: AgentCandidate,
855
+ allowed: set[str],
856
+ ) -> dict[str, Any]:
857
+ candidate = AgentCandidate.from_config(dict(config), target_name=target.name)
858
+ patch: dict[str, Any] = {}
859
+ for path in target.search_space:
860
+ if path not in allowed:
861
+ continue
862
+ value = candidate.get_path(path, _NO_VALUE)
863
+ if value is _NO_VALUE:
864
+ continue
865
+ if value == seed_candidate.get_path(path):
866
+ continue
867
+ if not _value_allowed(path, value, target.search_space):
868
+ continue
869
+ patch[path] = copy.deepcopy(value)
870
+ return patch
871
+
872
+
873
+ def _record_credit_from_evaluation(
874
+ evaluation: CandidateEvaluation,
875
+ *,
876
+ baseline_score: float,
877
+ ledger: dict[str, _PatchCredit],
878
+ source: str,
879
+ ) -> None:
880
+ if not evaluation.candidate.patch:
881
+ return
882
+ delta = evaluation.score - baseline_score
883
+ passed = evaluation.score >= baseline_score
884
+ for path, value in evaluation.candidate.patch.items():
885
+ _record_credit(
886
+ ledger,
887
+ path=path,
888
+ value=value,
889
+ score=evaluation.score,
890
+ delta=delta,
891
+ passed=passed,
892
+ source=source,
893
+ )
894
+
895
+
896
+ def _record_credit(
897
+ ledger: dict[str, _PatchCredit],
898
+ *,
899
+ path: str,
900
+ value: Any,
901
+ score: float,
902
+ delta: float,
903
+ passed: bool,
904
+ source: str,
905
+ ) -> None:
906
+ key = _credit_key(path, value)
907
+ credit = ledger.get(key)
908
+ if credit is None:
909
+ credit = _PatchCredit(path=path, value=copy.deepcopy(value))
910
+ ledger[key] = credit
911
+ credit.observations += 1
912
+ credit.total_delta += delta
913
+ credit.total_score += score
914
+ credit.best_score = max(credit.best_score, score)
915
+ if passed:
916
+ credit.passed += 1
917
+ else:
918
+ credit.failed += 1
919
+ credit.sources.add(source)
920
+
921
+
922
+ def _top_credit_patch(
923
+ ledger: Mapping[str, _PatchCredit],
924
+ *,
925
+ search_paths: Sequence[str],
926
+ ) -> dict[str, Any]:
927
+ patch: dict[str, Any] = {}
928
+ for path in search_paths:
929
+ credit = _best_credit(path, ledger)
930
+ if credit is None:
931
+ continue
932
+ if credit.mean_delta <= 0.0 and credit.passed <= credit.failed:
933
+ continue
934
+ patch[path] = copy.deepcopy(credit.value)
935
+ return patch
936
+
937
+
938
+ def _best_credit_value(path: str, ledger: Mapping[str, _PatchCredit]) -> Any:
939
+ credit = _best_credit(path, ledger)
940
+ if credit is None:
941
+ return _NO_VALUE
942
+ return copy.deepcopy(credit.value)
943
+
944
+
945
+ def _best_credit(
946
+ path: str,
947
+ ledger: Mapping[str, _PatchCredit],
948
+ ) -> Optional[_PatchCredit]:
949
+ candidates = [credit for credit in ledger.values() if credit.path == path]
950
+ if not candidates:
951
+ return None
952
+ return max(
953
+ candidates,
954
+ key=lambda credit: (
955
+ credit.mean_delta,
956
+ credit.best_score,
957
+ credit.mean_score,
958
+ credit.passed,
959
+ -credit.failed,
960
+ _canonical_value(credit.value),
961
+ ),
962
+ )
963
+
964
+
965
+ def _credit_for_value(
966
+ path: str,
967
+ value: Any,
968
+ ledger: Mapping[str, _PatchCredit],
969
+ ) -> Optional[_PatchCredit]:
970
+ return ledger.get(_credit_key(path, value))
971
+
972
+
973
+ def _path_groups(
974
+ search_paths: Sequence[str],
975
+ diagnoses: Sequence[ComponentDiagnosis],
976
+ ) -> dict[str, list[str]]:
977
+ groups: dict[str, list[str]] = {}
978
+ for path in search_paths:
979
+ group_key = _diagnostic_group_key(path, diagnoses) or path.split(".", 1)[0]
980
+ groups.setdefault(group_key, []).append(path)
981
+ return groups
982
+
983
+
984
+ def _diagnostic_group_key(
985
+ path: str,
986
+ diagnoses: Sequence[ComponentDiagnosis],
987
+ ) -> Optional[str]:
988
+ for diagnosis in diagnoses:
989
+ for suggested_path in diagnosis.suggested_paths:
990
+ if path == suggested_path or path.startswith(f"{suggested_path}."):
991
+ return f"{diagnosis.component}:{suggested_path}"
992
+ if path == diagnosis.component or path.startswith(f"{diagnosis.component}."):
993
+ return diagnosis.component
994
+ return None
995
+
996
+
997
+ def _first_non_seed_value(
998
+ seed_candidate: AgentCandidate,
999
+ search_space: Mapping[str, List[Any]],
1000
+ path: str,
1001
+ ) -> Any:
1002
+ current = seed_candidate.get_path(path)
1003
+ for value in search_space.get(path, []):
1004
+ if value != current:
1005
+ return copy.deepcopy(value)
1006
+ return _NO_VALUE
1007
+
1008
+
1009
+ def _last_non_seed_value(
1010
+ seed_candidate: AgentCandidate,
1011
+ search_space: Mapping[str, List[Any]],
1012
+ path: str,
1013
+ ) -> Any:
1014
+ current = seed_candidate.get_path(path)
1015
+ for value in reversed(search_space.get(path, [])):
1016
+ if value != current:
1017
+ return copy.deepcopy(value)
1018
+ return _NO_VALUE
1019
+
1020
+
1021
+ def _first_non_candidate_value(
1022
+ candidate: AgentCandidate,
1023
+ search_space: Mapping[str, List[Any]],
1024
+ path: str,
1025
+ ) -> Any:
1026
+ current = candidate.get_path(path)
1027
+ for value in search_space.get(path, []):
1028
+ if value != current:
1029
+ return copy.deepcopy(value)
1030
+ return _NO_VALUE
1031
+
1032
+
1033
+ def _append_proposal(
1034
+ proposals: list[_MemoryProposal],
1035
+ seen: set[str],
1036
+ proposal: _MemoryProposal,
1037
+ max_proposals: int,
1038
+ ) -> None:
1039
+ if len(proposals) >= max_proposals or not proposal.patch:
1040
+ return
1041
+ key = _canonical_patch(proposal.patch)
1042
+ if key in seen:
1043
+ return
1044
+ seen.add(key)
1045
+ proposals.append(proposal)
1046
+
1047
+
1048
+ def _candidate_id_for_patch(
1049
+ seed_candidate: AgentCandidate,
1050
+ patch: dict[str, Any],
1051
+ ) -> str:
1052
+ return seed_candidate.with_patch(patch).id
1053
+
1054
+
1055
+ def _value_allowed(
1056
+ path: str,
1057
+ value: Any,
1058
+ search_space: Mapping[str, List[Any]],
1059
+ ) -> bool:
1060
+ return any(candidate_value == value for candidate_value in search_space.get(path, []))
1061
+
1062
+
1063
+ def _history_baseline_score(
1064
+ history: Optional[AgentObservabilityWindow],
1065
+ ) -> float:
1066
+ if history is None or history.average_score is None:
1067
+ return 0.0
1068
+ return history.average_score
1069
+
1070
+
1071
+ def _ledger_summary(ledger: Mapping[str, _PatchCredit]) -> list[dict[str, Any]]:
1072
+ return [
1073
+ {
1074
+ "path": credit.path,
1075
+ "value": copy.deepcopy(credit.value),
1076
+ "observations": credit.observations,
1077
+ "mean_delta": credit.mean_delta,
1078
+ "mean_score": credit.mean_score,
1079
+ "best_score": credit.best_score,
1080
+ "passed": credit.passed,
1081
+ "failed": credit.failed,
1082
+ "sources": sorted(credit.sources),
1083
+ }
1084
+ for credit in sorted(
1085
+ ledger.values(),
1086
+ key=lambda item: (
1087
+ item.mean_delta,
1088
+ item.best_score,
1089
+ item.path,
1090
+ _canonical_value(item.value),
1091
+ ),
1092
+ reverse=True,
1093
+ )
1094
+ ]
1095
+
1096
+
1097
+ def _credit_key(path: str, value: Any) -> str:
1098
+ return f"{path}:{_canonical_value(value)}"
1099
+
1100
+
1101
+ def _canonical_patch(patch: Mapping[str, Any]) -> str:
1102
+ return json.dumps(patch, sort_keys=True, default=str)
1103
+
1104
+
1105
+ def _canonical_value(value: Any) -> str:
1106
+ return json.dumps(value, sort_keys=True, default=str)
1107
+
1108
+
1109
+ class _NoValue:
1110
+ pass
1111
+
1112
+
1113
+ _NO_VALUE = _NoValue()