agentevolve-optimizer 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (414) hide show
  1. agent_evolve/__init__.py +722 -0
  2. agent_evolve/agentic.py +2800 -0
  3. agent_evolve/api.py +767 -0
  4. agent_evolve/application/__init__.py +1580 -0
  5. agent_evolve/application/action_allocation.py +744 -0
  6. agent_evolve/application/action_allocation_frame.py +347 -0
  7. agent_evolve/application/action_allocation_frame_commit.py +185 -0
  8. agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
  9. agent_evolve/application/action_allocation_frame_v3.py +338 -0
  10. agent_evolve/application/action_archive_value.py +497 -0
  11. agent_evolve/application/action_evidence_consistency.py +455 -0
  12. agent_evolve/application/action_forecast_partitioning.py +1471 -0
  13. agent_evolve/application/action_metric_projection.py +211 -0
  14. agent_evolve/application/action_role_value.py +680 -0
  15. agent_evolve/application/action_score_authorities.py +363 -0
  16. agent_evolve/application/action_structural_signature.py +116 -0
  17. agent_evolve/application/action_target_realization.py +402 -0
  18. agent_evolve/application/agentic_evolution.py +7734 -0
  19. agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
  20. agent_evolve/application/anchor_residual_identification.py +463 -0
  21. agent_evolve/application/archive_conditioned_action_target.py +208 -0
  22. agent_evolve/application/artifact_journal.py +246 -0
  23. agent_evolve/application/artifact_replay.py +347 -0
  24. agent_evolve/application/budgeted_optimizer.py +1828 -0
  25. agent_evolve/application/calibrated_campaign.py +485 -0
  26. agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
  27. agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
  28. agent_evolve/application/campaign_capacity_recourse.py +254 -0
  29. agent_evolve/application/campaign_contextual_outcomes.py +119 -0
  30. agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
  31. agent_evolve/application/campaign_evidence_registry.py +262 -0
  32. agent_evolve/application/campaign_execution.py +2537 -0
  33. agent_evolve/application/campaign_generation_audit.py +942 -0
  34. agent_evolve/application/campaign_learning.py +1812 -0
  35. agent_evolve/application/campaign_learning_runtime.py +1977 -0
  36. agent_evolve/application/campaign_search_phase.py +227 -0
  37. agent_evolve/application/campaign_selector_context_extension.py +220 -0
  38. agent_evolve/application/campaign_variation_envelope.py +649 -0
  39. agent_evolve/application/campaign_variation_trace.py +451 -0
  40. agent_evolve/application/candidate_archive_consequence.py +128 -0
  41. agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
  42. agent_evolve/application/composite_outcome_updater.py +145 -0
  43. agent_evolve/application/composition_portfolio_selection.py +363 -0
  44. agent_evolve/application/concurrent_stage.py +144 -0
  45. agent_evolve/application/contextual_action_allocation.py +181 -0
  46. agent_evolve/application/contextual_campaign_outcomes.py +267 -0
  47. agent_evolve/application/contextual_campaign_planning.py +1366 -0
  48. agent_evolve/application/contextual_delayed_credit.py +651 -0
  49. agent_evolve/application/contextual_search_controller.py +2374 -0
  50. agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
  51. agent_evolve/application/decision_metric_projection.py +112 -0
  52. agent_evolve/application/derived_action_semantics.py +129 -0
  53. agent_evolve/application/detailed_evaluation.py +449 -0
  54. agent_evolve/application/earned_lineage.py +1011 -0
  55. agent_evolve/application/effective_choice_audit.py +484 -0
  56. agent_evolve/application/empirical_consequence_calibration.py +908 -0
  57. agent_evolve/application/evaluation_accounting.py +325 -0
  58. agent_evolve/application/evaluation_cache.py +199 -0
  59. agent_evolve/application/evaluation_escrow.py +547 -0
  60. agent_evolve/application/evaluation_recourse.py +253 -0
  61. agent_evolve/application/event_recorder.py +151 -0
  62. agent_evolve/application/evolution_campaign.py +1840 -0
  63. agent_evolve/application/executable_hypothesis.py +323 -0
  64. agent_evolve/application/factorial_branch_pilot.py +772 -0
  65. agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
  66. agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
  67. agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
  68. agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
  69. agent_evolve/application/finite_action_selection.py +188 -0
  70. agent_evolve/application/finite_action_set.py +306 -0
  71. agent_evolve/application/finite_action_transition.py +537 -0
  72. agent_evolve/application/finite_variation_eligibility.py +296 -0
  73. agent_evolve/application/forecast_geometry_portfolio.py +799 -0
  74. agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
  75. agent_evolve/application/front_proximity_admission.py +311 -0
  76. agent_evolve/application/front_proximity_parent_basis.py +458 -0
  77. agent_evolve/application/frozen_hurdle_score.py +659 -0
  78. agent_evolve/application/g3_causal_screen.py +2257 -0
  79. agent_evolve/application/g3_causal_validation.py +1046 -0
  80. agent_evolve/application/g3_postseal_curation.py +818 -0
  81. agent_evolve/application/gated_agentic_generator.py +205 -0
  82. agent_evolve/application/generation_feedback.py +293 -0
  83. agent_evolve/application/generative_proposal_journal.py +185 -0
  84. agent_evolve/application/geometry_conditional_elasticity.py +453 -0
  85. agent_evolve/application/global_wave_action_allocation.py +1151 -0
  86. agent_evolve/application/head_mass_conditional_seat.py +268 -0
  87. agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
  88. agent_evolve/application/identifiable_reflection_learning.py +395 -0
  89. agent_evolve/application/identifiable_reflection_request.py +364 -0
  90. agent_evolve/application/in_memory_residual_archive.py +341 -0
  91. agent_evolve/application/insight_memory.py +1804 -0
  92. agent_evolve/application/live_runtime_manifest.py +758 -0
  93. agent_evolve/application/llm_task_queue.py +769 -0
  94. agent_evolve/application/matched_finite_action_block.py +409 -0
  95. agent_evolve/application/materialized_action_broker.py +2328 -0
  96. agent_evolve/application/materialized_action_constraints.py +83 -0
  97. agent_evolve/application/materialized_variation.py +211 -0
  98. agent_evolve/application/multi_option_evolution.py +1536 -0
  99. agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
  100. agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
  101. agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
  102. agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
  103. agent_evolve/application/outcome_relation.py +193 -0
  104. agent_evolve/application/paired_allocation_comparison.py +241 -0
  105. agent_evolve/application/paired_block_schedule.py +127 -0
  106. agent_evolve/application/parent_measurement.py +226 -0
  107. agent_evolve/application/pareto_archive.py +811 -0
  108. agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
  109. agent_evolve/application/portfolio_evolution.py +2950 -0
  110. agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
  111. agent_evolve/application/portfolio_memory_attribution.py +581 -0
  112. agent_evolve/application/portfolio_memory_dose.py +788 -0
  113. agent_evolve/application/portfolio_memory_matched_control.py +938 -0
  114. agent_evolve/application/portfolio_memory_transfer.py +297 -0
  115. agent_evolve/application/portfolio_optimization_memory.py +363 -0
  116. agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
  117. agent_evolve/application/portfolio_projection.py +335 -0
  118. agent_evolve/application/portfolio_recombination.py +2032 -0
  119. agent_evolve/application/post_evolution_reflection.py +834 -0
  120. agent_evolve/application/postcommit_rank_authority.py +245 -0
  121. agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
  122. agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
  123. agent_evolve/application/prequential_residual_exploration.py +343 -0
  124. agent_evolve/application/prequential_score_portfolio.py +954 -0
  125. agent_evolve/application/projections.py +292 -0
  126. agent_evolve/application/protected_action_committee.py +1027 -0
  127. agent_evolve/application/protected_branch_pilot.py +376 -0
  128. agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
  129. agent_evolve/application/provider_replay.py +910 -0
  130. agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
  131. agent_evolve/application/recombination_residual_expert.py +403 -0
  132. agent_evolve/application/reflection_workflow.py +571 -0
  133. agent_evolve/application/region_conditional_credit.py +911 -0
  134. agent_evolve/application/residual_campaign_runtime.py +531 -0
  135. agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
  136. agent_evolve/application/residual_headroom_ledger.py +1544 -0
  137. agent_evolve/application/residual_learning_transaction.py +396 -0
  138. agent_evolve/application/residual_portfolio_evolution.py +1228 -0
  139. agent_evolve/application/residual_reachability.py +749 -0
  140. agent_evolve/application/residual_stage_credit.py +499 -0
  141. agent_evolve/application/same_prefix_paired_audit.py +1580 -0
  142. agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
  143. agent_evolve/application/sequential_lineage_allocation.py +1017 -0
  144. agent_evolve/application/sequential_market_replay.py +1395 -0
  145. agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
  146. agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
  147. agent_evolve/application/single_score_action_allocation.py +299 -0
  148. agent_evolve/application/source_exposure_allocation.py +906 -0
  149. agent_evolve/application/staged_memory.py +210 -0
  150. agent_evolve/application/stratified_cold_start_allocation.py +732 -0
  151. agent_evolve/application/support_guarded_hurdle_score.py +549 -0
  152. agent_evolve/application/target_conditioned_action_forecast.py +595 -0
  153. agent_evolve/application/target_conditioned_campaign.py +566 -0
  154. agent_evolve/application/treatment_assignment.py +201 -0
  155. agent_evolve/application/trusted_objective_evidence.py +217 -0
  156. agent_evolve/application/two_stage_action_evolution.py +1131 -0
  157. agent_evolve/application/v8lite_allocation_policy.py +1083 -0
  158. agent_evolve/application/v9_candidate_policy.py +1303 -0
  159. agent_evolve/bootstrap.py +108 -0
  160. agent_evolve/campaign_presets.py +517 -0
  161. agent_evolve/campaign_profiles.py +452 -0
  162. agent_evolve/campaign_variation_topology.py +288 -0
  163. agent_evolve/campaign_workload.py +950 -0
  164. agent_evolve/cli.py +797 -0
  165. agent_evolve/contract.py +241 -0
  166. agent_evolve/core/__init__.py +91 -0
  167. agent_evolve/core/action_semantics.py +411 -0
  168. agent_evolve/core/authored.py +105 -0
  169. agent_evolve/core/formatting.py +286 -0
  170. agent_evolve/core/optimization_semantics.py +324 -0
  171. agent_evolve/core/problem.py +167 -0
  172. agent_evolve/core/results.py +323 -0
  173. agent_evolve/core/stats.py +70 -0
  174. agent_evolve/core/telemetry.py +100 -0
  175. agent_evolve/domain/__init__.py +89 -0
  176. agent_evolve/domain/artifact.py +162 -0
  177. agent_evolve/domain/durable_text.py +68 -0
  178. agent_evolve/domain/event.py +1454 -0
  179. agent_evolve/domain/finite_action_set.py +426 -0
  180. agent_evolve/domain/finite_variation.py +526 -0
  181. agent_evolve/domain/generative_emission.py +559 -0
  182. agent_evolve/domain/ids.py +163 -0
  183. agent_evolve/domain/inline_text.py +106 -0
  184. agent_evolve/domain/insight.py +27 -0
  185. agent_evolve/domain/lineage.py +737 -0
  186. agent_evolve/domain/llm_task_queue.py +960 -0
  187. agent_evolve/domain/outcome.py +96 -0
  188. agent_evolve/domain/patch.py +854 -0
  189. agent_evolve/domain/typed_json.py +542 -0
  190. agent_evolve/domain/variation_space.py +158 -0
  191. agent_evolve/driver.py +1014 -0
  192. agent_evolve/harness/__init__.py +29 -0
  193. agent_evolve/harness/base.py +242 -0
  194. agent_evolve/harness/directives.py +163 -0
  195. agent_evolve/harness/generative_seal.py +479 -0
  196. agent_evolve/harness/registry.py +41 -0
  197. agent_evolve/infrastructure/__init__.py +39 -0
  198. agent_evolve/infrastructure/artifacts/__init__.py +6 -0
  199. agent_evolve/infrastructure/artifacts/_verification.py +67 -0
  200. agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
  201. agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
  202. agent_evolve/infrastructure/asyncio_runtime.py +109 -0
  203. agent_evolve/infrastructure/authored_runtime.py +188 -0
  204. agent_evolve/infrastructure/authored_worker.py +171 -0
  205. agent_evolve/infrastructure/clock.py +53 -0
  206. agent_evolve/infrastructure/events/__init__.py +6 -0
  207. agent_evolve/infrastructure/events/_validation.py +89 -0
  208. agent_evolve/infrastructure/events/in_memory.py +56 -0
  209. agent_evolve/infrastructure/events/jsonl.py +193 -0
  210. agent_evolve/infrastructure/exception_provenance.py +215 -0
  211. agent_evolve/infrastructure/ids.py +118 -0
  212. agent_evolve/infrastructure/lineage_codec.py +1836 -0
  213. agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
  214. agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
  215. agent_evolve/infrastructure/resource_lease.py +370 -0
  216. agent_evolve/infrastructure/sanitization/__init__.py +8 -0
  217. agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
  218. agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
  219. agent_evolve/infrastructure/stream_liveness.py +383 -0
  220. agent_evolve/infrastructure/subprocess_boundary.py +136 -0
  221. agent_evolve/integrations/__init__.py +1 -0
  222. agent_evolve/integrations/botorch/__init__.py +28 -0
  223. agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
  224. agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
  225. agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
  226. agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
  227. agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
  228. agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
  229. agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
  230. agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
  231. agent_evolve/integrations/completion.py +242 -0
  232. agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
  233. agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
  234. agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
  235. agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
  236. agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
  237. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
  238. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
  239. agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
  240. agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
  241. agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
  242. agent_evolve/integrations/pydantic_ai/harness.py +159 -0
  243. agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
  244. agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
  245. agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
  246. agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
  247. agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
  248. agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
  249. agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
  250. agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
  251. agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
  252. agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
  253. agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
  254. agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
  255. agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
  256. agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
  257. agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
  258. agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
  259. agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
  260. agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
  261. agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
  262. agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
  263. agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
  264. agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
  265. agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
  266. agent_evolve/integrations/pymoo_adapter.py +242 -0
  267. agent_evolve/policies/__init__.py +17 -0
  268. agent_evolve/policies/check.py +469 -0
  269. agent_evolve/policies/emit_scaffold.py +451 -0
  270. agent_evolve/policies/feedback/__init__.py +37 -0
  271. agent_evolve/policies/feedback/held_out_asn.py +1325 -0
  272. agent_evolve/policies/genetic.py +607 -0
  273. agent_evolve/policies/llm_backoff.py +183 -0
  274. agent_evolve/policies/llm_chooser.py +226 -0
  275. agent_evolve/policies/llm_generator.py +1760 -0
  276. agent_evolve/policies/llm_init.py +267 -0
  277. agent_evolve/policies/llm_operator.py +109 -0
  278. agent_evolve/policies/llm_prior.py +194 -0
  279. agent_evolve/policies/llm_surrogate.py +334 -0
  280. agent_evolve/policies/measurement_evidence.py +704 -0
  281. agent_evolve/policies/memory/__init__.py +223 -0
  282. agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
  283. agent_evolve/policies/memory/compatibility_matching.py +593 -0
  284. agent_evolve/policies/memory/global_falsification.py +1841 -0
  285. agent_evolve/policies/memory/prompt_shape.py +503 -0
  286. agent_evolve/policies/memory/randomized_subset.py +714 -0
  287. agent_evolve/policies/memory/staged_causal.py +1270 -0
  288. agent_evolve/policies/memory/treatment_compliance.py +759 -0
  289. agent_evolve/policies/objective_resolution/__init__.py +17 -0
  290. agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
  291. agent_evolve/policies/operator_portfolio.py +407 -0
  292. agent_evolve/policies/reguidance.py +1133 -0
  293. agent_evolve/policies/reward/__init__.py +83 -0
  294. agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
  295. agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
  296. agent_evolve/policies/reward/affine_hypervolume.py +490 -0
  297. agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
  298. agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
  299. agent_evolve/policies/reward/frozen_archive.py +360 -0
  300. agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
  301. agent_evolve/policies/search_state.py +208 -0
  302. agent_evolve/policies/selection/__init__.py +345 -0
  303. agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
  304. agent_evolve/policies/selection/affine_frontier_context.py +330 -0
  305. agent_evolve/policies/selection/affine_frontier_target.py +473 -0
  306. agent_evolve/policies/selection/archive_elite.py +1346 -0
  307. agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
  308. agent_evolve/policies/selection/calibrated_slate.py +1394 -0
  309. agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
  310. agent_evolve/policies/selection/common_candidate_pool.py +685 -0
  311. agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
  312. agent_evolve/policies/selection/disjoint_pairs.py +479 -0
  313. agent_evolve/policies/selection/elite_explorer.py +719 -0
  314. agent_evolve/policies/selection/finite_action.py +187 -0
  315. agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
  316. agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
  317. agent_evolve/policies/selection/forecast_calibration.py +922 -0
  318. agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
  319. agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
  320. agent_evolve/policies/selection/full_support_slate.py +91 -0
  321. agent_evolve/policies/selection/meaningful_direction.py +240 -0
  322. agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
  323. agent_evolve/policies/selection/model_anchored_slate.py +826 -0
  324. agent_evolve/policies/selection/phenotype_recourse.py +979 -0
  325. agent_evolve/policies/selection/proposal_support.py +368 -0
  326. agent_evolve/policies/selection/random_portfolio.py +254 -0
  327. agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
  328. agent_evolve/policies/selection/residual_frontier.py +463 -0
  329. agent_evolve/policies/selection/residual_frontier_target.py +605 -0
  330. agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
  331. agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
  332. agent_evolve/policies/selection/target_conditioned_features.py +812 -0
  333. agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
  334. agent_evolve/policies/selection/task_keyed_palette.py +906 -0
  335. agent_evolve/policies/semantics.py +147 -0
  336. agent_evolve/policies/structure.py +362 -0
  337. agent_evolve/policies/structured_output_budget.py +62 -0
  338. agent_evolve/policies/surrogate.py +696 -0
  339. agent_evolve/policies/variation/__init__.py +1 -0
  340. agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
  341. agent_evolve/policies/variation/crossover_inheritance.py +575 -0
  342. agent_evolve/policies/variation/disjoint_recombination.py +611 -0
  343. agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
  344. agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
  345. agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
  346. agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
  347. agent_evolve/policies/variation/typed_patch.py +1981 -0
  348. agent_evolve/policies/weighted_prior.py +394 -0
  349. agent_evolve/ports/__init__.py +383 -0
  350. agent_evolve/ports/action_allocation.py +733 -0
  351. agent_evolve/ports/action_allocation_frame.py +1153 -0
  352. agent_evolve/ports/action_allocation_frame_commit.py +294 -0
  353. agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
  354. agent_evolve/ports/action_allocation_frame_v3.py +995 -0
  355. agent_evolve/ports/action_forecast.py +1568 -0
  356. agent_evolve/ports/action_metric_projection.py +165 -0
  357. agent_evolve/ports/agentic_generator.py +1561 -0
  358. agent_evolve/ports/archive_context.py +136 -0
  359. agent_evolve/ports/artifact_sanitizer.py +44 -0
  360. agent_evolve/ports/artifact_store.py +225 -0
  361. agent_evolve/ports/clock.py +13 -0
  362. agent_evolve/ports/contextual_search_allocation.py +827 -0
  363. agent_evolve/ports/decision_metric_projection.py +258 -0
  364. agent_evolve/ports/event_store.py +55 -0
  365. agent_evolve/ports/executable_hypothesis.py +557 -0
  366. agent_evolve/ports/finite_acquisition.py +377 -0
  367. agent_evolve/ports/finite_acquisition_batch.py +296 -0
  368. agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
  369. agent_evolve/ports/finite_acquisition_json.py +247 -0
  370. agent_evolve/ports/finite_acquisition_space.py +168 -0
  371. agent_evolve/ports/finite_action_selection.py +348 -0
  372. agent_evolve/ports/finite_action_set.py +256 -0
  373. agent_evolve/ports/frontier_target.py +396 -0
  374. agent_evolve/ports/generation_failure.py +43 -0
  375. agent_evolve/ports/hard_feasibility.py +233 -0
  376. agent_evolve/ports/id_factory.py +34 -0
  377. agent_evolve/ports/llm_task_queue.py +93 -0
  378. agent_evolve/ports/objective_resolution.py +419 -0
  379. agent_evolve/ports/paired_allocation_comparison.py +401 -0
  380. agent_evolve/ports/paired_block_schedule.py +475 -0
  381. agent_evolve/ports/parent_measurement.py +336 -0
  382. agent_evolve/ports/portfolio_memory_dose.py +643 -0
  383. agent_evolve/ports/portfolio_selection.py +3169 -0
  384. agent_evolve/ports/postcommit_rank_authority.py +467 -0
  385. agent_evolve/ports/presented_action_evidence.py +794 -0
  386. agent_evolve/ports/resource_lease.py +162 -0
  387. agent_evolve/ports/structured_generator.py +734 -0
  388. agent_evolve/ports/structured_output_budget.py +120 -0
  389. agent_evolve/ports/subprocess_boundary.py +138 -0
  390. agent_evolve/ports/treatment_assignment.py +466 -0
  391. agent_evolve/ports/variation_catalog.py +76 -0
  392. agent_evolve/ports/variation_source.py +226 -0
  393. agent_evolve/proposal_mode.py +157 -0
  394. agent_evolve/proposers/__init__.py +10 -0
  395. agent_evolve/proposers/random_proposer.py +188 -0
  396. agent_evolve/provider_accounting.py +163 -0
  397. agent_evolve/py.typed +0 -0
  398. agent_evolve/reference_method.py +1570 -0
  399. agent_evolve/session/__init__.py +11 -0
  400. agent_evolve/session/authorship.py +864 -0
  401. agent_evolve/session/evaluate.py +236 -0
  402. agent_evolve/session/fidelity.py +237 -0
  403. agent_evolve/session/genetic_loop.py +742 -0
  404. agent_evolve/session/loop.py +803 -0
  405. agent_evolve/session/screening.py +671 -0
  406. agent_evolve/settings.py +376 -0
  407. agent_evolve/workload_kit.py +368 -0
  408. agent_evolve/workload_prompt.py +398 -0
  409. agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
  410. agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
  411. agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
  412. agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
  413. agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
  414. agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,803 @@
1
+ """Pareto-guided evolutionary loop.
2
+
3
+ Pure Python orchestration: it depends only on ``core`` and the ``Harness`` port,
4
+ and knows nothing about pydantic-ai or any other LLM runtime. Switching the
5
+ harness changes nothing in this file.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import random
12
+ import time
13
+ from dataclasses import dataclass
14
+ from typing import Any, Callable, Dict, List, MutableMapping, Optional, Sequence
15
+
16
+ from agent_evolve.core.formatting import (
17
+ CandidateResult,
18
+ prettify_configuration,
19
+ prettify_results,
20
+ result_to_candidate,
21
+ )
22
+ from agent_evolve.core.problem import ObjectiveSpec, Problem, validate_objective_specs
23
+ from agent_evolve.core.results import (
24
+ Candidate,
25
+ ProviderUsageSummary,
26
+ SearchResult,
27
+ compute_pareto_front,
28
+ objective_value,
29
+ select_minimax_rank,
30
+ )
31
+ from agent_evolve.core.stats import compute_performance_stats, sample_failed_for_constraint
32
+ from agent_evolve.harness.base import Harness
33
+ from agent_evolve.session.evaluate import evaluate_batch
34
+
35
+ LogFn = Callable[[str], None]
36
+ EventFn = Callable[[Dict[str, Any]], None]
37
+ RenderFn = Callable[[Dict[str, Any]], str]
38
+
39
+
40
+ @dataclass(frozen=True)
41
+ class LoopConfig:
42
+ """Hyper-parameters for one optimisation run."""
43
+
44
+ pop_size: int = 8
45
+ generations: int = 5
46
+ candidates_per_batch: int = 5
47
+ max_regen_rounds: int = 10
48
+ max_failed_examples: int = 5
49
+ seed: Optional[int] = None
50
+ llm_retries: int = 3
51
+ #: Reflection ablation switches (all True = the paper's full reflective loop).
52
+ #: Turning one off cleanly isolates that feedback arm; all-off is the
53
+ #: non-reflective generate-validate-regenerate baseline (the LLM still sees the
54
+ #: raw validator errors, but no LLM-synthesized insights/guide/patterns).
55
+ use_failure_insights: bool = True
56
+ use_constraint_instruction: bool = True
57
+ use_performance_insights: bool = True
58
+ #: Separate, more patient budget for provider rate-limit (HTTP 429) errors,
59
+ #: which are transient and should not exhaust the normal retry budget.
60
+ rate_limit_retries: int = 12
61
+ #: Starting configurations, evaluated before anything is proposed, so a run
62
+ #: can say whether what it proposed beat what the caller already had.
63
+ seeds: tuple = ()
64
+ #: Hard ceiling on artifacts measured. ``None`` means the generation
65
+ #: structure alone decides.
66
+ evaluation_budget: Optional[int] = None
67
+ #: Artifact-identity cache shared across the run. Supplying one makes
68
+ #: repeated materializations free and reports what that saved.
69
+ evaluation_cache: Optional[MutableMapping] = None
70
+
71
+
72
+ def _noop_log(msg: str) -> None:
73
+ pass
74
+
75
+
76
+ def _default_candidate_key(config: Dict[str, Any]) -> str:
77
+ return json.dumps(config, sort_keys=True, default=str)
78
+
79
+
80
+ def _is_rate_limit(exc: BaseException) -> bool:
81
+ text = f"{type(exc).__name__} {exc}".lower()
82
+ return "429" in text or "rate limit" in text or "rate_limited" in text
83
+
84
+
85
+ def _retry(fn: Callable, args: tuple, retries: int, log: LogFn,
86
+ rate_limit_retries: int = 12,
87
+ backoff_base: float = 3.0, backoff_cap: float = 30.0,
88
+ rate_limit_cap: float = 60.0) -> Any:
89
+ """Call *fn*, retrying on failure with exponential backoff.
90
+
91
+ Provider rate-limit (HTTP 429) errors are transient and use a separate, more
92
+ patient budget so they don't exhaust the normal retry budget.
93
+ """
94
+ normal = 0
95
+ limited = 0
96
+ while True:
97
+ try:
98
+ return fn(*args)
99
+ except Exception as exc: # noqa: BLE001 - retried then re-raised
100
+ if _is_rate_limit(exc):
101
+ limited += 1
102
+ if limited >= rate_limit_retries:
103
+ raise
104
+ delay = min(15.0 + 10.0 * limited, rate_limit_cap)
105
+ log(f" [rate-limit {limited}/{rate_limit_retries}] waiting {delay:.0f}s")
106
+ time.sleep(delay)
107
+ else:
108
+ normal += 1
109
+ if normal >= max(retries, 1):
110
+ raise
111
+ delay = min(backoff_base * (2 ** (normal - 1)), backoff_cap)
112
+ log(f" [retry {normal}/{retries}] {type(exc).__name__}: {exc}; waiting {delay:.0f}s")
113
+ time.sleep(delay)
114
+
115
+
116
+ @dataclass
117
+ class _RunState:
118
+ """Shared collaborators + mutable bookkeeping threaded through the loop."""
119
+
120
+ problem: Any
121
+ harness: Harness
122
+ objectives: Sequence[ObjectiveSpec]
123
+ config: LoopConfig
124
+ rng: random.Random
125
+ call: Callable
126
+ log: LogFn
127
+ on_event: Optional[EventFn]
128
+ render: Optional[RenderFn]
129
+ key_fn: Callable[[Dict[str, Any]], str]
130
+ seen: set
131
+ #: Count of valid candidates in the batch that last (re)wrote the constraint
132
+ #: guide; -1 means "no guide created yet" (port of `constraint_valid_count`).
133
+ constraint_valid_count: int = -1
134
+
135
+ def failed_str(self, results: Sequence[CandidateResult]) -> str:
136
+ return prettify_results(results, self.objectives, render=self.render)
137
+
138
+ def dedup(self, configs: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
139
+ """Drop already-seen configs so regen rounds don't re-evaluate duplicates."""
140
+ out: List[Dict[str, Any]] = []
141
+ for c in configs:
142
+ k = self.key_fn(c)
143
+ if k not in self.seen:
144
+ self.seen.add(k)
145
+ out.append(c)
146
+ rejected = len(configs) - len(out)
147
+ if rejected:
148
+ self.log(f" [dedup] rejected {rejected} duplicate candidate(s)")
149
+ return out
150
+
151
+
152
+ def _snapshot_best(
153
+ pareto: Sequence[Candidate], objectives: Sequence[ObjectiveSpec]
154
+ ) -> Candidate:
155
+ best = select_minimax_rank(list(pareto), objectives)
156
+ if best is None:
157
+ return Candidate(configuration={}, objectives={}, metadata={})
158
+ return best
159
+
160
+
161
+ def _emit(on_event: Optional[EventFn], event: Dict[str, Any]) -> None:
162
+ if on_event is not None:
163
+ on_event(event)
164
+
165
+
166
+ def _performance_stats_str(
167
+ valid_results: Sequence[CandidateResult],
168
+ objectives: Sequence[ObjectiveSpec],
169
+ render: Optional[RenderFn] = None,
170
+ ) -> tuple[str, str, int, int]:
171
+ """Return ``(stats_str, pareto_str, total_valid, pareto_size)``."""
172
+ stats = compute_performance_stats(valid_results, objectives)
173
+ if not stats:
174
+ return "", "None", 0, 0
175
+ lines: List[str] = [f"TOTAL VALID CANDIDATES: {len(valid_results)}"]
176
+ lines.append(f"PARETO FRONT SIZE: {stats.get('pareto_size', 0)}")
177
+ lines.append("")
178
+ lines.append("BEST AND WORST PER OBJECTIVE:")
179
+ for spec in objectives:
180
+ best = stats.get(f"best_{spec.name}")
181
+ worst = stats.get(f"worst_{spec.name}")
182
+ if best:
183
+ cfg = render(best.configuration) if render else prettify_configuration(best.configuration)
184
+ lines.append(f" Best {spec.name}: {objective_value(best.objectives, spec.name)}")
185
+ lines.append(f" Config: {cfg}")
186
+ if worst:
187
+ lines.append(f" Worst {spec.name}: {objective_value(worst.objectives, spec.name)}")
188
+ top_pareto = stats.get("top_3_pareto", [])
189
+ pareto_str = prettify_results(top_pareto, objectives, render=render) if top_pareto else "None"
190
+ return "\n".join(lines), pareto_str, len(valid_results), stats.get("pareto_size", 0)
191
+
192
+
193
+ def run_evolution_loop(
194
+ *,
195
+ problem: Problem,
196
+ harness: Harness,
197
+ config: LoopConfig,
198
+ log: LogFn = _noop_log,
199
+ on_event: Optional[EventFn] = None,
200
+ ) -> SearchResult:
201
+ """Run the full Pareto-guided evolutionary loop against *harness*."""
202
+ objectives = list(problem.objectives)
203
+ validate_objective_specs(objectives)
204
+ if config.pop_size <= 0:
205
+ raise ValueError("pop_size must be positive")
206
+ if config.generations <= 0:
207
+ raise ValueError("generations must be positive")
208
+ if config.candidates_per_batch <= 0:
209
+ raise ValueError("candidates_per_batch must be positive")
210
+ if config.max_regen_rounds < 0:
211
+ raise ValueError("max_regen_rounds must be non-negative")
212
+
213
+ rng = random.Random(config.seed)
214
+ retries = config.llm_retries
215
+
216
+ # Counted, never declared. A run that made no proposer call reports zero
217
+ # because zero was observed here, not because the field was left out.
218
+ proposer_calls = [0]
219
+
220
+ def call(fn: Callable, *args: Any) -> Any:
221
+ proposer_calls[0] += 1
222
+ return _retry(fn, args, retries, log, rate_limit_retries=config.rate_limit_retries)
223
+
224
+ key_fn = getattr(problem, "candidate_key", None) or _default_candidate_key
225
+ state = _RunState(
226
+ problem=problem,
227
+ harness=harness,
228
+ objectives=objectives,
229
+ config=config,
230
+ rng=rng,
231
+ call=call,
232
+ log=log,
233
+ on_event=on_event,
234
+ render=getattr(problem, "render_candidate", None),
235
+ key_fn=key_fn,
236
+ seen=set(),
237
+ )
238
+ # Seed the dedup set with the example/base config so it is never re-proposed.
239
+ example = getattr(problem, "example_config", None)
240
+ if isinstance(example, dict):
241
+ state.seen.add(key_fn(example))
242
+
243
+ all_valid: List[CandidateResult] = []
244
+ all_failed: List[CandidateResult] = []
245
+ all_candidates_meta: List[tuple] = []
246
+ history: List[Dict[str, Any]] = []
247
+ constraint_instruction = ""
248
+ performance_insights = ""
249
+ best_per_generation: List[Candidate] = []
250
+
251
+ log(
252
+ f"agent_evolve: {config.generations} generations, pop={config.pop_size}, "
253
+ f"batch={config.candidates_per_batch}, harness={getattr(harness, 'id', '?')}"
254
+ )
255
+
256
+ # -- Generation 0: the caller's own starting points -------------------
257
+ # Evaluated before anything is proposed, so the run can say plainly
258
+ # whether what it proposed beat what the caller already had.
259
+ seed_pareto: List[Candidate] = []
260
+ if config.seeds:
261
+ seed_configs = [dict(c) for c in config.seeds]
262
+ for c in seed_configs:
263
+ state.seen.add(key_fn(c))
264
+ seed_valid, seed_failed, _ = evaluate_batch(
265
+ problem, seed_configs, objectives, cache=config.evaluation_cache
266
+ )
267
+ all_valid.extend(seed_valid)
268
+ all_failed.extend(seed_failed)
269
+ for r in seed_valid + seed_failed:
270
+ all_candidates_meta.append((r, _candidate_metadata(
271
+ r, generation=0, authored_by=AUTHORED_BY_CALLER_SEED
272
+ )))
273
+ log(f" [seeds] {len(seed_valid)} valid, {len(seed_failed)} rejected")
274
+ seed_pareto = compute_pareto_front(
275
+ [result_to_candidate(r) for r in seed_valid], objectives
276
+ )
277
+
278
+ # -- Generation 1 -----------------------------------------------------
279
+ # With starting points that measured, the first proposal *breeds from
280
+ # them*. Sampling blind here instead -- which is what this did -- threw
281
+ # away the one thing the caller supplied and paid to evaluate, and made
282
+ # the run's first batch independent of the state it was told to start
283
+ # from. Without seeds the behaviour is unchanged: sample, then regenerate.
284
+ if seed_pareto:
285
+ if config.use_performance_insights:
286
+ stats_str, pareto_str, _, _ = _performance_stats_str(
287
+ all_valid, objectives, state.render
288
+ )
289
+ performance_insights = call(
290
+ harness.performance_insights, stats_str, pareto_str, None
291
+ )
292
+ gen1_valid, gen1_failed, constraint_instruction = _run_evolution_generation(
293
+ state=state,
294
+ gen=1,
295
+ prev_pareto=seed_pareto,
296
+ constraint_instruction=constraint_instruction,
297
+ performance_insights=performance_insights,
298
+ )
299
+ gen1_authored_by = AUTHORED_BY_OFFSPRING_PROPOSAL
300
+ else:
301
+ gen1_valid, gen1_failed, constraint_instruction = _run_initial_generation(
302
+ state=state,
303
+ constraint_instruction=constraint_instruction,
304
+ performance_insights=performance_insights,
305
+ )
306
+ gen1_authored_by = AUTHORED_BY_INITIAL_PROPOSAL
307
+
308
+ all_valid.extend(gen1_valid)
309
+ all_failed.extend(gen1_failed)
310
+ for r in gen1_valid + gen1_failed:
311
+ all_candidates_meta.append((r, _candidate_metadata(
312
+ r, generation=1, authored_by=gen1_authored_by
313
+ )))
314
+
315
+ pareto = compute_pareto_front([result_to_candidate(r) for r in all_valid], objectives)
316
+ best_per_generation.append(_snapshot_best(pareto, objectives))
317
+
318
+ if config.use_performance_insights and gen1_valid:
319
+ stats_str, pareto_str, _, _ = _performance_stats_str(all_valid, objectives, state.render)
320
+ # Carry the seed-derived insight forward rather than restarting from
321
+ # nothing: the chain from a measurement to the next proposal is the
322
+ # mechanism under test, and dropping a link in it is not an ablation,
323
+ # it is a bug.
324
+ performance_insights = call(
325
+ harness.performance_insights, stats_str, pareto_str, performance_insights or None
326
+ )
327
+
328
+ history.append(
329
+ {
330
+ "gen": 1,
331
+ "valid_count": len(gen1_valid),
332
+ "failed_count": len(gen1_failed),
333
+ "pareto_size": len(pareto),
334
+ }
335
+ )
336
+ _emit(on_event, {"kind": "generation_complete", "gen": 1, "pareto_size": len(pareto)})
337
+
338
+ def _spent() -> int:
339
+ """Artifacts actually measured, which is what the budget counts."""
340
+ cache = config.evaluation_cache
341
+ misses = getattr(cache, "misses", None)
342
+ if misses is not None:
343
+ return int(misses)
344
+ return sum(
345
+ 1 for r in all_valid + all_failed if getattr(r, "evaluation_attempted", False)
346
+ )
347
+
348
+ # -- Generations 2..N: evolution from the Pareto front ---------------
349
+ for gen in range(2, config.generations + 1):
350
+ if config.evaluation_budget is not None and _spent() >= config.evaluation_budget:
351
+ log(
352
+ f" [budget] {_spent()}/{config.evaluation_budget} evaluations "
353
+ "spent; stopping before generation "
354
+ f"{gen}"
355
+ )
356
+ break
357
+ gen_valid, gen_failed, constraint_instruction = _run_evolution_generation(
358
+ state=state,
359
+ gen=gen,
360
+ prev_pareto=pareto,
361
+ constraint_instruction=constraint_instruction,
362
+ performance_insights=performance_insights,
363
+ )
364
+
365
+ all_valid.extend(gen_valid)
366
+ all_failed.extend(gen_failed)
367
+ for r in gen_valid + gen_failed:
368
+ all_candidates_meta.append((r, _candidate_metadata(
369
+ r, generation=gen, authored_by=AUTHORED_BY_OFFSPRING_PROPOSAL
370
+ )))
371
+
372
+ pareto = compute_pareto_front([result_to_candidate(r) for r in all_valid], objectives)
373
+ best_per_generation.append(_snapshot_best(pareto, objectives))
374
+
375
+ if config.use_performance_insights and all_valid:
376
+ stats_str, pareto_str, _, _ = _performance_stats_str(all_valid, objectives, state.render)
377
+ performance_insights = call(
378
+ harness.performance_insights, stats_str, pareto_str, performance_insights
379
+ )
380
+
381
+ history.append(
382
+ {
383
+ "gen": gen,
384
+ "valid_count": len(gen_valid),
385
+ "failed_count": len(gen_failed),
386
+ "pareto_size": len(pareto),
387
+ }
388
+ )
389
+ _emit(on_event, {"kind": "generation_complete", "gen": gen, "pareto_size": len(pareto)})
390
+
391
+ result = _build_search_result(
392
+ all_valid,
393
+ all_candidates_meta,
394
+ objectives,
395
+ history,
396
+ best_per_generation=best_per_generation,
397
+ # Artifacts actually measured. A result served from the artifact
398
+ # cache cost nothing, so counting it here would overstate the bill
399
+ # and make a budget look breached when it was honoured exactly.
400
+ evaluations=(
401
+ int(getattr(config.evaluation_cache, 'misses', 0))
402
+ if config.evaluation_cache is not None
403
+ else sum(r.evaluation_attempted for r in (*all_valid, *all_failed))
404
+ ),
405
+ candidate_key=state.key_fn,
406
+ provider_usage=_provider_usage(harness, proposer_calls[0]),
407
+ )
408
+ log("")
409
+ log(
410
+ f"Summary: evaluations={result.evaluations}"
411
+ + (
412
+ f" (+{getattr(config.evaluation_cache, 'hits', 0)} served from cache)"
413
+ if getattr(config.evaluation_cache, "hits", 0)
414
+ else ""
415
+ )
416
+ + f", valid={len(all_valid)}, "
417
+ f"pareto={len(result.pareto_front)}, best={result.best.objectives}"
418
+ )
419
+ _emit(
420
+ on_event,
421
+ {
422
+ "kind": "search_complete",
423
+ "evaluations": result.evaluations,
424
+ "pareto_size": len(result.pareto_front),
425
+ "performance_insights": performance_insights,
426
+ "constraint_instruction": constraint_instruction,
427
+ },
428
+ )
429
+ return result
430
+
431
+
432
+ # ------------------------------------------------------------------
433
+ # Constraint-instruction learning (port of constraint_valid_count heuristic)
434
+ # ------------------------------------------------------------------
435
+
436
+ def _learn_constraint(
437
+ state: _RunState,
438
+ constraint_instruction: str,
439
+ last_round_failed: Sequence[CandidateResult],
440
+ all_failed: Sequence[CandidateResult],
441
+ batch_valid_count: int,
442
+ ) -> str:
443
+ """Create the constraint guide if absent, else update it when a batch improves.
444
+
445
+ Mirrors the original: only (re)write the guide when this batch produced more
446
+ valid candidates than the batch that last wrote it.
447
+ """
448
+ if not state.config.use_constraint_instruction:
449
+ return constraint_instruction
450
+ if not all_failed:
451
+ return constraint_instruction
452
+
453
+ create = not constraint_instruction
454
+ improved = batch_valid_count > state.constraint_valid_count
455
+ if not (create or improved):
456
+ return constraint_instruction
457
+
458
+ sampled = sample_failed_for_constraint(
459
+ last_round_failed, all_failed, state.config.max_failed_examples, state.rng
460
+ )
461
+ sampled_str = state.failed_str(sampled)
462
+ previous = None if create else constraint_instruction
463
+ new_ci = state.call(state.harness.constraint_instruction, sampled_str, previous)
464
+ if new_ci and new_ci != constraint_instruction:
465
+ state.constraint_valid_count = batch_valid_count
466
+ verb = "created" if create else "updated"
467
+ state.log(f" [constraint] guide {verb} (batch valid={batch_valid_count})")
468
+ return new_ci
469
+ return constraint_instruction
470
+
471
+
472
+ # ------------------------------------------------------------------
473
+ # Generation helpers
474
+ # ------------------------------------------------------------------
475
+
476
+ def _run_initial_generation(
477
+ *,
478
+ state: _RunState,
479
+ constraint_instruction: str,
480
+ performance_insights: str,
481
+ ) -> tuple:
482
+ cfg = state.config
483
+ gen_valid: List[CandidateResult] = []
484
+ gen_failed: List[CandidateResult] = []
485
+ last_round_failed: List[CandidateResult] = []
486
+
487
+ # ``max_regen_rounds`` means retries *after* one mandatory initial call.
488
+ for regen_round in range(cfg.max_regen_rounds + 1):
489
+ remaining = max(cfg.pop_size - len(gen_valid), 1)
490
+ n = min(cfg.candidates_per_batch, remaining)
491
+ if regen_round == 0:
492
+ configs = state.call(state.harness.generate_initial, n)
493
+ else:
494
+ configs = state.call(
495
+ state.harness.regenerate,
496
+ state.failed_str(last_round_failed),
497
+ n,
498
+ constraint_instruction,
499
+ performance_insights,
500
+ )
501
+ configs = state.dedup(configs)
502
+
503
+ valid_batch, failed_batch, _ = evaluate_batch(state.problem, configs, state.objectives, cache=state.config.evaluation_cache)
504
+ _emit_candidates(state.on_event, 1, regen_round, valid_batch, failed_batch)
505
+
506
+ if failed_batch:
507
+ _attach_failure_insights(state, failed_batch)
508
+ gen_failed.extend(failed_batch)
509
+ gen_valid.extend(valid_batch)
510
+ last_round_failed = failed_batch
511
+
512
+ constraint_instruction = _learn_constraint(
513
+ state, constraint_instruction, last_round_failed, gen_failed, len(valid_batch)
514
+ )
515
+
516
+ if len(gen_valid) >= cfg.pop_size:
517
+ break
518
+
519
+ # A harness may return more than requested. Every already-evaluated candidate
520
+ # remains in the ledger/archive even if the working population target was met.
521
+ return gen_valid, gen_failed, constraint_instruction
522
+
523
+
524
+ def _run_evolution_generation(
525
+ *,
526
+ state: _RunState,
527
+ gen: int,
528
+ prev_pareto: Sequence[Candidate],
529
+ constraint_instruction: str,
530
+ performance_insights: str,
531
+ ) -> tuple:
532
+ cfg = state.config
533
+ gen_valid: List[CandidateResult] = []
534
+ gen_failed: List[CandidateResult] = []
535
+
536
+ pareto_results = _pareto_as_results(prev_pareto)
537
+
538
+ if not prev_pareto:
539
+ configs = state.call(state.harness.generate_initial, cfg.pop_size)
540
+ else:
541
+ pareto_str = prettify_results(pareto_results[:5], state.objectives, render=state.render)
542
+ configs = state.call(
543
+ state.harness.generate_offspring,
544
+ pareto_str,
545
+ cfg.pop_size,
546
+ constraint_instruction,
547
+ performance_insights,
548
+ )
549
+ configs = state.dedup(configs)
550
+
551
+ valid_batch, failed_batch, _ = evaluate_batch(state.problem, configs, state.objectives, cache=state.config.evaluation_cache)
552
+ _emit_candidates(state.on_event, gen, 0, valid_batch, failed_batch)
553
+
554
+ if failed_batch:
555
+ _attach_failure_insights(state, failed_batch)
556
+ gen_failed.extend(failed_batch)
557
+ gen_valid.extend(valid_batch)
558
+ last_round_failed = failed_batch
559
+
560
+ regen_round = 0
561
+ while len(gen_valid) < cfg.pop_size and regen_round < cfg.max_regen_rounds:
562
+ if not last_round_failed:
563
+ break
564
+ failed_str = state.failed_str(last_round_failed)
565
+
566
+ if prev_pareto:
567
+ p_str = prettify_results(pareto_results[:3], state.objectives, render=state.render)
568
+ remaining = max(cfg.pop_size - len(gen_valid), 1)
569
+ configs = state.call(
570
+ state.harness.regenerate_offspring,
571
+ failed_str,
572
+ p_str,
573
+ min(cfg.candidates_per_batch, remaining),
574
+ constraint_instruction,
575
+ performance_insights,
576
+ )
577
+ else:
578
+ remaining = max(cfg.pop_size - len(gen_valid), 1)
579
+ configs = state.call(
580
+ state.harness.regenerate,
581
+ failed_str,
582
+ min(cfg.candidates_per_batch, remaining),
583
+ constraint_instruction,
584
+ performance_insights,
585
+ )
586
+ configs = state.dedup(configs)
587
+
588
+ valid_batch, failed_batch, _ = evaluate_batch(state.problem, configs, state.objectives, cache=state.config.evaluation_cache)
589
+ _emit_candidates(state.on_event, gen, regen_round + 1, valid_batch, failed_batch)
590
+
591
+ if failed_batch:
592
+ _attach_failure_insights(state, failed_batch)
593
+ gen_failed.extend(failed_batch)
594
+ gen_valid.extend(valid_batch)
595
+ last_round_failed = failed_batch
596
+
597
+ constraint_instruction = _learn_constraint(
598
+ state, constraint_instruction, last_round_failed, gen_failed, len(valid_batch)
599
+ )
600
+ regen_round += 1
601
+
602
+ return gen_valid, gen_failed, constraint_instruction
603
+
604
+
605
+ # ------------------------------------------------------------------
606
+ # Shared helpers
607
+ # ------------------------------------------------------------------
608
+
609
+ def _pareto_as_results(pareto: Sequence[Candidate]) -> List[CandidateResult]:
610
+ return [
611
+ CandidateResult(
612
+ configuration=c.configuration,
613
+ objectives=c.objectives,
614
+ is_valid=True,
615
+ evaluation_attempted=True,
616
+ )
617
+ for c in pareto
618
+ ]
619
+
620
+
621
+ # Who authored a candidate's configuration. Recorded per candidate so a caller
622
+ # can tell what the model actually produced from what it was given, without
623
+ # inferring it from position in the loop. Inferring authorship from position is
624
+ # how an arm gets labelled by the component that produced it rather than by the
625
+ # authority that decided it, which is a mistake this project has made twice.
626
+ AUTHORED_BY_CALLER_SEED = "caller_seed"
627
+ AUTHORED_BY_INITIAL_PROPOSAL = "proposer_initial"
628
+ AUTHORED_BY_OFFSPRING_PROPOSAL = "proposer_offspring"
629
+ AUTHORING_CALLS = (
630
+ AUTHORED_BY_CALLER_SEED,
631
+ AUTHORED_BY_INITIAL_PROPOSAL,
632
+ AUTHORED_BY_OFFSPRING_PROPOSAL,
633
+ )
634
+
635
+
636
+ def _provider_usage(harness: Any, proposer_calls: int) -> ProviderUsageSummary:
637
+ """Summarise what the run spent on the proposer.
638
+
639
+ ``calls`` is always counted here. Token and cost figures come from the
640
+ harness when it reports them and stay ``None`` when it does not -- ``None``
641
+ means "this proposer does not report it", which is deliberately not the
642
+ same value as zero. Conflating the two is how a run that never looked comes
643
+ to read like a run that measured nothing spent.
644
+ """
645
+
646
+ reported = getattr(harness, "usage", None)
647
+ figures: Dict[str, Any] = {}
648
+ if callable(reported):
649
+ try:
650
+ value = reported()
651
+ except Exception: # a harness defect must not lose the run's result
652
+ value = None
653
+ if isinstance(value, dict):
654
+ figures = value
655
+ def _figure(name: str) -> Optional[int]:
656
+ # Absent means unreported, which is not zero. A harness that reports a
657
+ # genuine zero passes 0 and it is preserved.
658
+ value = figures.get(name)
659
+ return None if value is None else int(value)
660
+
661
+ supplied = {
662
+ "input_tokens": _figure("input_tokens"),
663
+ "output_tokens": _figure("output_tokens"),
664
+ "cost_usd": figures.get("cost_usd"),
665
+ }
666
+ reporter = (
667
+ type(harness).__name__
668
+ if any(v is not None for v in supplied.values())
669
+ else None
670
+ )
671
+ return ProviderUsageSummary(
672
+ calls=proposer_calls,
673
+ model=figures.get("model"),
674
+ reported_by=reporter,
675
+ **supplied,
676
+ )
677
+
678
+
679
+ def _candidate_metadata(
680
+ result: CandidateResult, *, generation: int, authored_by: str
681
+ ) -> Dict[str, Any]:
682
+ """Preserve trace diagnostics and authorship in the public candidate ledger."""
683
+ if authored_by not in AUTHORING_CALLS:
684
+ raise ValueError(
685
+ f"authored_by must name an authoring call {AUTHORING_CALLS}, "
686
+ f"got {authored_by!r}"
687
+ )
688
+ metadata: Dict[str, Any] = {
689
+ "generation": generation,
690
+ "authored_by": authored_by,
691
+ "valid": result.is_valid,
692
+ "is_pareto": False,
693
+ "evaluation_attempted": result.evaluation_attempted,
694
+ }
695
+ if result.failure_phase:
696
+ metadata["failure_phase"] = result.failure_phase
697
+ if result.error_message:
698
+ metadata["error_message"] = result.error_message
699
+ if result.insight:
700
+ metadata["insight"] = result.insight
701
+ return metadata
702
+
703
+
704
+ def _attach_failure_insights(state: _RunState, failed: List[CandidateResult]) -> None:
705
+ if not state.config.use_failure_insights:
706
+ return
707
+ # A proposer may declare that it produces no insights. An uninformed
708
+ # baseline is the case that matters: one that synthesised guidance would
709
+ # not be uninformed, so its empty return is correct and not a fault.
710
+ if not getattr(state.harness, "provides_insights", True):
711
+ return
712
+ insights = state.call(state.harness.failure_insights, state.failed_str(failed), len(failed))
713
+ if isinstance(insights, list) and insights:
714
+ # Broadcast: some models collapse the per-candidate list to one item; reuse the
715
+ # last insight for any remaining failures so every failed candidate carries feedback
716
+ # (otherwise zip() silently dropped feedback for all but the first failure).
717
+ for i, r in enumerate(failed):
718
+ r.insight = str(insights[i]) if i < len(insights) else str(insights[-1])
719
+ if len(insights) != len(failed):
720
+ state.log(f"[agent_evolve] note: {len(insights)} insight(s) for "
721
+ f"{len(failed)} failures — broadcasting last to remainder")
722
+ else:
723
+ state.log("[agent_evolve] WARNING: failure_insights returned no usable list")
724
+
725
+
726
+ def _emit_candidates(
727
+ on_event: Optional[EventFn],
728
+ gen: Optional[int],
729
+ regen_round: int,
730
+ valid_batch: Sequence[CandidateResult],
731
+ failed_batch: Sequence[CandidateResult],
732
+ ) -> None:
733
+ if on_event is None:
734
+ return
735
+ for r in valid_batch:
736
+ on_event(
737
+ {
738
+ "kind": "candidate_result",
739
+ "gen": gen,
740
+ "regen_round": regen_round,
741
+ "valid": True,
742
+ "configuration": dict(r.configuration),
743
+ "evaluation_attempted": r.evaluation_attempted,
744
+ "objectives": dict(r.objectives),
745
+ }
746
+ )
747
+ for r in failed_batch:
748
+ on_event(
749
+ {
750
+ "kind": "candidate_result",
751
+ "gen": gen,
752
+ "regen_round": regen_round,
753
+ "valid": False,
754
+ "configuration": dict(r.configuration),
755
+ "evaluation_attempted": r.evaluation_attempted,
756
+ "failure_phase": r.failure_phase,
757
+ "error": r.error_message,
758
+ }
759
+ )
760
+
761
+
762
+ def _build_search_result(
763
+ all_valid: List[CandidateResult],
764
+ all_candidates_meta: List[tuple],
765
+ objectives: Sequence[ObjectiveSpec],
766
+ history: List[Dict[str, Any]],
767
+ *,
768
+ best_per_generation: Optional[List[Candidate]] = None,
769
+ evaluations: int = 0,
770
+ candidate_key: Callable[[Dict[str, Any]], str] = _default_candidate_key,
771
+ provider_usage: Optional[ProviderUsageSummary] = None,
772
+ ) -> SearchResult:
773
+ pareto_results = compute_pareto_front(
774
+ [result_to_candidate(r) for r in all_valid], objectives
775
+ )
776
+ pareto_configs = {candidate_key(c.configuration) for c in pareto_results}
777
+
778
+ all_candidates: List[Candidate] = []
779
+ for cr, meta in all_candidates_meta:
780
+ meta_copy = dict(meta)
781
+ if candidate_key(cr.configuration) in pareto_configs:
782
+ meta_copy["is_pareto"] = True
783
+ all_candidates.append(result_to_candidate(cr, meta_copy))
784
+
785
+ pareto_list = [
786
+ Candidate(configuration=c.configuration, objectives=c.objectives, metadata={"is_pareto": True})
787
+ for c in pareto_results
788
+ ]
789
+
790
+ best_candidate = select_minimax_rank(pareto_results, objectives)
791
+ if best_candidate is None:
792
+ best_candidate = Candidate(configuration={}, objectives={}, metadata={})
793
+
794
+ return SearchResult(
795
+ objectives=list(objectives),
796
+ best=best_candidate,
797
+ pareto_front=pareto_list,
798
+ all_candidates=all_candidates,
799
+ history=history,
800
+ best_per_generation=list(best_per_generation or []),
801
+ evaluations=evaluations,
802
+ provider_usage=provider_usage,
803
+ )