agentevolve-optimizer 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (414) hide show
  1. agent_evolve/__init__.py +722 -0
  2. agent_evolve/agentic.py +2800 -0
  3. agent_evolve/api.py +767 -0
  4. agent_evolve/application/__init__.py +1580 -0
  5. agent_evolve/application/action_allocation.py +744 -0
  6. agent_evolve/application/action_allocation_frame.py +347 -0
  7. agent_evolve/application/action_allocation_frame_commit.py +185 -0
  8. agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
  9. agent_evolve/application/action_allocation_frame_v3.py +338 -0
  10. agent_evolve/application/action_archive_value.py +497 -0
  11. agent_evolve/application/action_evidence_consistency.py +455 -0
  12. agent_evolve/application/action_forecast_partitioning.py +1471 -0
  13. agent_evolve/application/action_metric_projection.py +211 -0
  14. agent_evolve/application/action_role_value.py +680 -0
  15. agent_evolve/application/action_score_authorities.py +363 -0
  16. agent_evolve/application/action_structural_signature.py +116 -0
  17. agent_evolve/application/action_target_realization.py +402 -0
  18. agent_evolve/application/agentic_evolution.py +7734 -0
  19. agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
  20. agent_evolve/application/anchor_residual_identification.py +463 -0
  21. agent_evolve/application/archive_conditioned_action_target.py +208 -0
  22. agent_evolve/application/artifact_journal.py +246 -0
  23. agent_evolve/application/artifact_replay.py +347 -0
  24. agent_evolve/application/budgeted_optimizer.py +1828 -0
  25. agent_evolve/application/calibrated_campaign.py +485 -0
  26. agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
  27. agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
  28. agent_evolve/application/campaign_capacity_recourse.py +254 -0
  29. agent_evolve/application/campaign_contextual_outcomes.py +119 -0
  30. agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
  31. agent_evolve/application/campaign_evidence_registry.py +262 -0
  32. agent_evolve/application/campaign_execution.py +2537 -0
  33. agent_evolve/application/campaign_generation_audit.py +942 -0
  34. agent_evolve/application/campaign_learning.py +1812 -0
  35. agent_evolve/application/campaign_learning_runtime.py +1977 -0
  36. agent_evolve/application/campaign_search_phase.py +227 -0
  37. agent_evolve/application/campaign_selector_context_extension.py +220 -0
  38. agent_evolve/application/campaign_variation_envelope.py +649 -0
  39. agent_evolve/application/campaign_variation_trace.py +451 -0
  40. agent_evolve/application/candidate_archive_consequence.py +128 -0
  41. agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
  42. agent_evolve/application/composite_outcome_updater.py +145 -0
  43. agent_evolve/application/composition_portfolio_selection.py +363 -0
  44. agent_evolve/application/concurrent_stage.py +144 -0
  45. agent_evolve/application/contextual_action_allocation.py +181 -0
  46. agent_evolve/application/contextual_campaign_outcomes.py +267 -0
  47. agent_evolve/application/contextual_campaign_planning.py +1366 -0
  48. agent_evolve/application/contextual_delayed_credit.py +651 -0
  49. agent_evolve/application/contextual_search_controller.py +2374 -0
  50. agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
  51. agent_evolve/application/decision_metric_projection.py +112 -0
  52. agent_evolve/application/derived_action_semantics.py +129 -0
  53. agent_evolve/application/detailed_evaluation.py +449 -0
  54. agent_evolve/application/earned_lineage.py +1011 -0
  55. agent_evolve/application/effective_choice_audit.py +484 -0
  56. agent_evolve/application/empirical_consequence_calibration.py +908 -0
  57. agent_evolve/application/evaluation_accounting.py +325 -0
  58. agent_evolve/application/evaluation_cache.py +199 -0
  59. agent_evolve/application/evaluation_escrow.py +547 -0
  60. agent_evolve/application/evaluation_recourse.py +253 -0
  61. agent_evolve/application/event_recorder.py +151 -0
  62. agent_evolve/application/evolution_campaign.py +1840 -0
  63. agent_evolve/application/executable_hypothesis.py +323 -0
  64. agent_evolve/application/factorial_branch_pilot.py +772 -0
  65. agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
  66. agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
  67. agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
  68. agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
  69. agent_evolve/application/finite_action_selection.py +188 -0
  70. agent_evolve/application/finite_action_set.py +306 -0
  71. agent_evolve/application/finite_action_transition.py +537 -0
  72. agent_evolve/application/finite_variation_eligibility.py +296 -0
  73. agent_evolve/application/forecast_geometry_portfolio.py +799 -0
  74. agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
  75. agent_evolve/application/front_proximity_admission.py +311 -0
  76. agent_evolve/application/front_proximity_parent_basis.py +458 -0
  77. agent_evolve/application/frozen_hurdle_score.py +659 -0
  78. agent_evolve/application/g3_causal_screen.py +2257 -0
  79. agent_evolve/application/g3_causal_validation.py +1046 -0
  80. agent_evolve/application/g3_postseal_curation.py +818 -0
  81. agent_evolve/application/gated_agentic_generator.py +205 -0
  82. agent_evolve/application/generation_feedback.py +293 -0
  83. agent_evolve/application/generative_proposal_journal.py +185 -0
  84. agent_evolve/application/geometry_conditional_elasticity.py +453 -0
  85. agent_evolve/application/global_wave_action_allocation.py +1151 -0
  86. agent_evolve/application/head_mass_conditional_seat.py +268 -0
  87. agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
  88. agent_evolve/application/identifiable_reflection_learning.py +395 -0
  89. agent_evolve/application/identifiable_reflection_request.py +364 -0
  90. agent_evolve/application/in_memory_residual_archive.py +341 -0
  91. agent_evolve/application/insight_memory.py +1804 -0
  92. agent_evolve/application/live_runtime_manifest.py +758 -0
  93. agent_evolve/application/llm_task_queue.py +769 -0
  94. agent_evolve/application/matched_finite_action_block.py +409 -0
  95. agent_evolve/application/materialized_action_broker.py +2328 -0
  96. agent_evolve/application/materialized_action_constraints.py +83 -0
  97. agent_evolve/application/materialized_variation.py +211 -0
  98. agent_evolve/application/multi_option_evolution.py +1536 -0
  99. agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
  100. agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
  101. agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
  102. agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
  103. agent_evolve/application/outcome_relation.py +193 -0
  104. agent_evolve/application/paired_allocation_comparison.py +241 -0
  105. agent_evolve/application/paired_block_schedule.py +127 -0
  106. agent_evolve/application/parent_measurement.py +226 -0
  107. agent_evolve/application/pareto_archive.py +811 -0
  108. agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
  109. agent_evolve/application/portfolio_evolution.py +2950 -0
  110. agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
  111. agent_evolve/application/portfolio_memory_attribution.py +581 -0
  112. agent_evolve/application/portfolio_memory_dose.py +788 -0
  113. agent_evolve/application/portfolio_memory_matched_control.py +938 -0
  114. agent_evolve/application/portfolio_memory_transfer.py +297 -0
  115. agent_evolve/application/portfolio_optimization_memory.py +363 -0
  116. agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
  117. agent_evolve/application/portfolio_projection.py +335 -0
  118. agent_evolve/application/portfolio_recombination.py +2032 -0
  119. agent_evolve/application/post_evolution_reflection.py +834 -0
  120. agent_evolve/application/postcommit_rank_authority.py +245 -0
  121. agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
  122. agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
  123. agent_evolve/application/prequential_residual_exploration.py +343 -0
  124. agent_evolve/application/prequential_score_portfolio.py +954 -0
  125. agent_evolve/application/projections.py +292 -0
  126. agent_evolve/application/protected_action_committee.py +1027 -0
  127. agent_evolve/application/protected_branch_pilot.py +376 -0
  128. agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
  129. agent_evolve/application/provider_replay.py +910 -0
  130. agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
  131. agent_evolve/application/recombination_residual_expert.py +403 -0
  132. agent_evolve/application/reflection_workflow.py +571 -0
  133. agent_evolve/application/region_conditional_credit.py +911 -0
  134. agent_evolve/application/residual_campaign_runtime.py +531 -0
  135. agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
  136. agent_evolve/application/residual_headroom_ledger.py +1544 -0
  137. agent_evolve/application/residual_learning_transaction.py +396 -0
  138. agent_evolve/application/residual_portfolio_evolution.py +1228 -0
  139. agent_evolve/application/residual_reachability.py +749 -0
  140. agent_evolve/application/residual_stage_credit.py +499 -0
  141. agent_evolve/application/same_prefix_paired_audit.py +1580 -0
  142. agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
  143. agent_evolve/application/sequential_lineage_allocation.py +1017 -0
  144. agent_evolve/application/sequential_market_replay.py +1395 -0
  145. agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
  146. agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
  147. agent_evolve/application/single_score_action_allocation.py +299 -0
  148. agent_evolve/application/source_exposure_allocation.py +906 -0
  149. agent_evolve/application/staged_memory.py +210 -0
  150. agent_evolve/application/stratified_cold_start_allocation.py +732 -0
  151. agent_evolve/application/support_guarded_hurdle_score.py +549 -0
  152. agent_evolve/application/target_conditioned_action_forecast.py +595 -0
  153. agent_evolve/application/target_conditioned_campaign.py +566 -0
  154. agent_evolve/application/treatment_assignment.py +201 -0
  155. agent_evolve/application/trusted_objective_evidence.py +217 -0
  156. agent_evolve/application/two_stage_action_evolution.py +1131 -0
  157. agent_evolve/application/v8lite_allocation_policy.py +1083 -0
  158. agent_evolve/application/v9_candidate_policy.py +1303 -0
  159. agent_evolve/bootstrap.py +108 -0
  160. agent_evolve/campaign_presets.py +517 -0
  161. agent_evolve/campaign_profiles.py +452 -0
  162. agent_evolve/campaign_variation_topology.py +288 -0
  163. agent_evolve/campaign_workload.py +950 -0
  164. agent_evolve/cli.py +797 -0
  165. agent_evolve/contract.py +241 -0
  166. agent_evolve/core/__init__.py +91 -0
  167. agent_evolve/core/action_semantics.py +411 -0
  168. agent_evolve/core/authored.py +105 -0
  169. agent_evolve/core/formatting.py +286 -0
  170. agent_evolve/core/optimization_semantics.py +324 -0
  171. agent_evolve/core/problem.py +167 -0
  172. agent_evolve/core/results.py +323 -0
  173. agent_evolve/core/stats.py +70 -0
  174. agent_evolve/core/telemetry.py +100 -0
  175. agent_evolve/domain/__init__.py +89 -0
  176. agent_evolve/domain/artifact.py +162 -0
  177. agent_evolve/domain/durable_text.py +68 -0
  178. agent_evolve/domain/event.py +1454 -0
  179. agent_evolve/domain/finite_action_set.py +426 -0
  180. agent_evolve/domain/finite_variation.py +526 -0
  181. agent_evolve/domain/generative_emission.py +559 -0
  182. agent_evolve/domain/ids.py +163 -0
  183. agent_evolve/domain/inline_text.py +106 -0
  184. agent_evolve/domain/insight.py +27 -0
  185. agent_evolve/domain/lineage.py +737 -0
  186. agent_evolve/domain/llm_task_queue.py +960 -0
  187. agent_evolve/domain/outcome.py +96 -0
  188. agent_evolve/domain/patch.py +854 -0
  189. agent_evolve/domain/typed_json.py +542 -0
  190. agent_evolve/domain/variation_space.py +158 -0
  191. agent_evolve/driver.py +1014 -0
  192. agent_evolve/harness/__init__.py +29 -0
  193. agent_evolve/harness/base.py +242 -0
  194. agent_evolve/harness/directives.py +163 -0
  195. agent_evolve/harness/generative_seal.py +479 -0
  196. agent_evolve/harness/registry.py +41 -0
  197. agent_evolve/infrastructure/__init__.py +39 -0
  198. agent_evolve/infrastructure/artifacts/__init__.py +6 -0
  199. agent_evolve/infrastructure/artifacts/_verification.py +67 -0
  200. agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
  201. agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
  202. agent_evolve/infrastructure/asyncio_runtime.py +109 -0
  203. agent_evolve/infrastructure/authored_runtime.py +188 -0
  204. agent_evolve/infrastructure/authored_worker.py +171 -0
  205. agent_evolve/infrastructure/clock.py +53 -0
  206. agent_evolve/infrastructure/events/__init__.py +6 -0
  207. agent_evolve/infrastructure/events/_validation.py +89 -0
  208. agent_evolve/infrastructure/events/in_memory.py +56 -0
  209. agent_evolve/infrastructure/events/jsonl.py +193 -0
  210. agent_evolve/infrastructure/exception_provenance.py +215 -0
  211. agent_evolve/infrastructure/ids.py +118 -0
  212. agent_evolve/infrastructure/lineage_codec.py +1836 -0
  213. agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
  214. agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
  215. agent_evolve/infrastructure/resource_lease.py +370 -0
  216. agent_evolve/infrastructure/sanitization/__init__.py +8 -0
  217. agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
  218. agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
  219. agent_evolve/infrastructure/stream_liveness.py +383 -0
  220. agent_evolve/infrastructure/subprocess_boundary.py +136 -0
  221. agent_evolve/integrations/__init__.py +1 -0
  222. agent_evolve/integrations/botorch/__init__.py +28 -0
  223. agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
  224. agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
  225. agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
  226. agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
  227. agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
  228. agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
  229. agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
  230. agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
  231. agent_evolve/integrations/completion.py +242 -0
  232. agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
  233. agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
  234. agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
  235. agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
  236. agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
  237. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
  238. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
  239. agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
  240. agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
  241. agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
  242. agent_evolve/integrations/pydantic_ai/harness.py +159 -0
  243. agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
  244. agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
  245. agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
  246. agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
  247. agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
  248. agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
  249. agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
  250. agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
  251. agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
  252. agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
  253. agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
  254. agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
  255. agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
  256. agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
  257. agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
  258. agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
  259. agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
  260. agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
  261. agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
  262. agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
  263. agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
  264. agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
  265. agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
  266. agent_evolve/integrations/pymoo_adapter.py +242 -0
  267. agent_evolve/policies/__init__.py +17 -0
  268. agent_evolve/policies/check.py +469 -0
  269. agent_evolve/policies/emit_scaffold.py +451 -0
  270. agent_evolve/policies/feedback/__init__.py +37 -0
  271. agent_evolve/policies/feedback/held_out_asn.py +1325 -0
  272. agent_evolve/policies/genetic.py +607 -0
  273. agent_evolve/policies/llm_backoff.py +183 -0
  274. agent_evolve/policies/llm_chooser.py +226 -0
  275. agent_evolve/policies/llm_generator.py +1760 -0
  276. agent_evolve/policies/llm_init.py +267 -0
  277. agent_evolve/policies/llm_operator.py +109 -0
  278. agent_evolve/policies/llm_prior.py +194 -0
  279. agent_evolve/policies/llm_surrogate.py +334 -0
  280. agent_evolve/policies/measurement_evidence.py +704 -0
  281. agent_evolve/policies/memory/__init__.py +223 -0
  282. agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
  283. agent_evolve/policies/memory/compatibility_matching.py +593 -0
  284. agent_evolve/policies/memory/global_falsification.py +1841 -0
  285. agent_evolve/policies/memory/prompt_shape.py +503 -0
  286. agent_evolve/policies/memory/randomized_subset.py +714 -0
  287. agent_evolve/policies/memory/staged_causal.py +1270 -0
  288. agent_evolve/policies/memory/treatment_compliance.py +759 -0
  289. agent_evolve/policies/objective_resolution/__init__.py +17 -0
  290. agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
  291. agent_evolve/policies/operator_portfolio.py +407 -0
  292. agent_evolve/policies/reguidance.py +1133 -0
  293. agent_evolve/policies/reward/__init__.py +83 -0
  294. agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
  295. agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
  296. agent_evolve/policies/reward/affine_hypervolume.py +490 -0
  297. agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
  298. agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
  299. agent_evolve/policies/reward/frozen_archive.py +360 -0
  300. agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
  301. agent_evolve/policies/search_state.py +208 -0
  302. agent_evolve/policies/selection/__init__.py +345 -0
  303. agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
  304. agent_evolve/policies/selection/affine_frontier_context.py +330 -0
  305. agent_evolve/policies/selection/affine_frontier_target.py +473 -0
  306. agent_evolve/policies/selection/archive_elite.py +1346 -0
  307. agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
  308. agent_evolve/policies/selection/calibrated_slate.py +1394 -0
  309. agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
  310. agent_evolve/policies/selection/common_candidate_pool.py +685 -0
  311. agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
  312. agent_evolve/policies/selection/disjoint_pairs.py +479 -0
  313. agent_evolve/policies/selection/elite_explorer.py +719 -0
  314. agent_evolve/policies/selection/finite_action.py +187 -0
  315. agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
  316. agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
  317. agent_evolve/policies/selection/forecast_calibration.py +922 -0
  318. agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
  319. agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
  320. agent_evolve/policies/selection/full_support_slate.py +91 -0
  321. agent_evolve/policies/selection/meaningful_direction.py +240 -0
  322. agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
  323. agent_evolve/policies/selection/model_anchored_slate.py +826 -0
  324. agent_evolve/policies/selection/phenotype_recourse.py +979 -0
  325. agent_evolve/policies/selection/proposal_support.py +368 -0
  326. agent_evolve/policies/selection/random_portfolio.py +254 -0
  327. agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
  328. agent_evolve/policies/selection/residual_frontier.py +463 -0
  329. agent_evolve/policies/selection/residual_frontier_target.py +605 -0
  330. agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
  331. agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
  332. agent_evolve/policies/selection/target_conditioned_features.py +812 -0
  333. agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
  334. agent_evolve/policies/selection/task_keyed_palette.py +906 -0
  335. agent_evolve/policies/semantics.py +147 -0
  336. agent_evolve/policies/structure.py +362 -0
  337. agent_evolve/policies/structured_output_budget.py +62 -0
  338. agent_evolve/policies/surrogate.py +696 -0
  339. agent_evolve/policies/variation/__init__.py +1 -0
  340. agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
  341. agent_evolve/policies/variation/crossover_inheritance.py +575 -0
  342. agent_evolve/policies/variation/disjoint_recombination.py +611 -0
  343. agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
  344. agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
  345. agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
  346. agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
  347. agent_evolve/policies/variation/typed_patch.py +1981 -0
  348. agent_evolve/policies/weighted_prior.py +394 -0
  349. agent_evolve/ports/__init__.py +383 -0
  350. agent_evolve/ports/action_allocation.py +733 -0
  351. agent_evolve/ports/action_allocation_frame.py +1153 -0
  352. agent_evolve/ports/action_allocation_frame_commit.py +294 -0
  353. agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
  354. agent_evolve/ports/action_allocation_frame_v3.py +995 -0
  355. agent_evolve/ports/action_forecast.py +1568 -0
  356. agent_evolve/ports/action_metric_projection.py +165 -0
  357. agent_evolve/ports/agentic_generator.py +1561 -0
  358. agent_evolve/ports/archive_context.py +136 -0
  359. agent_evolve/ports/artifact_sanitizer.py +44 -0
  360. agent_evolve/ports/artifact_store.py +225 -0
  361. agent_evolve/ports/clock.py +13 -0
  362. agent_evolve/ports/contextual_search_allocation.py +827 -0
  363. agent_evolve/ports/decision_metric_projection.py +258 -0
  364. agent_evolve/ports/event_store.py +55 -0
  365. agent_evolve/ports/executable_hypothesis.py +557 -0
  366. agent_evolve/ports/finite_acquisition.py +377 -0
  367. agent_evolve/ports/finite_acquisition_batch.py +296 -0
  368. agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
  369. agent_evolve/ports/finite_acquisition_json.py +247 -0
  370. agent_evolve/ports/finite_acquisition_space.py +168 -0
  371. agent_evolve/ports/finite_action_selection.py +348 -0
  372. agent_evolve/ports/finite_action_set.py +256 -0
  373. agent_evolve/ports/frontier_target.py +396 -0
  374. agent_evolve/ports/generation_failure.py +43 -0
  375. agent_evolve/ports/hard_feasibility.py +233 -0
  376. agent_evolve/ports/id_factory.py +34 -0
  377. agent_evolve/ports/llm_task_queue.py +93 -0
  378. agent_evolve/ports/objective_resolution.py +419 -0
  379. agent_evolve/ports/paired_allocation_comparison.py +401 -0
  380. agent_evolve/ports/paired_block_schedule.py +475 -0
  381. agent_evolve/ports/parent_measurement.py +336 -0
  382. agent_evolve/ports/portfolio_memory_dose.py +643 -0
  383. agent_evolve/ports/portfolio_selection.py +3169 -0
  384. agent_evolve/ports/postcommit_rank_authority.py +467 -0
  385. agent_evolve/ports/presented_action_evidence.py +794 -0
  386. agent_evolve/ports/resource_lease.py +162 -0
  387. agent_evolve/ports/structured_generator.py +734 -0
  388. agent_evolve/ports/structured_output_budget.py +120 -0
  389. agent_evolve/ports/subprocess_boundary.py +138 -0
  390. agent_evolve/ports/treatment_assignment.py +466 -0
  391. agent_evolve/ports/variation_catalog.py +76 -0
  392. agent_evolve/ports/variation_source.py +226 -0
  393. agent_evolve/proposal_mode.py +157 -0
  394. agent_evolve/proposers/__init__.py +10 -0
  395. agent_evolve/proposers/random_proposer.py +188 -0
  396. agent_evolve/provider_accounting.py +163 -0
  397. agent_evolve/py.typed +0 -0
  398. agent_evolve/reference_method.py +1570 -0
  399. agent_evolve/session/__init__.py +11 -0
  400. agent_evolve/session/authorship.py +864 -0
  401. agent_evolve/session/evaluate.py +236 -0
  402. agent_evolve/session/fidelity.py +237 -0
  403. agent_evolve/session/genetic_loop.py +742 -0
  404. agent_evolve/session/loop.py +803 -0
  405. agent_evolve/session/screening.py +671 -0
  406. agent_evolve/settings.py +376 -0
  407. agent_evolve/workload_kit.py +368 -0
  408. agent_evolve/workload_prompt.py +398 -0
  409. agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
  410. agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
  411. agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
  412. agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
  413. agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
  414. agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
agent_evolve/api.py ADDED
@@ -0,0 +1,767 @@
1
+ """The public entry point: ``optimize(problem, budget=...) -> SearchResult``.
2
+
3
+ One required argument and one number a caller actually knows -- how many
4
+ evaluations they can afford. Everything else has a defensible default.
5
+
6
+ ``proposer`` is the one option worth understanding:
7
+
8
+ ``"random"`` samples the candidate schema. No credentials, no network, no
9
+ cost. It is also the control arm: a model that cannot beat it on
10
+ your problem is not earning its price. ``agent_evolve check``
11
+ runs exactly that comparison.
12
+ ``"llm"`` the model-driven proposer.
13
+ ``"auto"`` ``llm`` when a provider credential is present, otherwise
14
+ ``random``, said out loud through ``on_progress`` rather than
15
+ silently.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import dataclasses
21
+ from typing import Any, Callable, Literal, Optional
22
+
23
+ from agent_evolve import bootstrap
24
+ from agent_evolve.contract import as_problem
25
+ from agent_evolve.core.formatting import format_search_space_description
26
+ from agent_evolve.core.results import ProviderUsageSummary, SearchResult
27
+ from agent_evolve.harness.base import HarnessContext, LLMConfig
28
+ from agent_evolve.harness.directives import DefaultDirectives
29
+ from agent_evolve.harness.registry import harness_registry
30
+ from agent_evolve.session.evaluate import EvaluationCache
31
+ from agent_evolve.session.loop import LoopConfig, run_evolution_loop
32
+ from agent_evolve.settings import AgentEvolveSettings, credentials_present
33
+
34
+ __all__ = ["optimize"]
35
+
36
+ Proposer = Literal["auto", "llm", "random"]
37
+
38
+ #: Who turns a screen's evidence into a sampling prior. The llm forms fall
39
+ #: back to their rule comparator -- out loud -- when no model call is possible.
40
+ _PRIORS = ("rule", "rule-weighted", "llm", "llm-weighted",
41
+ "llm-weighted-committed")
42
+
43
+ #: Candidates proposed per generation. The budget decides how many generations
44
+ #: that buys, so the caller states the number they know and not this one.
45
+ _BATCH = 8
46
+
47
+ #: The largest budget any SEALED row was measured at. At or below it the
48
+ #: genetic sizing is a control arm and may not move; above it there is nothing
49
+ #: to hold still. A constant rather than a knob, because a knob is a way to
50
+ #: move a sealed arm by accident.
51
+ _SEALED_BUDGET_CEILING = 384
52
+
53
+
54
+ def _genetic_sizing(budget: int) -> tuple[int, int]:
55
+ """``(population, offspring per generation)``, from the budget alone.
56
+
57
+ Sized from the BUDGET, not from how many seeds the caller happened to
58
+ supply: one seed would otherwise give a population of two, which cannot
59
+ recombine into anything its parents do not already contain.
60
+
61
+ Two regimes, and the split is measured rather than tasteful.
62
+
63
+ Up to ``_SEALED_BUDGET_CEILING`` the population is the old expression --
64
+ capped at twelve, floored at four -- written here as the literal branch so
65
+ that every budget a sealed row was measured at runs the arithmetic it was
66
+ measured with. The byte fossil and the sizing table both pin it.
67
+
68
+ Above that ceiling the cap was a THROTTLE. At B = 2000 twelve members
69
+ converge long before the budget is gone: late generations propose
70
+ recombinations the population already holds, those hit the evaluation
71
+ cache, and the generation count -- which is a cap on generations, not on
72
+ charges -- runs out with the budget unspent. Measured, six of six cheap
73
+ cells spent 969 to 1212 of 2000 charges while the uniform comparator spent
74
+ 1696 to 1842, so the matched-budget comparison was decided by how much each
75
+ arm could spend and not by how well it was guided; on recall per
76
+ EVALUATION the same cells read at parity or better. So the population
77
+ grows with the budget (one member per 32 charges, floored at the old cap of
78
+ twelve and ceilinged at 64, where the per-generation selection cost starts
79
+ to be the thing being paid for) and the offspring count follows it. The
80
+ generations formula is untouched: it divides by the offspring count and
81
+ adapts on its own.
82
+ """
83
+
84
+ if int(budget) > _SEALED_BUDGET_CEILING:
85
+ pop = min(64, max(12, int(budget) // 32))
86
+ return pop, pop - 2
87
+ pop = max(4, min(budget // 4, 12))
88
+ return pop, max(2, pop - 2)
89
+
90
+
91
+ def _describe(problem: Any) -> str:
92
+ """Build the search-space description shown to the proposer."""
93
+ problem_description = None
94
+ if hasattr(problem, "search_space_description"):
95
+ problem_description = problem.search_space_description()
96
+
97
+ config_schema = getattr(problem, "config_schema", None)
98
+ candidate_model = getattr(problem, "candidate_model", None)
99
+ if config_schema is None and candidate_model is not None:
100
+ try:
101
+ config_schema = candidate_model.model_json_schema()
102
+ except Exception:
103
+ config_schema = None
104
+
105
+ return format_search_space_description(
106
+ list(problem.objectives),
107
+ config_schema=config_schema,
108
+ example_config=getattr(problem, "example_config", None),
109
+ constraints=None, # constraints flow through HarnessContext
110
+ problem_description=problem_description,
111
+ )
112
+
113
+
114
+ def _resolve_proposer(proposer: str, announce: Callable[[str], None]) -> str:
115
+ if proposer != "auto" and proposer not in ("llm", "random"):
116
+ # Any harness registered by name is also a proposer. This is how an
117
+ # out-of-tree integration is selected, without a second parameter that
118
+ # means almost the same thing.
119
+ bootstrap.load_integrations()
120
+ if proposer not in harness_registry.ids():
121
+ raise ValueError(
122
+ f"proposer must be 'auto', 'llm', 'random' or a registered "
123
+ f"harness id, got {proposer!r}. Registered: "
124
+ f"{sorted(harness_registry.ids())}"
125
+ )
126
+ if proposer != "auto":
127
+ return proposer
128
+ if credentials_present():
129
+ return "llm"
130
+ announce(
131
+ "No provider credential found, so candidates are being proposed at "
132
+ "random. This costs nothing, and it is the baseline a model has to "
133
+ "beat. Pass proposer='llm' with a credential to use a model."
134
+ )
135
+ return "random"
136
+
137
+
138
+ def _priced_usage(
139
+ route: str, input_tokens: int, output_tokens: int
140
+ ) -> tuple[Optional[str], str]:
141
+ """``(cost_usd, reported_by)`` for a run's token counts.
142
+
143
+ The two halves of a cost figure come from different places and the reporter
144
+ string says which is which: the token counts are the provider's own, the
145
+ per-million prices are this package's published table -- the same numbers
146
+ the CLI echoes before it spends anything. Keeping the provenance in the
147
+ field means nobody has to guess later whether a dollar figure was billed or
148
+ computed.
149
+
150
+ An unpriced route returns ``None``. A cost that cannot be derived is
151
+ reported as unknown rather than as zero, because zero reads as "nothing was
152
+ spent" when it means "nobody looked".
153
+ """
154
+ from decimal import Decimal
155
+
156
+ from agent_evolve.settings import model_price
157
+
158
+ measured = "openrouter response usage"
159
+ price = model_price(route)
160
+ if price is None:
161
+ return None, measured
162
+ per_m_in, per_m_out = price
163
+ million = Decimal(1_000_000)
164
+ cost = (
165
+ Decimal(str(per_m_in)) * Decimal(input_tokens) / million
166
+ + Decimal(str(per_m_out)) * Decimal(output_tokens) / million
167
+ )
168
+ return (
169
+ str(cost.quantize(Decimal("0.000001"))),
170
+ f"{measured}; cost derived from the package's published price table",
171
+ )
172
+
173
+
174
+ def _build_harness(kind: str, seed: Optional[int], settings: AgentEvolveSettings) -> Any:
175
+ bootstrap.load_integrations()
176
+ if kind == "random":
177
+ harness_id = "random"
178
+ elif kind == "llm":
179
+ harness_id = settings.harness
180
+ else:
181
+ harness_id = kind # an explicitly named registered harness
182
+ missing = bootstrap.requirement_failure(harness_id)
183
+ if missing is not None:
184
+ # Fail here, naming the fix, rather than deep inside the first model
185
+ # call with a bare ModuleNotFoundError.
186
+ raise RuntimeError(
187
+ f"the {harness_id!r} proposer {missing}. "
188
+ "Or run with proposer='random', which needs nothing."
189
+ )
190
+ try:
191
+ return harness_registry.create(harness_id, seed=seed)
192
+ except KeyError as error:
193
+ raise KeyError(bootstrap.explain_missing_harness(harness_id)) from error
194
+
195
+
196
+ def _resolve_strategy(strategy: str, has_seeds: bool, announce) -> str:
197
+ """Pick the search loop. ``auto`` prefers genetics wherever they are usable.
198
+
199
+ The authoring loop asks a model to write whole configurations from a text
200
+ rendering of the Pareto front. Measured against uniform random sampling on
201
+ every genome length tried, that loses (-0.086 to -0.531 excess capture)
202
+ while recombination over a population wins (+0.0042 to +0.1798). So
203
+ ``genetic`` is preferred wherever it can run, which is wherever the problem
204
+ supplies at least one seed to give a candidate its shape.
205
+ """
206
+
207
+ if strategy not in ("auto", "genetic", "authoring"):
208
+ raise ValueError(
209
+ f"strategy must be 'auto', 'genetic' or 'authoring', got {strategy!r}"
210
+ )
211
+ if strategy != "auto":
212
+ return strategy
213
+ if has_seeds:
214
+ return "genetic"
215
+ announce(
216
+ "No seeds were supplied, so candidates are authored from scratch rather "
217
+ "than recombined. Give Problem.seeds() one configuration to use the "
218
+ "genetic loop, which measures better against random search."
219
+ )
220
+ return "authoring"
221
+
222
+
223
+ def _llm_refusal_message(*, extra_missing: bool) -> str:
224
+ """The explicit-llm refusal, naming every way out that applies.
225
+
226
+ On a core install the stranger who asks for a model is missing TWO
227
+ things, and the fix a message names first should be the one they hit
228
+ first: the optional dependencies, then the credential. On an install
229
+ that already has the extra, naming it would be noise. The CI stranger
230
+ job holds the extra-missing rendering to actually naming the extra.
231
+ """
232
+
233
+ fix = (
234
+ "Install the model path's optional dependencies with: pip install "
235
+ "'agentevolve-optimizer[llm]'. Then set OPENROUTER_API_KEY (or "
236
+ "AGENTEVOLVE_DOTENV naming a file that does)"
237
+ if extra_missing else
238
+ "Set OPENROUTER_API_KEY (or AGENTEVOLVE_DOTENV naming a file that "
239
+ "does)"
240
+ )
241
+ return (
242
+ "proposer='llm' was asked for by name, but no provider credential "
243
+ f"is configured, so no model can be called. {fix}, or run with "
244
+ "proposer='random', which needs nothing -- or proposer='auto', "
245
+ "which chooses it out loud."
246
+ )
247
+
248
+
249
+ def _check_structure_budget(structure_budget: int, budget: int) -> None:
250
+ """The screen is charged against the search it informs, so it must fit."""
251
+
252
+ if structure_budget >= budget:
253
+ raise ValueError(
254
+ f"structure_budget ({structure_budget}) must leave room inside the "
255
+ f"budget ({budget}): the screen is charged against the same budget "
256
+ "as the search it informs"
257
+ )
258
+
259
+
260
+ def _resolve_guidance(
261
+ prior: Any,
262
+ structure_budget: Any,
263
+ *,
264
+ budget: int,
265
+ model_calls: bool,
266
+ announce: Callable[[str], None],
267
+ ) -> tuple[str, int]:
268
+ """Turn the ``"auto"`` sentinels into the stack the measurements bought.
269
+
270
+ Two rules, and which one applies is decided by whether a model call is
271
+ actually possible -- not by what the caller hoped for.
272
+
273
+ Without a model call the sentinels resolve to ``"rule"`` and ``0``, which
274
+ are the literal pre-sentinel defaults: the credential-free path draws the
275
+ same candidates in the same order, and the fossil stream cannot move.
276
+
277
+ With one, the screen is sized from the budget and ``prior`` becomes
278
+ ``"llm-weighted"`` exactly when that screen will run. Below 48 evaluations
279
+ both stay off: the six-arm ablation screened at 15 evaluations of 96, at a
280
+ small budget that share buys less than the initialization seam alone (the
281
+ measured winner there), and the prior seat only ever acts on a screen's
282
+ evidence.
283
+ """
284
+
285
+ if not model_calls:
286
+ return ("rule" if prior == "auto" else prior,
287
+ 0 if structure_budget == "auto" else structure_budget)
288
+ if structure_budget == "auto":
289
+ structure_budget = 0 if budget < 48 else min(16, max(8, budget // 6))
290
+ if structure_budget:
291
+ announce(
292
+ f"structure_budget={structure_budget} by default at budget "
293
+ f"{budget}: the six-arm ablation screened at 15 evaluations of "
294
+ "96, and the screen is charged against the same budget. Below "
295
+ "48 it is skipped. Pass structure_budget=0 to skip it here."
296
+ )
297
+ if prior == "auto":
298
+ # The prior seat only acts on a screen's evidence, so the model form
299
+ # is bought exactly when a screen will run. Announcing a model prior
300
+ # beside structure_budget=0 would be a promise the run never cashes.
301
+ if structure_budget:
302
+ prior = "llm-weighted"
303
+ announce(
304
+ "prior='llm-weighted' by default on a model run: the model "
305
+ "reads the crossed screen and the screen's own statistics "
306
+ "carry the weights (the six-arm ablation's guidance arm). "
307
+ "Pass prior='rule' for the credential-free comparator."
308
+ )
309
+ else:
310
+ prior = "rule"
311
+ return prior, structure_budget
312
+
313
+
314
+ def optimize(
315
+ problem: Any,
316
+ *,
317
+ budget: int = 40,
318
+ model: Optional[str] = None,
319
+ proposer: str = "auto",
320
+ strategy: str = "auto",
321
+ seed: Optional[int] = None,
322
+ seal: Optional[str] = None,
323
+ on_progress: Optional[Callable[[str], None]] = None,
324
+ structure_budget: int | str = "auto",
325
+ prior: str = "auto",
326
+ chooser: str = "off",
327
+ effort: Optional[str] = None,
328
+ journal: Any = None,
329
+ authorship: Any = "auto",
330
+ ) -> SearchResult:
331
+ """Optimize *problem* within *budget* evaluations.
332
+
333
+ *budget* counts artifacts measured, which is the expensive thing and the
334
+ only sizing number the caller supplies. The problem's seeds are evaluated
335
+ before anything is proposed, so the result always answers "did this beat
336
+ what I already had".
337
+
338
+ *structure_budget* spends that many evaluations -- charged against the same
339
+ *budget* -- on a crossed screen before the population is built; *prior*
340
+ names who turns the screen into a sampling prior: the credential-free
341
+ ``"rule"`` or ``"rule-weighted"``, or their model-backed forms ``"llm"`` /
342
+ ``"llm-weighted"``, which fall back to the rule comparator, out loud, when
343
+ no model call is possible. Both default to ``"auto"``, which resolves
344
+ against what the run can actually do: without a model call, to ``"rule"``
345
+ and ``0`` -- the literal pre-sentinel defaults, so the credential-free path
346
+ stays byte-identical -- and with one, to ``"llm-weighted"`` and a screen
347
+ sized from the budget, announced through *on_progress* rather than picked
348
+ silently. ``"llm-weighted-committed"`` is the tuning-round variant under
349
+ measurement: ``"llm-weighted"`` with the prompt's leave-a-locus-free
350
+ caution swapped for evidence-proportional commitment.
351
+
352
+ *chooser* names who picks parents and cut points inside a generation, and
353
+ defaults to ``"off"``. ``"llm"`` buys the per-offspring chooser, which is
354
+ the one mechanism here that has never earned its price: ten sealed null
355
+ verdicts, Theta(offspring) model calls rather than one, and 61% of the
356
+ six-arm ablation's whole ledger consumed for 0.94x the speed of doing
357
+ nothing. ``"off"`` runs the random control it never beat. It needs a run
358
+ that makes model calls; asking for it on a run that cannot is refused
359
+ rather than ignored.
360
+
361
+ *effort* pins the model's reasoning effort on every completion call, and
362
+ *journal* (a callable, or a path to a JSONL file) receives one record per
363
+ completed model call -- model served plus token usage -- so a run's spend
364
+ is verifiable from its own artifacts. Both belong to the genetic strategy.
365
+
366
+ *seal* names a file to write the run's proposal journal to: one chained,
367
+ self-authenticating line per model call, holding the exact configuration
368
+ that was emitted, the digest of the prompt that produced it, the digest of
369
+ the schema it was drawn from, and the verdict ``validate`` returned. The run
370
+ then replays from that file with no provider and no credential. Pass it when
371
+ the result has to be checkable by someone who was not there.
372
+ """
373
+ if not isinstance(budget, int) or isinstance(budget, bool) or budget < 1:
374
+ raise ValueError(f"budget must be a positive integer, got {budget!r}")
375
+ if structure_budget != "auto":
376
+ if (not isinstance(structure_budget, int)
377
+ or isinstance(structure_budget, bool) or structure_budget < 0):
378
+ raise ValueError(
379
+ f"structure_budget must be 'auto' or a non-negative integer, "
380
+ f"got {structure_budget!r}"
381
+ )
382
+ _check_structure_budget(structure_budget, budget)
383
+ if prior != "auto" and prior not in _PRIORS:
384
+ raise ValueError(
385
+ f"prior must be 'auto' or one of {sorted(_PRIORS)}, got {prior!r}")
386
+ if chooser not in ("off", "llm"):
387
+ raise ValueError(f"chooser must be 'off' or 'llm', got {chooser!r}")
388
+ if effort is not None and not isinstance(effort, str):
389
+ raise ValueError(
390
+ f"effort must be a provider effort level as a string, got {effort!r}"
391
+ )
392
+ from agent_evolve.session.authorship import AuthorshipConfig
393
+ if isinstance(authorship, AuthorshipConfig):
394
+ authorship_config = authorship
395
+ elif authorship == "auto":
396
+ # Resolved on the genetic branch: the model-authored surrogate is ON
397
+ # when a model call is possible (the sealed S1 luna-clear row held),
398
+ # off otherwise. The evidence-backed default, not the hopeful one.
399
+ authorship_config = None
400
+ elif isinstance(authorship, str):
401
+ authorship_config = AuthorshipConfig.preset(authorship)
402
+ else:
403
+ raise ValueError(
404
+ "authorship must be an AuthorshipConfig or a preset name, got "
405
+ f"{authorship!r}"
406
+ )
407
+
408
+ bound = as_problem(problem)
409
+ announce = on_progress or (lambda _message: None)
410
+ settings = AgentEvolveSettings.from_env()
411
+
412
+ # Arguments are validated before any branching. A caller who passes a
413
+ # nonsense proposer must be told so whichever loop ends up running --
414
+ # skipping validation on one path is how an invalid argument becomes a
415
+ # silent no-op.
416
+ kind = _resolve_proposer(proposer, announce)
417
+ if chooser == "llm" and kind != "llm":
418
+ # A chooser that cannot call a model is a chooser that never chooses,
419
+ # and the run would look exactly like the one that never asked for it.
420
+ raise ValueError(
421
+ f"chooser='llm' asks a model to pick parents and cut points, and "
422
+ f"this run resolved to the {kind!r} proposer, which makes no model "
423
+ "call. Pass proposer='llm' with a provider credential, or drop "
424
+ "chooser= to keep the random control."
425
+ )
426
+
427
+ seeds = tuple(dict(c) for c in bound.seeds())
428
+ chosen = _resolve_strategy(strategy, bool(seeds), announce)
429
+ if chosen == "genetic" and seal is not None:
430
+ # The seal journal holds generative proposals; the genetic loop's model
431
+ # calls are operator choices, which that format cannot represent. A
432
+ # journal the caller asked for and never got would be a silent no-op,
433
+ # so refuse loudly and name the two ways out.
434
+ raise ValueError(
435
+ "seal journaling is not supported by the genetic strategy yet: "
436
+ "the seal format records generative proposals, and the genetic "
437
+ "loop makes operator choices instead. Pass strategy='authoring' "
438
+ "to seal a generative run, or drop seal=."
439
+ )
440
+ if chosen != "genetic":
441
+ # The sentinels are read as "not asked for": ``auto`` is this package
442
+ # choosing, and refusing a run over a choice the caller never made
443
+ # would be the package arguing with itself.
444
+ engaged = [name for name, on in (
445
+ ("structure_budget", structure_budget not in ("auto", 0)),
446
+ ("prior", prior not in ("auto", "rule")),
447
+ ("chooser", chooser == "llm"),
448
+ ("effort", effort is not None),
449
+ ("journal", journal is not None),
450
+ ("authorship", authorship_config is not None
451
+ and authorship_config.engaged),
452
+ ) if on]
453
+ if engaged:
454
+ # A knob the run would silently ignore is a silent no-op -- the
455
+ # same defect class the seal refusal above exists to prevent.
456
+ raise ValueError(
457
+ f"{', '.join(engaged)} belong(s) to the genetic strategy, and "
458
+ "this run resolved to 'authoring'. Give the problem a seed to "
459
+ "use the genetic loop, or drop the genetic-only arguments."
460
+ )
461
+ if chosen == "genetic":
462
+ # Only the loop is imported locally. Importing EvaluationCache here too
463
+ # would make that name function-local for the whole body and break the
464
+ # authoring path below, which uses the module-level import.
465
+ from agent_evolve.session.genetic_loop import GeneticConfig, run_genetic_loop
466
+
467
+ journal_handle = None
468
+ journal_sink: Optional[Callable[[dict], None]] = None
469
+ if callable(journal):
470
+ journal_sink = journal
471
+ elif journal is not None:
472
+ import json as _json
473
+ from pathlib import Path
474
+
475
+ journal_path = Path(journal)
476
+ journal_path.parent.mkdir(parents=True, exist_ok=True)
477
+ # Opened eagerly even though the run may make no call: an empty
478
+ # journal is a measured zero, an absent file is "nobody looked".
479
+ journal_handle = journal_path.open("w", encoding="utf-8")
480
+
481
+ def journal_sink(record: dict) -> None:
482
+ journal_handle.write(_json.dumps(record, sort_keys=True) + "\n")
483
+ journal_handle.flush()
484
+
485
+ try:
486
+ chooser_policy = None
487
+ complete = None
488
+ # Provider usage is measured from the completion seam's own
489
+ # journal, never declared: zero means "counted and none occurred".
490
+ usage_ledger = {"calls": 0, "input": 0, "output": 0, "tokens_known": True}
491
+ if kind == "llm":
492
+ # The completion seam is built for the whole run, not for one
493
+ # consumer. It used to be constructed inside the chooser's own
494
+ # branch, which meant the seams that measured well -- authored
495
+ # initialization, the weighted prior -- could only be bought
496
+ # together with the one that measured null.
497
+ from agent_evolve.integrations.completion import completion_for
498
+
499
+ def _record_usage(record: dict) -> None:
500
+ usage_ledger["calls"] += 1
501
+ usage = record.get("usage") or {}
502
+ prompt_tokens = usage.get("prompt_tokens")
503
+ completion_tokens = usage.get("completion_tokens")
504
+ if isinstance(prompt_tokens, int) and isinstance(completion_tokens, int):
505
+ usage_ledger["input"] += prompt_tokens
506
+ usage_ledger["output"] += completion_tokens
507
+ else:
508
+ usage_ledger["tokens_known"] = False
509
+ if journal_sink is not None:
510
+ journal_sink(record)
511
+
512
+ # The shipped completion ceiling comes from the profile the
513
+ # product already declares for the route, not from the
514
+ # provider's undeclared default. Sending nothing was never
515
+ # "no cap": it was 65,536 on the default route, against the
516
+ # 128,000 the profile declares -- and the half that went
517
+ # missing was taken from the calls that reasoned longest.
518
+ # An unknown route still declares nothing, and then nothing
519
+ # is sent, so that path keeps the pre-cap body exactly.
520
+ from agent_evolve.integrations.pydantic_ai.model_execution_profile import ( # noqa: E501
521
+ declared_max_output_tokens)
522
+
523
+ route = model or settings.model
524
+ cap = declared_max_output_tokens(route)
525
+ complete = completion_for(route, settings,
526
+ journal=_record_usage, effort=effort,
527
+ max_output_tokens=cap)
528
+ if complete is None:
529
+ # The caller asked for a model BY NAME and no credential
530
+ # can honour it. Falling back to the classical path here
531
+ # ran to completion and said nothing -- a run launched to
532
+ # measure a model measured the control instead, and the
533
+ # only trace was `calls: 0`. Found by the release CI's
534
+ # stranger job, 2026-08-20.
535
+ import importlib.util
536
+ raise RuntimeError(_llm_refusal_message(
537
+ extra_missing=importlib.util.find_spec("pydantic_ai")
538
+ is None))
539
+ if chooser == "llm":
540
+ if complete is None:
541
+ announce(
542
+ "chooser='llm' needs a model call and none is "
543
+ "available; operator choices stay random."
544
+ )
545
+ else:
546
+ # Guided operator choice: the model picks parents and cut
547
+ # points, reasoning over the accumulated search state. It
548
+ # cannot author a candidate -- OperatorChoice has no field
549
+ # that could hold one. Opt-in, because it is the one seam
550
+ # here with ten sealed null verdicts against it.
551
+ from agent_evolve.policies.llm_chooser import llm_chooser
552
+ from agent_evolve.policies.semantics import domain_card
553
+ chooser_policy = llm_chooser(
554
+ complete, objectives=list(bound.objectives), budget=budget,
555
+ domain_context=domain_card(bound),
556
+ on_shortfall=lambda got, want: announce(
557
+ f"the model supplied {got} of {want} operator choices; "
558
+ "the rest were filled at random"),
559
+ )
560
+ if effort is not None and complete is None:
561
+ announce(
562
+ "effort pins model reasoning, and this run makes no model "
563
+ "calls, so it has no effect here."
564
+ )
565
+
566
+ # Resolved here and not earlier: what the sentinels mean depends on
567
+ # whether a model call is actually possible, which is not known
568
+ # until the seam above has either been built or come back empty.
569
+ prior, structure_budget = _resolve_guidance(
570
+ prior, structure_budget, budget=budget,
571
+ model_calls=complete is not None, announce=announce)
572
+ _check_structure_budget(structure_budget, budget)
573
+
574
+ prior_proposer: Any = None
575
+ if prior == "rule-weighted":
576
+ from agent_evolve.policies.weighted_prior import (
577
+ statistical_weighted_prior)
578
+ prior_proposer = statistical_weighted_prior
579
+ elif prior in ("llm", "llm-weighted", "llm-weighted-committed"):
580
+ if complete is None:
581
+ announce(
582
+ f"prior={prior!r} needs a model call and none is "
583
+ "available; using the credential-free rule comparator "
584
+ "instead."
585
+ )
586
+ if prior != "llm":
587
+ from agent_evolve.policies.weighted_prior import (
588
+ statistical_weighted_prior)
589
+ prior_proposer = statistical_weighted_prior
590
+ elif prior == "llm":
591
+ from agent_evolve.policies.llm_prior import llm_prior_proposer
592
+ from agent_evolve.policies.semantics import domain_card
593
+ prior_proposer = llm_prior_proposer(
594
+ complete, objectives=list(bound.objectives),
595
+ domain_context=domain_card(bound))
596
+ else:
597
+ from agent_evolve.policies.semantics import domain_card
598
+ from agent_evolve.policies.weighted_prior import (
599
+ llm_weighted_prior_proposer)
600
+ # The two model-weighted forms differ by ONE clause of the
601
+ # prompt; everything downstream of the reply is shared.
602
+ prior_proposer = llm_weighted_prior_proposer(
603
+ complete, objectives=list(bound.objectives),
604
+ domain_context=domain_card(bound),
605
+ style=("committed"
606
+ if prior == "llm-weighted-committed"
607
+ else "cautious"))
608
+
609
+ if authorship_config is None:
610
+ if complete is not None:
611
+ authorship_config = AuthorshipConfig(surrogate="llm",
612
+ initialization="llm")
613
+ announce(
614
+ "authorship: model-authored surrogate screening is ON "
615
+ "(the sealed luna-clear row held), and model-proposed "
616
+ "initialization is ON -- the six-arm ablation's "
617
+ "strongest arm, at 11x fewer evaluations to target, "
618
+ "better on 40 of 40 paired seeds, for one call; pass "
619
+ "authorship='off' to disable.")
620
+ else:
621
+ authorship_config = AuthorshipConfig()
622
+ # One rule, stated once, in `_genetic_sizing`: the sealed
623
+ # expression at and below the sealed ceiling, a population that
624
+ # grows with the budget above it.
625
+ pop, offspring = _genetic_sizing(budget)
626
+
627
+ from agent_evolve.policies.semantics import domain_card
628
+ from agent_evolve.session.authorship import build_authorship
629
+ policies = build_authorship(
630
+ authorship_config, complete=complete,
631
+ objectives=list(bound.objectives),
632
+ schema_text=domain_card(bound), seed=seed, announce=announce,
633
+ candidate_model=getattr(bound, "candidate_model", None),
634
+ init_template=(dict(seeds[0]) if seeds else None),
635
+ init_k=max(0, pop - len(seeds)),
636
+ budget=budget, population_size=pop)
637
+
638
+ cache = EvaluationCache()
639
+ cache.budget = budget
640
+ # `generations` is a cap, not a schedule. Duplicate offspring hit
641
+ # the evaluation cache without spending budget, so a fixed
642
+ # generation count would end the run with budget unspent; the
643
+ # loop's real stop condition is the budget.
644
+ result = run_genetic_loop(
645
+ problem=bound,
646
+ config=GeneticConfig(
647
+ population_size=pop,
648
+ offspring_per_generation=offspring,
649
+ generations=max(1, 4 * budget // max(1, offspring)),
650
+ seed=seed,
651
+ seeds=seeds,
652
+ evaluation_budget=budget,
653
+ evaluation_cache=cache,
654
+ structure_budget=structure_budget,
655
+ prior_proposer=prior_proposer,
656
+ screening=policies.screening,
657
+ portfolio=policies.portfolio,
658
+ initial_proposals=policies.initial_proposals,
659
+ generator=policies.generator,
660
+ reguidance=policies.reguidance,
661
+ ),
662
+ chooser=chooser_policy,
663
+ log=announce,
664
+ )
665
+ # Authoring that produced no policy object still produced
666
+ # counters, and the loop can only harvest what it was handed.
667
+ # These are the seams whose failure leaves nothing behind.
668
+ orphaned = tuple(note for note in (policies.init_author,
669
+ policies.generator_author,
670
+ policies.reguidance_author)
671
+ if note is not None)
672
+ if orphaned and result.telemetry is not None:
673
+ from agent_evolve.core.telemetry import harvest_telemetry
674
+ extra = harvest_telemetry(orphaned)
675
+ result = dataclasses.replace(
676
+ result,
677
+ telemetry=dataclasses.replace(
678
+ result.telemetry,
679
+ mechanisms=result.telemetry.mechanisms + extra.mechanisms))
680
+ if usage_ledger["calls"] and usage_ledger["tokens_known"]:
681
+ # Cost is DERIVED, and the reporter says so. The tokens are the
682
+ # provider's own count; the price is this package's published
683
+ # table (`MODEL_PRICES_PER_MTOK`), which is the same number the
684
+ # CLI echoes before spending anything. A route the table does
685
+ # not name reports `cost_usd: null` -- unknown stays unknown
686
+ # rather than becoming a guess with a dollar sign on it.
687
+ route = model or settings.model
688
+ cost_usd, reporter = _priced_usage(
689
+ route, usage_ledger["input"], usage_ledger["output"])
690
+ usage = ProviderUsageSummary(
691
+ calls=usage_ledger["calls"],
692
+ input_tokens=usage_ledger["input"],
693
+ output_tokens=usage_ledger["output"],
694
+ cost_usd=cost_usd,
695
+ model=route,
696
+ reported_by=reporter,
697
+ )
698
+ else:
699
+ usage = ProviderUsageSummary(
700
+ calls=usage_ledger["calls"],
701
+ model=(model or settings.model) if usage_ledger["calls"] else None,
702
+ )
703
+ return dataclasses.replace(result, provider_usage=usage)
704
+ finally:
705
+ # Closed even when the run raises: a journal truncated by a crash
706
+ # still records every call that did happen.
707
+ if journal_handle is not None:
708
+ journal_handle.close()
709
+
710
+ harness = _build_harness(kind, seed, settings)
711
+
712
+ ctx = HarnessContext(
713
+ objectives=list(bound.objectives),
714
+ search_space_desc=_describe(bound),
715
+ candidate_model=getattr(bound, "candidate_model", None),
716
+ constraints_description=getattr(bound, "constraints_description", "") or "",
717
+ directives=getattr(bound, "directives", None) or DefaultDirectives(),
718
+ )
719
+ harness.bind(
720
+ ctx,
721
+ LLMConfig(model=model or settings.model, temperature=settings.temperature),
722
+ )
723
+
724
+ seal_handle = None
725
+ if seal is not None:
726
+ from pathlib import Path
727
+
728
+ from agent_evolve.application.generative_proposal_journal import journal_line
729
+ from agent_evolve.proposal_mode import build_generative_proposer
730
+
731
+ path = Path(seal)
732
+ path.parent.mkdir(parents=True, exist_ok=True)
733
+ seal_handle = path.open("w", encoding="ascii")
734
+
735
+ def _write(record: dict) -> None:
736
+ seal_handle.write(journal_line(record) + "\n")
737
+ seal_handle.flush()
738
+
739
+ harness = build_generative_proposer(bound, delegate=harness, on_seal=_write)
740
+ harness.bind(
741
+ ctx,
742
+ LLMConfig(model=model or settings.model, temperature=settings.temperature),
743
+ )
744
+
745
+ cache = EvaluationCache() # `seeds` was already read above, once
746
+ cache.budget = budget
747
+ config = LoopConfig(
748
+ pop_size=min(budget, _BATCH),
749
+ generations=max(1, budget // _BATCH),
750
+ candidates_per_batch=_BATCH,
751
+ seed=seed,
752
+ seeds=seeds,
753
+ evaluation_budget=budget,
754
+ evaluation_cache=cache,
755
+ )
756
+ try:
757
+ return run_evolution_loop(
758
+ problem=bound,
759
+ harness=harness,
760
+ config=config,
761
+ log=announce,
762
+ )
763
+ finally:
764
+ # Closed even when the run raises: a journal truncated by a crash still
765
+ # records every call that did happen, and that is the honest artifact.
766
+ if seal_handle is not None:
767
+ seal_handle.close()