agentevolve-optimizer 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (414) hide show
  1. agent_evolve/__init__.py +722 -0
  2. agent_evolve/agentic.py +2800 -0
  3. agent_evolve/api.py +767 -0
  4. agent_evolve/application/__init__.py +1580 -0
  5. agent_evolve/application/action_allocation.py +744 -0
  6. agent_evolve/application/action_allocation_frame.py +347 -0
  7. agent_evolve/application/action_allocation_frame_commit.py +185 -0
  8. agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
  9. agent_evolve/application/action_allocation_frame_v3.py +338 -0
  10. agent_evolve/application/action_archive_value.py +497 -0
  11. agent_evolve/application/action_evidence_consistency.py +455 -0
  12. agent_evolve/application/action_forecast_partitioning.py +1471 -0
  13. agent_evolve/application/action_metric_projection.py +211 -0
  14. agent_evolve/application/action_role_value.py +680 -0
  15. agent_evolve/application/action_score_authorities.py +363 -0
  16. agent_evolve/application/action_structural_signature.py +116 -0
  17. agent_evolve/application/action_target_realization.py +402 -0
  18. agent_evolve/application/agentic_evolution.py +7734 -0
  19. agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
  20. agent_evolve/application/anchor_residual_identification.py +463 -0
  21. agent_evolve/application/archive_conditioned_action_target.py +208 -0
  22. agent_evolve/application/artifact_journal.py +246 -0
  23. agent_evolve/application/artifact_replay.py +347 -0
  24. agent_evolve/application/budgeted_optimizer.py +1828 -0
  25. agent_evolve/application/calibrated_campaign.py +485 -0
  26. agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
  27. agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
  28. agent_evolve/application/campaign_capacity_recourse.py +254 -0
  29. agent_evolve/application/campaign_contextual_outcomes.py +119 -0
  30. agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
  31. agent_evolve/application/campaign_evidence_registry.py +262 -0
  32. agent_evolve/application/campaign_execution.py +2537 -0
  33. agent_evolve/application/campaign_generation_audit.py +942 -0
  34. agent_evolve/application/campaign_learning.py +1812 -0
  35. agent_evolve/application/campaign_learning_runtime.py +1977 -0
  36. agent_evolve/application/campaign_search_phase.py +227 -0
  37. agent_evolve/application/campaign_selector_context_extension.py +220 -0
  38. agent_evolve/application/campaign_variation_envelope.py +649 -0
  39. agent_evolve/application/campaign_variation_trace.py +451 -0
  40. agent_evolve/application/candidate_archive_consequence.py +128 -0
  41. agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
  42. agent_evolve/application/composite_outcome_updater.py +145 -0
  43. agent_evolve/application/composition_portfolio_selection.py +363 -0
  44. agent_evolve/application/concurrent_stage.py +144 -0
  45. agent_evolve/application/contextual_action_allocation.py +181 -0
  46. agent_evolve/application/contextual_campaign_outcomes.py +267 -0
  47. agent_evolve/application/contextual_campaign_planning.py +1366 -0
  48. agent_evolve/application/contextual_delayed_credit.py +651 -0
  49. agent_evolve/application/contextual_search_controller.py +2374 -0
  50. agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
  51. agent_evolve/application/decision_metric_projection.py +112 -0
  52. agent_evolve/application/derived_action_semantics.py +129 -0
  53. agent_evolve/application/detailed_evaluation.py +449 -0
  54. agent_evolve/application/earned_lineage.py +1011 -0
  55. agent_evolve/application/effective_choice_audit.py +484 -0
  56. agent_evolve/application/empirical_consequence_calibration.py +908 -0
  57. agent_evolve/application/evaluation_accounting.py +325 -0
  58. agent_evolve/application/evaluation_cache.py +199 -0
  59. agent_evolve/application/evaluation_escrow.py +547 -0
  60. agent_evolve/application/evaluation_recourse.py +253 -0
  61. agent_evolve/application/event_recorder.py +151 -0
  62. agent_evolve/application/evolution_campaign.py +1840 -0
  63. agent_evolve/application/executable_hypothesis.py +323 -0
  64. agent_evolve/application/factorial_branch_pilot.py +772 -0
  65. agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
  66. agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
  67. agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
  68. agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
  69. agent_evolve/application/finite_action_selection.py +188 -0
  70. agent_evolve/application/finite_action_set.py +306 -0
  71. agent_evolve/application/finite_action_transition.py +537 -0
  72. agent_evolve/application/finite_variation_eligibility.py +296 -0
  73. agent_evolve/application/forecast_geometry_portfolio.py +799 -0
  74. agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
  75. agent_evolve/application/front_proximity_admission.py +311 -0
  76. agent_evolve/application/front_proximity_parent_basis.py +458 -0
  77. agent_evolve/application/frozen_hurdle_score.py +659 -0
  78. agent_evolve/application/g3_causal_screen.py +2257 -0
  79. agent_evolve/application/g3_causal_validation.py +1046 -0
  80. agent_evolve/application/g3_postseal_curation.py +818 -0
  81. agent_evolve/application/gated_agentic_generator.py +205 -0
  82. agent_evolve/application/generation_feedback.py +293 -0
  83. agent_evolve/application/generative_proposal_journal.py +185 -0
  84. agent_evolve/application/geometry_conditional_elasticity.py +453 -0
  85. agent_evolve/application/global_wave_action_allocation.py +1151 -0
  86. agent_evolve/application/head_mass_conditional_seat.py +268 -0
  87. agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
  88. agent_evolve/application/identifiable_reflection_learning.py +395 -0
  89. agent_evolve/application/identifiable_reflection_request.py +364 -0
  90. agent_evolve/application/in_memory_residual_archive.py +341 -0
  91. agent_evolve/application/insight_memory.py +1804 -0
  92. agent_evolve/application/live_runtime_manifest.py +758 -0
  93. agent_evolve/application/llm_task_queue.py +769 -0
  94. agent_evolve/application/matched_finite_action_block.py +409 -0
  95. agent_evolve/application/materialized_action_broker.py +2328 -0
  96. agent_evolve/application/materialized_action_constraints.py +83 -0
  97. agent_evolve/application/materialized_variation.py +211 -0
  98. agent_evolve/application/multi_option_evolution.py +1536 -0
  99. agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
  100. agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
  101. agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
  102. agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
  103. agent_evolve/application/outcome_relation.py +193 -0
  104. agent_evolve/application/paired_allocation_comparison.py +241 -0
  105. agent_evolve/application/paired_block_schedule.py +127 -0
  106. agent_evolve/application/parent_measurement.py +226 -0
  107. agent_evolve/application/pareto_archive.py +811 -0
  108. agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
  109. agent_evolve/application/portfolio_evolution.py +2950 -0
  110. agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
  111. agent_evolve/application/portfolio_memory_attribution.py +581 -0
  112. agent_evolve/application/portfolio_memory_dose.py +788 -0
  113. agent_evolve/application/portfolio_memory_matched_control.py +938 -0
  114. agent_evolve/application/portfolio_memory_transfer.py +297 -0
  115. agent_evolve/application/portfolio_optimization_memory.py +363 -0
  116. agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
  117. agent_evolve/application/portfolio_projection.py +335 -0
  118. agent_evolve/application/portfolio_recombination.py +2032 -0
  119. agent_evolve/application/post_evolution_reflection.py +834 -0
  120. agent_evolve/application/postcommit_rank_authority.py +245 -0
  121. agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
  122. agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
  123. agent_evolve/application/prequential_residual_exploration.py +343 -0
  124. agent_evolve/application/prequential_score_portfolio.py +954 -0
  125. agent_evolve/application/projections.py +292 -0
  126. agent_evolve/application/protected_action_committee.py +1027 -0
  127. agent_evolve/application/protected_branch_pilot.py +376 -0
  128. agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
  129. agent_evolve/application/provider_replay.py +910 -0
  130. agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
  131. agent_evolve/application/recombination_residual_expert.py +403 -0
  132. agent_evolve/application/reflection_workflow.py +571 -0
  133. agent_evolve/application/region_conditional_credit.py +911 -0
  134. agent_evolve/application/residual_campaign_runtime.py +531 -0
  135. agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
  136. agent_evolve/application/residual_headroom_ledger.py +1544 -0
  137. agent_evolve/application/residual_learning_transaction.py +396 -0
  138. agent_evolve/application/residual_portfolio_evolution.py +1228 -0
  139. agent_evolve/application/residual_reachability.py +749 -0
  140. agent_evolve/application/residual_stage_credit.py +499 -0
  141. agent_evolve/application/same_prefix_paired_audit.py +1580 -0
  142. agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
  143. agent_evolve/application/sequential_lineage_allocation.py +1017 -0
  144. agent_evolve/application/sequential_market_replay.py +1395 -0
  145. agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
  146. agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
  147. agent_evolve/application/single_score_action_allocation.py +299 -0
  148. agent_evolve/application/source_exposure_allocation.py +906 -0
  149. agent_evolve/application/staged_memory.py +210 -0
  150. agent_evolve/application/stratified_cold_start_allocation.py +732 -0
  151. agent_evolve/application/support_guarded_hurdle_score.py +549 -0
  152. agent_evolve/application/target_conditioned_action_forecast.py +595 -0
  153. agent_evolve/application/target_conditioned_campaign.py +566 -0
  154. agent_evolve/application/treatment_assignment.py +201 -0
  155. agent_evolve/application/trusted_objective_evidence.py +217 -0
  156. agent_evolve/application/two_stage_action_evolution.py +1131 -0
  157. agent_evolve/application/v8lite_allocation_policy.py +1083 -0
  158. agent_evolve/application/v9_candidate_policy.py +1303 -0
  159. agent_evolve/bootstrap.py +108 -0
  160. agent_evolve/campaign_presets.py +517 -0
  161. agent_evolve/campaign_profiles.py +452 -0
  162. agent_evolve/campaign_variation_topology.py +288 -0
  163. agent_evolve/campaign_workload.py +950 -0
  164. agent_evolve/cli.py +797 -0
  165. agent_evolve/contract.py +241 -0
  166. agent_evolve/core/__init__.py +91 -0
  167. agent_evolve/core/action_semantics.py +411 -0
  168. agent_evolve/core/authored.py +105 -0
  169. agent_evolve/core/formatting.py +286 -0
  170. agent_evolve/core/optimization_semantics.py +324 -0
  171. agent_evolve/core/problem.py +167 -0
  172. agent_evolve/core/results.py +323 -0
  173. agent_evolve/core/stats.py +70 -0
  174. agent_evolve/core/telemetry.py +100 -0
  175. agent_evolve/domain/__init__.py +89 -0
  176. agent_evolve/domain/artifact.py +162 -0
  177. agent_evolve/domain/durable_text.py +68 -0
  178. agent_evolve/domain/event.py +1454 -0
  179. agent_evolve/domain/finite_action_set.py +426 -0
  180. agent_evolve/domain/finite_variation.py +526 -0
  181. agent_evolve/domain/generative_emission.py +559 -0
  182. agent_evolve/domain/ids.py +163 -0
  183. agent_evolve/domain/inline_text.py +106 -0
  184. agent_evolve/domain/insight.py +27 -0
  185. agent_evolve/domain/lineage.py +737 -0
  186. agent_evolve/domain/llm_task_queue.py +960 -0
  187. agent_evolve/domain/outcome.py +96 -0
  188. agent_evolve/domain/patch.py +854 -0
  189. agent_evolve/domain/typed_json.py +542 -0
  190. agent_evolve/domain/variation_space.py +158 -0
  191. agent_evolve/driver.py +1014 -0
  192. agent_evolve/harness/__init__.py +29 -0
  193. agent_evolve/harness/base.py +242 -0
  194. agent_evolve/harness/directives.py +163 -0
  195. agent_evolve/harness/generative_seal.py +479 -0
  196. agent_evolve/harness/registry.py +41 -0
  197. agent_evolve/infrastructure/__init__.py +39 -0
  198. agent_evolve/infrastructure/artifacts/__init__.py +6 -0
  199. agent_evolve/infrastructure/artifacts/_verification.py +67 -0
  200. agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
  201. agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
  202. agent_evolve/infrastructure/asyncio_runtime.py +109 -0
  203. agent_evolve/infrastructure/authored_runtime.py +188 -0
  204. agent_evolve/infrastructure/authored_worker.py +171 -0
  205. agent_evolve/infrastructure/clock.py +53 -0
  206. agent_evolve/infrastructure/events/__init__.py +6 -0
  207. agent_evolve/infrastructure/events/_validation.py +89 -0
  208. agent_evolve/infrastructure/events/in_memory.py +56 -0
  209. agent_evolve/infrastructure/events/jsonl.py +193 -0
  210. agent_evolve/infrastructure/exception_provenance.py +215 -0
  211. agent_evolve/infrastructure/ids.py +118 -0
  212. agent_evolve/infrastructure/lineage_codec.py +1836 -0
  213. agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
  214. agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
  215. agent_evolve/infrastructure/resource_lease.py +370 -0
  216. agent_evolve/infrastructure/sanitization/__init__.py +8 -0
  217. agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
  218. agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
  219. agent_evolve/infrastructure/stream_liveness.py +383 -0
  220. agent_evolve/infrastructure/subprocess_boundary.py +136 -0
  221. agent_evolve/integrations/__init__.py +1 -0
  222. agent_evolve/integrations/botorch/__init__.py +28 -0
  223. agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
  224. agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
  225. agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
  226. agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
  227. agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
  228. agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
  229. agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
  230. agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
  231. agent_evolve/integrations/completion.py +242 -0
  232. agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
  233. agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
  234. agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
  235. agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
  236. agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
  237. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
  238. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
  239. agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
  240. agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
  241. agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
  242. agent_evolve/integrations/pydantic_ai/harness.py +159 -0
  243. agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
  244. agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
  245. agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
  246. agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
  247. agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
  248. agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
  249. agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
  250. agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
  251. agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
  252. agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
  253. agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
  254. agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
  255. agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
  256. agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
  257. agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
  258. agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
  259. agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
  260. agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
  261. agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
  262. agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
  263. agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
  264. agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
  265. agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
  266. agent_evolve/integrations/pymoo_adapter.py +242 -0
  267. agent_evolve/policies/__init__.py +17 -0
  268. agent_evolve/policies/check.py +469 -0
  269. agent_evolve/policies/emit_scaffold.py +451 -0
  270. agent_evolve/policies/feedback/__init__.py +37 -0
  271. agent_evolve/policies/feedback/held_out_asn.py +1325 -0
  272. agent_evolve/policies/genetic.py +607 -0
  273. agent_evolve/policies/llm_backoff.py +183 -0
  274. agent_evolve/policies/llm_chooser.py +226 -0
  275. agent_evolve/policies/llm_generator.py +1760 -0
  276. agent_evolve/policies/llm_init.py +267 -0
  277. agent_evolve/policies/llm_operator.py +109 -0
  278. agent_evolve/policies/llm_prior.py +194 -0
  279. agent_evolve/policies/llm_surrogate.py +334 -0
  280. agent_evolve/policies/measurement_evidence.py +704 -0
  281. agent_evolve/policies/memory/__init__.py +223 -0
  282. agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
  283. agent_evolve/policies/memory/compatibility_matching.py +593 -0
  284. agent_evolve/policies/memory/global_falsification.py +1841 -0
  285. agent_evolve/policies/memory/prompt_shape.py +503 -0
  286. agent_evolve/policies/memory/randomized_subset.py +714 -0
  287. agent_evolve/policies/memory/staged_causal.py +1270 -0
  288. agent_evolve/policies/memory/treatment_compliance.py +759 -0
  289. agent_evolve/policies/objective_resolution/__init__.py +17 -0
  290. agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
  291. agent_evolve/policies/operator_portfolio.py +407 -0
  292. agent_evolve/policies/reguidance.py +1133 -0
  293. agent_evolve/policies/reward/__init__.py +83 -0
  294. agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
  295. agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
  296. agent_evolve/policies/reward/affine_hypervolume.py +490 -0
  297. agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
  298. agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
  299. agent_evolve/policies/reward/frozen_archive.py +360 -0
  300. agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
  301. agent_evolve/policies/search_state.py +208 -0
  302. agent_evolve/policies/selection/__init__.py +345 -0
  303. agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
  304. agent_evolve/policies/selection/affine_frontier_context.py +330 -0
  305. agent_evolve/policies/selection/affine_frontier_target.py +473 -0
  306. agent_evolve/policies/selection/archive_elite.py +1346 -0
  307. agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
  308. agent_evolve/policies/selection/calibrated_slate.py +1394 -0
  309. agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
  310. agent_evolve/policies/selection/common_candidate_pool.py +685 -0
  311. agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
  312. agent_evolve/policies/selection/disjoint_pairs.py +479 -0
  313. agent_evolve/policies/selection/elite_explorer.py +719 -0
  314. agent_evolve/policies/selection/finite_action.py +187 -0
  315. agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
  316. agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
  317. agent_evolve/policies/selection/forecast_calibration.py +922 -0
  318. agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
  319. agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
  320. agent_evolve/policies/selection/full_support_slate.py +91 -0
  321. agent_evolve/policies/selection/meaningful_direction.py +240 -0
  322. agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
  323. agent_evolve/policies/selection/model_anchored_slate.py +826 -0
  324. agent_evolve/policies/selection/phenotype_recourse.py +979 -0
  325. agent_evolve/policies/selection/proposal_support.py +368 -0
  326. agent_evolve/policies/selection/random_portfolio.py +254 -0
  327. agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
  328. agent_evolve/policies/selection/residual_frontier.py +463 -0
  329. agent_evolve/policies/selection/residual_frontier_target.py +605 -0
  330. agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
  331. agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
  332. agent_evolve/policies/selection/target_conditioned_features.py +812 -0
  333. agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
  334. agent_evolve/policies/selection/task_keyed_palette.py +906 -0
  335. agent_evolve/policies/semantics.py +147 -0
  336. agent_evolve/policies/structure.py +362 -0
  337. agent_evolve/policies/structured_output_budget.py +62 -0
  338. agent_evolve/policies/surrogate.py +696 -0
  339. agent_evolve/policies/variation/__init__.py +1 -0
  340. agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
  341. agent_evolve/policies/variation/crossover_inheritance.py +575 -0
  342. agent_evolve/policies/variation/disjoint_recombination.py +611 -0
  343. agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
  344. agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
  345. agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
  346. agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
  347. agent_evolve/policies/variation/typed_patch.py +1981 -0
  348. agent_evolve/policies/weighted_prior.py +394 -0
  349. agent_evolve/ports/__init__.py +383 -0
  350. agent_evolve/ports/action_allocation.py +733 -0
  351. agent_evolve/ports/action_allocation_frame.py +1153 -0
  352. agent_evolve/ports/action_allocation_frame_commit.py +294 -0
  353. agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
  354. agent_evolve/ports/action_allocation_frame_v3.py +995 -0
  355. agent_evolve/ports/action_forecast.py +1568 -0
  356. agent_evolve/ports/action_metric_projection.py +165 -0
  357. agent_evolve/ports/agentic_generator.py +1561 -0
  358. agent_evolve/ports/archive_context.py +136 -0
  359. agent_evolve/ports/artifact_sanitizer.py +44 -0
  360. agent_evolve/ports/artifact_store.py +225 -0
  361. agent_evolve/ports/clock.py +13 -0
  362. agent_evolve/ports/contextual_search_allocation.py +827 -0
  363. agent_evolve/ports/decision_metric_projection.py +258 -0
  364. agent_evolve/ports/event_store.py +55 -0
  365. agent_evolve/ports/executable_hypothesis.py +557 -0
  366. agent_evolve/ports/finite_acquisition.py +377 -0
  367. agent_evolve/ports/finite_acquisition_batch.py +296 -0
  368. agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
  369. agent_evolve/ports/finite_acquisition_json.py +247 -0
  370. agent_evolve/ports/finite_acquisition_space.py +168 -0
  371. agent_evolve/ports/finite_action_selection.py +348 -0
  372. agent_evolve/ports/finite_action_set.py +256 -0
  373. agent_evolve/ports/frontier_target.py +396 -0
  374. agent_evolve/ports/generation_failure.py +43 -0
  375. agent_evolve/ports/hard_feasibility.py +233 -0
  376. agent_evolve/ports/id_factory.py +34 -0
  377. agent_evolve/ports/llm_task_queue.py +93 -0
  378. agent_evolve/ports/objective_resolution.py +419 -0
  379. agent_evolve/ports/paired_allocation_comparison.py +401 -0
  380. agent_evolve/ports/paired_block_schedule.py +475 -0
  381. agent_evolve/ports/parent_measurement.py +336 -0
  382. agent_evolve/ports/portfolio_memory_dose.py +643 -0
  383. agent_evolve/ports/portfolio_selection.py +3169 -0
  384. agent_evolve/ports/postcommit_rank_authority.py +467 -0
  385. agent_evolve/ports/presented_action_evidence.py +794 -0
  386. agent_evolve/ports/resource_lease.py +162 -0
  387. agent_evolve/ports/structured_generator.py +734 -0
  388. agent_evolve/ports/structured_output_budget.py +120 -0
  389. agent_evolve/ports/subprocess_boundary.py +138 -0
  390. agent_evolve/ports/treatment_assignment.py +466 -0
  391. agent_evolve/ports/variation_catalog.py +76 -0
  392. agent_evolve/ports/variation_source.py +226 -0
  393. agent_evolve/proposal_mode.py +157 -0
  394. agent_evolve/proposers/__init__.py +10 -0
  395. agent_evolve/proposers/random_proposer.py +188 -0
  396. agent_evolve/provider_accounting.py +163 -0
  397. agent_evolve/py.typed +0 -0
  398. agent_evolve/reference_method.py +1570 -0
  399. agent_evolve/session/__init__.py +11 -0
  400. agent_evolve/session/authorship.py +864 -0
  401. agent_evolve/session/evaluate.py +236 -0
  402. agent_evolve/session/fidelity.py +237 -0
  403. agent_evolve/session/genetic_loop.py +742 -0
  404. agent_evolve/session/loop.py +803 -0
  405. agent_evolve/session/screening.py +671 -0
  406. agent_evolve/settings.py +376 -0
  407. agent_evolve/workload_kit.py +368 -0
  408. agent_evolve/workload_prompt.py +398 -0
  409. agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
  410. agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
  411. agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
  412. agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
  413. agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
  414. agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
agent_evolve/cli.py ADDED
@@ -0,0 +1,797 @@
1
+ """``agent_evolve`` command line: ``init``, ``diagnose``, ``check``, ``run``,
2
+ ``version``.
3
+
4
+ ``check`` is the one worth reading about. It runs the model against an
5
+ uninformed sampler on *your* problem, at the same budget, with the same
6
+ evaluator, and reports whether the model actually beat it.
7
+
8
+ That command exists because this project repeatedly measured how easy it is to
9
+ believe an optimizer is working when nothing is: on one benchmark a single
10
+ median random draw already accounted for roughly 80 percent of the
11
+ hypervolume that a full run produced, and the entire model-guided phase past
12
+ the initial design was worth about one percent. No amount of reading a paper
13
+ tells anyone whether their own problem is like that. Half a minute of
14
+ ``agent_evolve check`` does.
15
+
16
+ A problem is named as ``module:attribute``, e.g.::
17
+
18
+ agent_evolve check examples.knapsack.problem_def:problem --budget 40
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import importlib
25
+ import json
26
+ import statistics
27
+ import sys
28
+ from typing import Any, List, Optional, Sequence
29
+
30
+ from agent_evolve.contract import as_problem
31
+ from agent_evolve.core.results import compute_pareto_front
32
+
33
+ __all__ = ["main"]
34
+
35
+
36
+ def _authorship_presets() -> tuple:
37
+ """The preset names, read from the table that defines them."""
38
+
39
+ from agent_evolve.session.authorship import PRESETS
40
+
41
+ return tuple(PRESETS)
42
+
43
+
44
+ def _structure_budget_arg(value: str) -> Any:
45
+ """``auto`` or an evaluation count; the sentinel is resolved by the API.
46
+
47
+ argparse converts a string default through ``type``, so a plain ``int``
48
+ type would try to parse the sentinel itself. Resolving it here instead
49
+ would put a second copy of the sizing rule in the CLI, where it could
50
+ drift from the one the library announces.
51
+ """
52
+
53
+ if value == "auto":
54
+ return "auto"
55
+ try:
56
+ return int(value)
57
+ except ValueError:
58
+ raise argparse.ArgumentTypeError(
59
+ "--structure-budget takes 'auto' or an evaluation count, got "
60
+ f"{value!r}"
61
+ ) from None
62
+
63
+
64
+ def _load_problem(spec: str) -> Any:
65
+ """Import ``module:attribute`` and return the problem object."""
66
+ if ":" not in spec:
67
+ raise SystemExit(
68
+ f"could not read {spec!r}. Name a problem as module:attribute, "
69
+ "e.g. examples.knapsack.problem_def:problem"
70
+ )
71
+ module_name, _, attribute = spec.partition(":")
72
+ sys.path.insert(0, "")
73
+ try:
74
+ module = importlib.import_module(module_name)
75
+ except ImportError as error:
76
+ raise SystemExit(f"could not import {module_name!r}: {error}") from error
77
+ try:
78
+ candidate = getattr(module, attribute)
79
+ except AttributeError as error:
80
+ raise SystemExit(f"{module_name!r} has no attribute {attribute!r}") from error
81
+ problem = candidate() if isinstance(candidate, type) else candidate
82
+ try:
83
+ return as_problem(problem)
84
+ except TypeError as error:
85
+ raise SystemExit(str(error)) from error
86
+
87
+
88
+ def _front_values(result: Any, objective: str) -> List[float]:
89
+ return [c.objectives[objective] for c in result.pareto_front if objective in c.objectives]
90
+
91
+
92
+ def _summarise(label: str, result: Any, objectives: Sequence[Any]) -> str:
93
+ lines = [f" {label}:"]
94
+ lines.append(f" evaluations {result.evaluations}")
95
+ lines.append(f" pareto front {len(result.pareto_front)}")
96
+ for spec in objectives:
97
+ values = _front_values(result, spec.name)
98
+ if not values:
99
+ continue
100
+ best = max(values) if spec.goal == "max" else min(values)
101
+ lines.append(f" best {spec.name} ({spec.goal}) {best:g}")
102
+ return "\n".join(lines)
103
+
104
+
105
+ def _verdict_rows(
106
+ model_result: Any,
107
+ baseline_results: Sequence[Any],
108
+ objectives: Sequence[Any],
109
+ ) -> List[dict]:
110
+ """The comparison itself, as data: one row per objective it could judge.
111
+
112
+ The prose verdict and the ``--json`` document are two renderings of this
113
+ one computation. A second copy of the rule -- which draws count, how ``p``
114
+ is formed, what "won" means -- could disagree with the first, and the
115
+ disagreement would be invisible until someone compared two outputs of the
116
+ same run.
117
+ """
118
+ rows: List[dict] = []
119
+ for spec in objectives:
120
+ model_values = _front_values(model_result, spec.name)
121
+ if not model_values:
122
+ continue
123
+ pick = max if spec.goal == "max" else min
124
+ model_best = pick(model_values)
125
+ draws = [
126
+ pick(_front_values(r, spec.name))
127
+ for r in baseline_results
128
+ if _front_values(r, spec.name)
129
+ ]
130
+ if not draws:
131
+ continue
132
+ better = sum(
133
+ 1 for d in draws if (d > model_best if spec.goal == "max" else d < model_best)
134
+ )
135
+ # Fraction of uninformed runs that matched or beat the model. This is
136
+ # the chance baseline every claim in this project travels with.
137
+ p = (better + 1) / (len(draws) + 1)
138
+ median = statistics.median(draws)
139
+ won = (model_best > median) if spec.goal == "max" else (model_best < median)
140
+ rows.append({
141
+ "objective": spec.name,
142
+ "goal": spec.goal,
143
+ "model_best": model_best,
144
+ "random_median": median,
145
+ "baseline_runs": len(draws),
146
+ "baseline_better": better,
147
+ "p": p,
148
+ "model_won": won,
149
+ })
150
+ return rows
151
+
152
+
153
+ def _winner(rows: Sequence[dict]) -> Optional[str]:
154
+ """``model``, ``baseline`` or ``mixed`` -- ``None`` when nothing was judged."""
155
+ if not rows:
156
+ return None
157
+ beaten = sum(1 for row in rows if row["model_won"])
158
+ if beaten == len(rows):
159
+ return "model"
160
+ if beaten == 0:
161
+ return "baseline"
162
+ return "mixed"
163
+
164
+
165
+ def _verdict(
166
+ model_result: Any,
167
+ baseline_results: Sequence[Any],
168
+ objectives: Sequence[Any],
169
+ ) -> List[str]:
170
+ """State plainly whether the model beat chance, per objective."""
171
+ out: List[str] = []
172
+ rows = _verdict_rows(model_result, baseline_results, objectives)
173
+ counted = len(rows)
174
+ beaten = sum(1 for row in rows if row["model_won"])
175
+ for row in rows:
176
+ out.append(
177
+ f" {row['objective']}: model {row['model_best']:g} vs random "
178
+ f"median {row['random_median']:g} over {row['baseline_runs']} runs "
179
+ f" (p = {row['p']:.2f})"
180
+ )
181
+ if counted:
182
+ out.append("")
183
+ if beaten == counted:
184
+ out.append(" The model beat the uninformed baseline on every objective.")
185
+ elif beaten == 0:
186
+ out.append(
187
+ " The model did not beat the uninformed baseline on any objective.\n"
188
+ " On this problem, at this budget, it is not earning its cost."
189
+ )
190
+ else:
191
+ out.append(
192
+ f" The model beat the uninformed baseline on {beaten} of "
193
+ f"{counted} objectives. Mixed, and worth a larger budget "
194
+ "before concluding either way."
195
+ )
196
+ out.append(
197
+ " A p near 1.00 means uninformed sampling routinely does as well.\n"
198
+ " Few runs is weak evidence; raise --repeats to sharpen it."
199
+ )
200
+ return out
201
+
202
+
203
+ def _model_line(model: Optional[str]) -> str:
204
+ """Name the model and its price before anything is billed.
205
+
206
+ A default nobody saw is a default nobody consented to, so the resolved
207
+ model is printed whether or not the caller chose it, and it is marked as a
208
+ default when they did not.
209
+ """
210
+ from agent_evolve.settings import AgentEvolveSettings, model_price
211
+
212
+ resolved = model or AgentEvolveSettings.from_env().model
213
+ origin = "" if model else " (default)"
214
+ price = model_price(resolved)
215
+ cost = (
216
+ f" ${price[0]:.2f}/M in, ${price[1]:.2f}/M out"
217
+ if price
218
+ else " price unknown"
219
+ )
220
+ return f"model {resolved}{origin}{cost}"
221
+
222
+
223
+ def _arm_outcome(result: Any, objectives: Sequence[Any], seed: int) -> dict:
224
+ """One arm's run as data: what it spent and the best it reached.
225
+
226
+ The Pareto front's rows are deliberately not here -- ``run --json`` is the
227
+ command that hands you a front, and duplicating it under five baseline
228
+ repeats would bury the one thing ``check`` exists to answer.
229
+ """
230
+ import dataclasses
231
+
232
+ best = {}
233
+ for spec in objectives:
234
+ values = _front_values(result, spec.name)
235
+ if values:
236
+ best[spec.name] = (max if spec.goal == "max" else min)(values)
237
+ return {
238
+ "seed": seed,
239
+ "evaluations": result.evaluations,
240
+ "pareto_front": len(result.pareto_front),
241
+ "best": best,
242
+ "provider_usage": (
243
+ dataclasses.asdict(result.provider_usage)
244
+ if result.provider_usage is not None else None
245
+ ),
246
+ }
247
+
248
+
249
+ def _check_json(
250
+ args: argparse.Namespace,
251
+ objectives: Sequence[Any],
252
+ baselines: Sequence[Any],
253
+ model_result: Any,
254
+ model_error: Optional[BaseException],
255
+ ) -> str:
256
+ """One machine-readable document for a ``check`` verdict.
257
+
258
+ Same conventions as ``run --json``: a block nobody could populate
259
+ serializes as ``null`` rather than being omitted, so a reader can tell
260
+ "measured nothing" from "nobody looked" -- ``verdict: null`` under
261
+ ``--baseline-only`` is the second of those, and it is the honest answer
262
+ when no model ever ran.
263
+
264
+ The resolved model and its price ride in the document, because ``check``'s
265
+ contract is that nobody is billed by a default they never saw, and a
266
+ machine-readable mode that dropped the price would quietly break it.
267
+ """
268
+ from agent_evolve.settings import AgentEvolveSettings, model_price
269
+
270
+ model_arm: Optional[dict] = None
271
+ resolved: Optional[str] = None
272
+ price = None
273
+ if not args.baseline_only:
274
+ resolved = args.model or AgentEvolveSettings.from_env().model
275
+ price = model_price(resolved)
276
+ if model_error is not None:
277
+ model_arm = {
278
+ "seed": 0,
279
+ "error": f"{type(model_error).__name__}: {model_error}",
280
+ }
281
+ else:
282
+ model_arm = _arm_outcome(model_result, objectives, seed=0)
283
+ model_arm["error"] = None
284
+
285
+ rows = (
286
+ _verdict_rows(model_result, baselines, objectives)
287
+ if model_result is not None else []
288
+ )
289
+ verdict = None
290
+ if model_arm is not None and model_error is None:
291
+ verdict = {
292
+ "objectives": rows,
293
+ "objectives_judged": len(rows),
294
+ "objectives_won": sum(1 for row in rows if row["model_won"]),
295
+ "winner": _winner(rows),
296
+ }
297
+
298
+ payload = {
299
+ "command": "check",
300
+ "problem": args.problem,
301
+ "budget": args.budget,
302
+ "repeats": args.repeats,
303
+ "baseline_only": bool(args.baseline_only),
304
+ "model": resolved,
305
+ "model_is_default": (None if resolved is None else args.model is None),
306
+ "model_price_per_mtok": (
307
+ None if price is None else {"input": price[0], "output": price[1]}
308
+ ),
309
+ "objectives": [
310
+ {"name": spec.name, "goal": spec.goal} for spec in objectives
311
+ ],
312
+ "arms": {
313
+ "baseline": {
314
+ "proposer": "random",
315
+ "runs": [
316
+ _arm_outcome(result, objectives, seed=i)
317
+ for i, result in enumerate(baselines)
318
+ ],
319
+ },
320
+ "model": model_arm,
321
+ },
322
+ "verdict": verdict,
323
+ "provider_usage": (
324
+ None if model_arm is None else model_arm.get("provider_usage")
325
+ ),
326
+ }
327
+ return json.dumps(payload, sort_keys=True, default=str)
328
+
329
+
330
+ def _cmd_check(args: argparse.Namespace) -> int:
331
+ from agent_evolve.api import optimize
332
+
333
+ problem = _load_problem(args.problem)
334
+ objectives = list(problem.objectives)
335
+ # `run --json`'s convention: one parseable document on stdout and nothing
336
+ # else. The prose does not disappear, it moves to stderr -- including the
337
+ # model's price, which `check` states BEFORE it spends, and which a
338
+ # machine-readable mode has no business making quieter. `2>/dev/null` then
339
+ # leaves exactly the document.
340
+ stream = sys.stderr if args.json else sys.stdout
341
+
342
+ def emit(message: str = "") -> None:
343
+ print(message, file=stream, flush=True)
344
+
345
+ quiet = (lambda _m: None) if not args.verbose else emit
346
+
347
+ emit(f"agent_evolve check: {args.problem}")
348
+ emit(f"budget {args.budget} evaluations per run, {args.repeats} baseline runs")
349
+ if args.baseline_only:
350
+ emit("baseline only: no model, no credentials, no cost\n")
351
+ else:
352
+ emit(f"{_model_line(args.model)}\n")
353
+
354
+ baselines = []
355
+ for i in range(args.repeats):
356
+ baselines.append(
357
+ optimize(problem, budget=args.budget, proposer="random", seed=i, on_progress=quiet)
358
+ )
359
+ emit(_summarise(f"random baseline (run 1 of {args.repeats})", baselines[0], objectives))
360
+
361
+ if args.baseline_only:
362
+ emit(
363
+ "\n Baseline only. Re-run without --baseline-only, with a provider "
364
+ "credential set, to compare a model against it."
365
+ )
366
+ if args.json:
367
+ print(_check_json(args, objectives, baselines, None, None))
368
+ return 0
369
+
370
+ emit()
371
+ try:
372
+ model_result = optimize(
373
+ problem,
374
+ budget=args.budget,
375
+ model=args.model,
376
+ proposer="llm",
377
+ seed=0,
378
+ on_progress=quiet,
379
+ )
380
+ except Exception as error: # noqa: BLE001 - reported, not raised, so the baseline still stands
381
+ emit(f" model run failed: {type(error).__name__}: {error}")
382
+ emit("\n The baseline above still stands, and cost nothing.")
383
+ if args.json:
384
+ print(_check_json(args, objectives, baselines, None, error))
385
+ return 1
386
+ emit(_summarise("model", model_result, objectives))
387
+ emit("\n verdict")
388
+ for line in _verdict(model_result, baselines, objectives):
389
+ emit(line)
390
+ if args.json:
391
+ print(_check_json(args, objectives, baselines, model_result, None))
392
+ return 0
393
+
394
+
395
+ def _cmd_diagnose(args: argparse.Namespace) -> int:
396
+ """Probe the problem itself: search space, pipeline health, and headroom.
397
+
398
+ Where ``check`` asks whether a *model* beats uninformed sampling, this asks
399
+ the prior question: whether *anything* could demonstrate an advantage on
400
+ this problem at this budget. It needs no model and no credentials, and it
401
+ spends at most ``--probe`` evaluations.
402
+ """
403
+ from agent_evolve.policies.check import check as check_problem
404
+
405
+ problem = _load_problem(args.problem)
406
+ print(f"agent_evolve diagnose: {args.problem}\n")
407
+ report = check_problem(problem, args.budget, probe=args.probe, seed=args.seed)
408
+ print(report.render())
409
+ return 0
410
+
411
+
412
+ def _cmd_run(args: argparse.Namespace) -> int:
413
+ from agent_evolve.api import optimize
414
+
415
+ problem = _load_problem(args.problem)
416
+ if args.json:
417
+ # One parseable document on stdout and nothing else: progress moves to
418
+ # stderr under --verbose and is dropped otherwise.
419
+ progress = (
420
+ (lambda m: print(m, file=sys.stderr, flush=True))
421
+ if args.verbose else (lambda _m: None)
422
+ )
423
+ else:
424
+ if args.proposer != "random":
425
+ print(_model_line(args.model))
426
+ progress = (lambda m: print(m, flush=True)) if args.verbose else print
427
+ result = optimize(
428
+ problem,
429
+ budget=args.budget,
430
+ model=args.model,
431
+ proposer=args.proposer,
432
+ strategy=args.strategy,
433
+ seed=args.seed,
434
+ seal=args.seal,
435
+ structure_budget=args.structure_budget,
436
+ prior=args.prior,
437
+ chooser=args.chooser,
438
+ effort=args.effort,
439
+ journal=args.journal,
440
+ authorship=args.authorship,
441
+ on_progress=progress,
442
+ )
443
+ if args.json:
444
+ from agent_evolve.core.formatting import result_to_json
445
+
446
+ print(result_to_json(result))
447
+ return 0
448
+ print(f"\nbest {result.best.configuration}")
449
+ print(f"objectives {result.best.objectives}")
450
+ print(f"pareto {len(result.pareto_front)}")
451
+ print(f"evaluations {result.evaluations}")
452
+ for i, c in enumerate(result.pareto_front, 1):
453
+ print(f" {i}. {c.configuration} -> {c.objectives}")
454
+ return 0
455
+
456
+
457
+ #: The five obligations with this project's problem removed and yours left to
458
+ #: write. It is the knapsack example's shape -- the same order, the same
459
+ #: comments about why each obligation exists -- with the knapsack taken out.
460
+ #:
461
+ #: It imports and it is a valid ``Problem`` the moment it lands, so ``diagnose``
462
+ #: can be run against it immediately; only ``evaluate`` refuses, by name,
463
+ #: because measuring is the one obligation nothing can guess for you. A template
464
+ #: that returned a plausible number instead would let a run look like it worked.
465
+ _SCAFFOLD = '''"""Your problem, as the five obligations ``agent_evolve`` asks for.
466
+
467
+ Fill in the parts marked TODO. The README section "Describing your problem:
468
+ five obligations" explains each one and what it buys you; the worked reference
469
+ is ``examples/knapsack/problem_def.py``.
470
+
471
+ candidate_model the schema a proposal must satisfy
472
+ objectives what is optimized, and which way
473
+ seeds() where to start
474
+ validate() cheap rejection that explains itself
475
+ materialize() candidate -> the artifact that gets measured
476
+ evaluate() artifact -> objective values
477
+
478
+ Then, before spending anything::
479
+
480
+ agent_evolve diagnose problem_def:problem --budget 40
481
+ agent_evolve run problem_def:problem --budget 40 --proposer random
482
+ """
483
+
484
+ from pydantic import BaseModel, Field
485
+
486
+ from agent_evolve import ObjectiveSpec, ValidationOutcome
487
+
488
+
489
+ class CandidateConfig(BaseModel):
490
+ """One candidate configuration.
491
+
492
+ TODO: your decision variables. Declare their domains here -- an enum, a
493
+ ``Literal``, a bounded number -- because everything that reads this schema,
494
+ including the uninformed sampler your run is measured against, draws only
495
+ from what it declares. A field with no finite reading is one the operators
496
+ leave frozen, and that is the commonest reason a run goes nowhere.
497
+ """
498
+
499
+ workers: int = Field(..., ge=1, le=64, description="TODO: describe this axis")
500
+
501
+
502
+ class MyProblem:
503
+ """TODO: one line saying what is being optimized, and under what limit."""
504
+
505
+ candidate_model = CandidateConfig
506
+
507
+ # -- 1. what is being optimized ---------------------------------------
508
+ @property
509
+ def objectives(self):
510
+ # TODO: name each objective and its direction. Direction is declared,
511
+ # never encoded by negating a value.
512
+ return [
513
+ ObjectiveSpec("throughput", "max"),
514
+ ObjectiveSpec("cost", "min"),
515
+ ]
516
+
517
+ # -- 2. where to start -------------------------------------------------
518
+ def seeds(self):
519
+ """Configurations you would have tried anyway.
520
+
521
+ Seeds are evaluated before anything is proposed, so the result answers
522
+ "did this beat what I already had" rather than leaving it assumed.
523
+ Return ``[]`` if you have none.
524
+ """
525
+ # TODO
526
+ return [{"workers": 8}]
527
+
528
+ # -- 3. cheap rejection that explains itself ---------------------------
529
+ def validate(self, config) -> ValidationOutcome:
530
+ """Reject what cannot work, and say what would.
531
+
532
+ The message is fed back to the proposer verbatim, so state what is
533
+ wrong AND what would be acceptable. A rejection costs no evaluation.
534
+ """
535
+ # TODO: your feasibility rules, e.g.
536
+ # return ValidationOutcome(
537
+ # False, "constraint",
538
+ # "workers above 32 needs the sharded strategy; reduce workers",
539
+ # )
540
+ return ValidationOutcome(True)
541
+
542
+ # -- 4. candidate -> the artifact that gets measured -------------------
543
+ def materialize(self, config):
544
+ """Canonicalise to the thing that actually gets measured.
545
+
546
+ Two configurations often produce the same artifact -- the same build,
547
+ the same mapping, the same deployment. Materializing first means the
548
+ second one is free instead of being paid for twice. Put anything cheap
549
+ and deterministic here and keep ``evaluate`` for the expensive part.
550
+ """
551
+ # TODO
552
+ return (config["workers"],)
553
+
554
+ # -- 5. measure it -----------------------------------------------------
555
+ def evaluate(self, artifact):
556
+ """Measure the artifact and return one value per objective."""
557
+ # TODO: run the build, the simulation, the benchmark -- the expensive
558
+ # thing. `budget` counts calls to this method, so this is what you are
559
+ # paying for.
560
+ raise NotImplementedError(
561
+ "evaluate() is the one obligation nothing can guess for you: "
562
+ "return {\\"throughput\\": ..., \\"cost\\": ...} for this artifact"
563
+ )
564
+
565
+ # -- optional: prose context for a model-driven proposer ---------------
566
+ def search_space_description(self):
567
+ # TODO, or delete: what a model should know about this space that the
568
+ # schema cannot say. Only the `llm` proposer reads it.
569
+ return ""
570
+
571
+
572
+ # `module:attribute` on the command line resolves to this name.
573
+ problem = MyProblem()
574
+ '''
575
+
576
+
577
+ def _cmd_init(args: argparse.Namespace) -> int:
578
+ """Write the five-obligation template, and refuse to overwrite anything.
579
+
580
+ A scaffold that clobbers is a scaffold nobody can run twice, and the file
581
+ it would clobber is the one thing in the directory nobody else can rewrite.
582
+ """
583
+ from pathlib import Path
584
+
585
+ target = Path(args.path)
586
+ if target.is_dir() or target.suffix != ".py":
587
+ target = target / "problem_def.py"
588
+ if target.exists():
589
+ raise SystemExit(
590
+ f"refusing to overwrite {target}. Name another path, or move the "
591
+ "existing file first."
592
+ )
593
+ target.parent.mkdir(parents=True, exist_ok=True)
594
+ target.write_text(_SCAFFOLD, encoding="utf-8")
595
+ module = target.stem
596
+ print(f"wrote {target}")
597
+ print("")
598
+ print("Fill in the parts marked TODO, then, before spending anything:")
599
+ print(f" agent_evolve diagnose {module}:problem --budget 40")
600
+ print(f" agent_evolve run {module}:problem --budget 40 --proposer random")
601
+ return 0
602
+
603
+
604
+ def _cmd_version(_args: argparse.Namespace) -> int:
605
+ try:
606
+ from importlib.metadata import version
607
+
608
+ # The DISTRIBUTION name; the import stays agent_evolve. See pyproject.
609
+ print(version("agentevolve-optimizer"))
610
+ except Exception:
611
+ print("unknown (not installed as a distribution)")
612
+ return 0
613
+
614
+
615
+ def main(argv: Optional[Sequence[str]] = None) -> int:
616
+ parser = argparse.ArgumentParser(
617
+ prog="agent_evolve",
618
+ description="Multi-objective optimization driven by a language model.",
619
+ )
620
+ sub = parser.add_subparsers(dest="command", required=True)
621
+
622
+ check = sub.add_parser(
623
+ "check",
624
+ help="does the model beat uninformed sampling on your problem?",
625
+ description=(
626
+ "Runs an uninformed sampler and a model against the same problem, "
627
+ "the same budget and the same evaluator, then says which won. Run "
628
+ "this before deciding to spend anything."
629
+ ),
630
+ )
631
+ check.add_argument("problem", help="module:attribute naming your problem")
632
+ check.add_argument("--budget", type=int, default=40, help="evaluations per run")
633
+ check.add_argument(
634
+ "--repeats", type=int, default=5,
635
+ help=(
636
+ "uninformed BASELINE runs (default 5, seeds 0..N-1). The model arm "
637
+ "is one run at seed 0 and this does not repeat it, so the spread "
638
+ "you see is chance's, not the model's"
639
+ ),
640
+ )
641
+ check.add_argument("--model", default=None, help="model id for the model arm")
642
+ check.add_argument(
643
+ "--baseline-only",
644
+ action="store_true",
645
+ help="run only the free baseline; no credentials needed",
646
+ )
647
+ check.add_argument(
648
+ "--json", action="store_true",
649
+ help=(
650
+ "print one machine-readable JSON document of the verdict on "
651
+ "stdout; the prose moves to stderr rather than being dropped"
652
+ ),
653
+ )
654
+ check.add_argument("--verbose", action="store_true")
655
+ check.set_defaults(func=_cmd_check)
656
+
657
+ diagnose = sub.add_parser(
658
+ "diagnose",
659
+ help="could ANY optimizer show an advantage on your problem at this budget?",
660
+ description=(
661
+ "Probes the problem with schema-uniform draws through its own "
662
+ "validate/materialize/evaluate pipeline and reports the locus and "
663
+ "domain structure, failure rate, evaluation cost, per-objective "
664
+ "spread, and whether best-of-budget random draws already reach the "
665
+ "best the probe found. Spends at most --probe evaluations; needs "
666
+ "no model and no credentials. Run it before `check`, which spends "
667
+ "model money to answer the next question."
668
+ ),
669
+ )
670
+ diagnose.add_argument("problem", help="module:attribute naming your problem")
671
+ diagnose.add_argument(
672
+ "--budget", type=int, default=40,
673
+ help="the optimizer budget being assessed (not the probe's spend)",
674
+ )
675
+ diagnose.add_argument(
676
+ "--probe", type=int, default=120,
677
+ help="schema-uniform draws the probe spends (default 120)",
678
+ )
679
+ diagnose.add_argument("--seed", type=int, default=None)
680
+ diagnose.set_defaults(func=_cmd_diagnose)
681
+
682
+ run = sub.add_parser("run", help="optimize a problem")
683
+ run.add_argument("problem", help="module:attribute naming your problem")
684
+ run.add_argument("--budget", type=int, default=40)
685
+ run.add_argument("--model", default=None)
686
+ run.add_argument(
687
+ "--proposer",
688
+ default="auto",
689
+ choices=("auto", "llm", "random"),
690
+ help="'random' needs no credentials and is the honest baseline",
691
+ )
692
+ run.add_argument(
693
+ "--strategy",
694
+ default="auto",
695
+ choices=("auto", "genetic", "authoring"),
696
+ help="'auto' prefers the genetic loop when the problem has seeds",
697
+ )
698
+ run.add_argument("--seed", type=int, default=None)
699
+ run.add_argument(
700
+ "--seal",
701
+ default=None,
702
+ metavar="PATH",
703
+ help=(
704
+ "write the run's chained proposal journal here; requires the "
705
+ "authoring strategy (the genetic loop refuses it by name)"
706
+ ),
707
+ )
708
+ run.add_argument(
709
+ "--structure-budget", type=_structure_budget_arg, default="auto",
710
+ dest="structure_budget",
711
+ help=(
712
+ "evaluations to spend on a crossed screen before the population; "
713
+ "charged against --budget, not free. 'auto' (default) skips it "
714
+ "below a budget of 48 and sizes it from the budget above that"
715
+ ),
716
+ )
717
+ run.add_argument(
718
+ "--prior",
719
+ default="auto",
720
+ choices=("auto", "rule", "rule-weighted", "llm", "llm-weighted"),
721
+ help=(
722
+ "who turns the screen into a sampling prior; the llm forms fall "
723
+ "back to their rule comparator, out loud, without a credential. "
724
+ "'auto' (default) is 'rule' offline and 'llm-weighted' on a model "
725
+ "run, announced either way"
726
+ ),
727
+ )
728
+ run.add_argument(
729
+ "--chooser",
730
+ default="off",
731
+ choices=("off", "llm"),
732
+ help=(
733
+ "who picks parents and cut points. 'llm' spends one model call per "
734
+ "offspring; it returned ten sealed null verdicts at 107-171x the "
735
+ "cost of the run it advises, and consumed 61%% of the six-arm "
736
+ "ablation's ledger for 0.94x the speed of doing nothing. 'off' "
737
+ "(default) is the random control it never beat"
738
+ ),
739
+ )
740
+ run.add_argument(
741
+ "--effort", default=None,
742
+ help="reasoning-effort pin for every model call (e.g. low, high)",
743
+ )
744
+ run.add_argument(
745
+ "--journal", default=None, metavar="PATH",
746
+ help="write one JSON line per completed model call (model, usage)",
747
+ )
748
+ run.add_argument(
749
+ "--authorship",
750
+ default="auto",
751
+ # Enumerated from the preset table, never repeated: a mechanism that
752
+ # is reachable from the library but not the CLI is half-shipped.
753
+ choices=("auto",) + tuple(_authorship_presets()),
754
+ help=(
755
+ "authored machinery: 'surrogate[-llm]' turns on virtual "
756
+ "pre-screening (model-written surrogates screen only when they "
757
+ "out-validate the rules); 'operators[-llm]' runs variation arms "
758
+ "under survival credit; 'generation-llm' lets the model write the "
759
+ "sampler every candidate is drawn from, 'generative' puts that "
760
+ "sampler under the authored screen; 'guided' is what 'auto' "
761
+ "resolves to on a model run (authored surrogate + model-proposed "
762
+ "initialization, the two measured winners); 'full' is surrogate + "
763
+ "operators + init, model-authored"
764
+ ),
765
+ )
766
+ run.add_argument(
767
+ "--json", action="store_true",
768
+ help="print one machine-readable JSON document instead of prose",
769
+ )
770
+ run.add_argument("--verbose", action="store_true")
771
+ run.set_defaults(func=_cmd_run)
772
+
773
+ init = sub.add_parser(
774
+ "init",
775
+ help="write a problem_def.py template: the five obligations, blank",
776
+ description=(
777
+ "Writes the five-obligation template -- the shipped knapsack "
778
+ "example's shape with the knapsack removed. PATH may be a "
779
+ "directory (problem_def.py is written inside it) or a .py file to "
780
+ "write. An existing file is never overwritten."
781
+ ),
782
+ )
783
+ init.add_argument(
784
+ "path", nargs="?", default=".",
785
+ help="directory to write problem_def.py into, or a .py path (default .)",
786
+ )
787
+ init.set_defaults(func=_cmd_init)
788
+
789
+ ver = sub.add_parser("version", help="print the installed version")
790
+ ver.set_defaults(func=_cmd_version)
791
+
792
+ args = parser.parse_args(argv)
793
+ return int(args.func(args))
794
+
795
+
796
+ if __name__ == "__main__": # pragma: no cover
797
+ raise SystemExit(main())