agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
agent_evolve/cli.py
ADDED
|
@@ -0,0 +1,797 @@
|
|
|
1
|
+
"""``agent_evolve`` command line: ``init``, ``diagnose``, ``check``, ``run``,
|
|
2
|
+
``version``.
|
|
3
|
+
|
|
4
|
+
``check`` is the one worth reading about. It runs the model against an
|
|
5
|
+
uninformed sampler on *your* problem, at the same budget, with the same
|
|
6
|
+
evaluator, and reports whether the model actually beat it.
|
|
7
|
+
|
|
8
|
+
That command exists because this project repeatedly measured how easy it is to
|
|
9
|
+
believe an optimizer is working when nothing is: on one benchmark a single
|
|
10
|
+
median random draw already accounted for roughly 80 percent of the
|
|
11
|
+
hypervolume that a full run produced, and the entire model-guided phase past
|
|
12
|
+
the initial design was worth about one percent. No amount of reading a paper
|
|
13
|
+
tells anyone whether their own problem is like that. Half a minute of
|
|
14
|
+
``agent_evolve check`` does.
|
|
15
|
+
|
|
16
|
+
A problem is named as ``module:attribute``, e.g.::
|
|
17
|
+
|
|
18
|
+
agent_evolve check examples.knapsack.problem_def:problem --budget 40
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import argparse
|
|
24
|
+
import importlib
|
|
25
|
+
import json
|
|
26
|
+
import statistics
|
|
27
|
+
import sys
|
|
28
|
+
from typing import Any, List, Optional, Sequence
|
|
29
|
+
|
|
30
|
+
from agent_evolve.contract import as_problem
|
|
31
|
+
from agent_evolve.core.results import compute_pareto_front
|
|
32
|
+
|
|
33
|
+
__all__ = ["main"]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _authorship_presets() -> tuple:
|
|
37
|
+
"""The preset names, read from the table that defines them."""
|
|
38
|
+
|
|
39
|
+
from agent_evolve.session.authorship import PRESETS
|
|
40
|
+
|
|
41
|
+
return tuple(PRESETS)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _structure_budget_arg(value: str) -> Any:
|
|
45
|
+
"""``auto`` or an evaluation count; the sentinel is resolved by the API.
|
|
46
|
+
|
|
47
|
+
argparse converts a string default through ``type``, so a plain ``int``
|
|
48
|
+
type would try to parse the sentinel itself. Resolving it here instead
|
|
49
|
+
would put a second copy of the sizing rule in the CLI, where it could
|
|
50
|
+
drift from the one the library announces.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
if value == "auto":
|
|
54
|
+
return "auto"
|
|
55
|
+
try:
|
|
56
|
+
return int(value)
|
|
57
|
+
except ValueError:
|
|
58
|
+
raise argparse.ArgumentTypeError(
|
|
59
|
+
"--structure-budget takes 'auto' or an evaluation count, got "
|
|
60
|
+
f"{value!r}"
|
|
61
|
+
) from None
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _load_problem(spec: str) -> Any:
|
|
65
|
+
"""Import ``module:attribute`` and return the problem object."""
|
|
66
|
+
if ":" not in spec:
|
|
67
|
+
raise SystemExit(
|
|
68
|
+
f"could not read {spec!r}. Name a problem as module:attribute, "
|
|
69
|
+
"e.g. examples.knapsack.problem_def:problem"
|
|
70
|
+
)
|
|
71
|
+
module_name, _, attribute = spec.partition(":")
|
|
72
|
+
sys.path.insert(0, "")
|
|
73
|
+
try:
|
|
74
|
+
module = importlib.import_module(module_name)
|
|
75
|
+
except ImportError as error:
|
|
76
|
+
raise SystemExit(f"could not import {module_name!r}: {error}") from error
|
|
77
|
+
try:
|
|
78
|
+
candidate = getattr(module, attribute)
|
|
79
|
+
except AttributeError as error:
|
|
80
|
+
raise SystemExit(f"{module_name!r} has no attribute {attribute!r}") from error
|
|
81
|
+
problem = candidate() if isinstance(candidate, type) else candidate
|
|
82
|
+
try:
|
|
83
|
+
return as_problem(problem)
|
|
84
|
+
except TypeError as error:
|
|
85
|
+
raise SystemExit(str(error)) from error
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _front_values(result: Any, objective: str) -> List[float]:
|
|
89
|
+
return [c.objectives[objective] for c in result.pareto_front if objective in c.objectives]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _summarise(label: str, result: Any, objectives: Sequence[Any]) -> str:
|
|
93
|
+
lines = [f" {label}:"]
|
|
94
|
+
lines.append(f" evaluations {result.evaluations}")
|
|
95
|
+
lines.append(f" pareto front {len(result.pareto_front)}")
|
|
96
|
+
for spec in objectives:
|
|
97
|
+
values = _front_values(result, spec.name)
|
|
98
|
+
if not values:
|
|
99
|
+
continue
|
|
100
|
+
best = max(values) if spec.goal == "max" else min(values)
|
|
101
|
+
lines.append(f" best {spec.name} ({spec.goal}) {best:g}")
|
|
102
|
+
return "\n".join(lines)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _verdict_rows(
|
|
106
|
+
model_result: Any,
|
|
107
|
+
baseline_results: Sequence[Any],
|
|
108
|
+
objectives: Sequence[Any],
|
|
109
|
+
) -> List[dict]:
|
|
110
|
+
"""The comparison itself, as data: one row per objective it could judge.
|
|
111
|
+
|
|
112
|
+
The prose verdict and the ``--json`` document are two renderings of this
|
|
113
|
+
one computation. A second copy of the rule -- which draws count, how ``p``
|
|
114
|
+
is formed, what "won" means -- could disagree with the first, and the
|
|
115
|
+
disagreement would be invisible until someone compared two outputs of the
|
|
116
|
+
same run.
|
|
117
|
+
"""
|
|
118
|
+
rows: List[dict] = []
|
|
119
|
+
for spec in objectives:
|
|
120
|
+
model_values = _front_values(model_result, spec.name)
|
|
121
|
+
if not model_values:
|
|
122
|
+
continue
|
|
123
|
+
pick = max if spec.goal == "max" else min
|
|
124
|
+
model_best = pick(model_values)
|
|
125
|
+
draws = [
|
|
126
|
+
pick(_front_values(r, spec.name))
|
|
127
|
+
for r in baseline_results
|
|
128
|
+
if _front_values(r, spec.name)
|
|
129
|
+
]
|
|
130
|
+
if not draws:
|
|
131
|
+
continue
|
|
132
|
+
better = sum(
|
|
133
|
+
1 for d in draws if (d > model_best if spec.goal == "max" else d < model_best)
|
|
134
|
+
)
|
|
135
|
+
# Fraction of uninformed runs that matched or beat the model. This is
|
|
136
|
+
# the chance baseline every claim in this project travels with.
|
|
137
|
+
p = (better + 1) / (len(draws) + 1)
|
|
138
|
+
median = statistics.median(draws)
|
|
139
|
+
won = (model_best > median) if spec.goal == "max" else (model_best < median)
|
|
140
|
+
rows.append({
|
|
141
|
+
"objective": spec.name,
|
|
142
|
+
"goal": spec.goal,
|
|
143
|
+
"model_best": model_best,
|
|
144
|
+
"random_median": median,
|
|
145
|
+
"baseline_runs": len(draws),
|
|
146
|
+
"baseline_better": better,
|
|
147
|
+
"p": p,
|
|
148
|
+
"model_won": won,
|
|
149
|
+
})
|
|
150
|
+
return rows
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _winner(rows: Sequence[dict]) -> Optional[str]:
|
|
154
|
+
"""``model``, ``baseline`` or ``mixed`` -- ``None`` when nothing was judged."""
|
|
155
|
+
if not rows:
|
|
156
|
+
return None
|
|
157
|
+
beaten = sum(1 for row in rows if row["model_won"])
|
|
158
|
+
if beaten == len(rows):
|
|
159
|
+
return "model"
|
|
160
|
+
if beaten == 0:
|
|
161
|
+
return "baseline"
|
|
162
|
+
return "mixed"
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _verdict(
|
|
166
|
+
model_result: Any,
|
|
167
|
+
baseline_results: Sequence[Any],
|
|
168
|
+
objectives: Sequence[Any],
|
|
169
|
+
) -> List[str]:
|
|
170
|
+
"""State plainly whether the model beat chance, per objective."""
|
|
171
|
+
out: List[str] = []
|
|
172
|
+
rows = _verdict_rows(model_result, baseline_results, objectives)
|
|
173
|
+
counted = len(rows)
|
|
174
|
+
beaten = sum(1 for row in rows if row["model_won"])
|
|
175
|
+
for row in rows:
|
|
176
|
+
out.append(
|
|
177
|
+
f" {row['objective']}: model {row['model_best']:g} vs random "
|
|
178
|
+
f"median {row['random_median']:g} over {row['baseline_runs']} runs "
|
|
179
|
+
f" (p = {row['p']:.2f})"
|
|
180
|
+
)
|
|
181
|
+
if counted:
|
|
182
|
+
out.append("")
|
|
183
|
+
if beaten == counted:
|
|
184
|
+
out.append(" The model beat the uninformed baseline on every objective.")
|
|
185
|
+
elif beaten == 0:
|
|
186
|
+
out.append(
|
|
187
|
+
" The model did not beat the uninformed baseline on any objective.\n"
|
|
188
|
+
" On this problem, at this budget, it is not earning its cost."
|
|
189
|
+
)
|
|
190
|
+
else:
|
|
191
|
+
out.append(
|
|
192
|
+
f" The model beat the uninformed baseline on {beaten} of "
|
|
193
|
+
f"{counted} objectives. Mixed, and worth a larger budget "
|
|
194
|
+
"before concluding either way."
|
|
195
|
+
)
|
|
196
|
+
out.append(
|
|
197
|
+
" A p near 1.00 means uninformed sampling routinely does as well.\n"
|
|
198
|
+
" Few runs is weak evidence; raise --repeats to sharpen it."
|
|
199
|
+
)
|
|
200
|
+
return out
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def _model_line(model: Optional[str]) -> str:
|
|
204
|
+
"""Name the model and its price before anything is billed.
|
|
205
|
+
|
|
206
|
+
A default nobody saw is a default nobody consented to, so the resolved
|
|
207
|
+
model is printed whether or not the caller chose it, and it is marked as a
|
|
208
|
+
default when they did not.
|
|
209
|
+
"""
|
|
210
|
+
from agent_evolve.settings import AgentEvolveSettings, model_price
|
|
211
|
+
|
|
212
|
+
resolved = model or AgentEvolveSettings.from_env().model
|
|
213
|
+
origin = "" if model else " (default)"
|
|
214
|
+
price = model_price(resolved)
|
|
215
|
+
cost = (
|
|
216
|
+
f" ${price[0]:.2f}/M in, ${price[1]:.2f}/M out"
|
|
217
|
+
if price
|
|
218
|
+
else " price unknown"
|
|
219
|
+
)
|
|
220
|
+
return f"model {resolved}{origin}{cost}"
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _arm_outcome(result: Any, objectives: Sequence[Any], seed: int) -> dict:
|
|
224
|
+
"""One arm's run as data: what it spent and the best it reached.
|
|
225
|
+
|
|
226
|
+
The Pareto front's rows are deliberately not here -- ``run --json`` is the
|
|
227
|
+
command that hands you a front, and duplicating it under five baseline
|
|
228
|
+
repeats would bury the one thing ``check`` exists to answer.
|
|
229
|
+
"""
|
|
230
|
+
import dataclasses
|
|
231
|
+
|
|
232
|
+
best = {}
|
|
233
|
+
for spec in objectives:
|
|
234
|
+
values = _front_values(result, spec.name)
|
|
235
|
+
if values:
|
|
236
|
+
best[spec.name] = (max if spec.goal == "max" else min)(values)
|
|
237
|
+
return {
|
|
238
|
+
"seed": seed,
|
|
239
|
+
"evaluations": result.evaluations,
|
|
240
|
+
"pareto_front": len(result.pareto_front),
|
|
241
|
+
"best": best,
|
|
242
|
+
"provider_usage": (
|
|
243
|
+
dataclasses.asdict(result.provider_usage)
|
|
244
|
+
if result.provider_usage is not None else None
|
|
245
|
+
),
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _check_json(
|
|
250
|
+
args: argparse.Namespace,
|
|
251
|
+
objectives: Sequence[Any],
|
|
252
|
+
baselines: Sequence[Any],
|
|
253
|
+
model_result: Any,
|
|
254
|
+
model_error: Optional[BaseException],
|
|
255
|
+
) -> str:
|
|
256
|
+
"""One machine-readable document for a ``check`` verdict.
|
|
257
|
+
|
|
258
|
+
Same conventions as ``run --json``: a block nobody could populate
|
|
259
|
+
serializes as ``null`` rather than being omitted, so a reader can tell
|
|
260
|
+
"measured nothing" from "nobody looked" -- ``verdict: null`` under
|
|
261
|
+
``--baseline-only`` is the second of those, and it is the honest answer
|
|
262
|
+
when no model ever ran.
|
|
263
|
+
|
|
264
|
+
The resolved model and its price ride in the document, because ``check``'s
|
|
265
|
+
contract is that nobody is billed by a default they never saw, and a
|
|
266
|
+
machine-readable mode that dropped the price would quietly break it.
|
|
267
|
+
"""
|
|
268
|
+
from agent_evolve.settings import AgentEvolveSettings, model_price
|
|
269
|
+
|
|
270
|
+
model_arm: Optional[dict] = None
|
|
271
|
+
resolved: Optional[str] = None
|
|
272
|
+
price = None
|
|
273
|
+
if not args.baseline_only:
|
|
274
|
+
resolved = args.model or AgentEvolveSettings.from_env().model
|
|
275
|
+
price = model_price(resolved)
|
|
276
|
+
if model_error is not None:
|
|
277
|
+
model_arm = {
|
|
278
|
+
"seed": 0,
|
|
279
|
+
"error": f"{type(model_error).__name__}: {model_error}",
|
|
280
|
+
}
|
|
281
|
+
else:
|
|
282
|
+
model_arm = _arm_outcome(model_result, objectives, seed=0)
|
|
283
|
+
model_arm["error"] = None
|
|
284
|
+
|
|
285
|
+
rows = (
|
|
286
|
+
_verdict_rows(model_result, baselines, objectives)
|
|
287
|
+
if model_result is not None else []
|
|
288
|
+
)
|
|
289
|
+
verdict = None
|
|
290
|
+
if model_arm is not None and model_error is None:
|
|
291
|
+
verdict = {
|
|
292
|
+
"objectives": rows,
|
|
293
|
+
"objectives_judged": len(rows),
|
|
294
|
+
"objectives_won": sum(1 for row in rows if row["model_won"]),
|
|
295
|
+
"winner": _winner(rows),
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
payload = {
|
|
299
|
+
"command": "check",
|
|
300
|
+
"problem": args.problem,
|
|
301
|
+
"budget": args.budget,
|
|
302
|
+
"repeats": args.repeats,
|
|
303
|
+
"baseline_only": bool(args.baseline_only),
|
|
304
|
+
"model": resolved,
|
|
305
|
+
"model_is_default": (None if resolved is None else args.model is None),
|
|
306
|
+
"model_price_per_mtok": (
|
|
307
|
+
None if price is None else {"input": price[0], "output": price[1]}
|
|
308
|
+
),
|
|
309
|
+
"objectives": [
|
|
310
|
+
{"name": spec.name, "goal": spec.goal} for spec in objectives
|
|
311
|
+
],
|
|
312
|
+
"arms": {
|
|
313
|
+
"baseline": {
|
|
314
|
+
"proposer": "random",
|
|
315
|
+
"runs": [
|
|
316
|
+
_arm_outcome(result, objectives, seed=i)
|
|
317
|
+
for i, result in enumerate(baselines)
|
|
318
|
+
],
|
|
319
|
+
},
|
|
320
|
+
"model": model_arm,
|
|
321
|
+
},
|
|
322
|
+
"verdict": verdict,
|
|
323
|
+
"provider_usage": (
|
|
324
|
+
None if model_arm is None else model_arm.get("provider_usage")
|
|
325
|
+
),
|
|
326
|
+
}
|
|
327
|
+
return json.dumps(payload, sort_keys=True, default=str)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _cmd_check(args: argparse.Namespace) -> int:
|
|
331
|
+
from agent_evolve.api import optimize
|
|
332
|
+
|
|
333
|
+
problem = _load_problem(args.problem)
|
|
334
|
+
objectives = list(problem.objectives)
|
|
335
|
+
# `run --json`'s convention: one parseable document on stdout and nothing
|
|
336
|
+
# else. The prose does not disappear, it moves to stderr -- including the
|
|
337
|
+
# model's price, which `check` states BEFORE it spends, and which a
|
|
338
|
+
# machine-readable mode has no business making quieter. `2>/dev/null` then
|
|
339
|
+
# leaves exactly the document.
|
|
340
|
+
stream = sys.stderr if args.json else sys.stdout
|
|
341
|
+
|
|
342
|
+
def emit(message: str = "") -> None:
|
|
343
|
+
print(message, file=stream, flush=True)
|
|
344
|
+
|
|
345
|
+
quiet = (lambda _m: None) if not args.verbose else emit
|
|
346
|
+
|
|
347
|
+
emit(f"agent_evolve check: {args.problem}")
|
|
348
|
+
emit(f"budget {args.budget} evaluations per run, {args.repeats} baseline runs")
|
|
349
|
+
if args.baseline_only:
|
|
350
|
+
emit("baseline only: no model, no credentials, no cost\n")
|
|
351
|
+
else:
|
|
352
|
+
emit(f"{_model_line(args.model)}\n")
|
|
353
|
+
|
|
354
|
+
baselines = []
|
|
355
|
+
for i in range(args.repeats):
|
|
356
|
+
baselines.append(
|
|
357
|
+
optimize(problem, budget=args.budget, proposer="random", seed=i, on_progress=quiet)
|
|
358
|
+
)
|
|
359
|
+
emit(_summarise(f"random baseline (run 1 of {args.repeats})", baselines[0], objectives))
|
|
360
|
+
|
|
361
|
+
if args.baseline_only:
|
|
362
|
+
emit(
|
|
363
|
+
"\n Baseline only. Re-run without --baseline-only, with a provider "
|
|
364
|
+
"credential set, to compare a model against it."
|
|
365
|
+
)
|
|
366
|
+
if args.json:
|
|
367
|
+
print(_check_json(args, objectives, baselines, None, None))
|
|
368
|
+
return 0
|
|
369
|
+
|
|
370
|
+
emit()
|
|
371
|
+
try:
|
|
372
|
+
model_result = optimize(
|
|
373
|
+
problem,
|
|
374
|
+
budget=args.budget,
|
|
375
|
+
model=args.model,
|
|
376
|
+
proposer="llm",
|
|
377
|
+
seed=0,
|
|
378
|
+
on_progress=quiet,
|
|
379
|
+
)
|
|
380
|
+
except Exception as error: # noqa: BLE001 - reported, not raised, so the baseline still stands
|
|
381
|
+
emit(f" model run failed: {type(error).__name__}: {error}")
|
|
382
|
+
emit("\n The baseline above still stands, and cost nothing.")
|
|
383
|
+
if args.json:
|
|
384
|
+
print(_check_json(args, objectives, baselines, None, error))
|
|
385
|
+
return 1
|
|
386
|
+
emit(_summarise("model", model_result, objectives))
|
|
387
|
+
emit("\n verdict")
|
|
388
|
+
for line in _verdict(model_result, baselines, objectives):
|
|
389
|
+
emit(line)
|
|
390
|
+
if args.json:
|
|
391
|
+
print(_check_json(args, objectives, baselines, model_result, None))
|
|
392
|
+
return 0
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _cmd_diagnose(args: argparse.Namespace) -> int:
|
|
396
|
+
"""Probe the problem itself: search space, pipeline health, and headroom.
|
|
397
|
+
|
|
398
|
+
Where ``check`` asks whether a *model* beats uninformed sampling, this asks
|
|
399
|
+
the prior question: whether *anything* could demonstrate an advantage on
|
|
400
|
+
this problem at this budget. It needs no model and no credentials, and it
|
|
401
|
+
spends at most ``--probe`` evaluations.
|
|
402
|
+
"""
|
|
403
|
+
from agent_evolve.policies.check import check as check_problem
|
|
404
|
+
|
|
405
|
+
problem = _load_problem(args.problem)
|
|
406
|
+
print(f"agent_evolve diagnose: {args.problem}\n")
|
|
407
|
+
report = check_problem(problem, args.budget, probe=args.probe, seed=args.seed)
|
|
408
|
+
print(report.render())
|
|
409
|
+
return 0
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _cmd_run(args: argparse.Namespace) -> int:
|
|
413
|
+
from agent_evolve.api import optimize
|
|
414
|
+
|
|
415
|
+
problem = _load_problem(args.problem)
|
|
416
|
+
if args.json:
|
|
417
|
+
# One parseable document on stdout and nothing else: progress moves to
|
|
418
|
+
# stderr under --verbose and is dropped otherwise.
|
|
419
|
+
progress = (
|
|
420
|
+
(lambda m: print(m, file=sys.stderr, flush=True))
|
|
421
|
+
if args.verbose else (lambda _m: None)
|
|
422
|
+
)
|
|
423
|
+
else:
|
|
424
|
+
if args.proposer != "random":
|
|
425
|
+
print(_model_line(args.model))
|
|
426
|
+
progress = (lambda m: print(m, flush=True)) if args.verbose else print
|
|
427
|
+
result = optimize(
|
|
428
|
+
problem,
|
|
429
|
+
budget=args.budget,
|
|
430
|
+
model=args.model,
|
|
431
|
+
proposer=args.proposer,
|
|
432
|
+
strategy=args.strategy,
|
|
433
|
+
seed=args.seed,
|
|
434
|
+
seal=args.seal,
|
|
435
|
+
structure_budget=args.structure_budget,
|
|
436
|
+
prior=args.prior,
|
|
437
|
+
chooser=args.chooser,
|
|
438
|
+
effort=args.effort,
|
|
439
|
+
journal=args.journal,
|
|
440
|
+
authorship=args.authorship,
|
|
441
|
+
on_progress=progress,
|
|
442
|
+
)
|
|
443
|
+
if args.json:
|
|
444
|
+
from agent_evolve.core.formatting import result_to_json
|
|
445
|
+
|
|
446
|
+
print(result_to_json(result))
|
|
447
|
+
return 0
|
|
448
|
+
print(f"\nbest {result.best.configuration}")
|
|
449
|
+
print(f"objectives {result.best.objectives}")
|
|
450
|
+
print(f"pareto {len(result.pareto_front)}")
|
|
451
|
+
print(f"evaluations {result.evaluations}")
|
|
452
|
+
for i, c in enumerate(result.pareto_front, 1):
|
|
453
|
+
print(f" {i}. {c.configuration} -> {c.objectives}")
|
|
454
|
+
return 0
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
#: The five obligations with this project's problem removed and yours left to
|
|
458
|
+
#: write. It is the knapsack example's shape -- the same order, the same
|
|
459
|
+
#: comments about why each obligation exists -- with the knapsack taken out.
|
|
460
|
+
#:
|
|
461
|
+
#: It imports and it is a valid ``Problem`` the moment it lands, so ``diagnose``
|
|
462
|
+
#: can be run against it immediately; only ``evaluate`` refuses, by name,
|
|
463
|
+
#: because measuring is the one obligation nothing can guess for you. A template
|
|
464
|
+
#: that returned a plausible number instead would let a run look like it worked.
|
|
465
|
+
_SCAFFOLD = '''"""Your problem, as the five obligations ``agent_evolve`` asks for.
|
|
466
|
+
|
|
467
|
+
Fill in the parts marked TODO. The README section "Describing your problem:
|
|
468
|
+
five obligations" explains each one and what it buys you; the worked reference
|
|
469
|
+
is ``examples/knapsack/problem_def.py``.
|
|
470
|
+
|
|
471
|
+
candidate_model the schema a proposal must satisfy
|
|
472
|
+
objectives what is optimized, and which way
|
|
473
|
+
seeds() where to start
|
|
474
|
+
validate() cheap rejection that explains itself
|
|
475
|
+
materialize() candidate -> the artifact that gets measured
|
|
476
|
+
evaluate() artifact -> objective values
|
|
477
|
+
|
|
478
|
+
Then, before spending anything::
|
|
479
|
+
|
|
480
|
+
agent_evolve diagnose problem_def:problem --budget 40
|
|
481
|
+
agent_evolve run problem_def:problem --budget 40 --proposer random
|
|
482
|
+
"""
|
|
483
|
+
|
|
484
|
+
from pydantic import BaseModel, Field
|
|
485
|
+
|
|
486
|
+
from agent_evolve import ObjectiveSpec, ValidationOutcome
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
class CandidateConfig(BaseModel):
|
|
490
|
+
"""One candidate configuration.
|
|
491
|
+
|
|
492
|
+
TODO: your decision variables. Declare their domains here -- an enum, a
|
|
493
|
+
``Literal``, a bounded number -- because everything that reads this schema,
|
|
494
|
+
including the uninformed sampler your run is measured against, draws only
|
|
495
|
+
from what it declares. A field with no finite reading is one the operators
|
|
496
|
+
leave frozen, and that is the commonest reason a run goes nowhere.
|
|
497
|
+
"""
|
|
498
|
+
|
|
499
|
+
workers: int = Field(..., ge=1, le=64, description="TODO: describe this axis")
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
class MyProblem:
|
|
503
|
+
"""TODO: one line saying what is being optimized, and under what limit."""
|
|
504
|
+
|
|
505
|
+
candidate_model = CandidateConfig
|
|
506
|
+
|
|
507
|
+
# -- 1. what is being optimized ---------------------------------------
|
|
508
|
+
@property
|
|
509
|
+
def objectives(self):
|
|
510
|
+
# TODO: name each objective and its direction. Direction is declared,
|
|
511
|
+
# never encoded by negating a value.
|
|
512
|
+
return [
|
|
513
|
+
ObjectiveSpec("throughput", "max"),
|
|
514
|
+
ObjectiveSpec("cost", "min"),
|
|
515
|
+
]
|
|
516
|
+
|
|
517
|
+
# -- 2. where to start -------------------------------------------------
|
|
518
|
+
def seeds(self):
|
|
519
|
+
"""Configurations you would have tried anyway.
|
|
520
|
+
|
|
521
|
+
Seeds are evaluated before anything is proposed, so the result answers
|
|
522
|
+
"did this beat what I already had" rather than leaving it assumed.
|
|
523
|
+
Return ``[]`` if you have none.
|
|
524
|
+
"""
|
|
525
|
+
# TODO
|
|
526
|
+
return [{"workers": 8}]
|
|
527
|
+
|
|
528
|
+
# -- 3. cheap rejection that explains itself ---------------------------
|
|
529
|
+
def validate(self, config) -> ValidationOutcome:
|
|
530
|
+
"""Reject what cannot work, and say what would.
|
|
531
|
+
|
|
532
|
+
The message is fed back to the proposer verbatim, so state what is
|
|
533
|
+
wrong AND what would be acceptable. A rejection costs no evaluation.
|
|
534
|
+
"""
|
|
535
|
+
# TODO: your feasibility rules, e.g.
|
|
536
|
+
# return ValidationOutcome(
|
|
537
|
+
# False, "constraint",
|
|
538
|
+
# "workers above 32 needs the sharded strategy; reduce workers",
|
|
539
|
+
# )
|
|
540
|
+
return ValidationOutcome(True)
|
|
541
|
+
|
|
542
|
+
# -- 4. candidate -> the artifact that gets measured -------------------
|
|
543
|
+
def materialize(self, config):
|
|
544
|
+
"""Canonicalise to the thing that actually gets measured.
|
|
545
|
+
|
|
546
|
+
Two configurations often produce the same artifact -- the same build,
|
|
547
|
+
the same mapping, the same deployment. Materializing first means the
|
|
548
|
+
second one is free instead of being paid for twice. Put anything cheap
|
|
549
|
+
and deterministic here and keep ``evaluate`` for the expensive part.
|
|
550
|
+
"""
|
|
551
|
+
# TODO
|
|
552
|
+
return (config["workers"],)
|
|
553
|
+
|
|
554
|
+
# -- 5. measure it -----------------------------------------------------
|
|
555
|
+
def evaluate(self, artifact):
|
|
556
|
+
"""Measure the artifact and return one value per objective."""
|
|
557
|
+
# TODO: run the build, the simulation, the benchmark -- the expensive
|
|
558
|
+
# thing. `budget` counts calls to this method, so this is what you are
|
|
559
|
+
# paying for.
|
|
560
|
+
raise NotImplementedError(
|
|
561
|
+
"evaluate() is the one obligation nothing can guess for you: "
|
|
562
|
+
"return {\\"throughput\\": ..., \\"cost\\": ...} for this artifact"
|
|
563
|
+
)
|
|
564
|
+
|
|
565
|
+
# -- optional: prose context for a model-driven proposer ---------------
|
|
566
|
+
def search_space_description(self):
|
|
567
|
+
# TODO, or delete: what a model should know about this space that the
|
|
568
|
+
# schema cannot say. Only the `llm` proposer reads it.
|
|
569
|
+
return ""
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
# `module:attribute` on the command line resolves to this name.
|
|
573
|
+
problem = MyProblem()
|
|
574
|
+
'''
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def _cmd_init(args: argparse.Namespace) -> int:
|
|
578
|
+
"""Write the five-obligation template, and refuse to overwrite anything.
|
|
579
|
+
|
|
580
|
+
A scaffold that clobbers is a scaffold nobody can run twice, and the file
|
|
581
|
+
it would clobber is the one thing in the directory nobody else can rewrite.
|
|
582
|
+
"""
|
|
583
|
+
from pathlib import Path
|
|
584
|
+
|
|
585
|
+
target = Path(args.path)
|
|
586
|
+
if target.is_dir() or target.suffix != ".py":
|
|
587
|
+
target = target / "problem_def.py"
|
|
588
|
+
if target.exists():
|
|
589
|
+
raise SystemExit(
|
|
590
|
+
f"refusing to overwrite {target}. Name another path, or move the "
|
|
591
|
+
"existing file first."
|
|
592
|
+
)
|
|
593
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
594
|
+
target.write_text(_SCAFFOLD, encoding="utf-8")
|
|
595
|
+
module = target.stem
|
|
596
|
+
print(f"wrote {target}")
|
|
597
|
+
print("")
|
|
598
|
+
print("Fill in the parts marked TODO, then, before spending anything:")
|
|
599
|
+
print(f" agent_evolve diagnose {module}:problem --budget 40")
|
|
600
|
+
print(f" agent_evolve run {module}:problem --budget 40 --proposer random")
|
|
601
|
+
return 0
|
|
602
|
+
|
|
603
|
+
|
|
604
|
+
def _cmd_version(_args: argparse.Namespace) -> int:
|
|
605
|
+
try:
|
|
606
|
+
from importlib.metadata import version
|
|
607
|
+
|
|
608
|
+
# The DISTRIBUTION name; the import stays agent_evolve. See pyproject.
|
|
609
|
+
print(version("agentevolve-optimizer"))
|
|
610
|
+
except Exception:
|
|
611
|
+
print("unknown (not installed as a distribution)")
|
|
612
|
+
return 0
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
616
|
+
parser = argparse.ArgumentParser(
|
|
617
|
+
prog="agent_evolve",
|
|
618
|
+
description="Multi-objective optimization driven by a language model.",
|
|
619
|
+
)
|
|
620
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
621
|
+
|
|
622
|
+
check = sub.add_parser(
|
|
623
|
+
"check",
|
|
624
|
+
help="does the model beat uninformed sampling on your problem?",
|
|
625
|
+
description=(
|
|
626
|
+
"Runs an uninformed sampler and a model against the same problem, "
|
|
627
|
+
"the same budget and the same evaluator, then says which won. Run "
|
|
628
|
+
"this before deciding to spend anything."
|
|
629
|
+
),
|
|
630
|
+
)
|
|
631
|
+
check.add_argument("problem", help="module:attribute naming your problem")
|
|
632
|
+
check.add_argument("--budget", type=int, default=40, help="evaluations per run")
|
|
633
|
+
check.add_argument(
|
|
634
|
+
"--repeats", type=int, default=5,
|
|
635
|
+
help=(
|
|
636
|
+
"uninformed BASELINE runs (default 5, seeds 0..N-1). The model arm "
|
|
637
|
+
"is one run at seed 0 and this does not repeat it, so the spread "
|
|
638
|
+
"you see is chance's, not the model's"
|
|
639
|
+
),
|
|
640
|
+
)
|
|
641
|
+
check.add_argument("--model", default=None, help="model id for the model arm")
|
|
642
|
+
check.add_argument(
|
|
643
|
+
"--baseline-only",
|
|
644
|
+
action="store_true",
|
|
645
|
+
help="run only the free baseline; no credentials needed",
|
|
646
|
+
)
|
|
647
|
+
check.add_argument(
|
|
648
|
+
"--json", action="store_true",
|
|
649
|
+
help=(
|
|
650
|
+
"print one machine-readable JSON document of the verdict on "
|
|
651
|
+
"stdout; the prose moves to stderr rather than being dropped"
|
|
652
|
+
),
|
|
653
|
+
)
|
|
654
|
+
check.add_argument("--verbose", action="store_true")
|
|
655
|
+
check.set_defaults(func=_cmd_check)
|
|
656
|
+
|
|
657
|
+
diagnose = sub.add_parser(
|
|
658
|
+
"diagnose",
|
|
659
|
+
help="could ANY optimizer show an advantage on your problem at this budget?",
|
|
660
|
+
description=(
|
|
661
|
+
"Probes the problem with schema-uniform draws through its own "
|
|
662
|
+
"validate/materialize/evaluate pipeline and reports the locus and "
|
|
663
|
+
"domain structure, failure rate, evaluation cost, per-objective "
|
|
664
|
+
"spread, and whether best-of-budget random draws already reach the "
|
|
665
|
+
"best the probe found. Spends at most --probe evaluations; needs "
|
|
666
|
+
"no model and no credentials. Run it before `check`, which spends "
|
|
667
|
+
"model money to answer the next question."
|
|
668
|
+
),
|
|
669
|
+
)
|
|
670
|
+
diagnose.add_argument("problem", help="module:attribute naming your problem")
|
|
671
|
+
diagnose.add_argument(
|
|
672
|
+
"--budget", type=int, default=40,
|
|
673
|
+
help="the optimizer budget being assessed (not the probe's spend)",
|
|
674
|
+
)
|
|
675
|
+
diagnose.add_argument(
|
|
676
|
+
"--probe", type=int, default=120,
|
|
677
|
+
help="schema-uniform draws the probe spends (default 120)",
|
|
678
|
+
)
|
|
679
|
+
diagnose.add_argument("--seed", type=int, default=None)
|
|
680
|
+
diagnose.set_defaults(func=_cmd_diagnose)
|
|
681
|
+
|
|
682
|
+
run = sub.add_parser("run", help="optimize a problem")
|
|
683
|
+
run.add_argument("problem", help="module:attribute naming your problem")
|
|
684
|
+
run.add_argument("--budget", type=int, default=40)
|
|
685
|
+
run.add_argument("--model", default=None)
|
|
686
|
+
run.add_argument(
|
|
687
|
+
"--proposer",
|
|
688
|
+
default="auto",
|
|
689
|
+
choices=("auto", "llm", "random"),
|
|
690
|
+
help="'random' needs no credentials and is the honest baseline",
|
|
691
|
+
)
|
|
692
|
+
run.add_argument(
|
|
693
|
+
"--strategy",
|
|
694
|
+
default="auto",
|
|
695
|
+
choices=("auto", "genetic", "authoring"),
|
|
696
|
+
help="'auto' prefers the genetic loop when the problem has seeds",
|
|
697
|
+
)
|
|
698
|
+
run.add_argument("--seed", type=int, default=None)
|
|
699
|
+
run.add_argument(
|
|
700
|
+
"--seal",
|
|
701
|
+
default=None,
|
|
702
|
+
metavar="PATH",
|
|
703
|
+
help=(
|
|
704
|
+
"write the run's chained proposal journal here; requires the "
|
|
705
|
+
"authoring strategy (the genetic loop refuses it by name)"
|
|
706
|
+
),
|
|
707
|
+
)
|
|
708
|
+
run.add_argument(
|
|
709
|
+
"--structure-budget", type=_structure_budget_arg, default="auto",
|
|
710
|
+
dest="structure_budget",
|
|
711
|
+
help=(
|
|
712
|
+
"evaluations to spend on a crossed screen before the population; "
|
|
713
|
+
"charged against --budget, not free. 'auto' (default) skips it "
|
|
714
|
+
"below a budget of 48 and sizes it from the budget above that"
|
|
715
|
+
),
|
|
716
|
+
)
|
|
717
|
+
run.add_argument(
|
|
718
|
+
"--prior",
|
|
719
|
+
default="auto",
|
|
720
|
+
choices=("auto", "rule", "rule-weighted", "llm", "llm-weighted"),
|
|
721
|
+
help=(
|
|
722
|
+
"who turns the screen into a sampling prior; the llm forms fall "
|
|
723
|
+
"back to their rule comparator, out loud, without a credential. "
|
|
724
|
+
"'auto' (default) is 'rule' offline and 'llm-weighted' on a model "
|
|
725
|
+
"run, announced either way"
|
|
726
|
+
),
|
|
727
|
+
)
|
|
728
|
+
run.add_argument(
|
|
729
|
+
"--chooser",
|
|
730
|
+
default="off",
|
|
731
|
+
choices=("off", "llm"),
|
|
732
|
+
help=(
|
|
733
|
+
"who picks parents and cut points. 'llm' spends one model call per "
|
|
734
|
+
"offspring; it returned ten sealed null verdicts at 107-171x the "
|
|
735
|
+
"cost of the run it advises, and consumed 61%% of the six-arm "
|
|
736
|
+
"ablation's ledger for 0.94x the speed of doing nothing. 'off' "
|
|
737
|
+
"(default) is the random control it never beat"
|
|
738
|
+
),
|
|
739
|
+
)
|
|
740
|
+
run.add_argument(
|
|
741
|
+
"--effort", default=None,
|
|
742
|
+
help="reasoning-effort pin for every model call (e.g. low, high)",
|
|
743
|
+
)
|
|
744
|
+
run.add_argument(
|
|
745
|
+
"--journal", default=None, metavar="PATH",
|
|
746
|
+
help="write one JSON line per completed model call (model, usage)",
|
|
747
|
+
)
|
|
748
|
+
run.add_argument(
|
|
749
|
+
"--authorship",
|
|
750
|
+
default="auto",
|
|
751
|
+
# Enumerated from the preset table, never repeated: a mechanism that
|
|
752
|
+
# is reachable from the library but not the CLI is half-shipped.
|
|
753
|
+
choices=("auto",) + tuple(_authorship_presets()),
|
|
754
|
+
help=(
|
|
755
|
+
"authored machinery: 'surrogate[-llm]' turns on virtual "
|
|
756
|
+
"pre-screening (model-written surrogates screen only when they "
|
|
757
|
+
"out-validate the rules); 'operators[-llm]' runs variation arms "
|
|
758
|
+
"under survival credit; 'generation-llm' lets the model write the "
|
|
759
|
+
"sampler every candidate is drawn from, 'generative' puts that "
|
|
760
|
+
"sampler under the authored screen; 'guided' is what 'auto' "
|
|
761
|
+
"resolves to on a model run (authored surrogate + model-proposed "
|
|
762
|
+
"initialization, the two measured winners); 'full' is surrogate + "
|
|
763
|
+
"operators + init, model-authored"
|
|
764
|
+
),
|
|
765
|
+
)
|
|
766
|
+
run.add_argument(
|
|
767
|
+
"--json", action="store_true",
|
|
768
|
+
help="print one machine-readable JSON document instead of prose",
|
|
769
|
+
)
|
|
770
|
+
run.add_argument("--verbose", action="store_true")
|
|
771
|
+
run.set_defaults(func=_cmd_run)
|
|
772
|
+
|
|
773
|
+
init = sub.add_parser(
|
|
774
|
+
"init",
|
|
775
|
+
help="write a problem_def.py template: the five obligations, blank",
|
|
776
|
+
description=(
|
|
777
|
+
"Writes the five-obligation template -- the shipped knapsack "
|
|
778
|
+
"example's shape with the knapsack removed. PATH may be a "
|
|
779
|
+
"directory (problem_def.py is written inside it) or a .py file to "
|
|
780
|
+
"write. An existing file is never overwritten."
|
|
781
|
+
),
|
|
782
|
+
)
|
|
783
|
+
init.add_argument(
|
|
784
|
+
"path", nargs="?", default=".",
|
|
785
|
+
help="directory to write problem_def.py into, or a .py path (default .)",
|
|
786
|
+
)
|
|
787
|
+
init.set_defaults(func=_cmd_init)
|
|
788
|
+
|
|
789
|
+
ver = sub.add_parser("version", help="print the installed version")
|
|
790
|
+
ver.set_defaults(func=_cmd_version)
|
|
791
|
+
|
|
792
|
+
args = parser.parse_args(argv)
|
|
793
|
+
return int(args.func(args))
|
|
794
|
+
|
|
795
|
+
|
|
796
|
+
if __name__ == "__main__": # pragma: no cover
|
|
797
|
+
raise SystemExit(main())
|