agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,803 @@
|
|
|
1
|
+
"""Pareto-guided evolutionary loop.
|
|
2
|
+
|
|
3
|
+
Pure Python orchestration: it depends only on ``core`` and the ``Harness`` port,
|
|
4
|
+
and knows nothing about pydantic-ai or any other LLM runtime. Switching the
|
|
5
|
+
harness changes nothing in this file.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import random
|
|
12
|
+
import time
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from typing import Any, Callable, Dict, List, MutableMapping, Optional, Sequence
|
|
15
|
+
|
|
16
|
+
from agent_evolve.core.formatting import (
|
|
17
|
+
CandidateResult,
|
|
18
|
+
prettify_configuration,
|
|
19
|
+
prettify_results,
|
|
20
|
+
result_to_candidate,
|
|
21
|
+
)
|
|
22
|
+
from agent_evolve.core.problem import ObjectiveSpec, Problem, validate_objective_specs
|
|
23
|
+
from agent_evolve.core.results import (
|
|
24
|
+
Candidate,
|
|
25
|
+
ProviderUsageSummary,
|
|
26
|
+
SearchResult,
|
|
27
|
+
compute_pareto_front,
|
|
28
|
+
objective_value,
|
|
29
|
+
select_minimax_rank,
|
|
30
|
+
)
|
|
31
|
+
from agent_evolve.core.stats import compute_performance_stats, sample_failed_for_constraint
|
|
32
|
+
from agent_evolve.harness.base import Harness
|
|
33
|
+
from agent_evolve.session.evaluate import evaluate_batch
|
|
34
|
+
|
|
35
|
+
LogFn = Callable[[str], None]
|
|
36
|
+
EventFn = Callable[[Dict[str, Any]], None]
|
|
37
|
+
RenderFn = Callable[[Dict[str, Any]], str]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class LoopConfig:
|
|
42
|
+
"""Hyper-parameters for one optimisation run."""
|
|
43
|
+
|
|
44
|
+
pop_size: int = 8
|
|
45
|
+
generations: int = 5
|
|
46
|
+
candidates_per_batch: int = 5
|
|
47
|
+
max_regen_rounds: int = 10
|
|
48
|
+
max_failed_examples: int = 5
|
|
49
|
+
seed: Optional[int] = None
|
|
50
|
+
llm_retries: int = 3
|
|
51
|
+
#: Reflection ablation switches (all True = the paper's full reflective loop).
|
|
52
|
+
#: Turning one off cleanly isolates that feedback arm; all-off is the
|
|
53
|
+
#: non-reflective generate-validate-regenerate baseline (the LLM still sees the
|
|
54
|
+
#: raw validator errors, but no LLM-synthesized insights/guide/patterns).
|
|
55
|
+
use_failure_insights: bool = True
|
|
56
|
+
use_constraint_instruction: bool = True
|
|
57
|
+
use_performance_insights: bool = True
|
|
58
|
+
#: Separate, more patient budget for provider rate-limit (HTTP 429) errors,
|
|
59
|
+
#: which are transient and should not exhaust the normal retry budget.
|
|
60
|
+
rate_limit_retries: int = 12
|
|
61
|
+
#: Starting configurations, evaluated before anything is proposed, so a run
|
|
62
|
+
#: can say whether what it proposed beat what the caller already had.
|
|
63
|
+
seeds: tuple = ()
|
|
64
|
+
#: Hard ceiling on artifacts measured. ``None`` means the generation
|
|
65
|
+
#: structure alone decides.
|
|
66
|
+
evaluation_budget: Optional[int] = None
|
|
67
|
+
#: Artifact-identity cache shared across the run. Supplying one makes
|
|
68
|
+
#: repeated materializations free and reports what that saved.
|
|
69
|
+
evaluation_cache: Optional[MutableMapping] = None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _noop_log(msg: str) -> None:
|
|
73
|
+
pass
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _default_candidate_key(config: Dict[str, Any]) -> str:
|
|
77
|
+
return json.dumps(config, sort_keys=True, default=str)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _is_rate_limit(exc: BaseException) -> bool:
|
|
81
|
+
text = f"{type(exc).__name__} {exc}".lower()
|
|
82
|
+
return "429" in text or "rate limit" in text or "rate_limited" in text
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _retry(fn: Callable, args: tuple, retries: int, log: LogFn,
|
|
86
|
+
rate_limit_retries: int = 12,
|
|
87
|
+
backoff_base: float = 3.0, backoff_cap: float = 30.0,
|
|
88
|
+
rate_limit_cap: float = 60.0) -> Any:
|
|
89
|
+
"""Call *fn*, retrying on failure with exponential backoff.
|
|
90
|
+
|
|
91
|
+
Provider rate-limit (HTTP 429) errors are transient and use a separate, more
|
|
92
|
+
patient budget so they don't exhaust the normal retry budget.
|
|
93
|
+
"""
|
|
94
|
+
normal = 0
|
|
95
|
+
limited = 0
|
|
96
|
+
while True:
|
|
97
|
+
try:
|
|
98
|
+
return fn(*args)
|
|
99
|
+
except Exception as exc: # noqa: BLE001 - retried then re-raised
|
|
100
|
+
if _is_rate_limit(exc):
|
|
101
|
+
limited += 1
|
|
102
|
+
if limited >= rate_limit_retries:
|
|
103
|
+
raise
|
|
104
|
+
delay = min(15.0 + 10.0 * limited, rate_limit_cap)
|
|
105
|
+
log(f" [rate-limit {limited}/{rate_limit_retries}] waiting {delay:.0f}s")
|
|
106
|
+
time.sleep(delay)
|
|
107
|
+
else:
|
|
108
|
+
normal += 1
|
|
109
|
+
if normal >= max(retries, 1):
|
|
110
|
+
raise
|
|
111
|
+
delay = min(backoff_base * (2 ** (normal - 1)), backoff_cap)
|
|
112
|
+
log(f" [retry {normal}/{retries}] {type(exc).__name__}: {exc}; waiting {delay:.0f}s")
|
|
113
|
+
time.sleep(delay)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass
|
|
117
|
+
class _RunState:
|
|
118
|
+
"""Shared collaborators + mutable bookkeeping threaded through the loop."""
|
|
119
|
+
|
|
120
|
+
problem: Any
|
|
121
|
+
harness: Harness
|
|
122
|
+
objectives: Sequence[ObjectiveSpec]
|
|
123
|
+
config: LoopConfig
|
|
124
|
+
rng: random.Random
|
|
125
|
+
call: Callable
|
|
126
|
+
log: LogFn
|
|
127
|
+
on_event: Optional[EventFn]
|
|
128
|
+
render: Optional[RenderFn]
|
|
129
|
+
key_fn: Callable[[Dict[str, Any]], str]
|
|
130
|
+
seen: set
|
|
131
|
+
#: Count of valid candidates in the batch that last (re)wrote the constraint
|
|
132
|
+
#: guide; -1 means "no guide created yet" (port of `constraint_valid_count`).
|
|
133
|
+
constraint_valid_count: int = -1
|
|
134
|
+
|
|
135
|
+
def failed_str(self, results: Sequence[CandidateResult]) -> str:
|
|
136
|
+
return prettify_results(results, self.objectives, render=self.render)
|
|
137
|
+
|
|
138
|
+
def dedup(self, configs: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
|
139
|
+
"""Drop already-seen configs so regen rounds don't re-evaluate duplicates."""
|
|
140
|
+
out: List[Dict[str, Any]] = []
|
|
141
|
+
for c in configs:
|
|
142
|
+
k = self.key_fn(c)
|
|
143
|
+
if k not in self.seen:
|
|
144
|
+
self.seen.add(k)
|
|
145
|
+
out.append(c)
|
|
146
|
+
rejected = len(configs) - len(out)
|
|
147
|
+
if rejected:
|
|
148
|
+
self.log(f" [dedup] rejected {rejected} duplicate candidate(s)")
|
|
149
|
+
return out
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _snapshot_best(
|
|
153
|
+
pareto: Sequence[Candidate], objectives: Sequence[ObjectiveSpec]
|
|
154
|
+
) -> Candidate:
|
|
155
|
+
best = select_minimax_rank(list(pareto), objectives)
|
|
156
|
+
if best is None:
|
|
157
|
+
return Candidate(configuration={}, objectives={}, metadata={})
|
|
158
|
+
return best
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _emit(on_event: Optional[EventFn], event: Dict[str, Any]) -> None:
|
|
162
|
+
if on_event is not None:
|
|
163
|
+
on_event(event)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _performance_stats_str(
|
|
167
|
+
valid_results: Sequence[CandidateResult],
|
|
168
|
+
objectives: Sequence[ObjectiveSpec],
|
|
169
|
+
render: Optional[RenderFn] = None,
|
|
170
|
+
) -> tuple[str, str, int, int]:
|
|
171
|
+
"""Return ``(stats_str, pareto_str, total_valid, pareto_size)``."""
|
|
172
|
+
stats = compute_performance_stats(valid_results, objectives)
|
|
173
|
+
if not stats:
|
|
174
|
+
return "", "None", 0, 0
|
|
175
|
+
lines: List[str] = [f"TOTAL VALID CANDIDATES: {len(valid_results)}"]
|
|
176
|
+
lines.append(f"PARETO FRONT SIZE: {stats.get('pareto_size', 0)}")
|
|
177
|
+
lines.append("")
|
|
178
|
+
lines.append("BEST AND WORST PER OBJECTIVE:")
|
|
179
|
+
for spec in objectives:
|
|
180
|
+
best = stats.get(f"best_{spec.name}")
|
|
181
|
+
worst = stats.get(f"worst_{spec.name}")
|
|
182
|
+
if best:
|
|
183
|
+
cfg = render(best.configuration) if render else prettify_configuration(best.configuration)
|
|
184
|
+
lines.append(f" Best {spec.name}: {objective_value(best.objectives, spec.name)}")
|
|
185
|
+
lines.append(f" Config: {cfg}")
|
|
186
|
+
if worst:
|
|
187
|
+
lines.append(f" Worst {spec.name}: {objective_value(worst.objectives, spec.name)}")
|
|
188
|
+
top_pareto = stats.get("top_3_pareto", [])
|
|
189
|
+
pareto_str = prettify_results(top_pareto, objectives, render=render) if top_pareto else "None"
|
|
190
|
+
return "\n".join(lines), pareto_str, len(valid_results), stats.get("pareto_size", 0)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def run_evolution_loop(
|
|
194
|
+
*,
|
|
195
|
+
problem: Problem,
|
|
196
|
+
harness: Harness,
|
|
197
|
+
config: LoopConfig,
|
|
198
|
+
log: LogFn = _noop_log,
|
|
199
|
+
on_event: Optional[EventFn] = None,
|
|
200
|
+
) -> SearchResult:
|
|
201
|
+
"""Run the full Pareto-guided evolutionary loop against *harness*."""
|
|
202
|
+
objectives = list(problem.objectives)
|
|
203
|
+
validate_objective_specs(objectives)
|
|
204
|
+
if config.pop_size <= 0:
|
|
205
|
+
raise ValueError("pop_size must be positive")
|
|
206
|
+
if config.generations <= 0:
|
|
207
|
+
raise ValueError("generations must be positive")
|
|
208
|
+
if config.candidates_per_batch <= 0:
|
|
209
|
+
raise ValueError("candidates_per_batch must be positive")
|
|
210
|
+
if config.max_regen_rounds < 0:
|
|
211
|
+
raise ValueError("max_regen_rounds must be non-negative")
|
|
212
|
+
|
|
213
|
+
rng = random.Random(config.seed)
|
|
214
|
+
retries = config.llm_retries
|
|
215
|
+
|
|
216
|
+
# Counted, never declared. A run that made no proposer call reports zero
|
|
217
|
+
# because zero was observed here, not because the field was left out.
|
|
218
|
+
proposer_calls = [0]
|
|
219
|
+
|
|
220
|
+
def call(fn: Callable, *args: Any) -> Any:
|
|
221
|
+
proposer_calls[0] += 1
|
|
222
|
+
return _retry(fn, args, retries, log, rate_limit_retries=config.rate_limit_retries)
|
|
223
|
+
|
|
224
|
+
key_fn = getattr(problem, "candidate_key", None) or _default_candidate_key
|
|
225
|
+
state = _RunState(
|
|
226
|
+
problem=problem,
|
|
227
|
+
harness=harness,
|
|
228
|
+
objectives=objectives,
|
|
229
|
+
config=config,
|
|
230
|
+
rng=rng,
|
|
231
|
+
call=call,
|
|
232
|
+
log=log,
|
|
233
|
+
on_event=on_event,
|
|
234
|
+
render=getattr(problem, "render_candidate", None),
|
|
235
|
+
key_fn=key_fn,
|
|
236
|
+
seen=set(),
|
|
237
|
+
)
|
|
238
|
+
# Seed the dedup set with the example/base config so it is never re-proposed.
|
|
239
|
+
example = getattr(problem, "example_config", None)
|
|
240
|
+
if isinstance(example, dict):
|
|
241
|
+
state.seen.add(key_fn(example))
|
|
242
|
+
|
|
243
|
+
all_valid: List[CandidateResult] = []
|
|
244
|
+
all_failed: List[CandidateResult] = []
|
|
245
|
+
all_candidates_meta: List[tuple] = []
|
|
246
|
+
history: List[Dict[str, Any]] = []
|
|
247
|
+
constraint_instruction = ""
|
|
248
|
+
performance_insights = ""
|
|
249
|
+
best_per_generation: List[Candidate] = []
|
|
250
|
+
|
|
251
|
+
log(
|
|
252
|
+
f"agent_evolve: {config.generations} generations, pop={config.pop_size}, "
|
|
253
|
+
f"batch={config.candidates_per_batch}, harness={getattr(harness, 'id', '?')}"
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
# -- Generation 0: the caller's own starting points -------------------
|
|
257
|
+
# Evaluated before anything is proposed, so the run can say plainly
|
|
258
|
+
# whether what it proposed beat what the caller already had.
|
|
259
|
+
seed_pareto: List[Candidate] = []
|
|
260
|
+
if config.seeds:
|
|
261
|
+
seed_configs = [dict(c) for c in config.seeds]
|
|
262
|
+
for c in seed_configs:
|
|
263
|
+
state.seen.add(key_fn(c))
|
|
264
|
+
seed_valid, seed_failed, _ = evaluate_batch(
|
|
265
|
+
problem, seed_configs, objectives, cache=config.evaluation_cache
|
|
266
|
+
)
|
|
267
|
+
all_valid.extend(seed_valid)
|
|
268
|
+
all_failed.extend(seed_failed)
|
|
269
|
+
for r in seed_valid + seed_failed:
|
|
270
|
+
all_candidates_meta.append((r, _candidate_metadata(
|
|
271
|
+
r, generation=0, authored_by=AUTHORED_BY_CALLER_SEED
|
|
272
|
+
)))
|
|
273
|
+
log(f" [seeds] {len(seed_valid)} valid, {len(seed_failed)} rejected")
|
|
274
|
+
seed_pareto = compute_pareto_front(
|
|
275
|
+
[result_to_candidate(r) for r in seed_valid], objectives
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
# -- Generation 1 -----------------------------------------------------
|
|
279
|
+
# With starting points that measured, the first proposal *breeds from
|
|
280
|
+
# them*. Sampling blind here instead -- which is what this did -- threw
|
|
281
|
+
# away the one thing the caller supplied and paid to evaluate, and made
|
|
282
|
+
# the run's first batch independent of the state it was told to start
|
|
283
|
+
# from. Without seeds the behaviour is unchanged: sample, then regenerate.
|
|
284
|
+
if seed_pareto:
|
|
285
|
+
if config.use_performance_insights:
|
|
286
|
+
stats_str, pareto_str, _, _ = _performance_stats_str(
|
|
287
|
+
all_valid, objectives, state.render
|
|
288
|
+
)
|
|
289
|
+
performance_insights = call(
|
|
290
|
+
harness.performance_insights, stats_str, pareto_str, None
|
|
291
|
+
)
|
|
292
|
+
gen1_valid, gen1_failed, constraint_instruction = _run_evolution_generation(
|
|
293
|
+
state=state,
|
|
294
|
+
gen=1,
|
|
295
|
+
prev_pareto=seed_pareto,
|
|
296
|
+
constraint_instruction=constraint_instruction,
|
|
297
|
+
performance_insights=performance_insights,
|
|
298
|
+
)
|
|
299
|
+
gen1_authored_by = AUTHORED_BY_OFFSPRING_PROPOSAL
|
|
300
|
+
else:
|
|
301
|
+
gen1_valid, gen1_failed, constraint_instruction = _run_initial_generation(
|
|
302
|
+
state=state,
|
|
303
|
+
constraint_instruction=constraint_instruction,
|
|
304
|
+
performance_insights=performance_insights,
|
|
305
|
+
)
|
|
306
|
+
gen1_authored_by = AUTHORED_BY_INITIAL_PROPOSAL
|
|
307
|
+
|
|
308
|
+
all_valid.extend(gen1_valid)
|
|
309
|
+
all_failed.extend(gen1_failed)
|
|
310
|
+
for r in gen1_valid + gen1_failed:
|
|
311
|
+
all_candidates_meta.append((r, _candidate_metadata(
|
|
312
|
+
r, generation=1, authored_by=gen1_authored_by
|
|
313
|
+
)))
|
|
314
|
+
|
|
315
|
+
pareto = compute_pareto_front([result_to_candidate(r) for r in all_valid], objectives)
|
|
316
|
+
best_per_generation.append(_snapshot_best(pareto, objectives))
|
|
317
|
+
|
|
318
|
+
if config.use_performance_insights and gen1_valid:
|
|
319
|
+
stats_str, pareto_str, _, _ = _performance_stats_str(all_valid, objectives, state.render)
|
|
320
|
+
# Carry the seed-derived insight forward rather than restarting from
|
|
321
|
+
# nothing: the chain from a measurement to the next proposal is the
|
|
322
|
+
# mechanism under test, and dropping a link in it is not an ablation,
|
|
323
|
+
# it is a bug.
|
|
324
|
+
performance_insights = call(
|
|
325
|
+
harness.performance_insights, stats_str, pareto_str, performance_insights or None
|
|
326
|
+
)
|
|
327
|
+
|
|
328
|
+
history.append(
|
|
329
|
+
{
|
|
330
|
+
"gen": 1,
|
|
331
|
+
"valid_count": len(gen1_valid),
|
|
332
|
+
"failed_count": len(gen1_failed),
|
|
333
|
+
"pareto_size": len(pareto),
|
|
334
|
+
}
|
|
335
|
+
)
|
|
336
|
+
_emit(on_event, {"kind": "generation_complete", "gen": 1, "pareto_size": len(pareto)})
|
|
337
|
+
|
|
338
|
+
def _spent() -> int:
|
|
339
|
+
"""Artifacts actually measured, which is what the budget counts."""
|
|
340
|
+
cache = config.evaluation_cache
|
|
341
|
+
misses = getattr(cache, "misses", None)
|
|
342
|
+
if misses is not None:
|
|
343
|
+
return int(misses)
|
|
344
|
+
return sum(
|
|
345
|
+
1 for r in all_valid + all_failed if getattr(r, "evaluation_attempted", False)
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
# -- Generations 2..N: evolution from the Pareto front ---------------
|
|
349
|
+
for gen in range(2, config.generations + 1):
|
|
350
|
+
if config.evaluation_budget is not None and _spent() >= config.evaluation_budget:
|
|
351
|
+
log(
|
|
352
|
+
f" [budget] {_spent()}/{config.evaluation_budget} evaluations "
|
|
353
|
+
"spent; stopping before generation "
|
|
354
|
+
f"{gen}"
|
|
355
|
+
)
|
|
356
|
+
break
|
|
357
|
+
gen_valid, gen_failed, constraint_instruction = _run_evolution_generation(
|
|
358
|
+
state=state,
|
|
359
|
+
gen=gen,
|
|
360
|
+
prev_pareto=pareto,
|
|
361
|
+
constraint_instruction=constraint_instruction,
|
|
362
|
+
performance_insights=performance_insights,
|
|
363
|
+
)
|
|
364
|
+
|
|
365
|
+
all_valid.extend(gen_valid)
|
|
366
|
+
all_failed.extend(gen_failed)
|
|
367
|
+
for r in gen_valid + gen_failed:
|
|
368
|
+
all_candidates_meta.append((r, _candidate_metadata(
|
|
369
|
+
r, generation=gen, authored_by=AUTHORED_BY_OFFSPRING_PROPOSAL
|
|
370
|
+
)))
|
|
371
|
+
|
|
372
|
+
pareto = compute_pareto_front([result_to_candidate(r) for r in all_valid], objectives)
|
|
373
|
+
best_per_generation.append(_snapshot_best(pareto, objectives))
|
|
374
|
+
|
|
375
|
+
if config.use_performance_insights and all_valid:
|
|
376
|
+
stats_str, pareto_str, _, _ = _performance_stats_str(all_valid, objectives, state.render)
|
|
377
|
+
performance_insights = call(
|
|
378
|
+
harness.performance_insights, stats_str, pareto_str, performance_insights
|
|
379
|
+
)
|
|
380
|
+
|
|
381
|
+
history.append(
|
|
382
|
+
{
|
|
383
|
+
"gen": gen,
|
|
384
|
+
"valid_count": len(gen_valid),
|
|
385
|
+
"failed_count": len(gen_failed),
|
|
386
|
+
"pareto_size": len(pareto),
|
|
387
|
+
}
|
|
388
|
+
)
|
|
389
|
+
_emit(on_event, {"kind": "generation_complete", "gen": gen, "pareto_size": len(pareto)})
|
|
390
|
+
|
|
391
|
+
result = _build_search_result(
|
|
392
|
+
all_valid,
|
|
393
|
+
all_candidates_meta,
|
|
394
|
+
objectives,
|
|
395
|
+
history,
|
|
396
|
+
best_per_generation=best_per_generation,
|
|
397
|
+
# Artifacts actually measured. A result served from the artifact
|
|
398
|
+
# cache cost nothing, so counting it here would overstate the bill
|
|
399
|
+
# and make a budget look breached when it was honoured exactly.
|
|
400
|
+
evaluations=(
|
|
401
|
+
int(getattr(config.evaluation_cache, 'misses', 0))
|
|
402
|
+
if config.evaluation_cache is not None
|
|
403
|
+
else sum(r.evaluation_attempted for r in (*all_valid, *all_failed))
|
|
404
|
+
),
|
|
405
|
+
candidate_key=state.key_fn,
|
|
406
|
+
provider_usage=_provider_usage(harness, proposer_calls[0]),
|
|
407
|
+
)
|
|
408
|
+
log("")
|
|
409
|
+
log(
|
|
410
|
+
f"Summary: evaluations={result.evaluations}"
|
|
411
|
+
+ (
|
|
412
|
+
f" (+{getattr(config.evaluation_cache, 'hits', 0)} served from cache)"
|
|
413
|
+
if getattr(config.evaluation_cache, "hits", 0)
|
|
414
|
+
else ""
|
|
415
|
+
)
|
|
416
|
+
+ f", valid={len(all_valid)}, "
|
|
417
|
+
f"pareto={len(result.pareto_front)}, best={result.best.objectives}"
|
|
418
|
+
)
|
|
419
|
+
_emit(
|
|
420
|
+
on_event,
|
|
421
|
+
{
|
|
422
|
+
"kind": "search_complete",
|
|
423
|
+
"evaluations": result.evaluations,
|
|
424
|
+
"pareto_size": len(result.pareto_front),
|
|
425
|
+
"performance_insights": performance_insights,
|
|
426
|
+
"constraint_instruction": constraint_instruction,
|
|
427
|
+
},
|
|
428
|
+
)
|
|
429
|
+
return result
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
# ------------------------------------------------------------------
|
|
433
|
+
# Constraint-instruction learning (port of constraint_valid_count heuristic)
|
|
434
|
+
# ------------------------------------------------------------------
|
|
435
|
+
|
|
436
|
+
def _learn_constraint(
|
|
437
|
+
state: _RunState,
|
|
438
|
+
constraint_instruction: str,
|
|
439
|
+
last_round_failed: Sequence[CandidateResult],
|
|
440
|
+
all_failed: Sequence[CandidateResult],
|
|
441
|
+
batch_valid_count: int,
|
|
442
|
+
) -> str:
|
|
443
|
+
"""Create the constraint guide if absent, else update it when a batch improves.
|
|
444
|
+
|
|
445
|
+
Mirrors the original: only (re)write the guide when this batch produced more
|
|
446
|
+
valid candidates than the batch that last wrote it.
|
|
447
|
+
"""
|
|
448
|
+
if not state.config.use_constraint_instruction:
|
|
449
|
+
return constraint_instruction
|
|
450
|
+
if not all_failed:
|
|
451
|
+
return constraint_instruction
|
|
452
|
+
|
|
453
|
+
create = not constraint_instruction
|
|
454
|
+
improved = batch_valid_count > state.constraint_valid_count
|
|
455
|
+
if not (create or improved):
|
|
456
|
+
return constraint_instruction
|
|
457
|
+
|
|
458
|
+
sampled = sample_failed_for_constraint(
|
|
459
|
+
last_round_failed, all_failed, state.config.max_failed_examples, state.rng
|
|
460
|
+
)
|
|
461
|
+
sampled_str = state.failed_str(sampled)
|
|
462
|
+
previous = None if create else constraint_instruction
|
|
463
|
+
new_ci = state.call(state.harness.constraint_instruction, sampled_str, previous)
|
|
464
|
+
if new_ci and new_ci != constraint_instruction:
|
|
465
|
+
state.constraint_valid_count = batch_valid_count
|
|
466
|
+
verb = "created" if create else "updated"
|
|
467
|
+
state.log(f" [constraint] guide {verb} (batch valid={batch_valid_count})")
|
|
468
|
+
return new_ci
|
|
469
|
+
return constraint_instruction
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
# ------------------------------------------------------------------
|
|
473
|
+
# Generation helpers
|
|
474
|
+
# ------------------------------------------------------------------
|
|
475
|
+
|
|
476
|
+
def _run_initial_generation(
|
|
477
|
+
*,
|
|
478
|
+
state: _RunState,
|
|
479
|
+
constraint_instruction: str,
|
|
480
|
+
performance_insights: str,
|
|
481
|
+
) -> tuple:
|
|
482
|
+
cfg = state.config
|
|
483
|
+
gen_valid: List[CandidateResult] = []
|
|
484
|
+
gen_failed: List[CandidateResult] = []
|
|
485
|
+
last_round_failed: List[CandidateResult] = []
|
|
486
|
+
|
|
487
|
+
# ``max_regen_rounds`` means retries *after* one mandatory initial call.
|
|
488
|
+
for regen_round in range(cfg.max_regen_rounds + 1):
|
|
489
|
+
remaining = max(cfg.pop_size - len(gen_valid), 1)
|
|
490
|
+
n = min(cfg.candidates_per_batch, remaining)
|
|
491
|
+
if regen_round == 0:
|
|
492
|
+
configs = state.call(state.harness.generate_initial, n)
|
|
493
|
+
else:
|
|
494
|
+
configs = state.call(
|
|
495
|
+
state.harness.regenerate,
|
|
496
|
+
state.failed_str(last_round_failed),
|
|
497
|
+
n,
|
|
498
|
+
constraint_instruction,
|
|
499
|
+
performance_insights,
|
|
500
|
+
)
|
|
501
|
+
configs = state.dedup(configs)
|
|
502
|
+
|
|
503
|
+
valid_batch, failed_batch, _ = evaluate_batch(state.problem, configs, state.objectives, cache=state.config.evaluation_cache)
|
|
504
|
+
_emit_candidates(state.on_event, 1, regen_round, valid_batch, failed_batch)
|
|
505
|
+
|
|
506
|
+
if failed_batch:
|
|
507
|
+
_attach_failure_insights(state, failed_batch)
|
|
508
|
+
gen_failed.extend(failed_batch)
|
|
509
|
+
gen_valid.extend(valid_batch)
|
|
510
|
+
last_round_failed = failed_batch
|
|
511
|
+
|
|
512
|
+
constraint_instruction = _learn_constraint(
|
|
513
|
+
state, constraint_instruction, last_round_failed, gen_failed, len(valid_batch)
|
|
514
|
+
)
|
|
515
|
+
|
|
516
|
+
if len(gen_valid) >= cfg.pop_size:
|
|
517
|
+
break
|
|
518
|
+
|
|
519
|
+
# A harness may return more than requested. Every already-evaluated candidate
|
|
520
|
+
# remains in the ledger/archive even if the working population target was met.
|
|
521
|
+
return gen_valid, gen_failed, constraint_instruction
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
def _run_evolution_generation(
|
|
525
|
+
*,
|
|
526
|
+
state: _RunState,
|
|
527
|
+
gen: int,
|
|
528
|
+
prev_pareto: Sequence[Candidate],
|
|
529
|
+
constraint_instruction: str,
|
|
530
|
+
performance_insights: str,
|
|
531
|
+
) -> tuple:
|
|
532
|
+
cfg = state.config
|
|
533
|
+
gen_valid: List[CandidateResult] = []
|
|
534
|
+
gen_failed: List[CandidateResult] = []
|
|
535
|
+
|
|
536
|
+
pareto_results = _pareto_as_results(prev_pareto)
|
|
537
|
+
|
|
538
|
+
if not prev_pareto:
|
|
539
|
+
configs = state.call(state.harness.generate_initial, cfg.pop_size)
|
|
540
|
+
else:
|
|
541
|
+
pareto_str = prettify_results(pareto_results[:5], state.objectives, render=state.render)
|
|
542
|
+
configs = state.call(
|
|
543
|
+
state.harness.generate_offspring,
|
|
544
|
+
pareto_str,
|
|
545
|
+
cfg.pop_size,
|
|
546
|
+
constraint_instruction,
|
|
547
|
+
performance_insights,
|
|
548
|
+
)
|
|
549
|
+
configs = state.dedup(configs)
|
|
550
|
+
|
|
551
|
+
valid_batch, failed_batch, _ = evaluate_batch(state.problem, configs, state.objectives, cache=state.config.evaluation_cache)
|
|
552
|
+
_emit_candidates(state.on_event, gen, 0, valid_batch, failed_batch)
|
|
553
|
+
|
|
554
|
+
if failed_batch:
|
|
555
|
+
_attach_failure_insights(state, failed_batch)
|
|
556
|
+
gen_failed.extend(failed_batch)
|
|
557
|
+
gen_valid.extend(valid_batch)
|
|
558
|
+
last_round_failed = failed_batch
|
|
559
|
+
|
|
560
|
+
regen_round = 0
|
|
561
|
+
while len(gen_valid) < cfg.pop_size and regen_round < cfg.max_regen_rounds:
|
|
562
|
+
if not last_round_failed:
|
|
563
|
+
break
|
|
564
|
+
failed_str = state.failed_str(last_round_failed)
|
|
565
|
+
|
|
566
|
+
if prev_pareto:
|
|
567
|
+
p_str = prettify_results(pareto_results[:3], state.objectives, render=state.render)
|
|
568
|
+
remaining = max(cfg.pop_size - len(gen_valid), 1)
|
|
569
|
+
configs = state.call(
|
|
570
|
+
state.harness.regenerate_offspring,
|
|
571
|
+
failed_str,
|
|
572
|
+
p_str,
|
|
573
|
+
min(cfg.candidates_per_batch, remaining),
|
|
574
|
+
constraint_instruction,
|
|
575
|
+
performance_insights,
|
|
576
|
+
)
|
|
577
|
+
else:
|
|
578
|
+
remaining = max(cfg.pop_size - len(gen_valid), 1)
|
|
579
|
+
configs = state.call(
|
|
580
|
+
state.harness.regenerate,
|
|
581
|
+
failed_str,
|
|
582
|
+
min(cfg.candidates_per_batch, remaining),
|
|
583
|
+
constraint_instruction,
|
|
584
|
+
performance_insights,
|
|
585
|
+
)
|
|
586
|
+
configs = state.dedup(configs)
|
|
587
|
+
|
|
588
|
+
valid_batch, failed_batch, _ = evaluate_batch(state.problem, configs, state.objectives, cache=state.config.evaluation_cache)
|
|
589
|
+
_emit_candidates(state.on_event, gen, regen_round + 1, valid_batch, failed_batch)
|
|
590
|
+
|
|
591
|
+
if failed_batch:
|
|
592
|
+
_attach_failure_insights(state, failed_batch)
|
|
593
|
+
gen_failed.extend(failed_batch)
|
|
594
|
+
gen_valid.extend(valid_batch)
|
|
595
|
+
last_round_failed = failed_batch
|
|
596
|
+
|
|
597
|
+
constraint_instruction = _learn_constraint(
|
|
598
|
+
state, constraint_instruction, last_round_failed, gen_failed, len(valid_batch)
|
|
599
|
+
)
|
|
600
|
+
regen_round += 1
|
|
601
|
+
|
|
602
|
+
return gen_valid, gen_failed, constraint_instruction
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
# ------------------------------------------------------------------
|
|
606
|
+
# Shared helpers
|
|
607
|
+
# ------------------------------------------------------------------
|
|
608
|
+
|
|
609
|
+
def _pareto_as_results(pareto: Sequence[Candidate]) -> List[CandidateResult]:
|
|
610
|
+
return [
|
|
611
|
+
CandidateResult(
|
|
612
|
+
configuration=c.configuration,
|
|
613
|
+
objectives=c.objectives,
|
|
614
|
+
is_valid=True,
|
|
615
|
+
evaluation_attempted=True,
|
|
616
|
+
)
|
|
617
|
+
for c in pareto
|
|
618
|
+
]
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
# Who authored a candidate's configuration. Recorded per candidate so a caller
|
|
622
|
+
# can tell what the model actually produced from what it was given, without
|
|
623
|
+
# inferring it from position in the loop. Inferring authorship from position is
|
|
624
|
+
# how an arm gets labelled by the component that produced it rather than by the
|
|
625
|
+
# authority that decided it, which is a mistake this project has made twice.
|
|
626
|
+
AUTHORED_BY_CALLER_SEED = "caller_seed"
|
|
627
|
+
AUTHORED_BY_INITIAL_PROPOSAL = "proposer_initial"
|
|
628
|
+
AUTHORED_BY_OFFSPRING_PROPOSAL = "proposer_offspring"
|
|
629
|
+
AUTHORING_CALLS = (
|
|
630
|
+
AUTHORED_BY_CALLER_SEED,
|
|
631
|
+
AUTHORED_BY_INITIAL_PROPOSAL,
|
|
632
|
+
AUTHORED_BY_OFFSPRING_PROPOSAL,
|
|
633
|
+
)
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
def _provider_usage(harness: Any, proposer_calls: int) -> ProviderUsageSummary:
|
|
637
|
+
"""Summarise what the run spent on the proposer.
|
|
638
|
+
|
|
639
|
+
``calls`` is always counted here. Token and cost figures come from the
|
|
640
|
+
harness when it reports them and stay ``None`` when it does not -- ``None``
|
|
641
|
+
means "this proposer does not report it", which is deliberately not the
|
|
642
|
+
same value as zero. Conflating the two is how a run that never looked comes
|
|
643
|
+
to read like a run that measured nothing spent.
|
|
644
|
+
"""
|
|
645
|
+
|
|
646
|
+
reported = getattr(harness, "usage", None)
|
|
647
|
+
figures: Dict[str, Any] = {}
|
|
648
|
+
if callable(reported):
|
|
649
|
+
try:
|
|
650
|
+
value = reported()
|
|
651
|
+
except Exception: # a harness defect must not lose the run's result
|
|
652
|
+
value = None
|
|
653
|
+
if isinstance(value, dict):
|
|
654
|
+
figures = value
|
|
655
|
+
def _figure(name: str) -> Optional[int]:
|
|
656
|
+
# Absent means unreported, which is not zero. A harness that reports a
|
|
657
|
+
# genuine zero passes 0 and it is preserved.
|
|
658
|
+
value = figures.get(name)
|
|
659
|
+
return None if value is None else int(value)
|
|
660
|
+
|
|
661
|
+
supplied = {
|
|
662
|
+
"input_tokens": _figure("input_tokens"),
|
|
663
|
+
"output_tokens": _figure("output_tokens"),
|
|
664
|
+
"cost_usd": figures.get("cost_usd"),
|
|
665
|
+
}
|
|
666
|
+
reporter = (
|
|
667
|
+
type(harness).__name__
|
|
668
|
+
if any(v is not None for v in supplied.values())
|
|
669
|
+
else None
|
|
670
|
+
)
|
|
671
|
+
return ProviderUsageSummary(
|
|
672
|
+
calls=proposer_calls,
|
|
673
|
+
model=figures.get("model"),
|
|
674
|
+
reported_by=reporter,
|
|
675
|
+
**supplied,
|
|
676
|
+
)
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
def _candidate_metadata(
|
|
680
|
+
result: CandidateResult, *, generation: int, authored_by: str
|
|
681
|
+
) -> Dict[str, Any]:
|
|
682
|
+
"""Preserve trace diagnostics and authorship in the public candidate ledger."""
|
|
683
|
+
if authored_by not in AUTHORING_CALLS:
|
|
684
|
+
raise ValueError(
|
|
685
|
+
f"authored_by must name an authoring call {AUTHORING_CALLS}, "
|
|
686
|
+
f"got {authored_by!r}"
|
|
687
|
+
)
|
|
688
|
+
metadata: Dict[str, Any] = {
|
|
689
|
+
"generation": generation,
|
|
690
|
+
"authored_by": authored_by,
|
|
691
|
+
"valid": result.is_valid,
|
|
692
|
+
"is_pareto": False,
|
|
693
|
+
"evaluation_attempted": result.evaluation_attempted,
|
|
694
|
+
}
|
|
695
|
+
if result.failure_phase:
|
|
696
|
+
metadata["failure_phase"] = result.failure_phase
|
|
697
|
+
if result.error_message:
|
|
698
|
+
metadata["error_message"] = result.error_message
|
|
699
|
+
if result.insight:
|
|
700
|
+
metadata["insight"] = result.insight
|
|
701
|
+
return metadata
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
def _attach_failure_insights(state: _RunState, failed: List[CandidateResult]) -> None:
|
|
705
|
+
if not state.config.use_failure_insights:
|
|
706
|
+
return
|
|
707
|
+
# A proposer may declare that it produces no insights. An uninformed
|
|
708
|
+
# baseline is the case that matters: one that synthesised guidance would
|
|
709
|
+
# not be uninformed, so its empty return is correct and not a fault.
|
|
710
|
+
if not getattr(state.harness, "provides_insights", True):
|
|
711
|
+
return
|
|
712
|
+
insights = state.call(state.harness.failure_insights, state.failed_str(failed), len(failed))
|
|
713
|
+
if isinstance(insights, list) and insights:
|
|
714
|
+
# Broadcast: some models collapse the per-candidate list to one item; reuse the
|
|
715
|
+
# last insight for any remaining failures so every failed candidate carries feedback
|
|
716
|
+
# (otherwise zip() silently dropped feedback for all but the first failure).
|
|
717
|
+
for i, r in enumerate(failed):
|
|
718
|
+
r.insight = str(insights[i]) if i < len(insights) else str(insights[-1])
|
|
719
|
+
if len(insights) != len(failed):
|
|
720
|
+
state.log(f"[agent_evolve] note: {len(insights)} insight(s) for "
|
|
721
|
+
f"{len(failed)} failures — broadcasting last to remainder")
|
|
722
|
+
else:
|
|
723
|
+
state.log("[agent_evolve] WARNING: failure_insights returned no usable list")
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
def _emit_candidates(
|
|
727
|
+
on_event: Optional[EventFn],
|
|
728
|
+
gen: Optional[int],
|
|
729
|
+
regen_round: int,
|
|
730
|
+
valid_batch: Sequence[CandidateResult],
|
|
731
|
+
failed_batch: Sequence[CandidateResult],
|
|
732
|
+
) -> None:
|
|
733
|
+
if on_event is None:
|
|
734
|
+
return
|
|
735
|
+
for r in valid_batch:
|
|
736
|
+
on_event(
|
|
737
|
+
{
|
|
738
|
+
"kind": "candidate_result",
|
|
739
|
+
"gen": gen,
|
|
740
|
+
"regen_round": regen_round,
|
|
741
|
+
"valid": True,
|
|
742
|
+
"configuration": dict(r.configuration),
|
|
743
|
+
"evaluation_attempted": r.evaluation_attempted,
|
|
744
|
+
"objectives": dict(r.objectives),
|
|
745
|
+
}
|
|
746
|
+
)
|
|
747
|
+
for r in failed_batch:
|
|
748
|
+
on_event(
|
|
749
|
+
{
|
|
750
|
+
"kind": "candidate_result",
|
|
751
|
+
"gen": gen,
|
|
752
|
+
"regen_round": regen_round,
|
|
753
|
+
"valid": False,
|
|
754
|
+
"configuration": dict(r.configuration),
|
|
755
|
+
"evaluation_attempted": r.evaluation_attempted,
|
|
756
|
+
"failure_phase": r.failure_phase,
|
|
757
|
+
"error": r.error_message,
|
|
758
|
+
}
|
|
759
|
+
)
|
|
760
|
+
|
|
761
|
+
|
|
762
|
+
def _build_search_result(
|
|
763
|
+
all_valid: List[CandidateResult],
|
|
764
|
+
all_candidates_meta: List[tuple],
|
|
765
|
+
objectives: Sequence[ObjectiveSpec],
|
|
766
|
+
history: List[Dict[str, Any]],
|
|
767
|
+
*,
|
|
768
|
+
best_per_generation: Optional[List[Candidate]] = None,
|
|
769
|
+
evaluations: int = 0,
|
|
770
|
+
candidate_key: Callable[[Dict[str, Any]], str] = _default_candidate_key,
|
|
771
|
+
provider_usage: Optional[ProviderUsageSummary] = None,
|
|
772
|
+
) -> SearchResult:
|
|
773
|
+
pareto_results = compute_pareto_front(
|
|
774
|
+
[result_to_candidate(r) for r in all_valid], objectives
|
|
775
|
+
)
|
|
776
|
+
pareto_configs = {candidate_key(c.configuration) for c in pareto_results}
|
|
777
|
+
|
|
778
|
+
all_candidates: List[Candidate] = []
|
|
779
|
+
for cr, meta in all_candidates_meta:
|
|
780
|
+
meta_copy = dict(meta)
|
|
781
|
+
if candidate_key(cr.configuration) in pareto_configs:
|
|
782
|
+
meta_copy["is_pareto"] = True
|
|
783
|
+
all_candidates.append(result_to_candidate(cr, meta_copy))
|
|
784
|
+
|
|
785
|
+
pareto_list = [
|
|
786
|
+
Candidate(configuration=c.configuration, objectives=c.objectives, metadata={"is_pareto": True})
|
|
787
|
+
for c in pareto_results
|
|
788
|
+
]
|
|
789
|
+
|
|
790
|
+
best_candidate = select_minimax_rank(pareto_results, objectives)
|
|
791
|
+
if best_candidate is None:
|
|
792
|
+
best_candidate = Candidate(configuration={}, objectives={}, metadata={})
|
|
793
|
+
|
|
794
|
+
return SearchResult(
|
|
795
|
+
objectives=list(objectives),
|
|
796
|
+
best=best_candidate,
|
|
797
|
+
pareto_front=pareto_list,
|
|
798
|
+
all_candidates=all_candidates,
|
|
799
|
+
history=history,
|
|
800
|
+
best_per_generation=list(best_per_generation or []),
|
|
801
|
+
evaluations=evaluations,
|
|
802
|
+
provider_usage=provider_usage,
|
|
803
|
+
)
|