agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
agent_evolve/api.py
ADDED
|
@@ -0,0 +1,767 @@
|
|
|
1
|
+
"""The public entry point: ``optimize(problem, budget=...) -> SearchResult``.
|
|
2
|
+
|
|
3
|
+
One required argument and one number a caller actually knows -- how many
|
|
4
|
+
evaluations they can afford. Everything else has a defensible default.
|
|
5
|
+
|
|
6
|
+
``proposer`` is the one option worth understanding:
|
|
7
|
+
|
|
8
|
+
``"random"`` samples the candidate schema. No credentials, no network, no
|
|
9
|
+
cost. It is also the control arm: a model that cannot beat it on
|
|
10
|
+
your problem is not earning its price. ``agent_evolve check``
|
|
11
|
+
runs exactly that comparison.
|
|
12
|
+
``"llm"`` the model-driven proposer.
|
|
13
|
+
``"auto"`` ``llm`` when a provider credential is present, otherwise
|
|
14
|
+
``random``, said out loud through ``on_progress`` rather than
|
|
15
|
+
silently.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import dataclasses
|
|
21
|
+
from typing import Any, Callable, Literal, Optional
|
|
22
|
+
|
|
23
|
+
from agent_evolve import bootstrap
|
|
24
|
+
from agent_evolve.contract import as_problem
|
|
25
|
+
from agent_evolve.core.formatting import format_search_space_description
|
|
26
|
+
from agent_evolve.core.results import ProviderUsageSummary, SearchResult
|
|
27
|
+
from agent_evolve.harness.base import HarnessContext, LLMConfig
|
|
28
|
+
from agent_evolve.harness.directives import DefaultDirectives
|
|
29
|
+
from agent_evolve.harness.registry import harness_registry
|
|
30
|
+
from agent_evolve.session.evaluate import EvaluationCache
|
|
31
|
+
from agent_evolve.session.loop import LoopConfig, run_evolution_loop
|
|
32
|
+
from agent_evolve.settings import AgentEvolveSettings, credentials_present
|
|
33
|
+
|
|
34
|
+
__all__ = ["optimize"]
|
|
35
|
+
|
|
36
|
+
Proposer = Literal["auto", "llm", "random"]
|
|
37
|
+
|
|
38
|
+
#: Who turns a screen's evidence into a sampling prior. The llm forms fall
|
|
39
|
+
#: back to their rule comparator -- out loud -- when no model call is possible.
|
|
40
|
+
_PRIORS = ("rule", "rule-weighted", "llm", "llm-weighted",
|
|
41
|
+
"llm-weighted-committed")
|
|
42
|
+
|
|
43
|
+
#: Candidates proposed per generation. The budget decides how many generations
|
|
44
|
+
#: that buys, so the caller states the number they know and not this one.
|
|
45
|
+
_BATCH = 8
|
|
46
|
+
|
|
47
|
+
#: The largest budget any SEALED row was measured at. At or below it the
|
|
48
|
+
#: genetic sizing is a control arm and may not move; above it there is nothing
|
|
49
|
+
#: to hold still. A constant rather than a knob, because a knob is a way to
|
|
50
|
+
#: move a sealed arm by accident.
|
|
51
|
+
_SEALED_BUDGET_CEILING = 384
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _genetic_sizing(budget: int) -> tuple[int, int]:
|
|
55
|
+
"""``(population, offspring per generation)``, from the budget alone.
|
|
56
|
+
|
|
57
|
+
Sized from the BUDGET, not from how many seeds the caller happened to
|
|
58
|
+
supply: one seed would otherwise give a population of two, which cannot
|
|
59
|
+
recombine into anything its parents do not already contain.
|
|
60
|
+
|
|
61
|
+
Two regimes, and the split is measured rather than tasteful.
|
|
62
|
+
|
|
63
|
+
Up to ``_SEALED_BUDGET_CEILING`` the population is the old expression --
|
|
64
|
+
capped at twelve, floored at four -- written here as the literal branch so
|
|
65
|
+
that every budget a sealed row was measured at runs the arithmetic it was
|
|
66
|
+
measured with. The byte fossil and the sizing table both pin it.
|
|
67
|
+
|
|
68
|
+
Above that ceiling the cap was a THROTTLE. At B = 2000 twelve members
|
|
69
|
+
converge long before the budget is gone: late generations propose
|
|
70
|
+
recombinations the population already holds, those hit the evaluation
|
|
71
|
+
cache, and the generation count -- which is a cap on generations, not on
|
|
72
|
+
charges -- runs out with the budget unspent. Measured, six of six cheap
|
|
73
|
+
cells spent 969 to 1212 of 2000 charges while the uniform comparator spent
|
|
74
|
+
1696 to 1842, so the matched-budget comparison was decided by how much each
|
|
75
|
+
arm could spend and not by how well it was guided; on recall per
|
|
76
|
+
EVALUATION the same cells read at parity or better. So the population
|
|
77
|
+
grows with the budget (one member per 32 charges, floored at the old cap of
|
|
78
|
+
twelve and ceilinged at 64, where the per-generation selection cost starts
|
|
79
|
+
to be the thing being paid for) and the offspring count follows it. The
|
|
80
|
+
generations formula is untouched: it divides by the offspring count and
|
|
81
|
+
adapts on its own.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
if int(budget) > _SEALED_BUDGET_CEILING:
|
|
85
|
+
pop = min(64, max(12, int(budget) // 32))
|
|
86
|
+
return pop, pop - 2
|
|
87
|
+
pop = max(4, min(budget // 4, 12))
|
|
88
|
+
return pop, max(2, pop - 2)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _describe(problem: Any) -> str:
|
|
92
|
+
"""Build the search-space description shown to the proposer."""
|
|
93
|
+
problem_description = None
|
|
94
|
+
if hasattr(problem, "search_space_description"):
|
|
95
|
+
problem_description = problem.search_space_description()
|
|
96
|
+
|
|
97
|
+
config_schema = getattr(problem, "config_schema", None)
|
|
98
|
+
candidate_model = getattr(problem, "candidate_model", None)
|
|
99
|
+
if config_schema is None and candidate_model is not None:
|
|
100
|
+
try:
|
|
101
|
+
config_schema = candidate_model.model_json_schema()
|
|
102
|
+
except Exception:
|
|
103
|
+
config_schema = None
|
|
104
|
+
|
|
105
|
+
return format_search_space_description(
|
|
106
|
+
list(problem.objectives),
|
|
107
|
+
config_schema=config_schema,
|
|
108
|
+
example_config=getattr(problem, "example_config", None),
|
|
109
|
+
constraints=None, # constraints flow through HarnessContext
|
|
110
|
+
problem_description=problem_description,
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _resolve_proposer(proposer: str, announce: Callable[[str], None]) -> str:
|
|
115
|
+
if proposer != "auto" and proposer not in ("llm", "random"):
|
|
116
|
+
# Any harness registered by name is also a proposer. This is how an
|
|
117
|
+
# out-of-tree integration is selected, without a second parameter that
|
|
118
|
+
# means almost the same thing.
|
|
119
|
+
bootstrap.load_integrations()
|
|
120
|
+
if proposer not in harness_registry.ids():
|
|
121
|
+
raise ValueError(
|
|
122
|
+
f"proposer must be 'auto', 'llm', 'random' or a registered "
|
|
123
|
+
f"harness id, got {proposer!r}. Registered: "
|
|
124
|
+
f"{sorted(harness_registry.ids())}"
|
|
125
|
+
)
|
|
126
|
+
if proposer != "auto":
|
|
127
|
+
return proposer
|
|
128
|
+
if credentials_present():
|
|
129
|
+
return "llm"
|
|
130
|
+
announce(
|
|
131
|
+
"No provider credential found, so candidates are being proposed at "
|
|
132
|
+
"random. This costs nothing, and it is the baseline a model has to "
|
|
133
|
+
"beat. Pass proposer='llm' with a credential to use a model."
|
|
134
|
+
)
|
|
135
|
+
return "random"
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _priced_usage(
|
|
139
|
+
route: str, input_tokens: int, output_tokens: int
|
|
140
|
+
) -> tuple[Optional[str], str]:
|
|
141
|
+
"""``(cost_usd, reported_by)`` for a run's token counts.
|
|
142
|
+
|
|
143
|
+
The two halves of a cost figure come from different places and the reporter
|
|
144
|
+
string says which is which: the token counts are the provider's own, the
|
|
145
|
+
per-million prices are this package's published table -- the same numbers
|
|
146
|
+
the CLI echoes before it spends anything. Keeping the provenance in the
|
|
147
|
+
field means nobody has to guess later whether a dollar figure was billed or
|
|
148
|
+
computed.
|
|
149
|
+
|
|
150
|
+
An unpriced route returns ``None``. A cost that cannot be derived is
|
|
151
|
+
reported as unknown rather than as zero, because zero reads as "nothing was
|
|
152
|
+
spent" when it means "nobody looked".
|
|
153
|
+
"""
|
|
154
|
+
from decimal import Decimal
|
|
155
|
+
|
|
156
|
+
from agent_evolve.settings import model_price
|
|
157
|
+
|
|
158
|
+
measured = "openrouter response usage"
|
|
159
|
+
price = model_price(route)
|
|
160
|
+
if price is None:
|
|
161
|
+
return None, measured
|
|
162
|
+
per_m_in, per_m_out = price
|
|
163
|
+
million = Decimal(1_000_000)
|
|
164
|
+
cost = (
|
|
165
|
+
Decimal(str(per_m_in)) * Decimal(input_tokens) / million
|
|
166
|
+
+ Decimal(str(per_m_out)) * Decimal(output_tokens) / million
|
|
167
|
+
)
|
|
168
|
+
return (
|
|
169
|
+
str(cost.quantize(Decimal("0.000001"))),
|
|
170
|
+
f"{measured}; cost derived from the package's published price table",
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _build_harness(kind: str, seed: Optional[int], settings: AgentEvolveSettings) -> Any:
|
|
175
|
+
bootstrap.load_integrations()
|
|
176
|
+
if kind == "random":
|
|
177
|
+
harness_id = "random"
|
|
178
|
+
elif kind == "llm":
|
|
179
|
+
harness_id = settings.harness
|
|
180
|
+
else:
|
|
181
|
+
harness_id = kind # an explicitly named registered harness
|
|
182
|
+
missing = bootstrap.requirement_failure(harness_id)
|
|
183
|
+
if missing is not None:
|
|
184
|
+
# Fail here, naming the fix, rather than deep inside the first model
|
|
185
|
+
# call with a bare ModuleNotFoundError.
|
|
186
|
+
raise RuntimeError(
|
|
187
|
+
f"the {harness_id!r} proposer {missing}. "
|
|
188
|
+
"Or run with proposer='random', which needs nothing."
|
|
189
|
+
)
|
|
190
|
+
try:
|
|
191
|
+
return harness_registry.create(harness_id, seed=seed)
|
|
192
|
+
except KeyError as error:
|
|
193
|
+
raise KeyError(bootstrap.explain_missing_harness(harness_id)) from error
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _resolve_strategy(strategy: str, has_seeds: bool, announce) -> str:
|
|
197
|
+
"""Pick the search loop. ``auto`` prefers genetics wherever they are usable.
|
|
198
|
+
|
|
199
|
+
The authoring loop asks a model to write whole configurations from a text
|
|
200
|
+
rendering of the Pareto front. Measured against uniform random sampling on
|
|
201
|
+
every genome length tried, that loses (-0.086 to -0.531 excess capture)
|
|
202
|
+
while recombination over a population wins (+0.0042 to +0.1798). So
|
|
203
|
+
``genetic`` is preferred wherever it can run, which is wherever the problem
|
|
204
|
+
supplies at least one seed to give a candidate its shape.
|
|
205
|
+
"""
|
|
206
|
+
|
|
207
|
+
if strategy not in ("auto", "genetic", "authoring"):
|
|
208
|
+
raise ValueError(
|
|
209
|
+
f"strategy must be 'auto', 'genetic' or 'authoring', got {strategy!r}"
|
|
210
|
+
)
|
|
211
|
+
if strategy != "auto":
|
|
212
|
+
return strategy
|
|
213
|
+
if has_seeds:
|
|
214
|
+
return "genetic"
|
|
215
|
+
announce(
|
|
216
|
+
"No seeds were supplied, so candidates are authored from scratch rather "
|
|
217
|
+
"than recombined. Give Problem.seeds() one configuration to use the "
|
|
218
|
+
"genetic loop, which measures better against random search."
|
|
219
|
+
)
|
|
220
|
+
return "authoring"
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _llm_refusal_message(*, extra_missing: bool) -> str:
|
|
224
|
+
"""The explicit-llm refusal, naming every way out that applies.
|
|
225
|
+
|
|
226
|
+
On a core install the stranger who asks for a model is missing TWO
|
|
227
|
+
things, and the fix a message names first should be the one they hit
|
|
228
|
+
first: the optional dependencies, then the credential. On an install
|
|
229
|
+
that already has the extra, naming it would be noise. The CI stranger
|
|
230
|
+
job holds the extra-missing rendering to actually naming the extra.
|
|
231
|
+
"""
|
|
232
|
+
|
|
233
|
+
fix = (
|
|
234
|
+
"Install the model path's optional dependencies with: pip install "
|
|
235
|
+
"'agentevolve-optimizer[llm]'. Then set OPENROUTER_API_KEY (or "
|
|
236
|
+
"AGENTEVOLVE_DOTENV naming a file that does)"
|
|
237
|
+
if extra_missing else
|
|
238
|
+
"Set OPENROUTER_API_KEY (or AGENTEVOLVE_DOTENV naming a file that "
|
|
239
|
+
"does)"
|
|
240
|
+
)
|
|
241
|
+
return (
|
|
242
|
+
"proposer='llm' was asked for by name, but no provider credential "
|
|
243
|
+
f"is configured, so no model can be called. {fix}, or run with "
|
|
244
|
+
"proposer='random', which needs nothing -- or proposer='auto', "
|
|
245
|
+
"which chooses it out loud."
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _check_structure_budget(structure_budget: int, budget: int) -> None:
|
|
250
|
+
"""The screen is charged against the search it informs, so it must fit."""
|
|
251
|
+
|
|
252
|
+
if structure_budget >= budget:
|
|
253
|
+
raise ValueError(
|
|
254
|
+
f"structure_budget ({structure_budget}) must leave room inside the "
|
|
255
|
+
f"budget ({budget}): the screen is charged against the same budget "
|
|
256
|
+
"as the search it informs"
|
|
257
|
+
)
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _resolve_guidance(
|
|
261
|
+
prior: Any,
|
|
262
|
+
structure_budget: Any,
|
|
263
|
+
*,
|
|
264
|
+
budget: int,
|
|
265
|
+
model_calls: bool,
|
|
266
|
+
announce: Callable[[str], None],
|
|
267
|
+
) -> tuple[str, int]:
|
|
268
|
+
"""Turn the ``"auto"`` sentinels into the stack the measurements bought.
|
|
269
|
+
|
|
270
|
+
Two rules, and which one applies is decided by whether a model call is
|
|
271
|
+
actually possible -- not by what the caller hoped for.
|
|
272
|
+
|
|
273
|
+
Without a model call the sentinels resolve to ``"rule"`` and ``0``, which
|
|
274
|
+
are the literal pre-sentinel defaults: the credential-free path draws the
|
|
275
|
+
same candidates in the same order, and the fossil stream cannot move.
|
|
276
|
+
|
|
277
|
+
With one, the screen is sized from the budget and ``prior`` becomes
|
|
278
|
+
``"llm-weighted"`` exactly when that screen will run. Below 48 evaluations
|
|
279
|
+
both stay off: the six-arm ablation screened at 15 evaluations of 96, at a
|
|
280
|
+
small budget that share buys less than the initialization seam alone (the
|
|
281
|
+
measured winner there), and the prior seat only ever acts on a screen's
|
|
282
|
+
evidence.
|
|
283
|
+
"""
|
|
284
|
+
|
|
285
|
+
if not model_calls:
|
|
286
|
+
return ("rule" if prior == "auto" else prior,
|
|
287
|
+
0 if structure_budget == "auto" else structure_budget)
|
|
288
|
+
if structure_budget == "auto":
|
|
289
|
+
structure_budget = 0 if budget < 48 else min(16, max(8, budget // 6))
|
|
290
|
+
if structure_budget:
|
|
291
|
+
announce(
|
|
292
|
+
f"structure_budget={structure_budget} by default at budget "
|
|
293
|
+
f"{budget}: the six-arm ablation screened at 15 evaluations of "
|
|
294
|
+
"96, and the screen is charged against the same budget. Below "
|
|
295
|
+
"48 it is skipped. Pass structure_budget=0 to skip it here."
|
|
296
|
+
)
|
|
297
|
+
if prior == "auto":
|
|
298
|
+
# The prior seat only acts on a screen's evidence, so the model form
|
|
299
|
+
# is bought exactly when a screen will run. Announcing a model prior
|
|
300
|
+
# beside structure_budget=0 would be a promise the run never cashes.
|
|
301
|
+
if structure_budget:
|
|
302
|
+
prior = "llm-weighted"
|
|
303
|
+
announce(
|
|
304
|
+
"prior='llm-weighted' by default on a model run: the model "
|
|
305
|
+
"reads the crossed screen and the screen's own statistics "
|
|
306
|
+
"carry the weights (the six-arm ablation's guidance arm). "
|
|
307
|
+
"Pass prior='rule' for the credential-free comparator."
|
|
308
|
+
)
|
|
309
|
+
else:
|
|
310
|
+
prior = "rule"
|
|
311
|
+
return prior, structure_budget
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def optimize(
|
|
315
|
+
problem: Any,
|
|
316
|
+
*,
|
|
317
|
+
budget: int = 40,
|
|
318
|
+
model: Optional[str] = None,
|
|
319
|
+
proposer: str = "auto",
|
|
320
|
+
strategy: str = "auto",
|
|
321
|
+
seed: Optional[int] = None,
|
|
322
|
+
seal: Optional[str] = None,
|
|
323
|
+
on_progress: Optional[Callable[[str], None]] = None,
|
|
324
|
+
structure_budget: int | str = "auto",
|
|
325
|
+
prior: str = "auto",
|
|
326
|
+
chooser: str = "off",
|
|
327
|
+
effort: Optional[str] = None,
|
|
328
|
+
journal: Any = None,
|
|
329
|
+
authorship: Any = "auto",
|
|
330
|
+
) -> SearchResult:
|
|
331
|
+
"""Optimize *problem* within *budget* evaluations.
|
|
332
|
+
|
|
333
|
+
*budget* counts artifacts measured, which is the expensive thing and the
|
|
334
|
+
only sizing number the caller supplies. The problem's seeds are evaluated
|
|
335
|
+
before anything is proposed, so the result always answers "did this beat
|
|
336
|
+
what I already had".
|
|
337
|
+
|
|
338
|
+
*structure_budget* spends that many evaluations -- charged against the same
|
|
339
|
+
*budget* -- on a crossed screen before the population is built; *prior*
|
|
340
|
+
names who turns the screen into a sampling prior: the credential-free
|
|
341
|
+
``"rule"`` or ``"rule-weighted"``, or their model-backed forms ``"llm"`` /
|
|
342
|
+
``"llm-weighted"``, which fall back to the rule comparator, out loud, when
|
|
343
|
+
no model call is possible. Both default to ``"auto"``, which resolves
|
|
344
|
+
against what the run can actually do: without a model call, to ``"rule"``
|
|
345
|
+
and ``0`` -- the literal pre-sentinel defaults, so the credential-free path
|
|
346
|
+
stays byte-identical -- and with one, to ``"llm-weighted"`` and a screen
|
|
347
|
+
sized from the budget, announced through *on_progress* rather than picked
|
|
348
|
+
silently. ``"llm-weighted-committed"`` is the tuning-round variant under
|
|
349
|
+
measurement: ``"llm-weighted"`` with the prompt's leave-a-locus-free
|
|
350
|
+
caution swapped for evidence-proportional commitment.
|
|
351
|
+
|
|
352
|
+
*chooser* names who picks parents and cut points inside a generation, and
|
|
353
|
+
defaults to ``"off"``. ``"llm"`` buys the per-offspring chooser, which is
|
|
354
|
+
the one mechanism here that has never earned its price: ten sealed null
|
|
355
|
+
verdicts, Theta(offspring) model calls rather than one, and 61% of the
|
|
356
|
+
six-arm ablation's whole ledger consumed for 0.94x the speed of doing
|
|
357
|
+
nothing. ``"off"`` runs the random control it never beat. It needs a run
|
|
358
|
+
that makes model calls; asking for it on a run that cannot is refused
|
|
359
|
+
rather than ignored.
|
|
360
|
+
|
|
361
|
+
*effort* pins the model's reasoning effort on every completion call, and
|
|
362
|
+
*journal* (a callable, or a path to a JSONL file) receives one record per
|
|
363
|
+
completed model call -- model served plus token usage -- so a run's spend
|
|
364
|
+
is verifiable from its own artifacts. Both belong to the genetic strategy.
|
|
365
|
+
|
|
366
|
+
*seal* names a file to write the run's proposal journal to: one chained,
|
|
367
|
+
self-authenticating line per model call, holding the exact configuration
|
|
368
|
+
that was emitted, the digest of the prompt that produced it, the digest of
|
|
369
|
+
the schema it was drawn from, and the verdict ``validate`` returned. The run
|
|
370
|
+
then replays from that file with no provider and no credential. Pass it when
|
|
371
|
+
the result has to be checkable by someone who was not there.
|
|
372
|
+
"""
|
|
373
|
+
if not isinstance(budget, int) or isinstance(budget, bool) or budget < 1:
|
|
374
|
+
raise ValueError(f"budget must be a positive integer, got {budget!r}")
|
|
375
|
+
if structure_budget != "auto":
|
|
376
|
+
if (not isinstance(structure_budget, int)
|
|
377
|
+
or isinstance(structure_budget, bool) or structure_budget < 0):
|
|
378
|
+
raise ValueError(
|
|
379
|
+
f"structure_budget must be 'auto' or a non-negative integer, "
|
|
380
|
+
f"got {structure_budget!r}"
|
|
381
|
+
)
|
|
382
|
+
_check_structure_budget(structure_budget, budget)
|
|
383
|
+
if prior != "auto" and prior not in _PRIORS:
|
|
384
|
+
raise ValueError(
|
|
385
|
+
f"prior must be 'auto' or one of {sorted(_PRIORS)}, got {prior!r}")
|
|
386
|
+
if chooser not in ("off", "llm"):
|
|
387
|
+
raise ValueError(f"chooser must be 'off' or 'llm', got {chooser!r}")
|
|
388
|
+
if effort is not None and not isinstance(effort, str):
|
|
389
|
+
raise ValueError(
|
|
390
|
+
f"effort must be a provider effort level as a string, got {effort!r}"
|
|
391
|
+
)
|
|
392
|
+
from agent_evolve.session.authorship import AuthorshipConfig
|
|
393
|
+
if isinstance(authorship, AuthorshipConfig):
|
|
394
|
+
authorship_config = authorship
|
|
395
|
+
elif authorship == "auto":
|
|
396
|
+
# Resolved on the genetic branch: the model-authored surrogate is ON
|
|
397
|
+
# when a model call is possible (the sealed S1 luna-clear row held),
|
|
398
|
+
# off otherwise. The evidence-backed default, not the hopeful one.
|
|
399
|
+
authorship_config = None
|
|
400
|
+
elif isinstance(authorship, str):
|
|
401
|
+
authorship_config = AuthorshipConfig.preset(authorship)
|
|
402
|
+
else:
|
|
403
|
+
raise ValueError(
|
|
404
|
+
"authorship must be an AuthorshipConfig or a preset name, got "
|
|
405
|
+
f"{authorship!r}"
|
|
406
|
+
)
|
|
407
|
+
|
|
408
|
+
bound = as_problem(problem)
|
|
409
|
+
announce = on_progress or (lambda _message: None)
|
|
410
|
+
settings = AgentEvolveSettings.from_env()
|
|
411
|
+
|
|
412
|
+
# Arguments are validated before any branching. A caller who passes a
|
|
413
|
+
# nonsense proposer must be told so whichever loop ends up running --
|
|
414
|
+
# skipping validation on one path is how an invalid argument becomes a
|
|
415
|
+
# silent no-op.
|
|
416
|
+
kind = _resolve_proposer(proposer, announce)
|
|
417
|
+
if chooser == "llm" and kind != "llm":
|
|
418
|
+
# A chooser that cannot call a model is a chooser that never chooses,
|
|
419
|
+
# and the run would look exactly like the one that never asked for it.
|
|
420
|
+
raise ValueError(
|
|
421
|
+
f"chooser='llm' asks a model to pick parents and cut points, and "
|
|
422
|
+
f"this run resolved to the {kind!r} proposer, which makes no model "
|
|
423
|
+
"call. Pass proposer='llm' with a provider credential, or drop "
|
|
424
|
+
"chooser= to keep the random control."
|
|
425
|
+
)
|
|
426
|
+
|
|
427
|
+
seeds = tuple(dict(c) for c in bound.seeds())
|
|
428
|
+
chosen = _resolve_strategy(strategy, bool(seeds), announce)
|
|
429
|
+
if chosen == "genetic" and seal is not None:
|
|
430
|
+
# The seal journal holds generative proposals; the genetic loop's model
|
|
431
|
+
# calls are operator choices, which that format cannot represent. A
|
|
432
|
+
# journal the caller asked for and never got would be a silent no-op,
|
|
433
|
+
# so refuse loudly and name the two ways out.
|
|
434
|
+
raise ValueError(
|
|
435
|
+
"seal journaling is not supported by the genetic strategy yet: "
|
|
436
|
+
"the seal format records generative proposals, and the genetic "
|
|
437
|
+
"loop makes operator choices instead. Pass strategy='authoring' "
|
|
438
|
+
"to seal a generative run, or drop seal=."
|
|
439
|
+
)
|
|
440
|
+
if chosen != "genetic":
|
|
441
|
+
# The sentinels are read as "not asked for": ``auto`` is this package
|
|
442
|
+
# choosing, and refusing a run over a choice the caller never made
|
|
443
|
+
# would be the package arguing with itself.
|
|
444
|
+
engaged = [name for name, on in (
|
|
445
|
+
("structure_budget", structure_budget not in ("auto", 0)),
|
|
446
|
+
("prior", prior not in ("auto", "rule")),
|
|
447
|
+
("chooser", chooser == "llm"),
|
|
448
|
+
("effort", effort is not None),
|
|
449
|
+
("journal", journal is not None),
|
|
450
|
+
("authorship", authorship_config is not None
|
|
451
|
+
and authorship_config.engaged),
|
|
452
|
+
) if on]
|
|
453
|
+
if engaged:
|
|
454
|
+
# A knob the run would silently ignore is a silent no-op -- the
|
|
455
|
+
# same defect class the seal refusal above exists to prevent.
|
|
456
|
+
raise ValueError(
|
|
457
|
+
f"{', '.join(engaged)} belong(s) to the genetic strategy, and "
|
|
458
|
+
"this run resolved to 'authoring'. Give the problem a seed to "
|
|
459
|
+
"use the genetic loop, or drop the genetic-only arguments."
|
|
460
|
+
)
|
|
461
|
+
if chosen == "genetic":
|
|
462
|
+
# Only the loop is imported locally. Importing EvaluationCache here too
|
|
463
|
+
# would make that name function-local for the whole body and break the
|
|
464
|
+
# authoring path below, which uses the module-level import.
|
|
465
|
+
from agent_evolve.session.genetic_loop import GeneticConfig, run_genetic_loop
|
|
466
|
+
|
|
467
|
+
journal_handle = None
|
|
468
|
+
journal_sink: Optional[Callable[[dict], None]] = None
|
|
469
|
+
if callable(journal):
|
|
470
|
+
journal_sink = journal
|
|
471
|
+
elif journal is not None:
|
|
472
|
+
import json as _json
|
|
473
|
+
from pathlib import Path
|
|
474
|
+
|
|
475
|
+
journal_path = Path(journal)
|
|
476
|
+
journal_path.parent.mkdir(parents=True, exist_ok=True)
|
|
477
|
+
# Opened eagerly even though the run may make no call: an empty
|
|
478
|
+
# journal is a measured zero, an absent file is "nobody looked".
|
|
479
|
+
journal_handle = journal_path.open("w", encoding="utf-8")
|
|
480
|
+
|
|
481
|
+
def journal_sink(record: dict) -> None:
|
|
482
|
+
journal_handle.write(_json.dumps(record, sort_keys=True) + "\n")
|
|
483
|
+
journal_handle.flush()
|
|
484
|
+
|
|
485
|
+
try:
|
|
486
|
+
chooser_policy = None
|
|
487
|
+
complete = None
|
|
488
|
+
# Provider usage is measured from the completion seam's own
|
|
489
|
+
# journal, never declared: zero means "counted and none occurred".
|
|
490
|
+
usage_ledger = {"calls": 0, "input": 0, "output": 0, "tokens_known": True}
|
|
491
|
+
if kind == "llm":
|
|
492
|
+
# The completion seam is built for the whole run, not for one
|
|
493
|
+
# consumer. It used to be constructed inside the chooser's own
|
|
494
|
+
# branch, which meant the seams that measured well -- authored
|
|
495
|
+
# initialization, the weighted prior -- could only be bought
|
|
496
|
+
# together with the one that measured null.
|
|
497
|
+
from agent_evolve.integrations.completion import completion_for
|
|
498
|
+
|
|
499
|
+
def _record_usage(record: dict) -> None:
|
|
500
|
+
usage_ledger["calls"] += 1
|
|
501
|
+
usage = record.get("usage") or {}
|
|
502
|
+
prompt_tokens = usage.get("prompt_tokens")
|
|
503
|
+
completion_tokens = usage.get("completion_tokens")
|
|
504
|
+
if isinstance(prompt_tokens, int) and isinstance(completion_tokens, int):
|
|
505
|
+
usage_ledger["input"] += prompt_tokens
|
|
506
|
+
usage_ledger["output"] += completion_tokens
|
|
507
|
+
else:
|
|
508
|
+
usage_ledger["tokens_known"] = False
|
|
509
|
+
if journal_sink is not None:
|
|
510
|
+
journal_sink(record)
|
|
511
|
+
|
|
512
|
+
# The shipped completion ceiling comes from the profile the
|
|
513
|
+
# product already declares for the route, not from the
|
|
514
|
+
# provider's undeclared default. Sending nothing was never
|
|
515
|
+
# "no cap": it was 65,536 on the default route, against the
|
|
516
|
+
# 128,000 the profile declares -- and the half that went
|
|
517
|
+
# missing was taken from the calls that reasoned longest.
|
|
518
|
+
# An unknown route still declares nothing, and then nothing
|
|
519
|
+
# is sent, so that path keeps the pre-cap body exactly.
|
|
520
|
+
from agent_evolve.integrations.pydantic_ai.model_execution_profile import ( # noqa: E501
|
|
521
|
+
declared_max_output_tokens)
|
|
522
|
+
|
|
523
|
+
route = model or settings.model
|
|
524
|
+
cap = declared_max_output_tokens(route)
|
|
525
|
+
complete = completion_for(route, settings,
|
|
526
|
+
journal=_record_usage, effort=effort,
|
|
527
|
+
max_output_tokens=cap)
|
|
528
|
+
if complete is None:
|
|
529
|
+
# The caller asked for a model BY NAME and no credential
|
|
530
|
+
# can honour it. Falling back to the classical path here
|
|
531
|
+
# ran to completion and said nothing -- a run launched to
|
|
532
|
+
# measure a model measured the control instead, and the
|
|
533
|
+
# only trace was `calls: 0`. Found by the release CI's
|
|
534
|
+
# stranger job, 2026-08-20.
|
|
535
|
+
import importlib.util
|
|
536
|
+
raise RuntimeError(_llm_refusal_message(
|
|
537
|
+
extra_missing=importlib.util.find_spec("pydantic_ai")
|
|
538
|
+
is None))
|
|
539
|
+
if chooser == "llm":
|
|
540
|
+
if complete is None:
|
|
541
|
+
announce(
|
|
542
|
+
"chooser='llm' needs a model call and none is "
|
|
543
|
+
"available; operator choices stay random."
|
|
544
|
+
)
|
|
545
|
+
else:
|
|
546
|
+
# Guided operator choice: the model picks parents and cut
|
|
547
|
+
# points, reasoning over the accumulated search state. It
|
|
548
|
+
# cannot author a candidate -- OperatorChoice has no field
|
|
549
|
+
# that could hold one. Opt-in, because it is the one seam
|
|
550
|
+
# here with ten sealed null verdicts against it.
|
|
551
|
+
from agent_evolve.policies.llm_chooser import llm_chooser
|
|
552
|
+
from agent_evolve.policies.semantics import domain_card
|
|
553
|
+
chooser_policy = llm_chooser(
|
|
554
|
+
complete, objectives=list(bound.objectives), budget=budget,
|
|
555
|
+
domain_context=domain_card(bound),
|
|
556
|
+
on_shortfall=lambda got, want: announce(
|
|
557
|
+
f"the model supplied {got} of {want} operator choices; "
|
|
558
|
+
"the rest were filled at random"),
|
|
559
|
+
)
|
|
560
|
+
if effort is not None and complete is None:
|
|
561
|
+
announce(
|
|
562
|
+
"effort pins model reasoning, and this run makes no model "
|
|
563
|
+
"calls, so it has no effect here."
|
|
564
|
+
)
|
|
565
|
+
|
|
566
|
+
# Resolved here and not earlier: what the sentinels mean depends on
|
|
567
|
+
# whether a model call is actually possible, which is not known
|
|
568
|
+
# until the seam above has either been built or come back empty.
|
|
569
|
+
prior, structure_budget = _resolve_guidance(
|
|
570
|
+
prior, structure_budget, budget=budget,
|
|
571
|
+
model_calls=complete is not None, announce=announce)
|
|
572
|
+
_check_structure_budget(structure_budget, budget)
|
|
573
|
+
|
|
574
|
+
prior_proposer: Any = None
|
|
575
|
+
if prior == "rule-weighted":
|
|
576
|
+
from agent_evolve.policies.weighted_prior import (
|
|
577
|
+
statistical_weighted_prior)
|
|
578
|
+
prior_proposer = statistical_weighted_prior
|
|
579
|
+
elif prior in ("llm", "llm-weighted", "llm-weighted-committed"):
|
|
580
|
+
if complete is None:
|
|
581
|
+
announce(
|
|
582
|
+
f"prior={prior!r} needs a model call and none is "
|
|
583
|
+
"available; using the credential-free rule comparator "
|
|
584
|
+
"instead."
|
|
585
|
+
)
|
|
586
|
+
if prior != "llm":
|
|
587
|
+
from agent_evolve.policies.weighted_prior import (
|
|
588
|
+
statistical_weighted_prior)
|
|
589
|
+
prior_proposer = statistical_weighted_prior
|
|
590
|
+
elif prior == "llm":
|
|
591
|
+
from agent_evolve.policies.llm_prior import llm_prior_proposer
|
|
592
|
+
from agent_evolve.policies.semantics import domain_card
|
|
593
|
+
prior_proposer = llm_prior_proposer(
|
|
594
|
+
complete, objectives=list(bound.objectives),
|
|
595
|
+
domain_context=domain_card(bound))
|
|
596
|
+
else:
|
|
597
|
+
from agent_evolve.policies.semantics import domain_card
|
|
598
|
+
from agent_evolve.policies.weighted_prior import (
|
|
599
|
+
llm_weighted_prior_proposer)
|
|
600
|
+
# The two model-weighted forms differ by ONE clause of the
|
|
601
|
+
# prompt; everything downstream of the reply is shared.
|
|
602
|
+
prior_proposer = llm_weighted_prior_proposer(
|
|
603
|
+
complete, objectives=list(bound.objectives),
|
|
604
|
+
domain_context=domain_card(bound),
|
|
605
|
+
style=("committed"
|
|
606
|
+
if prior == "llm-weighted-committed"
|
|
607
|
+
else "cautious"))
|
|
608
|
+
|
|
609
|
+
if authorship_config is None:
|
|
610
|
+
if complete is not None:
|
|
611
|
+
authorship_config = AuthorshipConfig(surrogate="llm",
|
|
612
|
+
initialization="llm")
|
|
613
|
+
announce(
|
|
614
|
+
"authorship: model-authored surrogate screening is ON "
|
|
615
|
+
"(the sealed luna-clear row held), and model-proposed "
|
|
616
|
+
"initialization is ON -- the six-arm ablation's "
|
|
617
|
+
"strongest arm, at 11x fewer evaluations to target, "
|
|
618
|
+
"better on 40 of 40 paired seeds, for one call; pass "
|
|
619
|
+
"authorship='off' to disable.")
|
|
620
|
+
else:
|
|
621
|
+
authorship_config = AuthorshipConfig()
|
|
622
|
+
# One rule, stated once, in `_genetic_sizing`: the sealed
|
|
623
|
+
# expression at and below the sealed ceiling, a population that
|
|
624
|
+
# grows with the budget above it.
|
|
625
|
+
pop, offspring = _genetic_sizing(budget)
|
|
626
|
+
|
|
627
|
+
from agent_evolve.policies.semantics import domain_card
|
|
628
|
+
from agent_evolve.session.authorship import build_authorship
|
|
629
|
+
policies = build_authorship(
|
|
630
|
+
authorship_config, complete=complete,
|
|
631
|
+
objectives=list(bound.objectives),
|
|
632
|
+
schema_text=domain_card(bound), seed=seed, announce=announce,
|
|
633
|
+
candidate_model=getattr(bound, "candidate_model", None),
|
|
634
|
+
init_template=(dict(seeds[0]) if seeds else None),
|
|
635
|
+
init_k=max(0, pop - len(seeds)),
|
|
636
|
+
budget=budget, population_size=pop)
|
|
637
|
+
|
|
638
|
+
cache = EvaluationCache()
|
|
639
|
+
cache.budget = budget
|
|
640
|
+
# `generations` is a cap, not a schedule. Duplicate offspring hit
|
|
641
|
+
# the evaluation cache without spending budget, so a fixed
|
|
642
|
+
# generation count would end the run with budget unspent; the
|
|
643
|
+
# loop's real stop condition is the budget.
|
|
644
|
+
result = run_genetic_loop(
|
|
645
|
+
problem=bound,
|
|
646
|
+
config=GeneticConfig(
|
|
647
|
+
population_size=pop,
|
|
648
|
+
offspring_per_generation=offspring,
|
|
649
|
+
generations=max(1, 4 * budget // max(1, offspring)),
|
|
650
|
+
seed=seed,
|
|
651
|
+
seeds=seeds,
|
|
652
|
+
evaluation_budget=budget,
|
|
653
|
+
evaluation_cache=cache,
|
|
654
|
+
structure_budget=structure_budget,
|
|
655
|
+
prior_proposer=prior_proposer,
|
|
656
|
+
screening=policies.screening,
|
|
657
|
+
portfolio=policies.portfolio,
|
|
658
|
+
initial_proposals=policies.initial_proposals,
|
|
659
|
+
generator=policies.generator,
|
|
660
|
+
reguidance=policies.reguidance,
|
|
661
|
+
),
|
|
662
|
+
chooser=chooser_policy,
|
|
663
|
+
log=announce,
|
|
664
|
+
)
|
|
665
|
+
# Authoring that produced no policy object still produced
|
|
666
|
+
# counters, and the loop can only harvest what it was handed.
|
|
667
|
+
# These are the seams whose failure leaves nothing behind.
|
|
668
|
+
orphaned = tuple(note for note in (policies.init_author,
|
|
669
|
+
policies.generator_author,
|
|
670
|
+
policies.reguidance_author)
|
|
671
|
+
if note is not None)
|
|
672
|
+
if orphaned and result.telemetry is not None:
|
|
673
|
+
from agent_evolve.core.telemetry import harvest_telemetry
|
|
674
|
+
extra = harvest_telemetry(orphaned)
|
|
675
|
+
result = dataclasses.replace(
|
|
676
|
+
result,
|
|
677
|
+
telemetry=dataclasses.replace(
|
|
678
|
+
result.telemetry,
|
|
679
|
+
mechanisms=result.telemetry.mechanisms + extra.mechanisms))
|
|
680
|
+
if usage_ledger["calls"] and usage_ledger["tokens_known"]:
|
|
681
|
+
# Cost is DERIVED, and the reporter says so. The tokens are the
|
|
682
|
+
# provider's own count; the price is this package's published
|
|
683
|
+
# table (`MODEL_PRICES_PER_MTOK`), which is the same number the
|
|
684
|
+
# CLI echoes before spending anything. A route the table does
|
|
685
|
+
# not name reports `cost_usd: null` -- unknown stays unknown
|
|
686
|
+
# rather than becoming a guess with a dollar sign on it.
|
|
687
|
+
route = model or settings.model
|
|
688
|
+
cost_usd, reporter = _priced_usage(
|
|
689
|
+
route, usage_ledger["input"], usage_ledger["output"])
|
|
690
|
+
usage = ProviderUsageSummary(
|
|
691
|
+
calls=usage_ledger["calls"],
|
|
692
|
+
input_tokens=usage_ledger["input"],
|
|
693
|
+
output_tokens=usage_ledger["output"],
|
|
694
|
+
cost_usd=cost_usd,
|
|
695
|
+
model=route,
|
|
696
|
+
reported_by=reporter,
|
|
697
|
+
)
|
|
698
|
+
else:
|
|
699
|
+
usage = ProviderUsageSummary(
|
|
700
|
+
calls=usage_ledger["calls"],
|
|
701
|
+
model=(model or settings.model) if usage_ledger["calls"] else None,
|
|
702
|
+
)
|
|
703
|
+
return dataclasses.replace(result, provider_usage=usage)
|
|
704
|
+
finally:
|
|
705
|
+
# Closed even when the run raises: a journal truncated by a crash
|
|
706
|
+
# still records every call that did happen.
|
|
707
|
+
if journal_handle is not None:
|
|
708
|
+
journal_handle.close()
|
|
709
|
+
|
|
710
|
+
harness = _build_harness(kind, seed, settings)
|
|
711
|
+
|
|
712
|
+
ctx = HarnessContext(
|
|
713
|
+
objectives=list(bound.objectives),
|
|
714
|
+
search_space_desc=_describe(bound),
|
|
715
|
+
candidate_model=getattr(bound, "candidate_model", None),
|
|
716
|
+
constraints_description=getattr(bound, "constraints_description", "") or "",
|
|
717
|
+
directives=getattr(bound, "directives", None) or DefaultDirectives(),
|
|
718
|
+
)
|
|
719
|
+
harness.bind(
|
|
720
|
+
ctx,
|
|
721
|
+
LLMConfig(model=model or settings.model, temperature=settings.temperature),
|
|
722
|
+
)
|
|
723
|
+
|
|
724
|
+
seal_handle = None
|
|
725
|
+
if seal is not None:
|
|
726
|
+
from pathlib import Path
|
|
727
|
+
|
|
728
|
+
from agent_evolve.application.generative_proposal_journal import journal_line
|
|
729
|
+
from agent_evolve.proposal_mode import build_generative_proposer
|
|
730
|
+
|
|
731
|
+
path = Path(seal)
|
|
732
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
733
|
+
seal_handle = path.open("w", encoding="ascii")
|
|
734
|
+
|
|
735
|
+
def _write(record: dict) -> None:
|
|
736
|
+
seal_handle.write(journal_line(record) + "\n")
|
|
737
|
+
seal_handle.flush()
|
|
738
|
+
|
|
739
|
+
harness = build_generative_proposer(bound, delegate=harness, on_seal=_write)
|
|
740
|
+
harness.bind(
|
|
741
|
+
ctx,
|
|
742
|
+
LLMConfig(model=model or settings.model, temperature=settings.temperature),
|
|
743
|
+
)
|
|
744
|
+
|
|
745
|
+
cache = EvaluationCache() # `seeds` was already read above, once
|
|
746
|
+
cache.budget = budget
|
|
747
|
+
config = LoopConfig(
|
|
748
|
+
pop_size=min(budget, _BATCH),
|
|
749
|
+
generations=max(1, budget // _BATCH),
|
|
750
|
+
candidates_per_batch=_BATCH,
|
|
751
|
+
seed=seed,
|
|
752
|
+
seeds=seeds,
|
|
753
|
+
evaluation_budget=budget,
|
|
754
|
+
evaluation_cache=cache,
|
|
755
|
+
)
|
|
756
|
+
try:
|
|
757
|
+
return run_evolution_loop(
|
|
758
|
+
problem=bound,
|
|
759
|
+
harness=harness,
|
|
760
|
+
config=config,
|
|
761
|
+
log=announce,
|
|
762
|
+
)
|
|
763
|
+
finally:
|
|
764
|
+
# Closed even when the run raises: a journal truncated by a crash still
|
|
765
|
+
# records every call that did happen, and that is the honest artifact.
|
|
766
|
+
if seal_handle is not None:
|
|
767
|
+
seal_handle.close()
|