agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,671 @@
|
|
|
1
|
+
"""Virtual pre-screening: order a pool of offspring before paying for any.
|
|
2
|
+
|
|
3
|
+
This module is deliberately starved: :func:`screen_offspring` receives
|
|
4
|
+
configurations and objective vectors and a predictor -- never the problem,
|
|
5
|
+
never the evaluation cache -- so "the surrogate cannot spend budget" is a
|
|
6
|
+
property of the import graph, not a convention. The only route from here to
|
|
7
|
+
a real evaluation is that the loop measures the candidates this module
|
|
8
|
+
merely ordered.
|
|
9
|
+
|
|
10
|
+
Because ORDERING is the whole of what this module consumes, it validates its
|
|
11
|
+
surrogates under :data:`~agent_evolve.policies.surrogate.ORDERING_GATE`:
|
|
12
|
+
rank fidelity rejects, and the error ratio against the train-mean predictor
|
|
13
|
+
is computed for arbitration among passers. The screen is the reason that
|
|
14
|
+
distinction exists -- under a gate that also rejected on magnitude it went
|
|
15
|
+
dark on the venues where saving an evaluation is worth anything, ordering 7
|
|
16
|
+
of 186 generations on an expensive venue against 54% on a cheap one.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import math
|
|
22
|
+
import statistics
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
from typing import Any, Dict, Mapping, Optional, Sequence, Tuple
|
|
25
|
+
|
|
26
|
+
from agent_evolve.core.problem import ObjectiveSpec
|
|
27
|
+
from agent_evolve.policies.surrogate import (
|
|
28
|
+
ORDERING_GATE,
|
|
29
|
+
GatePolicy,
|
|
30
|
+
Predict,
|
|
31
|
+
SurrogateBuilder,
|
|
32
|
+
validate_surrogate,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
__all__ = ["ScreenReport", "screen_offspring", "Screening", "ScreeningTelemetry"]
|
|
36
|
+
|
|
37
|
+
Config = Dict[str, Any]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class ScreenReport:
|
|
42
|
+
"""The pool, ordered by predicted worth. Indices address the caller's pool.
|
|
43
|
+
|
|
44
|
+
``screened_objectives`` names the objectives the order was actually
|
|
45
|
+
computed over, and ``declared_objectives`` names the problem's. They
|
|
46
|
+
differ when the gate certified the surrogate on only some of them. A
|
|
47
|
+
consumer that reads ``order`` without reading these two is free to
|
|
48
|
+
believe the pool was ranked on the whole problem when it was ranked on
|
|
49
|
+
part of it, so both travel with the order rather than beside it.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
order: Tuple[int, ...]
|
|
53
|
+
predicted: Tuple[Mapping[str, float], ...]
|
|
54
|
+
virtual_evaluations: int
|
|
55
|
+
surrogate_name: str
|
|
56
|
+
screened_objectives: Tuple[str, ...] = ()
|
|
57
|
+
declared_objectives: Tuple[str, ...] = ()
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def partial(self) -> bool:
|
|
61
|
+
"""Was this order computed over a STRICT SUBSET of the objectives?"""
|
|
62
|
+
|
|
63
|
+
return len(self.screened_objectives) < len(self.declared_objectives)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _oriented(
|
|
67
|
+
rows: Sequence[Mapping[str, float]], specs: Sequence[ObjectiveSpec]
|
|
68
|
+
) -> Optional[list[tuple]]:
|
|
69
|
+
"""Objective vectors as "smaller is better" float tuples, or ``None``.
|
|
70
|
+
|
|
71
|
+
``None`` means a row was missing an objective or carried something that
|
|
72
|
+
is not a finite number -- the same "screen nothing" answer this module
|
|
73
|
+
already gives for malformed predictions, rather than an exception from
|
|
74
|
+
the middle of a ranking loop.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
signs = [1.0 if spec.goal == "min" else -1.0 for spec in specs]
|
|
78
|
+
names = [spec.name for spec in specs]
|
|
79
|
+
out: list[tuple] = []
|
|
80
|
+
for row in rows:
|
|
81
|
+
vector = []
|
|
82
|
+
for sign, name in zip(signs, names):
|
|
83
|
+
value = row.get(name)
|
|
84
|
+
if (value is None or isinstance(value, bool)
|
|
85
|
+
or not isinstance(value, (int, float))
|
|
86
|
+
or not math.isfinite(float(value))):
|
|
87
|
+
return None
|
|
88
|
+
vector.append(sign * float(value))
|
|
89
|
+
out.append(tuple(vector))
|
|
90
|
+
return out
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _dominator_counts(
|
|
94
|
+
rows: Sequence[Mapping[str, float]],
|
|
95
|
+
against: Sequence[Mapping[str, float]],
|
|
96
|
+
specs: Sequence[ObjectiveSpec],
|
|
97
|
+
) -> Optional[list[int]]:
|
|
98
|
+
"""How many of ``rows + against`` dominate each of *rows*. Lower is better.
|
|
99
|
+
|
|
100
|
+
Exactly what ``sum(1 for other in field if dominates(other, row))`` says,
|
|
101
|
+
computed over DISTINCT objective vectors weighted by how many rows carry
|
|
102
|
+
each. Dominance is a property of the vector alone, so this returns the
|
|
103
|
+
same integers -- but a pool of n candidates whose predictions take k
|
|
104
|
+
distinct values costs O(k^2) instead of O(n^2), and a surrogate over a
|
|
105
|
+
discrete space collapses thousands of candidates onto tens of vectors.
|
|
106
|
+
The comparison itself works on pre-oriented float tuples: the general
|
|
107
|
+
``core.results.dominates`` re-validates every objective on every call
|
|
108
|
+
(a ``numbers.Real`` ABC check per number), which is right for a contract
|
|
109
|
+
boundary and ruinous inside a quadratic loop -- measured at 21.7s of a
|
|
110
|
+
25.6s run before this, on one screened pool of 2,000.
|
|
111
|
+
"""
|
|
112
|
+
|
|
113
|
+
keys = _oriented(rows, specs)
|
|
114
|
+
other_keys = _oriented(against, specs)
|
|
115
|
+
if keys is None or other_keys is None:
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
multiplicity: Dict[tuple, int] = {}
|
|
119
|
+
for key in keys:
|
|
120
|
+
multiplicity[key] = multiplicity.get(key, 0) + 1
|
|
121
|
+
for key in other_keys:
|
|
122
|
+
multiplicity[key] = multiplicity.get(key, 0) + 1
|
|
123
|
+
distinct = list(multiplicity.items())
|
|
124
|
+
|
|
125
|
+
def _beats(a: tuple, b: tuple) -> bool:
|
|
126
|
+
better = False
|
|
127
|
+
for x, y in zip(a, b):
|
|
128
|
+
if x > y:
|
|
129
|
+
return False
|
|
130
|
+
if x < y:
|
|
131
|
+
better = True
|
|
132
|
+
return better
|
|
133
|
+
|
|
134
|
+
counted = {
|
|
135
|
+
key: sum(weight for other, weight in distinct if _beats(other, key))
|
|
136
|
+
for key, _weight in distinct
|
|
137
|
+
}
|
|
138
|
+
return [counted[key] for key in keys]
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def screen_offspring(
|
|
142
|
+
pool: Sequence[Config],
|
|
143
|
+
population_objectives: Sequence[Mapping[str, float]],
|
|
144
|
+
specs: Sequence[ObjectiveSpec],
|
|
145
|
+
predict: Predict,
|
|
146
|
+
*,
|
|
147
|
+
surrogate_name: str = "surrogate",
|
|
148
|
+
objectives: Optional[Sequence[str]] = None,
|
|
149
|
+
) -> Optional[ScreenReport]:
|
|
150
|
+
"""Order *pool* by predicted domination against the measured population.
|
|
151
|
+
|
|
152
|
+
A candidate's rank counts how many points dominate its PREDICTION -- other
|
|
153
|
+
predictions and the population's real measurements together -- so a pool
|
|
154
|
+
member that merely reshuffles known-dominated territory sinks, and one
|
|
155
|
+
predicted past the current front rises. Never scalarized. ``None`` (from
|
|
156
|
+
the predictor, or on malformed predictions) means "screen nothing": the
|
|
157
|
+
caller falls back to measuring its original picks.
|
|
158
|
+
|
|
159
|
+
This function consumes an ORDER and nothing else: the returned
|
|
160
|
+
``predicted`` rows are telemetry, and the loop reads only ``order``. That
|
|
161
|
+
is why the gate this screen validates under is
|
|
162
|
+
:data:`~agent_evolve.policies.surrogate.ORDERING_GATE` -- rank fidelity
|
|
163
|
+
is what the output can be wrong about, and a calibration test on
|
|
164
|
+
magnitudes nobody reads can only reject artifacts that would have
|
|
165
|
+
ordered correctly.
|
|
166
|
+
|
|
167
|
+
``objectives`` restricts the domination test to the objectives the gate
|
|
168
|
+
certified this surrogate for; ``None`` means all of them. **The excluded
|
|
169
|
+
objectives are treated as UNKNOWN, not as satisfied**: they are neither
|
|
170
|
+
read from the prediction nor compared, so a surrogate that emits nonsense
|
|
171
|
+
on an objective it was not certified for cannot influence the order
|
|
172
|
+
through it. That is a deliberate asymmetry with a cost, stated here
|
|
173
|
+
because a caller must weigh it: domination over a subset is a STRICTER
|
|
174
|
+
relation than domination over the whole (more pairs compare, fewer are
|
|
175
|
+
incomparable), so a candidate that is excellent only on an excluded
|
|
176
|
+
objective is dominated on the subset and sinks. The screen is therefore
|
|
177
|
+
biased against exactly the trade-off it cannot see, and the caller's
|
|
178
|
+
exploration floor -- not this function -- is what keeps unscreened picks
|
|
179
|
+
in the generation (see :meth:`Screening.exploration_floor_for`).
|
|
180
|
+
"""
|
|
181
|
+
|
|
182
|
+
if not pool:
|
|
183
|
+
return None
|
|
184
|
+
names = [s.name for s in specs]
|
|
185
|
+
if objectives is None:
|
|
186
|
+
screened = list(names)
|
|
187
|
+
else:
|
|
188
|
+
wanted = set(objectives)
|
|
189
|
+
unknown = wanted - set(names)
|
|
190
|
+
if unknown:
|
|
191
|
+
raise ValueError(
|
|
192
|
+
"objectives to screen on must be declared objectives; "
|
|
193
|
+
f"{sorted(unknown)} are not among {names}")
|
|
194
|
+
screened = [name for name in names if name in wanted]
|
|
195
|
+
# Ordering on nothing is not ordering. A caller that reaches here with an
|
|
196
|
+
# empty subset has a gate bug, and screening the pool by index would hide
|
|
197
|
+
# it behind a plausible-looking order.
|
|
198
|
+
if not screened:
|
|
199
|
+
return None
|
|
200
|
+
specs = [spec for spec in specs if spec.name in set(screened)]
|
|
201
|
+
predictions = predict(list(pool))
|
|
202
|
+
if predictions is None or len(predictions) != len(pool):
|
|
203
|
+
return None
|
|
204
|
+
clean: list[dict[str, float]] = []
|
|
205
|
+
for predicted in predictions:
|
|
206
|
+
row = {}
|
|
207
|
+
for name in screened:
|
|
208
|
+
value = predicted.get(name) if isinstance(predicted, Mapping) else None
|
|
209
|
+
if (value is None or isinstance(value, bool)
|
|
210
|
+
or not isinstance(value, (int, float))
|
|
211
|
+
or not math.isfinite(float(value))):
|
|
212
|
+
return None
|
|
213
|
+
row[name] = float(value)
|
|
214
|
+
clean.append(row)
|
|
215
|
+
|
|
216
|
+
ranks = _dominator_counts(
|
|
217
|
+
clean, [dict(measured) for measured in population_objectives], specs)
|
|
218
|
+
if ranks is None:
|
|
219
|
+
return None
|
|
220
|
+
order = tuple(sorted(range(len(pool)), key=lambda i: (ranks[i], i)))
|
|
221
|
+
return ScreenReport(
|
|
222
|
+
order=order,
|
|
223
|
+
predicted=tuple(clean),
|
|
224
|
+
virtual_evaluations=len(pool),
|
|
225
|
+
surrogate_name=surrogate_name,
|
|
226
|
+
screened_objectives=tuple(screened),
|
|
227
|
+
declared_objectives=tuple(names),
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
class ScreeningTelemetry:
|
|
232
|
+
"""What the screen did, counted. Reaches the result via harvest."""
|
|
233
|
+
|
|
234
|
+
#: Why a builder was rejected, counted per (builder, split) verdict.
|
|
235
|
+
#: Without this a campaign cannot tell "the model is not predictive"
|
|
236
|
+
#: from "the gate never had enough data to look" -- which is exactly the
|
|
237
|
+
#: distinction that turned out to decide whether the mechanism runs at
|
|
238
|
+
#: all -- and every study that needed it had to monkeypatch the gate.
|
|
239
|
+
REJECTIONS = ("rejected_insufficient_rows", "rejected_insufficient_holdout",
|
|
240
|
+
"rejected_rank", "rejected_error", "rejected_builder_failed",
|
|
241
|
+
"rejected_no_predictions", "rejected_bad_prediction")
|
|
242
|
+
|
|
243
|
+
#: The cheap fidelity's own counters, kept BESIDE the charged ones and
|
|
244
|
+
#: never added to them. `proxy_rows_used` is how many gate rows came from
|
|
245
|
+
#: the cheap evaluator on the last refresh; `chosen_proxy` counts the
|
|
246
|
+
#: generations the cheap fidelity itself won the gate and did the
|
|
247
|
+
#: ordering.
|
|
248
|
+
PROXY = ("proxy_rows_used", "chosen_proxy")
|
|
249
|
+
#: ``rejected_unstable_subset`` is counted separately and is NOT a gate
|
|
250
|
+
#: reason: every split passed, but they certified different objectives,
|
|
251
|
+
#: so the artifact is not stably predictive on enough of them. It exists
|
|
252
|
+
#: because a partial verdict makes that failure possible for the first
|
|
253
|
+
#: time, and a campaign must be able to see it rather than read it as
|
|
254
|
+
#: "the gate never had data".
|
|
255
|
+
|
|
256
|
+
#: The prefix under which ``as_dict`` reports, per objective, how many
|
|
257
|
+
#: screens ordered on it. A run that screened on two of three objectives
|
|
258
|
+
#: says so here in a form no reader can mistake for "screened on all
|
|
259
|
+
#: three", and it says it WITHOUT this module knowing any objective name.
|
|
260
|
+
SCREENED_PREFIX = "screened_on:"
|
|
261
|
+
|
|
262
|
+
__slots__ = ("refreshes", "validated", "rejected_validation", "screens",
|
|
263
|
+
"screen_failures", "virtual_evaluations", "chosen_llm",
|
|
264
|
+
"chosen_rule", "revisions", "revisions_accepted",
|
|
265
|
+
"gate_calls", "screens_full", "screens_partial",
|
|
266
|
+
"rejected_unstable_subset",
|
|
267
|
+
"_screened") + REJECTIONS + PROXY
|
|
268
|
+
|
|
269
|
+
def __init__(self) -> None:
|
|
270
|
+
for name in self.__slots__:
|
|
271
|
+
setattr(self, name, 0)
|
|
272
|
+
#: objective name -> screens whose order was computed over it.
|
|
273
|
+
self._screened: Dict[str, int] = {}
|
|
274
|
+
|
|
275
|
+
def record(self, verdict: Any) -> None:
|
|
276
|
+
"""Count one gate verdict, passed or rejected and why."""
|
|
277
|
+
|
|
278
|
+
self.gate_calls += 1
|
|
279
|
+
if verdict.passed:
|
|
280
|
+
return
|
|
281
|
+
name = f"rejected_{verdict.reason or 'unknown'}"
|
|
282
|
+
if name in self.REJECTIONS:
|
|
283
|
+
setattr(self, name, getattr(self, name) + 1)
|
|
284
|
+
|
|
285
|
+
def record_screen(self, report: Any) -> None:
|
|
286
|
+
"""Count one screen, and WHICH objectives its order was computed over.
|
|
287
|
+
|
|
288
|
+
This is a correctness requirement, not a nicety: an order over a
|
|
289
|
+
subset is a different object from an order over the whole problem,
|
|
290
|
+
and a run that cannot distinguish them can report an endpoint it
|
|
291
|
+
cannot attribute.
|
|
292
|
+
"""
|
|
293
|
+
|
|
294
|
+
if report.partial:
|
|
295
|
+
self.screens_partial += 1
|
|
296
|
+
else:
|
|
297
|
+
self.screens_full += 1
|
|
298
|
+
for name in report.screened_objectives:
|
|
299
|
+
self._screened[name] = self._screened.get(name, 0) + 1
|
|
300
|
+
|
|
301
|
+
def as_dict(self) -> dict[str, int]:
|
|
302
|
+
counters = {name: getattr(self, name) for name in self.__slots__
|
|
303
|
+
if not name.startswith("_")}
|
|
304
|
+
for name, count in sorted(self._screened.items()):
|
|
305
|
+
counters[f"{self.SCREENED_PREFIX}{name}"] = count
|
|
306
|
+
return counters
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
class Screening:
|
|
310
|
+
"""The screening policy: builders, the gate, and the current predictor.
|
|
311
|
+
|
|
312
|
+
``builders`` is an ordered sequence of ``(name, authored_by, builder)``;
|
|
313
|
+
each generation, :meth:`refresh` re-validates them in order on the data
|
|
314
|
+
measured so far and installs the FIRST that passes the gate -- so an
|
|
315
|
+
authored surrogate listed ahead of the rules is used exactly when it
|
|
316
|
+
earns it, and the rules are the standing fallback.
|
|
317
|
+
"""
|
|
318
|
+
|
|
319
|
+
def __init__(
|
|
320
|
+
self,
|
|
321
|
+
builders: Sequence[Tuple[str, str, SurrogateBuilder]],
|
|
322
|
+
*,
|
|
323
|
+
pool_factor: int = 4,
|
|
324
|
+
exploration_floor: float = 0.25,
|
|
325
|
+
unscreened_objective_floor: float = 1.0,
|
|
326
|
+
validation_splits: int = 3,
|
|
327
|
+
revise: Any = None,
|
|
328
|
+
max_revisions: int = 2,
|
|
329
|
+
max_training_rows: int = 1024,
|
|
330
|
+
gate: GatePolicy = ORDERING_GATE,
|
|
331
|
+
) -> None:
|
|
332
|
+
if pool_factor < 2:
|
|
333
|
+
raise ValueError(f"pool_factor must be at least 2, got {pool_factor}")
|
|
334
|
+
if max_training_rows < 1:
|
|
335
|
+
raise ValueError(
|
|
336
|
+
f"max_training_rows must be at least 1, got {max_training_rows}"
|
|
337
|
+
)
|
|
338
|
+
if not 0.0 <= exploration_floor < 1.0:
|
|
339
|
+
raise ValueError(
|
|
340
|
+
f"exploration_floor must be in [0, 1), got {exploration_floor}"
|
|
341
|
+
)
|
|
342
|
+
if not 0.0 <= unscreened_objective_floor <= 1.0:
|
|
343
|
+
raise ValueError(
|
|
344
|
+
"unscreened_objective_floor must be in [0, 1], got "
|
|
345
|
+
f"{unscreened_objective_floor}"
|
|
346
|
+
)
|
|
347
|
+
if validation_splits < 1:
|
|
348
|
+
raise ValueError(
|
|
349
|
+
f"validation_splits must be at least 1, got {validation_splits}"
|
|
350
|
+
)
|
|
351
|
+
if not isinstance(gate, GatePolicy):
|
|
352
|
+
raise TypeError(f"gate must be a GatePolicy, got {type(gate).__name__}")
|
|
353
|
+
self.builders = tuple(builders)
|
|
354
|
+
self.pool_factor = int(pool_factor)
|
|
355
|
+
self.exploration_floor = float(exploration_floor)
|
|
356
|
+
#: How much of a generation is reserved from the screen when the gate
|
|
357
|
+
#: certified the surrogate on only SOME objectives, expressed as a
|
|
358
|
+
#: multiple of the share of objectives the screen is blind to.
|
|
359
|
+
#:
|
|
360
|
+
#: The screen orders by domination over the certified subset, and
|
|
361
|
+
#: domination over a subset is a stricter relation than domination
|
|
362
|
+
#: over the whole problem: a candidate that is excellent only on an
|
|
363
|
+
#: excluded objective is dominated on the subset and sinks. So a
|
|
364
|
+
#: partial screen is not merely less informed than a full one, it is
|
|
365
|
+
#: SYSTEMATICALLY biased against the objectives it cannot see, and
|
|
366
|
+
#: the flat 0.25 floor -- sized for a screen that might be wrong,
|
|
367
|
+
#: not for one that is wrong in a known direction -- is not the right
|
|
368
|
+
#: protection. At 1.0 (the default) the reserved share is the
|
|
369
|
+
#: unscreened share of the objectives: 1/3 of the generation stays
|
|
370
|
+
#: unscreened when 2 of 3 objectives are certified, 2/3 when 1 of 3
|
|
371
|
+
#: is, and the ordinary ``exploration_floor`` still applies as a
|
|
372
|
+
#: lower bound. At 0.0 a partial screen is treated exactly like a
|
|
373
|
+
#: full one, which is the arm this default was measured against.
|
|
374
|
+
self.unscreened_objective_floor = float(unscreened_objective_floor)
|
|
375
|
+
self.validation_splits = int(validation_splits)
|
|
376
|
+
#: What this consumer relies on, declared to the gate rather than
|
|
377
|
+
#: assumed by it. The screen consumes an ORDER (`screen_offspring`
|
|
378
|
+
#: reads `report.order` and nothing else), so rank fidelity is the
|
|
379
|
+
#: hard term and the error ratio arbitrates among passers. Overriding
|
|
380
|
+
#: this with a prediction-purpose policy restores the historical
|
|
381
|
+
#: behaviour, at the historical cost: a magnitude test on a small
|
|
382
|
+
#: holdout rejects most of the artifacts that would have ordered
|
|
383
|
+
#: correctly.
|
|
384
|
+
self.gate = gate
|
|
385
|
+
#: The evolving-surrogate hook: called with (evaluated, specs) when
|
|
386
|
+
#: the llm builder exists and did not win this refresh, at most
|
|
387
|
+
#: max_revisions times per run. Returns a replacement
|
|
388
|
+
#: (name, authored_by, builder) entry -- authored from the current
|
|
389
|
+
#: artifact plus its measured validation residuals -- or None. The
|
|
390
|
+
#: revision competes from the NEXT refresh under the same gate; a
|
|
391
|
+
#: model that cannot fix its artifact keeps losing to the rules.
|
|
392
|
+
self.revise = revise
|
|
393
|
+
self.max_revisions = int(max_revisions)
|
|
394
|
+
#: How many of the most recent measurements a refresh fits and
|
|
395
|
+
#: validates on. Refitting every builder on EVERYTHING measured so
|
|
396
|
+
#: far makes one refresh O(n) and a run O(n^2): at B=10,000 the screen
|
|
397
|
+
#: alone runs for over a quarter of an hour and never finishes a run,
|
|
398
|
+
#: which is precisely the regime an authored generator exists for.
|
|
399
|
+
#: The recent window is also the better statistics for a distribution
|
|
400
|
+
#: the search keeps moving. The default is far above any campaign run
|
|
401
|
+
#: to date (all at B <= 150), so every measured run is unaffected.
|
|
402
|
+
self.max_training_rows = int(max_training_rows)
|
|
403
|
+
self.telemetry = ScreeningTelemetry()
|
|
404
|
+
self.mechanism = "surrogate_screen"
|
|
405
|
+
self.authored_by = "none"
|
|
406
|
+
self._predict: Optional[Predict] = None
|
|
407
|
+
self._name = ""
|
|
408
|
+
#: The cheap fidelity, if the problem has one and the loop attached
|
|
409
|
+
#: it. `_proxy_rows` is gate EVIDENCE bought at that fidelity: it is
|
|
410
|
+
#: keyed so a real measurement always supersedes it, it never reaches
|
|
411
|
+
#: the archive, the population or the budget, and it is counted in
|
|
412
|
+
#: the source's own ledger.
|
|
413
|
+
self._proxy: Any = None
|
|
414
|
+
self._proxy_mode = "off"
|
|
415
|
+
self._proxy_rows: Dict[str, Tuple[Config, Mapping[str, float]]] = {}
|
|
416
|
+
|
|
417
|
+
# ------------------------------------------------------------------ proxy
|
|
418
|
+
def attach_proxy(self, source: Any, *, mode: str = "rows") -> None:
|
|
419
|
+
"""Let this screen spend a CHEAPER evaluation fidelity.
|
|
420
|
+
|
|
421
|
+
``mode="rows"`` -- the cheap evaluator buys gate EVIDENCE: rows the
|
|
422
|
+
campaign could not afford at full price, so the gate can reach a
|
|
423
|
+
verdict at budgets where the run has not yet measured its minimum
|
|
424
|
+
number of rows. This is the term that closes "too few rows", which is
|
|
425
|
+
a property of the budget and which no gate policy can fix.
|
|
426
|
+
|
|
427
|
+
``mode="screen"`` -- the cheap evaluator competes AS a surrogate,
|
|
428
|
+
first in the builder order, and is cross-validated against the run's
|
|
429
|
+
own real measurements exactly like an authored artifact. This is the
|
|
430
|
+
term that can close a rank veto, and only if the cheap fidelity
|
|
431
|
+
really does rank the expensive one.
|
|
432
|
+
|
|
433
|
+
``mode="both"`` -- both. ``mode="off"`` -- neither, and the screen is
|
|
434
|
+
then byte-identical to a screen with no proxy at all.
|
|
435
|
+
"""
|
|
436
|
+
|
|
437
|
+
if mode not in ("off", "rows", "screen", "both"):
|
|
438
|
+
raise ValueError(
|
|
439
|
+
"proxy mode must be 'off', 'rows', 'screen' or 'both', "
|
|
440
|
+
f"got {mode!r}")
|
|
441
|
+
if mode == "off" or source is None:
|
|
442
|
+
self._proxy, self._proxy_mode = None, "off"
|
|
443
|
+
return
|
|
444
|
+
self._proxy = source
|
|
445
|
+
self._proxy_mode = mode
|
|
446
|
+
if mode in ("screen", "both"):
|
|
447
|
+
from agent_evolve.session.fidelity import proxy_fidelity_builder
|
|
448
|
+
name = f"proxy:{getattr(source, 'name', 'proxy')}"
|
|
449
|
+
if not any(entry[0] == name for entry in self.builders):
|
|
450
|
+
self.builders = ((name, "proxy", proxy_fidelity_builder(source)),
|
|
451
|
+
) + self.builders
|
|
452
|
+
|
|
453
|
+
def prime(self, candidates: Sequence[Config],
|
|
454
|
+
measured_keys: Sequence[str] = ()) -> int:
|
|
455
|
+
"""Buy gate evidence at the cheap fidelity. Returns rows added.
|
|
456
|
+
|
|
457
|
+
Called by the loop with the candidates it is about to consider. Rows
|
|
458
|
+
already measured for real are skipped, and any cheap row whose
|
|
459
|
+
candidate later gets measured is dropped by :meth:`refresh` -- cheap
|
|
460
|
+
evidence exists to fill a hole, never to outvote the real thing.
|
|
461
|
+
"""
|
|
462
|
+
|
|
463
|
+
if self._proxy is None or self._proxy_mode not in ("rows", "both"):
|
|
464
|
+
return 0
|
|
465
|
+
known = set(measured_keys)
|
|
466
|
+
added = 0
|
|
467
|
+
for config, values in self._proxy.rows(candidates, exclude=known):
|
|
468
|
+
token = self._proxy.key(config)
|
|
469
|
+
if token in self._proxy_rows:
|
|
470
|
+
continue
|
|
471
|
+
self._proxy_rows[token] = (config, values)
|
|
472
|
+
added += 1
|
|
473
|
+
self._proxy.ledger.rows_used = len(self._proxy_rows)
|
|
474
|
+
return added
|
|
475
|
+
#: The objectives the gate certified the installed surrogate for.
|
|
476
|
+
#: ``None`` when nothing is installed. The screen orders on exactly
|
|
477
|
+
#: these and treats the rest as unknown.
|
|
478
|
+
self._objectives: Optional[Tuple[str, ...]] = None
|
|
479
|
+
|
|
480
|
+
def refresh(
|
|
481
|
+
self,
|
|
482
|
+
evaluated: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
483
|
+
specs: Sequence[ObjectiveSpec],
|
|
484
|
+
*,
|
|
485
|
+
seed: int = 0,
|
|
486
|
+
) -> bool:
|
|
487
|
+
"""Re-arbitrate: today's data decides WHO may screen, if anyone.
|
|
488
|
+
|
|
489
|
+
Every builder is validated on ``validation_splits`` INDEPENDENT
|
|
490
|
+
re-partitions of today's data and must pass ``self.gate`` on EVERY
|
|
491
|
+
one; among the survivors, the lowest median mse/baseline ratio wins
|
|
492
|
+
the generation. The all-splits requirement is the variance guard the
|
|
493
|
+
ladder1 E2 row demanded: a high-variance authored artifact can pass
|
|
494
|
+
one partition by luck and then mis-screen mid-run -- measured at the
|
|
495
|
+
cheapest scale, where authored screening HURT the endpoint under the
|
|
496
|
+
single-split gate. Surviving every re-partition of today's data is
|
|
497
|
+
the in-loop generalization of "pass on both frozen datasets
|
|
498
|
+
independently"; under cross-validation each split already scores the
|
|
499
|
+
artifact on every row, so what the splits vary is which rows it was
|
|
500
|
+
FITTED on, which is the instability the guard exists to catch.
|
|
501
|
+
Best-passing across splits stays the arbitration: listing the
|
|
502
|
+
authored builder first would be trust, this is measurement. The
|
|
503
|
+
gate's rank-agreement term applies on every split too, but it only
|
|
504
|
+
GATES -- the ratio arbitrating among passers stays pure mse/baseline
|
|
505
|
+
(rank-unfaithful passers were the measured failure, not mis-ranking
|
|
506
|
+
among passers), and under the ordering purpose that ratio is the ONLY
|
|
507
|
+
thing the error term does.
|
|
508
|
+
|
|
509
|
+
Only the most recent ``max_training_rows`` measurements take part:
|
|
510
|
+
see that field for why a refresh must not grow with the run.
|
|
511
|
+
"""
|
|
512
|
+
|
|
513
|
+
self.telemetry.refreshes += 1
|
|
514
|
+
self._predict = None
|
|
515
|
+
self._objectives = None
|
|
516
|
+
data = list(evaluated)
|
|
517
|
+
# Cheap-fidelity evidence, where the campaign has none of its own.
|
|
518
|
+
# REAL SUPERSEDES CHEAP, always and by key; the cheap rows are
|
|
519
|
+
# appended after the real ones so the recent-window trim below drops
|
|
520
|
+
# them first when the run has measured more than the window holds.
|
|
521
|
+
if len(data) > self.max_training_rows:
|
|
522
|
+
data = data[-self.max_training_rows:]
|
|
523
|
+
proxy_used = 0
|
|
524
|
+
if self._proxy is not None and self._proxy_rows:
|
|
525
|
+
measured = {self._proxy.key(config) for config, _values in data}
|
|
526
|
+
extra = [row for token, row in self._proxy_rows.items()
|
|
527
|
+
if token not in measured]
|
|
528
|
+
room = self.max_training_rows - len(data)
|
|
529
|
+
extra = extra[:room] if room > 0 else []
|
|
530
|
+
proxy_used = len(extra)
|
|
531
|
+
data = data + extra
|
|
532
|
+
self.telemetry.proxy_rows_used = proxy_used
|
|
533
|
+
names = [spec.name for spec in specs]
|
|
534
|
+
required = self.gate.objectives_required(len(names))
|
|
535
|
+
best: Optional[Tuple[Tuple[int, float], str, str,
|
|
536
|
+
SurrogateBuilder, Tuple[str, ...]]] = None
|
|
537
|
+
for name, authored_by, builder in self.builders:
|
|
538
|
+
verdicts = []
|
|
539
|
+
failed = False
|
|
540
|
+
for split in range(self.validation_splits):
|
|
541
|
+
verdict = validate_surrogate(
|
|
542
|
+
builder, data, specs, policy=self.gate,
|
|
543
|
+
seed=seed + split * 7919)
|
|
544
|
+
self.telemetry.record(verdict)
|
|
545
|
+
if not verdict.passed:
|
|
546
|
+
failed = True
|
|
547
|
+
break
|
|
548
|
+
verdicts.append(verdict)
|
|
549
|
+
if failed:
|
|
550
|
+
self.telemetry.rejected_validation += 1
|
|
551
|
+
continue
|
|
552
|
+
# The variance guard applies PER OBJECTIVE, because that is the
|
|
553
|
+
# granularity the verdict now has. An artifact certified on
|
|
554
|
+
# {area, latency} by one re-partition and on {area, energy} by
|
|
555
|
+
# the next is stable on {area} alone -- it has not shown it can
|
|
556
|
+
# order latency or energy across fits -- so the certified set is
|
|
557
|
+
# the INTERSECTION and it must still meet the policy's
|
|
558
|
+
# requirement. Under the conjunction every passing split
|
|
559
|
+
# certifies every objective, so the intersection is the whole set
|
|
560
|
+
# and this is a no-op.
|
|
561
|
+
certified = set(names)
|
|
562
|
+
for verdict in verdicts:
|
|
563
|
+
certified &= set(verdict.passing_objectives)
|
|
564
|
+
scope = tuple(n for n in names if n in certified)
|
|
565
|
+
if len(scope) < required or not scope:
|
|
566
|
+
self.telemetry.rejected_unstable_subset += 1
|
|
567
|
+
continue
|
|
568
|
+
ratios = []
|
|
569
|
+
for verdict in verdicts:
|
|
570
|
+
per = [verdict.mse_ratio[n] for n in scope
|
|
571
|
+
if n in verdict.mse_ratio]
|
|
572
|
+
ratios.append(sum(per) / len(per) if per else 1.0)
|
|
573
|
+
ratio = statistics.median(ratios) if ratios else 1.0
|
|
574
|
+
# More certified objectives beats a better error ratio: an
|
|
575
|
+
# artifact that can order the whole problem is a different
|
|
576
|
+
# instrument from one that can order a third of it, and the
|
|
577
|
+
# ratio -- an average over whichever objectives each artifact
|
|
578
|
+
# got certified on -- is not comparable across different scopes.
|
|
579
|
+
# Under the conjunction every survivor has the same scope, so
|
|
580
|
+
# this reduces to the historical "lowest median ratio wins".
|
|
581
|
+
key = (-len(scope), ratio)
|
|
582
|
+
if best is None or key < best[0]:
|
|
583
|
+
best = (key, name, authored_by, builder, scope)
|
|
584
|
+
# Revise only when the rules measurably beat the artifact -- a refresh
|
|
585
|
+
# where nothing passes the gate carries no feedback a revision could
|
|
586
|
+
# use, and the revision budget is small.
|
|
587
|
+
llm_won = best is not None and best[2] == "llm"
|
|
588
|
+
if (best is not None and not llm_won and self.revise is not None
|
|
589
|
+
and self.telemetry.revisions < self.max_revisions
|
|
590
|
+
and any(authored_by == "llm"
|
|
591
|
+
for _n, authored_by, _b in self.builders)):
|
|
592
|
+
self.telemetry.revisions += 1
|
|
593
|
+
try:
|
|
594
|
+
replacement = self.revise(data, specs)
|
|
595
|
+
except Exception:
|
|
596
|
+
replacement = None
|
|
597
|
+
if replacement is not None:
|
|
598
|
+
self.telemetry.revisions_accepted += 1
|
|
599
|
+
rebuilt = []
|
|
600
|
+
swapped = False
|
|
601
|
+
for entry in self.builders:
|
|
602
|
+
if not swapped and entry[1] == "llm":
|
|
603
|
+
rebuilt.append(tuple(replacement))
|
|
604
|
+
swapped = True
|
|
605
|
+
else:
|
|
606
|
+
rebuilt.append(entry)
|
|
607
|
+
self.builders = tuple(rebuilt)
|
|
608
|
+
|
|
609
|
+
if best is None:
|
|
610
|
+
return False
|
|
611
|
+
_key, name, authored_by, builder, scope = best
|
|
612
|
+
try:
|
|
613
|
+
self._predict = builder(data, specs)
|
|
614
|
+
except Exception:
|
|
615
|
+
self.telemetry.screen_failures += 1
|
|
616
|
+
return False
|
|
617
|
+
self._objectives = scope
|
|
618
|
+
self._name = name
|
|
619
|
+
self.authored_by = authored_by
|
|
620
|
+
self.telemetry.validated += 1
|
|
621
|
+
if authored_by == "llm":
|
|
622
|
+
self.telemetry.chosen_llm += 1
|
|
623
|
+
elif authored_by == "proxy":
|
|
624
|
+
self.telemetry.chosen_proxy += 1
|
|
625
|
+
else:
|
|
626
|
+
self.telemetry.chosen_rule += 1
|
|
627
|
+
return True
|
|
628
|
+
|
|
629
|
+
def screen(
|
|
630
|
+
self,
|
|
631
|
+
pool: Sequence[Config],
|
|
632
|
+
population_objectives: Sequence[Mapping[str, float]],
|
|
633
|
+
specs: Sequence[ObjectiveSpec],
|
|
634
|
+
) -> Optional[ScreenReport]:
|
|
635
|
+
if self._predict is None:
|
|
636
|
+
return None
|
|
637
|
+
self.telemetry.screens += 1
|
|
638
|
+
try:
|
|
639
|
+
report = screen_offspring(
|
|
640
|
+
pool, population_objectives, specs, self._predict,
|
|
641
|
+
surrogate_name=self._name,
|
|
642
|
+
objectives=self._objectives,
|
|
643
|
+
)
|
|
644
|
+
except Exception:
|
|
645
|
+
self.telemetry.screen_failures += 1
|
|
646
|
+
return None
|
|
647
|
+
if report is None:
|
|
648
|
+
self.telemetry.screen_failures += 1
|
|
649
|
+
return None
|
|
650
|
+
self.telemetry.virtual_evaluations += report.virtual_evaluations
|
|
651
|
+
self.telemetry.record_screen(report)
|
|
652
|
+
return report
|
|
653
|
+
|
|
654
|
+
def exploration_floor_for(self, report: ScreenReport) -> float:
|
|
655
|
+
"""The share of the generation to keep away from THIS screen.
|
|
656
|
+
|
|
657
|
+
``exploration_floor`` when the screen ordered on the whole problem.
|
|
658
|
+
When it ordered on a subset, the floor rises with the share of
|
|
659
|
+
objectives it could not see, scaled by
|
|
660
|
+
``unscreened_objective_floor`` -- see that field for why a partial
|
|
661
|
+
screen needs more protection than a full one rather than the same.
|
|
662
|
+
The caller applies it; this class does not touch the budget.
|
|
663
|
+
"""
|
|
664
|
+
|
|
665
|
+
declared = len(report.declared_objectives)
|
|
666
|
+
screened = len(report.screened_objectives)
|
|
667
|
+
if declared <= 0 or screened >= declared:
|
|
668
|
+
return self.exploration_floor
|
|
669
|
+
unscreened_share = (declared - screened) / declared
|
|
670
|
+
return max(self.exploration_floor,
|
|
671
|
+
self.unscreened_objective_floor * unscreened_share)
|