agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,696 @@
|
|
|
1
|
+
"""Cheap predictive models of the evaluator, and the gate they must pass.
|
|
2
|
+
|
|
3
|
+
A surrogate exists to spend VIRTUAL evaluations so real ones go further: the
|
|
4
|
+
loop builds more offspring than it can afford to measure, asks the surrogate
|
|
5
|
+
to order them, and pays the evaluator only for the promising ones. That is
|
|
6
|
+
only honest if the surrogate is actually predictive, so every surrogate --
|
|
7
|
+
rule or model-authored alike -- passes :func:`validate_surrogate` before it
|
|
8
|
+
may order anything, and is re-validated as data accumulates.
|
|
9
|
+
|
|
10
|
+
**The gate proves what its consumer relies on, and nothing else.** That is a
|
|
11
|
+
:class:`GatePolicy`, declared by the caller rather than assumed here. A
|
|
12
|
+
consumer that only ORDERS candidates (:mod:`agent_evolve.session.screening`
|
|
13
|
+
never reads a predicted magnitude -- it ranks by predicted domination) is
|
|
14
|
+
gated on RANK FIDELITY; the error ratio against the train-mean predictor is
|
|
15
|
+
still computed and returned, because it ARBITRATES among gate-passers, but a
|
|
16
|
+
magnitude test cannot reject an artifact whose magnitudes nobody consumes. A
|
|
17
|
+
consumer that reads PREDICTED VALUES is gated on both. Gating ordering on
|
|
18
|
+
magnitude was measured to switch the mechanism off wholesale: on an expensive
|
|
19
|
+
venue at B <= 24, 27-28% of gate calls never reached scoring and the
|
|
20
|
+
MSE-vs-train-mean term rejected 59-63% of the rest, while the rank term --
|
|
21
|
+
the one the consumer actually depends on -- rejected 1.4-2.4%.
|
|
22
|
+
|
|
23
|
+
**The verdict is per objective, and the policy says how many must pass.**
|
|
24
|
+
An artifact that orders two of three objectives well is a usable ordering
|
|
25
|
+
instrument on those two; requiring all three lets the least predictable one
|
|
26
|
+
veto the others outright, which was measured happening on a live co-design
|
|
27
|
+
venue (area 0.855, latency 0.606, energy 0.329 -- energy alone closed the
|
|
28
|
+
gate). ``GatePolicy.min_passing_objectives`` declares the requirement, the
|
|
29
|
+
default remains the conjunction, and a partial pass certifies a SCOPE:
|
|
30
|
+
``SurrogateValidation.passing_objectives`` is what the consumer may order on
|
|
31
|
+
and the rest stay unknown rather than assumed.
|
|
32
|
+
|
|
33
|
+
**The evidence is cross-validated, not a single 30% holdout.** Every row is
|
|
34
|
+
held out exactly once and the statistics pool across the folds, so a run with
|
|
35
|
+
16 measured rows scores its surrogate on 16 held-out points instead of 4. At
|
|
36
|
+
the budgets an expensive venue admits, that is the difference between a rank
|
|
37
|
+
statistic and a coin flip -- and it is why the ordering policy can afford a
|
|
38
|
+
rank threshold meaningfully above chance plus a minimum effective holdout,
|
|
39
|
+
rather than the "no worse than chance" floor a 2-point holdout forced. The
|
|
40
|
+
single-holdout scheme remains available and agrees with cross-validation
|
|
41
|
+
where there is enough data for the question not to matter.
|
|
42
|
+
|
|
43
|
+
The two shipped surrogates are dependency-free rules. They are the
|
|
44
|
+
comparators any model-authored surrogate has to beat: an authored form that
|
|
45
|
+
cannot out-predict a shrunk per-locus mean has no business ordering
|
|
46
|
+
candidates.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
from __future__ import annotations
|
|
50
|
+
|
|
51
|
+
import math
|
|
52
|
+
import random
|
|
53
|
+
from dataclasses import dataclass, field, replace
|
|
54
|
+
from typing import Any, Callable, Dict, List, Mapping, Optional, Sequence, Tuple
|
|
55
|
+
|
|
56
|
+
from agent_evolve.core.problem import ObjectiveSpec
|
|
57
|
+
from agent_evolve.policies.genetic import loci_of, read_locus
|
|
58
|
+
|
|
59
|
+
__all__ = [
|
|
60
|
+
"GatePolicy",
|
|
61
|
+
"ORDERING_GATE",
|
|
62
|
+
"PARTIAL_ORDERING_GATE",
|
|
63
|
+
"PREDICTION_GATE",
|
|
64
|
+
"Predict",
|
|
65
|
+
"SurrogateBuilder",
|
|
66
|
+
"SurrogateValidation",
|
|
67
|
+
"knn_surrogate",
|
|
68
|
+
"additive_surrogate",
|
|
69
|
+
"validate_surrogate",
|
|
70
|
+
]
|
|
71
|
+
|
|
72
|
+
Config = Dict[str, Any]
|
|
73
|
+
|
|
74
|
+
#: Batch predictor: configurations in, one objective dict per configuration
|
|
75
|
+
#: out, or ``None`` when prediction is unavailable (the caller then screens
|
|
76
|
+
#: nothing rather than screening on garbage).
|
|
77
|
+
Predict = Callable[[Sequence[Config]], Optional[Sequence[Mapping[str, float]]]]
|
|
78
|
+
|
|
79
|
+
#: Fits a predictor to evaluated data. Rule builders run in-process; authored
|
|
80
|
+
#: builders wrap the out-of-process runtime behind the same signature.
|
|
81
|
+
SurrogateBuilder = Callable[
|
|
82
|
+
[Sequence[Tuple[Config, Mapping[str, float]]], Sequence[ObjectiveSpec]],
|
|
83
|
+
Predict,
|
|
84
|
+
]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def knn_surrogate(
|
|
88
|
+
evaluated: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
89
|
+
specs: Sequence[ObjectiveSpec],
|
|
90
|
+
*,
|
|
91
|
+
k: int = 3,
|
|
92
|
+
) -> Predict:
|
|
93
|
+
"""Nearest neighbours by Hamming distance over loci; mean of their outcomes."""
|
|
94
|
+
|
|
95
|
+
data = [(cfg, dict(obj)) for cfg, obj in evaluated]
|
|
96
|
+
names = [s.name for s in specs]
|
|
97
|
+
|
|
98
|
+
def predict(pool: Sequence[Config]) -> Optional[Sequence[Mapping[str, float]]]:
|
|
99
|
+
if not data:
|
|
100
|
+
return None
|
|
101
|
+
out = []
|
|
102
|
+
for candidate in pool:
|
|
103
|
+
loci = loci_of(candidate)
|
|
104
|
+
|
|
105
|
+
def distance(cfg: Config) -> int:
|
|
106
|
+
return sum(
|
|
107
|
+
1 for lc in loci
|
|
108
|
+
if read_locus(cfg, lc) != read_locus(candidate, lc)
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
nearest = sorted(data, key=lambda pair: distance(pair[0]))[:k]
|
|
112
|
+
out.append({
|
|
113
|
+
name: sum(obj[name] for _c, obj in nearest) / len(nearest)
|
|
114
|
+
for name in names
|
|
115
|
+
})
|
|
116
|
+
return out
|
|
117
|
+
|
|
118
|
+
return predict
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def additive_surrogate(
|
|
122
|
+
evaluated: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
123
|
+
specs: Sequence[ObjectiveSpec],
|
|
124
|
+
) -> Predict:
|
|
125
|
+
"""Global mean plus shrunk per-locus-value deviations.
|
|
126
|
+
|
|
127
|
+
The program's own landscape studies measured 88-99.99% additive structure
|
|
128
|
+
on every venue censused, which is why a first-order model is the honest
|
|
129
|
+
default rather than a strawman. Deviations shrink by ``n/(n+1)`` so a
|
|
130
|
+
value seen once moves a prediction half as far as its raw mean would.
|
|
131
|
+
"""
|
|
132
|
+
|
|
133
|
+
names = [s.name for s in specs]
|
|
134
|
+
if not evaluated:
|
|
135
|
+
return lambda pool: None
|
|
136
|
+
grand = {
|
|
137
|
+
name: sum(float(obj[name]) for _c, obj in evaluated) / len(evaluated)
|
|
138
|
+
for name in names
|
|
139
|
+
}
|
|
140
|
+
per_value: Dict[Tuple[str, Any], Dict[str, float]] = {}
|
|
141
|
+
counts: Dict[Tuple[str, Any], int] = {}
|
|
142
|
+
for cfg, obj in evaluated:
|
|
143
|
+
for lc in loci_of(cfg):
|
|
144
|
+
key = (str(lc), read_locus(cfg, lc))
|
|
145
|
+
counts[key] = counts.get(key, 0) + 1
|
|
146
|
+
bucket = per_value.setdefault(key, {name: 0.0 for name in names})
|
|
147
|
+
for name in names:
|
|
148
|
+
bucket[name] += float(obj[name])
|
|
149
|
+
|
|
150
|
+
def predict(pool: Sequence[Config]) -> Optional[Sequence[Mapping[str, float]]]:
|
|
151
|
+
out = []
|
|
152
|
+
for candidate in pool:
|
|
153
|
+
estimate = dict(grand)
|
|
154
|
+
for lc in loci_of(candidate):
|
|
155
|
+
key = (str(lc), read_locus(candidate, lc))
|
|
156
|
+
n = counts.get(key, 0)
|
|
157
|
+
if not n:
|
|
158
|
+
continue
|
|
159
|
+
shrink = n / (n + 1.0)
|
|
160
|
+
for name in names:
|
|
161
|
+
deviation = per_value[key][name] / n - grand[name]
|
|
162
|
+
estimate[name] += shrink * deviation
|
|
163
|
+
out.append(estimate)
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
return predict
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
#: Purposes a consumer may declare. ``"ordering"`` means "I rank candidates
|
|
170
|
+
#: with this and never read a predicted magnitude"; ``"prediction"`` means "I
|
|
171
|
+
#: read the values themselves".
|
|
172
|
+
PURPOSES: Tuple[str, ...] = ("ordering", "prediction")
|
|
173
|
+
|
|
174
|
+
#: Validation schemes. ``"cross_validated"`` holds every row out exactly once
|
|
175
|
+
#: and pools the statistics; ``"holdout"`` is the single random split.
|
|
176
|
+
SCHEMES: Tuple[str, ...] = ("cross_validated", "holdout")
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@dataclass(frozen=True)
|
|
180
|
+
class GatePolicy:
|
|
181
|
+
"""What the consumer relies on, and therefore what the gate must prove.
|
|
182
|
+
|
|
183
|
+
The gate has two terms and they answer different questions. RANK
|
|
184
|
+
fidelity asks "does this artifact order candidates the way the evaluator
|
|
185
|
+
would" -- the only question a screen's output can be wrong about, since
|
|
186
|
+
a screen consumes an ORDER. The ERROR ratio (holdout MSE against the
|
|
187
|
+
train-mean predictor's) asks "are the magnitudes usable" -- which
|
|
188
|
+
matters to a consumer that reads predicted values and not to one that
|
|
189
|
+
sorts by them.
|
|
190
|
+
|
|
191
|
+
A policy declares which terms REJECT. Both terms are always COMPUTED and
|
|
192
|
+
returned: the error ratio arbitrates among gate-passers even under the
|
|
193
|
+
ordering purpose (:mod:`agent_evolve.session.screening` picks the lowest
|
|
194
|
+
median ratio among the artifacts that passed), so "does not gate" is not
|
|
195
|
+
"is not measured".
|
|
196
|
+
|
|
197
|
+
Fields:
|
|
198
|
+
|
|
199
|
+
``purpose``
|
|
200
|
+
``"ordering"`` -- rank fidelity gates, the error ratio arbitrates.
|
|
201
|
+
``"prediction"`` -- both gate.
|
|
202
|
+
``min_rank_correlation``
|
|
203
|
+
The Spearman every objective must reach. The ordering purpose sets
|
|
204
|
+
this meaningfully above chance rather than at the 0.0 ("no worse
|
|
205
|
+
than a coin") floor a 2-point holdout used to force.
|
|
206
|
+
``min_effective_holdout``
|
|
207
|
+
How many held-out points the statistics must be pooled over before
|
|
208
|
+
the verdict is trusted at all. A Spearman on two points is +-1 by
|
|
209
|
+
construction and is not evidence of anything.
|
|
210
|
+
``scheme`` / ``folds``
|
|
211
|
+
``"cross_validated"`` (default) with ``folds`` (0 = chosen from n),
|
|
212
|
+
or the single ``"holdout"`` split of ``holdout_fraction``.
|
|
213
|
+
``min_rows``
|
|
214
|
+
The absolute floor on measured rows below which no split is honest.
|
|
215
|
+
``min_passing_objectives``
|
|
216
|
+
How many objectives must clear the terms above for the verdict to
|
|
217
|
+
pass. ``0`` means EVERY declared objective -- the conjunction, and
|
|
218
|
+
the historical behaviour. Any positive value admits a PARTIAL
|
|
219
|
+
verdict: the artifact is certified for the objectives it cleared and
|
|
220
|
+
for no others, and ``SurrogateValidation.passing_objectives`` names
|
|
221
|
+
them. A consumer that acts on a partial verdict must consume that
|
|
222
|
+
list; one that ignores it would be asserting predictions the gate
|
|
223
|
+
explicitly refused to certify. Values are clamped up to 1 and down to
|
|
224
|
+
the number of declared objectives, so the field is a floor on
|
|
225
|
+
evidence and never a way to pass with nothing.
|
|
226
|
+
"""
|
|
227
|
+
|
|
228
|
+
purpose: str = "prediction"
|
|
229
|
+
min_rank_correlation: float = 0.0
|
|
230
|
+
min_effective_holdout: int = 2
|
|
231
|
+
scheme: str = "cross_validated"
|
|
232
|
+
folds: int = 0
|
|
233
|
+
holdout_fraction: float = 0.3
|
|
234
|
+
min_rows: int = 8
|
|
235
|
+
min_passing_objectives: int = 0
|
|
236
|
+
|
|
237
|
+
def __post_init__(self) -> None:
|
|
238
|
+
if self.purpose not in PURPOSES:
|
|
239
|
+
raise ValueError(
|
|
240
|
+
f"purpose must be one of {PURPOSES}, got {self.purpose!r}")
|
|
241
|
+
if self.scheme not in SCHEMES:
|
|
242
|
+
raise ValueError(
|
|
243
|
+
f"scheme must be one of {SCHEMES}, got {self.scheme!r}")
|
|
244
|
+
if not -1.0 <= self.min_rank_correlation <= 1.0:
|
|
245
|
+
raise ValueError(
|
|
246
|
+
"min_rank_correlation must be in [-1, 1], got "
|
|
247
|
+
f"{self.min_rank_correlation}")
|
|
248
|
+
if self.min_effective_holdout < 2:
|
|
249
|
+
raise ValueError(
|
|
250
|
+
"min_effective_holdout must be at least 2, got "
|
|
251
|
+
f"{self.min_effective_holdout}")
|
|
252
|
+
if self.folds < 0 or self.folds == 1:
|
|
253
|
+
raise ValueError(
|
|
254
|
+
f"folds must be 0 (choose from n) or at least 2, got {self.folds}")
|
|
255
|
+
if not 0.0 < self.holdout_fraction < 1.0:
|
|
256
|
+
raise ValueError(
|
|
257
|
+
f"holdout_fraction must be in (0, 1), got {self.holdout_fraction}")
|
|
258
|
+
if self.min_rows < 2:
|
|
259
|
+
raise ValueError(f"min_rows must be at least 2, got {self.min_rows}")
|
|
260
|
+
if self.min_passing_objectives < 0:
|
|
261
|
+
raise ValueError(
|
|
262
|
+
"min_passing_objectives must be 0 (every objective) or "
|
|
263
|
+
f"positive, got {self.min_passing_objectives}")
|
|
264
|
+
|
|
265
|
+
@property
|
|
266
|
+
def error_rejects(self) -> bool:
|
|
267
|
+
"""Does a worse-than-train-mean MSE reject, or only arbitrate?"""
|
|
268
|
+
|
|
269
|
+
return self.purpose == "prediction"
|
|
270
|
+
|
|
271
|
+
@property
|
|
272
|
+
def admits_partial(self) -> bool:
|
|
273
|
+
"""May a verdict pass while some declared objective failed?"""
|
|
274
|
+
|
|
275
|
+
return self.min_passing_objectives > 0
|
|
276
|
+
|
|
277
|
+
def objectives_required(self, declared: int) -> int:
|
|
278
|
+
"""How many of *declared* objectives must clear the terms.
|
|
279
|
+
|
|
280
|
+
``0`` means all of them. A positive setting is clamped into
|
|
281
|
+
``[1, declared]``: never zero, because certifying an artifact for no
|
|
282
|
+
objective at all would let a screen order by nothing while reporting
|
|
283
|
+
that it screened; and never more than exist, so a policy written for
|
|
284
|
+
three objectives does not deadlock a two-objective problem.
|
|
285
|
+
"""
|
|
286
|
+
|
|
287
|
+
if declared <= 0:
|
|
288
|
+
return 0
|
|
289
|
+
if not self.min_passing_objectives:
|
|
290
|
+
return declared
|
|
291
|
+
return max(1, min(self.min_passing_objectives, declared))
|
|
292
|
+
|
|
293
|
+
def folds_for(self, n: int) -> int:
|
|
294
|
+
"""How many folds *n* rows get: the declared count, or one from n.
|
|
295
|
+
|
|
296
|
+
Bounded at 5 rather than leave-one-out because a fold costs one
|
|
297
|
+
builder fit, and an authored builder's fit is an out-of-process
|
|
298
|
+
call -- the statistic is pooled over every row either way.
|
|
299
|
+
"""
|
|
300
|
+
|
|
301
|
+
if self.scheme != "cross_validated":
|
|
302
|
+
return 1
|
|
303
|
+
if self.folds:
|
|
304
|
+
return max(2, min(self.folds, n))
|
|
305
|
+
return max(2, min(5, n // 2))
|
|
306
|
+
|
|
307
|
+
def replace(self, **overrides: Any) -> "GatePolicy":
|
|
308
|
+
"""This policy with fields overridden. Validated like any other."""
|
|
309
|
+
|
|
310
|
+
return replace(self, **overrides)
|
|
311
|
+
|
|
312
|
+
@classmethod
|
|
313
|
+
def for_purpose(cls, purpose: str, **overrides: Any) -> "GatePolicy":
|
|
314
|
+
"""The standing policy for *purpose*, with optional overrides."""
|
|
315
|
+
|
|
316
|
+
if purpose not in PURPOSES:
|
|
317
|
+
raise ValueError(
|
|
318
|
+
f"purpose must be one of {PURPOSES}, got {purpose!r}")
|
|
319
|
+
if purpose == "ordering":
|
|
320
|
+
base: Dict[str, Any] = {
|
|
321
|
+
"min_rank_correlation": 0.5, "min_effective_holdout": 8}
|
|
322
|
+
else:
|
|
323
|
+
base = {"min_rank_correlation": 0.0, "min_effective_holdout": 2}
|
|
324
|
+
base.update(overrides)
|
|
325
|
+
base.pop("purpose", None)
|
|
326
|
+
return cls(purpose=purpose, **base)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
#: The gate for a consumer that ORDERS candidates and never reads a predicted
|
|
330
|
+
#: magnitude -- the surrogate screen. Rank fidelity 0.5 on every objective,
|
|
331
|
+
#: pooled over at least 8 held-out points. Under cross-validation the second
|
|
332
|
+
#: condition is met by any run that clears ``min_rows``, which is the point:
|
|
333
|
+
#: the threshold is affordable because the evidence is no longer a 2-point
|
|
334
|
+
#: split. 0.5 is above chance by a stated margin -- against the exact
|
|
335
|
+
#: permutation null a predictor with no information at all reaches it on 10.8%
|
|
336
|
+
#: of draws at n = 8, 4.9% at n = 12 and 1.3% at n = 20 -- and the screen
|
|
337
|
+
#: additionally requires it on EVERY objective and on EVERY validation split
|
|
338
|
+
#: (session.screening's variance guard).
|
|
339
|
+
ORDERING_GATE = GatePolicy.for_purpose("ordering")
|
|
340
|
+
|
|
341
|
+
#: The gate for a consumer that reads PREDICTED VALUES. Both terms reject,
|
|
342
|
+
#: at the historical thresholds.
|
|
343
|
+
PREDICTION_GATE = GatePolicy.for_purpose("prediction")
|
|
344
|
+
|
|
345
|
+
#: The ordering gate, admitting a PARTIAL verdict: an artifact certified on
|
|
346
|
+
#: at least two objectives may order on exactly those, and the objectives it
|
|
347
|
+
#: failed stay unknown rather than being ordered on regardless. The
|
|
348
|
+
#: conjunction was measured turning a usable artifact away over one
|
|
349
|
+
#: objective: on an expensive 3-objective co-design venue an authored
|
|
350
|
+
#: surrogate read rank fidelity 0.855 (area) / 0.606 (latency) / 0.329
|
|
351
|
+
#: (energy), and the all-objectives requirement let the third veto the two.
|
|
352
|
+
#: Two, not one: ordering by domination over a single objective is a total
|
|
353
|
+
#: order on that objective and discards the trade-off the problem is about,
|
|
354
|
+
#: which is a different mechanism from the one this gate certifies.
|
|
355
|
+
PARTIAL_ORDERING_GATE = ORDERING_GATE.replace(min_passing_objectives=2)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
@dataclass(frozen=True)
|
|
359
|
+
class SurrogateValidation:
|
|
360
|
+
"""The gate's verdict, with the numbers that produced it.
|
|
361
|
+
|
|
362
|
+
``per_objective_spearman`` records the rank agreement between predicted
|
|
363
|
+
and actual held-out values (average-rank ties; 0.0 when either side is
|
|
364
|
+
entirely tied, i.e. no measurable ordering). It is empty on verdicts
|
|
365
|
+
that never reached scoring (too little data, builder failure, unusable
|
|
366
|
+
predictions).
|
|
367
|
+
|
|
368
|
+
``holdout`` is the EFFECTIVE holdout: how many held-out points the
|
|
369
|
+
statistics were pooled over. Under cross-validation that is every row.
|
|
370
|
+
|
|
371
|
+
``reason`` is the machine-readable term that rejected -- ``""`` when the
|
|
372
|
+
verdict passed -- so a caller can count WHY a gate closed without
|
|
373
|
+
parsing ``detail``.
|
|
374
|
+
|
|
375
|
+
``passing_objectives`` names the objectives that cleared the policy's
|
|
376
|
+
terms, in declaration order. It is the SCOPE of the verdict, not a
|
|
377
|
+
detail: under a policy that admits partial verdicts a pass certifies the
|
|
378
|
+
artifact for these objectives and for no others, and a consumer that
|
|
379
|
+
orders on anything outside this tuple is using a prediction the gate
|
|
380
|
+
refused. Under the conjunction (``min_passing_objectives=0``) a passing
|
|
381
|
+
verdict lists every declared objective, so the field reads the same way
|
|
382
|
+
under both policies and no consumer needs to branch on the policy.
|
|
383
|
+
"""
|
|
384
|
+
|
|
385
|
+
passed: bool
|
|
386
|
+
per_objective_mse: Mapping[str, float]
|
|
387
|
+
baseline_mse: Mapping[str, float]
|
|
388
|
+
holdout: int
|
|
389
|
+
detail: str = ""
|
|
390
|
+
per_objective_spearman: Mapping[str, float] = field(default_factory=dict)
|
|
391
|
+
reason: str = ""
|
|
392
|
+
purpose: str = "prediction"
|
|
393
|
+
scheme: str = "holdout"
|
|
394
|
+
folds: int = 1
|
|
395
|
+
passing_objectives: Tuple[str, ...] = ()
|
|
396
|
+
declared_objectives: Tuple[str, ...] = ()
|
|
397
|
+
|
|
398
|
+
@property
|
|
399
|
+
def partial(self) -> bool:
|
|
400
|
+
"""Did this verdict pass on a STRICT SUBSET of the objectives?
|
|
401
|
+
|
|
402
|
+
A run must never be able to report "the screen was active" without
|
|
403
|
+
being able to answer this, which is why it is derived from the two
|
|
404
|
+
tuples rather than from a flag a caller could forget to set.
|
|
405
|
+
"""
|
|
406
|
+
|
|
407
|
+
return bool(self.passed) and (
|
|
408
|
+
len(self.passing_objectives) < len(self.declared_objectives))
|
|
409
|
+
|
|
410
|
+
@property
|
|
411
|
+
def mse_ratio(self) -> Dict[str, float]:
|
|
412
|
+
"""Holdout MSE over the train-mean baseline's, where the baseline is
|
|
413
|
+
non-degenerate. Below 1.0 is predictive. This is the ARBITRATION
|
|
414
|
+
number: it separates gate-passers under every purpose, including the
|
|
415
|
+
one where it does not reject."""
|
|
416
|
+
|
|
417
|
+
return {
|
|
418
|
+
name: self.per_objective_mse[name] / self.baseline_mse[name]
|
|
419
|
+
for name in self.per_objective_mse
|
|
420
|
+
if self.baseline_mse.get(name, 0.0) > 0.0
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _average_ranks(values: Sequence[float]) -> List[float]:
|
|
425
|
+
"""Ranks 1..n with tied values sharing the average of their positions."""
|
|
426
|
+
|
|
427
|
+
order = sorted(range(len(values)), key=lambda i: values[i])
|
|
428
|
+
ranks = [0.0] * len(values)
|
|
429
|
+
i = 0
|
|
430
|
+
while i < len(order):
|
|
431
|
+
j = i
|
|
432
|
+
while j + 1 < len(order) and values[order[j + 1]] == values[order[i]]:
|
|
433
|
+
j += 1
|
|
434
|
+
average = (i + j) / 2.0 + 1.0
|
|
435
|
+
for k in range(i, j + 1):
|
|
436
|
+
ranks[order[k]] = average
|
|
437
|
+
i = j + 1
|
|
438
|
+
return ranks
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def _spearman(predicted: Sequence[float], actual: Sequence[float]) -> float:
|
|
442
|
+
"""Spearman rank correlation: Pearson correlation of average ranks.
|
|
443
|
+
|
|
444
|
+
Hand-rolled and dependency-free. When either side is entirely tied
|
|
445
|
+
there is no ordering to agree with (or none offered), so the
|
|
446
|
+
correlation is reported as 0.0 -- no measurable agreement.
|
|
447
|
+
"""
|
|
448
|
+
|
|
449
|
+
predicted_ranks = _average_ranks(predicted)
|
|
450
|
+
actual_ranks = _average_ranks(actual)
|
|
451
|
+
mean = (len(predicted_ranks) + 1) / 2.0 # both sides rank 1..n
|
|
452
|
+
dp = [rank - mean for rank in predicted_ranks]
|
|
453
|
+
da = [rank - mean for rank in actual_ranks]
|
|
454
|
+
vp = sum(d * d for d in dp)
|
|
455
|
+
va = sum(d * d for d in da)
|
|
456
|
+
if vp <= 0.0 or va <= 0.0:
|
|
457
|
+
return 0.0
|
|
458
|
+
return sum(p * a for p, a in zip(dp, da)) / math.sqrt(vp * va)
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _splits(
|
|
462
|
+
n: int, policy: GatePolicy, seed: int
|
|
463
|
+
) -> List[Tuple[List[int], List[int]]]:
|
|
464
|
+
"""``(train_idx, holdout_idx)`` pairs: k folds, or the one random split.
|
|
465
|
+
|
|
466
|
+
Cross-validation shuffles once and cuts contiguous blocks, so every row
|
|
467
|
+
is held out exactly once and the union of the holdouts is the data.
|
|
468
|
+
"""
|
|
469
|
+
|
|
470
|
+
indices = list(range(n))
|
|
471
|
+
random.Random(seed).shuffle(indices)
|
|
472
|
+
k = policy.folds_for(n)
|
|
473
|
+
if k <= 1:
|
|
474
|
+
cut = max(2, int(n * policy.holdout_fraction))
|
|
475
|
+
return [(indices[cut:], indices[:cut])]
|
|
476
|
+
sizes = [n // k + (1 if i < n % k else 0) for i in range(k)]
|
|
477
|
+
out: List[Tuple[List[int], List[int]]] = []
|
|
478
|
+
start = 0
|
|
479
|
+
for size in sizes:
|
|
480
|
+
held = indices[start:start + size]
|
|
481
|
+
held_set = set(held)
|
|
482
|
+
out.append(([i for i in indices if i not in held_set], held))
|
|
483
|
+
start += size
|
|
484
|
+
return out
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
class _Pooled:
|
|
488
|
+
"""Squared errors and (predicted, actual) pairs, accumulated over folds."""
|
|
489
|
+
|
|
490
|
+
__slots__ = ("names", "mse", "baseline", "predicted", "actual", "count")
|
|
491
|
+
|
|
492
|
+
def __init__(self, names: Sequence[str]) -> None:
|
|
493
|
+
self.names = list(names)
|
|
494
|
+
self.mse = {name: 0.0 for name in names}
|
|
495
|
+
self.baseline = {name: 0.0 for name in names}
|
|
496
|
+
self.predicted: Dict[str, List[float]] = {name: [] for name in names}
|
|
497
|
+
self.actual: Dict[str, List[float]] = {name: [] for name in names}
|
|
498
|
+
self.count = 0
|
|
499
|
+
|
|
500
|
+
def add(self, holdout, predictions, train_mean) -> Optional[str]:
|
|
501
|
+
"""Fold in one split's results; the objective that was unusable, if any."""
|
|
502
|
+
|
|
503
|
+
for (_cfg, actual), predicted in zip(holdout, predictions):
|
|
504
|
+
for name in self.names:
|
|
505
|
+
value = (predicted.get(name)
|
|
506
|
+
if isinstance(predicted, Mapping) else None)
|
|
507
|
+
if (value is None or isinstance(value, bool)
|
|
508
|
+
or not isinstance(value, (int, float))
|
|
509
|
+
or not math.isfinite(float(value))):
|
|
510
|
+
return name
|
|
511
|
+
#: A prediction can be finite and still un-squarable: at
|
|
512
|
+
#: |x| ~ 1e200 the residual overflows a float and Python
|
|
513
|
+
#: raises OverflowError rather than returning inf, which
|
|
514
|
+
#: kills the run instead of failing the objective. An
|
|
515
|
+
#: artifact that predicts 1e200 is unusable for the same
|
|
516
|
+
#: reason a non-finite one is, so it takes the same exit.
|
|
517
|
+
#: Cross-validation made this reachable in practice: it
|
|
518
|
+
#: squares every row, where a single 30% holdout squared
|
|
519
|
+
#: about a third of them.
|
|
520
|
+
try:
|
|
521
|
+
residual = (float(value) - float(actual[name])) ** 2
|
|
522
|
+
baseline = (train_mean[name] - float(actual[name])) ** 2
|
|
523
|
+
except OverflowError:
|
|
524
|
+
return name
|
|
525
|
+
if not (math.isfinite(residual) and math.isfinite(baseline)):
|
|
526
|
+
return name
|
|
527
|
+
self.mse[name] += residual
|
|
528
|
+
self.baseline[name] += baseline
|
|
529
|
+
self.predicted[name].append(float(value))
|
|
530
|
+
self.actual[name].append(float(actual[name]))
|
|
531
|
+
self.count += len(holdout)
|
|
532
|
+
return None
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def validate_surrogate(
|
|
536
|
+
builder: SurrogateBuilder,
|
|
537
|
+
evaluated: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
538
|
+
specs: Sequence[ObjectiveSpec],
|
|
539
|
+
*,
|
|
540
|
+
policy: Optional[GatePolicy] = None,
|
|
541
|
+
seed: int = 0,
|
|
542
|
+
holdout_fraction: Optional[float] = None,
|
|
543
|
+
min_rank_correlation: Optional[float] = None,
|
|
544
|
+
) -> SurrogateValidation:
|
|
545
|
+
"""May this surrogate be used for *policy*'s purpose, on today's evidence?
|
|
546
|
+
|
|
547
|
+
Two terms, computed on every declared objective over data the surrogate
|
|
548
|
+
never fitted, and *policy* decides which of them REJECT:
|
|
549
|
+
|
|
550
|
+
- **Rank agreement**: Spearman correlation between predicted and actual
|
|
551
|
+
held-out values must reach ``policy.min_rank_correlation``. MSE
|
|
552
|
+
fidelity is not rank fidelity -- the study-2 analog trace read
|
|
553
|
+
measured authored surrogates that passed the MSE split on half their
|
|
554
|
+
losing seeds while still hurting the endpoint, i.e. they misordered
|
|
555
|
+
the candidate pool, and ordering is the only thing a screen does with
|
|
556
|
+
a surrogate. This term rejects under EVERY purpose.
|
|
557
|
+
- **Error**: strictly beat the train-mean predictor's MSE, on every
|
|
558
|
+
objective -- a surrogate that predicts latency and guesses energy
|
|
559
|
+
would order the pool by half the problem while claiming to order it
|
|
560
|
+
by all of it. This term rejects only under ``purpose="prediction"``.
|
|
561
|
+
Under ``purpose="ordering"`` it is computed and returned
|
|
562
|
+
(``mse_ratio``) and ARBITRATES among passers, because a consumer that
|
|
563
|
+
sorts by predictions never reads their magnitudes -- gating ordering
|
|
564
|
+
on magnitude was measured switching the screen off wholesale on the
|
|
565
|
+
venues where evaluations are expensive enough for it to matter.
|
|
566
|
+
|
|
567
|
+
Evidence comes from ``policy.scheme``: cross-validation (every row held
|
|
568
|
+
out once, statistics pooled -- the default, and the only honest way to
|
|
569
|
+
ask this question of 16 measured rows) or one random holdout split. A
|
|
570
|
+
verdict is refused outright below ``policy.min_rows`` measured rows or
|
|
571
|
+
below ``policy.min_effective_holdout`` pooled held-out points: a
|
|
572
|
+
Spearman on two points is +-1 by construction and gates nothing.
|
|
573
|
+
|
|
574
|
+
``holdout_fraction`` and ``min_rank_correlation`` remain accepted as
|
|
575
|
+
direct overrides of the corresponding policy fields.
|
|
576
|
+
"""
|
|
577
|
+
|
|
578
|
+
policy = policy or PREDICTION_GATE
|
|
579
|
+
overrides: Dict[str, Any] = {}
|
|
580
|
+
if holdout_fraction is not None:
|
|
581
|
+
overrides["holdout_fraction"] = holdout_fraction
|
|
582
|
+
if min_rank_correlation is not None:
|
|
583
|
+
overrides["min_rank_correlation"] = min_rank_correlation
|
|
584
|
+
if overrides:
|
|
585
|
+
policy = policy.replace(**overrides)
|
|
586
|
+
|
|
587
|
+
names = [s.name for s in specs]
|
|
588
|
+
n = len(evaluated)
|
|
589
|
+
stamp: Dict[str, Any] = {
|
|
590
|
+
"purpose": policy.purpose,
|
|
591
|
+
"scheme": policy.scheme,
|
|
592
|
+
"folds": policy.folds_for(n) if n else 0,
|
|
593
|
+
"declared_objectives": tuple(names),
|
|
594
|
+
}
|
|
595
|
+
if n < policy.min_rows:
|
|
596
|
+
return SurrogateValidation(
|
|
597
|
+
False, {}, {}, 0,
|
|
598
|
+
detail=(f"needs at least {policy.min_rows} evaluated points, "
|
|
599
|
+
f"has {n}"),
|
|
600
|
+
reason="insufficient_rows", **stamp,
|
|
601
|
+
)
|
|
602
|
+
|
|
603
|
+
pooled = _Pooled(names)
|
|
604
|
+
for train_idx, holdout_idx in _splits(n, policy, seed):
|
|
605
|
+
train = [evaluated[i] for i in train_idx]
|
|
606
|
+
holdout = [evaluated[i] for i in holdout_idx]
|
|
607
|
+
if not train or not holdout:
|
|
608
|
+
continue
|
|
609
|
+
try:
|
|
610
|
+
predict = builder(train, specs)
|
|
611
|
+
predictions = predict([cfg for cfg, _obj in holdout])
|
|
612
|
+
except Exception as error:
|
|
613
|
+
return SurrogateValidation(
|
|
614
|
+
False, {}, {}, pooled.count,
|
|
615
|
+
detail=f"builder failed: {type(error).__name__}: {error}"[:200],
|
|
616
|
+
reason="builder_failed", **stamp,
|
|
617
|
+
)
|
|
618
|
+
if predictions is None or len(predictions) != len(holdout):
|
|
619
|
+
return SurrogateValidation(
|
|
620
|
+
False, {}, {}, pooled.count, detail="no usable predictions",
|
|
621
|
+
reason="no_predictions", **stamp,
|
|
622
|
+
)
|
|
623
|
+
train_mean = {
|
|
624
|
+
name: sum(float(obj[name]) for _c, obj in train) / len(train)
|
|
625
|
+
for name in names
|
|
626
|
+
}
|
|
627
|
+
unusable = pooled.add(holdout, predictions, train_mean)
|
|
628
|
+
if unusable is not None:
|
|
629
|
+
return SurrogateValidation(
|
|
630
|
+
False, {}, {}, pooled.count,
|
|
631
|
+
detail=f"non-finite or missing prediction for {unusable!r}",
|
|
632
|
+
reason="bad_prediction", **stamp,
|
|
633
|
+
)
|
|
634
|
+
return _score(pooled, names, policy, stamp)
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
def _score(pooled: "_Pooled", names, policy: GatePolicy,
|
|
638
|
+
stamp: Mapping[str, Any]) -> SurrogateValidation:
|
|
639
|
+
count = pooled.count
|
|
640
|
+
if count < policy.min_effective_holdout:
|
|
641
|
+
return SurrogateValidation(
|
|
642
|
+
False, {}, {}, count,
|
|
643
|
+
detail=(f"effective holdout {count} is below "
|
|
644
|
+
f"{policy.min_effective_holdout}: a rank statistic on "
|
|
645
|
+
"that many points is not evidence"),
|
|
646
|
+
reason="insufficient_holdout", **stamp,
|
|
647
|
+
)
|
|
648
|
+
mse = {k: v / count for k, v in pooled.mse.items()}
|
|
649
|
+
baseline = {k: v / count for k, v in pooled.baseline.items()}
|
|
650
|
+
spearman = {
|
|
651
|
+
name: _spearman(pooled.predicted[name], pooled.actual[name])
|
|
652
|
+
for name in names
|
|
653
|
+
}
|
|
654
|
+
# Both terms are decided PER OBJECTIVE, and the policy then says how many
|
|
655
|
+
# objectives have to clear them. Under the conjunction
|
|
656
|
+
# (min_passing_objectives = 0, `required` = len(names)) that is exactly
|
|
657
|
+
# the historical `all(...) and all(...)`; under a partial policy the same
|
|
658
|
+
# per-objective decisions are what `passing_objectives` reports, so the
|
|
659
|
+
# verdict's scope and its pass/fail come from one computation and cannot
|
|
660
|
+
# disagree.
|
|
661
|
+
rank_pass = [name for name in names
|
|
662
|
+
if spearman[name] >= policy.min_rank_correlation]
|
|
663
|
+
if policy.error_rejects:
|
|
664
|
+
passing = [name for name in rank_pass if mse[name] < baseline[name]]
|
|
665
|
+
else:
|
|
666
|
+
passing = list(rank_pass)
|
|
667
|
+
required = policy.objectives_required(len(names))
|
|
668
|
+
detail = ""
|
|
669
|
+
reason = ""
|
|
670
|
+
if len(rank_pass) < required:
|
|
671
|
+
worst = min(names, key=lambda name: spearman[name])
|
|
672
|
+
reason = "rank"
|
|
673
|
+
shortfall = (f"{len(rank_pass)} of {len(names)} objectives reach it, "
|
|
674
|
+
f"{required} needed; " if policy.admits_partial else "")
|
|
675
|
+
detail = (
|
|
676
|
+
f"{shortfall}rank agreement {spearman[worst]:.3f} on {worst!r} is "
|
|
677
|
+
f"below {policy.min_rank_correlation:.3f}: MSE fidelity is not "
|
|
678
|
+
"rank fidelity"
|
|
679
|
+
)
|
|
680
|
+
elif len(passing) < required:
|
|
681
|
+
worst = max(names,
|
|
682
|
+
key=lambda name: (mse[name] / baseline[name]
|
|
683
|
+
if baseline[name] > 0.0 else math.inf))
|
|
684
|
+
reason = "error"
|
|
685
|
+
ratio = (f"{mse[worst] / baseline[worst]:.2f}x"
|
|
686
|
+
if baseline[worst] > 0.0 else "at a degenerate")
|
|
687
|
+
detail = (f"holdout error is {ratio} the train-mean baseline on "
|
|
688
|
+
f"{worst!r}: the magnitudes this purpose consumes are not "
|
|
689
|
+
"predictive")
|
|
690
|
+
passed = len(passing) >= required and required > 0
|
|
691
|
+
return SurrogateValidation(
|
|
692
|
+
passed, mse, baseline, count,
|
|
693
|
+
detail=detail, per_objective_spearman=spearman, reason=reason,
|
|
694
|
+
passing_objectives=tuple(passing) if passed else (),
|
|
695
|
+
**stamp,
|
|
696
|
+
)
|