agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1083 @@
|
|
|
1
|
+
"""V8-lite composition: repaired pilot and challenger, preserved terminal.
|
|
2
|
+
|
|
3
|
+
One sealed materialized-action market is allocated in three phases:
|
|
4
|
+
|
|
5
|
+
* PILOT — the rank-balanced causal pilot replaces the deterministic
|
|
6
|
+
one-lane-head-per-engine design (V70 failure a): engine coverage floor,
|
|
7
|
+
configurable top-heavy band mass, blocked randomization between the
|
|
8
|
+
native-rank and frozen-score orders, and exact seat propensities.
|
|
9
|
+
* ADAPTIVE / CONTINUATION — the calibrated positive-gain challenger ranks
|
|
10
|
+
every remaining candidate with no hard abstention (V70 failure b) and
|
|
11
|
+
values only archive-conditioned gain (V70 failure c). The existing
|
|
12
|
+
outcome-adaptive controller is retained as the protected fallback: when
|
|
13
|
+
the challenger's best score is non-positive, the incumbent decision is
|
|
14
|
+
used unchanged.
|
|
15
|
+
* TERMINAL — EXACTLY the V7 terminal hierarchical-exploitation rule, by
|
|
16
|
+
delegation to the frozen ``OutcomeAdaptiveActionRacingPolicy`` version 7
|
|
17
|
+
instance. The V7 rule worked prospectively and is not altered.
|
|
18
|
+
|
|
19
|
+
The policy knows no workload, objective, model, provider, or prompt.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import hashlib
|
|
25
|
+
import json
|
|
26
|
+
import math
|
|
27
|
+
import re
|
|
28
|
+
from dataclasses import dataclass, field
|
|
29
|
+
|
|
30
|
+
from agent_evolve.application.calibrated_positive_gain_opportunity import (
|
|
31
|
+
ArchiveConditionedGainPort,
|
|
32
|
+
CalibratedPositiveGainOpportunityPolicy,
|
|
33
|
+
CalibratedPositiveGainRanking,
|
|
34
|
+
ObjectivePoint,
|
|
35
|
+
ObservedConversionOutcome,
|
|
36
|
+
PositiveGainCandidate,
|
|
37
|
+
PositiveGainForecast,
|
|
38
|
+
)
|
|
39
|
+
from agent_evolve.application.outcome_adaptive_action_racing import (
|
|
40
|
+
OUTCOME_ADAPTIVE_ACTION_RACING_HORIZON_HIERARCHICAL_POLICY_VERSION,
|
|
41
|
+
AdaptiveActionDescriptor,
|
|
42
|
+
AdaptiveActionOutcome,
|
|
43
|
+
AdaptiveActionRacingDecision,
|
|
44
|
+
AdaptiveActionSetOutcome,
|
|
45
|
+
OutcomeAdaptiveActionRacingPolicy,
|
|
46
|
+
)
|
|
47
|
+
from agent_evolve.application.rank_balanced_causal_pilot import (
|
|
48
|
+
DEFAULT_PILOT_BAND_WEIGHTS,
|
|
49
|
+
SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_ID,
|
|
50
|
+
PilotSeatObservation,
|
|
51
|
+
RankBalancedCausalPilotPolicy,
|
|
52
|
+
RankBalancedPilotCandidate,
|
|
53
|
+
RankBalancedPilotDesign,
|
|
54
|
+
SequentialAdaptiveBandPilotPolicy,
|
|
55
|
+
)
|
|
56
|
+
from agent_evolve.domain.patch import require_sha256
|
|
57
|
+
from agent_evolve.domain.typed_json import FrozenJsonObject, freeze_json
|
|
58
|
+
|
|
59
|
+
V8LITE_ALLOCATION_POLICY_ID = "v8lite_allocation"
|
|
60
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID = "v8lite_r1"
|
|
61
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R2 = "v8lite_r2"
|
|
62
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R3 = "v8lite_r3"
|
|
63
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R4 = "v8lite_r4"
|
|
64
|
+
V8LITE_ALLOCATION_POLICY_VERSION = 1
|
|
65
|
+
_V8LITE_VERSION_IDS = frozenset(
|
|
66
|
+
{
|
|
67
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID,
|
|
68
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R2,
|
|
69
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R3,
|
|
70
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R4,
|
|
71
|
+
}
|
|
72
|
+
)
|
|
73
|
+
_TOKEN = re.compile(r"^[a-z][a-z0-9_.:-]{0,255}$")
|
|
74
|
+
_DEFINITION_DOMAIN = b"agent-evolve:v8lite-allocation-definition:v1\x00"
|
|
75
|
+
_DECISION_DOMAIN = b"agent-evolve:v8lite-allocation-decision:v1\x00"
|
|
76
|
+
|
|
77
|
+
V8LITE_PHASE_PILOT = "pilot"
|
|
78
|
+
V8LITE_PHASE_ADAPTIVE = "adaptive"
|
|
79
|
+
V8LITE_PHASE_PROTECTED_FALLBACK = "protected_fallback"
|
|
80
|
+
V8LITE_PHASE_TERMINAL = "terminal"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _canonical_json(value: object) -> bytes:
|
|
84
|
+
return json.dumps(
|
|
85
|
+
value,
|
|
86
|
+
allow_nan=False,
|
|
87
|
+
ensure_ascii=True,
|
|
88
|
+
separators=(",", ":"),
|
|
89
|
+
sort_keys=True,
|
|
90
|
+
).encode("ascii", errors="strict")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _hash(domain: bytes, value: object) -> str:
|
|
94
|
+
return hashlib.sha256(domain + _canonical_json(value)).hexdigest()
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _require_token(value: str, *, name: str) -> None:
|
|
98
|
+
if type(value) is not str or _TOKEN.fullmatch(value) is None:
|
|
99
|
+
raise ValueError(f"{name} must use the closed token grammar")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
@dataclass(frozen=True, slots=True)
|
|
103
|
+
class V8LiteAllocationConfig:
|
|
104
|
+
"""All V8-lite constants; no magic numbers live in policy code."""
|
|
105
|
+
|
|
106
|
+
# Pilot phase.
|
|
107
|
+
diagnostic_slots: int = 4
|
|
108
|
+
band_count: int = 3
|
|
109
|
+
band_weights: tuple[float, ...] = DEFAULT_PILOT_BAND_WEIGHTS
|
|
110
|
+
exploration_epsilon: float = 0.125
|
|
111
|
+
# Calibrated challenger.
|
|
112
|
+
lambda_: float = 1.0
|
|
113
|
+
beta: float = 1.0
|
|
114
|
+
mixture_weight: float = 0.5
|
|
115
|
+
prior_strength: float = 2.0
|
|
116
|
+
probability_floor: float = 1.0e-6
|
|
117
|
+
scenario_weights: tuple[float, float, float] = (0.25, 0.5, 0.25)
|
|
118
|
+
frozen_score_weight: float = 0.125
|
|
119
|
+
frozen_score_minimum_training_runs: int = 10
|
|
120
|
+
# Shared scale and preserved V7 terminal/fallback controller.
|
|
121
|
+
reference_gain_scale: float = 1.0e-4
|
|
122
|
+
reference_gain_evidence_sha256: str = "0" * 64
|
|
123
|
+
ucb_strength: float = 1.0
|
|
124
|
+
counterfactual_strength: float = 0.5
|
|
125
|
+
diversity_strength: float = 0.25
|
|
126
|
+
randomized_audit_slots: int = 1
|
|
127
|
+
randomized_audit_after_directed_steps: int = 1
|
|
128
|
+
minimum_post_audit_optimization_slots: int = 1
|
|
129
|
+
terminal_hierarchical_slots: int = 1
|
|
130
|
+
native_rank_strength: float = 0.25
|
|
131
|
+
exploration_pool_size: int = 4
|
|
132
|
+
trace_alternative_count: int = 9
|
|
133
|
+
random_seed: int = 0
|
|
134
|
+
# r2-only constants; excluded from to_record so the r1 identity
|
|
135
|
+
# bytes are preserved (the r2 identity adds them explicitly).
|
|
136
|
+
adaptation_temperature: float = 1.0
|
|
137
|
+
# r3 carries no new constant: the archive-geometry tie-break is a
|
|
138
|
+
# strict ordering, not a weighted term.
|
|
139
|
+
|
|
140
|
+
def __post_init__(self) -> None:
|
|
141
|
+
# Structural bounds are owned by the composed policies; this
|
|
142
|
+
# config only guards fields it consumes directly.
|
|
143
|
+
if type(self.diagnostic_slots) is not int or self.diagnostic_slots <= 0:
|
|
144
|
+
raise ValueError("diagnostic_slots must be positive")
|
|
145
|
+
if (
|
|
146
|
+
type(self.randomized_audit_slots) is not int
|
|
147
|
+
or self.randomized_audit_slots <= 0
|
|
148
|
+
):
|
|
149
|
+
raise ValueError("randomized_audit_slots must be positive")
|
|
150
|
+
|
|
151
|
+
def to_record(self) -> dict[str, object]:
|
|
152
|
+
self.__post_init__()
|
|
153
|
+
return {
|
|
154
|
+
"diagnostic_slots": self.diagnostic_slots,
|
|
155
|
+
"band_count": self.band_count,
|
|
156
|
+
"band_weights_hex": [
|
|
157
|
+
value.hex() for value in self.band_weights
|
|
158
|
+
],
|
|
159
|
+
"exploration_epsilon_hex": self.exploration_epsilon.hex(),
|
|
160
|
+
"lambda_hex": self.lambda_.hex(),
|
|
161
|
+
"beta_hex": self.beta.hex(),
|
|
162
|
+
"mixture_weight_hex": self.mixture_weight.hex(),
|
|
163
|
+
"prior_strength_hex": self.prior_strength.hex(),
|
|
164
|
+
"probability_floor_hex": self.probability_floor.hex(),
|
|
165
|
+
"scenario_weights_hex": [
|
|
166
|
+
value.hex() for value in self.scenario_weights
|
|
167
|
+
],
|
|
168
|
+
"frozen_score_weight_hex": self.frozen_score_weight.hex(),
|
|
169
|
+
"frozen_score_minimum_training_runs": (
|
|
170
|
+
self.frozen_score_minimum_training_runs
|
|
171
|
+
),
|
|
172
|
+
"reference_gain_scale_hex": (
|
|
173
|
+
self.reference_gain_scale.hex()
|
|
174
|
+
),
|
|
175
|
+
"reference_gain_evidence_sha256": (
|
|
176
|
+
self.reference_gain_evidence_sha256
|
|
177
|
+
),
|
|
178
|
+
"ucb_strength_hex": self.ucb_strength.hex(),
|
|
179
|
+
"counterfactual_strength_hex": (
|
|
180
|
+
self.counterfactual_strength.hex()
|
|
181
|
+
),
|
|
182
|
+
"diversity_strength_hex": self.diversity_strength.hex(),
|
|
183
|
+
"randomized_audit_slots": self.randomized_audit_slots,
|
|
184
|
+
"randomized_audit_after_directed_steps": (
|
|
185
|
+
self.randomized_audit_after_directed_steps
|
|
186
|
+
),
|
|
187
|
+
"minimum_post_audit_optimization_slots": (
|
|
188
|
+
self.minimum_post_audit_optimization_slots
|
|
189
|
+
),
|
|
190
|
+
"terminal_hierarchical_slots": (
|
|
191
|
+
self.terminal_hierarchical_slots
|
|
192
|
+
),
|
|
193
|
+
"native_rank_strength_hex": (
|
|
194
|
+
self.native_rank_strength.hex()
|
|
195
|
+
),
|
|
196
|
+
"exploration_pool_size": self.exploration_pool_size,
|
|
197
|
+
"trace_alternative_count": self.trace_alternative_count,
|
|
198
|
+
"random_seed": self.random_seed,
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
@dataclass(frozen=True, slots=True)
|
|
203
|
+
class V8LiteDecision:
|
|
204
|
+
"""One phase-attributed V8-lite allocation step."""
|
|
205
|
+
|
|
206
|
+
policy_id: str
|
|
207
|
+
policy_version_id: str
|
|
208
|
+
policy_definition_sha256: str
|
|
209
|
+
residual_request_sha256: str
|
|
210
|
+
phase: str
|
|
211
|
+
authority_policy_id: str
|
|
212
|
+
selected_action_sha256s: tuple[str, ...]
|
|
213
|
+
selection_propensity: float
|
|
214
|
+
evidence: FrozenJsonObject
|
|
215
|
+
delegated_decision: AdaptiveActionRacingDecision | None = field(
|
|
216
|
+
default=None,
|
|
217
|
+
repr=False,
|
|
218
|
+
compare=False,
|
|
219
|
+
)
|
|
220
|
+
pilot_design: RankBalancedPilotDesign | None = field(
|
|
221
|
+
default=None,
|
|
222
|
+
repr=False,
|
|
223
|
+
compare=False,
|
|
224
|
+
)
|
|
225
|
+
challenger_ranking: CalibratedPositiveGainRanking | None = field(
|
|
226
|
+
default=None,
|
|
227
|
+
repr=False,
|
|
228
|
+
compare=False,
|
|
229
|
+
)
|
|
230
|
+
decision_sha256: str = field(init=False)
|
|
231
|
+
|
|
232
|
+
def __post_init__(self) -> None:
|
|
233
|
+
_require_token(self.policy_id, name="policy_id")
|
|
234
|
+
_require_token(
|
|
235
|
+
self.policy_version_id,
|
|
236
|
+
name="policy_version_id",
|
|
237
|
+
)
|
|
238
|
+
require_sha256(
|
|
239
|
+
self.policy_definition_sha256,
|
|
240
|
+
"policy_definition_sha256",
|
|
241
|
+
)
|
|
242
|
+
require_sha256(
|
|
243
|
+
self.residual_request_sha256,
|
|
244
|
+
"residual_request_sha256",
|
|
245
|
+
)
|
|
246
|
+
if self.phase not in {
|
|
247
|
+
V8LITE_PHASE_PILOT,
|
|
248
|
+
V8LITE_PHASE_ADAPTIVE,
|
|
249
|
+
V8LITE_PHASE_PROTECTED_FALLBACK,
|
|
250
|
+
V8LITE_PHASE_TERMINAL,
|
|
251
|
+
}:
|
|
252
|
+
raise ValueError("phase is not a V8-lite phase")
|
|
253
|
+
_require_token(
|
|
254
|
+
self.authority_policy_id,
|
|
255
|
+
name="authority_policy_id",
|
|
256
|
+
)
|
|
257
|
+
if (
|
|
258
|
+
type(self.selected_action_sha256s) is not tuple
|
|
259
|
+
or not self.selected_action_sha256s
|
|
260
|
+
or self.selected_action_sha256s
|
|
261
|
+
!= tuple(sorted(set(self.selected_action_sha256s)))
|
|
262
|
+
):
|
|
263
|
+
raise ValueError(
|
|
264
|
+
"selected_action_sha256s must be a canonical exact tuple"
|
|
265
|
+
)
|
|
266
|
+
for value in self.selected_action_sha256s:
|
|
267
|
+
require_sha256(value, "selected_action_sha256")
|
|
268
|
+
if (
|
|
269
|
+
type(self.selection_propensity) is not float
|
|
270
|
+
or not math.isfinite(self.selection_propensity)
|
|
271
|
+
or not 0.0 < self.selection_propensity <= 1.0
|
|
272
|
+
):
|
|
273
|
+
raise ValueError(
|
|
274
|
+
"selection_propensity must be a positive probability"
|
|
275
|
+
)
|
|
276
|
+
if (
|
|
277
|
+
type(self.evidence) is not FrozenJsonObject
|
|
278
|
+
or freeze_json(self.evidence) is not self.evidence
|
|
279
|
+
):
|
|
280
|
+
raise TypeError("evidence must be an exact frozen object")
|
|
281
|
+
object.__setattr__(
|
|
282
|
+
self,
|
|
283
|
+
"decision_sha256",
|
|
284
|
+
_hash(_DECISION_DOMAIN, self._unsigned_record()),
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
def _unsigned_record(self) -> dict[str, object]:
|
|
288
|
+
return {
|
|
289
|
+
"schema_version": 1,
|
|
290
|
+
"policy": {
|
|
291
|
+
"policy_id": self.policy_id,
|
|
292
|
+
"policy_version_id": self.policy_version_id,
|
|
293
|
+
"definition_sha256": self.policy_definition_sha256,
|
|
294
|
+
},
|
|
295
|
+
"residual_request_sha256": self.residual_request_sha256,
|
|
296
|
+
"phase": self.phase,
|
|
297
|
+
"authority_policy_id": self.authority_policy_id,
|
|
298
|
+
"selected_action_sha256s": list(
|
|
299
|
+
self.selected_action_sha256s
|
|
300
|
+
),
|
|
301
|
+
"selection_propensity_hex": (
|
|
302
|
+
self.selection_propensity.hex()
|
|
303
|
+
),
|
|
304
|
+
"delegated_decision_sha256": (
|
|
305
|
+
None
|
|
306
|
+
if self.delegated_decision is None
|
|
307
|
+
else self.delegated_decision.decision_sha256
|
|
308
|
+
),
|
|
309
|
+
"pilot_design_sha256": (
|
|
310
|
+
None
|
|
311
|
+
if self.pilot_design is None
|
|
312
|
+
else self.pilot_design.design_sha256
|
|
313
|
+
),
|
|
314
|
+
"challenger_ranking_sha256": (
|
|
315
|
+
None
|
|
316
|
+
if self.challenger_ranking is None
|
|
317
|
+
else self.challenger_ranking.ranking_sha256
|
|
318
|
+
),
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
def to_record(self) -> dict[str, object]:
|
|
322
|
+
self.__post_init__()
|
|
323
|
+
return {
|
|
324
|
+
**self._unsigned_record(),
|
|
325
|
+
"decision_sha256": self.decision_sha256,
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
@dataclass(frozen=True, slots=True)
|
|
330
|
+
class V8LiteAllocationPolicy:
|
|
331
|
+
"""Compose the repaired pilot and challenger over the preserved V7."""
|
|
332
|
+
|
|
333
|
+
archive_gain_utility: ArchiveConditionedGainPort = field(
|
|
334
|
+
repr=False,
|
|
335
|
+
compare=False,
|
|
336
|
+
)
|
|
337
|
+
config: V8LiteAllocationConfig = V8LiteAllocationConfig()
|
|
338
|
+
policy_id: str = V8LITE_ALLOCATION_POLICY_ID
|
|
339
|
+
policy_version_id: str = V8LITE_ALLOCATION_POLICY_VERSION_ID
|
|
340
|
+
definition_sha256: str = field(init=False)
|
|
341
|
+
|
|
342
|
+
def __post_init__(self) -> None:
|
|
343
|
+
if type(self.config) is not V8LiteAllocationConfig:
|
|
344
|
+
raise TypeError("config must be an exact V8-lite config")
|
|
345
|
+
self.config.__post_init__()
|
|
346
|
+
_require_token(self.policy_id, name="policy_id")
|
|
347
|
+
if (
|
|
348
|
+
self.policy_id != V8LITE_ALLOCATION_POLICY_ID
|
|
349
|
+
or self.policy_version_id not in _V8LITE_VERSION_IDS
|
|
350
|
+
):
|
|
351
|
+
raise ValueError("policy identity is immutable")
|
|
352
|
+
# Constructing every component validates the whole configuration
|
|
353
|
+
# and binds component identities into this policy's definition.
|
|
354
|
+
pilot = self.pilot_policy()
|
|
355
|
+
challenger = self.challenger_policy()
|
|
356
|
+
terminal = self.terminal_policy()
|
|
357
|
+
object.__setattr__(
|
|
358
|
+
self,
|
|
359
|
+
"definition_sha256",
|
|
360
|
+
_hash(_DEFINITION_DOMAIN, self._identity(
|
|
361
|
+
pilot=pilot,
|
|
362
|
+
challenger=challenger,
|
|
363
|
+
terminal=terminal,
|
|
364
|
+
)),
|
|
365
|
+
)
|
|
366
|
+
|
|
367
|
+
@property
|
|
368
|
+
def revision_2(self) -> bool:
|
|
369
|
+
"""r3 is r2 plus the challenger's archive-geometry channel."""
|
|
370
|
+
|
|
371
|
+
return self.policy_version_id in {
|
|
372
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R2,
|
|
373
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R3,
|
|
374
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R4,
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
@property
|
|
378
|
+
def revision_3(self) -> bool:
|
|
379
|
+
return self.policy_version_id in {
|
|
380
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R3,
|
|
381
|
+
V8LITE_ALLOCATION_POLICY_VERSION_ID_R4,
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
@property
|
|
385
|
+
def revision_4(self) -> bool:
|
|
386
|
+
"""r4 is r3 plus the challenger's repaired tail-risk algebra."""
|
|
387
|
+
|
|
388
|
+
return (
|
|
389
|
+
self.policy_version_id
|
|
390
|
+
== V8LITE_ALLOCATION_POLICY_VERSION_ID_R4
|
|
391
|
+
)
|
|
392
|
+
|
|
393
|
+
def _identity(
|
|
394
|
+
self,
|
|
395
|
+
*,
|
|
396
|
+
pilot: object,
|
|
397
|
+
challenger: CalibratedPositiveGainOpportunityPolicy,
|
|
398
|
+
terminal: OutcomeAdaptiveActionRacingPolicy,
|
|
399
|
+
) -> dict[str, object]:
|
|
400
|
+
identity = {
|
|
401
|
+
"schema_version": 1,
|
|
402
|
+
"policy_id": self.policy_id,
|
|
403
|
+
"policy_version_id": self.policy_version_id,
|
|
404
|
+
"policy_version": V8LITE_ALLOCATION_POLICY_VERSION,
|
|
405
|
+
"config": self.config.to_record(),
|
|
406
|
+
"components": {
|
|
407
|
+
"pilot": {
|
|
408
|
+
"policy_id": pilot.policy_id,
|
|
409
|
+
"policy_version": pilot.policy_version,
|
|
410
|
+
"definition_sha256": pilot.definition_sha256,
|
|
411
|
+
},
|
|
412
|
+
"challenger": {
|
|
413
|
+
"policy_id": challenger.policy_id,
|
|
414
|
+
"policy_version": challenger.policy_version,
|
|
415
|
+
"definition_sha256": challenger.definition_sha256,
|
|
416
|
+
},
|
|
417
|
+
"terminal_and_fallback": {
|
|
418
|
+
"policy_id": terminal.policy_id,
|
|
419
|
+
"policy_version": terminal.policy_version,
|
|
420
|
+
"definition_sha256": terminal.definition_sha256,
|
|
421
|
+
},
|
|
422
|
+
},
|
|
423
|
+
"phases": {
|
|
424
|
+
"pilot": "rank_balanced_causal_pilot",
|
|
425
|
+
"continuation": (
|
|
426
|
+
"calibrated_positive_gain_challenger_with_"
|
|
427
|
+
"protected_v7_fallback"
|
|
428
|
+
),
|
|
429
|
+
"terminal": (
|
|
430
|
+
"exact_v7_terminal_hierarchical_exploitation_"
|
|
431
|
+
"by_delegation"
|
|
432
|
+
),
|
|
433
|
+
},
|
|
434
|
+
"challenger_fallback_condition": (
|
|
435
|
+
"top_calibrated_score_non_positive"
|
|
436
|
+
),
|
|
437
|
+
"v7_terminal_rule_altered": False,
|
|
438
|
+
"hard_abstention": False,
|
|
439
|
+
"unobserved_outcomes_accepted": False,
|
|
440
|
+
"workload_objective_model_provider_prompt_branches": False,
|
|
441
|
+
}
|
|
442
|
+
if self.revision_2:
|
|
443
|
+
identity["phases"]["pilot"] = (
|
|
444
|
+
"sequential_adaptive_band_pilot"
|
|
445
|
+
)
|
|
446
|
+
identity["r2"] = {
|
|
447
|
+
"adaptation_temperature_hex": (
|
|
448
|
+
self.config.adaptation_temperature.hex()
|
|
449
|
+
),
|
|
450
|
+
"pilot_width_cap": (
|
|
451
|
+
"max_engine_count_or_ceil_budget_over_three"
|
|
452
|
+
),
|
|
453
|
+
"challenger_within_cell_tie_break": (
|
|
454
|
+
"native_rank_quality"
|
|
455
|
+
),
|
|
456
|
+
"pilot_seats_sequential_with_revealed_outcomes": (
|
|
457
|
+
True
|
|
458
|
+
),
|
|
459
|
+
}
|
|
460
|
+
if self.revision_3:
|
|
461
|
+
identity["r3"] = {
|
|
462
|
+
"challenger_within_cell_tie_break": (
|
|
463
|
+
"max_min_chebyshev_dispersion_from_bought_anchors_"
|
|
464
|
+
"then_chebyshev_excess_over_current_archive_front"
|
|
465
|
+
),
|
|
466
|
+
"tie_break_carries_no_weight_or_constant": True,
|
|
467
|
+
"current_archive_required": True,
|
|
468
|
+
}
|
|
469
|
+
if self.revision_4:
|
|
470
|
+
identity["r4"] = {
|
|
471
|
+
"challenger_conversion_tail_risk": (
|
|
472
|
+
"expected_shortfall_below_the_conversion_branch_"
|
|
473
|
+
"own_expected_gain"
|
|
474
|
+
),
|
|
475
|
+
"magnitude_channel_sign_preserved_below_half": True,
|
|
476
|
+
"carries_no_new_constant": True,
|
|
477
|
+
}
|
|
478
|
+
return identity
|
|
479
|
+
|
|
480
|
+
def identity_record(self) -> dict[str, object]:
|
|
481
|
+
"""Public identity dict binding every composed component."""
|
|
482
|
+
|
|
483
|
+
self.__post_init__()
|
|
484
|
+
return {
|
|
485
|
+
**self._identity(
|
|
486
|
+
pilot=self.pilot_policy(),
|
|
487
|
+
challenger=self.challenger_policy(),
|
|
488
|
+
terminal=self.terminal_policy(),
|
|
489
|
+
),
|
|
490
|
+
"definition_sha256": self.definition_sha256,
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
def pilot_policy(
|
|
494
|
+
self,
|
|
495
|
+
) -> RankBalancedCausalPilotPolicy | SequentialAdaptiveBandPilotPolicy:
|
|
496
|
+
if self.revision_2:
|
|
497
|
+
return SequentialAdaptiveBandPilotPolicy(
|
|
498
|
+
band_count=self.config.band_count,
|
|
499
|
+
band_weights=self.config.band_weights,
|
|
500
|
+
exploration_epsilon=self.config.exploration_epsilon,
|
|
501
|
+
adaptation_temperature=(
|
|
502
|
+
self.config.adaptation_temperature
|
|
503
|
+
),
|
|
504
|
+
prior_strength=self.config.prior_strength,
|
|
505
|
+
random_seed=self.config.random_seed,
|
|
506
|
+
)
|
|
507
|
+
return RankBalancedCausalPilotPolicy(
|
|
508
|
+
band_count=self.config.band_count,
|
|
509
|
+
band_weights=self.config.band_weights,
|
|
510
|
+
exploration_epsilon=self.config.exploration_epsilon,
|
|
511
|
+
random_seed=self.config.random_seed,
|
|
512
|
+
)
|
|
513
|
+
|
|
514
|
+
def pilot_width_for(
|
|
515
|
+
self,
|
|
516
|
+
*,
|
|
517
|
+
evaluation_slots: int,
|
|
518
|
+
engine_count: int,
|
|
519
|
+
) -> int:
|
|
520
|
+
"""Seats the pilot claims before the challenger takes over."""
|
|
521
|
+
|
|
522
|
+
width = min(
|
|
523
|
+
self.config.diagnostic_slots,
|
|
524
|
+
evaluation_slots - self.config.randomized_audit_slots,
|
|
525
|
+
)
|
|
526
|
+
if self.revision_2:
|
|
527
|
+
# r2 cap: coverage never exceeds the engine count or one
|
|
528
|
+
# third of the budget, whichever is larger, so evidence-
|
|
529
|
+
# driven challenger seats start earlier at wide budgets.
|
|
530
|
+
width = min(
|
|
531
|
+
width,
|
|
532
|
+
max(
|
|
533
|
+
engine_count,
|
|
534
|
+
math.ceil(evaluation_slots / 3),
|
|
535
|
+
),
|
|
536
|
+
)
|
|
537
|
+
return width
|
|
538
|
+
|
|
539
|
+
def challenger_policy(self) -> CalibratedPositiveGainOpportunityPolicy:
|
|
540
|
+
return CalibratedPositiveGainOpportunityPolicy(
|
|
541
|
+
archive_gain_utility=self.archive_gain_utility,
|
|
542
|
+
lambda_=self.config.lambda_,
|
|
543
|
+
beta=self.config.beta,
|
|
544
|
+
mixture_weight=self.config.mixture_weight,
|
|
545
|
+
prior_strength=self.config.prior_strength,
|
|
546
|
+
band_count=self.config.band_count,
|
|
547
|
+
reference_gain_scale=self.config.reference_gain_scale,
|
|
548
|
+
probability_floor=self.config.probability_floor,
|
|
549
|
+
scenario_weights=self.config.scenario_weights,
|
|
550
|
+
frozen_score_weight=self.config.frozen_score_weight,
|
|
551
|
+
frozen_score_minimum_training_runs=(
|
|
552
|
+
self.config.frozen_score_minimum_training_runs
|
|
553
|
+
),
|
|
554
|
+
within_cell_rank_tie_break=self.revision_2,
|
|
555
|
+
anchor_geometry_tie_break=self.revision_3,
|
|
556
|
+
downside_shortfall_tail_risk=self.revision_4,
|
|
557
|
+
)
|
|
558
|
+
|
|
559
|
+
def terminal_policy(self) -> OutcomeAdaptiveActionRacingPolicy:
|
|
560
|
+
"""The frozen V7 incumbent used for terminal seats and fallback."""
|
|
561
|
+
|
|
562
|
+
return OutcomeAdaptiveActionRacingPolicy(
|
|
563
|
+
diagnostic_slots=self.config.diagnostic_slots,
|
|
564
|
+
randomized_audit_slots=self.config.randomized_audit_slots,
|
|
565
|
+
reference_gain_scale=self.config.reference_gain_scale,
|
|
566
|
+
reference_gain_evidence_sha256=(
|
|
567
|
+
self.config.reference_gain_evidence_sha256
|
|
568
|
+
),
|
|
569
|
+
prior_strength=self.config.prior_strength,
|
|
570
|
+
ucb_strength=self.config.ucb_strength,
|
|
571
|
+
counterfactual_strength=(
|
|
572
|
+
self.config.counterfactual_strength
|
|
573
|
+
),
|
|
574
|
+
diversity_strength=self.config.diversity_strength,
|
|
575
|
+
randomized_audit_after_directed_steps=(
|
|
576
|
+
self.config.randomized_audit_after_directed_steps
|
|
577
|
+
),
|
|
578
|
+
exploration_pool_size=self.config.exploration_pool_size,
|
|
579
|
+
trace_alternative_count=(
|
|
580
|
+
self.config.trace_alternative_count
|
|
581
|
+
),
|
|
582
|
+
minimum_post_audit_optimization_slots=(
|
|
583
|
+
self.config.minimum_post_audit_optimization_slots
|
|
584
|
+
),
|
|
585
|
+
terminal_hierarchical_slots=(
|
|
586
|
+
self.config.terminal_hierarchical_slots
|
|
587
|
+
),
|
|
588
|
+
native_rank_strength=self.config.native_rank_strength,
|
|
589
|
+
random_seed=self.config.random_seed,
|
|
590
|
+
policy_version=(
|
|
591
|
+
OUTCOME_ADAPTIVE_ACTION_RACING_HORIZON_HIERARCHICAL_POLICY_VERSION
|
|
592
|
+
),
|
|
593
|
+
)
|
|
594
|
+
|
|
595
|
+
@staticmethod
|
|
596
|
+
def _validated_forecasts(
|
|
597
|
+
forecasts: tuple[tuple[str, PositiveGainForecast], ...],
|
|
598
|
+
) -> dict[str, PositiveGainForecast]:
|
|
599
|
+
if type(forecasts) is not tuple:
|
|
600
|
+
raise TypeError("forecasts must be an exact tuple")
|
|
601
|
+
result: dict[str, PositiveGainForecast] = {}
|
|
602
|
+
for value in forecasts:
|
|
603
|
+
if type(value) is not tuple or len(value) != 2:
|
|
604
|
+
raise TypeError("forecasts must pair action and forecast")
|
|
605
|
+
action_sha256, forecast = value
|
|
606
|
+
require_sha256(action_sha256, "forecast action_sha256")
|
|
607
|
+
if type(forecast) is not PositiveGainForecast:
|
|
608
|
+
raise TypeError("forecast must be exact")
|
|
609
|
+
forecast.__post_init__()
|
|
610
|
+
if action_sha256 in result:
|
|
611
|
+
raise ValueError("forecasts repeat an action")
|
|
612
|
+
result[action_sha256] = forecast
|
|
613
|
+
return result
|
|
614
|
+
|
|
615
|
+
@staticmethod
|
|
616
|
+
def _validated_anchors(
|
|
617
|
+
anchor_points: tuple[tuple[str, ObjectivePoint], ...],
|
|
618
|
+
) -> dict[str, ObjectivePoint]:
|
|
619
|
+
"""Outcome-blind parent anchors, one per action at most."""
|
|
620
|
+
|
|
621
|
+
if type(anchor_points) is not tuple:
|
|
622
|
+
raise TypeError("anchor_points must be an exact tuple")
|
|
623
|
+
result: dict[str, ObjectivePoint] = {}
|
|
624
|
+
for value in anchor_points:
|
|
625
|
+
if type(value) is not tuple or len(value) != 2:
|
|
626
|
+
raise TypeError("anchor_points must pair action and point")
|
|
627
|
+
action_sha256, point = value
|
|
628
|
+
require_sha256(action_sha256, "anchor action_sha256")
|
|
629
|
+
if type(point) is not tuple or not point:
|
|
630
|
+
raise TypeError("anchor point must be an exact tuple")
|
|
631
|
+
if action_sha256 in result:
|
|
632
|
+
raise ValueError("anchor_points repeat an action")
|
|
633
|
+
result[action_sha256] = point
|
|
634
|
+
return result
|
|
635
|
+
|
|
636
|
+
def design_pilot(
|
|
637
|
+
self,
|
|
638
|
+
*,
|
|
639
|
+
residual_request_sha256: str,
|
|
640
|
+
actions: tuple[AdaptiveActionDescriptor, ...],
|
|
641
|
+
evaluation_slots: int,
|
|
642
|
+
) -> V8LiteDecision:
|
|
643
|
+
"""Phase one: rank-balanced randomized pilot over the market."""
|
|
644
|
+
|
|
645
|
+
self.__post_init__()
|
|
646
|
+
if self.revision_2:
|
|
647
|
+
raise ValueError(
|
|
648
|
+
"revision 2 pilots are sequential; use "
|
|
649
|
+
"design_pilot_seat"
|
|
650
|
+
)
|
|
651
|
+
if type(actions) is not tuple or not actions:
|
|
652
|
+
raise ValueError("actions must be a non-empty exact tuple")
|
|
653
|
+
for value in actions:
|
|
654
|
+
if type(value) is not AdaptiveActionDescriptor:
|
|
655
|
+
raise TypeError("actions must contain exact descriptors")
|
|
656
|
+
value.__post_init__()
|
|
657
|
+
if (
|
|
658
|
+
type(evaluation_slots) is not int
|
|
659
|
+
or not 2 <= evaluation_slots <= len(actions)
|
|
660
|
+
):
|
|
661
|
+
raise ValueError("evaluation_slots must fit the action market")
|
|
662
|
+
pilot_width = self.pilot_width_for(
|
|
663
|
+
evaluation_slots=evaluation_slots,
|
|
664
|
+
engine_count=len(
|
|
665
|
+
{value.lane_id for value in actions}
|
|
666
|
+
),
|
|
667
|
+
)
|
|
668
|
+
if pilot_width <= 0:
|
|
669
|
+
raise ValueError("pilot design leaves no pilot")
|
|
670
|
+
design = self.pilot_policy().design_pilot(
|
|
671
|
+
residual_request_sha256=residual_request_sha256,
|
|
672
|
+
candidates=tuple(
|
|
673
|
+
RankBalancedPilotCandidate(
|
|
674
|
+
action_sha256=value.action_sha256,
|
|
675
|
+
engine_id=value.lane_id,
|
|
676
|
+
native_rank=value.native_rank,
|
|
677
|
+
frozen_score=value.prior_score,
|
|
678
|
+
)
|
|
679
|
+
for value in actions
|
|
680
|
+
),
|
|
681
|
+
pilot_width=pilot_width,
|
|
682
|
+
)
|
|
683
|
+
return V8LiteDecision(
|
|
684
|
+
policy_id=self.policy_id,
|
|
685
|
+
policy_version_id=self.policy_version_id,
|
|
686
|
+
policy_definition_sha256=self.definition_sha256,
|
|
687
|
+
residual_request_sha256=residual_request_sha256,
|
|
688
|
+
phase=V8LITE_PHASE_PILOT,
|
|
689
|
+
authority_policy_id=design.policy_id,
|
|
690
|
+
selected_action_sha256s=design.selected_action_sha256s,
|
|
691
|
+
selection_propensity=design.design_propensity,
|
|
692
|
+
evidence=freeze_json(
|
|
693
|
+
{
|
|
694
|
+
"pilot_design": design.to_record(),
|
|
695
|
+
"candidate_outcomes_observed": False,
|
|
696
|
+
}
|
|
697
|
+
),
|
|
698
|
+
pilot_design=design,
|
|
699
|
+
)
|
|
700
|
+
|
|
701
|
+
def design_pilot_seat(
|
|
702
|
+
self,
|
|
703
|
+
*,
|
|
704
|
+
residual_request_sha256: str,
|
|
705
|
+
actions: tuple[AdaptiveActionDescriptor, ...],
|
|
706
|
+
evaluation_slots: int,
|
|
707
|
+
selected_action_sha256s: tuple[str, ...],
|
|
708
|
+
outcomes: tuple[AdaptiveActionOutcome, ...],
|
|
709
|
+
) -> V8LiteDecision:
|
|
710
|
+
"""Revision-2 sequential pilot: one seat given revealed history."""
|
|
711
|
+
|
|
712
|
+
self.__post_init__()
|
|
713
|
+
if not self.revision_2:
|
|
714
|
+
raise ValueError(
|
|
715
|
+
"sequential pilot seats require revision 2"
|
|
716
|
+
)
|
|
717
|
+
if type(actions) is not tuple or not actions:
|
|
718
|
+
raise ValueError("actions must be a non-empty exact tuple")
|
|
719
|
+
for value in actions:
|
|
720
|
+
if type(value) is not AdaptiveActionDescriptor:
|
|
721
|
+
raise TypeError("actions must contain exact descriptors")
|
|
722
|
+
value.__post_init__()
|
|
723
|
+
if (
|
|
724
|
+
type(evaluation_slots) is not int
|
|
725
|
+
or not 2 <= evaluation_slots <= len(actions)
|
|
726
|
+
):
|
|
727
|
+
raise ValueError("evaluation_slots must fit the action market")
|
|
728
|
+
by_action = {value.action_sha256: value for value in actions}
|
|
729
|
+
pilot_width = self.pilot_width_for(
|
|
730
|
+
evaluation_slots=evaluation_slots,
|
|
731
|
+
engine_count=len(
|
|
732
|
+
{value.lane_id for value in actions}
|
|
733
|
+
),
|
|
734
|
+
)
|
|
735
|
+
if pilot_width <= 0:
|
|
736
|
+
raise ValueError("pilot design leaves no pilot")
|
|
737
|
+
if len(selected_action_sha256s) >= pilot_width:
|
|
738
|
+
raise ValueError("the pilot is already complete")
|
|
739
|
+
outcome_by_action = {
|
|
740
|
+
value.action_sha256: value for value in outcomes
|
|
741
|
+
}
|
|
742
|
+
seat = self.pilot_policy().design_seat(
|
|
743
|
+
residual_request_sha256=residual_request_sha256,
|
|
744
|
+
candidates=tuple(
|
|
745
|
+
RankBalancedPilotCandidate(
|
|
746
|
+
action_sha256=value.action_sha256,
|
|
747
|
+
engine_id=value.lane_id,
|
|
748
|
+
native_rank=value.native_rank,
|
|
749
|
+
frozen_score=value.prior_score,
|
|
750
|
+
)
|
|
751
|
+
for value in actions
|
|
752
|
+
),
|
|
753
|
+
selected_action_sha256s=selected_action_sha256s,
|
|
754
|
+
observations=tuple(
|
|
755
|
+
PilotSeatObservation(
|
|
756
|
+
action_sha256=value.action_sha256,
|
|
757
|
+
feasible=value.feasible,
|
|
758
|
+
marginal_archive_gain=(
|
|
759
|
+
value.marginal_archive_gain
|
|
760
|
+
),
|
|
761
|
+
)
|
|
762
|
+
for value in sorted(
|
|
763
|
+
outcomes,
|
|
764
|
+
key=lambda item: item.action_sha256,
|
|
765
|
+
)
|
|
766
|
+
),
|
|
767
|
+
seat_ordinal=len(selected_action_sha256s) + 1,
|
|
768
|
+
)
|
|
769
|
+
if seat.selected_action_sha256 not in by_action:
|
|
770
|
+
raise AssertionError( # pragma: no cover
|
|
771
|
+
"pilot seat escaped the market"
|
|
772
|
+
)
|
|
773
|
+
return V8LiteDecision(
|
|
774
|
+
policy_id=self.policy_id,
|
|
775
|
+
policy_version_id=self.policy_version_id,
|
|
776
|
+
policy_definition_sha256=self.definition_sha256,
|
|
777
|
+
residual_request_sha256=residual_request_sha256,
|
|
778
|
+
phase=V8LITE_PHASE_PILOT,
|
|
779
|
+
authority_policy_id=(
|
|
780
|
+
SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_ID
|
|
781
|
+
),
|
|
782
|
+
selected_action_sha256s=(
|
|
783
|
+
seat.selected_action_sha256,
|
|
784
|
+
),
|
|
785
|
+
selection_propensity=seat.selection_propensity,
|
|
786
|
+
evidence=freeze_json(
|
|
787
|
+
{
|
|
788
|
+
"pilot_seat": seat.to_record(),
|
|
789
|
+
"pilot_width": pilot_width,
|
|
790
|
+
"sequential_adaptive_pilot": True,
|
|
791
|
+
"revealed_outcome_count": len(outcomes),
|
|
792
|
+
}
|
|
793
|
+
),
|
|
794
|
+
)
|
|
795
|
+
|
|
796
|
+
def select_next(
|
|
797
|
+
self,
|
|
798
|
+
*,
|
|
799
|
+
residual_request_sha256: str,
|
|
800
|
+
actions: tuple[AdaptiveActionDescriptor, ...],
|
|
801
|
+
evaluation_slots: int,
|
|
802
|
+
diagnostic_action_sha256s: tuple[str, ...],
|
|
803
|
+
diagnostic_joint_gain: float,
|
|
804
|
+
selected_action_sha256s: tuple[str, ...],
|
|
805
|
+
outcomes: tuple[AdaptiveActionOutcome, ...],
|
|
806
|
+
archive_points: tuple[ObjectivePoint, ...],
|
|
807
|
+
forecasts: tuple[tuple[str, PositiveGainForecast], ...] = (),
|
|
808
|
+
anchor_points: tuple[tuple[str, ObjectivePoint], ...] = (),
|
|
809
|
+
frozen_fit_training_run_count: int = 0,
|
|
810
|
+
prior_conversion_outcomes: tuple[
|
|
811
|
+
ObservedConversionOutcome,
|
|
812
|
+
...,
|
|
813
|
+
] = (),
|
|
814
|
+
set_outcomes: tuple[AdaptiveActionSetOutcome, ...] = (),
|
|
815
|
+
) -> V8LiteDecision:
|
|
816
|
+
"""Select one continuation action at the current cutoff.
|
|
817
|
+
|
|
818
|
+
``prior_conversion_outcomes`` carries conversion evidence that
|
|
819
|
+
was fully observed BEFORE this market opened (for example a
|
|
820
|
+
warm-start from earlier completed markets). It is re-ordinaled
|
|
821
|
+
ahead of the within-market outcomes, so the prequential
|
|
822
|
+
no-future-leak invariant is preserved: nothing later than the
|
|
823
|
+
current cutoff can enter the evidence stream.
|
|
824
|
+
"""
|
|
825
|
+
|
|
826
|
+
self.__post_init__()
|
|
827
|
+
require_sha256(
|
|
828
|
+
residual_request_sha256,
|
|
829
|
+
"residual_request_sha256",
|
|
830
|
+
)
|
|
831
|
+
if type(actions) is not tuple or not actions:
|
|
832
|
+
raise ValueError("actions must be a non-empty exact tuple")
|
|
833
|
+
for value in actions:
|
|
834
|
+
if type(value) is not AdaptiveActionDescriptor:
|
|
835
|
+
raise TypeError("actions must contain exact descriptors")
|
|
836
|
+
value.__post_init__()
|
|
837
|
+
if (
|
|
838
|
+
type(evaluation_slots) is not int
|
|
839
|
+
or not 2 <= evaluation_slots <= len(actions)
|
|
840
|
+
):
|
|
841
|
+
raise ValueError("evaluation_slots must fit the action market")
|
|
842
|
+
if (
|
|
843
|
+
type(selected_action_sha256s) is not tuple
|
|
844
|
+
or selected_action_sha256s
|
|
845
|
+
!= tuple(sorted(set(selected_action_sha256s)))
|
|
846
|
+
):
|
|
847
|
+
raise ValueError(
|
|
848
|
+
"selected_action_sha256s must be an exact canonical tuple"
|
|
849
|
+
)
|
|
850
|
+
seats_left = evaluation_slots - len(selected_action_sha256s)
|
|
851
|
+
if seats_left <= 0:
|
|
852
|
+
raise ValueError("the evaluation slate is already complete")
|
|
853
|
+
terminal = seats_left <= self.config.terminal_hierarchical_slots
|
|
854
|
+
incumbent = self.terminal_policy()
|
|
855
|
+
if terminal:
|
|
856
|
+
# EXACT V7 terminal hierarchical exploitation by delegation.
|
|
857
|
+
delegated = incumbent.select_next(
|
|
858
|
+
residual_request_sha256=residual_request_sha256,
|
|
859
|
+
actions=actions,
|
|
860
|
+
evaluation_slots=evaluation_slots,
|
|
861
|
+
diagnostic_action_sha256s=diagnostic_action_sha256s,
|
|
862
|
+
diagnostic_joint_gain=diagnostic_joint_gain,
|
|
863
|
+
selected_action_sha256s=selected_action_sha256s,
|
|
864
|
+
outcomes=outcomes,
|
|
865
|
+
set_outcomes=set_outcomes,
|
|
866
|
+
)
|
|
867
|
+
return self._wrap_delegated(
|
|
868
|
+
residual_request_sha256=residual_request_sha256,
|
|
869
|
+
phase=V8LITE_PHASE_TERMINAL,
|
|
870
|
+
delegated=delegated,
|
|
871
|
+
extra_evidence={
|
|
872
|
+
"terminal_delegation": True,
|
|
873
|
+
"seats_left_before_decision": seats_left,
|
|
874
|
+
},
|
|
875
|
+
)
|
|
876
|
+
|
|
877
|
+
forecast_by_action = self._validated_forecasts(forecasts)
|
|
878
|
+
anchor_by_action = self._validated_anchors(anchor_points)
|
|
879
|
+
by_action = {value.action_sha256: value for value in actions}
|
|
880
|
+
outcome_by_action = {
|
|
881
|
+
value.action_sha256: value for value in outcomes
|
|
882
|
+
}
|
|
883
|
+
if set(outcome_by_action) != set(selected_action_sha256s):
|
|
884
|
+
raise ValueError(
|
|
885
|
+
"observations must exactly cover all previously "
|
|
886
|
+
"selected actions"
|
|
887
|
+
)
|
|
888
|
+
if not set(selected_action_sha256s) <= set(by_action):
|
|
889
|
+
raise ValueError("selected action is outside the sealed market")
|
|
890
|
+
selected_phenotypes = {
|
|
891
|
+
by_action[value].phenotype_sha256
|
|
892
|
+
for value in selected_action_sha256s
|
|
893
|
+
}
|
|
894
|
+
remaining = tuple(
|
|
895
|
+
value
|
|
896
|
+
for value in actions
|
|
897
|
+
if value.action_sha256 not in outcome_by_action
|
|
898
|
+
and value.phenotype_sha256 not in selected_phenotypes
|
|
899
|
+
)
|
|
900
|
+
if not remaining:
|
|
901
|
+
raise ValueError("no unevaluated action can fill the slate")
|
|
902
|
+
if type(prior_conversion_outcomes) is not tuple or any(
|
|
903
|
+
type(value) is not ObservedConversionOutcome
|
|
904
|
+
for value in prior_conversion_outcomes
|
|
905
|
+
):
|
|
906
|
+
raise TypeError(
|
|
907
|
+
"prior_conversion_outcomes must contain exact "
|
|
908
|
+
"conversion outcomes"
|
|
909
|
+
)
|
|
910
|
+
prior_evidence = tuple(
|
|
911
|
+
ObservedConversionOutcome(
|
|
912
|
+
observation_ordinal=ordinal,
|
|
913
|
+
engine_id=value.engine_id,
|
|
914
|
+
native_rank=value.native_rank,
|
|
915
|
+
lane_size=value.lane_size,
|
|
916
|
+
feasible=value.feasible,
|
|
917
|
+
marginal_archive_gain=value.marginal_archive_gain,
|
|
918
|
+
)
|
|
919
|
+
for ordinal, value in enumerate(
|
|
920
|
+
prior_conversion_outcomes,
|
|
921
|
+
start=1,
|
|
922
|
+
)
|
|
923
|
+
)
|
|
924
|
+
conversion_outcomes = prior_evidence + tuple(
|
|
925
|
+
ObservedConversionOutcome(
|
|
926
|
+
observation_ordinal=ordinal,
|
|
927
|
+
engine_id=by_action[action_sha256].lane_id,
|
|
928
|
+
native_rank=by_action[action_sha256].native_rank,
|
|
929
|
+
lane_size=by_action[action_sha256].lane_size,
|
|
930
|
+
feasible=outcome_by_action[action_sha256].feasible,
|
|
931
|
+
marginal_archive_gain=(
|
|
932
|
+
outcome_by_action[
|
|
933
|
+
action_sha256
|
|
934
|
+
].marginal_archive_gain
|
|
935
|
+
),
|
|
936
|
+
)
|
|
937
|
+
for ordinal, action_sha256 in enumerate(
|
|
938
|
+
sorted(selected_action_sha256s),
|
|
939
|
+
start=len(prior_evidence) + 1,
|
|
940
|
+
)
|
|
941
|
+
)
|
|
942
|
+
ranking = self.challenger_policy().score_market(
|
|
943
|
+
candidates=tuple(
|
|
944
|
+
PositiveGainCandidate(
|
|
945
|
+
action_sha256=value.action_sha256,
|
|
946
|
+
engine_id=value.lane_id,
|
|
947
|
+
native_rank=value.native_rank,
|
|
948
|
+
lane_size=value.lane_size,
|
|
949
|
+
forecast=forecast_by_action.get(
|
|
950
|
+
value.action_sha256
|
|
951
|
+
),
|
|
952
|
+
frozen_score=value.prior_score,
|
|
953
|
+
anchor_point=anchor_by_action.get(
|
|
954
|
+
value.action_sha256
|
|
955
|
+
),
|
|
956
|
+
)
|
|
957
|
+
for value in remaining
|
|
958
|
+
),
|
|
959
|
+
archive_points=archive_points,
|
|
960
|
+
observed_outcomes=conversion_outcomes,
|
|
961
|
+
future_seats_remaining=seats_left - 1,
|
|
962
|
+
horizon_total=evaluation_slots,
|
|
963
|
+
frozen_fit_training_run_count=(
|
|
964
|
+
frozen_fit_training_run_count
|
|
965
|
+
),
|
|
966
|
+
# The anchors this market has already paid to sample. They
|
|
967
|
+
# join the archive-geometry channel's reference front, which is
|
|
968
|
+
# what stops a second seat landing in a region the first seat
|
|
969
|
+
# already bought.
|
|
970
|
+
covered_anchors=tuple(
|
|
971
|
+
anchor_by_action[action_sha256]
|
|
972
|
+
for action_sha256 in sorted(selected_action_sha256s)
|
|
973
|
+
if action_sha256 in anchor_by_action
|
|
974
|
+
),
|
|
975
|
+
)
|
|
976
|
+
top_action_sha256 = ranking.ranked_action_sha256s[0]
|
|
977
|
+
top_score = ranking.score_for(top_action_sha256)
|
|
978
|
+
if top_score.score > 0.0:
|
|
979
|
+
return V8LiteDecision(
|
|
980
|
+
policy_id=self.policy_id,
|
|
981
|
+
policy_version_id=self.policy_version_id,
|
|
982
|
+
policy_definition_sha256=self.definition_sha256,
|
|
983
|
+
residual_request_sha256=residual_request_sha256,
|
|
984
|
+
phase=V8LITE_PHASE_ADAPTIVE,
|
|
985
|
+
authority_policy_id=ranking.policy_id,
|
|
986
|
+
selected_action_sha256s=(top_action_sha256,),
|
|
987
|
+
selection_propensity=1.0,
|
|
988
|
+
evidence=freeze_json(
|
|
989
|
+
{
|
|
990
|
+
"challenger_ranking": ranking.to_record(
|
|
991
|
+
include_scores=True
|
|
992
|
+
),
|
|
993
|
+
"selected_score_sha256": (
|
|
994
|
+
top_score.score_sha256
|
|
995
|
+
),
|
|
996
|
+
"protected_fallback_used": False,
|
|
997
|
+
"seats_left_before_decision": seats_left,
|
|
998
|
+
"prior_conversion_evidence_count": len(
|
|
999
|
+
prior_evidence
|
|
1000
|
+
),
|
|
1001
|
+
"unobserved_candidate_outcomes_available": (
|
|
1002
|
+
False
|
|
1003
|
+
),
|
|
1004
|
+
}
|
|
1005
|
+
),
|
|
1006
|
+
challenger_ranking=ranking,
|
|
1007
|
+
)
|
|
1008
|
+
# Protected fallback: the existing controller decides unchanged.
|
|
1009
|
+
delegated = incumbent.select_next(
|
|
1010
|
+
residual_request_sha256=residual_request_sha256,
|
|
1011
|
+
actions=actions,
|
|
1012
|
+
evaluation_slots=evaluation_slots,
|
|
1013
|
+
diagnostic_action_sha256s=diagnostic_action_sha256s,
|
|
1014
|
+
diagnostic_joint_gain=diagnostic_joint_gain,
|
|
1015
|
+
selected_action_sha256s=selected_action_sha256s,
|
|
1016
|
+
outcomes=outcomes,
|
|
1017
|
+
set_outcomes=set_outcomes,
|
|
1018
|
+
)
|
|
1019
|
+
return self._wrap_delegated(
|
|
1020
|
+
residual_request_sha256=residual_request_sha256,
|
|
1021
|
+
phase=V8LITE_PHASE_PROTECTED_FALLBACK,
|
|
1022
|
+
delegated=delegated,
|
|
1023
|
+
extra_evidence={
|
|
1024
|
+
"protected_fallback_used": True,
|
|
1025
|
+
"fallback_reason": (
|
|
1026
|
+
"challenger_top_score_non_positive"
|
|
1027
|
+
),
|
|
1028
|
+
"challenger_ranking": ranking.to_record(
|
|
1029
|
+
include_scores=True
|
|
1030
|
+
),
|
|
1031
|
+
"seats_left_before_decision": seats_left,
|
|
1032
|
+
},
|
|
1033
|
+
challenger_ranking=ranking,
|
|
1034
|
+
)
|
|
1035
|
+
|
|
1036
|
+
def _wrap_delegated(
|
|
1037
|
+
self,
|
|
1038
|
+
*,
|
|
1039
|
+
residual_request_sha256: str,
|
|
1040
|
+
phase: str,
|
|
1041
|
+
delegated: AdaptiveActionRacingDecision,
|
|
1042
|
+
extra_evidence: dict[str, object],
|
|
1043
|
+
challenger_ranking: CalibratedPositiveGainRanking | None = None,
|
|
1044
|
+
) -> V8LiteDecision:
|
|
1045
|
+
return V8LiteDecision(
|
|
1046
|
+
policy_id=self.policy_id,
|
|
1047
|
+
policy_version_id=self.policy_version_id,
|
|
1048
|
+
policy_definition_sha256=self.definition_sha256,
|
|
1049
|
+
residual_request_sha256=residual_request_sha256,
|
|
1050
|
+
phase=phase,
|
|
1051
|
+
authority_policy_id=delegated.policy_id,
|
|
1052
|
+
selected_action_sha256s=(
|
|
1053
|
+
delegated.selected_action_sha256s
|
|
1054
|
+
),
|
|
1055
|
+
selection_propensity=delegated.selection_propensity,
|
|
1056
|
+
evidence=freeze_json(
|
|
1057
|
+
{
|
|
1058
|
+
**extra_evidence,
|
|
1059
|
+
"delegated_decision": delegated.to_record(
|
|
1060
|
+
include_evidence=True
|
|
1061
|
+
),
|
|
1062
|
+
}
|
|
1063
|
+
),
|
|
1064
|
+
delegated_decision=delegated,
|
|
1065
|
+
challenger_ranking=challenger_ranking,
|
|
1066
|
+
)
|
|
1067
|
+
|
|
1068
|
+
|
|
1069
|
+
__all__ = [
|
|
1070
|
+
"V8LITE_ALLOCATION_POLICY_ID",
|
|
1071
|
+
"V8LITE_ALLOCATION_POLICY_VERSION",
|
|
1072
|
+
"V8LITE_ALLOCATION_POLICY_VERSION_ID",
|
|
1073
|
+
"V8LITE_ALLOCATION_POLICY_VERSION_ID_R2",
|
|
1074
|
+
"V8LITE_ALLOCATION_POLICY_VERSION_ID_R3",
|
|
1075
|
+
"V8LITE_ALLOCATION_POLICY_VERSION_ID_R4",
|
|
1076
|
+
"V8LITE_PHASE_ADAPTIVE",
|
|
1077
|
+
"V8LITE_PHASE_PILOT",
|
|
1078
|
+
"V8LITE_PHASE_PROTECTED_FALLBACK",
|
|
1079
|
+
"V8LITE_PHASE_TERMINAL",
|
|
1080
|
+
"V8LiteAllocationConfig",
|
|
1081
|
+
"V8LiteAllocationPolicy",
|
|
1082
|
+
"V8LiteDecision",
|
|
1083
|
+
]
|