agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1133 @@
|
|
|
1
|
+
"""Measurement-conditioned revision of the graded prior, mid-run.
|
|
2
|
+
|
|
3
|
+
Every prior this package installs today is authored ONCE, before the run has
|
|
4
|
+
measured anything worth reading: the screen's statistical rule and the graded
|
|
5
|
+
prior it feeds both settle at t = 0 and are then held for the whole budget.
|
|
6
|
+
That is the right shape when the budget is short and the wrong shape when it
|
|
7
|
+
is not -- a prior fitted to the first rows keeps steering draws long after the
|
|
8
|
+
rows that justified it stopped being the run's evidence, and staleness is
|
|
9
|
+
measured to decide the row at the larger budgets (GE: the best arm at the
|
|
10
|
+
small budgets loses decisively by B = 160, with the late half of a run
|
|
11
|
+
gaining a fraction of what its early half gained).
|
|
12
|
+
|
|
13
|
+
This module is the other clock. At a DECLARED cadence in charged evaluations,
|
|
14
|
+
one model call reads the domain card, the run's own measured-evidence
|
|
15
|
+
rendering (:mod:`agent_evolve.policies.measurement_evidence`) and the weights
|
|
16
|
+
currently installed, and replies with a revised graded prior -- and, where the
|
|
17
|
+
run bought that channel, with the complete configurations it was REQUIRED to
|
|
18
|
+
propose beside it. The prior half is admitted or refused WHOLE and then
|
|
19
|
+
step-damped into the installed weights, so a revision is a tilt and never a
|
|
20
|
+
replacement; the proposals beside it are validated one at a time, and a short
|
|
21
|
+
list is counted rather than fatal.
|
|
22
|
+
|
|
23
|
+
The evidence is a BUNDLE of two renderings, and the second one is why this
|
|
24
|
+
channel was rebuilt. ``render_measurement_evidence`` supplies the front, the
|
|
25
|
+
progress line and the per-locus rank correlations; ``render_elite_table``
|
|
26
|
+
supplies value occupancy among the non-dominated configurations. At the 40-90
|
|
27
|
+
rows a revision actually holds, over a two-dozen-field space, those
|
|
28
|
+
correlations are noise -- the W1 pilot moved less between arms than the same
|
|
29
|
+
arm moved between two draws -- while the sealed prior that did separate was
|
|
30
|
+
authored in the occupancy format. Both halves read the same viewed rows, one
|
|
31
|
+
digest covers the whole bundle, and every event journals ``evidence_text``:
|
|
32
|
+
the bundle verbatim. A digest can falsify a reconstruction but cannot produce
|
|
33
|
+
one, and the oracle instrument measured that a late checkpoint's prompt is not
|
|
34
|
+
reconstructible from the cells beside it -- the row list a checkpoint reads
|
|
35
|
+
carries cache-served repeats that the charge log, which counts charged
|
|
36
|
+
evaluations, cannot recover.
|
|
37
|
+
|
|
38
|
+
Three constraints shape everything here, and each is structural rather than
|
|
39
|
+
procedural:
|
|
40
|
+
|
|
41
|
+
*Damping bounds the damage.* ``w_new = (1 - a) * w_prev + a * w_admitted``
|
|
42
|
+
with ``a < 1`` keeps every previously positive weight positive, so a revision
|
|
43
|
+
CANNOT introduce an exclusion; the worst case of a wrong revision is wasted
|
|
44
|
+
draws inside the declared domain. The same mixture keeps the concentration
|
|
45
|
+
cap: a convex combination of two vectors whose max/min ratio is at most ``r``
|
|
46
|
+
has ratio at most ``r`` (the mediant inequality), so admitting proposals under
|
|
47
|
+
``max_weight_ratio`` bounds every installed prior at that ratio forever.
|
|
48
|
+
|
|
49
|
+
The admission gate is narrow because of that, and deliberately so. A value the
|
|
50
|
+
reply leaves OUT of a parameter it names is not a zero -- the mixture leaves it
|
|
51
|
+
``(1 - a)`` of the share it held -- so ``excludes_front`` fires only on an
|
|
52
|
+
EXPLICIT zero weight for a value some rank-0 configuration holds. Reading
|
|
53
|
+
silence as exclusion made the SEMANTICS the binding constraint on this channel
|
|
54
|
+
rather than the model: 10 of 11 live refusals were ``excludes_front`` on
|
|
55
|
+
subset replies; on the losing taped pair the late revisions were refused
|
|
56
|
+
exactly where the oracle's hindsight alignment peaked (delta loglik 2.55 at
|
|
57
|
+
k = 2); and the oracle's OWN replies -- authored with the winning run's front
|
|
58
|
+
in hand -- were refused at the late checkpoints of BOTH studies (s101 at
|
|
59
|
+
k = 3; s105 at k = 2 and k = 3). A rule that refuses hindsight is measuring
|
|
60
|
+
itself. The rule is stated as the condition it rests on rather than assumed:
|
|
61
|
+
at ``a = 1`` there is no mixture, an omission really does become a zero, and
|
|
62
|
+
the gate reads silence the old way because nothing else is left to.
|
|
63
|
+
|
|
64
|
+
*A revision is a bet, and a bet is checked.* Each event records the weights it
|
|
65
|
+
replaced and where the trace stood. At the next checkpoint, if nothing
|
|
66
|
+
measured since is rank-0 in the pooled rows, the weights revert to the
|
|
67
|
+
pre-event snapshot before anything new is considered. Nothing here is
|
|
68
|
+
unrecoverable, which is exactly why the check can be this cheap.
|
|
69
|
+
|
|
70
|
+
*The control is a declared parameter.* ``evidence_view`` receives the rows the
|
|
71
|
+
run measured and returns the rows the model is shown, so the shuffled-evidence
|
|
72
|
+
arm -- same count, same shape, same cost, another run's rows -- is buildable
|
|
73
|
+
without editing the product. Every event journals the digest of the rendered
|
|
74
|
+
evidence, so no run can imply it reasoned over its own measurements when it
|
|
75
|
+
did not. ``gate_reads_view`` names which of those two row sets the front check
|
|
76
|
+
reads, and its default is the product's safety stance: the gate that protects
|
|
77
|
+
a LIVE run reads REALITY, so no revision can write a zero onto a value that
|
|
78
|
+
some configuration this run actually measured onto the front, whatever the
|
|
79
|
+
prompt happened to show. A CONTROL arm sets it ``True``, because a control
|
|
80
|
+
whose prompt reads donor rows while its gate reads this run's front accrues
|
|
81
|
+
``excludes_front`` refusals the arm it controls never meets, and its refusal
|
|
82
|
+
rate stops being comparable (W1 pilot, seed 20370103: two of four revisions
|
|
83
|
+
refused on the shuffled arm alone, on evidence that named no front value).
|
|
84
|
+
A control is the only sane user of ``True``.
|
|
85
|
+
|
|
86
|
+
This is the GRADED, mid-run form of the typed locus restriction, which is the
|
|
87
|
+
one measurement-conditioned channel that separated from the unguided null with
|
|
88
|
+
semantics removed; it is not the re-authoring channel that lost to its
|
|
89
|
+
shuffled-evidence control, and it must not be described as one.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
from __future__ import annotations
|
|
93
|
+
|
|
94
|
+
import json
|
|
95
|
+
import math
|
|
96
|
+
import re
|
|
97
|
+
from dataclasses import dataclass, field
|
|
98
|
+
from typing import Any, Callable, Dict, List, Mapping, Optional, Sequence, Tuple
|
|
99
|
+
|
|
100
|
+
from agent_evolve.core.problem import ObjectiveSpec
|
|
101
|
+
from agent_evolve.core.results import dominates
|
|
102
|
+
from agent_evolve.policies.genetic import (
|
|
103
|
+
Locus, loci_of, locus_domain, read_locus)
|
|
104
|
+
from agent_evolve.policies.measurement_evidence import (
|
|
105
|
+
MIN_EVIDENCE_ROWS,
|
|
106
|
+
MeasuredRow,
|
|
107
|
+
evidence_digest,
|
|
108
|
+
render_elite_table,
|
|
109
|
+
render_measurement_evidence,
|
|
110
|
+
)
|
|
111
|
+
from agent_evolve.policies.weighted_prior import WeightedRestriction
|
|
112
|
+
|
|
113
|
+
__all__ = ["ReguidanceTelemetry", "ReguidanceOutcome", "Reguidance", "PROMPT",
|
|
114
|
+
"IMMIGRANTS_CLAUSE", "ELITE_TABLE_TITLE", "EVIDENCE_VERSION",
|
|
115
|
+
"MECHANISM_VERSION", "TILT_CAP"]
|
|
116
|
+
|
|
117
|
+
Config = Dict[str, Any]
|
|
118
|
+
#: field -> (values, weights), the installed overlay's one representation.
|
|
119
|
+
Overlay = Dict[str, Tuple[Tuple[Any, ...], Tuple[float, ...]]]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@dataclass
|
|
123
|
+
class ReguidanceTelemetry:
|
|
124
|
+
"""What the revision channel did. Counted, never inferred."""
|
|
125
|
+
|
|
126
|
+
calls: int = 0
|
|
127
|
+
revisions_admitted: int = 0
|
|
128
|
+
revisions_refused: int = 0
|
|
129
|
+
revisions_reverted: int = 0
|
|
130
|
+
immigrants_proposed: int = 0
|
|
131
|
+
immigrants_accepted: int = 0
|
|
132
|
+
immigrants_rejected: int = 0
|
|
133
|
+
#: Members the reply OWED and did not write, summed over the events of a
|
|
134
|
+
#: run that bought the channel: the required-k clause's compliance meter.
|
|
135
|
+
#: A shortfall costs the reply nothing else -- the prior half of the same
|
|
136
|
+
#: reply is judged on its own -- so this is the only place the ask's
|
|
137
|
+
#: answer rate is visible.
|
|
138
|
+
immigrants_shortfall: int = 0
|
|
139
|
+
#: Parameters named in ``"weights"``, summed over every reply that came
|
|
140
|
+
#: back, admitted or refused. Divided by the events that carry a breadth
|
|
141
|
+
#: it is the mean tilt; the per-event ``tilt_breadth`` carries the median
|
|
142
|
+
#: a campaign actually reads. Counted, never capped: the focused-tilt ask
|
|
143
|
+
#: is an ask, and a second refusal mode would be a throttle.
|
|
144
|
+
breadth_total: int = 0
|
|
145
|
+
errors: int = 0
|
|
146
|
+
#: One record per event: the cadence position it fired at, the digest of
|
|
147
|
+
#: the evidence the model was shown, the verdict, and what changed.
|
|
148
|
+
events: List[Dict[str, Any]] = field(default_factory=list)
|
|
149
|
+
|
|
150
|
+
def as_dict(self) -> Dict[str, int]:
|
|
151
|
+
return {
|
|
152
|
+
"calls": self.calls,
|
|
153
|
+
"revisions_admitted": self.revisions_admitted,
|
|
154
|
+
"revisions_refused": self.revisions_refused,
|
|
155
|
+
"revisions_reverted": self.revisions_reverted,
|
|
156
|
+
"immigrants_proposed": self.immigrants_proposed,
|
|
157
|
+
"immigrants_accepted": self.immigrants_accepted,
|
|
158
|
+
"immigrants_rejected": self.immigrants_rejected,
|
|
159
|
+
"immigrants_shortfall": self.immigrants_shortfall,
|
|
160
|
+
"breadth_total": self.breadth_total,
|
|
161
|
+
"errors": self.errors,
|
|
162
|
+
"events": len(self.events),
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
@dataclass(frozen=True)
|
|
167
|
+
class ReguidanceOutcome:
|
|
168
|
+
"""What the loop should do with this checkpoint.
|
|
169
|
+
|
|
170
|
+
``restriction`` is ``None`` for "keep whatever you hold": no call fired, or
|
|
171
|
+
the reply was refused and nothing about the installed weights moved. A
|
|
172
|
+
:class:`~agent_evolve.policies.weighted_prior.WeightedRestriction` is the
|
|
173
|
+
prior the loop installs from here on -- including the empty one, which
|
|
174
|
+
samples exactly as no restriction does.
|
|
175
|
+
"""
|
|
176
|
+
|
|
177
|
+
restriction: Optional[Any] = None
|
|
178
|
+
immigrants: Tuple[Config, ...] = ()
|
|
179
|
+
note: Optional[Dict[str, Any]] = None
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
PROMPT = """{context}
|
|
183
|
+
|
|
184
|
+
You are REVISING the weighted sampling prior mid-run.
|
|
185
|
+
|
|
186
|
+
The optimizer draws every new candidate from a per-parameter weight table.
|
|
187
|
+
That table was set before these measurements existed; you are being shown the
|
|
188
|
+
measurements so it can be corrected.
|
|
189
|
+
|
|
190
|
+
OBJECTIVES (name and direction):
|
|
191
|
+
{goals}
|
|
192
|
+
|
|
193
|
+
SEARCH SPACE -- every parameter and the values it may take:
|
|
194
|
+
{domains}
|
|
195
|
+
|
|
196
|
+
THE WEIGHTS CURRENTLY INSTALLED -- a parameter that does not appear here is
|
|
197
|
+
sampled UNIFORMLY over its declared values:
|
|
198
|
+
{weights}
|
|
199
|
+
|
|
200
|
+
WHAT THE OPTIMIZER HAS MEASURED SO FAR -- {rows} configurations, {charges} \
|
|
201
|
+
charged evaluations:
|
|
202
|
+
|
|
203
|
+
{evidence}
|
|
204
|
+
|
|
205
|
+
Read the measurements, not the parameter names. Decide which parameters the
|
|
206
|
+
trace says are worth concentrating the remaining budget on, and where. Name
|
|
207
|
+
AT MOST {tilt_cap} parameters in "weights" -- the table's strongest cases -- and
|
|
208
|
+
leave the rest unlisted.
|
|
209
|
+
|
|
210
|
+
Reply with ONLY a JSON object of this shape, no prose and no code fence:
|
|
211
|
+
|
|
212
|
+
{{"weights": {{"<parameter>": {{"values": [...], "weights": [...]}}}},
|
|
213
|
+
"free": ["<parameter>", ...]}}
|
|
214
|
+
{immigrants}
|
|
215
|
+
Rules, and the harness checks every one of them:
|
|
216
|
+
- Name ONLY parameters that appear in the search space above, and ONLY values
|
|
217
|
+
that parameter declares. Anything else and the WHOLE reply is REFUSED.
|
|
218
|
+
- "values" and "weights" are parallel lists of the same, non-zero length.
|
|
219
|
+
Weights must be finite and non-negative.
|
|
220
|
+
- Within one parameter the heaviest value may outweigh the lightest POSITIVE
|
|
221
|
+
one by at most {max_ratio}x; more concentration than that and the whole
|
|
222
|
+
reply is REFUSED. Concentration is the point; a de-facto exclusion is not.
|
|
223
|
+
- For a parameter you name, a value you do NOT list keeps the mass it already
|
|
224
|
+
holds, reduced by the mixture below: silence damps a value, it never
|
|
225
|
+
excludes one. List the values the evidence speaks to and stay silent about
|
|
226
|
+
the rest. What IS refused is an EXPLICIT zero weight on a value held by any
|
|
227
|
+
configuration on the front above -- writing that zero is the one revision
|
|
228
|
+
that could throw away what the run has already measured to be good.
|
|
229
|
+
- "free" lists parameters whose weights should move back toward uniform,
|
|
230
|
+
because the measurements no longer justify biasing them.
|
|
231
|
+
- Your reply is not installed as written: it is MIXED with the weights above
|
|
232
|
+
at {damping:g}, so every value sampled now stays sampled and a revision is a
|
|
233
|
+
tilt rather than a replacement. Say what the evidence says; the mixture
|
|
234
|
+
supplies the caution."""
|
|
235
|
+
|
|
236
|
+
#: The joint-proposal channel, and the reason it is REQUIRED rather than
|
|
237
|
+
#: offered. Both oracle studies name the same standing gap in the model's own
|
|
238
|
+
#: words -- per-parameter weights cannot express the interaction structure the
|
|
239
|
+
#: front is built out of -- at EVERY checkpoint of both, and three times they
|
|
240
|
+
#: name these proposals as its only carrier. Offered, the clause went
|
|
241
|
+
#: unanswered: zero proposals across roughly thirty analog calls, by the live
|
|
242
|
+
#: model and by the hindsight oracle alike (every admitted checkpoint of both
|
|
243
|
+
#: studies reports an immigrant count of 0), while the same clause on the
|
|
244
|
+
#: six-field NAS venue was sometimes answered. Optionality, not capability,
|
|
245
|
+
#: was suppressing it -- so the clause states a count, and the harness meters
|
|
246
|
+
#: the answer instead of refusing over it.
|
|
247
|
+
#:
|
|
248
|
+
#: The novelty half is the SECOND thing the live measurement forced. Required,
|
|
249
|
+
#: the clause was answered on schedule -- twelve proposals per cell -- and
|
|
250
|
+
#: accepted 0 of 69: at roughly 320 measured rows, a recombination of the
|
|
251
|
+
#: elites the occupancy table shows is usually a configuration the run has
|
|
252
|
+
#: already charged, and the dedup drops it. "Never repeats" was already in the
|
|
253
|
+
#: prose; what was missing was the GROUND for it, because the model cannot
|
|
254
|
+
#: count rows it was shown a digest of. So the clause now states how many
|
|
255
|
+
#: configurations the run has measured and what a repeat costs. Nothing about
|
|
256
|
+
#: admission moved: a repeat is still dropped, and the drop is still counted.
|
|
257
|
+
IMMIGRANTS_CLAUSE = """
|
|
258
|
+
Your reply MUST also carry an "immigrants" key holding EXACTLY {m} COMPLETE
|
|
259
|
+
configurations worth measuring next -- recombinations or refinements of what
|
|
260
|
+
the occupancy table says the front rewards, never repeats of configurations
|
|
261
|
+
the run has already measured. A per-parameter table cannot say which values
|
|
262
|
+
belong TOGETHER; these {m} are where you say it. Every parameter present,
|
|
263
|
+
every value from that parameter's declared domain:
|
|
264
|
+
|
|
265
|
+
{{"immigrants": [{{"<parameter>": <value>, ...}}]}}
|
|
266
|
+
|
|
267
|
+
NOVELTY IS THE POINT: this run has ALREADY MEASURED {measured}
|
|
268
|
+
configurations, and the evidence above is drawn from them. A proposal that
|
|
269
|
+
repeats one of those {measured} is REJECTED without being measured and WASTES
|
|
270
|
+
the slot it took. Every one of the {m} must differ from every configuration
|
|
271
|
+
this run has measured, in at least one parameter -- recombine what the front
|
|
272
|
+
rewards into a joint setting the trace does not already contain.
|
|
273
|
+
"""
|
|
274
|
+
|
|
275
|
+
#: How many parameters one reply is ASKED to name in ``"weights"``. Not a
|
|
276
|
+
#: refusal threshold and deliberately not one: the harness counts breadth
|
|
277
|
+
#: (``tilt_breadth``) and never throttles it, because a second refusal mode is
|
|
278
|
+
#: what v3 exists to remove. The number is the oracle's own: over the five
|
|
279
|
+
#: usable hindsight checkpoints of the two studies it tilted 2 to 8 focused
|
|
280
|
+
#: parameters, where the live replies tilted or freed all 24 fields of the
|
|
281
|
+
#: analog venue at once -- a breadth that says nothing a uniform table does
|
|
282
|
+
#: not.
|
|
283
|
+
TILT_CAP = 4
|
|
284
|
+
|
|
285
|
+
#: The heading the elite-occupancy half of the evidence bundle carries. It is
|
|
286
|
+
#: a constant because the immigrants clause and the analysis tooling both name
|
|
287
|
+
#: the section, and a heading two places quote is a heading worth declaring.
|
|
288
|
+
ELITE_TABLE_TITLE = "WHAT THE FRONT IS BUILT OUT OF"
|
|
289
|
+
|
|
290
|
+
#: Which evidence bundle an event was conditioned on, journalled on every
|
|
291
|
+
#: event. ``"v2"`` is the measured trace PLUS the elite-occupancy table; the
|
|
292
|
+
#: unversioned bundle before it was the trace alone. A study that pools events
|
|
293
|
+
#: across the change would otherwise be pooling two different prompts.
|
|
294
|
+
EVIDENCE_VERSION = "v2"
|
|
295
|
+
|
|
296
|
+
#: Which MECHANISM authored an event, journalled beside the evidence version
|
|
297
|
+
#: so a cell self-identifies without its campaign's paperwork. ``"v3"`` is
|
|
298
|
+
#: silence-keeps-mass admission, required-k joint proposals and the
|
|
299
|
+
#: focused-tilt ask; ``"v2"`` before it refused a subset reply whole, offered
|
|
300
|
+
#: the proposals and asked for no focus. The two markers move INDEPENDENTLY:
|
|
301
|
+
#: v3 changed what the harness asks for and what it admits, not what it shows,
|
|
302
|
+
#: so the evidence version stays where it was and a study may pool bundles
|
|
303
|
+
#: across the mechanism change while refusing to pool the mechanisms.
|
|
304
|
+
MECHANISM_VERSION = "v3"
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
class Reguidance:
|
|
308
|
+
"""The revision channel: one call per checkpoint, damped into the prior.
|
|
309
|
+
|
|
310
|
+
Constructed complete -- the completion callable, the objectives, the
|
|
311
|
+
schema, the cadence -- so a run states what it bought instead of
|
|
312
|
+
assembling it from defaults at three call sites. It is a pure consumer of
|
|
313
|
+
``complete``: it never sees the problem, the evaluator, the cache or the
|
|
314
|
+
budget, so a revision cannot spend one.
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
def __init__(
|
|
318
|
+
self,
|
|
319
|
+
complete: Callable[[str], str],
|
|
320
|
+
*,
|
|
321
|
+
objectives: Sequence[ObjectiveSpec],
|
|
322
|
+
candidate_model: Any,
|
|
323
|
+
template: Mapping[str, Any],
|
|
324
|
+
domain_context: str = "",
|
|
325
|
+
every: int,
|
|
326
|
+
max_events: int = 4,
|
|
327
|
+
immigrants: int = 0,
|
|
328
|
+
damping: float = 0.5,
|
|
329
|
+
max_weight_ratio: float = 8.0,
|
|
330
|
+
evidence_view: Optional[Callable[
|
|
331
|
+
[Sequence[MeasuredRow]], Sequence[MeasuredRow]]] = None,
|
|
332
|
+
gate_reads_view: bool = False,
|
|
333
|
+
telemetry: Optional[ReguidanceTelemetry] = None,
|
|
334
|
+
min_rows: int = MIN_EVIDENCE_ROWS,
|
|
335
|
+
front_shown: int = 8,
|
|
336
|
+
effects_shown: int = 8,
|
|
337
|
+
) -> None:
|
|
338
|
+
if int(every) <= 0:
|
|
339
|
+
raise ValueError(
|
|
340
|
+
"reguidance fires on a declared cadence in charged "
|
|
341
|
+
f"evaluations and needs a positive one, got {every!r}")
|
|
342
|
+
if int(max_events) < 0 or int(immigrants) < 0 or int(min_rows) < 0:
|
|
343
|
+
raise ValueError(
|
|
344
|
+
"reguidance max_events, immigrants and min_rows are counts "
|
|
345
|
+
f"and must be non-negative, got {max_events!r}, "
|
|
346
|
+
f"{immigrants!r}, {min_rows!r}")
|
|
347
|
+
if not 0.0 <= float(damping) <= 1.0:
|
|
348
|
+
raise ValueError(
|
|
349
|
+
"reguidance damping mixes the reply into the installed "
|
|
350
|
+
f"weights and must lie in [0, 1], got {damping!r}")
|
|
351
|
+
if float(max_weight_ratio) < 1.0:
|
|
352
|
+
raise ValueError(
|
|
353
|
+
"reguidance max_weight_ratio caps how far one parameter may "
|
|
354
|
+
f"concentrate and must be at least 1, got {max_weight_ratio!r}")
|
|
355
|
+
|
|
356
|
+
self.complete = complete
|
|
357
|
+
self.objectives = tuple(objectives)
|
|
358
|
+
self.candidate_model = candidate_model
|
|
359
|
+
self.template = dict(template or {})
|
|
360
|
+
self.domain_context = domain_context
|
|
361
|
+
self.every = int(every)
|
|
362
|
+
self.max_events = int(max_events)
|
|
363
|
+
self.immigrants = int(immigrants)
|
|
364
|
+
self.damping = float(damping)
|
|
365
|
+
self.max_weight_ratio = float(max_weight_ratio)
|
|
366
|
+
self.evidence_view = evidence_view
|
|
367
|
+
#: Which rows the zero-on-front admission check reads. False -- the
|
|
368
|
+
#: default, and the only setting a shipped run should use -- reads the
|
|
369
|
+
#: rows this run really measured. True reads the VIEWED rows instead,
|
|
370
|
+
#: which is what makes a shuffled-evidence CONTROL's refusal rate
|
|
371
|
+
#: comparable to the arm it controls. See the module docstring.
|
|
372
|
+
self.gate_reads_view = bool(gate_reads_view)
|
|
373
|
+
self.min_rows = int(min_rows)
|
|
374
|
+
self.front_shown = int(front_shown)
|
|
375
|
+
self.effects_shown = int(effects_shown)
|
|
376
|
+
|
|
377
|
+
# The harvest contract (core.telemetry): counters, a name, an author.
|
|
378
|
+
self.telemetry = telemetry if telemetry is not None else ReguidanceTelemetry()
|
|
379
|
+
self.mechanism = "reguidance"
|
|
380
|
+
self.authored_by = "llm"
|
|
381
|
+
|
|
382
|
+
#: Per-FIELD vocabulary. Weights are keyed by ``Locus.field`` because
|
|
383
|
+
#: that is the key the sampler consults, so a sequence's elements
|
|
384
|
+
#: share one entry -- as they do in every other prior here.
|
|
385
|
+
self._field_domains: Dict[str, Tuple[Any, ...]] = _field_domains(
|
|
386
|
+
self.candidate_model, self.template)
|
|
387
|
+
#: Per-LOCUS domains, which is what the evidence renderer keys on.
|
|
388
|
+
self._locus_domains: Dict[str, Tuple[Any, ...]] = _locus_domains(
|
|
389
|
+
self.candidate_model, self.template)
|
|
390
|
+
|
|
391
|
+
self._installed: Overlay = {}
|
|
392
|
+
self._seeded = False
|
|
393
|
+
self._events_fired = 0
|
|
394
|
+
self._next_at = self.every
|
|
395
|
+
self._rows_at_last_event = 0
|
|
396
|
+
#: The outstanding bet: the weights an event replaced and the row
|
|
397
|
+
#: count at the time. ``None`` whenever no installed revision is
|
|
398
|
+
#: waiting to be judged.
|
|
399
|
+
self._pending: Optional[Dict[str, Any]] = None
|
|
400
|
+
|
|
401
|
+
# -- what the loop sees --------------------------------------------------
|
|
402
|
+
|
|
403
|
+
@property
|
|
404
|
+
def installed(self) -> Overlay:
|
|
405
|
+
"""The graded overlay this policy currently holds. A copy."""
|
|
406
|
+
|
|
407
|
+
return {k: (tuple(v[0]), tuple(v[1])) for k, v in self._installed.items()}
|
|
408
|
+
|
|
409
|
+
def maybe_revise(
|
|
410
|
+
self,
|
|
411
|
+
rows: Sequence[Tuple[Mapping[str, Any], Mapping[str, float]]],
|
|
412
|
+
population: Sequence[Tuple[Mapping[str, Any], Mapping[str, float]]],
|
|
413
|
+
specs: Sequence[ObjectiveSpec],
|
|
414
|
+
restriction: Any,
|
|
415
|
+
charges: int,
|
|
416
|
+
gen: int,
|
|
417
|
+
) -> ReguidanceOutcome:
|
|
418
|
+
"""One checkpoint. Cheap and silent unless the cadence says otherwise."""
|
|
419
|
+
|
|
420
|
+
measured = [(dict(config), dict(objectives)) for config, objectives in rows]
|
|
421
|
+
if not self._field_domains:
|
|
422
|
+
# Nothing the schema declares finitely: there is no weight table
|
|
423
|
+
# to revise, so no call is worth its cost.
|
|
424
|
+
return ReguidanceOutcome()
|
|
425
|
+
if not self._due(int(charges), len(measured)):
|
|
426
|
+
return ReguidanceOutcome()
|
|
427
|
+
specs = list(specs or self.objectives)
|
|
428
|
+
|
|
429
|
+
self._seed_from(restriction)
|
|
430
|
+
note: Dict[str, Any] = {
|
|
431
|
+
"gen": int(gen),
|
|
432
|
+
"at_charges": int(charges),
|
|
433
|
+
"rows": len(measured),
|
|
434
|
+
}
|
|
435
|
+
changed = self._revert_if_the_bet_lost(measured, specs, note)
|
|
436
|
+
|
|
437
|
+
self._events_fired += 1
|
|
438
|
+
self._rows_at_last_event = len(measured)
|
|
439
|
+
self._next_at = int(charges) + self.every
|
|
440
|
+
return self._revise(measured, population, specs, note, changed)
|
|
441
|
+
|
|
442
|
+
# -- cadence -------------------------------------------------------------
|
|
443
|
+
|
|
444
|
+
def _due(self, charges: int, rows: int) -> bool:
|
|
445
|
+
"""The declared rule, and only it.
|
|
446
|
+
|
|
447
|
+
Three conditions, each declared rather than inherited: the cadence in
|
|
448
|
+
CHARGED evaluations (what the campaign paid for), enough NEW measured
|
|
449
|
+
rows since the last event for the evidence to say anything the last
|
|
450
|
+
rendering did not, and the event cap. A stall trigger would be a
|
|
451
|
+
fourth clock and is deliberately not built here.
|
|
452
|
+
"""
|
|
453
|
+
|
|
454
|
+
if self._events_fired >= self.max_events:
|
|
455
|
+
return False
|
|
456
|
+
if charges < self._next_at:
|
|
457
|
+
return False
|
|
458
|
+
if rows <= 0 or rows - self._rows_at_last_event < self.min_rows:
|
|
459
|
+
return False
|
|
460
|
+
return True
|
|
461
|
+
|
|
462
|
+
# -- state ---------------------------------------------------------------
|
|
463
|
+
|
|
464
|
+
def _seed_from(self, restriction: Any) -> None:
|
|
465
|
+
"""Adopt whatever prior the loop already holds, once.
|
|
466
|
+
|
|
467
|
+
A hard restriction is the 0/1 special case of the graded form, so it
|
|
468
|
+
seeds as such. Its exclusions persist through any revision that stays
|
|
469
|
+
SILENT about them (damping keeps an untouched zero at zero) and regain
|
|
470
|
+
mass exactly when a reply weights them -- ``_damp`` states why that
|
|
471
|
+
direction is the deliberate one. ``None`` seeds the uniform table.
|
|
472
|
+
"""
|
|
473
|
+
|
|
474
|
+
if self._seeded:
|
|
475
|
+
return
|
|
476
|
+
self._seeded = True
|
|
477
|
+
weighted = getattr(restriction, "weighted", None)
|
|
478
|
+
if weighted:
|
|
479
|
+
self._installed = {str(name): (tuple(values), tuple(float(w) for w in weights))
|
|
480
|
+
for name, (values, weights) in dict(weighted).items()}
|
|
481
|
+
return
|
|
482
|
+
allowed = getattr(restriction, "allowed", None)
|
|
483
|
+
if allowed:
|
|
484
|
+
hard = WeightedRestriction.hard(dict(allowed))
|
|
485
|
+
self._installed = {str(name): (tuple(values), tuple(float(w) for w in weights))
|
|
486
|
+
for name, (values, weights) in dict(hard.weighted).items()}
|
|
487
|
+
|
|
488
|
+
def _revert_if_the_bet_lost(
|
|
489
|
+
self,
|
|
490
|
+
rows: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
491
|
+
specs: Sequence[ObjectiveSpec],
|
|
492
|
+
note: Dict[str, Any],
|
|
493
|
+
) -> bool:
|
|
494
|
+
"""Undo the last revision unless its window IMPROVED the front.
|
|
495
|
+
|
|
496
|
+
The claim a revision makes is narrow and therefore checkable: draws
|
|
497
|
+
from the tilted prior are worth more than draws from the one it
|
|
498
|
+
replaced. The first reading of "worth more" -- some post-event row is
|
|
499
|
+
rank-0 in the pooled rows -- was measured impotent on the first live
|
|
500
|
+
venue it met: across six revision-carrying runs on a three-objective
|
|
501
|
+
simulator it admitted 23 revisions and reverted 0, because on three
|
|
502
|
+
objectives almost every fresh point is non-dominated, and a run whose
|
|
503
|
+
revisions had locked it flat for 120 charges kept every one of them
|
|
504
|
+
(W1 pilot, seed 20370102). The bet is now the loop's own unwind
|
|
505
|
+
semantics: the revision stands only if some row measured after the
|
|
506
|
+
event STRICTLY DOMINATES a member of the pre-event front -- the
|
|
507
|
+
tilted prior must move the front, not merely land beside it.
|
|
508
|
+
"""
|
|
509
|
+
|
|
510
|
+
pending = self._pending
|
|
511
|
+
if pending is None:
|
|
512
|
+
return False
|
|
513
|
+
self._pending = None
|
|
514
|
+
cut = int(pending["rows_at_event"])
|
|
515
|
+
before = [dict(row[1]) for row in rows[:cut]]
|
|
516
|
+
front_before = [before[index]
|
|
517
|
+
for index in _front_indices(rows[:cut], specs)]
|
|
518
|
+
oriented = list(specs)
|
|
519
|
+
if any(dominates(dict(objectives), member, oriented)
|
|
520
|
+
for _config, objectives in rows[cut:]
|
|
521
|
+
for member in front_before):
|
|
522
|
+
return False
|
|
523
|
+
self._installed = {k: (tuple(v[0]), tuple(v[1]))
|
|
524
|
+
for k, v in dict(pending["weights"]).items()}
|
|
525
|
+
self.telemetry.revisions_reverted += 1
|
|
526
|
+
note["reverted"] = {"rows_at_event": cut,
|
|
527
|
+
"fields": sorted(self._installed)}
|
|
528
|
+
return True
|
|
529
|
+
|
|
530
|
+
# -- the call ------------------------------------------------------------
|
|
531
|
+
|
|
532
|
+
def _revise(
|
|
533
|
+
self,
|
|
534
|
+
rows: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
535
|
+
population: Sequence[Tuple[Mapping[str, Any], Mapping[str, float]]],
|
|
536
|
+
specs: Sequence[ObjectiveSpec],
|
|
537
|
+
note: Dict[str, Any],
|
|
538
|
+
changed: bool,
|
|
539
|
+
) -> ReguidanceOutcome:
|
|
540
|
+
view_rows = self._evidence_rows(rows, population)
|
|
541
|
+
evidence = self._evidence_bundle(view_rows, specs,
|
|
542
|
+
int(note["at_charges"]))
|
|
543
|
+
note["rows_shown"] = len(view_rows)
|
|
544
|
+
note["evidence"] = EVIDENCE_VERSION
|
|
545
|
+
note["mechanism"] = MECHANISM_VERSION
|
|
546
|
+
note["evidence_sha256"] = evidence_digest(evidence)
|
|
547
|
+
# The rendering itself, not a recipe for reconstructing it. The oracle
|
|
548
|
+
# instrument proved a late-checkpoint prompt UNRECONSTRUCTIBLE from the
|
|
549
|
+
# cells it was journalled beside: the row list a checkpoint reads
|
|
550
|
+
# includes cache-served repeats, and the charge log -- which counts
|
|
551
|
+
# charged evaluations -- cannot recover them. A digest can only falsify
|
|
552
|
+
# a reconstruction; the text makes the study exact.
|
|
553
|
+
note["evidence_text"] = evidence
|
|
554
|
+
|
|
555
|
+
prompt = self._prompt(evidence, note)
|
|
556
|
+
self.telemetry.calls += 1
|
|
557
|
+
try:
|
|
558
|
+
reply = self.complete(prompt)
|
|
559
|
+
except Exception as exc: # a policy must never kill a run
|
|
560
|
+
self.telemetry.errors += 1
|
|
561
|
+
note["error"] = f"{type(exc).__name__}: {exc}"
|
|
562
|
+
self.telemetry.events.append(note)
|
|
563
|
+
return self._outcome(note, changed)
|
|
564
|
+
|
|
565
|
+
# WHICH front the admission check protects. Reality by default: a value
|
|
566
|
+
# this run measured onto its own front keeps its mass however the
|
|
567
|
+
# prompt was composed. A control arm hands the gate the same rows it
|
|
568
|
+
# prompted with, so the two arms refuse for the same reasons.
|
|
569
|
+
gate_rows = rows
|
|
570
|
+
if self.gate_reads_view:
|
|
571
|
+
gate_rows = [(dict(row[0]), dict(row[1])) for row in view_rows]
|
|
572
|
+
note["gate_reads_view"] = True
|
|
573
|
+
# Breadth is METERED, not gated: it is read off every reply that came
|
|
574
|
+
# back, whatever the verdict, so a campaign's median tilt is taken
|
|
575
|
+
# over the replies the model wrote rather than over the subset the
|
|
576
|
+
# admission rule happened to keep.
|
|
577
|
+
breadth = _weights_breadth(reply)
|
|
578
|
+
note["tilt_breadth"] = breadth
|
|
579
|
+
self.telemetry.breadth_total += breadth
|
|
580
|
+
|
|
581
|
+
parsed, refusal = self._parse(reply, gate_rows, specs)
|
|
582
|
+
if parsed is None:
|
|
583
|
+
self.telemetry.revisions_refused += 1
|
|
584
|
+
note["refused"] = refusal
|
|
585
|
+
self.telemetry.events.append(note)
|
|
586
|
+
return self._outcome(note, changed)
|
|
587
|
+
|
|
588
|
+
weights, free, raw_immigrants = parsed
|
|
589
|
+
before = self.installed
|
|
590
|
+
mixed = self._damp(weights, free)
|
|
591
|
+
self._installed = mixed
|
|
592
|
+
self.telemetry.revisions_admitted += 1
|
|
593
|
+
self._pending = {"rows_at_event": len(rows), "weights": before}
|
|
594
|
+
note["admitted"] = True
|
|
595
|
+
note["damped_fields"] = sorted(mixed)
|
|
596
|
+
note["proposed_fields"] = sorted(weights)
|
|
597
|
+
note["freed_fields"] = sorted(free)
|
|
598
|
+
|
|
599
|
+
immigrants = self._immigrants(raw_immigrants, rows, note)
|
|
600
|
+
self.telemetry.events.append(note)
|
|
601
|
+
return self._outcome(note, True, immigrants)
|
|
602
|
+
|
|
603
|
+
def _outcome(self, note: Dict[str, Any], changed: bool,
|
|
604
|
+
immigrants: Tuple[Config, ...] = ()) -> ReguidanceOutcome:
|
|
605
|
+
"""``restriction=None`` means keep; anything else is what to install."""
|
|
606
|
+
|
|
607
|
+
restriction = WeightedRestriction(self.installed) if changed else None
|
|
608
|
+
return ReguidanceOutcome(restriction=restriction,
|
|
609
|
+
immigrants=tuple(immigrants), note=note)
|
|
610
|
+
|
|
611
|
+
def _evidence_rows(
|
|
612
|
+
self,
|
|
613
|
+
rows: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
614
|
+
population: Sequence[Tuple[Mapping[str, Any], Mapping[str, float]]],
|
|
615
|
+
) -> List[MeasuredRow]:
|
|
616
|
+
"""The rows the model is shown: identity, unless a view is declared."""
|
|
617
|
+
|
|
618
|
+
surviving = {_key(dict(config)) for config, _objectives in population}
|
|
619
|
+
measured: Sequence[MeasuredRow] = [
|
|
620
|
+
(config, objectives, _key(config) in surviving)
|
|
621
|
+
for config, objectives in rows
|
|
622
|
+
]
|
|
623
|
+
if self.evidence_view is not None:
|
|
624
|
+
try:
|
|
625
|
+
measured = self.evidence_view(measured)
|
|
626
|
+
except Exception: # a control that throws must not
|
|
627
|
+
measured = () # be able to kill a measurement
|
|
628
|
+
return [tuple(row) for row in measured] # type: ignore[misc]
|
|
629
|
+
|
|
630
|
+
def _evidence_bundle(
|
|
631
|
+
self,
|
|
632
|
+
view_rows: Sequence[MeasuredRow],
|
|
633
|
+
specs: Sequence[ObjectiveSpec],
|
|
634
|
+
charged: int,
|
|
635
|
+
) -> str:
|
|
636
|
+
"""The v2 bundle: the measured trace, then what the front is made of.
|
|
637
|
+
|
|
638
|
+
Both halves read the SAME rows -- the ones ``evidence_view`` returned
|
|
639
|
+
-- so the control parameter transforms the whole of what the model
|
|
640
|
+
sees rather than half of it, and one digest over the concatenation is
|
|
641
|
+
the identity of the whole prompt's evidence. Which is why the digest is
|
|
642
|
+
taken here, over the bundle, and not per section: a bundle whose halves
|
|
643
|
+
were separately digested could report "same evidence" while one half
|
|
644
|
+
had moved.
|
|
645
|
+
"""
|
|
646
|
+
|
|
647
|
+
measured = render_measurement_evidence(
|
|
648
|
+
view_rows, list(specs), self._locus_domains,
|
|
649
|
+
front_shown=self.front_shown, effects_shown=self.effects_shown,
|
|
650
|
+
charged=int(charged))
|
|
651
|
+
elite = render_elite_table(
|
|
652
|
+
view_rows, list(specs), self._locus_domains)
|
|
653
|
+
return f"{measured}\n\n {ELITE_TABLE_TITLE}:\n{elite}"
|
|
654
|
+
|
|
655
|
+
def _prompt(self, evidence: str, note: Mapping[str, Any]) -> str:
|
|
656
|
+
# The novelty ground is the run's OWN row count -- the same number the
|
|
657
|
+
# prompt states above the evidence -- because that is the count the
|
|
658
|
+
# dedup the proposals will meet actually holds. A view arm changes
|
|
659
|
+
# which rows are RENDERED, never how many the run has measured, so the
|
|
660
|
+
# two arms are asked for novelty against the same standard.
|
|
661
|
+
clause = ("" if self.immigrants <= 0
|
|
662
|
+
else IMMIGRANTS_CLAUSE.format(m=self.immigrants,
|
|
663
|
+
measured=int(note["rows"])))
|
|
664
|
+
return PROMPT.format(
|
|
665
|
+
context=self.domain_context.strip(),
|
|
666
|
+
goals="\n".join(f" {s.name}: {s.goal}imise" for s in self.objectives),
|
|
667
|
+
domains="\n".join(
|
|
668
|
+
f" {name}: {json.dumps(list(values), default=str)}"
|
|
669
|
+
+ self._shared_note(name)
|
|
670
|
+
for name, values in sorted(self._field_domains.items())),
|
|
671
|
+
weights=self._render_weights(),
|
|
672
|
+
rows=int(note["rows"]),
|
|
673
|
+
charges=int(note["at_charges"]),
|
|
674
|
+
evidence=evidence,
|
|
675
|
+
immigrants=clause,
|
|
676
|
+
tilt_cap=TILT_CAP,
|
|
677
|
+
max_ratio=f"{self.max_weight_ratio:g}",
|
|
678
|
+
damping=self.damping,
|
|
679
|
+
)
|
|
680
|
+
|
|
681
|
+
def _shared_note(self, name: str) -> str:
|
|
682
|
+
"""Say where a sequence parameter's one weight table applies.
|
|
683
|
+
|
|
684
|
+
The evidence names positions (``genome[3]``) because a correlation is
|
|
685
|
+
per position; the weight table is per PARAMETER, because that is the
|
|
686
|
+
key the sampler consults and because a table per position would be a
|
|
687
|
+
different prior for every genome length. Both facts are in the prompt,
|
|
688
|
+
so the difference cannot read as a contradiction.
|
|
689
|
+
"""
|
|
690
|
+
|
|
691
|
+
positions = [str(locus) for locus in loci_of(self.template)
|
|
692
|
+
if locus.field == name and locus.index is not None]
|
|
693
|
+
if len(positions) < 2:
|
|
694
|
+
return ""
|
|
695
|
+
return (f" (one weight table, used at every position: "
|
|
696
|
+
f"{positions[0]} .. {positions[-1]})")
|
|
697
|
+
|
|
698
|
+
def _render_weights(self) -> str:
|
|
699
|
+
if not self._installed:
|
|
700
|
+
return " (none installed: every parameter is sampled uniformly)"
|
|
701
|
+
lines = []
|
|
702
|
+
for name in sorted(self._installed):
|
|
703
|
+
values, weights = self._installed[name]
|
|
704
|
+
body = ", ".join(f"{json.dumps(v, default=str)}={float(w):.4g}"
|
|
705
|
+
for v, w in zip(values, weights))
|
|
706
|
+
lines.append(f" {name}: {body}")
|
|
707
|
+
absent = sorted(set(self._field_domains) - set(self._installed))
|
|
708
|
+
if absent:
|
|
709
|
+
lines.append(f" (uniform, no entry: {', '.join(absent)})")
|
|
710
|
+
return "\n".join(lines)
|
|
711
|
+
|
|
712
|
+
# -- parse and admission: whole-reply, never repaired --------------------
|
|
713
|
+
|
|
714
|
+
def _parse(
|
|
715
|
+
self,
|
|
716
|
+
reply: Any,
|
|
717
|
+
rows: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
718
|
+
specs: Sequence[ObjectiveSpec],
|
|
719
|
+
) -> Tuple[Optional[Tuple[Dict[str, List[Tuple[Any, float]]],
|
|
720
|
+
List[str], List[Any]]], str]:
|
|
721
|
+
"""The reply, judged whole, in the taxonomy the hard gate established.
|
|
722
|
+
|
|
723
|
+
The reasons are the ones
|
|
724
|
+
:func:`~agent_evolve.policies.measurement_evidence.admit_weighted_restriction`
|
|
725
|
+
refuses on, plus the two the GRADED form adds: an all-zero field (a
|
|
726
|
+
restriction that samples nothing) and ``excludes_front`` -- an EXPLICIT
|
|
727
|
+
zero weight on a value some rank-0 configuration holds, which is the
|
|
728
|
+
only way a reply can take mass off the measured front. A value the
|
|
729
|
+
reply simply omits is damped, not excluded, and is admitted; the module
|
|
730
|
+
docstring records what reading that omission as a zero cost. Nothing is
|
|
731
|
+
repaired anywhere in here: a repaired prior is the harness's prior
|
|
732
|
+
wearing the model's name.
|
|
733
|
+
"""
|
|
734
|
+
|
|
735
|
+
raw = _json_object(reply)
|
|
736
|
+
if raw is None:
|
|
737
|
+
return None, "unparsed"
|
|
738
|
+
entries = raw.get("weights")
|
|
739
|
+
if entries is None:
|
|
740
|
+
entries = {}
|
|
741
|
+
free_raw = raw.get("free")
|
|
742
|
+
if free_raw is None:
|
|
743
|
+
free_raw = []
|
|
744
|
+
if not isinstance(entries, dict) or not isinstance(free_raw, list):
|
|
745
|
+
return None, "unparsed"
|
|
746
|
+
# Every parameter mapped to a bare value is a CONFIGURATION, not a
|
|
747
|
+
# prior -- the failure mode that collapses this channel into artifact
|
|
748
|
+
# authoring, and the reason the weighted proposer checks for it too.
|
|
749
|
+
if entries and all(not isinstance(v, dict) for v in entries.values()):
|
|
750
|
+
return None, "wrote_candidate"
|
|
751
|
+
|
|
752
|
+
front = {index for index in _front_indices(rows, specs)}
|
|
753
|
+
front_values: Dict[str, List[Any]] = {}
|
|
754
|
+
for index in front:
|
|
755
|
+
for name, values in self._field_values(rows[index][0]).items():
|
|
756
|
+
for value in values:
|
|
757
|
+
if value not in front_values.setdefault(name, []):
|
|
758
|
+
front_values[name].append(value)
|
|
759
|
+
|
|
760
|
+
weights: Dict[str, List[Tuple[Any, float]]] = {}
|
|
761
|
+
for name, entry in entries.items():
|
|
762
|
+
name = str(name)
|
|
763
|
+
if name not in self._field_domains:
|
|
764
|
+
return None, f"undeclared parameter {name!r}"
|
|
765
|
+
domain = self._field_domains[name]
|
|
766
|
+
if not isinstance(entry, dict):
|
|
767
|
+
return None, f"malformed entry for {name!r}"
|
|
768
|
+
values = entry.get("values")
|
|
769
|
+
listed = entry.get("weights")
|
|
770
|
+
if (not isinstance(values, list) or not isinstance(listed, list)
|
|
771
|
+
or not values or len(values) != len(listed)):
|
|
772
|
+
return None, f"malformed entry for {name!r}"
|
|
773
|
+
clean: List[Tuple[Any, float]] = []
|
|
774
|
+
for value, weight in zip(values, listed):
|
|
775
|
+
if value not in domain:
|
|
776
|
+
return None, f"undeclared value for {name!r}"
|
|
777
|
+
if (isinstance(weight, bool)
|
|
778
|
+
or not isinstance(weight, (int, float))
|
|
779
|
+
or not math.isfinite(float(weight))
|
|
780
|
+
or float(weight) < 0.0):
|
|
781
|
+
return None, f"invalid_weight for {name!r}"
|
|
782
|
+
clean.append((value, float(weight)))
|
|
783
|
+
positive = [w for _v, w in clean if w > 0.0]
|
|
784
|
+
if not positive:
|
|
785
|
+
return None, f"all_zero for {name!r}"
|
|
786
|
+
ratio = max(positive) / min(positive)
|
|
787
|
+
if ratio > self.max_weight_ratio:
|
|
788
|
+
return None, (f"over_concentrated ({ratio:.3g}x > "
|
|
789
|
+
f"{self.max_weight_ratio:g}x) for {name!r}")
|
|
790
|
+
# Only a zero the reply WROTE. Silence about a value is not a
|
|
791
|
+
# zero -- damping leaves an unlisted value ``(1 - a)`` of the mass
|
|
792
|
+
# it holds -- so a subset reply excludes nothing and is admitted.
|
|
793
|
+
# At ``a == 1`` there is no mixture and an omission really does
|
|
794
|
+
# become a zero, so the gate carries the whole guarantee again and
|
|
795
|
+
# reads silence the way the installed weights will.
|
|
796
|
+
held = {_token(v): w for v, w in clean}
|
|
797
|
+
unlisted = 0.0 if self.damping >= 1.0 else None
|
|
798
|
+
for value in front_values.get(name, ()):
|
|
799
|
+
mass = held.get(_token(value), unlisted)
|
|
800
|
+
if mass is not None and mass <= 0.0:
|
|
801
|
+
return None, f"excludes_front for {name!r}"
|
|
802
|
+
weights[name] = clean
|
|
803
|
+
|
|
804
|
+
free: List[str] = []
|
|
805
|
+
for name in free_raw:
|
|
806
|
+
name = str(name)
|
|
807
|
+
if name not in self._field_domains:
|
|
808
|
+
return None, f"undeclared parameter {name!r}"
|
|
809
|
+
free.append(name)
|
|
810
|
+
|
|
811
|
+
if not weights and not free:
|
|
812
|
+
return None, "empty"
|
|
813
|
+
|
|
814
|
+
# The required-k clause is NOT enforced here, and that is the design:
|
|
815
|
+
# the two halves of a reply are judged separately, so a model that
|
|
816
|
+
# under-answers the joint-proposal ask does not also lose the prior it
|
|
817
|
+
# got right. ``_immigrants`` counts the shortfall.
|
|
818
|
+
immigrants = raw.get("immigrants")
|
|
819
|
+
if not isinstance(immigrants, list):
|
|
820
|
+
immigrants = []
|
|
821
|
+
return (weights, free, immigrants), "admitted"
|
|
822
|
+
|
|
823
|
+
def _field_values(self, config: Mapping[str, Any]) -> Dict[str, List[Any]]:
|
|
824
|
+
"""Which declared values a configuration holds, per FIELD.
|
|
825
|
+
|
|
826
|
+
A sequence field holds one value per element and they share the
|
|
827
|
+
field's entry, so a front member pins every value it uses anywhere in
|
|
828
|
+
that field.
|
|
829
|
+
"""
|
|
830
|
+
|
|
831
|
+
out: Dict[str, List[Any]] = {}
|
|
832
|
+
for locus in loci_of(dict(config)):
|
|
833
|
+
if locus.field not in self._field_domains:
|
|
834
|
+
continue
|
|
835
|
+
try:
|
|
836
|
+
value = read_locus(config, locus)
|
|
837
|
+
except Exception:
|
|
838
|
+
continue
|
|
839
|
+
out.setdefault(locus.field, []).append(value)
|
|
840
|
+
return out
|
|
841
|
+
|
|
842
|
+
# -- damping -------------------------------------------------------------
|
|
843
|
+
|
|
844
|
+
def _damp(self, proposal: Mapping[str, Sequence[Tuple[Any, float]]],
|
|
845
|
+
free: Sequence[str]) -> Overlay:
|
|
846
|
+
"""Mix the admitted reply into the installed weights. Two properties.
|
|
847
|
+
|
|
848
|
+
1. With ``damping < 1`` no exclusion can be INTRODUCED: every value
|
|
849
|
+
whose installed weight is positive keeps a positive mixed weight,
|
|
850
|
+
whatever the reply says about it. A revision is therefore a tilt,
|
|
851
|
+
and the worst case of a wrong one is bounded by the declared
|
|
852
|
+
domains rather than by a gate.
|
|
853
|
+
2. The concentration cap survives mixing. For positive vectors ``b``
|
|
854
|
+
and ``p`` with ``max/min <= r`` each, every mixed entry satisfies
|
|
855
|
+
``min(b_i, p_i) * (something) <=`` ... concretely, the mediant
|
|
856
|
+
inequality gives ``max_i((1-a)b_i + a*p_i) / min_i((1-a)b_i +
|
|
857
|
+
a*p_i) <= r``, so admitting under ``max_weight_ratio`` bounds the
|
|
858
|
+
INSTALLED ratio at that value for the whole run, however many
|
|
859
|
+
revisions land. (A base carrying zeros -- a hard restriction seeded
|
|
860
|
+
in -- is not an ``r``-ratio vector; the bound is over the support
|
|
861
|
+
the two share, which is where the cap has meaning.)
|
|
862
|
+
|
|
863
|
+
A field the reply neither weights nor frees is left exactly as it is:
|
|
864
|
+
silence about a parameter is not evidence about it. A VALUE the reply
|
|
865
|
+
omits from a field it does weight is the same shape one level down --
|
|
866
|
+
it keeps ``(1 - a)`` of its share rather than being zeroed -- which is
|
|
867
|
+
why the admission gate can afford to refuse written zeros only.
|
|
868
|
+
|
|
869
|
+
The mixture is directional in one place only: a value the BASE
|
|
870
|
+
excludes -- a hard restriction seeded in at the first event -- regains
|
|
871
|
+
mass when the reply weights it. That is the correction the unwind
|
|
872
|
+
machinery can only make by dropping the whole prior, it is bounded by
|
|
873
|
+
the same cap and the same front check as any other revision, and it is
|
|
874
|
+
the direction that cannot lose the optimum.
|
|
875
|
+
"""
|
|
876
|
+
|
|
877
|
+
freed = set(free)
|
|
878
|
+
out: Overlay = {}
|
|
879
|
+
names = list(dict.fromkeys(
|
|
880
|
+
list(proposal) + list(free) + list(self._installed)))
|
|
881
|
+
for name in names:
|
|
882
|
+
entry = self._installed.get(name)
|
|
883
|
+
domain = self._field_domains.get(name)
|
|
884
|
+
if not domain:
|
|
885
|
+
if entry is not None:
|
|
886
|
+
out[name] = entry
|
|
887
|
+
continue
|
|
888
|
+
if name not in proposal and name not in freed:
|
|
889
|
+
if entry is not None:
|
|
890
|
+
out[name] = entry
|
|
891
|
+
continue
|
|
892
|
+
base = _distribution(entry, domain)
|
|
893
|
+
if name in proposal:
|
|
894
|
+
table = list(proposal[name])
|
|
895
|
+
prop = _normalize([_lookup(table, value) for value in domain])
|
|
896
|
+
else:
|
|
897
|
+
prop = [1.0 / len(domain)] * len(domain)
|
|
898
|
+
mixed = tuple((1.0 - self.damping) * b + self.damping * p
|
|
899
|
+
for b, p in zip(base, prop))
|
|
900
|
+
if all(w == mixed[0] for w in mixed):
|
|
901
|
+
# Uniform is FREE, and free is the honest reading: an entry
|
|
902
|
+
# here would only make the sampler take the weighted path to
|
|
903
|
+
# reach the draw it would have made anyway.
|
|
904
|
+
continue
|
|
905
|
+
out[name] = (tuple(domain), mixed)
|
|
906
|
+
return out
|
|
907
|
+
|
|
908
|
+
# -- immigrants ----------------------------------------------------------
|
|
909
|
+
|
|
910
|
+
def _immigrants(
|
|
911
|
+
self,
|
|
912
|
+
raw: Sequence[Any],
|
|
913
|
+
rows: Sequence[Tuple[Config, Mapping[str, float]]],
|
|
914
|
+
note: Dict[str, Any],
|
|
915
|
+
) -> Tuple[Config, ...]:
|
|
916
|
+
"""The REQUIRED k, validated value-by-value like ``llm_init``.
|
|
917
|
+
|
|
918
|
+
Same rule, same counters, two additions. A member the run has already
|
|
919
|
+
measured is dropped: it would cost nothing (the cache holds it) and
|
|
920
|
+
buy nothing, and counting it as accepted would report guidance that
|
|
921
|
+
moved no draw. And a reply that writes FEWER than k is not refused --
|
|
922
|
+
the prior half of the same reply was judged on its own and is
|
|
923
|
+
installed on its own, so an under-answered ask cannot cost the run the
|
|
924
|
+
channel that did answer. The shortfall is COUNTED instead, on the
|
|
925
|
+
event (proposed against required) and in the run's telemetry, which is
|
|
926
|
+
where a campaign reads how often the required-k ask was met at all.
|
|
927
|
+
"""
|
|
928
|
+
|
|
929
|
+
if self.immigrants <= 0:
|
|
930
|
+
return ()
|
|
931
|
+
provided = list(raw or ())
|
|
932
|
+
measured = {_key(config) for config, _objectives in rows}
|
|
933
|
+
accepted: List[Config] = []
|
|
934
|
+
rejected: List[Dict[str, str]] = []
|
|
935
|
+
loci = loci_of(self.template)
|
|
936
|
+
for member in provided:
|
|
937
|
+
self.telemetry.immigrants_proposed += 1
|
|
938
|
+
reason = _immigrant_reason(member, self.template, loci,
|
|
939
|
+
self.candidate_model)
|
|
940
|
+
if reason is None and _key(dict(member)) in measured:
|
|
941
|
+
reason = "already_measured"
|
|
942
|
+
if reason is None and len(accepted) >= self.immigrants:
|
|
943
|
+
reason = "over_cap"
|
|
944
|
+
if reason is not None:
|
|
945
|
+
self.telemetry.immigrants_rejected += 1
|
|
946
|
+
rejected.append({"reason": reason})
|
|
947
|
+
continue
|
|
948
|
+
self.telemetry.immigrants_accepted += 1
|
|
949
|
+
accepted.append(dict(member))
|
|
950
|
+
shortfall = max(0, self.immigrants - len(provided))
|
|
951
|
+
self.telemetry.immigrants_shortfall += shortfall
|
|
952
|
+
# WHY the channel bought nothing, per event, in one line. The rejection
|
|
953
|
+
# list already carried the reason on each member; the split is what a
|
|
954
|
+
# campaign reads, because the three reasons name three different
|
|
955
|
+
# failures and one aggregate count names none of them. A member the run
|
|
956
|
+
# has already charged (``already_measured``) says the ask needs more
|
|
957
|
+
# novelty ground -- the live 0-of-69 signature; ``out_of_domain`` says
|
|
958
|
+
# the model misread a declared vocabulary; ``shape`` says it wrote
|
|
959
|
+
# something that is not a configuration of this schema at all. Sparse
|
|
960
|
+
# by construction: a reason that never fired is absent, not zero.
|
|
961
|
+
by_reason: Dict[str, int] = {}
|
|
962
|
+
for entry in rejected:
|
|
963
|
+
reason = entry["reason"]
|
|
964
|
+
by_reason[reason] = by_reason.get(reason, 0) + 1
|
|
965
|
+
note["immigrants"] = {"accepted": len(accepted),
|
|
966
|
+
"rejected": rejected,
|
|
967
|
+
"rejected_by_reason": by_reason,
|
|
968
|
+
"proposed": len(provided),
|
|
969
|
+
"required": self.immigrants,
|
|
970
|
+
"shortfall": shortfall}
|
|
971
|
+
return tuple(accepted)
|
|
972
|
+
|
|
973
|
+
|
|
974
|
+
# ------------------------------------------------------------------ helpers
|
|
975
|
+
|
|
976
|
+
def _key(config: Mapping[str, Any]) -> str:
|
|
977
|
+
return json.dumps(dict(config), sort_keys=True, default=str)
|
|
978
|
+
|
|
979
|
+
|
|
980
|
+
def _token(value: Any) -> str:
|
|
981
|
+
"""One declared value's identity, rendered as measurement_evidence does."""
|
|
982
|
+
|
|
983
|
+
return value if isinstance(value, str) else json.dumps(value, default=str)
|
|
984
|
+
|
|
985
|
+
|
|
986
|
+
def _json_object(reply: Any) -> Optional[Dict[str, Any]]:
|
|
987
|
+
"""The one JSON object a reply carries, or ``None``.
|
|
988
|
+
|
|
989
|
+
The single reader of a raw reply, so the admission gate and the breadth
|
|
990
|
+
meter cannot disagree about what the model actually wrote.
|
|
991
|
+
"""
|
|
992
|
+
|
|
993
|
+
text = reply if isinstance(reply, str) else ""
|
|
994
|
+
match = re.search(r"\{.*\}", text, re.S)
|
|
995
|
+
if match is None:
|
|
996
|
+
return None
|
|
997
|
+
try:
|
|
998
|
+
raw = json.loads(match.group(0))
|
|
999
|
+
except (ValueError, TypeError):
|
|
1000
|
+
return None
|
|
1001
|
+
return raw if isinstance(raw, dict) else None
|
|
1002
|
+
|
|
1003
|
+
|
|
1004
|
+
def _weights_breadth(reply: Any) -> int:
|
|
1005
|
+
"""How many parameters a reply names in ``"weights"``. 0 when unreadable.
|
|
1006
|
+
|
|
1007
|
+
A measurement, not a check: nothing in this module refuses over it. It
|
|
1008
|
+
exists because the live pilot's replies tilted or freed every field of a
|
|
1009
|
+
24-field venue at once, which a per-event count makes visible and an
|
|
1010
|
+
admitted/refused verdict does not.
|
|
1011
|
+
"""
|
|
1012
|
+
|
|
1013
|
+
raw = _json_object(reply)
|
|
1014
|
+
entries = raw.get("weights") if raw is not None else None
|
|
1015
|
+
return len(entries) if isinstance(entries, dict) else 0
|
|
1016
|
+
|
|
1017
|
+
|
|
1018
|
+
def _front_indices(
|
|
1019
|
+
rows: Sequence[Tuple[Mapping[str, Any], Mapping[str, float]]],
|
|
1020
|
+
specs: Sequence[ObjectiveSpec],
|
|
1021
|
+
) -> List[int]:
|
|
1022
|
+
"""Row indices dominated by nothing. Goal-aware, weight-free."""
|
|
1023
|
+
|
|
1024
|
+
objectives = [dict(row[1]) for row in rows]
|
|
1025
|
+
return [
|
|
1026
|
+
index for index, this in enumerate(objectives)
|
|
1027
|
+
if not any(dominates(other, this, list(specs))
|
|
1028
|
+
for position, other in enumerate(objectives)
|
|
1029
|
+
if position != index)
|
|
1030
|
+
]
|
|
1031
|
+
|
|
1032
|
+
|
|
1033
|
+
def _field_domains(candidate_model: Any,
|
|
1034
|
+
template: Mapping[str, Any]) -> Dict[str, Tuple[Any, ...]]:
|
|
1035
|
+
"""Per-field vocabularies, the key the sampler consults.
|
|
1036
|
+
|
|
1037
|
+
Sequence loci share their field's entry: a weight table keyed per element
|
|
1038
|
+
would be a different prior for every genome length, which the ragged-genome
|
|
1039
|
+
case makes meaningless.
|
|
1040
|
+
"""
|
|
1041
|
+
|
|
1042
|
+
if candidate_model is None or not template:
|
|
1043
|
+
return {}
|
|
1044
|
+
out: Dict[str, Tuple[Any, ...]] = {}
|
|
1045
|
+
try:
|
|
1046
|
+
loci = loci_of(dict(template))
|
|
1047
|
+
except Exception:
|
|
1048
|
+
return {}
|
|
1049
|
+
for locus in loci:
|
|
1050
|
+
if locus.field in out:
|
|
1051
|
+
continue
|
|
1052
|
+
try:
|
|
1053
|
+
domain = tuple(locus_domain(candidate_model, locus))
|
|
1054
|
+
except Exception:
|
|
1055
|
+
domain = ()
|
|
1056
|
+
if not domain and locus.index is not None:
|
|
1057
|
+
try:
|
|
1058
|
+
domain = tuple(locus_domain(candidate_model, Locus(locus.field)))
|
|
1059
|
+
except Exception:
|
|
1060
|
+
domain = ()
|
|
1061
|
+
if domain:
|
|
1062
|
+
out[locus.field] = domain
|
|
1063
|
+
return out
|
|
1064
|
+
|
|
1065
|
+
|
|
1066
|
+
def _locus_domains(candidate_model: Any,
|
|
1067
|
+
template: Mapping[str, Any]) -> Dict[str, Tuple[Any, ...]]:
|
|
1068
|
+
"""Per-locus domains, which is what the evidence renderer keys on."""
|
|
1069
|
+
|
|
1070
|
+
if candidate_model is None or not template:
|
|
1071
|
+
return {}
|
|
1072
|
+
out: Dict[str, Tuple[Any, ...]] = {}
|
|
1073
|
+
try:
|
|
1074
|
+
loci = loci_of(dict(template))
|
|
1075
|
+
except Exception:
|
|
1076
|
+
return {}
|
|
1077
|
+
for locus in loci:
|
|
1078
|
+
try:
|
|
1079
|
+
domain = tuple(locus_domain(candidate_model, locus))
|
|
1080
|
+
except Exception:
|
|
1081
|
+
domain = ()
|
|
1082
|
+
if domain:
|
|
1083
|
+
out[str(locus)] = domain
|
|
1084
|
+
return out
|
|
1085
|
+
|
|
1086
|
+
|
|
1087
|
+
def _lookup(table: Sequence[Tuple[Any, float]], value: Any) -> float:
|
|
1088
|
+
for candidate, weight in table:
|
|
1089
|
+
if candidate == value:
|
|
1090
|
+
return float(weight)
|
|
1091
|
+
return 0.0
|
|
1092
|
+
|
|
1093
|
+
|
|
1094
|
+
def _normalize(raw: Sequence[float]) -> List[float]:
|
|
1095
|
+
total = float(sum(raw))
|
|
1096
|
+
if total <= 0.0 or not raw:
|
|
1097
|
+
return [1.0 / max(1, len(raw))] * len(raw)
|
|
1098
|
+
return [float(w) / total for w in raw]
|
|
1099
|
+
|
|
1100
|
+
|
|
1101
|
+
def _distribution(
|
|
1102
|
+
entry: Optional[Tuple[Tuple[Any, ...], Tuple[float, ...]]],
|
|
1103
|
+
domain: Sequence[Any],
|
|
1104
|
+
) -> List[float]:
|
|
1105
|
+
"""The installed weights over *domain*, normalized; uniform when absent."""
|
|
1106
|
+
|
|
1107
|
+
if entry is None:
|
|
1108
|
+
return [1.0 / len(domain)] * len(domain)
|
|
1109
|
+
table = list(zip(entry[0], entry[1]))
|
|
1110
|
+
return _normalize([_lookup(table, value) for value in domain])
|
|
1111
|
+
|
|
1112
|
+
|
|
1113
|
+
def _immigrant_reason(member: Any, template: Mapping[str, Any],
|
|
1114
|
+
loci: Sequence[Locus], candidate_model: Any) -> Optional[str]:
|
|
1115
|
+
"""``None`` when the member is admissible; the refusal reason otherwise."""
|
|
1116
|
+
|
|
1117
|
+
if not isinstance(member, dict) or set(member) != set(template):
|
|
1118
|
+
return "shape"
|
|
1119
|
+
try:
|
|
1120
|
+
member_loci = loci_of(member)
|
|
1121
|
+
except Exception:
|
|
1122
|
+
return "shape"
|
|
1123
|
+
if member_loci != tuple(loci):
|
|
1124
|
+
return "shape"
|
|
1125
|
+
for locus in member_loci:
|
|
1126
|
+
value = read_locus(member, locus)
|
|
1127
|
+
domain = locus_domain(candidate_model, locus)
|
|
1128
|
+
if domain:
|
|
1129
|
+
if value not in domain:
|
|
1130
|
+
return "out_of_domain"
|
|
1131
|
+
elif value != read_locus(template, locus):
|
|
1132
|
+
return "out_of_domain"
|
|
1133
|
+
return None
|