agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1760 @@
|
|
|
1
|
+
"""The model writes the SAMPLER, not the samples.
|
|
2
|
+
|
|
3
|
+
Every guidance mechanism before this one paid per decision: a chooser picks
|
|
4
|
+
parents once per offspring, an immigration call authors a handful of members
|
|
5
|
+
once per segment. Leverage like that scales as 1/budget, which is exactly why
|
|
6
|
+
per-decision guidance washes out on cheap-evaluation venues -- at B=40 one
|
|
7
|
+
call shapes a tenth of the run, at B=40,000 it shapes a ten-thousandth.
|
|
8
|
+
|
|
9
|
+
An authored GENERATOR inverts that economics. The model is asked once, before
|
|
10
|
+
any evaluation, to write ``propose(archive, n, domains, seed)`` -- a
|
|
11
|
+
distribution, not a draw -- and that one fixed cost then shapes every
|
|
12
|
+
candidate the run ever considers. Its quality is directly measurable and was
|
|
13
|
+
already measured before this seam existed (W3's induced-sampler recall
|
|
14
|
+
0.094/0.211/0.205 against 0.0585 chance).
|
|
15
|
+
|
|
16
|
+
The authoring line holds by the same construction as everywhere else, and it
|
|
17
|
+
is worth stating precisely because a generator emits candidates:
|
|
18
|
+
|
|
19
|
+
- Every emitted candidate is validated VALUE-BY-VALUE against the declared
|
|
20
|
+
domains and the template's shape (``validate_pool``); rejects are counted
|
|
21
|
+
and their slots fall back to schema-uniform draws, so a broken generator
|
|
22
|
+
degrades to the credential-free sampler rather than to nothing.
|
|
23
|
+
- The emitted set is measured for COLLAPSE -- duplicates inside the batch and
|
|
24
|
+
candidates already measured earlier in the run -- because a generator that
|
|
25
|
+
returns the same configuration n times is not a sampler, and the difference
|
|
26
|
+
has to be visible in the telemetry rather than inferred from a flat curve.
|
|
27
|
+
- This module is starved exactly as ``session.screening`` is: it receives a
|
|
28
|
+
template, a candidate model, a restriction and an archive of configurations
|
|
29
|
+
-- never the problem, never the evaluation cache. Mass generation therefore
|
|
30
|
+
cannot spend budget: the only route from a generated candidate to a real
|
|
31
|
+
evaluation is the loop measuring the ``want`` of them it could already
|
|
32
|
+
afford, which is a property of the import graph rather than a convention.
|
|
33
|
+
- It EVOLVES with feedback: the revision hook shows the generator its own
|
|
34
|
+
source plus what the harness measured about its output -- acceptance rate,
|
|
35
|
+
duplicate and archive-overlap rates, and how many of its candidates
|
|
36
|
+
survived selection -- and asks for a rewrite under the identical gate.
|
|
37
|
+
|
|
38
|
+
What the seam does NOT ask the model to do, since Wave D measured what
|
|
39
|
+
happens when it does. On an assignment-structured genome (`upms_j14_m3`,
|
|
40
|
+
fourteen scalar loci with per-locus eligibility) the sealed generator emitted
|
|
41
|
+
7,104 candidates against 39,993 uniformly-filled pool slots and 23
|
|
42
|
+
acceptances; `upms_j13_m3` reproduced it. The failure decomposes into three
|
|
43
|
+
things the HARNESS already knows and the model was left to re-derive:
|
|
44
|
+
|
|
45
|
+
- the SHAPE. ``policies.emit_scaffold`` ships a ``build(picks)`` helper into
|
|
46
|
+
the sandbox, so authored code names loci and values and the harness
|
|
47
|
+
assembles the configuration -- and assembles a partially-correct emission
|
|
48
|
+
rather than dropping it whole, a per-LOCUS fallback in place of a
|
|
49
|
+
per-CANDIDATE one.
|
|
50
|
+
- the DOMAINS. Every locus's admissible set is echoed into the authoring
|
|
51
|
+
prompt (``render_domain_echo``), because a field-level card cannot say
|
|
52
|
+
what a per-position domain is.
|
|
53
|
+
- the RESOURCE BUDGET. A batch that overruns the sandbox returns nothing, so
|
|
54
|
+
its whole pool falls back to uniform -- invisible in a per-candidate
|
|
55
|
+
counter. The wall/CPU/memory contract is echoed too, and one overrun buys
|
|
56
|
+
a retry at ``n // 4`` rather than the loss of the pool.
|
|
57
|
+
|
|
58
|
+
Each is counted separately, ablatable separately, and off restores the
|
|
59
|
+
sealed behaviour exactly.
|
|
60
|
+
|
|
61
|
+
Two of those channels fire on EMISSION DEFECTS -- a rejected candidate, a
|
|
62
|
+
collapsed batch, a generator whose children never survive. That is a repair
|
|
63
|
+
loop for a broken sampler, and it is not the same thing as guidance: a
|
|
64
|
+
generator that emits perfectly valid candidates from a region the run has
|
|
65
|
+
already measured to be bad is never revised at all, because nothing about it
|
|
66
|
+
is defective. The prior it encodes is STATIC -- written once from the semantic
|
|
67
|
+
card, in a ladder cell from an empty archive -- and where a recallable domain
|
|
68
|
+
prior is strong that is worth a great deal, while where one is not it is worth
|
|
69
|
+
exactly nothing, three times measured.
|
|
70
|
+
|
|
71
|
+
``reauthor_every`` is the other trigger, and it fires on SEARCH PROGRESS
|
|
72
|
+
rather than on defects: after the run has MEASURED that many more rows, the
|
|
73
|
+
generator is re-authored against the measured
|
|
74
|
+
``(configuration -> objectives)`` trace -- the current front, what improved
|
|
75
|
+
and what did not, and which parameter the measurements say moves which cost
|
|
76
|
+
(:mod:`agent_evolve.policies.measurement_evidence`). ``locus_prior`` is the
|
|
77
|
+
second consumer of the same evidence: the model weighs the parameters and
|
|
78
|
+
values worth the remaining budget, the harness types the answer as a GRADED
|
|
79
|
+
bias over the declared domains -- per-locus value weights that exclude
|
|
80
|
+
NOTHING -- and a GATE refuses it, rather than trusts it, whenever it is
|
|
81
|
+
undeclared, empty, garbage-weighted or concentrated past the declared
|
|
82
|
+
weight-ratio cap. Because nothing is excluded, excluding a measured front
|
|
83
|
+
member is structurally impossible rather than gated against.
|
|
84
|
+
|
|
85
|
+
The evidence is WHAT THE RUN MEASURED, not what this generator produced. Every
|
|
86
|
+
charged evaluation the loop holds -- the initial population, a structure
|
|
87
|
+
screen, and the generator's own children alike -- arrives through
|
|
88
|
+
``note_measured``; only the children additionally move the attribution
|
|
89
|
+
counters, through ``record_measured``. Keeping those two apart is what lets
|
|
90
|
+
the channel speak at the first generation instead of the third: while the
|
|
91
|
+
evidence clock was the generator's own children, the locus prior could not be
|
|
92
|
+
authored before a median charge of 40 on a venue whose dominant knob is
|
|
93
|
+
legible by charge 19, which is the whole of the W11 defect. ``reauthor_every``
|
|
94
|
+
therefore governs how often an evidence call RECURS, and ``evidence_min_rows``
|
|
95
|
+
-- default: the fewest rows a determinable effect can be computed from --
|
|
96
|
+
governs when the first one may fire.
|
|
97
|
+
|
|
98
|
+
Both are OFF by default (``reauthor_every=0``), and off is the seam that ran
|
|
99
|
+
every sealed row to date. Both record what evidence the model was shown (its
|
|
100
|
+
digest), what it emitted, and whether the emission was accepted, so no run can
|
|
101
|
+
imply it reasoned over measurements when it did not.
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
from __future__ import annotations
|
|
105
|
+
|
|
106
|
+
import json
|
|
107
|
+
import random
|
|
108
|
+
from dataclasses import dataclass, fields
|
|
109
|
+
from typing import Any, Callable, Dict, List, Mapping, Optional, Sequence, Tuple
|
|
110
|
+
|
|
111
|
+
from agent_evolve.core.authored import CONTRACTS, AuthoredArtifact, authored_artifact
|
|
112
|
+
from agent_evolve.core.problem import ObjectiveSpec
|
|
113
|
+
from agent_evolve.infrastructure.authored_worker import ALLOWED_IMPORTS
|
|
114
|
+
from agent_evolve.policies.emit_scaffold import (
|
|
115
|
+
NOTES_GLOBAL,
|
|
116
|
+
SCAFFOLD_RULES,
|
|
117
|
+
coerce_candidate,
|
|
118
|
+
render_domain_echo,
|
|
119
|
+
scaffold_prelude,
|
|
120
|
+
)
|
|
121
|
+
from agent_evolve.policies.genetic import (
|
|
122
|
+
loci_of,
|
|
123
|
+
locus_domain,
|
|
124
|
+
read_locus,
|
|
125
|
+
uniform_candidate,
|
|
126
|
+
)
|
|
127
|
+
from agent_evolve.policies.llm_surrogate import (
|
|
128
|
+
AuthorTelemetry,
|
|
129
|
+
accept_block,
|
|
130
|
+
json_compact,
|
|
131
|
+
)
|
|
132
|
+
from agent_evolve.policies.measurement_evidence import (
|
|
133
|
+
MIN_EVIDENCE_ROWS,
|
|
134
|
+
WEIGHTED_RESTRICTION_PROMPT,
|
|
135
|
+
MeasuredRow,
|
|
136
|
+
admit_weighted_restriction,
|
|
137
|
+
apply_weighted_restriction,
|
|
138
|
+
evidence_digest,
|
|
139
|
+
parse_weighted_restriction,
|
|
140
|
+
render_measurement_evidence,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
__all__ = [
|
|
144
|
+
"GeneratorTelemetry",
|
|
145
|
+
"PoolReport",
|
|
146
|
+
"RejectionCensus",
|
|
147
|
+
"AuthoredGenerator",
|
|
148
|
+
"author_generator",
|
|
149
|
+
"revise_generator",
|
|
150
|
+
"reauthor_generator",
|
|
151
|
+
"render_generation_feedback",
|
|
152
|
+
"validate_pool",
|
|
153
|
+
"candidate_key",
|
|
154
|
+
"GENERATOR_PROMPT",
|
|
155
|
+
"GENERATOR_REVISION_PROMPT",
|
|
156
|
+
"GENERATOR_EVIDENCE_PROMPT",
|
|
157
|
+
]
|
|
158
|
+
|
|
159
|
+
Config = Dict[str, Any]
|
|
160
|
+
|
|
161
|
+
#: How many archive members a prompt-side call carries into the sandbox. The
|
|
162
|
+
#: contract says "the archive"; shipping all of it would make a 10,000-
|
|
163
|
+
#: evaluation run pay a growing JSON round trip every generation for
|
|
164
|
+
#: information the sampler cannot use anyway.
|
|
165
|
+
ARCHIVE_SHOWN = 32
|
|
166
|
+
|
|
167
|
+
#: A hard ceiling on one batch, so "a configured pool size up to very large"
|
|
168
|
+
#: cannot become an accidental out-of-memory. Mass generation is the point;
|
|
169
|
+
#: an unbounded request is not.
|
|
170
|
+
MAX_POOL = 200_000
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def candidate_key(config: Mapping[str, Any]) -> str:
|
|
174
|
+
"""Identity of a configuration, matching the loop's own dedup key."""
|
|
175
|
+
|
|
176
|
+
return json.dumps(config, sort_keys=True, default=str)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@dataclass
|
|
180
|
+
class GeneratorTelemetry:
|
|
181
|
+
"""What mass generation actually did. Counted, never inferred.
|
|
182
|
+
|
|
183
|
+
Rates are deliberately absent: :func:`~agent_evolve.core.telemetry.
|
|
184
|
+
harvest_telemetry` coerces counters to ``int``, so every rate this
|
|
185
|
+
mechanism claims is published as its exact numerator and denominator
|
|
186
|
+
(``duplicates`` over ``emitted``) and computed by whoever reads them.
|
|
187
|
+
:class:`PoolReport` carries the same ratios as floats for callers in
|
|
188
|
+
process.
|
|
189
|
+
"""
|
|
190
|
+
|
|
191
|
+
batches: int = 0
|
|
192
|
+
runtime_failures: int = 0
|
|
193
|
+
emitted: int = 0
|
|
194
|
+
accepted: int = 0
|
|
195
|
+
rejected_shape: int = 0
|
|
196
|
+
rejected_out_of_domain: int = 0
|
|
197
|
+
duplicates: int = 0
|
|
198
|
+
archive_overlap: int = 0
|
|
199
|
+
filled_uniform: int = 0
|
|
200
|
+
measured: int = 0
|
|
201
|
+
survived: int = 0
|
|
202
|
+
revisions: int = 0
|
|
203
|
+
revisions_accepted: int = 0
|
|
204
|
+
#: Candidates the harness ASSEMBLED rather than rejected: the emitted
|
|
205
|
+
#: member addressed at least one locus with an admissible value, and the
|
|
206
|
+
#: rest of the configuration was filled from the template and the
|
|
207
|
+
#: domains. Counted apart from ``accepted`` because a repaired candidate
|
|
208
|
+
#: carries less of the model's guidance than a clean one, and a mechanism
|
|
209
|
+
#: that only works after repair must not read as one that works.
|
|
210
|
+
repaired: int = 0
|
|
211
|
+
#: Individual loci the harness had to decide inside those candidates.
|
|
212
|
+
repaired_loci: int = 0
|
|
213
|
+
#: What the in-sandbox emit scaffold reported about its own work: loci
|
|
214
|
+
#: the authored code left unset, and values it asked for that were not in
|
|
215
|
+
#: that locus's domain. Both are counted at the point of construction, so
|
|
216
|
+
#: they are visible even when nothing is rejected at all.
|
|
217
|
+
scaffold_filled: int = 0
|
|
218
|
+
scaffold_out_of_domain: int = 0
|
|
219
|
+
#: Locus names the authored code used that the schema does not declare.
|
|
220
|
+
scaffold_unknown_locus: int = 0
|
|
221
|
+
#: Revisions that were authored, ran, and did NOT improve the measured
|
|
222
|
+
#: defect -- the population the rejected-edit memory is built from.
|
|
223
|
+
revisions_rejected: int = 0
|
|
224
|
+
#: Batches that blew the sandbox's wall/CPU/memory budget and were retried
|
|
225
|
+
#: at a smaller ``n``, and how many of those retries came back usable.
|
|
226
|
+
runtime_retries: int = 0
|
|
227
|
+
runtime_recovered: int = 0
|
|
228
|
+
#: The MEASUREMENT-CONDITIONED channel, counted apart from the
|
|
229
|
+
#: defect-triggered one above, because they are different mechanisms and a
|
|
230
|
+
#: campaign that cannot tell them apart cannot attribute anything.
|
|
231
|
+
#: ``reauthorings`` fired; ``reauthorings_accepted`` came back as a usable
|
|
232
|
+
#: artifact; ``evidence_rows_shown`` is how many measured rows the model
|
|
233
|
+
#: was actually given across those calls.
|
|
234
|
+
reauthorings: int = 0
|
|
235
|
+
reauthorings_accepted: int = 0
|
|
236
|
+
evidence_rows_shown: int = 0
|
|
237
|
+
#: The locus-importance channel. ``priors_refused`` is the number the GATE
|
|
238
|
+
#: threw out and is the counter that says the gate is doing its job;
|
|
239
|
+
#: ``priors_unwound`` counts admitted priors later dropped for not paying.
|
|
240
|
+
priors_proposed: int = 0
|
|
241
|
+
priors_admitted: int = 0
|
|
242
|
+
priors_refused: int = 0
|
|
243
|
+
priors_unwound: int = 0
|
|
244
|
+
|
|
245
|
+
def as_dict(self) -> dict[str, int]:
|
|
246
|
+
return {f.name: int(getattr(self, f.name)) for f in fields(self)}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
@dataclass
|
|
250
|
+
class RejectionCensus:
|
|
251
|
+
"""WHICH loci rejected, WHY, and one concrete offending sample of each.
|
|
252
|
+
|
|
253
|
+
Wave D's counters said 6,021 shape and 1,060 out-of-domain and could say
|
|
254
|
+
nothing more, so the revision prompt could only tell the model that
|
|
255
|
+
something was wrong -- which is why revision fired on 77 of 80 cells and
|
|
256
|
+
repaired none of them. A revision is a repair instruction, and a repair
|
|
257
|
+
instruction needs the address of the fault: the locus, the reason, and a
|
|
258
|
+
value the model can recognise as its own.
|
|
259
|
+
|
|
260
|
+
The routing is the point (program section 9-B5, the SHE borrowing):
|
|
261
|
+
a defect is diagnosed against the artifact responsible for it, not
|
|
262
|
+
aggregated into a rate that names nobody.
|
|
263
|
+
"""
|
|
264
|
+
|
|
265
|
+
shape_reasons: Dict[str, int] = None # type: ignore[assignment]
|
|
266
|
+
out_of_domain_by_locus: Dict[str, int] = None # type: ignore[assignment]
|
|
267
|
+
repaired_by_locus: Dict[str, int] = None # type: ignore[assignment]
|
|
268
|
+
samples: Dict[str, Any] = None # type: ignore[assignment]
|
|
269
|
+
|
|
270
|
+
def __post_init__(self) -> None:
|
|
271
|
+
for name in ("shape_reasons", "out_of_domain_by_locus",
|
|
272
|
+
"repaired_by_locus", "samples"):
|
|
273
|
+
if getattr(self, name) is None:
|
|
274
|
+
setattr(self, name, {})
|
|
275
|
+
|
|
276
|
+
def shape(self, reason: str, sample: Any = None) -> None:
|
|
277
|
+
self.shape_reasons[reason] = self.shape_reasons.get(reason, 0) + 1
|
|
278
|
+
self.samples.setdefault(f"shape:{reason}", _sample_of(sample))
|
|
279
|
+
|
|
280
|
+
def out_of_domain(self, locus: str, value: Any) -> None:
|
|
281
|
+
self.out_of_domain_by_locus[locus] = (
|
|
282
|
+
self.out_of_domain_by_locus.get(locus, 0) + 1)
|
|
283
|
+
self.sample(locus, value)
|
|
284
|
+
|
|
285
|
+
def sample(self, locus: str, value: Any) -> None:
|
|
286
|
+
"""One concrete offending value at *locus*; the first one sticks."""
|
|
287
|
+
|
|
288
|
+
self.samples.setdefault(f"domain:{locus}", _sample_of(value))
|
|
289
|
+
|
|
290
|
+
def repaired(self, locus: str) -> None:
|
|
291
|
+
self.repaired_by_locus[locus] = self.repaired_by_locus.get(locus, 0) + 1
|
|
292
|
+
|
|
293
|
+
def merge(self, other: "RejectionCensus") -> None:
|
|
294
|
+
for reason, count in other.shape_reasons.items():
|
|
295
|
+
self.shape_reasons[reason] = self.shape_reasons.get(reason, 0) + count
|
|
296
|
+
for locus, count in other.out_of_domain_by_locus.items():
|
|
297
|
+
self.out_of_domain_by_locus[locus] = (
|
|
298
|
+
self.out_of_domain_by_locus.get(locus, 0) + count)
|
|
299
|
+
for locus, count in other.repaired_by_locus.items():
|
|
300
|
+
self.repaired_by_locus[locus] = (
|
|
301
|
+
self.repaired_by_locus.get(locus, 0) + count)
|
|
302
|
+
for key, value in other.samples.items():
|
|
303
|
+
self.samples.setdefault(key, value)
|
|
304
|
+
|
|
305
|
+
@property
|
|
306
|
+
def empty(self) -> bool:
|
|
307
|
+
return not (self.shape_reasons or self.out_of_domain_by_locus
|
|
308
|
+
or self.repaired_by_locus)
|
|
309
|
+
|
|
310
|
+
def signature(self) -> str:
|
|
311
|
+
"""A stable name for THIS defect, so a repeat is recognisable.
|
|
312
|
+
|
|
313
|
+
Two revisions that leave the same loci failing for the same reasons
|
|
314
|
+
have not changed anything the harness can measure, whatever else they
|
|
315
|
+
changed -- and that is exactly what the rejected-edit memory must be
|
|
316
|
+
able to say back to the next revision.
|
|
317
|
+
"""
|
|
318
|
+
|
|
319
|
+
parts = ([f"shape:{k}" for k in sorted(self.shape_reasons)]
|
|
320
|
+
+ [f"domain:{k}" for k in sorted(self.out_of_domain_by_locus)])
|
|
321
|
+
return "|".join(parts) or "clean"
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _sample_of(value: Any) -> Any:
|
|
325
|
+
"""A JSON-safe, bounded rendering of one offending value."""
|
|
326
|
+
|
|
327
|
+
try:
|
|
328
|
+
json.dumps(value)
|
|
329
|
+
except (TypeError, ValueError):
|
|
330
|
+
return repr(value)[:120]
|
|
331
|
+
if isinstance(value, str) and len(value) > 120:
|
|
332
|
+
return value[:120]
|
|
333
|
+
return value
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
@dataclass(frozen=True)
|
|
337
|
+
class PoolReport:
|
|
338
|
+
"""One mass-generation batch, as the harness received it."""
|
|
339
|
+
|
|
340
|
+
accepted: Tuple[Config, ...] = ()
|
|
341
|
+
emitted: int = 0
|
|
342
|
+
rejected_shape: int = 0
|
|
343
|
+
rejected_out_of_domain: int = 0
|
|
344
|
+
duplicates: int = 0
|
|
345
|
+
archive_overlap: int = 0
|
|
346
|
+
#: Accepted by ASSEMBLY rather than as emitted (see ``repair`` below).
|
|
347
|
+
repaired: int = 0
|
|
348
|
+
repaired_loci: int = 0
|
|
349
|
+
census: RejectionCensus = None # type: ignore[assignment]
|
|
350
|
+
|
|
351
|
+
def __post_init__(self) -> None:
|
|
352
|
+
if self.census is None:
|
|
353
|
+
object.__setattr__(self, "census", RejectionCensus())
|
|
354
|
+
|
|
355
|
+
def _rate(self, count: int) -> float:
|
|
356
|
+
return (count / self.emitted) if self.emitted else 0.0
|
|
357
|
+
|
|
358
|
+
@property
|
|
359
|
+
def acceptance_rate(self) -> float:
|
|
360
|
+
"""Fraction of the emitted batch the VALIDATION let through.
|
|
361
|
+
|
|
362
|
+
Distinct from :attr:`novelty_rate`: a generator can be perfectly
|
|
363
|
+
in-domain and still emit one configuration a thousand times, and the
|
|
364
|
+
two guards have to be readable apart.
|
|
365
|
+
"""
|
|
366
|
+
|
|
367
|
+
return self._rate(self.emitted - self.rejected_shape
|
|
368
|
+
- self.rejected_out_of_domain)
|
|
369
|
+
|
|
370
|
+
@property
|
|
371
|
+
def duplicate_rate(self) -> float:
|
|
372
|
+
"""Fraction of the emitted batch that repeated an earlier member."""
|
|
373
|
+
|
|
374
|
+
return self._rate(self.duplicates)
|
|
375
|
+
|
|
376
|
+
@property
|
|
377
|
+
def archive_overlap_rate(self) -> float:
|
|
378
|
+
"""Fraction of the emitted batch already measured in this run."""
|
|
379
|
+
|
|
380
|
+
return self._rate(self.archive_overlap)
|
|
381
|
+
|
|
382
|
+
@property
|
|
383
|
+
def novelty_rate(self) -> float:
|
|
384
|
+
"""Fraction of the emitted batch that was both valid and new."""
|
|
385
|
+
|
|
386
|
+
return self._rate(len(self.accepted))
|
|
387
|
+
|
|
388
|
+
@property
|
|
389
|
+
def defect_rate(self) -> float:
|
|
390
|
+
"""Fraction of the batch the harness had to reject OR assemble.
|
|
391
|
+
|
|
392
|
+
The one number a revision must move. Repairs are counted as defects
|
|
393
|
+
here even though they were accepted: a candidate the harness had to
|
|
394
|
+
finish is a candidate the model did not write, and a revision that
|
|
395
|
+
turns rejections into repairs has moved the failure rather than
|
|
396
|
+
fixed it.
|
|
397
|
+
"""
|
|
398
|
+
|
|
399
|
+
return self._rate(self.rejected_shape + self.rejected_out_of_domain
|
|
400
|
+
+ self.repaired)
|
|
401
|
+
|
|
402
|
+
def as_note(self) -> Dict[str, Any]:
|
|
403
|
+
"""The per-generation history record: counts plus the guard's rates."""
|
|
404
|
+
|
|
405
|
+
return {
|
|
406
|
+
"emitted": self.emitted,
|
|
407
|
+
"accepted": len(self.accepted),
|
|
408
|
+
"rejected_shape": self.rejected_shape,
|
|
409
|
+
"rejected_out_of_domain": self.rejected_out_of_domain,
|
|
410
|
+
"duplicates": self.duplicates,
|
|
411
|
+
"archive_overlap": self.archive_overlap,
|
|
412
|
+
"repaired": self.repaired,
|
|
413
|
+
"repaired_loci": self.repaired_loci,
|
|
414
|
+
"acceptance_rate": round(self.acceptance_rate, 4),
|
|
415
|
+
"duplicate_rate": round(self.duplicate_rate, 4),
|
|
416
|
+
"archive_overlap_rate": round(self.archive_overlap_rate, 4),
|
|
417
|
+
"novelty_rate": round(self.novelty_rate, 4),
|
|
418
|
+
"defect_rate": round(self.defect_rate, 4),
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def validate_pool(
|
|
423
|
+
emitted: Any,
|
|
424
|
+
*,
|
|
425
|
+
template: Config,
|
|
426
|
+
domains: Mapping[str, Sequence[Any]],
|
|
427
|
+
seen: Optional[Any] = None,
|
|
428
|
+
limit: Optional[int] = None,
|
|
429
|
+
repair: bool = False,
|
|
430
|
+
rng: Optional[random.Random] = None,
|
|
431
|
+
) -> PoolReport:
|
|
432
|
+
"""Accept the candidates a generator may actually emit into the run.
|
|
433
|
+
|
|
434
|
+
Every candidate is checked value by value: the shape must equal the
|
|
435
|
+
template's, a locus with a declared domain must hold a declared value,
|
|
436
|
+
and a locus the schema does not constrain must keep the template's value
|
|
437
|
+
-- the same rule ``llm_init`` applies to authored initial members, which
|
|
438
|
+
is the point: one authoring line, not one per seam.
|
|
439
|
+
|
|
440
|
+
The diversity guard rides along, because a batch is a SET and its
|
|
441
|
+
degeneracies are only visible batch-wide: a candidate repeating an
|
|
442
|
+
earlier member of the same batch counts as a duplicate, and one whose
|
|
443
|
+
key is in *seen* (everything the run has measured) counts as archive
|
|
444
|
+
overlap. Both are dropped -- they consume a pool slot and can teach the
|
|
445
|
+
run nothing -- and both are counted, so a generator that has collapsed
|
|
446
|
+
onto one configuration reads as ``duplicates == emitted - 1`` instead of
|
|
447
|
+
as an unremarkable flat curve.
|
|
448
|
+
|
|
449
|
+
*limit* caps how many members are considered at all, so a generator that
|
|
450
|
+
answers "give me 2,000" with 200,000 costs the harness the 2,000 it
|
|
451
|
+
asked for. ``PoolReport.emitted`` counts what was considered, which is
|
|
452
|
+
the denominator every rate here is against.
|
|
453
|
+
|
|
454
|
+
Every reject is also ADDRESSED, into :class:`RejectionCensus`: which
|
|
455
|
+
locus, which reason, and one concrete offending sample. A counter that
|
|
456
|
+
says "6,021 shape" cannot instruct a revision; "job_13 is missing from
|
|
457
|
+
every candidate you emitted, sample {...}" can.
|
|
458
|
+
|
|
459
|
+
*repair* turns the per-candidate fallback into a per-LOCUS one. Off (the
|
|
460
|
+
default, and what every sealed row was measured under) a candidate with
|
|
461
|
+
one bad locus is dropped whole and its pool slot is filled by a
|
|
462
|
+
schema-uniform draw, so thirteen good choices are discarded with the
|
|
463
|
+
fourteenth. On, the harness assembles the candidate out of whatever the
|
|
464
|
+
member did address -- flat locus keys, the template's own nesting, or a
|
|
465
|
+
bare sequence aligned with the loci -- and decides only the loci the
|
|
466
|
+
member got wrong or left out. Repairs are accepted, but counted apart in
|
|
467
|
+
``repaired``/``repaired_loci`` and censused per locus, because a
|
|
468
|
+
candidate the harness finished is not a candidate the model wrote.
|
|
469
|
+
"""
|
|
470
|
+
|
|
471
|
+
if not isinstance(emitted, list):
|
|
472
|
+
return PoolReport()
|
|
473
|
+
template_loci = loci_of(template)
|
|
474
|
+
template_fields = set(template)
|
|
475
|
+
known = set(seen) if seen is not None else set()
|
|
476
|
+
batch: set[str] = set()
|
|
477
|
+
draw = rng if rng is not None else random.Random(0)
|
|
478
|
+
|
|
479
|
+
accepted: List[Config] = []
|
|
480
|
+
census = RejectionCensus()
|
|
481
|
+
counts = {"shape": 0, "domain": 0, "duplicate": 0, "overlap": 0,
|
|
482
|
+
"repaired": 0, "repaired_loci": 0}
|
|
483
|
+
considered = 0
|
|
484
|
+
for member in emitted:
|
|
485
|
+
if limit is not None and considered >= limit:
|
|
486
|
+
break
|
|
487
|
+
considered += 1
|
|
488
|
+
candidate, reason, locus, value = _read_candidate(
|
|
489
|
+
member, template=template, template_loci=template_loci,
|
|
490
|
+
template_fields=template_fields, domains=domains)
|
|
491
|
+
if candidate is None and repair:
|
|
492
|
+
fixed, repairs = coerce_candidate(
|
|
493
|
+
member, template=template, domains=domains, rng=draw,
|
|
494
|
+
loci=template_loci)
|
|
495
|
+
if fixed is not None:
|
|
496
|
+
for kind, loci_list in repairs.items():
|
|
497
|
+
for name in loci_list:
|
|
498
|
+
census.repaired(name)
|
|
499
|
+
if kind == "out_of_domain":
|
|
500
|
+
census.out_of_domain(name, _picked(member, name))
|
|
501
|
+
counts["repaired"] += 1
|
|
502
|
+
counts["repaired_loci"] += sum(len(v) for v in repairs.values())
|
|
503
|
+
candidate, reason = fixed, ""
|
|
504
|
+
if candidate is None:
|
|
505
|
+
if reason == "domain":
|
|
506
|
+
counts["domain"] += 1
|
|
507
|
+
census.out_of_domain(str(locus), value)
|
|
508
|
+
else:
|
|
509
|
+
counts["shape"] += 1
|
|
510
|
+
census.shape(reason, member)
|
|
511
|
+
continue
|
|
512
|
+
key = candidate_key(candidate)
|
|
513
|
+
if key in batch:
|
|
514
|
+
counts["duplicate"] += 1
|
|
515
|
+
continue
|
|
516
|
+
# Marked as seen in this batch whatever happens next, so a candidate
|
|
517
|
+
# that is BOTH already measured and repeated eight times reads as one
|
|
518
|
+
# overlap and seven duplicates. Two different defects, two counters.
|
|
519
|
+
batch.add(key)
|
|
520
|
+
if key in known:
|
|
521
|
+
counts["overlap"] += 1
|
|
522
|
+
continue
|
|
523
|
+
accepted.append(dict(candidate))
|
|
524
|
+
|
|
525
|
+
return PoolReport(
|
|
526
|
+
accepted=tuple(accepted),
|
|
527
|
+
emitted=considered,
|
|
528
|
+
rejected_shape=counts["shape"],
|
|
529
|
+
rejected_out_of_domain=counts["domain"],
|
|
530
|
+
duplicates=counts["duplicate"],
|
|
531
|
+
archive_overlap=counts["overlap"],
|
|
532
|
+
repaired=counts["repaired"],
|
|
533
|
+
repaired_loci=counts["repaired_loci"],
|
|
534
|
+
census=census,
|
|
535
|
+
)
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
def _read_candidate(member, *, template, template_loci, template_fields,
|
|
539
|
+
domains):
|
|
540
|
+
"""``(config, reason, locus, value)`` -- the value-by-value gate itself.
|
|
541
|
+
|
|
542
|
+
``config`` is the member unchanged when it passes. Otherwise ``reason``
|
|
543
|
+
names WHICH gate it failed, in the vocabulary a revision can act on:
|
|
544
|
+
``not a mapping``, ``missing loci``, ``unexpected fields``, ``wrong
|
|
545
|
+
length``, or ``domain`` with the offending locus and value.
|
|
546
|
+
"""
|
|
547
|
+
|
|
548
|
+
if not isinstance(member, dict):
|
|
549
|
+
return None, "not a mapping", None, None
|
|
550
|
+
fields_seen = set(member)
|
|
551
|
+
if fields_seen != template_fields:
|
|
552
|
+
missing = sorted(template_fields - fields_seen)
|
|
553
|
+
extra = sorted(fields_seen - template_fields)
|
|
554
|
+
if missing and extra:
|
|
555
|
+
reason = (f"missing fields {missing[:4]} and unexpected fields "
|
|
556
|
+
f"{extra[:4]}")
|
|
557
|
+
elif missing:
|
|
558
|
+
reason = f"missing fields {missing[:6]}"
|
|
559
|
+
else:
|
|
560
|
+
reason = f"unexpected fields {extra[:6]}"
|
|
561
|
+
return None, reason, None, None
|
|
562
|
+
try:
|
|
563
|
+
member_loci = loci_of(member)
|
|
564
|
+
except Exception:
|
|
565
|
+
return None, "not a mapping", None, None
|
|
566
|
+
if member_loci != template_loci:
|
|
567
|
+
want, got = len(template_loci), len(member_loci)
|
|
568
|
+
return None, (f"wrong sequence length ({got} loci, the archive's "
|
|
569
|
+
f"members have {want})"), None, None
|
|
570
|
+
for locus in member_loci:
|
|
571
|
+
value = read_locus(member, locus)
|
|
572
|
+
domain = domains.get(str(locus)) or ()
|
|
573
|
+
if domain:
|
|
574
|
+
if value not in domain:
|
|
575
|
+
return None, "domain", locus, value
|
|
576
|
+
elif value != read_locus(template, locus):
|
|
577
|
+
return None, "domain", locus, value
|
|
578
|
+
return member, "", None, None
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
def _picked(member: Any, locus: str) -> Any:
|
|
582
|
+
"""The value *member* carried at *locus*, for the census's sample."""
|
|
583
|
+
|
|
584
|
+
try:
|
|
585
|
+
if isinstance(member, dict):
|
|
586
|
+
if locus in member:
|
|
587
|
+
return member[locus]
|
|
588
|
+
if locus.endswith("]") and "[" in locus:
|
|
589
|
+
field, index = locus[:-1].split("[", 1)
|
|
590
|
+
return member[field][int(index)]
|
|
591
|
+
except Exception:
|
|
592
|
+
return None
|
|
593
|
+
return None
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
GENERATOR_PROMPT = """You are writing the CANDIDATE GENERATOR for a black-box \
|
|
597
|
+
multi-objective optimizer. It calls your function every generation to draw the \
|
|
598
|
+
pool of configurations it will consider; you are writing the DISTRIBUTION those \
|
|
599
|
+
draws come from, not any particular draw.
|
|
600
|
+
|
|
601
|
+
OBJECTIVES (name and direction):
|
|
602
|
+
{goals}
|
|
603
|
+
|
|
604
|
+
SEARCH SPACE:
|
|
605
|
+
{schema}
|
|
606
|
+
{loci}
|
|
607
|
+
Write ONE Python function with EXACTLY this signature:
|
|
608
|
+
|
|
609
|
+
{contract}
|
|
610
|
+
|
|
611
|
+
Rules:
|
|
612
|
+
- `domains` maps every locus to its allowed values under the current sampling
|
|
613
|
+
prior; sequence fields appear per position as `name[i]`. Each configuration
|
|
614
|
+
you return must have the SAME SHAPE as the archive members and take every
|
|
615
|
+
value from `domains` at that locus -- anything else is validated out and
|
|
616
|
+
its slot falls back to a uniform random draw.
|
|
617
|
+
{scaffold}- `archive` holds configurations already measured in this run (it may be
|
|
618
|
+
short early on). Use it for context; do NOT return copies of it, and do not
|
|
619
|
+
return the same configuration twice. A batch that collapses is measured and
|
|
620
|
+
reported back to you.
|
|
621
|
+
- Return EXACTLY `n` configurations. `n` can be large (thousands): this is
|
|
622
|
+
mass generation, so keep it cheap and vectorless -- plain loops over
|
|
623
|
+
`domains`.
|
|
624
|
+
- Use what the parameter NAMES AND MEANINGS say about this domain to bias
|
|
625
|
+
where mass lands: known good regions, couplings that must co-move,
|
|
626
|
+
trade-offs worth spreading along. That knowledge is the only reason your
|
|
627
|
+
sampler can beat drawing uniformly from the same domains, which is exactly
|
|
628
|
+
what it is measured against.
|
|
629
|
+
- Derive all randomness from `seed` (e.g. `random.Random(seed)`), so a pool
|
|
630
|
+
is reproducible.
|
|
631
|
+
- Standard library only; imports limited to: {imports}.
|
|
632
|
+
- No I/O, no globals, deterministic.
|
|
633
|
+
{limits}
|
|
634
|
+
Reply with ONLY one fenced Python code block and no other text."""
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
#: The resource contract, echoed for the same reason the domains are: a
|
|
638
|
+
#: function that exceeds its sandbox budget returns NOTHING, so the whole pool
|
|
639
|
+
#: falls back to schema-uniform draws and the mechanism contributes zero. Wave
|
|
640
|
+
#: D's `upms_j14_m3` telemetry is the evidence -- 7,104 candidates emitted
|
|
641
|
+
#: against 39,993 pool slots filled uniformly means most BATCHES emitted
|
|
642
|
+
#: nothing at all, which is what a timeout looks like when the counter is
|
|
643
|
+
#: per-candidate. An unstated budget is a budget the author cannot honour.
|
|
644
|
+
LIMITS_RULES = """\
|
|
645
|
+
- HARD RESOURCE LIMITS, enforced by the sandbox: {wall} s wall-clock, {cpu} s
|
|
646
|
+
CPU and {memory} MB of memory for ONE call, at up to n={max_n}. Exceeding
|
|
647
|
+
any of them returns NOTHING -- not a partial pool, nothing -- and the run
|
|
648
|
+
falls back to drawing every candidate uniformly, which is exactly the
|
|
649
|
+
baseline you are being measured against. Budget for the WORST case, not the
|
|
650
|
+
typical one: prefer O(n) construction from `domains` to any search, sort or
|
|
651
|
+
simulation over candidates, and if you want a local improvement step, cap
|
|
652
|
+
its total work by a constant you choose rather than by convergence."""
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
GENERATOR_REVISION_PROMPT = """You previously wrote this candidate generator \
|
|
656
|
+
for a black-box multi-objective optimizer:
|
|
657
|
+
|
|
658
|
+
```python
|
|
659
|
+
{source}
|
|
660
|
+
```
|
|
661
|
+
|
|
662
|
+
The harness ran it, validated everything it emitted, and measured what
|
|
663
|
+
survived. Here is what actually happened:
|
|
664
|
+
|
|
665
|
+
{feedback}
|
|
666
|
+
{loci}
|
|
667
|
+
Revise the function. Read the numbers literally: a rejected or repaired
|
|
668
|
+
candidate names the LOCUS it failed at and the value it tried, so fix that
|
|
669
|
+
locus rather than the sampler in general; duplicates mean the sampler is
|
|
670
|
+
collapsing; archive overlap means it keeps re-proposing configurations
|
|
671
|
+
already measured; no survivors means the region it concentrates on is not
|
|
672
|
+
competitive and the mass should move. Same rules as before: exactly this
|
|
673
|
+
signature
|
|
674
|
+
|
|
675
|
+
{contract}
|
|
676
|
+
|
|
677
|
+
exactly `n` configurations, every value from `domains` at that locus, the same
|
|
678
|
+
shape as the archive members, randomness derived from `seed`, standard library
|
|
679
|
+
only ({imports}), deterministic, no I/O.
|
|
680
|
+
{scaffold}{limits}
|
|
681
|
+
Reply with ONLY one fenced Python code block and no other text."""
|
|
682
|
+
|
|
683
|
+
|
|
684
|
+
GENERATOR_EVIDENCE_PROMPT = """You wrote this candidate generator for a \
|
|
685
|
+
black-box multi-objective optimizer:
|
|
686
|
+
|
|
687
|
+
```python
|
|
688
|
+
{source}
|
|
689
|
+
```
|
|
690
|
+
|
|
691
|
+
The optimizer has been running it and MEASURING what it drew. Here is the
|
|
692
|
+
evidence -- the run's own measurements, and nothing else:
|
|
693
|
+
|
|
694
|
+
{evidence}
|
|
695
|
+
|
|
696
|
+
Now reason about where to sample next, and rewrite the function so its mass
|
|
697
|
+
lands there. Concretely: which parameters do the measurements say actually
|
|
698
|
+
move the costs, and in which direction? Which regions has the run already
|
|
699
|
+
measured and found not competitive, so that re-proposing them wastes the rest
|
|
700
|
+
of the budget? Where is the front, and what is the smallest change to a front
|
|
701
|
+
member that has not been measured yet?
|
|
702
|
+
|
|
703
|
+
Your previous version was written before any of this was measured. It is not
|
|
704
|
+
being corrected for a defect -- it is being asked to use information that did
|
|
705
|
+
not exist when it was written. If the measurements do not support a change,
|
|
706
|
+
say so by returning a function that differs only where they do.
|
|
707
|
+
|
|
708
|
+
Same contract as before: exactly this signature
|
|
709
|
+
|
|
710
|
+
{contract}
|
|
711
|
+
|
|
712
|
+
exactly `n` configurations, every value taken from `domains` at that locus,
|
|
713
|
+
the same shape as the archive members, randomness derived from `seed`,
|
|
714
|
+
standard library only ({imports}), deterministic, no I/O.
|
|
715
|
+
{scaffold}{limits}
|
|
716
|
+
Reply with ONLY one fenced Python code block and no other text."""
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def _locus_block(domains: Optional[Mapping[str, Sequence[Any]]]) -> str:
|
|
720
|
+
"""The per-locus domain echo, as a prompt section (empty when unknown)."""
|
|
721
|
+
|
|
722
|
+
if not domains:
|
|
723
|
+
return ""
|
|
724
|
+
echo = render_domain_echo(domains)
|
|
725
|
+
if not echo:
|
|
726
|
+
return ""
|
|
727
|
+
return ("\nLOCI AND THEIR ADMISSIBLE VALUES (the exact `domains` mapping "
|
|
728
|
+
"you will be passed;\nthese key names ARE the shape -- a "
|
|
729
|
+
"configuration has exactly these loci and no others):\n"
|
|
730
|
+
f"{echo}\n")
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
def _scaffold_block(scaffold: bool) -> str:
|
|
734
|
+
return (SCAFFOLD_RULES + "\n") if scaffold else ""
|
|
735
|
+
|
|
736
|
+
|
|
737
|
+
def _limits_block(limits: Any, max_n: Optional[int]) -> str:
|
|
738
|
+
"""The sandbox's own budget, echoed (empty when the caller knows none)."""
|
|
739
|
+
|
|
740
|
+
if limits is None or not max_n:
|
|
741
|
+
return ""
|
|
742
|
+
try:
|
|
743
|
+
return LIMITS_RULES.format(
|
|
744
|
+
wall=f"{float(limits.wall_time_s):g}",
|
|
745
|
+
cpu=f"{float(limits.cpu_seconds):g}",
|
|
746
|
+
memory=int(int(limits.memory_bytes) / (1024 * 1024)),
|
|
747
|
+
max_n=int(max_n)) + "\n"
|
|
748
|
+
except (AttributeError, TypeError, ValueError):
|
|
749
|
+
return ""
|
|
750
|
+
|
|
751
|
+
|
|
752
|
+
def author_generator(
|
|
753
|
+
complete: Callable[[str], str],
|
|
754
|
+
*,
|
|
755
|
+
objectives: Sequence[ObjectiveSpec],
|
|
756
|
+
schema_text: str,
|
|
757
|
+
attempts: int = 2,
|
|
758
|
+
telemetry: Optional[AuthorTelemetry] = None,
|
|
759
|
+
domains: Optional[Mapping[str, Sequence[Any]]] = None,
|
|
760
|
+
scaffold: bool = True,
|
|
761
|
+
limits: Any = None,
|
|
762
|
+
max_n: Optional[int] = None,
|
|
763
|
+
) -> Optional[AuthoredArtifact]:
|
|
764
|
+
"""Ask the model to write ``propose``; accept whole or not at all.
|
|
765
|
+
|
|
766
|
+
*domains* is the per-locus admissible set the run will actually pass, so
|
|
767
|
+
the prompt can ECHO it rather than leave the model to infer per-position
|
|
768
|
+
domains from a field-level card -- the difference that turns an
|
|
769
|
+
out-of-domain value from a guess into a prompt failure. *scaffold*
|
|
770
|
+
advertises the in-sandbox emit harness (``build``), which is what makes a
|
|
771
|
+
shape error impossible to construct rather than caught after the fact.
|
|
772
|
+
"""
|
|
773
|
+
|
|
774
|
+
tel = telemetry if telemetry is not None else AuthorTelemetry()
|
|
775
|
+
contract = CONTRACTS["generator"]
|
|
776
|
+
prompt = GENERATOR_PROMPT.format(
|
|
777
|
+
goals="\n".join(f" {s.name}: {s.goal}imise" for s in objectives),
|
|
778
|
+
schema=schema_text,
|
|
779
|
+
loci=_locus_block(domains),
|
|
780
|
+
scaffold=_scaffold_block(scaffold),
|
|
781
|
+
limits=_limits_block(limits, max_n),
|
|
782
|
+
contract=contract.description,
|
|
783
|
+
imports=", ".join(sorted(ALLOWED_IMPORTS)),
|
|
784
|
+
)
|
|
785
|
+
return _author(complete, prompt, contract=contract, attempts=attempts,
|
|
786
|
+
telemetry=tel, name="llm_generator")
|
|
787
|
+
|
|
788
|
+
|
|
789
|
+
def revise_generator(
|
|
790
|
+
complete: Callable[[str], str],
|
|
791
|
+
*,
|
|
792
|
+
artifact: AuthoredArtifact,
|
|
793
|
+
feedback: str,
|
|
794
|
+
attempts: int = 1,
|
|
795
|
+
telemetry: Optional[AuthorTelemetry] = None,
|
|
796
|
+
domains: Optional[Mapping[str, Sequence[Any]]] = None,
|
|
797
|
+
scaffold: bool = True,
|
|
798
|
+
limits: Any = None,
|
|
799
|
+
max_n: Optional[int] = None,
|
|
800
|
+
) -> Optional[AuthoredArtifact]:
|
|
801
|
+
"""One revision round: the artifact, its measured behaviour, a rewrite.
|
|
802
|
+
|
|
803
|
+
The gate treats a revision exactly like a fresh authorship -- fenced
|
|
804
|
+
block only, import allowlist, correct entry point, whole-reply rejection
|
|
805
|
+
-- so a model that answers a revision with prose, or with code that
|
|
806
|
+
imports the filesystem, keeps the generator it already had.
|
|
807
|
+
"""
|
|
808
|
+
|
|
809
|
+
tel = telemetry if telemetry is not None else AuthorTelemetry()
|
|
810
|
+
contract = CONTRACTS["generator"]
|
|
811
|
+
prompt = GENERATOR_REVISION_PROMPT.format(
|
|
812
|
+
source=artifact.source,
|
|
813
|
+
feedback=feedback,
|
|
814
|
+
loci=_locus_block(domains),
|
|
815
|
+
scaffold=_scaffold_block(scaffold),
|
|
816
|
+
limits=_limits_block(limits, max_n),
|
|
817
|
+
contract=contract.description,
|
|
818
|
+
imports=", ".join(sorted(ALLOWED_IMPORTS)),
|
|
819
|
+
)
|
|
820
|
+
return _author(complete, prompt, contract=contract, attempts=attempts,
|
|
821
|
+
telemetry=tel, name=f"{artifact.name}_rev")
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
def reauthor_generator(
|
|
825
|
+
complete: Callable[[str], str],
|
|
826
|
+
*,
|
|
827
|
+
artifact: AuthoredArtifact,
|
|
828
|
+
evidence: str,
|
|
829
|
+
attempts: int = 1,
|
|
830
|
+
telemetry: Optional[AuthorTelemetry] = None,
|
|
831
|
+
scaffold: bool = True,
|
|
832
|
+
limits: Any = None,
|
|
833
|
+
max_n: Optional[int] = None,
|
|
834
|
+
) -> Optional[AuthoredArtifact]:
|
|
835
|
+
"""Re-author the sampler against the run's MEASURED trace.
|
|
836
|
+
|
|
837
|
+
Structurally identical to :func:`revise_generator` -- same contract, same
|
|
838
|
+
whole-reply gate, same degradation to the artifact already in hand -- and
|
|
839
|
+
different in the one way that matters: the prompt carries measurements
|
|
840
|
+
instead of emission counters, so the model is asked to reason about the
|
|
841
|
+
search rather than to repair its own output. The scaffold and resource
|
|
842
|
+
contracts are echoed exactly as they are for authorship and revision: a
|
|
843
|
+
re-authored sampler runs in the same sandbox as the one it replaces.
|
|
844
|
+
"""
|
|
845
|
+
|
|
846
|
+
tel = telemetry if telemetry is not None else AuthorTelemetry()
|
|
847
|
+
contract = CONTRACTS["generator"]
|
|
848
|
+
prompt = GENERATOR_EVIDENCE_PROMPT.format(
|
|
849
|
+
source=artifact.source,
|
|
850
|
+
evidence=evidence,
|
|
851
|
+
scaffold=_scaffold_block(scaffold),
|
|
852
|
+
limits=_limits_block(limits, max_n),
|
|
853
|
+
contract=contract.description,
|
|
854
|
+
imports=", ".join(sorted(ALLOWED_IMPORTS)),
|
|
855
|
+
)
|
|
856
|
+
return _author(complete, prompt, contract=contract, attempts=attempts,
|
|
857
|
+
telemetry=tel, name=f"{artifact.name}_evidence")
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
def _author(complete, prompt, *, contract, attempts, telemetry, name):
|
|
861
|
+
for _attempt in range(max(1, attempts)):
|
|
862
|
+
telemetry.calls += 1
|
|
863
|
+
try:
|
|
864
|
+
text = complete(prompt)
|
|
865
|
+
except Exception:
|
|
866
|
+
telemetry.errors += 1
|
|
867
|
+
continue
|
|
868
|
+
source = accept_block(text, contract=contract, telemetry=telemetry)
|
|
869
|
+
if source is None:
|
|
870
|
+
continue
|
|
871
|
+
telemetry.accepted += 1
|
|
872
|
+
telemetry.sources.append(source)
|
|
873
|
+
return authored_artifact(contract.kind, source, name=name,
|
|
874
|
+
authored_by="llm")
|
|
875
|
+
return None
|
|
876
|
+
|
|
877
|
+
|
|
878
|
+
#: How many offending loci one feedback block names before it stops. A
|
|
879
|
+
#: revision cannot act on four hundred addresses; it can act on the worst few.
|
|
880
|
+
FEEDBACK_LOCI = 6
|
|
881
|
+
|
|
882
|
+
|
|
883
|
+
def _defect_lines(
|
|
884
|
+
census: Optional[RejectionCensus],
|
|
885
|
+
domains: Optional[Mapping[str, Sequence[Any]]] = None,
|
|
886
|
+
) -> List[str]:
|
|
887
|
+
"""WHICH loci failed and WHY, worst first, each with a real sample."""
|
|
888
|
+
|
|
889
|
+
if census is None or census.empty:
|
|
890
|
+
return []
|
|
891
|
+
lines: List[str] = [" WHERE IT FAILED (the harness's own addresses):"]
|
|
892
|
+
for reason, count in sorted(census.shape_reasons.items(),
|
|
893
|
+
key=lambda kv: -kv[1])[:FEEDBACK_LOCI]:
|
|
894
|
+
sample = census.samples.get(f"shape:{reason}")
|
|
895
|
+
lines.append(f" shape -- {reason}: {count} candidate(s); "
|
|
896
|
+
f"you emitted {json_compact(sample)}")
|
|
897
|
+
ranked = sorted(census.out_of_domain_by_locus.items(),
|
|
898
|
+
key=lambda kv: -kv[1])
|
|
899
|
+
for locus, count in ranked[:FEEDBACK_LOCI]:
|
|
900
|
+
sample = census.samples.get(f"domain:{locus}")
|
|
901
|
+
allowed = list((domains or {}).get(locus) or ())
|
|
902
|
+
rendered = (f"; its domain is {allowed[:8]}"
|
|
903
|
+
+ (f" ({len(allowed)} values)" if len(allowed) > 8 else "")
|
|
904
|
+
) if allowed else ""
|
|
905
|
+
lines.append(f" locus {locus} -- out of domain {count} time(s); "
|
|
906
|
+
f"you used {json_compact(sample)}{rendered}")
|
|
907
|
+
if len(ranked) > FEEDBACK_LOCI:
|
|
908
|
+
lines.append(f" ... and {len(ranked) - FEEDBACK_LOCI} further loci "
|
|
909
|
+
f"out of domain")
|
|
910
|
+
repaired = sorted(census.repaired_by_locus.items(), key=lambda kv: -kv[1])
|
|
911
|
+
if repaired:
|
|
912
|
+
worst = ", ".join(f"{locus} ({count})"
|
|
913
|
+
for locus, count in repaired[:FEEDBACK_LOCI])
|
|
914
|
+
lines.append(f" the harness had to DECIDE these loci for you: "
|
|
915
|
+
f"{worst}")
|
|
916
|
+
return lines
|
|
917
|
+
|
|
918
|
+
|
|
919
|
+
def _edit_lines(rejected_edits: Sequence[Mapping[str, Any]]) -> List[str]:
|
|
920
|
+
"""The rejected-edit memory: fixes already tried that did not fix it.
|
|
921
|
+
|
|
922
|
+
Wave D measured revision firing on 77 of 80 cells on the broken
|
|
923
|
+
instances and repairing none of them. A revision loop with no memory of
|
|
924
|
+
its own failures can only re-propose them; naming the edit, the defect it
|
|
925
|
+
was supposed to fix, and the fact that the defect survived it is the
|
|
926
|
+
cheapest thing that makes the next attempt different.
|
|
927
|
+
"""
|
|
928
|
+
|
|
929
|
+
if not rejected_edits:
|
|
930
|
+
return []
|
|
931
|
+
lines = [" EDITS ALREADY TRIED THAT DID NOT FIX THIS -- do not repeat "
|
|
932
|
+
"them or anything equivalent:"]
|
|
933
|
+
for edit in rejected_edits[-3:]:
|
|
934
|
+
lines.append(
|
|
935
|
+
f" revision {edit.get('revision')} (source {edit.get('sha')}): "
|
|
936
|
+
f"defect rate {float(edit.get('before', 0.0)):.0%} -> "
|
|
937
|
+
f"{float(edit.get('after', 0.0)):.0%}, and the same loci still "
|
|
938
|
+
f"fail ({edit.get('signature')}).")
|
|
939
|
+
excerpt = str(edit.get("excerpt") or "").strip()
|
|
940
|
+
if excerpt:
|
|
941
|
+
lines.append(" it looked like: "
|
|
942
|
+
+ " ".join(excerpt.split())[:240])
|
|
943
|
+
return lines
|
|
944
|
+
|
|
945
|
+
|
|
946
|
+
def render_generation_feedback(
|
|
947
|
+
telemetry: GeneratorTelemetry,
|
|
948
|
+
last: Optional[PoolReport] = None,
|
|
949
|
+
survivors: Sequence[Tuple[Config, Mapping[str, float]]] = (),
|
|
950
|
+
*,
|
|
951
|
+
census: Optional[RejectionCensus] = None,
|
|
952
|
+
domains: Optional[Mapping[str, Sequence[Any]]] = None,
|
|
953
|
+
rejected_edits: Sequence[Mapping[str, Any]] = (),
|
|
954
|
+
) -> str:
|
|
955
|
+
"""The measured story a generator revision needs, as text.
|
|
956
|
+
|
|
957
|
+
Three things, all counted: how much of what it emitted the harness could
|
|
958
|
+
use, how much of it was novel, and what happened to the candidates that
|
|
959
|
+
were measured. Survivors are shown rather than "the best" -- ranking
|
|
960
|
+
candidates across objectives would need weights nobody declared, while
|
|
961
|
+
surviving truncation is the run's own weight-free verdict.
|
|
962
|
+
|
|
963
|
+
Then the two things Wave D's counters could not say, and without which
|
|
964
|
+
revision repaired nothing: WHICH locus rejected and WHY, with a value the
|
|
965
|
+
model will recognise as its own (*census*, *domains*), and which edits
|
|
966
|
+
have already been tried against this same defect and failed
|
|
967
|
+
(*rejected_edits*).
|
|
968
|
+
"""
|
|
969
|
+
|
|
970
|
+
lines = [
|
|
971
|
+
f" batches generated: {telemetry.batches}"
|
|
972
|
+
f" (runtime failures: {telemetry.runtime_failures})",
|
|
973
|
+
f" candidates emitted: {telemetry.emitted}; "
|
|
974
|
+
f"accepted by the harness: {telemetry.accepted}",
|
|
975
|
+
f" rejected -- wrong shape: {telemetry.rejected_shape}; "
|
|
976
|
+
f"value outside its declared domain: "
|
|
977
|
+
f"{telemetry.rejected_out_of_domain}",
|
|
978
|
+
f" dropped -- duplicate within the batch: {telemetry.duplicates}; "
|
|
979
|
+
f"already measured in this run: {telemetry.archive_overlap}",
|
|
980
|
+
f" pool slots the harness had to fill with uniform random draws: "
|
|
981
|
+
f"{telemetry.filled_uniform}",
|
|
982
|
+
f" of yours that were measured: {telemetry.measured}; "
|
|
983
|
+
f"survived selection into the next population: {telemetry.survived}",
|
|
984
|
+
]
|
|
985
|
+
if telemetry.repaired or telemetry.repaired_loci:
|
|
986
|
+
lines.append(
|
|
987
|
+
f" candidates the harness had to ASSEMBLE for you rather than "
|
|
988
|
+
f"reject: {telemetry.repaired} "
|
|
989
|
+
f"({telemetry.repaired_loci} individual loci decided for you)")
|
|
990
|
+
if (telemetry.scaffold_filled or telemetry.scaffold_out_of_domain
|
|
991
|
+
or telemetry.scaffold_unknown_locus):
|
|
992
|
+
lines.append(
|
|
993
|
+
f" inside your own code, `build` filled "
|
|
994
|
+
f"{telemetry.scaffold_filled} locus/loci you left unset, "
|
|
995
|
+
f"overrode {telemetry.scaffold_out_of_domain} out-of-domain "
|
|
996
|
+
f"value(s) and ignored {telemetry.scaffold_unknown_locus} locus "
|
|
997
|
+
f"name(s) the schema does not declare")
|
|
998
|
+
if last is not None and last.emitted:
|
|
999
|
+
lines.append(
|
|
1000
|
+
f" most recent batch: {last.duplicate_rate:.0%} duplicates, "
|
|
1001
|
+
f"{last.archive_overlap_rate:.0%} already measured, "
|
|
1002
|
+
f"{last.novelty_rate:.0%} usable and new")
|
|
1003
|
+
lines.extend(_defect_lines(
|
|
1004
|
+
census if census is not None
|
|
1005
|
+
else (last.census if last is not None else None), domains))
|
|
1006
|
+
lines.extend(_edit_lines(rejected_edits))
|
|
1007
|
+
for config, objectives in survivors:
|
|
1008
|
+
rendered = ", ".join(f"{k}={float(v):.6g}"
|
|
1009
|
+
for k, v in sorted(objectives.items()))
|
|
1010
|
+
lines.append(f" survived: {json_compact(config)} -> {rendered}")
|
|
1011
|
+
return "\n".join(lines)
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
class AuthoredGenerator:
|
|
1015
|
+
"""The authored sampler as a loop policy: mass generation, then the guard.
|
|
1016
|
+
|
|
1017
|
+
One call per generation produces the whole pool out of process; the
|
|
1018
|
+
harness validates it, drops what collapsed, fills any shortfall
|
|
1019
|
+
schema-uniformly, and hands back exactly ``pool_for(want)``
|
|
1020
|
+
configurations. Nothing here can reach an evaluation: the pool is
|
|
1021
|
+
candidates, and the loop measures at most the ``want`` it could already
|
|
1022
|
+
afford.
|
|
1023
|
+
"""
|
|
1024
|
+
|
|
1025
|
+
def __init__(
|
|
1026
|
+
self,
|
|
1027
|
+
artifact: AuthoredArtifact,
|
|
1028
|
+
runtime: Any,
|
|
1029
|
+
*,
|
|
1030
|
+
pool_factor: int = 4,
|
|
1031
|
+
pool_size: int = 0,
|
|
1032
|
+
max_pool: int = MAX_POOL,
|
|
1033
|
+
archive_shown: int = ARCHIVE_SHOWN,
|
|
1034
|
+
revise: Optional[Callable[[AuthoredArtifact, str],
|
|
1035
|
+
Optional[AuthoredArtifact]]] = None,
|
|
1036
|
+
max_revisions: int = 1,
|
|
1037
|
+
min_measured_for_revision: int = 4,
|
|
1038
|
+
min_novelty: float = 0.5,
|
|
1039
|
+
scaffold: bool = True,
|
|
1040
|
+
repair: bool = True,
|
|
1041
|
+
revision_guard: bool = False,
|
|
1042
|
+
shrink_on_overrun: int = 4,
|
|
1043
|
+
objectives: Sequence[ObjectiveSpec] = (),
|
|
1044
|
+
reauthor: Optional[Callable[[AuthoredArtifact, str],
|
|
1045
|
+
Optional[AuthoredArtifact]]] = None,
|
|
1046
|
+
reauthor_every: int = 0,
|
|
1047
|
+
max_reauthorings: int = 0,
|
|
1048
|
+
evidence_view: Optional[Callable[[Sequence[MeasuredRow]],
|
|
1049
|
+
Sequence[MeasuredRow]]] = None,
|
|
1050
|
+
evidence_min_rows: int = 0,
|
|
1051
|
+
evidence_front_shown: int = 8,
|
|
1052
|
+
evidence_effects_shown: int = 8,
|
|
1053
|
+
prior_author: Optional[Callable[[str], str]] = None,
|
|
1054
|
+
max_priors: int = 1,
|
|
1055
|
+
prior_max_weight_ratio: float = 8.0,
|
|
1056
|
+
prior_unwind_batches: int = 2,
|
|
1057
|
+
) -> None:
|
|
1058
|
+
if pool_factor < 1:
|
|
1059
|
+
raise ValueError(f"pool_factor must be at least 1, got {pool_factor}")
|
|
1060
|
+
if pool_size < 0:
|
|
1061
|
+
raise ValueError(f"pool_size must be non-negative, got {pool_size}")
|
|
1062
|
+
if reauthor_every < 0:
|
|
1063
|
+
raise ValueError(
|
|
1064
|
+
f"reauthor_every must be non-negative, got {reauthor_every}")
|
|
1065
|
+
if evidence_min_rows < 0:
|
|
1066
|
+
raise ValueError(
|
|
1067
|
+
"evidence_min_rows is the fewest measured rows the evidence "
|
|
1068
|
+
"channel will author from and must be non-negative, got "
|
|
1069
|
+
f"{evidence_min_rows}")
|
|
1070
|
+
if prior_max_weight_ratio < 1.0:
|
|
1071
|
+
raise ValueError(
|
|
1072
|
+
"prior_max_weight_ratio caps a graded prior's concentration "
|
|
1073
|
+
f"and must be at least 1, got {prior_max_weight_ratio}")
|
|
1074
|
+
if prior_author is not None and reauthor_every <= 0:
|
|
1075
|
+
# The evidence channel has one cadence and both consumers ride it.
|
|
1076
|
+
# A prior asked for on no cadence would fire never or every
|
|
1077
|
+
# generation depending on who read the code, which is exactly the
|
|
1078
|
+
# magic number this knob exists to replace.
|
|
1079
|
+
raise ValueError(
|
|
1080
|
+
"a locus prior is authored from the measured trace on the "
|
|
1081
|
+
"reauthor_every cadence; set reauthor_every > 0")
|
|
1082
|
+
self.artifact = artifact
|
|
1083
|
+
self.runtime = runtime
|
|
1084
|
+
self.pool_factor = int(pool_factor)
|
|
1085
|
+
self.pool_size = int(pool_size)
|
|
1086
|
+
self.max_pool = int(max_pool)
|
|
1087
|
+
self.archive_shown = int(archive_shown)
|
|
1088
|
+
self.revise = revise
|
|
1089
|
+
self.max_revisions = int(max_revisions)
|
|
1090
|
+
self.min_measured_for_revision = int(min_measured_for_revision)
|
|
1091
|
+
self.min_novelty = float(min_novelty)
|
|
1092
|
+
#: Ship the emit harness into the sandbox, so the authored code
|
|
1093
|
+
#: constructs candidates locus by locus instead of transcribing a
|
|
1094
|
+
#: shape. Shape was 6,021 of 7,104 emissions on `upms_j14_m3`.
|
|
1095
|
+
self.scaffold = bool(scaffold)
|
|
1096
|
+
#: Assemble a candidate out of whatever the emission got right rather
|
|
1097
|
+
#: than dropping it whole -- a per-LOCUS fallback in place of a
|
|
1098
|
+
#: per-CANDIDATE one. Off restores the sealed-row semantics exactly.
|
|
1099
|
+
self.repair = bool(repair)
|
|
1100
|
+
#: Keep a revision only if a frozen replay says it MEASURABLY helped
|
|
1101
|
+
#: (see :meth:`_guard_admits`). Off by default: the one-shot and
|
|
1102
|
+
#: capped-revision arms are what every sealed row is defined on.
|
|
1103
|
+
self.revision_guard = bool(revision_guard)
|
|
1104
|
+
#: Divisor for the one retry a resource overrun gets. 0 or 1 disables
|
|
1105
|
+
#: it and a timeout costs the whole pool, as it did when Wave D
|
|
1106
|
+
#: measured 39,993 uniformly-filled slots against 7,104 emissions.
|
|
1107
|
+
self.shrink_on_overrun = int(shrink_on_overrun)
|
|
1108
|
+
#: The MEASUREMENT-CONDITIONED channel. ``reauthor_every`` is a
|
|
1109
|
+
#: cadence in CHARGED EVALUATIONS -- a declared, typed knob rather
|
|
1110
|
+
#: than a magic number -- and ``0`` (the default) is the seam every
|
|
1111
|
+
#: sealed row to date ran: no evidence call ever fires.
|
|
1112
|
+
self.objectives = tuple(objectives)
|
|
1113
|
+
self.reauthor = reauthor
|
|
1114
|
+
self.reauthor_every = int(reauthor_every)
|
|
1115
|
+
self.max_reauthorings = int(max_reauthorings)
|
|
1116
|
+
#: What the model is SHOWN. The identity view is the product; a view
|
|
1117
|
+
#: returning another run's rows is the shuffled-evidence control, and
|
|
1118
|
+
#: it is a parameter rather than a patch precisely so the control can
|
|
1119
|
+
#: be built without editing this file.
|
|
1120
|
+
self.evidence_view = evidence_view
|
|
1121
|
+
#: WHEN the channel may speak for the FIRST time, in measured rows.
|
|
1122
|
+
#: The cadence above says how often an evidence call recurs; it cannot
|
|
1123
|
+
#: also say when the first one is allowed, because a run holds
|
|
1124
|
+
#: measurements before this generator has produced any (the initial
|
|
1125
|
+
#: population is the whole of the evidence at generation 1) and a
|
|
1126
|
+
#: cadence read as "wait for that many of MY OWN children" makes the
|
|
1127
|
+
#: channel arrive generations after the evidence did -- the W11 defect.
|
|
1128
|
+
#: ``0`` (the default) means AS SOON AS THE GATE CAN BE MET:
|
|
1129
|
+
#: :data:`~agent_evolve.policies.measurement_evidence.MIN_EVIDENCE_ROWS`
|
|
1130
|
+
#: rows, the fewest a determinable effect can be computed from. Setting
|
|
1131
|
+
#: it equal to ``reauthor_every`` restores the pure-cadence rule
|
|
1132
|
+
#: exactly, so the older behaviour is a declared configuration rather
|
|
1133
|
+
#: than a lost one.
|
|
1134
|
+
self.evidence_min_rows = (int(evidence_min_rows) if evidence_min_rows
|
|
1135
|
+
else MIN_EVIDENCE_ROWS)
|
|
1136
|
+
self.evidence_front_shown = int(evidence_front_shown)
|
|
1137
|
+
self.evidence_effects_shown = int(evidence_effects_shown)
|
|
1138
|
+
#: The locus-importance channel, and the gate that refuses it.
|
|
1139
|
+
self.prior_author = prior_author
|
|
1140
|
+
self.max_priors = int(max_priors)
|
|
1141
|
+
self.prior_max_weight_ratio = float(prior_max_weight_ratio)
|
|
1142
|
+
self.prior_unwind_batches = int(prior_unwind_batches)
|
|
1143
|
+
self.telemetry = GeneratorTelemetry()
|
|
1144
|
+
self.mechanism = "authored_generator"
|
|
1145
|
+
self.authored_by = artifact.authored_by
|
|
1146
|
+
self.last_report: Optional[PoolReport] = None
|
|
1147
|
+
self.census = RejectionCensus()
|
|
1148
|
+
self._seen: set[str] = set()
|
|
1149
|
+
self._survivors: List[Tuple[Config, Mapping[str, float]]] = []
|
|
1150
|
+
self._domains: Dict[str, List[Any]] = {}
|
|
1151
|
+
self._last_call: Optional[Tuple[List[Config], int, Dict[str, List[Any]],
|
|
1152
|
+
int, Config]] = None
|
|
1153
|
+
self._rejected_edits: List[Dict[str, Any]] = []
|
|
1154
|
+
self._pending_edit: Optional[Dict[str, Any]] = None
|
|
1155
|
+
#: The measured trace, in measurement order. This is the evidence, and
|
|
1156
|
+
#: it is the ONLY thing this class knows about outcomes: it arrives
|
|
1157
|
+
#: through `note_measured` -- which the loop calls for EVERY charged
|
|
1158
|
+
#: evaluation it holds, whoever produced it -- so the generator still
|
|
1159
|
+
#: never sees the problem, the evaluator or the cache.
|
|
1160
|
+
self._rows: List[MeasuredRow] = []
|
|
1161
|
+
self._evidence_at = 0
|
|
1162
|
+
#: How many evidence ticks have fired. The first one is gated on
|
|
1163
|
+
#: evidence (``evidence_min_rows``); every later one on the cadence.
|
|
1164
|
+
self._evidence_ticks = 0
|
|
1165
|
+
self._prior_asked_at = -1
|
|
1166
|
+
self._prior: Any = None
|
|
1167
|
+
self._prior_batches = 0
|
|
1168
|
+
self._survived_at_prior = 0
|
|
1169
|
+
#: One record per evidence-conditioned call: what was shown (digest
|
|
1170
|
+
#: and row count), what came back, and whether it was accepted.
|
|
1171
|
+
#: Telemetry as correctness -- a run cannot claim this channel fired
|
|
1172
|
+
#: without the record that says what it saw.
|
|
1173
|
+
self.evidence_log: List[Dict[str, Any]] = []
|
|
1174
|
+
self._noted = 0
|
|
1175
|
+
|
|
1176
|
+
# -- sizing -------------------------------------------------------------
|
|
1177
|
+
|
|
1178
|
+
def pool_for(self, want: int) -> int:
|
|
1179
|
+
"""How many candidates one generation asks for. Never below *want*."""
|
|
1180
|
+
|
|
1181
|
+
size = self.pool_size or self.pool_factor * max(1, want)
|
|
1182
|
+
return max(1, min(self.max_pool, max(int(want), size)))
|
|
1183
|
+
|
|
1184
|
+
# -- the archive it is conditioned on and measured against --------------
|
|
1185
|
+
|
|
1186
|
+
def note_archive(self, configs: Sequence[Config]) -> None:
|
|
1187
|
+
"""Record configurations the run has measured. No quality claim."""
|
|
1188
|
+
|
|
1189
|
+
for config in configs:
|
|
1190
|
+
self._seen.add(candidate_key(config))
|
|
1191
|
+
|
|
1192
|
+
def note_measured(
|
|
1193
|
+
self,
|
|
1194
|
+
config: Config,
|
|
1195
|
+
*,
|
|
1196
|
+
objectives: Mapping[str, float],
|
|
1197
|
+
survived: bool = False,
|
|
1198
|
+
) -> None:
|
|
1199
|
+
"""One charged evaluation the RUN holds. Evidence, not attribution.
|
|
1200
|
+
|
|
1201
|
+
Evidence is *what has been measured*, not *what this component
|
|
1202
|
+
produced*. The distinction is the whole of the W11 fix: the initial
|
|
1203
|
+
population, and anything else the loop charged before or beside this
|
|
1204
|
+
generator, is measurement the model can reason over and must be shown
|
|
1205
|
+
-- while the survival counters that decide whether the generator is
|
|
1206
|
+
deficient stay credited to its own children alone (:meth:
|
|
1207
|
+
`record_measured`). Conflating the two either blinds the channel for
|
|
1208
|
+
two generations or forges the attribution; keeping them apart costs
|
|
1209
|
+
one method.
|
|
1210
|
+
|
|
1211
|
+
The row is kept whether or not it survived, because "this region was
|
|
1212
|
+
measured and is NOT competitive" is exactly the evidence a survivor
|
|
1213
|
+
list cannot carry.
|
|
1214
|
+
|
|
1215
|
+
It does NOT touch the novelty ledger. What the run has already tried is
|
|
1216
|
+
``note_archive``'s declared job, the loop already calls it, and having
|
|
1217
|
+
a second entry point quietly feed the same set would change the
|
|
1218
|
+
duplicate and overlap counters of runs with this channel OFF -- which
|
|
1219
|
+
must stay byte-identical to the sealed seam.
|
|
1220
|
+
"""
|
|
1221
|
+
|
|
1222
|
+
self._rows.append((dict(config), dict(objectives), bool(survived)))
|
|
1223
|
+
|
|
1224
|
+
def record_measured(
|
|
1225
|
+
self,
|
|
1226
|
+
config: Config,
|
|
1227
|
+
*,
|
|
1228
|
+
survived: bool,
|
|
1229
|
+
objectives: Optional[Mapping[str, float]] = None,
|
|
1230
|
+
) -> None:
|
|
1231
|
+
"""Credit at survival time, exactly as the operator portfolio does.
|
|
1232
|
+
|
|
1233
|
+
This generator's OWN child: it is both evidence and attribution, so
|
|
1234
|
+
the row is noted and the counters that judge this generator move.
|
|
1235
|
+
"""
|
|
1236
|
+
|
|
1237
|
+
self._seen.add(candidate_key(config))
|
|
1238
|
+
self.telemetry.measured += 1
|
|
1239
|
+
if objectives is not None:
|
|
1240
|
+
self.note_measured(config, objectives=objectives,
|
|
1241
|
+
survived=survived)
|
|
1242
|
+
if survived:
|
|
1243
|
+
self.telemetry.survived += 1
|
|
1244
|
+
if objectives is not None:
|
|
1245
|
+
self._survivors.append((dict(config), dict(objectives)))
|
|
1246
|
+
del self._survivors[:-3]
|
|
1247
|
+
|
|
1248
|
+
# -- generation ---------------------------------------------------------
|
|
1249
|
+
|
|
1250
|
+
def propose(
|
|
1251
|
+
self,
|
|
1252
|
+
*,
|
|
1253
|
+
template: Config,
|
|
1254
|
+
candidate_model: Any,
|
|
1255
|
+
restriction: Any,
|
|
1256
|
+
archive: Sequence[Config],
|
|
1257
|
+
want: int,
|
|
1258
|
+
rng: random.Random,
|
|
1259
|
+
seed: int = 0,
|
|
1260
|
+
) -> List[Config]:
|
|
1261
|
+
"""A validated pool of ``pool_for(want)`` configurations.
|
|
1262
|
+
|
|
1263
|
+
The head of the pool is what the loop would measure with no screen at
|
|
1264
|
+
all, so a screen that follows keeps its exploration floor over the
|
|
1265
|
+
generator's OWN first picks rather than over an unrelated draw.
|
|
1266
|
+
"""
|
|
1267
|
+
|
|
1268
|
+
n = self.pool_for(want)
|
|
1269
|
+
declared = {
|
|
1270
|
+
str(locus): list(locus_domain(candidate_model, locus,
|
|
1271
|
+
restriction=restriction))
|
|
1272
|
+
for locus in loci_of(template)
|
|
1273
|
+
}
|
|
1274
|
+
self._domains = declared
|
|
1275
|
+
# Search-progress triggers, on the DECLARED domains: the evidence and
|
|
1276
|
+
# the prior both describe the space the problem published, not a space
|
|
1277
|
+
# a previous prior already narrowed, or a second prior would compound
|
|
1278
|
+
# the first one's bet without ever measuring it.
|
|
1279
|
+
# ONE cadence tick, read once and consumed by both channels: asking
|
|
1280
|
+
# each of them separately would let whichever ran first advance the
|
|
1281
|
+
# anchor and starve the other, which is a rule nobody declared.
|
|
1282
|
+
due = self._due()
|
|
1283
|
+
self._maybe_reauthor(declared, due)
|
|
1284
|
+
self._maybe_author_prior(declared, due)
|
|
1285
|
+
if due:
|
|
1286
|
+
self._evidence_at = len(self._rows)
|
|
1287
|
+
self._evidence_ticks += 1
|
|
1288
|
+
domains = self._effective_domains(declared)
|
|
1289
|
+
shown = [dict(config) for config in list(archive)[:self.archive_shown]]
|
|
1290
|
+
self._maybe_revise()
|
|
1291
|
+
self._last_call = (shown, n, domains, int(seed), dict(template))
|
|
1292
|
+
|
|
1293
|
+
self.telemetry.batches += 1
|
|
1294
|
+
# The generator SAMPLES from the (possibly biased) domains and is
|
|
1295
|
+
# VALIDATED against the declared ones. A prior is guidance about where
|
|
1296
|
+
# to spend, not a new definition of what is legal, so a candidate
|
|
1297
|
+
# outside the prior but inside the schema is admitted rather than
|
|
1298
|
+
# counted as a defect -- otherwise installing a prior would
|
|
1299
|
+
# manufacture rejections and fire the defect-repair channel on a
|
|
1300
|
+
# generator that did exactly what it was asked.
|
|
1301
|
+
report = self._run(self.artifact, shown, n, domains, seed,
|
|
1302
|
+
template=template, rng=rng, count=True,
|
|
1303
|
+
validate_domains=declared)
|
|
1304
|
+
self.last_report = report
|
|
1305
|
+
self.census.merge(report.census)
|
|
1306
|
+
self._score_pending_edit(report)
|
|
1307
|
+
self.telemetry.emitted += report.emitted
|
|
1308
|
+
self.telemetry.accepted += len(report.accepted)
|
|
1309
|
+
self.telemetry.rejected_shape += report.rejected_shape
|
|
1310
|
+
self.telemetry.rejected_out_of_domain += report.rejected_out_of_domain
|
|
1311
|
+
self.telemetry.duplicates += report.duplicates
|
|
1312
|
+
self.telemetry.archive_overlap += report.archive_overlap
|
|
1313
|
+
self.telemetry.repaired += report.repaired
|
|
1314
|
+
self.telemetry.repaired_loci += report.repaired_loci
|
|
1315
|
+
|
|
1316
|
+
pool = [dict(config) for config in report.accepted[:n]]
|
|
1317
|
+
while len(pool) < n:
|
|
1318
|
+
pool.append(uniform_candidate(template, candidate_model, rng=rng,
|
|
1319
|
+
restriction=restriction))
|
|
1320
|
+
self.telemetry.filled_uniform += 1
|
|
1321
|
+
return pool
|
|
1322
|
+
|
|
1323
|
+
def _run(self, artifact, archive, n, domains, seed, *, template, rng,
|
|
1324
|
+
count: bool,
|
|
1325
|
+
validate_domains: Optional[Mapping[str, Sequence[Any]]] = None,
|
|
1326
|
+
) -> PoolReport:
|
|
1327
|
+
"""One emission through the scaffold, validated. Optionally counted.
|
|
1328
|
+
|
|
1329
|
+
*count* is false for the revision guard's frozen replay, which must
|
|
1330
|
+
measure a challenger without the run's telemetry recording an
|
|
1331
|
+
emission the loop never saw.
|
|
1332
|
+
|
|
1333
|
+
*validate_domains*, when given, is what the pool is judged against --
|
|
1334
|
+
the DECLARED domains -- while *domains* is what the sampler draws
|
|
1335
|
+
from (a weighted prior may have biased them). Sampling guidance must
|
|
1336
|
+
never redefine what is legal.
|
|
1337
|
+
"""
|
|
1338
|
+
|
|
1339
|
+
rows = self._emit(artifact, archive, n, domains, seed,
|
|
1340
|
+
template=template, count=count)
|
|
1341
|
+
return validate_pool(
|
|
1342
|
+
rows, template=template,
|
|
1343
|
+
domains=validate_domains if validate_domains is not None else domains,
|
|
1344
|
+
seen=self._seen, limit=n, repair=self.repair,
|
|
1345
|
+
rng=rng if rng is not None else random.Random(seed))
|
|
1346
|
+
|
|
1347
|
+
def _emit(self, artifact, archive, n, domains, seed, *, template,
|
|
1348
|
+
count: bool = True) -> Any:
|
|
1349
|
+
prelude = (scaffold_prelude(template, domains, nonce=int(seed))
|
|
1350
|
+
if self.scaffold else None)
|
|
1351
|
+
try:
|
|
1352
|
+
outcome = self.runtime.call(
|
|
1353
|
+
artifact, [[archive, int(n), domains, int(seed)]],
|
|
1354
|
+
prelude=prelude, notes_global=NOTES_GLOBAL)
|
|
1355
|
+
except TypeError:
|
|
1356
|
+
# A runtime that predates the prelude channel: the artifact still
|
|
1357
|
+
# runs, the scaffold simply is not there, and the harness-side
|
|
1358
|
+
# repair remains the only guard. Degrade, never fail.
|
|
1359
|
+
try:
|
|
1360
|
+
outcome = self.runtime.call(
|
|
1361
|
+
artifact, [[archive, int(n), domains, int(seed)]])
|
|
1362
|
+
except Exception:
|
|
1363
|
+
if count:
|
|
1364
|
+
self.telemetry.runtime_failures += 1
|
|
1365
|
+
return []
|
|
1366
|
+
except Exception: # a runtime that cannot even ship
|
|
1367
|
+
if count: # the call is a countable
|
|
1368
|
+
self.telemetry.runtime_failures += 1 # event, not an emergency
|
|
1369
|
+
return []
|
|
1370
|
+
if count:
|
|
1371
|
+
self._absorb_notes(getattr(outcome, "notes", None))
|
|
1372
|
+
if (not outcome.ok and self.shrink_on_overrun > 1
|
|
1373
|
+
and outcome.status in ("timeout", "memory")
|
|
1374
|
+
and int(n) > self.shrink_on_overrun):
|
|
1375
|
+
# A resource overrun is the one failure whose CAUSE the harness
|
|
1376
|
+
# can act on: `propose` is a distribution, so asking it for fewer
|
|
1377
|
+
# draws is the same request at a fraction of the work. A quarter
|
|
1378
|
+
# of a guided pool beats none of one, the shortfall still falls
|
|
1379
|
+
# back to schema-uniform, and both events stay counted.
|
|
1380
|
+
if count:
|
|
1381
|
+
self.telemetry.runtime_failures += 1
|
|
1382
|
+
self.telemetry.runtime_retries += 1
|
|
1383
|
+
smaller = max(1, int(n) // self.shrink_on_overrun)
|
|
1384
|
+
try:
|
|
1385
|
+
outcome = self.runtime.call(
|
|
1386
|
+
artifact, [[archive, smaller, domains, int(seed)]],
|
|
1387
|
+
prelude=prelude, notes_global=NOTES_GLOBAL)
|
|
1388
|
+
except Exception:
|
|
1389
|
+
return []
|
|
1390
|
+
if count and outcome.ok:
|
|
1391
|
+
self.telemetry.runtime_recovered += 1
|
|
1392
|
+
self._absorb_notes(getattr(outcome, "notes", None))
|
|
1393
|
+
if not outcome.ok:
|
|
1394
|
+
return []
|
|
1395
|
+
[rows] = outcome.results
|
|
1396
|
+
return rows if isinstance(rows, list) else []
|
|
1397
|
+
if not outcome.ok:
|
|
1398
|
+
if count:
|
|
1399
|
+
self.telemetry.runtime_failures += 1
|
|
1400
|
+
return []
|
|
1401
|
+
[rows] = outcome.results
|
|
1402
|
+
if not isinstance(rows, list):
|
|
1403
|
+
if count:
|
|
1404
|
+
self.telemetry.runtime_failures += 1
|
|
1405
|
+
return []
|
|
1406
|
+
return rows
|
|
1407
|
+
|
|
1408
|
+
def _absorb_notes(self, notes: Any) -> None:
|
|
1409
|
+
"""The scaffold's own counters, from inside the sandbox."""
|
|
1410
|
+
|
|
1411
|
+
if not isinstance(notes, Mapping):
|
|
1412
|
+
return
|
|
1413
|
+
self.telemetry.scaffold_filled += int(notes.get("filled") or 0)
|
|
1414
|
+
self.telemetry.scaffold_out_of_domain += int(
|
|
1415
|
+
notes.get("out_of_domain") or 0)
|
|
1416
|
+
self.telemetry.scaffold_unknown_locus += int(
|
|
1417
|
+
notes.get("unknown_locus") or 0)
|
|
1418
|
+
by_locus = notes.get("by_locus")
|
|
1419
|
+
if isinstance(by_locus, Mapping):
|
|
1420
|
+
for locus, row in by_locus.items():
|
|
1421
|
+
if not isinstance(row, Mapping):
|
|
1422
|
+
continue
|
|
1423
|
+
name = str(locus)
|
|
1424
|
+
for _ in range(int(row.get("filled") or 0)):
|
|
1425
|
+
self.census.repaired(name)
|
|
1426
|
+
count = int(row.get("out_of_domain") or 0)
|
|
1427
|
+
if count:
|
|
1428
|
+
self.census.out_of_domain_by_locus[name] = (
|
|
1429
|
+
self.census.out_of_domain_by_locus.get(name, 0) + count)
|
|
1430
|
+
for sample in (notes.get("samples") or ()):
|
|
1431
|
+
if isinstance(sample, Mapping) and "locus" in sample:
|
|
1432
|
+
self.census.sample(str(sample["locus"]), sample.get("value"))
|
|
1433
|
+
|
|
1434
|
+
# -- revision from measured feedback ------------------------------------
|
|
1435
|
+
|
|
1436
|
+
def deficient(self) -> bool:
|
|
1437
|
+
"""The preregistered trigger: has the harness MEASURED a defect?
|
|
1438
|
+
|
|
1439
|
+
Three kinds, in order of how little interpretation they need. A
|
|
1440
|
+
rejected candidate or a runtime failure is a broken contract, however
|
|
1441
|
+
rare. A batch whose novelty falls under ``min_novelty`` has collapsed
|
|
1442
|
+
-- onto itself or onto the archive -- which is different from the
|
|
1443
|
+
occasional collision any honest sampler makes in a small space, and
|
|
1444
|
+
the threshold is what keeps those two apart. And a generator whose
|
|
1445
|
+
measured children never survive is futile even when it is faultless.
|
|
1446
|
+
|
|
1447
|
+
A revision fires on evidence or not at all: time passing is not
|
|
1448
|
+
evidence, and W3 measured that revision LEVELS the rungs, so it must
|
|
1449
|
+
never fire quietly on a generator that is working.
|
|
1450
|
+
"""
|
|
1451
|
+
|
|
1452
|
+
tel = self.telemetry
|
|
1453
|
+
if tel.batches == 0:
|
|
1454
|
+
return False
|
|
1455
|
+
if (tel.rejected_shape or tel.rejected_out_of_domain
|
|
1456
|
+
or tel.runtime_failures or tel.repaired
|
|
1457
|
+
or tel.scaffold_out_of_domain or tel.scaffold_unknown_locus):
|
|
1458
|
+
return True
|
|
1459
|
+
last = self.last_report
|
|
1460
|
+
if (last is not None and last.emitted
|
|
1461
|
+
and last.novelty_rate < self.min_novelty):
|
|
1462
|
+
return True
|
|
1463
|
+
return (tel.measured >= self.min_measured_for_revision
|
|
1464
|
+
and tel.survived == 0)
|
|
1465
|
+
|
|
1466
|
+
def _maybe_revise(self) -> None:
|
|
1467
|
+
if self.revise is None or self.telemetry.revisions >= self.max_revisions:
|
|
1468
|
+
return
|
|
1469
|
+
if not self.deficient():
|
|
1470
|
+
return
|
|
1471
|
+
self.telemetry.revisions += 1
|
|
1472
|
+
feedback = self.feedback()
|
|
1473
|
+
try:
|
|
1474
|
+
replacement = self.revise(self.artifact, feedback)
|
|
1475
|
+
except Exception: # a revision must not kill a run
|
|
1476
|
+
replacement = None
|
|
1477
|
+
if replacement is None:
|
|
1478
|
+
return
|
|
1479
|
+
if self.revision_guard and not self._guard_admits(replacement):
|
|
1480
|
+
self.telemetry.revisions_rejected += 1
|
|
1481
|
+
self._remember_rejected_edit(replacement, guarded=True)
|
|
1482
|
+
return
|
|
1483
|
+
self._pending_edit = {
|
|
1484
|
+
"revision": self.telemetry.revisions,
|
|
1485
|
+
"sha": replacement.source_sha256[:8],
|
|
1486
|
+
"excerpt": replacement.source[:400],
|
|
1487
|
+
"before": (self.last_report.defect_rate
|
|
1488
|
+
if self.last_report is not None else 0.0),
|
|
1489
|
+
"signature": self.census.signature(),
|
|
1490
|
+
}
|
|
1491
|
+
self.telemetry.revisions_accepted += 1
|
|
1492
|
+
self.artifact = replacement
|
|
1493
|
+
|
|
1494
|
+
def feedback(self) -> str:
|
|
1495
|
+
"""The measured story this generator would hand a revision."""
|
|
1496
|
+
|
|
1497
|
+
return render_generation_feedback(
|
|
1498
|
+
self.telemetry, self.last_report, self._survivors,
|
|
1499
|
+
census=self.census, domains=self._domains,
|
|
1500
|
+
rejected_edits=self._rejected_edits)
|
|
1501
|
+
|
|
1502
|
+
# -- the guard, and the memory of what it (or measurement) rejected -----
|
|
1503
|
+
|
|
1504
|
+
def _guard_admits(self, replacement: AuthoredArtifact) -> bool:
|
|
1505
|
+
"""Does a FROZEN replay say the revision measurably helped?
|
|
1506
|
+
|
|
1507
|
+
The generator seam never sees the problem or the evaluation cache, so
|
|
1508
|
+
the only honest validation available to it is its own emission,
|
|
1509
|
+
replayed against the identical inputs the incumbent was last measured
|
|
1510
|
+
on: same archive, same ``n``, same domains, same seed. Admission takes
|
|
1511
|
+
a conjunction, so a revision cannot buy defect reduction with
|
|
1512
|
+
collapse: the defect rate must strictly fall AND the novelty rate --
|
|
1513
|
+
the frozen-validation score, the fraction of the batch that was
|
|
1514
|
+
usable and new -- must not fall.
|
|
1515
|
+
|
|
1516
|
+
No model call and no evaluation is spent here; the incumbent's side of
|
|
1517
|
+
the comparison is the batch already measured.
|
|
1518
|
+
"""
|
|
1519
|
+
|
|
1520
|
+
incumbent = self.last_report
|
|
1521
|
+
if self._last_call is None or incumbent is None:
|
|
1522
|
+
return True # nothing to compare against yet
|
|
1523
|
+
archive, n, domains, seed, template = self._last_call
|
|
1524
|
+
try:
|
|
1525
|
+
trial = self._run(replacement, archive, n, domains, seed,
|
|
1526
|
+
template=template, rng=random.Random(seed),
|
|
1527
|
+
count=False)
|
|
1528
|
+
except Exception:
|
|
1529
|
+
return False
|
|
1530
|
+
if not trial.emitted:
|
|
1531
|
+
return False
|
|
1532
|
+
return (trial.defect_rate < incumbent.defect_rate
|
|
1533
|
+
and trial.novelty_rate >= incumbent.novelty_rate)
|
|
1534
|
+
|
|
1535
|
+
def _remember_rejected_edit(self, artifact: AuthoredArtifact, *,
|
|
1536
|
+
guarded: bool, after: float = -1.0) -> None:
|
|
1537
|
+
before = (self.last_report.defect_rate
|
|
1538
|
+
if self.last_report is not None else 0.0)
|
|
1539
|
+
self._rejected_edits.append({
|
|
1540
|
+
"revision": self.telemetry.revisions,
|
|
1541
|
+
"sha": artifact.source_sha256[:8],
|
|
1542
|
+
"excerpt": artifact.source[:400],
|
|
1543
|
+
"before": before,
|
|
1544
|
+
"after": before if after < 0 else after,
|
|
1545
|
+
"signature": self.census.signature(),
|
|
1546
|
+
"guarded": bool(guarded),
|
|
1547
|
+
})
|
|
1548
|
+
del self._rejected_edits[:-3]
|
|
1549
|
+
|
|
1550
|
+
def _score_pending_edit(self, report: PoolReport) -> None:
|
|
1551
|
+
"""Did the revision we accepted last time actually fix anything?
|
|
1552
|
+
|
|
1553
|
+
Measured on the first batch the replacement emitted. If the defect
|
|
1554
|
+
rate did not fall, the edit joins the rejected-edit memory and every
|
|
1555
|
+
later revision is told, by name, that it was tried and failed.
|
|
1556
|
+
"""
|
|
1557
|
+
|
|
1558
|
+
pending, self._pending_edit = self._pending_edit, None
|
|
1559
|
+
if pending is None:
|
|
1560
|
+
return
|
|
1561
|
+
if report.defect_rate < float(pending["before"]):
|
|
1562
|
+
return
|
|
1563
|
+
self.telemetry.revisions_rejected += 1
|
|
1564
|
+
pending["after"] = report.defect_rate
|
|
1565
|
+
pending["guarded"] = False
|
|
1566
|
+
self._rejected_edits.append(pending)
|
|
1567
|
+
del self._rejected_edits[:-3]
|
|
1568
|
+
|
|
1569
|
+
# -- the SEARCH-PROGRESS channel: reasoning over measurements ------------
|
|
1570
|
+
|
|
1571
|
+
def _evidence_rows(self) -> List[MeasuredRow]:
|
|
1572
|
+
"""The rows the model will be shown. Identity, unless a view is set."""
|
|
1573
|
+
|
|
1574
|
+
rows: Sequence[MeasuredRow] = tuple(self._rows)
|
|
1575
|
+
if self.evidence_view is not None:
|
|
1576
|
+
try:
|
|
1577
|
+
rows = self.evidence_view(rows)
|
|
1578
|
+
except Exception: # a control that throws must not
|
|
1579
|
+
rows = () # be able to kill a measurement
|
|
1580
|
+
return [row for row in rows]
|
|
1581
|
+
|
|
1582
|
+
def _render_evidence(self, rows, domains) -> str:
|
|
1583
|
+
return render_measurement_evidence(
|
|
1584
|
+
rows, self.objectives, domains,
|
|
1585
|
+
front_shown=self.evidence_front_shown,
|
|
1586
|
+
effects_shown=self.evidence_effects_shown,
|
|
1587
|
+
# What the RUN charged and this generator was told about, not what
|
|
1588
|
+
# it produced: the model is entitled to know how much of the
|
|
1589
|
+
# budget bought the rows in front of it.
|
|
1590
|
+
charged=len(self._rows))
|
|
1591
|
+
|
|
1592
|
+
def _due(self) -> bool:
|
|
1593
|
+
"""Is an evidence-conditioned call due, and on WHOSE clock?
|
|
1594
|
+
|
|
1595
|
+
Two conditions, in the order they bind.
|
|
1596
|
+
|
|
1597
|
+
The channel cannot reason about rows it does not hold, so the FIRST
|
|
1598
|
+
call waits on EVIDENCE and nothing else: ``evidence_min_rows`` measured
|
|
1599
|
+
rows, which by default is the fewest a determinable effect can be
|
|
1600
|
+
computed from. Everything after it waits on the declared CADENCE --
|
|
1601
|
+
``reauthor_every`` further measured rows since the last call.
|
|
1602
|
+
|
|
1603
|
+
Splitting the two is the W11 fix. A single cadence had to answer both
|
|
1604
|
+
questions at once, and answering "when may it first speak?" with "when
|
|
1605
|
+
my own children number N" made the channel arrive two generations
|
|
1606
|
+
after the evidence did: on the EDA venue the prior was authored at a
|
|
1607
|
+
median charge of 40 against a 43.5-charge target, with the run's first
|
|
1608
|
+
20 charges structurally invisible to it. Rows are counted, not
|
|
1609
|
+
children; ``reauthor_every == 0`` still means the channel never fires.
|
|
1610
|
+
"""
|
|
1611
|
+
|
|
1612
|
+
if self.reauthor_every <= 0:
|
|
1613
|
+
return False
|
|
1614
|
+
if len(self._rows) < self.evidence_min_rows:
|
|
1615
|
+
return False
|
|
1616
|
+
if self._evidence_ticks == 0:
|
|
1617
|
+
return True
|
|
1618
|
+
return len(self._rows) - self._evidence_at >= self.reauthor_every
|
|
1619
|
+
|
|
1620
|
+
def _log(self, kind: str, *, rows: int, evidence: str,
|
|
1621
|
+
emitted: Optional[str], accepted: bool, **extra: Any) -> None:
|
|
1622
|
+
record: Dict[str, Any] = {
|
|
1623
|
+
"kind": kind,
|
|
1624
|
+
# Two different clocks, both recorded, neither standing in for the
|
|
1625
|
+
# other: `at_measured` is this generator's OWN children (the
|
|
1626
|
+
# attribution clock) and `at_rows` is every charged measurement the
|
|
1627
|
+
# run had reported to it (the evidence clock). Before W11 they were
|
|
1628
|
+
# the same number, which is precisely why the channel's lateness
|
|
1629
|
+
# was invisible in its own telemetry.
|
|
1630
|
+
"at_measured": int(self.telemetry.measured),
|
|
1631
|
+
"at_rows": len(self._rows),
|
|
1632
|
+
"rows_shown": int(rows),
|
|
1633
|
+
"evidence_sha256": evidence_digest(evidence),
|
|
1634
|
+
"evidence_chars": len(evidence),
|
|
1635
|
+
"emitted": emitted,
|
|
1636
|
+
"accepted": bool(accepted),
|
|
1637
|
+
}
|
|
1638
|
+
record.update(extra)
|
|
1639
|
+
self.evidence_log.append(record)
|
|
1640
|
+
|
|
1641
|
+
def _maybe_reauthor(self, domains: Mapping[str, Sequence[Any]],
|
|
1642
|
+
due: bool) -> None:
|
|
1643
|
+
"""Re-author the sampler against the measured trace, on cadence.
|
|
1644
|
+
|
|
1645
|
+
The trigger is SEARCH PROGRESS, not an emission defect: a generator
|
|
1646
|
+
that emits perfectly valid candidates out of a region the run has
|
|
1647
|
+
already measured to be uncompetitive is never deficient, and is
|
|
1648
|
+
exactly the case the defect trigger cannot see.
|
|
1649
|
+
"""
|
|
1650
|
+
|
|
1651
|
+
if (self.reauthor is None or not due
|
|
1652
|
+
or self.telemetry.reauthorings >= self.max_reauthorings):
|
|
1653
|
+
return
|
|
1654
|
+
rows = self._evidence_rows()
|
|
1655
|
+
if not rows:
|
|
1656
|
+
return
|
|
1657
|
+
self.telemetry.reauthorings += 1
|
|
1658
|
+
self.telemetry.evidence_rows_shown += len(rows)
|
|
1659
|
+
evidence = self._render_evidence(rows, domains)
|
|
1660
|
+
try:
|
|
1661
|
+
replacement = self.reauthor(self.artifact, evidence)
|
|
1662
|
+
except Exception: # a re-authoring must not kill a run
|
|
1663
|
+
replacement = None
|
|
1664
|
+
self._log("reauthor", rows=len(rows), evidence=evidence,
|
|
1665
|
+
emitted=(None if replacement is None
|
|
1666
|
+
else replacement.source_sha256),
|
|
1667
|
+
accepted=replacement is not None,
|
|
1668
|
+
replaced=self.artifact.source_sha256)
|
|
1669
|
+
if replacement is None:
|
|
1670
|
+
return
|
|
1671
|
+
self.telemetry.reauthorings_accepted += 1
|
|
1672
|
+
self.artifact = replacement
|
|
1673
|
+
|
|
1674
|
+
def _maybe_author_prior(self, domains: Mapping[str, Sequence[Any]],
|
|
1675
|
+
due: bool) -> None:
|
|
1676
|
+
"""Ask which loci matter, type the answer, and let the GATE refuse it."""
|
|
1677
|
+
|
|
1678
|
+
self._unwind_prior_if_it_stopped_paying()
|
|
1679
|
+
if (self.prior_author is None or not due
|
|
1680
|
+
or self._prior is not None
|
|
1681
|
+
or self.telemetry.priors_proposed >= self.max_priors):
|
|
1682
|
+
return
|
|
1683
|
+
rows = self._evidence_rows()
|
|
1684
|
+
if not rows:
|
|
1685
|
+
return
|
|
1686
|
+
self.telemetry.priors_proposed += 1
|
|
1687
|
+
evidence = self._render_evidence(rows, domains)
|
|
1688
|
+
prompt = WEIGHTED_RESTRICTION_PROMPT.format(
|
|
1689
|
+
goals="\n".join(f" {s.name}: {s.goal}imise" for s in self.objectives),
|
|
1690
|
+
domains="\n".join(
|
|
1691
|
+
f" {name}: {json_compact(list(values))}"
|
|
1692
|
+
for name, values in sorted(dict(domains).items())),
|
|
1693
|
+
evidence=evidence,
|
|
1694
|
+
max_ratio=f"{float(self.prior_max_weight_ratio):g}")
|
|
1695
|
+
try:
|
|
1696
|
+
reply = self.prior_author(prompt)
|
|
1697
|
+
except Exception:
|
|
1698
|
+
reply = ""
|
|
1699
|
+
verdict = admit_weighted_restriction(
|
|
1700
|
+
parse_weighted_restriction(reply),
|
|
1701
|
+
domains=domains,
|
|
1702
|
+
max_weight_ratio=self.prior_max_weight_ratio)
|
|
1703
|
+
self._log("locus_prior", rows=len(rows), evidence=evidence,
|
|
1704
|
+
emitted=(None if verdict.proposal is None
|
|
1705
|
+
else evidence_digest(json_compact(
|
|
1706
|
+
verdict.proposal.as_note()))),
|
|
1707
|
+
accepted=verdict.admitted, verdict=verdict.as_note())
|
|
1708
|
+
if not verdict.admitted:
|
|
1709
|
+
self.telemetry.priors_refused += 1
|
|
1710
|
+
return
|
|
1711
|
+
self.telemetry.priors_admitted += 1
|
|
1712
|
+
self._prior = verdict.prior
|
|
1713
|
+
self._prior_batches = 0
|
|
1714
|
+
self._survived_at_prior = self.telemetry.survived
|
|
1715
|
+
|
|
1716
|
+
def _effective_domains(
|
|
1717
|
+
self, declared: Mapping[str, Sequence[Any]]
|
|
1718
|
+
) -> Dict[str, List[Any]]:
|
|
1719
|
+
if self._prior is None:
|
|
1720
|
+
return {k: list(v) for k, v in dict(declared).items()}
|
|
1721
|
+
self._prior_batches += 1
|
|
1722
|
+
return apply_weighted_restriction(declared, self._prior)
|
|
1723
|
+
|
|
1724
|
+
def _unwind_prior_if_it_stopped_paying(self) -> None:
|
|
1725
|
+
"""An admitted prior is still a bet, and a bet must be checkable.
|
|
1726
|
+
|
|
1727
|
+
A graded restriction cannot exclude a measured front member -- every
|
|
1728
|
+
declared value keeps positive mass -- so nothing about it is
|
|
1729
|
+
unrecoverable. It can still be WRONG, and wrong only shows up as
|
|
1730
|
+
spend: the prior is held only while it pays. After
|
|
1731
|
+
``prior_unwind_batches`` generations drawn under it with not one new
|
|
1732
|
+
survivor, it is dropped and the run finishes on the declared domains.
|
|
1733
|
+
"""
|
|
1734
|
+
|
|
1735
|
+
if self._prior is None or self.prior_unwind_batches <= 0:
|
|
1736
|
+
return
|
|
1737
|
+
if self._prior_batches < self.prior_unwind_batches:
|
|
1738
|
+
return
|
|
1739
|
+
if self.telemetry.survived > self._survived_at_prior:
|
|
1740
|
+
return
|
|
1741
|
+
self._prior = None
|
|
1742
|
+
self.telemetry.priors_unwound += 1
|
|
1743
|
+
|
|
1744
|
+
def note(self) -> Dict[str, Any]:
|
|
1745
|
+
"""The per-generation history record for the last batch."""
|
|
1746
|
+
|
|
1747
|
+
note = (self.last_report.as_note() if self.last_report is not None
|
|
1748
|
+
else PoolReport().as_note())
|
|
1749
|
+
note["artifact"] = f"{self.artifact.name}:{self.artifact.source_sha256[:8]}"
|
|
1750
|
+
note["revisions"] = self.telemetry.revisions_accepted
|
|
1751
|
+
if self.reauthor_every > 0:
|
|
1752
|
+
# What the model saw and what it emitted, in the run's own record.
|
|
1753
|
+
# Only what is NEW since the last generation's note: the full log
|
|
1754
|
+
# stays on the object, and the history is a diary rather than n
|
|
1755
|
+
# copies of the same list.
|
|
1756
|
+
fresh = self.evidence_log[self._noted:]
|
|
1757
|
+
self._noted = len(self.evidence_log)
|
|
1758
|
+
note["evidence"] = [dict(record) for record in fresh]
|
|
1759
|
+
note["prior_active"] = self._prior is not None
|
|
1760
|
+
return note
|