agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,864 @@
|
|
|
1
|
+
"""The authorship substrate's one public knob, and its factory.
|
|
2
|
+
|
|
3
|
+
``AuthorshipConfig`` names who authors which machinery -- surrogates today,
|
|
4
|
+
variation operators next -- and :func:`build_authorship` turns it into the
|
|
5
|
+
policy objects the genetic loop consumes. Everything defaults to off, and
|
|
6
|
+
off is byte-identical to the pre-substrate loop (the fossil test holds it).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from types import SimpleNamespace
|
|
13
|
+
from typing import Any, Callable, Mapping, Optional, Sequence
|
|
14
|
+
|
|
15
|
+
from agent_evolve.infrastructure.authored_runtime import RuntimeLimits
|
|
16
|
+
|
|
17
|
+
__all__ = ["AuthorshipConfig", "AuthorshipPolicies", "PRESETS",
|
|
18
|
+
"build_authorship"]
|
|
19
|
+
|
|
20
|
+
_SURROGATE = ("off", "rule", "llm")
|
|
21
|
+
_OPERATORS = ("off", "rule", "llm")
|
|
22
|
+
_INITIALIZATION = ("off", "llm")
|
|
23
|
+
_INIT_STYLE = ("joint", "split")
|
|
24
|
+
_GENERATION = ("off", "llm")
|
|
25
|
+
_ADAPTATION = ("off", "llm")
|
|
26
|
+
|
|
27
|
+
#: The named compositions, as field settings. A table rather than a method
|
|
28
|
+
#: body so that everything downstream -- the CLI's ``--authorship`` choices,
|
|
29
|
+
#: any campaign script -- ENUMERATES what exists instead of repeating a list
|
|
30
|
+
#: that then drifts.
|
|
31
|
+
PRESETS: Mapping[str, Mapping[str, str]] = {
|
|
32
|
+
"off": {},
|
|
33
|
+
"surrogate": {"surrogate": "rule"},
|
|
34
|
+
"surrogate-llm": {"surrogate": "llm"},
|
|
35
|
+
"operators": {"operators": "rule"},
|
|
36
|
+
"operators-llm": {"operators": "llm"},
|
|
37
|
+
"init-llm": {"initialization": "llm"},
|
|
38
|
+
"generation-llm": {"generation": "llm"},
|
|
39
|
+
# The authored sampler under the frozen screen stack: the model writes
|
|
40
|
+
# where candidates come from, and the variance-guarded authored surrogate
|
|
41
|
+
# decides which of them are worth measuring.
|
|
42
|
+
"generative": {"generation": "llm", "surrogate": "llm"},
|
|
43
|
+
# What `authorship="auto"` resolves to when a model call is possible, named
|
|
44
|
+
# so a campaign can state it instead of inheriting it: the two seams the
|
|
45
|
+
# six-arm ablation and the sealed luna-clear row measured as winners, and
|
|
46
|
+
# neither of the per-decision seams that did not.
|
|
47
|
+
"guided": {"surrogate": "llm", "initialization": "llm"},
|
|
48
|
+
# `guided`, plus the one channel that reads what the run MEASURED: the
|
|
49
|
+
# graded prior is revised on a declared cadence instead of being authored
|
|
50
|
+
# once at t = 0 and held for the whole budget.
|
|
51
|
+
"adaptive": {"surrogate": "llm", "initialization": "llm",
|
|
52
|
+
"adaptation": "llm"},
|
|
53
|
+
"full": {"surrogate": "llm", "operators": "llm", "initialization": "llm"},
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True)
|
|
58
|
+
class AuthorshipConfig:
|
|
59
|
+
"""Who authors which machinery, and under what bounds.
|
|
60
|
+
|
|
61
|
+
``surrogate="rule"`` turns on virtual pre-screening with the shipped,
|
|
62
|
+
credential-free surrogates behind the validation gate. The ``"llm"``
|
|
63
|
+
values put the model in the authoring seat for that piece of machinery --
|
|
64
|
+
the surrogate, the variation operators, the initial population, or
|
|
65
|
+
``generation``, where it writes the sampler every candidate is drawn
|
|
66
|
+
from. Naming a value that has not landed is an error today rather than a
|
|
67
|
+
silent no-op forever.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
surrogate: str = "off"
|
|
71
|
+
operators: str = "off"
|
|
72
|
+
initialization: str = "off"
|
|
73
|
+
#: WHAT the one initialization call asks for. ``"joint"`` -- the default
|
|
74
|
+
#: and every sealed row -- asks for one list that is individually strong
|
|
75
|
+
#: AND collectively diverse. ``"split"`` asks the same call for two
|
|
76
|
+
#: labelled halves, ``(k + 1) // 2`` strongest individual bets and the
|
|
77
|
+
#: rest coverage, which is the tuning round's answer to a measured
|
|
78
|
+
#: conflation: with one ask, deliberation buys diversity and the pool
|
|
79
|
+
#: median pays (reasoning effort ``none`` beat ``medium`` on 23 of 24
|
|
80
|
+
#: paired seeds), so effort had no clean channel into exploitation. The
|
|
81
|
+
#: halves are labelled in the init telemetry, so the exploit subset can be
|
|
82
|
+
#: scored on its own. See policies.llm_init.
|
|
83
|
+
init_style: str = "joint"
|
|
84
|
+
generation: str = "off"
|
|
85
|
+
pool_factor: int = 4
|
|
86
|
+
exploration_floor: float = 0.25
|
|
87
|
+
#: How many objectives the screen's gate must certify before the screen
|
|
88
|
+
#: may order anything: ``0`` = every declared objective (the conjunction),
|
|
89
|
+
#: a positive value = that many, with the screen ordering on exactly the
|
|
90
|
+
#: certified ones. See policies.surrogate.GatePolicy.min_passing_objectives.
|
|
91
|
+
#:
|
|
92
|
+
#: The default is the MEASURED arm, not taste: partial screening at 2
|
|
93
|
+
#: beat the shipped conjunction 102/54 (sign test p = 1.5e-4, better >
|
|
94
|
+
#: worse in all four cells where the arms can differ; wave-K
|
|
95
|
+
#: aug14_partial_screen.md), and it is INERT when the gate certifies
|
|
96
|
+
#: every objective -- on a 2-objective venue the partial gate IS the
|
|
97
|
+
#: conjunction, identical in every counter (100/100 identical runs).
|
|
98
|
+
#: ``0`` restores the historical conjunction exactly.
|
|
99
|
+
screen_min_passing_objectives: int = 2
|
|
100
|
+
#: The share of a generation reserved from a PARTIAL screen, as a multiple
|
|
101
|
+
#: of the share of objectives it could not see. See
|
|
102
|
+
#: session.screening.Screening.unscreened_objective_floor.
|
|
103
|
+
screen_unscreened_objective_floor: float = 1.0
|
|
104
|
+
authoring_attempts: int = 2
|
|
105
|
+
max_authored_fraction: float = 0.5
|
|
106
|
+
#: How many of the most recent measurements a screen refresh fits and
|
|
107
|
+
#: validates on (see session.screening.Screening.max_training_rows). The
|
|
108
|
+
#: screen re-arbitrates every generation, so this is what decides whether
|
|
109
|
+
#: a high-budget run's screening cost is constant or grows with the run.
|
|
110
|
+
screen_training_rows: int = 1024
|
|
111
|
+
#: Mass generation's pool: ``generation_pool_factor`` times the offspring
|
|
112
|
+
#: a generation can afford, or exactly ``generation_pool_size`` when that
|
|
113
|
+
#: is set. The pool costs no evaluations, so it is sized by what the
|
|
114
|
+
#: sampler and the screen can chew through, not by the budget.
|
|
115
|
+
generation_pool_factor: int = 4
|
|
116
|
+
generation_pool_size: int = 0
|
|
117
|
+
#: One-shot authoring is what the ladder cells measure; revision LEVELS
|
|
118
|
+
#: the rungs (W3), so it is capped here and ablatable to 0.
|
|
119
|
+
generation_revisions: int = 1
|
|
120
|
+
#: Keep an authored revision only when a FROZEN replay measures it
|
|
121
|
+
#: better -- strictly lower defect rate, no loss of novelty. Off by
|
|
122
|
+
#: default because every sealed row is defined on unguarded revision;
|
|
123
|
+
#: it is the third arm of the revision-value row, not a silent change.
|
|
124
|
+
generation_revision_guard: bool = False
|
|
125
|
+
#: Ship the emit harness into the sandbox (the authored code builds
|
|
126
|
+
#: candidates through ``build``) and assemble a partially-correct
|
|
127
|
+
#: emission rather than dropping it whole. Both default ON: they are the
|
|
128
|
+
#: fix for the measured assignment-genome authoring failure, and both are
|
|
129
|
+
#: ablatable so the fix can be measured against its own absence.
|
|
130
|
+
generation_scaffold: bool = True
|
|
131
|
+
generation_repair: bool = True
|
|
132
|
+
#: Echo the sandbox's own wall/CPU/memory budget into the authoring
|
|
133
|
+
#: prompt, and retry a batch that overran it at ``n // 4`` rather than
|
|
134
|
+
#: losing the whole pool. An unstated budget is a budget the author
|
|
135
|
+
#: cannot honour, and on `upms_j14_m3` the overrun -- not the shape --
|
|
136
|
+
#: is what emptied most batches.
|
|
137
|
+
generation_limits_echo: bool = True
|
|
138
|
+
#: MEASUREMENT-CONDITIONED RE-AUTHORING. After the archive grows by this
|
|
139
|
+
#: many CHARGED evaluations, the generator is re-authored against the
|
|
140
|
+
#: run's own measured trace -- the front, what improved, which parameter
|
|
141
|
+
#: the measurements say moves which cost -- instead of against emission
|
|
142
|
+
#: counters. ``0`` (the default) is the static-prior seam every sealed row
|
|
143
|
+
#: to date ran, and off is byte-identical to it: no evidence call fires,
|
|
144
|
+
#: no evidence is rendered, no counter moves.
|
|
145
|
+
#:
|
|
146
|
+
#: This is the channel `generation_revisions` is not. Revision fires on
|
|
147
|
+
#: EMISSION DEFECTS (rejects, collapse, no survivors); a generator drawing
|
|
148
|
+
#: valid candidates out of a region already measured to be bad is not
|
|
149
|
+
#: deficient by that test and is never revised. The cadence is declared
|
|
150
|
+
#: here, in evaluations, so a campaign states it rather than inheriting a
|
|
151
|
+
#: number from a code path.
|
|
152
|
+
generation_reauthor_every: int = 0
|
|
153
|
+
#: WHEN the channel may speak for the FIRST time, in MEASURED ROWS -- the
|
|
154
|
+
#: charged evaluations the run holds, whoever produced them, initial
|
|
155
|
+
#: population included. The cadence above says how often an evidence call
|
|
156
|
+
#: RECURS and cannot also say when the first one is allowed: read as "wait
|
|
157
|
+
#: for that many of the generator's own children" it made the channel
|
|
158
|
+
#: arrive two generations after the evidence did (W11 -- on the EDA venue
|
|
159
|
+
#: the prior landed at a median charge of 40 against a 43.5-charge target,
|
|
160
|
+
#: with the run's first 20 charges structurally invisible to it).
|
|
161
|
+
#:
|
|
162
|
+
#: ``0`` (the default) means AS SOON AS THE GATE CAN BE MET:
|
|
163
|
+
#: ``measurement_evidence.MIN_EVIDENCE_ROWS`` rows, the fewest from which a
|
|
164
|
+
#: determinable per-locus effect can be computed at all. Set it to a
|
|
165
|
+
#: venue's own legibility point to wait for one; set it equal to
|
|
166
|
+
#: ``generation_reauthor_every`` to restore the pure-cadence rule exactly.
|
|
167
|
+
generation_evidence_min_rows: int = 0
|
|
168
|
+
#: How many measurement-conditioned re-authorings one run may pay for.
|
|
169
|
+
generation_reauthorings: int = 0
|
|
170
|
+
#: THE LOCUS-IMPORTANCE CHANNEL. On the same cadence, ask the model which
|
|
171
|
+
#: parameters and values the measurements justify concentrating the
|
|
172
|
+
#: remaining budget on, type the answer as a GRADED bias over the
|
|
173
|
+
#: DECLARED domains -- per-locus value weights that exclude NOTHING --
|
|
174
|
+
#: and let the gate refuse it (see
|
|
175
|
+
#: policies.measurement_evidence.admit_weighted_restriction). Requires a
|
|
176
|
+
#: cadence. Because nothing is excluded, the prior can waste budget but
|
|
177
|
+
#: can never drop the optimum, and it is unwound when it stops producing
|
|
178
|
+
#: survivors.
|
|
179
|
+
generation_locus_prior: bool = False
|
|
180
|
+
#: How many priors one run may have admitted, and the concentration cap:
|
|
181
|
+
#: within one parameter the heaviest value may outweigh the lightest by
|
|
182
|
+
#: at most this ratio, so a graded bias cannot become a de-facto
|
|
183
|
+
#: exclusion.
|
|
184
|
+
generation_locus_priors: int = 1
|
|
185
|
+
generation_prior_max_weight_ratio: float = 8.0
|
|
186
|
+
#: MEASUREMENT-CONDITIONED REVISION OF THE SAMPLING PRIOR. ``"off"`` is
|
|
187
|
+
#: the static-prior seam every sealed row ran: whatever prior the run
|
|
188
|
+
#: installs before it measures anything is held for the whole budget.
|
|
189
|
+
#: ``"llm"`` buys the other clock -- on the cadence below, one call reads
|
|
190
|
+
#: the run's measured trace and the weights in force and returns a
|
|
191
|
+
#: revision, which is damped into them rather than replacing them. It
|
|
192
|
+
#: revises the CLASSICAL breeding path's prior, which is what
|
|
193
|
+
#: initialization and mutation both draw through; it is not the
|
|
194
|
+
#: sampler-re-authoring channel, which lost to its shuffled-evidence
|
|
195
|
+
#: control.
|
|
196
|
+
adaptation: str = "off"
|
|
197
|
+
#: The cadence, in CHARGED evaluations. ``0`` resolves from the run's own
|
|
198
|
+
#: shape in :func:`build_authorship` (a couple of generations, or a sixth
|
|
199
|
+
#: of the budget, whichever is larger) and the resolved value is
|
|
200
|
+
#: announced, so a campaign can state it instead of inheriting it.
|
|
201
|
+
adapt_every: int = 0
|
|
202
|
+
#: How many revisions one run may pay for.
|
|
203
|
+
adapt_max: int = 4
|
|
204
|
+
#: How many complete configurations a revision call may also return, each
|
|
205
|
+
#: validated value-by-value like an authored initial member. ``0`` -- the
|
|
206
|
+
#: default -- runs revision alone: immigrants are a separately flagged
|
|
207
|
+
#: second cell, because two mechanisms bought together measure one number.
|
|
208
|
+
adapt_immigrants: int = 0
|
|
209
|
+
#: How far a reply moves the installed weights. ``0.5`` mixes them evenly;
|
|
210
|
+
#: ``1.0`` installs the reply as written (an ablation arm, not a default)
|
|
211
|
+
#: and ``0.0`` is the no-op arm. Below 1 no revision can introduce an
|
|
212
|
+
#: exclusion, which is what bounds a wrong revision to wasted draws.
|
|
213
|
+
adapt_damping: float = 0.5
|
|
214
|
+
#: The CONTROL seam, declared: it receives the rows this run measured and
|
|
215
|
+
#: returns the rows the model is shown. Identity (None) is the product; a
|
|
216
|
+
#: view returning another run's rows -- same count, same shape, same cost
|
|
217
|
+
#: -- is the shuffled-evidence control, buildable without editing the
|
|
218
|
+
#: product.
|
|
219
|
+
adapt_evidence_view: Any = None
|
|
220
|
+
#: Which rows the revision channel's zero-on-front admission check reads.
|
|
221
|
+
#: ``False`` -- the default and the product's stance -- reads the rows the
|
|
222
|
+
#: run really measured, so the gate that protects a live run reads reality
|
|
223
|
+
#: whatever ``adapt_evidence_view`` showed the model. ``True`` reads the
|
|
224
|
+
#: VIEWED rows, which only a CONTROL arm wants: a control prompted with
|
|
225
|
+
#: donor rows and gated on this run's front accrues refusals the arm it
|
|
226
|
+
#: controls never meets, and its refusal rate stops being comparable.
|
|
227
|
+
adapt_gate_reads_view: bool = False
|
|
228
|
+
limits: RuntimeLimits = field(default_factory=RuntimeLimits)
|
|
229
|
+
|
|
230
|
+
def __post_init__(self) -> None:
|
|
231
|
+
if self.surrogate not in _SURROGATE:
|
|
232
|
+
raise ValueError(
|
|
233
|
+
f"authorship.surrogate must be one of {_SURROGATE}, got "
|
|
234
|
+
f"{self.surrogate!r}"
|
|
235
|
+
)
|
|
236
|
+
if self.initialization not in _INITIALIZATION:
|
|
237
|
+
raise ValueError(
|
|
238
|
+
f"authorship.initialization must be one of {_INITIALIZATION}, "
|
|
239
|
+
f"got {self.initialization!r}"
|
|
240
|
+
)
|
|
241
|
+
if self.init_style not in _INIT_STYLE:
|
|
242
|
+
raise ValueError(
|
|
243
|
+
f"authorship.init_style must be one of {_INIT_STYLE}, got "
|
|
244
|
+
f"{self.init_style!r}"
|
|
245
|
+
)
|
|
246
|
+
if self.init_style != "joint" and self.initialization == "off":
|
|
247
|
+
# The same rule the adaptation knobs follow: a setting nothing
|
|
248
|
+
# reads is a campaign reporting something it never bought.
|
|
249
|
+
raise ValueError(
|
|
250
|
+
f"authorship.init_style={self.init_style!r} shapes the ask the "
|
|
251
|
+
"model-proposed initial population is authored from and this "
|
|
252
|
+
"run has authorship.initialization='off'; set "
|
|
253
|
+
"initialization='llm' or drop the setting"
|
|
254
|
+
)
|
|
255
|
+
if self.operators not in _OPERATORS:
|
|
256
|
+
raise ValueError(
|
|
257
|
+
f"authorship.operators must be one of {_OPERATORS}, got "
|
|
258
|
+
f"{self.operators!r}"
|
|
259
|
+
)
|
|
260
|
+
if self.generation not in _GENERATION:
|
|
261
|
+
raise ValueError(
|
|
262
|
+
f"authorship.generation must be one of {_GENERATION}, got "
|
|
263
|
+
f"{self.generation!r}"
|
|
264
|
+
)
|
|
265
|
+
if self.adaptation not in _ADAPTATION:
|
|
266
|
+
raise ValueError(
|
|
267
|
+
f"authorship.adaptation must be one of {_ADAPTATION}, got "
|
|
268
|
+
f"{self.adaptation!r}"
|
|
269
|
+
)
|
|
270
|
+
if self.adapt_every < 0 or self.adapt_max < 0 or self.adapt_immigrants < 0:
|
|
271
|
+
raise ValueError(
|
|
272
|
+
"authorship.adapt_every (a cadence in charged evaluations), "
|
|
273
|
+
"adapt_max and adapt_immigrants are counts and must be "
|
|
274
|
+
f"non-negative, got {self.adapt_every}, {self.adapt_max}, "
|
|
275
|
+
f"{self.adapt_immigrants}")
|
|
276
|
+
if not 0.0 <= self.adapt_damping <= 1.0:
|
|
277
|
+
raise ValueError(
|
|
278
|
+
"authorship.adapt_damping mixes a revision into the installed "
|
|
279
|
+
f"weights and must lie in [0, 1], got {self.adapt_damping}")
|
|
280
|
+
if self.adaptation == "off":
|
|
281
|
+
# A knob the run would silently ignore is a silent no-op: the
|
|
282
|
+
# cadence would be read by nothing and the campaign would report a
|
|
283
|
+
# setting it never bought.
|
|
284
|
+
asked = [name for name, value, default in (
|
|
285
|
+
("adapt_every", self.adapt_every, 0),
|
|
286
|
+
("adapt_max", self.adapt_max, 4),
|
|
287
|
+
("adapt_immigrants", self.adapt_immigrants, 0),
|
|
288
|
+
("adapt_damping", self.adapt_damping, 0.5),
|
|
289
|
+
("adapt_evidence_view", self.adapt_evidence_view, None),
|
|
290
|
+
("adapt_gate_reads_view", self.adapt_gate_reads_view, False),
|
|
291
|
+
) if value != default]
|
|
292
|
+
if asked:
|
|
293
|
+
raise ValueError(
|
|
294
|
+
f"authorship.{', '.join(asked)} configure(s) the "
|
|
295
|
+
"measurement-conditioned revision channel and this run "
|
|
296
|
+
"has authorship.adaptation='off'; set adaptation='llm' or "
|
|
297
|
+
"drop the setting")
|
|
298
|
+
if self.generation_reauthor_every < 0:
|
|
299
|
+
raise ValueError(
|
|
300
|
+
"authorship.generation_reauthor_every is a cadence in charged "
|
|
301
|
+
"evaluations and must be non-negative, got "
|
|
302
|
+
f"{self.generation_reauthor_every}")
|
|
303
|
+
if self.generation_evidence_min_rows < 0:
|
|
304
|
+
raise ValueError(
|
|
305
|
+
"authorship.generation_evidence_min_rows is the fewest "
|
|
306
|
+
"measured rows the evidence channel will author from and must "
|
|
307
|
+
f"be non-negative, got {self.generation_evidence_min_rows}")
|
|
308
|
+
if self.generation_prior_max_weight_ratio < 1.0:
|
|
309
|
+
raise ValueError(
|
|
310
|
+
"authorship.generation_prior_max_weight_ratio caps how far a "
|
|
311
|
+
"graded prior may concentrate (heaviest over lightest value "
|
|
312
|
+
"of one parameter) and must be at least 1, got "
|
|
313
|
+
f"{self.generation_prior_max_weight_ratio}")
|
|
314
|
+
if self.generation_locus_prior and self.generation_reauthor_every <= 0:
|
|
315
|
+
raise ValueError(
|
|
316
|
+
"authorship.generation_locus_prior is authored from the "
|
|
317
|
+
"measured trace on the generation_reauthor_every cadence; set "
|
|
318
|
+
"that cadence, or the prior would fire on no declared rule")
|
|
319
|
+
if (self.generation_reauthor_every or self.generation_reauthorings
|
|
320
|
+
or self.generation_locus_prior) and self.generation == "off":
|
|
321
|
+
raise ValueError(
|
|
322
|
+
"the measurement-conditioned channel re-authors the GENERATOR; "
|
|
323
|
+
"it needs authorship.generation='llm'")
|
|
324
|
+
if self.generation != "off" and self.operators != "off":
|
|
325
|
+
# Both construct the generation's candidates. Accepting the pair
|
|
326
|
+
# would silently run one of them and bill the caller for two.
|
|
327
|
+
raise ValueError(
|
|
328
|
+
"authorship.generation and authorship.operators both "
|
|
329
|
+
"construct the generation's candidates -- the generator draws "
|
|
330
|
+
"the pool, the operator arms recombine parents into it. Ask "
|
|
331
|
+
"for one."
|
|
332
|
+
)
|
|
333
|
+
|
|
334
|
+
@property
|
|
335
|
+
def engaged(self) -> bool:
|
|
336
|
+
return (self.surrogate != "off" or self.operators != "off"
|
|
337
|
+
or self.initialization != "off" or self.generation != "off"
|
|
338
|
+
or self.adaptation != "off")
|
|
339
|
+
|
|
340
|
+
@classmethod
|
|
341
|
+
def preset(cls, name: str) -> "AuthorshipConfig":
|
|
342
|
+
if name not in PRESETS:
|
|
343
|
+
raise ValueError(
|
|
344
|
+
f"authorship preset must be one of {sorted(PRESETS)}, got "
|
|
345
|
+
f"{name!r}"
|
|
346
|
+
)
|
|
347
|
+
return cls(**PRESETS[name])
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
@dataclass(frozen=True)
|
|
351
|
+
class AuthorshipPolicies:
|
|
352
|
+
"""What the factory built: policy objects the loop consumes directly."""
|
|
353
|
+
|
|
354
|
+
screening: Optional[Any] = None
|
|
355
|
+
portfolio: Optional[Any] = None
|
|
356
|
+
initial_proposals: tuple = ()
|
|
357
|
+
init_author: Optional[Any] = None
|
|
358
|
+
generator: Optional[Any] = None
|
|
359
|
+
#: Set only when generation was asked for and produced no generator: the
|
|
360
|
+
#: authoring counters would otherwise have nowhere to live, and "the
|
|
361
|
+
#: model failed to author a sampler" would be indistinguishable from
|
|
362
|
+
#: "nobody asked". When a generator exists it carries its own note.
|
|
363
|
+
generator_author: Optional[Any] = None
|
|
364
|
+
#: The measurement-conditioned revision policy the loop consults once per
|
|
365
|
+
#: generation; None when adaptation is off.
|
|
366
|
+
reguidance: Optional[Any] = None
|
|
367
|
+
#: Set only when adaptation was asked for and no policy could be built --
|
|
368
|
+
#: the same counters-need-a-home pattern as ``generator_author``. A live
|
|
369
|
+
#: policy carries its own counters.
|
|
370
|
+
reguidance_author: Optional[Any] = None
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def build_authorship(
|
|
374
|
+
config: AuthorshipConfig,
|
|
375
|
+
*,
|
|
376
|
+
complete: Any = None,
|
|
377
|
+
objectives: Sequence[Any] = (),
|
|
378
|
+
schema_text: str = "",
|
|
379
|
+
seed: Optional[int] = None,
|
|
380
|
+
announce: Optional[Callable[[str], None]] = None,
|
|
381
|
+
candidate_model: Any = None,
|
|
382
|
+
init_template: Any = None,
|
|
383
|
+
init_k: int = 0,
|
|
384
|
+
budget: Optional[int] = None,
|
|
385
|
+
population_size: Optional[int] = None,
|
|
386
|
+
) -> AuthorshipPolicies:
|
|
387
|
+
"""The policy objects for *config*; fields are ``None`` where nothing is on.
|
|
388
|
+
|
|
389
|
+
With an ``"llm"`` value the model is asked ONCE, before any evaluation,
|
|
390
|
+
to author from the schema and objective meanings; authored machinery then
|
|
391
|
+
competes against the shipped rules under measurement -- the validation
|
|
392
|
+
gate for surrogates, survival credit for operators. No usable authorship
|
|
393
|
+
(no credential, no code block, forbidden imports) degrades to the rules,
|
|
394
|
+
out loud, never silently.
|
|
395
|
+
"""
|
|
396
|
+
|
|
397
|
+
say = announce or (lambda _m: None)
|
|
398
|
+
proposals, init_note = _build_initialization(
|
|
399
|
+
config, complete, schema_text, say, candidate_model,
|
|
400
|
+
init_template, init_k)
|
|
401
|
+
generator, generator_note = _build_generator(
|
|
402
|
+
config, complete, objectives, schema_text, say,
|
|
403
|
+
candidate_model, init_template)
|
|
404
|
+
reguidance, reguidance_note = _build_reguidance(
|
|
405
|
+
config, complete, objectives, schema_text, say, candidate_model,
|
|
406
|
+
init_template, budget, population_size)
|
|
407
|
+
return AuthorshipPolicies(
|
|
408
|
+
screening=_build_screening(config, complete, objectives,
|
|
409
|
+
schema_text, say),
|
|
410
|
+
portfolio=_build_portfolio(config, complete, objectives,
|
|
411
|
+
schema_text, say),
|
|
412
|
+
initial_proposals=proposals,
|
|
413
|
+
init_author=init_note,
|
|
414
|
+
generator=generator,
|
|
415
|
+
generator_author=generator_note,
|
|
416
|
+
reguidance=reguidance,
|
|
417
|
+
reguidance_author=reguidance_note,
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _build_reguidance(
|
|
422
|
+
config: AuthorshipConfig,
|
|
423
|
+
complete: Any,
|
|
424
|
+
objectives: Sequence[Any],
|
|
425
|
+
schema_text: str,
|
|
426
|
+
say: Callable[[str], None],
|
|
427
|
+
candidate_model: Any,
|
|
428
|
+
template: Any,
|
|
429
|
+
budget: Optional[int],
|
|
430
|
+
population_size: Optional[int],
|
|
431
|
+
) -> tuple:
|
|
432
|
+
"""``(policy, orphaned_note)``; both ``None`` when adaptation is off.
|
|
433
|
+
|
|
434
|
+
The cadence is DECLARED, and when the caller declares ``0`` it is resolved
|
|
435
|
+
here from the run's own shape and announced: a couple of generations of
|
|
436
|
+
offspring, or a sixth of the budget, whichever is larger, so a short run
|
|
437
|
+
still gets its first revision after the trace says something and a long
|
|
438
|
+
one gets several. A campaign that wants a different clock states it.
|
|
439
|
+
"""
|
|
440
|
+
|
|
441
|
+
if config.adaptation == "off":
|
|
442
|
+
return None, None
|
|
443
|
+
from agent_evolve.policies.reguidance import Reguidance, ReguidanceTelemetry
|
|
444
|
+
|
|
445
|
+
telemetry = ReguidanceTelemetry()
|
|
446
|
+
note = SimpleNamespace(telemetry=telemetry, mechanism="reguidance",
|
|
447
|
+
authored_by="llm")
|
|
448
|
+
if complete is None:
|
|
449
|
+
say("authorship.adaptation='llm' needs a model call and none is "
|
|
450
|
+
"available; the sampling prior stays as it was installed.")
|
|
451
|
+
return None, note
|
|
452
|
+
if not template:
|
|
453
|
+
say("authorship.adaptation='llm' received no candidate template, so "
|
|
454
|
+
"there are no declared domains to re-weight; the sampling prior "
|
|
455
|
+
"stays as it was installed.")
|
|
456
|
+
return None, note
|
|
457
|
+
if config.generation != "off":
|
|
458
|
+
# Allowed, and announced: the authored sampler draws from lists the
|
|
459
|
+
# harness hands it, and a graded prior reaches it as a narrowed
|
|
460
|
+
# DOMAIN, not as weights.
|
|
461
|
+
say("authorship.adaptation revises the weighted sampling prior the "
|
|
462
|
+
"breeding path draws through; an authored sampler "
|
|
463
|
+
"(authorship.generation='llm') reads narrowed domains, not "
|
|
464
|
+
"weights.")
|
|
465
|
+
|
|
466
|
+
every = (config.adapt_every if config.adapt_every > 0
|
|
467
|
+
else (max(2 * (population_size or 8), (budget or 0) // 6) or 16))
|
|
468
|
+
say(f"authorship.adaptation='llm': the sampling prior is revised every "
|
|
469
|
+
f"{every} charged evaluations, at most {config.adapt_max} times, "
|
|
470
|
+
f"damped at {config.adapt_damping:g}"
|
|
471
|
+
+ (f", with up to {config.adapt_immigrants} immigrant "
|
|
472
|
+
"configuration(s) per revision." if config.adapt_immigrants
|
|
473
|
+
else "."))
|
|
474
|
+
if config.adapt_gate_reads_view:
|
|
475
|
+
# Never silent: a run whose gate reads a view instead of its own
|
|
476
|
+
# measurements is a CONTROL, and a control that does not say so is
|
|
477
|
+
# indistinguishable from the product.
|
|
478
|
+
say("authorship.adapt_gate_reads_view=True: the zero-on-front check "
|
|
479
|
+
"reads the rows this run SHOWED the model, not the rows it "
|
|
480
|
+
"measured. That is a control arm's setting -- it makes the "
|
|
481
|
+
"control's refusal rate comparable to the arm it controls -- and "
|
|
482
|
+
"it is not the protection a live run wants.")
|
|
483
|
+
return Reguidance(
|
|
484
|
+
complete,
|
|
485
|
+
objectives=list(objectives),
|
|
486
|
+
candidate_model=candidate_model,
|
|
487
|
+
template=dict(template),
|
|
488
|
+
domain_context=schema_text,
|
|
489
|
+
every=int(every),
|
|
490
|
+
max_events=config.adapt_max,
|
|
491
|
+
immigrants=config.adapt_immigrants,
|
|
492
|
+
damping=config.adapt_damping,
|
|
493
|
+
max_weight_ratio=8.0,
|
|
494
|
+
evidence_view=config.adapt_evidence_view,
|
|
495
|
+
gate_reads_view=config.adapt_gate_reads_view,
|
|
496
|
+
telemetry=telemetry,
|
|
497
|
+
), None
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def _build_initialization(config, complete, schema_text, say,
|
|
501
|
+
candidate_model, template, k):
|
|
502
|
+
if config.initialization == "off":
|
|
503
|
+
return (), None
|
|
504
|
+
from agent_evolve.policies.llm_init import (InitTelemetry,
|
|
505
|
+
author_initial_population)
|
|
506
|
+
telemetry = InitTelemetry()
|
|
507
|
+
note = SimpleNamespace(telemetry=telemetry, mechanism="init_author",
|
|
508
|
+
authored_by="llm")
|
|
509
|
+
if complete is None:
|
|
510
|
+
say("authorship.initialization='llm' needs a model call and none is "
|
|
511
|
+
"available; initialization stays schema-uniform.")
|
|
512
|
+
return (), note
|
|
513
|
+
if template is None or not k:
|
|
514
|
+
say("authorship.initialization='llm' received no template/size; "
|
|
515
|
+
"initialization stays schema-uniform.")
|
|
516
|
+
return (), note
|
|
517
|
+
if config.init_style != "joint":
|
|
518
|
+
k_exploit = (int(k) + 1) // 2
|
|
519
|
+
say(f"authorship.init_style='{config.init_style}': the one "
|
|
520
|
+
f"initialization call asks for {k_exploit} strongest individual "
|
|
521
|
+
f"bets and {int(k) - k_exploit} coverage members, labelled, "
|
|
522
|
+
"instead of one strong-and-diverse list.")
|
|
523
|
+
proposals = author_initial_population(
|
|
524
|
+
complete, candidate_model=candidate_model, template=dict(template),
|
|
525
|
+
k=int(k), domain_context=schema_text, telemetry=telemetry,
|
|
526
|
+
style=config.init_style)
|
|
527
|
+
if not proposals:
|
|
528
|
+
say("the model proposed no usable initial members; initialization "
|
|
529
|
+
"stays schema-uniform.")
|
|
530
|
+
return tuple(proposals), note
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def _build_generator(
|
|
534
|
+
config: AuthorshipConfig,
|
|
535
|
+
complete: Any,
|
|
536
|
+
objectives: Sequence[Any],
|
|
537
|
+
schema_text: str,
|
|
538
|
+
say: Callable[[str], None],
|
|
539
|
+
candidate_model: Any = None,
|
|
540
|
+
template: Any = None,
|
|
541
|
+
) -> tuple:
|
|
542
|
+
"""``(generator, author_note)``; both ``None`` when generation is off.
|
|
543
|
+
|
|
544
|
+
One authoring call before any evaluation buys a distribution that shapes
|
|
545
|
+
every candidate the run draws -- the amortization that makes this the
|
|
546
|
+
mechanism for cheap-evaluation venues, where per-decision guidance has
|
|
547
|
+
leverage 1/budget. No usable authorship degrades to the shipped
|
|
548
|
+
schema-uniform sampler, out loud: the loop simply keeps drawing the way
|
|
549
|
+
it always did, and the note carries the counters that say why.
|
|
550
|
+
"""
|
|
551
|
+
|
|
552
|
+
if config.generation == "off":
|
|
553
|
+
return None, None
|
|
554
|
+
from agent_evolve.policies.llm_generator import AuthoredGenerator
|
|
555
|
+
from agent_evolve.policies.llm_surrogate import AuthorTelemetry
|
|
556
|
+
|
|
557
|
+
telemetry = AuthorTelemetry()
|
|
558
|
+
author_note = SimpleNamespace(
|
|
559
|
+
telemetry=telemetry, mechanism="generator_author", authored_by="llm"
|
|
560
|
+
)
|
|
561
|
+
if complete is None:
|
|
562
|
+
say("authorship.generation='llm' needs a model call and none is "
|
|
563
|
+
"available; candidates stay schema-uniform.")
|
|
564
|
+
return None, author_note
|
|
565
|
+
from agent_evolve.infrastructure.authored_runtime import AuthoredRuntime
|
|
566
|
+
from agent_evolve.policies.llm_generator import (author_generator,
|
|
567
|
+
revise_generator)
|
|
568
|
+
|
|
569
|
+
# The per-locus admissible sets the run will actually pass, echoed into
|
|
570
|
+
# the authoring prompt. Derived from the problem's own schema exactly as
|
|
571
|
+
# the sampler derives them; absent a template there are no loci to name,
|
|
572
|
+
# and the prompt keeps the field-level card it always had.
|
|
573
|
+
domains = _generator_domains(candidate_model, template)
|
|
574
|
+
artifact = author_generator(
|
|
575
|
+
complete, objectives=list(objectives), schema_text=schema_text,
|
|
576
|
+
attempts=config.authoring_attempts, telemetry=telemetry,
|
|
577
|
+
domains=domains, scaffold=config.generation_scaffold,
|
|
578
|
+
limits=(config.limits if config.generation_limits_echo else None),
|
|
579
|
+
max_n=_generator_max_n(config))
|
|
580
|
+
if artifact is None:
|
|
581
|
+
say(f"the model authored no usable generator in "
|
|
582
|
+
f"{config.authoring_attempts} attempt(s); candidates stay "
|
|
583
|
+
"schema-uniform.")
|
|
584
|
+
return None, author_note
|
|
585
|
+
|
|
586
|
+
holder: dict = {}
|
|
587
|
+
|
|
588
|
+
def revise(current: Any, feedback: str) -> Any:
|
|
589
|
+
# The evolving generator: its own source plus what the harness
|
|
590
|
+
# measured about the candidates it emitted -- which loci rejected,
|
|
591
|
+
# why, with a sample, and which edits already failed. Same gate as
|
|
592
|
+
# authoring. The echo is the RUN's domains when a batch has been
|
|
593
|
+
# emitted (a restriction may have narrowed them), the declared ones
|
|
594
|
+
# otherwise.
|
|
595
|
+
live = getattr(holder.get("generator"), "_domains", None)
|
|
596
|
+
return revise_generator(
|
|
597
|
+
complete, artifact=current, feedback=feedback,
|
|
598
|
+
telemetry=telemetry, domains=live or domains,
|
|
599
|
+
scaffold=config.generation_scaffold,
|
|
600
|
+
limits=(config.limits if config.generation_limits_echo else None),
|
|
601
|
+
max_n=_generator_max_n(config))
|
|
602
|
+
|
|
603
|
+
def reauthor(current: Any, evidence: str) -> Any:
|
|
604
|
+
# The OTHER channel: the same artifact, the same gate, and a prompt
|
|
605
|
+
# carrying the run's measured trace instead of its emission counters.
|
|
606
|
+
from agent_evolve.policies.llm_generator import reauthor_generator
|
|
607
|
+
return reauthor_generator(
|
|
608
|
+
complete, artifact=current, evidence=evidence,
|
|
609
|
+
telemetry=telemetry, scaffold=config.generation_scaffold,
|
|
610
|
+
limits=(config.limits if config.generation_limits_echo else None),
|
|
611
|
+
max_n=_generator_max_n(config))
|
|
612
|
+
|
|
613
|
+
conditioned = config.generation_reauthor_every > 0
|
|
614
|
+
generator = AuthoredGenerator(
|
|
615
|
+
artifact,
|
|
616
|
+
AuthoredRuntime(limits=config.limits),
|
|
617
|
+
pool_factor=config.generation_pool_factor,
|
|
618
|
+
pool_size=config.generation_pool_size,
|
|
619
|
+
revise=revise if config.generation_revisions > 0 else None,
|
|
620
|
+
max_revisions=config.generation_revisions,
|
|
621
|
+
scaffold=config.generation_scaffold,
|
|
622
|
+
repair=config.generation_repair,
|
|
623
|
+
revision_guard=config.generation_revision_guard,
|
|
624
|
+
shrink_on_overrun=(4 if config.generation_limits_echo else 0),
|
|
625
|
+
objectives=tuple(objectives),
|
|
626
|
+
reauthor=(reauthor if conditioned
|
|
627
|
+
and config.generation_reauthorings > 0 else None),
|
|
628
|
+
reauthor_every=config.generation_reauthor_every,
|
|
629
|
+
evidence_min_rows=config.generation_evidence_min_rows,
|
|
630
|
+
max_reauthorings=config.generation_reauthorings,
|
|
631
|
+
prior_author=(complete if conditioned
|
|
632
|
+
and config.generation_locus_prior else None),
|
|
633
|
+
max_priors=config.generation_locus_priors,
|
|
634
|
+
prior_max_weight_ratio=config.generation_prior_max_weight_ratio,
|
|
635
|
+
)
|
|
636
|
+
holder["generator"] = generator
|
|
637
|
+
generator.author = author_note
|
|
638
|
+
return generator, None
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
def _generator_max_n(config: "AuthorshipConfig") -> int:
|
|
642
|
+
"""The largest pool one call may be asked for, for the prompt's echo.
|
|
643
|
+
|
|
644
|
+
Sized from the same two knobs the sampler is: an explicit pool size when
|
|
645
|
+
one is set, otherwise the factor times the offspring a generation can
|
|
646
|
+
afford. The population sizing caps offspring at ten, so this is an upper
|
|
647
|
+
bound rather than a guess.
|
|
648
|
+
"""
|
|
649
|
+
|
|
650
|
+
if config.generation_pool_size:
|
|
651
|
+
return int(config.generation_pool_size)
|
|
652
|
+
return int(config.generation_pool_factor) * 10
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def _generator_domains(candidate_model: Any, template: Any) -> dict:
|
|
656
|
+
"""``{locus name: admissible values}`` for the authoring prompt's echo.
|
|
657
|
+
|
|
658
|
+
Empty whenever the caller supplied no template or no candidate model --
|
|
659
|
+
there is then nothing to echo, and the prompt is exactly the one every
|
|
660
|
+
sealed row was authored under.
|
|
661
|
+
"""
|
|
662
|
+
|
|
663
|
+
if candidate_model is None or not template:
|
|
664
|
+
return {}
|
|
665
|
+
from agent_evolve.policies.genetic import loci_of, locus_domain
|
|
666
|
+
|
|
667
|
+
try:
|
|
668
|
+
return {str(locus): list(locus_domain(candidate_model, locus))
|
|
669
|
+
for locus in loci_of(dict(template))}
|
|
670
|
+
except Exception:
|
|
671
|
+
return {}
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
def _build_screening(
|
|
675
|
+
config: AuthorshipConfig,
|
|
676
|
+
complete: Any,
|
|
677
|
+
objectives: Sequence[Any],
|
|
678
|
+
schema_text: str,
|
|
679
|
+
say: Callable[[str], None],
|
|
680
|
+
) -> Optional[Any]:
|
|
681
|
+
if config.surrogate == "off":
|
|
682
|
+
return None
|
|
683
|
+
from agent_evolve.policies.surrogate import (
|
|
684
|
+
ORDERING_GATE,
|
|
685
|
+
additive_surrogate,
|
|
686
|
+
knn_surrogate,
|
|
687
|
+
)
|
|
688
|
+
from agent_evolve.session.screening import Screening
|
|
689
|
+
|
|
690
|
+
# The screen sorts candidates and never reads a predicted magnitude, so
|
|
691
|
+
# it is gated on rank fidelity and arbitrated on error. The revision hook
|
|
692
|
+
# below must judge the artifact under the SAME policy, or the residual
|
|
693
|
+
# feedback the model receives describes a different bar from the one it
|
|
694
|
+
# is actually held to.
|
|
695
|
+
screening_gate = ORDERING_GATE
|
|
696
|
+
if config.screen_min_passing_objectives:
|
|
697
|
+
# A partial verdict certifies the artifact for the objectives it can
|
|
698
|
+
# order and leaves the rest unknown; the screen then orders on those
|
|
699
|
+
# and reserves more of the generation for unscreened picks.
|
|
700
|
+
screening_gate = ORDERING_GATE.replace(
|
|
701
|
+
min_passing_objectives=config.screen_min_passing_objectives)
|
|
702
|
+
builders: list = []
|
|
703
|
+
author_note: Optional[SimpleNamespace] = None
|
|
704
|
+
if config.surrogate == "llm":
|
|
705
|
+
from agent_evolve.infrastructure.authored_runtime import AuthoredRuntime
|
|
706
|
+
from agent_evolve.policies.llm_surrogate import (
|
|
707
|
+
AuthorTelemetry,
|
|
708
|
+
author_surrogate,
|
|
709
|
+
authored_surrogate_builder,
|
|
710
|
+
)
|
|
711
|
+
|
|
712
|
+
telemetry = AuthorTelemetry()
|
|
713
|
+
author_note = SimpleNamespace(
|
|
714
|
+
telemetry=telemetry, mechanism="surrogate_author", authored_by="llm"
|
|
715
|
+
)
|
|
716
|
+
if complete is None:
|
|
717
|
+
say(
|
|
718
|
+
"authorship.surrogate='llm' needs a model call and none is "
|
|
719
|
+
"available; the rule surrogates carry the screen."
|
|
720
|
+
)
|
|
721
|
+
else:
|
|
722
|
+
artifact = author_surrogate(
|
|
723
|
+
complete,
|
|
724
|
+
objectives=list(objectives),
|
|
725
|
+
schema_text=schema_text,
|
|
726
|
+
attempts=config.authoring_attempts,
|
|
727
|
+
telemetry=telemetry,
|
|
728
|
+
)
|
|
729
|
+
if artifact is None:
|
|
730
|
+
say(
|
|
731
|
+
"the model authored no usable surrogate in "
|
|
732
|
+
f"{config.authoring_attempts} attempt(s); the rule "
|
|
733
|
+
"surrogates carry the screen."
|
|
734
|
+
)
|
|
735
|
+
else:
|
|
736
|
+
runtime = AuthoredRuntime(limits=config.limits)
|
|
737
|
+
builders.append((
|
|
738
|
+
f"llm:{artifact.source_sha256[:8]}",
|
|
739
|
+
"llm",
|
|
740
|
+
authored_surrogate_builder(artifact, runtime),
|
|
741
|
+
))
|
|
742
|
+
|
|
743
|
+
builders.extend([
|
|
744
|
+
("additive", "rule", additive_surrogate),
|
|
745
|
+
("knn", "rule", knn_surrogate),
|
|
746
|
+
])
|
|
747
|
+
|
|
748
|
+
revise = None
|
|
749
|
+
if config.surrogate == "llm" and complete is not None and builders[0][1] == "llm":
|
|
750
|
+
# The evolving surrogate: when the authored artifact loses to the
|
|
751
|
+
# rules, the model sees its own source plus the measured residuals
|
|
752
|
+
# and revises it. Structure from meaning, correction from data.
|
|
753
|
+
# builders[0] being llm guarantees the authoring above succeeded.
|
|
754
|
+
state = {"artifact": artifact}
|
|
755
|
+
|
|
756
|
+
def revise(evaluated, specs):
|
|
757
|
+
from agent_evolve.policies.llm_surrogate import (
|
|
758
|
+
render_validation_feedback, revise_surrogate)
|
|
759
|
+
from agent_evolve.policies.surrogate import validate_surrogate
|
|
760
|
+
|
|
761
|
+
current = state["artifact"]
|
|
762
|
+
if current is None:
|
|
763
|
+
return None
|
|
764
|
+
builder = authored_surrogate_builder(current, runtime)
|
|
765
|
+
# Under the gate the SCREEN uses, or the feedback would describe a
|
|
766
|
+
# bar the artifact is not actually judged against.
|
|
767
|
+
verdict = validate_surrogate(builder, evaluated, specs, seed=1,
|
|
768
|
+
policy=screening_gate)
|
|
769
|
+
try:
|
|
770
|
+
predictions = builder(list(evaluated), specs)(
|
|
771
|
+
[candidate_config for candidate_config, _obj in evaluated])
|
|
772
|
+
except Exception:
|
|
773
|
+
predictions = None
|
|
774
|
+
feedback = render_validation_feedback(
|
|
775
|
+
verdict, evaluated, predictions)
|
|
776
|
+
revised = revise_surrogate(
|
|
777
|
+
complete, artifact=current, feedback=feedback,
|
|
778
|
+
telemetry=telemetry)
|
|
779
|
+
if revised is None:
|
|
780
|
+
return None
|
|
781
|
+
state["artifact"] = revised
|
|
782
|
+
return (f"llm:{revised.source_sha256[:8]}", "llm",
|
|
783
|
+
authored_surrogate_builder(revised, runtime))
|
|
784
|
+
|
|
785
|
+
screening = Screening(
|
|
786
|
+
builders=tuple(builders),
|
|
787
|
+
pool_factor=config.pool_factor,
|
|
788
|
+
exploration_floor=config.exploration_floor,
|
|
789
|
+
unscreened_objective_floor=config.screen_unscreened_objective_floor,
|
|
790
|
+
revise=revise,
|
|
791
|
+
max_training_rows=config.screen_training_rows,
|
|
792
|
+
gate=screening_gate,
|
|
793
|
+
)
|
|
794
|
+
if author_note is not None:
|
|
795
|
+
# Harvested beside the screen's own counters: how authoring went is
|
|
796
|
+
# part of the run's story even when nothing usable came back.
|
|
797
|
+
screening.author = author_note
|
|
798
|
+
return screening
|
|
799
|
+
|
|
800
|
+
|
|
801
|
+
def _build_portfolio(
|
|
802
|
+
config: AuthorshipConfig,
|
|
803
|
+
complete: Any,
|
|
804
|
+
objectives: Sequence[Any],
|
|
805
|
+
schema_text: str,
|
|
806
|
+
say: Callable[[str], None],
|
|
807
|
+
) -> Optional[Any]:
|
|
808
|
+
if config.operators == "off":
|
|
809
|
+
return None
|
|
810
|
+
from agent_evolve.policies.operator_portfolio import (
|
|
811
|
+
OperatorPortfolio,
|
|
812
|
+
VariationArm,
|
|
813
|
+
classical_arm,
|
|
814
|
+
segment_arm,
|
|
815
|
+
)
|
|
816
|
+
|
|
817
|
+
arms: list = [classical_arm(), segment_arm()]
|
|
818
|
+
runtime = None
|
|
819
|
+
author_note: Optional[SimpleNamespace] = None
|
|
820
|
+
if config.operators == "llm":
|
|
821
|
+
from agent_evolve.policies.llm_operator import author_operators
|
|
822
|
+
from agent_evolve.policies.llm_surrogate import AuthorTelemetry
|
|
823
|
+
|
|
824
|
+
telemetry = AuthorTelemetry()
|
|
825
|
+
author_note = SimpleNamespace(
|
|
826
|
+
telemetry=telemetry, mechanism="operator_author", authored_by="llm"
|
|
827
|
+
)
|
|
828
|
+
if complete is None:
|
|
829
|
+
say(
|
|
830
|
+
"authorship.operators='llm' needs a model call and none is "
|
|
831
|
+
"available; the rule arms carry the portfolio."
|
|
832
|
+
)
|
|
833
|
+
else:
|
|
834
|
+
artifacts = author_operators(
|
|
835
|
+
complete,
|
|
836
|
+
objectives=list(objectives),
|
|
837
|
+
schema_text=schema_text,
|
|
838
|
+
attempts=config.authoring_attempts,
|
|
839
|
+
telemetry=telemetry,
|
|
840
|
+
)
|
|
841
|
+
if not artifacts:
|
|
842
|
+
say(
|
|
843
|
+
"the model authored no usable operator in "
|
|
844
|
+
f"{config.authoring_attempts} attempt(s); the rule arms "
|
|
845
|
+
"carry the portfolio."
|
|
846
|
+
)
|
|
847
|
+
else:
|
|
848
|
+
from agent_evolve.infrastructure.authored_runtime import (
|
|
849
|
+
AuthoredRuntime)
|
|
850
|
+
runtime = AuthoredRuntime(limits=config.limits)
|
|
851
|
+
arms.extend(
|
|
852
|
+
VariationArm(
|
|
853
|
+
name=f"{artifact.name}:{artifact.source_sha256[:8]}",
|
|
854
|
+
kind="authored", artifact=artifact,
|
|
855
|
+
)
|
|
856
|
+
for artifact in artifacts
|
|
857
|
+
)
|
|
858
|
+
portfolio = OperatorPortfolio(
|
|
859
|
+
arms, runtime=runtime,
|
|
860
|
+
max_authored_fraction=config.max_authored_fraction,
|
|
861
|
+
)
|
|
862
|
+
if author_note is not None:
|
|
863
|
+
portfolio.author = author_note
|
|
864
|
+
return portfolio
|