agentevolve-optimizer 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (414) hide show
  1. agent_evolve/__init__.py +722 -0
  2. agent_evolve/agentic.py +2800 -0
  3. agent_evolve/api.py +767 -0
  4. agent_evolve/application/__init__.py +1580 -0
  5. agent_evolve/application/action_allocation.py +744 -0
  6. agent_evolve/application/action_allocation_frame.py +347 -0
  7. agent_evolve/application/action_allocation_frame_commit.py +185 -0
  8. agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
  9. agent_evolve/application/action_allocation_frame_v3.py +338 -0
  10. agent_evolve/application/action_archive_value.py +497 -0
  11. agent_evolve/application/action_evidence_consistency.py +455 -0
  12. agent_evolve/application/action_forecast_partitioning.py +1471 -0
  13. agent_evolve/application/action_metric_projection.py +211 -0
  14. agent_evolve/application/action_role_value.py +680 -0
  15. agent_evolve/application/action_score_authorities.py +363 -0
  16. agent_evolve/application/action_structural_signature.py +116 -0
  17. agent_evolve/application/action_target_realization.py +402 -0
  18. agent_evolve/application/agentic_evolution.py +7734 -0
  19. agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
  20. agent_evolve/application/anchor_residual_identification.py +463 -0
  21. agent_evolve/application/archive_conditioned_action_target.py +208 -0
  22. agent_evolve/application/artifact_journal.py +246 -0
  23. agent_evolve/application/artifact_replay.py +347 -0
  24. agent_evolve/application/budgeted_optimizer.py +1828 -0
  25. agent_evolve/application/calibrated_campaign.py +485 -0
  26. agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
  27. agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
  28. agent_evolve/application/campaign_capacity_recourse.py +254 -0
  29. agent_evolve/application/campaign_contextual_outcomes.py +119 -0
  30. agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
  31. agent_evolve/application/campaign_evidence_registry.py +262 -0
  32. agent_evolve/application/campaign_execution.py +2537 -0
  33. agent_evolve/application/campaign_generation_audit.py +942 -0
  34. agent_evolve/application/campaign_learning.py +1812 -0
  35. agent_evolve/application/campaign_learning_runtime.py +1977 -0
  36. agent_evolve/application/campaign_search_phase.py +227 -0
  37. agent_evolve/application/campaign_selector_context_extension.py +220 -0
  38. agent_evolve/application/campaign_variation_envelope.py +649 -0
  39. agent_evolve/application/campaign_variation_trace.py +451 -0
  40. agent_evolve/application/candidate_archive_consequence.py +128 -0
  41. agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
  42. agent_evolve/application/composite_outcome_updater.py +145 -0
  43. agent_evolve/application/composition_portfolio_selection.py +363 -0
  44. agent_evolve/application/concurrent_stage.py +144 -0
  45. agent_evolve/application/contextual_action_allocation.py +181 -0
  46. agent_evolve/application/contextual_campaign_outcomes.py +267 -0
  47. agent_evolve/application/contextual_campaign_planning.py +1366 -0
  48. agent_evolve/application/contextual_delayed_credit.py +651 -0
  49. agent_evolve/application/contextual_search_controller.py +2374 -0
  50. agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
  51. agent_evolve/application/decision_metric_projection.py +112 -0
  52. agent_evolve/application/derived_action_semantics.py +129 -0
  53. agent_evolve/application/detailed_evaluation.py +449 -0
  54. agent_evolve/application/earned_lineage.py +1011 -0
  55. agent_evolve/application/effective_choice_audit.py +484 -0
  56. agent_evolve/application/empirical_consequence_calibration.py +908 -0
  57. agent_evolve/application/evaluation_accounting.py +325 -0
  58. agent_evolve/application/evaluation_cache.py +199 -0
  59. agent_evolve/application/evaluation_escrow.py +547 -0
  60. agent_evolve/application/evaluation_recourse.py +253 -0
  61. agent_evolve/application/event_recorder.py +151 -0
  62. agent_evolve/application/evolution_campaign.py +1840 -0
  63. agent_evolve/application/executable_hypothesis.py +323 -0
  64. agent_evolve/application/factorial_branch_pilot.py +772 -0
  65. agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
  66. agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
  67. agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
  68. agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
  69. agent_evolve/application/finite_action_selection.py +188 -0
  70. agent_evolve/application/finite_action_set.py +306 -0
  71. agent_evolve/application/finite_action_transition.py +537 -0
  72. agent_evolve/application/finite_variation_eligibility.py +296 -0
  73. agent_evolve/application/forecast_geometry_portfolio.py +799 -0
  74. agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
  75. agent_evolve/application/front_proximity_admission.py +311 -0
  76. agent_evolve/application/front_proximity_parent_basis.py +458 -0
  77. agent_evolve/application/frozen_hurdle_score.py +659 -0
  78. agent_evolve/application/g3_causal_screen.py +2257 -0
  79. agent_evolve/application/g3_causal_validation.py +1046 -0
  80. agent_evolve/application/g3_postseal_curation.py +818 -0
  81. agent_evolve/application/gated_agentic_generator.py +205 -0
  82. agent_evolve/application/generation_feedback.py +293 -0
  83. agent_evolve/application/generative_proposal_journal.py +185 -0
  84. agent_evolve/application/geometry_conditional_elasticity.py +453 -0
  85. agent_evolve/application/global_wave_action_allocation.py +1151 -0
  86. agent_evolve/application/head_mass_conditional_seat.py +268 -0
  87. agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
  88. agent_evolve/application/identifiable_reflection_learning.py +395 -0
  89. agent_evolve/application/identifiable_reflection_request.py +364 -0
  90. agent_evolve/application/in_memory_residual_archive.py +341 -0
  91. agent_evolve/application/insight_memory.py +1804 -0
  92. agent_evolve/application/live_runtime_manifest.py +758 -0
  93. agent_evolve/application/llm_task_queue.py +769 -0
  94. agent_evolve/application/matched_finite_action_block.py +409 -0
  95. agent_evolve/application/materialized_action_broker.py +2328 -0
  96. agent_evolve/application/materialized_action_constraints.py +83 -0
  97. agent_evolve/application/materialized_variation.py +211 -0
  98. agent_evolve/application/multi_option_evolution.py +1536 -0
  99. agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
  100. agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
  101. agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
  102. agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
  103. agent_evolve/application/outcome_relation.py +193 -0
  104. agent_evolve/application/paired_allocation_comparison.py +241 -0
  105. agent_evolve/application/paired_block_schedule.py +127 -0
  106. agent_evolve/application/parent_measurement.py +226 -0
  107. agent_evolve/application/pareto_archive.py +811 -0
  108. agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
  109. agent_evolve/application/portfolio_evolution.py +2950 -0
  110. agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
  111. agent_evolve/application/portfolio_memory_attribution.py +581 -0
  112. agent_evolve/application/portfolio_memory_dose.py +788 -0
  113. agent_evolve/application/portfolio_memory_matched_control.py +938 -0
  114. agent_evolve/application/portfolio_memory_transfer.py +297 -0
  115. agent_evolve/application/portfolio_optimization_memory.py +363 -0
  116. agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
  117. agent_evolve/application/portfolio_projection.py +335 -0
  118. agent_evolve/application/portfolio_recombination.py +2032 -0
  119. agent_evolve/application/post_evolution_reflection.py +834 -0
  120. agent_evolve/application/postcommit_rank_authority.py +245 -0
  121. agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
  122. agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
  123. agent_evolve/application/prequential_residual_exploration.py +343 -0
  124. agent_evolve/application/prequential_score_portfolio.py +954 -0
  125. agent_evolve/application/projections.py +292 -0
  126. agent_evolve/application/protected_action_committee.py +1027 -0
  127. agent_evolve/application/protected_branch_pilot.py +376 -0
  128. agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
  129. agent_evolve/application/provider_replay.py +910 -0
  130. agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
  131. agent_evolve/application/recombination_residual_expert.py +403 -0
  132. agent_evolve/application/reflection_workflow.py +571 -0
  133. agent_evolve/application/region_conditional_credit.py +911 -0
  134. agent_evolve/application/residual_campaign_runtime.py +531 -0
  135. agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
  136. agent_evolve/application/residual_headroom_ledger.py +1544 -0
  137. agent_evolve/application/residual_learning_transaction.py +396 -0
  138. agent_evolve/application/residual_portfolio_evolution.py +1228 -0
  139. agent_evolve/application/residual_reachability.py +749 -0
  140. agent_evolve/application/residual_stage_credit.py +499 -0
  141. agent_evolve/application/same_prefix_paired_audit.py +1580 -0
  142. agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
  143. agent_evolve/application/sequential_lineage_allocation.py +1017 -0
  144. agent_evolve/application/sequential_market_replay.py +1395 -0
  145. agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
  146. agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
  147. agent_evolve/application/single_score_action_allocation.py +299 -0
  148. agent_evolve/application/source_exposure_allocation.py +906 -0
  149. agent_evolve/application/staged_memory.py +210 -0
  150. agent_evolve/application/stratified_cold_start_allocation.py +732 -0
  151. agent_evolve/application/support_guarded_hurdle_score.py +549 -0
  152. agent_evolve/application/target_conditioned_action_forecast.py +595 -0
  153. agent_evolve/application/target_conditioned_campaign.py +566 -0
  154. agent_evolve/application/treatment_assignment.py +201 -0
  155. agent_evolve/application/trusted_objective_evidence.py +217 -0
  156. agent_evolve/application/two_stage_action_evolution.py +1131 -0
  157. agent_evolve/application/v8lite_allocation_policy.py +1083 -0
  158. agent_evolve/application/v9_candidate_policy.py +1303 -0
  159. agent_evolve/bootstrap.py +108 -0
  160. agent_evolve/campaign_presets.py +517 -0
  161. agent_evolve/campaign_profiles.py +452 -0
  162. agent_evolve/campaign_variation_topology.py +288 -0
  163. agent_evolve/campaign_workload.py +950 -0
  164. agent_evolve/cli.py +797 -0
  165. agent_evolve/contract.py +241 -0
  166. agent_evolve/core/__init__.py +91 -0
  167. agent_evolve/core/action_semantics.py +411 -0
  168. agent_evolve/core/authored.py +105 -0
  169. agent_evolve/core/formatting.py +286 -0
  170. agent_evolve/core/optimization_semantics.py +324 -0
  171. agent_evolve/core/problem.py +167 -0
  172. agent_evolve/core/results.py +323 -0
  173. agent_evolve/core/stats.py +70 -0
  174. agent_evolve/core/telemetry.py +100 -0
  175. agent_evolve/domain/__init__.py +89 -0
  176. agent_evolve/domain/artifact.py +162 -0
  177. agent_evolve/domain/durable_text.py +68 -0
  178. agent_evolve/domain/event.py +1454 -0
  179. agent_evolve/domain/finite_action_set.py +426 -0
  180. agent_evolve/domain/finite_variation.py +526 -0
  181. agent_evolve/domain/generative_emission.py +559 -0
  182. agent_evolve/domain/ids.py +163 -0
  183. agent_evolve/domain/inline_text.py +106 -0
  184. agent_evolve/domain/insight.py +27 -0
  185. agent_evolve/domain/lineage.py +737 -0
  186. agent_evolve/domain/llm_task_queue.py +960 -0
  187. agent_evolve/domain/outcome.py +96 -0
  188. agent_evolve/domain/patch.py +854 -0
  189. agent_evolve/domain/typed_json.py +542 -0
  190. agent_evolve/domain/variation_space.py +158 -0
  191. agent_evolve/driver.py +1014 -0
  192. agent_evolve/harness/__init__.py +29 -0
  193. agent_evolve/harness/base.py +242 -0
  194. agent_evolve/harness/directives.py +163 -0
  195. agent_evolve/harness/generative_seal.py +479 -0
  196. agent_evolve/harness/registry.py +41 -0
  197. agent_evolve/infrastructure/__init__.py +39 -0
  198. agent_evolve/infrastructure/artifacts/__init__.py +6 -0
  199. agent_evolve/infrastructure/artifacts/_verification.py +67 -0
  200. agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
  201. agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
  202. agent_evolve/infrastructure/asyncio_runtime.py +109 -0
  203. agent_evolve/infrastructure/authored_runtime.py +188 -0
  204. agent_evolve/infrastructure/authored_worker.py +171 -0
  205. agent_evolve/infrastructure/clock.py +53 -0
  206. agent_evolve/infrastructure/events/__init__.py +6 -0
  207. agent_evolve/infrastructure/events/_validation.py +89 -0
  208. agent_evolve/infrastructure/events/in_memory.py +56 -0
  209. agent_evolve/infrastructure/events/jsonl.py +193 -0
  210. agent_evolve/infrastructure/exception_provenance.py +215 -0
  211. agent_evolve/infrastructure/ids.py +118 -0
  212. agent_evolve/infrastructure/lineage_codec.py +1836 -0
  213. agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
  214. agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
  215. agent_evolve/infrastructure/resource_lease.py +370 -0
  216. agent_evolve/infrastructure/sanitization/__init__.py +8 -0
  217. agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
  218. agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
  219. agent_evolve/infrastructure/stream_liveness.py +383 -0
  220. agent_evolve/infrastructure/subprocess_boundary.py +136 -0
  221. agent_evolve/integrations/__init__.py +1 -0
  222. agent_evolve/integrations/botorch/__init__.py +28 -0
  223. agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
  224. agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
  225. agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
  226. agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
  227. agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
  228. agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
  229. agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
  230. agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
  231. agent_evolve/integrations/completion.py +242 -0
  232. agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
  233. agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
  234. agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
  235. agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
  236. agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
  237. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
  238. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
  239. agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
  240. agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
  241. agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
  242. agent_evolve/integrations/pydantic_ai/harness.py +159 -0
  243. agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
  244. agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
  245. agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
  246. agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
  247. agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
  248. agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
  249. agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
  250. agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
  251. agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
  252. agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
  253. agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
  254. agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
  255. agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
  256. agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
  257. agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
  258. agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
  259. agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
  260. agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
  261. agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
  262. agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
  263. agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
  264. agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
  265. agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
  266. agent_evolve/integrations/pymoo_adapter.py +242 -0
  267. agent_evolve/policies/__init__.py +17 -0
  268. agent_evolve/policies/check.py +469 -0
  269. agent_evolve/policies/emit_scaffold.py +451 -0
  270. agent_evolve/policies/feedback/__init__.py +37 -0
  271. agent_evolve/policies/feedback/held_out_asn.py +1325 -0
  272. agent_evolve/policies/genetic.py +607 -0
  273. agent_evolve/policies/llm_backoff.py +183 -0
  274. agent_evolve/policies/llm_chooser.py +226 -0
  275. agent_evolve/policies/llm_generator.py +1760 -0
  276. agent_evolve/policies/llm_init.py +267 -0
  277. agent_evolve/policies/llm_operator.py +109 -0
  278. agent_evolve/policies/llm_prior.py +194 -0
  279. agent_evolve/policies/llm_surrogate.py +334 -0
  280. agent_evolve/policies/measurement_evidence.py +704 -0
  281. agent_evolve/policies/memory/__init__.py +223 -0
  282. agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
  283. agent_evolve/policies/memory/compatibility_matching.py +593 -0
  284. agent_evolve/policies/memory/global_falsification.py +1841 -0
  285. agent_evolve/policies/memory/prompt_shape.py +503 -0
  286. agent_evolve/policies/memory/randomized_subset.py +714 -0
  287. agent_evolve/policies/memory/staged_causal.py +1270 -0
  288. agent_evolve/policies/memory/treatment_compliance.py +759 -0
  289. agent_evolve/policies/objective_resolution/__init__.py +17 -0
  290. agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
  291. agent_evolve/policies/operator_portfolio.py +407 -0
  292. agent_evolve/policies/reguidance.py +1133 -0
  293. agent_evolve/policies/reward/__init__.py +83 -0
  294. agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
  295. agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
  296. agent_evolve/policies/reward/affine_hypervolume.py +490 -0
  297. agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
  298. agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
  299. agent_evolve/policies/reward/frozen_archive.py +360 -0
  300. agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
  301. agent_evolve/policies/search_state.py +208 -0
  302. agent_evolve/policies/selection/__init__.py +345 -0
  303. agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
  304. agent_evolve/policies/selection/affine_frontier_context.py +330 -0
  305. agent_evolve/policies/selection/affine_frontier_target.py +473 -0
  306. agent_evolve/policies/selection/archive_elite.py +1346 -0
  307. agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
  308. agent_evolve/policies/selection/calibrated_slate.py +1394 -0
  309. agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
  310. agent_evolve/policies/selection/common_candidate_pool.py +685 -0
  311. agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
  312. agent_evolve/policies/selection/disjoint_pairs.py +479 -0
  313. agent_evolve/policies/selection/elite_explorer.py +719 -0
  314. agent_evolve/policies/selection/finite_action.py +187 -0
  315. agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
  316. agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
  317. agent_evolve/policies/selection/forecast_calibration.py +922 -0
  318. agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
  319. agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
  320. agent_evolve/policies/selection/full_support_slate.py +91 -0
  321. agent_evolve/policies/selection/meaningful_direction.py +240 -0
  322. agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
  323. agent_evolve/policies/selection/model_anchored_slate.py +826 -0
  324. agent_evolve/policies/selection/phenotype_recourse.py +979 -0
  325. agent_evolve/policies/selection/proposal_support.py +368 -0
  326. agent_evolve/policies/selection/random_portfolio.py +254 -0
  327. agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
  328. agent_evolve/policies/selection/residual_frontier.py +463 -0
  329. agent_evolve/policies/selection/residual_frontier_target.py +605 -0
  330. agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
  331. agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
  332. agent_evolve/policies/selection/target_conditioned_features.py +812 -0
  333. agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
  334. agent_evolve/policies/selection/task_keyed_palette.py +906 -0
  335. agent_evolve/policies/semantics.py +147 -0
  336. agent_evolve/policies/structure.py +362 -0
  337. agent_evolve/policies/structured_output_budget.py +62 -0
  338. agent_evolve/policies/surrogate.py +696 -0
  339. agent_evolve/policies/variation/__init__.py +1 -0
  340. agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
  341. agent_evolve/policies/variation/crossover_inheritance.py +575 -0
  342. agent_evolve/policies/variation/disjoint_recombination.py +611 -0
  343. agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
  344. agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
  345. agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
  346. agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
  347. agent_evolve/policies/variation/typed_patch.py +1981 -0
  348. agent_evolve/policies/weighted_prior.py +394 -0
  349. agent_evolve/ports/__init__.py +383 -0
  350. agent_evolve/ports/action_allocation.py +733 -0
  351. agent_evolve/ports/action_allocation_frame.py +1153 -0
  352. agent_evolve/ports/action_allocation_frame_commit.py +294 -0
  353. agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
  354. agent_evolve/ports/action_allocation_frame_v3.py +995 -0
  355. agent_evolve/ports/action_forecast.py +1568 -0
  356. agent_evolve/ports/action_metric_projection.py +165 -0
  357. agent_evolve/ports/agentic_generator.py +1561 -0
  358. agent_evolve/ports/archive_context.py +136 -0
  359. agent_evolve/ports/artifact_sanitizer.py +44 -0
  360. agent_evolve/ports/artifact_store.py +225 -0
  361. agent_evolve/ports/clock.py +13 -0
  362. agent_evolve/ports/contextual_search_allocation.py +827 -0
  363. agent_evolve/ports/decision_metric_projection.py +258 -0
  364. agent_evolve/ports/event_store.py +55 -0
  365. agent_evolve/ports/executable_hypothesis.py +557 -0
  366. agent_evolve/ports/finite_acquisition.py +377 -0
  367. agent_evolve/ports/finite_acquisition_batch.py +296 -0
  368. agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
  369. agent_evolve/ports/finite_acquisition_json.py +247 -0
  370. agent_evolve/ports/finite_acquisition_space.py +168 -0
  371. agent_evolve/ports/finite_action_selection.py +348 -0
  372. agent_evolve/ports/finite_action_set.py +256 -0
  373. agent_evolve/ports/frontier_target.py +396 -0
  374. agent_evolve/ports/generation_failure.py +43 -0
  375. agent_evolve/ports/hard_feasibility.py +233 -0
  376. agent_evolve/ports/id_factory.py +34 -0
  377. agent_evolve/ports/llm_task_queue.py +93 -0
  378. agent_evolve/ports/objective_resolution.py +419 -0
  379. agent_evolve/ports/paired_allocation_comparison.py +401 -0
  380. agent_evolve/ports/paired_block_schedule.py +475 -0
  381. agent_evolve/ports/parent_measurement.py +336 -0
  382. agent_evolve/ports/portfolio_memory_dose.py +643 -0
  383. agent_evolve/ports/portfolio_selection.py +3169 -0
  384. agent_evolve/ports/postcommit_rank_authority.py +467 -0
  385. agent_evolve/ports/presented_action_evidence.py +794 -0
  386. agent_evolve/ports/resource_lease.py +162 -0
  387. agent_evolve/ports/structured_generator.py +734 -0
  388. agent_evolve/ports/structured_output_budget.py +120 -0
  389. agent_evolve/ports/subprocess_boundary.py +138 -0
  390. agent_evolve/ports/treatment_assignment.py +466 -0
  391. agent_evolve/ports/variation_catalog.py +76 -0
  392. agent_evolve/ports/variation_source.py +226 -0
  393. agent_evolve/proposal_mode.py +157 -0
  394. agent_evolve/proposers/__init__.py +10 -0
  395. agent_evolve/proposers/random_proposer.py +188 -0
  396. agent_evolve/provider_accounting.py +163 -0
  397. agent_evolve/py.typed +0 -0
  398. agent_evolve/reference_method.py +1570 -0
  399. agent_evolve/session/__init__.py +11 -0
  400. agent_evolve/session/authorship.py +864 -0
  401. agent_evolve/session/evaluate.py +236 -0
  402. agent_evolve/session/fidelity.py +237 -0
  403. agent_evolve/session/genetic_loop.py +742 -0
  404. agent_evolve/session/loop.py +803 -0
  405. agent_evolve/session/screening.py +671 -0
  406. agent_evolve/settings.py +376 -0
  407. agent_evolve/workload_kit.py +368 -0
  408. agent_evolve/workload_prompt.py +398 -0
  409. agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
  410. agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
  411. agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
  412. agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
  413. agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
  414. agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,671 @@
1
+ """Virtual pre-screening: order a pool of offspring before paying for any.
2
+
3
+ This module is deliberately starved: :func:`screen_offspring` receives
4
+ configurations and objective vectors and a predictor -- never the problem,
5
+ never the evaluation cache -- so "the surrogate cannot spend budget" is a
6
+ property of the import graph, not a convention. The only route from here to
7
+ a real evaluation is that the loop measures the candidates this module
8
+ merely ordered.
9
+
10
+ Because ORDERING is the whole of what this module consumes, it validates its
11
+ surrogates under :data:`~agent_evolve.policies.surrogate.ORDERING_GATE`:
12
+ rank fidelity rejects, and the error ratio against the train-mean predictor
13
+ is computed for arbitration among passers. The screen is the reason that
14
+ distinction exists -- under a gate that also rejected on magnitude it went
15
+ dark on the venues where saving an evaluation is worth anything, ordering 7
16
+ of 186 generations on an expensive venue against 54% on a cheap one.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import math
22
+ import statistics
23
+ from dataclasses import dataclass
24
+ from typing import Any, Dict, Mapping, Optional, Sequence, Tuple
25
+
26
+ from agent_evolve.core.problem import ObjectiveSpec
27
+ from agent_evolve.policies.surrogate import (
28
+ ORDERING_GATE,
29
+ GatePolicy,
30
+ Predict,
31
+ SurrogateBuilder,
32
+ validate_surrogate,
33
+ )
34
+
35
+ __all__ = ["ScreenReport", "screen_offspring", "Screening", "ScreeningTelemetry"]
36
+
37
+ Config = Dict[str, Any]
38
+
39
+
40
+ @dataclass(frozen=True)
41
+ class ScreenReport:
42
+ """The pool, ordered by predicted worth. Indices address the caller's pool.
43
+
44
+ ``screened_objectives`` names the objectives the order was actually
45
+ computed over, and ``declared_objectives`` names the problem's. They
46
+ differ when the gate certified the surrogate on only some of them. A
47
+ consumer that reads ``order`` without reading these two is free to
48
+ believe the pool was ranked on the whole problem when it was ranked on
49
+ part of it, so both travel with the order rather than beside it.
50
+ """
51
+
52
+ order: Tuple[int, ...]
53
+ predicted: Tuple[Mapping[str, float], ...]
54
+ virtual_evaluations: int
55
+ surrogate_name: str
56
+ screened_objectives: Tuple[str, ...] = ()
57
+ declared_objectives: Tuple[str, ...] = ()
58
+
59
+ @property
60
+ def partial(self) -> bool:
61
+ """Was this order computed over a STRICT SUBSET of the objectives?"""
62
+
63
+ return len(self.screened_objectives) < len(self.declared_objectives)
64
+
65
+
66
+ def _oriented(
67
+ rows: Sequence[Mapping[str, float]], specs: Sequence[ObjectiveSpec]
68
+ ) -> Optional[list[tuple]]:
69
+ """Objective vectors as "smaller is better" float tuples, or ``None``.
70
+
71
+ ``None`` means a row was missing an objective or carried something that
72
+ is not a finite number -- the same "screen nothing" answer this module
73
+ already gives for malformed predictions, rather than an exception from
74
+ the middle of a ranking loop.
75
+ """
76
+
77
+ signs = [1.0 if spec.goal == "min" else -1.0 for spec in specs]
78
+ names = [spec.name for spec in specs]
79
+ out: list[tuple] = []
80
+ for row in rows:
81
+ vector = []
82
+ for sign, name in zip(signs, names):
83
+ value = row.get(name)
84
+ if (value is None or isinstance(value, bool)
85
+ or not isinstance(value, (int, float))
86
+ or not math.isfinite(float(value))):
87
+ return None
88
+ vector.append(sign * float(value))
89
+ out.append(tuple(vector))
90
+ return out
91
+
92
+
93
+ def _dominator_counts(
94
+ rows: Sequence[Mapping[str, float]],
95
+ against: Sequence[Mapping[str, float]],
96
+ specs: Sequence[ObjectiveSpec],
97
+ ) -> Optional[list[int]]:
98
+ """How many of ``rows + against`` dominate each of *rows*. Lower is better.
99
+
100
+ Exactly what ``sum(1 for other in field if dominates(other, row))`` says,
101
+ computed over DISTINCT objective vectors weighted by how many rows carry
102
+ each. Dominance is a property of the vector alone, so this returns the
103
+ same integers -- but a pool of n candidates whose predictions take k
104
+ distinct values costs O(k^2) instead of O(n^2), and a surrogate over a
105
+ discrete space collapses thousands of candidates onto tens of vectors.
106
+ The comparison itself works on pre-oriented float tuples: the general
107
+ ``core.results.dominates`` re-validates every objective on every call
108
+ (a ``numbers.Real`` ABC check per number), which is right for a contract
109
+ boundary and ruinous inside a quadratic loop -- measured at 21.7s of a
110
+ 25.6s run before this, on one screened pool of 2,000.
111
+ """
112
+
113
+ keys = _oriented(rows, specs)
114
+ other_keys = _oriented(against, specs)
115
+ if keys is None or other_keys is None:
116
+ return None
117
+
118
+ multiplicity: Dict[tuple, int] = {}
119
+ for key in keys:
120
+ multiplicity[key] = multiplicity.get(key, 0) + 1
121
+ for key in other_keys:
122
+ multiplicity[key] = multiplicity.get(key, 0) + 1
123
+ distinct = list(multiplicity.items())
124
+
125
+ def _beats(a: tuple, b: tuple) -> bool:
126
+ better = False
127
+ for x, y in zip(a, b):
128
+ if x > y:
129
+ return False
130
+ if x < y:
131
+ better = True
132
+ return better
133
+
134
+ counted = {
135
+ key: sum(weight for other, weight in distinct if _beats(other, key))
136
+ for key, _weight in distinct
137
+ }
138
+ return [counted[key] for key in keys]
139
+
140
+
141
+ def screen_offspring(
142
+ pool: Sequence[Config],
143
+ population_objectives: Sequence[Mapping[str, float]],
144
+ specs: Sequence[ObjectiveSpec],
145
+ predict: Predict,
146
+ *,
147
+ surrogate_name: str = "surrogate",
148
+ objectives: Optional[Sequence[str]] = None,
149
+ ) -> Optional[ScreenReport]:
150
+ """Order *pool* by predicted domination against the measured population.
151
+
152
+ A candidate's rank counts how many points dominate its PREDICTION -- other
153
+ predictions and the population's real measurements together -- so a pool
154
+ member that merely reshuffles known-dominated territory sinks, and one
155
+ predicted past the current front rises. Never scalarized. ``None`` (from
156
+ the predictor, or on malformed predictions) means "screen nothing": the
157
+ caller falls back to measuring its original picks.
158
+
159
+ This function consumes an ORDER and nothing else: the returned
160
+ ``predicted`` rows are telemetry, and the loop reads only ``order``. That
161
+ is why the gate this screen validates under is
162
+ :data:`~agent_evolve.policies.surrogate.ORDERING_GATE` -- rank fidelity
163
+ is what the output can be wrong about, and a calibration test on
164
+ magnitudes nobody reads can only reject artifacts that would have
165
+ ordered correctly.
166
+
167
+ ``objectives`` restricts the domination test to the objectives the gate
168
+ certified this surrogate for; ``None`` means all of them. **The excluded
169
+ objectives are treated as UNKNOWN, not as satisfied**: they are neither
170
+ read from the prediction nor compared, so a surrogate that emits nonsense
171
+ on an objective it was not certified for cannot influence the order
172
+ through it. That is a deliberate asymmetry with a cost, stated here
173
+ because a caller must weigh it: domination over a subset is a STRICTER
174
+ relation than domination over the whole (more pairs compare, fewer are
175
+ incomparable), so a candidate that is excellent only on an excluded
176
+ objective is dominated on the subset and sinks. The screen is therefore
177
+ biased against exactly the trade-off it cannot see, and the caller's
178
+ exploration floor -- not this function -- is what keeps unscreened picks
179
+ in the generation (see :meth:`Screening.exploration_floor_for`).
180
+ """
181
+
182
+ if not pool:
183
+ return None
184
+ names = [s.name for s in specs]
185
+ if objectives is None:
186
+ screened = list(names)
187
+ else:
188
+ wanted = set(objectives)
189
+ unknown = wanted - set(names)
190
+ if unknown:
191
+ raise ValueError(
192
+ "objectives to screen on must be declared objectives; "
193
+ f"{sorted(unknown)} are not among {names}")
194
+ screened = [name for name in names if name in wanted]
195
+ # Ordering on nothing is not ordering. A caller that reaches here with an
196
+ # empty subset has a gate bug, and screening the pool by index would hide
197
+ # it behind a plausible-looking order.
198
+ if not screened:
199
+ return None
200
+ specs = [spec for spec in specs if spec.name in set(screened)]
201
+ predictions = predict(list(pool))
202
+ if predictions is None or len(predictions) != len(pool):
203
+ return None
204
+ clean: list[dict[str, float]] = []
205
+ for predicted in predictions:
206
+ row = {}
207
+ for name in screened:
208
+ value = predicted.get(name) if isinstance(predicted, Mapping) else None
209
+ if (value is None or isinstance(value, bool)
210
+ or not isinstance(value, (int, float))
211
+ or not math.isfinite(float(value))):
212
+ return None
213
+ row[name] = float(value)
214
+ clean.append(row)
215
+
216
+ ranks = _dominator_counts(
217
+ clean, [dict(measured) for measured in population_objectives], specs)
218
+ if ranks is None:
219
+ return None
220
+ order = tuple(sorted(range(len(pool)), key=lambda i: (ranks[i], i)))
221
+ return ScreenReport(
222
+ order=order,
223
+ predicted=tuple(clean),
224
+ virtual_evaluations=len(pool),
225
+ surrogate_name=surrogate_name,
226
+ screened_objectives=tuple(screened),
227
+ declared_objectives=tuple(names),
228
+ )
229
+
230
+
231
+ class ScreeningTelemetry:
232
+ """What the screen did, counted. Reaches the result via harvest."""
233
+
234
+ #: Why a builder was rejected, counted per (builder, split) verdict.
235
+ #: Without this a campaign cannot tell "the model is not predictive"
236
+ #: from "the gate never had enough data to look" -- which is exactly the
237
+ #: distinction that turned out to decide whether the mechanism runs at
238
+ #: all -- and every study that needed it had to monkeypatch the gate.
239
+ REJECTIONS = ("rejected_insufficient_rows", "rejected_insufficient_holdout",
240
+ "rejected_rank", "rejected_error", "rejected_builder_failed",
241
+ "rejected_no_predictions", "rejected_bad_prediction")
242
+
243
+ #: The cheap fidelity's own counters, kept BESIDE the charged ones and
244
+ #: never added to them. `proxy_rows_used` is how many gate rows came from
245
+ #: the cheap evaluator on the last refresh; `chosen_proxy` counts the
246
+ #: generations the cheap fidelity itself won the gate and did the
247
+ #: ordering.
248
+ PROXY = ("proxy_rows_used", "chosen_proxy")
249
+ #: ``rejected_unstable_subset`` is counted separately and is NOT a gate
250
+ #: reason: every split passed, but they certified different objectives,
251
+ #: so the artifact is not stably predictive on enough of them. It exists
252
+ #: because a partial verdict makes that failure possible for the first
253
+ #: time, and a campaign must be able to see it rather than read it as
254
+ #: "the gate never had data".
255
+
256
+ #: The prefix under which ``as_dict`` reports, per objective, how many
257
+ #: screens ordered on it. A run that screened on two of three objectives
258
+ #: says so here in a form no reader can mistake for "screened on all
259
+ #: three", and it says it WITHOUT this module knowing any objective name.
260
+ SCREENED_PREFIX = "screened_on:"
261
+
262
+ __slots__ = ("refreshes", "validated", "rejected_validation", "screens",
263
+ "screen_failures", "virtual_evaluations", "chosen_llm",
264
+ "chosen_rule", "revisions", "revisions_accepted",
265
+ "gate_calls", "screens_full", "screens_partial",
266
+ "rejected_unstable_subset",
267
+ "_screened") + REJECTIONS + PROXY
268
+
269
+ def __init__(self) -> None:
270
+ for name in self.__slots__:
271
+ setattr(self, name, 0)
272
+ #: objective name -> screens whose order was computed over it.
273
+ self._screened: Dict[str, int] = {}
274
+
275
+ def record(self, verdict: Any) -> None:
276
+ """Count one gate verdict, passed or rejected and why."""
277
+
278
+ self.gate_calls += 1
279
+ if verdict.passed:
280
+ return
281
+ name = f"rejected_{verdict.reason or 'unknown'}"
282
+ if name in self.REJECTIONS:
283
+ setattr(self, name, getattr(self, name) + 1)
284
+
285
+ def record_screen(self, report: Any) -> None:
286
+ """Count one screen, and WHICH objectives its order was computed over.
287
+
288
+ This is a correctness requirement, not a nicety: an order over a
289
+ subset is a different object from an order over the whole problem,
290
+ and a run that cannot distinguish them can report an endpoint it
291
+ cannot attribute.
292
+ """
293
+
294
+ if report.partial:
295
+ self.screens_partial += 1
296
+ else:
297
+ self.screens_full += 1
298
+ for name in report.screened_objectives:
299
+ self._screened[name] = self._screened.get(name, 0) + 1
300
+
301
+ def as_dict(self) -> dict[str, int]:
302
+ counters = {name: getattr(self, name) for name in self.__slots__
303
+ if not name.startswith("_")}
304
+ for name, count in sorted(self._screened.items()):
305
+ counters[f"{self.SCREENED_PREFIX}{name}"] = count
306
+ return counters
307
+
308
+
309
+ class Screening:
310
+ """The screening policy: builders, the gate, and the current predictor.
311
+
312
+ ``builders`` is an ordered sequence of ``(name, authored_by, builder)``;
313
+ each generation, :meth:`refresh` re-validates them in order on the data
314
+ measured so far and installs the FIRST that passes the gate -- so an
315
+ authored surrogate listed ahead of the rules is used exactly when it
316
+ earns it, and the rules are the standing fallback.
317
+ """
318
+
319
+ def __init__(
320
+ self,
321
+ builders: Sequence[Tuple[str, str, SurrogateBuilder]],
322
+ *,
323
+ pool_factor: int = 4,
324
+ exploration_floor: float = 0.25,
325
+ unscreened_objective_floor: float = 1.0,
326
+ validation_splits: int = 3,
327
+ revise: Any = None,
328
+ max_revisions: int = 2,
329
+ max_training_rows: int = 1024,
330
+ gate: GatePolicy = ORDERING_GATE,
331
+ ) -> None:
332
+ if pool_factor < 2:
333
+ raise ValueError(f"pool_factor must be at least 2, got {pool_factor}")
334
+ if max_training_rows < 1:
335
+ raise ValueError(
336
+ f"max_training_rows must be at least 1, got {max_training_rows}"
337
+ )
338
+ if not 0.0 <= exploration_floor < 1.0:
339
+ raise ValueError(
340
+ f"exploration_floor must be in [0, 1), got {exploration_floor}"
341
+ )
342
+ if not 0.0 <= unscreened_objective_floor <= 1.0:
343
+ raise ValueError(
344
+ "unscreened_objective_floor must be in [0, 1], got "
345
+ f"{unscreened_objective_floor}"
346
+ )
347
+ if validation_splits < 1:
348
+ raise ValueError(
349
+ f"validation_splits must be at least 1, got {validation_splits}"
350
+ )
351
+ if not isinstance(gate, GatePolicy):
352
+ raise TypeError(f"gate must be a GatePolicy, got {type(gate).__name__}")
353
+ self.builders = tuple(builders)
354
+ self.pool_factor = int(pool_factor)
355
+ self.exploration_floor = float(exploration_floor)
356
+ #: How much of a generation is reserved from the screen when the gate
357
+ #: certified the surrogate on only SOME objectives, expressed as a
358
+ #: multiple of the share of objectives the screen is blind to.
359
+ #:
360
+ #: The screen orders by domination over the certified subset, and
361
+ #: domination over a subset is a stricter relation than domination
362
+ #: over the whole problem: a candidate that is excellent only on an
363
+ #: excluded objective is dominated on the subset and sinks. So a
364
+ #: partial screen is not merely less informed than a full one, it is
365
+ #: SYSTEMATICALLY biased against the objectives it cannot see, and
366
+ #: the flat 0.25 floor -- sized for a screen that might be wrong,
367
+ #: not for one that is wrong in a known direction -- is not the right
368
+ #: protection. At 1.0 (the default) the reserved share is the
369
+ #: unscreened share of the objectives: 1/3 of the generation stays
370
+ #: unscreened when 2 of 3 objectives are certified, 2/3 when 1 of 3
371
+ #: is, and the ordinary ``exploration_floor`` still applies as a
372
+ #: lower bound. At 0.0 a partial screen is treated exactly like a
373
+ #: full one, which is the arm this default was measured against.
374
+ self.unscreened_objective_floor = float(unscreened_objective_floor)
375
+ self.validation_splits = int(validation_splits)
376
+ #: What this consumer relies on, declared to the gate rather than
377
+ #: assumed by it. The screen consumes an ORDER (`screen_offspring`
378
+ #: reads `report.order` and nothing else), so rank fidelity is the
379
+ #: hard term and the error ratio arbitrates among passers. Overriding
380
+ #: this with a prediction-purpose policy restores the historical
381
+ #: behaviour, at the historical cost: a magnitude test on a small
382
+ #: holdout rejects most of the artifacts that would have ordered
383
+ #: correctly.
384
+ self.gate = gate
385
+ #: The evolving-surrogate hook: called with (evaluated, specs) when
386
+ #: the llm builder exists and did not win this refresh, at most
387
+ #: max_revisions times per run. Returns a replacement
388
+ #: (name, authored_by, builder) entry -- authored from the current
389
+ #: artifact plus its measured validation residuals -- or None. The
390
+ #: revision competes from the NEXT refresh under the same gate; a
391
+ #: model that cannot fix its artifact keeps losing to the rules.
392
+ self.revise = revise
393
+ self.max_revisions = int(max_revisions)
394
+ #: How many of the most recent measurements a refresh fits and
395
+ #: validates on. Refitting every builder on EVERYTHING measured so
396
+ #: far makes one refresh O(n) and a run O(n^2): at B=10,000 the screen
397
+ #: alone runs for over a quarter of an hour and never finishes a run,
398
+ #: which is precisely the regime an authored generator exists for.
399
+ #: The recent window is also the better statistics for a distribution
400
+ #: the search keeps moving. The default is far above any campaign run
401
+ #: to date (all at B <= 150), so every measured run is unaffected.
402
+ self.max_training_rows = int(max_training_rows)
403
+ self.telemetry = ScreeningTelemetry()
404
+ self.mechanism = "surrogate_screen"
405
+ self.authored_by = "none"
406
+ self._predict: Optional[Predict] = None
407
+ self._name = ""
408
+ #: The cheap fidelity, if the problem has one and the loop attached
409
+ #: it. `_proxy_rows` is gate EVIDENCE bought at that fidelity: it is
410
+ #: keyed so a real measurement always supersedes it, it never reaches
411
+ #: the archive, the population or the budget, and it is counted in
412
+ #: the source's own ledger.
413
+ self._proxy: Any = None
414
+ self._proxy_mode = "off"
415
+ self._proxy_rows: Dict[str, Tuple[Config, Mapping[str, float]]] = {}
416
+
417
+ # ------------------------------------------------------------------ proxy
418
+ def attach_proxy(self, source: Any, *, mode: str = "rows") -> None:
419
+ """Let this screen spend a CHEAPER evaluation fidelity.
420
+
421
+ ``mode="rows"`` -- the cheap evaluator buys gate EVIDENCE: rows the
422
+ campaign could not afford at full price, so the gate can reach a
423
+ verdict at budgets where the run has not yet measured its minimum
424
+ number of rows. This is the term that closes "too few rows", which is
425
+ a property of the budget and which no gate policy can fix.
426
+
427
+ ``mode="screen"`` -- the cheap evaluator competes AS a surrogate,
428
+ first in the builder order, and is cross-validated against the run's
429
+ own real measurements exactly like an authored artifact. This is the
430
+ term that can close a rank veto, and only if the cheap fidelity
431
+ really does rank the expensive one.
432
+
433
+ ``mode="both"`` -- both. ``mode="off"`` -- neither, and the screen is
434
+ then byte-identical to a screen with no proxy at all.
435
+ """
436
+
437
+ if mode not in ("off", "rows", "screen", "both"):
438
+ raise ValueError(
439
+ "proxy mode must be 'off', 'rows', 'screen' or 'both', "
440
+ f"got {mode!r}")
441
+ if mode == "off" or source is None:
442
+ self._proxy, self._proxy_mode = None, "off"
443
+ return
444
+ self._proxy = source
445
+ self._proxy_mode = mode
446
+ if mode in ("screen", "both"):
447
+ from agent_evolve.session.fidelity import proxy_fidelity_builder
448
+ name = f"proxy:{getattr(source, 'name', 'proxy')}"
449
+ if not any(entry[0] == name for entry in self.builders):
450
+ self.builders = ((name, "proxy", proxy_fidelity_builder(source)),
451
+ ) + self.builders
452
+
453
+ def prime(self, candidates: Sequence[Config],
454
+ measured_keys: Sequence[str] = ()) -> int:
455
+ """Buy gate evidence at the cheap fidelity. Returns rows added.
456
+
457
+ Called by the loop with the candidates it is about to consider. Rows
458
+ already measured for real are skipped, and any cheap row whose
459
+ candidate later gets measured is dropped by :meth:`refresh` -- cheap
460
+ evidence exists to fill a hole, never to outvote the real thing.
461
+ """
462
+
463
+ if self._proxy is None or self._proxy_mode not in ("rows", "both"):
464
+ return 0
465
+ known = set(measured_keys)
466
+ added = 0
467
+ for config, values in self._proxy.rows(candidates, exclude=known):
468
+ token = self._proxy.key(config)
469
+ if token in self._proxy_rows:
470
+ continue
471
+ self._proxy_rows[token] = (config, values)
472
+ added += 1
473
+ self._proxy.ledger.rows_used = len(self._proxy_rows)
474
+ return added
475
+ #: The objectives the gate certified the installed surrogate for.
476
+ #: ``None`` when nothing is installed. The screen orders on exactly
477
+ #: these and treats the rest as unknown.
478
+ self._objectives: Optional[Tuple[str, ...]] = None
479
+
480
+ def refresh(
481
+ self,
482
+ evaluated: Sequence[Tuple[Config, Mapping[str, float]]],
483
+ specs: Sequence[ObjectiveSpec],
484
+ *,
485
+ seed: int = 0,
486
+ ) -> bool:
487
+ """Re-arbitrate: today's data decides WHO may screen, if anyone.
488
+
489
+ Every builder is validated on ``validation_splits`` INDEPENDENT
490
+ re-partitions of today's data and must pass ``self.gate`` on EVERY
491
+ one; among the survivors, the lowest median mse/baseline ratio wins
492
+ the generation. The all-splits requirement is the variance guard the
493
+ ladder1 E2 row demanded: a high-variance authored artifact can pass
494
+ one partition by luck and then mis-screen mid-run -- measured at the
495
+ cheapest scale, where authored screening HURT the endpoint under the
496
+ single-split gate. Surviving every re-partition of today's data is
497
+ the in-loop generalization of "pass on both frozen datasets
498
+ independently"; under cross-validation each split already scores the
499
+ artifact on every row, so what the splits vary is which rows it was
500
+ FITTED on, which is the instability the guard exists to catch.
501
+ Best-passing across splits stays the arbitration: listing the
502
+ authored builder first would be trust, this is measurement. The
503
+ gate's rank-agreement term applies on every split too, but it only
504
+ GATES -- the ratio arbitrating among passers stays pure mse/baseline
505
+ (rank-unfaithful passers were the measured failure, not mis-ranking
506
+ among passers), and under the ordering purpose that ratio is the ONLY
507
+ thing the error term does.
508
+
509
+ Only the most recent ``max_training_rows`` measurements take part:
510
+ see that field for why a refresh must not grow with the run.
511
+ """
512
+
513
+ self.telemetry.refreshes += 1
514
+ self._predict = None
515
+ self._objectives = None
516
+ data = list(evaluated)
517
+ # Cheap-fidelity evidence, where the campaign has none of its own.
518
+ # REAL SUPERSEDES CHEAP, always and by key; the cheap rows are
519
+ # appended after the real ones so the recent-window trim below drops
520
+ # them first when the run has measured more than the window holds.
521
+ if len(data) > self.max_training_rows:
522
+ data = data[-self.max_training_rows:]
523
+ proxy_used = 0
524
+ if self._proxy is not None and self._proxy_rows:
525
+ measured = {self._proxy.key(config) for config, _values in data}
526
+ extra = [row for token, row in self._proxy_rows.items()
527
+ if token not in measured]
528
+ room = self.max_training_rows - len(data)
529
+ extra = extra[:room] if room > 0 else []
530
+ proxy_used = len(extra)
531
+ data = data + extra
532
+ self.telemetry.proxy_rows_used = proxy_used
533
+ names = [spec.name for spec in specs]
534
+ required = self.gate.objectives_required(len(names))
535
+ best: Optional[Tuple[Tuple[int, float], str, str,
536
+ SurrogateBuilder, Tuple[str, ...]]] = None
537
+ for name, authored_by, builder in self.builders:
538
+ verdicts = []
539
+ failed = False
540
+ for split in range(self.validation_splits):
541
+ verdict = validate_surrogate(
542
+ builder, data, specs, policy=self.gate,
543
+ seed=seed + split * 7919)
544
+ self.telemetry.record(verdict)
545
+ if not verdict.passed:
546
+ failed = True
547
+ break
548
+ verdicts.append(verdict)
549
+ if failed:
550
+ self.telemetry.rejected_validation += 1
551
+ continue
552
+ # The variance guard applies PER OBJECTIVE, because that is the
553
+ # granularity the verdict now has. An artifact certified on
554
+ # {area, latency} by one re-partition and on {area, energy} by
555
+ # the next is stable on {area} alone -- it has not shown it can
556
+ # order latency or energy across fits -- so the certified set is
557
+ # the INTERSECTION and it must still meet the policy's
558
+ # requirement. Under the conjunction every passing split
559
+ # certifies every objective, so the intersection is the whole set
560
+ # and this is a no-op.
561
+ certified = set(names)
562
+ for verdict in verdicts:
563
+ certified &= set(verdict.passing_objectives)
564
+ scope = tuple(n for n in names if n in certified)
565
+ if len(scope) < required or not scope:
566
+ self.telemetry.rejected_unstable_subset += 1
567
+ continue
568
+ ratios = []
569
+ for verdict in verdicts:
570
+ per = [verdict.mse_ratio[n] for n in scope
571
+ if n in verdict.mse_ratio]
572
+ ratios.append(sum(per) / len(per) if per else 1.0)
573
+ ratio = statistics.median(ratios) if ratios else 1.0
574
+ # More certified objectives beats a better error ratio: an
575
+ # artifact that can order the whole problem is a different
576
+ # instrument from one that can order a third of it, and the
577
+ # ratio -- an average over whichever objectives each artifact
578
+ # got certified on -- is not comparable across different scopes.
579
+ # Under the conjunction every survivor has the same scope, so
580
+ # this reduces to the historical "lowest median ratio wins".
581
+ key = (-len(scope), ratio)
582
+ if best is None or key < best[0]:
583
+ best = (key, name, authored_by, builder, scope)
584
+ # Revise only when the rules measurably beat the artifact -- a refresh
585
+ # where nothing passes the gate carries no feedback a revision could
586
+ # use, and the revision budget is small.
587
+ llm_won = best is not None and best[2] == "llm"
588
+ if (best is not None and not llm_won and self.revise is not None
589
+ and self.telemetry.revisions < self.max_revisions
590
+ and any(authored_by == "llm"
591
+ for _n, authored_by, _b in self.builders)):
592
+ self.telemetry.revisions += 1
593
+ try:
594
+ replacement = self.revise(data, specs)
595
+ except Exception:
596
+ replacement = None
597
+ if replacement is not None:
598
+ self.telemetry.revisions_accepted += 1
599
+ rebuilt = []
600
+ swapped = False
601
+ for entry in self.builders:
602
+ if not swapped and entry[1] == "llm":
603
+ rebuilt.append(tuple(replacement))
604
+ swapped = True
605
+ else:
606
+ rebuilt.append(entry)
607
+ self.builders = tuple(rebuilt)
608
+
609
+ if best is None:
610
+ return False
611
+ _key, name, authored_by, builder, scope = best
612
+ try:
613
+ self._predict = builder(data, specs)
614
+ except Exception:
615
+ self.telemetry.screen_failures += 1
616
+ return False
617
+ self._objectives = scope
618
+ self._name = name
619
+ self.authored_by = authored_by
620
+ self.telemetry.validated += 1
621
+ if authored_by == "llm":
622
+ self.telemetry.chosen_llm += 1
623
+ elif authored_by == "proxy":
624
+ self.telemetry.chosen_proxy += 1
625
+ else:
626
+ self.telemetry.chosen_rule += 1
627
+ return True
628
+
629
+ def screen(
630
+ self,
631
+ pool: Sequence[Config],
632
+ population_objectives: Sequence[Mapping[str, float]],
633
+ specs: Sequence[ObjectiveSpec],
634
+ ) -> Optional[ScreenReport]:
635
+ if self._predict is None:
636
+ return None
637
+ self.telemetry.screens += 1
638
+ try:
639
+ report = screen_offspring(
640
+ pool, population_objectives, specs, self._predict,
641
+ surrogate_name=self._name,
642
+ objectives=self._objectives,
643
+ )
644
+ except Exception:
645
+ self.telemetry.screen_failures += 1
646
+ return None
647
+ if report is None:
648
+ self.telemetry.screen_failures += 1
649
+ return None
650
+ self.telemetry.virtual_evaluations += report.virtual_evaluations
651
+ self.telemetry.record_screen(report)
652
+ return report
653
+
654
+ def exploration_floor_for(self, report: ScreenReport) -> float:
655
+ """The share of the generation to keep away from THIS screen.
656
+
657
+ ``exploration_floor`` when the screen ordered on the whole problem.
658
+ When it ordered on a subset, the floor rises with the share of
659
+ objectives it could not see, scaled by
660
+ ``unscreened_objective_floor`` -- see that field for why a partial
661
+ screen needs more protection than a full one rather than the same.
662
+ The caller applies it; this class does not touch the budget.
663
+ """
664
+
665
+ declared = len(report.declared_objectives)
666
+ screened = len(report.screened_objectives)
667
+ if declared <= 0 or screened >= declared:
668
+ return self.exploration_floor
669
+ unscreened_share = (declared - screened) / declared
670
+ return max(self.exploration_floor,
671
+ self.unscreened_objective_floor * unscreened_share)