agentevolve-optimizer 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (414) hide show
  1. agent_evolve/__init__.py +722 -0
  2. agent_evolve/agentic.py +2800 -0
  3. agent_evolve/api.py +767 -0
  4. agent_evolve/application/__init__.py +1580 -0
  5. agent_evolve/application/action_allocation.py +744 -0
  6. agent_evolve/application/action_allocation_frame.py +347 -0
  7. agent_evolve/application/action_allocation_frame_commit.py +185 -0
  8. agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
  9. agent_evolve/application/action_allocation_frame_v3.py +338 -0
  10. agent_evolve/application/action_archive_value.py +497 -0
  11. agent_evolve/application/action_evidence_consistency.py +455 -0
  12. agent_evolve/application/action_forecast_partitioning.py +1471 -0
  13. agent_evolve/application/action_metric_projection.py +211 -0
  14. agent_evolve/application/action_role_value.py +680 -0
  15. agent_evolve/application/action_score_authorities.py +363 -0
  16. agent_evolve/application/action_structural_signature.py +116 -0
  17. agent_evolve/application/action_target_realization.py +402 -0
  18. agent_evolve/application/agentic_evolution.py +7734 -0
  19. agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
  20. agent_evolve/application/anchor_residual_identification.py +463 -0
  21. agent_evolve/application/archive_conditioned_action_target.py +208 -0
  22. agent_evolve/application/artifact_journal.py +246 -0
  23. agent_evolve/application/artifact_replay.py +347 -0
  24. agent_evolve/application/budgeted_optimizer.py +1828 -0
  25. agent_evolve/application/calibrated_campaign.py +485 -0
  26. agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
  27. agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
  28. agent_evolve/application/campaign_capacity_recourse.py +254 -0
  29. agent_evolve/application/campaign_contextual_outcomes.py +119 -0
  30. agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
  31. agent_evolve/application/campaign_evidence_registry.py +262 -0
  32. agent_evolve/application/campaign_execution.py +2537 -0
  33. agent_evolve/application/campaign_generation_audit.py +942 -0
  34. agent_evolve/application/campaign_learning.py +1812 -0
  35. agent_evolve/application/campaign_learning_runtime.py +1977 -0
  36. agent_evolve/application/campaign_search_phase.py +227 -0
  37. agent_evolve/application/campaign_selector_context_extension.py +220 -0
  38. agent_evolve/application/campaign_variation_envelope.py +649 -0
  39. agent_evolve/application/campaign_variation_trace.py +451 -0
  40. agent_evolve/application/candidate_archive_consequence.py +128 -0
  41. agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
  42. agent_evolve/application/composite_outcome_updater.py +145 -0
  43. agent_evolve/application/composition_portfolio_selection.py +363 -0
  44. agent_evolve/application/concurrent_stage.py +144 -0
  45. agent_evolve/application/contextual_action_allocation.py +181 -0
  46. agent_evolve/application/contextual_campaign_outcomes.py +267 -0
  47. agent_evolve/application/contextual_campaign_planning.py +1366 -0
  48. agent_evolve/application/contextual_delayed_credit.py +651 -0
  49. agent_evolve/application/contextual_search_controller.py +2374 -0
  50. agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
  51. agent_evolve/application/decision_metric_projection.py +112 -0
  52. agent_evolve/application/derived_action_semantics.py +129 -0
  53. agent_evolve/application/detailed_evaluation.py +449 -0
  54. agent_evolve/application/earned_lineage.py +1011 -0
  55. agent_evolve/application/effective_choice_audit.py +484 -0
  56. agent_evolve/application/empirical_consequence_calibration.py +908 -0
  57. agent_evolve/application/evaluation_accounting.py +325 -0
  58. agent_evolve/application/evaluation_cache.py +199 -0
  59. agent_evolve/application/evaluation_escrow.py +547 -0
  60. agent_evolve/application/evaluation_recourse.py +253 -0
  61. agent_evolve/application/event_recorder.py +151 -0
  62. agent_evolve/application/evolution_campaign.py +1840 -0
  63. agent_evolve/application/executable_hypothesis.py +323 -0
  64. agent_evolve/application/factorial_branch_pilot.py +772 -0
  65. agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
  66. agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
  67. agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
  68. agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
  69. agent_evolve/application/finite_action_selection.py +188 -0
  70. agent_evolve/application/finite_action_set.py +306 -0
  71. agent_evolve/application/finite_action_transition.py +537 -0
  72. agent_evolve/application/finite_variation_eligibility.py +296 -0
  73. agent_evolve/application/forecast_geometry_portfolio.py +799 -0
  74. agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
  75. agent_evolve/application/front_proximity_admission.py +311 -0
  76. agent_evolve/application/front_proximity_parent_basis.py +458 -0
  77. agent_evolve/application/frozen_hurdle_score.py +659 -0
  78. agent_evolve/application/g3_causal_screen.py +2257 -0
  79. agent_evolve/application/g3_causal_validation.py +1046 -0
  80. agent_evolve/application/g3_postseal_curation.py +818 -0
  81. agent_evolve/application/gated_agentic_generator.py +205 -0
  82. agent_evolve/application/generation_feedback.py +293 -0
  83. agent_evolve/application/generative_proposal_journal.py +185 -0
  84. agent_evolve/application/geometry_conditional_elasticity.py +453 -0
  85. agent_evolve/application/global_wave_action_allocation.py +1151 -0
  86. agent_evolve/application/head_mass_conditional_seat.py +268 -0
  87. agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
  88. agent_evolve/application/identifiable_reflection_learning.py +395 -0
  89. agent_evolve/application/identifiable_reflection_request.py +364 -0
  90. agent_evolve/application/in_memory_residual_archive.py +341 -0
  91. agent_evolve/application/insight_memory.py +1804 -0
  92. agent_evolve/application/live_runtime_manifest.py +758 -0
  93. agent_evolve/application/llm_task_queue.py +769 -0
  94. agent_evolve/application/matched_finite_action_block.py +409 -0
  95. agent_evolve/application/materialized_action_broker.py +2328 -0
  96. agent_evolve/application/materialized_action_constraints.py +83 -0
  97. agent_evolve/application/materialized_variation.py +211 -0
  98. agent_evolve/application/multi_option_evolution.py +1536 -0
  99. agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
  100. agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
  101. agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
  102. agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
  103. agent_evolve/application/outcome_relation.py +193 -0
  104. agent_evolve/application/paired_allocation_comparison.py +241 -0
  105. agent_evolve/application/paired_block_schedule.py +127 -0
  106. agent_evolve/application/parent_measurement.py +226 -0
  107. agent_evolve/application/pareto_archive.py +811 -0
  108. agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
  109. agent_evolve/application/portfolio_evolution.py +2950 -0
  110. agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
  111. agent_evolve/application/portfolio_memory_attribution.py +581 -0
  112. agent_evolve/application/portfolio_memory_dose.py +788 -0
  113. agent_evolve/application/portfolio_memory_matched_control.py +938 -0
  114. agent_evolve/application/portfolio_memory_transfer.py +297 -0
  115. agent_evolve/application/portfolio_optimization_memory.py +363 -0
  116. agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
  117. agent_evolve/application/portfolio_projection.py +335 -0
  118. agent_evolve/application/portfolio_recombination.py +2032 -0
  119. agent_evolve/application/post_evolution_reflection.py +834 -0
  120. agent_evolve/application/postcommit_rank_authority.py +245 -0
  121. agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
  122. agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
  123. agent_evolve/application/prequential_residual_exploration.py +343 -0
  124. agent_evolve/application/prequential_score_portfolio.py +954 -0
  125. agent_evolve/application/projections.py +292 -0
  126. agent_evolve/application/protected_action_committee.py +1027 -0
  127. agent_evolve/application/protected_branch_pilot.py +376 -0
  128. agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
  129. agent_evolve/application/provider_replay.py +910 -0
  130. agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
  131. agent_evolve/application/recombination_residual_expert.py +403 -0
  132. agent_evolve/application/reflection_workflow.py +571 -0
  133. agent_evolve/application/region_conditional_credit.py +911 -0
  134. agent_evolve/application/residual_campaign_runtime.py +531 -0
  135. agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
  136. agent_evolve/application/residual_headroom_ledger.py +1544 -0
  137. agent_evolve/application/residual_learning_transaction.py +396 -0
  138. agent_evolve/application/residual_portfolio_evolution.py +1228 -0
  139. agent_evolve/application/residual_reachability.py +749 -0
  140. agent_evolve/application/residual_stage_credit.py +499 -0
  141. agent_evolve/application/same_prefix_paired_audit.py +1580 -0
  142. agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
  143. agent_evolve/application/sequential_lineage_allocation.py +1017 -0
  144. agent_evolve/application/sequential_market_replay.py +1395 -0
  145. agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
  146. agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
  147. agent_evolve/application/single_score_action_allocation.py +299 -0
  148. agent_evolve/application/source_exposure_allocation.py +906 -0
  149. agent_evolve/application/staged_memory.py +210 -0
  150. agent_evolve/application/stratified_cold_start_allocation.py +732 -0
  151. agent_evolve/application/support_guarded_hurdle_score.py +549 -0
  152. agent_evolve/application/target_conditioned_action_forecast.py +595 -0
  153. agent_evolve/application/target_conditioned_campaign.py +566 -0
  154. agent_evolve/application/treatment_assignment.py +201 -0
  155. agent_evolve/application/trusted_objective_evidence.py +217 -0
  156. agent_evolve/application/two_stage_action_evolution.py +1131 -0
  157. agent_evolve/application/v8lite_allocation_policy.py +1083 -0
  158. agent_evolve/application/v9_candidate_policy.py +1303 -0
  159. agent_evolve/bootstrap.py +108 -0
  160. agent_evolve/campaign_presets.py +517 -0
  161. agent_evolve/campaign_profiles.py +452 -0
  162. agent_evolve/campaign_variation_topology.py +288 -0
  163. agent_evolve/campaign_workload.py +950 -0
  164. agent_evolve/cli.py +797 -0
  165. agent_evolve/contract.py +241 -0
  166. agent_evolve/core/__init__.py +91 -0
  167. agent_evolve/core/action_semantics.py +411 -0
  168. agent_evolve/core/authored.py +105 -0
  169. agent_evolve/core/formatting.py +286 -0
  170. agent_evolve/core/optimization_semantics.py +324 -0
  171. agent_evolve/core/problem.py +167 -0
  172. agent_evolve/core/results.py +323 -0
  173. agent_evolve/core/stats.py +70 -0
  174. agent_evolve/core/telemetry.py +100 -0
  175. agent_evolve/domain/__init__.py +89 -0
  176. agent_evolve/domain/artifact.py +162 -0
  177. agent_evolve/domain/durable_text.py +68 -0
  178. agent_evolve/domain/event.py +1454 -0
  179. agent_evolve/domain/finite_action_set.py +426 -0
  180. agent_evolve/domain/finite_variation.py +526 -0
  181. agent_evolve/domain/generative_emission.py +559 -0
  182. agent_evolve/domain/ids.py +163 -0
  183. agent_evolve/domain/inline_text.py +106 -0
  184. agent_evolve/domain/insight.py +27 -0
  185. agent_evolve/domain/lineage.py +737 -0
  186. agent_evolve/domain/llm_task_queue.py +960 -0
  187. agent_evolve/domain/outcome.py +96 -0
  188. agent_evolve/domain/patch.py +854 -0
  189. agent_evolve/domain/typed_json.py +542 -0
  190. agent_evolve/domain/variation_space.py +158 -0
  191. agent_evolve/driver.py +1014 -0
  192. agent_evolve/harness/__init__.py +29 -0
  193. agent_evolve/harness/base.py +242 -0
  194. agent_evolve/harness/directives.py +163 -0
  195. agent_evolve/harness/generative_seal.py +479 -0
  196. agent_evolve/harness/registry.py +41 -0
  197. agent_evolve/infrastructure/__init__.py +39 -0
  198. agent_evolve/infrastructure/artifacts/__init__.py +6 -0
  199. agent_evolve/infrastructure/artifacts/_verification.py +67 -0
  200. agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
  201. agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
  202. agent_evolve/infrastructure/asyncio_runtime.py +109 -0
  203. agent_evolve/infrastructure/authored_runtime.py +188 -0
  204. agent_evolve/infrastructure/authored_worker.py +171 -0
  205. agent_evolve/infrastructure/clock.py +53 -0
  206. agent_evolve/infrastructure/events/__init__.py +6 -0
  207. agent_evolve/infrastructure/events/_validation.py +89 -0
  208. agent_evolve/infrastructure/events/in_memory.py +56 -0
  209. agent_evolve/infrastructure/events/jsonl.py +193 -0
  210. agent_evolve/infrastructure/exception_provenance.py +215 -0
  211. agent_evolve/infrastructure/ids.py +118 -0
  212. agent_evolve/infrastructure/lineage_codec.py +1836 -0
  213. agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
  214. agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
  215. agent_evolve/infrastructure/resource_lease.py +370 -0
  216. agent_evolve/infrastructure/sanitization/__init__.py +8 -0
  217. agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
  218. agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
  219. agent_evolve/infrastructure/stream_liveness.py +383 -0
  220. agent_evolve/infrastructure/subprocess_boundary.py +136 -0
  221. agent_evolve/integrations/__init__.py +1 -0
  222. agent_evolve/integrations/botorch/__init__.py +28 -0
  223. agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
  224. agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
  225. agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
  226. agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
  227. agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
  228. agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
  229. agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
  230. agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
  231. agent_evolve/integrations/completion.py +242 -0
  232. agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
  233. agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
  234. agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
  235. agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
  236. agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
  237. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
  238. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
  239. agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
  240. agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
  241. agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
  242. agent_evolve/integrations/pydantic_ai/harness.py +159 -0
  243. agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
  244. agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
  245. agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
  246. agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
  247. agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
  248. agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
  249. agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
  250. agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
  251. agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
  252. agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
  253. agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
  254. agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
  255. agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
  256. agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
  257. agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
  258. agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
  259. agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
  260. agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
  261. agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
  262. agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
  263. agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
  264. agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
  265. agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
  266. agent_evolve/integrations/pymoo_adapter.py +242 -0
  267. agent_evolve/policies/__init__.py +17 -0
  268. agent_evolve/policies/check.py +469 -0
  269. agent_evolve/policies/emit_scaffold.py +451 -0
  270. agent_evolve/policies/feedback/__init__.py +37 -0
  271. agent_evolve/policies/feedback/held_out_asn.py +1325 -0
  272. agent_evolve/policies/genetic.py +607 -0
  273. agent_evolve/policies/llm_backoff.py +183 -0
  274. agent_evolve/policies/llm_chooser.py +226 -0
  275. agent_evolve/policies/llm_generator.py +1760 -0
  276. agent_evolve/policies/llm_init.py +267 -0
  277. agent_evolve/policies/llm_operator.py +109 -0
  278. agent_evolve/policies/llm_prior.py +194 -0
  279. agent_evolve/policies/llm_surrogate.py +334 -0
  280. agent_evolve/policies/measurement_evidence.py +704 -0
  281. agent_evolve/policies/memory/__init__.py +223 -0
  282. agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
  283. agent_evolve/policies/memory/compatibility_matching.py +593 -0
  284. agent_evolve/policies/memory/global_falsification.py +1841 -0
  285. agent_evolve/policies/memory/prompt_shape.py +503 -0
  286. agent_evolve/policies/memory/randomized_subset.py +714 -0
  287. agent_evolve/policies/memory/staged_causal.py +1270 -0
  288. agent_evolve/policies/memory/treatment_compliance.py +759 -0
  289. agent_evolve/policies/objective_resolution/__init__.py +17 -0
  290. agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
  291. agent_evolve/policies/operator_portfolio.py +407 -0
  292. agent_evolve/policies/reguidance.py +1133 -0
  293. agent_evolve/policies/reward/__init__.py +83 -0
  294. agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
  295. agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
  296. agent_evolve/policies/reward/affine_hypervolume.py +490 -0
  297. agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
  298. agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
  299. agent_evolve/policies/reward/frozen_archive.py +360 -0
  300. agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
  301. agent_evolve/policies/search_state.py +208 -0
  302. agent_evolve/policies/selection/__init__.py +345 -0
  303. agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
  304. agent_evolve/policies/selection/affine_frontier_context.py +330 -0
  305. agent_evolve/policies/selection/affine_frontier_target.py +473 -0
  306. agent_evolve/policies/selection/archive_elite.py +1346 -0
  307. agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
  308. agent_evolve/policies/selection/calibrated_slate.py +1394 -0
  309. agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
  310. agent_evolve/policies/selection/common_candidate_pool.py +685 -0
  311. agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
  312. agent_evolve/policies/selection/disjoint_pairs.py +479 -0
  313. agent_evolve/policies/selection/elite_explorer.py +719 -0
  314. agent_evolve/policies/selection/finite_action.py +187 -0
  315. agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
  316. agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
  317. agent_evolve/policies/selection/forecast_calibration.py +922 -0
  318. agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
  319. agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
  320. agent_evolve/policies/selection/full_support_slate.py +91 -0
  321. agent_evolve/policies/selection/meaningful_direction.py +240 -0
  322. agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
  323. agent_evolve/policies/selection/model_anchored_slate.py +826 -0
  324. agent_evolve/policies/selection/phenotype_recourse.py +979 -0
  325. agent_evolve/policies/selection/proposal_support.py +368 -0
  326. agent_evolve/policies/selection/random_portfolio.py +254 -0
  327. agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
  328. agent_evolve/policies/selection/residual_frontier.py +463 -0
  329. agent_evolve/policies/selection/residual_frontier_target.py +605 -0
  330. agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
  331. agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
  332. agent_evolve/policies/selection/target_conditioned_features.py +812 -0
  333. agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
  334. agent_evolve/policies/selection/task_keyed_palette.py +906 -0
  335. agent_evolve/policies/semantics.py +147 -0
  336. agent_evolve/policies/structure.py +362 -0
  337. agent_evolve/policies/structured_output_budget.py +62 -0
  338. agent_evolve/policies/surrogate.py +696 -0
  339. agent_evolve/policies/variation/__init__.py +1 -0
  340. agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
  341. agent_evolve/policies/variation/crossover_inheritance.py +575 -0
  342. agent_evolve/policies/variation/disjoint_recombination.py +611 -0
  343. agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
  344. agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
  345. agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
  346. agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
  347. agent_evolve/policies/variation/typed_patch.py +1981 -0
  348. agent_evolve/policies/weighted_prior.py +394 -0
  349. agent_evolve/ports/__init__.py +383 -0
  350. agent_evolve/ports/action_allocation.py +733 -0
  351. agent_evolve/ports/action_allocation_frame.py +1153 -0
  352. agent_evolve/ports/action_allocation_frame_commit.py +294 -0
  353. agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
  354. agent_evolve/ports/action_allocation_frame_v3.py +995 -0
  355. agent_evolve/ports/action_forecast.py +1568 -0
  356. agent_evolve/ports/action_metric_projection.py +165 -0
  357. agent_evolve/ports/agentic_generator.py +1561 -0
  358. agent_evolve/ports/archive_context.py +136 -0
  359. agent_evolve/ports/artifact_sanitizer.py +44 -0
  360. agent_evolve/ports/artifact_store.py +225 -0
  361. agent_evolve/ports/clock.py +13 -0
  362. agent_evolve/ports/contextual_search_allocation.py +827 -0
  363. agent_evolve/ports/decision_metric_projection.py +258 -0
  364. agent_evolve/ports/event_store.py +55 -0
  365. agent_evolve/ports/executable_hypothesis.py +557 -0
  366. agent_evolve/ports/finite_acquisition.py +377 -0
  367. agent_evolve/ports/finite_acquisition_batch.py +296 -0
  368. agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
  369. agent_evolve/ports/finite_acquisition_json.py +247 -0
  370. agent_evolve/ports/finite_acquisition_space.py +168 -0
  371. agent_evolve/ports/finite_action_selection.py +348 -0
  372. agent_evolve/ports/finite_action_set.py +256 -0
  373. agent_evolve/ports/frontier_target.py +396 -0
  374. agent_evolve/ports/generation_failure.py +43 -0
  375. agent_evolve/ports/hard_feasibility.py +233 -0
  376. agent_evolve/ports/id_factory.py +34 -0
  377. agent_evolve/ports/llm_task_queue.py +93 -0
  378. agent_evolve/ports/objective_resolution.py +419 -0
  379. agent_evolve/ports/paired_allocation_comparison.py +401 -0
  380. agent_evolve/ports/paired_block_schedule.py +475 -0
  381. agent_evolve/ports/parent_measurement.py +336 -0
  382. agent_evolve/ports/portfolio_memory_dose.py +643 -0
  383. agent_evolve/ports/portfolio_selection.py +3169 -0
  384. agent_evolve/ports/postcommit_rank_authority.py +467 -0
  385. agent_evolve/ports/presented_action_evidence.py +794 -0
  386. agent_evolve/ports/resource_lease.py +162 -0
  387. agent_evolve/ports/structured_generator.py +734 -0
  388. agent_evolve/ports/structured_output_budget.py +120 -0
  389. agent_evolve/ports/subprocess_boundary.py +138 -0
  390. agent_evolve/ports/treatment_assignment.py +466 -0
  391. agent_evolve/ports/variation_catalog.py +76 -0
  392. agent_evolve/ports/variation_source.py +226 -0
  393. agent_evolve/proposal_mode.py +157 -0
  394. agent_evolve/proposers/__init__.py +10 -0
  395. agent_evolve/proposers/random_proposer.py +188 -0
  396. agent_evolve/provider_accounting.py +163 -0
  397. agent_evolve/py.typed +0 -0
  398. agent_evolve/reference_method.py +1570 -0
  399. agent_evolve/session/__init__.py +11 -0
  400. agent_evolve/session/authorship.py +864 -0
  401. agent_evolve/session/evaluate.py +236 -0
  402. agent_evolve/session/fidelity.py +237 -0
  403. agent_evolve/session/genetic_loop.py +742 -0
  404. agent_evolve/session/loop.py +803 -0
  405. agent_evolve/session/screening.py +671 -0
  406. agent_evolve/settings.py +376 -0
  407. agent_evolve/workload_kit.py +368 -0
  408. agent_evolve/workload_prompt.py +398 -0
  409. agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
  410. agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
  411. agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
  412. agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
  413. agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
  414. agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1760 @@
1
+ """The model writes the SAMPLER, not the samples.
2
+
3
+ Every guidance mechanism before this one paid per decision: a chooser picks
4
+ parents once per offspring, an immigration call authors a handful of members
5
+ once per segment. Leverage like that scales as 1/budget, which is exactly why
6
+ per-decision guidance washes out on cheap-evaluation venues -- at B=40 one
7
+ call shapes a tenth of the run, at B=40,000 it shapes a ten-thousandth.
8
+
9
+ An authored GENERATOR inverts that economics. The model is asked once, before
10
+ any evaluation, to write ``propose(archive, n, domains, seed)`` -- a
11
+ distribution, not a draw -- and that one fixed cost then shapes every
12
+ candidate the run ever considers. Its quality is directly measurable and was
13
+ already measured before this seam existed (W3's induced-sampler recall
14
+ 0.094/0.211/0.205 against 0.0585 chance).
15
+
16
+ The authoring line holds by the same construction as everywhere else, and it
17
+ is worth stating precisely because a generator emits candidates:
18
+
19
+ - Every emitted candidate is validated VALUE-BY-VALUE against the declared
20
+ domains and the template's shape (``validate_pool``); rejects are counted
21
+ and their slots fall back to schema-uniform draws, so a broken generator
22
+ degrades to the credential-free sampler rather than to nothing.
23
+ - The emitted set is measured for COLLAPSE -- duplicates inside the batch and
24
+ candidates already measured earlier in the run -- because a generator that
25
+ returns the same configuration n times is not a sampler, and the difference
26
+ has to be visible in the telemetry rather than inferred from a flat curve.
27
+ - This module is starved exactly as ``session.screening`` is: it receives a
28
+ template, a candidate model, a restriction and an archive of configurations
29
+ -- never the problem, never the evaluation cache. Mass generation therefore
30
+ cannot spend budget: the only route from a generated candidate to a real
31
+ evaluation is the loop measuring the ``want`` of them it could already
32
+ afford, which is a property of the import graph rather than a convention.
33
+ - It EVOLVES with feedback: the revision hook shows the generator its own
34
+ source plus what the harness measured about its output -- acceptance rate,
35
+ duplicate and archive-overlap rates, and how many of its candidates
36
+ survived selection -- and asks for a rewrite under the identical gate.
37
+
38
+ What the seam does NOT ask the model to do, since Wave D measured what
39
+ happens when it does. On an assignment-structured genome (`upms_j14_m3`,
40
+ fourteen scalar loci with per-locus eligibility) the sealed generator emitted
41
+ 7,104 candidates against 39,993 uniformly-filled pool slots and 23
42
+ acceptances; `upms_j13_m3` reproduced it. The failure decomposes into three
43
+ things the HARNESS already knows and the model was left to re-derive:
44
+
45
+ - the SHAPE. ``policies.emit_scaffold`` ships a ``build(picks)`` helper into
46
+ the sandbox, so authored code names loci and values and the harness
47
+ assembles the configuration -- and assembles a partially-correct emission
48
+ rather than dropping it whole, a per-LOCUS fallback in place of a
49
+ per-CANDIDATE one.
50
+ - the DOMAINS. Every locus's admissible set is echoed into the authoring
51
+ prompt (``render_domain_echo``), because a field-level card cannot say
52
+ what a per-position domain is.
53
+ - the RESOURCE BUDGET. A batch that overruns the sandbox returns nothing, so
54
+ its whole pool falls back to uniform -- invisible in a per-candidate
55
+ counter. The wall/CPU/memory contract is echoed too, and one overrun buys
56
+ a retry at ``n // 4`` rather than the loss of the pool.
57
+
58
+ Each is counted separately, ablatable separately, and off restores the
59
+ sealed behaviour exactly.
60
+
61
+ Two of those channels fire on EMISSION DEFECTS -- a rejected candidate, a
62
+ collapsed batch, a generator whose children never survive. That is a repair
63
+ loop for a broken sampler, and it is not the same thing as guidance: a
64
+ generator that emits perfectly valid candidates from a region the run has
65
+ already measured to be bad is never revised at all, because nothing about it
66
+ is defective. The prior it encodes is STATIC -- written once from the semantic
67
+ card, in a ladder cell from an empty archive -- and where a recallable domain
68
+ prior is strong that is worth a great deal, while where one is not it is worth
69
+ exactly nothing, three times measured.
70
+
71
+ ``reauthor_every`` is the other trigger, and it fires on SEARCH PROGRESS
72
+ rather than on defects: after the run has MEASURED that many more rows, the
73
+ generator is re-authored against the measured
74
+ ``(configuration -> objectives)`` trace -- the current front, what improved
75
+ and what did not, and which parameter the measurements say moves which cost
76
+ (:mod:`agent_evolve.policies.measurement_evidence`). ``locus_prior`` is the
77
+ second consumer of the same evidence: the model weighs the parameters and
78
+ values worth the remaining budget, the harness types the answer as a GRADED
79
+ bias over the declared domains -- per-locus value weights that exclude
80
+ NOTHING -- and a GATE refuses it, rather than trusts it, whenever it is
81
+ undeclared, empty, garbage-weighted or concentrated past the declared
82
+ weight-ratio cap. Because nothing is excluded, excluding a measured front
83
+ member is structurally impossible rather than gated against.
84
+
85
+ The evidence is WHAT THE RUN MEASURED, not what this generator produced. Every
86
+ charged evaluation the loop holds -- the initial population, a structure
87
+ screen, and the generator's own children alike -- arrives through
88
+ ``note_measured``; only the children additionally move the attribution
89
+ counters, through ``record_measured``. Keeping those two apart is what lets
90
+ the channel speak at the first generation instead of the third: while the
91
+ evidence clock was the generator's own children, the locus prior could not be
92
+ authored before a median charge of 40 on a venue whose dominant knob is
93
+ legible by charge 19, which is the whole of the W11 defect. ``reauthor_every``
94
+ therefore governs how often an evidence call RECURS, and ``evidence_min_rows``
95
+ -- default: the fewest rows a determinable effect can be computed from --
96
+ governs when the first one may fire.
97
+
98
+ Both are OFF by default (``reauthor_every=0``), and off is the seam that ran
99
+ every sealed row to date. Both record what evidence the model was shown (its
100
+ digest), what it emitted, and whether the emission was accepted, so no run can
101
+ imply it reasoned over measurements when it did not.
102
+ """
103
+
104
+ from __future__ import annotations
105
+
106
+ import json
107
+ import random
108
+ from dataclasses import dataclass, fields
109
+ from typing import Any, Callable, Dict, List, Mapping, Optional, Sequence, Tuple
110
+
111
+ from agent_evolve.core.authored import CONTRACTS, AuthoredArtifact, authored_artifact
112
+ from agent_evolve.core.problem import ObjectiveSpec
113
+ from agent_evolve.infrastructure.authored_worker import ALLOWED_IMPORTS
114
+ from agent_evolve.policies.emit_scaffold import (
115
+ NOTES_GLOBAL,
116
+ SCAFFOLD_RULES,
117
+ coerce_candidate,
118
+ render_domain_echo,
119
+ scaffold_prelude,
120
+ )
121
+ from agent_evolve.policies.genetic import (
122
+ loci_of,
123
+ locus_domain,
124
+ read_locus,
125
+ uniform_candidate,
126
+ )
127
+ from agent_evolve.policies.llm_surrogate import (
128
+ AuthorTelemetry,
129
+ accept_block,
130
+ json_compact,
131
+ )
132
+ from agent_evolve.policies.measurement_evidence import (
133
+ MIN_EVIDENCE_ROWS,
134
+ WEIGHTED_RESTRICTION_PROMPT,
135
+ MeasuredRow,
136
+ admit_weighted_restriction,
137
+ apply_weighted_restriction,
138
+ evidence_digest,
139
+ parse_weighted_restriction,
140
+ render_measurement_evidence,
141
+ )
142
+
143
+ __all__ = [
144
+ "GeneratorTelemetry",
145
+ "PoolReport",
146
+ "RejectionCensus",
147
+ "AuthoredGenerator",
148
+ "author_generator",
149
+ "revise_generator",
150
+ "reauthor_generator",
151
+ "render_generation_feedback",
152
+ "validate_pool",
153
+ "candidate_key",
154
+ "GENERATOR_PROMPT",
155
+ "GENERATOR_REVISION_PROMPT",
156
+ "GENERATOR_EVIDENCE_PROMPT",
157
+ ]
158
+
159
+ Config = Dict[str, Any]
160
+
161
+ #: How many archive members a prompt-side call carries into the sandbox. The
162
+ #: contract says "the archive"; shipping all of it would make a 10,000-
163
+ #: evaluation run pay a growing JSON round trip every generation for
164
+ #: information the sampler cannot use anyway.
165
+ ARCHIVE_SHOWN = 32
166
+
167
+ #: A hard ceiling on one batch, so "a configured pool size up to very large"
168
+ #: cannot become an accidental out-of-memory. Mass generation is the point;
169
+ #: an unbounded request is not.
170
+ MAX_POOL = 200_000
171
+
172
+
173
+ def candidate_key(config: Mapping[str, Any]) -> str:
174
+ """Identity of a configuration, matching the loop's own dedup key."""
175
+
176
+ return json.dumps(config, sort_keys=True, default=str)
177
+
178
+
179
+ @dataclass
180
+ class GeneratorTelemetry:
181
+ """What mass generation actually did. Counted, never inferred.
182
+
183
+ Rates are deliberately absent: :func:`~agent_evolve.core.telemetry.
184
+ harvest_telemetry` coerces counters to ``int``, so every rate this
185
+ mechanism claims is published as its exact numerator and denominator
186
+ (``duplicates`` over ``emitted``) and computed by whoever reads them.
187
+ :class:`PoolReport` carries the same ratios as floats for callers in
188
+ process.
189
+ """
190
+
191
+ batches: int = 0
192
+ runtime_failures: int = 0
193
+ emitted: int = 0
194
+ accepted: int = 0
195
+ rejected_shape: int = 0
196
+ rejected_out_of_domain: int = 0
197
+ duplicates: int = 0
198
+ archive_overlap: int = 0
199
+ filled_uniform: int = 0
200
+ measured: int = 0
201
+ survived: int = 0
202
+ revisions: int = 0
203
+ revisions_accepted: int = 0
204
+ #: Candidates the harness ASSEMBLED rather than rejected: the emitted
205
+ #: member addressed at least one locus with an admissible value, and the
206
+ #: rest of the configuration was filled from the template and the
207
+ #: domains. Counted apart from ``accepted`` because a repaired candidate
208
+ #: carries less of the model's guidance than a clean one, and a mechanism
209
+ #: that only works after repair must not read as one that works.
210
+ repaired: int = 0
211
+ #: Individual loci the harness had to decide inside those candidates.
212
+ repaired_loci: int = 0
213
+ #: What the in-sandbox emit scaffold reported about its own work: loci
214
+ #: the authored code left unset, and values it asked for that were not in
215
+ #: that locus's domain. Both are counted at the point of construction, so
216
+ #: they are visible even when nothing is rejected at all.
217
+ scaffold_filled: int = 0
218
+ scaffold_out_of_domain: int = 0
219
+ #: Locus names the authored code used that the schema does not declare.
220
+ scaffold_unknown_locus: int = 0
221
+ #: Revisions that were authored, ran, and did NOT improve the measured
222
+ #: defect -- the population the rejected-edit memory is built from.
223
+ revisions_rejected: int = 0
224
+ #: Batches that blew the sandbox's wall/CPU/memory budget and were retried
225
+ #: at a smaller ``n``, and how many of those retries came back usable.
226
+ runtime_retries: int = 0
227
+ runtime_recovered: int = 0
228
+ #: The MEASUREMENT-CONDITIONED channel, counted apart from the
229
+ #: defect-triggered one above, because they are different mechanisms and a
230
+ #: campaign that cannot tell them apart cannot attribute anything.
231
+ #: ``reauthorings`` fired; ``reauthorings_accepted`` came back as a usable
232
+ #: artifact; ``evidence_rows_shown`` is how many measured rows the model
233
+ #: was actually given across those calls.
234
+ reauthorings: int = 0
235
+ reauthorings_accepted: int = 0
236
+ evidence_rows_shown: int = 0
237
+ #: The locus-importance channel. ``priors_refused`` is the number the GATE
238
+ #: threw out and is the counter that says the gate is doing its job;
239
+ #: ``priors_unwound`` counts admitted priors later dropped for not paying.
240
+ priors_proposed: int = 0
241
+ priors_admitted: int = 0
242
+ priors_refused: int = 0
243
+ priors_unwound: int = 0
244
+
245
+ def as_dict(self) -> dict[str, int]:
246
+ return {f.name: int(getattr(self, f.name)) for f in fields(self)}
247
+
248
+
249
+ @dataclass
250
+ class RejectionCensus:
251
+ """WHICH loci rejected, WHY, and one concrete offending sample of each.
252
+
253
+ Wave D's counters said 6,021 shape and 1,060 out-of-domain and could say
254
+ nothing more, so the revision prompt could only tell the model that
255
+ something was wrong -- which is why revision fired on 77 of 80 cells and
256
+ repaired none of them. A revision is a repair instruction, and a repair
257
+ instruction needs the address of the fault: the locus, the reason, and a
258
+ value the model can recognise as its own.
259
+
260
+ The routing is the point (program section 9-B5, the SHE borrowing):
261
+ a defect is diagnosed against the artifact responsible for it, not
262
+ aggregated into a rate that names nobody.
263
+ """
264
+
265
+ shape_reasons: Dict[str, int] = None # type: ignore[assignment]
266
+ out_of_domain_by_locus: Dict[str, int] = None # type: ignore[assignment]
267
+ repaired_by_locus: Dict[str, int] = None # type: ignore[assignment]
268
+ samples: Dict[str, Any] = None # type: ignore[assignment]
269
+
270
+ def __post_init__(self) -> None:
271
+ for name in ("shape_reasons", "out_of_domain_by_locus",
272
+ "repaired_by_locus", "samples"):
273
+ if getattr(self, name) is None:
274
+ setattr(self, name, {})
275
+
276
+ def shape(self, reason: str, sample: Any = None) -> None:
277
+ self.shape_reasons[reason] = self.shape_reasons.get(reason, 0) + 1
278
+ self.samples.setdefault(f"shape:{reason}", _sample_of(sample))
279
+
280
+ def out_of_domain(self, locus: str, value: Any) -> None:
281
+ self.out_of_domain_by_locus[locus] = (
282
+ self.out_of_domain_by_locus.get(locus, 0) + 1)
283
+ self.sample(locus, value)
284
+
285
+ def sample(self, locus: str, value: Any) -> None:
286
+ """One concrete offending value at *locus*; the first one sticks."""
287
+
288
+ self.samples.setdefault(f"domain:{locus}", _sample_of(value))
289
+
290
+ def repaired(self, locus: str) -> None:
291
+ self.repaired_by_locus[locus] = self.repaired_by_locus.get(locus, 0) + 1
292
+
293
+ def merge(self, other: "RejectionCensus") -> None:
294
+ for reason, count in other.shape_reasons.items():
295
+ self.shape_reasons[reason] = self.shape_reasons.get(reason, 0) + count
296
+ for locus, count in other.out_of_domain_by_locus.items():
297
+ self.out_of_domain_by_locus[locus] = (
298
+ self.out_of_domain_by_locus.get(locus, 0) + count)
299
+ for locus, count in other.repaired_by_locus.items():
300
+ self.repaired_by_locus[locus] = (
301
+ self.repaired_by_locus.get(locus, 0) + count)
302
+ for key, value in other.samples.items():
303
+ self.samples.setdefault(key, value)
304
+
305
+ @property
306
+ def empty(self) -> bool:
307
+ return not (self.shape_reasons or self.out_of_domain_by_locus
308
+ or self.repaired_by_locus)
309
+
310
+ def signature(self) -> str:
311
+ """A stable name for THIS defect, so a repeat is recognisable.
312
+
313
+ Two revisions that leave the same loci failing for the same reasons
314
+ have not changed anything the harness can measure, whatever else they
315
+ changed -- and that is exactly what the rejected-edit memory must be
316
+ able to say back to the next revision.
317
+ """
318
+
319
+ parts = ([f"shape:{k}" for k in sorted(self.shape_reasons)]
320
+ + [f"domain:{k}" for k in sorted(self.out_of_domain_by_locus)])
321
+ return "|".join(parts) or "clean"
322
+
323
+
324
+ def _sample_of(value: Any) -> Any:
325
+ """A JSON-safe, bounded rendering of one offending value."""
326
+
327
+ try:
328
+ json.dumps(value)
329
+ except (TypeError, ValueError):
330
+ return repr(value)[:120]
331
+ if isinstance(value, str) and len(value) > 120:
332
+ return value[:120]
333
+ return value
334
+
335
+
336
+ @dataclass(frozen=True)
337
+ class PoolReport:
338
+ """One mass-generation batch, as the harness received it."""
339
+
340
+ accepted: Tuple[Config, ...] = ()
341
+ emitted: int = 0
342
+ rejected_shape: int = 0
343
+ rejected_out_of_domain: int = 0
344
+ duplicates: int = 0
345
+ archive_overlap: int = 0
346
+ #: Accepted by ASSEMBLY rather than as emitted (see ``repair`` below).
347
+ repaired: int = 0
348
+ repaired_loci: int = 0
349
+ census: RejectionCensus = None # type: ignore[assignment]
350
+
351
+ def __post_init__(self) -> None:
352
+ if self.census is None:
353
+ object.__setattr__(self, "census", RejectionCensus())
354
+
355
+ def _rate(self, count: int) -> float:
356
+ return (count / self.emitted) if self.emitted else 0.0
357
+
358
+ @property
359
+ def acceptance_rate(self) -> float:
360
+ """Fraction of the emitted batch the VALIDATION let through.
361
+
362
+ Distinct from :attr:`novelty_rate`: a generator can be perfectly
363
+ in-domain and still emit one configuration a thousand times, and the
364
+ two guards have to be readable apart.
365
+ """
366
+
367
+ return self._rate(self.emitted - self.rejected_shape
368
+ - self.rejected_out_of_domain)
369
+
370
+ @property
371
+ def duplicate_rate(self) -> float:
372
+ """Fraction of the emitted batch that repeated an earlier member."""
373
+
374
+ return self._rate(self.duplicates)
375
+
376
+ @property
377
+ def archive_overlap_rate(self) -> float:
378
+ """Fraction of the emitted batch already measured in this run."""
379
+
380
+ return self._rate(self.archive_overlap)
381
+
382
+ @property
383
+ def novelty_rate(self) -> float:
384
+ """Fraction of the emitted batch that was both valid and new."""
385
+
386
+ return self._rate(len(self.accepted))
387
+
388
+ @property
389
+ def defect_rate(self) -> float:
390
+ """Fraction of the batch the harness had to reject OR assemble.
391
+
392
+ The one number a revision must move. Repairs are counted as defects
393
+ here even though they were accepted: a candidate the harness had to
394
+ finish is a candidate the model did not write, and a revision that
395
+ turns rejections into repairs has moved the failure rather than
396
+ fixed it.
397
+ """
398
+
399
+ return self._rate(self.rejected_shape + self.rejected_out_of_domain
400
+ + self.repaired)
401
+
402
+ def as_note(self) -> Dict[str, Any]:
403
+ """The per-generation history record: counts plus the guard's rates."""
404
+
405
+ return {
406
+ "emitted": self.emitted,
407
+ "accepted": len(self.accepted),
408
+ "rejected_shape": self.rejected_shape,
409
+ "rejected_out_of_domain": self.rejected_out_of_domain,
410
+ "duplicates": self.duplicates,
411
+ "archive_overlap": self.archive_overlap,
412
+ "repaired": self.repaired,
413
+ "repaired_loci": self.repaired_loci,
414
+ "acceptance_rate": round(self.acceptance_rate, 4),
415
+ "duplicate_rate": round(self.duplicate_rate, 4),
416
+ "archive_overlap_rate": round(self.archive_overlap_rate, 4),
417
+ "novelty_rate": round(self.novelty_rate, 4),
418
+ "defect_rate": round(self.defect_rate, 4),
419
+ }
420
+
421
+
422
+ def validate_pool(
423
+ emitted: Any,
424
+ *,
425
+ template: Config,
426
+ domains: Mapping[str, Sequence[Any]],
427
+ seen: Optional[Any] = None,
428
+ limit: Optional[int] = None,
429
+ repair: bool = False,
430
+ rng: Optional[random.Random] = None,
431
+ ) -> PoolReport:
432
+ """Accept the candidates a generator may actually emit into the run.
433
+
434
+ Every candidate is checked value by value: the shape must equal the
435
+ template's, a locus with a declared domain must hold a declared value,
436
+ and a locus the schema does not constrain must keep the template's value
437
+ -- the same rule ``llm_init`` applies to authored initial members, which
438
+ is the point: one authoring line, not one per seam.
439
+
440
+ The diversity guard rides along, because a batch is a SET and its
441
+ degeneracies are only visible batch-wide: a candidate repeating an
442
+ earlier member of the same batch counts as a duplicate, and one whose
443
+ key is in *seen* (everything the run has measured) counts as archive
444
+ overlap. Both are dropped -- they consume a pool slot and can teach the
445
+ run nothing -- and both are counted, so a generator that has collapsed
446
+ onto one configuration reads as ``duplicates == emitted - 1`` instead of
447
+ as an unremarkable flat curve.
448
+
449
+ *limit* caps how many members are considered at all, so a generator that
450
+ answers "give me 2,000" with 200,000 costs the harness the 2,000 it
451
+ asked for. ``PoolReport.emitted`` counts what was considered, which is
452
+ the denominator every rate here is against.
453
+
454
+ Every reject is also ADDRESSED, into :class:`RejectionCensus`: which
455
+ locus, which reason, and one concrete offending sample. A counter that
456
+ says "6,021 shape" cannot instruct a revision; "job_13 is missing from
457
+ every candidate you emitted, sample {...}" can.
458
+
459
+ *repair* turns the per-candidate fallback into a per-LOCUS one. Off (the
460
+ default, and what every sealed row was measured under) a candidate with
461
+ one bad locus is dropped whole and its pool slot is filled by a
462
+ schema-uniform draw, so thirteen good choices are discarded with the
463
+ fourteenth. On, the harness assembles the candidate out of whatever the
464
+ member did address -- flat locus keys, the template's own nesting, or a
465
+ bare sequence aligned with the loci -- and decides only the loci the
466
+ member got wrong or left out. Repairs are accepted, but counted apart in
467
+ ``repaired``/``repaired_loci`` and censused per locus, because a
468
+ candidate the harness finished is not a candidate the model wrote.
469
+ """
470
+
471
+ if not isinstance(emitted, list):
472
+ return PoolReport()
473
+ template_loci = loci_of(template)
474
+ template_fields = set(template)
475
+ known = set(seen) if seen is not None else set()
476
+ batch: set[str] = set()
477
+ draw = rng if rng is not None else random.Random(0)
478
+
479
+ accepted: List[Config] = []
480
+ census = RejectionCensus()
481
+ counts = {"shape": 0, "domain": 0, "duplicate": 0, "overlap": 0,
482
+ "repaired": 0, "repaired_loci": 0}
483
+ considered = 0
484
+ for member in emitted:
485
+ if limit is not None and considered >= limit:
486
+ break
487
+ considered += 1
488
+ candidate, reason, locus, value = _read_candidate(
489
+ member, template=template, template_loci=template_loci,
490
+ template_fields=template_fields, domains=domains)
491
+ if candidate is None and repair:
492
+ fixed, repairs = coerce_candidate(
493
+ member, template=template, domains=domains, rng=draw,
494
+ loci=template_loci)
495
+ if fixed is not None:
496
+ for kind, loci_list in repairs.items():
497
+ for name in loci_list:
498
+ census.repaired(name)
499
+ if kind == "out_of_domain":
500
+ census.out_of_domain(name, _picked(member, name))
501
+ counts["repaired"] += 1
502
+ counts["repaired_loci"] += sum(len(v) for v in repairs.values())
503
+ candidate, reason = fixed, ""
504
+ if candidate is None:
505
+ if reason == "domain":
506
+ counts["domain"] += 1
507
+ census.out_of_domain(str(locus), value)
508
+ else:
509
+ counts["shape"] += 1
510
+ census.shape(reason, member)
511
+ continue
512
+ key = candidate_key(candidate)
513
+ if key in batch:
514
+ counts["duplicate"] += 1
515
+ continue
516
+ # Marked as seen in this batch whatever happens next, so a candidate
517
+ # that is BOTH already measured and repeated eight times reads as one
518
+ # overlap and seven duplicates. Two different defects, two counters.
519
+ batch.add(key)
520
+ if key in known:
521
+ counts["overlap"] += 1
522
+ continue
523
+ accepted.append(dict(candidate))
524
+
525
+ return PoolReport(
526
+ accepted=tuple(accepted),
527
+ emitted=considered,
528
+ rejected_shape=counts["shape"],
529
+ rejected_out_of_domain=counts["domain"],
530
+ duplicates=counts["duplicate"],
531
+ archive_overlap=counts["overlap"],
532
+ repaired=counts["repaired"],
533
+ repaired_loci=counts["repaired_loci"],
534
+ census=census,
535
+ )
536
+
537
+
538
+ def _read_candidate(member, *, template, template_loci, template_fields,
539
+ domains):
540
+ """``(config, reason, locus, value)`` -- the value-by-value gate itself.
541
+
542
+ ``config`` is the member unchanged when it passes. Otherwise ``reason``
543
+ names WHICH gate it failed, in the vocabulary a revision can act on:
544
+ ``not a mapping``, ``missing loci``, ``unexpected fields``, ``wrong
545
+ length``, or ``domain`` with the offending locus and value.
546
+ """
547
+
548
+ if not isinstance(member, dict):
549
+ return None, "not a mapping", None, None
550
+ fields_seen = set(member)
551
+ if fields_seen != template_fields:
552
+ missing = sorted(template_fields - fields_seen)
553
+ extra = sorted(fields_seen - template_fields)
554
+ if missing and extra:
555
+ reason = (f"missing fields {missing[:4]} and unexpected fields "
556
+ f"{extra[:4]}")
557
+ elif missing:
558
+ reason = f"missing fields {missing[:6]}"
559
+ else:
560
+ reason = f"unexpected fields {extra[:6]}"
561
+ return None, reason, None, None
562
+ try:
563
+ member_loci = loci_of(member)
564
+ except Exception:
565
+ return None, "not a mapping", None, None
566
+ if member_loci != template_loci:
567
+ want, got = len(template_loci), len(member_loci)
568
+ return None, (f"wrong sequence length ({got} loci, the archive's "
569
+ f"members have {want})"), None, None
570
+ for locus in member_loci:
571
+ value = read_locus(member, locus)
572
+ domain = domains.get(str(locus)) or ()
573
+ if domain:
574
+ if value not in domain:
575
+ return None, "domain", locus, value
576
+ elif value != read_locus(template, locus):
577
+ return None, "domain", locus, value
578
+ return member, "", None, None
579
+
580
+
581
+ def _picked(member: Any, locus: str) -> Any:
582
+ """The value *member* carried at *locus*, for the census's sample."""
583
+
584
+ try:
585
+ if isinstance(member, dict):
586
+ if locus in member:
587
+ return member[locus]
588
+ if locus.endswith("]") and "[" in locus:
589
+ field, index = locus[:-1].split("[", 1)
590
+ return member[field][int(index)]
591
+ except Exception:
592
+ return None
593
+ return None
594
+
595
+
596
+ GENERATOR_PROMPT = """You are writing the CANDIDATE GENERATOR for a black-box \
597
+ multi-objective optimizer. It calls your function every generation to draw the \
598
+ pool of configurations it will consider; you are writing the DISTRIBUTION those \
599
+ draws come from, not any particular draw.
600
+
601
+ OBJECTIVES (name and direction):
602
+ {goals}
603
+
604
+ SEARCH SPACE:
605
+ {schema}
606
+ {loci}
607
+ Write ONE Python function with EXACTLY this signature:
608
+
609
+ {contract}
610
+
611
+ Rules:
612
+ - `domains` maps every locus to its allowed values under the current sampling
613
+ prior; sequence fields appear per position as `name[i]`. Each configuration
614
+ you return must have the SAME SHAPE as the archive members and take every
615
+ value from `domains` at that locus -- anything else is validated out and
616
+ its slot falls back to a uniform random draw.
617
+ {scaffold}- `archive` holds configurations already measured in this run (it may be
618
+ short early on). Use it for context; do NOT return copies of it, and do not
619
+ return the same configuration twice. A batch that collapses is measured and
620
+ reported back to you.
621
+ - Return EXACTLY `n` configurations. `n` can be large (thousands): this is
622
+ mass generation, so keep it cheap and vectorless -- plain loops over
623
+ `domains`.
624
+ - Use what the parameter NAMES AND MEANINGS say about this domain to bias
625
+ where mass lands: known good regions, couplings that must co-move,
626
+ trade-offs worth spreading along. That knowledge is the only reason your
627
+ sampler can beat drawing uniformly from the same domains, which is exactly
628
+ what it is measured against.
629
+ - Derive all randomness from `seed` (e.g. `random.Random(seed)`), so a pool
630
+ is reproducible.
631
+ - Standard library only; imports limited to: {imports}.
632
+ - No I/O, no globals, deterministic.
633
+ {limits}
634
+ Reply with ONLY one fenced Python code block and no other text."""
635
+
636
+
637
+ #: The resource contract, echoed for the same reason the domains are: a
638
+ #: function that exceeds its sandbox budget returns NOTHING, so the whole pool
639
+ #: falls back to schema-uniform draws and the mechanism contributes zero. Wave
640
+ #: D's `upms_j14_m3` telemetry is the evidence -- 7,104 candidates emitted
641
+ #: against 39,993 pool slots filled uniformly means most BATCHES emitted
642
+ #: nothing at all, which is what a timeout looks like when the counter is
643
+ #: per-candidate. An unstated budget is a budget the author cannot honour.
644
+ LIMITS_RULES = """\
645
+ - HARD RESOURCE LIMITS, enforced by the sandbox: {wall} s wall-clock, {cpu} s
646
+ CPU and {memory} MB of memory for ONE call, at up to n={max_n}. Exceeding
647
+ any of them returns NOTHING -- not a partial pool, nothing -- and the run
648
+ falls back to drawing every candidate uniformly, which is exactly the
649
+ baseline you are being measured against. Budget for the WORST case, not the
650
+ typical one: prefer O(n) construction from `domains` to any search, sort or
651
+ simulation over candidates, and if you want a local improvement step, cap
652
+ its total work by a constant you choose rather than by convergence."""
653
+
654
+
655
+ GENERATOR_REVISION_PROMPT = """You previously wrote this candidate generator \
656
+ for a black-box multi-objective optimizer:
657
+
658
+ ```python
659
+ {source}
660
+ ```
661
+
662
+ The harness ran it, validated everything it emitted, and measured what
663
+ survived. Here is what actually happened:
664
+
665
+ {feedback}
666
+ {loci}
667
+ Revise the function. Read the numbers literally: a rejected or repaired
668
+ candidate names the LOCUS it failed at and the value it tried, so fix that
669
+ locus rather than the sampler in general; duplicates mean the sampler is
670
+ collapsing; archive overlap means it keeps re-proposing configurations
671
+ already measured; no survivors means the region it concentrates on is not
672
+ competitive and the mass should move. Same rules as before: exactly this
673
+ signature
674
+
675
+ {contract}
676
+
677
+ exactly `n` configurations, every value from `domains` at that locus, the same
678
+ shape as the archive members, randomness derived from `seed`, standard library
679
+ only ({imports}), deterministic, no I/O.
680
+ {scaffold}{limits}
681
+ Reply with ONLY one fenced Python code block and no other text."""
682
+
683
+
684
+ GENERATOR_EVIDENCE_PROMPT = """You wrote this candidate generator for a \
685
+ black-box multi-objective optimizer:
686
+
687
+ ```python
688
+ {source}
689
+ ```
690
+
691
+ The optimizer has been running it and MEASURING what it drew. Here is the
692
+ evidence -- the run's own measurements, and nothing else:
693
+
694
+ {evidence}
695
+
696
+ Now reason about where to sample next, and rewrite the function so its mass
697
+ lands there. Concretely: which parameters do the measurements say actually
698
+ move the costs, and in which direction? Which regions has the run already
699
+ measured and found not competitive, so that re-proposing them wastes the rest
700
+ of the budget? Where is the front, and what is the smallest change to a front
701
+ member that has not been measured yet?
702
+
703
+ Your previous version was written before any of this was measured. It is not
704
+ being corrected for a defect -- it is being asked to use information that did
705
+ not exist when it was written. If the measurements do not support a change,
706
+ say so by returning a function that differs only where they do.
707
+
708
+ Same contract as before: exactly this signature
709
+
710
+ {contract}
711
+
712
+ exactly `n` configurations, every value taken from `domains` at that locus,
713
+ the same shape as the archive members, randomness derived from `seed`,
714
+ standard library only ({imports}), deterministic, no I/O.
715
+ {scaffold}{limits}
716
+ Reply with ONLY one fenced Python code block and no other text."""
717
+
718
+
719
+ def _locus_block(domains: Optional[Mapping[str, Sequence[Any]]]) -> str:
720
+ """The per-locus domain echo, as a prompt section (empty when unknown)."""
721
+
722
+ if not domains:
723
+ return ""
724
+ echo = render_domain_echo(domains)
725
+ if not echo:
726
+ return ""
727
+ return ("\nLOCI AND THEIR ADMISSIBLE VALUES (the exact `domains` mapping "
728
+ "you will be passed;\nthese key names ARE the shape -- a "
729
+ "configuration has exactly these loci and no others):\n"
730
+ f"{echo}\n")
731
+
732
+
733
+ def _scaffold_block(scaffold: bool) -> str:
734
+ return (SCAFFOLD_RULES + "\n") if scaffold else ""
735
+
736
+
737
+ def _limits_block(limits: Any, max_n: Optional[int]) -> str:
738
+ """The sandbox's own budget, echoed (empty when the caller knows none)."""
739
+
740
+ if limits is None or not max_n:
741
+ return ""
742
+ try:
743
+ return LIMITS_RULES.format(
744
+ wall=f"{float(limits.wall_time_s):g}",
745
+ cpu=f"{float(limits.cpu_seconds):g}",
746
+ memory=int(int(limits.memory_bytes) / (1024 * 1024)),
747
+ max_n=int(max_n)) + "\n"
748
+ except (AttributeError, TypeError, ValueError):
749
+ return ""
750
+
751
+
752
+ def author_generator(
753
+ complete: Callable[[str], str],
754
+ *,
755
+ objectives: Sequence[ObjectiveSpec],
756
+ schema_text: str,
757
+ attempts: int = 2,
758
+ telemetry: Optional[AuthorTelemetry] = None,
759
+ domains: Optional[Mapping[str, Sequence[Any]]] = None,
760
+ scaffold: bool = True,
761
+ limits: Any = None,
762
+ max_n: Optional[int] = None,
763
+ ) -> Optional[AuthoredArtifact]:
764
+ """Ask the model to write ``propose``; accept whole or not at all.
765
+
766
+ *domains* is the per-locus admissible set the run will actually pass, so
767
+ the prompt can ECHO it rather than leave the model to infer per-position
768
+ domains from a field-level card -- the difference that turns an
769
+ out-of-domain value from a guess into a prompt failure. *scaffold*
770
+ advertises the in-sandbox emit harness (``build``), which is what makes a
771
+ shape error impossible to construct rather than caught after the fact.
772
+ """
773
+
774
+ tel = telemetry if telemetry is not None else AuthorTelemetry()
775
+ contract = CONTRACTS["generator"]
776
+ prompt = GENERATOR_PROMPT.format(
777
+ goals="\n".join(f" {s.name}: {s.goal}imise" for s in objectives),
778
+ schema=schema_text,
779
+ loci=_locus_block(domains),
780
+ scaffold=_scaffold_block(scaffold),
781
+ limits=_limits_block(limits, max_n),
782
+ contract=contract.description,
783
+ imports=", ".join(sorted(ALLOWED_IMPORTS)),
784
+ )
785
+ return _author(complete, prompt, contract=contract, attempts=attempts,
786
+ telemetry=tel, name="llm_generator")
787
+
788
+
789
+ def revise_generator(
790
+ complete: Callable[[str], str],
791
+ *,
792
+ artifact: AuthoredArtifact,
793
+ feedback: str,
794
+ attempts: int = 1,
795
+ telemetry: Optional[AuthorTelemetry] = None,
796
+ domains: Optional[Mapping[str, Sequence[Any]]] = None,
797
+ scaffold: bool = True,
798
+ limits: Any = None,
799
+ max_n: Optional[int] = None,
800
+ ) -> Optional[AuthoredArtifact]:
801
+ """One revision round: the artifact, its measured behaviour, a rewrite.
802
+
803
+ The gate treats a revision exactly like a fresh authorship -- fenced
804
+ block only, import allowlist, correct entry point, whole-reply rejection
805
+ -- so a model that answers a revision with prose, or with code that
806
+ imports the filesystem, keeps the generator it already had.
807
+ """
808
+
809
+ tel = telemetry if telemetry is not None else AuthorTelemetry()
810
+ contract = CONTRACTS["generator"]
811
+ prompt = GENERATOR_REVISION_PROMPT.format(
812
+ source=artifact.source,
813
+ feedback=feedback,
814
+ loci=_locus_block(domains),
815
+ scaffold=_scaffold_block(scaffold),
816
+ limits=_limits_block(limits, max_n),
817
+ contract=contract.description,
818
+ imports=", ".join(sorted(ALLOWED_IMPORTS)),
819
+ )
820
+ return _author(complete, prompt, contract=contract, attempts=attempts,
821
+ telemetry=tel, name=f"{artifact.name}_rev")
822
+
823
+
824
+ def reauthor_generator(
825
+ complete: Callable[[str], str],
826
+ *,
827
+ artifact: AuthoredArtifact,
828
+ evidence: str,
829
+ attempts: int = 1,
830
+ telemetry: Optional[AuthorTelemetry] = None,
831
+ scaffold: bool = True,
832
+ limits: Any = None,
833
+ max_n: Optional[int] = None,
834
+ ) -> Optional[AuthoredArtifact]:
835
+ """Re-author the sampler against the run's MEASURED trace.
836
+
837
+ Structurally identical to :func:`revise_generator` -- same contract, same
838
+ whole-reply gate, same degradation to the artifact already in hand -- and
839
+ different in the one way that matters: the prompt carries measurements
840
+ instead of emission counters, so the model is asked to reason about the
841
+ search rather than to repair its own output. The scaffold and resource
842
+ contracts are echoed exactly as they are for authorship and revision: a
843
+ re-authored sampler runs in the same sandbox as the one it replaces.
844
+ """
845
+
846
+ tel = telemetry if telemetry is not None else AuthorTelemetry()
847
+ contract = CONTRACTS["generator"]
848
+ prompt = GENERATOR_EVIDENCE_PROMPT.format(
849
+ source=artifact.source,
850
+ evidence=evidence,
851
+ scaffold=_scaffold_block(scaffold),
852
+ limits=_limits_block(limits, max_n),
853
+ contract=contract.description,
854
+ imports=", ".join(sorted(ALLOWED_IMPORTS)),
855
+ )
856
+ return _author(complete, prompt, contract=contract, attempts=attempts,
857
+ telemetry=tel, name=f"{artifact.name}_evidence")
858
+
859
+
860
+ def _author(complete, prompt, *, contract, attempts, telemetry, name):
861
+ for _attempt in range(max(1, attempts)):
862
+ telemetry.calls += 1
863
+ try:
864
+ text = complete(prompt)
865
+ except Exception:
866
+ telemetry.errors += 1
867
+ continue
868
+ source = accept_block(text, contract=contract, telemetry=telemetry)
869
+ if source is None:
870
+ continue
871
+ telemetry.accepted += 1
872
+ telemetry.sources.append(source)
873
+ return authored_artifact(contract.kind, source, name=name,
874
+ authored_by="llm")
875
+ return None
876
+
877
+
878
+ #: How many offending loci one feedback block names before it stops. A
879
+ #: revision cannot act on four hundred addresses; it can act on the worst few.
880
+ FEEDBACK_LOCI = 6
881
+
882
+
883
+ def _defect_lines(
884
+ census: Optional[RejectionCensus],
885
+ domains: Optional[Mapping[str, Sequence[Any]]] = None,
886
+ ) -> List[str]:
887
+ """WHICH loci failed and WHY, worst first, each with a real sample."""
888
+
889
+ if census is None or census.empty:
890
+ return []
891
+ lines: List[str] = [" WHERE IT FAILED (the harness's own addresses):"]
892
+ for reason, count in sorted(census.shape_reasons.items(),
893
+ key=lambda kv: -kv[1])[:FEEDBACK_LOCI]:
894
+ sample = census.samples.get(f"shape:{reason}")
895
+ lines.append(f" shape -- {reason}: {count} candidate(s); "
896
+ f"you emitted {json_compact(sample)}")
897
+ ranked = sorted(census.out_of_domain_by_locus.items(),
898
+ key=lambda kv: -kv[1])
899
+ for locus, count in ranked[:FEEDBACK_LOCI]:
900
+ sample = census.samples.get(f"domain:{locus}")
901
+ allowed = list((domains or {}).get(locus) or ())
902
+ rendered = (f"; its domain is {allowed[:8]}"
903
+ + (f" ({len(allowed)} values)" if len(allowed) > 8 else "")
904
+ ) if allowed else ""
905
+ lines.append(f" locus {locus} -- out of domain {count} time(s); "
906
+ f"you used {json_compact(sample)}{rendered}")
907
+ if len(ranked) > FEEDBACK_LOCI:
908
+ lines.append(f" ... and {len(ranked) - FEEDBACK_LOCI} further loci "
909
+ f"out of domain")
910
+ repaired = sorted(census.repaired_by_locus.items(), key=lambda kv: -kv[1])
911
+ if repaired:
912
+ worst = ", ".join(f"{locus} ({count})"
913
+ for locus, count in repaired[:FEEDBACK_LOCI])
914
+ lines.append(f" the harness had to DECIDE these loci for you: "
915
+ f"{worst}")
916
+ return lines
917
+
918
+
919
+ def _edit_lines(rejected_edits: Sequence[Mapping[str, Any]]) -> List[str]:
920
+ """The rejected-edit memory: fixes already tried that did not fix it.
921
+
922
+ Wave D measured revision firing on 77 of 80 cells on the broken
923
+ instances and repairing none of them. A revision loop with no memory of
924
+ its own failures can only re-propose them; naming the edit, the defect it
925
+ was supposed to fix, and the fact that the defect survived it is the
926
+ cheapest thing that makes the next attempt different.
927
+ """
928
+
929
+ if not rejected_edits:
930
+ return []
931
+ lines = [" EDITS ALREADY TRIED THAT DID NOT FIX THIS -- do not repeat "
932
+ "them or anything equivalent:"]
933
+ for edit in rejected_edits[-3:]:
934
+ lines.append(
935
+ f" revision {edit.get('revision')} (source {edit.get('sha')}): "
936
+ f"defect rate {float(edit.get('before', 0.0)):.0%} -> "
937
+ f"{float(edit.get('after', 0.0)):.0%}, and the same loci still "
938
+ f"fail ({edit.get('signature')}).")
939
+ excerpt = str(edit.get("excerpt") or "").strip()
940
+ if excerpt:
941
+ lines.append(" it looked like: "
942
+ + " ".join(excerpt.split())[:240])
943
+ return lines
944
+
945
+
946
+ def render_generation_feedback(
947
+ telemetry: GeneratorTelemetry,
948
+ last: Optional[PoolReport] = None,
949
+ survivors: Sequence[Tuple[Config, Mapping[str, float]]] = (),
950
+ *,
951
+ census: Optional[RejectionCensus] = None,
952
+ domains: Optional[Mapping[str, Sequence[Any]]] = None,
953
+ rejected_edits: Sequence[Mapping[str, Any]] = (),
954
+ ) -> str:
955
+ """The measured story a generator revision needs, as text.
956
+
957
+ Three things, all counted: how much of what it emitted the harness could
958
+ use, how much of it was novel, and what happened to the candidates that
959
+ were measured. Survivors are shown rather than "the best" -- ranking
960
+ candidates across objectives would need weights nobody declared, while
961
+ surviving truncation is the run's own weight-free verdict.
962
+
963
+ Then the two things Wave D's counters could not say, and without which
964
+ revision repaired nothing: WHICH locus rejected and WHY, with a value the
965
+ model will recognise as its own (*census*, *domains*), and which edits
966
+ have already been tried against this same defect and failed
967
+ (*rejected_edits*).
968
+ """
969
+
970
+ lines = [
971
+ f" batches generated: {telemetry.batches}"
972
+ f" (runtime failures: {telemetry.runtime_failures})",
973
+ f" candidates emitted: {telemetry.emitted}; "
974
+ f"accepted by the harness: {telemetry.accepted}",
975
+ f" rejected -- wrong shape: {telemetry.rejected_shape}; "
976
+ f"value outside its declared domain: "
977
+ f"{telemetry.rejected_out_of_domain}",
978
+ f" dropped -- duplicate within the batch: {telemetry.duplicates}; "
979
+ f"already measured in this run: {telemetry.archive_overlap}",
980
+ f" pool slots the harness had to fill with uniform random draws: "
981
+ f"{telemetry.filled_uniform}",
982
+ f" of yours that were measured: {telemetry.measured}; "
983
+ f"survived selection into the next population: {telemetry.survived}",
984
+ ]
985
+ if telemetry.repaired or telemetry.repaired_loci:
986
+ lines.append(
987
+ f" candidates the harness had to ASSEMBLE for you rather than "
988
+ f"reject: {telemetry.repaired} "
989
+ f"({telemetry.repaired_loci} individual loci decided for you)")
990
+ if (telemetry.scaffold_filled or telemetry.scaffold_out_of_domain
991
+ or telemetry.scaffold_unknown_locus):
992
+ lines.append(
993
+ f" inside your own code, `build` filled "
994
+ f"{telemetry.scaffold_filled} locus/loci you left unset, "
995
+ f"overrode {telemetry.scaffold_out_of_domain} out-of-domain "
996
+ f"value(s) and ignored {telemetry.scaffold_unknown_locus} locus "
997
+ f"name(s) the schema does not declare")
998
+ if last is not None and last.emitted:
999
+ lines.append(
1000
+ f" most recent batch: {last.duplicate_rate:.0%} duplicates, "
1001
+ f"{last.archive_overlap_rate:.0%} already measured, "
1002
+ f"{last.novelty_rate:.0%} usable and new")
1003
+ lines.extend(_defect_lines(
1004
+ census if census is not None
1005
+ else (last.census if last is not None else None), domains))
1006
+ lines.extend(_edit_lines(rejected_edits))
1007
+ for config, objectives in survivors:
1008
+ rendered = ", ".join(f"{k}={float(v):.6g}"
1009
+ for k, v in sorted(objectives.items()))
1010
+ lines.append(f" survived: {json_compact(config)} -> {rendered}")
1011
+ return "\n".join(lines)
1012
+
1013
+
1014
+ class AuthoredGenerator:
1015
+ """The authored sampler as a loop policy: mass generation, then the guard.
1016
+
1017
+ One call per generation produces the whole pool out of process; the
1018
+ harness validates it, drops what collapsed, fills any shortfall
1019
+ schema-uniformly, and hands back exactly ``pool_for(want)``
1020
+ configurations. Nothing here can reach an evaluation: the pool is
1021
+ candidates, and the loop measures at most the ``want`` it could already
1022
+ afford.
1023
+ """
1024
+
1025
+ def __init__(
1026
+ self,
1027
+ artifact: AuthoredArtifact,
1028
+ runtime: Any,
1029
+ *,
1030
+ pool_factor: int = 4,
1031
+ pool_size: int = 0,
1032
+ max_pool: int = MAX_POOL,
1033
+ archive_shown: int = ARCHIVE_SHOWN,
1034
+ revise: Optional[Callable[[AuthoredArtifact, str],
1035
+ Optional[AuthoredArtifact]]] = None,
1036
+ max_revisions: int = 1,
1037
+ min_measured_for_revision: int = 4,
1038
+ min_novelty: float = 0.5,
1039
+ scaffold: bool = True,
1040
+ repair: bool = True,
1041
+ revision_guard: bool = False,
1042
+ shrink_on_overrun: int = 4,
1043
+ objectives: Sequence[ObjectiveSpec] = (),
1044
+ reauthor: Optional[Callable[[AuthoredArtifact, str],
1045
+ Optional[AuthoredArtifact]]] = None,
1046
+ reauthor_every: int = 0,
1047
+ max_reauthorings: int = 0,
1048
+ evidence_view: Optional[Callable[[Sequence[MeasuredRow]],
1049
+ Sequence[MeasuredRow]]] = None,
1050
+ evidence_min_rows: int = 0,
1051
+ evidence_front_shown: int = 8,
1052
+ evidence_effects_shown: int = 8,
1053
+ prior_author: Optional[Callable[[str], str]] = None,
1054
+ max_priors: int = 1,
1055
+ prior_max_weight_ratio: float = 8.0,
1056
+ prior_unwind_batches: int = 2,
1057
+ ) -> None:
1058
+ if pool_factor < 1:
1059
+ raise ValueError(f"pool_factor must be at least 1, got {pool_factor}")
1060
+ if pool_size < 0:
1061
+ raise ValueError(f"pool_size must be non-negative, got {pool_size}")
1062
+ if reauthor_every < 0:
1063
+ raise ValueError(
1064
+ f"reauthor_every must be non-negative, got {reauthor_every}")
1065
+ if evidence_min_rows < 0:
1066
+ raise ValueError(
1067
+ "evidence_min_rows is the fewest measured rows the evidence "
1068
+ "channel will author from and must be non-negative, got "
1069
+ f"{evidence_min_rows}")
1070
+ if prior_max_weight_ratio < 1.0:
1071
+ raise ValueError(
1072
+ "prior_max_weight_ratio caps a graded prior's concentration "
1073
+ f"and must be at least 1, got {prior_max_weight_ratio}")
1074
+ if prior_author is not None and reauthor_every <= 0:
1075
+ # The evidence channel has one cadence and both consumers ride it.
1076
+ # A prior asked for on no cadence would fire never or every
1077
+ # generation depending on who read the code, which is exactly the
1078
+ # magic number this knob exists to replace.
1079
+ raise ValueError(
1080
+ "a locus prior is authored from the measured trace on the "
1081
+ "reauthor_every cadence; set reauthor_every > 0")
1082
+ self.artifact = artifact
1083
+ self.runtime = runtime
1084
+ self.pool_factor = int(pool_factor)
1085
+ self.pool_size = int(pool_size)
1086
+ self.max_pool = int(max_pool)
1087
+ self.archive_shown = int(archive_shown)
1088
+ self.revise = revise
1089
+ self.max_revisions = int(max_revisions)
1090
+ self.min_measured_for_revision = int(min_measured_for_revision)
1091
+ self.min_novelty = float(min_novelty)
1092
+ #: Ship the emit harness into the sandbox, so the authored code
1093
+ #: constructs candidates locus by locus instead of transcribing a
1094
+ #: shape. Shape was 6,021 of 7,104 emissions on `upms_j14_m3`.
1095
+ self.scaffold = bool(scaffold)
1096
+ #: Assemble a candidate out of whatever the emission got right rather
1097
+ #: than dropping it whole -- a per-LOCUS fallback in place of a
1098
+ #: per-CANDIDATE one. Off restores the sealed-row semantics exactly.
1099
+ self.repair = bool(repair)
1100
+ #: Keep a revision only if a frozen replay says it MEASURABLY helped
1101
+ #: (see :meth:`_guard_admits`). Off by default: the one-shot and
1102
+ #: capped-revision arms are what every sealed row is defined on.
1103
+ self.revision_guard = bool(revision_guard)
1104
+ #: Divisor for the one retry a resource overrun gets. 0 or 1 disables
1105
+ #: it and a timeout costs the whole pool, as it did when Wave D
1106
+ #: measured 39,993 uniformly-filled slots against 7,104 emissions.
1107
+ self.shrink_on_overrun = int(shrink_on_overrun)
1108
+ #: The MEASUREMENT-CONDITIONED channel. ``reauthor_every`` is a
1109
+ #: cadence in CHARGED EVALUATIONS -- a declared, typed knob rather
1110
+ #: than a magic number -- and ``0`` (the default) is the seam every
1111
+ #: sealed row to date ran: no evidence call ever fires.
1112
+ self.objectives = tuple(objectives)
1113
+ self.reauthor = reauthor
1114
+ self.reauthor_every = int(reauthor_every)
1115
+ self.max_reauthorings = int(max_reauthorings)
1116
+ #: What the model is SHOWN. The identity view is the product; a view
1117
+ #: returning another run's rows is the shuffled-evidence control, and
1118
+ #: it is a parameter rather than a patch precisely so the control can
1119
+ #: be built without editing this file.
1120
+ self.evidence_view = evidence_view
1121
+ #: WHEN the channel may speak for the FIRST time, in measured rows.
1122
+ #: The cadence above says how often an evidence call recurs; it cannot
1123
+ #: also say when the first one is allowed, because a run holds
1124
+ #: measurements before this generator has produced any (the initial
1125
+ #: population is the whole of the evidence at generation 1) and a
1126
+ #: cadence read as "wait for that many of MY OWN children" makes the
1127
+ #: channel arrive generations after the evidence did -- the W11 defect.
1128
+ #: ``0`` (the default) means AS SOON AS THE GATE CAN BE MET:
1129
+ #: :data:`~agent_evolve.policies.measurement_evidence.MIN_EVIDENCE_ROWS`
1130
+ #: rows, the fewest a determinable effect can be computed from. Setting
1131
+ #: it equal to ``reauthor_every`` restores the pure-cadence rule
1132
+ #: exactly, so the older behaviour is a declared configuration rather
1133
+ #: than a lost one.
1134
+ self.evidence_min_rows = (int(evidence_min_rows) if evidence_min_rows
1135
+ else MIN_EVIDENCE_ROWS)
1136
+ self.evidence_front_shown = int(evidence_front_shown)
1137
+ self.evidence_effects_shown = int(evidence_effects_shown)
1138
+ #: The locus-importance channel, and the gate that refuses it.
1139
+ self.prior_author = prior_author
1140
+ self.max_priors = int(max_priors)
1141
+ self.prior_max_weight_ratio = float(prior_max_weight_ratio)
1142
+ self.prior_unwind_batches = int(prior_unwind_batches)
1143
+ self.telemetry = GeneratorTelemetry()
1144
+ self.mechanism = "authored_generator"
1145
+ self.authored_by = artifact.authored_by
1146
+ self.last_report: Optional[PoolReport] = None
1147
+ self.census = RejectionCensus()
1148
+ self._seen: set[str] = set()
1149
+ self._survivors: List[Tuple[Config, Mapping[str, float]]] = []
1150
+ self._domains: Dict[str, List[Any]] = {}
1151
+ self._last_call: Optional[Tuple[List[Config], int, Dict[str, List[Any]],
1152
+ int, Config]] = None
1153
+ self._rejected_edits: List[Dict[str, Any]] = []
1154
+ self._pending_edit: Optional[Dict[str, Any]] = None
1155
+ #: The measured trace, in measurement order. This is the evidence, and
1156
+ #: it is the ONLY thing this class knows about outcomes: it arrives
1157
+ #: through `note_measured` -- which the loop calls for EVERY charged
1158
+ #: evaluation it holds, whoever produced it -- so the generator still
1159
+ #: never sees the problem, the evaluator or the cache.
1160
+ self._rows: List[MeasuredRow] = []
1161
+ self._evidence_at = 0
1162
+ #: How many evidence ticks have fired. The first one is gated on
1163
+ #: evidence (``evidence_min_rows``); every later one on the cadence.
1164
+ self._evidence_ticks = 0
1165
+ self._prior_asked_at = -1
1166
+ self._prior: Any = None
1167
+ self._prior_batches = 0
1168
+ self._survived_at_prior = 0
1169
+ #: One record per evidence-conditioned call: what was shown (digest
1170
+ #: and row count), what came back, and whether it was accepted.
1171
+ #: Telemetry as correctness -- a run cannot claim this channel fired
1172
+ #: without the record that says what it saw.
1173
+ self.evidence_log: List[Dict[str, Any]] = []
1174
+ self._noted = 0
1175
+
1176
+ # -- sizing -------------------------------------------------------------
1177
+
1178
+ def pool_for(self, want: int) -> int:
1179
+ """How many candidates one generation asks for. Never below *want*."""
1180
+
1181
+ size = self.pool_size or self.pool_factor * max(1, want)
1182
+ return max(1, min(self.max_pool, max(int(want), size)))
1183
+
1184
+ # -- the archive it is conditioned on and measured against --------------
1185
+
1186
+ def note_archive(self, configs: Sequence[Config]) -> None:
1187
+ """Record configurations the run has measured. No quality claim."""
1188
+
1189
+ for config in configs:
1190
+ self._seen.add(candidate_key(config))
1191
+
1192
+ def note_measured(
1193
+ self,
1194
+ config: Config,
1195
+ *,
1196
+ objectives: Mapping[str, float],
1197
+ survived: bool = False,
1198
+ ) -> None:
1199
+ """One charged evaluation the RUN holds. Evidence, not attribution.
1200
+
1201
+ Evidence is *what has been measured*, not *what this component
1202
+ produced*. The distinction is the whole of the W11 fix: the initial
1203
+ population, and anything else the loop charged before or beside this
1204
+ generator, is measurement the model can reason over and must be shown
1205
+ -- while the survival counters that decide whether the generator is
1206
+ deficient stay credited to its own children alone (:meth:
1207
+ `record_measured`). Conflating the two either blinds the channel for
1208
+ two generations or forges the attribution; keeping them apart costs
1209
+ one method.
1210
+
1211
+ The row is kept whether or not it survived, because "this region was
1212
+ measured and is NOT competitive" is exactly the evidence a survivor
1213
+ list cannot carry.
1214
+
1215
+ It does NOT touch the novelty ledger. What the run has already tried is
1216
+ ``note_archive``'s declared job, the loop already calls it, and having
1217
+ a second entry point quietly feed the same set would change the
1218
+ duplicate and overlap counters of runs with this channel OFF -- which
1219
+ must stay byte-identical to the sealed seam.
1220
+ """
1221
+
1222
+ self._rows.append((dict(config), dict(objectives), bool(survived)))
1223
+
1224
+ def record_measured(
1225
+ self,
1226
+ config: Config,
1227
+ *,
1228
+ survived: bool,
1229
+ objectives: Optional[Mapping[str, float]] = None,
1230
+ ) -> None:
1231
+ """Credit at survival time, exactly as the operator portfolio does.
1232
+
1233
+ This generator's OWN child: it is both evidence and attribution, so
1234
+ the row is noted and the counters that judge this generator move.
1235
+ """
1236
+
1237
+ self._seen.add(candidate_key(config))
1238
+ self.telemetry.measured += 1
1239
+ if objectives is not None:
1240
+ self.note_measured(config, objectives=objectives,
1241
+ survived=survived)
1242
+ if survived:
1243
+ self.telemetry.survived += 1
1244
+ if objectives is not None:
1245
+ self._survivors.append((dict(config), dict(objectives)))
1246
+ del self._survivors[:-3]
1247
+
1248
+ # -- generation ---------------------------------------------------------
1249
+
1250
+ def propose(
1251
+ self,
1252
+ *,
1253
+ template: Config,
1254
+ candidate_model: Any,
1255
+ restriction: Any,
1256
+ archive: Sequence[Config],
1257
+ want: int,
1258
+ rng: random.Random,
1259
+ seed: int = 0,
1260
+ ) -> List[Config]:
1261
+ """A validated pool of ``pool_for(want)`` configurations.
1262
+
1263
+ The head of the pool is what the loop would measure with no screen at
1264
+ all, so a screen that follows keeps its exploration floor over the
1265
+ generator's OWN first picks rather than over an unrelated draw.
1266
+ """
1267
+
1268
+ n = self.pool_for(want)
1269
+ declared = {
1270
+ str(locus): list(locus_domain(candidate_model, locus,
1271
+ restriction=restriction))
1272
+ for locus in loci_of(template)
1273
+ }
1274
+ self._domains = declared
1275
+ # Search-progress triggers, on the DECLARED domains: the evidence and
1276
+ # the prior both describe the space the problem published, not a space
1277
+ # a previous prior already narrowed, or a second prior would compound
1278
+ # the first one's bet without ever measuring it.
1279
+ # ONE cadence tick, read once and consumed by both channels: asking
1280
+ # each of them separately would let whichever ran first advance the
1281
+ # anchor and starve the other, which is a rule nobody declared.
1282
+ due = self._due()
1283
+ self._maybe_reauthor(declared, due)
1284
+ self._maybe_author_prior(declared, due)
1285
+ if due:
1286
+ self._evidence_at = len(self._rows)
1287
+ self._evidence_ticks += 1
1288
+ domains = self._effective_domains(declared)
1289
+ shown = [dict(config) for config in list(archive)[:self.archive_shown]]
1290
+ self._maybe_revise()
1291
+ self._last_call = (shown, n, domains, int(seed), dict(template))
1292
+
1293
+ self.telemetry.batches += 1
1294
+ # The generator SAMPLES from the (possibly biased) domains and is
1295
+ # VALIDATED against the declared ones. A prior is guidance about where
1296
+ # to spend, not a new definition of what is legal, so a candidate
1297
+ # outside the prior but inside the schema is admitted rather than
1298
+ # counted as a defect -- otherwise installing a prior would
1299
+ # manufacture rejections and fire the defect-repair channel on a
1300
+ # generator that did exactly what it was asked.
1301
+ report = self._run(self.artifact, shown, n, domains, seed,
1302
+ template=template, rng=rng, count=True,
1303
+ validate_domains=declared)
1304
+ self.last_report = report
1305
+ self.census.merge(report.census)
1306
+ self._score_pending_edit(report)
1307
+ self.telemetry.emitted += report.emitted
1308
+ self.telemetry.accepted += len(report.accepted)
1309
+ self.telemetry.rejected_shape += report.rejected_shape
1310
+ self.telemetry.rejected_out_of_domain += report.rejected_out_of_domain
1311
+ self.telemetry.duplicates += report.duplicates
1312
+ self.telemetry.archive_overlap += report.archive_overlap
1313
+ self.telemetry.repaired += report.repaired
1314
+ self.telemetry.repaired_loci += report.repaired_loci
1315
+
1316
+ pool = [dict(config) for config in report.accepted[:n]]
1317
+ while len(pool) < n:
1318
+ pool.append(uniform_candidate(template, candidate_model, rng=rng,
1319
+ restriction=restriction))
1320
+ self.telemetry.filled_uniform += 1
1321
+ return pool
1322
+
1323
+ def _run(self, artifact, archive, n, domains, seed, *, template, rng,
1324
+ count: bool,
1325
+ validate_domains: Optional[Mapping[str, Sequence[Any]]] = None,
1326
+ ) -> PoolReport:
1327
+ """One emission through the scaffold, validated. Optionally counted.
1328
+
1329
+ *count* is false for the revision guard's frozen replay, which must
1330
+ measure a challenger without the run's telemetry recording an
1331
+ emission the loop never saw.
1332
+
1333
+ *validate_domains*, when given, is what the pool is judged against --
1334
+ the DECLARED domains -- while *domains* is what the sampler draws
1335
+ from (a weighted prior may have biased them). Sampling guidance must
1336
+ never redefine what is legal.
1337
+ """
1338
+
1339
+ rows = self._emit(artifact, archive, n, domains, seed,
1340
+ template=template, count=count)
1341
+ return validate_pool(
1342
+ rows, template=template,
1343
+ domains=validate_domains if validate_domains is not None else domains,
1344
+ seen=self._seen, limit=n, repair=self.repair,
1345
+ rng=rng if rng is not None else random.Random(seed))
1346
+
1347
+ def _emit(self, artifact, archive, n, domains, seed, *, template,
1348
+ count: bool = True) -> Any:
1349
+ prelude = (scaffold_prelude(template, domains, nonce=int(seed))
1350
+ if self.scaffold else None)
1351
+ try:
1352
+ outcome = self.runtime.call(
1353
+ artifact, [[archive, int(n), domains, int(seed)]],
1354
+ prelude=prelude, notes_global=NOTES_GLOBAL)
1355
+ except TypeError:
1356
+ # A runtime that predates the prelude channel: the artifact still
1357
+ # runs, the scaffold simply is not there, and the harness-side
1358
+ # repair remains the only guard. Degrade, never fail.
1359
+ try:
1360
+ outcome = self.runtime.call(
1361
+ artifact, [[archive, int(n), domains, int(seed)]])
1362
+ except Exception:
1363
+ if count:
1364
+ self.telemetry.runtime_failures += 1
1365
+ return []
1366
+ except Exception: # a runtime that cannot even ship
1367
+ if count: # the call is a countable
1368
+ self.telemetry.runtime_failures += 1 # event, not an emergency
1369
+ return []
1370
+ if count:
1371
+ self._absorb_notes(getattr(outcome, "notes", None))
1372
+ if (not outcome.ok and self.shrink_on_overrun > 1
1373
+ and outcome.status in ("timeout", "memory")
1374
+ and int(n) > self.shrink_on_overrun):
1375
+ # A resource overrun is the one failure whose CAUSE the harness
1376
+ # can act on: `propose` is a distribution, so asking it for fewer
1377
+ # draws is the same request at a fraction of the work. A quarter
1378
+ # of a guided pool beats none of one, the shortfall still falls
1379
+ # back to schema-uniform, and both events stay counted.
1380
+ if count:
1381
+ self.telemetry.runtime_failures += 1
1382
+ self.telemetry.runtime_retries += 1
1383
+ smaller = max(1, int(n) // self.shrink_on_overrun)
1384
+ try:
1385
+ outcome = self.runtime.call(
1386
+ artifact, [[archive, smaller, domains, int(seed)]],
1387
+ prelude=prelude, notes_global=NOTES_GLOBAL)
1388
+ except Exception:
1389
+ return []
1390
+ if count and outcome.ok:
1391
+ self.telemetry.runtime_recovered += 1
1392
+ self._absorb_notes(getattr(outcome, "notes", None))
1393
+ if not outcome.ok:
1394
+ return []
1395
+ [rows] = outcome.results
1396
+ return rows if isinstance(rows, list) else []
1397
+ if not outcome.ok:
1398
+ if count:
1399
+ self.telemetry.runtime_failures += 1
1400
+ return []
1401
+ [rows] = outcome.results
1402
+ if not isinstance(rows, list):
1403
+ if count:
1404
+ self.telemetry.runtime_failures += 1
1405
+ return []
1406
+ return rows
1407
+
1408
+ def _absorb_notes(self, notes: Any) -> None:
1409
+ """The scaffold's own counters, from inside the sandbox."""
1410
+
1411
+ if not isinstance(notes, Mapping):
1412
+ return
1413
+ self.telemetry.scaffold_filled += int(notes.get("filled") or 0)
1414
+ self.telemetry.scaffold_out_of_domain += int(
1415
+ notes.get("out_of_domain") or 0)
1416
+ self.telemetry.scaffold_unknown_locus += int(
1417
+ notes.get("unknown_locus") or 0)
1418
+ by_locus = notes.get("by_locus")
1419
+ if isinstance(by_locus, Mapping):
1420
+ for locus, row in by_locus.items():
1421
+ if not isinstance(row, Mapping):
1422
+ continue
1423
+ name = str(locus)
1424
+ for _ in range(int(row.get("filled") or 0)):
1425
+ self.census.repaired(name)
1426
+ count = int(row.get("out_of_domain") or 0)
1427
+ if count:
1428
+ self.census.out_of_domain_by_locus[name] = (
1429
+ self.census.out_of_domain_by_locus.get(name, 0) + count)
1430
+ for sample in (notes.get("samples") or ()):
1431
+ if isinstance(sample, Mapping) and "locus" in sample:
1432
+ self.census.sample(str(sample["locus"]), sample.get("value"))
1433
+
1434
+ # -- revision from measured feedback ------------------------------------
1435
+
1436
+ def deficient(self) -> bool:
1437
+ """The preregistered trigger: has the harness MEASURED a defect?
1438
+
1439
+ Three kinds, in order of how little interpretation they need. A
1440
+ rejected candidate or a runtime failure is a broken contract, however
1441
+ rare. A batch whose novelty falls under ``min_novelty`` has collapsed
1442
+ -- onto itself or onto the archive -- which is different from the
1443
+ occasional collision any honest sampler makes in a small space, and
1444
+ the threshold is what keeps those two apart. And a generator whose
1445
+ measured children never survive is futile even when it is faultless.
1446
+
1447
+ A revision fires on evidence or not at all: time passing is not
1448
+ evidence, and W3 measured that revision LEVELS the rungs, so it must
1449
+ never fire quietly on a generator that is working.
1450
+ """
1451
+
1452
+ tel = self.telemetry
1453
+ if tel.batches == 0:
1454
+ return False
1455
+ if (tel.rejected_shape or tel.rejected_out_of_domain
1456
+ or tel.runtime_failures or tel.repaired
1457
+ or tel.scaffold_out_of_domain or tel.scaffold_unknown_locus):
1458
+ return True
1459
+ last = self.last_report
1460
+ if (last is not None and last.emitted
1461
+ and last.novelty_rate < self.min_novelty):
1462
+ return True
1463
+ return (tel.measured >= self.min_measured_for_revision
1464
+ and tel.survived == 0)
1465
+
1466
+ def _maybe_revise(self) -> None:
1467
+ if self.revise is None or self.telemetry.revisions >= self.max_revisions:
1468
+ return
1469
+ if not self.deficient():
1470
+ return
1471
+ self.telemetry.revisions += 1
1472
+ feedback = self.feedback()
1473
+ try:
1474
+ replacement = self.revise(self.artifact, feedback)
1475
+ except Exception: # a revision must not kill a run
1476
+ replacement = None
1477
+ if replacement is None:
1478
+ return
1479
+ if self.revision_guard and not self._guard_admits(replacement):
1480
+ self.telemetry.revisions_rejected += 1
1481
+ self._remember_rejected_edit(replacement, guarded=True)
1482
+ return
1483
+ self._pending_edit = {
1484
+ "revision": self.telemetry.revisions,
1485
+ "sha": replacement.source_sha256[:8],
1486
+ "excerpt": replacement.source[:400],
1487
+ "before": (self.last_report.defect_rate
1488
+ if self.last_report is not None else 0.0),
1489
+ "signature": self.census.signature(),
1490
+ }
1491
+ self.telemetry.revisions_accepted += 1
1492
+ self.artifact = replacement
1493
+
1494
+ def feedback(self) -> str:
1495
+ """The measured story this generator would hand a revision."""
1496
+
1497
+ return render_generation_feedback(
1498
+ self.telemetry, self.last_report, self._survivors,
1499
+ census=self.census, domains=self._domains,
1500
+ rejected_edits=self._rejected_edits)
1501
+
1502
+ # -- the guard, and the memory of what it (or measurement) rejected -----
1503
+
1504
+ def _guard_admits(self, replacement: AuthoredArtifact) -> bool:
1505
+ """Does a FROZEN replay say the revision measurably helped?
1506
+
1507
+ The generator seam never sees the problem or the evaluation cache, so
1508
+ the only honest validation available to it is its own emission,
1509
+ replayed against the identical inputs the incumbent was last measured
1510
+ on: same archive, same ``n``, same domains, same seed. Admission takes
1511
+ a conjunction, so a revision cannot buy defect reduction with
1512
+ collapse: the defect rate must strictly fall AND the novelty rate --
1513
+ the frozen-validation score, the fraction of the batch that was
1514
+ usable and new -- must not fall.
1515
+
1516
+ No model call and no evaluation is spent here; the incumbent's side of
1517
+ the comparison is the batch already measured.
1518
+ """
1519
+
1520
+ incumbent = self.last_report
1521
+ if self._last_call is None or incumbent is None:
1522
+ return True # nothing to compare against yet
1523
+ archive, n, domains, seed, template = self._last_call
1524
+ try:
1525
+ trial = self._run(replacement, archive, n, domains, seed,
1526
+ template=template, rng=random.Random(seed),
1527
+ count=False)
1528
+ except Exception:
1529
+ return False
1530
+ if not trial.emitted:
1531
+ return False
1532
+ return (trial.defect_rate < incumbent.defect_rate
1533
+ and trial.novelty_rate >= incumbent.novelty_rate)
1534
+
1535
+ def _remember_rejected_edit(self, artifact: AuthoredArtifact, *,
1536
+ guarded: bool, after: float = -1.0) -> None:
1537
+ before = (self.last_report.defect_rate
1538
+ if self.last_report is not None else 0.0)
1539
+ self._rejected_edits.append({
1540
+ "revision": self.telemetry.revisions,
1541
+ "sha": artifact.source_sha256[:8],
1542
+ "excerpt": artifact.source[:400],
1543
+ "before": before,
1544
+ "after": before if after < 0 else after,
1545
+ "signature": self.census.signature(),
1546
+ "guarded": bool(guarded),
1547
+ })
1548
+ del self._rejected_edits[:-3]
1549
+
1550
+ def _score_pending_edit(self, report: PoolReport) -> None:
1551
+ """Did the revision we accepted last time actually fix anything?
1552
+
1553
+ Measured on the first batch the replacement emitted. If the defect
1554
+ rate did not fall, the edit joins the rejected-edit memory and every
1555
+ later revision is told, by name, that it was tried and failed.
1556
+ """
1557
+
1558
+ pending, self._pending_edit = self._pending_edit, None
1559
+ if pending is None:
1560
+ return
1561
+ if report.defect_rate < float(pending["before"]):
1562
+ return
1563
+ self.telemetry.revisions_rejected += 1
1564
+ pending["after"] = report.defect_rate
1565
+ pending["guarded"] = False
1566
+ self._rejected_edits.append(pending)
1567
+ del self._rejected_edits[:-3]
1568
+
1569
+ # -- the SEARCH-PROGRESS channel: reasoning over measurements ------------
1570
+
1571
+ def _evidence_rows(self) -> List[MeasuredRow]:
1572
+ """The rows the model will be shown. Identity, unless a view is set."""
1573
+
1574
+ rows: Sequence[MeasuredRow] = tuple(self._rows)
1575
+ if self.evidence_view is not None:
1576
+ try:
1577
+ rows = self.evidence_view(rows)
1578
+ except Exception: # a control that throws must not
1579
+ rows = () # be able to kill a measurement
1580
+ return [row for row in rows]
1581
+
1582
+ def _render_evidence(self, rows, domains) -> str:
1583
+ return render_measurement_evidence(
1584
+ rows, self.objectives, domains,
1585
+ front_shown=self.evidence_front_shown,
1586
+ effects_shown=self.evidence_effects_shown,
1587
+ # What the RUN charged and this generator was told about, not what
1588
+ # it produced: the model is entitled to know how much of the
1589
+ # budget bought the rows in front of it.
1590
+ charged=len(self._rows))
1591
+
1592
+ def _due(self) -> bool:
1593
+ """Is an evidence-conditioned call due, and on WHOSE clock?
1594
+
1595
+ Two conditions, in the order they bind.
1596
+
1597
+ The channel cannot reason about rows it does not hold, so the FIRST
1598
+ call waits on EVIDENCE and nothing else: ``evidence_min_rows`` measured
1599
+ rows, which by default is the fewest a determinable effect can be
1600
+ computed from. Everything after it waits on the declared CADENCE --
1601
+ ``reauthor_every`` further measured rows since the last call.
1602
+
1603
+ Splitting the two is the W11 fix. A single cadence had to answer both
1604
+ questions at once, and answering "when may it first speak?" with "when
1605
+ my own children number N" made the channel arrive two generations
1606
+ after the evidence did: on the EDA venue the prior was authored at a
1607
+ median charge of 40 against a 43.5-charge target, with the run's first
1608
+ 20 charges structurally invisible to it. Rows are counted, not
1609
+ children; ``reauthor_every == 0`` still means the channel never fires.
1610
+ """
1611
+
1612
+ if self.reauthor_every <= 0:
1613
+ return False
1614
+ if len(self._rows) < self.evidence_min_rows:
1615
+ return False
1616
+ if self._evidence_ticks == 0:
1617
+ return True
1618
+ return len(self._rows) - self._evidence_at >= self.reauthor_every
1619
+
1620
+ def _log(self, kind: str, *, rows: int, evidence: str,
1621
+ emitted: Optional[str], accepted: bool, **extra: Any) -> None:
1622
+ record: Dict[str, Any] = {
1623
+ "kind": kind,
1624
+ # Two different clocks, both recorded, neither standing in for the
1625
+ # other: `at_measured` is this generator's OWN children (the
1626
+ # attribution clock) and `at_rows` is every charged measurement the
1627
+ # run had reported to it (the evidence clock). Before W11 they were
1628
+ # the same number, which is precisely why the channel's lateness
1629
+ # was invisible in its own telemetry.
1630
+ "at_measured": int(self.telemetry.measured),
1631
+ "at_rows": len(self._rows),
1632
+ "rows_shown": int(rows),
1633
+ "evidence_sha256": evidence_digest(evidence),
1634
+ "evidence_chars": len(evidence),
1635
+ "emitted": emitted,
1636
+ "accepted": bool(accepted),
1637
+ }
1638
+ record.update(extra)
1639
+ self.evidence_log.append(record)
1640
+
1641
+ def _maybe_reauthor(self, domains: Mapping[str, Sequence[Any]],
1642
+ due: bool) -> None:
1643
+ """Re-author the sampler against the measured trace, on cadence.
1644
+
1645
+ The trigger is SEARCH PROGRESS, not an emission defect: a generator
1646
+ that emits perfectly valid candidates out of a region the run has
1647
+ already measured to be uncompetitive is never deficient, and is
1648
+ exactly the case the defect trigger cannot see.
1649
+ """
1650
+
1651
+ if (self.reauthor is None or not due
1652
+ or self.telemetry.reauthorings >= self.max_reauthorings):
1653
+ return
1654
+ rows = self._evidence_rows()
1655
+ if not rows:
1656
+ return
1657
+ self.telemetry.reauthorings += 1
1658
+ self.telemetry.evidence_rows_shown += len(rows)
1659
+ evidence = self._render_evidence(rows, domains)
1660
+ try:
1661
+ replacement = self.reauthor(self.artifact, evidence)
1662
+ except Exception: # a re-authoring must not kill a run
1663
+ replacement = None
1664
+ self._log("reauthor", rows=len(rows), evidence=evidence,
1665
+ emitted=(None if replacement is None
1666
+ else replacement.source_sha256),
1667
+ accepted=replacement is not None,
1668
+ replaced=self.artifact.source_sha256)
1669
+ if replacement is None:
1670
+ return
1671
+ self.telemetry.reauthorings_accepted += 1
1672
+ self.artifact = replacement
1673
+
1674
+ def _maybe_author_prior(self, domains: Mapping[str, Sequence[Any]],
1675
+ due: bool) -> None:
1676
+ """Ask which loci matter, type the answer, and let the GATE refuse it."""
1677
+
1678
+ self._unwind_prior_if_it_stopped_paying()
1679
+ if (self.prior_author is None or not due
1680
+ or self._prior is not None
1681
+ or self.telemetry.priors_proposed >= self.max_priors):
1682
+ return
1683
+ rows = self._evidence_rows()
1684
+ if not rows:
1685
+ return
1686
+ self.telemetry.priors_proposed += 1
1687
+ evidence = self._render_evidence(rows, domains)
1688
+ prompt = WEIGHTED_RESTRICTION_PROMPT.format(
1689
+ goals="\n".join(f" {s.name}: {s.goal}imise" for s in self.objectives),
1690
+ domains="\n".join(
1691
+ f" {name}: {json_compact(list(values))}"
1692
+ for name, values in sorted(dict(domains).items())),
1693
+ evidence=evidence,
1694
+ max_ratio=f"{float(self.prior_max_weight_ratio):g}")
1695
+ try:
1696
+ reply = self.prior_author(prompt)
1697
+ except Exception:
1698
+ reply = ""
1699
+ verdict = admit_weighted_restriction(
1700
+ parse_weighted_restriction(reply),
1701
+ domains=domains,
1702
+ max_weight_ratio=self.prior_max_weight_ratio)
1703
+ self._log("locus_prior", rows=len(rows), evidence=evidence,
1704
+ emitted=(None if verdict.proposal is None
1705
+ else evidence_digest(json_compact(
1706
+ verdict.proposal.as_note()))),
1707
+ accepted=verdict.admitted, verdict=verdict.as_note())
1708
+ if not verdict.admitted:
1709
+ self.telemetry.priors_refused += 1
1710
+ return
1711
+ self.telemetry.priors_admitted += 1
1712
+ self._prior = verdict.prior
1713
+ self._prior_batches = 0
1714
+ self._survived_at_prior = self.telemetry.survived
1715
+
1716
+ def _effective_domains(
1717
+ self, declared: Mapping[str, Sequence[Any]]
1718
+ ) -> Dict[str, List[Any]]:
1719
+ if self._prior is None:
1720
+ return {k: list(v) for k, v in dict(declared).items()}
1721
+ self._prior_batches += 1
1722
+ return apply_weighted_restriction(declared, self._prior)
1723
+
1724
+ def _unwind_prior_if_it_stopped_paying(self) -> None:
1725
+ """An admitted prior is still a bet, and a bet must be checkable.
1726
+
1727
+ A graded restriction cannot exclude a measured front member -- every
1728
+ declared value keeps positive mass -- so nothing about it is
1729
+ unrecoverable. It can still be WRONG, and wrong only shows up as
1730
+ spend: the prior is held only while it pays. After
1731
+ ``prior_unwind_batches`` generations drawn under it with not one new
1732
+ survivor, it is dropped and the run finishes on the declared domains.
1733
+ """
1734
+
1735
+ if self._prior is None or self.prior_unwind_batches <= 0:
1736
+ return
1737
+ if self._prior_batches < self.prior_unwind_batches:
1738
+ return
1739
+ if self.telemetry.survived > self._survived_at_prior:
1740
+ return
1741
+ self._prior = None
1742
+ self.telemetry.priors_unwound += 1
1743
+
1744
+ def note(self) -> Dict[str, Any]:
1745
+ """The per-generation history record for the last batch."""
1746
+
1747
+ note = (self.last_report.as_note() if self.last_report is not None
1748
+ else PoolReport().as_note())
1749
+ note["artifact"] = f"{self.artifact.name}:{self.artifact.source_sha256[:8]}"
1750
+ note["revisions"] = self.telemetry.revisions_accepted
1751
+ if self.reauthor_every > 0:
1752
+ # What the model saw and what it emitted, in the run's own record.
1753
+ # Only what is NEW since the last generation's note: the full log
1754
+ # stays on the object, and the history is a diary rather than n
1755
+ # copies of the same list.
1756
+ fresh = self.evidence_log[self._noted:]
1757
+ self._noted = len(self.evidence_log)
1758
+ note["evidence"] = [dict(record) for record in fresh]
1759
+ note["prior_active"] = self._prior is not None
1760
+ return note