agentevolve-optimizer 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (414) hide show
  1. agent_evolve/__init__.py +722 -0
  2. agent_evolve/agentic.py +2800 -0
  3. agent_evolve/api.py +767 -0
  4. agent_evolve/application/__init__.py +1580 -0
  5. agent_evolve/application/action_allocation.py +744 -0
  6. agent_evolve/application/action_allocation_frame.py +347 -0
  7. agent_evolve/application/action_allocation_frame_commit.py +185 -0
  8. agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
  9. agent_evolve/application/action_allocation_frame_v3.py +338 -0
  10. agent_evolve/application/action_archive_value.py +497 -0
  11. agent_evolve/application/action_evidence_consistency.py +455 -0
  12. agent_evolve/application/action_forecast_partitioning.py +1471 -0
  13. agent_evolve/application/action_metric_projection.py +211 -0
  14. agent_evolve/application/action_role_value.py +680 -0
  15. agent_evolve/application/action_score_authorities.py +363 -0
  16. agent_evolve/application/action_structural_signature.py +116 -0
  17. agent_evolve/application/action_target_realization.py +402 -0
  18. agent_evolve/application/agentic_evolution.py +7734 -0
  19. agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
  20. agent_evolve/application/anchor_residual_identification.py +463 -0
  21. agent_evolve/application/archive_conditioned_action_target.py +208 -0
  22. agent_evolve/application/artifact_journal.py +246 -0
  23. agent_evolve/application/artifact_replay.py +347 -0
  24. agent_evolve/application/budgeted_optimizer.py +1828 -0
  25. agent_evolve/application/calibrated_campaign.py +485 -0
  26. agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
  27. agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
  28. agent_evolve/application/campaign_capacity_recourse.py +254 -0
  29. agent_evolve/application/campaign_contextual_outcomes.py +119 -0
  30. agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
  31. agent_evolve/application/campaign_evidence_registry.py +262 -0
  32. agent_evolve/application/campaign_execution.py +2537 -0
  33. agent_evolve/application/campaign_generation_audit.py +942 -0
  34. agent_evolve/application/campaign_learning.py +1812 -0
  35. agent_evolve/application/campaign_learning_runtime.py +1977 -0
  36. agent_evolve/application/campaign_search_phase.py +227 -0
  37. agent_evolve/application/campaign_selector_context_extension.py +220 -0
  38. agent_evolve/application/campaign_variation_envelope.py +649 -0
  39. agent_evolve/application/campaign_variation_trace.py +451 -0
  40. agent_evolve/application/candidate_archive_consequence.py +128 -0
  41. agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
  42. agent_evolve/application/composite_outcome_updater.py +145 -0
  43. agent_evolve/application/composition_portfolio_selection.py +363 -0
  44. agent_evolve/application/concurrent_stage.py +144 -0
  45. agent_evolve/application/contextual_action_allocation.py +181 -0
  46. agent_evolve/application/contextual_campaign_outcomes.py +267 -0
  47. agent_evolve/application/contextual_campaign_planning.py +1366 -0
  48. agent_evolve/application/contextual_delayed_credit.py +651 -0
  49. agent_evolve/application/contextual_search_controller.py +2374 -0
  50. agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
  51. agent_evolve/application/decision_metric_projection.py +112 -0
  52. agent_evolve/application/derived_action_semantics.py +129 -0
  53. agent_evolve/application/detailed_evaluation.py +449 -0
  54. agent_evolve/application/earned_lineage.py +1011 -0
  55. agent_evolve/application/effective_choice_audit.py +484 -0
  56. agent_evolve/application/empirical_consequence_calibration.py +908 -0
  57. agent_evolve/application/evaluation_accounting.py +325 -0
  58. agent_evolve/application/evaluation_cache.py +199 -0
  59. agent_evolve/application/evaluation_escrow.py +547 -0
  60. agent_evolve/application/evaluation_recourse.py +253 -0
  61. agent_evolve/application/event_recorder.py +151 -0
  62. agent_evolve/application/evolution_campaign.py +1840 -0
  63. agent_evolve/application/executable_hypothesis.py +323 -0
  64. agent_evolve/application/factorial_branch_pilot.py +772 -0
  65. agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
  66. agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
  67. agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
  68. agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
  69. agent_evolve/application/finite_action_selection.py +188 -0
  70. agent_evolve/application/finite_action_set.py +306 -0
  71. agent_evolve/application/finite_action_transition.py +537 -0
  72. agent_evolve/application/finite_variation_eligibility.py +296 -0
  73. agent_evolve/application/forecast_geometry_portfolio.py +799 -0
  74. agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
  75. agent_evolve/application/front_proximity_admission.py +311 -0
  76. agent_evolve/application/front_proximity_parent_basis.py +458 -0
  77. agent_evolve/application/frozen_hurdle_score.py +659 -0
  78. agent_evolve/application/g3_causal_screen.py +2257 -0
  79. agent_evolve/application/g3_causal_validation.py +1046 -0
  80. agent_evolve/application/g3_postseal_curation.py +818 -0
  81. agent_evolve/application/gated_agentic_generator.py +205 -0
  82. agent_evolve/application/generation_feedback.py +293 -0
  83. agent_evolve/application/generative_proposal_journal.py +185 -0
  84. agent_evolve/application/geometry_conditional_elasticity.py +453 -0
  85. agent_evolve/application/global_wave_action_allocation.py +1151 -0
  86. agent_evolve/application/head_mass_conditional_seat.py +268 -0
  87. agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
  88. agent_evolve/application/identifiable_reflection_learning.py +395 -0
  89. agent_evolve/application/identifiable_reflection_request.py +364 -0
  90. agent_evolve/application/in_memory_residual_archive.py +341 -0
  91. agent_evolve/application/insight_memory.py +1804 -0
  92. agent_evolve/application/live_runtime_manifest.py +758 -0
  93. agent_evolve/application/llm_task_queue.py +769 -0
  94. agent_evolve/application/matched_finite_action_block.py +409 -0
  95. agent_evolve/application/materialized_action_broker.py +2328 -0
  96. agent_evolve/application/materialized_action_constraints.py +83 -0
  97. agent_evolve/application/materialized_variation.py +211 -0
  98. agent_evolve/application/multi_option_evolution.py +1536 -0
  99. agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
  100. agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
  101. agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
  102. agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
  103. agent_evolve/application/outcome_relation.py +193 -0
  104. agent_evolve/application/paired_allocation_comparison.py +241 -0
  105. agent_evolve/application/paired_block_schedule.py +127 -0
  106. agent_evolve/application/parent_measurement.py +226 -0
  107. agent_evolve/application/pareto_archive.py +811 -0
  108. agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
  109. agent_evolve/application/portfolio_evolution.py +2950 -0
  110. agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
  111. agent_evolve/application/portfolio_memory_attribution.py +581 -0
  112. agent_evolve/application/portfolio_memory_dose.py +788 -0
  113. agent_evolve/application/portfolio_memory_matched_control.py +938 -0
  114. agent_evolve/application/portfolio_memory_transfer.py +297 -0
  115. agent_evolve/application/portfolio_optimization_memory.py +363 -0
  116. agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
  117. agent_evolve/application/portfolio_projection.py +335 -0
  118. agent_evolve/application/portfolio_recombination.py +2032 -0
  119. agent_evolve/application/post_evolution_reflection.py +834 -0
  120. agent_evolve/application/postcommit_rank_authority.py +245 -0
  121. agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
  122. agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
  123. agent_evolve/application/prequential_residual_exploration.py +343 -0
  124. agent_evolve/application/prequential_score_portfolio.py +954 -0
  125. agent_evolve/application/projections.py +292 -0
  126. agent_evolve/application/protected_action_committee.py +1027 -0
  127. agent_evolve/application/protected_branch_pilot.py +376 -0
  128. agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
  129. agent_evolve/application/provider_replay.py +910 -0
  130. agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
  131. agent_evolve/application/recombination_residual_expert.py +403 -0
  132. agent_evolve/application/reflection_workflow.py +571 -0
  133. agent_evolve/application/region_conditional_credit.py +911 -0
  134. agent_evolve/application/residual_campaign_runtime.py +531 -0
  135. agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
  136. agent_evolve/application/residual_headroom_ledger.py +1544 -0
  137. agent_evolve/application/residual_learning_transaction.py +396 -0
  138. agent_evolve/application/residual_portfolio_evolution.py +1228 -0
  139. agent_evolve/application/residual_reachability.py +749 -0
  140. agent_evolve/application/residual_stage_credit.py +499 -0
  141. agent_evolve/application/same_prefix_paired_audit.py +1580 -0
  142. agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
  143. agent_evolve/application/sequential_lineage_allocation.py +1017 -0
  144. agent_evolve/application/sequential_market_replay.py +1395 -0
  145. agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
  146. agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
  147. agent_evolve/application/single_score_action_allocation.py +299 -0
  148. agent_evolve/application/source_exposure_allocation.py +906 -0
  149. agent_evolve/application/staged_memory.py +210 -0
  150. agent_evolve/application/stratified_cold_start_allocation.py +732 -0
  151. agent_evolve/application/support_guarded_hurdle_score.py +549 -0
  152. agent_evolve/application/target_conditioned_action_forecast.py +595 -0
  153. agent_evolve/application/target_conditioned_campaign.py +566 -0
  154. agent_evolve/application/treatment_assignment.py +201 -0
  155. agent_evolve/application/trusted_objective_evidence.py +217 -0
  156. agent_evolve/application/two_stage_action_evolution.py +1131 -0
  157. agent_evolve/application/v8lite_allocation_policy.py +1083 -0
  158. agent_evolve/application/v9_candidate_policy.py +1303 -0
  159. agent_evolve/bootstrap.py +108 -0
  160. agent_evolve/campaign_presets.py +517 -0
  161. agent_evolve/campaign_profiles.py +452 -0
  162. agent_evolve/campaign_variation_topology.py +288 -0
  163. agent_evolve/campaign_workload.py +950 -0
  164. agent_evolve/cli.py +797 -0
  165. agent_evolve/contract.py +241 -0
  166. agent_evolve/core/__init__.py +91 -0
  167. agent_evolve/core/action_semantics.py +411 -0
  168. agent_evolve/core/authored.py +105 -0
  169. agent_evolve/core/formatting.py +286 -0
  170. agent_evolve/core/optimization_semantics.py +324 -0
  171. agent_evolve/core/problem.py +167 -0
  172. agent_evolve/core/results.py +323 -0
  173. agent_evolve/core/stats.py +70 -0
  174. agent_evolve/core/telemetry.py +100 -0
  175. agent_evolve/domain/__init__.py +89 -0
  176. agent_evolve/domain/artifact.py +162 -0
  177. agent_evolve/domain/durable_text.py +68 -0
  178. agent_evolve/domain/event.py +1454 -0
  179. agent_evolve/domain/finite_action_set.py +426 -0
  180. agent_evolve/domain/finite_variation.py +526 -0
  181. agent_evolve/domain/generative_emission.py +559 -0
  182. agent_evolve/domain/ids.py +163 -0
  183. agent_evolve/domain/inline_text.py +106 -0
  184. agent_evolve/domain/insight.py +27 -0
  185. agent_evolve/domain/lineage.py +737 -0
  186. agent_evolve/domain/llm_task_queue.py +960 -0
  187. agent_evolve/domain/outcome.py +96 -0
  188. agent_evolve/domain/patch.py +854 -0
  189. agent_evolve/domain/typed_json.py +542 -0
  190. agent_evolve/domain/variation_space.py +158 -0
  191. agent_evolve/driver.py +1014 -0
  192. agent_evolve/harness/__init__.py +29 -0
  193. agent_evolve/harness/base.py +242 -0
  194. agent_evolve/harness/directives.py +163 -0
  195. agent_evolve/harness/generative_seal.py +479 -0
  196. agent_evolve/harness/registry.py +41 -0
  197. agent_evolve/infrastructure/__init__.py +39 -0
  198. agent_evolve/infrastructure/artifacts/__init__.py +6 -0
  199. agent_evolve/infrastructure/artifacts/_verification.py +67 -0
  200. agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
  201. agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
  202. agent_evolve/infrastructure/asyncio_runtime.py +109 -0
  203. agent_evolve/infrastructure/authored_runtime.py +188 -0
  204. agent_evolve/infrastructure/authored_worker.py +171 -0
  205. agent_evolve/infrastructure/clock.py +53 -0
  206. agent_evolve/infrastructure/events/__init__.py +6 -0
  207. agent_evolve/infrastructure/events/_validation.py +89 -0
  208. agent_evolve/infrastructure/events/in_memory.py +56 -0
  209. agent_evolve/infrastructure/events/jsonl.py +193 -0
  210. agent_evolve/infrastructure/exception_provenance.py +215 -0
  211. agent_evolve/infrastructure/ids.py +118 -0
  212. agent_evolve/infrastructure/lineage_codec.py +1836 -0
  213. agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
  214. agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
  215. agent_evolve/infrastructure/resource_lease.py +370 -0
  216. agent_evolve/infrastructure/sanitization/__init__.py +8 -0
  217. agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
  218. agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
  219. agent_evolve/infrastructure/stream_liveness.py +383 -0
  220. agent_evolve/infrastructure/subprocess_boundary.py +136 -0
  221. agent_evolve/integrations/__init__.py +1 -0
  222. agent_evolve/integrations/botorch/__init__.py +28 -0
  223. agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
  224. agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
  225. agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
  226. agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
  227. agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
  228. agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
  229. agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
  230. agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
  231. agent_evolve/integrations/completion.py +242 -0
  232. agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
  233. agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
  234. agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
  235. agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
  236. agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
  237. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
  238. agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
  239. agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
  240. agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
  241. agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
  242. agent_evolve/integrations/pydantic_ai/harness.py +159 -0
  243. agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
  244. agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
  245. agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
  246. agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
  247. agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
  248. agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
  249. agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
  250. agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
  251. agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
  252. agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
  253. agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
  254. agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
  255. agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
  256. agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
  257. agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
  258. agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
  259. agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
  260. agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
  261. agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
  262. agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
  263. agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
  264. agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
  265. agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
  266. agent_evolve/integrations/pymoo_adapter.py +242 -0
  267. agent_evolve/policies/__init__.py +17 -0
  268. agent_evolve/policies/check.py +469 -0
  269. agent_evolve/policies/emit_scaffold.py +451 -0
  270. agent_evolve/policies/feedback/__init__.py +37 -0
  271. agent_evolve/policies/feedback/held_out_asn.py +1325 -0
  272. agent_evolve/policies/genetic.py +607 -0
  273. agent_evolve/policies/llm_backoff.py +183 -0
  274. agent_evolve/policies/llm_chooser.py +226 -0
  275. agent_evolve/policies/llm_generator.py +1760 -0
  276. agent_evolve/policies/llm_init.py +267 -0
  277. agent_evolve/policies/llm_operator.py +109 -0
  278. agent_evolve/policies/llm_prior.py +194 -0
  279. agent_evolve/policies/llm_surrogate.py +334 -0
  280. agent_evolve/policies/measurement_evidence.py +704 -0
  281. agent_evolve/policies/memory/__init__.py +223 -0
  282. agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
  283. agent_evolve/policies/memory/compatibility_matching.py +593 -0
  284. agent_evolve/policies/memory/global_falsification.py +1841 -0
  285. agent_evolve/policies/memory/prompt_shape.py +503 -0
  286. agent_evolve/policies/memory/randomized_subset.py +714 -0
  287. agent_evolve/policies/memory/staged_causal.py +1270 -0
  288. agent_evolve/policies/memory/treatment_compliance.py +759 -0
  289. agent_evolve/policies/objective_resolution/__init__.py +17 -0
  290. agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
  291. agent_evolve/policies/operator_portfolio.py +407 -0
  292. agent_evolve/policies/reguidance.py +1133 -0
  293. agent_evolve/policies/reward/__init__.py +83 -0
  294. agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
  295. agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
  296. agent_evolve/policies/reward/affine_hypervolume.py +490 -0
  297. agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
  298. agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
  299. agent_evolve/policies/reward/frozen_archive.py +360 -0
  300. agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
  301. agent_evolve/policies/search_state.py +208 -0
  302. agent_evolve/policies/selection/__init__.py +345 -0
  303. agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
  304. agent_evolve/policies/selection/affine_frontier_context.py +330 -0
  305. agent_evolve/policies/selection/affine_frontier_target.py +473 -0
  306. agent_evolve/policies/selection/archive_elite.py +1346 -0
  307. agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
  308. agent_evolve/policies/selection/calibrated_slate.py +1394 -0
  309. agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
  310. agent_evolve/policies/selection/common_candidate_pool.py +685 -0
  311. agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
  312. agent_evolve/policies/selection/disjoint_pairs.py +479 -0
  313. agent_evolve/policies/selection/elite_explorer.py +719 -0
  314. agent_evolve/policies/selection/finite_action.py +187 -0
  315. agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
  316. agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
  317. agent_evolve/policies/selection/forecast_calibration.py +922 -0
  318. agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
  319. agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
  320. agent_evolve/policies/selection/full_support_slate.py +91 -0
  321. agent_evolve/policies/selection/meaningful_direction.py +240 -0
  322. agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
  323. agent_evolve/policies/selection/model_anchored_slate.py +826 -0
  324. agent_evolve/policies/selection/phenotype_recourse.py +979 -0
  325. agent_evolve/policies/selection/proposal_support.py +368 -0
  326. agent_evolve/policies/selection/random_portfolio.py +254 -0
  327. agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
  328. agent_evolve/policies/selection/residual_frontier.py +463 -0
  329. agent_evolve/policies/selection/residual_frontier_target.py +605 -0
  330. agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
  331. agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
  332. agent_evolve/policies/selection/target_conditioned_features.py +812 -0
  333. agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
  334. agent_evolve/policies/selection/task_keyed_palette.py +906 -0
  335. agent_evolve/policies/semantics.py +147 -0
  336. agent_evolve/policies/structure.py +362 -0
  337. agent_evolve/policies/structured_output_budget.py +62 -0
  338. agent_evolve/policies/surrogate.py +696 -0
  339. agent_evolve/policies/variation/__init__.py +1 -0
  340. agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
  341. agent_evolve/policies/variation/crossover_inheritance.py +575 -0
  342. agent_evolve/policies/variation/disjoint_recombination.py +611 -0
  343. agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
  344. agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
  345. agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
  346. agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
  347. agent_evolve/policies/variation/typed_patch.py +1981 -0
  348. agent_evolve/policies/weighted_prior.py +394 -0
  349. agent_evolve/ports/__init__.py +383 -0
  350. agent_evolve/ports/action_allocation.py +733 -0
  351. agent_evolve/ports/action_allocation_frame.py +1153 -0
  352. agent_evolve/ports/action_allocation_frame_commit.py +294 -0
  353. agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
  354. agent_evolve/ports/action_allocation_frame_v3.py +995 -0
  355. agent_evolve/ports/action_forecast.py +1568 -0
  356. agent_evolve/ports/action_metric_projection.py +165 -0
  357. agent_evolve/ports/agentic_generator.py +1561 -0
  358. agent_evolve/ports/archive_context.py +136 -0
  359. agent_evolve/ports/artifact_sanitizer.py +44 -0
  360. agent_evolve/ports/artifact_store.py +225 -0
  361. agent_evolve/ports/clock.py +13 -0
  362. agent_evolve/ports/contextual_search_allocation.py +827 -0
  363. agent_evolve/ports/decision_metric_projection.py +258 -0
  364. agent_evolve/ports/event_store.py +55 -0
  365. agent_evolve/ports/executable_hypothesis.py +557 -0
  366. agent_evolve/ports/finite_acquisition.py +377 -0
  367. agent_evolve/ports/finite_acquisition_batch.py +296 -0
  368. agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
  369. agent_evolve/ports/finite_acquisition_json.py +247 -0
  370. agent_evolve/ports/finite_acquisition_space.py +168 -0
  371. agent_evolve/ports/finite_action_selection.py +348 -0
  372. agent_evolve/ports/finite_action_set.py +256 -0
  373. agent_evolve/ports/frontier_target.py +396 -0
  374. agent_evolve/ports/generation_failure.py +43 -0
  375. agent_evolve/ports/hard_feasibility.py +233 -0
  376. agent_evolve/ports/id_factory.py +34 -0
  377. agent_evolve/ports/llm_task_queue.py +93 -0
  378. agent_evolve/ports/objective_resolution.py +419 -0
  379. agent_evolve/ports/paired_allocation_comparison.py +401 -0
  380. agent_evolve/ports/paired_block_schedule.py +475 -0
  381. agent_evolve/ports/parent_measurement.py +336 -0
  382. agent_evolve/ports/portfolio_memory_dose.py +643 -0
  383. agent_evolve/ports/portfolio_selection.py +3169 -0
  384. agent_evolve/ports/postcommit_rank_authority.py +467 -0
  385. agent_evolve/ports/presented_action_evidence.py +794 -0
  386. agent_evolve/ports/resource_lease.py +162 -0
  387. agent_evolve/ports/structured_generator.py +734 -0
  388. agent_evolve/ports/structured_output_budget.py +120 -0
  389. agent_evolve/ports/subprocess_boundary.py +138 -0
  390. agent_evolve/ports/treatment_assignment.py +466 -0
  391. agent_evolve/ports/variation_catalog.py +76 -0
  392. agent_evolve/ports/variation_source.py +226 -0
  393. agent_evolve/proposal_mode.py +157 -0
  394. agent_evolve/proposers/__init__.py +10 -0
  395. agent_evolve/proposers/random_proposer.py +188 -0
  396. agent_evolve/provider_accounting.py +163 -0
  397. agent_evolve/py.typed +0 -0
  398. agent_evolve/reference_method.py +1570 -0
  399. agent_evolve/session/__init__.py +11 -0
  400. agent_evolve/session/authorship.py +864 -0
  401. agent_evolve/session/evaluate.py +236 -0
  402. agent_evolve/session/fidelity.py +237 -0
  403. agent_evolve/session/genetic_loop.py +742 -0
  404. agent_evolve/session/loop.py +803 -0
  405. agent_evolve/session/screening.py +671 -0
  406. agent_evolve/settings.py +376 -0
  407. agent_evolve/workload_kit.py +368 -0
  408. agent_evolve/workload_prompt.py +398 -0
  409. agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
  410. agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
  411. agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
  412. agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
  413. agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
  414. agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1372 @@
1
+ """Rank-balanced randomized pilot selection with exact seat propensities.
2
+
3
+ This policy repairs the deterministic one-lane-head-per-engine pilot. A
4
+ stale frozen prior score can promote a deep native rank over an engine's own
5
+ top ranks; a head-only pilot then never observes where the engine's native
6
+ ordering actually converts. The repaired design keeps engine coverage but:
7
+
8
+ * spreads seats across native rank BANDS via a low-discrepancy schedule, so
9
+ interior ranks are purchased, not only heads;
10
+ * block-randomizes the within-band head between the native-rank order and the
11
+ frozen-score order, so rank-source disagreement receives exposure; and
12
+ * mixes an exploration floor toward uniform-within-engine.
13
+
14
+ Every seat is drawn from an explicitly defined mixture distribution over the
15
+ engine's remaining candidates. The exact propensity of every support member
16
+ (selected or not) is emitted as an exact rational, so later causal analysis
17
+ can inverse-propensity-weight any realized pilot.
18
+
19
+ The policy knows no workload, objective, model, provider, or prompt.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import hashlib
25
+ import json
26
+ import math
27
+ import re
28
+ from dataclasses import dataclass, field
29
+ from fractions import Fraction
30
+
31
+ from agent_evolve.domain.patch import require_sha256
32
+
33
+
34
+ RANK_BALANCED_CAUSAL_PILOT_POLICY_ID = "rank_balanced_causal_pilot"
35
+ RANK_BALANCED_CAUSAL_PILOT_POLICY_VERSION = 1
36
+ _TOKEN = re.compile(r"^[a-z][a-z0-9_.:-]{0,255}$")
37
+ _DEFINITION_DOMAIN = (
38
+ b"agent-evolve:rank-balanced-causal-pilot-definition:v1\x00"
39
+ )
40
+ _MARKET_DOMAIN = b"agent-evolve:rank-balanced-causal-pilot-market:v1\x00"
41
+ _DESIGN_DOMAIN = b"agent-evolve:rank-balanced-causal-pilot-design:v1\x00"
42
+
43
+
44
+ def _canonical_json(value: object) -> bytes:
45
+ return json.dumps(
46
+ value,
47
+ allow_nan=False,
48
+ ensure_ascii=True,
49
+ separators=(",", ":"),
50
+ sort_keys=True,
51
+ ).encode("ascii", errors="strict")
52
+
53
+
54
+ def _hash(domain: bytes, value: object) -> str:
55
+ return hashlib.sha256(domain + _canonical_json(value)).hexdigest()
56
+
57
+
58
+ def _require_token(value: str, *, name: str) -> None:
59
+ if type(value) is not str or _TOKEN.fullmatch(value) is None:
60
+ raise ValueError(f"{name} must use the closed token grammar")
61
+
62
+
63
+ def _stable_unit_interval(*parts: object) -> float:
64
+ payload = _canonical_json(list(parts))
65
+ numerator = int.from_bytes(hashlib.sha256(payload).digest()[:8], "big")
66
+ return numerator / float(2**64)
67
+
68
+
69
+ def rank_band_index(
70
+ native_rank: int,
71
+ lane_size: int,
72
+ band_count: int,
73
+ ) -> int:
74
+ """Map one native rank into its lane band (band zero is the top)."""
75
+
76
+ if (
77
+ type(native_rank) is not int
78
+ or type(lane_size) is not int
79
+ or type(band_count) is not int
80
+ or native_rank <= 0
81
+ or lane_size <= 0
82
+ or band_count <= 0
83
+ or native_rank > lane_size
84
+ ):
85
+ raise ValueError("rank band inputs must be consistent positives")
86
+ return min(
87
+ ((native_rank - 1) * band_count) // lane_size,
88
+ band_count - 1,
89
+ )
90
+
91
+
92
+ def _validated_band_weights(
93
+ band_count: int,
94
+ band_weights: tuple[float, ...] | None,
95
+ ) -> tuple[Fraction, ...]:
96
+ if band_weights is None:
97
+ return tuple(
98
+ Fraction(1, band_count) for _ in range(band_count)
99
+ )
100
+ if (
101
+ type(band_weights) is not tuple
102
+ or len(band_weights) != band_count
103
+ or any(
104
+ type(value) is not float
105
+ or not math.isfinite(value)
106
+ or value <= 0.0
107
+ for value in band_weights
108
+ )
109
+ ):
110
+ raise ValueError(
111
+ "band_weights must be one positive float per band"
112
+ )
113
+ weights = tuple(Fraction(value) for value in band_weights)
114
+ if sum(weights, Fraction(0)) != Fraction(1):
115
+ raise ValueError("band_weights must sum to exactly one")
116
+ return weights
117
+
118
+
119
+ def rank_band_schedule(
120
+ lane_size: int,
121
+ band_count: int,
122
+ band_weights: tuple[float, ...] | None = None,
123
+ ) -> tuple[int, ...]:
124
+ """Low-discrepancy native-rank visitation order for one lane.
125
+
126
+ Ranks are grouped into contiguous bands and visited by an exact
127
+ weighted quota walk (D'Hondt over per-band visit counts): each step
128
+ visits the open band maximizing ``weight / (visits + 1)``, taking that
129
+ band's next-best native rank. With equal weights (``band_weights``
130
+ omitted) this round-robins band heads: a six-item lane with two bands
131
+ yields the validated order ``(1, 4, 2, 5, 3, 6)`` and three bands
132
+ yield ``(1, 3, 5, 2, 4, 6)``. Unequal weights concentrate early
133
+ seats on heavier bands while every positively weighted band retains a
134
+ nonzero visitation floor.
135
+ """
136
+
137
+ if (
138
+ type(lane_size) is not int
139
+ or type(band_count) is not int
140
+ or lane_size <= 0
141
+ or band_count <= 0
142
+ ):
143
+ raise ValueError("schedule inputs must be positive integers")
144
+ weights = _validated_band_weights(band_count, band_weights)
145
+ bands: list[list[int]] = [[] for _ in range(band_count)]
146
+ for rank in range(1, lane_size + 1):
147
+ bands[rank_band_index(rank, lane_size, band_count)].append(rank)
148
+ visits = [0] * band_count
149
+ schedule: list[int] = []
150
+ while len(schedule) < lane_size:
151
+ open_bands = [
152
+ band
153
+ for band in range(band_count)
154
+ if visits[band] < len(bands[band])
155
+ ]
156
+ chosen = min(
157
+ open_bands,
158
+ key=lambda band: (
159
+ -(weights[band] / (visits[band] + 1)),
160
+ band,
161
+ ),
162
+ )
163
+ schedule.append(bands[chosen][visits[chosen]])
164
+ visits[chosen] += 1
165
+ return tuple(schedule)
166
+
167
+
168
+ @dataclass(frozen=True, slots=True)
169
+ class RankBalancedPilotCandidate:
170
+ """Portable, outcome-blind view of one pilotable candidate."""
171
+
172
+ action_sha256: str
173
+ engine_id: str
174
+ native_rank: int
175
+ frozen_score: float | None = None
176
+ forecast_summary: float | None = None
177
+
178
+ def __post_init__(self) -> None:
179
+ require_sha256(self.action_sha256, "action_sha256")
180
+ _require_token(self.engine_id, name="engine_id")
181
+ if type(self.native_rank) is not int or self.native_rank <= 0:
182
+ raise ValueError("native_rank must be a positive integer")
183
+ if self.frozen_score is not None and (
184
+ type(self.frozen_score) is not float
185
+ or not math.isfinite(self.frozen_score)
186
+ ):
187
+ raise ValueError("frozen_score must be a finite float or None")
188
+ if self.forecast_summary is not None and (
189
+ type(self.forecast_summary) is not float
190
+ or not math.isfinite(self.forecast_summary)
191
+ ):
192
+ raise ValueError(
193
+ "forecast_summary must be a finite float or None"
194
+ )
195
+
196
+ def to_record(self) -> dict[str, object]:
197
+ self.__post_init__()
198
+ return {
199
+ "action_sha256": self.action_sha256,
200
+ "engine_id": self.engine_id,
201
+ "native_rank": self.native_rank,
202
+ "frozen_score_hex": (
203
+ None
204
+ if self.frozen_score is None
205
+ else self.frozen_score.hex()
206
+ ),
207
+ "forecast_summary_hex": (
208
+ None
209
+ if self.forecast_summary is None
210
+ else self.forecast_summary.hex()
211
+ ),
212
+ }
213
+
214
+
215
+ @dataclass(frozen=True, slots=True)
216
+ class RankBalancedPilotPropensity:
217
+ """Exact mixture propensity of one support member at one seat."""
218
+
219
+ action_sha256: str
220
+ propensity_numerator: int
221
+ propensity_denominator: int
222
+
223
+ def __post_init__(self) -> None:
224
+ require_sha256(self.action_sha256, "action_sha256")
225
+ if (
226
+ type(self.propensity_numerator) is not int
227
+ or type(self.propensity_denominator) is not int
228
+ or self.propensity_numerator < 0
229
+ or self.propensity_denominator <= 0
230
+ ):
231
+ raise ValueError("propensity must be an exact rational")
232
+
233
+ @property
234
+ def exact(self) -> Fraction:
235
+ return Fraction(
236
+ self.propensity_numerator,
237
+ self.propensity_denominator,
238
+ )
239
+
240
+ @property
241
+ def propensity(self) -> float:
242
+ return float(self.exact)
243
+
244
+ def to_record(self) -> dict[str, object]:
245
+ self.__post_init__()
246
+ return {
247
+ "action_sha256": self.action_sha256,
248
+ "propensity_numerator": self.propensity_numerator,
249
+ "propensity_denominator": self.propensity_denominator,
250
+ "propensity_hex": self.propensity.hex(),
251
+ }
252
+
253
+
254
+ @dataclass(frozen=True, slots=True)
255
+ class RankBalancedPilotSeat:
256
+ """One realized pilot seat plus its complete support distribution."""
257
+
258
+ seat_ordinal: int
259
+ engine_id: str
260
+ engine_seat_index: int
261
+ target_band_index: int
262
+ effective_band_index: int
263
+ selected_action_sha256: str
264
+ branch: str
265
+ directed_order: str
266
+ support_propensities: tuple[RankBalancedPilotPropensity, ...]
267
+
268
+ def __post_init__(self) -> None:
269
+ if type(self.seat_ordinal) is not int or self.seat_ordinal <= 0:
270
+ raise ValueError("seat_ordinal must be positive")
271
+ _require_token(self.engine_id, name="engine_id")
272
+ if (
273
+ type(self.engine_seat_index) is not int
274
+ or self.engine_seat_index < 0
275
+ ):
276
+ raise ValueError("engine_seat_index must be non-negative")
277
+ for name in ("target_band_index", "effective_band_index"):
278
+ value = getattr(self, name)
279
+ if type(value) is not int or value < 0:
280
+ raise ValueError(f"{name} must be non-negative")
281
+ require_sha256(
282
+ self.selected_action_sha256,
283
+ "selected_action_sha256",
284
+ )
285
+ for value, name in (
286
+ (self.branch, "branch"),
287
+ (self.directed_order, "directed_order"),
288
+ ):
289
+ _require_token(value, name=name)
290
+ if (
291
+ type(self.support_propensities) is not tuple
292
+ or not self.support_propensities
293
+ or any(
294
+ type(value) is not RankBalancedPilotPropensity
295
+ for value in self.support_propensities
296
+ )
297
+ ):
298
+ raise TypeError(
299
+ "support_propensities must be exact and non-empty"
300
+ )
301
+ support_ids = tuple(
302
+ value.action_sha256 for value in self.support_propensities
303
+ )
304
+ if support_ids != tuple(sorted(set(support_ids))):
305
+ raise ValueError("support propensities must be canonical")
306
+ if self.selected_action_sha256 not in set(support_ids):
307
+ raise ValueError("selected action must be in the seat support")
308
+ for value in self.support_propensities:
309
+ value.__post_init__()
310
+ if sum(
311
+ (value.exact for value in self.support_propensities),
312
+ Fraction(0),
313
+ ) != Fraction(1):
314
+ raise ValueError("seat propensities must sum to exactly one")
315
+ if self.selected_propensity_exact <= 0:
316
+ raise ValueError("selected propensity must be positive")
317
+
318
+ @property
319
+ def selected_propensity_exact(self) -> Fraction:
320
+ return next(
321
+ value.exact
322
+ for value in self.support_propensities
323
+ if value.action_sha256 == self.selected_action_sha256
324
+ )
325
+
326
+ @property
327
+ def selection_propensity(self) -> float:
328
+ return float(self.selected_propensity_exact)
329
+
330
+ def to_record(self) -> dict[str, object]:
331
+ self.__post_init__()
332
+ return {
333
+ "seat_ordinal": self.seat_ordinal,
334
+ "engine_id": self.engine_id,
335
+ "engine_seat_index": self.engine_seat_index,
336
+ "target_band_index": self.target_band_index,
337
+ "effective_band_index": self.effective_band_index,
338
+ "selected_action_sha256": self.selected_action_sha256,
339
+ "branch": self.branch,
340
+ "directed_order": self.directed_order,
341
+ "selection_propensity_hex": self.selection_propensity.hex(),
342
+ "support_propensities": [
343
+ value.to_record()
344
+ for value in self.support_propensities
345
+ ],
346
+ }
347
+
348
+
349
+ @dataclass(frozen=True, slots=True)
350
+ class RankBalancedPilotDesign:
351
+ """One deterministic-given-seed realized pilot with causal receipts."""
352
+
353
+ policy_id: str
354
+ policy_version: int
355
+ policy_definition_sha256: str
356
+ residual_request_sha256: str
357
+ market_sha256: str
358
+ pilot_width: int
359
+ seats: tuple[RankBalancedPilotSeat, ...]
360
+ design_sha256: str = field(init=False)
361
+
362
+ def __post_init__(self) -> None:
363
+ _require_token(self.policy_id, name="policy_id")
364
+ if type(self.policy_version) is not int or self.policy_version <= 0:
365
+ raise ValueError("policy_version must be positive")
366
+ for value, name in (
367
+ (self.policy_definition_sha256, "policy_definition_sha256"),
368
+ (self.residual_request_sha256, "residual_request_sha256"),
369
+ (self.market_sha256, "market_sha256"),
370
+ ):
371
+ require_sha256(value, name)
372
+ if type(self.pilot_width) is not int or self.pilot_width <= 0:
373
+ raise ValueError("pilot_width must be positive")
374
+ if (
375
+ type(self.seats) is not tuple
376
+ or len(self.seats) != self.pilot_width
377
+ or any(
378
+ type(value) is not RankBalancedPilotSeat
379
+ for value in self.seats
380
+ )
381
+ ):
382
+ raise TypeError("seats must exactly fill the pilot width")
383
+ for ordinal, value in enumerate(self.seats, start=1):
384
+ value.__post_init__()
385
+ if value.seat_ordinal != ordinal:
386
+ raise ValueError("seat ordinals must be sequential")
387
+ selected = tuple(
388
+ value.selected_action_sha256 for value in self.seats
389
+ )
390
+ if len(selected) != len(set(selected)):
391
+ raise ValueError("a pilot cannot repeat an action")
392
+ object.__setattr__(
393
+ self,
394
+ "design_sha256",
395
+ _hash(_DESIGN_DOMAIN, self._unsigned_record()),
396
+ )
397
+
398
+ @property
399
+ def selected_action_sha256s(self) -> tuple[str, ...]:
400
+ return tuple(
401
+ sorted(
402
+ value.selected_action_sha256 for value in self.seats
403
+ )
404
+ )
405
+
406
+ @property
407
+ def design_propensity(self) -> float:
408
+ product = Fraction(1)
409
+ for value in self.seats:
410
+ product *= value.selected_propensity_exact
411
+ return float(product)
412
+
413
+ def _unsigned_record(self) -> dict[str, object]:
414
+ return {
415
+ "schema_version": 1,
416
+ "policy": {
417
+ "policy_id": self.policy_id,
418
+ "policy_version": self.policy_version,
419
+ "definition_sha256": self.policy_definition_sha256,
420
+ },
421
+ "residual_request_sha256": self.residual_request_sha256,
422
+ "market_sha256": self.market_sha256,
423
+ "pilot_width": self.pilot_width,
424
+ "seats": [value.to_record() for value in self.seats],
425
+ "candidate_outcomes_observed": False,
426
+ }
427
+
428
+ def to_record(self) -> dict[str, object]:
429
+ self.__post_init__()
430
+ return {
431
+ **self._unsigned_record(),
432
+ "selected_action_sha256s": list(
433
+ self.selected_action_sha256s
434
+ ),
435
+ "design_propensity_hex": self.design_propensity.hex(),
436
+ "design_sha256": self.design_sha256,
437
+ }
438
+
439
+
440
+ #: Default per-band seat mass. The V70 full-market census measured
441
+ #: positive rates concentrated in the top native-rank bands (ranks 1-2
442
+ #: converted at roughly triple the deep-rank rate), so the default mass
443
+ #: leans onto the top and middle thirds while every band keeps a nonzero
444
+ #: visitation floor. All values are exact dyadic floats.
445
+ DEFAULT_PILOT_BAND_WEIGHTS = (0.5, 0.3125, 0.1875)
446
+
447
+
448
+ @dataclass(frozen=True, slots=True)
449
+ class RankBalancedCausalPilotPolicy:
450
+ """Engine-covering, band-scheduled, block-randomized pilot design."""
451
+
452
+ band_count: int = 3
453
+ band_weights: tuple[float, ...] = DEFAULT_PILOT_BAND_WEIGHTS
454
+ exploration_epsilon: float = 0.125
455
+ random_seed: int = 0
456
+ policy_id: str = RANK_BALANCED_CAUSAL_PILOT_POLICY_ID
457
+ policy_version: int = RANK_BALANCED_CAUSAL_PILOT_POLICY_VERSION
458
+ definition_sha256: str = field(init=False)
459
+
460
+ def __post_init__(self) -> None:
461
+ if type(self.band_count) is not int or self.band_count <= 0:
462
+ raise ValueError("band_count must be positive")
463
+ _validated_band_weights(self.band_count, self.band_weights)
464
+ if (
465
+ type(self.exploration_epsilon) is not float
466
+ or not math.isfinite(self.exploration_epsilon)
467
+ or not 0.0 <= self.exploration_epsilon < 1.0
468
+ ):
469
+ raise ValueError("exploration_epsilon must lie in [0, 1)")
470
+ if type(self.random_seed) is not int or self.random_seed < 0:
471
+ raise ValueError("random_seed must be non-negative")
472
+ _require_token(self.policy_id, name="policy_id")
473
+ if self.policy_version != RANK_BALANCED_CAUSAL_PILOT_POLICY_VERSION:
474
+ raise ValueError("policy_version is unsupported")
475
+ object.__setattr__(
476
+ self,
477
+ "definition_sha256",
478
+ _hash(
479
+ _DEFINITION_DOMAIN,
480
+ {
481
+ "schema_version": 1,
482
+ "policy_id": self.policy_id,
483
+ "policy_version": self.policy_version,
484
+ "band_count": self.band_count,
485
+ "band_weights_hex": [
486
+ value.hex() for value in self.band_weights
487
+ ],
488
+ "exploration_epsilon_hex": (
489
+ self.exploration_epsilon.hex()
490
+ ),
491
+ "random_seed": self.random_seed,
492
+ "seat_allocation": (
493
+ "engine_coverage_floor_then_dhondt_by_"
494
+ "engine_requested_width"
495
+ ),
496
+ "within_engine_schedule": (
497
+ "weighted_quota_rank_band_heads_low_discrepancy"
498
+ ),
499
+ "within_band_order": (
500
+ "blocked_randomization_native_rank_vs_"
501
+ "frozen_score"
502
+ ),
503
+ "exploration_floor": (
504
+ "epsilon_uniform_within_engine_remaining"
505
+ ),
506
+ "propensities": (
507
+ "exact_rational_mixture_conditional_on_prefix"
508
+ ),
509
+ "candidate_outcomes_observed": False,
510
+ "workload_objective_model_provider_prompt_branches": (
511
+ False
512
+ ),
513
+ },
514
+ ),
515
+ )
516
+
517
+ @staticmethod
518
+ def _validated_engines(
519
+ candidates: tuple[RankBalancedPilotCandidate, ...],
520
+ ) -> dict[str, tuple[RankBalancedPilotCandidate, ...]]:
521
+ if type(candidates) is not tuple or not candidates:
522
+ raise ValueError("candidates must be a non-empty exact tuple")
523
+ seen: set[str] = set()
524
+ engines: dict[str, list[RankBalancedPilotCandidate]] = {}
525
+ for value in candidates:
526
+ if type(value) is not RankBalancedPilotCandidate:
527
+ raise TypeError(
528
+ "candidates must contain exact pilot candidates"
529
+ )
530
+ value.__post_init__()
531
+ if value.action_sha256 in seen:
532
+ raise ValueError("candidate identities repeat")
533
+ seen.add(value.action_sha256)
534
+ engines.setdefault(value.engine_id, []).append(value)
535
+ result: dict[str, tuple[RankBalancedPilotCandidate, ...]] = {}
536
+ for engine_id in sorted(engines):
537
+ lane = tuple(
538
+ sorted(
539
+ engines[engine_id],
540
+ key=lambda value: value.native_rank,
541
+ )
542
+ )
543
+ if tuple(value.native_rank for value in lane) != tuple(
544
+ range(1, len(lane) + 1)
545
+ ):
546
+ raise ValueError(
547
+ "each engine must be a contiguous ranked list"
548
+ )
549
+ frozen_flags = {
550
+ value.frozen_score is not None for value in lane
551
+ }
552
+ if len(frozen_flags) != 1:
553
+ raise ValueError(
554
+ "an engine must carry frozen scores for all "
555
+ "candidates or none"
556
+ )
557
+ result[engine_id] = lane
558
+ return result
559
+
560
+ def _market_sha256(
561
+ self,
562
+ engines: dict[str, tuple[RankBalancedPilotCandidate, ...]],
563
+ ) -> str:
564
+ return _hash(
565
+ _MARKET_DOMAIN,
566
+ {
567
+ engine_id: [value.to_record() for value in lane]
568
+ for engine_id, lane in engines.items()
569
+ },
570
+ )
571
+
572
+ @staticmethod
573
+ def _seat_engine_sequence(
574
+ engines: dict[str, tuple[RankBalancedPilotCandidate, ...]],
575
+ pilot_width: int,
576
+ ) -> tuple[str, ...]:
577
+ """Coverage-floor seats first, then D'Hondt extra seats."""
578
+
579
+ widths = {
580
+ engine_id: len(lane) for engine_id, lane in engines.items()
581
+ }
582
+ floor_order = sorted(
583
+ widths,
584
+ key=lambda engine_id: (-widths[engine_id], engine_id),
585
+ )
586
+ sequence: list[str] = []
587
+ awarded = {engine_id: 0 for engine_id in widths}
588
+ for engine_id in floor_order:
589
+ if len(sequence) >= pilot_width:
590
+ break
591
+ sequence.append(engine_id)
592
+ awarded[engine_id] += 1
593
+ while len(sequence) < pilot_width:
594
+ open_engines = [
595
+ engine_id
596
+ for engine_id in widths
597
+ if awarded[engine_id] < widths[engine_id]
598
+ ]
599
+ if not open_engines:
600
+ raise ValueError(
601
+ "pilot width exceeds the candidate market"
602
+ )
603
+ chosen = min(
604
+ open_engines,
605
+ key=lambda engine_id: (
606
+ -Fraction(widths[engine_id], awarded[engine_id] + 1),
607
+ -widths[engine_id],
608
+ # Ascending engine id wins ties deterministically.
609
+ engine_id,
610
+ ),
611
+ )
612
+ sequence.append(chosen)
613
+ awarded[chosen] += 1
614
+ return tuple(sequence)
615
+
616
+ def design_pilot(
617
+ self,
618
+ *,
619
+ residual_request_sha256: str,
620
+ candidates: tuple[RankBalancedPilotCandidate, ...],
621
+ pilot_width: int,
622
+ ) -> RankBalancedPilotDesign:
623
+ """Realize one seeded pilot and its exact causal propensities."""
624
+
625
+ self.__post_init__()
626
+ require_sha256(
627
+ residual_request_sha256,
628
+ "residual_request_sha256",
629
+ )
630
+ engines = self._validated_engines(candidates)
631
+ if (
632
+ type(pilot_width) is not int
633
+ or not 1 <= pilot_width <= len(candidates)
634
+ ):
635
+ raise ValueError("pilot_width must fit the candidate market")
636
+ market_sha256 = self._market_sha256(engines)
637
+ engine_sequence = self._seat_engine_sequence(
638
+ engines,
639
+ pilot_width,
640
+ )
641
+ epsilon = Fraction(self.exploration_epsilon)
642
+ selected: set[str] = set()
643
+ block_native_first: dict[tuple[str, int], bool] = {}
644
+ seats: list[RankBalancedPilotSeat] = []
645
+ engine_seat_counts: dict[str, int] = {}
646
+ for seat_ordinal, engine_id in enumerate(
647
+ engine_sequence,
648
+ start=1,
649
+ ):
650
+ lane = engines[engine_id]
651
+ engine_seat_index = engine_seat_counts.get(engine_id, 0)
652
+ engine_seat_counts[engine_id] = engine_seat_index + 1
653
+ remaining = tuple(
654
+ value
655
+ for value in lane
656
+ if value.action_sha256 not in selected
657
+ )
658
+ if not remaining: # pragma: no cover - engine seats are capped
659
+ raise AssertionError("engine seat exceeded its lane")
660
+ band_sequence = tuple(
661
+ rank_band_index(
662
+ rank,
663
+ len(lane),
664
+ self.band_count,
665
+ )
666
+ for rank in rank_band_schedule(
667
+ len(lane),
668
+ self.band_count,
669
+ self.band_weights,
670
+ )
671
+ )
672
+ target_band = band_sequence[engine_seat_index % len(lane)]
673
+ effective_band = target_band
674
+ band_members: tuple[RankBalancedPilotCandidate, ...] = ()
675
+ for offset in range(len(lane)):
676
+ effective_band = band_sequence[
677
+ (engine_seat_index + offset) % len(lane)
678
+ ]
679
+ band_members = tuple(
680
+ value
681
+ for value in remaining
682
+ if rank_band_index(
683
+ value.native_rank,
684
+ len(lane),
685
+ self.band_count,
686
+ )
687
+ == effective_band
688
+ )
689
+ if band_members:
690
+ break
691
+ if not band_members: # pragma: no cover - remaining is non-empty
692
+ raise AssertionError("no band covers a remaining candidate")
693
+ has_frozen = lane[0].frozen_score is not None
694
+ native_head = min(
695
+ band_members,
696
+ key=lambda value: (
697
+ value.native_rank,
698
+ value.action_sha256,
699
+ ),
700
+ )
701
+ if has_frozen:
702
+ frozen_head = min(
703
+ band_members,
704
+ key=lambda value: (
705
+ -value.frozen_score,
706
+ value.native_rank,
707
+ value.action_sha256,
708
+ ),
709
+ )
710
+ else:
711
+ frozen_head = native_head
712
+ block_key = (engine_id, engine_seat_index // 2)
713
+ block_first = engine_seat_index % 2 == 0
714
+ if has_frozen and block_first:
715
+ block_native_first[block_key] = (
716
+ _stable_unit_interval(
717
+ self.random_seed,
718
+ market_sha256,
719
+ residual_request_sha256,
720
+ "pilot_block_order",
721
+ engine_id,
722
+ engine_seat_index // 2,
723
+ )
724
+ < 0.5
725
+ )
726
+ native_first = block_native_first.get(block_key, True)
727
+ if not has_frozen:
728
+ directed_order = "native_rank"
729
+ directed: dict[str, Fraction] = {
730
+ native_head.action_sha256: Fraction(1)
731
+ }
732
+ directed_head = native_head
733
+ elif block_first:
734
+ # The block order is drawn at this seat, so both orders
735
+ # are equally likely before the draw.
736
+ directed_order = (
737
+ "block_first_native_rank"
738
+ if native_first
739
+ else "block_first_frozen_score"
740
+ )
741
+ directed = {native_head.action_sha256: Fraction(0)}
742
+ directed[native_head.action_sha256] += Fraction(1, 2)
743
+ directed[frozen_head.action_sha256] = directed.get(
744
+ frozen_head.action_sha256,
745
+ Fraction(0),
746
+ ) + Fraction(1, 2)
747
+ directed_head = (
748
+ native_head if native_first else frozen_head
749
+ )
750
+ else:
751
+ # The block order was drawn and logged one seat earlier,
752
+ # so this seat's directed head is deterministic.
753
+ forced_native = not native_first
754
+ directed_order = (
755
+ "block_second_native_rank"
756
+ if forced_native
757
+ else "block_second_frozen_score"
758
+ )
759
+ directed_head = (
760
+ native_head if forced_native else frozen_head
761
+ )
762
+ directed = {directed_head.action_sha256: Fraction(1)}
763
+ support = tuple(
764
+ sorted(
765
+ remaining,
766
+ key=lambda value: value.action_sha256,
767
+ )
768
+ )
769
+ uniform_share = epsilon / len(support)
770
+ propensity_values: list[RankBalancedPilotPropensity] = []
771
+ for value in support:
772
+ mixture = (Fraction(1) - epsilon) * directed.get(
773
+ value.action_sha256,
774
+ Fraction(0),
775
+ ) + uniform_share
776
+ propensity_values.append(
777
+ RankBalancedPilotPropensity(
778
+ action_sha256=value.action_sha256,
779
+ propensity_numerator=mixture.numerator,
780
+ propensity_denominator=mixture.denominator,
781
+ )
782
+ )
783
+ propensities = tuple(propensity_values)
784
+ exploration_draw = _stable_unit_interval(
785
+ self.random_seed,
786
+ market_sha256,
787
+ residual_request_sha256,
788
+ "pilot_exploration_branch",
789
+ seat_ordinal,
790
+ )
791
+ if exploration_draw < self.exploration_epsilon:
792
+ choice_draw = _stable_unit_interval(
793
+ self.random_seed,
794
+ market_sha256,
795
+ residual_request_sha256,
796
+ "pilot_exploration_choice",
797
+ seat_ordinal,
798
+ )
799
+ chosen = support[
800
+ min(
801
+ int(choice_draw * len(support)),
802
+ len(support) - 1,
803
+ )
804
+ ]
805
+ branch = "exploration"
806
+ else:
807
+ chosen = directed_head
808
+ branch = "directed"
809
+ selected.add(chosen.action_sha256)
810
+ seats.append(
811
+ RankBalancedPilotSeat(
812
+ seat_ordinal=seat_ordinal,
813
+ engine_id=engine_id,
814
+ engine_seat_index=engine_seat_index,
815
+ target_band_index=target_band,
816
+ effective_band_index=effective_band,
817
+ selected_action_sha256=chosen.action_sha256,
818
+ branch=branch,
819
+ directed_order=directed_order,
820
+ support_propensities=propensities,
821
+ )
822
+ )
823
+ return RankBalancedPilotDesign(
824
+ policy_id=self.policy_id,
825
+ policy_version=self.policy_version,
826
+ policy_definition_sha256=self.definition_sha256,
827
+ residual_request_sha256=residual_request_sha256,
828
+ market_sha256=market_sha256,
829
+ pilot_width=pilot_width,
830
+ seats=tuple(seats),
831
+ )
832
+
833
+
834
+ SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_ID = (
835
+ "sequential_adaptive_band_pilot"
836
+ )
837
+ SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_VERSION = 1
838
+ _SEQUENTIAL_DEFINITION_DOMAIN = (
839
+ b"agent-evolve:sequential-adaptive-band-pilot-definition:v1\x00"
840
+ )
841
+
842
+
843
+ @dataclass(frozen=True, slots=True)
844
+ class PilotSeatObservation:
845
+ """One revealed outcome of an earlier pilot seat."""
846
+
847
+ action_sha256: str
848
+ feasible: bool
849
+ marginal_archive_gain: float
850
+
851
+ def __post_init__(self) -> None:
852
+ require_sha256(self.action_sha256, "action_sha256")
853
+ if type(self.feasible) is not bool:
854
+ raise TypeError("feasible must be exact")
855
+ if (
856
+ type(self.marginal_archive_gain) is not float
857
+ or not math.isfinite(self.marginal_archive_gain)
858
+ or self.marginal_archive_gain < 0.0
859
+ ):
860
+ raise ValueError(
861
+ "marginal_archive_gain must be finite and non-negative"
862
+ )
863
+
864
+ @property
865
+ def positive(self) -> bool:
866
+ return self.marginal_archive_gain > 0.0
867
+
868
+
869
+ @dataclass(frozen=True, slots=True)
870
+ class SequentialAdaptiveBandPilotPolicy:
871
+ """Sequential pilot: engine and band mass adapt to revealed outcomes.
872
+
873
+ Repairs measured on the deterministic one-shot pilot:
874
+
875
+ * engine floor seats are ordered by the engines' posterior positive
876
+ rate (Beta shrinkage toward the within-market global posterior),
877
+ tie-broken by ascending engine id — never by lane width, which was
878
+ measured anti-correlated with conversion on many-engine markets;
879
+ * the target band is SAMPLED per seat from the base band weights
880
+ multiplied by the band-level posterior positive rate (pooled
881
+ across engines, temperature-controlled), so revealed top-band
882
+ successes concentrate later mass toward heads while top-band
883
+ zeros preserve interior exploration; and
884
+ * blocked randomization and the epsilon-uniform floor are retained,
885
+ so every propensity remains an exact rational of the mixture.
886
+
887
+ Adaptivity uses only outcomes revealed BEFORE the seat; the policy
888
+ remains workload-, model-, and provider-blind.
889
+ """
890
+
891
+ band_count: int = 3
892
+ band_weights: tuple[float, ...] = DEFAULT_PILOT_BAND_WEIGHTS
893
+ exploration_epsilon: float = 0.125
894
+ adaptation_temperature: float = 1.0
895
+ prior_strength: float = 2.0
896
+ root_prior_probability: float = 0.5
897
+ random_seed: int = 0
898
+ policy_id: str = SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_ID
899
+ policy_version: int = SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_VERSION
900
+ definition_sha256: str = field(init=False)
901
+
902
+ def __post_init__(self) -> None:
903
+ if type(self.band_count) is not int or self.band_count <= 0:
904
+ raise ValueError("band_count must be positive")
905
+ _validated_band_weights(self.band_count, self.band_weights)
906
+ if (
907
+ type(self.exploration_epsilon) is not float
908
+ or not math.isfinite(self.exploration_epsilon)
909
+ or not 0.0 <= self.exploration_epsilon < 1.0
910
+ ):
911
+ raise ValueError("exploration_epsilon must lie in [0, 1)")
912
+ if (
913
+ type(self.adaptation_temperature) is not float
914
+ or not math.isfinite(self.adaptation_temperature)
915
+ or self.adaptation_temperature < 0.0
916
+ ):
917
+ raise ValueError(
918
+ "adaptation_temperature must be finite and non-negative"
919
+ )
920
+ if (
921
+ type(self.prior_strength) is not float
922
+ or not math.isfinite(self.prior_strength)
923
+ or self.prior_strength <= 0.0
924
+ ):
925
+ raise ValueError("prior_strength must be positive")
926
+ if (
927
+ type(self.root_prior_probability) is not float
928
+ or not math.isfinite(self.root_prior_probability)
929
+ or not 0.0 < self.root_prior_probability < 1.0
930
+ ):
931
+ raise ValueError(
932
+ "root_prior_probability must lie in (0, 1)"
933
+ )
934
+ if type(self.random_seed) is not int or self.random_seed < 0:
935
+ raise ValueError("random_seed must be non-negative")
936
+ _require_token(self.policy_id, name="policy_id")
937
+ if (
938
+ self.policy_version
939
+ != SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_VERSION
940
+ ):
941
+ raise ValueError("policy_version is unsupported")
942
+ object.__setattr__(
943
+ self,
944
+ "definition_sha256",
945
+ _hash(
946
+ _SEQUENTIAL_DEFINITION_DOMAIN,
947
+ {
948
+ "schema_version": 1,
949
+ "policy_id": self.policy_id,
950
+ "policy_version": self.policy_version,
951
+ "band_count": self.band_count,
952
+ "band_weights_hex": [
953
+ value.hex() for value in self.band_weights
954
+ ],
955
+ "exploration_epsilon_hex": (
956
+ self.exploration_epsilon.hex()
957
+ ),
958
+ "adaptation_temperature_hex": (
959
+ self.adaptation_temperature.hex()
960
+ ),
961
+ "prior_strength_hex": self.prior_strength.hex(),
962
+ "root_prior_probability_hex": (
963
+ self.root_prior_probability.hex()
964
+ ),
965
+ "random_seed": self.random_seed,
966
+ "engine_floor_order": (
967
+ "posterior_positive_rate_then_engine_id_"
968
+ "never_lane_width"
969
+ ),
970
+ "band_mass": (
971
+ "base_weights_times_pooled_band_posterior_"
972
+ "temperature_controlled_sampled_per_seat"
973
+ ),
974
+ "within_band_order": (
975
+ "blocked_randomization_native_rank_vs_"
976
+ "frozen_score"
977
+ ),
978
+ "exploration_floor": (
979
+ "epsilon_uniform_within_engine_remaining"
980
+ ),
981
+ "propensities": (
982
+ "exact_rational_mixture_conditional_on_"
983
+ "revealed_history"
984
+ ),
985
+ "future_outcomes_visible": False,
986
+ "workload_objective_model_provider_prompt_branches": (
987
+ False
988
+ ),
989
+ },
990
+ ),
991
+ )
992
+
993
+ @staticmethod
994
+ def _posterior(
995
+ *,
996
+ prior_mean: float,
997
+ prior_strength: float,
998
+ successes: float,
999
+ failures: float,
1000
+ ) -> float:
1001
+ return (prior_strength * prior_mean + successes) / (
1002
+ prior_strength + successes + failures
1003
+ )
1004
+
1005
+ def design_seat(
1006
+ self,
1007
+ *,
1008
+ residual_request_sha256: str,
1009
+ candidates: tuple[RankBalancedPilotCandidate, ...],
1010
+ selected_action_sha256s: tuple[str, ...],
1011
+ observations: tuple[PilotSeatObservation, ...],
1012
+ seat_ordinal: int,
1013
+ ) -> RankBalancedPilotSeat:
1014
+ """Choose one pilot seat given the revealed pilot history."""
1015
+
1016
+ self.__post_init__()
1017
+ require_sha256(
1018
+ residual_request_sha256,
1019
+ "residual_request_sha256",
1020
+ )
1021
+ engines = RankBalancedCausalPilotPolicy._validated_engines(
1022
+ candidates
1023
+ )
1024
+ by_action = {
1025
+ value.action_sha256: value for value in candidates
1026
+ }
1027
+ if (
1028
+ type(selected_action_sha256s) is not tuple
1029
+ or selected_action_sha256s
1030
+ != tuple(sorted(set(selected_action_sha256s)))
1031
+ or not set(selected_action_sha256s) <= set(by_action)
1032
+ ):
1033
+ raise ValueError(
1034
+ "selected_action_sha256s must be a canonical subset"
1035
+ )
1036
+ if type(observations) is not tuple or any(
1037
+ type(value) is not PilotSeatObservation
1038
+ for value in observations
1039
+ ):
1040
+ raise TypeError(
1041
+ "observations must contain exact seat observations"
1042
+ )
1043
+ observed_ids = set()
1044
+ for value in observations:
1045
+ value.__post_init__()
1046
+ if value.action_sha256 in observed_ids:
1047
+ raise ValueError("observations repeat an action")
1048
+ observed_ids.add(value.action_sha256)
1049
+ if not observed_ids <= set(selected_action_sha256s):
1050
+ raise ValueError(
1051
+ "observations must cover only selected actions"
1052
+ )
1053
+ if type(seat_ordinal) is not int or seat_ordinal <= 0:
1054
+ raise ValueError("seat_ordinal must be positive")
1055
+ market_sha256 = _hash(
1056
+ _MARKET_DOMAIN,
1057
+ {
1058
+ engine_id: [value.to_record() for value in lane]
1059
+ for engine_id, lane in engines.items()
1060
+ },
1061
+ )
1062
+
1063
+ # Posterior evidence from revealed pilot outcomes only.
1064
+ global_positive = sum(
1065
+ value.positive for value in observations
1066
+ )
1067
+ global_count = len(observations)
1068
+ global_posterior = self._posterior(
1069
+ prior_mean=self.root_prior_probability,
1070
+ prior_strength=self.prior_strength,
1071
+ successes=float(global_positive),
1072
+ failures=float(global_count - global_positive),
1073
+ )
1074
+ engine_counts: dict[str, list[float]] = {}
1075
+ band_counts: dict[int, list[float]] = {}
1076
+ for value in observations:
1077
+ candidate = by_action[value.action_sha256]
1078
+ lane = engines[candidate.engine_id]
1079
+ band = rank_band_index(
1080
+ candidate.native_rank,
1081
+ len(lane),
1082
+ self.band_count,
1083
+ )
1084
+ row = engine_counts.setdefault(
1085
+ candidate.engine_id,
1086
+ [0.0, 0.0],
1087
+ )
1088
+ row[0] += 1.0
1089
+ row[1] += float(value.positive)
1090
+ band_row = band_counts.setdefault(band, [0.0, 0.0])
1091
+ band_row[0] += 1.0
1092
+ band_row[1] += float(value.positive)
1093
+
1094
+ def engine_posterior(engine_id: str) -> float:
1095
+ count, positive = engine_counts.get(
1096
+ engine_id,
1097
+ [0.0, 0.0],
1098
+ )
1099
+ return self._posterior(
1100
+ prior_mean=global_posterior,
1101
+ prior_strength=self.prior_strength,
1102
+ successes=positive,
1103
+ failures=count - positive,
1104
+ )
1105
+
1106
+ def band_posterior(band: int) -> float:
1107
+ count, positive = band_counts.get(band, [0.0, 0.0])
1108
+ return self._posterior(
1109
+ prior_mean=global_posterior,
1110
+ prior_strength=self.prior_strength,
1111
+ successes=positive,
1112
+ failures=count - positive,
1113
+ )
1114
+
1115
+ selected = set(selected_action_sha256s)
1116
+ open_engines = {
1117
+ engine_id: tuple(
1118
+ value
1119
+ for value in lane
1120
+ if value.action_sha256 not in selected
1121
+ )
1122
+ for engine_id, lane in engines.items()
1123
+ }
1124
+ open_engines = {
1125
+ engine_id: remaining
1126
+ for engine_id, remaining in open_engines.items()
1127
+ if remaining
1128
+ }
1129
+ if not open_engines:
1130
+ raise ValueError("no engine has a remaining candidate")
1131
+ seated_engines = {
1132
+ by_action[value].engine_id
1133
+ for value in selected_action_sha256s
1134
+ }
1135
+ unseated = [
1136
+ engine_id
1137
+ for engine_id in open_engines
1138
+ if engine_id not in seated_engines
1139
+ ]
1140
+ # Coverage floor: while any engine is unseated, seats go to
1141
+ # unseated engines, ordered by posterior conversion evidence.
1142
+ pool = unseated if unseated else sorted(open_engines)
1143
+ engine_id = min(
1144
+ pool,
1145
+ key=lambda value: (-engine_posterior(value), value),
1146
+ )
1147
+ lane = engines[engine_id]
1148
+ remaining = open_engines[engine_id]
1149
+ engine_seat_index = sum(
1150
+ by_action[value].engine_id == engine_id
1151
+ for value in selected_action_sha256s
1152
+ )
1153
+
1154
+ # Adapted band mass over non-empty bands: exact rationals.
1155
+ non_empty: dict[int, tuple[RankBalancedPilotCandidate, ...]] = {}
1156
+ for value in remaining:
1157
+ band = rank_band_index(
1158
+ value.native_rank,
1159
+ len(lane),
1160
+ self.band_count,
1161
+ )
1162
+ non_empty.setdefault(band, ())
1163
+ non_empty[band] = (*non_empty[band], value)
1164
+ weights: dict[int, Fraction] = {}
1165
+ for band in sorted(non_empty):
1166
+ multiplier = band_posterior(band) ** (
1167
+ self.adaptation_temperature
1168
+ )
1169
+ weights[band] = Fraction(
1170
+ float(self.band_weights[band])
1171
+ ) * Fraction(float(multiplier))
1172
+ total_weight = sum(weights.values(), Fraction(0))
1173
+ if total_weight <= 0: # pragma: no cover - weights are positive
1174
+ raise AssertionError("band mass vanished")
1175
+ band_mass = {
1176
+ band: value / total_weight
1177
+ for band, value in weights.items()
1178
+ }
1179
+
1180
+ has_frozen = lane[0].frozen_score is not None
1181
+ block_key_index = engine_seat_index // 2
1182
+ block_first = engine_seat_index % 2 == 0
1183
+ if has_frozen and block_first:
1184
+ native_first = (
1185
+ _stable_unit_interval(
1186
+ self.random_seed,
1187
+ market_sha256,
1188
+ residual_request_sha256,
1189
+ "adaptive_pilot_block_order",
1190
+ engine_id,
1191
+ block_key_index,
1192
+ )
1193
+ < 0.5
1194
+ )
1195
+ elif has_frozen:
1196
+ native_first = (
1197
+ _stable_unit_interval(
1198
+ self.random_seed,
1199
+ market_sha256,
1200
+ residual_request_sha256,
1201
+ "adaptive_pilot_block_order",
1202
+ engine_id,
1203
+ block_key_index,
1204
+ )
1205
+ < 0.5
1206
+ )
1207
+ else:
1208
+ native_first = True
1209
+
1210
+ directed: dict[str, Fraction] = {}
1211
+ directed_heads: dict[int, RankBalancedPilotCandidate] = {}
1212
+ for band, members in non_empty.items():
1213
+ native_head = min(
1214
+ members,
1215
+ key=lambda value: (
1216
+ value.native_rank,
1217
+ value.action_sha256,
1218
+ ),
1219
+ )
1220
+ if has_frozen:
1221
+ frozen_head = min(
1222
+ members,
1223
+ key=lambda value: (
1224
+ -value.frozen_score,
1225
+ value.native_rank,
1226
+ value.action_sha256,
1227
+ ),
1228
+ )
1229
+ else:
1230
+ frozen_head = native_head
1231
+ if has_frozen and block_first:
1232
+ for head in (native_head, frozen_head):
1233
+ directed[head.action_sha256] = directed.get(
1234
+ head.action_sha256,
1235
+ Fraction(0),
1236
+ )
1237
+ directed[native_head.action_sha256] += (
1238
+ band_mass[band] / 2
1239
+ )
1240
+ directed[frozen_head.action_sha256] += (
1241
+ band_mass[band] / 2
1242
+ )
1243
+ directed_heads[band] = (
1244
+ native_head if native_first else frozen_head
1245
+ )
1246
+ else:
1247
+ head = (
1248
+ native_head
1249
+ if (not has_frozen) or native_first
1250
+ else frozen_head
1251
+ )
1252
+ # Block-second seats reuse the block order drawn at
1253
+ # the first seat of the block, deterministically.
1254
+ if has_frozen and not block_first:
1255
+ head = (
1256
+ frozen_head if native_first else native_head
1257
+ )
1258
+ directed[head.action_sha256] = directed.get(
1259
+ head.action_sha256,
1260
+ Fraction(0),
1261
+ ) + band_mass[band]
1262
+ directed_heads[band] = head
1263
+
1264
+ epsilon = Fraction(self.exploration_epsilon)
1265
+ support = tuple(
1266
+ sorted(
1267
+ remaining,
1268
+ key=lambda value: value.action_sha256,
1269
+ )
1270
+ )
1271
+ uniform_share = epsilon / len(support)
1272
+ propensity_values: list[RankBalancedPilotPropensity] = []
1273
+ for value in support:
1274
+ mixture = (Fraction(1) - epsilon) * directed.get(
1275
+ value.action_sha256,
1276
+ Fraction(0),
1277
+ ) + uniform_share
1278
+ propensity_values.append(
1279
+ RankBalancedPilotPropensity(
1280
+ action_sha256=value.action_sha256,
1281
+ propensity_numerator=mixture.numerator,
1282
+ propensity_denominator=mixture.denominator,
1283
+ )
1284
+ )
1285
+
1286
+ exploration_draw = _stable_unit_interval(
1287
+ self.random_seed,
1288
+ market_sha256,
1289
+ residual_request_sha256,
1290
+ "adaptive_pilot_exploration_branch",
1291
+ seat_ordinal,
1292
+ )
1293
+ if exploration_draw < self.exploration_epsilon:
1294
+ choice_draw = _stable_unit_interval(
1295
+ self.random_seed,
1296
+ market_sha256,
1297
+ residual_request_sha256,
1298
+ "adaptive_pilot_exploration_choice",
1299
+ seat_ordinal,
1300
+ )
1301
+ chosen = support[
1302
+ min(
1303
+ int(choice_draw * len(support)),
1304
+ len(support) - 1,
1305
+ )
1306
+ ]
1307
+ branch = "exploration"
1308
+ effective_band = rank_band_index(
1309
+ chosen.native_rank,
1310
+ len(lane),
1311
+ self.band_count,
1312
+ )
1313
+ else:
1314
+ band_draw = _stable_unit_interval(
1315
+ self.random_seed,
1316
+ market_sha256,
1317
+ residual_request_sha256,
1318
+ "adaptive_pilot_band",
1319
+ seat_ordinal,
1320
+ )
1321
+ cumulative = Fraction(0)
1322
+ effective_band = sorted(non_empty)[-1]
1323
+ for band in sorted(non_empty):
1324
+ cumulative += band_mass[band]
1325
+ if band_draw < float(cumulative):
1326
+ effective_band = band
1327
+ break
1328
+ chosen = directed_heads[effective_band]
1329
+ branch = "directed"
1330
+ if not has_frozen:
1331
+ directed_order = "native_rank"
1332
+ elif block_first:
1333
+ directed_order = (
1334
+ "block_first_native_rank"
1335
+ if native_first
1336
+ else "block_first_frozen_score"
1337
+ )
1338
+ else:
1339
+ directed_order = (
1340
+ "block_second_frozen_score"
1341
+ if native_first
1342
+ else "block_second_native_rank"
1343
+ )
1344
+ return RankBalancedPilotSeat(
1345
+ seat_ordinal=seat_ordinal,
1346
+ engine_id=engine_id,
1347
+ engine_seat_index=engine_seat_index,
1348
+ target_band_index=effective_band,
1349
+ effective_band_index=effective_band,
1350
+ selected_action_sha256=chosen.action_sha256,
1351
+ branch=branch,
1352
+ directed_order=directed_order,
1353
+ support_propensities=tuple(propensity_values),
1354
+ )
1355
+
1356
+
1357
+ __all__ = [
1358
+ "DEFAULT_PILOT_BAND_WEIGHTS",
1359
+ "PilotSeatObservation",
1360
+ "RANK_BALANCED_CAUSAL_PILOT_POLICY_ID",
1361
+ "RANK_BALANCED_CAUSAL_PILOT_POLICY_VERSION",
1362
+ "SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_ID",
1363
+ "SEQUENTIAL_ADAPTIVE_BAND_PILOT_POLICY_VERSION",
1364
+ "SequentialAdaptiveBandPilotPolicy",
1365
+ "RankBalancedCausalPilotPolicy",
1366
+ "RankBalancedPilotCandidate",
1367
+ "RankBalancedPilotDesign",
1368
+ "RankBalancedPilotPropensity",
1369
+ "RankBalancedPilotSeat",
1370
+ "rank_band_index",
1371
+ "rank_band_schedule",
1372
+ ]