flextool 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (322) hide show
  1. flextool/__init__.py +41 -0
  2. flextool/_mem_sampler.py +193 -0
  3. flextool/_resources.py +43 -0
  4. flextool/calibrate/__init__.py +51 -0
  5. flextool/calibrate/__main__.py +11 -0
  6. flextool/calibrate/_cli.py +316 -0
  7. flextool/calibrate/_db_alt.py +166 -0
  8. flextool/calibrate/_final_outputs.py +110 -0
  9. flextool/calibrate/_guard.py +151 -0
  10. flextool/calibrate/_loop.py +558 -0
  11. flextool/calibrate/_readers.py +223 -0
  12. flextool/calibrate/_report.py +263 -0
  13. flextool/calibrate/_sizing.py +699 -0
  14. flextool/calibrate/_solve.py +134 -0
  15. flextool/calibrate/_solve_status.py +495 -0
  16. flextool/cli/__init__.py +9 -0
  17. flextool/cli/_console.py +51 -0
  18. flextool/cli/_timing.py +147 -0
  19. flextool/cli/cmd_execute_flextool_workflow.py +187 -0
  20. flextool/cli/cmd_export_to_tabular.py +56 -0
  21. flextool/cli/cmd_import_sensitivities.py +75 -0
  22. flextool/cli/cmd_migrate_database.py +13 -0
  23. flextool/cli/cmd_open_results_db.py +269 -0
  24. flextool/cli/cmd_read_matpower.py +66 -0
  25. flextool/cli/cmd_read_old_flextool.py +63 -0
  26. flextool/cli/cmd_read_self_describing_tabular_input.py +50 -0
  27. flextool/cli/cmd_read_tabular_input.py +81 -0
  28. flextool/cli/cmd_run_flextool.py +1095 -0
  29. flextool/cli/cmd_scenario_results.py +284 -0
  30. flextool/cli/cmd_solve_mps.py +169 -0
  31. flextool/cli/cmd_update_flextool.py +17 -0
  32. flextool/cli/cmd_write_outputs.py +125 -0
  33. flextool/common_utils/__init__.py +1 -0
  34. flextool/common_utils/plot_mem_shape.py +77 -0
  35. flextool/common_utils/precision.py +451 -0
  36. flextool/decomposition/__init__.py +0 -0
  37. flextool/decomposition/region_decomposition.py +128 -0
  38. flextool/decomposition/region_filter.py +1261 -0
  39. flextool/engine_polars/__init__.py +110 -0
  40. flextool/engine_polars/_axis_enums.py +742 -0
  41. flextool/engine_polars/_benders.py +3462 -0
  42. flextool/engine_polars/_block_layout.py +1479 -0
  43. flextool/engine_polars/_blocks.py +1515 -0
  44. flextool/engine_polars/_commodity_ladder.py +660 -0
  45. flextool/engine_polars/_cumulative_invest.py +1165 -0
  46. flextool/engine_polars/_db_loader.py +153 -0
  47. flextool/engine_polars/_db_reader.py +127 -0
  48. flextool/engine_polars/_dc_power_flow.py +445 -0
  49. flextool/engine_polars/_delay.py +442 -0
  50. flextool/engine_polars/_derived_arithmetic.py +432 -0
  51. flextool/engine_polars/_derived_block.py +990 -0
  52. flextool/engine_polars/_derived_branch.py +769 -0
  53. flextool/engine_polars/_derived_existing.py +1353 -0
  54. flextool/engine_polars/_derived_npv.py +1297 -0
  55. flextool/engine_polars/_derived_params.py +9850 -0
  56. flextool/engine_polars/_derived_profile.py +881 -0
  57. flextool/engine_polars/_derived_walks.py +276 -0
  58. flextool/engine_polars/_determinism.py +70 -0
  59. flextool/engine_polars/_direct_params.py +2186 -0
  60. flextool/engine_polars/_dump_csvs.py +1009 -0
  61. flextool/engine_polars/_emit_arc_unions.py +1631 -0
  62. flextool/engine_polars/_emit_calc_params.py +729 -0
  63. flextool/engine_polars/_emit_chain_params.py +709 -0
  64. flextool/engine_polars/_emit_co2_accumulators.py +400 -0
  65. flextool/engine_polars/_emit_dispatchers.py +690 -0
  66. flextool/engine_polars/_emit_energy_margin.py +125 -0
  67. flextool/engine_polars/_emit_energy_margin_adder.py +290 -0
  68. flextool/engine_polars/_emit_entity_annual.py +428 -0
  69. flextool/engine_polars/_emit_inflow_scaling.py +1420 -0
  70. flextool/engine_polars/_emit_leaf_sets.py +550 -0
  71. flextool/engine_polars/_emit_lp_scaling.py +665 -0
  72. flextool/engine_polars/_emit_mid_sets.py +859 -0
  73. flextool/engine_polars/_emit_pdt_params.py +759 -0
  74. flextool/engine_polars/_emit_per_solve.py +774 -0
  75. flextool/engine_polars/_emit_period_calc.py +504 -0
  76. flextool/engine_polars/_emit_period_params.py +2398 -0
  77. flextool/engine_polars/_emit_provider_io.py +141 -0
  78. flextool/engine_polars/_emit_reserve.py +574 -0
  79. flextool/engine_polars/_emit_solve_time.py +311 -0
  80. flextool/engine_polars/_emit_solve_writers.py +1249 -0
  81. flextool/engine_polars/_flex_data_accumulator.py +388 -0
  82. flextool/engine_polars/_flex_data_provider.py +478 -0
  83. flextool/engine_polars/_group_slack.py +1253 -0
  84. flextool/engine_polars/_inmemory_reader.py +140 -0
  85. flextool/engine_polars/_input_source.py +336 -0
  86. flextool/engine_polars/_invest_seeds.py +191 -0
  87. flextool/engine_polars/_native_input_writer.py +100 -0
  88. flextool/engine_polars/_native_run_model.py +1348 -0
  89. flextool/engine_polars/_orchestration.py +4314 -0
  90. flextool/engine_polars/_output_writer.py +439 -0
  91. flextool/engine_polars/_param_shapes.py +1595 -0
  92. flextool/engine_polars/_parquet_bundle.py +723 -0
  93. flextool/engine_polars/_pdt_join.py +167 -0
  94. flextool/engine_polars/_pdt_lookup.py +547 -0
  95. flextool/engine_polars/_per_solve_sets.py +335 -0
  96. flextool/engine_polars/_projection_params.py +2056 -0
  97. flextool/engine_polars/_provider_keys.py +173 -0
  98. flextool/engine_polars/_provider_translators.py +225 -0
  99. flextool/engine_polars/_recursive_solve.py +703 -0
  100. flextool/engine_polars/_region_filter.py +2508 -0
  101. flextool/engine_polars/_reserve.py +649 -0
  102. flextool/engine_polars/_solve_acceptance.py +331 -0
  103. flextool/engine_polars/_solve_config.py +1001 -0
  104. flextool/engine_polars/_solve_context.py +885 -0
  105. flextool/engine_polars/_solve_handoff.py +164 -0
  106. flextool/engine_polars/_solve_state.py +232 -0
  107. flextool/engine_polars/_solver_base.py +36 -0
  108. flextool/engine_polars/_solver_dispatch.py +511 -0
  109. flextool/engine_polars/_spinedb_reader.py +1165 -0
  110. flextool/engine_polars/_stochastic.py +593 -0
  111. flextool/engine_polars/_subprocess_solve.py +1838 -0
  112. flextool/engine_polars/_timeline.py +1416 -0
  113. flextool/engine_polars/_vectorize.py +438 -0
  114. flextool/engine_polars/_warm.py +858 -0
  115. flextool/engine_polars/autoscale/__init__.py +107 -0
  116. flextool/engine_polars/autoscale/_config.py +218 -0
  117. flextool/engine_polars/autoscale/_layer2.py +1253 -0
  118. flextool/engine_polars/autoscale/_layer2_types.py +584 -0
  119. flextool/engine_polars/autoscale/_quantity_types.py +621 -0
  120. flextool/engine_polars/autoscale/_report.py +336 -0
  121. flextool/engine_polars/chain.py +259 -0
  122. flextool/engine_polars/input.py +6638 -0
  123. flextool/engine_polars/model.py +4754 -0
  124. flextool/env_check.py +388 -0
  125. flextool/export_to_tabular/__init__.py +5 -0
  126. flextool/export_to_tabular/db_reader.py +224 -0
  127. flextool/export_to_tabular/excel_writer.py +3559 -0
  128. flextool/export_to_tabular/export_settings.yaml +377 -0
  129. flextool/export_to_tabular/export_to_excel.py +227 -0
  130. flextool/export_to_tabular/formatting.py +543 -0
  131. flextool/export_to_tabular/sheet_config.py +876 -0
  132. flextool/gui/__init__.py +0 -0
  133. flextool/gui/__main__.py +118 -0
  134. flextool/gui/calibrate_commands.py +184 -0
  135. flextool/gui/calibrate_jobs.py +424 -0
  136. flextool/gui/check_tree.py +142 -0
  137. flextool/gui/cli_format.py +83 -0
  138. flextool/gui/config_parser.py +68 -0
  139. flextool/gui/data_models.py +362 -0
  140. flextool/gui/db_editor_integration.py +202 -0
  141. flextool/gui/db_version_check.py +269 -0
  142. flextool/gui/dialogs/__init__.py +0 -0
  143. flextool/gui/dialogs/add_dialog.py +1098 -0
  144. flextool/gui/dialogs/calibrate_dialog.py +1259 -0
  145. flextool/gui/dialogs/file_picker.py +473 -0
  146. flextool/gui/dialogs/group_picker.py +299 -0
  147. flextool/gui/dialogs/migration_consent_dialog.py +106 -0
  148. flextool/gui/dialogs/migration_progress_dialog.py +237 -0
  149. flextool/gui/dialogs/plot_dialog.py +459 -0
  150. flextool/gui/dialogs/plot_settings_picker.py +2184 -0
  151. flextool/gui/dialogs/project_dialog.py +426 -0
  152. flextool/gui/dialogs/update_dialog.py +212 -0
  153. flextool/gui/downsampling.py +88 -0
  154. flextool/gui/error_handling.py +50 -0
  155. flextool/gui/execution_manager.py +1715 -0
  156. flextool/gui/execution_window.py +1377 -0
  157. flextool/gui/hover_tooltip.py +111 -0
  158. flextool/gui/input_sources.py +730 -0
  159. flextool/gui/main_window.py +6181 -0
  160. flextool/gui/network_graph.py +215 -0
  161. flextool/gui/output_actions.py +393 -0
  162. flextool/gui/output_log_window.py +159 -0
  163. flextool/gui/platform_utils.py +421 -0
  164. flextool/gui/plot_cache.py +88 -0
  165. flextool/gui/plot_canvas.py +543 -0
  166. flextool/gui/plot_config_reader.py +272 -0
  167. flextool/gui/project_utils.py +100 -0
  168. flextool/gui/result_viewer.py +4394 -0
  169. flextool/gui/scenario_key.py +162 -0
  170. flextool/gui/scenario_lists.py +516 -0
  171. flextool/gui/settings_io.py +360 -0
  172. flextool/gui/solve_reader.py +103 -0
  173. flextool/gui/tree_reorder.py +88 -0
  174. flextool/gui/ui_metrics.py +420 -0
  175. flextool/input_derivation/__init__.py +281 -0
  176. flextool/input_derivation/_commodity_ladder.py +375 -0
  177. flextool/input_derivation/_commodity_ladder_sets.py +70 -0
  178. flextool/input_derivation/_dc_power_flow.py +377 -0
  179. flextool/input_derivation/_method_constants.py +77 -0
  180. flextool/input_derivation/_process_method.py +258 -0
  181. flextool/input_derivation/_specs.py +1026 -0
  182. flextool/input_derivation/_validators.py +321 -0
  183. flextool/lean_parquet.py +159 -0
  184. flextool/model_builder/__init__.py +5 -0
  185. flextool/model_builder/build_model.py +589 -0
  186. flextool/model_builder/encoding.py +67 -0
  187. flextool/model_builder/names.py +34 -0
  188. flextool/model_builder/profiles.py +129 -0
  189. flextool/plot_outputs/__init__.py +14 -0
  190. flextool/plot_outputs/axis_helpers.py +355 -0
  191. flextool/plot_outputs/color_template.py +888 -0
  192. flextool/plot_outputs/config.py +171 -0
  193. flextool/plot_outputs/format_helpers.py +345 -0
  194. flextool/plot_outputs/legend_helpers.py +143 -0
  195. flextool/plot_outputs/orchestrator.py +1141 -0
  196. flextool/plot_outputs/perf.py +37 -0
  197. flextool/plot_outputs/plan.py +1787 -0
  198. flextool/plot_outputs/plot_bars.py +1510 -0
  199. flextool/plot_outputs/plot_bars_detail.py +753 -0
  200. flextool/plot_outputs/plot_lines.py +951 -0
  201. flextool/plot_outputs/shared_manifest.py +564 -0
  202. flextool/plot_outputs/subplot_helpers.py +137 -0
  203. flextool/process_inputs/__init__.py +188 -0
  204. flextool/process_inputs/import_old_excel_input.json +4159 -0
  205. flextool/process_inputs/read_matpower.py +451 -0
  206. flextool/process_inputs/read_old_flextool.py +1288 -0
  207. flextool/process_inputs/read_self_describing_excel.py +1423 -0
  208. flextool/process_inputs/read_tabular_with_specification.py +1114 -0
  209. flextool/process_inputs/write_old_flextool_to_db.py +3077 -0
  210. flextool/process_inputs/write_self_describing_to_db.py +977 -0
  211. flextool/process_inputs/write_to_input_db.py +269 -0
  212. flextool/process_outputs/__init__.py +7 -0
  213. flextool/process_outputs/_annualize.py +55 -0
  214. flextool/process_outputs/_inmemory_helpers.py +292 -0
  215. flextool/process_outputs/_output_meta.py +672 -0
  216. flextool/process_outputs/calc_capacity_flows.py +107 -0
  217. flextool/process_outputs/calc_connections.py +136 -0
  218. flextool/process_outputs/calc_costs.py +260 -0
  219. flextool/process_outputs/calc_group_flows.py +192 -0
  220. flextool/process_outputs/calc_slacks.py +103 -0
  221. flextool/process_outputs/calc_storage_vre.py +160 -0
  222. flextool/process_outputs/drop_levels.py +208 -0
  223. flextool/process_outputs/handoff_writers.py +1315 -0
  224. flextool/process_outputs/out_ancillary.py +544 -0
  225. flextool/process_outputs/out_capacity.py +179 -0
  226. flextool/process_outputs/out_costs.py +334 -0
  227. flextool/process_outputs/out_flowgroup.py +189 -0
  228. flextool/process_outputs/out_flows.py +301 -0
  229. flextool/process_outputs/out_group.py +475 -0
  230. flextool/process_outputs/out_node.py +190 -0
  231. flextool/process_outputs/persist_realized_slice.py +601 -0
  232. flextool/process_outputs/process_results.py +24 -0
  233. flextool/process_outputs/read_highs_solution.py +2256 -0
  234. flextool/process_outputs/read_parameters.py +1799 -0
  235. flextool/process_outputs/read_sets.py +1095 -0
  236. flextool/process_outputs/read_variables.py +553 -0
  237. flextool/process_outputs/solve_order.py +81 -0
  238. flextool/process_outputs/spinedb_replay.py +412 -0
  239. flextool/process_outputs/union_realized_slice.py +224 -0
  240. flextool/process_outputs/write_outputs.py +1286 -0
  241. flextool/process_outputs/write_spinedb.py +1267 -0
  242. flextool/representative_periods/__init__.py +5 -0
  243. flextool/representative_periods/clustering.py +165 -0
  244. flextool/representative_periods/force_include.py +563 -0
  245. flextool/representative_periods/netload.py +365 -0
  246. flextool/representative_periods/netload_inputs.py +345 -0
  247. flextool/representative_periods/netload_iterate.py +722 -0
  248. flextool/representative_periods/preprocess.py +948 -0
  249. flextool/representative_periods/scenario_stack.py +195 -0
  250. flextool/representative_periods/weights.py +124 -0
  251. flextool/scenario_comparison/__init__.py +13 -0
  252. flextool/scenario_comparison/config_builder.py +158 -0
  253. flextool/scenario_comparison/constants.py +20 -0
  254. flextool/scenario_comparison/data_models.py +222 -0
  255. flextool/scenario_comparison/db_reader.py +399 -0
  256. flextool/scenario_comparison/dispatch_data.py +1002 -0
  257. flextool/scenario_comparison/dispatch_mappings.py +205 -0
  258. flextool/scenario_comparison/dispatch_plots.py +691 -0
  259. flextool/scenario_comparison/input_entity_colors.py +319 -0
  260. flextool/scenario_comparison/orchestrator.py +453 -0
  261. flextool/scenario_comparison/plan_union.py +244 -0
  262. flextool/scenario_comparison/plot_settings_seed.py +205 -0
  263. flextool/schemas/AXIS_CONTRACT.md +71 -0
  264. flextool/schemas/canonical_databases/howto_aggregate_output.json +6225 -0
  265. flextool/schemas/canonical_databases/howto_connections.json +5606 -0
  266. flextool/schemas/canonical_databases/howto_demand.json +5518 -0
  267. flextool/schemas/canonical_databases/howto_hydro_reservoir.json +6239 -0
  268. flextool/schemas/canonical_databases/howto_hydro_reservoir_with_pump.json +5933 -0
  269. flextool/schemas/canonical_databases/howto_non_sync_and_curtailment.json +5794 -0
  270. flextool/schemas/canonical_databases/howto_ramp_and_start_up.json +5707 -0
  271. flextool/schemas/canonical_databases/howto_stochastics.json +6032 -0
  272. flextool/schemas/canonical_databases/templates_examples.json +13532 -0
  273. flextool/schemas/canonical_databases/templates_time_settings_only.json +5340 -0
  274. flextool/schemas/comparison_settings_template.json +197 -0
  275. flextool/schemas/default_plot_settings.yaml +260 -0
  276. flextool/schemas/default_plots.yaml +2293 -0
  277. flextool/schemas/flextool_axis_contract.json +303 -0
  278. flextool/schemas/flextool_axis_contract.schema.json +247 -0
  279. flextool/schemas/old_flextool_import_template.json +4443 -0
  280. flextool/schemas/output_info_template.json +48 -0
  281. flextool/schemas/output_settings_template.json +256 -0
  282. flextool/schemas/pre_v26/flextool_template_constant_default.json +2105 -0
  283. flextool/schemas/pre_v26/flextool_template_default_optional_output.json +2152 -0
  284. flextool/schemas/pre_v26/flextool_template_default_value.json +2094 -0
  285. flextool/schemas/pre_v26/flextool_template_drop_down.json +2080 -0
  286. flextool/schemas/pre_v26/flextool_template_lifetime_method.json +1990 -0
  287. flextool/schemas/pre_v26/flextool_template_optional_outputs.json +2094 -0
  288. flextool/schemas/pre_v26/flextool_template_output_node_flows.json +2105 -0
  289. flextool/schemas/pre_v26/flextool_template_results_master.json +493 -0
  290. flextool/schemas/pre_v26/flextool_template_rolling_start_remove.json +2087 -0
  291. flextool/schemas/pre_v26/flextool_template_rolling_window.json +2059 -0
  292. flextool/schemas/pre_v26/flextool_template_storage_binding_defaults.json +46 -0
  293. flextool/schemas/pre_v26/flextool_template_v2.json +1990 -0
  294. flextool/schemas/pre_v26/flextool_template_v25.json +3864 -0
  295. flextool/schemas/spinedb_results_schema.json +581 -0
  296. flextool/schemas/spinedb_schema.json +4636 -0
  297. flextool/solver_config/copt.opt.template +18 -0
  298. flextool/solver_config/cplex.opt.template +25 -0
  299. flextool/solver_config/gurobi.opt.template +18 -0
  300. flextool/solver_config/highs.opt.template +18 -0
  301. flextool/solver_config/xpress.opt.template +26 -0
  302. flextool/spinedb_backend/__init__.py +26 -0
  303. flextool/spinedb_backend/_axis_enums.py +1119 -0
  304. flextool/spinedb_backend/_backend.py +1139 -0
  305. flextool/update_flextool/__init__.py +12 -0
  306. flextool/update_flextool/canonical_databases.py +251 -0
  307. flextool/update_flextool/db_migration.py +7108 -0
  308. flextool/update_flextool/ensure_settings_db.py +138 -0
  309. flextool/update_flextool/export_database.py +103 -0
  310. flextool/update_flextool/extend_tests_fixture.py +772 -0
  311. flextool/update_flextool/generate_canonical.py +274 -0
  312. flextool/update_flextool/initialize_database.py +42 -0
  313. flextool/update_flextool/install_info.py +225 -0
  314. flextool/update_flextool/self_update.py +464 -0
  315. flextool/update_flextool/sync_master_json_template.py +125 -0
  316. flextool/update_flextool/test_fixtures.py +187 -0
  317. flextool-4.0.0.dist-info/METADATA +217 -0
  318. flextool-4.0.0.dist-info/RECORD +322 -0
  319. flextool-4.0.0.dist-info/WHEEL +5 -0
  320. flextool-4.0.0.dist-info/entry_points.txt +17 -0
  321. flextool-4.0.0.dist-info/licenses/LICENSE.txt +19 -0
  322. flextool-4.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,4314 @@
1
+ """Native polar_high orchestrator — master loop conductor.
2
+
3
+ The orchestrator:
4
+
5
+ * Combines the foundation modules (``_solve_config`` + ``_timeline`` +
6
+ ``_recursive_solve`` + ``_stochastic``).
7
+ * Drives the per-solve preprocessing through the L0-L9 batch +
8
+ ``preprocessing_solve_time`` + ``solve_writers``.
9
+ * Runs the actual solve via ``polar_high.Problem.solve`` (HiGHS).
10
+ * Captures :class:`SolveHandoff` per solve via the native
11
+ :func:`build_handoff_from_solution`, threads it forward as
12
+ ``prior_handoff``, and routes it into the in-memory handoff slot of
13
+ the runner so the consume side (``preprocessing_solve_time``,
14
+ ``handoff_writers``) reads from it.
15
+
16
+ Design choices
17
+ --------------
18
+
19
+ * The orchestrator drives ``_native_run_model.native_run_model`` once
20
+ per top-level invocation, with a **polar_high-as-inner-solver** wrapper
21
+ that:
22
+ - Reads the per-solve snapshot via ``load_flextool``.
23
+ - Builds the LP via ``build_flextool``.
24
+ - Solves via ``polar_high`` (HiGHS).
25
+ - Captures handoff via ``build_handoff_from_solution``.
26
+ - Deposits handoff into ``state.handoffs`` for the next iteration's
27
+ preprocessing.
28
+
29
+ * **Storage-fixing handoff is in-memory by default** when
30
+ ``state.handoffs`` is non-None. The file-copy path
31
+ (``shutil.copy`` of ``solve_data_<parent>/fix_storage_*.csv``) is
32
+ consulted only when ``state.handoffs is None``.
33
+
34
+ * **Roll-counter reset semantics**: every top-level
35
+ :func:`run_orchestration` call invokes
36
+ ``state.solve.roll_counter = state.solve.make_roll_counter()`` first
37
+ so test re-use of the same SolveConfig doesn't desync (R-O5).
38
+
39
+ * **``model_solve`` validation**: enforced loud-and-early. Empty
40
+ ``model_solve`` or more-than-one model raises
41
+ :class:`FlexToolConfigError`.
42
+
43
+ * **``run_chain``** is a thin compat shim that always delegates here.
44
+ """
45
+ from __future__ import annotations
46
+
47
+ import json
48
+ import logging
49
+ import math
50
+ import os
51
+ import re
52
+ import shutil
53
+ import tempfile
54
+ import textwrap
55
+ import threading
56
+ import time
57
+ from dataclasses import dataclass, field
58
+ from pathlib import Path
59
+ from typing import TYPE_CHECKING, Any, Mapping, Sequence
60
+
61
+ from flextool.engine_polars._solve_acceptance import classify_acceptance
62
+ from flextool.engine_polars._solve_handoff import SolveHandoff
63
+ from flextool.engine_polars._solve_state import (
64
+ FlexToolConfigError,
65
+ PathConfig,
66
+ RunnerState,
67
+ )
68
+ from flextool.engine_polars._determinism import (
69
+ DETERMINISM_OPTIONS,
70
+ SIMPLEX_SCALE_STRATEGY_ADVANCED,
71
+ )
72
+ from flextool.engine_polars.autoscale import (
73
+ Layer2Plan as _AutoscaleLayer2Plan,
74
+ Layer3Plan as _AutoscaleLayer3Plan,
75
+ RangeReport as _AutoscaleRangeReport,
76
+ ScalingMode as _AutoscaleScalingMode,
77
+ USER_BOUND_SCALE_MAX as _USER_BOUND_SCALE_MAX,
78
+ USER_BOUND_SCALE_MIN as _USER_BOUND_SCALE_MIN,
79
+ apply_layer2 as _autoscale_apply_layer2,
80
+ apply_layer2_with_exponents as _autoscale_apply_layer2_with_exponents,
81
+ apply_scaling as _autoscale_apply_scaling,
82
+ detect_ranges as _autoscale_compute_ranges,
83
+ format_console_summary as _autoscale_format_console_summary,
84
+ mode_enables_layer1 as _autoscale_mode_enables_layer1,
85
+ mode_enables_layer3 as _autoscale_mode_enables_layer3,
86
+ recommend_scaling as _autoscale_recommend_scaling,
87
+ resolve_scaling_config as _autoscale_resolve_config,
88
+ resolve_user_bound_scale_override as _resolve_user_bound_scale_override,
89
+ unscale_solution as _autoscale_unscale_solution,
90
+ write_report as _autoscale_write_report,
91
+ )
92
+
93
+
94
+ def _wrap_log_prose(text: str, width: int = 100, indent: str = " ") -> str:
95
+ """Wrap a prose log message at ``width`` chars, continuation indented."""
96
+ return textwrap.fill(
97
+ text, width=width, subsequent_indent=indent,
98
+ break_long_words=False, break_on_hyphens=False,
99
+ )
100
+
101
+
102
+ def _format_solver_args_line(
103
+ effective_options: Mapping[str, Any] | None,
104
+ *,
105
+ prefer_first: "Sequence[str]" = (),
106
+ ) -> str | None:
107
+ """Render the effective solver options as a one-liner.
108
+
109
+ Emitted just before HiGHS' own "Running HiGHS …" banner so the run
110
+ log shows exactly the options the solver received — the full merged,
111
+ post-precedence set of everything the operator touched:
112
+ ``solver_config/highs.opt`` (project floor) ∪ the DB-authored
113
+ ``solver_arguments`` ∪ the CLI-flag overrides. Engine-pinned
114
+ *baseline* options (determinism keys + the Curtis-Reid scale) are
115
+ deliberately excluded — they are internal and always present, so
116
+ listing them would bury the operator's intent. Pass
117
+ ``effective_options`` already resolved with ``baseline=None`` so this
118
+ exclusion holds (see :func:`_resolve_effective_highs_options`).
119
+
120
+ *prefer_first* names keys (typically the ``solver_arguments`` keys)
121
+ to surface at the front of the line; the remaining keys follow in
122
+ ``effective_options`` order. Each key's value is the winning one
123
+ after precedence resolution.
124
+
125
+ Returns ``None`` when *effective_options* is empty (nothing the
126
+ operator set → no line).
127
+ """
128
+ if not effective_options:
129
+ return None
130
+ order = [k for k in prefer_first if k in effective_options]
131
+ order += [k for k in effective_options if k not in order]
132
+ body = ", ".join(f"{k}={effective_options[k]}" for k in order)
133
+ return f"Solver args: {body}"
134
+
135
+
136
+ # Legacy ``scale_the_objective`` default — historically the
137
+ # ``flextool_base.dat``'s ``param scale_the_objective default 1e-6`` and the
138
+ # pre-autoscale analyser's fallback when the rough-objective heuristic
139
+ # returned a non-finite value. Phase 2b drops the legacy data-driven
140
+ # analyser entirely; we keep the build-time cost multiplier so that:
141
+ #
142
+ # * The MPS export (post-Layer-2, pre-solve) still carries pre-scaled
143
+ # cost coefficients in the same magnitude as before — handoff to
144
+ # external solvers continues to see a recognisable objective.
145
+ # * The output writers' un-scaling path (``_resolve_inv_scale_the_objective``)
146
+ # continues to find ``solve_data/scale_the_objective.csv`` with a non-1
147
+ # value; missing-file fallback in that helper is also ``1.0 / 1e-6``,
148
+ # so the un-scaling round-trip is byte-stable.
149
+ # * The user can still override per-solve via DB
150
+ # ``solve.scale_the_objective`` — autoscale Layer 3's
151
+ # ``user_objective_scale`` is HiGHS-internal and stacks on top
152
+ # (HiGHS un-scales internally on output, so layering the build-time
153
+ # scale and the HiGHS-side scale is well-defined).
154
+ _LEGACY_DEFAULT_OBJECTIVE_SCALE = 1e-6
155
+
156
+
157
+ def _resolve_effective_obj_scale(user_value: object | None) -> float:
158
+ """Coerce a raw DB ``solve.scale_the_objective`` value to a float.
159
+
160
+ Returns the user value when finite and strictly positive; otherwise
161
+ falls back to :data:`_LEGACY_DEFAULT_OBJECTIVE_SCALE` (1e-6). This
162
+ mirrors the defensive contract the retired
163
+ ``scaling.resolve_effective_scaling`` had: any malformed user value
164
+ (None, non-numeric string, 0, NaN, negative) falls back rather than
165
+ crashing the cascade — the failure mode would be HiGHS's own
166
+ division by zero on output un-scaling.
167
+ """
168
+ if user_value is None:
169
+ return _LEGACY_DEFAULT_OBJECTIVE_SCALE
170
+ try:
171
+ candidate = float(user_value)
172
+ except (TypeError, ValueError):
173
+ return _LEGACY_DEFAULT_OBJECTIVE_SCALE
174
+ if not math.isfinite(candidate) or candidate <= 0.0:
175
+ return _LEGACY_DEFAULT_OBJECTIVE_SCALE
176
+ return candidate
177
+
178
+
179
+ def _baseline_highs_options(
180
+ *,
181
+ user_bound_scale_override: int | None = None,
182
+ scaling_mode: "_AutoscaleScalingMode | None" = None,
183
+ ) -> dict[str, object]:
184
+ """Build the base HiGHS solver-option dict (determinism + matrix scale).
185
+
186
+ Replaces the retired ``scaling.recommended_highs_options`` helper.
187
+ Sets:
188
+
189
+ * ``simplex_scale_strategy`` — defaults to
190
+ :data:`SIMPLEX_SCALE_STRATEGY_ADVANCED` (Curtis-Reid matrix
191
+ equilibration). When ``scaling_mode == ScalingMode.OFF`` we force
192
+ ``simplex_scale_strategy=0`` so HiGHS' own equilibration is also
193
+ disabled — that's the only mode in which FlexTool touches the
194
+ HiGHS-internal scaling knob. Layer 3's :func:`apply_scaling` may
195
+ re-assert this value on the cold path; on warm rebuilds the value
196
+ set here is the authoritative one.
197
+ * :data:`DETERMINISM_OPTIONS` — ``random_seed`` / ``parallel`` /
198
+ ``solver`` / ``presolve`` pins for byte-deterministic LP solutions.
199
+ * ``user_bound_scale`` — only when ``user_bound_scale_override`` is
200
+ a non-zero integer (CLI ``--user-bound-scale N`` / DB
201
+ ``solve.user_bound_scale``). Clamped to the HiGHS-safe range
202
+ ``[USER_BOUND_SCALE_MIN, USER_BOUND_SCALE_MAX]``. When unset, the
203
+ autoscaler's Layer 3 may still emit its own value based on the
204
+ coefficient ranges.
205
+
206
+ ``user_cost_scale`` is intentionally NOT set — costs are already
207
+ multiplied by ``scale_the_objective`` inside ``build_flextool``, and
208
+ Layer 3 may add ``user_objective_scale`` on top; layering a third
209
+ cost-side knob would compound confusingly.
210
+ """
211
+ if scaling_mode is _AutoscaleScalingMode.OFF:
212
+ # OFF mode: disable HiGHS' internal equilibration too. ``parallel``
213
+ # / ``random_seed`` / ``solver`` / ``presolve`` pins from
214
+ # DETERMINISM_OPTIONS still apply — OFF is about scaling, not
215
+ # determinism.
216
+ simplex_scale = 0
217
+ else:
218
+ simplex_scale = SIMPLEX_SCALE_STRATEGY_ADVANCED
219
+ options: dict[str, object] = {
220
+ "simplex_scale_strategy": simplex_scale,
221
+ **DETERMINISM_OPTIONS,
222
+ }
223
+ if user_bound_scale_override is not None and user_bound_scale_override != 0:
224
+ n = int(user_bound_scale_override)
225
+ if n > _USER_BOUND_SCALE_MAX:
226
+ n = _USER_BOUND_SCALE_MAX
227
+ if n < _USER_BOUND_SCALE_MIN:
228
+ n = _USER_BOUND_SCALE_MIN
229
+ options["user_bound_scale"] = n
230
+ return options
231
+
232
+
233
+ def _autoscale_emit_layer1(
234
+ sol: "Solution | None",
235
+ *,
236
+ solve_name: str,
237
+ logger: logging.Logger,
238
+ work_folder: str | os.PathLike | None,
239
+ layer2_plan: "_AutoscaleLayer2Plan | None" = None,
240
+ layer3_plan: "_AutoscaleLayer3Plan | None" = None,
241
+ ) -> None:
242
+ """Layer 1 (detect) post-solve emitter.
243
+
244
+ Reads polar-high's already-computed ``Solution.streamed_lp_ranges``
245
+ (no duplicate matrix walk), runs the autoscaler's Layer 1
246
+ detection, and logs the four ranges + trigger flag. Optionally
247
+ writes a YAML audit report when the config carries a path.
248
+
249
+ Phase 1b is detection-only — this function does NOT modify the LP
250
+ or the solve options. Layer 2 / Layer 3 will hook in alongside
251
+ it in later phases.
252
+
253
+ Failures here are non-fatal: a missing ``streamed_lp_ranges`` (the
254
+ subprocess path — polar-high streams ranges during the in-process
255
+ solve, which the subprocess child does not share back) skips the
256
+ layer with a debug-level note rather than breaking the solve.
257
+ """
258
+ # ``cli_args=None`` is intentional: the CLI surface
259
+ # (``cmd_run_flextool``) mirrors ``--scaling`` /
260
+ # ``--user-bound-scale`` into the ``FLEXTOOL_SCALING`` /
261
+ # ``FLEXTOOL_USER_BOUND_SCALE`` env vars before invoking the
262
+ # orchestrator, matching the existing env-threading convention
263
+ # documented on the ``run_chain_from_db`` call site. Cascade-
264
+ # internal hops therefore observe operator intent without
265
+ # plumbing the parsed ``args`` namespace through every helper.
266
+ cfg = _autoscale_resolve_config(None)
267
+ if not _autoscale_mode_enables_layer1(cfg.mode):
268
+ return
269
+ if sol is None:
270
+ return
271
+ streamed = getattr(sol, "streamed_lp_ranges", None)
272
+ if not isinstance(streamed, dict):
273
+ logger.debug(
274
+ "autoscale Layer 1 skipped for %s: no streamed_lp_ranges on Solution",
275
+ solve_name,
276
+ )
277
+ return
278
+
279
+ try:
280
+ ranges = _autoscale_compute_ranges(sol, cfg)
281
+ except Exception: # pragma: no cover — guard against future API drift
282
+ logger.exception(
283
+ "autoscale Layer 1 failed for %s; continuing without it", solve_name,
284
+ )
285
+ return
286
+
287
+ def _fmt(span: tuple[float, float]) -> str:
288
+ import math as _math
289
+ if _math.isnan(span[0]) or _math.isnan(span[1]):
290
+ return "empty"
291
+ return f"{span[0]:.1e}, {span[1]:.1e}"
292
+
293
+ logger.info(
294
+ "autoscale Layer 1 [%s]: Matrix [%s], Cost [%s], Bound [%s], "
295
+ "RHS [%s], cross=%s, trigger=%s",
296
+ solve_name,
297
+ _fmt(ranges.matrix), _fmt(ranges.cost),
298
+ _fmt(ranges.bound), _fmt(ranges.rhs),
299
+ (f"{ranges.cross_group_max_ratio:.1e}"
300
+ if ranges.cross_group_max_ratio == ranges.cross_group_max_ratio
301
+ else "n/a"),
302
+ ranges.trigger,
303
+ )
304
+
305
+ # Default report location next to the existing scaling_report file
306
+ # when no explicit path is configured — keeps both diagnostics
307
+ # together for the operator.
308
+ yaml_path = cfg.report_yaml_path
309
+ if yaml_path is None and work_folder is not None:
310
+ yaml_path = Path(work_folder) / "solve_data" / f"autoscale_{solve_name}.yaml"
311
+ if yaml_path is not None:
312
+ try:
313
+ report_tree: dict = {"layer1": ranges}
314
+ if layer2_plan is not None:
315
+ from flextool.engine_polars.autoscale._report import (
316
+ render_layer2 as _render_l2,
317
+ )
318
+ report_tree["layer2"] = _render_l2(layer2_plan)
319
+ if layer3_plan is not None:
320
+ from flextool.engine_polars.autoscale._report import (
321
+ render_layer3 as _render_l3,
322
+ )
323
+ report_tree["layer3"] = _render_l3(layer3_plan)
324
+ _autoscale_write_report(report_tree, yaml_path)
325
+ except Exception: # pragma: no cover — non-fatal
326
+ logger.exception(
327
+ "autoscale Layer 1 report write failed (%s)", yaml_path,
328
+ )
329
+
330
+
331
+ def _autoscale_apply_layer3_pre_solve(
332
+ pb: "Problem",
333
+ *,
334
+ layer2_plan: "_AutoscaleLayer2Plan | None",
335
+ solve_name: str,
336
+ logger: logging.Logger,
337
+ ) -> "_AutoscaleLayer3Plan | None":
338
+ """Layer 3 (HiGHS-native top-up) pre-solve apply.
339
+
340
+ Runs unconditionally when the autoscaler is enabled — Layer 3 is
341
+ cheap and sets HiGHS options that take effect only when needed
342
+ (``user_*_scale`` defaults to 0 = no-op). The recommendation is
343
+ derived from the *post-Layer-2* coefficient ranges so the residual
344
+ spread (after Layer 2's per-type rescale) drives the global
345
+ HiGHS-side scaling.
346
+
347
+ Returns ``None`` when the autoscaler is disabled or the readout
348
+ fails; the caller continues without setting any Layer 3 options.
349
+
350
+ Layer 2's mutation of the LP arrays is the same the polar-high
351
+ streaming solve sees — Layer 3 just picks exponents from those
352
+ arrays. Precedence-respect: when the caller (highs.opt file,
353
+ ``set_solver_options`` from elsewhere) has already set
354
+ ``user_bound_scale`` or ``user_objective_scale``, Layer 3 skips that
355
+ axis and the caller's value wins.
356
+ """
357
+ cfg = _autoscale_resolve_config(None)
358
+ if not _autoscale_mode_enables_layer3(cfg.mode):
359
+ return None
360
+ try:
361
+ ranges_post_l2 = _autoscale_compute_ranges(pb, cfg)
362
+ except Exception: # pragma: no cover — guard against future API drift
363
+ logger.exception(
364
+ "autoscale Layer 3 pre-solve range readout failed for %s; "
365
+ "skipping Layer 3 (Layer 1 / Layer 2 still applied)",
366
+ solve_name,
367
+ )
368
+ return None
369
+ try:
370
+ plan = _autoscale_recommend_scaling(ranges_post_l2, cfg, problem=pb)
371
+ except Exception: # pragma: no cover
372
+ logger.exception(
373
+ "autoscale Layer 3 recommendation failed for %s; "
374
+ "skipping (HiGHS internal scaling still applies)",
375
+ solve_name,
376
+ )
377
+ return None
378
+ try:
379
+ _autoscale_apply_scaling(pb, plan)
380
+ except Exception: # pragma: no cover
381
+ logger.exception(
382
+ "autoscale Layer 3 option apply failed for %s; HiGHS "
383
+ "internal scaling will fill in",
384
+ solve_name,
385
+ )
386
+ return plan # still return for report visibility
387
+ logger.info(
388
+ "autoscale Layer 3 [%s]: user_objective_scale=%d, "
389
+ "user_bound_scale=%d, simplex_scale_strategy=%d (%s)",
390
+ solve_name,
391
+ plan.user_objective_scale,
392
+ plan.user_bound_scale,
393
+ plan.simplex_scale_strategy,
394
+ plan.reasoning,
395
+ )
396
+ return plan
397
+
398
+
399
+ def _autoscale_apply_layer2_pre_solve(
400
+ pb: "Problem",
401
+ *,
402
+ solve_name: str,
403
+ logger: logging.Logger,
404
+ ) -> "tuple[_AutoscaleLayer2Plan | None, _AutoscaleRangeReport | None]":
405
+ """Layer 2 (semantic per-type scaling) pre-solve apply.
406
+
407
+ Runs only when the autoscaler is enabled AND the pre-solve Layer-1
408
+ detector trips (per ``AutoScaleConfig.threshold_decades``). When
409
+ triggered, mutates ``pb`` in place and returns the inverse-plan
410
+ that ``_autoscale_unscale_post_solve`` consumes immediately after
411
+ ``pb.solve(...)``.
412
+
413
+ Returns ``(plan, ranges_pre)`` where:
414
+
415
+ * ``plan`` is ``None`` when Layer 2 was skipped (config off or the
416
+ Layer-1 trigger did not fire).
417
+ * ``ranges_pre`` is the pre-Layer-2 :class:`RangeReport` (always
418
+ present when the autoscaler is enabled and the readout succeeded;
419
+ ``None`` only when disabled or the readout itself failed). The
420
+ caller threads it into the console summary and the non-optimal
421
+ hint so both reports describe the LP the autoscaler *decided on*
422
+ rather than re-reading after Layer 2 mutated the arrays.
423
+
424
+ The detector here uses :func:`_autoscale_compute_ranges` on the
425
+ pre-solve ``Problem`` — :mod:`_ranges` falls back to
426
+ ``Problem._build_lp_arrays`` for that path. Layer-1 emission
427
+ after solve still happens via the existing post-solve hook so the
428
+ operator-facing log line and YAML report describe the *scaled*
429
+ LP that HiGHS actually saw.
430
+ """
431
+ cfg = _autoscale_resolve_config(None)
432
+ # Layer 2 is FlexTool-side semantic per-quantity scaling: only fires
433
+ # in ``ScalingMode.FULL``. In ``BASIC`` we still want Layer 1's
434
+ # pre-solve ranges to be available for the console summary, so we
435
+ # compute them when Layer 3 is enabled too.
436
+ if not _autoscale_mode_enables_layer1(cfg.mode):
437
+ return None, None
438
+ try:
439
+ ranges_pre = _autoscale_compute_ranges(pb, cfg)
440
+ except Exception: # pragma: no cover — guard against future API drift
441
+ if os.environ.get("FLEXTOOL_AUTOSCALE_STRICT") == "1":
442
+ raise
443
+ logger.exception(
444
+ "autoscale Layer 2 pre-solve range readout failed for %s; "
445
+ "skipping Layer 2 (Layer 1 post-solve still fires)",
446
+ solve_name,
447
+ )
448
+ return None, None
449
+ if cfg.mode is not _AutoscaleScalingMode.FULL:
450
+ # BASIC mode: skip Layer 2's LP-array mutation but still report
451
+ # the pre-solve ranges so Layer 3 / the console summary see them.
452
+ return None, ranges_pre
453
+ if not ranges_pre.trigger:
454
+ return None, ranges_pre
455
+ try:
456
+ plan = _autoscale_apply_layer2(pb, cfg)
457
+ except Exception: # pragma: no cover
458
+ if os.environ.get("FLEXTOOL_AUTOSCALE_STRICT") == "1":
459
+ # Opt-in fail-fast for tests / CI: surface Layer-2 errors
460
+ # (notably autoscale-registry gaps — an unregistered constraint
461
+ # / variable / parameter raising KeyError in
462
+ # bucket_coefficients) loudly instead of silently reverting to
463
+ # an un-scaled LP. This is the behavioral backstop for dynamic
464
+ # (f-string) constraint names that the static-literal grep in
465
+ # test_registry_coverage cannot see (e.g. ramp_*_constraint).
466
+ # Production runs leave the flag unset and keep degrading
467
+ # gracefully so a registry gap never blocks a user's solve.
468
+ raise
469
+ logger.exception(
470
+ "autoscale Layer 2 apply failed for %s; reverting solve to "
471
+ "un-scaled LP (Layer 1 post-solve still fires)",
472
+ solve_name,
473
+ )
474
+ return None, ranges_pre
475
+ logger.info(
476
+ "autoscale Layer 2 [%s]: exponents=%s, rows=%d, skipped_rows=%d, "
477
+ "integer_cols=%d",
478
+ solve_name,
479
+ {t.value: e for t, e in plan.type_exponents.items()},
480
+ plan.row_factors.shape[0],
481
+ len(plan.skipped_rows),
482
+ len(plan.skipped_integer_cols),
483
+ )
484
+ return plan, ranges_pre
485
+
486
+
487
+ @dataclass
488
+ class _AutoscaleShapeCacheEntry:
489
+ """Cached autoscale DECISION for one structural fingerprint.
490
+
491
+ Rolling dispatch solves that share the same structural fingerprint
492
+ (``_warm._fingerprint``) emit an LP of identical shape, so the
493
+ autoscaler's per-layer DECISION is invariant across rolls:
494
+
495
+ * ``layer2_exponents`` — the per-:class:`QuantityType` power-of-two
496
+ exponents Layer 2 chose on the first solve of this shape. ``None``
497
+ when Layer 2 did not trigger (or BASIC/OFF mode). Replaying these
498
+ via :func:`apply_layer2_with_exponents` reinstalls byte-identical
499
+ side vectors on a subsequent roll's freshly-built Problem WITHOUT
500
+ re-walking coefficients.
501
+ * ``layer3_plan`` — the :class:`Layer3Plan` (``user_*_scale`` etc.)
502
+ Layer 3 recommended. Re-applied verbatim via
503
+ :func:`apply_scaling` (cheap option-set, no range walk). ``None``
504
+ when Layer 3 was disabled or its readout failed.
505
+ * ``ranges_pre`` / ``ranges_post`` — the pre-/post-Layer-2
506
+ :class:`RangeReport`s, cached so the per-roll Layer-1 YAML emit and
507
+ the (deduped) console summary keep their range context without a
508
+ re-walk.
509
+
510
+ The KEY property: a cache HIT re-applies all of the above WITHOUT a
511
+ single :func:`detect_ranges` / ``bucket_coefficients`` traversal,
512
+ which is where the per-roll multi-GB ``priv_dirty`` spikes came from.
513
+ """
514
+
515
+ layer2_exponents: "dict | None"
516
+ layer2_buckets_before: "dict"
517
+ layer2_buckets_after: "dict"
518
+ layer3_plan: "_AutoscaleLayer3Plan | None"
519
+ ranges_pre: "_AutoscaleRangeReport | None"
520
+ ranges_post: "_AutoscaleRangeReport | None"
521
+
522
+
523
+ def _autoscale_disable_cache() -> bool:
524
+ """True when ``FLEXTOOL_DISABLE_AUTOSCALE_CACHE=1`` — always recompute
525
+ (the pre-cache, per-roll-traversal behaviour)."""
526
+ return os.environ.get("FLEXTOOL_DISABLE_AUTOSCALE_CACHE") == "1"
527
+
528
+
529
+ def _autoscale_apply_layer2_from_cache(
530
+ pb: "Problem",
531
+ entry: "_AutoscaleShapeCacheEntry",
532
+ *,
533
+ solve_name: str,
534
+ logger: logging.Logger,
535
+ ) -> "_AutoscaleLayer2Plan | None":
536
+ """Cache-HIT Layer-2 re-apply — NO coefficient walk.
537
+
538
+ Reinstalls the cached per-type exponents onto THIS roll's Problem via
539
+ :func:`apply_layer2_with_exponents` (O(#families); zero
540
+ ``detect_ranges`` / ``bucket_coefficients``). Returns the replayed
541
+ :class:`Layer2Plan` (needed by :func:`_autoscale_unscale_post_solve`)
542
+ or ``None`` when Layer 2 did not trigger for this shape.
543
+ """
544
+ if entry.layer2_exponents is None:
545
+ return None
546
+ try:
547
+ plan = _autoscale_apply_layer2_with_exponents(
548
+ pb,
549
+ entry.layer2_exponents,
550
+ type_buckets_before=entry.layer2_buckets_before,
551
+ type_buckets_after=entry.layer2_buckets_after,
552
+ )
553
+ except Exception: # pragma: no cover — guard against API drift
554
+ if os.environ.get("FLEXTOOL_AUTOSCALE_STRICT") == "1":
555
+ raise
556
+ logger.exception(
557
+ "autoscale Layer 2 cached replay failed for %s; reverting to "
558
+ "un-scaled LP for this roll",
559
+ solve_name,
560
+ )
561
+ return None
562
+ logger.debug(
563
+ "autoscale Layer 2 [%s]: replayed cached exponents=%s (no range walk)",
564
+ solve_name,
565
+ {t.value: e for t, e in plan.type_exponents.items()},
566
+ )
567
+ return plan
568
+
569
+
570
+ def _autoscale_apply_layer3_from_cache(
571
+ pb: "Problem",
572
+ entry: "_AutoscaleShapeCacheEntry",
573
+ *,
574
+ solve_name: str,
575
+ logger: logging.Logger,
576
+ ) -> "_AutoscaleLayer3Plan | None":
577
+ """Cache-HIT Layer-3 re-apply — NO post-Layer-2 range walk.
578
+
579
+ Re-applies the cached :class:`Layer3Plan`'s HiGHS options to THIS
580
+ roll's Problem via :func:`apply_scaling` (a plain ``set_solver_options``
581
+ merge). Returns the cached plan for report visibility, or ``None``
582
+ when Layer 3 produced no plan on the first solve of this shape.
583
+ """
584
+ plan = entry.layer3_plan
585
+ if plan is None:
586
+ return None
587
+ try:
588
+ _autoscale_apply_scaling(pb, plan)
589
+ except Exception: # pragma: no cover
590
+ logger.exception(
591
+ "autoscale Layer 3 cached option apply failed for %s; HiGHS "
592
+ "internal scaling will fill in",
593
+ solve_name,
594
+ )
595
+ return plan
596
+ logger.debug(
597
+ "autoscale Layer 3 [%s]: replayed cached user_objective_scale=%d, "
598
+ "user_bound_scale=%d (no range walk)",
599
+ solve_name,
600
+ plan.user_objective_scale,
601
+ plan.user_bound_scale,
602
+ )
603
+ return plan
604
+
605
+
606
+ def _autoscale_lp_shape_signature(pb: "Problem", base_solve_name: str) -> tuple:
607
+ """Structural signature of a BUILT Problem for the autoscale cache key.
608
+
609
+ Invariant across rolls of one rolling solve (same matrix shape +
610
+ family layout) but distinct for genuinely different LPs. Cheap —
611
+ O(#var families + #cstr families), NO coefficient walk. Scoped by
612
+ ``base_solve_name`` so only rolls of the SAME named rolling solve can
613
+ share a cached scaling decision (guards against a same-shape /
614
+ different-magnitude collision between unrelated solves).
615
+ """
616
+ var_sig = tuple(sorted(
617
+ (name, int(v.frame.height), bool(v.integer))
618
+ for name, v in pb._vars.items()
619
+ ))
620
+ cstr_sig = tuple(
621
+ (cname, 1 if over is None else int(over.height))
622
+ for cname, _proto, over in pb._cstrs
623
+ )
624
+ return (base_solve_name, int(pb._next_col), var_sig, cstr_sig)
625
+
626
+
627
+ def _basis_cache_active(decomposition: "str | None", solver_name: str) -> bool:
628
+ """Gate for the Phase-4 in-process warm-start basis cache.
629
+
630
+ True only when (a) the operator opted in with ``FLEXTOOL_WARM_START=1``,
631
+ (b) the in-process solver is HiGHS — the only solver whose live handle
632
+ supports ``setBasis`` / ``getBasis`` for basis transfer, and (c) the
633
+ active solve is NOT Benders-decomposed. A Benders solve never builds a
634
+ monolithic ``WarmProblem`` (it returns early through
635
+ ``_run_benders_solve`` before the warm branch), and its region
636
+ subproblems are not keyed by this cache; the ``decomposition`` check is
637
+ a belt-and-suspenders guard on top of that early return.
638
+
639
+ When this returns False the caller does nothing new, so the off-path
640
+ (and every non-HiGHS / Benders path) stays byte-identical.
641
+ """
642
+ return (
643
+ os.environ.get("FLEXTOOL_WARM_START") == "1"
644
+ and solver_name == "highs"
645
+ and decomposition != "benders"
646
+ )
647
+
648
+
649
+ def _basis_cache_dir(work_folder: "Path | str | None") -> Path:
650
+ """Resolve the shared warm-start basis cache directory (Phase-2 pattern).
651
+
652
+ ``FLEXTOOL_BASIS_CACHE_DIR`` env > ``<work_folder>/basis_cache`` >
653
+ ``<tmp>/flextool_basis_cache``. Mirrors the resolution used by the
654
+ subprocess ``.bas`` arm in :mod:`_subprocess_solve` so the in-process
655
+ ``.nbasis`` files land alongside their ``.bas`` counterparts. Creates
656
+ the directory (``parents=True, exist_ok=True``).
657
+ """
658
+ cache_env = os.environ.get("FLEXTOOL_BASIS_CACHE_DIR")
659
+ if cache_env:
660
+ cache_dir = Path(cache_env)
661
+ elif work_folder is not None:
662
+ cache_dir = Path(work_folder) / "basis_cache"
663
+ else:
664
+ cache_dir = Path(tempfile.gettempdir()) / "flextool_basis_cache"
665
+ cache_dir.mkdir(parents=True, exist_ok=True)
666
+ return cache_dir
667
+
668
+
669
+ def _autoscale_emit_console_summary(
670
+ *,
671
+ ranges_pre: "_AutoscaleRangeReport | None",
672
+ ranges_post: "_AutoscaleRangeReport | None",
673
+ layer2_plan: "_AutoscaleLayer2Plan | None",
674
+ layer3_plan: "_AutoscaleLayer3Plan | None",
675
+ solve_name: str,
676
+ already_emitted: set[str],
677
+ memrec: "_MemoryRecorder | None" = None,
678
+ logger: logging.Logger | None = None,
679
+ ) -> None:
680
+ """Emit the one-line user-visible autoscale summary.
681
+
682
+ Uses ``print(...)`` rather than ``logger.info`` so the line surfaces
683
+ at the default log level in the same stream where FlexTool's other
684
+ phase-progress lines (``Input: …``, the HiGHS banner) appear. We
685
+ de-duplicate by solve name so a rolling solve emits the line once
686
+ per base-solve, not once per roll — Layer 1/2/3 decisions are
687
+ identical across rolls of the same base solve when the autoscaler
688
+ is enabled.
689
+
690
+ When ``memrec`` is supplied, a ``polar-high scaling`` phase-progress
691
+ row is emitted immediately after the summary so the log carries a
692
+ timestamp/memory checkpoint at the moment scaling finishes and HiGHS
693
+ is about to run (the long, output-silent solve sits right after it).
694
+ The checkpoint is gated by the same dedup as the summary, so it fires
695
+ once per base solve — exactly when the summary text is printed.
696
+ """
697
+ cfg = _autoscale_resolve_config(None)
698
+ if not _autoscale_mode_enables_layer1(cfg.mode):
699
+ return
700
+ if ranges_pre is None:
701
+ return
702
+ if solve_name in already_emitted:
703
+ return
704
+ line = _autoscale_format_console_summary(
705
+ ranges_pre=ranges_pre,
706
+ ranges_post=ranges_post,
707
+ layer2_plan=layer2_plan,
708
+ layer3_plan=layer3_plan,
709
+ threshold_decades=cfg.threshold_decades,
710
+ )
711
+ print(line, flush=True)
712
+ if memrec is not None:
713
+ try:
714
+ memrec.checkpoint(
715
+ "polar_high_scaling", logger, user_label="polar-high scaling",
716
+ )
717
+ except Exception:
718
+ pass
719
+ already_emitted.add(solve_name)
720
+
721
+
722
+
723
+ def _autoscale_unscale_post_solve(
724
+ sol: "Solution | None",
725
+ plan: "_AutoscaleLayer2Plan | None",
726
+ *,
727
+ solve_name: str,
728
+ logger: logging.Logger,
729
+ ) -> None:
730
+ """Layer 2 unscale guard — invoke immediately after ``pb.solve(...)``.
731
+
732
+ Idempotent when ``plan is None`` or ``sol is None``: the eager
733
+ unscale keeps the rest of the pipeline (output writers, range
734
+ re-readouts) blind to the Layer-2 substitution.
735
+ """
736
+ if plan is None or sol is None:
737
+ return
738
+ try:
739
+ _autoscale_unscale_solution(sol, plan)
740
+ except Exception: # pragma: no cover
741
+ logger.exception(
742
+ "autoscale Layer 2 unscale failed for %s; downstream output "
743
+ "values are still in scaled units — re-run with autoscale off",
744
+ solve_name,
745
+ )
746
+
747
+
748
+ if TYPE_CHECKING:
749
+ from collections.abc import Callable
750
+
751
+ import polars as pl
752
+ from polar_high import Problem, Solution
753
+
754
+ from flextool.engine_polars.input import FlexData
755
+
756
+
757
+ # ---------------------------------------------------------------------------
758
+ # Opt-in memory diagnostics
759
+ # ---------------------------------------------------------------------------
760
+
761
+
762
+ # Whitelist of ``user_label`` strings that are emitted to the log in
763
+ # regular (non-verbose) mode. Set ``FLEXTOOL_MEMORY_VERBOSE=1`` to emit
764
+ # every checkpoint (the full pre-cleanup trace).
765
+ #
766
+ # ``Solve start: <name>, k/N`` markers are emitted as plain text lines
767
+ # (no checkpoint), so they don't appear here.
768
+ _MEM_WHITELIST_LABELS: frozenset[str] = frozenset({
769
+ "Run start",
770
+ "Inputs prepared",
771
+ "FlexData built",
772
+ "Matrix built by polar-high",
773
+ "polar-high scaling",
774
+ "Solver",
775
+ "Outputs written",
776
+ "Solve cleanup",
777
+ })
778
+
779
+
780
+ def _is_whitelisted_mem_label(user_label: str | None) -> bool:
781
+ """True when ``user_label`` matches a whitelisted phase label."""
782
+ if not user_label:
783
+ return False
784
+ return user_label in _MEM_WHITELIST_LABELS
785
+
786
+
787
+ class _MemoryRecorder:
788
+ """Opt-in tracemalloc + RSS checkpoint recorder.
789
+
790
+ Activated by ``FLEXTOOL_MEMORY_DIAGNOSTICS=1``. When the env var is
791
+ not set, callers should construct :class:`_NoopMemoryRecorder`
792
+ instead (or simply skip construction); :meth:`checkpoint` here is the
793
+ hot-path no-op fallback only when ``enabled`` is False.
794
+
795
+ Each :meth:`checkpoint` call appends one row to
796
+ ``<work_folder>/solve_data/memory_diagnostics.csv`` with schema::
797
+
798
+ checkpoint,t_elapsed_s,traced_current_mb,traced_peak_mb,rss_mb
799
+
800
+ The file is open/append/closed per row (same atomicity pattern as
801
+ :class:`flextool.cli._timing.TimingRecorder.record`)
802
+ so a crash mid-cascade still leaves a parseable trail.
803
+
804
+ A one-liner is also logged at INFO level via the supplied logger so
805
+ progress is visible in stdout even when the GUI buffers.
806
+
807
+ ``tracemalloc.start()`` is invoked lazily on first checkpoint to keep
808
+ the cost localised to instrumented runs.
809
+ """
810
+
811
+ _HEADER = (
812
+ "checkpoint",
813
+ "t_elapsed_s",
814
+ "traced_current_mb",
815
+ "traced_peak_mb",
816
+ "rss_mb",
817
+ )
818
+
819
+ def __init__(self, csv_path: Path | None = None,
820
+ enabled: bool = True,
821
+ verbose: bool = True) -> None:
822
+ """Construct a phase-progress recorder.
823
+
824
+ Parameters
825
+ ----------
826
+ csv_path
827
+ Where to write the per-checkpoint CSV. ``None`` skips CSV
828
+ emission (verbose log lines still fire).
829
+ enabled
830
+ Full diagnostic mode — starts tracemalloc on first checkpoint
831
+ so ``traced_peak`` becomes meaningful, and writes the CSV.
832
+ When ``False`` we still emit human-readable log lines with
833
+ RSS + section time + Δrss (RSS reads from ``/proc`` are
834
+ essentially free); ``peak`` shows as ``-`` since tracemalloc
835
+ isn't running.
836
+ verbose
837
+ Emit log lines (one per checkpoint). Set ``False`` only if
838
+ you want a fully silent recorder (rare; debugging).
839
+ """
840
+ self.enabled = enabled
841
+ self.verbose = verbose
842
+ self.t0 = time.perf_counter()
843
+ self._t_prev = self.t0
844
+ self._rss_prev_mb: float = 0.0
845
+ self._peak_prev_mb: float = 0.0
846
+ self._sys_prev_mb: float = 0.0
847
+ self._swap_prev_mb: float = 0.0
848
+ self._header_emitted: bool = False
849
+ self._path = Path(csv_path) if csv_path is not None else None
850
+ self._started = False
851
+ # FLEXTOOL_PYRAMID_PROFILE=1: previous RSS for per-batch delta.
852
+ self._pyramid_prev_rss_mb: float = 0.0
853
+ if self.enabled and self._path is not None:
854
+ self._path.parent.mkdir(parents=True, exist_ok=True)
855
+ import csv as _csv
856
+ with open(self._path, "w", newline="") as f:
857
+ _csv.writer(f).writerow(self._HEADER)
858
+
859
+ @staticmethod
860
+ def _read_rss_mb() -> float:
861
+ """Read committed memory (anon RSS + swap, in MB) from
862
+ ``/proc/self/status``.
863
+
864
+ We deliberately don't report ``VmRSS`` (= ``RssAnon`` +
865
+ ``RssFile`` + ``RssShmem``) because file-backed pages are
866
+ evictable cache from the kernel's POV and don't reflect the
867
+ process's true memory commitment. ``RssAnon`` (anonymous
868
+ resident, i.e. heap + private mappings) plus ``VmSwap`` (the
869
+ same anonymous pages that have been swapped out) gives the
870
+ right picture of "memory this process actually needs" —
871
+ what systemd-oomd's PSI signal effectively responds to, and
872
+ what tracks the system monitor's "Used" number more closely
873
+ than raw ``VmRSS``.
874
+
875
+ Returns 0.0 if /proc isn't available (non-Linux) or the
876
+ relevant lines aren't found. We never want diagnostics to
877
+ raise.
878
+ """
879
+ anon_kb = 0.0
880
+ swap_kb = 0.0
881
+ try:
882
+ with open("/proc/self/status", "r") as f:
883
+ for line in f:
884
+ if line.startswith("RssAnon:"):
885
+ parts = line.split()
886
+ if len(parts) >= 2:
887
+ anon_kb = float(parts[1])
888
+ elif line.startswith("VmSwap:"):
889
+ parts = line.split()
890
+ if len(parts) >= 2:
891
+ swap_kb = float(parts[1])
892
+ except OSError:
893
+ pass
894
+ return (anon_kb + swap_kb) / 1024.0
895
+
896
+ @staticmethod
897
+ def _read_sys_swap_used_mb() -> float:
898
+ """System-level used swap (MB) = ``SwapTotal - SwapFree`` from
899
+ ``/proc/meminfo``. Returns 0.0 when there's no swap configured
900
+ (``SwapTotal == 0``) or ``/proc`` isn't available.
901
+ """
902
+ total_kb = 0.0
903
+ free_kb = 0.0
904
+ try:
905
+ with open("/proc/meminfo", "r") as f:
906
+ for line in f:
907
+ if line.startswith("SwapTotal:"):
908
+ parts = line.split()
909
+ if len(parts) >= 2:
910
+ total_kb = float(parts[1])
911
+ elif line.startswith("SwapFree:"):
912
+ parts = line.split()
913
+ if len(parts) >= 2:
914
+ free_kb = float(parts[1])
915
+ if total_kb and free_kb:
916
+ break
917
+ except OSError:
918
+ pass
919
+ if total_kb <= 0:
920
+ return 0.0
921
+ return max(0.0, (total_kb - free_kb) / 1024.0)
922
+
923
+ @staticmethod
924
+ def _read_sys_used_mb() -> float:
925
+ """System-level used memory (MB), matching what most monitors
926
+ ("htop", KSysGuard, GNOME) show as "Used".
927
+
928
+ Computed as ``MemTotal - MemAvailable`` from ``/proc/meminfo``.
929
+ ``MemAvailable`` is the kernel's own estimate of how much
930
+ memory could be allocated to a new process without swapping —
931
+ it already accounts for evictable page-cache + reclaimable
932
+ slab + lazily-freed anonymous pages (MADV_FREE). Subtracting
933
+ from total gives the closest single-number match to what the
934
+ user sees in their desktop's system monitor.
935
+
936
+ Includes contributions from every process on the host, not
937
+ just this one — which is the right metric for desktop-crash
938
+ awareness (the desktop crashes when total ``MemAvailable``
939
+ approaches zero, regardless of which process is consuming the
940
+ pages).
941
+
942
+ Returns 0.0 if /proc isn't available.
943
+ """
944
+ total_kb = 0.0
945
+ avail_kb = 0.0
946
+ try:
947
+ with open("/proc/meminfo", "r") as f:
948
+ for line in f:
949
+ if line.startswith("MemTotal:"):
950
+ parts = line.split()
951
+ if len(parts) >= 2:
952
+ total_kb = float(parts[1])
953
+ elif line.startswith("MemAvailable:"):
954
+ parts = line.split()
955
+ if len(parts) >= 2:
956
+ avail_kb = float(parts[1])
957
+ if total_kb and avail_kb:
958
+ break
959
+ except OSError:
960
+ pass
961
+ if total_kb <= 0:
962
+ return 0.0
963
+ return (total_kb - avail_kb) / 1024.0
964
+
965
+ # Fixed widths used to align the mem output into a table.
966
+ # Label column fits the longest whitelisted label
967
+ # ("Matrix built by polar-high", 26 chars).
968
+ _LABEL_W = 28 # label column (left-aligned)
969
+ # Each data cell renders "<absolute> (<+/-delta>)" right-aligned
970
+ # within these widths. Sized for normal use; cells widen
971
+ # naturally when values overflow (right-align preserves the gutter).
972
+ _TIME_CELL_W = 17 # fits e.g. "9999.9s (+999.9)"
973
+ _MEM_CELL_W = 19 # fits e.g. "-100.00 GB (+99.99)"
974
+
975
+ @staticmethod
976
+ def _pick_size_unit(mb: float) -> tuple[float, str]:
977
+ """Pick GB vs MB display unit for an absolute MB value.
978
+ Returns ``(divisor, label)`` — e.g. ``(1024.0, "GB")``.
979
+ """
980
+ if abs(mb) >= 1024.0:
981
+ return 1024.0, "GB"
982
+ return 1.0, "MB"
983
+
984
+ @classmethod
985
+ def _fmt_mem_cell(cls, mb: float | None, delta_mb: float | None) -> str:
986
+ """Render a memory cell as ``"<absolute> (<±delta>)"`` right-
987
+ aligned within ``_MEM_CELL_W``. Both numbers are rendered in
988
+ the unit chosen for the absolute, so the cell reads as a single
989
+ consistent magnitude (e.g. ``"5.18 GB (+0.21)"`` — both GB).
990
+ When ``mb`` is None, a single dash fills the cell.
991
+ """
992
+ if mb is None:
993
+ return f"{'-':>{cls._MEM_CELL_W}}"
994
+ div, label = cls._pick_size_unit(mb)
995
+ # Two decimals when GB, integer when MB — matches the example.
996
+ if label == "GB":
997
+ val = f"{mb / div:.2f} {label}"
998
+ else:
999
+ val = f"{mb / div:.0f} {label}"
1000
+ if delta_mb is None:
1001
+ cell = f"{val} (-)"
1002
+ else:
1003
+ if abs(delta_mb) < (0.005 if label == "GB" else 0.5):
1004
+ delta_str = "+0"
1005
+ else:
1006
+ sign = "+" if delta_mb >= 0 else "-"
1007
+ a = abs(delta_mb) / div
1008
+ fmt = ".2f" if label == "GB" else ".0f"
1009
+ delta_str = f"{sign}{a:{fmt}}"
1010
+ cell = f"{val} ({delta_str})"
1011
+ return f"{cell:>{cls._MEM_CELL_W}}"
1012
+
1013
+ @classmethod
1014
+ def _fmt_time_cell(cls, t_elapsed: float, t_section: float | None) -> str:
1015
+ """Render the time cell as ``"<elapsed>s (+<section>)"`` right-
1016
+ aligned within ``_TIME_CELL_W``. First-row ``t_section is None``
1017
+ renders ``"<elapsed>s (-)"``.
1018
+ """
1019
+ if t_section is None:
1020
+ cell = f"{t_elapsed:.1f}s (-)"
1021
+ else:
1022
+ sign = "+" if t_section >= 0 else "-"
1023
+ cell = f"{t_elapsed:.1f}s ({sign}{abs(t_section):.1f})"
1024
+ return f"{cell:>{cls._TIME_CELL_W}}"
1025
+
1026
+ def _emit_header(self) -> None:
1027
+ """Print the column-header line for the phase-progress table."""
1028
+ blank_label = " " * self._LABEL_W
1029
+ header = (
1030
+ f"{blank_label} "
1031
+ f"{'time':^{self._TIME_CELL_W}} "
1032
+ f"| {'RSS memory':^{self._MEM_CELL_W}} "
1033
+ f"| {'system memory':^{self._MEM_CELL_W}} "
1034
+ f"| {'system swap':^{self._MEM_CELL_W}}"
1035
+ )
1036
+ try:
1037
+ print(header, flush=True)
1038
+ except OSError:
1039
+ pass
1040
+
1041
+ def checkpoint(self, label: str, logger: logging.Logger,
1042
+ user_label: str | None = None) -> None:
1043
+ """Record a phase checkpoint.
1044
+
1045
+ ``label`` is the canonical machine-readable identifier persisted
1046
+ to the CSV (when full diagnostics is enabled). ``user_label``
1047
+ (optional) is the human-friendly phrasing emitted to the log;
1048
+ when absent, ``label`` is used.
1049
+
1050
+ Log lines always emit (RSS read from ``/proc`` is essentially
1051
+ free). When full diagnostics is enabled (env-var
1052
+ ``FLEXTOOL_MEMORY_DIAGNOSTICS=1``) the ``traced_peak`` column
1053
+ and the CSV emission are populated by tracemalloc; otherwise
1054
+ the peak shows as ``-``.
1055
+ """
1056
+ peak_mb: float | None = None
1057
+ current_mb: float | None = None
1058
+ if self.enabled:
1059
+ import tracemalloc
1060
+ if not self._started:
1061
+ tracemalloc.start()
1062
+ self._started = True
1063
+ current, peak = tracemalloc.get_traced_memory()
1064
+ current_mb = current / (1024.0 * 1024.0)
1065
+ peak_mb = peak / (1024.0 * 1024.0)
1066
+ rss_mb = self._read_rss_mb()
1067
+ sys_mb = self._read_sys_used_mb()
1068
+ swap_mb = self._read_sys_swap_used_mb()
1069
+ t_elapsed = time.perf_counter() - self.t0
1070
+ # Section deltas relative to previous *emitted* checkpoint. When
1071
+ # we suppress an emission we leave ``_t_prev`` /
1072
+ # ``_rss_prev_mb`` / ``_sys_prev_mb`` / ``_peak_prev_mb`` alone,
1073
+ # so the next emitted line shows the cumulative delta covering
1074
+ # all the suppressed activity in between — exactly what the
1075
+ # user wants to see in the compact mode.
1076
+ t_section = t_elapsed - (self._t_prev - self.t0)
1077
+ delta_rss = rss_mb - self._rss_prev_mb
1078
+ delta_sys = sys_mb - self._sys_prev_mb
1079
+ delta_swap = swap_mb - self._swap_prev_mb
1080
+ # CSV row — only when full diagnostics is enabled and a path was
1081
+ # configured. Always written for every checkpoint (independent
1082
+ # of the log-line whitelist) so the CSV remains a complete
1083
+ # trace.
1084
+ if self.enabled and self._path is not None and peak_mb is not None:
1085
+ row = (
1086
+ str(label),
1087
+ f"{t_elapsed:.6f}",
1088
+ f"{current_mb:.3f}",
1089
+ f"{peak_mb:.3f}",
1090
+ f"{rss_mb:.3f}",
1091
+ )
1092
+ import csv as _csv
1093
+ try:
1094
+ with open(self._path, "a", newline="") as f:
1095
+ _csv.writer(f).writerow(row)
1096
+ except OSError:
1097
+ pass
1098
+ # FLEXTOOL_PYRAMID_PROFILE=1: emit a tab-separated stderr line
1099
+ # for emit_solve_time.* labels (polar-high precedent). Reuses
1100
+ # the already-computed rss_mb / t_elapsed; zero cost when unset.
1101
+ if (os.environ.get("FLEXTOOL_PYRAMID_PROFILE") == "1"
1102
+ and label.startswith("emit_solve_time.")):
1103
+ delta_pyramid = rss_mb - self._pyramid_prev_rss_mb
1104
+ rss_gb = rss_mb / 1024.0
1105
+ delta_gb = delta_pyramid / 1024.0
1106
+ try:
1107
+ import sys as _sys
1108
+ _sys.stderr.write(
1109
+ f"[pyramid profile]\tphase={label}\t"
1110
+ f"rss_gb={rss_gb:.2f}\tdelta_gb={delta_gb:+.2f}\t"
1111
+ f"t_s={t_elapsed:.1f}\n"
1112
+ )
1113
+ _sys.stderr.flush()
1114
+ except OSError:
1115
+ pass
1116
+ self._pyramid_prev_rss_mb = rss_mb
1117
+ # Decide whether to emit the log line. Regular mode shows only
1118
+ # the whitelisted phase labels; ``FLEXTOOL_MEMORY_VERBOSE=1``
1119
+ # restores the full per-checkpoint trace.
1120
+ verbose_mode = bool(os.environ.get("FLEXTOOL_MEMORY_VERBOSE"))
1121
+ emit = self.verbose and (
1122
+ verbose_mode or _is_whitelisted_mem_label(user_label or label)
1123
+ )
1124
+ if emit:
1125
+ display = user_label or label
1126
+ label_col = f"{display:<{self._LABEL_W}}"
1127
+ is_first = self._t_prev == self.t0
1128
+ # Emit the column header once, immediately before the first
1129
+ # data line. Inline labels are dropped from data lines
1130
+ # (cleaner, narrower); the header is the legend.
1131
+ if not self._header_emitted:
1132
+ self._emit_header()
1133
+ self._header_emitted = True
1134
+ elif display == "Solver":
1135
+ # Blank line + header repeat above each Solver row so
1136
+ # the dominant solve phase visually separates from the
1137
+ # per-group prep block printed above it.
1138
+ try:
1139
+ print("", flush=True)
1140
+ except OSError:
1141
+ pass
1142
+ self._emit_header()
1143
+ # First-row convention: time delta renders "(-)" (no prior
1144
+ # checkpoint to subtract from); memory deltas equal their
1145
+ # absolute (prev=0), so e.g. "+230.1" appears alongside the
1146
+ # absolute "230.1 MB" — which is correct: the process has
1147
+ # consumed exactly that much since the recorder started.
1148
+ time_cell = self._fmt_time_cell(
1149
+ t_elapsed, None if is_first else t_section
1150
+ )
1151
+ rss_cell = self._fmt_mem_cell(rss_mb, delta_rss)
1152
+ sys_cell = self._fmt_mem_cell(sys_mb, delta_sys)
1153
+ swap_cell = self._fmt_mem_cell(swap_mb, delta_swap)
1154
+ line = (
1155
+ f"{label_col} "
1156
+ f"{time_cell} "
1157
+ f"| {rss_cell} "
1158
+ f"| {sys_cell} "
1159
+ f"| {swap_cell}"
1160
+ )
1161
+ try:
1162
+ print(line, flush=True)
1163
+ except OSError:
1164
+ pass
1165
+ # Advance prev-section bookkeeping ONLY on actual emission so
1166
+ # the next emitted line's delta covers all the suppressed
1167
+ # activity since the last visible checkpoint.
1168
+ self._t_prev = time.perf_counter()
1169
+ self._rss_prev_mb = rss_mb
1170
+ self._sys_prev_mb = sys_mb
1171
+ self._swap_prev_mb = swap_mb
1172
+ if peak_mb is not None:
1173
+ self._peak_prev_mb = peak_mb
1174
+
1175
+
1176
+ class _NoopMemoryRecorder:
1177
+ """Zero-overhead drop-in when ``FLEXTOOL_MEMORY_DIAGNOSTICS`` is unset.
1178
+
1179
+ Retained for callers that explicitly want a fully silent recorder
1180
+ (rare; debugging). The default code path now uses
1181
+ :class:`_MemoryRecorder` with ``enabled=False`` instead — that mode
1182
+ still emits user-visible log lines (RSS + section time + Δrss)
1183
+ while skipping CSV emission and tracemalloc startup.
1184
+ """
1185
+
1186
+ enabled = False
1187
+
1188
+ def checkpoint(self, label: str, logger: logging.Logger,
1189
+ user_label: str | None = None) -> None: # noqa: D401, ARG002
1190
+ return None
1191
+
1192
+
1193
+ # Module-level recorder reference. ``run_orchestration`` constructs the
1194
+ # per-run recorder and publishes it here so deeper-stack modules (e.g.
1195
+ # :mod:`flextool.engine_polars.input`'s ``_apply_db_overrides``) can
1196
+ # emit phase progress in the unified ``[mem]`` format without each
1197
+ # carrying a recorder kwarg. Reset to ``None`` when the run completes
1198
+ # so a subsequent run starts clean.
1199
+ _PHASE_RECORDER: "_MemoryRecorder | None" = None
1200
+
1201
+
1202
+ def set_phase_recorder(rec: "_MemoryRecorder | None") -> None:
1203
+ """Publish the current run's phase recorder so deeper callers can
1204
+ emit checkpoints without explicit plumbing. Pass ``None`` to clear.
1205
+ """
1206
+ global _PHASE_RECORDER
1207
+ _PHASE_RECORDER = rec
1208
+
1209
+
1210
+ def get_phase_recorder() -> "_MemoryRecorder | None":
1211
+ """Return the current run's phase recorder, or ``None`` when none
1212
+ is active (e.g. unit tests that bypass ``run_orchestration``).
1213
+ """
1214
+ return _PHASE_RECORDER
1215
+
1216
+
1217
+ # ---------------------------------------------------------------------------
1218
+ # Heap release (glibc malloc_trim)
1219
+ # ---------------------------------------------------------------------------
1220
+ #
1221
+ # The polars/Rust allocator routes through glibc malloc, and glibc's main
1222
+ # arena holds freed pages internally instead of returning them to the OS.
1223
+ # After a heavy allocation+free cycle (``write_workdir_inputs``,
1224
+ # ``load_flextool``, the broadcast cascade) we leak hundreds of MB to
1225
+ # multiple GB of unmapped-but-untrimmed heap. Direct measurement on
1226
+ # H2_trade y2050 (2026-05-13): RSS 3.8 GB → 2.25 GB after a single
1227
+ # ``malloc_trim(0)`` call (1.6 GB / 41 % drop). ``pa.default_memory_pool
1228
+ # ().release_unused()`` and ``gc.collect()`` had zero effect — polars
1229
+ # does not route through pyarrow's pool, so only the libc-level trim
1230
+ # releases anything.
1231
+ #
1232
+ # The helper is a no-op on non-glibc systems (musl Alpine containers,
1233
+ # macOS, Windows). Safe to call freely; cost is ~10-50ms per call.
1234
+ _libc_malloc_trim = None
1235
+
1236
+
1237
+ def _try_malloc_trim() -> bool:
1238
+ """Call ``libc.so.6.malloc_trim(0)`` if available; return True on success.
1239
+
1240
+ Cached lookup after the first call. Failures (non-glibc systems,
1241
+ missing libc, etc.) are logged once at DEBUG level and the helper
1242
+ becomes a permanent no-op for the process lifetime.
1243
+ """
1244
+ global _libc_malloc_trim
1245
+ if _libc_malloc_trim is False:
1246
+ return False
1247
+ if _libc_malloc_trim is None:
1248
+ try:
1249
+ import ctypes
1250
+ libc = ctypes.CDLL("libc.so.6")
1251
+ # malloc_trim(size_t pad) -> int. pad=0 means trim aggressively.
1252
+ _libc_malloc_trim = libc.malloc_trim
1253
+ except (OSError, AttributeError):
1254
+ _libc_malloc_trim = False
1255
+ return False
1256
+ try:
1257
+ _libc_malloc_trim(0)
1258
+ return True
1259
+ except Exception: # noqa: BLE001
1260
+ _libc_malloc_trim = False
1261
+ return False
1262
+
1263
+
1264
+ def _phase_prof(label: str) -> None:
1265
+ """Env-gated (FLEXTOOL_PHASE_PROFILE=1) epoch-stamped RSS print to stderr. No-op otherwise.
1266
+ Epoch matches flextool/_mem_sampler.py's `epoch=` field for 1:1 alignment with mem.log."""
1267
+ import os
1268
+ import sys
1269
+ import time
1270
+ if os.environ.get("FLEXTOOL_PHASE_PROFILE") != "1":
1271
+ return
1272
+ try:
1273
+ with open("/proc/self/status") as _f:
1274
+ for _ln in _f:
1275
+ if _ln.startswith("VmRSS:"):
1276
+ sys.stderr.write(f"[phase profile] epoch={time.time():.3f}\tstep={label}\trss_gb={int(_ln.split()[1])/(1024*1024):.3f}\n")
1277
+ sys.stderr.flush()
1278
+ break
1279
+ except Exception:
1280
+ pass
1281
+
1282
+
1283
+ # ---------------------------------------------------------------------------
1284
+ # Cross-level retention audit (env-gated diagnostic + regression hook)
1285
+ #
1286
+ # A prior solve-level's ``Solution.highs`` (and ``flex_data_provider``) must be
1287
+ # released BEFORE the next solve builds its FlexData + LP — otherwise the two
1288
+ # levels' footprints coexist (storage + dispatch ≈ 2x peak; the DES 7/9
1289
+ # near-OOM). This records any prior step whose level is EXHAUSTED (no upcoming
1290
+ # solve of that level) yet still holds a live ``solution.highs`` at the instant
1291
+ # a new solve is about to build. A non-empty record == the cross-level
1292
+ # retention bug is present. Gated by ``FLEXTOOL_LEVEL_RELEASE_AUDIT=1``;
1293
+ # consumed by tests/engine_polars/test_cross_level_highs_release.py.
1294
+ # ---------------------------------------------------------------------------
1295
+ _LEVEL_RELEASE_AUDIT: "list[dict]" = []
1296
+
1297
+
1298
+ def _audit_prior_level_release(*, steps, step_level_keys, all_level_keys,
1299
+ iter_idx, this_level, complete_solve_name):
1300
+ """Append a violation record for exhausted-level prior steps still holding
1301
+ a live ``solution.highs``. No-op unless ``FLEXTOOL_LEVEL_RELEASE_AUDIT=1``."""
1302
+ if os.environ.get("FLEXTOOL_LEVEL_RELEASE_AUDIT") != "1":
1303
+ return
1304
+ upcoming = (
1305
+ set(all_level_keys[iter_idx + 1:])
1306
+ if (iter_idx is not None and all_level_keys) else set()
1307
+ )
1308
+ violators = []
1309
+ for _k, _step in (steps or {}).items():
1310
+ _lvl = step_level_keys.get(_k)
1311
+ sol = getattr(_step, "solution", None)
1312
+ if sol is None or getattr(sol, "highs", None) is None:
1313
+ continue
1314
+ if _lvl is not None and _lvl != this_level and _lvl not in upcoming:
1315
+ violators.append(_k)
1316
+ _LEVEL_RELEASE_AUDIT.append({
1317
+ "kind": "exhausted_entry",
1318
+ "solve": complete_solve_name,
1319
+ "iter_idx": iter_idx,
1320
+ "violators": violators,
1321
+ })
1322
+
1323
+
1324
+ def _audit_cold_rebuild_release(*, steps, complete_solve_name):
1325
+ """At a COLD rebuild (warm_used False), no parked step's HiGHS is the
1326
+ reuse source, so any prior step still holding a live ``solution.highs``
1327
+ is a same-level stacking risk. Record them (post-release should be
1328
+ empty). No-op unless ``FLEXTOOL_LEVEL_RELEASE_AUDIT=1``."""
1329
+ if os.environ.get("FLEXTOOL_LEVEL_RELEASE_AUDIT") != "1":
1330
+ return
1331
+ violators = [
1332
+ _k for _k, _step in (steps or {}).items()
1333
+ if getattr(getattr(_step, "solution", None), "highs", None) is not None
1334
+ ]
1335
+ _LEVEL_RELEASE_AUDIT.append({
1336
+ "kind": "cold_rebuild",
1337
+ "solve": complete_solve_name,
1338
+ "violators": violators,
1339
+ })
1340
+
1341
+
1342
+ # ---------------------------------------------------------------------------
1343
+ # Result types
1344
+ # ---------------------------------------------------------------------------
1345
+
1346
+
1347
+ @dataclass
1348
+ class SnapshotSolution:
1349
+ """Lightweight Solution-API stand-in carrying only the captured
1350
+ decision-variable frames a sub-solve needs to survive polar-high's
1351
+ internal between-solve release of ``Solution._vars``.
1352
+
1353
+ The cascade captures :attr:`OrchestrationStep.captured_vars` at the
1354
+ earliest reliable point in the per-solve loop (immediately before
1355
+ the step is deposited in ``self._all_steps``). Polar-high may
1356
+ release the live :class:`polar_high.Solution._vars` dict for
1357
+ earlier sub-solves between the deposit and the end-of-cascade
1358
+ consumers (e.g. :func:`flextool.process_outputs.read_parameters.
1359
+ _entity_all_capacity`), so end-of-cascade writers that look at
1360
+ every sub-solve's decision variables fall back to this snapshot.
1361
+
1362
+ The wrapper only satisfies the duck-typed contract that
1363
+ ``_entity_all_capacity`` needs:
1364
+
1365
+ * ``solution._vars`` — dict-like with the captured names as keys
1366
+ (supports ``"v_invest_p" in solution._vars``).
1367
+ * ``solution.value(name)`` — returns the captured polars long-form
1368
+ DataFrame.
1369
+
1370
+ All other ``Solution`` attributes (``obj``, ``optimal``, ``highs``,
1371
+ ``col_value``, …) are intentionally absent; callers that need
1372
+ those should read the live :attr:`OrchestrationStep.solution`
1373
+ (last step only by default, or every step under
1374
+ ``keep_solutions=True``).
1375
+ """
1376
+
1377
+ _vars: "dict[str, pl.DataFrame]"
1378
+
1379
+ def value(self, name: str) -> "pl.DataFrame":
1380
+ return self._vars[name]
1381
+
1382
+
1383
+ @dataclass
1384
+ class OrchestrationStep:
1385
+ """Per-solve result of :func:`run_orchestration`.
1386
+
1387
+ Mirrors :class:`flextool.engine_polars.chain.ChainStep` but produced
1388
+ by the native orchestrator path. ``handoff`` is the carrier used to
1389
+ seed the *next* solve's preprocessing.
1390
+
1391
+ Attributes
1392
+ ----------
1393
+ solve_name : str
1394
+ The complete (sub-)solve identifier emitted by flextool's
1395
+ orchestration loop (e.g. ``"y2025_5week"`` or
1396
+ ``"dispatch_fullYear_roll_roll_3"``).
1397
+ solution : polar_high.Solution | None
1398
+ The HiGHS solution. By default (``keep_solutions=False`` on
1399
+ :func:`run_chain_from_db` / :func:`run_orchestration`) only the
1400
+ LAST sub-solve in a cascade retains its ``solution`` — earlier
1401
+ steps clear this slot to release the HiGHS instance + variable
1402
+ arrays. Set ``keep_solutions=True`` to retain ``solution`` on
1403
+ every step (Phase C.5 — memory discipline). Also ``None`` on
1404
+ the failed-solve path.
1405
+ handoff : SolveHandoff
1406
+ polar_high-derived handoff carriers, threaded forward. Always
1407
+ populated (kilobyte-sized; safe to retain across the cascade).
1408
+ obj : float | None
1409
+ Objective value (cached for quick comparison; equal to
1410
+ ``solution.obj``). Always populated when the solve succeeded —
1411
+ survives the ``keep_solutions=False`` slim pass, so cascade-
1412
+ wide objective sweeps work without ``keep_solutions=True``.
1413
+ optimal : bool | None
1414
+ Phase C.5 — slim summary mirror of ``solution.optimal`` that
1415
+ survives the per-step memory release. ``None`` only on the
1416
+ failed-solve path (where ``solution`` is also ``None``). This is
1417
+ the STRICT solver verdict (HiGHS ``kOptimal``): a near-optimal
1418
+ crossover-off solve that :func:`classify_acceptance` accepted for
1419
+ consumption is ``False`` here — read :attr:`near_optimal` (or the
1420
+ ``optimal or near_optimal`` union) when the question is "is this
1421
+ step's solution usable?", not "did the solver certify optimality?".
1422
+ near_optimal : bool
1423
+ True when the solve was NOT strictly ``kOptimal`` yet
1424
+ :func:`classify_acceptance` accepted it as in-practice-optimal
1425
+ (feasible primal, small primal--dual gap) — the crossover-off
1426
+ interior-point case. Such a solve wrote a usable solution and must
1427
+ NOT be treated as a failure by the CLI exit-code scan. Always
1428
+ ``False`` for the Benders path (its acceptance is governed by
1429
+ ``is_benders`` + gap/tol) and for the failed-solve path.
1430
+ warm_used : bool
1431
+ Δ.12d — True if this solve was produced by warm-updating the
1432
+ prior solve's :class:`polar_high.WarmProblem` instance; False
1433
+ if it was a cold rebuild. Always False for the first solve
1434
+ and for ``warm=False`` runs. Always populated (slim summary).
1435
+ flex_data : FlexData | None
1436
+ Δ.31 — the polars input bundle this sub-solve consumed. Held
1437
+ on the step so downstream :func:`flextool.process_outputs.
1438
+ write_outputs` can build the parameter / set namespaces in
1439
+ memory instead of re-parsing the workdir CSVs. Subject to the
1440
+ same ``keep_solutions`` gating as ``solution`` (Phase C.5):
1441
+ only the LAST step retains ``flex_data`` by default. ``None``
1442
+ on the failed-load path.
1443
+ flex_data_provider : FlexDataProvider | None
1444
+ The per-sub-solve :class:`FlexDataProvider` populated by the
1445
+ cascade's writers. Subject to the same ``keep_solutions``
1446
+ gating as ``solution`` / ``flex_data`` (Phase C.5): only the
1447
+ LAST step retains it by default. Consumed by ``--csv-dump``
1448
+ in ``cmd_run_flextool`` to snapshot the cascade's derived
1449
+ frames to disk.
1450
+ """
1451
+
1452
+ solve_name: str
1453
+ solution: "Solution | None"
1454
+ handoff: SolveHandoff
1455
+ obj: float | None = None
1456
+ optimal: bool | None = None
1457
+ near_optimal: bool = False
1458
+ warm_used: bool = False
1459
+ flex_data: "FlexData | None" = None
1460
+ flex_data_provider: "object | None" = None
1461
+ is_benders: bool = False
1462
+ """True when this step was produced by the Benders region driver and
1463
+ carries only a :class:`SnapshotSolution` invest carrier, not a full
1464
+ :class:`Solution` (so it cannot yet drive processed outputs)."""
1465
+ benders_gap: float | None = None
1466
+ """Benders steps only — the relative optimality gap reached at the
1467
+ incumbent ``(best_UB − LB)/max(1, |best_UB|)``. Paired with
1468
+ ``benders_tol`` / ``benders_iterations`` so the CLI exit-code scan can
1469
+ report a NON-convergence (feasible incumbent, gap never met ``tol``) as a
1470
+ loud warning rather than a bogus "infeasible/unbounded"."""
1471
+ benders_tol: float | None = None
1472
+ """Benders steps only — the relative-gap tolerance that was in force."""
1473
+ benders_iterations: int | None = None
1474
+ """Benders steps only — the number of master/subproblem iterations run."""
1475
+ captured_vars: "dict[str, pl.DataFrame]" = field(default_factory=dict)
1476
+ """Per-sub-solve snapshot of the decision-variable frames that
1477
+ end-of-cascade writers (``_entity_all_capacity`` and friends) need
1478
+ after polar-high has released the live ``Solution._vars`` dict for
1479
+ earlier sub-solves.
1480
+
1481
+ Captured at the latest reliable point in the per-solve loop —
1482
+ immediately before the step is deposited — so it always reflects
1483
+ the sub-solve's own values regardless of how polar-high or
1484
+ downstream slimming touches ``self.solution``. Empty dict when
1485
+ the solve failed or when there was no Solution to capture from.
1486
+
1487
+ Access via :attr:`effective_solution` rather than reading directly:
1488
+ callers normally want the live Solution when it's still populated
1489
+ and the snapshot only as a fallback.
1490
+ """
1491
+
1492
+ @property
1493
+ def effective_solution(self) -> "Solution | SnapshotSolution | None":
1494
+ """Return the live :attr:`solution` when its ``_vars`` is still
1495
+ populated; otherwise a :class:`SnapshotSolution` over the
1496
+ captured frames. ``None`` only when both are unavailable
1497
+ (failed-solve path, or solution slimmed AND no capture taken).
1498
+
1499
+ End-of-cascade consumers that walk every sub-solve's decision
1500
+ variables (e.g. ``read_parameters_multi`` →
1501
+ ``_entity_all_capacity``) should read this instead of
1502
+ :attr:`solution` so non-last sub-solves remain observable
1503
+ after polar-high's between-solve ``_vars`` release.
1504
+ """
1505
+ sol = self.solution
1506
+ live_vars = getattr(sol, "_vars", None) if sol is not None else None
1507
+ if live_vars:
1508
+ return sol
1509
+ if self.captured_vars:
1510
+ return SnapshotSolution(_vars=dict(self.captured_vars))
1511
+ return sol
1512
+
1513
+
1514
+ # ---------------------------------------------------------------------------
1515
+ # Scaling-output helper (shared by cascade & single-solve paths)
1516
+ # ---------------------------------------------------------------------------
1517
+
1518
+
1519
+ def _write_scale_csv(
1520
+ *,
1521
+ solve_data_dir: Path,
1522
+ solve_name: str,
1523
+ effective_obj_scale: float,
1524
+ logger: logging.Logger,
1525
+ ) -> None:
1526
+ """Emit ``solve_data/scale_the_objective.csv`` for the given solve.
1527
+
1528
+ Required by the downstream parquet / CSV writers — they read it via
1529
+ :func:`flextool.process_outputs.read_highs_solution.
1530
+ _resolve_inv_scale_the_objective` to un-scale variable values and
1531
+ duals back to user-facing units. Best-effort: a write failure logs
1532
+ a warning but does not raise.
1533
+
1534
+ The legacy human-readable ``scaling_report.txt`` diagnostic was
1535
+ retired in Phase 2b; the autoscaler's ``solve_data/autoscale_<solve>.yaml``
1536
+ (written from :func:`_autoscale_emit_layer1`) is the new
1537
+ machine-readable audit. Callers should write the CSV exactly once
1538
+ per base solve — its value is invariant across rolls of the same
1539
+ base solve.
1540
+ """
1541
+ try:
1542
+ from flextool.engine_polars._emit_solve_writers import (
1543
+ derive_scale_the_objective,
1544
+ )
1545
+ sd = Path(solve_data_dir)
1546
+ sd.mkdir(parents=True, exist_ok=True)
1547
+ path = sd / "scale_the_objective.csv"
1548
+ derive_scale_the_objective(effective_obj_scale).write_csv(
1549
+ path, line_terminator="\r\n",
1550
+ )
1551
+ except Exception as exc: # noqa: BLE001
1552
+ logger.warning(
1553
+ "scale_the_objective.csv write failed for %s: %s",
1554
+ solve_name, exc,
1555
+ )
1556
+
1557
+
1558
+ # ---------------------------------------------------------------------------
1559
+ # Master loop
1560
+ # ---------------------------------------------------------------------------
1561
+
1562
+
1563
+ def _validate_model_solve(state: RunnerState) -> list[str]:
1564
+ """Validate ``state.solve.model_solve`` and return the solve list.
1565
+
1566
+ There must be exactly one model with at least one solve. Multi-
1567
+ model is documented as unsupported.
1568
+ """
1569
+ if not state.solve.model_solve:
1570
+ raise FlexToolConfigError(
1571
+ "No model. Make sure the 'model' class defines solves [Array]."
1572
+ )
1573
+ if len(state.solve.model_solve) > 1:
1574
+ raise FlexToolConfigError(
1575
+ "Trying to run more than one model — not supported. "
1576
+ "model_solve must contain exactly one model."
1577
+ )
1578
+ solves = next(iter(state.solve.model_solve.values()))
1579
+ if not solves:
1580
+ raise FlexToolConfigError("No solves in model.")
1581
+ return solves
1582
+
1583
+
1584
+ def _benders_consume_guard_message(
1585
+ base_solve_name: str,
1586
+ prev_captured: str | None,
1587
+ benders_solve_names: "set[str]",
1588
+ benders_invest_handoff_names: "set[str]",
1589
+ ) -> str | None:
1590
+ """Loud guard for the cross-scheme handoff (v60/v62, narrowed for TIER 1).
1591
+
1592
+ A Benders *investment* solve now deposits a real ``SolveHandoff``
1593
+ (``realized_invest`` / ``realized_existing`` / ``divest_cumulative``)
1594
+ so a downstream rolling-dispatch solve can consume its invested
1595
+ capacity — that path is SUPPORTED and must NOT fire the guard.
1596
+
1597
+ The guard now fires only when *base_solve_name* would consume a
1598
+ Benders predecessor that deposited **no usable investment handoff**
1599
+ — i.e. the predecessor is in *benders_solve_names* but NOT in
1600
+ *benders_invest_handoff_names* (its handoff carries no
1601
+ ``realized_invest`` / ``divest_cumulative`` to hand forward, so the
1602
+ downstream solve would silently load nothing investy from it).
1603
+
1604
+ Returns ``None`` when the situation is safe:
1605
+
1606
+ * no predecessor, or the predecessor was not a Benders solve;
1607
+ * the predecessor is a roll of the *same* base solve (rolls of one
1608
+ Benders solve share its base name — not a cross-scheme consume);
1609
+ * the predecessor deposited a usable invest handoff (TIER 1 path).
1610
+ """
1611
+ if prev_captured is None or prev_captured not in benders_solve_names:
1612
+ return None
1613
+ if base_solve_name == re.sub(r"_roll_\d+$", "", prev_captured):
1614
+ return None
1615
+ if prev_captured in benders_invest_handoff_names:
1616
+ return None
1617
+ return (
1618
+ f"Solve '{base_solve_name}' follows '{prev_captured}', which ran "
1619
+ f"under decomposition=benders but produced NO invested/divested "
1620
+ f"capacity to hand forward (its handoff carries no realized_invest "
1621
+ f"/ divest_cumulative). Consuming it would silently load nothing "
1622
+ f"from that solve. Make the downstream solve benders too, order "
1623
+ f"the chain so the Benders solve is terminal, or ensure the "
1624
+ f"Benders solve actually invests."
1625
+ )
1626
+
1627
+
1628
+ def _bootstrap_dirs(work_folder: Path, logger: logging.Logger) -> None:
1629
+ """Create ``solve_data/``, ``output_raw/``, ``output_plots/`` under
1630
+ *work_folder* if they don't exist.
1631
+
1632
+ Mirrors lines 63-76 of the flextool reference.
1633
+ """
1634
+ for sub in ("solve_data", "output_raw", "output_plots"):
1635
+ try:
1636
+ (work_folder / sub).mkdir(parents=True, exist_ok=False)
1637
+ except FileExistsError:
1638
+ logger.debug(f"{sub} folder existed")
1639
+
1640
+
1641
+ def run_orchestration(
1642
+ state: RunnerState,
1643
+ work_folder: Path | str,
1644
+ *,
1645
+ runner_factory=None,
1646
+ db_url: str | None = None,
1647
+ scenario_name: str | None = None,
1648
+ warm: bool = True,
1649
+ keep_solutions: bool = False,
1650
+ csv_dump: bool = False,
1651
+ ) -> dict[str, OrchestrationStep]:
1652
+ """Drive the master loop natively.
1653
+
1654
+ Per-step:
1655
+
1656
+ 1. Bootstrap directories (idempotent).
1657
+ 2. Validate ``state.solve.model_solve`` (exactly one model, ≥1 solve).
1658
+ 3. Reset ``state.solve.roll_counter`` for repeatable test runs (R-O5).
1659
+ 4. Drive flextool's ``orchestration.run_model`` with a polar_high
1660
+ cascade solver — each per-solve iteration loads the snapshot via
1661
+ ``load_flextool``, builds the LP via ``build_flextool``, solves
1662
+ via HiGHS, captures the handoff, and deposits it into
1663
+ ``state.handoffs`` (which the consume side already reads from).
1664
+
1665
+ Parameters
1666
+ ----------
1667
+ state : RunnerState
1668
+ Native polar_high state carrier. ``state.solve`` and
1669
+ ``state.timeline`` must be populated (call
1670
+ :func:`run_chain_from_db` for the canonical end-to-end path that
1671
+ sets these up from a DB). The function may flip
1672
+ ``state.handoffs`` from ``None`` to ``{}`` to enable the in-memory
1673
+ capture/consume path; this is done unconditionally — the native
1674
+ orchestrator always uses in-memory handoff (storage-fixing falls
1675
+ through to the file-copy path only when ``state.handoffs`` is
1676
+ explicitly set to ``None`` after this returns).
1677
+ work_folder : Path | str
1678
+ Directory the snapshot tree lives under. Created if missing.
1679
+ Per-solve preprocessing CSVs are emitted under
1680
+ ``work_folder/solve_data/``.
1681
+ runner_factory : callable | None
1682
+ Optional override for constructing the underlying
1683
+ :class:`FlexToolRunner` — used by tests that want to short-
1684
+ circuit flextool's preprocessing. Default uses the canonical
1685
+ constructor.
1686
+ warm : bool, default False
1687
+ Δ.12d — when True, attempt warm LP updates between consecutive
1688
+ structurally-compatible per-solve iterations using
1689
+ :class:`polar_high.WarmProblem`. Reuses one WarmProblem across
1690
+ the cascade, applying ``_apply_warm_updates`` between solves
1691
+ and falling back to a cold rebuild whenever the structural
1692
+ fingerprint changes or any unmapped Param differs. Decisions
1693
+ are recorded per-step on :attr:`OrchestrationStep.warm_used`.
1694
+ Default ``False`` preserves the original cold-rebuild
1695
+ behaviour.
1696
+
1697
+ Returns
1698
+ -------
1699
+ dict[str, OrchestrationStep]
1700
+ Mapping ``complete_solve_name → OrchestrationStep`` in solve
1701
+ order (Python dict insertion order preserved).
1702
+
1703
+ Raises
1704
+ ------
1705
+ FlexToolConfigError
1706
+ Empty / multi-model ``model_solve``.
1707
+ FlexToolSolveError
1708
+ Any per-solve LP infeasibility / non-optimal status.
1709
+ """
1710
+ work_folder = Path(work_folder)
1711
+ work_folder.mkdir(parents=True, exist_ok=True)
1712
+ # Rebuild ``paths`` for this run's work_folder, but CARRY FORWARD
1713
+ # ``solver_config_dir`` from the incoming state — the per-solve loop
1714
+ # reads it to resolve ``<dir>/highs.opt`` for the in-process HiGHS
1715
+ # options. ``run_chain_from_db`` sets it (to the work-folder
1716
+ # ``solver_config`` copy); a bare ``PathConfig(work_folder=...)`` here
1717
+ # would silently drop the ``highs.opt`` floor on the in-process path.
1718
+ _prior_solver_config_dir = (
1719
+ state.paths.solver_config_dir if state.paths is not None else None
1720
+ )
1721
+ state.paths = PathConfig(
1722
+ work_folder=work_folder,
1723
+ solver_config_dir=_prior_solver_config_dir,
1724
+ )
1725
+
1726
+ logger = state.logger
1727
+ _bootstrap_dirs(work_folder, logger)
1728
+ solves = _validate_model_solve(state)
1729
+
1730
+ # Reset roll-counter so repeated calls with the same SolveConfig
1731
+ # don't desync. R-O5 in the orchestration risk register.
1732
+ state.solve.roll_counter = state.solve.make_roll_counter()
1733
+
1734
+ # Always enable in-memory handoff for the native path.
1735
+ if state.handoffs is None:
1736
+ state.handoffs = {}
1737
+
1738
+ # Stash the ``csv_dump`` flag on the state so per-iter sites in
1739
+ # ``_drive_cascade`` can consult it (gates ``data.dump_csvs``).
1740
+ state.csv_dump = bool(csv_dump) # type: ignore[attr-defined]
1741
+
1742
+ # Per-solve autoscaler mode (v64 ``solve.scaling``): capture whether
1743
+ # the operator pinned the mode explicitly via ``--scaling`` /
1744
+ # ``FLEXTOOL_SCALING`` (the CLI mirrors the flag into the env var)
1745
+ # BEFORE the per-solve loop can overwrite it. When set, the operator
1746
+ # override wins over every solve's DB value; when unset, each solve's
1747
+ # ``solve.scaling`` drives ``FLEXTOOL_SCALING`` in ``_drive_cascade``.
1748
+ # Restore on exit so the DB-derived env never leaks to the caller.
1749
+ state.scaling_user_env = os.environ.get( # type: ignore[attr-defined]
1750
+ "FLEXTOOL_SCALING")
1751
+
1752
+ # The cascade solver runs polar_high on each solve and captures the
1753
+ # handoff. We use flextool's orchestration loop driver because it
1754
+ # encodes the recursive/rolling/stochastic expansion + per-solve
1755
+ # preprocessing chain we still consume. Our cascade solver is the
1756
+ # `solver.run(...)` callback inside that loop.
1757
+ try:
1758
+ return _drive_cascade(state, work_folder, solves, runner_factory,
1759
+ db_url=db_url, scenario_name=scenario_name,
1760
+ warm=warm, keep_solutions=keep_solutions)
1761
+ finally:
1762
+ if state.scaling_user_env is None:
1763
+ os.environ.pop("FLEXTOOL_SCALING", None)
1764
+ else:
1765
+ os.environ["FLEXTOOL_SCALING"] = state.scaling_user_env
1766
+
1767
+
1768
+ def _drive_cascade(
1769
+ state: RunnerState,
1770
+ work_folder: Path,
1771
+ solves: list[str],
1772
+ runner_factory,
1773
+ *,
1774
+ db_url: str | None = None,
1775
+ scenario_name: str | None = None,
1776
+ keep_solutions: bool = False,
1777
+ warm: bool = True,
1778
+ ) -> dict[str, OrchestrationStep]:
1779
+ """Drive the flextool master loop with a polar_high cascade solver.
1780
+
1781
+ For every per-solve iteration:
1782
+
1783
+ 1. Read the snapshot via ``load_flextool``.
1784
+ 2. Build the LP via ``build_flextool`` (cold rebuild) OR warm-update
1785
+ the prior iteration's :class:`polar_high.WarmProblem`.
1786
+ 3. Solve via HiGHS.
1787
+ 4. Build the handoff via ``build_handoff_from_solution``.
1788
+ 5. Deposit it into ``state.handoffs`` so the next iteration's
1789
+ preprocessing picks it up.
1790
+
1791
+ Parameter ``warm`` toggles per-iteration warm-LP updates: when True,
1792
+ the cascade reuses one ``WarmProblem`` across consecutive
1793
+ structurally-compatible iterations. See
1794
+ :mod:`flextool.engine_polars._warm` for the structural-fingerprint
1795
+ + Param-classification machinery. Cold rebuild (``warm=False``)
1796
+ remains the default for backward compatibility with every existing
1797
+ caller.
1798
+
1799
+ Emits an :class:`OrchestrationStep` per solve and runs the LAST
1800
+ solve too (the polar_high-side bookkeeping is the deliverable,
1801
+ not just an intermediate).
1802
+ """
1803
+ # Late imports — keep the orchestration module's import surface narrow
1804
+ # for callers that only need the dataclass.
1805
+ from flextool.engine_polars._solver_base import SolverRunner
1806
+ from flextool.engine_polars._native_run_model import native_run_model
1807
+
1808
+ from polar_high import NamedBasis, Problem, WarmProblem
1809
+ from flextool.engine_polars.input import (
1810
+ build_handoff_from_solution,
1811
+ load_flextool,
1812
+ )
1813
+ from flextool.engine_polars.model import build_flextool
1814
+ from flextool.engine_polars._output_writer import (
1815
+ OutputWriterState,
1816
+ write_outputs_for_solve,
1817
+ )
1818
+ from flextool.engine_polars._warm import (
1819
+ _IncompatibleUpdate,
1820
+ _apply_warm_updates,
1821
+ _build_warm_problem,
1822
+ _fingerprint,
1823
+ )
1824
+
1825
+ results: dict[str, OrchestrationStep] = {}
1826
+ # Δ.1: adapter that reuses flextool's process_outputs writers. The
1827
+ # state carrier collects ``periods_already_emitted`` across the
1828
+ # cascade so we don't have to round-trip through SolveHandoff.
1829
+ writer_state = OutputWriterState()
1830
+
1831
+ # The runner_factory hook lets tests inject a mock; the default uses
1832
+ # FlexToolRunner constructed against the same DB the state was
1833
+ # loaded from. Since RunnerState doesn't carry a DB URL
1834
+ # by default, callers must supply this via runner_factory or use
1835
+ # ``run_chain_from_db`` which constructs the runner explicitly.
1836
+ if runner_factory is None:
1837
+ raise FlexToolConfigError(
1838
+ "run_orchestration requires a runner_factory to construct "
1839
+ "the underlying FlexToolRunner. Use run_chain_from_db for "
1840
+ "the canonical end-to-end path that wires this for you."
1841
+ )
1842
+ _drive_rec = get_phase_recorder()
1843
+ _drive_logger = state.logger
1844
+ runner = runner_factory()
1845
+ if _drive_rec is not None:
1846
+ _drive_rec.checkpoint(
1847
+ "flextool_runner_constructed", _drive_logger,
1848
+ user_label="FlexToolRunner constructed",
1849
+ )
1850
+ # Push our state's handoff slot onto the runner's state so the
1851
+ # cascade and any consume hooks share the same dict.
1852
+ runner.state.handoffs = state.handoffs
1853
+ # Per-level Provider cache (Design A). ``native_run_model`` lazily
1854
+ # initialises this on first iter, but seeding it here makes the
1855
+ # invariant ``state._level_providers is dict`` explicit at every
1856
+ # entry point (cascade + fast_load) instead of relying on hasattr
1857
+ # probes downstream.
1858
+ runner.state._level_providers = {}
1859
+ # Step 2.5 — forward the cascade-input Provider seeded in
1860
+ # ``run_chain_from_db`` onto runner.state so the per-sub-solve hook
1861
+ # at :mod:`flextool.engine_polars._native_run_model` (line 365-370)
1862
+ # picks it up. ``None`` is allowed for entry points that bypass
1863
+ # ``run_chain_from_db`` — the hook then builds an empty Provider.
1864
+ _cip = getattr(state, "cascade_input_provider", None)
1865
+ if _cip is not None:
1866
+ runner.state.cascade_input_provider = _cip
1867
+ # Phase 5c — forward the engine_polars-side ``override_provider``
1868
+ # callable onto ``runner.state`` so the per-sub-solve hook in
1869
+ # :mod:`flextool.engine_polars._native_run_model` (Phase 5b) picks
1870
+ # it up. ``None`` keeps the no-override default.
1871
+ _op = getattr(state, "override_provider", None)
1872
+ if _op is not None:
1873
+ runner.state.override_provider = _op
1874
+ runner.state.logger.setLevel(logging.ERROR)
1875
+ # Forward the opt-in memory recorder (no-op when env var unset) so
1876
+ # ``_PolarHighCascadeSolver.run`` can fire the first-iter checkpoints.
1877
+ runner.state._memory_recorder = getattr( # type: ignore[attr-defined]
1878
+ state, "_memory_recorder", _NoopMemoryRecorder()
1879
+ )
1880
+
1881
+ # Δ.12c — build a SpineDbReader once and reuse it across the cascade.
1882
+ # When db_url + scenario_name are supplied (run_chain_from_db wires
1883
+ # them), the override chain fires for every per-solve load — covering
1884
+ # the seeds the workdir CSV path can't provide once Δ.12-drop /
1885
+ # Δ.12c have retired the redundant CSVs. When the caller didn't
1886
+ # supply them, we fall back to load_flextool's per-call autoresolve
1887
+ # (which works for fixtures whose work_folder follows the
1888
+ # ``work_<scenario>`` convention).
1889
+ cascade_db_reader = None
1890
+ if db_url is not None and scenario_name is not None:
1891
+ from flextool.engine_polars._spinedb_reader import SpineDbReader
1892
+ # Phase 4.6 — thread axis_enums + contract from the cascade
1893
+ # provider if available so the reader casts on emit.
1894
+ _cip_for_reader = getattr(state, "cascade_input_provider", None)
1895
+ _cascade_axis_enums = getattr(_cip_for_reader, "axis_enums", None) \
1896
+ if _cip_for_reader is not None else None
1897
+ _cascade_contract = getattr(_cip_for_reader, "contract", None) \
1898
+ if _cip_for_reader is not None else None
1899
+ try:
1900
+ cascade_db_reader = SpineDbReader(
1901
+ db_url, scenario=scenario_name,
1902
+ axis_enums=_cascade_axis_enums,
1903
+ contract=_cascade_contract,
1904
+ )
1905
+ except Exception: # noqa: BLE001
1906
+ cascade_db_reader = None
1907
+ if _drive_rec is not None:
1908
+ _drive_rec.checkpoint(
1909
+ "cascade_spinedb_reader_constructed", _drive_logger,
1910
+ user_label="Inputs prepared",
1911
+ )
1912
+
1913
+ # ``input/p_all_entity_unitsize`` — solve-invariant entity-unitsize
1914
+ # cascade (virtual_unitsize OR existing OR 1000.0) over ALL entities
1915
+ # (unit ∪ node ∪ connection). Computed ONCE here against the
1916
+ # whole-model (unfiltered-scenario) ``cascade_db_reader`` and seeded
1917
+ # into the cascade-input Provider, which every per-sub-solve Provider
1918
+ # copies frame-by-frame (see ``_native_run_model`` seed loop). This
1919
+ # (a) avoids the per-roll recompute in ``apply_derived_b`` and
1920
+ # (b) gives the output reader a complete carrier covering every
1921
+ # entity — including invest candidates absent from the LAST solve's
1922
+ # pss/invest sets — so ``read_parameters`` no longer KeyErrors on
1923
+ # such entities. ``_cip`` is the same Provider object forwarded onto
1924
+ # ``runner.state.cascade_input_provider`` above and read back by the
1925
+ # cascade as ``cascade_input_provider``.
1926
+ if cascade_db_reader is not None and _cip is not None:
1927
+ try:
1928
+ from flextool.engine_polars._derived_params import (
1929
+ _entity_unitsize_lf,
1930
+ )
1931
+ _all_us_df = (
1932
+ _entity_unitsize_lf(cascade_db_reader)
1933
+ .rename({"us": "value"})
1934
+ .collect()
1935
+ )
1936
+ if _all_us_df.height > 0:
1937
+ _cip.put("input/p_all_entity_unitsize", _all_us_df)
1938
+ except Exception: # noqa: BLE001
1939
+ # Non-fatal: the per-solve ``apply_derived_b`` fallback
1940
+ # recomputes ``p_all_entity_unitsize`` from ``source`` and
1941
+ # the output reader falls back to its reconstruction path.
1942
+ pass
1943
+
1944
+ class _PolarHighCascadeSolver(SolverRunner):
1945
+ def __init__(self, runner_state):
1946
+ super().__init__(runner_state)
1947
+ self._all_steps: dict[str, OrchestrationStep] = results
1948
+ # Δ.12d — warm-LP carry-over state. ``_warm_problem`` holds
1949
+ # the live :class:`polar_high.WarmProblem` reused across
1950
+ # consecutive structurally-compatible iterations; ``_prior_data``
1951
+ # / ``_prior_fp`` snapshot the previous iteration's FlexData +
1952
+ # fingerprint for the diff scan in
1953
+ # :func:`_apply_warm_updates`. All three stay None when
1954
+ # ``warm=False`` (the existing cold-cascade behaviour) AND
1955
+ # are reset to None on every cold rebuild.
1956
+ self._warm_problem: "WarmProblem | None" = None
1957
+ self._prior_data = None
1958
+ self._prior_fp: "tuple | None" = None
1959
+ # Per-iter slim of the PRIOR step's parked Solution — see the
1960
+ # block just before ``self._all_steps[step_key] = ...`` in
1961
+ # :meth:`run`. Tracks the step_key parked on the previous
1962
+ # iter so we can null its heavy ``_vars`` + ``highs`` once the
1963
+ # per-iter writers and ``build_handoff_from_solution`` have
1964
+ # finished consuming it. Bounds peak RSS during the cascade
1965
+ # — without this, every iter's full ``Var.frame`` dataframe
1966
+ # set stays parked until the post-loop slim at the bottom of
1967
+ # :func:`_native_run_model`, which on multi-roll runs is too
1968
+ # late (storage→dispatch OOMs).
1969
+ self._prev_step_key: "str | None" = None
1970
+ # Phase 2 — per-step level_key sidecar. Populated when
1971
+ # parking each step so the warm-path "keep one
1972
+ # ``Solution.highs`` + one ``flex_data_provider`` per level"
1973
+ # slim can iterate prior steps and resolve their level.
1974
+ # Keyed by the same ``step_key`` used in ``self._all_steps``.
1975
+ self._step_level_keys: "dict[str, tuple]" = {}
1976
+ # Per-base-solve gating for the scaling CSV. The CSV value
1977
+ # (effective_obj_scale) is invariant across rolls of the same
1978
+ # base solve, so we track which base solve names already have
1979
+ # ``scale_the_objective.csv`` written and skip subsequent
1980
+ # rolls. Phase 2b dropped the legacy diagnostic TXT report
1981
+ # (``FLEXTOOL_SCALING_REPORT=1``); the autoscaler's per-solve
1982
+ # YAML report (``solve_data/autoscale_<solve>.yaml``) is the
1983
+ # replacement and is gated inside
1984
+ # :func:`_autoscale_emit_layer1`.
1985
+ self._scale_csv_written: set[str] = set()
1986
+ # Autoscale console summary dedup: one line per base solve.
1987
+ # Layer 1/2/3 decisions are identical across rolls of the
1988
+ # same base solve, so the operator-facing summary fires once.
1989
+ self._autoscale_summary_emitted: set[str] = set()
1990
+ # Warm-path autoscale plan cache: Layer 2 writes side
1991
+ # vectors on the Problem at first-build; the WarmProblem's
1992
+ # canonical matrix bakes them in (and ``_param_cells``
1993
+ # caches the scaled factors for tracked Params). The plan
1994
+ # stays valid across subsequent ``_apply_warm_updates``
1995
+ # reuses because ``WarmProblem.update_param`` updates HiGHS
1996
+ # cells via the cached factors — no re-scaling, no plan
1997
+ # re-evaluation. The plan is needed on every warm solve so
1998
+ # :func:`_autoscale_unscale_post_solve` can restore the
1999
+ # solution to physical coordinates. Cleared whenever
2000
+ # ``self._warm_problem`` is dropped.
2001
+ self._autoscale_warm_layer2_plan: "_AutoscaleLayer2Plan | None" = None
2002
+ # Cache of the pre-Layer-2 RangeReport from the warm first
2003
+ # build, surfaced as ``Solution.streamed_lp_ranges`` after
2004
+ # every warm solve so downstream telemetry (the LP-bound-
2005
+ # range smoke in ``test_invest_chain_regression``, the
2006
+ # autoscale Layer-1 YAML) sees the same four (min, max)
2007
+ # pairs the autoscaler decided on. Mirrors the cold path's
2008
+ # post-``run_one_solve`` ``sol.streamed_lp_ranges = …``
2009
+ # assignment (see the longer comment around the cold-path
2010
+ # write below); dropped together with the warm problem.
2011
+ self._autoscale_warm_ranges_pre: (
2012
+ "_AutoscaleRangeReport | None"
2013
+ ) = None
2014
+ # Per-structural-shape autoscale DECISION cache. Keyed by the
2015
+ # BUILT LP's structural signature
2016
+ # (:func:`_autoscale_lp_shape_signature` — matrix shape +
2017
+ # per-family layout, scoped by base solve name), which stays
2018
+ # invariant across rolls of a rolling solve where
2019
+ # ``_fingerprint(data)`` would slide; each
2020
+ # value is an :class:`_AutoscaleShapeCacheEntry` carrying the
2021
+ # Layer-2 exponents, the Layer-3 plan, and the pre/post
2022
+ # RangeReports computed on the FIRST solve of that shape. On
2023
+ # every subsequent same-shape solve (notably the COLD-rebuild-
2024
+ # per-roll path where ladder Params force a cold rebuild yet
2025
+ # the matrix shape is invariant) the cached decision is
2026
+ # re-applied via :func:`_autoscale_apply_layer2_from_cache` /
2027
+ # :func:`_autoscale_apply_layer3_from_cache` WITHOUT any
2028
+ # ``detect_ranges`` / ``bucket_coefficients`` traversal — that
2029
+ # traversal was the source of the per-roll multi-GB transient
2030
+ # peaks (the autoscale memory pyramid). Disable with
2031
+ # ``FLEXTOOL_DISABLE_AUTOSCALE_CACHE=1`` (always recompute).
2032
+ self._autoscale_shape_cache: (
2033
+ "dict[tuple, _AutoscaleShapeCacheEntry]"
2034
+ ) = {}
2035
+ # Phase 4 Step 4b — in-process warm-start basis cache. Keyed
2036
+ # by the built LP's name-set fingerprint
2037
+ # (``Problem.basis_name_fingerprint``); each value is a
2038
+ # :class:`polar_high.NamedBasis` captured off a fresh build's
2039
+ # optimal solve and re-injected into an identically-named fresh
2040
+ # build on a later cross-run solve (UC2 sweep / UC3 resume).
2041
+ # Populated only under ``FLEXTOOL_WARM_START=1`` on the
2042
+ # in-process HiGHS fresh-build path and mirrored on disk as
2043
+ # ``<cache_dir>/<fp>.nbasis`` (JSON) alongside the subprocess
2044
+ # arm's ``<fp>.bas``. Stays empty + untouched when the opt-in
2045
+ # is off (byte-identical off-path).
2046
+ self._inproc_basis_cache: "dict[str, NamedBasis]" = {}
2047
+ # v60/v62 per-solve decomposition — complete-solve names that
2048
+ # ran under ``decomposition=benders``. Used by the consume-side
2049
+ # guard in :meth:`run` to raise loudly if a downstream solve
2050
+ # would try to consume a Benders solve's (absent) handoff —
2051
+ # cross-scheme handoff is a deferred follow-up.
2052
+ self._benders_solve_names: set[str] = set()
2053
+ # TIER 1 — complete-solve names whose Benders solve deposited a
2054
+ # USABLE investment handoff (non-empty
2055
+ # ``result.invest_solution_vars`` → ``realized_invest`` /
2056
+ # ``divest_cumulative`` carriers). A downstream solve may
2057
+ # consume these; the consume-side guard fires only for Benders
2058
+ # predecessors NOT in this set.
2059
+ self._benders_invest_handoff_names: set[str] = set()
2060
+
2061
+ def _run_benders_solve(
2062
+ self,
2063
+ complete_solve_name: str,
2064
+ base_solve_name: str,
2065
+ data,
2066
+ ) -> int:
2067
+ """Run *complete_solve_name* via the Benders region driver.
2068
+
2069
+ Selected per solve from ``solve.decomposition = benders``.
2070
+ Decomposes over the groups whose
2071
+ ``group.decomposition_method`` is ``benders_regional`` (the
2072
+ ``decomp_<REG>`` groups the PLEXOS-to-FlexTool writer emits),
2073
+ using the per-solve knobs resolved by
2074
+ :meth:`SolveConfig.benders_config_for`.
2075
+
2076
+ The solve runs and reports convergence/objective + the valid
2077
+ LB/UB sandwich, then (TIER 1) assembles its owner-selected
2078
+ whole-system invest/divest decisions
2079
+ (``result.invest_solution_vars``) into a :class:`SolveHandoff`
2080
+ and deposits it into ``state.handoffs`` so a DOWNSTREAM
2081
+ rolling-dispatch solve consumes the invested capacity. When
2082
+ the model has no investment the dict is empty, no handoff is
2083
+ built, and the deposited step carries a
2084
+ :class:`SnapshotSolution` with empty ``_vars``. Every Benders
2085
+ solve is recorded in ``self._benders_solve_names``; those that
2086
+ handed forward a usable invest handoff are also recorded in
2087
+ ``self._benders_invest_handoff_names`` so the (now narrowed)
2088
+ consume-side guard in :meth:`run` fires only for a downstream
2089
+ consumer of a Benders solve that produced no investment.
2090
+ """
2091
+ from flextool.engine_polars._benders import solve_benders
2092
+ from flextool.decomposition.region_filter import (
2093
+ discover_decomposition_regions_from_db,
2094
+ )
2095
+
2096
+ # Regions are discovered from the DB (the same source the
2097
+ # group-level decomposition_method lives in). The DB-driven
2098
+ # run path (run_chain_from_db) always supplies ``db_url``; a
2099
+ # caller that bypasses it cannot resolve the region groups, so
2100
+ # fail with an actionable message rather than guess.
2101
+ if db_url is None:
2102
+ raise FlexToolConfigError(
2103
+ f"Solve '{base_solve_name}' requests "
2104
+ f"decomposition=benders but the run was started "
2105
+ f"without a database URL (db_url is None). Benders "
2106
+ f"region decomposition is only available on the "
2107
+ f"DB-driven run path (run_chain_from_db)."
2108
+ )
2109
+ regions = discover_decomposition_regions_from_db(db_url)
2110
+ if len(regions) < 2:
2111
+ raise FlexToolConfigError(
2112
+ f"Solve '{base_solve_name}' requests "
2113
+ f"decomposition=benders but the model declares "
2114
+ f"{len(regions)} group(s) with "
2115
+ f"decomposition_method='benders_regional' "
2116
+ f"({regions or '(none)'}); at least two region groups "
2117
+ f"are required."
2118
+ )
2119
+
2120
+ max_iter, tol, in_out_weight_db = state.solve.benders_config_for(
2121
+ base_solve_name
2122
+ )
2123
+ # Resolve the objective scale the SAME way the monolithic path
2124
+ # does (the solve's ``scale_the_objective``; malformed/unset =>
2125
+ # 1e-6). Both the Benders master and the region subproblems
2126
+ # build at this scale so the cut coefficients stay consistent.
2127
+ _user_obj_scale = state.solve.scale_the_objective.get(
2128
+ complete_solve_name
2129
+ ) or state.solve.scale_the_objective.get(base_solve_name)
2130
+ obj_scale = _resolve_effective_obj_scale(_user_obj_scale)
2131
+
2132
+ # Benders progress is surfaced via ``print`` (flushed), the
2133
+ # same channel as the cascade's "Solve start" markers — the
2134
+ # per-solve logger is pinned to ERROR in ``_drive_cascade``, so
2135
+ # logger.info/.warning would be swallowed.
2136
+ _tag = f"[benders {base_solve_name}]"
2137
+
2138
+ # ``_emit`` is called from the main thread (iteration / summary
2139
+ # lines) AND from polar-high worker threads (the per-region
2140
+ # ``subsolve_callback`` fires from the parallel region pass), so
2141
+ # the lock keeps lines from interleaving mid-string.
2142
+ _emit_lock = threading.Lock()
2143
+
2144
+ def _emit(line: str) -> None:
2145
+ with _emit_lock:
2146
+ try:
2147
+ print(line, flush=True)
2148
+ except OSError:
2149
+ pass
2150
+
2151
+ # Put the config INSIDE the tag so it stays visible even though the
2152
+ # (long, non-wrapping) region list pushes the rest off-screen.
2153
+ _emit(
2154
+ f"[benders {base_solve_name}, max_iter={max_iter}, "
2155
+ f"tol={tol:g}, obj_scale={obj_scale:g}] "
2156
+ f"start: {len(regions)} regions {regions}"
2157
+ )
2158
+
2159
+ # MASTER-HOSTED node announcement (plan C8 / risk R10 tripwire):
2160
+ # balance/state nodes in NO region group are hosted natively in
2161
+ # the Benders MASTER, and that re-partition must never happen
2162
+ # silently — the driver's _logger.info is swallowed here (the
2163
+ # cascade pins per-solve loggers to ERROR), so surface the list
2164
+ # on the same print channel as the other Benders lines. With
2165
+ # every node grouped the set is empty and NOTHING extra is
2166
+ # emitted (and the solve itself takes today's exact path — this
2167
+ # announcement adds no solves either way, it is derived from
2168
+ # the already-loaded input frames).
2169
+ from flextool.engine_polars._region_filter import (
2170
+ compute_master_hosted_nodes,
2171
+ load_region_membership,
2172
+ )
2173
+
2174
+ _master_hosted = sorted(compute_master_hosted_nodes(
2175
+ data, load_region_membership(data, regions)
2176
+ ))
2177
+ if _master_hosted:
2178
+ _emit(
2179
+ f"{_tag} Benders: {len(_master_hosted)} master-hosted "
2180
+ f"node(s): {_master_hosted}"
2181
+ )
2182
+
2183
+ # Live per-iteration callback — one line as each outer Benders
2184
+ # iteration completes, reporting the valid LB/UB sandwich + gap
2185
+ # (all in REAL units, ÷s).
2186
+ def _on_iteration(entry: dict) -> None:
2187
+ _emit(
2188
+ f"{_tag} iter {entry['iter']:>3}/{max_iter}: "
2189
+ f"LB={entry['lower_bound']:.6g} "
2190
+ f"UB={entry['best_upper_bound']:.6g} "
2191
+ f"gap={entry['gap']:.4g}"
2192
+ )
2193
+
2194
+ # Per-region sub-solve FINISH line — replaces the raw per-sub-solve
2195
+ # HiGHS banner with a useful "which region finished" log. Fires
2196
+ # from polar-high worker threads (parallel region pass); ``_emit``
2197
+ # is lock-guarded. (``iter`` 0 is the autarkic bootstrap pass.)
2198
+ def _on_subsolve(entry: dict) -> None:
2199
+ _emit(
2200
+ f"{_tag} iter {entry['iter']:>3}: "
2201
+ f"{entry['region']} done (obj={entry['obj']:.6g})"
2202
+ )
2203
+
2204
+ result = solve_benders(
2205
+ data,
2206
+ regions,
2207
+ max_iters=max_iter,
2208
+ tol=tol,
2209
+ monolith_objective=None,
2210
+ scale_the_objective=obj_scale,
2211
+ in_out_weight=in_out_weight_db,
2212
+ progress_callback=_on_iteration,
2213
+ subsolve_callback=_on_subsolve,
2214
+ )
2215
+
2216
+ # Final summary — convergence + the valid LB/UB sandwich that
2217
+ # is the headline benefit of Benders over the subgradient path
2218
+ # (a certified lower bound, not just a dual estimate).
2219
+ _status = "CONVERGED" if result.converged else "DID NOT CONVERGE"
2220
+ _lb = result.lower_bound
2221
+ _ub = result.upper_bound
2222
+ _emit(
2223
+ f"{_tag} {_status} after {result.iterations}/{max_iter} "
2224
+ f"iterations"
2225
+ )
2226
+ _emit(f"{_tag} lower bound (valid) = {_lb:.6g}")
2227
+ _emit(f"{_tag} upper bound (best) = {_ub:.6g}")
2228
+ _emit(
2229
+ f"{_tag} optimality gap = {result.gap:.4g} "
2230
+ f"(relative, (UB-LB)/|UB|)"
2231
+ )
2232
+ if result.region_costs:
2233
+ _rc = ", ".join(
2234
+ f"{r}={c:.6g}"
2235
+ for r, c in result.region_costs.items()
2236
+ )
2237
+ _emit(f"{_tag} region costs: {_rc}")
2238
+ if not result.converged:
2239
+ # Also log at ERROR so non-convergence still surfaces in
2240
+ # error-only contexts (CI, log scrapers); the chain is not
2241
+ # aborted — the decomposition produced a usable point.
2242
+ self.state.logger.error(
2243
+ "Solve '%s': Benders decomposition did NOT converge "
2244
+ "after %d/%d iterations (gap %.4g).",
2245
+ base_solve_name, result.iterations, max_iter, result.gap,
2246
+ )
2247
+
2248
+ # TIER 1 — turn the assembled, owner-selected whole-system
2249
+ # invest/divest frames into a handoff so a DOWNSTREAM rolling-
2250
+ # dispatch solve consumes this Benders solve's invested
2251
+ # capacity. ``result.invest_solution_vars`` is a dict keyed by
2252
+ # ``v_invest_p`` / ``v_invest_n`` / ``v_divest_p`` /
2253
+ # ``v_divest_n`` to long-form (entity, d, value) frames whose
2254
+ # columns match ``polar_high.Solution.value(name)`` exactly;
2255
+ # it is ``{}`` when the model has no investment.
2256
+ from flextool.engine_polars.input import (
2257
+ build_handoff_from_solution,
2258
+ )
2259
+
2260
+ self._benders_solve_names.add(complete_solve_name)
2261
+
2262
+ # Carrier: a SnapshotSolution duck-types the ``_vars``
2263
+ # membership + ``.value(name)`` contract the invest/divest
2264
+ # extraction in ``build_handoff_from_solution`` needs (it
2265
+ # reads ONLY ``name in sol._vars`` and ``sol.value(name)`` on
2266
+ # that path — no Var internals, no ``col_value``).
2267
+ snap = SnapshotSolution(_vars=dict(result.invest_solution_vars))
2268
+
2269
+ handoff = None
2270
+ if result.invest_solution_vars:
2271
+ # Mirror the monolithic path's prior-handoff resolution
2272
+ # (the predecessor this solve loaded), so the cumulative
2273
+ # invest/existing carriers chain forward correctly.
2274
+ _prior = (
2275
+ self.state.handoffs.get(self.state.last_captured_solve)
2276
+ if self.state.last_captured_solve is not None else None
2277
+ )
2278
+ # CRITICAL: pass ``flex_data=None``. With a real
2279
+ # ``flex_data`` the builder also runs the co2 /
2280
+ # cumulative-commodity / fix_storage branches, the last of
2281
+ # which is gated on storage nodes (not on ``v_flow in
2282
+ # _vars``) and calls ``sol.constraint_dual(...)`` —
2283
+ # absent on SnapshotSolution → AttributeError. TIER 1
2284
+ # needs none of those derivations from the invest solve;
2285
+ # ``flex_data=None`` collapses them to prior-handoff
2286
+ # passthroughs and keeps the call on the pure invest/divest
2287
+ # path. ``provider`` is safe (it only feeds the
2288
+ # ``solve_data/`` CSV reads that source the invest periods
2289
+ # / entity sets, independent of ``sol``) and is REQUIRED
2290
+ # for the carriers to populate.
2291
+ handoff = build_handoff_from_solution(
2292
+ snap,
2293
+ self.state.paths.work_folder,
2294
+ complete_solve_name,
2295
+ prior_handoff=_prior,
2296
+ flex_data=None,
2297
+ parent_handoff=None,
2298
+ provider=getattr(
2299
+ self.state, "current_provider", None,
2300
+ ),
2301
+ )
2302
+ # Deposit so the next solve's load picks it up via the
2303
+ # ``last_captured_solve`` translation in _native_run_model
2304
+ # (which re-reads ``state.handoffs[last_captured_solve]``
2305
+ # after this method returns 0).
2306
+ self.state.handoffs[complete_solve_name] = handoff
2307
+ # Track that this Benders solve handed forward usable
2308
+ # investment so the narrowed consume-side guard does NOT
2309
+ # fire for a downstream consumer of it.
2310
+ self._benders_invest_handoff_names.add(complete_solve_name)
2311
+
2312
+ self._all_steps[complete_solve_name] = OrchestrationStep(
2313
+ solve_name=complete_solve_name,
2314
+ solution=snap,
2315
+ handoff=handoff,
2316
+ obj=result.total_objective,
2317
+ optimal=result.converged,
2318
+ warm_used=False,
2319
+ is_benders=True,
2320
+ benders_gap=result.gap,
2321
+ benders_tol=result.tol,
2322
+ benders_iterations=result.iterations,
2323
+ flex_data=data,
2324
+ flex_data_provider=getattr(
2325
+ self.state, "current_provider", None,
2326
+ ),
2327
+ )
2328
+ self._prev_step_key = complete_solve_name
2329
+ return 0
2330
+
2331
+ def run(self, complete_solve_name: str) -> int:
2332
+ _phase_prof("run_enter")
2333
+ # Cross-level eviction — release any EXHAUSTED prior solve-level's
2334
+ # live HiGHS instance + flex_data_provider BEFORE this solve builds
2335
+ # its FlexData/LP, so two level footprints never coexist (the DES
2336
+ # storage+dispatch ≈ 2x peak that drove the 7/9 near-OOM). This
2337
+ # hoists the post-solve slim's exhausted-level branch
2338
+ # (``:2632-2647``) ahead of the allocation instead of running it
2339
+ # after — the prior level's per-iter writers + handoff already
2340
+ # consumed its solution on its own iter, so releasing here is safe.
2341
+ # A level is "exhausted" when no upcoming iter shares its level_key
2342
+ # and it is not the level THIS solve belongs to; same-level steps
2343
+ # are kept for warm reuse. ``self._warm_problem`` for an exhausted
2344
+ # level was already nulled at the level boundary
2345
+ # (_native_run_model.py:497-505), so nulling the step's
2346
+ # ``solution.highs`` here drops the last reference and frees it.
2347
+ # malloc_trim reclaims the HiGHS (glibc) heap; the polars-side
2348
+ # provider is freed by dropping the Python ref.
2349
+ if not keep_solutions:
2350
+ _ilk = getattr(self.state, "_all_level_keys", ())
2351
+ _iidx = getattr(self.state, "_current_iter_index", None)
2352
+ _tlvl = getattr(self.state, "_current_level_key", None)
2353
+ _upcoming = (
2354
+ set(_ilk[_iidx + 1:])
2355
+ if (_iidx is not None and _ilk) else set()
2356
+ )
2357
+ # Disable knob for A/B peak-memory measurement and the
2358
+ # regression test's negative control (mirrors the
2359
+ # ``POLAR_HIGH_DISABLE_PRUNE_DOWN`` style escape hatch).
2360
+ if os.environ.get("FLEXTOOL_DISABLE_XLEVEL_RELEASE") != "1":
2361
+ _released = False
2362
+ for _k, _step in (getattr(self, "_all_steps", None) or {}).items():
2363
+ _lvl = self._step_level_keys.get(_k)
2364
+ if _lvl is None or _lvl == _tlvl or _lvl in _upcoming:
2365
+ continue # current level or still-upcoming: keep
2366
+ _sol = getattr(_step, "solution", None)
2367
+ if _sol is not None and getattr(_sol, "highs", None) is not None:
2368
+ _sol.highs = None
2369
+ _released = True
2370
+ if getattr(_step, "flex_data_provider", None) is not None:
2371
+ _step.flex_data_provider = None
2372
+ _released = True
2373
+ # Also evict the exhausted level's entry from the
2374
+ # per-level FlexDataProvider cache — it is keyed by
2375
+ # level_key and only reused by FUTURE same-level rolls,
2376
+ # of which an exhausted level has none. Without this
2377
+ # the cache (``state._level_providers``) pins the
2378
+ # level's polars FlexData for the whole cascade even
2379
+ # after the step ref above is dropped.
2380
+ _lp = getattr(self.state, "_level_providers", None)
2381
+ if isinstance(_lp, dict) and _lp.pop(_lvl, None) is not None:
2382
+ _released = True
2383
+ if _released:
2384
+ _try_malloc_trim()
2385
+ # Cross-level retention audit — runs AFTER the eviction above so
2386
+ # a correct release records zero violators. Gated by the same
2387
+ # ``not keep_solutions`` as the eviction: the release invariant
2388
+ # only applies when slimming is active (``keep_solutions=True``
2389
+ # deliberately retains every level's solution).
2390
+ _audit_prior_level_release(
2391
+ steps=getattr(self, "_all_steps", None),
2392
+ step_level_keys=getattr(self, "_step_level_keys", {}),
2393
+ all_level_keys=getattr(self.state, "_all_level_keys", ()),
2394
+ iter_idx=getattr(self.state, "_current_iter_index", None),
2395
+ this_level=getattr(self.state, "_current_level_key", None),
2396
+ complete_solve_name=complete_solve_name,
2397
+ )
2398
+ # Optional per-iter phase-timing (opt-in via env var). Emits
2399
+ # `per_iter` rows to the workdir's timings.csv covering
2400
+ # lp_build / solve / handoff and a warm_used marker. See
2401
+ # specs/warm_start_phase_breakdown_handoff.md.
2402
+ _phase_timing = (
2403
+ os.environ.get("FLEXTOOL_PHASE_TIMING") == "1"
2404
+ and getattr(self.state, "timing_recorder", None) is not None
2405
+ )
2406
+ _tr = self.state.timing_recorder if _phase_timing else None
2407
+ _roll_idx = getattr(self.state, "current_roll_index", "")
2408
+ if _roll_idx is None:
2409
+ _roll_idx = ""
2410
+ _t_build_start = time.perf_counter() if _phase_timing else 0.0
2411
+ # Δ.12 — wire ``handoff=`` through ``load_flextool`` so the
2412
+ # in-memory carriers from the prior solve flow into this
2413
+ # solve's FlexData directly. Replaces the previous
2414
+ # implicit dependency on flextool's per-solve preprocessing
2415
+ # rewriting ``solve_data/p_entity_*.csv`` between solves.
2416
+ # After Δ.12 the cascade reads these five carrier-derived
2417
+ # fields from the in-memory ``SolveHandoff`` rather than the
2418
+ # workdir CSVs:
2419
+ #
2420
+ # * ``p_entity_invested``
2421
+ # * ``p_entity_divested``
2422
+ # * ``p_entity_previously_invested_capacity``
2423
+ # * ``p_roll_continue_state``
2424
+ # * ``p_fix_storage_quantity``
2425
+ prior_for_load = (
2426
+ self.state.handoffs.get(self.state.last_captured_solve)
2427
+ if self.state.last_captured_solve is not None else None
2428
+ )
2429
+ _sub_solve_provider = getattr(
2430
+ self.state, "current_provider", None,
2431
+ )
2432
+ data = load_flextool(
2433
+ self.state.paths.work_folder,
2434
+ handoff=prior_for_load,
2435
+ db_reader=cascade_db_reader,
2436
+ provider=_sub_solve_provider,
2437
+ )
2438
+ # Release heap held by the broadcast cascade scratch frames.
2439
+ # On H2_trade y2050 this drops RSS ~1.6 GB / 41 %; expected
2440
+ # to scale with timeline size. No-op on non-glibc.
2441
+ _try_malloc_trim()
2442
+ # Memory checkpoint — fires on level-boundary iters (the
2443
+ # last roll of a roll group) so the recorded delta aggregates
2444
+ # across all rolls in the group. ``_native_run_model`` sets
2445
+ # the flag before each ``solver.run()`` call.
2446
+ _emit_phase = bool(getattr(
2447
+ self.state, "emit_phase_checkpoints_this_iter", False,
2448
+ ))
2449
+ _memrec_local = getattr(self.state, "_memory_recorder", None)
2450
+ if _memrec_local is not None and _emit_phase:
2451
+ _memrec_local.checkpoint(
2452
+ "load_flextool_end", self.state.logger,
2453
+ user_label="FlexData built",
2454
+ )
2455
+
2456
+ # --- LP scaling -------------------------------------------------
2457
+ # Phase 2b — the legacy ``scaling.analyze_solve`` /
2458
+ # ``ScaleTable`` / ``resolve_effective_scaling`` pipeline has
2459
+ # been retired in favour of the autoscale package; the
2460
+ # cascade now resolves the per-solve effective objective
2461
+ # scale directly from the user's DB override (defaulting to
2462
+ # the legacy 1e-6 when absent) and lets autoscale Layer 3
2463
+ # handle residual cost / bound magnitudes inside HiGHS.
2464
+ base_solve_name = re.sub(r"_roll_\d+$", "", complete_solve_name)
2465
+
2466
+ # v60/v62 per-solve decomposition routing -------------------
2467
+ # Consume-side guard: if the immediately-preceding captured
2468
+ # solve ran under decomposition=benders and THIS solve is a
2469
+ # different base solve, it would consume the (intentionally
2470
+ # absent) Benders handoff. Cross-scheme handoff
2471
+ # (Benders → monolithic dispatch) is a deferred follow-up,
2472
+ # so we fail loudly rather than silently load handoff=None.
2473
+ _guard_msg = _benders_consume_guard_message(
2474
+ base_solve_name,
2475
+ self.state.last_captured_solve,
2476
+ self._benders_solve_names,
2477
+ self._benders_invest_handoff_names,
2478
+ )
2479
+ if _guard_msg is not None:
2480
+ raise FlexToolConfigError(_guard_msg)
2481
+ # When this solve resolves to decomposition=benders, run it
2482
+ # through the Benders region coordinator instead of building
2483
+ # and solving a monolithic LP.
2484
+ if state.solve.decomposition_for(base_solve_name) == "benders":
2485
+ return self._run_benders_solve(
2486
+ complete_solve_name, base_solve_name, data,
2487
+ )
2488
+
2489
+ user_obj_scale = state.solve.scale_the_objective.get(complete_solve_name)
2490
+ effective_obj_scale = _resolve_effective_obj_scale(user_obj_scale)
2491
+ # ``user_bound_scale`` resolution priority:
2492
+ # ``FLEXTOOL_USER_BOUND_SCALE`` env var (set by
2493
+ # ``--user-bound-scale`` CLI flag) > DB ``solve.user_bound_scale``
2494
+ # > autoscale Layer 3's automatic recommendation > HiGHS'
2495
+ # own internal scaling. HiGHS' "Consider setting the
2496
+ # user_bound_scale option to <N>" warning still prints a
2497
+ # value if any case slips through Layer 3; pass it via
2498
+ # ``--user-bound-scale``.
2499
+ _cli_ubs = os.environ.get("FLEXTOOL_USER_BOUND_SCALE")
2500
+ user_bound_scale_override = _resolve_user_bound_scale_override(
2501
+ _cli_ubs if _cli_ubs is not None
2502
+ else state.solve.user_bound_scale.get(complete_solve_name)
2503
+ )
2504
+ # Per-solve autoscaler mode (v64 ``solve.scaling``): unless the
2505
+ # operator pinned it via ``--scaling`` / ``FLEXTOOL_SCALING``
2506
+ # (captured in ``state.scaling_user_env`` before the loop), the
2507
+ # solve's DB value drives ``FLEXTOOL_SCALING``. Set it fresh
2508
+ # each sub-solve (default ``full`` when the solve does not author
2509
+ # it) so per-solve differences and the cascade-internal env
2510
+ # readers below stay consistent. Precedence: operator override >
2511
+ # ``solve.scaling`` > default ``full``.
2512
+ if getattr(state, "scaling_user_env", None) is None:
2513
+ _db_scaling = state.solve.scaling_for(base_solve_name)
2514
+ os.environ["FLEXTOOL_SCALING"] = _db_scaling or "full"
2515
+ # Resolve the autoscaler's mode once for this sub-solve so the
2516
+ # baseline-options builder, the cold/warm LP construction, and
2517
+ # the Layer 2 / Layer 3 helpers all see the same value. Cascade-
2518
+ # internal call sites read ``FLEXTOOL_SCALING`` from env.
2519
+ _scaling_cfg = _autoscale_resolve_config(None)
2520
+ _scaling_mode = _scaling_cfg.mode
2521
+
2522
+ # HiGHS solver options. ``simplex_scale_strategy`` =
2523
+ # advanced (Curtis-Reid) is always-on; ``user_bound_scale``
2524
+ # is only emitted when explicitly requested via
2525
+ # ``--user-bound-scale`` CLI / DB override (see priority
2526
+ # block above). Default: HiGHS does its own scaling.
2527
+ # Cap solve time via env var if the operator requested it.
2528
+ _diag_tlim = os.environ.get("FLEXTOOL_HIGHS_TIME_LIMIT")
2529
+ # Allow operator to override HiGHS ``mip_rel_gap`` via
2530
+ # ``--solver-mip-gap GAP`` CLI flag (env-var-plumbed). Only
2531
+ # bites MIP solves; pure-LP solves ignore it.
2532
+ _cli_mip_gap = os.environ.get("FLEXTOOL_HIGHS_MIP_GAP")
2533
+ # Allow operator to override HiGHS ``presolve`` via
2534
+ # ``--presolve {on,off,choose}`` CLI flag (env-var-plumbed).
2535
+ _cli_presolve = os.environ.get("FLEXTOOL_HIGHS_PRESOLVE")
2536
+ # Allow operator to override HiGHS ``threads`` via
2537
+ # ``--highs-threads N`` CLI flag (env-var-plumbed). N > 1
2538
+ # flips ``parallel`` to ``on`` and trades determinism for
2539
+ # wall-clock speedup; N == 1 keeps the DETERMINISM_OPTIONS
2540
+ # pinning intact. Resolving once per ``_finalise_highs_options``
2541
+ # call (= once per sub-solve) ensures every Highs instance in
2542
+ # the process sees the same value, sidestepping HiGHS'
2543
+ # "global scheduler already initialised" rejection path.
2544
+ _cli_threads = os.environ.get("FLEXTOOL_HIGHS_THREADS")
2545
+ # ``--solver-log-level {silent,normal,verbose}`` CLI flag,
2546
+ # env-var-plumbed. Replaces the v55-era ``solve.solver_log_level``
2547
+ # DB knob removed in Batch C.7. ``silent`` flips HiGHS'
2548
+ # ``output_flag`` off; ``verbose`` additionally bumps
2549
+ # ``log_dev_level=2`` for per-iteration solver telemetry.
2550
+ _cli_log_level = os.environ.get("FLEXTOOL_SOLVER_LOG_LEVEL")
2551
+
2552
+ def _build_cli_overrides() -> dict[str, object]:
2553
+ """Translate CLI env-var-plumbed flags into a HiGHS
2554
+ options dict that the effective-options resolver layers
2555
+ on top of ``solver_arguments`` and ``highs.opt``.
2556
+ """
2557
+ cli: dict[str, object] = {}
2558
+ if _diag_tlim:
2559
+ try:
2560
+ cli["time_limit"] = float(_diag_tlim)
2561
+ except ValueError:
2562
+ pass
2563
+ if _cli_mip_gap:
2564
+ try:
2565
+ cli["mip_rel_gap"] = float(_cli_mip_gap)
2566
+ except ValueError:
2567
+ pass
2568
+ if _cli_presolve in ("on", "off", "choose"):
2569
+ cli["presolve"] = _cli_presolve
2570
+ if _cli_threads is not None:
2571
+ try:
2572
+ n = int(_cli_threads)
2573
+ except ValueError:
2574
+ n = 1
2575
+ if n > 1:
2576
+ cli["threads"] = n
2577
+ # User opted out of the determinism pin; HiGHS
2578
+ # needs ``parallel="on"`` before it will actually
2579
+ # use the threads.
2580
+ cli["parallel"] = "on"
2581
+ # n == 1 (or n <= 0) keeps the deterministic defaults
2582
+ # from DETERMINISM_OPTIONS — no override needed.
2583
+ if _cli_log_level == "silent":
2584
+ cli["output_flag"] = False
2585
+ elif _cli_log_level == "verbose":
2586
+ cli["output_flag"] = True
2587
+ cli["log_dev_level"] = 2
2588
+ elif _cli_log_level == "normal":
2589
+ cli["output_flag"] = True
2590
+ # Anything else (None, unknown) leaves HiGHS defaults
2591
+ # standing — same as the pre-C.7 behaviour where the
2592
+ # DB-side knob fed nothing.
2593
+ return cli
2594
+
2595
+ # Per-solve ``solver_arguments`` 1d-map (Batch C.1). Empty
2596
+ # dict when no entry authored on the active solve.
2597
+ _solver_args_map = state.solve.solver_settings.arguments.get(
2598
+ complete_solve_name, {}
2599
+ )
2600
+ # ``solver_config/highs.opt`` floor parsed by the resolver.
2601
+ # ``run_chain_from_db`` seeds ``state.paths.solver_config_dir``
2602
+ # to the work-folder ``solver_config`` copy, so the CLI path
2603
+ # reads the floor; direct native callers that build a bare
2604
+ # ``PathConfig`` leave it None and the resolver treats that as
2605
+ # an empty floor.
2606
+ _highs_opt_path = (
2607
+ state.paths.solver_config_dir / "highs.opt"
2608
+ if state.paths.solver_config_dir is not None
2609
+ else None
2610
+ )
2611
+
2612
+ # --- LP build & solve ------------------------------------------
2613
+ # Δ.12d — warm-LP per-iteration decision. When ``warm`` is
2614
+ # True AND the prior iteration left a live WarmProblem whose
2615
+ # fingerprint matches this iteration's data, we push the
2616
+ # Param diff into the live LP. Any ``_IncompatibleUpdate``
2617
+ # (unmapped Param differs, gate transitions, …) drops back
2618
+ # to a cold rebuild. Cold rebuild also fires on the first
2619
+ # iteration and on any structural fingerprint mismatch.
2620
+ #
2621
+ # Phase 3 — warm-LP is a HiGHS-only design (polar-high's
2622
+ # WarmProblem wraps a single live HiGHS instance). When the
2623
+ # active solve picks a commercial solver we disable warm
2624
+ # reuse for this iteration, log a one-time warning, and
2625
+ # cold-rebuild + dispatch through ``run_one_solve``.
2626
+ from flextool.engine_polars._solve_config import (
2627
+ SolverConfig as _SolverConfig,
2628
+ )
2629
+ _active_solver_cfg = state.solve.solver_configs.get(
2630
+ complete_solve_name, _SolverConfig()
2631
+ )
2632
+ _warm_disabled_by_solver = (
2633
+ warm and _active_solver_cfg.name != "highs"
2634
+ )
2635
+ # ``FLEXTOOL_SAVE_MEMORY=1`` opts into polar-high's
2636
+ # ``save_memory=True`` solve path, which drops the polar-side
2637
+ # LP source and round-trips the HiGHS instance through MPS
2638
+ # mid-solve. The Problem is then in a "released" state and
2639
+ # can no longer be warm-reused, so every iteration must
2640
+ # cold-rebuild. Resolve once per sub-solve (cheap) so the
2641
+ # knob can be toggled between runs of the same cascade.
2642
+ _save_memory = os.environ.get("FLEXTOOL_SAVE_MEMORY") == "1"
2643
+ _warm_disabled_by_save_memory = warm and _save_memory
2644
+ if _warm_disabled_by_save_memory and not getattr(
2645
+ self, "_warm_disabled_by_save_memory_warned", False
2646
+ ):
2647
+ state.logger.warning(
2648
+ _wrap_log_prose(
2649
+ "FLEXTOOL_SAVE_MEMORY=1: warm-LP reuse disabled; "
2650
+ "every sub-solve will cold-rebuild, write MPS, and "
2651
+ "dispatch to a subprocess HiGHS. Expect ~+30-60 s "
2652
+ "I/O per sub-solve in exchange for HiGHS' "
2653
+ "active-solve memory living outside this Python "
2654
+ "process."
2655
+ ),
2656
+ )
2657
+ self._warm_disabled_by_save_memory_warned = True
2658
+ if _warm_disabled_by_solver and not getattr(
2659
+ self, "_warm_disabled_warned", False
2660
+ ):
2661
+ state.logger.warning(
2662
+ _wrap_log_prose(
2663
+ f"warm-start is unavailable for solver "
2664
+ f"{_active_solver_cfg.name!r}; falling back to cold "
2665
+ f"rebuilds per sub-solve, expect slower per-iter "
2666
+ f"wall-clock."
2667
+ ),
2668
+ )
2669
+ self._warm_disabled_warned = True
2670
+ # HiGHS soft-promote: warm=False on HiGHS without
2671
+ # FLEXTOOL_SAVE_MEMORY=1 used to fall through to an in-
2672
+ # process cold rebuild that built a fresh ``highspy.Highs``
2673
+ # inside this Python process — undoing the entire reason
2674
+ # warm reuse exists in the first place (peak RSS). Retired
2675
+ # path: route every HiGHS cold solve through the same
2676
+ # ``cmd_solve_mps`` subprocess the save-memory branch uses,
2677
+ # bounding peak RSS to ``write_mps``'s footprint. Mutate
2678
+ # only the local ``_save_memory`` — do NOT touch the env
2679
+ # var, which would leak to sibling solves.
2680
+ if (
2681
+ (not warm)
2682
+ and _active_solver_cfg.name == "highs"
2683
+ and not _save_memory
2684
+ ):
2685
+ if not getattr(
2686
+ self, "_warm_disabled_softpromote_warned", False,
2687
+ ):
2688
+ state.logger.warning(
2689
+ _wrap_log_prose(
2690
+ "Warm reuse disabled (warm=False); HiGHS solve "
2691
+ "will route through the cmd_solve_mps "
2692
+ "subprocess to bound memory footprint. Set "
2693
+ "FLEXTOOL_SAVE_MEMORY=1 explicitly to silence "
2694
+ "this warning."
2695
+ ),
2696
+ )
2697
+ self._warm_disabled_softpromote_warned = True
2698
+ _save_memory = True
2699
+ warm_used = False
2700
+ warm_active = (
2701
+ warm
2702
+ and not _warm_disabled_by_solver
2703
+ and not _warm_disabled_by_save_memory
2704
+ )
2705
+ if warm_active:
2706
+ fp = _fingerprint(data)
2707
+ # Phase 4 Step 4b bookkeeping — ``did_fresh_build`` gates
2708
+ # the post-solve basis CAPTURE to fresh builds only (a warm
2709
+ # reuse re-run is already optimally warm and its inner-
2710
+ # problem names are stale for rolling, so capturing there
2711
+ # would be wrong-keyed); ``fresh_fp`` carries the name-set
2712
+ # cache key from the inject block to the capture block.
2713
+ # Both are inert unless the basis-cache gate is active.
2714
+ did_fresh_build = False
2715
+ fresh_fp = None
2716
+ tried_warm = (
2717
+ self._warm_problem is not None
2718
+ and self._prior_data is not None
2719
+ and self._prior_fp == fp
2720
+ )
2721
+ if tried_warm:
2722
+ try:
2723
+ _apply_warm_updates(self._warm_problem,
2724
+ self._prior_data, data)
2725
+ warm_used = True
2726
+ except _IncompatibleUpdate as _warm_exc:
2727
+ # Drop the stale warm problem so the next
2728
+ # branch builds a fresh one. The cached Layer 2
2729
+ # plan dies with it — the next first-build will
2730
+ # regenerate it from the fresh LP.
2731
+ #
2732
+ # Diagnostic: surface WHICH condition forced the
2733
+ # cold rebuild. On rolling cascades the ladder
2734
+ # Params (commit 7b5ccb3e) trip this every roll,
2735
+ # which re-runs the full pre-solve autoscale
2736
+ # traversal (the between-solves memory pyramid).
2737
+ # The exception message names the offending
2738
+ # Param / reason; log it at WARNING so a single
2739
+ # run pins the cause without tracemalloc.
2740
+ state.logger.warning(
2741
+ "warm reuse fell back to COLD REBUILD for %s "
2742
+ "(re-runs pre-solve autoscale traversal): %s",
2743
+ complete_solve_name, _warm_exc,
2744
+ )
2745
+ self._warm_problem = None
2746
+ self._autoscale_warm_layer2_plan = None
2747
+ self._autoscale_warm_ranges_pre = None
2748
+ if not warm_used:
2749
+ # Cross-level eviction — same-level COLD-rebuild case
2750
+ # (e.g. ladder period switch, 7b5ccb3e). We are about to
2751
+ # build a fresh LP and are NOT warm-reusing, so no parked
2752
+ # step's HiGHS is a reuse source and ``self._warm_problem``
2753
+ # was just nulled above. Any prior step still holding a
2754
+ # live ``solution.highs`` — typically the previous
2755
+ # same-level roll; exhausted *other* levels were already
2756
+ # freed at the top of run() — is now the ONLY reference to
2757
+ # that HiGHS. Release it BEFORE ``_build_warm_problem``
2758
+ # allocates, else two same-level footprints coexist (the
2759
+ # DES 7/9 dispatch-on-dispatch stack). Per-iter writers +
2760
+ # handoff already consumed each prior step on its own iter,
2761
+ # so this is safe. ``flex_data_provider`` is NOT dropped
2762
+ # here — same-level rolls reuse the per-level provider
2763
+ # cache (``state._level_providers``).
2764
+ if (
2765
+ not keep_solutions
2766
+ and os.environ.get("FLEXTOOL_DISABLE_XLEVEL_RELEASE") != "1"
2767
+ ):
2768
+ _cr_released = False
2769
+ for _ck, _cstep in (getattr(self, "_all_steps", None) or {}).items():
2770
+ _csol = getattr(_cstep, "solution", None)
2771
+ if _csol is not None and getattr(_csol, "highs", None) is not None:
2772
+ _csol.highs = None
2773
+ _cr_released = True
2774
+ if _cr_released:
2775
+ _try_malloc_trim()
2776
+ if not keep_solutions:
2777
+ _audit_cold_rebuild_release(
2778
+ steps=getattr(self, "_all_steps", None),
2779
+ complete_solve_name=complete_solve_name,
2780
+ )
2781
+ # Build the warm problem first WITHOUT solver
2782
+ # options so we can inspect LP ranges, then push the
2783
+ # finalised HiGHS options through ``set_solver_options``
2784
+ # on the underlying Problem.
2785
+ _phase_prof("build_start")
2786
+ self._warm_problem = _build_warm_problem(
2787
+ data,
2788
+ scale_the_objective=effective_obj_scale,
2789
+ solver_options=None,
2790
+ )
2791
+ _phase_prof("build_done")
2792
+ if _memrec_local is not None and _emit_phase:
2793
+ _memrec_local.checkpoint(
2794
+ "lp_build_end", self.state.logger,
2795
+ user_label="Matrix built by polar-high",
2796
+ )
2797
+ inner_pb = self._warm_problem.problem
2798
+ from flextool.engine_polars._solver_dispatch import (
2799
+ _resolve_effective_highs_options,
2800
+ )
2801
+ highs_options = _resolve_effective_highs_options(
2802
+ solver_arguments_map=_solver_args_map,
2803
+ highs_opt_path=_highs_opt_path,
2804
+ cli_overrides=_build_cli_overrides(),
2805
+ baseline=_baseline_highs_options(
2806
+ user_bound_scale_override=user_bound_scale_override,
2807
+ scaling_mode=_scaling_mode,
2808
+ ),
2809
+ )
2810
+ inner_pb.set_solver_options(highs_options)
2811
+ # Autoscale Layer 2 + Layer 3 on the warm-active
2812
+ # first-build branch. Same call sequence as the
2813
+ # cold path below — see the longer-form comment
2814
+ # there for the rationale. Skipping these on warm
2815
+ # solves used to leave HiGHS staring at an unscaled
2816
+ # LP, costing both numerical health and ~tens of GB
2817
+ # of internal simplex working set on
2818
+ # poorly-conditioned LPs.
2819
+ #
2820
+ # First-build only: Layer 2 writes side vectors on
2821
+ # the Problem and ``WarmProblem._initial_build``
2822
+ # bakes them into the canonical matrix (with
2823
+ # ``_param_cells`` caching the scaled factors for
2824
+ # tracked Params). Subsequent
2825
+ # ``_apply_warm_updates`` Param mutations update
2826
+ # HiGHS coefficients via those cached factors — no
2827
+ # re-canonicalisation, no Layer 2 re-apply.
2828
+ # ``self._autoscale_warm_layer2_plan`` caches the
2829
+ # plan for use by
2830
+ # :func:`_autoscale_unscale_post_solve` after every
2831
+ # warm solve (first build AND reuses).
2832
+ #
2833
+ # Per-shape autoscale DECISION cache: keyed on the
2834
+ # BUILT LP's structural signature (matrix shape +
2835
+ # per-family layout), which is invariant across rolls
2836
+ # of the same rolling solve even though
2837
+ # ``_fingerprint(data)`` slides (a windowed period/dt
2838
+ # field's height tracks the rolling horizon). On a HIT
2839
+ # we replay the cached Layer-2 exponents + Layer-3 plan
2840
+ # WITHOUT any ``detect_ranges`` / ``bucket_coefficients``
2841
+ # walk (the per-roll multi-GB spike). A cold rebuild of
2842
+ # an already-seen shape (the ladder-Param
2843
+ # ``_IncompatibleUpdate`` path) therefore skips the
2844
+ # traversals entirely.
2845
+ _shape_key = _autoscale_lp_shape_signature(
2846
+ inner_pb, base_solve_name,
2847
+ )
2848
+ _cache_entry = (
2849
+ None if _autoscale_disable_cache()
2850
+ else self._autoscale_shape_cache.get(_shape_key)
2851
+ )
2852
+ _phase_prof("autoscale_l2_start")
2853
+ if _cache_entry is not None:
2854
+ # HIT — replay decision, no range walk.
2855
+ self._autoscale_warm_layer2_plan = (
2856
+ _autoscale_apply_layer2_from_cache(
2857
+ inner_pb, _cache_entry,
2858
+ solve_name=complete_solve_name,
2859
+ logger=self.state.logger,
2860
+ )
2861
+ )
2862
+ _autoscale_ranges_pre = _cache_entry.ranges_pre
2863
+ self._autoscale_warm_ranges_pre = _autoscale_ranges_pre
2864
+ _phase_prof("autoscale_l3_start")
2865
+ _autoscale_layer3_plan = (
2866
+ _autoscale_apply_layer3_from_cache(
2867
+ inner_pb, _cache_entry,
2868
+ solve_name=complete_solve_name,
2869
+ logger=self.state.logger,
2870
+ )
2871
+ )
2872
+ _autoscale_ranges_post = _cache_entry.ranges_post
2873
+ else:
2874
+ # MISS — run the full traversals, then cache.
2875
+ (
2876
+ self._autoscale_warm_layer2_plan,
2877
+ _autoscale_ranges_pre,
2878
+ ) = _autoscale_apply_layer2_pre_solve(
2879
+ inner_pb,
2880
+ solve_name=complete_solve_name,
2881
+ logger=self.state.logger,
2882
+ )
2883
+ # Cache the pre-Layer-2 RangeReport so subsequent
2884
+ # warm reuses can still attach it as
2885
+ # ``Solution.streamed_lp_ranges`` (the cascade only
2886
+ # builds the LP once, so the four ranges are
2887
+ # invariant across rolls of the same warm problem).
2888
+ self._autoscale_warm_ranges_pre = _autoscale_ranges_pre
2889
+ _phase_prof("autoscale_l3_start")
2890
+ _autoscale_layer3_plan = _autoscale_apply_layer3_pre_solve(
2891
+ inner_pb,
2892
+ layer2_plan=self._autoscale_warm_layer2_plan,
2893
+ solve_name=complete_solve_name,
2894
+ logger=self.state.logger,
2895
+ )
2896
+ _autoscale_ranges_post = None
2897
+ if not _autoscale_disable_cache():
2898
+ _l2p = self._autoscale_warm_layer2_plan
2899
+ self._autoscale_shape_cache[_shape_key] = (
2900
+ _AutoscaleShapeCacheEntry(
2901
+ layer2_exponents=(
2902
+ dict(_l2p.type_exponents)
2903
+ if _l2p is not None else None
2904
+ ),
2905
+ layer2_buckets_before=(
2906
+ dict(_l2p.type_buckets_before)
2907
+ if _l2p is not None else {}
2908
+ ),
2909
+ layer2_buckets_after=(
2910
+ dict(_l2p.type_buckets_after)
2911
+ if _l2p is not None else {}
2912
+ ),
2913
+ layer3_plan=_autoscale_layer3_plan,
2914
+ ranges_pre=_autoscale_ranges_pre,
2915
+ ranges_post=None,
2916
+ )
2917
+ )
2918
+ _phase_prof("autoscale_summary_start")
2919
+ _autoscale_emit_console_summary(
2920
+ ranges_pre=_autoscale_ranges_pre,
2921
+ ranges_post=_autoscale_ranges_post,
2922
+ layer2_plan=self._autoscale_warm_layer2_plan,
2923
+ layer3_plan=_autoscale_layer3_plan,
2924
+ solve_name=base_solve_name,
2925
+ already_emitted=self._autoscale_summary_emitted,
2926
+ memrec=_memrec_local if _emit_phase else None,
2927
+ logger=self.state.logger,
2928
+ )
2929
+ _phase_prof("autoscale_done")
2930
+ # Phase 4 Step 4b — in-process warm-start basis
2931
+ # INJECT (fresh build only). Keyed on the LP name-set
2932
+ # fingerprint, which is Layer-2-invariant, so it is
2933
+ # computed HERE — after Layer 2 — to reuse the memoized
2934
+ # ``canonicalise`` rather than assemble the names
2935
+ # twice. On a cache hit (in-process dict first, else
2936
+ # the on-disk ``<fp>.nbasis``) we record the basis on
2937
+ # the WarmProblem; the actual ``setBasis`` fires once
2938
+ # in ``_initial_build`` before the first solve and
2939
+ # safely falls back to a cold solve on any fingerprint
2940
+ # mismatch. Warm-start must NEVER break a solve, so
2941
+ # every step here is defensive — any failure logs and
2942
+ # continues cold.
2943
+ did_fresh_build = True
2944
+ if _basis_cache_active(
2945
+ state.solve.decomposition_for(base_solve_name),
2946
+ _active_solver_cfg.name,
2947
+ ):
2948
+ try:
2949
+ fresh_fp = inner_pb.basis_name_fingerprint()
2950
+ if fresh_fp:
2951
+ nb = self._inproc_basis_cache.get(fresh_fp)
2952
+ if nb is None:
2953
+ _bc_dir = _basis_cache_dir(
2954
+ self.state.paths.work_folder,
2955
+ )
2956
+ _nb_path = (
2957
+ _bc_dir / f"{fresh_fp}.nbasis"
2958
+ )
2959
+ if _nb_path.exists():
2960
+ _raw = json.loads(
2961
+ _nb_path.read_text()
2962
+ )
2963
+ nb = NamedBasis(
2964
+ col_status=_raw["col_status"],
2965
+ row_status=_raw["row_status"],
2966
+ fingerprint=_raw["fingerprint"],
2967
+ )
2968
+ if nb is not None:
2969
+ self._warm_problem.set_named_basis(
2970
+ nb, policy="exact",
2971
+ )
2972
+ self.state.logger.info(
2973
+ "in-process warm-basis cache hit "
2974
+ "%s", fresh_fp,
2975
+ )
2976
+ except Exception as _basis_exc: # noqa: BLE001
2977
+ self.state.logger.warning(
2978
+ "in-process warm-basis inject skipped "
2979
+ "(%s); solving cold", _basis_exc,
2980
+ )
2981
+ # ``WarmProblem.solve`` always keeps the HiGHS instance
2982
+ # alive on ``Solution.highs`` — that's the whole point
2983
+ # of warm reuse — so the output writer adapter
2984
+ # (``write_all_variables`` / ``write_all_handoffs``)
2985
+ # sees the live solver as it does for cold rebuilds
2986
+ # under ``keep_solver=True``. No extra kwarg required.
2987
+ # Blank line so HiGHS' "Running HiGHS …" banner (and the
2988
+ # grey solver-output block in the GUI) separates from the
2989
+ # scaling/LP-build rows above it.
2990
+ print("", flush=True)
2991
+ # Echo the effective solver options right before HiGHS'
2992
+ # own banner: the full merged, post-precedence set the
2993
+ # solver actually received (``highs.opt`` floor ∪ DB
2994
+ # ``solver_arguments`` ∪ CLI overrides), so the run is
2995
+ # replicable from the log alone. Resolved with
2996
+ # ``baseline=None`` so the engine-internal determinism /
2997
+ # scale keys are excluded — only what the operator
2998
+ # touched is shown. Uses ``print`` (not the logger) to
2999
+ # sit in the same stdout stream as the blank separator
3000
+ # above and HiGHS' native banner below — the GUI's
3001
+ # ``execution_window`` parses these printed markers.
3002
+ from flextool.engine_polars._solver_dispatch import (
3003
+ _resolve_effective_highs_options,
3004
+ )
3005
+ _eff_touched = _resolve_effective_highs_options(
3006
+ solver_arguments_map=_solver_args_map,
3007
+ highs_opt_path=_highs_opt_path,
3008
+ cli_overrides=_build_cli_overrides(),
3009
+ baseline=None,
3010
+ )
3011
+ _sa_line = _format_solver_args_line(
3012
+ _eff_touched, prefer_first=list(_solver_args_map or {}),
3013
+ )
3014
+ if _sa_line:
3015
+ print(_sa_line, flush=True)
3016
+ _t_solve_start = (
3017
+ time.perf_counter() if _phase_timing else 0.0
3018
+ )
3019
+ _phase_prof("solve_start")
3020
+ sol = self._warm_problem.solve()
3021
+ _t_solve_end = (
3022
+ time.perf_counter() if _phase_timing else 0.0
3023
+ )
3024
+ _phase_prof("after_solve")
3025
+ # Attach the cached pre-Layer-2 RangeReport as
3026
+ # ``streamed_lp_ranges`` on the warm Solution so the
3027
+ # Layer-1 emit hook and downstream telemetry
3028
+ # (e.g. ``test_invest_chain_lp_bound_range_smoke``) see
3029
+ # the four (min, max) pairs the autoscaler decided on.
3030
+ # ``WarmProblem.solve`` returns a bare Solution with no
3031
+ # streamed ranges of its own — the polar-high in-process
3032
+ # streaming-solve path that populates that attribute is
3033
+ # bypassed when HiGHS' ``Highs.run`` is invoked directly
3034
+ # on the live instance, so the warm path has to surface
3035
+ # the ranges itself. Mirrors the cold-path assignment
3036
+ # below (search for ``sol.streamed_lp_ranges = {``).
3037
+ _warm_ranges = self._autoscale_warm_ranges_pre
3038
+ if (
3039
+ _warm_ranges is not None
3040
+ and getattr(sol, "streamed_lp_ranges", None) is None
3041
+ ):
3042
+ try:
3043
+ sol.streamed_lp_ranges = {
3044
+ "matrix": _warm_ranges.matrix,
3045
+ "cost": _warm_ranges.cost,
3046
+ "col_bound": _warm_ranges.bound,
3047
+ "row_bound": _warm_ranges.rhs,
3048
+ }
3049
+ except Exception: # pragma: no cover — Solution may
3050
+ # forbid the assignment in a future polar-high
3051
+ pass
3052
+ # Eager unscale on every warm solve (first-build AND
3053
+ # reuses) so output writers see physical-coordinate
3054
+ # primal / duals / reduced costs. No-op when the
3055
+ # cached plan is None (Layer 2 was off or didn't
3056
+ # trigger at first-build).
3057
+ _phase_prof("unscale_start")
3058
+ if self._autoscale_warm_layer2_plan is not None:
3059
+ _autoscale_unscale_post_solve(
3060
+ sol, self._autoscale_warm_layer2_plan,
3061
+ solve_name=complete_solve_name,
3062
+ logger=self.state.logger,
3063
+ )
3064
+ _phase_prof("unscale_done")
3065
+ # Phase 4 Step 4b — in-process warm-start basis CAPTURE
3066
+ # (fresh build only). ``sol.highs`` is the live retained
3067
+ # handle (``WarmProblem.solve`` always keeps the solver),
3068
+ # so ``get_named_basis`` succeeds. Persist atomically to
3069
+ # the shared cache dir so a later cross-run solve of the
3070
+ # same structural model can inject it (UC2 sweep / UC3
3071
+ # resume). Never capture on the warm-reuse branch
3072
+ # (``did_fresh_build`` stays False there) — those inner
3073
+ # names are stale and the LP is already optimally warm.
3074
+ # Capture failure is non-fatal (log + continue).
3075
+ if (
3076
+ did_fresh_build
3077
+ and fresh_fp
3078
+ and _basis_cache_active(
3079
+ state.solve.decomposition_for(base_solve_name),
3080
+ _active_solver_cfg.name,
3081
+ )
3082
+ ):
3083
+ try:
3084
+ nb_out = sol.get_named_basis()
3085
+ self._inproc_basis_cache[fresh_fp] = nb_out
3086
+ _bc_dir = _basis_cache_dir(
3087
+ self.state.paths.work_folder,
3088
+ )
3089
+ _tmp = (
3090
+ _bc_dir
3091
+ / f"{fresh_fp}.nbasis.tmp.{os.getpid()}"
3092
+ )
3093
+ _tmp.write_text(json.dumps({
3094
+ "col_status": nb_out.col_status,
3095
+ "row_status": nb_out.row_status,
3096
+ "fingerprint": nb_out.fingerprint,
3097
+ }))
3098
+ os.replace(
3099
+ _tmp, _bc_dir / f"{fresh_fp}.nbasis",
3100
+ )
3101
+ self.state.logger.info(
3102
+ "in-process warm-basis cache captured %s",
3103
+ fresh_fp,
3104
+ )
3105
+ except Exception as _basis_exc: # noqa: BLE001
3106
+ self.state.logger.warning(
3107
+ "in-process warm-basis capture skipped (%s)",
3108
+ _basis_exc,
3109
+ )
3110
+ self._prior_data = data
3111
+ self._prior_fp = fp
3112
+ else:
3113
+ pb = Problem()
3114
+ _phase_prof("build_start")
3115
+ build_flextool(pb, data, scale_the_objective=effective_obj_scale)
3116
+ _phase_prof("build_done")
3117
+ if _memrec_local is not None and _emit_phase:
3118
+ _memrec_local.checkpoint(
3119
+ "lp_build_end", self.state.logger,
3120
+ user_label="Matrix built by polar-high",
3121
+ )
3122
+ from flextool.engine_polars._solver_dispatch import (
3123
+ _resolve_effective_highs_options,
3124
+ )
3125
+ highs_options = _resolve_effective_highs_options(
3126
+ solver_arguments_map=_solver_args_map,
3127
+ highs_opt_path=_highs_opt_path,
3128
+ cli_overrides=_build_cli_overrides(),
3129
+ baseline=_baseline_highs_options(
3130
+ user_bound_scale_override=user_bound_scale_override,
3131
+ scaling_mode=_scaling_mode,
3132
+ ),
3133
+ )
3134
+ pb.set_solver_options(highs_options)
3135
+ # ── DIAGNOSTIC: per-substep RSS in the pre-write_mps gap ──
3136
+ # OOM in this gap (post "Matrix built", pre write_mps) is
3137
+ # invisible to both ``_MemoryRecorder`` (single checkpoint
3138
+ # for "Matrix built") and ``POLAR_HIGH_WRITE_MPS_PROFILE``
3139
+ # (only fires inside write_mps). This closure samples
3140
+ # ``psutil.Process().memory_info().rss`` at each substep
3141
+ # below and writes to stderr in the same format as the
3142
+ # polar-high profile. Activate with
3143
+ # ``FLEXTOOL_AUTOSCALE_PROFILE=1``; zero overhead when off.
3144
+ _autoscale_profile = (
3145
+ os.environ.get("FLEXTOOL_AUTOSCALE_PROFILE") == "1"
3146
+ )
3147
+ if _autoscale_profile:
3148
+ try:
3149
+ import psutil as _psutil
3150
+ _ap_proc = _psutil.Process()
3151
+ _ap_t0 = time.monotonic()
3152
+ _ap_prev = _ap_proc.memory_info().rss / (1024 ** 3)
3153
+ import sys as _sys
3154
+ def _ap(phase: str, **extras) -> None:
3155
+ nonlocal _ap_prev
3156
+ rss = _ap_proc.memory_info().rss / (1024 ** 3)
3157
+ delta = rss - _ap_prev
3158
+ wall = time.monotonic() - _ap_t0
3159
+ sign = "+" if delta >= 0 else ""
3160
+ extras_str = "\t".join(
3161
+ f"{k}={v}" for k, v in extras.items()
3162
+ )
3163
+ print(
3164
+ f"[autoscale profile]\tphase={phase}\t"
3165
+ f"rss_gb={rss:.2f}\tdelta_gb={sign}{delta:.2f}"
3166
+ f"\twall_s={wall:.2f}"
3167
+ + (f"\t{extras_str}" if extras_str else ""),
3168
+ file=_sys.stderr, flush=True,
3169
+ )
3170
+ _ap_prev = rss
3171
+ _ap("enter")
3172
+ except ImportError:
3173
+ _autoscale_profile = False
3174
+ print(
3175
+ "FLEXTOOL_AUTOSCALE_PROFILE=1 but psutil not "
3176
+ "installed; profiling disabled.",
3177
+ file=__import__("sys").stderr, flush=True,
3178
+ )
3179
+ # autoscale Layer 2 (semantic per-type) pre-solve apply.
3180
+ # Trigger gate is the same Layer-1 four-range readout —
3181
+ # see ``_autoscale_apply_layer2_pre_solve``. Plan is
3182
+ # consumed by ``_autoscale_unscale_post_solve`` once the
3183
+ # solve returns so downstream output writers see the
3184
+ # un-scaled solution.
3185
+ # Per-shape autoscale DECISION cache (cold ``warm=False``
3186
+ # / save_memory path). Keyed on the BUILT LP's structural
3187
+ # signature (matrix shape + per-family layout), which is
3188
+ # invariant across rolls of the same rolling solve even
3189
+ # though ``_fingerprint(data)`` slides as the rolling
3190
+ # horizon moves. Cheap to compute (no coefficient walk) so
3191
+ # repeated same-shape cold solves replay the cached decision
3192
+ # without the per-roll ``detect_ranges`` /
3193
+ # ``bucket_coefficients`` traversal. Honours
3194
+ # ``FLEXTOOL_DISABLE_AUTOSCALE_CACHE=1``.
3195
+ _cold_key = (
3196
+ None if _autoscale_disable_cache()
3197
+ else _autoscale_lp_shape_signature(pb, base_solve_name)
3198
+ )
3199
+ _cache_entry = (
3200
+ self._autoscale_shape_cache.get(_cold_key)
3201
+ if _cold_key is not None else None
3202
+ )
3203
+ _phase_prof("autoscale_l2_start")
3204
+ if _cache_entry is not None:
3205
+ # HIT — replay decision, NO range walk.
3206
+ _autoscale_layer2_plan = _autoscale_apply_layer2_from_cache(
3207
+ pb, _cache_entry,
3208
+ solve_name=complete_solve_name,
3209
+ logger=self.state.logger,
3210
+ )
3211
+ _autoscale_ranges_pre = _cache_entry.ranges_pre
3212
+ if _autoscale_profile:
3213
+ _ap("layer2_applied_cached",
3214
+ n_cstrs=len(pb._cstrs),
3215
+ n_vars=len(pb._vars))
3216
+ _phase_prof("autoscale_l3_start")
3217
+ _autoscale_layer3_plan = _autoscale_apply_layer3_from_cache(
3218
+ pb, _cache_entry,
3219
+ solve_name=complete_solve_name,
3220
+ logger=self.state.logger,
3221
+ )
3222
+ if _autoscale_profile:
3223
+ _ap("layer3_applied_cached")
3224
+ _phase_prof("autoscale_summary_start")
3225
+ _autoscale_ranges_post = _cache_entry.ranges_post
3226
+ if _autoscale_profile:
3227
+ _ap("ranges_post_cached",
3228
+ ranges_post_ran=str(_autoscale_ranges_post is not None))
3229
+ else:
3230
+ # MISS — run the full traversals, then cache.
3231
+ (
3232
+ _autoscale_layer2_plan,
3233
+ _autoscale_ranges_pre,
3234
+ ) = _autoscale_apply_layer2_pre_solve(
3235
+ pb,
3236
+ solve_name=complete_solve_name,
3237
+ logger=self.state.logger,
3238
+ )
3239
+ if _autoscale_profile:
3240
+ _ap("layer2_applied",
3241
+ n_cstrs=len(pb._cstrs),
3242
+ n_vars=len(pb._vars))
3243
+ # Layer 3 (HiGHS-native top-up): set user_objective_scale,
3244
+ # user_bound_scale, and simplex_scale_strategy from the
3245
+ # post-Layer-2 ranges so HiGHS sees a final LP that is
3246
+ # already inside its comfort zone. Layer 3 is HiGHS-
3247
+ # internal (no inverse transform on the solution); the
3248
+ # writeModel MPS export remains unscaled.
3249
+ _phase_prof("autoscale_l3_start")
3250
+ _autoscale_layer3_plan = _autoscale_apply_layer3_pre_solve(
3251
+ pb,
3252
+ layer2_plan=_autoscale_layer2_plan,
3253
+ solve_name=complete_solve_name,
3254
+ logger=self.state.logger,
3255
+ )
3256
+ if _autoscale_profile:
3257
+ _ap("layer3_applied")
3258
+ # Console summary: one user-visible line per base solve
3259
+ # describing the autoscaler's pre/post ranges and the
3260
+ # Layer 2 / Layer 3 decisions. Read post-Layer-2 ranges
3261
+ # from the (mutated) Problem so the "after" view reflects
3262
+ # what HiGHS will see; Layer 3 acts inside HiGHS so its
3263
+ # values are surfaced separately in the same line.
3264
+ _phase_prof("autoscale_summary_start")
3265
+ _autoscale_ranges_post: "_AutoscaleRangeReport | None" = None
3266
+ if (
3267
+ _autoscale_ranges_pre is not None
3268
+ and _autoscale_ranges_pre.trigger
3269
+ ):
3270
+ try:
3271
+ _autoscale_cfg_for_post = _autoscale_resolve_config(
3272
+ None,
3273
+ )
3274
+ _autoscale_ranges_post = _autoscale_compute_ranges(
3275
+ pb, _autoscale_cfg_for_post,
3276
+ )
3277
+ except Exception: # pragma: no cover — non-fatal
3278
+ self.state.logger.exception(
3279
+ "autoscale post-Layer-2 range readout failed "
3280
+ "for %s; console summary will omit the "
3281
+ "'ranges post' segment",
3282
+ complete_solve_name,
3283
+ )
3284
+ if _autoscale_profile:
3285
+ _ap("ranges_post_computed",
3286
+ ranges_post_ran=str(_autoscale_ranges_post is not None))
3287
+ if _cold_key is not None:
3288
+ _l2p = _autoscale_layer2_plan
3289
+ self._autoscale_shape_cache[_cold_key] = (
3290
+ _AutoscaleShapeCacheEntry(
3291
+ layer2_exponents=(
3292
+ dict(_l2p.type_exponents)
3293
+ if _l2p is not None else None
3294
+ ),
3295
+ layer2_buckets_before=(
3296
+ dict(_l2p.type_buckets_before)
3297
+ if _l2p is not None else {}
3298
+ ),
3299
+ layer2_buckets_after=(
3300
+ dict(_l2p.type_buckets_after)
3301
+ if _l2p is not None else {}
3302
+ ),
3303
+ layer3_plan=_autoscale_layer3_plan,
3304
+ ranges_pre=_autoscale_ranges_pre,
3305
+ ranges_post=_autoscale_ranges_post,
3306
+ )
3307
+ )
3308
+ _autoscale_emit_console_summary(
3309
+ ranges_pre=_autoscale_ranges_pre,
3310
+ ranges_post=_autoscale_ranges_post,
3311
+ layer2_plan=_autoscale_layer2_plan,
3312
+ layer3_plan=_autoscale_layer3_plan,
3313
+ solve_name=base_solve_name,
3314
+ already_emitted=self._autoscale_summary_emitted,
3315
+ memrec=_memrec_local if _emit_phase else None,
3316
+ logger=self.state.logger,
3317
+ )
3318
+ _phase_prof("autoscale_done")
3319
+ if _autoscale_profile:
3320
+ _ap("console_summary_done")
3321
+ # Phase 3 — multi-solver dispatch. ``run_one_solve`` calls
3322
+ # ``pb.solve(keep_solver=True)`` for the default HiGHS path
3323
+ # (byte-identical to the pre-Phase-3 behaviour); routes to
3324
+ # ``solve_via_subprocess`` (HiGHS CLI / commercial CLI) on
3325
+ # every cold path, which always returns a real
3326
+ # ``polar_high.Solution`` with the HiGHS instance read back
3327
+ # from the MPS. The cascade-level SolverConfig lookup uses
3328
+ # the active solve name with the standard
3329
+ # default-when-absent fallback.
3330
+ from flextool.engine_polars._solver_dispatch import (
3331
+ run_one_solve,
3332
+ )
3333
+ _t_solve_start = (
3334
+ time.perf_counter() if _phase_timing else 0.0
3335
+ )
3336
+ # Pre-solve range capture: every cold path now goes
3337
+ # subprocess, and the subprocess Solution carries no
3338
+ # ``streamed_lp_ranges`` (polar-high populates that
3339
+ # during its in-process streaming solve, which the
3340
+ # subprocess child doesn't share with us). Re-use the
3341
+ # post-Layer-2 RangeReport already computed above to
3342
+ # synthesize the dict :func:`_autoscale_emit_layer1`
3343
+ # consumes, so the per-solve ``autoscale_<solve>.yaml``
3344
+ # still lands under ``solve_data/``.
3345
+ _ranges_for_l1 = _autoscale_ranges_post or _autoscale_ranges_pre
3346
+ # Blank line so HiGHS' "Running HiGHS …" banner (and the
3347
+ # grey solver-output block in the GUI) separates from the
3348
+ # scaling/LP-build rows above it.
3349
+ print("", flush=True)
3350
+ # Echo the effective solver options right before HiGHS'
3351
+ # own banner: the full merged, post-precedence set the
3352
+ # solver actually received (``highs.opt`` floor ∪ DB
3353
+ # ``solver_arguments`` ∪ CLI overrides), so the run is
3354
+ # replicable from the log alone. Resolved with
3355
+ # ``baseline=None`` so the engine-internal determinism /
3356
+ # scale keys are excluded — only what the operator
3357
+ # touched is shown. Uses ``print`` (not the logger) to
3358
+ # sit in the same stdout stream as the blank separator
3359
+ # above and HiGHS' native banner below — the GUI's
3360
+ # ``execution_window`` parses these printed markers.
3361
+ _eff_touched = _resolve_effective_highs_options(
3362
+ solver_arguments_map=_solver_args_map,
3363
+ highs_opt_path=_highs_opt_path,
3364
+ cli_overrides=_build_cli_overrides(),
3365
+ baseline=None,
3366
+ )
3367
+ _sa_line = _format_solver_args_line(
3368
+ _eff_touched, prefer_first=list(_solver_args_map or {}),
3369
+ )
3370
+ if _sa_line:
3371
+ print(_sa_line, flush=True)
3372
+ _phase_prof("solve_start")
3373
+ sol = run_one_solve(
3374
+ pb, _active_solver_cfg, logger=state.logger,
3375
+ save_memory=_save_memory,
3376
+ solve_name=complete_solve_name,
3377
+ work_folder=state.paths.work_folder,
3378
+ )
3379
+ _t_solve_end = (
3380
+ time.perf_counter() if _phase_timing else 0.0
3381
+ )
3382
+ _phase_prof("after_solve")
3383
+ # Attach the pre-solve ranges as
3384
+ # ``streamed_lp_ranges`` on the Solution so the L1
3385
+ # emit hook (which expects a dict per polar-high's
3386
+ # in-process contract) sees the four (min, max) pairs
3387
+ # the solver actually saw — matrix / cost / col_bound /
3388
+ # row_bound, matching ``ranges_from_streamed``'s key
3389
+ # contract.
3390
+ if (
3391
+ _ranges_for_l1 is not None
3392
+ and getattr(sol, "streamed_lp_ranges", None) is None
3393
+ ):
3394
+ try:
3395
+ sol.streamed_lp_ranges = {
3396
+ "matrix": _ranges_for_l1.matrix,
3397
+ "cost": _ranges_for_l1.cost,
3398
+ "col_bound": _ranges_for_l1.bound,
3399
+ "row_bound": _ranges_for_l1.rhs,
3400
+ }
3401
+ except Exception: # pragma: no cover — Solution may
3402
+ # forbid the assignment in a future polar-high
3403
+ pass
3404
+ # Eager unscale — restore primal / duals / reduced costs
3405
+ # to the un-scaled coordinate so output writers and
3406
+ # subsequent rolling iterations see physical values.
3407
+ _phase_prof("unscale_start")
3408
+ _autoscale_unscale_post_solve(
3409
+ sol, _autoscale_layer2_plan,
3410
+ solve_name=complete_solve_name,
3411
+ logger=self.state.logger,
3412
+ )
3413
+ _phase_prof("unscale_done")
3414
+ # autoscale Layer 1 (detect) — log the four LP coefficient
3415
+ # ranges + trigger flag now that ``streamed_lp_ranges`` is
3416
+ # populated. Detection-only in Phase 1b; Layer 2 / Layer 3
3417
+ # actions land in later phases.
3418
+ _phase_prof("layer1emit_start")
3419
+ _autoscale_emit_layer1(
3420
+ sol,
3421
+ solve_name=complete_solve_name,
3422
+ logger=self.state.logger,
3423
+ work_folder=self.state.paths.work_folder
3424
+ if self.state.paths is not None else None,
3425
+ layer2_plan=locals().get("_autoscale_layer2_plan"),
3426
+ layer3_plan=locals().get("_autoscale_layer3_plan"),
3427
+ )
3428
+ _phase_prof("layer1emit_done")
3429
+ # Memory checkpoint after the solve completes. Fires on
3430
+ # level-boundary iters only; on within-group rolling iters
3431
+ # the suppressed deltas accumulate into the next emitted
3432
+ # group's lines.
3433
+ if _memrec_local is not None and _emit_phase:
3434
+ _memrec_local.checkpoint(
3435
+ "solve_end", self.state.logger,
3436
+ user_label="Solver",
3437
+ )
3438
+ # Emit per-iter lp_build / solve / warm_used rows now that
3439
+ # the solve is done and we have valid timestamps from
3440
+ # whichever branch ran. handoff is recorded at end-of-run.
3441
+ if _phase_timing:
3442
+ _tr.record(
3443
+ "per_iter",
3444
+ subphase="lp_build",
3445
+ solve=complete_solve_name,
3446
+ roll_index=_roll_idx,
3447
+ seconds=_t_solve_start - _t_build_start,
3448
+ t_start=_t_build_start,
3449
+ )
3450
+ _tr.record(
3451
+ "per_iter",
3452
+ subphase="solve",
3453
+ solve=complete_solve_name,
3454
+ roll_index=_roll_idx,
3455
+ seconds=_t_solve_end - _t_solve_start,
3456
+ t_start=_t_solve_start,
3457
+ )
3458
+ _tr.record(
3459
+ "per_iter",
3460
+ subphase="warm_used",
3461
+ solve=complete_solve_name,
3462
+ roll_index=_roll_idx,
3463
+ seconds=1.0 if warm_used else 0.0,
3464
+ )
3465
+ _t_handoff_start = (
3466
+ time.perf_counter() if _phase_timing else 0.0
3467
+ )
3468
+ # Write the ``scale_the_objective.csv`` (consumed by the
3469
+ # output writers' un-scaling path) BEFORE the optimality
3470
+ # check, so a non-optimal / time-limited solve still leaves
3471
+ # behind a coherent solve_data/ directory.
3472
+ #
3473
+ # Per-base-solve gating: the CSV value is invariant across
3474
+ # rolls of the same base solve, so emit it only on the first
3475
+ # roll. The legacy ``scaling_report.txt`` diagnostic (and
3476
+ # its ``FLEXTOOL_SCALING_REPORT=1`` gate) was retired in
3477
+ # Phase 2b — the autoscaler's per-solve YAML report (written
3478
+ # from ``_autoscale_emit_layer1``) is the replacement.
3479
+ _t_scale_start = time.perf_counter() if _phase_timing else 0.0
3480
+ _write_csv = base_solve_name not in self._scale_csv_written
3481
+ if _write_csv:
3482
+ _write_scale_csv(
3483
+ solve_data_dir=self.state.paths.work_folder / "solve_data",
3484
+ solve_name=complete_solve_name,
3485
+ effective_obj_scale=effective_obj_scale,
3486
+ logger=self.state.logger,
3487
+ )
3488
+ self._scale_csv_written.add(base_solve_name)
3489
+ if _phase_timing:
3490
+ _tr.record(
3491
+ "handoff_part",
3492
+ subphase="scale_csv_report",
3493
+ solve=complete_solve_name,
3494
+ roll_index=_roll_idx,
3495
+ seconds=time.perf_counter() - _t_scale_start,
3496
+ t_start=_t_scale_start,
3497
+ )
3498
+ # Accept/reject the solve on its actual solver diagnostics, not
3499
+ # a bare ``kOptimal`` check. An interior-point solve run without
3500
+ # crossover (the model generator's ``run_crossover`` choice) can
3501
+ # return a feasible, in-practice-optimal primal that HiGHS refuses
3502
+ # to *certify* (status Unknown) because postsolve left the dual
3503
+ # objective slightly inconsistent. ``classify_acceptance`` accepts
3504
+ # that case (using the feasible primal) while still rejecting
3505
+ # genuine failures — and reports only what the diagnostics
3506
+ # actually show, never a raw-range scaling guess as the cause.
3507
+ _acc = classify_acceptance(
3508
+ sol,
3509
+ ranges_post=locals().get("_autoscale_ranges_post"),
3510
+ solve_name=complete_solve_name,
3511
+ )
3512
+ if not _acc.accepted:
3513
+ self.state.logger.error(_acc.message)
3514
+ if _acc.scaling_hint:
3515
+ print(_acc.scaling_hint, flush=True)
3516
+ return 1
3517
+ if _acc.near_optimal:
3518
+ self.state.logger.info(_acc.message)
3519
+
3520
+ prior = prior_for_load
3521
+ # ``--csv-dump``: gate ``data.dump_csvs`` behind the
3522
+ # orchestrator-level ``csv_dump`` flag. In default mode the
3523
+ # cascade stays in-memory; ``--csv-dump`` materialises the
3524
+ # full FlexData → CSV snapshot for debug.
3525
+ _t_dump_start = time.perf_counter() if _phase_timing else 0.0
3526
+ try:
3527
+ if getattr(self.state, "csv_dump", False):
3528
+ data.dump_csvs(self.state.paths.work_folder)
3529
+ except Exception as exc: # noqa: BLE001
3530
+ self.state.logger.warning(
3531
+ f"dump_csvs failed for {complete_solve_name}: {exc}"
3532
+ )
3533
+ if _phase_timing:
3534
+ _tr.record(
3535
+ "handoff_part",
3536
+ subphase="dump_csvs",
3537
+ solve=complete_solve_name,
3538
+ roll_index=_roll_idx,
3539
+ seconds=time.perf_counter() - _t_dump_start,
3540
+ t_start=_t_dump_start,
3541
+ )
3542
+ # Emit TIER A ``output_raw`` artefacts BEFORE the in-memory
3543
+ # handoff is built. ``write_all_handoffs`` (called by the
3544
+ # adapter) refreshes ``solve_data/period_capacity.csv`` and
3545
+ # other handoff CSVs; ``build_handoff_from_solution`` then
3546
+ # reads those refreshed files for the in-memory handoff.
3547
+ _t_wofs_start = time.perf_counter() if _phase_timing else 0.0
3548
+ _phase_prof("write_outputs_start")
3549
+ try:
3550
+ # Phase G — pass in-memory FlexData and the cascade-known
3551
+ # ``is_first_solve`` boolean so handoff writers + extractors
3552
+ # can short-circuit ~12 + ~10 per-iter CSV reads (audit:
3553
+ # specs/in_memory_carriers_audit.md). CSV fallback paths
3554
+ # remain in place for callers that synthesize a Solution
3555
+ # without a FlexData (e.g. unit tests).
3556
+ _is_first = (
3557
+ self.state.last_captured_solve is None
3558
+ or len(self.state.handoffs or {}) == 0
3559
+ )
3560
+ write_outputs_for_solve(
3561
+ sol,
3562
+ work_folder=self.state.paths.work_folder,
3563
+ solve_name=complete_solve_name,
3564
+ prior_handoff=prior,
3565
+ writer_state=writer_state,
3566
+ flex_data=data,
3567
+ is_first_solve=_is_first,
3568
+ scale_the_objective=effective_obj_scale,
3569
+ provider=getattr(
3570
+ self.state, "current_provider", None,
3571
+ ),
3572
+ csv_dump=getattr(self.state, "csv_dump", False),
3573
+ )
3574
+ except Exception as exc: # noqa: BLE001
3575
+ self.state.logger.warning(
3576
+ f"write_outputs_for_solve failed for "
3577
+ f"{complete_solve_name}: {exc}"
3578
+ )
3579
+ if _phase_timing:
3580
+ _tr.record(
3581
+ "handoff_part",
3582
+ subphase="write_outputs_for_solve",
3583
+ solve=complete_solve_name,
3584
+ roll_index=_roll_idx,
3585
+ seconds=time.perf_counter() - _t_wofs_start,
3586
+ t_start=_t_wofs_start,
3587
+ )
3588
+ _phase_prof("write_outputs_done")
3589
+
3590
+ # Phase 4 (Gap F) — thread the in-memory FlexData + the
3591
+ # upper-level parent handoff so the extractors / fix_storage
3592
+ # parent overlay skip their workdir CSV reads where the same
3593
+ # data is already in scope. ``parent_handoff`` is the upper
3594
+ # nesting parent (used for fix_storage propagation, deposited
3595
+ # by ``_native_run_model``); ``prior_handoff`` is the
3596
+ # sequence predecessor (used for cumulative accumulators).
3597
+ parent_complete = getattr(
3598
+ self.state, "current_parent_complete", None
3599
+ )
3600
+ parent_handoff = (
3601
+ self.state.handoffs.get(parent_complete)
3602
+ if parent_complete is not None
3603
+ and self.state.handoffs is not None else None
3604
+ )
3605
+ _t_bhf_start = time.perf_counter() if _phase_timing else 0.0
3606
+ _phase_prof("build_handoff_start")
3607
+ handoff = build_handoff_from_solution(
3608
+ sol, self.state.paths.work_folder, complete_solve_name,
3609
+ prior_handoff=prior,
3610
+ flex_data=data,
3611
+ parent_handoff=parent_handoff,
3612
+ provider=getattr(self.state, "current_provider", None),
3613
+ )
3614
+ _phase_prof("build_handoff_done")
3615
+ if _phase_timing:
3616
+ _tr.record(
3617
+ "handoff_part",
3618
+ subphase="build_handoff_from_solution",
3619
+ solve=complete_solve_name,
3620
+ roll_index=_roll_idx,
3621
+ seconds=time.perf_counter() - _t_bhf_start,
3622
+ t_start=_t_bhf_start,
3623
+ )
3624
+ # Deposit so the next iteration's translator picks it up
3625
+ # AND we have it for the result dict.
3626
+ self.state.handoffs[complete_solve_name] = handoff
3627
+ # Un-scale the objective value back to user-facing units.
3628
+ # ``build_flextool`` multiplied the objective coefficients by
3629
+ # ``effective_obj_scale``, so HiGHS reports a scaled value.
3630
+ # Overwrite ``sol.obj`` in place so the public
3631
+ # ``step.solution.obj`` matches the unscaled ``step.obj`` /
3632
+ # the ``v_obj__{solve}.parquet`` value that the legacy
3633
+ # writer un-scales via ``_resolve_inv_scale_the_objective``.
3634
+ # Without this, callers reading ``step.solution.obj`` see
3635
+ # the LP-internal (scaled-by-1e-6) magnitude — a parity
3636
+ # break with the legacy flextool objective.
3637
+ unscaled_obj = (
3638
+ sol.obj / effective_obj_scale if sol.obj is not None else None
3639
+ )
3640
+ if sol is not None and unscaled_obj is not None:
3641
+ sol.obj = unscaled_obj
3642
+ # In rolling solves, every iteration's ``complete_solve_name``
3643
+ # is the parent solve name (see ``recursive_solves.py:259``:
3644
+ # ``complete_solves[roll_name] = complete_solve_name``). Use
3645
+ # the actual per-roll name from ``solve_data/solve_current.csv``
3646
+ # — the file flextool rewrites between rolls — so
3647
+ # ``_all_steps`` has one entry per roll instead of every roll
3648
+ # overwriting the parent key. ``write_outputs`` keys its
3649
+ # union over sub-solves on this dict, and the parquet
3650
+ # writers use the same per-roll name (see
3651
+ # ``read_highs_solution._actual_solve_name``).
3652
+ from flextool.process_outputs.read_highs_solution import (
3653
+ _actual_solve_name,
3654
+ )
3655
+ step_key = _actual_solve_name(
3656
+ self.state.paths.work_folder, complete_solve_name,
3657
+ provider=getattr(self.state, "current_provider", None),
3658
+ )
3659
+ # NOTE: the previous "slim PRIOR iter's _vars + highs at
3660
+ # the start of iter N" block has been retired. Both paths
3661
+ # now do their slim AFTER ``Outputs written`` below:
3662
+ #
3663
+ # * Warm path: per-level retention slim (Phase 2) — keeps
3664
+ # one ``Solution.highs`` + one ``flex_data_provider`` per
3665
+ # live level, drops everything else.
3666
+ # * Cold path (``_save_memory``): eager prior-iter slim —
3667
+ # nulls everything heavy on the prior step.
3668
+ #
3669
+ # Both blocks live at the bottom of this method, just after
3670
+ # the ``outputs_written_end`` memory checkpoint.
3671
+ # Capture per-sub-solve decision-variable frames before the
3672
+ # step is deposited. Polar-high may release ``sol._vars``
3673
+ # internally between sub-solves (a memory optimisation on
3674
+ # the ``polar-high-opt`` branch), so the LAST iter is the
3675
+ # only one whose live ``solution._vars`` survives to
3676
+ # end-of-cascade. End-of-cascade writers
3677
+ # (``_entity_all_capacity`` and friends) need EACH sub-
3678
+ # solve's own ``v_invest_p`` / ``v_invest_n`` /
3679
+ # ``v_divest_p`` / ``v_divest_n``; snapshot those four
3680
+ # long-form polars frames here so they survive the
3681
+ # between-solve release. Cost: 4 small DataFrames per
3682
+ # sub-solve (typically a few hundred rows each). See
3683
+ # :class:`SnapshotSolution` for the wrapper consumers read.
3684
+ captured_vars: "dict[str, pl.DataFrame]" = {}
3685
+ _phase_prof("captured_vars_start")
3686
+ if sol is not None and getattr(sol, "_vars", None):
3687
+ for _vname in (
3688
+ "v_invest_p", "v_invest_n",
3689
+ "v_divest_p", "v_divest_n",
3690
+ ):
3691
+ if _vname in sol._vars:
3692
+ try:
3693
+ captured_vars[_vname] = sol.value(_vname)
3694
+ except Exception: # noqa: BLE001
3695
+ # Best-effort: if a variable can't be
3696
+ # materialised here, the end-of-cascade
3697
+ # writer's empty-fallback branch handles
3698
+ # it (see ``_entity_all_capacity._try_value``).
3699
+ pass
3700
+ _phase_prof("captured_vars_done")
3701
+ self._all_steps[step_key] = OrchestrationStep(
3702
+ solve_name=step_key,
3703
+ solution=sol,
3704
+ handoff=handoff,
3705
+ obj=unscaled_obj,
3706
+ optimal=bool(getattr(sol, "optimal", False)) if sol is not None else None,
3707
+ # ``_acc`` (from ``classify_acceptance`` above) is authoritative:
3708
+ # reaching here means the solve was accepted (a reject returned 1
3709
+ # and raised before this point). Record the near-optimal verdict
3710
+ # so the CLI exit-code scan does not re-reject a usable, accepted
3711
+ # crossover-off solve off the strict ``optimal`` mirror alone.
3712
+ near_optimal=bool(getattr(_acc, "near_optimal", False)),
3713
+ warm_used=warm_used,
3714
+ flex_data=data,
3715
+ flex_data_provider=getattr(
3716
+ self.state, "current_provider", None,
3717
+ ),
3718
+ captured_vars=captured_vars,
3719
+ )
3720
+ # Track the just-parked step_key so the next iter can slim
3721
+ # THIS iter's Solution (see block above).
3722
+ self._prev_step_key = step_key
3723
+ # Phase 2 — record this step's level_key for the warm-path
3724
+ # per-level slim below. ``state._current_level_key`` was
3725
+ # set by ``_native_run_model`` immediately before this call.
3726
+ _this_level_key = getattr(
3727
+ self.state, "_current_level_key", None,
3728
+ )
3729
+ if _this_level_key is not None:
3730
+ self._step_level_keys[step_key] = _this_level_key
3731
+ if _phase_timing:
3732
+ _tr.record(
3733
+ "per_iter",
3734
+ subphase="handoff",
3735
+ solve=complete_solve_name,
3736
+ roll_index=_roll_idx,
3737
+ seconds=time.perf_counter() - _t_handoff_start,
3738
+ t_start=_t_handoff_start,
3739
+ )
3740
+ # End-of-iter heap trim — release per-roll scratch frames so
3741
+ # the next iter's load_flextool doesn't compound on stale heap.
3742
+ # No-op on non-glibc. Cost: ~10-50ms per iter.
3743
+ _try_malloc_trim()
3744
+ # Memory checkpoint after the per-iter writers + handoff
3745
+ # capture finish. Same gating as the other phase checkpoints:
3746
+ # level-boundary iters only.
3747
+ if _memrec_local is not None and _emit_phase:
3748
+ _memrec_local.checkpoint(
3749
+ "outputs_written_end", self.state.logger,
3750
+ user_label="Outputs written",
3751
+ )
3752
+ # Phase 2 — warm-path per-level retention slim. Per the
3753
+ # user's design:
3754
+ #
3755
+ # * Keep ONE ``Solution.highs`` (the live ``polar_high.Solution``
3756
+ # wrapping the WarmProblem's HiGHS instance) per level —
3757
+ # the MOST RECENT parked one of each level.
3758
+ # * Keep ONE ``flex_data_provider`` per level — same gating.
3759
+ # * Drop both as soon as the pipeline has no more upcoming
3760
+ # solves of that level (``state._all_level_keys[i+1:]``).
3761
+ # * Drop ``Solution._vars`` after ``Outputs written`` on
3762
+ # warm too (warm-start uses HiGHS' basis, not these
3763
+ # polars frames).
3764
+ # * Drop ``flex_data`` after the solve consumes it.
3765
+ #
3766
+ # ``flex_data`` and ``solution`` (the Python object, minus
3767
+ # the heavy ``.highs`` + ``._vars`` slots) on the LAST step
3768
+ # overall survive the per-iter slim because cmd_run_flextool
3769
+ # passes them to ``write_outputs``. We never null the
3770
+ # ``step.solution`` object itself here — only its
3771
+ # ``_vars`` dict and its ``.highs`` reference.
3772
+ #
3773
+ # Memory-pressure-yielding for kept Highs instances is a
3774
+ # future concern, explicitly out of scope per the user.
3775
+ #
3776
+ # ``keep_solutions=True`` opts out of per-iter slimming
3777
+ # entirely — callers like ``tests/test_scenarios.py`` and
3778
+ # any other ``solve_steps``-style end-of-cascade walker
3779
+ # union per-step ``flex_data`` + ``solution`` after the
3780
+ # cascade returns and would crash on the nulled fields
3781
+ # otherwise.
3782
+ if not _save_memory and not keep_solutions:
3783
+ _all_level_keys = getattr(
3784
+ self.state, "_all_level_keys", ()
3785
+ )
3786
+ _iter_idx = getattr(
3787
+ self.state, "_current_iter_index", None,
3788
+ )
3789
+ _upcoming_levels: "set" = set()
3790
+ if _iter_idx is not None and _all_level_keys:
3791
+ _upcoming_levels = set(
3792
+ _all_level_keys[_iter_idx + 1:]
3793
+ )
3794
+ _this_level = self._step_level_keys.get(step_key)
3795
+ # Walk every parked step. For each, decide whether to
3796
+ # keep its ``solution.highs`` + ``flex_data_provider``.
3797
+ for _k, _step in self._all_steps.items():
3798
+ _step_lvl = self._step_level_keys.get(_k)
3799
+ _is_just_parked = (_k == step_key)
3800
+ # ``solution._vars`` and ``flex_data`` are dropped on
3801
+ # every PRIOR step regardless of level (the per-iter
3802
+ # writers + handoff carrier already consumed them).
3803
+ if not _is_just_parked and _step.solution is not None:
3804
+ try:
3805
+ _step.solution._vars = {}
3806
+ except Exception: # noqa: BLE001
3807
+ pass
3808
+ if not _is_just_parked:
3809
+ _step.flex_data = None
3810
+ # ``solution.highs`` + ``flex_data_provider`` are
3811
+ # kept only on the MOST RECENT parked step of each
3812
+ # level whose pipeline still has upcoming iters.
3813
+ # Just-parked step's level always has at least one
3814
+ # member (itself) so the "drop entire level" rule
3815
+ # fires only on PRIOR steps whose level is exhausted.
3816
+ if _is_just_parked:
3817
+ # Even the just-parked step drops its highs +
3818
+ # flex_data_provider when its level has no
3819
+ # more upcoming iters. Saves the level's last
3820
+ # parked Highs for the duration of subsequent
3821
+ # other-level work that would otherwise pin it.
3822
+ if (
3823
+ _step_lvl is not None
3824
+ and _step_lvl not in _upcoming_levels
3825
+ and _this_level != _step_lvl
3826
+ ):
3827
+ # Can't happen: just-parked step's level
3828
+ # IS ``_this_level``. Defensive no-op.
3829
+ pass
3830
+ continue
3831
+ # Prior step. Drop its highs / provider when EITHER:
3832
+ # (a) the level it belongs to is exhausted
3833
+ # (``_step_lvl not in _upcoming_levels`` and
3834
+ # ``_step_lvl != _this_level``), OR
3835
+ # (b) the level it belongs to is the same as the
3836
+ # just-parked step's level — in which case
3837
+ # the just-parked step is the new "most recent
3838
+ # of this level" and this older sibling is
3839
+ # superseded.
3840
+ _level_exhausted = (
3841
+ _step_lvl is not None
3842
+ and _step_lvl != _this_level
3843
+ and _step_lvl not in _upcoming_levels
3844
+ )
3845
+ _same_level_older = (
3846
+ _step_lvl is not None
3847
+ and _step_lvl == _this_level
3848
+ )
3849
+ if _level_exhausted or _same_level_older:
3850
+ if _step.solution is not None:
3851
+ try:
3852
+ # Null the WHOLE solution, not just
3853
+ # ``.highs``. A superseded prior step's
3854
+ # ``polar_high.Solution`` retains the
3855
+ # ``col_names``/``row_names`` lists and the
3856
+ # ``col_value``/``row_dual``/``col_dual``
3857
+ # numpy arrays — all O(LP size) (~1.26 GB
3858
+ # per roll on the real DES). Nulling only
3859
+ # ``.highs`` released the HiGHS handle but
3860
+ # left those arrays parked for the whole
3861
+ # cascade; dropping the whole object is the
3862
+ # per-roll floor-ratchet release. Warm
3863
+ # reuse is unaffected — it runs off
3864
+ # ``self._warm_problem``, never off parked
3865
+ # step solutions; and ``handoff`` /
3866
+ # ``captured_vars`` (the only per-step state
3867
+ # later consumers need) live on the step,
3868
+ # not inside ``solution``.
3869
+ _step.solution = None
3870
+ except Exception: # noqa: BLE001
3871
+ pass
3872
+ _step.flex_data_provider = None
3873
+ # Trim the libc heap after potentially dropping multiple
3874
+ # large Highs instances + polars frames.
3875
+ _try_malloc_trim()
3876
+ # Cold-path (save-memory) eager slim of the PRIOR iter's
3877
+ # parked OrchestrationStep. We can't slim the JUST-parked
3878
+ # step here — the orchestration cli (``cmd_run_flextool``)
3879
+ # passes the LAST step's ``flex_data`` + ``solution`` to
3880
+ # :func:`write_outputs`, and from inside the per-iter callback
3881
+ # we don't yet know which iter is last. Slimming the PRIOR
3882
+ # iter instead leaves one step's heavy state live at any
3883
+ # given time (the just-parked one) and guarantees the LAST
3884
+ # step survives the cascade intact.
3885
+ #
3886
+ # By the time we reach here the PRIOR iter's per-iter
3887
+ # consumers all ran on its own iter:
3888
+ #
3889
+ # * ``write_outputs_for_solve`` (the writers that needed
3890
+ # ``sol.highs.allVariableNames()`` / ``getSolution()``
3891
+ # / ``getLp().row_names_``) — done.
3892
+ # * ``build_handoff_from_solution`` — done; carrier stored
3893
+ # in ``step.handoff`` survives this slim.
3894
+ # * ``captured_vars`` snapshot — done; lives on
3895
+ # ``step.captured_vars`` independently of ``sol._vars``.
3896
+ #
3897
+ # On cold the cascade rebuilds the LP from scratch every
3898
+ # sub-solve (warm reuse is disabled via the
3899
+ # ``_warm_disabled_by_save_memory`` branch at the top of
3900
+ # this method), so the prior iter has no further consumer.
3901
+ # Drop everything heavy on it — that is the root-cause fix
3902
+ # for the cross-solve RSS climb on ``--save-memory`` runs.
3903
+ #
3904
+ # ``flex_data_provider`` is dropped here by default; set
3905
+ # ``FLEXTOOL_COLD_KEEP_PROVIDER=1`` to retain it across the
3906
+ # cold-path cascade (trades higher RSS for skipping the
3907
+ # per-iter Spine DB re-read). Real-model measurement at
3908
+ # the time of writing did not produce a default-changing
3909
+ # signal — the knob exists for workloads where the DB
3910
+ # re-read dominates wall time.
3911
+ # ``keep_solutions=True`` opts out (see the warm-slim
3912
+ # block above for the same rationale — callers that union
3913
+ # per-step state after the cascade returns crash on nulled
3914
+ # fields).
3915
+ if _save_memory and not keep_solutions and self._prev_step_key is not None:
3916
+ _prev_step = self._all_steps.get(self._prev_step_key)
3917
+ if _prev_step is not None and _prev_step is not self._all_steps.get(step_key):
3918
+ _prev_step.flex_data = None
3919
+ if os.environ.get("FLEXTOOL_COLD_KEEP_PROVIDER") != "1":
3920
+ _prev_step.flex_data_provider = None
3921
+ _psol = _prev_step.solution
3922
+ if _psol is not None:
3923
+ try:
3924
+ _psol._vars = {}
3925
+ except Exception: # noqa: BLE001
3926
+ pass
3927
+ try:
3928
+ _psol.highs = None
3929
+ except Exception: # noqa: BLE001
3930
+ pass
3931
+ return 0
3932
+
3933
+ # Drive the cascade via the native ``native_run_model``. Native
3934
+ # emitters thread ``sub_solve_provider`` through every emit_* call.
3935
+ native_run_model(runner.state, _PolarHighCascadeSolver(runner.state))
3936
+ # Mirror the in-memory handoff dict back onto our state in case
3937
+ # callers want to inspect it.
3938
+ state.handoffs = runner.state.handoffs
3939
+
3940
+ # Phase C.5 — slim every step except the LAST, releasing the heaviest
3941
+ # per-step state (Solution + FlexData + FlexDataProvider) once
3942
+ # downstream consumers (handoff extraction, raw-output write) have
3943
+ # run for that sub-solve. ``keep_solutions=True`` opts out — used
3944
+ # by tests that need per-step ``solution`` / ``flex_data`` access.
3945
+ if not keep_solutions and results:
3946
+ last_key = next(reversed(results))
3947
+ for k, step in results.items():
3948
+ if k == last_key:
3949
+ continue
3950
+ step.solution = None
3951
+ step.flex_data = None
3952
+ step.flex_data_provider = None
3953
+ # Free the HiGHS heap once the per-step references are gone.
3954
+ # Cheap and a no-op on non-glibc.
3955
+ _try_malloc_trim()
3956
+ return results
3957
+
3958
+
3959
+ # ---------------------------------------------------------------------------
3960
+ # Solver-config work-folder seeding
3961
+ # ---------------------------------------------------------------------------
3962
+
3963
+ # The five solvers whose ``<solver>.opt`` files FlexTool manages. Kept in
3964
+ # sync with the seed list in
3965
+ # :func:`flextool.update_flextool.self_update.ensure_runtime_files`.
3966
+ _SOLVER_CONFIG_NAMES = ("highs", "gurobi", "cplex", "xpress", "copt")
3967
+
3968
+
3969
+ def _seed_work_folder_solver_config(
3970
+ work_folder: Path,
3971
+ source_dir: Path,
3972
+ logger: logging.Logger,
3973
+ ) -> Path:
3974
+ """Populate ``<work_folder>/solver_config/`` with per-solver ``.opt`` files.
3975
+
3976
+ For each of :data:`_SOLVER_CONFIG_NAMES`, copy a ``<solver>.opt`` into
3977
+ ``<work_folder>/solver_config/`` so the run is self-contained and
3978
+ repeatable, and both the in-process and subprocess solve paths read the
3979
+ *same* files.
3980
+
3981
+ Copy-if-missing: a ``<solver>.opt`` already present in the work folder is
3982
+ left untouched, so a user can hand-tweak a work-folder opt file and re-run
3983
+ with it.
3984
+
3985
+ Source per file (first that exists wins):
3986
+
3987
+ 1. ``<source_dir>/<solver>.opt`` — the operator's active/edited file (or an
3988
+ explicitly-passed ``solver_config_dir``).
3989
+ 2. the packaged ``solver_config/<solver>.opt.template``.
3990
+
3991
+ If neither exists for a given solver, that solver is skipped silently.
3992
+
3993
+ Returns the destination directory (``<work_folder>/solver_config``). Any
3994
+ failure is logged as a WARNING and swallowed — seeding must never fail the
3995
+ run.
3996
+ """
3997
+ dest_dir = work_folder / "solver_config"
3998
+ try:
3999
+ from flextool._resources import package_data_path
4000
+
4001
+ dest_dir.mkdir(parents=True, exist_ok=True)
4002
+ for _solver in _SOLVER_CONFIG_NAMES:
4003
+ dest = dest_dir / f"{_solver}.opt"
4004
+ if dest.exists():
4005
+ # Do not clobber a work-folder file the user may have edited.
4006
+ continue
4007
+ src = source_dir / f"{_solver}.opt"
4008
+ if not src.is_file():
4009
+ try:
4010
+ src = Path(
4011
+ package_data_path(f"solver_config/{_solver}.opt.template")
4012
+ )
4013
+ except Exception:
4014
+ src = None # type: ignore[assignment]
4015
+ if src is None or not src.is_file():
4016
+ # No operator file and no bundled template — skip silently.
4017
+ continue
4018
+ shutil.copy2(str(src), str(dest))
4019
+ except Exception as exc: # pragma: no cover — defensive; never fail a run
4020
+ logger.warning(
4021
+ "Seeding %s failed (%s); solver options fall back to defaults.",
4022
+ dest_dir,
4023
+ exc,
4024
+ )
4025
+ return dest_dir
4026
+
4027
+
4028
+ # ---------------------------------------------------------------------------
4029
+ # run_chain_from_db — top-level entry point
4030
+ # ---------------------------------------------------------------------------
4031
+
4032
+
4033
+ def run_chain_from_db(
4034
+ input_db_url: str | Path,
4035
+ scenario_name: str | None = None,
4036
+ work_folder: Path | str | None = None,
4037
+ *,
4038
+ flextool_dir: Path | str | None = None,
4039
+ solver_config_dir: Path | str | None = None,
4040
+ logger: logging.Logger | None = None,
4041
+ warm: bool = True,
4042
+ keep_solutions: bool = False,
4043
+ csv_dump: bool = False,
4044
+ override_provider: "Callable[[], dict[str, pl.DataFrame]] | None" = None,
4045
+ ) -> dict[str, OrchestrationStep]:
4046
+ """Run a flextool multi-solve scenario end-to-end natively.
4047
+
4048
+ Combines:
4049
+
4050
+ 1. :func:`flextool.engine_polars._native_input_writer.write_workdir_inputs`
4051
+ populates the cascade-input Provider with every derived frame
4052
+ from the Spine DB (pure in-memory; no workdir CSVs are
4053
+ written).
4054
+ 2. ``_orchestration.run_orchestration`` to drive the per-solve loop
4055
+ with a polar_high-as-inner-solver wrapper.
4056
+ 3. Returns one :class:`OrchestrationStep` per per-solve iteration.
4057
+
4058
+ For tests / scripts that want a single function call to go from a
4059
+ DB scenario to a dict of (Solution, SolveHandoff) pairs.
4060
+
4061
+ Parameters
4062
+ ----------
4063
+ input_db_url : str | Path
4064
+ Spine SQLite URL or path. A bare path is upgraded to ``sqlite:///``.
4065
+ scenario_name : str, optional
4066
+ Scenario filter to apply. ``None`` picks the first scenario.
4067
+ work_folder : Path | str, optional
4068
+ Where to materialise the CSVs. ``None`` uses an auto-cleaned
4069
+ tempdir.
4070
+ flextool_dir, solver_config_dir : Path, optional
4071
+ Override the default flextool install location. Default:
4072
+ ``/home/jkiviluo/sources/flextool/{flextool,bin}``.
4073
+ logger : logging.Logger, optional
4074
+ Logger to use. ``None`` constructs one named after the scenario.
4075
+ warm : bool, default False
4076
+ Δ.12d — when True, reuse one :class:`polar_high.WarmProblem`
4077
+ across consecutive structurally-compatible per-solve iterations
4078
+ in the cascade, applying ``_apply_warm_updates`` between solves
4079
+ rather than cold-rebuilding. See
4080
+ :func:`run_orchestration` for full semantics.
4081
+ keep_solutions : bool, default False
4082
+ Phase C.5 — when False (default), only the LAST step in the
4083
+ returned dict retains ``solution`` / ``flex_data`` /
4084
+ ``flex_data_provider``; earlier steps clear those slots to
4085
+ release the HiGHS instance + variable arrays + writer-frame
4086
+ snapshot for that sub-solve. All slim fields (``solve_name``,
4087
+ ``obj``, ``optimal``, ``warm_used``, ``handoff``) remain
4088
+ populated on every step. Set ``True`` to retain the full
4089
+ per-step state — required by tests that need per-step
4090
+ ``solution`` / ``flex_data`` access (parity sweeps, warm
4091
+ comparisons, etc.).
4092
+
4093
+ Returns
4094
+ -------
4095
+ dict[str, OrchestrationStep]
4096
+ Mapping ``complete_solve_name → OrchestrationStep``. Iterate
4097
+ in insertion order to walk the chain.
4098
+ """
4099
+ from flextool.engine_polars._db_loader import FlexToolRunner
4100
+
4101
+ if logger is None:
4102
+ logger = logging.getLogger(
4103
+ f"flextool.engine_polars.run_chain_from_db[{scenario_name}]"
4104
+ )
4105
+
4106
+ # v52 multi-solver dispatch (Phase 2 startup hint). Probe each
4107
+ # solver in polar-high's catalog with a trivial 1-var LP so users
4108
+ # see at a glance which are licensed on this machine (vs wrapper-
4109
+ # installed-but-no-license vs not-installed-at-all) before the
4110
+ # solve loop selects one. See ``specs/flextool-multi-solver-handoff.md``
4111
+ # Step 4b. Cached at module level so repeat cascade runs don't
4112
+ # re-probe.
4113
+ try:
4114
+ from flextool.engine_polars._solver_dispatch import (
4115
+ probe_solver_licenses,
4116
+ )
4117
+ statuses = probe_solver_licenses()
4118
+ # HiGHS is an open-source dependency that's always present and
4119
+ # always usable -- nothing to opt in to. Drop it from the
4120
+ # "Available solvers" line so it only reports the commercial
4121
+ # adapters whose installation status actually varies.
4122
+ statuses = {n: s for n, s in (statuses or {}).items() if n != "highs"}
4123
+ if statuses:
4124
+ formatted = ", ".join(f"{n}={s}" for n, s in statuses.items())
4125
+ # Trailing blank line so the mem-checkpoint table that
4126
+ # follows visually separates from the header block.
4127
+ logger.info("Solvers: %s\n", formatted)
4128
+ except ImportError: # pragma: no cover — older polar_high without dispatch
4129
+ pass
4130
+
4131
+ db_url = str(input_db_url)
4132
+ if "://" not in db_url:
4133
+ db_url = f"sqlite:///{db_url}"
4134
+
4135
+ if work_folder is None:
4136
+ work_folder = Path(tempfile.mkdtemp(prefix="flextool_run_chain_"))
4137
+ else:
4138
+ work_folder = Path(work_folder)
4139
+ work_folder.mkdir(parents=True, exist_ok=True)
4140
+
4141
+ # ``flextool_dir`` defaults to the installed flextool package directory
4142
+ # (resolved via importlib.resources so it works both editable and wheel).
4143
+ # ``solver_config_dir`` defaults to ``<cwd>/solver_config`` — this is where the user's
4144
+ # editable ``highs.opt`` lives; the package's ``highs.opt.template``
4145
+ # is only used to seed that file on first run.
4146
+ from flextool._resources import package_data_path
4147
+ flextool_dir_resolved = (
4148
+ Path(flextool_dir) if flextool_dir is not None
4149
+ else package_data_path("")
4150
+ )
4151
+ # Seeding SOURCE precedence: explicit ``solver_config_dir`` arg >
4152
+ # a pre-existing ``$FLEXTOOL_SOLVER_CONFIG_DIR`` override (the hook the
4153
+ # subprocess path already honours as its highest-priority source — read
4154
+ # it BEFORE we overwrite it below, so a user/CI override's *content* is
4155
+ # carried into the work-folder copy rather than silently lost) >
4156
+ # ``<cwd>/solver_config``.
4157
+ _env_solver_config = os.environ.get("FLEXTOOL_SOLVER_CONFIG_DIR")
4158
+ if solver_config_dir is not None:
4159
+ solver_config_dir_resolved = Path(solver_config_dir)
4160
+ elif _env_solver_config:
4161
+ solver_config_dir_resolved = Path(_env_solver_config)
4162
+ else:
4163
+ solver_config_dir_resolved = Path.cwd() / "solver_config"
4164
+
4165
+ # Make the run self-contained and repeatable: copy the active
4166
+ # ``<solver>.opt`` files into ``<work_folder>/solver_config/`` (copy-if-
4167
+ # missing so a hand-tweaked work-folder file survives a re-run), then
4168
+ # point BOTH solve paths at that copy. The GUI's cwd is frequently not
4169
+ # the folder that holds the project's ``highs.opt``; reading from the
4170
+ # work-folder copy instead of ``<cwd>/solver_config`` ensures project-
4171
+ # level solver options are actually applied, and leaves a durable on-disk
4172
+ # record of exactly which options the run used.
4173
+ _workdir_solver_config = _seed_work_folder_solver_config(
4174
+ work_folder, solver_config_dir_resolved, logger
4175
+ )
4176
+ solver_config_dir_resolved = _workdir_solver_config
4177
+ # The subprocess/commercial solve path resolves its config dir
4178
+ # independently via ``_resolve_solver_config_dir()``, which honours
4179
+ # ``$FLEXTOOL_SOLVER_CONFIG_DIR`` first; point it at the same work-folder
4180
+ # copy so both paths read a single source of truth.
4181
+ os.environ["FLEXTOOL_SOLVER_CONFIG_DIR"] = str(_workdir_solver_config)
4182
+
4183
+ # Cascade-input Provider population from the Spine DB. Pure
4184
+ # in-memory: ``write_workdir_inputs`` runs the input_derivation
4185
+ # pipeline whose emitters populate the Provider directly, so no
4186
+ # CSVs hit disk.
4187
+ from flextool.engine_polars._native_input_writer import (
4188
+ write_workdir_inputs,
4189
+ )
4190
+
4191
+ # Phase-progress recorder. Always emits user-visible log lines
4192
+ # (RSS + section time + Δrss) so users following the run can see
4193
+ # what each phase is doing. Setting FLEXTOOL_MEMORY_DIAGNOSTICS=1
4194
+ # additionally enables tracemalloc (so ``traced_peak`` is meaningful)
4195
+ # and writes the per-checkpoint CSV under ``solve_data/`` for
4196
+ # post-hoc analysis.
4197
+ _mem_enabled = os.environ.get("FLEXTOOL_MEMORY_DIAGNOSTICS") == "1"
4198
+ if _mem_enabled:
4199
+ (work_folder / "solve_data").mkdir(parents=True, exist_ok=True)
4200
+ _memrec = _MemoryRecorder(
4201
+ work_folder / "solve_data" / "memory_diagnostics.csv",
4202
+ enabled=True,
4203
+ )
4204
+ else:
4205
+ _memrec = _MemoryRecorder(csv_path=None, enabled=False)
4206
+ # Publish so deeper modules (input.py's _apply_db_overrides) emit
4207
+ # in the unified [mem] format.
4208
+ set_phase_recorder(_memrec)
4209
+ _memrec.checkpoint("cascade_start", logger,
4210
+ user_label="Run start")
4211
+
4212
+ # Construct the cascade-input Provider and let
4213
+ # ``write_workdir_inputs`` populate it from the Spine DB. The
4214
+ # Provider is then attached to the cascade ``RunnerState`` below
4215
+ # so the per-sub-solve hook in
4216
+ # :mod:`flextool.engine_polars._native_run_model` picks it up at
4217
+ # provider seed time.
4218
+ from flextool.engine_polars._flex_data_provider import FlexDataProvider
4219
+ cascade_input_provider = FlexDataProvider()
4220
+
4221
+ write_workdir_inputs(
4222
+ db_url,
4223
+ scenario_name,
4224
+ work_folder,
4225
+ logger=logger,
4226
+ provider=cascade_input_provider,
4227
+ memory_recorder=_memrec,
4228
+ )
4229
+ # input_derivation allocates and frees a lot of polars scratch
4230
+ # state; glibc's heap retains the freed pages. Release them
4231
+ # before the polars-heavy ``load_flextool`` starts so the heap
4232
+ # watermark doesn't compound.
4233
+ _try_malloc_trim()
4234
+ _memrec.checkpoint("write_workdir_inputs_end", logger,
4235
+ user_label="Input data prepared (after malloc_trim)")
4236
+
4237
+ # Construct the underlying FlexToolRunner — still needed to carry
4238
+ # the cross-cutting ``RunnerState`` (timeline, solve config, handoff
4239
+ # dict) into ``native_run_model``, which drives the per-solve
4240
+ # preprocessing chain (``preprocessing_solve_time``,
4241
+ # ``solve_writers``, ``handoff_writers``). No write_input call.
4242
+ def _runner_factory() -> "FlexToolRunner":
4243
+ runner = FlexToolRunner(
4244
+ input_db_url=db_url,
4245
+ scenario_name=scenario_name,
4246
+ flextool_dir=flextool_dir_resolved,
4247
+ solver_config_dir=solver_config_dir_resolved,
4248
+ work_folder=work_folder,
4249
+ )
4250
+ runner.state.logger.setLevel(logging.ERROR)
4251
+ return runner
4252
+
4253
+ # Build a minimal native RunnerState so callers can introspect
4254
+ # state.handoffs after the run. The real per-solve mutation happens
4255
+ # on the underlying flextool runner's state (driven inside
4256
+ # _drive_cascade); we mirror the handoffs dict back.
4257
+ from flextool.engine_polars._solve_config import SolveConfig
4258
+ from flextool.engine_polars._timeline import TimelineConfig
4259
+
4260
+ sc = SolveConfig.load_from_db_url(db_url, scenario_name, logger=logger)
4261
+ _memrec.checkpoint("solve_config_loaded", logger,
4262
+ user_label="SolveConfig loaded (from DB)")
4263
+ tc = TimelineConfig.load_from_db_url(db_url, scenario_name, logger=logger)
4264
+ tc.create_assumptive_parts(sc)
4265
+ tc.create_timeline_from_timestep_duration(sc)
4266
+ _memrec.checkpoint("timeline_constructed", logger,
4267
+ user_label="TimelineConfig constructed (from DB)")
4268
+
4269
+ state = RunnerState(
4270
+ # ``solver_config_dir`` MUST be carried here: this is the state the
4271
+ # per-solve loop reads when it resolves ``<dir>/highs.opt`` for the
4272
+ # in-process HiGHS options (``_highs_opt_path`` in the solve loop).
4273
+ # Omitting it leaves ``solver_config_dir`` None, so the ``highs.opt``
4274
+ # floor is silently dropped on the in-process path even though the
4275
+ # factory runner below got the dir. It points at the work-folder
4276
+ # copy seeded above.
4277
+ paths=PathConfig(
4278
+ work_folder=work_folder,
4279
+ solver_config_dir=solver_config_dir_resolved,
4280
+ ),
4281
+ solve=sc,
4282
+ logger=logger,
4283
+ timeline=tc,
4284
+ handoffs={},
4285
+ )
4286
+ # Stash the memory recorder so ``run_orchestration`` →
4287
+ # ``_PolarHighCascadeSolver`` can fire the remaining first-iter
4288
+ # checkpoints (load / build / solve) without having to plumb it
4289
+ # through additional keyword arguments.
4290
+ state._memory_recorder = _memrec # type: ignore[attr-defined]
4291
+ # Step 2.5 — seed the cascade-input Provider onto the state so
4292
+ # ``_drive_cascade`` can forward it onto ``runner.state`` (which the
4293
+ # per-sub-solve Provider hook in
4294
+ # :mod:`flextool.engine_polars._native_run_model` consults).
4295
+ state.cascade_input_provider = cascade_input_provider # type: ignore[attr-defined]
4296
+ # Phase 5c — attach the optional external override provider onto
4297
+ # the cascade ``RunnerState``. ``_drive_cascade`` forwards it onto
4298
+ # the underlying ``runner.state``; the per-sub-solve hook in
4299
+ # :mod:`flextool.engine_polars._native_run_model` (Phase 5b) invokes
4300
+ # it after the parent-handoff translator at every iteration.
4301
+ state.override_provider = override_provider
4302
+
4303
+ return run_orchestration(
4304
+ state, work_folder, runner_factory=_runner_factory,
4305
+ db_url=db_url, scenario_name=scenario_name, warm=warm,
4306
+ keep_solutions=keep_solutions, csv_dump=csv_dump,
4307
+ )
4308
+
4309
+
4310
+ __all__ = [
4311
+ "OrchestrationStep",
4312
+ "run_orchestration",
4313
+ "run_chain_from_db",
4314
+ ]