flextool 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (322) hide show
  1. flextool/__init__.py +41 -0
  2. flextool/_mem_sampler.py +193 -0
  3. flextool/_resources.py +43 -0
  4. flextool/calibrate/__init__.py +51 -0
  5. flextool/calibrate/__main__.py +11 -0
  6. flextool/calibrate/_cli.py +316 -0
  7. flextool/calibrate/_db_alt.py +166 -0
  8. flextool/calibrate/_final_outputs.py +110 -0
  9. flextool/calibrate/_guard.py +151 -0
  10. flextool/calibrate/_loop.py +558 -0
  11. flextool/calibrate/_readers.py +223 -0
  12. flextool/calibrate/_report.py +263 -0
  13. flextool/calibrate/_sizing.py +699 -0
  14. flextool/calibrate/_solve.py +134 -0
  15. flextool/calibrate/_solve_status.py +495 -0
  16. flextool/cli/__init__.py +9 -0
  17. flextool/cli/_console.py +51 -0
  18. flextool/cli/_timing.py +147 -0
  19. flextool/cli/cmd_execute_flextool_workflow.py +187 -0
  20. flextool/cli/cmd_export_to_tabular.py +56 -0
  21. flextool/cli/cmd_import_sensitivities.py +75 -0
  22. flextool/cli/cmd_migrate_database.py +13 -0
  23. flextool/cli/cmd_open_results_db.py +269 -0
  24. flextool/cli/cmd_read_matpower.py +66 -0
  25. flextool/cli/cmd_read_old_flextool.py +63 -0
  26. flextool/cli/cmd_read_self_describing_tabular_input.py +50 -0
  27. flextool/cli/cmd_read_tabular_input.py +81 -0
  28. flextool/cli/cmd_run_flextool.py +1095 -0
  29. flextool/cli/cmd_scenario_results.py +284 -0
  30. flextool/cli/cmd_solve_mps.py +169 -0
  31. flextool/cli/cmd_update_flextool.py +17 -0
  32. flextool/cli/cmd_write_outputs.py +125 -0
  33. flextool/common_utils/__init__.py +1 -0
  34. flextool/common_utils/plot_mem_shape.py +77 -0
  35. flextool/common_utils/precision.py +451 -0
  36. flextool/decomposition/__init__.py +0 -0
  37. flextool/decomposition/region_decomposition.py +128 -0
  38. flextool/decomposition/region_filter.py +1261 -0
  39. flextool/engine_polars/__init__.py +110 -0
  40. flextool/engine_polars/_axis_enums.py +742 -0
  41. flextool/engine_polars/_benders.py +3462 -0
  42. flextool/engine_polars/_block_layout.py +1479 -0
  43. flextool/engine_polars/_blocks.py +1515 -0
  44. flextool/engine_polars/_commodity_ladder.py +660 -0
  45. flextool/engine_polars/_cumulative_invest.py +1165 -0
  46. flextool/engine_polars/_db_loader.py +153 -0
  47. flextool/engine_polars/_db_reader.py +127 -0
  48. flextool/engine_polars/_dc_power_flow.py +445 -0
  49. flextool/engine_polars/_delay.py +442 -0
  50. flextool/engine_polars/_derived_arithmetic.py +432 -0
  51. flextool/engine_polars/_derived_block.py +990 -0
  52. flextool/engine_polars/_derived_branch.py +769 -0
  53. flextool/engine_polars/_derived_existing.py +1353 -0
  54. flextool/engine_polars/_derived_npv.py +1297 -0
  55. flextool/engine_polars/_derived_params.py +9850 -0
  56. flextool/engine_polars/_derived_profile.py +881 -0
  57. flextool/engine_polars/_derived_walks.py +276 -0
  58. flextool/engine_polars/_determinism.py +70 -0
  59. flextool/engine_polars/_direct_params.py +2186 -0
  60. flextool/engine_polars/_dump_csvs.py +1009 -0
  61. flextool/engine_polars/_emit_arc_unions.py +1631 -0
  62. flextool/engine_polars/_emit_calc_params.py +729 -0
  63. flextool/engine_polars/_emit_chain_params.py +709 -0
  64. flextool/engine_polars/_emit_co2_accumulators.py +400 -0
  65. flextool/engine_polars/_emit_dispatchers.py +690 -0
  66. flextool/engine_polars/_emit_energy_margin.py +125 -0
  67. flextool/engine_polars/_emit_energy_margin_adder.py +290 -0
  68. flextool/engine_polars/_emit_entity_annual.py +428 -0
  69. flextool/engine_polars/_emit_inflow_scaling.py +1420 -0
  70. flextool/engine_polars/_emit_leaf_sets.py +550 -0
  71. flextool/engine_polars/_emit_lp_scaling.py +665 -0
  72. flextool/engine_polars/_emit_mid_sets.py +859 -0
  73. flextool/engine_polars/_emit_pdt_params.py +759 -0
  74. flextool/engine_polars/_emit_per_solve.py +774 -0
  75. flextool/engine_polars/_emit_period_calc.py +504 -0
  76. flextool/engine_polars/_emit_period_params.py +2398 -0
  77. flextool/engine_polars/_emit_provider_io.py +141 -0
  78. flextool/engine_polars/_emit_reserve.py +574 -0
  79. flextool/engine_polars/_emit_solve_time.py +311 -0
  80. flextool/engine_polars/_emit_solve_writers.py +1249 -0
  81. flextool/engine_polars/_flex_data_accumulator.py +388 -0
  82. flextool/engine_polars/_flex_data_provider.py +478 -0
  83. flextool/engine_polars/_group_slack.py +1253 -0
  84. flextool/engine_polars/_inmemory_reader.py +140 -0
  85. flextool/engine_polars/_input_source.py +336 -0
  86. flextool/engine_polars/_invest_seeds.py +191 -0
  87. flextool/engine_polars/_native_input_writer.py +100 -0
  88. flextool/engine_polars/_native_run_model.py +1348 -0
  89. flextool/engine_polars/_orchestration.py +4314 -0
  90. flextool/engine_polars/_output_writer.py +439 -0
  91. flextool/engine_polars/_param_shapes.py +1595 -0
  92. flextool/engine_polars/_parquet_bundle.py +723 -0
  93. flextool/engine_polars/_pdt_join.py +167 -0
  94. flextool/engine_polars/_pdt_lookup.py +547 -0
  95. flextool/engine_polars/_per_solve_sets.py +335 -0
  96. flextool/engine_polars/_projection_params.py +2056 -0
  97. flextool/engine_polars/_provider_keys.py +173 -0
  98. flextool/engine_polars/_provider_translators.py +225 -0
  99. flextool/engine_polars/_recursive_solve.py +703 -0
  100. flextool/engine_polars/_region_filter.py +2508 -0
  101. flextool/engine_polars/_reserve.py +649 -0
  102. flextool/engine_polars/_solve_acceptance.py +331 -0
  103. flextool/engine_polars/_solve_config.py +1001 -0
  104. flextool/engine_polars/_solve_context.py +885 -0
  105. flextool/engine_polars/_solve_handoff.py +164 -0
  106. flextool/engine_polars/_solve_state.py +232 -0
  107. flextool/engine_polars/_solver_base.py +36 -0
  108. flextool/engine_polars/_solver_dispatch.py +511 -0
  109. flextool/engine_polars/_spinedb_reader.py +1165 -0
  110. flextool/engine_polars/_stochastic.py +593 -0
  111. flextool/engine_polars/_subprocess_solve.py +1838 -0
  112. flextool/engine_polars/_timeline.py +1416 -0
  113. flextool/engine_polars/_vectorize.py +438 -0
  114. flextool/engine_polars/_warm.py +858 -0
  115. flextool/engine_polars/autoscale/__init__.py +107 -0
  116. flextool/engine_polars/autoscale/_config.py +218 -0
  117. flextool/engine_polars/autoscale/_layer2.py +1253 -0
  118. flextool/engine_polars/autoscale/_layer2_types.py +584 -0
  119. flextool/engine_polars/autoscale/_quantity_types.py +621 -0
  120. flextool/engine_polars/autoscale/_report.py +336 -0
  121. flextool/engine_polars/chain.py +259 -0
  122. flextool/engine_polars/input.py +6638 -0
  123. flextool/engine_polars/model.py +4754 -0
  124. flextool/env_check.py +388 -0
  125. flextool/export_to_tabular/__init__.py +5 -0
  126. flextool/export_to_tabular/db_reader.py +224 -0
  127. flextool/export_to_tabular/excel_writer.py +3559 -0
  128. flextool/export_to_tabular/export_settings.yaml +377 -0
  129. flextool/export_to_tabular/export_to_excel.py +227 -0
  130. flextool/export_to_tabular/formatting.py +543 -0
  131. flextool/export_to_tabular/sheet_config.py +876 -0
  132. flextool/gui/__init__.py +0 -0
  133. flextool/gui/__main__.py +118 -0
  134. flextool/gui/calibrate_commands.py +184 -0
  135. flextool/gui/calibrate_jobs.py +424 -0
  136. flextool/gui/check_tree.py +142 -0
  137. flextool/gui/cli_format.py +83 -0
  138. flextool/gui/config_parser.py +68 -0
  139. flextool/gui/data_models.py +362 -0
  140. flextool/gui/db_editor_integration.py +202 -0
  141. flextool/gui/db_version_check.py +269 -0
  142. flextool/gui/dialogs/__init__.py +0 -0
  143. flextool/gui/dialogs/add_dialog.py +1098 -0
  144. flextool/gui/dialogs/calibrate_dialog.py +1259 -0
  145. flextool/gui/dialogs/file_picker.py +473 -0
  146. flextool/gui/dialogs/group_picker.py +299 -0
  147. flextool/gui/dialogs/migration_consent_dialog.py +106 -0
  148. flextool/gui/dialogs/migration_progress_dialog.py +237 -0
  149. flextool/gui/dialogs/plot_dialog.py +459 -0
  150. flextool/gui/dialogs/plot_settings_picker.py +2184 -0
  151. flextool/gui/dialogs/project_dialog.py +426 -0
  152. flextool/gui/dialogs/update_dialog.py +212 -0
  153. flextool/gui/downsampling.py +88 -0
  154. flextool/gui/error_handling.py +50 -0
  155. flextool/gui/execution_manager.py +1715 -0
  156. flextool/gui/execution_window.py +1377 -0
  157. flextool/gui/hover_tooltip.py +111 -0
  158. flextool/gui/input_sources.py +730 -0
  159. flextool/gui/main_window.py +6181 -0
  160. flextool/gui/network_graph.py +215 -0
  161. flextool/gui/output_actions.py +393 -0
  162. flextool/gui/output_log_window.py +159 -0
  163. flextool/gui/platform_utils.py +421 -0
  164. flextool/gui/plot_cache.py +88 -0
  165. flextool/gui/plot_canvas.py +543 -0
  166. flextool/gui/plot_config_reader.py +272 -0
  167. flextool/gui/project_utils.py +100 -0
  168. flextool/gui/result_viewer.py +4394 -0
  169. flextool/gui/scenario_key.py +162 -0
  170. flextool/gui/scenario_lists.py +516 -0
  171. flextool/gui/settings_io.py +360 -0
  172. flextool/gui/solve_reader.py +103 -0
  173. flextool/gui/tree_reorder.py +88 -0
  174. flextool/gui/ui_metrics.py +420 -0
  175. flextool/input_derivation/__init__.py +281 -0
  176. flextool/input_derivation/_commodity_ladder.py +375 -0
  177. flextool/input_derivation/_commodity_ladder_sets.py +70 -0
  178. flextool/input_derivation/_dc_power_flow.py +377 -0
  179. flextool/input_derivation/_method_constants.py +77 -0
  180. flextool/input_derivation/_process_method.py +258 -0
  181. flextool/input_derivation/_specs.py +1026 -0
  182. flextool/input_derivation/_validators.py +321 -0
  183. flextool/lean_parquet.py +159 -0
  184. flextool/model_builder/__init__.py +5 -0
  185. flextool/model_builder/build_model.py +589 -0
  186. flextool/model_builder/encoding.py +67 -0
  187. flextool/model_builder/names.py +34 -0
  188. flextool/model_builder/profiles.py +129 -0
  189. flextool/plot_outputs/__init__.py +14 -0
  190. flextool/plot_outputs/axis_helpers.py +355 -0
  191. flextool/plot_outputs/color_template.py +888 -0
  192. flextool/plot_outputs/config.py +171 -0
  193. flextool/plot_outputs/format_helpers.py +345 -0
  194. flextool/plot_outputs/legend_helpers.py +143 -0
  195. flextool/plot_outputs/orchestrator.py +1141 -0
  196. flextool/plot_outputs/perf.py +37 -0
  197. flextool/plot_outputs/plan.py +1787 -0
  198. flextool/plot_outputs/plot_bars.py +1510 -0
  199. flextool/plot_outputs/plot_bars_detail.py +753 -0
  200. flextool/plot_outputs/plot_lines.py +951 -0
  201. flextool/plot_outputs/shared_manifest.py +564 -0
  202. flextool/plot_outputs/subplot_helpers.py +137 -0
  203. flextool/process_inputs/__init__.py +188 -0
  204. flextool/process_inputs/import_old_excel_input.json +4159 -0
  205. flextool/process_inputs/read_matpower.py +451 -0
  206. flextool/process_inputs/read_old_flextool.py +1288 -0
  207. flextool/process_inputs/read_self_describing_excel.py +1423 -0
  208. flextool/process_inputs/read_tabular_with_specification.py +1114 -0
  209. flextool/process_inputs/write_old_flextool_to_db.py +3077 -0
  210. flextool/process_inputs/write_self_describing_to_db.py +977 -0
  211. flextool/process_inputs/write_to_input_db.py +269 -0
  212. flextool/process_outputs/__init__.py +7 -0
  213. flextool/process_outputs/_annualize.py +55 -0
  214. flextool/process_outputs/_inmemory_helpers.py +292 -0
  215. flextool/process_outputs/_output_meta.py +672 -0
  216. flextool/process_outputs/calc_capacity_flows.py +107 -0
  217. flextool/process_outputs/calc_connections.py +136 -0
  218. flextool/process_outputs/calc_costs.py +260 -0
  219. flextool/process_outputs/calc_group_flows.py +192 -0
  220. flextool/process_outputs/calc_slacks.py +103 -0
  221. flextool/process_outputs/calc_storage_vre.py +160 -0
  222. flextool/process_outputs/drop_levels.py +208 -0
  223. flextool/process_outputs/handoff_writers.py +1315 -0
  224. flextool/process_outputs/out_ancillary.py +544 -0
  225. flextool/process_outputs/out_capacity.py +179 -0
  226. flextool/process_outputs/out_costs.py +334 -0
  227. flextool/process_outputs/out_flowgroup.py +189 -0
  228. flextool/process_outputs/out_flows.py +301 -0
  229. flextool/process_outputs/out_group.py +475 -0
  230. flextool/process_outputs/out_node.py +190 -0
  231. flextool/process_outputs/persist_realized_slice.py +601 -0
  232. flextool/process_outputs/process_results.py +24 -0
  233. flextool/process_outputs/read_highs_solution.py +2256 -0
  234. flextool/process_outputs/read_parameters.py +1799 -0
  235. flextool/process_outputs/read_sets.py +1095 -0
  236. flextool/process_outputs/read_variables.py +553 -0
  237. flextool/process_outputs/solve_order.py +81 -0
  238. flextool/process_outputs/spinedb_replay.py +412 -0
  239. flextool/process_outputs/union_realized_slice.py +224 -0
  240. flextool/process_outputs/write_outputs.py +1286 -0
  241. flextool/process_outputs/write_spinedb.py +1267 -0
  242. flextool/representative_periods/__init__.py +5 -0
  243. flextool/representative_periods/clustering.py +165 -0
  244. flextool/representative_periods/force_include.py +563 -0
  245. flextool/representative_periods/netload.py +365 -0
  246. flextool/representative_periods/netload_inputs.py +345 -0
  247. flextool/representative_periods/netload_iterate.py +722 -0
  248. flextool/representative_periods/preprocess.py +948 -0
  249. flextool/representative_periods/scenario_stack.py +195 -0
  250. flextool/representative_periods/weights.py +124 -0
  251. flextool/scenario_comparison/__init__.py +13 -0
  252. flextool/scenario_comparison/config_builder.py +158 -0
  253. flextool/scenario_comparison/constants.py +20 -0
  254. flextool/scenario_comparison/data_models.py +222 -0
  255. flextool/scenario_comparison/db_reader.py +399 -0
  256. flextool/scenario_comparison/dispatch_data.py +1002 -0
  257. flextool/scenario_comparison/dispatch_mappings.py +205 -0
  258. flextool/scenario_comparison/dispatch_plots.py +691 -0
  259. flextool/scenario_comparison/input_entity_colors.py +319 -0
  260. flextool/scenario_comparison/orchestrator.py +453 -0
  261. flextool/scenario_comparison/plan_union.py +244 -0
  262. flextool/scenario_comparison/plot_settings_seed.py +205 -0
  263. flextool/schemas/AXIS_CONTRACT.md +71 -0
  264. flextool/schemas/canonical_databases/howto_aggregate_output.json +6225 -0
  265. flextool/schemas/canonical_databases/howto_connections.json +5606 -0
  266. flextool/schemas/canonical_databases/howto_demand.json +5518 -0
  267. flextool/schemas/canonical_databases/howto_hydro_reservoir.json +6239 -0
  268. flextool/schemas/canonical_databases/howto_hydro_reservoir_with_pump.json +5933 -0
  269. flextool/schemas/canonical_databases/howto_non_sync_and_curtailment.json +5794 -0
  270. flextool/schemas/canonical_databases/howto_ramp_and_start_up.json +5707 -0
  271. flextool/schemas/canonical_databases/howto_stochastics.json +6032 -0
  272. flextool/schemas/canonical_databases/templates_examples.json +13532 -0
  273. flextool/schemas/canonical_databases/templates_time_settings_only.json +5340 -0
  274. flextool/schemas/comparison_settings_template.json +197 -0
  275. flextool/schemas/default_plot_settings.yaml +260 -0
  276. flextool/schemas/default_plots.yaml +2293 -0
  277. flextool/schemas/flextool_axis_contract.json +303 -0
  278. flextool/schemas/flextool_axis_contract.schema.json +247 -0
  279. flextool/schemas/old_flextool_import_template.json +4443 -0
  280. flextool/schemas/output_info_template.json +48 -0
  281. flextool/schemas/output_settings_template.json +256 -0
  282. flextool/schemas/pre_v26/flextool_template_constant_default.json +2105 -0
  283. flextool/schemas/pre_v26/flextool_template_default_optional_output.json +2152 -0
  284. flextool/schemas/pre_v26/flextool_template_default_value.json +2094 -0
  285. flextool/schemas/pre_v26/flextool_template_drop_down.json +2080 -0
  286. flextool/schemas/pre_v26/flextool_template_lifetime_method.json +1990 -0
  287. flextool/schemas/pre_v26/flextool_template_optional_outputs.json +2094 -0
  288. flextool/schemas/pre_v26/flextool_template_output_node_flows.json +2105 -0
  289. flextool/schemas/pre_v26/flextool_template_results_master.json +493 -0
  290. flextool/schemas/pre_v26/flextool_template_rolling_start_remove.json +2087 -0
  291. flextool/schemas/pre_v26/flextool_template_rolling_window.json +2059 -0
  292. flextool/schemas/pre_v26/flextool_template_storage_binding_defaults.json +46 -0
  293. flextool/schemas/pre_v26/flextool_template_v2.json +1990 -0
  294. flextool/schemas/pre_v26/flextool_template_v25.json +3864 -0
  295. flextool/schemas/spinedb_results_schema.json +581 -0
  296. flextool/schemas/spinedb_schema.json +4636 -0
  297. flextool/solver_config/copt.opt.template +18 -0
  298. flextool/solver_config/cplex.opt.template +25 -0
  299. flextool/solver_config/gurobi.opt.template +18 -0
  300. flextool/solver_config/highs.opt.template +18 -0
  301. flextool/solver_config/xpress.opt.template +26 -0
  302. flextool/spinedb_backend/__init__.py +26 -0
  303. flextool/spinedb_backend/_axis_enums.py +1119 -0
  304. flextool/spinedb_backend/_backend.py +1139 -0
  305. flextool/update_flextool/__init__.py +12 -0
  306. flextool/update_flextool/canonical_databases.py +251 -0
  307. flextool/update_flextool/db_migration.py +7108 -0
  308. flextool/update_flextool/ensure_settings_db.py +138 -0
  309. flextool/update_flextool/export_database.py +103 -0
  310. flextool/update_flextool/extend_tests_fixture.py +772 -0
  311. flextool/update_flextool/generate_canonical.py +274 -0
  312. flextool/update_flextool/initialize_database.py +42 -0
  313. flextool/update_flextool/install_info.py +225 -0
  314. flextool/update_flextool/self_update.py +464 -0
  315. flextool/update_flextool/sync_master_json_template.py +125 -0
  316. flextool/update_flextool/test_fixtures.py +187 -0
  317. flextool-4.0.0.dist-info/METADATA +217 -0
  318. flextool-4.0.0.dist-info/RECORD +322 -0
  319. flextool-4.0.0.dist-info/WHEEL +5 -0
  320. flextool-4.0.0.dist-info/entry_points.txt +17 -0
  321. flextool-4.0.0.dist-info/licenses/LICENSE.txt +19 -0
  322. flextool-4.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,2256 @@
1
+ """
2
+ Direct HiGHS solution → parquet extractor.
3
+
4
+ Pipeline: HiGHS (in-memory) → VARIABLE_SPECS harvest →
5
+ ``output_parquet/*.parquet`` → downstream pandas.
6
+
7
+ Design
8
+ ------
9
+ HiGHS exposes the MPS column names through
10
+ ``highspy.Highs.allVariableNames()`` and the solution through
11
+ ``Highs.getSolution().col_value``. The two arrays are index-aligned.
12
+
13
+ MPS variable names look like::
14
+
15
+ <var_name>[<idx1>,<idx2>,...,<period>,<time>] # time-indexed vars
16
+ <var_name>[<idx1>,...,<period>] # period-only vars (e.g. v_invest)
17
+
18
+ Splitting on '[', ',' and ']' recovers the indices. This assumes entity
19
+ names never contain commas or brackets — a requirement for unambiguous
20
+ MPS emission.
21
+
22
+ We retain only the ``dt_realize_dispatch`` timesteps — rolling-window
23
+ solves compute a longer horizon but only realize the first chunk. The
24
+ filter runs inside the single pass over HiGHS names (O(1) set lookup)
25
+ so non-realized rows are never materialised.
26
+
27
+ Storage layout: wide. Row MultiIndex = ``(solve, period, time)`` (or
28
+ ``(solve, period)`` for non-time vars), column (Multi)Index = the
29
+ remaining variable indices. Matches the shape of
30
+ ``read_variables.read_variables`` outputs exactly, so downstream
31
+ ``calc_*.py`` code is unchanged. Persisted via
32
+ ``flextool.lean_parquet.write_lean_parquet`` — compact level-name
33
+ metadata in the parquet footer.
34
+
35
+ Public API
36
+ ----------
37
+ * :data:`VARIABLE_SPECS` — list of :class:`VariableSpec` describing every
38
+ variable that has parquet coverage. Add a new line here to add a new
39
+ variable.
40
+ * :func:`extract_variable` — single variable → wide DataFrame.
41
+ * :func:`write_variable_parquet` — single variable → parquet file.
42
+ * :func:`write_all_variables` — iterate :data:`VARIABLE_SPECS` and write
43
+ one parquet per variable for a given solve.
44
+ * Loading: use :func:`flextool.lean_parquet.read_lean_parquet` directly.
45
+
46
+ Usage from solver_runner (after a successful ``h.run()``)::
47
+
48
+ from flextool.process_outputs.read_highs_solution import write_all_variables
49
+ write_all_variables(
50
+ h, solve_name=current_solve,
51
+ output_dir=wf / "output_raw",
52
+ realized_dispatch_csv=wf / "solve_data/realized_dispatch.csv",
53
+ )
54
+ """
55
+ from __future__ import annotations
56
+
57
+ import argparse
58
+ import logging
59
+ import os
60
+ import re
61
+ from pathlib import Path
62
+ from typing import NamedTuple, Sequence, TYPE_CHECKING
63
+
64
+ import pandas as pd
65
+ import polars as pl
66
+
67
+ from flextool.lean_parquet import write_lean_parquet
68
+
69
+ if TYPE_CHECKING:
70
+ import highspy
71
+
72
+ from flextool.engine_polars.input import FlexData
73
+
74
+ _logger = logging.getLogger(__name__)
75
+
76
+
77
+ # ---------------------------------------------------------------------------
78
+ # Provider-aware lookup helper — Provider-first, then ``None`` (caller
79
+ # falls back to its own disk read). Provider key uses the
80
+ # parent-qualified convention (``"<parent>/<basename>"`` without
81
+ # ``.csv``).
82
+
83
+ def _provider_lookup(provider: "object | None", path: "Path | str"):
84
+ """Return the polars frame for *path* sourced from the Provider, or
85
+ ``None`` when the Provider doesn't carry it (caller falls back to
86
+ its own disk read).
87
+ """
88
+ p = Path(path)
89
+ parent = p.parent.name
90
+ stem = p.stem
91
+ name = f"{parent}/{stem}" if parent else stem
92
+ if provider is not None and provider.has(name):
93
+ return provider.get(name)
94
+ return None
95
+
96
+
97
+ # ---------------------------------------------------------------------------
98
+ # Variable registry
99
+ # ---------------------------------------------------------------------------
100
+
101
+
102
+ class VariableSpec(NamedTuple):
103
+ """Describes one HiGHS result quantity for parquet extraction.
104
+
105
+ Covers three source arrays on the solved ``Highs`` instance:
106
+
107
+ * ``col_value`` (default) — primal variable values, keyed by
108
+ ``allVariableNames()``.
109
+ * ``col_dual`` — variable reduced costs, same keying as
110
+ ``col_value``. Used for ``v_invest.dual`` / ``v_divest.dual``.
111
+ * ``row_dual`` — constraint dual values, keyed by
112
+ ``getLp().row_names_``. Used for investment-cap duals,
113
+ ``nodeBalance_eq``, CO2-limit duals, …
114
+
115
+ Attributes
116
+ ----------
117
+ name:
118
+ Name as it appears in the MPS. Variable name for
119
+ ``col_value``/``col_dual`` sources; constraint name for
120
+ ``row_dual``. Must match the prefix in ``<name>[idx1,...]``.
121
+ col_names:
122
+ Names of the leading "column" indices before the trailing
123
+ ``period[, time]`` (or just ``period``, or nothing — see
124
+ :attr:`has_time` and :attr:`has_period`).
125
+ has_time:
126
+ True when indexed by both period and time. False for
127
+ period-only quantities.
128
+ has_period:
129
+ True (default) when the trailing index (after ``col_names``)
130
+ includes at least a period. False for quantities indexed only
131
+ by the column fields (e.g. ``co2_max_total[g]``) — in that case
132
+ the row index collapses to just ``(solve,)``.
133
+ source:
134
+ Which HiGHS array the values come from — one of
135
+ ``"col_value"`` (default), ``"col_dual"``, ``"row_dual"``.
136
+ value_scale:
137
+ Multiplier applied to every raw value. Typically ``1e6``
138
+ (``1 / scale_the_objective``) for duals.
139
+ output_name:
140
+ Parquet file prefix. Defaults to :attr:`name`. Set when the
141
+ constraint name differs from the desired output identifier —
142
+ e.g. constraint ``maxInvest_entity_period`` is written as
143
+ ``v_dual_maxInvest_period``.
144
+ """
145
+
146
+ name: str
147
+ col_names: tuple[str, ...]
148
+ has_time: bool = True
149
+ has_period: bool = True
150
+ source: str = "col_value"
151
+ value_scale: float = 1.0
152
+ output_name: str | None = None
153
+ # Column fields that appear AFTER the period (and time) in the
154
+ # bracket list. Needed for variables whose declared subscript order
155
+ # puts a column index after the period — e.g. ``v_trade[c, n, d, i]``
156
+ # where ``i`` (tier) is a column but sits to the right of ``d``
157
+ # (period). Parsed-out values are concatenated onto ``col_names`` to
158
+ # form the full column tuple.
159
+ trailing_col_names: tuple[str, ...] = ()
160
+ # Multi-source fan-out: when the output quantity is the sum of two or
161
+ # more HiGHS variables (e.g. a two-tier slack split into
162
+ # ``vq_foo_primary`` + ``vq_foo_escape``), list the source variable
163
+ # names here. ``name`` then becomes a pure logical identifier used
164
+ # for the parquet file name; the extractor reads each source
165
+ # separately, aligns on the row+column MultiIndex, and adds them.
166
+ # ``None`` (default) preserves the legacy single-source behaviour.
167
+ # All sources must share the same index shape — ``col_names``,
168
+ # ``has_time``, ``has_period``, ``trailing_col_names`` apply
169
+ # uniformly to every source in the tuple.
170
+ derived_from: tuple[str, ...] | None = None
171
+ # Agent 9 — row-scaling un-scaling. When non-None, the extracted
172
+ # frame is multiplied element-wise by the corresponding row scaler
173
+ # read from ``solve_data/{node,group}_capacity_for_scaling.csv``
174
+ # before parquet emission. Column names in ``col_names`` supply
175
+ # the entity axis of the scaler; scalers are per-(entity, period).
176
+ # Recognised values:
177
+ # * "node_cap" — multiply cell[(d, t), n] by node_cap[n, d].
178
+ # Used for vq_state_up / vq_state_down. Also
179
+ # divides by node_cap when the quantity is a dual
180
+ # of a row-scaled balance constraint (see
181
+ # ``unscale_dual=True`` below for that case).
182
+ # * "group_cap" — multiply by group_cap[g, d]; used for
183
+ # vq_non_synchronous, vq_state_up_group,
184
+ # vq_capacity_margin (no t axis).
185
+ # Mode A (flag off): node_cap / group_cap CSVs default to 1 so this
186
+ # is a no-op. Mode B: recovers absolute CSV magnitudes matching the
187
+ # pre-row-scaling (Agent 1) baselines.
188
+ unscale_by: str | None = None
189
+ # Agent 1.8 — block-aware output expansion. When non-None, the
190
+ # extracted frame is broadcast from the variable's temporal-resolution
191
+ # block down to the finest timeline. The raw MPS only emits values
192
+ # at the block's coarse ``(period, step)`` pairs; fine steps covered
193
+ # by a coarse step receive the same value so downstream CSV / parquet
194
+ # readers always see a rectangular fine-grid frame (design rule
195
+ # from Agent 1.8: "print all at finest resolution and drop the block
196
+ # dimension").
197
+ # Recognised values:
198
+ # * "process_block" — lookup per ``col_names[0]`` in
199
+ # ``solve_data/process_block.csv``. Used for
200
+ # v_flow, v_ramp, v_online_*, v_startup_*,
201
+ # v_shutdown_* (every process-scoped var).
202
+ # * "node_block" — lookup in ``solve_data/entity_block.csv``.
203
+ # Used for v_state, v_angle.
204
+ # In the degenerate case every entity maps to ``"default"`` so the
205
+ # overlap lookup is identity and the broadcast is a no-op.
206
+ expand_by: str | None = None
207
+ # Canonical-row period axis for period-only (``has_period and not
208
+ # has_time``) variables. ``"dispatch"`` (default) sources the row
209
+ # order from the roll's realized-DISPATCH periods
210
+ # (``flex_data.realized_dispatch`` / ``p_years_from_start_d.csv``);
211
+ # ``"invest"`` sources it from the roll's realized-INVEST periods
212
+ # (``realized_invest_periods_of_current_solve.csv`` / provider). Only
213
+ # ``v_invest`` / ``v_divest`` use ``"invest"`` — they are decided on
214
+ # the investment axis, so a dispatch-only roll must contribute no
215
+ # rows (see the registry note on those specs). Ignored for
216
+ # ``has_time`` and ``has_period=False`` variables.
217
+ period_source: str = "dispatch"
218
+
219
+
220
+ # Registry — add a new line here to add a new variable to the pipeline.
221
+ #
222
+ # ``scale_the_objective`` was a hardcoded ``1e-6`` in ``flextool_base.dat``
223
+ # (legacy); Agent 12 centralised it in Python — the value is written per
224
+ # solve to ``solve_data/scale_the_objective.csv`` from the Agent-8
225
+ # ScaleTable. Every dual of an objective-scaled constraint needs
226
+ # multiplication by ``1/scale_the_objective`` to undo the scaling.
227
+ #
228
+ # ``_INV_SCALE_THE_OBJECTIVE`` is retained as the **default** multiplier
229
+ # (matches the legacy 1e-6 scalar) — used as the sentinel value wired
230
+ # into ``VariableSpec.value_scale`` for dual specs. At write time,
231
+ # :func:`_resolve_inv_scale_the_objective` reads the current solve's
232
+ # ``solve_data/scale_the_objective.csv`` and replaces this default with
233
+ # the live reciprocal so per-solve scalar changes propagate correctly.
234
+ _INV_SCALE_THE_OBJECTIVE = 1e6
235
+
236
+ _DEFAULT_SCALE_THE_OBJECTIVE = 1e-6
237
+
238
+
239
+ def _resolve_inv_scale_the_objective(
240
+ work_folder: Path | str | None,
241
+ scale_the_objective: float | None = None,
242
+ ) -> float:
243
+ """Return ``1 / scale_the_objective`` for the current solve.
244
+
245
+ Phase G — when ``scale_the_objective`` is supplied directly (cascade-
246
+ threaded), use it and skip the disk read entirely. Otherwise reads
247
+ ``<work_folder>/solve_data/scale_the_objective.csv`` (Agent 12;
248
+ emitted by :func:`flextool.engine_polars._emit_solve_writers.write_scale_the_objective`).
249
+ Falls back to ``1 / 1e-6`` when the file is missing / empty /
250
+ unreadable — mirrors the ``default 1e-6`` clause on
251
+ ``param scale_the_objective`` in ``flextool.mod``.
252
+ """
253
+ if scale_the_objective is not None and scale_the_objective > 0:
254
+ return 1.0 / float(scale_the_objective)
255
+ if work_folder is None:
256
+ return 1.0 / _DEFAULT_SCALE_THE_OBJECTIVE
257
+ path = Path(work_folder) / "solve_data" / "scale_the_objective.csv"
258
+ if not path.exists():
259
+ return 1.0 / _DEFAULT_SCALE_THE_OBJECTIVE
260
+ try:
261
+ df = pd.read_csv(path)
262
+ except Exception:
263
+ return 1.0 / _DEFAULT_SCALE_THE_OBJECTIVE
264
+ if df.empty or "value" not in df.columns:
265
+ return 1.0 / _DEFAULT_SCALE_THE_OBJECTIVE
266
+ try:
267
+ val = float(df["value"].iloc[0])
268
+ except (ValueError, TypeError):
269
+ return 1.0 / _DEFAULT_SCALE_THE_OBJECTIVE
270
+ if not (val > 0):
271
+ return 1.0 / _DEFAULT_SCALE_THE_OBJECTIVE
272
+ return 1.0 / val
273
+
274
+
275
+ # Helper — every investment-constraint-dual entry uses the same scale to
276
+ # undo ``scale_the_objective`` and writes its parquet under ``v_dual_<…>``.
277
+ def _invest_dual(
278
+ mps_name: str,
279
+ col: str,
280
+ output_suffix: str,
281
+ *,
282
+ has_period: bool = True,
283
+ derived_from: tuple[str, ...] | None = None,
284
+ ) -> VariableSpec:
285
+ # ``derived_from`` is set for invest-cap families whose producer
286
+ # splits the LHS into process-side (``..._p``) and node-side
287
+ # (``..._n``) constraints (model.py / _cumulative_invest.py). The
288
+ # bracketed entity columns of the two sides are disjoint (process
289
+ # names vs. node names), so ``df.add(fill_value=0.0)`` cleanly
290
+ # unions them into one wide frame at the output boundary.
291
+ return VariableSpec(
292
+ name=mps_name,
293
+ col_names=(col,),
294
+ has_time=False,
295
+ has_period=has_period,
296
+ source="row_dual",
297
+ value_scale=_INV_SCALE_THE_OBJECTIVE,
298
+ output_name=f"v_dual_{output_suffix}",
299
+ derived_from=derived_from,
300
+ )
301
+
302
+
303
+ VARIABLE_SPECS: list[VariableSpec] = [
304
+ # -- Time-indexed decision variables ------------------------------------
305
+ # Agent 1.8: ``expand_by`` broadcasts coarse-block values to every
306
+ # covered fine timestep. Degenerate (every entity on 'default'): no-op.
307
+ VariableSpec("v_flow", ("process", "source", "sink"), expand_by="process_block"),
308
+ # Reverse-flow auxiliary for method_2way_1var_off arcs. The signed
309
+ # net flow on such an arc is ``v_flow - v_flow_back``; the output layer
310
+ # folds it into ``r.flow_dt`` (calc_capacity_flows) so the reported
311
+ # connection flow carries the correct sign (negative = sink→source).
312
+ VariableSpec("v_flow_back", ("process", "source", "sink"), expand_by="process_block"),
313
+ VariableSpec("v_ramp", ("process", "source", "sink"), expand_by="process_block"),
314
+ # Reserve participants are pinned to the default block in V1 (Agent
315
+ # 1.7), so v_reserve effectively needs no expansion — but the
316
+ # broadcast is still safely identity there.
317
+ VariableSpec("v_reserve", ("process", "reserve", "updown", "node"), expand_by="process_block"),
318
+ VariableSpec("v_state", ("node",), expand_by="node_block"),
319
+ VariableSpec("v_online_linear", ("process",), expand_by="process_block"),
320
+ VariableSpec("v_startup_linear", ("process",), expand_by="process_block"),
321
+ VariableSpec("v_shutdown_linear", ("process",), expand_by="process_block"),
322
+ VariableSpec("v_online_integer", ("process",), expand_by="process_block"),
323
+ VariableSpec("v_startup_integer", ("process",), expand_by="process_block"),
324
+ VariableSpec("v_shutdown_integer", ("process",), expand_by="process_block"),
325
+ VariableSpec("v_angle", ("node",), expand_by="node_block"),
326
+
327
+ # -- Time-indexed slack / penalty variables -----------------------------
328
+ # Agent 9 ``unscale_by="node_cap"`` / ``"group_cap"`` un-scales row
329
+ # scaling when ``use_row_scaling=yes`` (no-op in Mode A where the
330
+ # scaler defaults to 1). See flextool/SLACK_CONVENTION.md for the
331
+ # single-variable slack convention.
332
+ # Agent 1.8: vq_state_up / vq_state_down appear in the node balance,
333
+ # which is emitted at the node's block — broadcast via node_block.
334
+ VariableSpec(
335
+ "vq_state_up", ("node",),
336
+ unscale_by="node_cap", expand_by="node_block",
337
+ ),
338
+ VariableSpec(
339
+ "vq_state_down", ("node",),
340
+ unscale_by="node_cap", expand_by="node_block",
341
+ ),
342
+ # Column level names must match the CSV reader
343
+ # (``read_variables._read_from_csv``) so cross-reader mul aligns
344
+ # cleanly — both readers use ('reserve', 'updown', 'node_group').
345
+ # Reserves + inertia pinned to default block (Agent 1.7 V1) — no
346
+ # expansion needed.
347
+ VariableSpec("vq_reserve", ("reserve", "updown", "node_group")),
348
+ VariableSpec("vq_inertia", ("group",)),
349
+ VariableSpec(
350
+ "vq_non_synchronous", ("group",),
351
+ unscale_by="group_cap",
352
+ ),
353
+ # group_loss_share_constraint is emitted at every fine (d, t), so
354
+ # vq_state_up_group lives on the fine timeline directly.
355
+ VariableSpec(
356
+ "vq_state_up_group", ("group",),
357
+ unscale_by="group_cap",
358
+ ),
359
+
360
+ # -- Period-only (no time) decision / slack variables -------------------
361
+ # ``v_invest`` / ``v_divest`` are realized on the INVEST axis, not the
362
+ # dispatch axis. In nested / rolling-dispatch cascades the dispatch
363
+ # rolls realize a period for dispatch but NO investment, yet the
364
+ # densified column union (8e2de938 / a9058c66) still materialises a
365
+ # zero-valued ``(roll, dispatch_period)`` row for every invest-eligible
366
+ # entity. After ``drop_levels`` strips the ``solve`` level and dedups
367
+ # ``keep='last'``, that trailing dispatch-roll zero overwrites the
368
+ # invest step's real value for the shared period, zeroing the
369
+ # ``invested`` breakdown in ``unit_capacity__d`` (and friends). Pin
370
+ # the canonical row axis to the roll's realized-INVEST periods so a
371
+ # dispatch-only roll emits no spurious invest rows and the invest
372
+ # step's value is the sole contributor at union time.
373
+ VariableSpec("v_invest", ("entity",), has_time=False, period_source="invest"),
374
+ VariableSpec("v_divest", ("entity",), has_time=False, period_source="invest"),
375
+ # No t axis; the row scaler is still keyed by (g, d).
376
+ VariableSpec(
377
+ "vq_capacity_margin", ("group",), has_time=False,
378
+ unscale_by="group_cap",
379
+ ),
380
+
381
+ # -- Commodity-ladder period-level trade ---------------------------------
382
+ # ``v_trade[c, n, d, i]`` — no time, no branch. ``tier`` sits after
383
+ # the period in the MPS bracket order so it's declared via
384
+ # ``trailing_col_names`` (as opposed to ``col_names`` which are
385
+ # parsed from the leading positions). Output column MultiIndex is
386
+ # ``(commodity, node, tier)`` — the logical column tuple.
387
+ VariableSpec(
388
+ "v_trade", ("commodity", "node"),
389
+ has_time=False,
390
+ trailing_col_names=("tier",),
391
+ ),
392
+
393
+ # -- Investment-cap duals (period-only, simple 1/scale transform) -------
394
+ # Most of these families are emitted by a split process/node producer
395
+ # — ``maxInvest_entity_period_p`` (model.py:2319) for the process arm
396
+ # and ``maxInvest_entity_period_n`` (model.py:2336) for the node arm
397
+ # etc. — so the reader matches both via ``derived_from`` and unions
398
+ # the resulting frames.
399
+ _invest_dual("maxInvest_entity_period", "entity", "maxInvest_period",
400
+ derived_from=("maxInvest_entity_period_p",
401
+ "maxInvest_entity_period_n")),
402
+ # ``maxInvest_entity_total`` is asymmetric: process side keeps the
403
+ # bare name (model.py:2442), node side gets the ``_n`` suffix
404
+ # (model.py:2494). Period is summed out inside ``Sum(over=("d",))``
405
+ # — no period component in either constraint name.
406
+ _invest_dual("maxInvest_entity_total", "entity", "maxInvest_total",
407
+ has_period=False,
408
+ derived_from=("maxInvest_entity_total",
409
+ "maxInvest_entity_total_n")),
410
+ _invest_dual("maxCumulative_capacity", "entity", "maxCumulative",
411
+ derived_from=("maxCumulative_capacity_p",
412
+ "maxCumulative_capacity_n")),
413
+ _invest_dual("maxInvestGroup_entity_period", "group", "maxInvestGroup_period",
414
+ derived_from=("maxInvestGroup_entity_period_p",
415
+ "maxInvestGroup_entity_period_n")),
416
+ _invest_dual("maxInvestGroup_entity_total", "group", "maxInvestGroup_total",
417
+ derived_from=("maxInvestGroup_entity_total_p",
418
+ "maxInvestGroup_entity_total_n")),
419
+ # ``maxInvestGroup_entity_cumulative`` keeps a single non-suffixed
420
+ # name (_cumulative_invest.py:990).
421
+ _invest_dual("maxInvestGroup_entity_cumulative", "group", "maxInvestGroup_cumulative"),
422
+
423
+ # -- Investment-floor (min-side) duals ----------------------------------
424
+ # Mirror of the maxInvest families above, reading the row duals of the
425
+ # ``>=`` lower-floor constraints emitted by ``_emit_*_minmax`` /
426
+ # ``_emit_group_invest_*`` (flextool/engine_polars/_cumulative_invest.py).
427
+ # Purely additive (Increment 3): makes ``v_dual_min*`` available for the
428
+ # later synthesis increment; nothing consumes these yet. Same scale
429
+ # sentinel + ``source="row_dual"`` as the max-side; absent families
430
+ # degrade to an empty frame identically.
431
+ #
432
+ # ``minInvest_entity_period`` splits process (``_p``) / node (``_n``)
433
+ # arms (_cumulative_invest.py:383,397) — same shape as the max-side.
434
+ _invest_dual("minInvest_entity_period", "entity", "minInvest_period",
435
+ derived_from=("minInvest_entity_period_p",
436
+ "minInvest_entity_period_n")),
437
+ # ``minInvest_entity_total`` is NOT name-symmetric with its max-side
438
+ # counterpart: the process arm carries the ``_p`` suffix
439
+ # (``minInvest_entity_total_p``, _cumulative_invest.py:460) rather than
440
+ # the bare name ``maxInvest_entity_total`` keeps. More importantly, the
441
+ # min invest-total constraint is indexed per ``(entity, period)`` —
442
+ # ``over = (p|n, d)`` (_cumulative_invest.py:450,490) with the LHS
443
+ # summing over ``d_invest`` — so it KEEPS a period axis. The max-side
444
+ # total sums the period out (``Sum(over=("d",))``, model.py:2578) and is
445
+ # therefore ``has_period=False``. We use the default ``has_period=True``
446
+ # to match the actual min constraint's row bracket.
447
+ _invest_dual("minInvest_entity_total", "entity", "minInvest_total",
448
+ derived_from=("minInvest_entity_total_p",
449
+ "minInvest_entity_total_n")),
450
+ _invest_dual("minCumulative_capacity", "entity", "minCumulative",
451
+ derived_from=("minCumulative_capacity_p",
452
+ "minCumulative_capacity_n")),
453
+ _invest_dual("minInvestGroup_entity_period", "group", "minInvestGroup_period",
454
+ derived_from=("minInvestGroup_entity_period_p",
455
+ "minInvestGroup_entity_period_n")),
456
+ _invest_dual("minInvestGroup_entity_total", "group", "minInvestGroup_total",
457
+ derived_from=("minInvestGroup_entity_total_p",
458
+ "minInvestGroup_entity_total_n")),
459
+ # ``minInvestGroup_entity_cumulative`` keeps a single non-suffixed
460
+ # name (_cumulative_invest.py:1001), like the max-side cumulative.
461
+ _invest_dual("minInvestGroup_entity_cumulative", "group", "minInvestGroup_cumulative"),
462
+
463
+ # -- CO2 emission-cap duals ---------------------------------------------
464
+ # The model writes ``co2_max_*.dual / scale_the_objective``. Downstream
465
+ # Python processing applies the extra ``/1000`` (scaled RHS) and
466
+ # ``/inflation`` corrections — not our concern here.
467
+ VariableSpec(
468
+ name="co2_max_period", col_names=("group",),
469
+ has_time=False, source="row_dual",
470
+ value_scale=_INV_SCALE_THE_OBJECTIVE,
471
+ output_name="v_dual_co2_max_period",
472
+ ),
473
+ VariableSpec(
474
+ name="co2_max_total", col_names=("group",),
475
+ has_time=False, has_period=False, source="row_dual",
476
+ value_scale=_INV_SCALE_THE_OBJECTIVE,
477
+ output_name="v_dual_co2_max_total",
478
+ ),
479
+
480
+ # Not in VARIABLE_SPECS (handled by dedicated writers below):
481
+ # * v_dual_node_balance — per-period inflation scaling needs a
482
+ # custom writer (see write_v_dual_node_balance below).
483
+ # * v_dual_reserve__upDown__group__period__t — max() across up to 3
484
+ # constraint duals per (r,ud,g,d,t).
485
+ # * v_dual_invest_{unit,connection,node} — v_invest.dual split by
486
+ # entity class (see write_v_dual_invest_by_class below).
487
+ # * v_obj — scalar, see write_v_obj below.
488
+ ]
489
+
490
+
491
+ # ---------------------------------------------------------------------------
492
+ # Core helpers
493
+ # ---------------------------------------------------------------------------
494
+
495
+
496
+ def _load_realized_set(
497
+ realized_dispatch_csv: Path | str | None,
498
+ *,
499
+ provider: "object | None" = None,
500
+ ) -> set[tuple[str, str]] | None:
501
+ """Return ``{(period, time), …}`` from ``realized_dispatch.csv``, or None.
502
+
503
+ Used for O(1) membership checks in :func:`extract_variable`. For
504
+ synthesis of empty-frame row order (which must match the canonical
505
+ iteration order in phase-1 printfs), use
506
+ :func:`_load_realized_list` instead — an ordered variant.
507
+ """
508
+ if realized_dispatch_csv is None:
509
+ return None
510
+ path = Path(realized_dispatch_csv)
511
+ # Step 1-e — Provider-aware: under the in-memory cascade the file
512
+ # isn't on disk but the per-sub-solve Provider has the frame. The
513
+ # transitional seed-funnel fallback in :func:`_provider_lookup`
514
+ # keeps unplumbed callers working during the dual-write window.
515
+ seeded = _provider_lookup(provider, path)
516
+ if seeded is not None:
517
+ period_col = "period"
518
+ time_col = "step" if "step" in seeded.columns else "time"
519
+ return set(zip(
520
+ seeded[period_col].cast(str).to_list(),
521
+ seeded[time_col].cast(str).to_list(),
522
+ ))
523
+ if not path.exists():
524
+ # Expected fallback path on every solve where the previous
525
+ # iteration didn't realise any timesteps yet -- write all.
526
+ # Not a warning; just informational.
527
+ _logger.debug("realized_dispatch file missing, writing all timesteps: %s", path)
528
+ return None
529
+ realized = pd.read_csv(path)
530
+ period_col = "period"
531
+ time_col = "step" if "step" in realized.columns else "time"
532
+ return set(
533
+ zip(
534
+ realized[period_col].astype(str).to_list(),
535
+ realized[time_col].astype(str).to_list(),
536
+ )
537
+ )
538
+
539
+
540
+ def _load_realized_list(
541
+ realized_dispatch_csv: Path | str | None,
542
+ *,
543
+ provider: "object | None" = None,
544
+ ) -> list[tuple[str, str]] | None:
545
+ """Return ``[(period, time), …]`` from ``realized_dispatch.csv`` in file order.
546
+
547
+ The CSV file order matches the canonical
548
+ ``for {(d, t) in dt_realize_dispatch}`` iteration order that every
549
+ dt-indexed phase-1 printf uses — so synthesizing empty-frame rows
550
+ in this order reproduces the CSV-read parameter's row order exactly.
551
+ """
552
+ if realized_dispatch_csv is None:
553
+ return None
554
+ path = Path(realized_dispatch_csv)
555
+ seeded = _provider_lookup(provider, path)
556
+ if seeded is not None:
557
+ period_col = "period"
558
+ time_col = "step" if "step" in seeded.columns else "time"
559
+ return list(zip(
560
+ seeded[period_col].cast(str).to_list(),
561
+ seeded[time_col].cast(str).to_list(),
562
+ ))
563
+ if not path.exists():
564
+ return None
565
+ realized = pd.read_csv(path)
566
+ period_col = "period"
567
+ time_col = "step" if "step" in realized.columns else "time"
568
+ return list(
569
+ zip(
570
+ realized[period_col].astype(str).to_list(),
571
+ realized[time_col].astype(str).to_list(),
572
+ )
573
+ )
574
+
575
+
576
+ def _load_realized_periods(
577
+ realized_periods_csv: Path | str | None,
578
+ *,
579
+ provider: "object | None" = None,
580
+ ) -> set[str] | None:
581
+ """Return ``{period, …}`` from a ``period``-only CSV, or None."""
582
+ if realized_periods_csv is None:
583
+ return None
584
+ path = Path(realized_periods_csv)
585
+ seeded = _provider_lookup(provider, path)
586
+ if seeded is not None:
587
+ return set(seeded["period"].cast(str).to_list())
588
+ if not path.exists():
589
+ _logger.debug("realized periods file missing, writing all periods: %s", path)
590
+ return None
591
+ realized = pd.read_csv(path)
592
+ return set(realized["period"].astype(str).to_list())
593
+
594
+
595
+ def _load_realized_periods_list(
596
+ realized_periods_csv: Path | str | None,
597
+ ) -> list[str] | None:
598
+ """Return ``[period, …]`` in CSV file order."""
599
+ if realized_periods_csv is None:
600
+ return None
601
+ path = Path(realized_periods_csv)
602
+ if not path.exists():
603
+ return None
604
+ realized = pd.read_csv(path)
605
+ return list(realized["period"].astype(str).to_list())
606
+
607
+
608
+ def _load_realized_invest_periods_list(
609
+ realized_periods_csv: Path | str | None,
610
+ *,
611
+ provider: "object | None" = None,
612
+ ) -> list[str] | None:
613
+ """Return the roll's realized-INVEST ``[period, …]`` in source order.
614
+
615
+ Canonical row source for the investment-axis variables
616
+ (``v_invest`` / ``v_divest``). Prefers the in-memory provider frame
617
+ (``solve_data/realized_invest_periods_of_current_solve``); falls back
618
+ to the CSV. Returns the (possibly EMPTY) ordered period list when the
619
+ source is present — an empty list means this roll realizes no
620
+ investment, so the variable extractor must emit no rows for it.
621
+ Returns ``None`` only when no source is available at all, signalling
622
+ the caller to fall back to the dispatch-axis canonical order (e.g.
623
+ single-solve fixtures that never emit the invest-period set).
624
+ """
625
+ seeded = (
626
+ _provider_lookup(provider, realized_periods_csv)
627
+ if realized_periods_csv is not None
628
+ else None
629
+ )
630
+ if seeded is not None:
631
+ return list(seeded["period"].cast(str).to_list())
632
+ if realized_periods_csv is None:
633
+ return None
634
+ path = Path(realized_periods_csv)
635
+ if not path.exists():
636
+ return None
637
+ realized = pd.read_csv(path)
638
+ return list(realized["period"].astype(str).to_list())
639
+
640
+
641
+ def _name_regex(var_name: str) -> re.Pattern[str]:
642
+ """Return a compiled regex matching ``<var_name>[...]``."""
643
+ return re.compile(rf"^{re.escape(var_name)}\[(.+)\]$")
644
+
645
+
646
+ def _load_canonical_dt_order(
647
+ work_folder: Path | str | None,
648
+ solve_name: str,
649
+ *,
650
+ flex_data: "FlexData | None" = None,
651
+ ) -> list[tuple[str, str]] | None:
652
+ """Return ``[(period, time), …]`` in the canonical iteration order.
653
+
654
+ When ``flex_data`` is supplied, prefer the in-memory
655
+ ``flex_data.realized_dispatch`` (already a polars frame of
656
+ ``(period, step)``); skip the CSV read entirely.
657
+
658
+ Source priority (file fallback):
659
+ 1. ``solve_data/p_step_duration.csv`` — written by the phase-1
660
+ printf ``for {s in solve_current, (d, t) in dt_realize_dispatch}``.
661
+ 2. ``solve_data/dt_realize_dispatch_set.csv`` — the polars
662
+ cascade's authoritative emission set (mirrors the .mod's
663
+ ``dt_realize_dispatch`` after the ``output_horizon`` toggle).
664
+ Carries forecast-branch rows for stochastic scenarios where
665
+ ``realized_dispatch.csv`` is anchor-only by design.
666
+
667
+ Every dt-indexed phase-1 printf in ``flextool.mod`` uses the same
668
+ set iteration, so this sequence is the row order ALL parameter
669
+ CSVs of dt arity have.
670
+
671
+ Filtered to ``solve_name`` when the CSV carries a ``solve`` column;
672
+ ``dt_realize_dispatch_set.csv`` is per-solve already (no solve col).
673
+ Returns ``None`` when neither file is present.
674
+ """
675
+ if flex_data is not None and getattr(flex_data, "realized_dispatch", None) is not None:
676
+ try:
677
+ rd = flex_data.realized_dispatch
678
+ cols = rd.columns
679
+ time_col = "step" if "step" in cols else ("time" if "time" in cols else cols[1])
680
+ return list(zip(
681
+ rd["period"].cast(str).to_list(),
682
+ rd[time_col].cast(str).to_list(),
683
+ ))
684
+ except Exception: # noqa: BLE001
685
+ pass
686
+ if work_folder is None:
687
+ return None
688
+ sd = Path(work_folder) / "solve_data"
689
+ psd = sd / "p_step_duration.csv"
690
+ if psd.exists():
691
+ df = pd.read_csv(psd, usecols=["solve", "period", "time"], dtype=str)
692
+ df = df[df["solve"] == str(solve_name)]
693
+ return list(zip(df["period"].to_list(), df["time"].to_list()))
694
+ drd = sd / "dt_realize_dispatch_set.csv"
695
+ if drd.exists():
696
+ df = pd.read_csv(drd, usecols=["period", "time"], dtype=str)
697
+ return list(zip(df["period"].to_list(), df["time"].to_list()))
698
+ return None
699
+
700
+
701
+ def _load_canonical_d_order(
702
+ work_folder: Path | str | None,
703
+ solve_name: str,
704
+ *,
705
+ flex_data: "FlexData | None" = None,
706
+ ) -> list[str] | None:
707
+ """Return ``[period, …]`` in the canonical iteration order.
708
+
709
+ When ``flex_data.realized_dispatch`` is in memory, derive the
710
+ ordered distinct period list directly (skipping the disk read).
711
+
712
+ Source (file fallback): ``solve_data/p_years_from_start_d.csv`` —
713
+ written by the phase-1 printf ``for {s in solve_current, d in
714
+ d_realize_dispatch_or_invest}``. Period-indexed parameter CSVs
715
+ use the same iteration. Filtered to ``solve_name``. Returns
716
+ ``None`` if the file is absent.
717
+ """
718
+ if flex_data is not None and getattr(flex_data, "realized_dispatch", None) is not None:
719
+ try:
720
+ rd = flex_data.realized_dispatch
721
+ seen: list[str] = []
722
+ seen_set: set[str] = set()
723
+ for p in rd["period"].cast(str).to_list():
724
+ if p not in seen_set:
725
+ seen.append(p)
726
+ seen_set.add(p)
727
+ return seen
728
+ except Exception: # noqa: BLE001
729
+ pass
730
+ if work_folder is None:
731
+ return None
732
+ path = Path(work_folder) / "solve_data" / "p_years_from_start_d.csv"
733
+ if not path.exists():
734
+ return None
735
+ df = pd.read_csv(path, usecols=["solve", "period"], dtype=str)
736
+ df = df[df["solve"] == str(solve_name)]
737
+ return list(df["period"].to_list())
738
+
739
+
740
+ def _row_index_names(*, has_period: bool, has_time: bool) -> list[str]:
741
+ if not has_period:
742
+ return ["solve"]
743
+ if has_time:
744
+ return ["solve", "period", "time"]
745
+ return ["solve", "period"]
746
+
747
+
748
+ def _empty_columns(col_names: Sequence[str]) -> pd.Index:
749
+ if len(col_names) >= 2:
750
+ return pd.MultiIndex.from_tuples([], names=list(col_names))
751
+ return pd.Index([], name=col_names[0])
752
+
753
+
754
+ def empty_variable_frame(
755
+ solve_name: str,
756
+ col_names: Sequence[str],
757
+ *,
758
+ has_period: bool = True,
759
+ has_time: bool = True,
760
+ realized_dt: "list[tuple[str, str]] | set[tuple[str, str]] | None" = None,
761
+ realized_p: "list[str] | set[str] | None" = None,
762
+ ) -> pd.DataFrame:
763
+ """Same-shape empty frame: full ``(solve, period[, time])`` row index, zero columns.
764
+
765
+ Built so downstream pandas ops (``DataFrame.mul(axis=1, level=0)``)
766
+ don't see a ``(0, 0)`` operand on one side and a populated row index
767
+ on the other. Construction is a single ``DataFrame`` call — no
768
+ Python loops over rows.
769
+
770
+ ``realized_dt`` / ``realized_p`` should be **ordered** (from
771
+ :func:`_load_realized_list` / :func:`_load_realized_periods_list`)
772
+ so the synthesised row order matches the canonical
773
+ ``for {(d, t) in dt_realize_dispatch}`` iteration order. A ``set``
774
+ is accepted (for back-compat) but yields arbitrary iteration order
775
+ — only safe when the caller doesn't need cross-reader row alignment.
776
+ """
777
+ row_index_names = _row_index_names(has_period=has_period, has_time=has_time)
778
+ empty_cols = _empty_columns(col_names)
779
+
780
+ if has_period and has_time and realized_dt is not None:
781
+ rows = pd.MultiIndex.from_tuples(
782
+ [(solve_name, d, t) for (d, t) in realized_dt],
783
+ names=row_index_names,
784
+ )
785
+ elif has_period and not has_time and realized_p is not None:
786
+ rows = pd.MultiIndex.from_tuples(
787
+ [(solve_name, d) for d in realized_p],
788
+ names=row_index_names,
789
+ )
790
+ elif not has_period:
791
+ rows = pd.Index([solve_name], name=row_index_names[0])
792
+ else:
793
+ if len(row_index_names) == 1:
794
+ rows = pd.Index([], name=row_index_names[0])
795
+ else:
796
+ rows = pd.MultiIndex.from_tuples([], names=row_index_names)
797
+
798
+ return pd.DataFrame(index=rows, columns=empty_cols, dtype=float)
799
+
800
+
801
+ def extract_variable(
802
+ h: "highspy.Highs",
803
+ name: str,
804
+ col_names: Sequence[str],
805
+ *,
806
+ solve_name: str,
807
+ has_time: bool = True,
808
+ has_period: bool = True,
809
+ source: str = "col_value",
810
+ value_scale: float = 1.0,
811
+ realized_dispatch_csv: Path | str | None = None,
812
+ realized_periods_csv: Path | str | None = None,
813
+ trailing_col_names: Sequence[str] = (),
814
+ flex_data: "FlexData | None" = None,
815
+ provider: "object | None" = None,
816
+ col_names_cache: Sequence[str] | None = None,
817
+ row_names_cache: Sequence[str] | None = None,
818
+ col_value: "object | None" = None,
819
+ col_dual: "object | None" = None,
820
+ row_dual: "object | None" = None,
821
+ period_source: str = "dispatch",
822
+ ) -> pd.DataFrame:
823
+ """Extract one quantity from a solved HiGHS instance as a wide DataFrame.
824
+
825
+ Parameters mirror :class:`VariableSpec` plus ``solve_name`` (tag
826
+ inserted into the row MultiIndex — typically the current solve name).
827
+ ``source`` selects which aligned array HiGHS exposes —
828
+ ``"col_value"``, ``"col_dual"`` (both keyed by
829
+ ``allVariableNames()``) or ``"row_dual"`` (keyed by
830
+ ``getLp().row_names_``). ``value_scale`` is applied once per raw
831
+ value (typically ``1e6`` for dual sources, to undo
832
+ ``scale_the_objective``).
833
+
834
+ ``realized_dispatch_csv`` filters (period, time) pairs for time-
835
+ indexed quantities; ``realized_periods_csv`` filters periods for
836
+ period-only quantities. ``has_period=False`` means the quantity
837
+ has no period index at all — row index collapses to just
838
+ ``(solve,)`` (used for ``co2_max_total[g]``).
839
+
840
+ Returns
841
+ -------
842
+ DataFrame
843
+ Wide layout — row index ``(solve,)`` / ``(solve, period)`` /
844
+ ``(solve, period, time)`` depending on the has_* flags. Column
845
+ MultiIndex when ``len(col_names) >= 2``, a single-level ``Index``
846
+ otherwise. Missing combinations are filled with 0.0.
847
+ """
848
+ # Cached arrays (hoisted out of the per-spec loop by
849
+ # ``write_all_variables``) are used when provided; otherwise fall
850
+ # back to fetching from the live HiGHS instance so the standalone /
851
+ # single-spec code path keeps working unchanged.
852
+ if source == "row_dual":
853
+ # Constraint names are stored on the LP struct, not exposed via
854
+ # a bulk getter on Highs itself — ``getLp().row_names_`` is the
855
+ # fast path (no per-row Python call).
856
+ names = row_names_cache if row_names_cache is not None else h.getLp().row_names_
857
+ values = row_dual if row_dual is not None else h.getSolution().row_dual
858
+ elif source == "col_dual":
859
+ names = col_names_cache if col_names_cache is not None else h.allVariableNames()
860
+ values = col_dual if col_dual is not None else h.getSolution().col_dual
861
+ elif source == "col_value":
862
+ names = col_names_cache if col_names_cache is not None else h.allVariableNames()
863
+ values = col_value if col_value is not None else h.getSolution().col_value
864
+ else:
865
+ raise ValueError(
866
+ f"Unknown source '{source}' — expected one of "
867
+ "'col_value', 'col_dual', 'row_dual'"
868
+ )
869
+ if len(names) != len(values):
870
+ raise RuntimeError(
871
+ f"HiGHS name / value length mismatch for '{name}' "
872
+ f"(source={source}): {len(names)} names vs {len(values)} values"
873
+ )
874
+
875
+ prefix = f"{name}["
876
+ pattern = _name_regex(name)
877
+ trailing = (2 if has_time else 1) if has_period else 0
878
+ n_trailing_cols = len(trailing_col_names)
879
+ expected_arity = len(col_names) + trailing + n_trailing_cols
880
+ row_index_names = _row_index_names(has_period=has_period, has_time=has_time)
881
+ # Full column-name tuple — leading cols (before period) + trailing
882
+ # cols (after period/time). Used for the column (Multi)Index and
883
+ # for the empty-frame shape.
884
+ full_col_names: tuple[str, ...] = tuple(col_names) + tuple(trailing_col_names)
885
+
886
+ # Canonical row order: read directly from a phase-1 printf CSV that
887
+ # iterates the same set the per-solve parameter CSVs do. Building
888
+ # the wide frame against this order from the get-go means no
889
+ # post-hoc sort/reindex — both readers produce the exact same row
890
+ # sequence because both ultimately come from the same phase-1
891
+ # ``for {s, (d, t) in dt_realize_dispatch}`` iteration.
892
+ work_folder = (
893
+ Path(realized_dispatch_csv).parent.parent
894
+ if realized_dispatch_csv is not None
895
+ else (
896
+ Path(realized_periods_csv).parent.parent
897
+ if realized_periods_csv is not None
898
+ else None
899
+ )
900
+ )
901
+ if has_period and has_time:
902
+ canonical_rows: list[tuple[str, ...]] | None = (
903
+ _load_canonical_dt_order(work_folder, solve_name, flex_data=flex_data)
904
+ )
905
+ # Fallback for callers that only provide ``realized_dispatch_csv``
906
+ # (e.g. unit tests with a synthetic CSV outside ``solve_data/``).
907
+ if canonical_rows is None and realized_dispatch_csv is not None:
908
+ canonical_rows = _load_realized_list(
909
+ realized_dispatch_csv, provider=provider,
910
+ )
911
+ elif has_period:
912
+ canonical_d = _load_canonical_d_order(
913
+ work_folder, solve_name, flex_data=flex_data,
914
+ )
915
+ if canonical_d is None and realized_periods_csv is not None:
916
+ canonical_d = _load_realized_periods_list(realized_periods_csv)
917
+ if period_source == "invest":
918
+ # Investment-axis variables (``v_invest`` / ``v_divest``): a
919
+ # roll that realizes DISPATCH but NO investment (the dispatch
920
+ # children of a nested invest cascade) must emit NO rows.
921
+ # Otherwise the densified column union (8e2de938 / a9058c66)
922
+ # materialises a zero-valued ``(roll, dispatch_period)`` row
923
+ # for every invest-eligible entity, and the ``drop_levels``
924
+ # keep='last' dedup lets that trailing dispatch-roll zero
925
+ # overwrite the invest step's real value for the shared period
926
+ # (zeroing the ``invested`` breakdown in ``unit_capacity__d``).
927
+ #
928
+ # The discriminator is the roll's realized-INVEST set: when it
929
+ # is empty this is a dispatch-only roll → emit nothing; when it
930
+ # is non-empty (the invest-committing step, or a rolling-invest
931
+ # solve) keep the FULL realized-dispatch period axis so periods
932
+ # whose investment is legitimately 0 (e.g. y2020 p2025, no NEW
933
+ # decision) still produce their zero row. When no realized-
934
+ # invest source is available at all (single-solve fixtures that
935
+ # never emit the set), fall back to the dispatch axis unchanged.
936
+ realized_invest = _load_realized_invest_periods_list(
937
+ realized_periods_csv, provider=provider,
938
+ )
939
+ if realized_invest is not None and not realized_invest:
940
+ canonical_d = []
941
+ canonical_rows = [(d,) for d in canonical_d] if canonical_d is not None else None
942
+ else:
943
+ canonical_rows = [()] # one row: just (solve,)
944
+
945
+ # Single pass over HiGHS: dict ``(d[, t], *col_vals) → value`` plus
946
+ # first-appearance unique col_vals tracking.
947
+ values_by_key: dict[tuple[str, ...], float] = {}
948
+ seen_cols_set: set[tuple[str, ...]] = set()
949
+ seen_cols: list[tuple[str, ...]] = []
950
+
951
+ for item_name, val in zip(names, values):
952
+ if not item_name.startswith(prefix):
953
+ continue
954
+ m = pattern.match(item_name)
955
+ if not m:
956
+ _logger.warning("Unrecognised %s name: %s", name, item_name)
957
+ continue
958
+ # GLPSOL/MPS quoting: any symbolic name containing a colon (e.g.
959
+ # ISO 8601 timestamps like ``2050-01-01T00:00:00``) is wrapped in
960
+ # single quotes when written to the .mps and HiGHS preserves
961
+ # those quotes verbatim in ``allVariableNames()``. The canonical
962
+ # row order (read from ``solve_data/p_step_duration.csv``) has
963
+ # bare timestamps, so without stripping here every time-indexed
964
+ # row_key would silently miss the canonical lookup at line ~773
965
+ # and the resulting parquet would be filled with zeros.
966
+ parts = [p.strip("'") for p in m.group(1).split(",")]
967
+ if len(parts) != expected_arity:
968
+ _logger.warning(
969
+ "Unexpected %s arity (%d, expected %d): %s",
970
+ name, len(parts), expected_arity, item_name,
971
+ )
972
+ continue
973
+ col_end = len(col_names)
974
+ leading_col_vals = tuple(parts[:col_end])
975
+ if has_period and has_time:
976
+ row_key: tuple[str, ...] = (parts[col_end], parts[col_end + 1])
977
+ row_len = 2
978
+ elif has_period:
979
+ row_key = (parts[col_end],)
980
+ row_len = 1
981
+ else:
982
+ row_key = ()
983
+ row_len = 0
984
+ trailing_start = col_end + row_len
985
+ trailing_col_vals = tuple(
986
+ parts[trailing_start:trailing_start + n_trailing_cols]
987
+ )
988
+ col_vals = leading_col_vals + trailing_col_vals
989
+ if col_vals not in seen_cols_set:
990
+ seen_cols.append(col_vals)
991
+ seen_cols_set.add(col_vals)
992
+ values_by_key[row_key + col_vals] = float(val) * value_scale
993
+
994
+ # No variable values matched — produce the same N_rows × 0-cols
995
+ # frame phase-3 CSV writers would have produced (with the canonical
996
+ # row index), so downstream ``DataFrame.mul(axis=1, level=0)``
997
+ # against a populated parameter frame doesn't get an empty operand.
998
+ if not seen_cols:
999
+ return empty_variable_frame(
1000
+ solve_name, full_col_names,
1001
+ has_period=has_period, has_time=has_time,
1002
+ realized_dt=canonical_rows if (has_period and has_time) else None,
1003
+ realized_p=(
1004
+ [r[0] for r in canonical_rows]
1005
+ if (has_period and not has_time and canonical_rows is not None)
1006
+ else None
1007
+ ),
1008
+ )
1009
+
1010
+ # Build the wide matrix by canonical row × first-appearance col
1011
+ # position lookup. Single dict iteration; the lookup itself is
1012
+ # O(1) per entry.
1013
+ n_cols_total = len(full_col_names)
1014
+ if canonical_rows is None:
1015
+ # Defensive: no canonical source available — fall back to
1016
+ # first-appearance row order from the HiGHS scan.
1017
+ seen_rows_set: set[tuple[str, ...]] = set()
1018
+ canonical_rows = []
1019
+ for k in values_by_key:
1020
+ row_key = k[: -n_cols_total] if n_cols_total else k
1021
+ if row_key not in seen_rows_set:
1022
+ canonical_rows.append(row_key)
1023
+ seen_rows_set.add(row_key)
1024
+
1025
+ import numpy as np
1026
+ row_pos = {r: i for i, r in enumerate(canonical_rows)}
1027
+ col_pos = {c: j for j, c in enumerate(seen_cols)}
1028
+ matrix = np.zeros((len(canonical_rows), len(seen_cols)), dtype=float)
1029
+ for key, val in values_by_key.items():
1030
+ if n_cols_total:
1031
+ row_key = key[: -n_cols_total]
1032
+ col_key = key[-n_cols_total:]
1033
+ else:
1034
+ row_key = key
1035
+ col_key = ()
1036
+ i = row_pos.get(row_key)
1037
+ if i is None:
1038
+ continue # row not in canonical (e.g. storage-reference timestep)
1039
+ matrix[i, col_pos[col_key]] = val
1040
+ # Normalise IEEE negative zeros to positive zero — HiGHS occasionally
1041
+ # returns ``-0.0`` for variables pinned at the lower bound, and pandas
1042
+ # ``assert_frame_equal`` distinguishes ``-0.0`` from ``0.0``.
1043
+ matrix += 0.0
1044
+
1045
+ if n_cols_total >= 2:
1046
+ col_idx: pd.Index = pd.MultiIndex.from_tuples(
1047
+ seen_cols, names=list(full_col_names),
1048
+ )
1049
+ else:
1050
+ col_idx = pd.Index(
1051
+ [c[0] for c in seen_cols], name=full_col_names[0],
1052
+ )
1053
+
1054
+ if not has_period:
1055
+ row_idx: pd.Index = pd.Index([solve_name], name=row_index_names[0])
1056
+ else:
1057
+ row_idx = pd.MultiIndex.from_tuples(
1058
+ [(solve_name, *r) for r in canonical_rows], names=row_index_names,
1059
+ )
1060
+ return pd.DataFrame(matrix, index=row_idx, columns=col_idx)
1061
+
1062
+
1063
+ def write_variable_parquet(
1064
+ h: "highspy.Highs",
1065
+ spec: VariableSpec,
1066
+ *,
1067
+ solve_name: str,
1068
+ output_dir: Path | str,
1069
+ realized_dispatch_csv: Path | str | None = None,
1070
+ realized_periods_csv: Path | str | None = None,
1071
+ file_name: str | None = None,
1072
+ flex_data: "FlexData | None" = None,
1073
+ scale_the_objective: float | None = None,
1074
+ provider: "object | None" = None,
1075
+ col_names_cache: Sequence[str] | None = None,
1076
+ row_names_cache: Sequence[str] | None = None,
1077
+ col_value: "object | None" = None,
1078
+ col_dual: "object | None" = None,
1079
+ row_dual: "object | None" = None,
1080
+ ) -> Path:
1081
+ """Extract the quantity described by *spec* and write a per-solve parquet.
1082
+
1083
+ File name defaults to ``{spec.output_name or spec.name}__{solve}.parquet``
1084
+ so parallel / rolling / nested solves don't collide. Merging
1085
+ per-solve files into a single ``{output_name}.parquet`` is a cheap
1086
+ post-processing step (one ``pd.concat``).
1087
+ """
1088
+ output_dir = Path(output_dir)
1089
+ output_dir.mkdir(parents=True, exist_ok=True)
1090
+ # Agent 12: when the spec uses ``_INV_SCALE_THE_OBJECTIVE`` as a
1091
+ # sentinel default (value_scale == 1e6, matching the legacy 1e-6
1092
+ # hardcoded objective scalar), substitute the live reciprocal of the
1093
+ # current solve's ``scale_the_objective``. Specs with an explicit
1094
+ # non-sentinel ``value_scale`` (e.g. ``-1e6`` for node balance duals)
1095
+ # still honour that — see :func:`write_v_dual_node_balance`.
1096
+ work_folder = (
1097
+ Path(realized_dispatch_csv).parent.parent
1098
+ if realized_dispatch_csv is not None
1099
+ else (
1100
+ Path(realized_periods_csv).parent.parent
1101
+ if realized_periods_csv is not None
1102
+ else None
1103
+ )
1104
+ )
1105
+ effective_scale = spec.value_scale
1106
+ if spec.value_scale == _INV_SCALE_THE_OBJECTIVE:
1107
+ effective_scale = _resolve_inv_scale_the_objective(
1108
+ work_folder, scale_the_objective=scale_the_objective,
1109
+ )
1110
+ # Multi-source fan-out: when the output quantity is the sum of two or
1111
+ # more HiGHS variables (e.g. two-tier slack), extract each source and
1112
+ # add them. Per-source frames share the same index shape, so
1113
+ # ``DataFrame.add(fill_value=0.0)`` is safe and correct: columns that
1114
+ # appear only in one source contribute that source's value plus 0.
1115
+ if spec.derived_from:
1116
+ df: pd.DataFrame | None = None
1117
+ for src_name in spec.derived_from:
1118
+ src_df = extract_variable(
1119
+ h, src_name, spec.col_names,
1120
+ solve_name=solve_name,
1121
+ has_time=spec.has_time,
1122
+ has_period=spec.has_period,
1123
+ source=spec.source,
1124
+ value_scale=effective_scale,
1125
+ realized_dispatch_csv=realized_dispatch_csv,
1126
+ realized_periods_csv=realized_periods_csv,
1127
+ trailing_col_names=spec.trailing_col_names,
1128
+ flex_data=flex_data,
1129
+ provider=provider,
1130
+ col_names_cache=col_names_cache,
1131
+ row_names_cache=row_names_cache,
1132
+ col_value=col_value,
1133
+ col_dual=col_dual,
1134
+ row_dual=row_dual,
1135
+ period_source=spec.period_source,
1136
+ )
1137
+ df = src_df if df is None else df.add(src_df, fill_value=0.0)
1138
+ assert df is not None # guaranteed: derived_from is non-empty
1139
+ else:
1140
+ df = extract_variable(
1141
+ h, spec.name, spec.col_names,
1142
+ solve_name=solve_name,
1143
+ has_time=spec.has_time,
1144
+ has_period=spec.has_period,
1145
+ source=spec.source,
1146
+ value_scale=effective_scale,
1147
+ realized_dispatch_csv=realized_dispatch_csv,
1148
+ realized_periods_csv=realized_periods_csv,
1149
+ trailing_col_names=spec.trailing_col_names,
1150
+ flex_data=flex_data,
1151
+ provider=provider,
1152
+ col_names_cache=col_names_cache,
1153
+ row_names_cache=row_names_cache,
1154
+ col_value=col_value,
1155
+ col_dual=col_dual,
1156
+ row_dual=row_dual,
1157
+ period_source=spec.period_source,
1158
+ )
1159
+ # Agent 1.8 — block-aware output expansion. Broadcast coarse-block
1160
+ # values to every covered fine timestep so parquet output stays
1161
+ # rectangular at the finest resolution. Degenerate case (every
1162
+ # entity on 'default'): no-op, bit-identical to pre-Agent-1.8.
1163
+ # Apply BEFORE unscale so the row scaler (keyed at the fine grid)
1164
+ # multiplies the broadcasted values consistently.
1165
+ if spec.expand_by is not None:
1166
+ df = _apply_block_expand(
1167
+ df, spec.expand_by, work_folder,
1168
+ flex_data=flex_data, provider=provider,
1169
+ )
1170
+ # Agent 9 — row-scaling un-scaling applied at the output boundary.
1171
+ # Source CSV comes from the same work folder as realized_*_csv
1172
+ # (``work_folder`` was inferred above for the scale_the_objective
1173
+ # resolution — reuse it).
1174
+ if spec.unscale_by is not None:
1175
+ df = _apply_unscale(df, spec.unscale_by, work_folder, solve_name, flex_data=flex_data)
1176
+ if file_name is None:
1177
+ file_name = f"{spec.output_name or spec.name}__{solve_name}.parquet"
1178
+ path = output_dir / file_name
1179
+ write_lean_parquet(df, path)
1180
+ _logger.debug(
1181
+ "Wrote %s for solve '%s' -> %s (shape %s)",
1182
+ spec.output_name or spec.name, solve_name, path, df.shape,
1183
+ )
1184
+ return path
1185
+
1186
+
1187
+ # ---------------------------------------------------------------------------
1188
+ # Custom writers for outputs that don't fit the plain VariableSpec pattern
1189
+ # (scalar, entity-class split, per-period transform).
1190
+ # ---------------------------------------------------------------------------
1191
+
1192
+
1193
+ def _load_entity_class(work_folder: Path, set_name: str) -> set[str]:
1194
+ """Load the members of an entity class directly from ``input/``.
1195
+
1196
+ Sourcing from ``input/`` keeps the Category C custom writers
1197
+ independent of phase 3 (which runs AFTER our writers inside
1198
+ ``_run_highs``) — important for the planned phase-3 retirement.
1199
+
1200
+ ``set_name`` maps to files like ``input/process_unit.csv``,
1201
+ ``input/process_connection.csv``, ``input/node.csv``. The input
1202
+ files are long-format with a single column named after the set.
1203
+ """
1204
+ path = work_folder / "input" / f"{set_name}.csv"
1205
+ if not path.exists():
1206
+ return set()
1207
+ df = pd.read_csv(path)
1208
+ if df.empty or len(df.columns) == 0:
1209
+ return set()
1210
+ return set(df.iloc[:, 0].astype(str).tolist())
1211
+
1212
+
1213
+ def _load_inflation_factor(
1214
+ work_folder: Path,
1215
+ flex_data: "FlexData | None" = None,
1216
+ ) -> dict[str, float]:
1217
+ """``{period: p_inflation_factor_operations_yearly}``.
1218
+
1219
+ Phase G — when ``flex_data`` is supplied, trust the in-memory carrier
1220
+ (``None`` ⇒ empty dict, matching the disk fallback's missing-file
1221
+ branch). CSV fallback retained for callers without FlexData.
1222
+
1223
+ Written by the model during phase 1 (derived parameter moved above
1224
+ ``solve;``).
1225
+ """
1226
+ if flex_data is not None:
1227
+ param = getattr(flex_data, "p_inflation_op", None)
1228
+ if param is None:
1229
+ return {}
1230
+ try:
1231
+ f = param.frame
1232
+ period_col = f.columns[0]
1233
+ return dict(zip(
1234
+ f[period_col].cast(str).to_list(),
1235
+ f["value"].cast(float).to_list(),
1236
+ ))
1237
+ except Exception: # noqa: BLE001
1238
+ pass
1239
+ path = work_folder / "solve_data" / "solve__p_inflation_factor_operations_yearly.csv"
1240
+ if not path.exists():
1241
+ return {}
1242
+ df = pd.read_csv(path)
1243
+ period_col = "period"
1244
+ value_col = [c for c in df.columns if c not in ("solve", "period")][0]
1245
+ return dict(zip(df[period_col].astype(str), df[value_col].astype(float)))
1246
+
1247
+
1248
+ def _load_complete_period_share_of_year(
1249
+ work_folder: Path,
1250
+ flex_data: "FlexData | None" = None,
1251
+ ) -> dict[str, float]:
1252
+ """``{period: complete_period_share_of_year}`` — Phase G prefers
1253
+ ``flex_data.p_period_share`` (trusted when supplied; ``None`` ⇒ {}).
1254
+ CSV fallback retained."""
1255
+ if flex_data is not None:
1256
+ param = getattr(flex_data, "p_period_share", None)
1257
+ if param is None:
1258
+ return {}
1259
+ try:
1260
+ f = param.frame
1261
+ period_col = f.columns[0]
1262
+ return dict(zip(
1263
+ f[period_col].cast(str).to_list(),
1264
+ f["value"].cast(float).to_list(),
1265
+ ))
1266
+ except Exception: # noqa: BLE001
1267
+ pass
1268
+ path = work_folder / "solve_data" / "complete_period_share_of_year.csv"
1269
+ if not path.exists():
1270
+ return {}
1271
+ df = pd.read_csv(path)
1272
+ period_col = "period"
1273
+ value_col = [c for c in df.columns if c not in ("solve", "period")][0]
1274
+ return dict(zip(df[period_col].astype(str), df[value_col].astype(float)))
1275
+
1276
+
1277
+ # ---------------------------------------------------------------------------
1278
+ # Row-scaler CSVs (Agent 9)
1279
+ # ---------------------------------------------------------------------------
1280
+
1281
+
1282
+ def _load_row_scaler(
1283
+ work_folder: Path | str | None,
1284
+ kind: str,
1285
+ solve_name: str,
1286
+ *,
1287
+ flex_data: "FlexData | None" = None,
1288
+ ) -> pd.DataFrame | None:
1289
+ """Read ``solve_data/solve__{node,group}_capacity_for_scaling.csv``.
1290
+
1291
+ Phase G — prefers ``flex_data.p_{node,group}_capacity_for_scaling``
1292
+ (Param, already in memory). CSV fallback retained.
1293
+
1294
+ Format (wide, produced by the AMPL phase-1 printf block at
1295
+ ``flextool.mod:4805``)::
1296
+
1297
+ solve,period,entity1,entity2,...
1298
+ <solve>,<period>,<scaler>,...
1299
+
1300
+ Returned frame has ``(solve, period)`` as row MultiIndex and the
1301
+ entity names as columns. Filtered to ``solve_name`` on read.
1302
+
1303
+ Returns ``None`` when the CSV is missing or empty — callers then
1304
+ treat the scaler as 1 everywhere (no-op).
1305
+
1306
+ ``kind`` is ``"node"`` or ``"group"``.
1307
+ """
1308
+ if flex_data is not None:
1309
+ attr = f"p_{kind}_capacity_for_scaling"
1310
+ param = getattr(flex_data, attr, None)
1311
+ if param is not None:
1312
+ try:
1313
+ # Param frame: columns (entity, "d", "value"). Pivot to
1314
+ # wide ``(solve, period)``-row × entity-column shape used
1315
+ # downstream. ``solve`` is the active solve_name.
1316
+ f = param.frame
1317
+ cols = f.columns
1318
+ # Drop value column from the pivot index set.
1319
+ if "value" in cols:
1320
+ entity_col = cols[0]
1321
+ period_col = cols[1]
1322
+ wide = f.pivot(
1323
+ on=entity_col, index=period_col, values="value",
1324
+ )
1325
+ pdf = wide.to_pandas()
1326
+ pdf = pdf.set_index(period_col)
1327
+ pdf.index = pd.MultiIndex.from_tuples(
1328
+ [(str(solve_name), str(p)) for p in pdf.index],
1329
+ names=["solve", "period"],
1330
+ )
1331
+ pdf = pdf.apply(pd.to_numeric, errors="coerce")
1332
+ return pdf
1333
+ except Exception: # noqa: BLE001
1334
+ pass
1335
+ if work_folder is None:
1336
+ return None
1337
+ path = Path(work_folder) / "solve_data" / f"solve__{kind}_capacity_for_scaling.csv"
1338
+ if not path.exists():
1339
+ return None
1340
+ try:
1341
+ df = pd.read_csv(path)
1342
+ except Exception:
1343
+ return None
1344
+ if df.empty or "solve" not in df.columns or "period" not in df.columns:
1345
+ return None
1346
+ df = df[df["solve"].astype(str) == str(solve_name)]
1347
+ if df.empty:
1348
+ return None
1349
+ df = df.set_index(["solve", "period"])
1350
+ # Column dtype: float. Empty-entity columns possible if the model
1351
+ # emitted headers but no values; ignore parsing failures.
1352
+ df = df.apply(pd.to_numeric, errors="coerce")
1353
+ return df
1354
+
1355
+
1356
+ # ---------------------------------------------------------------------------
1357
+ # Agent 1.8 — block-aware output expansion
1358
+ # ---------------------------------------------------------------------------
1359
+
1360
+
1361
+ def _load_entity_block_map(
1362
+ work_folder: Path | str | None, kind: str,
1363
+ *,
1364
+ provider: "object | None" = None,
1365
+ ) -> dict[str, str]:
1366
+ """Return ``{entity: block}`` read from ``solve_data/{entity,process}_block.csv``.
1367
+
1368
+ * ``kind="node_block"`` → reads ``entity_block.csv`` (columns
1369
+ ``entity, block``) — every node maps to its temporal-resolution
1370
+ block.
1371
+ * ``kind="process_block"`` → reads ``process_block.csv`` (columns
1372
+ ``process, block``) — per-process unified block (Agent 1.6).
1373
+
1374
+ When *provider* is supplied and carries the frame (cascade path —
1375
+ block CSVs are kept in memory rather than flushed to disk), the
1376
+ Provider lookup wins over the disk read. Missing both →
1377
+ empty / default fall-through (caller treats every entity as on the
1378
+ ``"default"`` block, i.e. identity overlap).
1379
+ """
1380
+ if work_folder is None and provider is None:
1381
+ return {}
1382
+ if kind == "node_block":
1383
+ path = (Path(work_folder) if work_folder is not None else Path()) / "solve_data" / "entity_block.csv"
1384
+ key_col = "entity"
1385
+ elif kind == "process_block":
1386
+ path = (Path(work_folder) if work_folder is not None else Path()) / "solve_data" / "process_block.csv"
1387
+ key_col = "process"
1388
+ else:
1389
+ return {}
1390
+ # Δ.31 — provider-first: the cascade keeps block frames in-memory
1391
+ # because ``emit_block_data_for_solve`` registers them on the
1392
+ # Provider rather than flushing to disk. Without this lookup the
1393
+ # output writer would broadcast nothing for daily-block fixtures
1394
+ # (lh2_three_region).
1395
+ pframe = _provider_lookup(provider, path)
1396
+ if pframe is not None and pframe.height > 0:
1397
+ cols = pframe.columns
1398
+ if key_col in cols and "block" in cols:
1399
+ return dict(zip(
1400
+ pframe[key_col].cast(pl.Utf8).to_list(),
1401
+ pframe["block"].cast(pl.Utf8).to_list(),
1402
+ ))
1403
+ if not path.exists():
1404
+ return {}
1405
+ try:
1406
+ df = pd.read_csv(path, dtype=str)
1407
+ except Exception:
1408
+ return {}
1409
+ if df.empty or key_col not in df.columns or "block" not in df.columns:
1410
+ return {}
1411
+ return dict(zip(df[key_col].astype(str), df["block"].astype(str)))
1412
+
1413
+
1414
+ def _load_overlap_fine_to_coarse(
1415
+ work_folder: Path | str | None,
1416
+ *,
1417
+ provider: "object | None" = None,
1418
+ ) -> dict[tuple[str, str, str], str]:
1419
+ """Return ``{(period, block_coarse, step_fine): step_coarse}``.
1420
+
1421
+ Read from ``solve_data/overlap_set.csv`` (columns ``period,
1422
+ block_coarse, step_coarse, block_fine, step_fine, fraction``).
1423
+ Only rows where ``block_fine == 'default'`` are kept — those are the
1424
+ rows used to broadcast a coarse-block value to every fine timestep
1425
+ it covers.
1426
+
1427
+ Provider-first: when *provider* carries the frame, prefer it over
1428
+ the disk read (cascade path keeps the block CSVs in memory).
1429
+ Missing both → empty dict (caller treats every entity as on the
1430
+ default block → identity broadcast).
1431
+ """
1432
+ if work_folder is None and provider is None:
1433
+ return {}
1434
+ path = (Path(work_folder) if work_folder is not None else Path()) / "solve_data" / "overlap_set.csv"
1435
+ pframe = _provider_lookup(provider, path)
1436
+ df: pd.DataFrame | None = None
1437
+ if pframe is not None and pframe.height > 0:
1438
+ df = pframe.to_pandas()
1439
+ elif path.exists():
1440
+ try:
1441
+ df = pd.read_csv(path, dtype=str)
1442
+ except Exception:
1443
+ return {}
1444
+ if df is None:
1445
+ return {}
1446
+ required = {"period", "block_coarse", "step_coarse", "block_fine", "step_fine"}
1447
+ if not required.issubset(df.columns):
1448
+ return {}
1449
+ df = df[df["block_fine"].astype(str) == "default"]
1450
+ if df.empty:
1451
+ return {}
1452
+ return {
1453
+ (str(p), str(bc), str(sf)): str(sc)
1454
+ for p, bc, sf, sc in zip(
1455
+ df["period"], df["block_coarse"],
1456
+ df["step_fine"], df["step_coarse"],
1457
+ )
1458
+ }
1459
+
1460
+
1461
+ def _apply_block_expand(
1462
+ df: pd.DataFrame,
1463
+ expand_by: str,
1464
+ work_folder: Path | str | None,
1465
+ *,
1466
+ flex_data: "FlexData | None" = None,
1467
+ provider: "object | None" = None,
1468
+ ) -> pd.DataFrame:
1469
+ """Broadcast coarse-block variable values to covered fine timesteps.
1470
+
1471
+ For each column ``e`` whose entity maps to a non-default block ``b``,
1472
+ every fine row ``(d, tf)`` has its value replaced with the coarse
1473
+ value at ``(d, tc)`` where ``tc = overlap[(d, b, tf)]``. Entities on
1474
+ the default block are left untouched (identity broadcast).
1475
+
1476
+ The DataFrame's row index must be ``(solve, period, time)``; the
1477
+ column (Multi)Index's first level is the entity name.
1478
+
1479
+ Degenerate case (every entity on ``'default'``): no columns trigger
1480
+ the broadcast → returns *df* unchanged, bit-identical to pre-Agent-
1481
+ 1.8 state.
1482
+ """
1483
+ if df.empty or df.shape[1] == 0:
1484
+ return df
1485
+ if expand_by not in ("process_block", "node_block"):
1486
+ return df
1487
+ if not isinstance(df.index, pd.MultiIndex):
1488
+ return df
1489
+ level_names = df.index.names or []
1490
+ if "period" not in level_names or "time" not in level_names:
1491
+ return df
1492
+
1493
+ entity_block = _load_entity_block_map(
1494
+ work_folder, expand_by, provider=provider,
1495
+ )
1496
+ # Fast path: no entity on a non-default block → nothing to do.
1497
+ if not any(v != "default" for v in entity_block.values()):
1498
+ return df
1499
+
1500
+ overlap = _load_overlap_fine_to_coarse(work_folder, provider=provider)
1501
+ if not overlap:
1502
+ return df
1503
+
1504
+ # Entity level on the column index (row 0 of MultiIndex; the only
1505
+ # level otherwise). Agent 1.8's expand_by is always keyed on
1506
+ # ``col_names[0]`` by construction.
1507
+ if isinstance(df.columns, pd.MultiIndex):
1508
+ entity_names = df.columns.get_level_values(0).astype(str).tolist()
1509
+ else:
1510
+ entity_names = df.columns.astype(str).tolist()
1511
+
1512
+ # Working in-place would mutate the caller's frame — copy once.
1513
+ out = df.copy()
1514
+
1515
+ # Row tuples (period, time) in the frame's order — we'll vector-assign
1516
+ # into each expand-target column via numpy positional writes.
1517
+ periods = df.index.get_level_values("period").astype(str).to_numpy()
1518
+ times = df.index.get_level_values("time").astype(str).to_numpy()
1519
+ row_count = len(periods)
1520
+
1521
+ # Column axis for block-aware expansion: entity per column position.
1522
+ for col_pos, entity in enumerate(entity_names):
1523
+ block = entity_block.get(entity, "default")
1524
+ if block == "default":
1525
+ continue # identity broadcast → leave column alone
1526
+ # Build a per-row source index — position of the coarse row to copy
1527
+ # from. For each (d, tf), find tc via overlap; then locate the row
1528
+ # in the frame. Rows whose (d, tf) has no overlap entry keep their
1529
+ # own value (defensive — shouldn't happen with consistent data).
1530
+ source_pos = list(range(row_count))
1531
+ # Build a (period, time) → row_pos map once.
1532
+ row_pos_map: dict[tuple[str, str], int] = {
1533
+ (p, t): i for i, (p, t) in enumerate(zip(periods, times))
1534
+ }
1535
+ for i in range(row_count):
1536
+ d = periods[i]
1537
+ tf = times[i]
1538
+ tc = overlap.get((d, block, tf))
1539
+ if tc is None:
1540
+ continue
1541
+ src = row_pos_map.get((d, tc))
1542
+ if src is None:
1543
+ continue
1544
+ source_pos[i] = src
1545
+ # Apply the column-scoped rewrite: new values = existing values
1546
+ # indexed by source_pos. ``out.iloc[:, col_pos]`` returns a view;
1547
+ # reassign so pandas records the update without triggering a
1548
+ # SettingWithCopy warning.
1549
+ col_values = out.iloc[:, col_pos].to_numpy()
1550
+ out.iloc[:, col_pos] = col_values[source_pos]
1551
+ return out
1552
+
1553
+
1554
+ def _apply_unscale(
1555
+ df: pd.DataFrame,
1556
+ unscale_by: str,
1557
+ work_folder: Path | str | None,
1558
+ solve_name: str,
1559
+ *,
1560
+ flex_data: "FlexData | None" = None,
1561
+ ) -> pd.DataFrame:
1562
+ """Multiply *df* by the row scaler identified by *unscale_by*.
1563
+
1564
+ Handles two scaler kinds:
1565
+
1566
+ * ``"node_cap"`` — ``solve_data/solve__node_capacity_for_scaling.csv`` keyed
1567
+ by (period, node). ``df`` row index is ``(solve, period[, time])``
1568
+ and columns are node names; we broadcast the period row of the
1569
+ scaler across the time dimension.
1570
+ * ``"group_cap"`` — ``solve_data/solve__group_capacity_for_scaling.csv``
1571
+ keyed by (period, group). Columns = group names. Same
1572
+ period-broadcast for time-indexed frames; for the no-t case (only
1573
+ ``vq_capacity_margin`` today) the row index is just ``(solve,
1574
+ period)`` and the element-wise multiply aligns directly.
1575
+
1576
+ Missing CSV / unknown columns / Mode A (scaler = 1) all collapse to
1577
+ a safe no-op: rows without a matching scaler are unchanged.
1578
+ """
1579
+ if df.empty or df.shape[1] == 0:
1580
+ return df
1581
+ kind = {"node_cap": "node", "group_cap": "group"}.get(unscale_by)
1582
+ if kind is None:
1583
+ return df
1584
+ scaler = _load_row_scaler(work_folder, kind, solve_name, flex_data=flex_data)
1585
+ if scaler is None or scaler.empty:
1586
+ return df
1587
+
1588
+ # Re-key the scaler by period alone (drop the solve level — we already
1589
+ # filtered on solve_name). Columns = entity.
1590
+ scaler_by_period = scaler.droplevel("solve") if "solve" in (scaler.index.names or []) else scaler
1591
+
1592
+ # Build an aligned multiplier with the same shape as ``df``.
1593
+ # 1) Keep only columns of ``scaler_by_period`` that appear in ``df``.
1594
+ # Data column name is the entity — for MultiIndex column frames
1595
+ # the first level holds the entity name used in the CSV header.
1596
+ entity_level = 0 # col_names[0] by construction for unscaled slacks
1597
+ if isinstance(df.columns, pd.MultiIndex):
1598
+ entity_names = df.columns.get_level_values(entity_level).astype(str).tolist()
1599
+ else:
1600
+ entity_names = df.columns.astype(str).tolist()
1601
+ present = [e for e in entity_names if e in scaler_by_period.columns]
1602
+ if not present:
1603
+ return df
1604
+
1605
+ # 2) For each row of ``df``, fetch the scaler row for that period.
1606
+ # Rows whose period has no scaler row get multiplier = 1
1607
+ # (no-op). Time-indexed df: broadcast the same period row across
1608
+ # all timesteps of that period.
1609
+ if isinstance(df.index, pd.MultiIndex) and "period" in (df.index.names or []):
1610
+ periods = df.index.get_level_values("period").astype(str)
1611
+ else:
1612
+ # Shouldn't happen for un-scaled slacks (all have period at least),
1613
+ # but guard for the no-period edge.
1614
+ return df
1615
+
1616
+ # Construct the multiplier frame: same rows as df, columns = entity_names
1617
+ # order. Use reindex on scaler_by_period to align columns → NaN for
1618
+ # missing entities → fill with 1 so they remain unchanged.
1619
+ mult = scaler_by_period.reindex(columns=entity_names).astype(float)
1620
+ mult = mult.reindex(periods.astype(str)).fillna(1.0)
1621
+ mult.index = df.index
1622
+ # Preserve the df column index (could be MultiIndex); numpy-level
1623
+ # multiply keeps the dtype and index.
1624
+ mult.columns = df.columns
1625
+
1626
+ return df * mult
1627
+
1628
+
1629
+ def write_v_obj(
1630
+ h: "highspy.Highs",
1631
+ *,
1632
+ solve_name: str,
1633
+ output_dir: Path | str,
1634
+ work_folder: Path | str | None = None,
1635
+ flex_data: "FlexData | None" = None,
1636
+ scale_the_objective: float | None = None,
1637
+ ) -> Path:
1638
+ """Write ``v_obj__{solve}.parquet`` — objective value for this solve.
1639
+
1640
+ Model writes ``total_cost.val / scale_the_objective``. HiGHS's
1641
+ ``getObjectiveValue()`` returns the raw (scaled) value; we undo the
1642
+ scaling. Agent 12: ``scale_the_objective`` is now per-solve
1643
+ (``solve_data/scale_the_objective.csv``); pass ``work_folder`` so
1644
+ the live value is read, else fall back to the legacy ``1e-6`` scalar.
1645
+ """
1646
+ output_dir = Path(output_dir)
1647
+ output_dir.mkdir(parents=True, exist_ok=True)
1648
+ wf = Path(work_folder) if work_folder is not None else output_dir.parent
1649
+ inv_scale = _resolve_inv_scale_the_objective(
1650
+ wf, scale_the_objective=scale_the_objective,
1651
+ )
1652
+ # Prefer the autoscale-stashed objective when present: Layer 2's
1653
+ # ``_push_unscaled_to_highs`` calls ``h.setSolution`` to mirror the
1654
+ # unscaled primal back onto the live solver handle, but that call
1655
+ # zeroes HiGHS's cached ``getObjectiveValue()`` (verified against
1656
+ # highspy 1.14.0). ``_flextool_unscaled_objective`` is the obj
1657
+ # captured immediately before that ``setSolution`` and is the
1658
+ # post-Layer-2-substitution / post-user_bound_scale-unscale value.
1659
+ raw_obj = getattr(h, "_flextool_unscaled_objective", None)
1660
+ if raw_obj is None:
1661
+ raw_obj = float(h.getObjectiveValue())
1662
+ obj = float(raw_obj) * inv_scale
1663
+ df = pd.DataFrame(
1664
+ {"objective": [obj]},
1665
+ index=pd.Index([solve_name], name="solve"),
1666
+ )
1667
+ path = output_dir / f"v_obj__{solve_name}.parquet"
1668
+ write_lean_parquet(df, path)
1669
+ # Emit the canonical ``total_cost.val`` stdout line so that callers
1670
+ # parsing the FlexTool stdout (e.g. test_representative_periods)
1671
+ # still see the objective value.
1672
+ print(f"total_cost.val = {obj:.12g}")
1673
+ _logger.debug("Wrote v_obj for solve '%s' -> %s (%.10g)", solve_name, path, obj)
1674
+ return path
1675
+
1676
+
1677
+ def write_v_dual_invest_by_class(
1678
+ h: "highspy.Highs",
1679
+ *,
1680
+ solve_name: str,
1681
+ output_dir: Path | str,
1682
+ realized_periods_csv: Path | str | None = None,
1683
+ work_folder: Path | str | None = None,
1684
+ flex_data: "FlexData | None" = None,
1685
+ scale_the_objective: float | None = None,
1686
+ ) -> list[Path]:
1687
+ """Write v_invest reduced costs split by entity class.
1688
+
1689
+ Produces three parquet files — ``v_dual_invest_unit__{solve}``,
1690
+ ``v_dual_invest_connection__{solve}``, ``v_dual_invest_node__{solve}``
1691
+ — each containing the ``v_invest.dual`` values for entities in the
1692
+ corresponding class (``process_unit``, ``process_connection``,
1693
+ ``node``). Matches the three separate CSVs phase 3 writes.
1694
+
1695
+ HiGHS returns the column reduced cost in *scaled-objective* units
1696
+ (the objective is multiplied by ``scale_the_objective`` at build
1697
+ time, default 1e-6). Multiply by ``1 / scale_the_objective`` so the
1698
+ written reduced cost is in true NPV currency per v_invest unit —
1699
+ consistent with every row-dual writer (cf.
1700
+ :func:`write_v_dual_node_balance`).
1701
+ """
1702
+ output_dir = Path(output_dir)
1703
+ output_dir.mkdir(parents=True, exist_ok=True)
1704
+ wf = Path(work_folder) if work_folder is not None else output_dir.parent
1705
+
1706
+ # ``_resolve_inv_scale_the_objective`` returns ``1 / scale_the_objective``
1707
+ # (guarding None / non-positive → default 1e-6), exactly as the
1708
+ # node-balance writer uses it.
1709
+ inv_scale = _resolve_inv_scale_the_objective(
1710
+ wf, scale_the_objective=scale_the_objective,
1711
+ )
1712
+ duals = extract_variable(
1713
+ h, "v_invest", ("entity",),
1714
+ solve_name=solve_name, has_time=False, source="col_dual",
1715
+ value_scale=inv_scale,
1716
+ realized_periods_csv=realized_periods_csv,
1717
+ flex_data=flex_data,
1718
+ )
1719
+
1720
+ classes = {
1721
+ "process_unit": _load_entity_class(wf, "process_unit"),
1722
+ "process_connection": _load_entity_class(wf, "process_connection"),
1723
+ "node": _load_entity_class(wf, "node"),
1724
+ }
1725
+ output_suffixes = {
1726
+ "process_unit": "unit",
1727
+ "process_connection": "connection",
1728
+ "node": "node",
1729
+ }
1730
+
1731
+ paths: list[Path] = []
1732
+ for cls_name, members in classes.items():
1733
+ keep = [c for c in duals.columns if c in members]
1734
+ subset = duals[keep].copy() if keep else duals.iloc[:, :0]
1735
+ # Preserve the single-level column index name for round-trip
1736
+ subset.columns.name = "entity"
1737
+ fname = f"v_dual_invest_{output_suffixes[cls_name]}__{solve_name}.parquet"
1738
+ path = output_dir / fname
1739
+ write_lean_parquet(subset, path)
1740
+ _logger.debug(
1741
+ "Wrote v_dual_invest_%s for solve '%s' -> %s (shape %s)",
1742
+ output_suffixes[cls_name], solve_name, path, subset.shape,
1743
+ )
1744
+ paths.append(path)
1745
+ return paths
1746
+
1747
+
1748
+ def _divide_by_inflation_and_row_scaler(
1749
+ df: "pd.DataFrame",
1750
+ *,
1751
+ wf: "Path",
1752
+ solve_name: str,
1753
+ flex_data: "FlexData | None",
1754
+ ) -> "pd.DataFrame":
1755
+ """Un-discount and un-row-scale a nodal-dual frame in place-equivalent.
1756
+
1757
+ Divides each ``(solve, period, time)`` row by its period's
1758
+ ``inflation_factor_operations_yearly`` (recover nominal currency) and
1759
+ each ``(period, node)`` cell by ``node_capacity_for_scaling`` (Mode A:
1760
+ scaler ≡ 1 → no effect). Shared verbatim by the per-(d,t)
1761
+ ``nodeBalance_eq`` path and the per-block ``nodeBalanceBlock_eq`` path
1762
+ so both report user-facing currency/MWh under the same convention.
1763
+ """
1764
+ if df.empty:
1765
+ return df
1766
+ inflation = _load_inflation_factor(wf, flex_data=flex_data)
1767
+ if inflation:
1768
+ # Divide each row by its period's inflation factor. Rows whose
1769
+ # period isn't in the dict keep a unity divisor.
1770
+ periods = df.index.get_level_values("period")
1771
+ divisors = pd.Series(
1772
+ [inflation.get(str(p), 1.0) for p in periods],
1773
+ index=df.index,
1774
+ dtype=float,
1775
+ )
1776
+ df = df.div(divisors, axis=0)
1777
+ scaler = _load_row_scaler(wf, "node", solve_name, flex_data=flex_data)
1778
+ if scaler is not None and not scaler.empty:
1779
+ scaler_by_period = (
1780
+ scaler.droplevel("solve")
1781
+ if "solve" in (scaler.index.names or [])
1782
+ else scaler
1783
+ )
1784
+ # Align columns to df's node names; missing → 1.
1785
+ node_names = df.columns.astype(str).tolist()
1786
+ mult = scaler_by_period.reindex(columns=node_names).astype(float)
1787
+ periods = df.index.get_level_values("period").astype(str)
1788
+ mult = mult.reindex(periods.astype(str)).fillna(1.0)
1789
+ mult.index = df.index
1790
+ mult.columns = df.columns
1791
+ df = df.div(mult)
1792
+ return df
1793
+
1794
+
1795
+ def _broadcast_block_node_duals(
1796
+ h: "highspy.Highs",
1797
+ *,
1798
+ solve_name: str,
1799
+ inv_scale: float,
1800
+ realized_dispatch_csv: "Path | str | None",
1801
+ flex_data: "FlexData | None",
1802
+ wf: "Path",
1803
+ ) -> "pd.DataFrame | None":
1804
+ """Per-(d,t) nodal prices for ``nodeStateBlock`` nodes.
1805
+
1806
+ Nodes balanced on a coarse block (``new_stepduration`` variable-
1807
+ resolution feature) carry ``nodeBalanceBlock_eq[n, d, b_first]`` — one
1808
+ row per (node, period, block), summed over the block's timesteps — NOT
1809
+ the per-(d,t) ``nodeBalance_eq``. Its dual is the marginal value of
1810
+ one MWh of energy anywhere in the block (the block RHS is total block
1811
+ energy in MWh, so the dual is already currency/MWh, the same unit as
1812
+ the per-step ``nodeBalance_eq`` dual — no block-size or step-duration
1813
+ factor). We therefore extract it under the identical sign/scaling
1814
+ convention as ``nodeBalance_eq`` and **broadcast** each block's price
1815
+ to every ``(d, t)`` in the block via ``period_block_time``, giving a
1816
+ per-(d,t) price that is constant within each block.
1817
+
1818
+ Returns a ``(solve, period, time)`` × block-node DataFrame, or
1819
+ ``None`` when the run has no block-balanced nodes.
1820
+ """
1821
+ block_set = getattr(flex_data, "nodeStateBlock", None) if flex_data else None
1822
+ pbt = getattr(flex_data, "period_block_time", None) if flex_data else None
1823
+ if (block_set is None or block_set.height == 0
1824
+ or pbt is None or pbt.height == 0):
1825
+ return None
1826
+
1827
+ blk = extract_variable(
1828
+ h, "nodeBalanceBlock_eq", ("node",),
1829
+ solve_name=solve_name, has_time=True, source="row_dual",
1830
+ value_scale=-inv_scale,
1831
+ realized_dispatch_csv=realized_dispatch_csv,
1832
+ flex_data=flex_data,
1833
+ )
1834
+ if blk.empty or blk.shape[1] == 0:
1835
+ return None
1836
+
1837
+ # Same un-discount / un-row-scale convention as nodeBalance_eq. The
1838
+ # raw extract places each block's dual on its ``b_first`` row and 0 on
1839
+ # the other (canonical) rows; both inflation and the row-scaler are
1840
+ # per-period (constant across a block's timesteps), so applying them
1841
+ # here — before the broadcast — is equivalent to applying them after.
1842
+ blk = _divide_by_inflation_and_row_scaler(
1843
+ blk, wf=wf, solve_name=solve_name, flex_data=flex_data,
1844
+ )
1845
+
1846
+ # Broadcast the b_first price across every (d, t) in the block.
1847
+ # ``period_block_time`` is (d, b_first, t); the block dual lives on the
1848
+ # (period, b_first) row, so join on (period == d, time == b_first) and
1849
+ # re-key to (period, t).
1850
+ pbt_pd = pbt.select(["d", "b_first", "t"]).to_pandas()
1851
+ blk_long = (
1852
+ blk.stack(future_stack=True)
1853
+ .rename("value")
1854
+ .reset_index()
1855
+ ) # columns: solve, period, time, node, value
1856
+ merged = blk_long.merge(
1857
+ pbt_pd, left_on=["period", "time"], right_on=["d", "b_first"],
1858
+ how="inner",
1859
+ )
1860
+ if merged.empty:
1861
+ return None
1862
+ wide = merged.pivot_table(
1863
+ index=["solve", "period", "t"], columns="node", values="value",
1864
+ aggfunc="first",
1865
+ )
1866
+ wide.index = wide.index.set_names(["solve", "period", "time"])
1867
+ wide.columns.name = "node"
1868
+ return wide
1869
+
1870
+
1871
+ def write_v_dual_node_balance(
1872
+ h: "highspy.Highs",
1873
+ *,
1874
+ solve_name: str,
1875
+ output_dir: Path | str,
1876
+ realized_dispatch_csv: Path | str | None = None,
1877
+ work_folder: Path | str | None = None,
1878
+ flex_data: "FlexData | None" = None,
1879
+ scale_the_objective: float | None = None,
1880
+ ) -> Path:
1881
+ """Write ``v_dual_node_balance__{solve}.parquet``.
1882
+
1883
+ Model formula (for n not in nodeStateBlock)::
1884
+
1885
+ -nodeBalance_eq[n, d, t].dual
1886
+ / p_inflation_factor_operations_yearly[d]
1887
+ / scale_the_objective
1888
+
1889
+ Equivalently, raw-dual × (−1e6 / inflation[d]). Nodes in
1890
+ ``nodeStateBlock`` (coarse ``new_stepduration`` variable-resolution
1891
+ nodes) are balanced by ``nodeBalanceBlock_eq`` (one row per block,
1892
+ summed over ``period_block_time``) instead. Their per-block dual —
1893
+ in the same currency/MWh units and under the same sign/scale
1894
+ convention as ``nodeBalance_eq`` — is extracted and **broadcast** to
1895
+ every ``(d, t)`` in the block (constant within a block); see
1896
+ :func:`_broadcast_block_node_duals`. Without this those nodes would
1897
+ have no dual column and the ``node_prices_dt_e`` output would raise a
1898
+ KeyError indexing them.
1899
+
1900
+ The polars LP emits ``nodeBalance_eq`` with arity-3
1901
+ ``(node, period, time)``; this writer reads that arity directly.
1902
+ """
1903
+ output_dir = Path(output_dir)
1904
+ output_dir.mkdir(parents=True, exist_ok=True)
1905
+ wf = Path(work_folder) if work_folder is not None else output_dir.parent
1906
+
1907
+ # Raw duals × -(1 / scale_the_objective); per-period inflation
1908
+ # division applied after. Agent 12: resolve the live scalar so a
1909
+ # non-default scale_the_objective propagates.
1910
+ inv_scale = _resolve_inv_scale_the_objective(
1911
+ wf, scale_the_objective=scale_the_objective,
1912
+ )
1913
+ df = extract_variable(
1914
+ h, "nodeBalance_eq", ("node",),
1915
+ solve_name=solve_name, has_time=True, source="row_dual",
1916
+ value_scale=-inv_scale,
1917
+ realized_dispatch_csv=realized_dispatch_csv,
1918
+ flex_data=flex_data,
1919
+ )
1920
+
1921
+ # Per-(d,t) ``nodeBalance_eq`` duals (storage + per-step balance nodes),
1922
+ # un-discounted and un-row-scaled to user-facing currency/MWh.
1923
+ df = _divide_by_inflation_and_row_scaler(
1924
+ df, wf=wf, solve_name=solve_name, flex_data=flex_data,
1925
+ )
1926
+
1927
+ # Per-block ``nodeBalanceBlock_eq`` duals for ``nodeStateBlock`` nodes
1928
+ # (coarse variable-resolution nodes), broadcast to per-(d,t) and merged
1929
+ # in. Without this these nodes have NO dual column and the downstream
1930
+ # ``node_prices_dt_e`` output (out_node) raises a KeyError indexing
1931
+ # them. None when the run has no block-balanced nodes → existing
1932
+ # (non-block) outputs are byte-identical.
1933
+ block_prices = _broadcast_block_node_duals(
1934
+ h, solve_name=solve_name, inv_scale=inv_scale,
1935
+ realized_dispatch_csv=realized_dispatch_csv,
1936
+ flex_data=flex_data, wf=wf,
1937
+ )
1938
+ if block_prices is not None and not block_prices.empty:
1939
+ if df.empty:
1940
+ df = block_prices
1941
+ else:
1942
+ block_prices = block_prices.reindex(df.index)
1943
+ df = pd.concat([df, block_prices], axis=1)
1944
+ # Deterministic node-column order once both families are present.
1945
+ df = df.reindex(sorted(df.columns.astype(str)), axis=1)
1946
+ df.columns.name = "node"
1947
+
1948
+ path = output_dir / f"v_dual_node_balance__{solve_name}.parquet"
1949
+ write_lean_parquet(df, path)
1950
+ _logger.debug(
1951
+ "Wrote v_dual_node_balance for solve '%s' -> %s (shape %s)",
1952
+ solve_name, path, df.shape,
1953
+ )
1954
+ return path
1955
+
1956
+
1957
+ def write_v_dual_reserve_balance(
1958
+ h: "highspy.Highs",
1959
+ *,
1960
+ solve_name: str,
1961
+ output_dir: Path | str,
1962
+ realized_dispatch_csv: Path | str | None = None,
1963
+ work_folder: Path | str | None = None,
1964
+ flex_data: "FlexData | None" = None,
1965
+ ) -> Path:
1966
+ """Write ``v_dual_reserve__upDown__group__period__t__{solve}.parquet``.
1967
+
1968
+ Model formula takes ``max()`` over up to three constraint duals per
1969
+ ``(r, ud, g, r_m, d, t)``:
1970
+
1971
+ * ``reserveBalance_timeseries_eq`` (most common method)
1972
+ * ``reserveBalance_dynamic_eq``
1973
+ * ``reserveBalance_up_n_1_eq`` / ``reserveBalance_down_n_1_eq``
1974
+ (N-1 security)
1975
+
1976
+ applied after period scaling
1977
+ ``× complete_period_share_of_year[d] / p_inflation_factor_operations_yearly[d]``.
1978
+
1979
+ SIMPLIFIED HERE: currently extracts only the timeseries-equation
1980
+ duals. For ``(r, ud, g)`` that use the ``timeseries_only`` method
1981
+ (the common case covered by all the test scenarios) this is exact.
1982
+ Groups using ``dynamic`` or ``n_1`` methods will be under-reported
1983
+ until the full ``max()`` logic is ported.
1984
+ """
1985
+ output_dir = Path(output_dir)
1986
+ output_dir.mkdir(parents=True, exist_ok=True)
1987
+ wf = Path(work_folder) if work_folder is not None else output_dir.parent
1988
+ out_path = output_dir / (
1989
+ f"v_dual_reserve__upDown__group__period__t__{solve_name}.parquet"
1990
+ )
1991
+
1992
+ # reserveBalance_timeseries_eq indices: r, ud, g, d, t (5). The
1993
+ # method is implicit in the constraint name prefix — there is no
1994
+ # ``r_m`` column inside the brackets. Sibling constraints
1995
+ # ``reserveBalance_dynamic_eq`` / ``reserveBalance_*_n_1_eq`` carry
1996
+ # the same (r, ud, g, d, t) axes; the ``max() over methods`` combine
1997
+ # documented above is a future extension.
1998
+ df = extract_variable(
1999
+ h, "reserveBalance_timeseries_eq",
2000
+ ("reserve", "updown", "node_group"),
2001
+ solve_name=solve_name, has_time=True, source="row_dual",
2002
+ realized_dispatch_csv=realized_dispatch_csv,
2003
+ flex_data=flex_data,
2004
+ )
2005
+
2006
+ if df.empty:
2007
+ write_lean_parquet(df, out_path)
2008
+ _logger.debug(
2009
+ "Wrote v_dual_reserve_balance for solve '%s' -> %s (empty)",
2010
+ solve_name, out_path,
2011
+ )
2012
+ return out_path
2013
+
2014
+ # Apply × period_share / inflation per period.
2015
+ inflation = _load_inflation_factor(wf, flex_data=flex_data)
2016
+ period_share = _load_complete_period_share_of_year(wf, flex_data=flex_data)
2017
+ periods = df.index.get_level_values("period")
2018
+ factor = pd.Series(
2019
+ [
2020
+ period_share.get(str(p), 1.0) / inflation.get(str(p), 1.0)
2021
+ for p in periods
2022
+ ],
2023
+ index=df.index, dtype=float,
2024
+ )
2025
+ df = df.mul(factor, axis=0)
2026
+
2027
+ write_lean_parquet(df, out_path)
2028
+ _logger.debug(
2029
+ "Wrote v_dual_reserve_balance for solve '%s' -> %s (shape %s)",
2030
+ solve_name, out_path, df.shape,
2031
+ )
2032
+ return out_path
2033
+
2034
+
2035
+ def _is_first_solve_from_p_model(work_folder: Path) -> bool:
2036
+ """True iff ``solve_data/p_model.csv`` says ``solveFirst`` is 1 (or missing)."""
2037
+ path = work_folder / "solve_data" / "p_model.csv"
2038
+ if not path.exists():
2039
+ return True
2040
+ df = pd.read_csv(path)
2041
+ matches = df.loc[df["modelParam"] == "solveFirst", "p_model"]
2042
+ if matches.empty:
2043
+ return True
2044
+ return bool(int(matches.iloc[0]))
2045
+
2046
+
2047
+ def _actual_solve_name(work_folder: Path, fallback: str,
2048
+ *, provider: "object | None" = None) -> str:
2049
+ """Return the roll-level solve name from ``solve_data/solve_current.csv``.
2050
+
2051
+ In rolling-window scenarios, ``solver.run(complete_solve[solve])`` is
2052
+ invoked with the PARENT solve name (the ``complete_solve``) while
2053
+ every phase-1 CSV in ``solve_data/`` stores the child ROLL name
2054
+ under its ``solve`` column (the set ``solve_current`` in the model).
2055
+ When filtering those CSVs or emitting CSV rows whose format the model
2056
+ produced, use the ROLL name — otherwise Python output won't line up
2057
+ with phase 3's.
2058
+ """
2059
+ path = work_folder / "solve_data" / "solve_current.csv"
2060
+ # Step 1-e — Provider-aware: under the in-memory cascade the file
2061
+ # isn't on disk, but the per-sub-solve Provider has the frame. The
2062
+ # transitional seed-funnel fallback in :func:`_provider_lookup`
2063
+ # keeps unplumbed callers working during the dual-write window.
2064
+ seeded = _provider_lookup(provider, path)
2065
+ if seeded is not None:
2066
+ if seeded.height == 0 or len(seeded.columns) == 0:
2067
+ return fallback
2068
+ return str(seeded[0, 0])
2069
+ if not path.exists():
2070
+ return fallback
2071
+ df = pd.read_csv(path)
2072
+ if df.empty or len(df.columns) == 0:
2073
+ return fallback
2074
+ return str(df.iloc[0, 0])
2075
+
2076
+
2077
+
2078
+
2079
+ def write_all_variables(
2080
+ h: "highspy.Highs",
2081
+ *,
2082
+ solve_name: str,
2083
+ output_dir: Path | str,
2084
+ realized_dispatch_csv: Path | str | None = None,
2085
+ realized_periods_csv: Path | str | None = None,
2086
+ specs: Sequence[VariableSpec] | None = None,
2087
+ flex_data: "FlexData | None" = None,
2088
+ scale_the_objective: float | None = None,
2089
+ provider: "object | None" = None,
2090
+ ) -> list[Path]:
2091
+ """Iterate :data:`VARIABLE_SPECS` (or a custom list) and write parquets.
2092
+
2093
+ Returns the list of paths written, one per variable. Each variable
2094
+ is independent — failure on one is logged and does not abort the
2095
+ remaining ones, so one bad variable can't lose the whole run.
2096
+ """
2097
+ specs = specs if specs is not None else VARIABLE_SPECS
2098
+ written: list[Path] = []
2099
+ # Derive the work folder from the realized-dispatch CSV (preferred)
2100
+ # so custom writers (esp. write_v_obj) can find
2101
+ # ``solve_data/scale_the_objective.csv`` at the live path.
2102
+ _derived_wf: Path | None = (
2103
+ Path(realized_dispatch_csv).parent.parent
2104
+ if realized_dispatch_csv is not None
2105
+ else (
2106
+ Path(realized_periods_csv).parent.parent
2107
+ if realized_periods_csv is not None
2108
+ else None
2109
+ )
2110
+ )
2111
+ # Hoist the expensive HiGHS bulk fetches out of the per-spec loop.
2112
+ # extract_variable() otherwise re-materialises ``allVariableNames()``
2113
+ # (a fresh multi-million-element Python list), ``getSolution()`` (full
2114
+ # col_value/col_dual/row_dual array copies) and ``getLp().row_names_``
2115
+ # (a full LP copy incl. all row names) ONCE PER SPEC (×len(specs)).
2116
+ # Fetching once here and threading the cached arrays through
2117
+ # write_variable_parquet -> extract_variable collapses that to a
2118
+ # single fetch per solve. Output content is unchanged — the same
2119
+ # name/value arrays are indexed, just shared. Set
2120
+ # ``FLEXTOOL_DISABLE_OUTPUT_HOIST=1`` to keep the old per-spec fetch
2121
+ # (A/B comparison / quick revert).
2122
+ col_names_cache: Sequence[str] | None = None
2123
+ row_names_cache: Sequence[str] | None = None
2124
+ col_value = col_dual = row_dual = None
2125
+ if os.environ.get("FLEXTOOL_DISABLE_OUTPUT_HOIST") != "1":
2126
+ try:
2127
+ col_names_cache = h.allVariableNames()
2128
+ _sol = h.getSolution()
2129
+ col_value = _sol.col_value
2130
+ col_dual = _sol.col_dual
2131
+ row_dual = _sol.row_dual
2132
+ row_names_cache = h.getLp().row_names_
2133
+ except Exception as exc: # pragma: no cover - defensive fallback
2134
+ _logger.warning(
2135
+ "output-hoist pre-fetch failed (solve '%s'): %s — "
2136
+ "falling back to per-spec fetch", solve_name, exc,
2137
+ )
2138
+ col_names_cache = row_names_cache = None
2139
+ col_value = col_dual = row_dual = None
2140
+
2141
+ for spec in specs:
2142
+ try:
2143
+ path = write_variable_parquet(
2144
+ h, spec,
2145
+ solve_name=solve_name,
2146
+ output_dir=output_dir,
2147
+ realized_dispatch_csv=realized_dispatch_csv,
2148
+ realized_periods_csv=realized_periods_csv,
2149
+ flex_data=flex_data,
2150
+ scale_the_objective=scale_the_objective,
2151
+ provider=provider,
2152
+ col_names_cache=col_names_cache,
2153
+ row_names_cache=row_names_cache,
2154
+ col_value=col_value,
2155
+ col_dual=col_dual,
2156
+ row_dual=row_dual,
2157
+ )
2158
+ written.append(path)
2159
+ except Exception as exc:
2160
+ _logger.warning(
2161
+ "parquet extraction failed for %s (solve '%s'): %s",
2162
+ spec.output_name or spec.name, solve_name, exc,
2163
+ )
2164
+
2165
+ # Custom writers for outputs that don't fit a plain VariableSpec.
2166
+ _custom_writers = (
2167
+ ("v_obj", lambda: write_v_obj(
2168
+ h, solve_name=solve_name, output_dir=output_dir,
2169
+ work_folder=_derived_wf,
2170
+ flex_data=flex_data,
2171
+ scale_the_objective=scale_the_objective,
2172
+ )),
2173
+ ("v_dual_invest_{unit,connection,node}", lambda: write_v_dual_invest_by_class(
2174
+ h, solve_name=solve_name, output_dir=output_dir,
2175
+ realized_periods_csv=realized_periods_csv,
2176
+ flex_data=flex_data,
2177
+ scale_the_objective=scale_the_objective,
2178
+ )),
2179
+ ("v_dual_node_balance", lambda: write_v_dual_node_balance(
2180
+ h, solve_name=solve_name, output_dir=output_dir,
2181
+ realized_dispatch_csv=realized_dispatch_csv,
2182
+ flex_data=flex_data,
2183
+ scale_the_objective=scale_the_objective,
2184
+ )),
2185
+ ("v_dual_reserve_balance", lambda: write_v_dual_reserve_balance(
2186
+ h, solve_name=solve_name, output_dir=output_dir,
2187
+ realized_dispatch_csv=realized_dispatch_csv,
2188
+ flex_data=flex_data,
2189
+ )),
2190
+ # ``entity_all_capacity`` moved to ``handoff_writers.py`` — it's
2191
+ # conceptually a solve-to-solve handoff (accumulates across
2192
+ # solves) and needs the same CSV layout phase 3 produces, not a
2193
+ # per-solve parquet.
2194
+ )
2195
+ for label, fn in _custom_writers:
2196
+ try:
2197
+ result = fn()
2198
+ if isinstance(result, list):
2199
+ written.extend(result)
2200
+ else:
2201
+ written.append(result)
2202
+ except Exception as exc:
2203
+ _logger.warning(
2204
+ "custom writer failed for %s (solve '%s'): %s",
2205
+ label, solve_name, exc,
2206
+ )
2207
+ _logger.info(
2208
+ "Wrote %d output variables for solve '%s' -> %s",
2209
+ len(written), solve_name, output_dir,
2210
+ )
2211
+ return written
2212
+
2213
+
2214
+ # ---------------------------------------------------------------------------
2215
+ # Standalone / offline test entry point
2216
+ # ---------------------------------------------------------------------------
2217
+
2218
+
2219
+ def _standalone_from_files(
2220
+ mps_file: Path, sol_file: Path, solve_name: str, out_dir: Path
2221
+ ) -> list[Path]:
2222
+ """Re-read a HiGHS model + solution offline and write all registered vars."""
2223
+ import highspy # local import — the runtime path imports lazily
2224
+
2225
+ h = highspy.Highs()
2226
+ status = h.readModel(str(mps_file))
2227
+ if status != highspy.HighsStatus.kOk:
2228
+ raise RuntimeError(f"HiGHS failed to read MPS: {mps_file}")
2229
+ status = h.readSolution(str(sol_file), 0)
2230
+ if status != highspy.HighsStatus.kOk:
2231
+ raise RuntimeError(
2232
+ f"HiGHS failed to read solution file: {sol_file} "
2233
+ f"(status={status}). Try re-solving instead."
2234
+ )
2235
+ return write_all_variables(h, solve_name=solve_name, output_dir=out_dir)
2236
+
2237
+
2238
+ def main() -> None:
2239
+ parser = argparse.ArgumentParser(description=__doc__.splitlines()[1])
2240
+ parser.add_argument("--mps", type=Path, required=True, help="MPS file")
2241
+ parser.add_argument(
2242
+ "--sol", type=Path, required=True,
2243
+ help="HiGHS solution file (write_solution_style=0)",
2244
+ )
2245
+ parser.add_argument("--solve", required=True, help="Solve name tag")
2246
+ parser.add_argument("--out", type=Path, required=True, help="Output dir")
2247
+ args = parser.parse_args()
2248
+
2249
+ logging.basicConfig(level=logging.INFO)
2250
+ paths = _standalone_from_files(args.mps, args.sol, args.solve, args.out)
2251
+ for p in paths:
2252
+ print(f"Wrote {p}")
2253
+
2254
+
2255
+ if __name__ == "__main__":
2256
+ main()