flextool 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (322) hide show
  1. flextool/__init__.py +41 -0
  2. flextool/_mem_sampler.py +193 -0
  3. flextool/_resources.py +43 -0
  4. flextool/calibrate/__init__.py +51 -0
  5. flextool/calibrate/__main__.py +11 -0
  6. flextool/calibrate/_cli.py +316 -0
  7. flextool/calibrate/_db_alt.py +166 -0
  8. flextool/calibrate/_final_outputs.py +110 -0
  9. flextool/calibrate/_guard.py +151 -0
  10. flextool/calibrate/_loop.py +558 -0
  11. flextool/calibrate/_readers.py +223 -0
  12. flextool/calibrate/_report.py +263 -0
  13. flextool/calibrate/_sizing.py +699 -0
  14. flextool/calibrate/_solve.py +134 -0
  15. flextool/calibrate/_solve_status.py +495 -0
  16. flextool/cli/__init__.py +9 -0
  17. flextool/cli/_console.py +51 -0
  18. flextool/cli/_timing.py +147 -0
  19. flextool/cli/cmd_execute_flextool_workflow.py +187 -0
  20. flextool/cli/cmd_export_to_tabular.py +56 -0
  21. flextool/cli/cmd_import_sensitivities.py +75 -0
  22. flextool/cli/cmd_migrate_database.py +13 -0
  23. flextool/cli/cmd_open_results_db.py +269 -0
  24. flextool/cli/cmd_read_matpower.py +66 -0
  25. flextool/cli/cmd_read_old_flextool.py +63 -0
  26. flextool/cli/cmd_read_self_describing_tabular_input.py +50 -0
  27. flextool/cli/cmd_read_tabular_input.py +81 -0
  28. flextool/cli/cmd_run_flextool.py +1095 -0
  29. flextool/cli/cmd_scenario_results.py +284 -0
  30. flextool/cli/cmd_solve_mps.py +169 -0
  31. flextool/cli/cmd_update_flextool.py +17 -0
  32. flextool/cli/cmd_write_outputs.py +125 -0
  33. flextool/common_utils/__init__.py +1 -0
  34. flextool/common_utils/plot_mem_shape.py +77 -0
  35. flextool/common_utils/precision.py +451 -0
  36. flextool/decomposition/__init__.py +0 -0
  37. flextool/decomposition/region_decomposition.py +128 -0
  38. flextool/decomposition/region_filter.py +1261 -0
  39. flextool/engine_polars/__init__.py +110 -0
  40. flextool/engine_polars/_axis_enums.py +742 -0
  41. flextool/engine_polars/_benders.py +3462 -0
  42. flextool/engine_polars/_block_layout.py +1479 -0
  43. flextool/engine_polars/_blocks.py +1515 -0
  44. flextool/engine_polars/_commodity_ladder.py +660 -0
  45. flextool/engine_polars/_cumulative_invest.py +1165 -0
  46. flextool/engine_polars/_db_loader.py +153 -0
  47. flextool/engine_polars/_db_reader.py +127 -0
  48. flextool/engine_polars/_dc_power_flow.py +445 -0
  49. flextool/engine_polars/_delay.py +442 -0
  50. flextool/engine_polars/_derived_arithmetic.py +432 -0
  51. flextool/engine_polars/_derived_block.py +990 -0
  52. flextool/engine_polars/_derived_branch.py +769 -0
  53. flextool/engine_polars/_derived_existing.py +1353 -0
  54. flextool/engine_polars/_derived_npv.py +1297 -0
  55. flextool/engine_polars/_derived_params.py +9850 -0
  56. flextool/engine_polars/_derived_profile.py +881 -0
  57. flextool/engine_polars/_derived_walks.py +276 -0
  58. flextool/engine_polars/_determinism.py +70 -0
  59. flextool/engine_polars/_direct_params.py +2186 -0
  60. flextool/engine_polars/_dump_csvs.py +1009 -0
  61. flextool/engine_polars/_emit_arc_unions.py +1631 -0
  62. flextool/engine_polars/_emit_calc_params.py +729 -0
  63. flextool/engine_polars/_emit_chain_params.py +709 -0
  64. flextool/engine_polars/_emit_co2_accumulators.py +400 -0
  65. flextool/engine_polars/_emit_dispatchers.py +690 -0
  66. flextool/engine_polars/_emit_energy_margin.py +125 -0
  67. flextool/engine_polars/_emit_energy_margin_adder.py +290 -0
  68. flextool/engine_polars/_emit_entity_annual.py +428 -0
  69. flextool/engine_polars/_emit_inflow_scaling.py +1420 -0
  70. flextool/engine_polars/_emit_leaf_sets.py +550 -0
  71. flextool/engine_polars/_emit_lp_scaling.py +665 -0
  72. flextool/engine_polars/_emit_mid_sets.py +859 -0
  73. flextool/engine_polars/_emit_pdt_params.py +759 -0
  74. flextool/engine_polars/_emit_per_solve.py +774 -0
  75. flextool/engine_polars/_emit_period_calc.py +504 -0
  76. flextool/engine_polars/_emit_period_params.py +2398 -0
  77. flextool/engine_polars/_emit_provider_io.py +141 -0
  78. flextool/engine_polars/_emit_reserve.py +574 -0
  79. flextool/engine_polars/_emit_solve_time.py +311 -0
  80. flextool/engine_polars/_emit_solve_writers.py +1249 -0
  81. flextool/engine_polars/_flex_data_accumulator.py +388 -0
  82. flextool/engine_polars/_flex_data_provider.py +478 -0
  83. flextool/engine_polars/_group_slack.py +1253 -0
  84. flextool/engine_polars/_inmemory_reader.py +140 -0
  85. flextool/engine_polars/_input_source.py +336 -0
  86. flextool/engine_polars/_invest_seeds.py +191 -0
  87. flextool/engine_polars/_native_input_writer.py +100 -0
  88. flextool/engine_polars/_native_run_model.py +1348 -0
  89. flextool/engine_polars/_orchestration.py +4314 -0
  90. flextool/engine_polars/_output_writer.py +439 -0
  91. flextool/engine_polars/_param_shapes.py +1595 -0
  92. flextool/engine_polars/_parquet_bundle.py +723 -0
  93. flextool/engine_polars/_pdt_join.py +167 -0
  94. flextool/engine_polars/_pdt_lookup.py +547 -0
  95. flextool/engine_polars/_per_solve_sets.py +335 -0
  96. flextool/engine_polars/_projection_params.py +2056 -0
  97. flextool/engine_polars/_provider_keys.py +173 -0
  98. flextool/engine_polars/_provider_translators.py +225 -0
  99. flextool/engine_polars/_recursive_solve.py +703 -0
  100. flextool/engine_polars/_region_filter.py +2508 -0
  101. flextool/engine_polars/_reserve.py +649 -0
  102. flextool/engine_polars/_solve_acceptance.py +331 -0
  103. flextool/engine_polars/_solve_config.py +1001 -0
  104. flextool/engine_polars/_solve_context.py +885 -0
  105. flextool/engine_polars/_solve_handoff.py +164 -0
  106. flextool/engine_polars/_solve_state.py +232 -0
  107. flextool/engine_polars/_solver_base.py +36 -0
  108. flextool/engine_polars/_solver_dispatch.py +511 -0
  109. flextool/engine_polars/_spinedb_reader.py +1165 -0
  110. flextool/engine_polars/_stochastic.py +593 -0
  111. flextool/engine_polars/_subprocess_solve.py +1838 -0
  112. flextool/engine_polars/_timeline.py +1416 -0
  113. flextool/engine_polars/_vectorize.py +438 -0
  114. flextool/engine_polars/_warm.py +858 -0
  115. flextool/engine_polars/autoscale/__init__.py +107 -0
  116. flextool/engine_polars/autoscale/_config.py +218 -0
  117. flextool/engine_polars/autoscale/_layer2.py +1253 -0
  118. flextool/engine_polars/autoscale/_layer2_types.py +584 -0
  119. flextool/engine_polars/autoscale/_quantity_types.py +621 -0
  120. flextool/engine_polars/autoscale/_report.py +336 -0
  121. flextool/engine_polars/chain.py +259 -0
  122. flextool/engine_polars/input.py +6638 -0
  123. flextool/engine_polars/model.py +4754 -0
  124. flextool/env_check.py +388 -0
  125. flextool/export_to_tabular/__init__.py +5 -0
  126. flextool/export_to_tabular/db_reader.py +224 -0
  127. flextool/export_to_tabular/excel_writer.py +3559 -0
  128. flextool/export_to_tabular/export_settings.yaml +377 -0
  129. flextool/export_to_tabular/export_to_excel.py +227 -0
  130. flextool/export_to_tabular/formatting.py +543 -0
  131. flextool/export_to_tabular/sheet_config.py +876 -0
  132. flextool/gui/__init__.py +0 -0
  133. flextool/gui/__main__.py +118 -0
  134. flextool/gui/calibrate_commands.py +184 -0
  135. flextool/gui/calibrate_jobs.py +424 -0
  136. flextool/gui/check_tree.py +142 -0
  137. flextool/gui/cli_format.py +83 -0
  138. flextool/gui/config_parser.py +68 -0
  139. flextool/gui/data_models.py +362 -0
  140. flextool/gui/db_editor_integration.py +202 -0
  141. flextool/gui/db_version_check.py +269 -0
  142. flextool/gui/dialogs/__init__.py +0 -0
  143. flextool/gui/dialogs/add_dialog.py +1098 -0
  144. flextool/gui/dialogs/calibrate_dialog.py +1259 -0
  145. flextool/gui/dialogs/file_picker.py +473 -0
  146. flextool/gui/dialogs/group_picker.py +299 -0
  147. flextool/gui/dialogs/migration_consent_dialog.py +106 -0
  148. flextool/gui/dialogs/migration_progress_dialog.py +237 -0
  149. flextool/gui/dialogs/plot_dialog.py +459 -0
  150. flextool/gui/dialogs/plot_settings_picker.py +2184 -0
  151. flextool/gui/dialogs/project_dialog.py +426 -0
  152. flextool/gui/dialogs/update_dialog.py +212 -0
  153. flextool/gui/downsampling.py +88 -0
  154. flextool/gui/error_handling.py +50 -0
  155. flextool/gui/execution_manager.py +1715 -0
  156. flextool/gui/execution_window.py +1377 -0
  157. flextool/gui/hover_tooltip.py +111 -0
  158. flextool/gui/input_sources.py +730 -0
  159. flextool/gui/main_window.py +6181 -0
  160. flextool/gui/network_graph.py +215 -0
  161. flextool/gui/output_actions.py +393 -0
  162. flextool/gui/output_log_window.py +159 -0
  163. flextool/gui/platform_utils.py +421 -0
  164. flextool/gui/plot_cache.py +88 -0
  165. flextool/gui/plot_canvas.py +543 -0
  166. flextool/gui/plot_config_reader.py +272 -0
  167. flextool/gui/project_utils.py +100 -0
  168. flextool/gui/result_viewer.py +4394 -0
  169. flextool/gui/scenario_key.py +162 -0
  170. flextool/gui/scenario_lists.py +516 -0
  171. flextool/gui/settings_io.py +360 -0
  172. flextool/gui/solve_reader.py +103 -0
  173. flextool/gui/tree_reorder.py +88 -0
  174. flextool/gui/ui_metrics.py +420 -0
  175. flextool/input_derivation/__init__.py +281 -0
  176. flextool/input_derivation/_commodity_ladder.py +375 -0
  177. flextool/input_derivation/_commodity_ladder_sets.py +70 -0
  178. flextool/input_derivation/_dc_power_flow.py +377 -0
  179. flextool/input_derivation/_method_constants.py +77 -0
  180. flextool/input_derivation/_process_method.py +258 -0
  181. flextool/input_derivation/_specs.py +1026 -0
  182. flextool/input_derivation/_validators.py +321 -0
  183. flextool/lean_parquet.py +159 -0
  184. flextool/model_builder/__init__.py +5 -0
  185. flextool/model_builder/build_model.py +589 -0
  186. flextool/model_builder/encoding.py +67 -0
  187. flextool/model_builder/names.py +34 -0
  188. flextool/model_builder/profiles.py +129 -0
  189. flextool/plot_outputs/__init__.py +14 -0
  190. flextool/plot_outputs/axis_helpers.py +355 -0
  191. flextool/plot_outputs/color_template.py +888 -0
  192. flextool/plot_outputs/config.py +171 -0
  193. flextool/plot_outputs/format_helpers.py +345 -0
  194. flextool/plot_outputs/legend_helpers.py +143 -0
  195. flextool/plot_outputs/orchestrator.py +1141 -0
  196. flextool/plot_outputs/perf.py +37 -0
  197. flextool/plot_outputs/plan.py +1787 -0
  198. flextool/plot_outputs/plot_bars.py +1510 -0
  199. flextool/plot_outputs/plot_bars_detail.py +753 -0
  200. flextool/plot_outputs/plot_lines.py +951 -0
  201. flextool/plot_outputs/shared_manifest.py +564 -0
  202. flextool/plot_outputs/subplot_helpers.py +137 -0
  203. flextool/process_inputs/__init__.py +188 -0
  204. flextool/process_inputs/import_old_excel_input.json +4159 -0
  205. flextool/process_inputs/read_matpower.py +451 -0
  206. flextool/process_inputs/read_old_flextool.py +1288 -0
  207. flextool/process_inputs/read_self_describing_excel.py +1423 -0
  208. flextool/process_inputs/read_tabular_with_specification.py +1114 -0
  209. flextool/process_inputs/write_old_flextool_to_db.py +3077 -0
  210. flextool/process_inputs/write_self_describing_to_db.py +977 -0
  211. flextool/process_inputs/write_to_input_db.py +269 -0
  212. flextool/process_outputs/__init__.py +7 -0
  213. flextool/process_outputs/_annualize.py +55 -0
  214. flextool/process_outputs/_inmemory_helpers.py +292 -0
  215. flextool/process_outputs/_output_meta.py +672 -0
  216. flextool/process_outputs/calc_capacity_flows.py +107 -0
  217. flextool/process_outputs/calc_connections.py +136 -0
  218. flextool/process_outputs/calc_costs.py +260 -0
  219. flextool/process_outputs/calc_group_flows.py +192 -0
  220. flextool/process_outputs/calc_slacks.py +103 -0
  221. flextool/process_outputs/calc_storage_vre.py +160 -0
  222. flextool/process_outputs/drop_levels.py +208 -0
  223. flextool/process_outputs/handoff_writers.py +1315 -0
  224. flextool/process_outputs/out_ancillary.py +544 -0
  225. flextool/process_outputs/out_capacity.py +179 -0
  226. flextool/process_outputs/out_costs.py +334 -0
  227. flextool/process_outputs/out_flowgroup.py +189 -0
  228. flextool/process_outputs/out_flows.py +301 -0
  229. flextool/process_outputs/out_group.py +475 -0
  230. flextool/process_outputs/out_node.py +190 -0
  231. flextool/process_outputs/persist_realized_slice.py +601 -0
  232. flextool/process_outputs/process_results.py +24 -0
  233. flextool/process_outputs/read_highs_solution.py +2256 -0
  234. flextool/process_outputs/read_parameters.py +1799 -0
  235. flextool/process_outputs/read_sets.py +1095 -0
  236. flextool/process_outputs/read_variables.py +553 -0
  237. flextool/process_outputs/solve_order.py +81 -0
  238. flextool/process_outputs/spinedb_replay.py +412 -0
  239. flextool/process_outputs/union_realized_slice.py +224 -0
  240. flextool/process_outputs/write_outputs.py +1286 -0
  241. flextool/process_outputs/write_spinedb.py +1267 -0
  242. flextool/representative_periods/__init__.py +5 -0
  243. flextool/representative_periods/clustering.py +165 -0
  244. flextool/representative_periods/force_include.py +563 -0
  245. flextool/representative_periods/netload.py +365 -0
  246. flextool/representative_periods/netload_inputs.py +345 -0
  247. flextool/representative_periods/netload_iterate.py +722 -0
  248. flextool/representative_periods/preprocess.py +948 -0
  249. flextool/representative_periods/scenario_stack.py +195 -0
  250. flextool/representative_periods/weights.py +124 -0
  251. flextool/scenario_comparison/__init__.py +13 -0
  252. flextool/scenario_comparison/config_builder.py +158 -0
  253. flextool/scenario_comparison/constants.py +20 -0
  254. flextool/scenario_comparison/data_models.py +222 -0
  255. flextool/scenario_comparison/db_reader.py +399 -0
  256. flextool/scenario_comparison/dispatch_data.py +1002 -0
  257. flextool/scenario_comparison/dispatch_mappings.py +205 -0
  258. flextool/scenario_comparison/dispatch_plots.py +691 -0
  259. flextool/scenario_comparison/input_entity_colors.py +319 -0
  260. flextool/scenario_comparison/orchestrator.py +453 -0
  261. flextool/scenario_comparison/plan_union.py +244 -0
  262. flextool/scenario_comparison/plot_settings_seed.py +205 -0
  263. flextool/schemas/AXIS_CONTRACT.md +71 -0
  264. flextool/schemas/canonical_databases/howto_aggregate_output.json +6225 -0
  265. flextool/schemas/canonical_databases/howto_connections.json +5606 -0
  266. flextool/schemas/canonical_databases/howto_demand.json +5518 -0
  267. flextool/schemas/canonical_databases/howto_hydro_reservoir.json +6239 -0
  268. flextool/schemas/canonical_databases/howto_hydro_reservoir_with_pump.json +5933 -0
  269. flextool/schemas/canonical_databases/howto_non_sync_and_curtailment.json +5794 -0
  270. flextool/schemas/canonical_databases/howto_ramp_and_start_up.json +5707 -0
  271. flextool/schemas/canonical_databases/howto_stochastics.json +6032 -0
  272. flextool/schemas/canonical_databases/templates_examples.json +13532 -0
  273. flextool/schemas/canonical_databases/templates_time_settings_only.json +5340 -0
  274. flextool/schemas/comparison_settings_template.json +197 -0
  275. flextool/schemas/default_plot_settings.yaml +260 -0
  276. flextool/schemas/default_plots.yaml +2293 -0
  277. flextool/schemas/flextool_axis_contract.json +303 -0
  278. flextool/schemas/flextool_axis_contract.schema.json +247 -0
  279. flextool/schemas/old_flextool_import_template.json +4443 -0
  280. flextool/schemas/output_info_template.json +48 -0
  281. flextool/schemas/output_settings_template.json +256 -0
  282. flextool/schemas/pre_v26/flextool_template_constant_default.json +2105 -0
  283. flextool/schemas/pre_v26/flextool_template_default_optional_output.json +2152 -0
  284. flextool/schemas/pre_v26/flextool_template_default_value.json +2094 -0
  285. flextool/schemas/pre_v26/flextool_template_drop_down.json +2080 -0
  286. flextool/schemas/pre_v26/flextool_template_lifetime_method.json +1990 -0
  287. flextool/schemas/pre_v26/flextool_template_optional_outputs.json +2094 -0
  288. flextool/schemas/pre_v26/flextool_template_output_node_flows.json +2105 -0
  289. flextool/schemas/pre_v26/flextool_template_results_master.json +493 -0
  290. flextool/schemas/pre_v26/flextool_template_rolling_start_remove.json +2087 -0
  291. flextool/schemas/pre_v26/flextool_template_rolling_window.json +2059 -0
  292. flextool/schemas/pre_v26/flextool_template_storage_binding_defaults.json +46 -0
  293. flextool/schemas/pre_v26/flextool_template_v2.json +1990 -0
  294. flextool/schemas/pre_v26/flextool_template_v25.json +3864 -0
  295. flextool/schemas/spinedb_results_schema.json +581 -0
  296. flextool/schemas/spinedb_schema.json +4636 -0
  297. flextool/solver_config/copt.opt.template +18 -0
  298. flextool/solver_config/cplex.opt.template +25 -0
  299. flextool/solver_config/gurobi.opt.template +18 -0
  300. flextool/solver_config/highs.opt.template +18 -0
  301. flextool/solver_config/xpress.opt.template +26 -0
  302. flextool/spinedb_backend/__init__.py +26 -0
  303. flextool/spinedb_backend/_axis_enums.py +1119 -0
  304. flextool/spinedb_backend/_backend.py +1139 -0
  305. flextool/update_flextool/__init__.py +12 -0
  306. flextool/update_flextool/canonical_databases.py +251 -0
  307. flextool/update_flextool/db_migration.py +7108 -0
  308. flextool/update_flextool/ensure_settings_db.py +138 -0
  309. flextool/update_flextool/export_database.py +103 -0
  310. flextool/update_flextool/extend_tests_fixture.py +772 -0
  311. flextool/update_flextool/generate_canonical.py +274 -0
  312. flextool/update_flextool/initialize_database.py +42 -0
  313. flextool/update_flextool/install_info.py +225 -0
  314. flextool/update_flextool/self_update.py +464 -0
  315. flextool/update_flextool/sync_master_json_template.py +125 -0
  316. flextool/update_flextool/test_fixtures.py +187 -0
  317. flextool-4.0.0.dist-info/METADATA +217 -0
  318. flextool-4.0.0.dist-info/RECORD +322 -0
  319. flextool-4.0.0.dist-info/WHEEL +5 -0
  320. flextool-4.0.0.dist-info/entry_points.txt +17 -0
  321. flextool-4.0.0.dist-info/licenses/LICENSE.txt +19 -0
  322. flextool-4.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1631 @@
1
+ """Process-arc-union + period-param leaf writers.
2
+
3
+ Cheap, leaf-like writers whose inputs are already-native L0-L9
4
+ ``solve_data/*.csv`` outputs (or plain ``input/*.csv``) and whose
5
+ semantics are pure projection / join / filter with no ``PdtLookup``-class
6
+ machinery behind them.
7
+
8
+ Ported writers (legacy LOC budget ~535):
9
+
10
+ From ``process_arc_unions.py``:
11
+
12
+ * ``write_process_source_sink_param_t`` (~38 LOC)
13
+ * ``write_node_time_param_in_use`` (~44 LOC)
14
+ * ``write_process_source_delayed_partition`` (~18 LOC)
15
+ * ``write_process_source_sink_profile_method_connection`` (~35 LOC)
16
+ * ``write_process_method_sources_sinks`` (~56 LOC)
17
+ * ``write_ed_history_realized_first`` (~56 LOC)
18
+ * ``write_process_source_sink_ramp_method`` (~32 LOC)
19
+ * ``write_process_source_sink_coeff_zero`` (~24 LOC)
20
+ * ``write_process_source_sink_delayed_partition`` (~18 LOC)
21
+
22
+ From ``entity_period_calc_params.py``:
23
+
24
+ * ``write_pProcess_source_sink`` (~56 LOC)
25
+
26
+ The four ``write_pdtProcess`` / ``write_pdtNode`` / ``write_pdtProcess_source``
27
+ / ``write_pdtProcess_sink`` writers from ``entity_period_calc_params``
28
+ were *also* on the candidate list but lean on the ~200 LOC ``PdtLookup``
29
+ class hierarchy from ``preprocessing/pd_lookups.py``. Porting them
30
+ sensibly requires lifting that whole machinery — deferred to the next
31
+ dispatch.
32
+
33
+ Each ``write_*`` is a thin wrapper around a ``derive_*`` (or, where the
34
+ legacy emits multiple CSVs from one shared computation, around a small
35
+ ``_compute_*`` helper). The ``derive_*`` returns a fresh
36
+ ``pl.DataFrame`` in the same in-memory contract as
37
+ :mod:`._emit_leaf_sets` / :mod:`._emit_mid_sets` /
38
+ :mod:`._emit_calc_params`.
39
+
40
+ Style mirrors :mod:`._emit_calc_params` — eager polars reads of tiny
41
+ CSVs with positional column renames, expression chains where natural,
42
+ small python loops where the iteration order is precision-load-bearing
43
+ (matches the legacy ``dict.fromkeys`` ordered-dedup pattern).
44
+
45
+ Precision-parity pattern
46
+ ------------------------
47
+
48
+ ``write_pProcess_source_sink`` writes a value column. Legacy formats
49
+ it via ``f"{repr(v)}"`` with ``v`` already a python float — we mirror
50
+ exactly by pre-stringifying with ``repr(float(v))``. See
51
+ :mod:`._emit_calc_params` module docstring for the precision-parity
52
+ rationale (round-trip-exactness of ``repr(float)`` and divergence
53
+ from polars' default float formatting).
54
+ """
55
+ from __future__ import annotations
56
+
57
+ from pathlib import Path
58
+
59
+ import polars as pl
60
+
61
+
62
+ # ---------------------------------------------------------------------------
63
+ # CSV I/O — same conventions as the sibling _emit_*.py modules.
64
+ # ---------------------------------------------------------------------------
65
+
66
+ # Provider-aware open helper — re-exported from the shared module.
67
+ # Step 2.5 Phase B collapsed the local copy that carried a disk-fallback
68
+ # arm; cascade code uses the Provider-only shim.
69
+
70
+ from flextool.engine_polars._emit_provider_io import ( # noqa: E402
71
+ _emit,
72
+ _provider_key,
73
+ )
74
+
75
+
76
+ def _cell_str(value: "object | None") -> str:
77
+ """Reproduce a ``csv.reader`` cell string for a native frame value.
78
+
79
+ ``DataFrame.write_csv`` renders ``null`` as the empty string and every
80
+ other scalar as its textual form; ``csv.reader`` then reads those
81
+ strings back. Mirror that here so dict/set keys and string
82
+ comparisons stay byte-identical to the legacy CSV round-trip.
83
+ """
84
+ return "" if value is None else str(value)
85
+
86
+
87
+ def _read_csv(path: Path, columns: list[str],
88
+ *,
89
+ provider: "object | None" = None) -> pl.DataFrame:
90
+ """Provider-only — Step 2.5 Phase C dropped the disk-fallback arm.
91
+
92
+ Returns the Provider's frame sliced to *columns* with positional
93
+ rename; returns an empty all-Utf8 frame when the Provider misses
94
+ the key (matches legacy missing-CSV behaviour).
95
+ """
96
+ from flextool.engine_polars._emit_provider_io import (
97
+ _provider_lookup_positional,
98
+ )
99
+ seeded = _provider_lookup_positional(
100
+ provider, _provider_key(path), path, columns,
101
+ )
102
+ if seeded is not None:
103
+ return seeded
104
+ return pl.DataFrame(
105
+ {c: [] for c in columns}, schema={c: pl.Utf8 for c in columns},
106
+ )
107
+
108
+
109
+ def _drop_blank_rows(df: pl.DataFrame, required_cols: list[str]) -> pl.DataFrame:
110
+ expr = pl.col(required_cols[0]) != ""
111
+ for c in required_cols[1:]:
112
+ expr = expr & (pl.col(c) != "")
113
+ return df.filter(expr)
114
+
115
+
116
+ def _read_n_col_rows(path: Path, columns: list[str],
117
+ *,
118
+ provider: "object | None" = None) -> list[tuple[str, ...]]:
119
+ """Read a CSV as a list of tuples preserving CSV row order.
120
+
121
+ Mirrors the legacy ``_read_n_col`` helper — used where the legacy
122
+ code iterates a list (not a set) to drive deterministic dedupe via
123
+ ``dict.fromkeys``.
124
+ """
125
+ df = _read_csv(path, columns, provider=provider)
126
+ df = _drop_blank_rows(df, columns)
127
+ return list(df.iter_rows())
128
+
129
+
130
+ # ---------------------------------------------------------------------------
131
+ # Constants — mirror flextool_base.dat enums. Pinned here as frozensets
132
+ # so the native module has no transitive import from the legacy
133
+ # preprocessing tree. If those enums change, update both sites in
134
+ # lockstep; the parity tests will catch drift.
135
+ # ---------------------------------------------------------------------------
136
+
137
+ # NOTE on enum-iteration parity: legacy code stores these as frozensets
138
+ # and iterates them directly with ``for param in FOO``. Python's
139
+ # string-hash randomization makes the resulting iteration order
140
+ # session-dependent, BUT within a single pytest process both the legacy
141
+ # and native writers see the SAME randomized order — and the native
142
+ # writer's job is byte-identical parity within the same process, not
143
+ # stable order across runs. We therefore mirror the legacy storage
144
+ # (frozenset, with the exact same element tuple) to guarantee identical
145
+ # iteration order in any given session.
146
+
147
+ # flextool_base.dat:153 — PROCESS_TIME_PARAM
148
+ _PROCESS_TIME_PARAM: frozenset[str] = frozenset((
149
+ "efficiency", "efficiency_at_min_load", "min_load",
150
+ "other_operational_cost", "availability",
151
+ ))
152
+
153
+ # flextool_base.dat:178 — NODE_TIME_PARAM
154
+ _NODE_TIME_PARAM: frozenset[str] = frozenset((
155
+ "inflow", "penalty_down", "penalty_up", "self_discharge_loss",
156
+ "availability", "storage_state_reference_value",
157
+ ))
158
+
159
+ # flextool_base.dat:179 — NODE_TIME_PARAM_REQUIRED
160
+ _NODE_TIME_PARAM_REQUIRED: frozenset[str] = frozenset((
161
+ "inflow", "penalty_down", "penalty_up",
162
+ ))
163
+
164
+ # preprocessing/_method_constants.py L90 / L93 — speed/cost-gated subsets.
165
+ _RAMP_LIMIT_METHOD: frozenset[str] = frozenset(("ramp_limit", "both"))
166
+ _RAMP_COST_METHOD: frozenset[str] = frozenset(("ramp_cost", "both"))
167
+
168
+
169
+ # ===========================================================================
170
+ # process_arc_unions — leaf-like writers
171
+ # ===========================================================================
172
+
173
+
174
+ # ---- node__TimeParam_in_use (mod L1208-1214) ------------------------------
175
+
176
+ def derive_node_time_param_in_use(
177
+ input_dir: Path, solve_data_dir: Path,
178
+ *,
179
+ provider: "object | None" = None,
180
+ ) -> pl.DataFrame:
181
+ """node × nodeTimeParam filtered by per-node membership in
182
+ nodeBalance / nodeBalancePeriod / nodeState
183
+ or by ``(n, 'use_reference_value') in node__storage_solve_horizon_method``.
184
+ """
185
+ nodes = (
186
+ _drop_blank_rows(
187
+ _read_csv(input_dir / "node.csv", ["node"], provider=provider),
188
+ ["node"],
189
+ ).get_column("node").to_list()
190
+ )
191
+ n_balance = frozenset(
192
+ _drop_blank_rows(
193
+ _read_csv(solve_data_dir / "nodeBalance.csv", ["node"],
194
+ provider=provider),
195
+ ["node"],
196
+ ).get_column("node").to_list()
197
+ )
198
+ n_balance_period = frozenset(
199
+ _drop_blank_rows(
200
+ _read_csv(solve_data_dir / "nodeBalancePeriod.csv", ["node"],
201
+ provider=provider),
202
+ ["node"],
203
+ ).get_column("node").to_list()
204
+ )
205
+ n_state = frozenset(
206
+ _drop_blank_rows(
207
+ _read_csv(solve_data_dir / "nodeState.csv", ["node"],
208
+ provider=provider),
209
+ ["node"],
210
+ ).get_column("node").to_list()
211
+ )
212
+ storage_method = _drop_blank_rows(
213
+ _read_csv(
214
+ input_dir / "node__storage_solve_horizon_method.csv",
215
+ ["node", "method"],
216
+ provider=provider,
217
+ ),
218
+ ["node", "method"],
219
+ )
220
+ n_storage_use_ref = frozenset(
221
+ storage_method.filter(pl.col("method") == "use_reference_value")
222
+ .get_column("node").to_list()
223
+ )
224
+
225
+ rows: list[tuple[str, str]] = []
226
+ for n in nodes:
227
+ is_bal = n in n_balance
228
+ is_bal_period = n in n_balance_period
229
+ is_state = n in n_state
230
+ is_use_ref = n in n_storage_use_ref
231
+ for param in _NODE_TIME_PARAM:
232
+ if (is_bal or is_bal_period) and param in _NODE_TIME_PARAM_REQUIRED:
233
+ rows.append((n, param))
234
+ elif is_state and param in ("self_discharge_loss", "availability"):
235
+ rows.append((n, param))
236
+ elif is_use_ref and param == "storage_state_reference_value":
237
+ rows.append((n, param))
238
+ deduped = list(dict.fromkeys(rows))
239
+ return pl.DataFrame(
240
+ {"node": [r[0] for r in deduped], "param": [r[1] for r in deduped]},
241
+ schema={"node": pl.Utf8, "param": pl.Utf8},
242
+ )
243
+
244
+
245
+ def emit_node_time_param_in_use(input_dir: Path, solve_data_dir: Path,
246
+ *, provider) -> None:
247
+ """Emit ``node_time_param_in_use`` to the Provider."""
248
+ _emit(provider, "solve_data/node__TimeParam_in_use.csv",
249
+ derive_node_time_param_in_use(
250
+ input_dir, solve_data_dir, provider=provider,
251
+ ))
252
+
253
+
254
+ # ---- process_source_{delayed,undelayed} (mod L1092-1093) -------------------
255
+
256
+ def derive_process_source_delayed_partition(
257
+ input_dir: Path, solve_data_dir: Path,
258
+ *,
259
+ provider: "object | None" = None,
260
+ ) -> tuple[pl.DataFrame, pl.DataFrame]:
261
+ """Partition ``process__source`` by membership in ``process_delayed``.
262
+
263
+ Returns (delayed, undelayed) frames, each with columns (process, source).
264
+ """
265
+ pairs = _read_csv(
266
+ input_dir / "process__source.csv", ["process", "source"],
267
+ provider=provider,
268
+ )
269
+ pairs = _drop_blank_rows(pairs, ["process", "source"])
270
+ delayed_set = frozenset(
271
+ _drop_blank_rows(
272
+ _read_csv(
273
+ solve_data_dir / "process_delayed.csv", ["process"],
274
+ provider=provider,
275
+ ),
276
+ ["process"],
277
+ ).get_column("process").to_list()
278
+ )
279
+ delayed = pairs.filter(pl.col("process").is_in(list(delayed_set)))
280
+ undelayed = pairs.filter(~pl.col("process").is_in(list(delayed_set)))
281
+ return delayed, undelayed
282
+
283
+
284
+ def emit_process_source_delayed_partition(
285
+ input_dir: Path, solve_data_dir: Path,
286
+ *, provider,
287
+ ) -> None:
288
+ """Emit ``process_source_delayed_partition`` to the Provider."""
289
+ delayed, undelayed = derive_process_source_delayed_partition(
290
+ input_dir, solve_data_dir, provider=provider,
291
+ )
292
+ _emit(provider, "solve_data/process_source_delayed.csv", delayed)
293
+ _emit(provider, "solve_data/process_source_undelayed.csv", undelayed)
294
+
295
+
296
+ # ---- process__source__sink__profile__profile_method_connection
297
+ # (mod L1060-1063) -------------------------------------------------------
298
+
299
+ def derive_process_source_sink_profile_method_connection(
300
+ input_dir: Path, solve_data_dir: Path,
301
+ *,
302
+ provider: "object | None" = None,
303
+ ) -> pl.DataFrame:
304
+ """``process_source_sink × profile × profile_method`` filtered by
305
+ ``(p, profile, method) in process__profile__profile_method``.
306
+ """
307
+ triples = _read_n_col_rows(
308
+ solve_data_dir / "process_source_sink.csv",
309
+ ["process", "source", "sink"],
310
+ provider=provider,
311
+ )
312
+ pp_pm = _read_n_col_rows(
313
+ input_dir / "process__profile__profile_method.csv",
314
+ ["process", "profile", "profile_method"],
315
+ provider=provider,
316
+ )
317
+ fm_for_p: dict[str, list[tuple[str, str]]] = {}
318
+ for p, f, m in pp_pm:
319
+ fm_for_p.setdefault(p, []).append((f, m))
320
+
321
+ rows: list[tuple[str, str, str, str, str]] = []
322
+ for p, src, sink in triples:
323
+ for f, m in fm_for_p.get(p, ()):
324
+ rows.append((p, src, sink, f, m))
325
+ return pl.DataFrame(
326
+ {
327
+ "process": [r[0] for r in rows],
328
+ "source": [r[1] for r in rows],
329
+ "sink": [r[2] for r in rows],
330
+ "profile": [r[3] for r in rows],
331
+ "profile_method": [r[4] for r in rows],
332
+ },
333
+ schema={
334
+ "process": pl.Utf8, "source": pl.Utf8, "sink": pl.Utf8,
335
+ "profile": pl.Utf8, "profile_method": pl.Utf8,
336
+ },
337
+ )
338
+
339
+
340
+ def emit_process_source_sink_profile_method_connection(
341
+ input_dir: Path, solve_data_dir: Path,
342
+ *, provider,
343
+ ) -> None:
344
+ """Emit ``process_source_sink_profile_method_connection`` to the Provider."""
345
+ _emit(
346
+ provider,
347
+ "solve_data/process__source__sink__profile__profile_method_connection.csv",
348
+ derive_process_source_sink_profile_method_connection(
349
+ input_dir, solve_data_dir, provider=provider,
350
+ ),
351
+ )
352
+
353
+
354
+ # ---- ed_history_realized_first (mod L993) ---------------------------------
355
+
356
+ def derive_ed_history_realized_first(
357
+ input_dir: Path, solve_data_dir: Path,
358
+ *,
359
+ provider: "object | None" = None,
360
+ ) -> pl.DataFrame:
361
+ """entity × realized periods, but only on the first solve.
362
+
363
+ Honours the ``solveFirst`` flag on ``p_model``: non-first solves
364
+ emit an empty frame.
365
+ """
366
+ # solveFirst gate: short-circuit to empty.
367
+ solve_first = False
368
+ pm = _read_csv(
369
+ solve_data_dir / "p_model.csv", ["key", "value"], provider=provider,
370
+ )
371
+ if pm.height > 0:
372
+ row = pm.filter(pl.col("key") == "solveFirst")
373
+ if row.height > 0:
374
+ try:
375
+ solve_first = bool(int(row.get_column("value")[0]))
376
+ except (ValueError, TypeError):
377
+ solve_first = False
378
+ if not solve_first:
379
+ return pl.DataFrame(
380
+ {"entity": [], "period": []},
381
+ schema={"entity": pl.Utf8, "period": pl.Utf8},
382
+ )
383
+
384
+ entities = (
385
+ _drop_blank_rows(
386
+ _read_csv(input_dir / "entity.csv", ["entity"], provider=provider),
387
+ ["entity"],
388
+ ).get_column("entity").to_list()
389
+ )
390
+ d_realize_invest = frozenset(
391
+ _drop_blank_rows(
392
+ _read_csv(
393
+ solve_data_dir / "realized_invest_periods_of_current_solve.csv",
394
+ ["period"],
395
+ provider=provider,
396
+ ),
397
+ ["period"],
398
+ ).get_column("period").to_list()
399
+ )
400
+ d_fix_storage = frozenset(
401
+ _drop_blank_rows(
402
+ _read_csv(
403
+ solve_data_dir / "d_fix_storage_period_set.csv", ["period"],
404
+ provider=provider,
405
+ ),
406
+ ["period"],
407
+ ).get_column("period").to_list()
408
+ )
409
+ d_realized = frozenset(
410
+ _drop_blank_rows(
411
+ _read_csv(
412
+ solve_data_dir / "d_realized_period_set.csv", ["period"],
413
+ provider=provider,
414
+ ),
415
+ ["period"],
416
+ ).get_column("period").to_list()
417
+ )
418
+ realized_periods = d_realize_invest | d_fix_storage | d_realized
419
+
420
+ pb = _drop_blank_rows(
421
+ _read_csv(
422
+ solve_data_dir / "period__branch.csv", ["period", "branch"],
423
+ provider=provider,
424
+ ),
425
+ ["period", "branch"],
426
+ )
427
+ diag_periods = frozenset(
428
+ d for d, b in pb.iter_rows() if d == b
429
+ )
430
+
431
+ rows: list[tuple[str, str]] = [
432
+ (e, d) for e in entities
433
+ for d in realized_periods if d in diag_periods
434
+ ]
435
+ return pl.DataFrame(
436
+ {"entity": [r[0] for r in rows], "period": [r[1] for r in rows]},
437
+ schema={"entity": pl.Utf8, "period": pl.Utf8},
438
+ )
439
+
440
+
441
+ def emit_ed_history_realized_first(
442
+ input_dir: Path, solve_data_dir: Path,
443
+ *, provider,
444
+ ) -> None:
445
+ """Emit ``ed_history_realized_first`` to the Provider."""
446
+ _emit(provider, "solve_data/ed_history_realized_first.csv",
447
+ derive_ed_history_realized_first(
448
+ input_dir, solve_data_dir, provider=provider,
449
+ ))
450
+
451
+
452
+ # ---- process_source_sink_coeff_zero (mod L1973) ---------------------------
453
+
454
+ def derive_process_source_sink_coeff_zero(
455
+ solve_data_dir: Path,
456
+ *,
457
+ provider: "object | None" = None,
458
+ ) -> pl.DataFrame:
459
+ """``process_source_sink`` filtered by zero flow coefficient on EITHER side."""
460
+ triples = _read_n_col_rows(
461
+ solve_data_dir / "process_source_sink.csv",
462
+ ["process", "source", "sink"],
463
+ provider=provider,
464
+ )
465
+ src_zero = frozenset(_read_n_col_rows(
466
+ solve_data_dir / "process_source_coeff_zero.csv",
467
+ ["process", "source"],
468
+ provider=provider,
469
+ ))
470
+ sink_zero = frozenset(_read_n_col_rows(
471
+ solve_data_dir / "process_sink_coeff_zero.csv",
472
+ ["process", "sink"],
473
+ provider=provider,
474
+ ))
475
+ rows = [
476
+ (p, src, sink) for p, src, sink in triples
477
+ if (p, src) in src_zero or (p, sink) in sink_zero
478
+ ]
479
+ return pl.DataFrame(
480
+ {
481
+ "process": [r[0] for r in rows],
482
+ "source": [r[1] for r in rows],
483
+ "sink": [r[2] for r in rows],
484
+ },
485
+ schema={"process": pl.Utf8, "source": pl.Utf8, "sink": pl.Utf8},
486
+ )
487
+
488
+
489
+ def emit_process_source_sink_coeff_zero(
490
+ solve_data_dir: Path,
491
+ *, provider,
492
+ ) -> None:
493
+ """Emit ``process_source_sink_coeff_zero``."""
494
+ _emit(provider, "solve_data/process_source_sink_coeff_zero.csv",
495
+ derive_process_source_sink_coeff_zero(solve_data_dir,
496
+ provider=provider))
497
+
498
+
499
+ # ---- process_source_sink_{delayed,undelayed} (mod L1096-1097) --------------
500
+
501
+ def derive_process_source_sink_delayed_partition(
502
+ solve_data_dir: Path,
503
+ *,
504
+ provider: "object | None" = None,
505
+ ) -> tuple[pl.DataFrame, pl.DataFrame]:
506
+ """Partition ``process_source_sink`` by membership in ``process_delayed``."""
507
+ triples = _read_n_col_rows(
508
+ solve_data_dir / "process_source_sink.csv",
509
+ ["process", "source", "sink"],
510
+ provider=provider,
511
+ )
512
+ delayed_set = frozenset(
513
+ _drop_blank_rows(
514
+ _read_csv(
515
+ solve_data_dir / "process_delayed.csv", ["process"],
516
+ provider=provider,
517
+ ),
518
+ ["process"],
519
+ ).get_column("process").to_list()
520
+ )
521
+ delayed_rows = [r for r in triples if r[0] in delayed_set]
522
+ undelayed_rows = [r for r in triples if r[0] not in delayed_set]
523
+
524
+ def _to_df(rows: list[tuple[str, ...]]) -> pl.DataFrame:
525
+ return pl.DataFrame(
526
+ {
527
+ "process": [r[0] for r in rows],
528
+ "source": [r[1] for r in rows],
529
+ "sink": [r[2] for r in rows],
530
+ },
531
+ schema={"process": pl.Utf8, "source": pl.Utf8, "sink": pl.Utf8},
532
+ )
533
+
534
+ return _to_df(delayed_rows), _to_df(undelayed_rows)
535
+
536
+
537
+ def emit_process_source_sink_delayed_partition(
538
+ solve_data_dir: Path,
539
+ *, provider,
540
+ ) -> None:
541
+ """Emit the ``process_source_sink_delayed``/``_undelayed`` partition."""
542
+ delayed, undelayed = derive_process_source_sink_delayed_partition(
543
+ solve_data_dir, provider=provider,
544
+ )
545
+ _emit(provider, "solve_data/process_source_sink_delayed.csv", delayed)
546
+ _emit(provider, "solve_data/process_source_sink_undelayed.csv", undelayed)
547
+
548
+
549
+ # ---------------------------------------------------------------------------
550
+ # process_source_sink_ramp_family (mod L1660-1688)
551
+ # ---------------------------------------------------------------------------
552
+
553
+ def _read_p_proc_side_lookup(path: Path,
554
+ *,
555
+ provider: "object | None" = None,
556
+ ) -> dict[tuple[str, str, str], float]:
557
+ """Read p_process_source / p_process_sink: (process, side, param) → value.
558
+
559
+ Provider-only after Step 2.5 Phase C — returns empty when the
560
+ Provider misses the key (matches legacy missing-CSV behaviour).
561
+ """
562
+ out: dict[tuple[str, str, str], float] = {}
563
+ if provider is None or not provider.has(_provider_key(path)):
564
+ return out
565
+ df = _read_csv(path, ["process", "side", "param", "value"], provider=provider)
566
+ df = _drop_blank_rows(df, ["process", "side", "param"])
567
+ for p, s, param, v in df.iter_rows():
568
+ try:
569
+ out[(p, s, param)] = float(v)
570
+ except (TypeError, ValueError):
571
+ continue
572
+ return out
573
+
574
+
575
+ def _compute_ramp_family(
576
+ input_dir: Path, solve_data_dir: Path,
577
+ *,
578
+ provider: "object | None" = None,
579
+ ) -> dict[str, list[tuple[str, str, str]]]:
580
+ """Emit the 5 ramp-family triple sets.
581
+
582
+ Returns ``{filename → rows}``. Rows preserve ``process_source_sink``
583
+ order; legacy emits no dedup (input is already unique).
584
+ """
585
+ triples = _read_n_col_rows(
586
+ solve_data_dir / "process_source_sink.csv",
587
+ ["process", "source", "sink"],
588
+ provider=provider,
589
+ )
590
+ pnrm_rows = _read_n_col_rows(
591
+ input_dir / "process__node__ramp_method.csv",
592
+ ["process", "node", "ramp_method"],
593
+ provider=provider,
594
+ )
595
+ pnrm: dict[tuple[str, str], set[str]] = {}
596
+ for p, n, m in pnrm_rows:
597
+ pnrm.setdefault((p, n), set()).add(m)
598
+
599
+ p_proc_source = _read_p_proc_side_lookup(
600
+ input_dir / "p_process_source.csv", provider=provider,
601
+ )
602
+ p_proc_sink = _read_p_proc_side_lookup(
603
+ input_dir / "p_process_sink.csv", provider=provider,
604
+ )
605
+
606
+ def _has_method(p: str, n: str, methods: frozenset[str]) -> bool:
607
+ return bool(pnrm.get((p, n), set()) & methods)
608
+
609
+ rsu = [
610
+ (p, src, sink) for p, src, sink in triples
611
+ if _has_method(p, src, _RAMP_LIMIT_METHOD)
612
+ and p_proc_source.get((p, src, "ramp_speed_up"), 0.0) > 0
613
+ ]
614
+ siu = [
615
+ (p, src, sink) for p, src, sink in triples
616
+ if _has_method(p, sink, _RAMP_LIMIT_METHOD)
617
+ and p_proc_sink.get((p, sink, "ramp_speed_up"), 0.0) > 0
618
+ ]
619
+ rsd = [
620
+ (p, src, sink) for p, src, sink in triples
621
+ if _has_method(p, src, _RAMP_LIMIT_METHOD)
622
+ and p_proc_source.get((p, src, "ramp_speed_down"), 0.0) > 0
623
+ ]
624
+ sid = [
625
+ (p, src, sink) for p, src, sink in triples
626
+ if _has_method(p, sink, _RAMP_LIMIT_METHOD)
627
+ and p_proc_sink.get((p, sink, "ramp_speed_down"), 0.0) > 0
628
+ ]
629
+ cost = [
630
+ (p, src, sink) for p, src, sink in triples
631
+ if _has_method(p, src, _RAMP_COST_METHOD)
632
+ or _has_method(p, sink, _RAMP_COST_METHOD)
633
+ ]
634
+ return {
635
+ "process_source_sink_ramp_limit_source_up.csv": rsu,
636
+ "process_source_sink_ramp_limit_sink_up.csv": siu,
637
+ "process_source_sink_ramp_limit_source_down.csv": rsd,
638
+ "process_source_sink_ramp_limit_sink_down.csv": sid,
639
+ "process_source_sink_ramp_cost.csv": cost,
640
+ }
641
+
642
+
643
+ def _triples_frame(rows: list[tuple[str, str, str]]) -> pl.DataFrame:
644
+ return pl.DataFrame(
645
+ {
646
+ "process": [r[0] for r in rows],
647
+ "source": [r[1] for r in rows],
648
+ "sink": [r[2] for r in rows],
649
+ },
650
+ schema={"process": pl.Utf8, "source": pl.Utf8, "sink": pl.Utf8},
651
+ )
652
+
653
+
654
+ def emit_process_source_sink_ramp_family(
655
+ input_dir: Path, solve_data_dir: Path,
656
+ *, provider,
657
+ ) -> None:
658
+ """Emit ``process_source_sink_ramp_family`` to the Provider."""
659
+ by_file = _compute_ramp_family(input_dir, solve_data_dir, provider=provider)
660
+ for fname, rows in by_file.items():
661
+ _emit(provider, f"solve_data/{fname}", _triples_frame(rows))
662
+
663
+
664
+ # ===========================================================================
665
+ # Phase 1 follow-up 4 — param_in_use family + dispatch-fully-inside set.
666
+ # ===========================================================================
667
+
668
+
669
+ # Per-class param taxonomies — mirror flextool_base.dat. Pinned here to
670
+ # avoid a transitive import from the legacy preprocessing tree (matches
671
+ # the pattern used for ramp / method constants above).
672
+ #
673
+ # Update both sites in lockstep if base.dat changes; the parity tests
674
+ # catch drift.
675
+
676
+ # flextool_base.dat:144-152 — processPeriodParam family.
677
+ _PROCESS_PERIOD_PARAM: frozenset[str] = frozenset((
678
+ "fixed_cost", "other_operational_cost", "lifetime", "existing",
679
+ "discount_rate", "invest_cost", "salvage_value",
680
+ "invest_max_period", "invest_min_period",
681
+ "cumulative_max_capacity", "cumulative_min_capacity",
682
+ "retire_forced", "retire_max_period", "retire_min_period", "startup_cost",
683
+ ))
684
+ _PROCESS_PERIOD_PARAM_REQUIRED: frozenset[str] = frozenset((
685
+ "fixed_cost", "other_operational_cost", "lifetime", "existing",
686
+ ))
687
+ _PROCESS_PERIOD_PARAM_INVEST: frozenset[str] = frozenset((
688
+ "discount_rate", "invest_cost", "salvage_value",
689
+ "invest_max_period", "invest_min_period",
690
+ "cumulative_max_capacity", "cumulative_min_capacity",
691
+ "retire_forced", "retire_max_period", "retire_min_period",
692
+ ))
693
+
694
+ # flextool_base.dat:153-154 — processTimeParam family.
695
+ _PROCESS_TIME_PARAM_REQUIRED: frozenset[str] = frozenset((
696
+ "efficiency", "other_operational_cost", "availability",
697
+ ))
698
+
699
+ # flextool_base.dat:158-161 — sourceSinkTime/PeriodParam family
700
+ # (period == time taxonomy in this version of base.dat).
701
+ _SOURCE_SINK_TIME_PARAM: frozenset[str] = frozenset((
702
+ "efficiency", "efficiency_at_min_load", "min_load", "other_operational_cost",
703
+ ))
704
+ _SOURCE_SINK_TIME_PARAM_REQUIRED: frozenset[str] = frozenset((
705
+ "efficiency", "other_operational_cost",
706
+ ))
707
+
708
+ # flextool_base.dat:168-177 — nodePeriodParam family.
709
+ _NODE_PERIOD_PARAM: frozenset[str] = frozenset((
710
+ "annual_flow", "peak_inflow", "fixed_cost", "discount_rate",
711
+ "invest_cost", "salvage_value",
712
+ "invest_max_period", "invest_min_period", "lifetime",
713
+ "cumulative_max_capacity", "cumulative_min_capacity",
714
+ "retire_forced", "retire_max_period", "retire_min_period",
715
+ "virtual_unitsize",
716
+ "storage_state_reference_price", "existing", "penalty_up", "penalty_down",
717
+ ))
718
+ _NODE_PERIOD_PARAM_REQUIRED: frozenset[str] = frozenset((
719
+ "annual_flow", "peak_inflow", "fixed_cost", "lifetime",
720
+ "storage_state_reference_price", "existing",
721
+ "penalty_up", "penalty_down",
722
+ ))
723
+ _NODE_PERIOD_PARAM_INVEST: frozenset[str] = frozenset((
724
+ "discount_rate", "invest_cost", "salvage_value",
725
+ "invest_max_period", "invest_min_period",
726
+ "cumulative_max_capacity", "cumulative_min_capacity",
727
+ "retire_forced", "retire_max_period", "retire_min_period",
728
+ "virtual_unitsize",
729
+ ))
730
+
731
+
732
+ # ---- write_param_in_use_sets (mod L1247 / L1369) --------------------------
733
+ #
734
+ # Emits seven param-in-use CSVs. Legacy implementation iterates a small
735
+ # python dictionary keyed by (entity, param) and dedupes with
736
+ # ``dict.fromkeys``; native polars wouldn't be faster for this shape —
737
+ # the inputs are tiny enum cross-products. We mirror the legacy loops
738
+ # directly inside ``derive_*`` for code-shape parity.
739
+
740
+ def _read_singles_list(path: Path,
741
+ *,
742
+ provider: "object | None" = None) -> list[str]:
743
+ """Read column 0 of a small CSV into a list (preserves CSV order)."""
744
+ return [
745
+ r[0] for r in _read_n_col_rows(path, ["c0"], provider=provider)
746
+ ]
747
+
748
+
749
+ def _derive_node_period_param_in_use(
750
+ nodes: list[str], invest_set: frozenset[str], divest_set: frozenset[str],
751
+ ) -> list[tuple[str, str]]:
752
+ rows: list[tuple[str, str]] = []
753
+ for n in nodes:
754
+ is_invest = n in invest_set or n in divest_set
755
+ for param in _NODE_PERIOD_PARAM:
756
+ if param in _NODE_PERIOD_PARAM_REQUIRED:
757
+ rows.append((n, param))
758
+ elif is_invest and param in _NODE_PERIOD_PARAM_INVEST:
759
+ rows.append((n, param))
760
+ return list(dict.fromkeys(rows))
761
+
762
+
763
+ def _derive_process_period_param_in_use(
764
+ processes: list[str], invest_set: frozenset[str],
765
+ divest_set: frozenset[str], process_online: frozenset[str],
766
+ ) -> list[tuple[str, str]]:
767
+ rows: list[tuple[str, str]] = []
768
+ for p in processes:
769
+ is_invest = p in invest_set or p in divest_set
770
+ is_online = p in process_online
771
+ for param in _PROCESS_PERIOD_PARAM:
772
+ if param in _PROCESS_PERIOD_PARAM_REQUIRED:
773
+ rows.append((p, param))
774
+ elif is_invest and param in _PROCESS_PERIOD_PARAM_INVEST:
775
+ rows.append((p, param))
776
+ elif is_online and param == "startup_cost":
777
+ rows.append((p, param))
778
+ return list(dict.fromkeys(rows))
779
+
780
+
781
+ def _derive_process_time_param_in_use(
782
+ processes: list[str], p_with_min_load: frozenset[str],
783
+ ) -> list[tuple[str, str]]:
784
+ rows: list[tuple[str, str]] = []
785
+ for p in processes:
786
+ for param in _PROCESS_TIME_PARAM:
787
+ if param in _PROCESS_TIME_PARAM_REQUIRED:
788
+ rows.append((p, param))
789
+ elif (p in p_with_min_load
790
+ and param in ("min_load", "efficiency_at_min_load")):
791
+ rows.append((p, param))
792
+ return list(dict.fromkeys(rows))
793
+
794
+
795
+ def _derive_pss_param_in_use(
796
+ pairs: list[tuple[str, str]], p_with_min_load: frozenset[str],
797
+ enum: frozenset[str], required: frozenset[str],
798
+ ) -> list[tuple[str, str, str]]:
799
+ rows: list[tuple[str, str, str]] = []
800
+ for p, side in pairs:
801
+ for param in enum:
802
+ if param in required:
803
+ rows.append((p, side, param))
804
+ elif (p in p_with_min_load
805
+ and param in ("min_load", "efficiency_at_min_load")):
806
+ rows.append((p, side, param))
807
+ return list(dict.fromkeys(rows))
808
+
809
+
810
+ def _rows_to_frame_2(rows: list[tuple[str, str]],
811
+ cols: tuple[str, str]) -> pl.DataFrame:
812
+ return pl.DataFrame(
813
+ {cols[0]: [r[0] for r in rows], cols[1]: [r[1] for r in rows]},
814
+ schema={cols[0]: pl.Utf8, cols[1]: pl.Utf8},
815
+ )
816
+
817
+
818
+ def _rows_to_frame_3(rows: list[tuple[str, str, str]],
819
+ cols: tuple[str, str, str]) -> pl.DataFrame:
820
+ return pl.DataFrame(
821
+ {
822
+ cols[0]: [r[0] for r in rows],
823
+ cols[1]: [r[1] for r in rows],
824
+ cols[2]: [r[2] for r in rows],
825
+ },
826
+ schema={c: pl.Utf8 for c in cols},
827
+ )
828
+
829
+
830
+ def emit_param_in_use_sets(input_dir: Path, solve_data_dir: Path,
831
+ *, provider) -> None:
832
+ """Emit ``param_in_use_sets`` to the Provider."""
833
+ nodes = _read_singles_list(input_dir / "node.csv", provider=provider)
834
+ processes = _read_singles_list(input_dir / "process.csv", provider=provider)
835
+ invest_set = frozenset(
836
+ _read_singles_list(
837
+ solve_data_dir / "entityInvest.csv", provider=provider,
838
+ )
839
+ )
840
+ divest_set = frozenset(
841
+ _read_singles_list(
842
+ solve_data_dir / "entityDivest.csv", provider=provider,
843
+ )
844
+ )
845
+ ctm = _read_n_col_rows(
846
+ solve_data_dir / "process__ct_method.csv", ["process", "method"],
847
+ provider=provider,
848
+ )
849
+ p_with_min_load = frozenset(
850
+ p for p, m in ctm if m == "min_load_efficiency"
851
+ )
852
+ process_online = frozenset(
853
+ _read_singles_list(
854
+ solve_data_dir / "process_online.csv", provider=provider,
855
+ )
856
+ )
857
+ sources = [
858
+ (p, src) for p, src in _read_n_col_rows(
859
+ input_dir / "process__source.csv", ["process", "source"],
860
+ provider=provider,
861
+ )
862
+ ]
863
+ sinks = [
864
+ (p, snk) for p, snk in _read_n_col_rows(
865
+ input_dir / "process__sink.csv", ["process", "sink"],
866
+ provider=provider,
867
+ )
868
+ ]
869
+
870
+ _emit(provider, "solve_data/node__PeriodParam_in_use.csv",
871
+ _rows_to_frame_2(
872
+ _derive_node_period_param_in_use(nodes, invest_set, divest_set),
873
+ ("node", "param"),
874
+ ))
875
+ _emit(provider, "solve_data/process__PeriodParam_in_use.csv",
876
+ _rows_to_frame_2(
877
+ _derive_process_period_param_in_use(
878
+ processes, invest_set, divest_set, process_online,
879
+ ),
880
+ ("process", "param"),
881
+ ))
882
+ _emit(provider, "solve_data/process_TimeParam_in_use.csv",
883
+ _rows_to_frame_2(
884
+ _derive_process_time_param_in_use(processes, p_with_min_load),
885
+ ("process", "param"),
886
+ ))
887
+ _emit(provider, "solve_data/process_source_sourceSinkTimeParam_in_use.csv",
888
+ _rows_to_frame_3(
889
+ _derive_pss_param_in_use(
890
+ sources, p_with_min_load,
891
+ _SOURCE_SINK_TIME_PARAM, _SOURCE_SINK_TIME_PARAM_REQUIRED,
892
+ ),
893
+ ("process", "source", "param"),
894
+ ))
895
+ _emit(provider, "solve_data/process_sink_sourceSinkTimeParam_in_use.csv",
896
+ _rows_to_frame_3(
897
+ _derive_pss_param_in_use(
898
+ sinks, p_with_min_load,
899
+ _SOURCE_SINK_TIME_PARAM, _SOURCE_SINK_TIME_PARAM_REQUIRED,
900
+ ),
901
+ ("process", "sink", "param"),
902
+ ))
903
+
904
+
905
+ # ---- write_node_group_dispatch_process_fully_inside (mod L1789-1794) ------
906
+
907
+ def derive_node_group_dispatch_process_fully_inside(
908
+ input_dir: Path, solve_data_dir: Path,
909
+ *,
910
+ provider: "object | None" = None,
911
+ ) -> pl.DataFrame:
912
+ """For each ``g ∈ nodeGroupDispatch`` × ``p ∈ process``, include if
913
+ BOTH some source and some sink of ``p`` are in ``group__node[g]``
914
+ AND ``p`` is not a self-loop (no ``(p, n, n)`` in ``process_source_sink``).
915
+ """
916
+ ngd = _read_singles_list(
917
+ input_dir / "nodeGroupDispatch.csv", provider=provider,
918
+ )
919
+ procs = _read_singles_list(
920
+ input_dir / "process.csv", provider=provider,
921
+ )
922
+ process_source_pairs = _read_n_col_rows(
923
+ input_dir / "process__source.csv", ["process", "source"],
924
+ provider=provider,
925
+ )
926
+ process_sink_pairs = _read_n_col_rows(
927
+ input_dir / "process__sink.csv", ["process", "sink"],
928
+ provider=provider,
929
+ )
930
+ gn = _read_n_col_rows(
931
+ input_dir / "group__node.csv", ["group", "node"], provider=provider,
932
+ )
933
+ triples = _read_n_col_rows(
934
+ solve_data_dir / "process_source_sink.csv",
935
+ ["process", "source", "sink"],
936
+ provider=provider,
937
+ )
938
+
939
+ nodes_in_g: dict[str, set[str]] = {}
940
+ for g, n in gn:
941
+ nodes_in_g.setdefault(g, set()).add(n)
942
+ sources_of_p: dict[str, set[str]] = {}
943
+ for p, src in process_source_pairs:
944
+ sources_of_p.setdefault(p, set()).add(src)
945
+ sinks_of_p: dict[str, set[str]] = {}
946
+ for p, snk in process_sink_pairs:
947
+ sinks_of_p.setdefault(p, set()).add(snk)
948
+ self_loop_processes = frozenset(
949
+ p for p, src, snk in triples if src == snk
950
+ )
951
+
952
+ rows: list[tuple[str, str]] = []
953
+ for g in ngd:
954
+ gnodes = nodes_in_g.get(g, set())
955
+ if not gnodes:
956
+ continue
957
+ for p in procs:
958
+ if p in self_loop_processes:
959
+ continue
960
+ srcs = sources_of_p.get(p, set())
961
+ snks = sinks_of_p.get(p, set())
962
+ if (srcs & gnodes) and (snks & gnodes):
963
+ rows.append((g, p))
964
+ return _rows_to_frame_2(rows, ("group", "process"))
965
+
966
+
967
+ def emit_node_group_dispatch_process_fully_inside(
968
+ input_dir: Path, solve_data_dir: Path,
969
+ *, provider,
970
+ ) -> None:
971
+ """Emit ``node_group_dispatch_process_fully_inside`` to the Provider."""
972
+ _emit(provider, "solve_data/nodeGroupDispatch__process_fully_inside.csv",
973
+ derive_node_group_dispatch_process_fully_inside(
974
+ input_dir, solve_data_dir, provider=provider,
975
+ ))
976
+
977
+
978
+ # ===========================================================================
979
+ # Phase 1 follow-up 5 — small_set_derivations + arc-union small writers
980
+ # ===========================================================================
981
+
982
+ # Helpers used by the small writers below. These are byte-for-byte parity
983
+ # with the legacy ``_read_singles`` / ``_read_pairs`` / ``_write_csv``
984
+ # helpers in ``process_arc_unions``; we keep them local to this module
985
+ # rather than reaching into the legacy module so the native port has no
986
+ # transitive import from preprocessing.
987
+
988
+
989
+ def _read_singles_csv(path: Path,
990
+ *,
991
+ provider: "object | None" = None) -> list[str]:
992
+ df = provider.get(_provider_key(path))
993
+ if df is None:
994
+ return []
995
+ out: list[str] = []
996
+ for row in df.iter_rows():
997
+ if not row:
998
+ continue
999
+ c0 = _cell_str(row[0])
1000
+ if c0:
1001
+ out.append(c0)
1002
+ return out
1003
+
1004
+
1005
+ def _read_pairs_csv(path: Path,
1006
+ *,
1007
+ provider: "object | None" = None) -> list[tuple[str, str]]:
1008
+ df = provider.get(_provider_key(path))
1009
+ if df is None:
1010
+ return []
1011
+ out: list[tuple[str, str]] = []
1012
+ for row in df.iter_rows():
1013
+ if len(row) >= 2:
1014
+ c0, c1 = _cell_str(row[0]), _cell_str(row[1])
1015
+ if c0 and c1:
1016
+ out.append((c0, c1))
1017
+ return out
1018
+
1019
+
1020
+ def _read_n_col_csv(path: Path, n: int,
1021
+ *,
1022
+ provider: "object | None" = None) -> list[tuple[str, ...]]:
1023
+ df = provider.get(_provider_key(path))
1024
+ if df is None:
1025
+ return []
1026
+ out: list[tuple[str, ...]] = []
1027
+ for row in df.iter_rows():
1028
+ if len(row) >= n:
1029
+ cells = tuple(_cell_str(row[i]) for i in range(n))
1030
+ if all(cells):
1031
+ out.append(cells)
1032
+ return out
1033
+
1034
+
1035
+ def _rows_to_frame(rows, header: tuple[str, ...]) -> pl.DataFrame:
1036
+ """Build an all-Utf8 ``pl.DataFrame`` from rows + a header tuple.
1037
+
1038
+ Header becomes column names; each tuple element a string cell.
1039
+ Uses column-of-tuples projection so empty-row frames still carry
1040
+ the requested schema.
1041
+ """
1042
+ n = len(header)
1043
+ cols: list[list[str]] = [[] for _ in range(n)]
1044
+ for r in rows:
1045
+ for i in range(n):
1046
+ cols[i].append(r[i])
1047
+ return pl.DataFrame(
1048
+ {header[i]: cols[i] for i in range(n)},
1049
+ schema={h: pl.Utf8 for h in header},
1050
+ )
1051
+
1052
+
1053
+ # ---- write_small_set_derivations (mod L999, L1061, L1132, L1174, L1222-3) --
1054
+
1055
+ def derive_process_source_sink_profile_method(
1056
+ solve_data_dir: Path,
1057
+ *,
1058
+ provider: "object | None" = None,
1059
+ ) -> pl.DataFrame:
1060
+ """4-way union of the *profile_method* sub-CSVs (5-col frame)."""
1061
+ seen_pf: dict[tuple[str, ...], None] = {}
1062
+ for fname in (
1063
+ "process__profileProcess__toSink__profile__profile_method.csv",
1064
+ "process__source__toProfileProcess__profile__profile_method.csv",
1065
+ "process__source__sink__profile__profile_method_connection.csv",
1066
+ "process__source__sink__profile__profile_method_direct.csv",
1067
+ ):
1068
+ for r in _read_n_col_csv(
1069
+ solve_data_dir / fname, 5, provider=provider,
1070
+ ):
1071
+ seen_pf.setdefault(r, None)
1072
+ return _rows_to_frame(
1073
+ list(seen_pf.keys()),
1074
+ ("process", "source", "sink", "profile", "profile_method"),
1075
+ )
1076
+
1077
+
1078
+ def emit_small_set_derivations(solve_data_dir: Path,
1079
+ *, provider) -> None:
1080
+ """Emit the small per-solve set derivations consumed downstream."""
1081
+ _emit(provider,
1082
+ "solve_data/process__source__sink__profile__profile_method.csv",
1083
+ derive_process_source_sink_profile_method(
1084
+ solve_data_dir, provider=provider,
1085
+ ))
1086
+
1087
+
1088
+ # ---- write_p_process_delay_weight (mod L1096-1099) ------------------------
1089
+
1090
+ def derive_p_process_delay_weight(
1091
+ input_dir: Path, solve_data_dir: Path,
1092
+ *,
1093
+ provider: "object | None" = None,
1094
+ ) -> pl.DataFrame:
1095
+ """``p_process_delay_weight`` 3-col frame; see writer docstring."""
1096
+ delayed_duration = _read_pairs_csv(
1097
+ solve_data_dir / "process_delayed__duration.csv", provider=provider,
1098
+ )
1099
+ delay_single = frozenset(
1100
+ _read_pairs_csv(
1101
+ input_dir / "process_delay_single.csv", provider=provider,
1102
+ )
1103
+ )
1104
+ weighted: dict[tuple[str, str], float] = {}
1105
+ pdw_path = input_dir / "p_process_delay_weighted.csv"
1106
+ _df = provider.get(_provider_key(pdw_path))
1107
+ if _df is not None:
1108
+ for r in _df.iter_rows():
1109
+ if len(r) >= 3:
1110
+ c0, c1 = _cell_str(r[0]), _cell_str(r[1])
1111
+ if c0 and c1:
1112
+ try:
1113
+ weighted[(c0, c1)] = float(r[2])
1114
+ except (ValueError, TypeError):
1115
+ continue
1116
+ rows: list[tuple[str, str, str]] = []
1117
+ for p, td in delayed_duration:
1118
+ v = 1.0 if (p, td) in delay_single else weighted.get((p, td), 0.0)
1119
+ rows.append((p, td, repr(v)))
1120
+ return _rows_to_frame(rows, ("process", "delay_duration", "value"))
1121
+
1122
+
1123
+ def emit_p_process_delay_weight(
1124
+ input_dir: Path, solve_data_dir: Path,
1125
+ *, provider,
1126
+ ) -> None:
1127
+ """Emit ``p_process_delay_weight`` to the Provider."""
1128
+ _emit(provider, "solve_data/p_process_delay_weight.csv",
1129
+ derive_p_process_delay_weight(
1130
+ input_dir, solve_data_dir, provider=provider,
1131
+ ))
1132
+
1133
+
1134
+ # ---- write_peedt (mod L1084) ----------------------------------------------
1135
+
1136
+ def derive_peedt(solve_data_dir: Path,
1137
+ *,
1138
+ provider: "object | None" = None) -> pl.DataFrame:
1139
+ """``peedt = process_source_sink × steps_in_use`` (5-col frame).
1140
+
1141
+ Hot-path for full-year fixtures — up to ~280k rows.
1142
+ """
1143
+ triples = _read_n_col_csv(
1144
+ solve_data_dir / "process_source_sink.csv", 3, provider=provider,
1145
+ )
1146
+ dt_pairs = _read_n_col_csv(
1147
+ solve_data_dir / "steps_in_use.csv", 2, provider=provider,
1148
+ )
1149
+ procs: list[str] = []
1150
+ srcs: list[str] = []
1151
+ snks: list[str] = []
1152
+ ds: list[str] = []
1153
+ ts: list[str] = []
1154
+ for p, src, snk in triples:
1155
+ for d, t in dt_pairs:
1156
+ procs.append(p)
1157
+ srcs.append(src)
1158
+ snks.append(snk)
1159
+ ds.append(d)
1160
+ ts.append(t)
1161
+ return pl.DataFrame(
1162
+ {
1163
+ "process": procs,
1164
+ "source": srcs,
1165
+ "sink": snks,
1166
+ "period": ds,
1167
+ "time": ts,
1168
+ },
1169
+ schema={
1170
+ "process": pl.Utf8, "source": pl.Utf8, "sink": pl.Utf8,
1171
+ "period": pl.Utf8, "time": pl.Utf8,
1172
+ },
1173
+ )
1174
+
1175
+
1176
+ def emit_peedt(solve_data_dir: Path,
1177
+ *, provider) -> None:
1178
+ """Emit ``peedt`` — period×entity×entity×dt index frame."""
1179
+ _emit(provider, "solve_data/peedt.csv",
1180
+ derive_peedt(solve_data_dir, provider=provider))
1181
+
1182
+
1183
+ # ===========================================================================
1184
+ # Phase 1 follow-up 6 — flow-bound + state-slack + storage reference price
1185
+ # + 12-CSV nodeGroupDispatch dispatch set family.
1186
+ #
1187
+ # All five writers in this section emit either a parameter table (long-form
1188
+ # ``(keys..., value)`` with ``repr(float)`` precision parity) or a set of
1189
+ # tuples; semantics mirror flextool.mod L1596-L1803 exactly. Each native
1190
+ # implementation reads the same input/solve_data CSVs as its legacy peer
1191
+ # and writes byte-identical output (CSV row order + float formatting).
1192
+ # ===========================================================================
1193
+
1194
+
1195
+ # ---- write_p_flow_max (mod L1661-1677) ------------------------------------
1196
+
1197
+
1198
+ def derive_p_flow_max(
1199
+ input_dir: Path, solve_data_dir: Path,
1200
+ *,
1201
+ provider: "object | None" = None,
1202
+ ) -> pl.DataFrame:
1203
+ """``p_flow_max`` 6-col frame; see :func:`write_p_flow_max`."""
1204
+ coeff_zero = frozenset(_read_n_col_csv(
1205
+ solve_data_dir / "process_source_sink_coeff_zero.csv", 3,
1206
+ provider=provider,
1207
+ ))
1208
+ has_indirect = frozenset(
1209
+ p for p, _m in _read_pairs_csv(
1210
+ solve_data_dir / "process__method_indirect.csv",
1211
+ provider=provider,
1212
+ )
1213
+ )
1214
+ process_source = frozenset(_read_pairs_csv(
1215
+ input_dir / "process__source.csv", provider=provider,
1216
+ ))
1217
+ process_sink = frozenset(_read_pairs_csv(
1218
+ input_dir / "process__sink.csv", provider=provider,
1219
+ ))
1220
+ has_min_load = frozenset(
1221
+ p for p, m in _read_pairs_csv(
1222
+ solve_data_dir / "process__ct_method.csv", provider=provider,
1223
+ )
1224
+ if m == "min_load_efficiency"
1225
+ )
1226
+
1227
+ dcm: dict[tuple[str, str], float] = {}
1228
+ pdcm_path = solve_data_dir / "p_entity_dispatch_capacity_max.csv"
1229
+ _df = provider.get(_provider_key(pdcm_path))
1230
+ if _df is not None:
1231
+ for r in _df.iter_rows():
1232
+ if len(r) >= 3:
1233
+ c0, c1 = _cell_str(r[0]), _cell_str(r[1])
1234
+ if c0 and c1:
1235
+ try:
1236
+ dcm[(c0, c1)] = float(r[2])
1237
+ except (ValueError, TypeError):
1238
+ continue
1239
+ unitsize: dict[str, float] = {}
1240
+ pus_path = solve_data_dir / "p_entity_unitsize.csv"
1241
+ _df = provider.get(_provider_key(pus_path))
1242
+ if _df is not None:
1243
+ for r in _df.iter_rows():
1244
+ if len(r) >= 2:
1245
+ c0 = _cell_str(r[0])
1246
+ if c0:
1247
+ try:
1248
+ unitsize[c0] = float(r[1])
1249
+ except (ValueError, TypeError):
1250
+ continue
1251
+
1252
+ slope: dict[tuple[str, str, str], float] = {}
1253
+ section: dict[tuple[str, str, str], float] = {}
1254
+ for fname, target in (
1255
+ ("pdtProcess_slope.csv", slope),
1256
+ ("pdtProcess_section.csv", section),
1257
+ ):
1258
+ path = solve_data_dir / fname
1259
+ _df = provider.get(_provider_key(path))
1260
+ if _df is not None:
1261
+ for r in _df.iter_rows():
1262
+ if len(r) >= 4:
1263
+ c0, c1, c2 = (
1264
+ _cell_str(r[0]), _cell_str(r[1]), _cell_str(r[2]),
1265
+ )
1266
+ if c0 and c1 and c2:
1267
+ try:
1268
+ target[(c0, c1, c2)] = float(r[3])
1269
+ except (ValueError, TypeError):
1270
+ continue
1271
+
1272
+ src_max_coef: dict[tuple[str, str], float] = {}
1273
+ pms_path = input_dir / "p_process_source_capacity_max_coeff.csv"
1274
+ _df = provider.get(_provider_key(pms_path))
1275
+ if _df is not None:
1276
+ for r in _df.iter_rows():
1277
+ if len(r) >= 3:
1278
+ c0, c1 = _cell_str(r[0]), _cell_str(r[1])
1279
+ if c0 and c1:
1280
+ try:
1281
+ src_max_coef[(c0, c1)] = float(r[2])
1282
+ except (ValueError, TypeError):
1283
+ continue
1284
+ sink_max_coef: dict[tuple[str, str], float] = {}
1285
+ pmk_path = input_dir / "p_process_sink_capacity_max_coeff.csv"
1286
+ _df = provider.get(_provider_key(pmk_path))
1287
+ if _df is not None:
1288
+ for r in _df.iter_rows():
1289
+ if len(r) >= 3:
1290
+ c0, c1 = _cell_str(r[0]), _cell_str(r[1])
1291
+ if c0 and c1:
1292
+ try:
1293
+ sink_max_coef[(c0, c1)] = float(r[2])
1294
+ except (ValueError, TypeError):
1295
+ continue
1296
+
1297
+ # p_unconstrained_flow_cap = max over models of
1298
+ # p_max_flow_for_unconstrained_variables[m]; default 1e6 if absent.
1299
+ p_uflow = 1_000_000.0
1300
+ pmfu_path = input_dir / "p_max_flow_for_unconstrained_variables.csv"
1301
+ _df = provider.get(_provider_key(pmfu_path))
1302
+ if _df is not None:
1303
+ max_v: float | None = None
1304
+ for r in _df.iter_rows():
1305
+ if len(r) >= 2 and _cell_str(r[0]):
1306
+ try:
1307
+ v = float(r[1])
1308
+ except (ValueError, TypeError):
1309
+ continue
1310
+ if max_v is None or v > max_v:
1311
+ max_v = v
1312
+ if max_v is not None:
1313
+ p_uflow = max_v
1314
+
1315
+ peedt = _read_n_col_csv(
1316
+ solve_data_dir / "peedt.csv", 5, provider=provider,
1317
+ )
1318
+ rows: list[tuple[str, ...]] = []
1319
+ for p, src, sink, d, t in peedt:
1320
+ if (p, src, sink) in coeff_zero:
1321
+ value = p_uflow
1322
+ else:
1323
+ us = unitsize.get(p, 1.0)
1324
+ dcm_v = dcm.get((p, d), 0.0)
1325
+ if p in has_indirect and (p, src) in process_source:
1326
+ if p in has_min_load:
1327
+ eff_term = (slope.get((p, d, t), 0.0)
1328
+ + section.get((p, d, t), 0.0))
1329
+ else:
1330
+ eff_term = slope.get((p, d, t), 0.0)
1331
+ src_coef = src_max_coef.get((p, src), 1.0)
1332
+ base = eff_term * (dcm_v / us) / src_coef
1333
+ else:
1334
+ base = dcm_v / us
1335
+ sink_coef = (sink_max_coef.get((p, sink), 1.0)
1336
+ if (p, sink) in process_sink else 1.0)
1337
+ value = base * sink_coef
1338
+ rows.append((p, src, sink, d, t, repr(value)))
1339
+ return _rows_to_frame(
1340
+ rows, ("process", "source", "sink", "period", "time", "value"),
1341
+ )
1342
+
1343
+
1344
+ def emit_p_flow_max(input_dir: Path, solve_data_dir: Path,
1345
+ *, provider) -> None:
1346
+ """Emit ``p_flow_max`` to the Provider."""
1347
+ _emit(provider, "solve_data/p_flow_max.csv",
1348
+ derive_p_flow_max(input_dir, solve_data_dir, provider=provider))
1349
+
1350
+
1351
+ # ---- write_p_storage_state_reference_price (mod L1693-1698) ---------------
1352
+
1353
+
1354
+ def derive_p_storage_state_reference_price(
1355
+ input_dir: Path, solve_data_dir: Path,
1356
+ *,
1357
+ provider: "object | None" = None,
1358
+ ) -> pl.DataFrame:
1359
+ """``p_storage_state_reference_price`` 3-col frame; see writer docstring."""
1360
+ # (n, d2, t2) → value, keyed by (node, period, step) from
1361
+ # ``handoff/fix_storage_price`` (canonical schema
1362
+ # ``[node, period, step, p_fix_storage_price]``). Phase 4.1f —
1363
+ # replaces the legacy ``solve_data/fix_storage_price.csv`` Provider
1364
+ # read; the translator seeds the handoff key at iteration start
1365
+ # (parent's data shadowing sequential when nested).
1366
+ from flextool.engine_polars import _provider_keys as K
1367
+ from flextool.engine_polars._provider_translators import (
1368
+ read_handoff_frame,
1369
+ )
1370
+ fix_price: dict[tuple[str, str, str], float] = {}
1371
+ fsp_df = read_handoff_frame(provider, K.HANDOFF_FIX_STORAGE_PRICE)
1372
+ if fsp_df is not None and fsp_df.height > 0:
1373
+ for n_, d_, t_, v_ in fsp_df.select(
1374
+ "node", "period", "step", "p_fix_storage_price",
1375
+ ).iter_rows():
1376
+ if n_ and d_ and t_ and v_ is not None and v_ != "":
1377
+ try:
1378
+ fix_price[(n_, d_, t_)] = float(v_)
1379
+ except (ValueError, TypeError):
1380
+ continue
1381
+
1382
+ ptl = _read_pairs_csv(
1383
+ solve_data_dir / "last_timesteps.csv", provider=provider,
1384
+ )
1385
+ ptl_for_d: dict[str, list[str]] = {}
1386
+ for d, t in ptl:
1387
+ ptl_for_d.setdefault(d, []).append(t)
1388
+ pb_d2_for_d: dict[str, list[str]] = {}
1389
+ for d2, d in _read_pairs_csv(
1390
+ solve_data_dir / "period__branch.csv", provider=provider,
1391
+ ):
1392
+ pb_d2_for_d.setdefault(d, []).append(d2)
1393
+ dtt_for_dt: dict[tuple[str, str], list[str]] = {}
1394
+ for d, t, t2 in _read_n_col_csv(
1395
+ solve_data_dir / "timeline_matching_map.csv", 3, provider=provider,
1396
+ ):
1397
+ dtt_for_dt.setdefault((d, t), []).append(t2)
1398
+
1399
+ use_ref = frozenset(
1400
+ n for n, m in _read_pairs_csv(
1401
+ input_dir / "node__storage_solve_horizon_method.csv",
1402
+ provider=provider,
1403
+ ) if m == "use_reference_price"
1404
+ )
1405
+
1406
+ pd_ref_price: dict[tuple[str, str], float] = {}
1407
+ pdn_path = solve_data_dir / "pdNode.csv"
1408
+ _df = provider.get(_provider_key(pdn_path))
1409
+ if _df is not None:
1410
+ for r in _df.iter_rows():
1411
+ if len(r) >= 4:
1412
+ c0, c2 = _cell_str(r[0]), _cell_str(r[2])
1413
+ if (c0
1414
+ and _cell_str(r[1]) == "storage_state_reference_price"
1415
+ and c2):
1416
+ try:
1417
+ pd_ref_price[(c0, c2)] = float(r[3])
1418
+ except (ValueError, TypeError):
1419
+ continue
1420
+
1421
+ nodes_state = _read_singles_csv(
1422
+ solve_data_dir / "nodeState.csv", provider=provider,
1423
+ )
1424
+ period_in_use = _read_singles_csv(
1425
+ solve_data_dir / "period_in_use_set.csv", provider=provider,
1426
+ )
1427
+
1428
+ rows: list[tuple[str, str, str]] = []
1429
+ for n in nodes_state:
1430
+ for d in period_in_use:
1431
+ sum_v = 0.0
1432
+ has_match = False
1433
+ for d2 in pb_d2_for_d.get(d, []):
1434
+ for t in ptl_for_d.get(d, []):
1435
+ for t2 in dtt_for_dt.get((d, t), []):
1436
+ v = fix_price.get((n, d2, t2))
1437
+ if v is not None:
1438
+ has_match = True
1439
+ sum_v += v
1440
+ if has_match:
1441
+ value = sum_v
1442
+ elif n in use_ref:
1443
+ value = pd_ref_price.get((n, d), 0.0)
1444
+ else:
1445
+ value = 0.0
1446
+ rows.append((n, d, repr(value)))
1447
+ return _rows_to_frame(rows, ("node", "period", "value"))
1448
+
1449
+
1450
+ def emit_p_storage_state_reference_price(
1451
+ input_dir: Path, solve_data_dir: Path,
1452
+ *, provider,
1453
+ ) -> None:
1454
+ """Emit ``p_storage_state_reference_price`` to the Provider."""
1455
+ _emit(provider, "solve_data/p_storage_state_reference_price.csv",
1456
+ derive_p_storage_state_reference_price(
1457
+ input_dir, solve_data_dir, provider=provider,
1458
+ ))
1459
+
1460
+
1461
+ # ---- write_node_group_dispatch_sets (mod L1596-1657) ----------------------
1462
+
1463
+
1464
+ def _compute_node_group_dispatch_sets(
1465
+ input_dir: Path, solve_data_dir: Path,
1466
+ *,
1467
+ provider: "object | None" = None,
1468
+ ) -> dict[str, tuple[tuple[str, ...], list[tuple[str, ...]]]]:
1469
+ """One shared scan; returns ``{filename → (header, rows)}`` for the
1470
+ 12 nodeGroupDispatch CSVs.
1471
+ """
1472
+ ngd = _read_singles_csv(
1473
+ input_dir / "nodeGroupDispatch.csv", provider=provider,
1474
+ )
1475
+ fag = frozenset(_read_singles_csv(
1476
+ input_dir / "flowAggregator.csv", provider=provider,
1477
+ ))
1478
+ p_unit = frozenset(_read_singles_csv(
1479
+ input_dir / "process_unit.csv", provider=provider,
1480
+ ))
1481
+ p_conn = frozenset(_read_singles_csv(
1482
+ input_dir / "process_connection.csv", provider=provider,
1483
+ ))
1484
+
1485
+ g_nodes_acc: dict[str, dict[str, None]] = {}
1486
+ for g, n in _read_pairs_csv(
1487
+ input_dir / "group__node.csv", provider=provider,
1488
+ ):
1489
+ g_nodes_acc.setdefault(g, {})[n] = None
1490
+ g_nodes: dict[str, frozenset[str]] = {
1491
+ g: frozenset(d.keys()) for g, d in g_nodes_acc.items()
1492
+ }
1493
+
1494
+ # flowGroup_process_node restricted to flowAggregator groups: (p, n) → [ga, ...]
1495
+ pn_to_aggregators: dict[tuple[str, str], list[str]] = {}
1496
+ for g, p, n in _read_n_col_csv(
1497
+ input_dir / "flowGroup__process__node.csv", 3, provider=provider,
1498
+ ):
1499
+ if g in fag:
1500
+ pn_to_aggregators.setdefault((p, n), []).append(g)
1501
+
1502
+ pss_always = _read_n_col_csv(
1503
+ solve_data_dir / "process_source_sink_alwaysProcess.csv", 3,
1504
+ provider=provider,
1505
+ )
1506
+ fully_inside = frozenset(_read_pairs_csv(
1507
+ solve_data_dir / "nodeGroupDispatch__process_fully_inside.csv",
1508
+ provider=provider,
1509
+ ))
1510
+
1511
+ def _emit_4tuple(*, kind: frozenset[str], side: str,
1512
+ not_aggregated: bool) -> list[tuple[str, ...]]:
1513
+ out: list[tuple[str, ...]] = []
1514
+ for g in ngd:
1515
+ gnodes = g_nodes.get(g, frozenset())
1516
+ if not gnodes:
1517
+ continue
1518
+ for p, src, sink in pss_always:
1519
+ if p not in kind:
1520
+ continue
1521
+ if (g, p) in fully_inside:
1522
+ continue
1523
+ n = sink if side == "sink" else src
1524
+ if n not in gnodes:
1525
+ continue
1526
+ if not_aggregated and (p, n) in pn_to_aggregators:
1527
+ continue
1528
+ out.append((g, p, src, sink))
1529
+ return out
1530
+
1531
+ def _emit_5tuple(*, kind: frozenset[str], side: str
1532
+ ) -> list[tuple[str, ...]]:
1533
+ out: list[tuple[str, ...]] = []
1534
+ for g in ngd:
1535
+ gnodes = g_nodes.get(g, frozenset())
1536
+ if not gnodes:
1537
+ continue
1538
+ for p, src, sink in pss_always:
1539
+ if p not in kind:
1540
+ continue
1541
+ if (g, p) in fully_inside:
1542
+ continue
1543
+ n = sink if side == "sink" else src
1544
+ if n not in gnodes:
1545
+ continue
1546
+ for ga in pn_to_aggregators.get((p, n), ()):
1547
+ out.append((g, ga, p, src, sink))
1548
+ return out
1549
+
1550
+ rows1 = _emit_4tuple(kind=p_unit, side="sink", not_aggregated=True)
1551
+ rows2 = _emit_4tuple(kind=p_unit, side="source", not_aggregated=True)
1552
+ rows3 = _emit_5tuple(kind=p_unit, side="sink")
1553
+ rows4 = _emit_5tuple(kind=p_unit, side="source")
1554
+ rows5 = _emit_4tuple(kind=p_conn, side="source", not_aggregated=True)
1555
+ rows6 = _emit_4tuple(kind=p_conn, side="sink", not_aggregated=True)
1556
+ rows8 = _emit_5tuple(kind=p_conn, side="sink")
1557
+ rows9 = _emit_5tuple(kind=p_conn, side="source")
1558
+
1559
+ # Set 7 — projection of 5 ∪ 6 to (g, connection).
1560
+ seen7: dict[tuple[str, str], None] = {}
1561
+ for g, p, _, _ in rows5:
1562
+ seen7.setdefault((g, p), None)
1563
+ for g, p, _, _ in rows6:
1564
+ seen7.setdefault((g, p), None)
1565
+ # Set 10 — projection of 8 ∪ 9 to (g, ga).
1566
+ seen10: dict[tuple[str, str], None] = {}
1567
+ for g, ga, _, _, _ in rows8:
1568
+ seen10.setdefault((g, ga), None)
1569
+ for g, ga, _, _, _ in rows9:
1570
+ seen10.setdefault((g, ga), None)
1571
+ # Set 11 — projection of rows3 to (g, ga).
1572
+ seen11: dict[tuple[str, str], None] = {}
1573
+ for g, ga, _, _, _ in rows3:
1574
+ seen11.setdefault((g, ga), None)
1575
+ # Set 12 — projection of rows4 to (g, ga).
1576
+ seen12: dict[tuple[str, str], None] = {}
1577
+ for g, ga, _, _, _ in rows4:
1578
+ seen12.setdefault((g, ga), None)
1579
+
1580
+ return {
1581
+ "nodeGroupDispatch__process__unit__to_node_Not_in_aggregate.csv": (
1582
+ ("group", "process", "unit", "node"), rows1,
1583
+ ),
1584
+ "nodeGroupDispatch__process__node__to_unit_Not_in_aggregate.csv": (
1585
+ ("group", "process", "node", "unit"), rows2,
1586
+ ),
1587
+ "nodeGroupDispatch__group_aggregate__process__unit__to_node.csv": (
1588
+ ("group", "group_aggregate", "unit", "source", "sink"), rows3,
1589
+ ),
1590
+ "nodeGroupDispatch__group_aggregate__process__node__to_unit.csv": (
1591
+ ("group", "group_aggregate", "unit", "source", "sink"), rows4,
1592
+ ),
1593
+ "nodeGroupDispatch__process__node__to_connection_Not_in_aggregate.csv": (
1594
+ ("group", "process", "node", "connection"), rows5,
1595
+ ),
1596
+ "nodeGroupDispatch__process__connection__to_node_Not_in_aggregate.csv": (
1597
+ ("group", "process", "connection", "node"), rows6,
1598
+ ),
1599
+ "nodeGroupDispatch__connection_Not_in_aggregate.csv": (
1600
+ ("group", "connection"), list(seen7.keys()),
1601
+ ),
1602
+ "nodeGroupDispatch__group_aggregate__process__connection__to_node.csv": (
1603
+ ("group", "group_aggregate", "connection", "source", "sink"), rows8,
1604
+ ),
1605
+ "nodeGroupDispatch__group_aggregate__process__node__to_connection.csv": (
1606
+ ("group", "group_aggregate", "connection", "source", "sink"), rows9,
1607
+ ),
1608
+ "nodeGroupDispatch__group_aggregate_Connection.csv": (
1609
+ ("group", "group_aggregate"), list(seen10.keys()),
1610
+ ),
1611
+ "nodeGroupDispatch__group_aggregate_Unit_to_group.csv": (
1612
+ ("group", "group_aggregate"), list(seen11.keys()),
1613
+ ),
1614
+ "nodeGroupDispatch__group_aggregate_Group_to_unit.csv": (
1615
+ ("group", "group_aggregate"), list(seen12.keys()),
1616
+ ),
1617
+ }
1618
+
1619
+
1620
+ def emit_node_group_dispatch_sets(
1621
+ input_dir: Path, solve_data_dir: Path,
1622
+ *, provider,
1623
+ ) -> None:
1624
+ """Emit ``node_group_dispatch_sets`` to the Provider."""
1625
+ by_file = _compute_node_group_dispatch_sets(
1626
+ input_dir, solve_data_dir, provider=provider,
1627
+ )
1628
+ for fname, (header, rows) in by_file.items():
1629
+ _emit(provider, f"solve_data/{fname}", _rows_to_frame(rows, header))
1630
+
1631
+