flextool 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (322) hide show
  1. flextool/__init__.py +41 -0
  2. flextool/_mem_sampler.py +193 -0
  3. flextool/_resources.py +43 -0
  4. flextool/calibrate/__init__.py +51 -0
  5. flextool/calibrate/__main__.py +11 -0
  6. flextool/calibrate/_cli.py +316 -0
  7. flextool/calibrate/_db_alt.py +166 -0
  8. flextool/calibrate/_final_outputs.py +110 -0
  9. flextool/calibrate/_guard.py +151 -0
  10. flextool/calibrate/_loop.py +558 -0
  11. flextool/calibrate/_readers.py +223 -0
  12. flextool/calibrate/_report.py +263 -0
  13. flextool/calibrate/_sizing.py +699 -0
  14. flextool/calibrate/_solve.py +134 -0
  15. flextool/calibrate/_solve_status.py +495 -0
  16. flextool/cli/__init__.py +9 -0
  17. flextool/cli/_console.py +51 -0
  18. flextool/cli/_timing.py +147 -0
  19. flextool/cli/cmd_execute_flextool_workflow.py +187 -0
  20. flextool/cli/cmd_export_to_tabular.py +56 -0
  21. flextool/cli/cmd_import_sensitivities.py +75 -0
  22. flextool/cli/cmd_migrate_database.py +13 -0
  23. flextool/cli/cmd_open_results_db.py +269 -0
  24. flextool/cli/cmd_read_matpower.py +66 -0
  25. flextool/cli/cmd_read_old_flextool.py +63 -0
  26. flextool/cli/cmd_read_self_describing_tabular_input.py +50 -0
  27. flextool/cli/cmd_read_tabular_input.py +81 -0
  28. flextool/cli/cmd_run_flextool.py +1095 -0
  29. flextool/cli/cmd_scenario_results.py +284 -0
  30. flextool/cli/cmd_solve_mps.py +169 -0
  31. flextool/cli/cmd_update_flextool.py +17 -0
  32. flextool/cli/cmd_write_outputs.py +125 -0
  33. flextool/common_utils/__init__.py +1 -0
  34. flextool/common_utils/plot_mem_shape.py +77 -0
  35. flextool/common_utils/precision.py +451 -0
  36. flextool/decomposition/__init__.py +0 -0
  37. flextool/decomposition/region_decomposition.py +128 -0
  38. flextool/decomposition/region_filter.py +1261 -0
  39. flextool/engine_polars/__init__.py +110 -0
  40. flextool/engine_polars/_axis_enums.py +742 -0
  41. flextool/engine_polars/_benders.py +3462 -0
  42. flextool/engine_polars/_block_layout.py +1479 -0
  43. flextool/engine_polars/_blocks.py +1515 -0
  44. flextool/engine_polars/_commodity_ladder.py +660 -0
  45. flextool/engine_polars/_cumulative_invest.py +1165 -0
  46. flextool/engine_polars/_db_loader.py +153 -0
  47. flextool/engine_polars/_db_reader.py +127 -0
  48. flextool/engine_polars/_dc_power_flow.py +445 -0
  49. flextool/engine_polars/_delay.py +442 -0
  50. flextool/engine_polars/_derived_arithmetic.py +432 -0
  51. flextool/engine_polars/_derived_block.py +990 -0
  52. flextool/engine_polars/_derived_branch.py +769 -0
  53. flextool/engine_polars/_derived_existing.py +1353 -0
  54. flextool/engine_polars/_derived_npv.py +1297 -0
  55. flextool/engine_polars/_derived_params.py +9850 -0
  56. flextool/engine_polars/_derived_profile.py +881 -0
  57. flextool/engine_polars/_derived_walks.py +276 -0
  58. flextool/engine_polars/_determinism.py +70 -0
  59. flextool/engine_polars/_direct_params.py +2186 -0
  60. flextool/engine_polars/_dump_csvs.py +1009 -0
  61. flextool/engine_polars/_emit_arc_unions.py +1631 -0
  62. flextool/engine_polars/_emit_calc_params.py +729 -0
  63. flextool/engine_polars/_emit_chain_params.py +709 -0
  64. flextool/engine_polars/_emit_co2_accumulators.py +400 -0
  65. flextool/engine_polars/_emit_dispatchers.py +690 -0
  66. flextool/engine_polars/_emit_energy_margin.py +125 -0
  67. flextool/engine_polars/_emit_energy_margin_adder.py +290 -0
  68. flextool/engine_polars/_emit_entity_annual.py +428 -0
  69. flextool/engine_polars/_emit_inflow_scaling.py +1420 -0
  70. flextool/engine_polars/_emit_leaf_sets.py +550 -0
  71. flextool/engine_polars/_emit_lp_scaling.py +665 -0
  72. flextool/engine_polars/_emit_mid_sets.py +859 -0
  73. flextool/engine_polars/_emit_pdt_params.py +759 -0
  74. flextool/engine_polars/_emit_per_solve.py +774 -0
  75. flextool/engine_polars/_emit_period_calc.py +504 -0
  76. flextool/engine_polars/_emit_period_params.py +2398 -0
  77. flextool/engine_polars/_emit_provider_io.py +141 -0
  78. flextool/engine_polars/_emit_reserve.py +574 -0
  79. flextool/engine_polars/_emit_solve_time.py +311 -0
  80. flextool/engine_polars/_emit_solve_writers.py +1249 -0
  81. flextool/engine_polars/_flex_data_accumulator.py +388 -0
  82. flextool/engine_polars/_flex_data_provider.py +478 -0
  83. flextool/engine_polars/_group_slack.py +1253 -0
  84. flextool/engine_polars/_inmemory_reader.py +140 -0
  85. flextool/engine_polars/_input_source.py +336 -0
  86. flextool/engine_polars/_invest_seeds.py +191 -0
  87. flextool/engine_polars/_native_input_writer.py +100 -0
  88. flextool/engine_polars/_native_run_model.py +1348 -0
  89. flextool/engine_polars/_orchestration.py +4314 -0
  90. flextool/engine_polars/_output_writer.py +439 -0
  91. flextool/engine_polars/_param_shapes.py +1595 -0
  92. flextool/engine_polars/_parquet_bundle.py +723 -0
  93. flextool/engine_polars/_pdt_join.py +167 -0
  94. flextool/engine_polars/_pdt_lookup.py +547 -0
  95. flextool/engine_polars/_per_solve_sets.py +335 -0
  96. flextool/engine_polars/_projection_params.py +2056 -0
  97. flextool/engine_polars/_provider_keys.py +173 -0
  98. flextool/engine_polars/_provider_translators.py +225 -0
  99. flextool/engine_polars/_recursive_solve.py +703 -0
  100. flextool/engine_polars/_region_filter.py +2508 -0
  101. flextool/engine_polars/_reserve.py +649 -0
  102. flextool/engine_polars/_solve_acceptance.py +331 -0
  103. flextool/engine_polars/_solve_config.py +1001 -0
  104. flextool/engine_polars/_solve_context.py +885 -0
  105. flextool/engine_polars/_solve_handoff.py +164 -0
  106. flextool/engine_polars/_solve_state.py +232 -0
  107. flextool/engine_polars/_solver_base.py +36 -0
  108. flextool/engine_polars/_solver_dispatch.py +511 -0
  109. flextool/engine_polars/_spinedb_reader.py +1165 -0
  110. flextool/engine_polars/_stochastic.py +593 -0
  111. flextool/engine_polars/_subprocess_solve.py +1838 -0
  112. flextool/engine_polars/_timeline.py +1416 -0
  113. flextool/engine_polars/_vectorize.py +438 -0
  114. flextool/engine_polars/_warm.py +858 -0
  115. flextool/engine_polars/autoscale/__init__.py +107 -0
  116. flextool/engine_polars/autoscale/_config.py +218 -0
  117. flextool/engine_polars/autoscale/_layer2.py +1253 -0
  118. flextool/engine_polars/autoscale/_layer2_types.py +584 -0
  119. flextool/engine_polars/autoscale/_quantity_types.py +621 -0
  120. flextool/engine_polars/autoscale/_report.py +336 -0
  121. flextool/engine_polars/chain.py +259 -0
  122. flextool/engine_polars/input.py +6638 -0
  123. flextool/engine_polars/model.py +4754 -0
  124. flextool/env_check.py +388 -0
  125. flextool/export_to_tabular/__init__.py +5 -0
  126. flextool/export_to_tabular/db_reader.py +224 -0
  127. flextool/export_to_tabular/excel_writer.py +3559 -0
  128. flextool/export_to_tabular/export_settings.yaml +377 -0
  129. flextool/export_to_tabular/export_to_excel.py +227 -0
  130. flextool/export_to_tabular/formatting.py +543 -0
  131. flextool/export_to_tabular/sheet_config.py +876 -0
  132. flextool/gui/__init__.py +0 -0
  133. flextool/gui/__main__.py +118 -0
  134. flextool/gui/calibrate_commands.py +184 -0
  135. flextool/gui/calibrate_jobs.py +424 -0
  136. flextool/gui/check_tree.py +142 -0
  137. flextool/gui/cli_format.py +83 -0
  138. flextool/gui/config_parser.py +68 -0
  139. flextool/gui/data_models.py +362 -0
  140. flextool/gui/db_editor_integration.py +202 -0
  141. flextool/gui/db_version_check.py +269 -0
  142. flextool/gui/dialogs/__init__.py +0 -0
  143. flextool/gui/dialogs/add_dialog.py +1098 -0
  144. flextool/gui/dialogs/calibrate_dialog.py +1259 -0
  145. flextool/gui/dialogs/file_picker.py +473 -0
  146. flextool/gui/dialogs/group_picker.py +299 -0
  147. flextool/gui/dialogs/migration_consent_dialog.py +106 -0
  148. flextool/gui/dialogs/migration_progress_dialog.py +237 -0
  149. flextool/gui/dialogs/plot_dialog.py +459 -0
  150. flextool/gui/dialogs/plot_settings_picker.py +2184 -0
  151. flextool/gui/dialogs/project_dialog.py +426 -0
  152. flextool/gui/dialogs/update_dialog.py +212 -0
  153. flextool/gui/downsampling.py +88 -0
  154. flextool/gui/error_handling.py +50 -0
  155. flextool/gui/execution_manager.py +1715 -0
  156. flextool/gui/execution_window.py +1377 -0
  157. flextool/gui/hover_tooltip.py +111 -0
  158. flextool/gui/input_sources.py +730 -0
  159. flextool/gui/main_window.py +6181 -0
  160. flextool/gui/network_graph.py +215 -0
  161. flextool/gui/output_actions.py +393 -0
  162. flextool/gui/output_log_window.py +159 -0
  163. flextool/gui/platform_utils.py +421 -0
  164. flextool/gui/plot_cache.py +88 -0
  165. flextool/gui/plot_canvas.py +543 -0
  166. flextool/gui/plot_config_reader.py +272 -0
  167. flextool/gui/project_utils.py +100 -0
  168. flextool/gui/result_viewer.py +4394 -0
  169. flextool/gui/scenario_key.py +162 -0
  170. flextool/gui/scenario_lists.py +516 -0
  171. flextool/gui/settings_io.py +360 -0
  172. flextool/gui/solve_reader.py +103 -0
  173. flextool/gui/tree_reorder.py +88 -0
  174. flextool/gui/ui_metrics.py +420 -0
  175. flextool/input_derivation/__init__.py +281 -0
  176. flextool/input_derivation/_commodity_ladder.py +375 -0
  177. flextool/input_derivation/_commodity_ladder_sets.py +70 -0
  178. flextool/input_derivation/_dc_power_flow.py +377 -0
  179. flextool/input_derivation/_method_constants.py +77 -0
  180. flextool/input_derivation/_process_method.py +258 -0
  181. flextool/input_derivation/_specs.py +1026 -0
  182. flextool/input_derivation/_validators.py +321 -0
  183. flextool/lean_parquet.py +159 -0
  184. flextool/model_builder/__init__.py +5 -0
  185. flextool/model_builder/build_model.py +589 -0
  186. flextool/model_builder/encoding.py +67 -0
  187. flextool/model_builder/names.py +34 -0
  188. flextool/model_builder/profiles.py +129 -0
  189. flextool/plot_outputs/__init__.py +14 -0
  190. flextool/plot_outputs/axis_helpers.py +355 -0
  191. flextool/plot_outputs/color_template.py +888 -0
  192. flextool/plot_outputs/config.py +171 -0
  193. flextool/plot_outputs/format_helpers.py +345 -0
  194. flextool/plot_outputs/legend_helpers.py +143 -0
  195. flextool/plot_outputs/orchestrator.py +1141 -0
  196. flextool/plot_outputs/perf.py +37 -0
  197. flextool/plot_outputs/plan.py +1787 -0
  198. flextool/plot_outputs/plot_bars.py +1510 -0
  199. flextool/plot_outputs/plot_bars_detail.py +753 -0
  200. flextool/plot_outputs/plot_lines.py +951 -0
  201. flextool/plot_outputs/shared_manifest.py +564 -0
  202. flextool/plot_outputs/subplot_helpers.py +137 -0
  203. flextool/process_inputs/__init__.py +188 -0
  204. flextool/process_inputs/import_old_excel_input.json +4159 -0
  205. flextool/process_inputs/read_matpower.py +451 -0
  206. flextool/process_inputs/read_old_flextool.py +1288 -0
  207. flextool/process_inputs/read_self_describing_excel.py +1423 -0
  208. flextool/process_inputs/read_tabular_with_specification.py +1114 -0
  209. flextool/process_inputs/write_old_flextool_to_db.py +3077 -0
  210. flextool/process_inputs/write_self_describing_to_db.py +977 -0
  211. flextool/process_inputs/write_to_input_db.py +269 -0
  212. flextool/process_outputs/__init__.py +7 -0
  213. flextool/process_outputs/_annualize.py +55 -0
  214. flextool/process_outputs/_inmemory_helpers.py +292 -0
  215. flextool/process_outputs/_output_meta.py +672 -0
  216. flextool/process_outputs/calc_capacity_flows.py +107 -0
  217. flextool/process_outputs/calc_connections.py +136 -0
  218. flextool/process_outputs/calc_costs.py +260 -0
  219. flextool/process_outputs/calc_group_flows.py +192 -0
  220. flextool/process_outputs/calc_slacks.py +103 -0
  221. flextool/process_outputs/calc_storage_vre.py +160 -0
  222. flextool/process_outputs/drop_levels.py +208 -0
  223. flextool/process_outputs/handoff_writers.py +1315 -0
  224. flextool/process_outputs/out_ancillary.py +544 -0
  225. flextool/process_outputs/out_capacity.py +179 -0
  226. flextool/process_outputs/out_costs.py +334 -0
  227. flextool/process_outputs/out_flowgroup.py +189 -0
  228. flextool/process_outputs/out_flows.py +301 -0
  229. flextool/process_outputs/out_group.py +475 -0
  230. flextool/process_outputs/out_node.py +190 -0
  231. flextool/process_outputs/persist_realized_slice.py +601 -0
  232. flextool/process_outputs/process_results.py +24 -0
  233. flextool/process_outputs/read_highs_solution.py +2256 -0
  234. flextool/process_outputs/read_parameters.py +1799 -0
  235. flextool/process_outputs/read_sets.py +1095 -0
  236. flextool/process_outputs/read_variables.py +553 -0
  237. flextool/process_outputs/solve_order.py +81 -0
  238. flextool/process_outputs/spinedb_replay.py +412 -0
  239. flextool/process_outputs/union_realized_slice.py +224 -0
  240. flextool/process_outputs/write_outputs.py +1286 -0
  241. flextool/process_outputs/write_spinedb.py +1267 -0
  242. flextool/representative_periods/__init__.py +5 -0
  243. flextool/representative_periods/clustering.py +165 -0
  244. flextool/representative_periods/force_include.py +563 -0
  245. flextool/representative_periods/netload.py +365 -0
  246. flextool/representative_periods/netload_inputs.py +345 -0
  247. flextool/representative_periods/netload_iterate.py +722 -0
  248. flextool/representative_periods/preprocess.py +948 -0
  249. flextool/representative_periods/scenario_stack.py +195 -0
  250. flextool/representative_periods/weights.py +124 -0
  251. flextool/scenario_comparison/__init__.py +13 -0
  252. flextool/scenario_comparison/config_builder.py +158 -0
  253. flextool/scenario_comparison/constants.py +20 -0
  254. flextool/scenario_comparison/data_models.py +222 -0
  255. flextool/scenario_comparison/db_reader.py +399 -0
  256. flextool/scenario_comparison/dispatch_data.py +1002 -0
  257. flextool/scenario_comparison/dispatch_mappings.py +205 -0
  258. flextool/scenario_comparison/dispatch_plots.py +691 -0
  259. flextool/scenario_comparison/input_entity_colors.py +319 -0
  260. flextool/scenario_comparison/orchestrator.py +453 -0
  261. flextool/scenario_comparison/plan_union.py +244 -0
  262. flextool/scenario_comparison/plot_settings_seed.py +205 -0
  263. flextool/schemas/AXIS_CONTRACT.md +71 -0
  264. flextool/schemas/canonical_databases/howto_aggregate_output.json +6225 -0
  265. flextool/schemas/canonical_databases/howto_connections.json +5606 -0
  266. flextool/schemas/canonical_databases/howto_demand.json +5518 -0
  267. flextool/schemas/canonical_databases/howto_hydro_reservoir.json +6239 -0
  268. flextool/schemas/canonical_databases/howto_hydro_reservoir_with_pump.json +5933 -0
  269. flextool/schemas/canonical_databases/howto_non_sync_and_curtailment.json +5794 -0
  270. flextool/schemas/canonical_databases/howto_ramp_and_start_up.json +5707 -0
  271. flextool/schemas/canonical_databases/howto_stochastics.json +6032 -0
  272. flextool/schemas/canonical_databases/templates_examples.json +13532 -0
  273. flextool/schemas/canonical_databases/templates_time_settings_only.json +5340 -0
  274. flextool/schemas/comparison_settings_template.json +197 -0
  275. flextool/schemas/default_plot_settings.yaml +260 -0
  276. flextool/schemas/default_plots.yaml +2293 -0
  277. flextool/schemas/flextool_axis_contract.json +303 -0
  278. flextool/schemas/flextool_axis_contract.schema.json +247 -0
  279. flextool/schemas/old_flextool_import_template.json +4443 -0
  280. flextool/schemas/output_info_template.json +48 -0
  281. flextool/schemas/output_settings_template.json +256 -0
  282. flextool/schemas/pre_v26/flextool_template_constant_default.json +2105 -0
  283. flextool/schemas/pre_v26/flextool_template_default_optional_output.json +2152 -0
  284. flextool/schemas/pre_v26/flextool_template_default_value.json +2094 -0
  285. flextool/schemas/pre_v26/flextool_template_drop_down.json +2080 -0
  286. flextool/schemas/pre_v26/flextool_template_lifetime_method.json +1990 -0
  287. flextool/schemas/pre_v26/flextool_template_optional_outputs.json +2094 -0
  288. flextool/schemas/pre_v26/flextool_template_output_node_flows.json +2105 -0
  289. flextool/schemas/pre_v26/flextool_template_results_master.json +493 -0
  290. flextool/schemas/pre_v26/flextool_template_rolling_start_remove.json +2087 -0
  291. flextool/schemas/pre_v26/flextool_template_rolling_window.json +2059 -0
  292. flextool/schemas/pre_v26/flextool_template_storage_binding_defaults.json +46 -0
  293. flextool/schemas/pre_v26/flextool_template_v2.json +1990 -0
  294. flextool/schemas/pre_v26/flextool_template_v25.json +3864 -0
  295. flextool/schemas/spinedb_results_schema.json +581 -0
  296. flextool/schemas/spinedb_schema.json +4636 -0
  297. flextool/solver_config/copt.opt.template +18 -0
  298. flextool/solver_config/cplex.opt.template +25 -0
  299. flextool/solver_config/gurobi.opt.template +18 -0
  300. flextool/solver_config/highs.opt.template +18 -0
  301. flextool/solver_config/xpress.opt.template +26 -0
  302. flextool/spinedb_backend/__init__.py +26 -0
  303. flextool/spinedb_backend/_axis_enums.py +1119 -0
  304. flextool/spinedb_backend/_backend.py +1139 -0
  305. flextool/update_flextool/__init__.py +12 -0
  306. flextool/update_flextool/canonical_databases.py +251 -0
  307. flextool/update_flextool/db_migration.py +7108 -0
  308. flextool/update_flextool/ensure_settings_db.py +138 -0
  309. flextool/update_flextool/export_database.py +103 -0
  310. flextool/update_flextool/extend_tests_fixture.py +772 -0
  311. flextool/update_flextool/generate_canonical.py +274 -0
  312. flextool/update_flextool/initialize_database.py +42 -0
  313. flextool/update_flextool/install_info.py +225 -0
  314. flextool/update_flextool/self_update.py +464 -0
  315. flextool/update_flextool/sync_master_json_template.py +125 -0
  316. flextool/update_flextool/test_fixtures.py +187 -0
  317. flextool-4.0.0.dist-info/METADATA +217 -0
  318. flextool-4.0.0.dist-info/RECORD +322 -0
  319. flextool-4.0.0.dist-info/WHEEL +5 -0
  320. flextool-4.0.0.dist-info/entry_points.txt +17 -0
  321. flextool-4.0.0.dist-info/licenses/LICENSE.txt +19 -0
  322. flextool-4.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,729 @@
1
+ """Calculated-param + process-method families.
2
+
3
+ Two emitters:
4
+
5
+ * ``entity_total_caps`` — 4 e_*_total params keyed on
6
+ entityInvest / entityDivest, summed across p_process[e, param] and
7
+ p_node[e, param]; **MPS-precision-sensitive** (emits
8
+ ``repr(float(v))``, so we pre-stringify the value column with the
9
+ same expression and then ``write_csv``).
10
+ * ``process_method_sets`` — process-method-driven derived sets: 3
11
+ method-enum projections, ``process_VRE``, the ``process_*_to_*``
12
+ family of 10 method-gated cross-products, and 2 profile-method joins.
13
+
14
+ Each ``derive_*`` returns a fresh ``pl.DataFrame`` (in-memory
15
+ contract); ``write_*`` wrappers materialise the frame to the
16
+ ``solve_data/*.csv`` path.
17
+
18
+ Style mirrors :mod:`._emit_leaf_sets` / :mod:`._emit_mid_sets`:
19
+ eager polars reads of tiny CSVs, expression chains,
20
+ ``unique(maintain_order=True)`` for ordered dedup.
21
+
22
+ Precision-parity pattern (calculated-param families)
23
+ ----------------------------------------------------
24
+
25
+ ``entity_total_caps`` writes ``f"{key},{repr(float(v))}\\n"`` per row.
26
+ ``repr(float)`` is round-trip-exact. Polars' default ``write_csv``
27
+ float formatting does *not* match ``repr`` for all values (it strips
28
+ trailing zeros differently for some doubles), so we explicitly
29
+ pre-stringify the value column with the same ``repr(float(v))`` step
30
+ before writing. Verified byte-identical against fixtures with
31
+ explicit float values.
32
+ """
33
+ from __future__ import annotations
34
+
35
+ from pathlib import Path
36
+
37
+ import polars as pl
38
+
39
+ from ._axis_enums import schema_dtype
40
+ from ._emit_provider_io import _emit
41
+
42
+
43
+ # ---------------------------------------------------------------------------
44
+ # CSV I/O — same conventions as _emit_{leaf,mid}_sets:
45
+ # * eager read, missing file → empty frame with requested schema
46
+ # * positional column rename (handle legacy headers that differ in label)
47
+ # * empty frame still writes header line
48
+ # ---------------------------------------------------------------------------
49
+
50
+ def _read_csv(path: Path, columns: list[str],
51
+ *, provider: "object | None" = None) -> pl.DataFrame:
52
+ """Read a tiny flextool CSV via the Provider.
53
+
54
+ Returns the Provider's frame sliced to *columns*; returns an empty
55
+ all-``Utf8`` frame when the Provider misses the key. Step 2.5
56
+ Phase C dropped the disk-fallback arm.
57
+ """
58
+ from flextool.engine_polars._emit_provider_io import (
59
+ _provider_key,
60
+ _provider_lookup_positional,
61
+ )
62
+ seeded = _provider_lookup_positional(
63
+ provider, _provider_key(path), path, columns,
64
+ )
65
+ if seeded is not None:
66
+ return seeded
67
+ return pl.DataFrame(
68
+ {c: [] for c in columns},
69
+ schema={c: schema_dtype(None, c) for c in columns},
70
+ )
71
+
72
+
73
+ def _drop_blank_rows(df: pl.DataFrame, required_cols: list[str]) -> pl.DataFrame:
74
+ expr = pl.col(required_cols[0]) != ""
75
+ for c in required_cols[1:]:
76
+ expr = expr & (pl.col(c) != "")
77
+ return df.filter(expr)
78
+
79
+
80
+ # ===========================================================================
81
+ # Family 13 — entity_total_caps (legacy: preprocessing/entity_total_caps.py)
82
+ # ===========================================================================
83
+
84
+ # (output filename, source-keys CSV, p_*-table param name)
85
+ _ENTITY_TOTAL_SPEC: list[tuple[str, str, str]] = [
86
+ ("e_invest_max_total.csv", "entityInvest.csv", "invest_max_total"),
87
+ ("e_divest_max_total.csv", "entityDivest.csv", "retire_max_total"),
88
+ ("e_invest_min_total.csv", "entityInvest.csv", "invest_min_total"),
89
+ ("e_divest_min_total.csv", "entityDivest.csv", "retire_min_total"),
90
+ ]
91
+
92
+
93
+ def _read_param_lookup(path: Path,
94
+ *, provider: "object | None" = None,
95
+ ) -> dict[tuple[str, str], float]:
96
+ """Read a 3-col (entity, paramName, value) CSV into a python dict.
97
+
98
+ Mirrors ``entity_total_caps._read_param_table``: silently skip
99
+ rows whose value isn't parseable as float. Keeping a python
100
+ dict (rather than a polars frame) keeps the per-entity summation
101
+ below straightforward — input tables are tiny.
102
+ """
103
+ df = _read_csv(path, ["entity", "paramName", "value"], provider=provider)
104
+ out: dict[tuple[str, str], float] = {}
105
+ for e, p, v in df.iter_rows():
106
+ if not e or not p:
107
+ continue
108
+ try:
109
+ out[(e, p)] = float(v)
110
+ except (TypeError, ValueError):
111
+ continue
112
+ return out
113
+
114
+
115
+ def derive_entity_total_cap(
116
+ keys: list[str],
117
+ process_set: frozenset[str],
118
+ node_set: frozenset[str],
119
+ p_process: dict[tuple[str, str], float],
120
+ p_node: dict[tuple[str, str], float],
121
+ param_name: str,
122
+ ) -> pl.DataFrame:
123
+ """Compute a single ``(entity, value)`` table for one e_*_total param.
124
+
125
+ ``value`` = ``p_process[e, param_name]`` (if e ∈ process, default 0)
126
+ + ``p_node[e, param_name]`` (if e ∈ node, default 0).
127
+
128
+ Returns a 2-column frame with both columns as ``pl.Utf8`` —
129
+ ``value`` is pre-stringified with ``repr(float(v))`` to preserve
130
+ bit-exact MPS-precision parity with legacy code. See module
131
+ docstring for the precision-parity rationale.
132
+ """
133
+ entities: list[str] = []
134
+ values: list[str] = []
135
+ for e in keys:
136
+ v = 0.0
137
+ if e in process_set:
138
+ v += p_process.get((e, param_name), 0.0)
139
+ if e in node_set:
140
+ v += p_node.get((e, param_name), 0.0)
141
+ entities.append(e)
142
+ values.append(repr(v))
143
+ return pl.DataFrame(
144
+ {"entity": entities, "value": values},
145
+ schema={"entity": pl.Utf8, "value": pl.Utf8},
146
+ )
147
+
148
+
149
+ def emit_entity_total_caps(input_dir: Path, solve_data_dir: Path,
150
+ *, provider) -> None:
151
+ """Emit ``entity_total_caps`` to the Provider."""
152
+ process_set = frozenset(
153
+ _drop_blank_rows(_read_csv(input_dir / "process.csv", ["process"],
154
+ provider=provider), ["process"])
155
+ .get_column("process").to_list()
156
+ )
157
+ node_set = frozenset(
158
+ _drop_blank_rows(_read_csv(input_dir / "node.csv", ["node"],
159
+ provider=provider), ["node"])
160
+ .get_column("node").to_list()
161
+ )
162
+ p_process = _read_param_lookup(input_dir / "p_process.csv",
163
+ provider=provider)
164
+ p_node = _read_param_lookup(input_dir / "p_node.csv", provider=provider)
165
+
166
+ key_cache: dict[str, list[str]] = {}
167
+ for _, src, _ in _ENTITY_TOTAL_SPEC:
168
+ if src not in key_cache:
169
+ df = _read_csv(solve_data_dir / src, ["entity"], provider=provider)
170
+ df = _drop_blank_rows(df, ["entity"])
171
+ key_cache[src] = df.get_column("entity").to_list()
172
+
173
+ for fname, src, param in _ENTITY_TOTAL_SPEC:
174
+ out = derive_entity_total_cap(
175
+ key_cache[src], process_set, node_set,
176
+ p_process, p_node, param,
177
+ )
178
+ _emit(provider, f"solve_data/{fname}", out)
179
+
180
+
181
+ # ===========================================================================
182
+ # Family 14 — process_method_sets (legacy: preprocessing/process_method_sets.py)
183
+ # ===========================================================================
184
+
185
+ # Method-enum subsets mirrored from
186
+ # :mod:`flextool.input_derivation._method_constants`. Pinned here as
187
+ # frozensets so the module has no transitive import. If the source
188
+ # constants change, update both sites in lockstep (the parity tests
189
+ # would catch drift).
190
+
191
+ _METHOD_LP: frozenset[str] = frozenset((
192
+ "method_1way_1var_LP", "method_1way_nvar_LP",
193
+ ))
194
+ _METHOD_MIP: frozenset[str] = frozenset((
195
+ "method_1way_1var_MIP", "method_1way_nvar_MIP",
196
+ "method_2way_2var_MIP_exclude",
197
+ ))
198
+ _METHOD_INDIRECT: frozenset[str] = frozenset((
199
+ "method_1way_nvar_off", "method_1way_nvar_LP",
200
+ "method_1way_nvar_MIP", "method_2way_nvar_off",
201
+ ))
202
+ _METHOD_DIRECT: frozenset[str] = frozenset((
203
+ "method_1way_1var_off", "method_1way_1var_LP", "method_1way_1var_MIP",
204
+ "method_2way_1var_off", "method_2way_2var_off",
205
+ "method_2way_2var_exclude", "method_2way_2var_MIP_exclude",
206
+ ))
207
+ _METHOD_2WAY_1VAR: frozenset[str] = frozenset(("method_2way_1var_off",))
208
+ _METHOD_2WAY_2VAR: frozenset[str] = frozenset((
209
+ "method_2way_2var_off", "method_2way_2var_exclude",
210
+ "method_2way_2var_MIP_exclude",
211
+ ))
212
+ _METHOD_2WAY_NVAR: frozenset[str] = frozenset(("method_2way_nvar_off",))
213
+ _METHOD_1WAY_1VAR: frozenset[str] = frozenset((
214
+ "method_1way_1var_off", "method_1way_1var_LP", "method_1way_1var_MIP",
215
+ ))
216
+
217
+
218
+ # ---- process-method projections (mod L1121-L1194) -------------------------
219
+
220
+ def derive_process_online_linear(input_dir: Path,
221
+ *, provider: "object | None" = None,
222
+ ) -> pl.DataFrame:
223
+ """``process_online_linear`` = projection of process_method onto
224
+ rows whose method ∈ METHOD_LP."""
225
+ pm = _read_csv(input_dir / "process_method.csv", ["process", "method"],
226
+ provider=provider)
227
+ pm = _drop_blank_rows(pm, ["process", "method"])
228
+ return (
229
+ pm.filter(pl.col("method").is_in(list(_METHOD_LP)))
230
+ .select("process")
231
+ .unique(maintain_order=True)
232
+ )
233
+
234
+
235
+ def derive_process_online_integer(input_dir: Path,
236
+ *, provider: "object | None" = None,
237
+ ) -> pl.DataFrame:
238
+ """``process_online_integer`` = METHOD_MIP filter on process_method."""
239
+ pm = _read_csv(input_dir / "process_method.csv", ["process", "method"],
240
+ provider=provider)
241
+ pm = _drop_blank_rows(pm, ["process", "method"])
242
+ return (
243
+ pm.filter(pl.col("method").is_in(list(_METHOD_MIP)))
244
+ .select("process")
245
+ .unique(maintain_order=True)
246
+ )
247
+
248
+
249
+ def derive_process_method_indirect(input_dir: Path,
250
+ *, provider: "object | None" = None,
251
+ ) -> pl.DataFrame:
252
+ """``process__method_indirect`` = METHOD_INDIRECT filter, both columns kept."""
253
+ pm = _read_csv(input_dir / "process_method.csv", ["process", "method"],
254
+ provider=provider)
255
+ pm = _drop_blank_rows(pm, ["process", "method"])
256
+ return (
257
+ pm.filter(pl.col("method").is_in(list(_METHOD_INDIRECT)))
258
+ .unique(maintain_order=True)
259
+ )
260
+
261
+
262
+ def emit_process_method_projections(input_dir: Path,
263
+ *, provider) -> None:
264
+ """Emit ``process_online_linear``/``_integer``/``__method_indirect``."""
265
+ _emit(provider, "solve_data/process_online_linear.csv",
266
+ derive_process_online_linear(input_dir, provider=provider))
267
+ _emit(provider, "solve_data/process_online_integer.csv",
268
+ derive_process_online_integer(input_dir, provider=provider))
269
+ _emit(provider, "solve_data/process__method_indirect.csv",
270
+ derive_process_method_indirect(input_dir, provider=provider))
271
+
272
+
273
+ # ---- process_VRE (mod L2248) ----------------------------------------------
274
+
275
+ def derive_process_VRE(input_dir: Path,
276
+ *, provider: "object | None" = None,
277
+ ) -> pl.DataFrame:
278
+ """``process_VRE`` = process_unit ∩ no-source ∩ has-upper-limit-profile.
279
+
280
+ flextool.mod:2248 — VRE units have no source arc (free-energy
281
+ primary input) and at least one ``upper_limit`` profile method.
282
+ """
283
+ units = _read_csv(input_dir / "process_unit.csv", ["process"],
284
+ provider=provider)
285
+ units = _drop_blank_rows(units, ["process"])
286
+ sources = _read_csv(input_dir / "process__source.csv",
287
+ ["process", "source"], provider=provider)
288
+ sources = _drop_blank_rows(sources, ["process", "source"])
289
+ profiles = _read_csv(
290
+ input_dir / "process__node__profile__profile_method.csv",
291
+ ["process", "node", "profile", "profile_method"],
292
+ provider=provider,
293
+ )
294
+ profiles = _drop_blank_rows(
295
+ profiles, ["process", "node", "profile", "profile_method"],
296
+ )
297
+ has_source = frozenset(sources.get_column("process").to_list())
298
+ has_upper_limit = frozenset(
299
+ profiles.filter(pl.col("profile_method") == "upper_limit")
300
+ .get_column("process").to_list()
301
+ )
302
+ return (
303
+ units.filter(
304
+ ~pl.col("process").is_in(list(has_source))
305
+ & pl.col("process").is_in(list(has_upper_limit))
306
+ )
307
+ .select("process")
308
+ .unique(maintain_order=True)
309
+ )
310
+
311
+
312
+ def emit_process_VRE(input_dir: Path,
313
+ *, provider) -> None:
314
+ """Emit ``process_VRE`` = process_unit ∩ no-source ∩ has-upper-limit-profile."""
315
+ _emit(provider, "solve_data/process_VRE.csv",
316
+ derive_process_VRE(input_dir, provider=provider))
317
+
318
+
319
+ # ---- process_*_to_* family (mod L993-L1052) -------------------------------
320
+ #
321
+ # Each entry below is a method-enum-existence join. We iterate a base
322
+ # 2-tuple set (process_source, process_sink, or process) and admit rows
323
+ # whose process has at least one method in a specific enum subset.
324
+ #
325
+ # Output shape is dimen-3: (process_outer, process, source/sink) or
326
+ # (process, source/sink, process_aux). The legacy module fixes the
327
+ # header column names per output — we mirror those exactly.
328
+ # ---------------------------------------------------------------------------
329
+
330
+
331
+ def _processes_with_method_in(
332
+ pm: pl.DataFrame, allowed: frozenset[str],
333
+ ) -> frozenset[str]:
334
+ return frozenset(
335
+ pm.filter(pl.col("method").is_in(list(allowed)))
336
+ .get_column("process").to_list()
337
+ )
338
+
339
+
340
+ def _build_arc_map(arc_df: pl.DataFrame, key: str, value: str) -> dict[str, list[str]]:
341
+ """Group an arc CSV by ``key`` preserving CSV order."""
342
+ out: dict[str, list[str]] = {}
343
+ for k, v in arc_df.select(key, value).iter_rows():
344
+ out.setdefault(k, []).append(v)
345
+ return out
346
+
347
+
348
+ def _arc_method_inputs(input_dir: Path,
349
+ *, provider: "object | None" = None,
350
+ ) -> dict:
351
+ """Shared scan for the 10 process_*_to_* derives.
352
+
353
+ Returns the bundle of in-memory sets/lists used by every
354
+ ``derive_process_*`` below. Each public derive_X also calls this
355
+ helper directly so it remains standalone for accumulator capture.
356
+ """
357
+ pm = _drop_blank_rows(
358
+ _read_csv(input_dir / "process_method.csv", ["process", "method"],
359
+ provider=provider),
360
+ ["process", "method"],
361
+ )
362
+ sources = _drop_blank_rows(
363
+ _read_csv(input_dir / "process__source.csv", ["process", "source"],
364
+ provider=provider),
365
+ ["process", "source"],
366
+ )
367
+ sinks = _drop_blank_rows(
368
+ _read_csv(input_dir / "process__sink.csv", ["process", "sink"],
369
+ provider=provider),
370
+ ["process", "sink"],
371
+ )
372
+ processes = _drop_blank_rows(
373
+ _read_csv(input_dir / "process.csv", ["process"], provider=provider),
374
+ ["process"],
375
+ ).get_column("process").to_list()
376
+
377
+ has_source = frozenset(sources.get_column("process").to_list())
378
+ has_sink = frozenset(sinks.get_column("process").to_list())
379
+
380
+ return {
381
+ "pm": pm,
382
+ "sources": sources,
383
+ "sinks": sinks,
384
+ "processes": processes,
385
+ "p_with_2way_nvar": _processes_with_method_in(pm, _METHOD_2WAY_NVAR),
386
+ "p_with_direct": _processes_with_method_in(pm, _METHOD_DIRECT),
387
+ "p_with_2way_2var": _processes_with_method_in(pm, _METHOD_2WAY_2VAR),
388
+ "p_with_1way_1var": _processes_with_method_in(pm, _METHOD_1WAY_1VAR),
389
+ "has_source": has_source,
390
+ "has_sink": has_sink,
391
+ "process_no_source": frozenset(p for p in processes if p not in has_source),
392
+ "process_no_sink": frozenset(p for p in processes if p not in has_sink),
393
+ "sinks_by_process": _build_arc_map(sinks, "process", "sink"),
394
+ "sources_by_process": _build_arc_map(sources, "process", "source"),
395
+ "sink_rows": list(sinks.iter_rows()),
396
+ "source_rows": list(sources.iter_rows()),
397
+ }
398
+
399
+
400
+ def _to_frame(rows: list[tuple[str, ...]],
401
+ header: tuple[str, ...]) -> pl.DataFrame:
402
+ """Dedup + materialise to a pl.DataFrame with all-Utf8 columns."""
403
+ deduped = list(dict.fromkeys(rows))
404
+ cols = {h: [r[i] for r in deduped] for i, h in enumerate(header)}
405
+ return pl.DataFrame(cols, schema={h: pl.Utf8 for h in header})
406
+
407
+
408
+ # ---- 10 derive_X for the process_*_to_* family ----
409
+
410
+ def derive_process_sink_toProcess(input_dir: Path,
411
+ *, provider: "object | None" = None,
412
+ ) -> pl.DataFrame:
413
+ """``process_sink_toProcess`` — METHOD_2WAY_NVAR filter on (p, sink)."""
414
+ inp = _arc_method_inputs(input_dir, provider=provider)
415
+ rows = [
416
+ (p, sink, p)
417
+ for p, sink in inp["sink_rows"]
418
+ if p in inp["p_with_2way_nvar"]
419
+ ]
420
+ return _to_frame(rows, ("process", "sink", "process_aux"))
421
+
422
+
423
+ def derive_process_process_toSource(input_dir: Path,
424
+ *, provider: "object | None" = None,
425
+ ) -> pl.DataFrame:
426
+ """``process_process_toSource`` — METHOD_2WAY_NVAR filter on (p, source)."""
427
+ inp = _arc_method_inputs(input_dir, provider=provider)
428
+ rows = [
429
+ (p, p, source)
430
+ for p, source in inp["source_rows"]
431
+ if p in inp["p_with_2way_nvar"]
432
+ ]
433
+ return _to_frame(rows, ("process_outer", "process", "source"))
434
+
435
+
436
+ def derive_process_source_toSink(input_dir: Path,
437
+ *, provider: "object | None" = None,
438
+ ) -> pl.DataFrame:
439
+ """``process_source_toSink`` — METHOD_DIRECT cross-product
440
+ of source rows and sinks_by_process."""
441
+ inp = _arc_method_inputs(input_dir, provider=provider)
442
+ rows = [
443
+ (p, source, sink)
444
+ for p, source in inp["source_rows"]
445
+ if p in inp["p_with_direct"]
446
+ for sink in inp["sinks_by_process"].get(p, ())
447
+ ]
448
+ return _to_frame(rows, ("process", "source", "sink"))
449
+
450
+
451
+ def derive_process_source_toProcess_direct(input_dir: Path,
452
+ *, provider: "object | None" = None,
453
+ ) -> pl.DataFrame:
454
+ """``process_source_toProcess_direct`` — METHOD_DIRECT on (p, source)."""
455
+ inp = _arc_method_inputs(input_dir, provider=provider)
456
+ rows = [
457
+ (p, source, p)
458
+ for p, source in inp["source_rows"]
459
+ if p in inp["p_with_direct"]
460
+ ]
461
+ return _to_frame(rows, ("process", "source", "process_aux"))
462
+
463
+
464
+ def derive_process_process_toSink_direct(input_dir: Path,
465
+ *, provider: "object | None" = None,
466
+ ) -> pl.DataFrame:
467
+ """``process_process_toSink_direct`` — METHOD_DIRECT on (p, sink)."""
468
+ inp = _arc_method_inputs(input_dir, provider=provider)
469
+ rows = [
470
+ (p, p, sink)
471
+ for p, sink in inp["sink_rows"]
472
+ if p in inp["p_with_direct"]
473
+ ]
474
+ return _to_frame(rows, ("process_outer", "process", "sink"))
475
+
476
+
477
+ def derive_process_sink_toProcess_direct(input_dir: Path,
478
+ *, provider: "object | None" = None,
479
+ ) -> pl.DataFrame:
480
+ """``process_sink_toProcess_direct`` — METHOD_2WAY_2VAR on (p, sink)."""
481
+ inp = _arc_method_inputs(input_dir, provider=provider)
482
+ rows = [
483
+ (p, sink, p)
484
+ for p, sink in inp["sink_rows"]
485
+ if p in inp["p_with_2way_2var"]
486
+ ]
487
+ return _to_frame(rows, ("process", "sink", "process_aux"))
488
+
489
+
490
+ def derive_process_sink_toSource(input_dir: Path,
491
+ *, provider: "object | None" = None,
492
+ ) -> pl.DataFrame:
493
+ """``process_sink_toSource`` — METHOD_2WAY_2VAR cross-product
494
+ of sink rows and sources_by_process."""
495
+ inp = _arc_method_inputs(input_dir, provider=provider)
496
+ rows = [
497
+ (p, sink, source)
498
+ for p, sink in inp["sink_rows"]
499
+ if p in inp["p_with_2way_2var"]
500
+ for source in inp["sources_by_process"].get(p, ())
501
+ ]
502
+ return _to_frame(rows, ("process", "sink", "source"))
503
+
504
+
505
+ def derive_process_process_toSink_noConversion(input_dir: Path,
506
+ *, provider: "object | None" = None,
507
+ ) -> pl.DataFrame:
508
+ """``process_process_toSink_noConversion`` — METHOD_1WAY_1VAR ∧ no source."""
509
+ inp = _arc_method_inputs(input_dir, provider=provider)
510
+ rows = [
511
+ (p, p, sink)
512
+ for p, sink in inp["sink_rows"]
513
+ if p in inp["p_with_1way_1var"] and p in inp["process_no_source"]
514
+ ]
515
+ return _to_frame(rows, ("process_outer", "process", "sink"))
516
+
517
+
518
+ def derive_process_source_toProcess_noConversion(input_dir: Path,
519
+ *, provider: "object | None" = None,
520
+ ) -> pl.DataFrame:
521
+ """``process_source_toProcess_noConversion`` — METHOD_1WAY_1VAR ∧ no sink."""
522
+ inp = _arc_method_inputs(input_dir, provider=provider)
523
+ rows = [
524
+ (p, source, p)
525
+ for p, source in inp["source_rows"]
526
+ if p in inp["p_with_1way_1var"] and p in inp["process_no_sink"]
527
+ ]
528
+ return _to_frame(rows, ("process", "source", "process_aux"))
529
+
530
+
531
+ def derive_process_process_toSource_direct(input_dir: Path,
532
+ *, provider: "object | None" = None,
533
+ ) -> pl.DataFrame:
534
+ """``process_process_toSource_direct`` — METHOD_2WAY_2VAR on (p, source)."""
535
+ inp = _arc_method_inputs(input_dir, provider=provider)
536
+ rows = [
537
+ (p, p, source)
538
+ for p, source in inp["source_rows"]
539
+ if p in inp["p_with_2way_2var"]
540
+ ]
541
+ return _to_frame(rows, ("process_outer", "process", "source"))
542
+
543
+
544
+ def emit_process_arc_method_joins(input_dir: Path,
545
+ *, provider) -> None:
546
+ """Emit the three process×arc method-join derivations."""
547
+ _emit(provider, "solve_data/process_sink_toProcess.csv",
548
+ derive_process_sink_toProcess(input_dir, provider=provider))
549
+ _emit(provider, "solve_data/process_process_toSource.csv",
550
+ derive_process_process_toSource(input_dir, provider=provider))
551
+ _emit(provider, "solve_data/process_source_toSink.csv",
552
+ derive_process_source_toSink(input_dir, provider=provider))
553
+ _emit(provider, "solve_data/process_source_toProcess_direct.csv",
554
+ derive_process_source_toProcess_direct(input_dir,
555
+ provider=provider))
556
+ _emit(provider, "solve_data/process_process_toSink_direct.csv",
557
+ derive_process_process_toSink_direct(input_dir, provider=provider))
558
+ _emit(provider, "solve_data/process_sink_toProcess_direct.csv",
559
+ derive_process_sink_toProcess_direct(input_dir, provider=provider))
560
+ _emit(provider, "solve_data/process_sink_toSource.csv",
561
+ derive_process_sink_toSource(input_dir, provider=provider))
562
+ _emit(provider, "solve_data/process_process_toSink_noConversion.csv",
563
+ derive_process_process_toSink_noConversion(input_dir,
564
+ provider=provider))
565
+ _emit(provider, "solve_data/process_source_toProcess_noConversion.csv",
566
+ derive_process_source_toProcess_noConversion(input_dir,
567
+ provider=provider))
568
+ _emit(provider, "solve_data/process_process_toSource_direct.csv",
569
+ derive_process_process_toSource_direct(input_dir,
570
+ provider=provider))
571
+
572
+
573
+ # ---- profile-method joins (mod L961, L969) --------------------------------
574
+
575
+ def _profile_method_inputs(input_dir: Path,
576
+ *, provider: "object | None" = None,
577
+ ) -> tuple[
578
+ pl.DataFrame, pl.DataFrame, pl.DataFrame, pl.DataFrame, list[str],
579
+ frozenset[str], frozenset[str], frozenset[str],
580
+ dict[str, set[str]], dict[str, set[str]],
581
+ list[tuple[str, str, str, str]],
582
+ ]:
583
+ """Shared scan for both profile-method join derives.
584
+
585
+ Each public ``derive_*`` calls this so it is standalone for
586
+ accumulator capture; the wrapper :func:`write_process_profile_method_joins`
587
+ calls it once and feeds the same scan to both derives.
588
+ """
589
+ pm = _drop_blank_rows(
590
+ _read_csv(input_dir / "process_method.csv", ["process", "method"],
591
+ provider=provider),
592
+ ["process", "method"],
593
+ )
594
+ sources = _drop_blank_rows(
595
+ _read_csv(input_dir / "process__source.csv", ["process", "source"],
596
+ provider=provider),
597
+ ["process", "source"],
598
+ )
599
+ sinks = _drop_blank_rows(
600
+ _read_csv(input_dir / "process__sink.csv", ["process", "sink"],
601
+ provider=provider),
602
+ ["process", "sink"],
603
+ )
604
+ profiles = _drop_blank_rows(
605
+ _read_csv(
606
+ input_dir / "process__node__profile__profile_method.csv",
607
+ ["process", "node", "profile", "profile_method"],
608
+ provider=provider,
609
+ ),
610
+ ["process", "node", "profile", "profile_method"],
611
+ )
612
+ processes = _drop_blank_rows(
613
+ _read_csv(input_dir / "process.csv", ["process"], provider=provider),
614
+ ["process"],
615
+ ).get_column("process").to_list()
616
+
617
+ p_with_indirect = _processes_with_method_in(pm, _METHOD_INDIRECT)
618
+ has_sources = frozenset(sources.get_column("process").to_list())
619
+ has_sinks = frozenset(sinks.get_column("process").to_list())
620
+
621
+ sinks_by_process: dict[str, set[str]] = {}
622
+ for p, n in sinks.iter_rows():
623
+ sinks_by_process.setdefault(p, set()).add(n)
624
+ sources_by_process: dict[str, set[str]] = {}
625
+ for p, n in sources.iter_rows():
626
+ sources_by_process.setdefault(p, set()).add(n)
627
+
628
+ profiles_rows = list(profiles.iter_rows()) # (p, n, f, fm) tuples
629
+
630
+ return (
631
+ pm, sources, sinks, profiles, processes,
632
+ p_with_indirect, has_sources, has_sinks,
633
+ sinks_by_process, sources_by_process,
634
+ profiles_rows,
635
+ )
636
+
637
+
638
+ def derive_process_profileProcess_toSink_profile_profile_method(
639
+ input_dir: Path,
640
+ *, provider: "object | None" = None,
641
+ ) -> pl.DataFrame:
642
+ """``process__profileProcess__toSink__profile__profile_method``:
643
+ join profile rows against (process, sink) arcs, gated by
644
+ "process has any indirect method OR process has no source rows".
645
+
646
+ Iteration order mirrors the legacy module exactly so first-seen
647
+ dedup preserves the legacy CSV row order.
648
+ """
649
+ (_pm, _sources, _sinks, _profiles, processes,
650
+ p_with_indirect, has_sources, _has_sinks,
651
+ sinks_by_process, _sources_by_process,
652
+ profiles_rows) = _profile_method_inputs(input_dir, provider=provider)
653
+
654
+ rows_to_sink: list[tuple[str, str, str, str, str]] = []
655
+ for p in processes:
656
+ if not (p in p_with_indirect or p not in has_sources):
657
+ continue
658
+ psinks = sinks_by_process.get(p, set())
659
+ for p2, n, f, fm in profiles_rows:
660
+ if p2 == p and n in psinks:
661
+ rows_to_sink.append((p, p2, n, f, fm))
662
+ deduped = list(dict.fromkeys(rows_to_sink))
663
+ return pl.DataFrame(
664
+ {
665
+ "process_outer": [r[0] for r in deduped],
666
+ "process": [r[1] for r in deduped],
667
+ "sink": [r[2] for r in deduped],
668
+ "profile": [r[3] for r in deduped],
669
+ "profile_method": [r[4] for r in deduped],
670
+ },
671
+ schema={
672
+ "process_outer": pl.Utf8, "process": pl.Utf8, "sink": pl.Utf8,
673
+ "profile": pl.Utf8, "profile_method": pl.Utf8,
674
+ },
675
+ )
676
+
677
+
678
+ def derive_process_source_toProfileProcess_profile_profile_method(
679
+ input_dir: Path,
680
+ *, provider: "object | None" = None,
681
+ ) -> pl.DataFrame:
682
+ """``process__source__toProfileProcess__profile__profile_method``:
683
+ join profile rows against (process, source) arcs, gated by
684
+ "process has any indirect method OR process has no sink rows"."""
685
+ (_pm, sources, _sinks, _profiles, _processes,
686
+ p_with_indirect, _has_sources, has_sinks,
687
+ _sinks_by_process, _sources_by_process,
688
+ profiles_rows) = _profile_method_inputs(input_dir, provider=provider)
689
+
690
+ rows_to_source: list[tuple[str, str, str, str, str]] = []
691
+ for p, source in sources.iter_rows():
692
+ if not (p in p_with_indirect or p not in has_sinks):
693
+ continue
694
+ for p2, src2, f, fm in profiles_rows:
695
+ if p2 == p and src2 == source:
696
+ rows_to_source.append((p, source, p2, f, fm))
697
+ deduped = list(dict.fromkeys(rows_to_source))
698
+ return pl.DataFrame(
699
+ {
700
+ "process": [r[0] for r in deduped],
701
+ "source": [r[1] for r in deduped],
702
+ "process_aux": [r[2] for r in deduped],
703
+ "profile": [r[3] for r in deduped],
704
+ "profile_method": [r[4] for r in deduped],
705
+ },
706
+ schema={
707
+ "process": pl.Utf8, "source": pl.Utf8, "process_aux": pl.Utf8,
708
+ "profile": pl.Utf8, "profile_method": pl.Utf8,
709
+ },
710
+ )
711
+
712
+
713
+ def emit_process_profile_method_joins(
714
+ input_dir: Path,
715
+ *, provider,
716
+ ) -> None:
717
+ """Emit the process×profile-method join frames."""
718
+ _emit(
719
+ provider,
720
+ "solve_data/process__profileProcess__toSink__profile__profile_method.csv",
721
+ derive_process_profileProcess_toSink_profile_profile_method(
722
+ input_dir, provider=provider),
723
+ )
724
+ _emit(
725
+ provider,
726
+ "solve_data/process__source__toProfileProcess__profile__profile_method.csv",
727
+ derive_process_source_toProfileProcess_profile_profile_method(
728
+ input_dir, provider=provider),
729
+ )