ado-cplex-mip 1.0.3__py2.py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ Metadata-Version: 2.4
2
+ Name: ado-cplex-mip
3
+ Version: 1.0.3
4
+ Summary: ado custom experiment for benchmarking CPLEX MIP solver performance across parameter configurations
5
+ Requires-Dist: ado-core>=2.0.0
6
+ Requires-Dist: cplex
@@ -0,0 +1,6 @@
1
+ cplex_mip_experiments/__init__.py,sha256=TYzDILEvuRLJtb422cVnq-gMYTPJ18ia6VUc36t_ItA,70
2
+ cplex_mip_experiments/solve_mip.py,sha256=ED1vbd9J4c83P8w5DEDw3enF6MmUx_Kgz8MR_1n1E9w,36174
3
+ ado_cplex_mip-1.0.3.dist-info/METADATA,sha256=j29VoOPONgVCnE2hu6v3HLhPpZumr8edPO-mdPC0jeE,218
4
+ ado_cplex_mip-1.0.3.dist-info/WHEEL,sha256=QzEo54-UZrE9UYFkp71YQxN7B7ZqQByJkpywraEnP2s,105
5
+ ado_cplex_mip-1.0.3.dist-info/entry_points.txt,sha256=K-1yneqRNMgPPcoAC-2AvVOyR-pCZg3cJSntmQkikwY,69
6
+ ado_cplex_mip-1.0.3.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.31.0
3
+ Root-Is-Purelib: true
4
+ Tag: py2-none-any
5
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [ado.custom_experiments]
2
+ cplex_mip = cplex_mip_experiments.solve_mip
@@ -0,0 +1,2 @@
1
+ # Copyright IBM Corporation 2025, 2026
2
+ # SPDX-License-Identifier: MIT
@@ -0,0 +1,993 @@
1
+ # Copyright IBM Corporation 2025, 2026
2
+ # SPDX-License-Identifier: MIT
3
+
4
+ import logging
5
+ import os
6
+ import pathlib
7
+ import sys
8
+ import tempfile
9
+ import time
10
+ from typing import Any, Literal
11
+
12
+ from ado.modules.actuators.custom_experiments import custom_experiment
13
+ from ado.schema.domain import PropertyDomain, VariableTypeEnum
14
+ from ado.schema.property import ConstitutiveProperty
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ CutPassesAllLevel = Literal["cplex_default", -1, 0, 1, 2, 3]
19
+
20
+ MpsFile = ConstitutiveProperty(
21
+ identifier="mps_file",
22
+ metadata={
23
+ "description": (
24
+ "Path to the MPS instance file to solve (.mps or .mps.gz). "
25
+ "Example: a MIPLIB benchmark such as bab6.mps.gz."
26
+ )
27
+ },
28
+ propertyDomain=PropertyDomain(
29
+ variableType=VariableTypeEnum.OPEN_CATEGORICAL_VARIABLE_TYPE,
30
+ values=["/path/to/instance.mps.gz"],
31
+ ),
32
+ )
33
+
34
+ NSeeds = ConstitutiveProperty(
35
+ identifier="n_seeds",
36
+ metadata={
37
+ "description": (
38
+ "Number of random seeds to use. CPLEX is run once per seed "
39
+ "with seeds 0, 1, …, n_seeds-1. Output properties are vectors "
40
+ "of length n_seeds, one element per seed run."
41
+ )
42
+ },
43
+ propertyDomain=PropertyDomain(
44
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
45
+ domainRange=[1, 100],
46
+ interval=1,
47
+ ),
48
+ )
49
+
50
+ NodeSelection = ConstitutiveProperty(
51
+ identifier="node_selection",
52
+ metadata={
53
+ "description": (
54
+ "CPLEX node selection strategy (CPX_PARAM_NODESEL): "
55
+ "0=depth-first, 1=best-bound (default), "
56
+ "2=best-estimate, 3=best-estimate-alternative."
57
+ )
58
+ },
59
+ propertyDomain=PropertyDomain(
60
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
61
+ values=[0, 1, 2, 3],
62
+ ),
63
+ )
64
+
65
+ VariableSelection = ConstitutiveProperty(
66
+ identifier="variable_selection",
67
+ metadata={
68
+ "description": (
69
+ "CPLEX branching variable selection strategy (CPX_PARAM_VARSEL): "
70
+ "-1=minimum infeasibility, 0=automatic (default), "
71
+ "1=maximum infeasibility, 2=pseudo-cost, "
72
+ "3=strong branching, 4=pseudo-reduced-cost."
73
+ )
74
+ },
75
+ propertyDomain=PropertyDomain(
76
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
77
+ values=[-1, 0, 1, 2, 3, 4],
78
+ ),
79
+ )
80
+
81
+ HeuristicFrequency = ConstitutiveProperty(
82
+ identifier="heuristic_frequency",
83
+ metadata={
84
+ "description": (
85
+ "CPLEX MIP heuristic application frequency (CPX_PARAM_HEURFREQ): "
86
+ "-1=none, 0=automatic (default), n=apply every n nodes."
87
+ )
88
+ },
89
+ propertyDomain=PropertyDomain(
90
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
91
+ values=[-1, 0, 10, 50, 100],
92
+ ),
93
+ )
94
+
95
+ TimeLimit = ConstitutiveProperty(
96
+ identifier="time_limit_s",
97
+ metadata={
98
+ "description": (
99
+ "CPLEX time limit per seed run in seconds (CPX_PARAM_TILIM). "
100
+ "Default is 1e75 (no limit); CPLEX runs until the optimal solution is found. "
101
+ "Any positive value up to and including 1e75 is accepted."
102
+ )
103
+ },
104
+ propertyDomain=PropertyDomain(
105
+ variableType=VariableTypeEnum.CONTINUOUS_VARIABLE_TYPE,
106
+ # Upper bound is strictly above 1e75: float ULP makes (1e75 + 1) == 1e75, which
107
+ # breaks ``value < max(domainRange)`` validation for the default 1e75.
108
+ domainRange=[0, 1e76],
109
+ ),
110
+ )
111
+
112
+ NThreads = ConstitutiveProperty(
113
+ identifier="n_threads",
114
+ metadata={
115
+ "description": (
116
+ "Number of parallel threads for CPLEX B&B (CPX_PARAM_THREADS). "
117
+ "1=single-threaded (fully deterministic), 2/4/8=parallel. "
118
+ "With n_threads>1 results may vary across runs even with the same seed."
119
+ )
120
+ },
121
+ propertyDomain=PropertyDomain(
122
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
123
+ values=[1, 2, 4, 8],
124
+ ),
125
+ )
126
+
127
+ RinsFrequency = ConstitutiveProperty(
128
+ identifier="rins_frequency",
129
+ metadata={
130
+ "description": (
131
+ "Frequency of RINS (Relaxation Induced Neighborhood Search) heuristic "
132
+ "(CPX_PARAM_RINSHEUR): -1=disabled, 0=automatic, n=apply every n nodes."
133
+ )
134
+ },
135
+ propertyDomain=PropertyDomain(
136
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
137
+ values=[-1, 0, 5, 25, 100],
138
+ ),
139
+ )
140
+
141
+ CutPasses = ConstitutiveProperty(
142
+ identifier="cut_passes",
143
+ metadata={
144
+ "description": (
145
+ "Maximum number of cutting-plane passes at the root node "
146
+ "(CPX_PARAM_CUTPASSES): -1=no cuts, 0=automatic, n=at most n passes."
147
+ )
148
+ },
149
+ propertyDomain=PropertyDomain(
150
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
151
+ domainRange=[-1, 201],
152
+ interval=1,
153
+ ),
154
+ )
155
+
156
+ CutPassesAll = ConstitutiveProperty(
157
+ identifier="cut_passes_all",
158
+ metadata={
159
+ "description": (
160
+ "Uniform aggressiveness for every CPLEX MIP cut family (interactive "
161
+ "`set mip cuts all …`). Each `parameters.mip.cuts.*` value is set to "
162
+ "this level capped by that family's maximum (-1 off, 0 automatic, "
163
+ "1 moderate, 2 aggressive; some families support 3 very aggressive). "
164
+ "`cplex_default` does not change cut parameters (solver defaults)."
165
+ )
166
+ },
167
+ propertyDomain=PropertyDomain(
168
+ variableType=VariableTypeEnum.CATEGORICAL_VARIABLE_TYPE,
169
+ values=["cplex_default", -1, 0, 1, 2, 3],
170
+ ),
171
+ )
172
+
173
+ MipEmphasis = ConstitutiveProperty(
174
+ identifier="mip_emphasis",
175
+ metadata={
176
+ "description": (
177
+ "MIP optimization emphasis (CPX_PARAM_MIPEMPHASIS / "
178
+ "`set mip emphasis mip`): 0=balanced optimality and feasibility, "
179
+ "1=integer feasibility, 2=optimality, 3=best bound, "
180
+ "4=hidden feasible solutions, 5=heuristic."
181
+ )
182
+ },
183
+ propertyDomain=PropertyDomain(
184
+ variableType=VariableTypeEnum.DISCRETE_VARIABLE_TYPE,
185
+ values=[0, 1, 2, 3, 4, 5],
186
+ ),
187
+ )
188
+
189
+ Parallel = ConstitutiveProperty(
190
+ identifier="parallel",
191
+ metadata={
192
+ "description": (
193
+ "If True, run each of the n_seeds solver instances as a Ray remote task. "
194
+ "If False, run all seeds in serial. Ray failures on individual seeds "
195
+ "produce null metrics and a ray_task_failed solve_status for that seed "
196
+ "without failing the whole measurement (partial-OK policy)."
197
+ )
198
+ },
199
+ propertyDomain=PropertyDomain(
200
+ variableType=VariableTypeEnum.BINARY_VARIABLE_TYPE,
201
+ values=[False, True],
202
+ ),
203
+ )
204
+
205
+ WarmStartFile = ConstitutiveProperty(
206
+ identifier="warm_start_file",
207
+ metadata={
208
+ "description": (
209
+ "Path to a CPLEX MIP-start file (.mst, or .sol with the same XML structure). "
210
+ "Empty string disables warm start. Applied before solve on every seed run. "
211
+ "For remote execution, use a bare filename and ship the file via "
212
+ "execution context additionalFiles."
213
+ )
214
+ },
215
+ propertyDomain=PropertyDomain(
216
+ variableType=VariableTypeEnum.OPEN_CATEGORICAL_VARIABLE_TYPE,
217
+ values=["", "/path/to/warm_start.mst"],
218
+ ),
219
+ )
220
+
221
+ ExportSolution = ConstitutiveProperty(
222
+ identifier="export_solution",
223
+ metadata={
224
+ "description": (
225
+ "If True, export the incumbent as MST XML in best_solution_mst. "
226
+ "If False (default), best_solution_mst is still returned but each "
227
+ "seed element is an empty string (no export work, no storage bloat)."
228
+ )
229
+ },
230
+ propertyDomain=PropertyDomain(
231
+ variableType=VariableTypeEnum.BINARY_VARIABLE_TYPE,
232
+ values=[False, True],
233
+ ),
234
+ )
235
+
236
+ ProgressInterval = ConstitutiveProperty(
237
+ identifier="progress_interval_s",
238
+ metadata={
239
+ "description": (
240
+ "Interval in seconds for capturing intermediate MIP progress metrics. "
241
+ "When > 0, a callback records best_objective, best_bound, nodes_explored, "
242
+ "and mip_gap at each interval. Outputs progress_time_grid and aligned "
243
+ "time-series (objective_over_time, etc.). When 0, progress capture is disabled."
244
+ )
245
+ },
246
+ propertyDomain=PropertyDomain(
247
+ variableType=VariableTypeEnum.CONTINUOUS_VARIABLE_TYPE,
248
+ domainRange=[0, 86400], # 0 = disabled, up to 24h
249
+ ),
250
+ )
251
+
252
+ # Max aggressiveness level per ``parameters.mip.cuts.*`` family (CPLEX 22.1 API).
253
+ _MIP_CUT_FAMILY_MAX_LEVEL: dict[str, int] = {
254
+ "bqp": 3,
255
+ "cliques": 3,
256
+ "covers": 3,
257
+ "disjunctive": 3,
258
+ "flowcovers": 2,
259
+ "gomory": 2,
260
+ "gubcovers": 2,
261
+ "implied": 2,
262
+ "liftproj": 3,
263
+ "localimplied": 3,
264
+ "mcfcut": 2,
265
+ "mircut": 2,
266
+ "nodecuts": 3,
267
+ "pathcut": 2,
268
+ "rlt": 3,
269
+ "zerohalfcut": 2,
270
+ }
271
+
272
+
273
+ def _apply_cut_passes_all(model: object, level: int) -> None:
274
+ """Set every MIP cut family to ``level``, capped by that family's maximum."""
275
+ cuts = model.parameters.mip.cuts
276
+ for name, max_level in _MIP_CUT_FAMILY_MAX_LEVEL.items():
277
+ capped = min(max(level, -1), max_level)
278
+ getattr(cuts, name).set(capped)
279
+
280
+
281
+ # Sentinel for "no incumbent" from CPLEX (e.g. 1e75).
282
+ _NO_INCUMBENT_SENTINEL = 1e70
283
+
284
+
285
+ def _normalize_cplex_value(value: float | None) -> float | None:
286
+ """Return ``None`` when CPLEX uses a large sentinel for a missing value."""
287
+ if value is None:
288
+ return None
289
+ if abs(value) >= _NO_INCUMBENT_SENTINEL:
290
+ return None
291
+ return float(value)
292
+
293
+
294
+ def _load_warm_start(model: object, warm_start_file: str) -> str | None:
295
+ """Load a MIP start from disk. Return an error status string on failure."""
296
+ if not warm_start_file:
297
+ return None
298
+ if not pathlib.Path(warm_start_file).is_file():
299
+ return f"warm_start_file_not_found: {warm_start_file}"
300
+ try:
301
+ model.MIP_starts.read(warm_start_file)
302
+ model.parameters.advance.set(1)
303
+ except Exception as exc: # noqa: BLE001
304
+ return f"warm_start_read_error: {exc}"
305
+ return None
306
+
307
+
308
+ def _export_incumbent_mst(model: object, objective_value: float | None) -> str:
309
+ """Export the incumbent as warm-start-ready MST XML, with SOL fallback."""
310
+ import cplex
311
+
312
+ if _normalize_cplex_value(objective_value) is None:
313
+ return ""
314
+
315
+ try:
316
+ values = model.solution.get_values()
317
+ except cplex.exceptions.CplexSolverError:
318
+ return ""
319
+
320
+ if not values:
321
+ return ""
322
+
323
+ try:
324
+ model.MIP_starts.delete()
325
+ model.MIP_starts.add(values)
326
+ with tempfile.NamedTemporaryFile(suffix=".mst", delete=False) as handle:
327
+ temp_path = handle.name
328
+ try:
329
+ model.MIP_starts.write(temp_path)
330
+ return pathlib.Path(temp_path).read_text(encoding="utf-8")
331
+ finally:
332
+ os.unlink(temp_path)
333
+ except Exception: # noqa: BLE001
334
+ logger.debug("MST export failed; falling back to SOL format", exc_info=True)
335
+
336
+ try:
337
+ with tempfile.NamedTemporaryFile(suffix=".sol", delete=False) as handle:
338
+ temp_path = handle.name
339
+ try:
340
+ model.solution.write(temp_path)
341
+ return pathlib.Path(temp_path).read_text(encoding="utf-8")
342
+ finally:
343
+ os.unlink(temp_path)
344
+ except Exception: # noqa: BLE001
345
+ logger.debug("SOL export fallback failed", exc_info=True)
346
+ return ""
347
+
348
+
349
+ def _structured_seed_failure(
350
+ *,
351
+ solve_time: float,
352
+ solve_status: str,
353
+ progress_samples: list[dict[str, Any]] | None = None,
354
+ ) -> dict[str, Any]:
355
+ """Build a single-seed result dict for a failed or aborted run."""
356
+ return {
357
+ "solve_time_s": solve_time,
358
+ "objective_value": None,
359
+ "best_bound": None,
360
+ "mip_gap": None,
361
+ "nodes_explored": 0,
362
+ "solve_status": solve_status,
363
+ "best_solution_mst": "",
364
+ "progress_samples": progress_samples or [],
365
+ }
366
+
367
+
368
+ def _make_progress_callback(
369
+ interval_seconds: float,
370
+ samples: list[dict[str, Any]],
371
+ ) -> type:
372
+ """Create a MIPInfoCallback subclass that records progress at fixed intervals."""
373
+ import cplex
374
+
375
+ class ProgressCallbackImpl(cplex.callbacks.MIPInfoCallback):
376
+ def __init__(self, env: object) -> None:
377
+ super().__init__(env)
378
+ self._interval = interval_seconds
379
+ self._samples = samples
380
+ self._last_t: float | None = None
381
+
382
+ def __call__(self) -> None:
383
+ t = self.get_time() - self.get_start_time()
384
+ if self._last_t is None or t >= self._last_t + self._interval:
385
+ self._last_t = t
386
+ best_obj = self.get_incumbent_objective_value()
387
+ best_bound = self.get_best_objective_value()
388
+ nodes = self.get_num_nodes()
389
+ if abs(best_obj) >= _NO_INCUMBENT_SENTINEL:
390
+ best_obj = None
391
+ if abs(best_bound) >= _NO_INCUMBENT_SENTINEL:
392
+ best_bound = None
393
+ if best_obj is not None and best_bound is not None and best_obj != 0:
394
+ gap = abs(best_bound - best_obj) / abs(best_obj)
395
+ else:
396
+ gap = None
397
+ self._samples.append(
398
+ {
399
+ "elapsed": t,
400
+ "best_objective": best_obj,
401
+ "best_bound": best_bound,
402
+ "nodes_explored": nodes,
403
+ "mip_gap": gap,
404
+ }
405
+ )
406
+
407
+ return ProgressCallbackImpl
408
+
409
+
410
+ def _align_to_grid(
411
+ samples: list[dict[str, Any]],
412
+ time_grid: list[float],
413
+ ) -> dict[str, list[float | None]]:
414
+ """Forward-fill samples onto a fixed time grid.
415
+
416
+ For each grid point t, use the last sample with elapsed <= t.
417
+ Returns dict of metric -> list of values (one per grid point).
418
+ """
419
+ aligned: dict[str, list[float | None]] = {
420
+ "best_objective": [],
421
+ "best_bound": [],
422
+ "nodes_explored": [],
423
+ "mip_gap": [],
424
+ }
425
+ sample_idx = 0
426
+ last: dict[str, float | None] = {
427
+ "best_objective": None,
428
+ "best_bound": None,
429
+ "nodes_explored": None,
430
+ "mip_gap": None,
431
+ }
432
+ for t in time_grid:
433
+ while sample_idx < len(samples) and samples[sample_idx]["elapsed"] <= t:
434
+ s = samples[sample_idx]
435
+ # None means "not updated at this sample"; forward-fill prior values.
436
+ if s["best_objective"] is not None:
437
+ last["best_objective"] = s["best_objective"]
438
+ if s["best_bound"] is not None:
439
+ last["best_bound"] = s["best_bound"]
440
+ nodes = s["nodes_explored"]
441
+ if nodes is not None:
442
+ last["nodes_explored"] = float(nodes)
443
+ if s["mip_gap"] is not None:
444
+ last["mip_gap"] = s["mip_gap"]
445
+ sample_idx += 1
446
+ aligned["best_objective"].append(last["best_objective"])
447
+ aligned["best_bound"].append(last["best_bound"])
448
+ aligned["nodes_explored"].append(last["nodes_explored"])
449
+ aligned["mip_gap"].append(last["mip_gap"])
450
+ return aligned
451
+
452
+
453
+ def _build_time_grid(
454
+ interval_s: float,
455
+ time_limit_s: float,
456
+ max_elapsed: float,
457
+ ) -> list[float]:
458
+ """Build uniform grid ``[0, interval_s, 2*interval_s, ...]`` covering all samples.
459
+
460
+ ``max_elapsed`` is the latest ``elapsed`` among callbacks and the post-solve
461
+ terminal row (per ``solve_mip``). The grid must reach at least ``max_elapsed``
462
+ on its last point; otherwise ``_align_to_grid`` would never apply samples with
463
+ ``elapsed`` between the previous multiple of ``interval_s`` and ``max_elapsed``.
464
+
465
+ When ``time_limit_s`` is finite, the initial cap is at least the limit and at
466
+ least ``max_elapsed`` (handles small overrun past ``TILIM``).
467
+ """
468
+ if interval_s <= 0:
469
+ return []
470
+ cap = float(time_limit_s) if time_limit_s < 1e70 else float(max_elapsed)
471
+ cap = max(cap, float(max_elapsed))
472
+ grid: list[float] = []
473
+ t = 0.0
474
+ # 1e-9 is added to avoid issues with rounding errors when
475
+ # accumulating t+interval_s
476
+ # e.g. 0.2+0.1 in python is not 0.3
477
+ while t <= cap + 1e-9:
478
+ grid.append(t)
479
+ t += interval_s
480
+ while grid and grid[-1] + 1e-9 < float(max_elapsed):
481
+ grid.append(grid[-1] + interval_s)
482
+ return grid
483
+
484
+
485
+ def _append_terminal_progress_sample(
486
+ *,
487
+ model: object,
488
+ progress_samples: list[dict[str, Any]],
489
+ solve_time: float,
490
+ objective_value: float | None,
491
+ mip_gap: float | None,
492
+ nodes_explored: int,
493
+ ) -> None:
494
+ """Append one sample at solve end so the aligned grid includes the final MIP state.
495
+
496
+ Periodic MIPInfoCallback samples can omit the last jump to optimality if the
497
+ solver finishes between two callback ticks. This terminal sample ensures
498
+ the aligned grid always reflects the final incumbent and best bound.
499
+
500
+ Best bound selection:
501
+ - ``mip_gap == 0`` (proven optimal): best bound converges to ``objective_value``.
502
+ - Non-optimal (e.g. time limit): forward-fill the last callback-recorded bound.
503
+ The post-solve CPLEX solution API may return the incumbent rather than the
504
+ LP-relaxation dual bound, so it is only used as a fallback when no callback
505
+ samples exist (``progress_interval_s == 0``).
506
+ """
507
+ if mip_gap is not None and mip_gap == 0.0 and objective_value is not None:
508
+ best_bound: float | None = objective_value
509
+ else:
510
+ best_bound = None
511
+ for prev in reversed(progress_samples):
512
+ prev_bound = prev.get("best_bound")
513
+ if prev_bound is not None:
514
+ best_bound = prev_bound
515
+ break
516
+ if best_bound is None:
517
+ # No callback samples available (progress_interval_s == 0).
518
+ # The post-solve API may return the incumbent for non-optimal solves,
519
+ # so this is a best-effort fallback only.
520
+ import cplex
521
+
522
+ try:
523
+ api_bound = float(model.solution.MIP.get_best_objective_value())
524
+ if abs(api_bound) < _NO_INCUMBENT_SENTINEL:
525
+ best_bound = api_bound
526
+ except (
527
+ cplex.exceptions.CplexSolverError,
528
+ AttributeError,
529
+ TypeError,
530
+ ValueError,
531
+ ):
532
+ pass
533
+
534
+ progress_samples.append(
535
+ {
536
+ "elapsed": float(solve_time),
537
+ "best_objective": objective_value,
538
+ "best_bound": best_bound,
539
+ "nodes_explored": nodes_explored,
540
+ "mip_gap": mip_gap,
541
+ }
542
+ )
543
+
544
+
545
+ def estimate_mip_memory_bytes(mps_file_path: str) -> int:
546
+ """Estimate the Ray memory resource request for a single CPLEX seed task.
547
+
548
+ Uses a power-law formula to scale the estimate with MPS file size while
549
+ dampening growth for large instances. The result is used both as the Ray
550
+ task memory reservation and as the basis for the CPLEX WorkMem limit.
551
+
552
+ Formula:
553
+ estimated_peak_gb = floor_gb + (file_size_mb ** 0.75) * 0.5
554
+ total_requested_gb = estimated_peak_gb * 1.20 (20% OS/Python headroom)
555
+
556
+ The exponent 0.75 reflects that peak B&B memory grows sub-linearly with
557
+ model size: larger models have deeper trees but also more pruning. The
558
+ floor of 4 GB covers solver initialisation overhead for trivial instances.
559
+
560
+ Args:
561
+ mps_file_path: Path to the MPS/LP instance file.
562
+
563
+ Returns:
564
+ Memory request in bytes for ``ray.remote(memory=...)``.
565
+ """
566
+ import os
567
+
568
+ file_size_bytes = os.path.getsize(mps_file_path)
569
+ file_size_mb = file_size_bytes / (1024**2)
570
+ floor_gb = 4.0
571
+ scale_factor = 0.5
572
+ exponent = 0.75
573
+ estimated_peak_gb = floor_gb + (file_size_mb**exponent) * scale_factor
574
+ total_requested_gb = estimated_peak_gb * 1.20
575
+ return int(total_requested_gb * (1024**3))
576
+
577
+
578
+ def _collect_parallel_seed_results(
579
+ refs: list[object],
580
+ ) -> list[dict[str, Any]]:
581
+ """Collect per-seed Ray task results with partial-OK failure handling.
582
+
583
+ If an individual seed task fails (for example runtime environment setup on a
584
+ worker node), a structured failure entry is returned for that seed index and
585
+ collection continues for the remaining refs.
586
+
587
+ Args:
588
+ refs: Ray object refs returned by ``remote_fn.remote(seed)``.
589
+
590
+ Returns:
591
+ One single-seed result dict per ref, in seed order.
592
+ """
593
+ import ray
594
+
595
+ results: list[dict[str, Any]] = []
596
+ for seed_index, ref in enumerate(refs):
597
+ try:
598
+ results.append(ray.get(ref))
599
+ except Exception as exc: # noqa: PERF203
600
+ logger.warning("Ray seed task %d failed: %s", seed_index, exc)
601
+ results.append(
602
+ _structured_seed_failure(
603
+ solve_time=0.0,
604
+ solve_status=f"ray_task_failed: {exc}",
605
+ )
606
+ )
607
+ return results
608
+
609
+
610
+ def _run_single_seed(
611
+ *,
612
+ mps_file: str,
613
+ seed: int,
614
+ seed_index: int,
615
+ n_seeds: int,
616
+ node_selection: int,
617
+ variable_selection: int,
618
+ heuristic_frequency: int,
619
+ time_limit_s: float,
620
+ n_threads: int,
621
+ rins_frequency: int,
622
+ cut_passes: int,
623
+ cut_passes_all: CutPassesAllLevel = "cplex_default",
624
+ mip_emphasis: int = 0,
625
+ progress_interval_s: float = 0,
626
+ workmem_mb: int = 0,
627
+ warm_start_file: str = "",
628
+ export_solution: bool = False,
629
+ ) -> dict[str, Any]:
630
+ """Run CPLEX on a single MPS instance with the given random seed and parameters.
631
+
632
+ Args:
633
+ mps_file: Path to the MPS instance file.
634
+ seed: CPLEX random seed (CPX_PARAM_RANDOMSEED).
635
+ seed_index: Index of this seed (0-based) for logging.
636
+ n_seeds: Total number of seeds for logging.
637
+ node_selection: Node selection strategy (CPX_PARAM_NODESEL).
638
+ variable_selection: Variable selection strategy (CPX_PARAM_VARSEL).
639
+ heuristic_frequency: Heuristic frequency (CPX_PARAM_HEURFREQ).
640
+ time_limit_s: Time limit in seconds (CPX_PARAM_TILIM).
641
+ cut_passes_all: Uniform MIP cut aggressiveness for all families, capped
642
+ per family, or ``cplex_default`` to leave CPLEX defaults unchanged.
643
+ mip_emphasis: MIP emphasis (CPX_PARAM_MIPEMPHASIS): 0-5, see property
644
+ metadata; default 0 matches CPLEX balanced emphasis.
645
+ progress_interval_s: If > 0, capture progress at this interval (seconds).
646
+ workmem_mb: When > 0, sets CPLEX WorkMem (CPX_PARAM_WORKMEM) to this
647
+ value in MB and enables compressed on-disk node files
648
+ (CPX_PARAM_NODEFILEIND=3) so the solver spills to disk rather than
649
+ crashing OOM. Should be ~80% of the Ray task memory reservation.
650
+ warm_start_file: Optional CPLEX MIP-start file path; empty disables warm start.
651
+ export_solution: If True, populate best_solution_mst with MST XML.
652
+
653
+ Returns:
654
+ Dictionary with keys: solve_time_s, objective_value, best_bound, mip_gap,
655
+ nodes_explored, solve_status, best_solution_mst, and progress_samples when
656
+ progress_interval_s > 0.
657
+ """
658
+ import cplex
659
+
660
+ solver_n = seed_index + 1
661
+ logger.info("Start solver %d of %d", solver_n, n_seeds)
662
+
663
+ model = cplex.Cplex()
664
+ model.set_log_stream(sys.stdout)
665
+ model.set_error_stream(sys.stderr)
666
+ model.set_warning_stream(sys.stderr)
667
+ model.set_results_stream(sys.stdout)
668
+
669
+ model.read(mps_file)
670
+ warm_start_error = _load_warm_start(model, warm_start_file)
671
+ if warm_start_error is not None:
672
+ logger.warning(
673
+ "Warm start failed on seed %d: %s",
674
+ seed,
675
+ warm_start_error,
676
+ )
677
+ logger.info("End solver %d of %d", solver_n, n_seeds)
678
+ return _structured_seed_failure(
679
+ solve_time=0.0,
680
+ solve_status=warm_start_error,
681
+ )
682
+
683
+ model.parameters.randomseed.set(seed)
684
+ model.parameters.threads.set(n_threads)
685
+ model.parameters.emphasis.mip.set(mip_emphasis)
686
+ model.parameters.mip.strategy.nodeselect.set(node_selection)
687
+ model.parameters.mip.strategy.variableselect.set(variable_selection)
688
+ model.parameters.mip.strategy.heuristicfreq.set(heuristic_frequency)
689
+ model.parameters.mip.strategy.rinsheur.set(rins_frequency)
690
+ model.parameters.mip.limits.cutpasses.set(cut_passes)
691
+ model.parameters.timelimit.set(time_limit_s)
692
+ model.parameters.mip.display.set(4)
693
+ model.parameters.mip.interval.set(100)
694
+ if workmem_mb > 0:
695
+ model.parameters.workmem.set(float(workmem_mb))
696
+ model.parameters.mip.strategy.file.set(3)
697
+ if cut_passes_all != "cplex_default":
698
+ _apply_cut_passes_all(model, int(cut_passes_all))
699
+
700
+ progress_samples: list[dict[str, Any]] = []
701
+ if progress_interval_s > 0:
702
+ model.register_callback(
703
+ _make_progress_callback(progress_interval_s, progress_samples)
704
+ )
705
+
706
+ logger.debug(
707
+ "Solving %s with seed=%d, n_threads=%d, node_selection=%d, "
708
+ "variable_selection=%d, heuristic_frequency=%d, rins_frequency=%d, "
709
+ "cut_passes=%d, cut_passes_all=%r, mip_emphasis=%d, time_limit_s=%.1g",
710
+ mps_file,
711
+ seed,
712
+ n_threads,
713
+ node_selection,
714
+ variable_selection,
715
+ heuristic_frequency,
716
+ rins_frequency,
717
+ cut_passes,
718
+ cut_passes_all,
719
+ mip_emphasis,
720
+ time_limit_s,
721
+ )
722
+
723
+ t0 = time.perf_counter()
724
+ try:
725
+ model.solve()
726
+ except cplex.exceptions.CplexSolverError as exc:
727
+ solve_time = time.perf_counter() - t0
728
+ # CPLEX error 1016 (CPXERR_RESTRICTED_VERSION) is the community-edition
729
+ # size limit. Re-raise so the framework produces InvalidMeasurementResult,
730
+ # preventing memoization from reusing this failed result.
731
+ error_code = (
732
+ exc.args[2] if len(exc.args) > 2 else getattr(exc, "error_code", None)
733
+ )
734
+ if error_code == 1016: # CPXERR_RESTRICTED_VERSION
735
+ logger.warning(
736
+ "CPLEX Community Edition limits exceeded on seed %d: %s", seed, exc
737
+ )
738
+ raise
739
+ # For other CPLEX errors, return a structured result.
740
+ status = f"cplex_error_{error_code}" if error_code else f"cplex_error: {exc}"
741
+ logger.warning("CPLEX solver error on seed %d: %s", seed, exc)
742
+ logger.info("End solver %d of %d", solver_n, n_seeds)
743
+ return _structured_seed_failure(
744
+ solve_time=solve_time,
745
+ solve_status=status,
746
+ )
747
+ solve_time = time.perf_counter() - t0
748
+
749
+ status = model.solution.get_status_string()
750
+ nodes = model.solution.progress.get_num_nodes_processed()
751
+
752
+ try:
753
+ obj = _normalize_cplex_value(model.solution.get_objective_value())
754
+ except cplex.exceptions.CplexSolverError:
755
+ obj = None
756
+
757
+ try:
758
+ gap = model.solution.MIP.get_mip_relative_gap()
759
+ except cplex.exceptions.CplexSolverError:
760
+ gap = None
761
+
762
+ best_solution_mst = _export_incumbent_mst(model, obj) if export_solution else ""
763
+
764
+ logger.debug(
765
+ "Seed %d finished: status=%s, time=%.2fs, obj=%s, gap=%s, nodes=%d",
766
+ seed,
767
+ status,
768
+ solve_time,
769
+ obj,
770
+ gap,
771
+ nodes,
772
+ )
773
+
774
+ logger.info("End solver %d of %d", solver_n, n_seeds)
775
+
776
+ # Always append the terminal sample. When progress_interval_s > 0 it is
777
+ # included in time-series alignment; in all cases it is the canonical source
778
+ # for the scalar best_bound derived below.
779
+ _append_terminal_progress_sample(
780
+ model=model,
781
+ progress_samples=progress_samples,
782
+ solve_time=solve_time,
783
+ objective_value=obj,
784
+ mip_gap=gap,
785
+ nodes_explored=nodes,
786
+ )
787
+
788
+ # Scalar best_bound: last non-None best_bound in progress_samples.
789
+ # The terminal sample sets this to objective_value at optimality (mip_gap == 0)
790
+ # and forward-fills the last callback-recorded LP-relaxation bound otherwise,
791
+ # avoiding the post-solve CPLEX API which may return the incumbent.
792
+ best_bound: float | None = None
793
+ for prev in reversed(progress_samples):
794
+ if prev.get("best_bound") is not None:
795
+ best_bound = prev["best_bound"]
796
+ break
797
+
798
+ return {
799
+ "solve_time_s": solve_time,
800
+ "objective_value": obj,
801
+ "best_bound": best_bound,
802
+ "mip_gap": gap,
803
+ "nodes_explored": nodes,
804
+ "solve_status": status,
805
+ "best_solution_mst": best_solution_mst,
806
+ "progress_samples": progress_samples,
807
+ }
808
+
809
+
810
+ @custom_experiment(
811
+ required_properties=[MpsFile],
812
+ optional_properties=[
813
+ NSeeds,
814
+ NodeSelection,
815
+ VariableSelection,
816
+ HeuristicFrequency,
817
+ TimeLimit,
818
+ NThreads,
819
+ RinsFrequency,
820
+ CutPasses,
821
+ CutPassesAll,
822
+ MipEmphasis,
823
+ Parallel,
824
+ ProgressInterval,
825
+ WarmStartFile,
826
+ ExportSolution,
827
+ ],
828
+ output_property_identifiers=[
829
+ "solve_times",
830
+ "objective_values",
831
+ "mip_gaps",
832
+ "nodes_explored",
833
+ "solve_statuses",
834
+ "best_bounds",
835
+ "best_solution_mst",
836
+ "progress_time_grid",
837
+ "objective_over_time",
838
+ "best_bound_over_time",
839
+ "nodes_explored_over_time",
840
+ "mip_gap_over_time",
841
+ ],
842
+ metadata={
843
+ "description": (
844
+ "Solves a MIP instance with CPLEX across N random seeds and reports "
845
+ "vectors of performance metrics. Each output property is a list of "
846
+ "length n_seeds, capturing solve time, objective value, MIP gap, "
847
+ "nodes explored, and solver status per seed. This enables analysis "
848
+ "of both parameter effects and seed-induced variability."
849
+ )
850
+ },
851
+ parameterization={},
852
+ )
853
+ def solve_mip(
854
+ mps_file: str,
855
+ n_seeds: int = 5,
856
+ node_selection: int = 1,
857
+ variable_selection: int = 0,
858
+ heuristic_frequency: int = 0,
859
+ time_limit_s: float = 1e75,
860
+ n_threads: int = 1,
861
+ rins_frequency: int = 0,
862
+ cut_passes: int = 0,
863
+ cut_passes_all: CutPassesAllLevel = "cplex_default",
864
+ mip_emphasis: int = 0,
865
+ parallel: bool = True,
866
+ progress_interval_s: float = 0,
867
+ warm_start_file: str = "",
868
+ export_solution: bool = False,
869
+ ) -> dict[str, list]:
870
+ """Solve a MIP instance with CPLEX across multiple random seeds.
871
+
872
+ Args:
873
+ mps_file: Path to the MPS instance file (.mps or .mps.gz).
874
+ n_seeds: Number of random seeds to use (seeds 0 to n_seeds-1).
875
+ node_selection: CPLEX node selection strategy (CPX_PARAM_NODESEL).
876
+ variable_selection: CPLEX variable selection strategy (CPX_PARAM_VARSEL).
877
+ heuristic_frequency: CPLEX heuristic frequency (CPX_PARAM_HEURFREQ).
878
+ time_limit_s: CPLEX time limit per seed run in seconds. Default 1e75 means
879
+ no practical limit; CPLEX runs until the optimal solution is found.
880
+ n_threads: Number of parallel B&B threads (CPX_PARAM_THREADS). Default 1
881
+ ensures fully deterministic execution.
882
+ rins_frequency: RINS heuristic frequency (CPX_PARAM_RINSHEUR). -1=disabled,
883
+ 0=automatic, n=apply every n nodes.
884
+ cut_passes: Max cutting-plane passes at root (CPX_PARAM_CUTPASSES).
885
+ -1=no cuts, 0=automatic, n=at most n passes.
886
+ cut_passes_all: Same aggressiveness for all ``mip.cuts`` families, capped
887
+ per family; ``cplex_default`` leaves CPLEX defaults unchanged.
888
+ mip_emphasis: CPLEX MIP emphasis (0=balanced through 5=heuristic); default 0.
889
+ parallel: If True, run each seed as a Ray remote task (requires ray_remote).
890
+ If False, run seeds in serial. With parallel=True, Ray task failures on
891
+ individual seeds are recorded as ``ray_task_failed: ...`` in
892
+ solve_statuses with null metrics for that seed; successful seeds are
893
+ still returned (partial-OK policy).
894
+ progress_interval_s: If > 0, capture intermediate progress at this interval
895
+ (seconds). Outputs progress_time_grid and aligned time-series.
896
+ warm_start_file: Optional CPLEX MIP-start file applied before each seed solve.
897
+ export_solution: If True, populate best_solution_mst with MST XML per seed.
898
+
899
+ Returns:
900
+ Dictionary with vector-valued outputs (one element per seed):
901
+ - solve_times: Wall-clock solve times in seconds.
902
+ - objective_values: Best objective values found.
903
+ - best_bounds: Final MIP best bounds.
904
+ - best_solution_mst: MST XML strings for warm-start round-trip, or ``""``.
905
+ - mip_gaps: Final relative MIP gaps.
906
+ - nodes_explored: B&B nodes processed.
907
+ - solve_statuses: CPLEX status strings.
908
+ When progress_interval_s > 0, also:
909
+ - progress_time_grid: Shared time points (seconds).
910
+ - objective_over_time, best_bound_over_time, nodes_explored_over_time,
911
+ mip_gap_over_time: list[list] of aligned values [seed][time_idx].
912
+ A terminal sample at solve completion is appended so forward-filled
913
+ grid values can reflect the final incumbent and best bound, not only
914
+ the last periodic callback.
915
+ """
916
+ import ray
917
+
918
+ mem_bytes = estimate_mip_memory_bytes(mps_file)
919
+ # 80% of the Ray task reservation: CPLEX WorkMem budget in MB, leaving 20%
920
+ # headroom for the Python interpreter and Ray worker overhead.
921
+ workmem_mb = int(mem_bytes / (1024**2) * 0.80)
922
+
923
+ def _run_one(seed: int) -> dict[str, Any]:
924
+ return _run_single_seed(
925
+ mps_file=mps_file,
926
+ seed=seed,
927
+ seed_index=seed,
928
+ n_seeds=n_seeds,
929
+ node_selection=node_selection,
930
+ variable_selection=variable_selection,
931
+ heuristic_frequency=heuristic_frequency,
932
+ time_limit_s=time_limit_s,
933
+ n_threads=n_threads,
934
+ rins_frequency=rins_frequency,
935
+ cut_passes=cut_passes,
936
+ cut_passes_all=cut_passes_all,
937
+ mip_emphasis=mip_emphasis,
938
+ progress_interval_s=progress_interval_s,
939
+ warm_start_file=warm_start_file,
940
+ export_solution=export_solution,
941
+ workmem_mb=workmem_mb,
942
+ )
943
+
944
+ if parallel:
945
+ if not ray.is_initialized():
946
+ raise ValueError(
947
+ "parallel=True requires the experiment to run with ray_remote "
948
+ "(use_ray=True). Ray is not initialized."
949
+ )
950
+ remote_fn = ray.remote(num_cpus=n_threads, memory=mem_bytes)(_run_one)
951
+ refs = [remote_fn.remote(seed) for seed in range(n_seeds)]
952
+ results = _collect_parallel_seed_results(refs)
953
+ else:
954
+ results = []
955
+ for seed in range(n_seeds):
956
+ result = _run_one(seed)
957
+ results.append(result)
958
+
959
+ out: dict[str, list] = {
960
+ "solve_times": [r["solve_time_s"] for r in results],
961
+ "objective_values": [r["objective_value"] for r in results],
962
+ "best_bounds": [r["best_bound"] for r in results],
963
+ "best_solution_mst": [r["best_solution_mst"] for r in results],
964
+ "mip_gaps": [r["mip_gap"] for r in results],
965
+ "nodes_explored": [r["nodes_explored"] for r in results],
966
+ "solve_statuses": [r["solve_status"] for r in results],
967
+ }
968
+
969
+ if progress_interval_s > 0:
970
+ all_samples = [r.get("progress_samples", []) for r in results]
971
+ max_elapsed = max(
972
+ (s["elapsed"] for samples in all_samples for s in samples),
973
+ default=0.0,
974
+ )
975
+ time_grid = _build_time_grid(progress_interval_s, time_limit_s, max_elapsed)
976
+ aligned_per_seed = [
977
+ _align_to_grid(samples, time_grid) for samples in all_samples
978
+ ]
979
+ out["progress_time_grid"] = time_grid
980
+ out["objective_over_time"] = [a["best_objective"] for a in aligned_per_seed]
981
+ out["best_bound_over_time"] = [a["best_bound"] for a in aligned_per_seed]
982
+ out["nodes_explored_over_time"] = [
983
+ a["nodes_explored"] for a in aligned_per_seed
984
+ ]
985
+ out["mip_gap_over_time"] = [a["mip_gap"] for a in aligned_per_seed]
986
+ else:
987
+ out["progress_time_grid"] = []
988
+ out["objective_over_time"] = []
989
+ out["best_bound_over_time"] = []
990
+ out["nodes_explored_over_time"] = []
991
+ out["mip_gap_over_time"] = []
992
+
993
+ return out