cld-reducer 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cld_reducer/NOTICE +35 -0
- cld_reducer/__init__.py +16 -0
- cld_reducer/_solver.py +140 -0
- cld_reducer/algorithms/__init__.py +1 -0
- cld_reducer/algorithms/assignment_minimum.py +351 -0
- cld_reducer/api.py +97 -0
- cld_reducer/cli.py +79 -0
- cld_reducer/cliques.py +53 -0
- cld_reducer/exceptions.py +13 -0
- cld_reducer/labels.py +33 -0
- cld_reducer/result.py +34 -0
- cld_reducer/validation.py +290 -0
- cld_reducer-0.2.0.dist-info/METADATA +164 -0
- cld_reducer-0.2.0.dist-info/RECORD +17 -0
- cld_reducer-0.2.0.dist-info/WHEEL +4 -0
- cld_reducer-0.2.0.dist-info/entry_points.txt +2 -0
- cld_reducer-0.2.0.dist-info/licenses/LICENSE +21 -0
cld_reducer/NOTICE
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
Example data sources and reuse record
|
|
2
|
+
|
|
3
|
+
The MIT license covers the package code. The examples below contain public
|
|
4
|
+
summary data from the cited papers. On 2026-10-10, John Ennis confirmed that
|
|
5
|
+
the data are public and approved their inclusion in these distributions.
|
|
6
|
+
This record does not assign a new license to the source publications.
|
|
7
|
+
|
|
8
|
+
piepho2004_wheat_pairs.csv / piepho2004_wheat
|
|
9
|
+
Source: Piepho (2004), An Algorithm for a Letter-Based Representation of All-
|
|
10
|
+
Pairwise Comparisons, doi:10.1198/1061860043515; reproduced in Table 7 of
|
|
11
|
+
Ennis, Fayle, and Ennis (2012), Assignment-Minimum Clique Coverings,
|
|
12
|
+
doi:10.1145/2133803.2275596.
|
|
13
|
+
Content: significant/non-significant decisions for the 190 unordered pairs of
|
|
14
|
+
20 wheat treatments. The repository stores these decisions as CSV rows.
|
|
15
|
+
R data-raw/datasets.R reads labels as text and decisions as logical values,
|
|
16
|
+
then saves the same rows as piepho2004_wheat.rda. Python examples copy the CSV.
|
|
17
|
+
Reuse record: maintainer confirmation and approval on 2026-10-10.
|
|
18
|
+
|
|
19
|
+
simple_abc_to_ac_pairs.csv / simple_abc_pairs
|
|
20
|
+
simple_abc_to_ac_means.csv / simple_abc_means
|
|
21
|
+
Source: Ennis, Fayle, and Ennis (2012), Assignment-Minimum Clique Coverings,
|
|
22
|
+
doi:10.1145/2133803.2275596. The R help pages attribute the five-group
|
|
23
|
+
ABC-to-AC teaching example and its means to this paper.
|
|
24
|
+
Content: ten pairwise decisions and five numeric means. R data-raw/datasets.R
|
|
25
|
+
reads the CSV files and saves the corresponding data frames. Python examples
|
|
26
|
+
copy the CSV files.
|
|
27
|
+
Reuse record: maintainer confirmation and approval on 2026-10-10.
|
|
28
|
+
|
|
29
|
+
Retain this source record with the example data.
|
|
30
|
+
|
|
31
|
+
Distribution scope
|
|
32
|
+
The R archive contains both example data sets. The Python source archive contains
|
|
33
|
+
their CSV files. The Python wheel omits the CSV files but includes simple-example
|
|
34
|
+
output from the README in its metadata. The npm archive contains no wheat data,
|
|
35
|
+
but its README reproduces the simple five-group example.
|
cld_reducer/__init__.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Tools for reducing compact letter displays."""
|
|
2
|
+
|
|
3
|
+
from .api import reduce_from_adjacency, reduce_letters
|
|
4
|
+
from .exceptions import CLDReducerError, InvalidInputError, SolverError
|
|
5
|
+
from .result import CLDReductionResult
|
|
6
|
+
|
|
7
|
+
__version__ = "0.2.0"
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"CLDReducerError",
|
|
11
|
+
"CLDReductionResult",
|
|
12
|
+
"InvalidInputError",
|
|
13
|
+
"SolverError",
|
|
14
|
+
"reduce_from_adjacency",
|
|
15
|
+
"reduce_letters",
|
|
16
|
+
]
|
cld_reducer/_solver.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""HiGHS through highspy, with the settings of docs/algorithm.md section 6."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
import highspy
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class Settings:
|
|
13
|
+
"""Presolve is the one setting that differs between languages (R turns it off).
|
|
14
|
+
|
|
15
|
+
Tests switch it off to run the conformance suite a second time.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
presolve: str = "on"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
settings = Settings()
|
|
22
|
+
# The options of the most recent solve, read back from HiGHS. Tests check them.
|
|
23
|
+
last_options: dict[str, object] = {}
|
|
24
|
+
|
|
25
|
+
OPTION_NAMES = (
|
|
26
|
+
"presolve",
|
|
27
|
+
"mip_rel_gap",
|
|
28
|
+
"mip_abs_gap",
|
|
29
|
+
"primal_feasibility_tolerance",
|
|
30
|
+
"mip_feasibility_tolerance",
|
|
31
|
+
"threads",
|
|
32
|
+
"time_limit",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
OPTIMAL = "optimal"
|
|
36
|
+
INFEASIBLE = "infeasible"
|
|
37
|
+
TIME_LIMIT = "time_limit"
|
|
38
|
+
FAILED = "failed"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class Problem:
|
|
43
|
+
"""The model of docs/algorithm.md section 4 in row-wise sparse form.
|
|
44
|
+
|
|
45
|
+
The first `num_x` columns are the membership variables x; the rest are the y variables.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
num_cols: int
|
|
49
|
+
num_x: int
|
|
50
|
+
cost: np.ndarray
|
|
51
|
+
start: np.ndarray
|
|
52
|
+
index: np.ndarray
|
|
53
|
+
value: np.ndarray
|
|
54
|
+
row_lower: np.ndarray
|
|
55
|
+
row_upper: np.ndarray
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass(frozen=True)
|
|
59
|
+
class Outcome:
|
|
60
|
+
"""What one solve returned: a status, the HiGHS status text, and the column values."""
|
|
61
|
+
|
|
62
|
+
status: str
|
|
63
|
+
text: str
|
|
64
|
+
values: np.ndarray | None = None
|
|
65
|
+
objective: float | None = None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def run(
|
|
69
|
+
problem: Problem,
|
|
70
|
+
col_lower: np.ndarray,
|
|
71
|
+
col_upper: np.ndarray,
|
|
72
|
+
*,
|
|
73
|
+
sum_limit: int | None = None,
|
|
74
|
+
time_limit: float | None = None,
|
|
75
|
+
) -> Outcome:
|
|
76
|
+
"""Solve the model with the given column bounds.
|
|
77
|
+
|
|
78
|
+
`sum_limit` adds the row `sum(x) <= sum_limit`. `time_limit` is the HiGHS time limit in
|
|
79
|
+
seconds for this solve.
|
|
80
|
+
"""
|
|
81
|
+
start, index, value = problem.start, problem.index, problem.value
|
|
82
|
+
row_lower, row_upper = problem.row_lower, problem.row_upper
|
|
83
|
+
if sum_limit is not None:
|
|
84
|
+
extra = np.arange(problem.num_x, dtype=np.int32)
|
|
85
|
+
start = np.concatenate([start, [start[-1] + problem.num_x]]).astype(np.int32)
|
|
86
|
+
index = np.concatenate([index, extra])
|
|
87
|
+
value = np.concatenate([value, np.ones(problem.num_x)])
|
|
88
|
+
row_lower = np.concatenate([row_lower, [-highspy.kHighsInf]])
|
|
89
|
+
row_upper = np.concatenate([row_upper, [float(sum_limit)]])
|
|
90
|
+
|
|
91
|
+
lp = highspy.HighsLp()
|
|
92
|
+
lp.num_col_ = problem.num_cols
|
|
93
|
+
lp.num_row_ = len(row_lower)
|
|
94
|
+
lp.col_cost_ = problem.cost
|
|
95
|
+
lp.col_lower_ = col_lower
|
|
96
|
+
lp.col_upper_ = col_upper
|
|
97
|
+
lp.integrality_ = [highspy.HighsVarType.kInteger] * problem.num_cols
|
|
98
|
+
lp.row_lower_ = row_lower
|
|
99
|
+
lp.row_upper_ = row_upper
|
|
100
|
+
lp.a_matrix_.format_ = highspy.MatrixFormat.kRowwise
|
|
101
|
+
lp.a_matrix_.num_col_ = problem.num_cols
|
|
102
|
+
lp.a_matrix_.num_row_ = len(row_lower)
|
|
103
|
+
lp.a_matrix_.start_ = start
|
|
104
|
+
lp.a_matrix_.index_ = index
|
|
105
|
+
lp.a_matrix_.value_ = value
|
|
106
|
+
lp.sense_ = highspy.ObjSense.kMinimize
|
|
107
|
+
|
|
108
|
+
h = highspy.Highs()
|
|
109
|
+
h.setOptionValue("output_flag", False)
|
|
110
|
+
h.setOptionValue("presolve", settings.presolve)
|
|
111
|
+
h.setOptionValue("mip_rel_gap", 0.0)
|
|
112
|
+
h.setOptionValue("mip_abs_gap", 0.0)
|
|
113
|
+
h.setOptionValue("primal_feasibility_tolerance", 1e-9)
|
|
114
|
+
h.setOptionValue("mip_feasibility_tolerance", 1e-9)
|
|
115
|
+
h.setOptionValue("threads", 1)
|
|
116
|
+
if time_limit is not None:
|
|
117
|
+
h.setOptionValue("time_limit", float(time_limit))
|
|
118
|
+
h.passModel(lp)
|
|
119
|
+
h.run()
|
|
120
|
+
|
|
121
|
+
last_options.clear()
|
|
122
|
+
for name in OPTION_NAMES:
|
|
123
|
+
option = h.getOptionValue(name)
|
|
124
|
+
# highspy 1.15 returns (status, value).
|
|
125
|
+
last_options[name] = option[1] if isinstance(option, tuple) else option
|
|
126
|
+
|
|
127
|
+
status = h.getModelStatus()
|
|
128
|
+
text = h.modelStatusToString(status)
|
|
129
|
+
if status == highspy.HighsModelStatus.kOptimal:
|
|
130
|
+
values = np.asarray(h.getSolution().col_value, dtype=np.float64)
|
|
131
|
+
return Outcome(OPTIMAL, text, values, float(h.getInfo().objective_function_value))
|
|
132
|
+
# No model here is unbounded, so HiGHS's "unbounded or infeasible" means infeasible.
|
|
133
|
+
if status in (
|
|
134
|
+
highspy.HighsModelStatus.kInfeasible,
|
|
135
|
+
highspy.HighsModelStatus.kUnboundedOrInfeasible,
|
|
136
|
+
):
|
|
137
|
+
return Outcome(INFEASIBLE, text)
|
|
138
|
+
if status == highspy.HighsModelStatus.kTimeLimit:
|
|
139
|
+
return Outcome(TIME_LIMIT, text)
|
|
140
|
+
return Outcome(FAILED, text)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""CLD reduction algorithm implementations."""
|
|
@@ -0,0 +1,351 @@
|
|
|
1
|
+
"""Assignment-minimum compact letter display reduction.
|
|
2
|
+
|
|
3
|
+
Solves the assignment-minimum clique covering problem defined in Ennis, Fayle,
|
|
4
|
+
& Ennis (2012), "Assignment-Minimum Clique Coverings", ACM JEA 17, Art. 1.5
|
|
5
|
+
(https://doi.org/10.1145/2133803.2275596). The paper uses a backtracking
|
|
6
|
+
algorithm (FIND-AM); this module solves the same problem as a binary
|
|
7
|
+
mixed-integer program with HiGHS, then picks the canonical optimum of
|
|
8
|
+
docs/algorithm.md section 5 so that every implementation returns the same display.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import time
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from math import isfinite
|
|
16
|
+
from numbers import Integral, Real
|
|
17
|
+
|
|
18
|
+
import numpy as np
|
|
19
|
+
import pandas as pd
|
|
20
|
+
from highspy import kHighsInf
|
|
21
|
+
|
|
22
|
+
from .. import _solver
|
|
23
|
+
from ..cliques import maximal_cliques
|
|
24
|
+
from ..exceptions import SolverError
|
|
25
|
+
from ..labels import make_letter_labels
|
|
26
|
+
from ..result import CLDReductionResult
|
|
27
|
+
from ..validation import (
|
|
28
|
+
count_assignments,
|
|
29
|
+
normalize_means,
|
|
30
|
+
reconstruct_adjacency_from_assignments,
|
|
31
|
+
validate_adjacency,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
# The clock behind the shared time budget. Tests replace it.
|
|
35
|
+
_now = time.monotonic
|
|
36
|
+
|
|
37
|
+
_INTEGRALITY_TOLERANCE = 1e-6
|
|
38
|
+
_INVALID_SOLUTION = "HiGHS returned an invalid solution"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class _Model:
|
|
43
|
+
problem: _solver.Problem
|
|
44
|
+
members: list[tuple[int, int]] # (clique, group) of each x variable, in canonical order
|
|
45
|
+
edges: list[tuple[int, int]]
|
|
46
|
+
group_columns: list[list[int]] # x variables of each group
|
|
47
|
+
edge_ends: list[list[tuple[int, int]]] # per edge: the two x variables of each covering clique
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def reduce_assignment_minimum(
|
|
51
|
+
adjacency: np.ndarray,
|
|
52
|
+
groups: list[str],
|
|
53
|
+
means: pd.Series | None = None,
|
|
54
|
+
*,
|
|
55
|
+
method: str = "assignment_minimum",
|
|
56
|
+
time_limit: float | None = None,
|
|
57
|
+
max_cliques: int | None = 10_000,
|
|
58
|
+
) -> CLDReductionResult:
|
|
59
|
+
"""Reduce a CLD by minimizing total letter assignments.
|
|
60
|
+
|
|
61
|
+
Parameters
|
|
62
|
+
----------
|
|
63
|
+
adjacency:
|
|
64
|
+
Symmetric boolean matrix where `True` means two groups are not
|
|
65
|
+
significantly different and may share a letter.
|
|
66
|
+
groups:
|
|
67
|
+
Group labels in matrix order.
|
|
68
|
+
means:
|
|
69
|
+
Optional group means used only to produce stable, mean-ordered letters.
|
|
70
|
+
method:
|
|
71
|
+
Method label stored in the result metadata.
|
|
72
|
+
time_limit:
|
|
73
|
+
Optional time budget in seconds, shared by all solves of this call.
|
|
74
|
+
max_cliques:
|
|
75
|
+
Optional cap on maximal cliques to enumerate before failing with a
|
|
76
|
+
controlled `SolverError`. Pass `None` to disable the cap.
|
|
77
|
+
"""
|
|
78
|
+
adjacency, groups = validate_adjacency(adjacency, groups)
|
|
79
|
+
means = normalize_means(means, groups)
|
|
80
|
+
time_limit, max_cliques = _validate_solver_controls(time_limit, max_cliques)
|
|
81
|
+
|
|
82
|
+
cliques = maximal_cliques(adjacency, max_cliques=max_cliques)
|
|
83
|
+
model = _build_model(adjacency, cliques)
|
|
84
|
+
selected, minimum = _solve_canonical(model, time_limit)
|
|
85
|
+
columns = _selected_columns(cliques, model, selected)
|
|
86
|
+
tokens = _assign_letter_tokens(columns, len(groups), means, groups)
|
|
87
|
+
assignments = {group: tokens[index] for index, group in enumerate(groups)}
|
|
88
|
+
letters = {group: _format_letter_tokens(value) for group, value in assignments.items()}
|
|
89
|
+
reconstructed = reconstruct_adjacency_from_assignments(assignments, groups)
|
|
90
|
+
if not np.array_equal(reconstructed, adjacency):
|
|
91
|
+
msg = "optimized letters did not preserve the input pairwise relationships"
|
|
92
|
+
raise SolverError(msg)
|
|
93
|
+
|
|
94
|
+
assignments_before = len(model.members)
|
|
95
|
+
assignments_after = count_assignments(assignments)
|
|
96
|
+
reduction_pct = (
|
|
97
|
+
(assignments_before - assignments_after) / assignments_before * 100
|
|
98
|
+
if assignments_before
|
|
99
|
+
else 0.0
|
|
100
|
+
)
|
|
101
|
+
stats = {
|
|
102
|
+
"assignments_before": assignments_before,
|
|
103
|
+
"assignments_after": int(assignments_after),
|
|
104
|
+
"reduction_pct": reduction_pct,
|
|
105
|
+
"num_letters_before": len(cliques),
|
|
106
|
+
"num_letters_after": len(columns),
|
|
107
|
+
"num_groups": len(groups),
|
|
108
|
+
"num_edges": len(model.edges),
|
|
109
|
+
"solver_status": "Optimal",
|
|
110
|
+
"objective": minimum,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
return CLDReductionResult(
|
|
114
|
+
letters=letters,
|
|
115
|
+
assignments=assignments,
|
|
116
|
+
stats=stats,
|
|
117
|
+
method=method,
|
|
118
|
+
groups=tuple(groups),
|
|
119
|
+
relationship_preserved=True,
|
|
120
|
+
adjacency=tuple(tuple(bool(value) for value in row) for row in adjacency),
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _validate_solver_controls(
|
|
125
|
+
time_limit: float | None, max_cliques: int | None
|
|
126
|
+
) -> tuple[float | None, int | None]:
|
|
127
|
+
if time_limit is not None and (
|
|
128
|
+
isinstance(time_limit, bool)
|
|
129
|
+
or not isinstance(time_limit, Real)
|
|
130
|
+
or not isfinite(float(time_limit))
|
|
131
|
+
or time_limit <= 0
|
|
132
|
+
):
|
|
133
|
+
msg = "time_limit must be positive when provided"
|
|
134
|
+
raise SolverError(msg)
|
|
135
|
+
if max_cliques is not None and (
|
|
136
|
+
isinstance(max_cliques, bool) or not isinstance(max_cliques, Integral) or max_cliques < 1
|
|
137
|
+
):
|
|
138
|
+
msg = "max_cliques must be a positive integer or None"
|
|
139
|
+
raise SolverError(msg)
|
|
140
|
+
return (
|
|
141
|
+
float(time_limit) if time_limit is not None else None,
|
|
142
|
+
int(max_cliques) if max_cliques is not None else None,
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _build_model(adjacency: np.ndarray, cliques: list[tuple[int, ...]]) -> _Model:
|
|
147
|
+
"""Build the model of docs/algorithm.md section 4."""
|
|
148
|
+
num_groups = adjacency.shape[0]
|
|
149
|
+
members = [(c, g) for c, clique in enumerate(cliques) for g in clique]
|
|
150
|
+
x_index = {member: k for k, member in enumerate(members)}
|
|
151
|
+
cliques_of: list[list[int]] = [[] for _ in range(num_groups)]
|
|
152
|
+
for c, clique in enumerate(cliques):
|
|
153
|
+
for g in clique:
|
|
154
|
+
cliques_of[g].append(c)
|
|
155
|
+
edges = [(i, j) for i in range(num_groups) for j in range(i + 1, num_groups) if adjacency[i, j]]
|
|
156
|
+
|
|
157
|
+
num_x = len(members)
|
|
158
|
+
y_pairs: list[tuple[int, int]] = [] # (edge, clique)
|
|
159
|
+
for e, (i, j) in enumerate(edges):
|
|
160
|
+
shared = sorted(set(cliques_of[i]) & set(cliques_of[j]))
|
|
161
|
+
y_pairs.extend((e, c) for c in shared)
|
|
162
|
+
|
|
163
|
+
rows: list[list[tuple[int, float]]] = []
|
|
164
|
+
row_lower: list[float] = []
|
|
165
|
+
row_upper: list[float] = []
|
|
166
|
+
# Every group is in at least one clique.
|
|
167
|
+
for g in range(num_groups):
|
|
168
|
+
rows.append([(x_index[(c, g)], 1.0) for c in cliques_of[g]])
|
|
169
|
+
row_lower.append(1.0)
|
|
170
|
+
row_upper.append(kHighsInf)
|
|
171
|
+
# Every non-significant edge is covered by at least one clique that holds both ends.
|
|
172
|
+
by_edge: dict[int, list[int]] = {}
|
|
173
|
+
for k, (e, _) in enumerate(y_pairs):
|
|
174
|
+
by_edge.setdefault(e, []).append(num_x + k)
|
|
175
|
+
for e in range(len(edges)):
|
|
176
|
+
rows.append([(col, 1.0) for col in by_edge[e]])
|
|
177
|
+
row_lower.append(1.0)
|
|
178
|
+
row_upper.append(kHighsInf)
|
|
179
|
+
# A covering clique needs both ends: y <= x for each end.
|
|
180
|
+
for k, (e, c) in enumerate(y_pairs):
|
|
181
|
+
i, j = edges[e]
|
|
182
|
+
for g in (i, j):
|
|
183
|
+
rows.append([(num_x + k, 1.0), (x_index[(c, g)], -1.0)])
|
|
184
|
+
row_lower.append(-kHighsInf)
|
|
185
|
+
row_upper.append(0.0)
|
|
186
|
+
|
|
187
|
+
start = [0]
|
|
188
|
+
index: list[int] = []
|
|
189
|
+
value: list[float] = []
|
|
190
|
+
for row in rows:
|
|
191
|
+
for col, coefficient in row:
|
|
192
|
+
index.append(col)
|
|
193
|
+
value.append(coefficient)
|
|
194
|
+
start.append(len(index))
|
|
195
|
+
num_cols = num_x + len(y_pairs)
|
|
196
|
+
cost = np.zeros(num_cols)
|
|
197
|
+
cost[:num_x] = 1.0
|
|
198
|
+
problem = _solver.Problem(
|
|
199
|
+
num_cols=num_cols,
|
|
200
|
+
num_x=num_x,
|
|
201
|
+
cost=cost,
|
|
202
|
+
start=np.asarray(start, dtype=np.int32),
|
|
203
|
+
index=np.asarray(index, dtype=np.int32),
|
|
204
|
+
value=np.asarray(value, dtype=np.float64),
|
|
205
|
+
row_lower=np.asarray(row_lower, dtype=np.float64),
|
|
206
|
+
row_upper=np.asarray(row_upper, dtype=np.float64),
|
|
207
|
+
)
|
|
208
|
+
edge_ends: list[list[tuple[int, int]]] = [[] for _ in edges]
|
|
209
|
+
for e, c in y_pairs:
|
|
210
|
+
edge_ends[e].append((x_index[(c, edges[e][0])], x_index[(c, edges[e][1])]))
|
|
211
|
+
return _Model(
|
|
212
|
+
problem=problem,
|
|
213
|
+
members=members,
|
|
214
|
+
edges=edges,
|
|
215
|
+
group_columns=[[x_index[(c, g)] for c in cliques_of[g]] for g in range(num_groups)],
|
|
216
|
+
edge_ends=edge_ends,
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _solve_canonical(model: _Model, time_limit: float | None) -> tuple[np.ndarray, int]:
|
|
221
|
+
"""Solve once for the minimum, then fix the memberships in (clique, group) order.
|
|
222
|
+
|
|
223
|
+
This is the sequential fixing procedure of docs/algorithm.md section 5. Returns the
|
|
224
|
+
selected x variables (a boolean array) and the minimum number of assignments.
|
|
225
|
+
"""
|
|
226
|
+
problem = model.problem
|
|
227
|
+
num_x = problem.num_x
|
|
228
|
+
deadline = _now() + time_limit if time_limit is not None else None
|
|
229
|
+
col_lower = np.zeros(problem.num_cols)
|
|
230
|
+
col_upper = np.ones(problem.num_cols)
|
|
231
|
+
|
|
232
|
+
outcome = _solve(model, col_lower, col_upper, None, deadline)
|
|
233
|
+
selected = _check_solution(model, outcome, col_lower, col_upper, None)
|
|
234
|
+
minimum = int(round(float(outcome.objective)))
|
|
235
|
+
selected_x = selected
|
|
236
|
+
|
|
237
|
+
for v in range(num_x):
|
|
238
|
+
if selected_x[v]:
|
|
239
|
+
col_lower[v] = 1.0
|
|
240
|
+
continue
|
|
241
|
+
trial_lower = col_lower.copy()
|
|
242
|
+
trial_lower[v] = 1.0
|
|
243
|
+
outcome = _solve(model, trial_lower, col_upper, minimum, deadline)
|
|
244
|
+
if outcome.status == _solver.INFEASIBLE:
|
|
245
|
+
col_upper[v] = 0.0
|
|
246
|
+
continue
|
|
247
|
+
selected_x = _check_solution(model, outcome, trial_lower, col_upper, minimum)
|
|
248
|
+
col_lower = trial_lower
|
|
249
|
+
return selected_x, minimum
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _solve(
|
|
253
|
+
model: _Model,
|
|
254
|
+
col_lower: np.ndarray,
|
|
255
|
+
col_upper: np.ndarray,
|
|
256
|
+
sum_limit: int | None,
|
|
257
|
+
deadline: float | None,
|
|
258
|
+
) -> _solver.Outcome:
|
|
259
|
+
remaining = None
|
|
260
|
+
if deadline is not None:
|
|
261
|
+
remaining = deadline - _now()
|
|
262
|
+
if remaining <= 0:
|
|
263
|
+
msg = "assignment-minimum MILP failed: Time limit reached"
|
|
264
|
+
raise SolverError(msg)
|
|
265
|
+
outcome = _solver.run(
|
|
266
|
+
model.problem, col_lower, col_upper, sum_limit=sum_limit, time_limit=remaining
|
|
267
|
+
)
|
|
268
|
+
if outcome.status == _solver.INFEASIBLE and sum_limit is not None:
|
|
269
|
+
return outcome
|
|
270
|
+
if outcome.status != _solver.OPTIMAL:
|
|
271
|
+
msg = f"assignment-minimum MILP failed: {outcome.text}"
|
|
272
|
+
raise SolverError(msg)
|
|
273
|
+
return outcome
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _check_solution(
|
|
277
|
+
model: _Model,
|
|
278
|
+
outcome: _solver.Outcome,
|
|
279
|
+
col_lower: np.ndarray,
|
|
280
|
+
col_upper: np.ndarray,
|
|
281
|
+
expected_sum: int | None,
|
|
282
|
+
) -> np.ndarray:
|
|
283
|
+
"""Docs/algorithm.md section 6, checks 2 to 5. Returns the rounded x as booleans."""
|
|
284
|
+
num_x = model.problem.num_x
|
|
285
|
+
values = outcome.values
|
|
286
|
+
if values is None or len(values) != model.problem.num_cols:
|
|
287
|
+
raise SolverError(_INVALID_SOLUTION)
|
|
288
|
+
x = np.asarray(values[:num_x], dtype=np.float64)
|
|
289
|
+
# Each membership must be 0 or 1 within the tolerance; an integral 2 or -1 is invalid too.
|
|
290
|
+
near_zero = np.abs(x) <= _INTEGRALITY_TOLERANCE
|
|
291
|
+
near_one = np.abs(x - 1.0) <= _INTEGRALITY_TOLERANCE
|
|
292
|
+
if not np.all(np.isfinite(x) & (near_zero | near_one)):
|
|
293
|
+
raise SolverError(_INVALID_SOLUTION)
|
|
294
|
+
rounded = x > 0.5
|
|
295
|
+
if np.any(rounded & (col_upper[:num_x] < 0.5)) or np.any(~rounded & (col_lower[:num_x] > 0.5)):
|
|
296
|
+
raise SolverError(_INVALID_SOLUTION)
|
|
297
|
+
for columns in model.group_columns:
|
|
298
|
+
if not any(rounded[k] for k in columns):
|
|
299
|
+
raise SolverError(_INVALID_SOLUTION)
|
|
300
|
+
for ends in model.edge_ends:
|
|
301
|
+
if not any(rounded[a] and rounded[b] for a, b in ends):
|
|
302
|
+
raise SolverError(_INVALID_SOLUTION)
|
|
303
|
+
total = int(rounded.sum())
|
|
304
|
+
wanted = expected_sum if expected_sum is not None else int(round(float(outcome.objective)))
|
|
305
|
+
if total != wanted:
|
|
306
|
+
raise SolverError(_INVALID_SOLUTION)
|
|
307
|
+
return rounded
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _selected_columns(
|
|
311
|
+
cliques: list[tuple[int, ...]], model: _Model, selected: np.ndarray
|
|
312
|
+
) -> list[list[int]]:
|
|
313
|
+
"""Members of each clique's letter, dropping empty columns (canonical clique order)."""
|
|
314
|
+
columns: list[list[int]] = [[] for _ in cliques]
|
|
315
|
+
for k, (c, g) in enumerate(model.members):
|
|
316
|
+
if selected[k]:
|
|
317
|
+
columns[c].append(g)
|
|
318
|
+
return [column for column in columns if column]
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _assign_letter_tokens(
|
|
322
|
+
columns: list[list[int]],
|
|
323
|
+
num_groups: int,
|
|
324
|
+
means: pd.Series | None,
|
|
325
|
+
groups: list[str],
|
|
326
|
+
) -> list[tuple[str, ...]]:
|
|
327
|
+
order = sorted(range(len(columns)), key=lambda c: _column_sort_key(columns[c], groups, means))
|
|
328
|
+
labels = make_letter_labels(len(order))
|
|
329
|
+
tokens: list[list[str]] = [[] for _ in range(num_groups)]
|
|
330
|
+
for label, column in zip(labels, order, strict=True):
|
|
331
|
+
for group_index in columns[column]:
|
|
332
|
+
tokens[group_index].append(label)
|
|
333
|
+
return [tuple(value) for value in tokens]
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def _format_letter_tokens(tokens: tuple[str, ...]) -> str:
|
|
337
|
+
if all(len(token) == 1 for token in tokens):
|
|
338
|
+
return "".join(tokens)
|
|
339
|
+
return " ".join(tokens)
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _column_sort_key(
|
|
343
|
+
members: list[int],
|
|
344
|
+
groups: list[str],
|
|
345
|
+
means: pd.Series | None,
|
|
346
|
+
) -> tuple[float, int]:
|
|
347
|
+
lowest = min(members)
|
|
348
|
+
if means is not None:
|
|
349
|
+
highest_mean = max(float(means[groups[index]]) for index in members)
|
|
350
|
+
return (-highest_mean, lowest)
|
|
351
|
+
return (float(lowest), lowest)
|
cld_reducer/api.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Public API for CLD reduction."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping, Sequence
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import pandas as pd
|
|
9
|
+
|
|
10
|
+
from .algorithms.assignment_minimum import reduce_assignment_minimum
|
|
11
|
+
from .exceptions import InvalidInputError
|
|
12
|
+
from .result import CLDReductionResult
|
|
13
|
+
from .validation import (
|
|
14
|
+
adjacency_from_pairs,
|
|
15
|
+
groups_from_pairs,
|
|
16
|
+
normalize_means,
|
|
17
|
+
normalize_pairwise_frame,
|
|
18
|
+
validate_adjacency,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def reduce_letters(
|
|
23
|
+
post_hoc_results: pd.DataFrame | Sequence[Mapping[str, Any]],
|
|
24
|
+
means: Mapping[Any, float] | pd.Series | pd.DataFrame | None = None,
|
|
25
|
+
*,
|
|
26
|
+
method: str = "assignment_minimum",
|
|
27
|
+
group1: str = "group1",
|
|
28
|
+
group2: str = "group2",
|
|
29
|
+
significant: str = "significant",
|
|
30
|
+
time_limit: float | None = None,
|
|
31
|
+
max_cliques: int | None = 10_000,
|
|
32
|
+
) -> CLDReductionResult:
|
|
33
|
+
"""Reduce compact letter assignments from pairwise post-hoc results.
|
|
34
|
+
|
|
35
|
+
Parameters
|
|
36
|
+
----------
|
|
37
|
+
post_hoc_results:
|
|
38
|
+
Pairwise comparison results. Each row must identify two groups and
|
|
39
|
+
whether the comparison is statistically significant.
|
|
40
|
+
means:
|
|
41
|
+
Optional group means used for stable display ordering.
|
|
42
|
+
method:
|
|
43
|
+
Reduction algorithm. Currently only `"assignment_minimum"` is supported.
|
|
44
|
+
group1, group2, significant:
|
|
45
|
+
Column names in `post_hoc_results`.
|
|
46
|
+
time_limit:
|
|
47
|
+
Optional solver time limit in seconds.
|
|
48
|
+
max_cliques:
|
|
49
|
+
Optional cap on maximal cliques to enumerate before failing with a
|
|
50
|
+
controlled solver error. Pass `None` to disable the cap.
|
|
51
|
+
"""
|
|
52
|
+
frame = normalize_pairwise_frame(
|
|
53
|
+
post_hoc_results,
|
|
54
|
+
group1=group1,
|
|
55
|
+
group2=group2,
|
|
56
|
+
significant=significant,
|
|
57
|
+
)
|
|
58
|
+
groups = groups_from_pairs(frame, means)
|
|
59
|
+
normalized_means = normalize_means(means, groups)
|
|
60
|
+
adjacency = adjacency_from_pairs(frame, groups)
|
|
61
|
+
return reduce_from_adjacency(
|
|
62
|
+
adjacency,
|
|
63
|
+
groups=groups,
|
|
64
|
+
means=normalized_means,
|
|
65
|
+
method=method,
|
|
66
|
+
time_limit=time_limit,
|
|
67
|
+
max_cliques=max_cliques,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def reduce_from_adjacency(
|
|
72
|
+
adjacency: Any,
|
|
73
|
+
groups: Sequence[Any] | None = None,
|
|
74
|
+
means: Mapping[Any, float] | pd.Series | pd.DataFrame | None = None,
|
|
75
|
+
*,
|
|
76
|
+
method: str = "assignment_minimum",
|
|
77
|
+
time_limit: float | None = None,
|
|
78
|
+
max_cliques: int | None = 10_000,
|
|
79
|
+
) -> CLDReductionResult:
|
|
80
|
+
"""Reduce compact letters directly from a non-significance adjacency matrix.
|
|
81
|
+
|
|
82
|
+
`adjacency[i, j] == True` means groups `i` and `j` are not significantly
|
|
83
|
+
different and must share at least one letter in the returned CLD.
|
|
84
|
+
"""
|
|
85
|
+
matrix, normalized_groups = validate_adjacency(adjacency, groups)
|
|
86
|
+
normalized_means = normalize_means(means, normalized_groups)
|
|
87
|
+
if method in {"assignment_minimum", "assignment-minimum"}:
|
|
88
|
+
return reduce_assignment_minimum(
|
|
89
|
+
matrix,
|
|
90
|
+
normalized_groups,
|
|
91
|
+
normalized_means,
|
|
92
|
+
method="assignment_minimum",
|
|
93
|
+
time_limit=time_limit,
|
|
94
|
+
max_cliques=max_cliques,
|
|
95
|
+
)
|
|
96
|
+
msg = f"unsupported CLD reduction method: {method!r}"
|
|
97
|
+
raise InvalidInputError(msg)
|