cld-reducer 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cld_reducer/NOTICE ADDED
@@ -0,0 +1,35 @@
1
+ Example data sources and reuse record
2
+
3
+ The MIT license covers the package code. The examples below contain public
4
+ summary data from the cited papers. On 2026-10-10, John Ennis confirmed that
5
+ the data are public and approved their inclusion in these distributions.
6
+ This record does not assign a new license to the source publications.
7
+
8
+ piepho2004_wheat_pairs.csv / piepho2004_wheat
9
+ Source: Piepho (2004), An Algorithm for a Letter-Based Representation of All-
10
+ Pairwise Comparisons, doi:10.1198/1061860043515; reproduced in Table 7 of
11
+ Ennis, Fayle, and Ennis (2012), Assignment-Minimum Clique Coverings,
12
+ doi:10.1145/2133803.2275596.
13
+ Content: significant/non-significant decisions for the 190 unordered pairs of
14
+ 20 wheat treatments. The repository stores these decisions as CSV rows.
15
+ R data-raw/datasets.R reads labels as text and decisions as logical values,
16
+ then saves the same rows as piepho2004_wheat.rda. Python examples copy the CSV.
17
+ Reuse record: maintainer confirmation and approval on 2026-10-10.
18
+
19
+ simple_abc_to_ac_pairs.csv / simple_abc_pairs
20
+ simple_abc_to_ac_means.csv / simple_abc_means
21
+ Source: Ennis, Fayle, and Ennis (2012), Assignment-Minimum Clique Coverings,
22
+ doi:10.1145/2133803.2275596. The R help pages attribute the five-group
23
+ ABC-to-AC teaching example and its means to this paper.
24
+ Content: ten pairwise decisions and five numeric means. R data-raw/datasets.R
25
+ reads the CSV files and saves the corresponding data frames. Python examples
26
+ copy the CSV files.
27
+ Reuse record: maintainer confirmation and approval on 2026-10-10.
28
+
29
+ Retain this source record with the example data.
30
+
31
+ Distribution scope
32
+ The R archive contains both example data sets. The Python source archive contains
33
+ their CSV files. The Python wheel omits the CSV files but includes simple-example
34
+ output from the README in its metadata. The npm archive contains no wheat data,
35
+ but its README reproduces the simple five-group example.
@@ -0,0 +1,16 @@
1
+ """Tools for reducing compact letter displays."""
2
+
3
+ from .api import reduce_from_adjacency, reduce_letters
4
+ from .exceptions import CLDReducerError, InvalidInputError, SolverError
5
+ from .result import CLDReductionResult
6
+
7
+ __version__ = "0.2.0"
8
+
9
+ __all__ = [
10
+ "CLDReducerError",
11
+ "CLDReductionResult",
12
+ "InvalidInputError",
13
+ "SolverError",
14
+ "reduce_from_adjacency",
15
+ "reduce_letters",
16
+ ]
cld_reducer/_solver.py ADDED
@@ -0,0 +1,140 @@
1
+ """HiGHS through highspy, with the settings of docs/algorithm.md section 6."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+
7
+ import highspy
8
+ import numpy as np
9
+
10
+
11
+ @dataclass
12
+ class Settings:
13
+ """Presolve is the one setting that differs between languages (R turns it off).
14
+
15
+ Tests switch it off to run the conformance suite a second time.
16
+ """
17
+
18
+ presolve: str = "on"
19
+
20
+
21
+ settings = Settings()
22
+ # The options of the most recent solve, read back from HiGHS. Tests check them.
23
+ last_options: dict[str, object] = {}
24
+
25
+ OPTION_NAMES = (
26
+ "presolve",
27
+ "mip_rel_gap",
28
+ "mip_abs_gap",
29
+ "primal_feasibility_tolerance",
30
+ "mip_feasibility_tolerance",
31
+ "threads",
32
+ "time_limit",
33
+ )
34
+
35
+ OPTIMAL = "optimal"
36
+ INFEASIBLE = "infeasible"
37
+ TIME_LIMIT = "time_limit"
38
+ FAILED = "failed"
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class Problem:
43
+ """The model of docs/algorithm.md section 4 in row-wise sparse form.
44
+
45
+ The first `num_x` columns are the membership variables x; the rest are the y variables.
46
+ """
47
+
48
+ num_cols: int
49
+ num_x: int
50
+ cost: np.ndarray
51
+ start: np.ndarray
52
+ index: np.ndarray
53
+ value: np.ndarray
54
+ row_lower: np.ndarray
55
+ row_upper: np.ndarray
56
+
57
+
58
+ @dataclass(frozen=True)
59
+ class Outcome:
60
+ """What one solve returned: a status, the HiGHS status text, and the column values."""
61
+
62
+ status: str
63
+ text: str
64
+ values: np.ndarray | None = None
65
+ objective: float | None = None
66
+
67
+
68
+ def run(
69
+ problem: Problem,
70
+ col_lower: np.ndarray,
71
+ col_upper: np.ndarray,
72
+ *,
73
+ sum_limit: int | None = None,
74
+ time_limit: float | None = None,
75
+ ) -> Outcome:
76
+ """Solve the model with the given column bounds.
77
+
78
+ `sum_limit` adds the row `sum(x) <= sum_limit`. `time_limit` is the HiGHS time limit in
79
+ seconds for this solve.
80
+ """
81
+ start, index, value = problem.start, problem.index, problem.value
82
+ row_lower, row_upper = problem.row_lower, problem.row_upper
83
+ if sum_limit is not None:
84
+ extra = np.arange(problem.num_x, dtype=np.int32)
85
+ start = np.concatenate([start, [start[-1] + problem.num_x]]).astype(np.int32)
86
+ index = np.concatenate([index, extra])
87
+ value = np.concatenate([value, np.ones(problem.num_x)])
88
+ row_lower = np.concatenate([row_lower, [-highspy.kHighsInf]])
89
+ row_upper = np.concatenate([row_upper, [float(sum_limit)]])
90
+
91
+ lp = highspy.HighsLp()
92
+ lp.num_col_ = problem.num_cols
93
+ lp.num_row_ = len(row_lower)
94
+ lp.col_cost_ = problem.cost
95
+ lp.col_lower_ = col_lower
96
+ lp.col_upper_ = col_upper
97
+ lp.integrality_ = [highspy.HighsVarType.kInteger] * problem.num_cols
98
+ lp.row_lower_ = row_lower
99
+ lp.row_upper_ = row_upper
100
+ lp.a_matrix_.format_ = highspy.MatrixFormat.kRowwise
101
+ lp.a_matrix_.num_col_ = problem.num_cols
102
+ lp.a_matrix_.num_row_ = len(row_lower)
103
+ lp.a_matrix_.start_ = start
104
+ lp.a_matrix_.index_ = index
105
+ lp.a_matrix_.value_ = value
106
+ lp.sense_ = highspy.ObjSense.kMinimize
107
+
108
+ h = highspy.Highs()
109
+ h.setOptionValue("output_flag", False)
110
+ h.setOptionValue("presolve", settings.presolve)
111
+ h.setOptionValue("mip_rel_gap", 0.0)
112
+ h.setOptionValue("mip_abs_gap", 0.0)
113
+ h.setOptionValue("primal_feasibility_tolerance", 1e-9)
114
+ h.setOptionValue("mip_feasibility_tolerance", 1e-9)
115
+ h.setOptionValue("threads", 1)
116
+ if time_limit is not None:
117
+ h.setOptionValue("time_limit", float(time_limit))
118
+ h.passModel(lp)
119
+ h.run()
120
+
121
+ last_options.clear()
122
+ for name in OPTION_NAMES:
123
+ option = h.getOptionValue(name)
124
+ # highspy 1.15 returns (status, value).
125
+ last_options[name] = option[1] if isinstance(option, tuple) else option
126
+
127
+ status = h.getModelStatus()
128
+ text = h.modelStatusToString(status)
129
+ if status == highspy.HighsModelStatus.kOptimal:
130
+ values = np.asarray(h.getSolution().col_value, dtype=np.float64)
131
+ return Outcome(OPTIMAL, text, values, float(h.getInfo().objective_function_value))
132
+ # No model here is unbounded, so HiGHS's "unbounded or infeasible" means infeasible.
133
+ if status in (
134
+ highspy.HighsModelStatus.kInfeasible,
135
+ highspy.HighsModelStatus.kUnboundedOrInfeasible,
136
+ ):
137
+ return Outcome(INFEASIBLE, text)
138
+ if status == highspy.HighsModelStatus.kTimeLimit:
139
+ return Outcome(TIME_LIMIT, text)
140
+ return Outcome(FAILED, text)
@@ -0,0 +1 @@
1
+ """CLD reduction algorithm implementations."""
@@ -0,0 +1,351 @@
1
+ """Assignment-minimum compact letter display reduction.
2
+
3
+ Solves the assignment-minimum clique covering problem defined in Ennis, Fayle,
4
+ & Ennis (2012), "Assignment-Minimum Clique Coverings", ACM JEA 17, Art. 1.5
5
+ (https://doi.org/10.1145/2133803.2275596). The paper uses a backtracking
6
+ algorithm (FIND-AM); this module solves the same problem as a binary
7
+ mixed-integer program with HiGHS, then picks the canonical optimum of
8
+ docs/algorithm.md section 5 so that every implementation returns the same display.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import time
14
+ from dataclasses import dataclass
15
+ from math import isfinite
16
+ from numbers import Integral, Real
17
+
18
+ import numpy as np
19
+ import pandas as pd
20
+ from highspy import kHighsInf
21
+
22
+ from .. import _solver
23
+ from ..cliques import maximal_cliques
24
+ from ..exceptions import SolverError
25
+ from ..labels import make_letter_labels
26
+ from ..result import CLDReductionResult
27
+ from ..validation import (
28
+ count_assignments,
29
+ normalize_means,
30
+ reconstruct_adjacency_from_assignments,
31
+ validate_adjacency,
32
+ )
33
+
34
+ # The clock behind the shared time budget. Tests replace it.
35
+ _now = time.monotonic
36
+
37
+ _INTEGRALITY_TOLERANCE = 1e-6
38
+ _INVALID_SOLUTION = "HiGHS returned an invalid solution"
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class _Model:
43
+ problem: _solver.Problem
44
+ members: list[tuple[int, int]] # (clique, group) of each x variable, in canonical order
45
+ edges: list[tuple[int, int]]
46
+ group_columns: list[list[int]] # x variables of each group
47
+ edge_ends: list[list[tuple[int, int]]] # per edge: the two x variables of each covering clique
48
+
49
+
50
+ def reduce_assignment_minimum(
51
+ adjacency: np.ndarray,
52
+ groups: list[str],
53
+ means: pd.Series | None = None,
54
+ *,
55
+ method: str = "assignment_minimum",
56
+ time_limit: float | None = None,
57
+ max_cliques: int | None = 10_000,
58
+ ) -> CLDReductionResult:
59
+ """Reduce a CLD by minimizing total letter assignments.
60
+
61
+ Parameters
62
+ ----------
63
+ adjacency:
64
+ Symmetric boolean matrix where `True` means two groups are not
65
+ significantly different and may share a letter.
66
+ groups:
67
+ Group labels in matrix order.
68
+ means:
69
+ Optional group means used only to produce stable, mean-ordered letters.
70
+ method:
71
+ Method label stored in the result metadata.
72
+ time_limit:
73
+ Optional time budget in seconds, shared by all solves of this call.
74
+ max_cliques:
75
+ Optional cap on maximal cliques to enumerate before failing with a
76
+ controlled `SolverError`. Pass `None` to disable the cap.
77
+ """
78
+ adjacency, groups = validate_adjacency(adjacency, groups)
79
+ means = normalize_means(means, groups)
80
+ time_limit, max_cliques = _validate_solver_controls(time_limit, max_cliques)
81
+
82
+ cliques = maximal_cliques(adjacency, max_cliques=max_cliques)
83
+ model = _build_model(adjacency, cliques)
84
+ selected, minimum = _solve_canonical(model, time_limit)
85
+ columns = _selected_columns(cliques, model, selected)
86
+ tokens = _assign_letter_tokens(columns, len(groups), means, groups)
87
+ assignments = {group: tokens[index] for index, group in enumerate(groups)}
88
+ letters = {group: _format_letter_tokens(value) for group, value in assignments.items()}
89
+ reconstructed = reconstruct_adjacency_from_assignments(assignments, groups)
90
+ if not np.array_equal(reconstructed, adjacency):
91
+ msg = "optimized letters did not preserve the input pairwise relationships"
92
+ raise SolverError(msg)
93
+
94
+ assignments_before = len(model.members)
95
+ assignments_after = count_assignments(assignments)
96
+ reduction_pct = (
97
+ (assignments_before - assignments_after) / assignments_before * 100
98
+ if assignments_before
99
+ else 0.0
100
+ )
101
+ stats = {
102
+ "assignments_before": assignments_before,
103
+ "assignments_after": int(assignments_after),
104
+ "reduction_pct": reduction_pct,
105
+ "num_letters_before": len(cliques),
106
+ "num_letters_after": len(columns),
107
+ "num_groups": len(groups),
108
+ "num_edges": len(model.edges),
109
+ "solver_status": "Optimal",
110
+ "objective": minimum,
111
+ }
112
+
113
+ return CLDReductionResult(
114
+ letters=letters,
115
+ assignments=assignments,
116
+ stats=stats,
117
+ method=method,
118
+ groups=tuple(groups),
119
+ relationship_preserved=True,
120
+ adjacency=tuple(tuple(bool(value) for value in row) for row in adjacency),
121
+ )
122
+
123
+
124
+ def _validate_solver_controls(
125
+ time_limit: float | None, max_cliques: int | None
126
+ ) -> tuple[float | None, int | None]:
127
+ if time_limit is not None and (
128
+ isinstance(time_limit, bool)
129
+ or not isinstance(time_limit, Real)
130
+ or not isfinite(float(time_limit))
131
+ or time_limit <= 0
132
+ ):
133
+ msg = "time_limit must be positive when provided"
134
+ raise SolverError(msg)
135
+ if max_cliques is not None and (
136
+ isinstance(max_cliques, bool) or not isinstance(max_cliques, Integral) or max_cliques < 1
137
+ ):
138
+ msg = "max_cliques must be a positive integer or None"
139
+ raise SolverError(msg)
140
+ return (
141
+ float(time_limit) if time_limit is not None else None,
142
+ int(max_cliques) if max_cliques is not None else None,
143
+ )
144
+
145
+
146
+ def _build_model(adjacency: np.ndarray, cliques: list[tuple[int, ...]]) -> _Model:
147
+ """Build the model of docs/algorithm.md section 4."""
148
+ num_groups = adjacency.shape[0]
149
+ members = [(c, g) for c, clique in enumerate(cliques) for g in clique]
150
+ x_index = {member: k for k, member in enumerate(members)}
151
+ cliques_of: list[list[int]] = [[] for _ in range(num_groups)]
152
+ for c, clique in enumerate(cliques):
153
+ for g in clique:
154
+ cliques_of[g].append(c)
155
+ edges = [(i, j) for i in range(num_groups) for j in range(i + 1, num_groups) if adjacency[i, j]]
156
+
157
+ num_x = len(members)
158
+ y_pairs: list[tuple[int, int]] = [] # (edge, clique)
159
+ for e, (i, j) in enumerate(edges):
160
+ shared = sorted(set(cliques_of[i]) & set(cliques_of[j]))
161
+ y_pairs.extend((e, c) for c in shared)
162
+
163
+ rows: list[list[tuple[int, float]]] = []
164
+ row_lower: list[float] = []
165
+ row_upper: list[float] = []
166
+ # Every group is in at least one clique.
167
+ for g in range(num_groups):
168
+ rows.append([(x_index[(c, g)], 1.0) for c in cliques_of[g]])
169
+ row_lower.append(1.0)
170
+ row_upper.append(kHighsInf)
171
+ # Every non-significant edge is covered by at least one clique that holds both ends.
172
+ by_edge: dict[int, list[int]] = {}
173
+ for k, (e, _) in enumerate(y_pairs):
174
+ by_edge.setdefault(e, []).append(num_x + k)
175
+ for e in range(len(edges)):
176
+ rows.append([(col, 1.0) for col in by_edge[e]])
177
+ row_lower.append(1.0)
178
+ row_upper.append(kHighsInf)
179
+ # A covering clique needs both ends: y <= x for each end.
180
+ for k, (e, c) in enumerate(y_pairs):
181
+ i, j = edges[e]
182
+ for g in (i, j):
183
+ rows.append([(num_x + k, 1.0), (x_index[(c, g)], -1.0)])
184
+ row_lower.append(-kHighsInf)
185
+ row_upper.append(0.0)
186
+
187
+ start = [0]
188
+ index: list[int] = []
189
+ value: list[float] = []
190
+ for row in rows:
191
+ for col, coefficient in row:
192
+ index.append(col)
193
+ value.append(coefficient)
194
+ start.append(len(index))
195
+ num_cols = num_x + len(y_pairs)
196
+ cost = np.zeros(num_cols)
197
+ cost[:num_x] = 1.0
198
+ problem = _solver.Problem(
199
+ num_cols=num_cols,
200
+ num_x=num_x,
201
+ cost=cost,
202
+ start=np.asarray(start, dtype=np.int32),
203
+ index=np.asarray(index, dtype=np.int32),
204
+ value=np.asarray(value, dtype=np.float64),
205
+ row_lower=np.asarray(row_lower, dtype=np.float64),
206
+ row_upper=np.asarray(row_upper, dtype=np.float64),
207
+ )
208
+ edge_ends: list[list[tuple[int, int]]] = [[] for _ in edges]
209
+ for e, c in y_pairs:
210
+ edge_ends[e].append((x_index[(c, edges[e][0])], x_index[(c, edges[e][1])]))
211
+ return _Model(
212
+ problem=problem,
213
+ members=members,
214
+ edges=edges,
215
+ group_columns=[[x_index[(c, g)] for c in cliques_of[g]] for g in range(num_groups)],
216
+ edge_ends=edge_ends,
217
+ )
218
+
219
+
220
+ def _solve_canonical(model: _Model, time_limit: float | None) -> tuple[np.ndarray, int]:
221
+ """Solve once for the minimum, then fix the memberships in (clique, group) order.
222
+
223
+ This is the sequential fixing procedure of docs/algorithm.md section 5. Returns the
224
+ selected x variables (a boolean array) and the minimum number of assignments.
225
+ """
226
+ problem = model.problem
227
+ num_x = problem.num_x
228
+ deadline = _now() + time_limit if time_limit is not None else None
229
+ col_lower = np.zeros(problem.num_cols)
230
+ col_upper = np.ones(problem.num_cols)
231
+
232
+ outcome = _solve(model, col_lower, col_upper, None, deadline)
233
+ selected = _check_solution(model, outcome, col_lower, col_upper, None)
234
+ minimum = int(round(float(outcome.objective)))
235
+ selected_x = selected
236
+
237
+ for v in range(num_x):
238
+ if selected_x[v]:
239
+ col_lower[v] = 1.0
240
+ continue
241
+ trial_lower = col_lower.copy()
242
+ trial_lower[v] = 1.0
243
+ outcome = _solve(model, trial_lower, col_upper, minimum, deadline)
244
+ if outcome.status == _solver.INFEASIBLE:
245
+ col_upper[v] = 0.0
246
+ continue
247
+ selected_x = _check_solution(model, outcome, trial_lower, col_upper, minimum)
248
+ col_lower = trial_lower
249
+ return selected_x, minimum
250
+
251
+
252
+ def _solve(
253
+ model: _Model,
254
+ col_lower: np.ndarray,
255
+ col_upper: np.ndarray,
256
+ sum_limit: int | None,
257
+ deadline: float | None,
258
+ ) -> _solver.Outcome:
259
+ remaining = None
260
+ if deadline is not None:
261
+ remaining = deadline - _now()
262
+ if remaining <= 0:
263
+ msg = "assignment-minimum MILP failed: Time limit reached"
264
+ raise SolverError(msg)
265
+ outcome = _solver.run(
266
+ model.problem, col_lower, col_upper, sum_limit=sum_limit, time_limit=remaining
267
+ )
268
+ if outcome.status == _solver.INFEASIBLE and sum_limit is not None:
269
+ return outcome
270
+ if outcome.status != _solver.OPTIMAL:
271
+ msg = f"assignment-minimum MILP failed: {outcome.text}"
272
+ raise SolverError(msg)
273
+ return outcome
274
+
275
+
276
+ def _check_solution(
277
+ model: _Model,
278
+ outcome: _solver.Outcome,
279
+ col_lower: np.ndarray,
280
+ col_upper: np.ndarray,
281
+ expected_sum: int | None,
282
+ ) -> np.ndarray:
283
+ """Docs/algorithm.md section 6, checks 2 to 5. Returns the rounded x as booleans."""
284
+ num_x = model.problem.num_x
285
+ values = outcome.values
286
+ if values is None or len(values) != model.problem.num_cols:
287
+ raise SolverError(_INVALID_SOLUTION)
288
+ x = np.asarray(values[:num_x], dtype=np.float64)
289
+ # Each membership must be 0 or 1 within the tolerance; an integral 2 or -1 is invalid too.
290
+ near_zero = np.abs(x) <= _INTEGRALITY_TOLERANCE
291
+ near_one = np.abs(x - 1.0) <= _INTEGRALITY_TOLERANCE
292
+ if not np.all(np.isfinite(x) & (near_zero | near_one)):
293
+ raise SolverError(_INVALID_SOLUTION)
294
+ rounded = x > 0.5
295
+ if np.any(rounded & (col_upper[:num_x] < 0.5)) or np.any(~rounded & (col_lower[:num_x] > 0.5)):
296
+ raise SolverError(_INVALID_SOLUTION)
297
+ for columns in model.group_columns:
298
+ if not any(rounded[k] for k in columns):
299
+ raise SolverError(_INVALID_SOLUTION)
300
+ for ends in model.edge_ends:
301
+ if not any(rounded[a] and rounded[b] for a, b in ends):
302
+ raise SolverError(_INVALID_SOLUTION)
303
+ total = int(rounded.sum())
304
+ wanted = expected_sum if expected_sum is not None else int(round(float(outcome.objective)))
305
+ if total != wanted:
306
+ raise SolverError(_INVALID_SOLUTION)
307
+ return rounded
308
+
309
+
310
+ def _selected_columns(
311
+ cliques: list[tuple[int, ...]], model: _Model, selected: np.ndarray
312
+ ) -> list[list[int]]:
313
+ """Members of each clique's letter, dropping empty columns (canonical clique order)."""
314
+ columns: list[list[int]] = [[] for _ in cliques]
315
+ for k, (c, g) in enumerate(model.members):
316
+ if selected[k]:
317
+ columns[c].append(g)
318
+ return [column for column in columns if column]
319
+
320
+
321
+ def _assign_letter_tokens(
322
+ columns: list[list[int]],
323
+ num_groups: int,
324
+ means: pd.Series | None,
325
+ groups: list[str],
326
+ ) -> list[tuple[str, ...]]:
327
+ order = sorted(range(len(columns)), key=lambda c: _column_sort_key(columns[c], groups, means))
328
+ labels = make_letter_labels(len(order))
329
+ tokens: list[list[str]] = [[] for _ in range(num_groups)]
330
+ for label, column in zip(labels, order, strict=True):
331
+ for group_index in columns[column]:
332
+ tokens[group_index].append(label)
333
+ return [tuple(value) for value in tokens]
334
+
335
+
336
+ def _format_letter_tokens(tokens: tuple[str, ...]) -> str:
337
+ if all(len(token) == 1 for token in tokens):
338
+ return "".join(tokens)
339
+ return " ".join(tokens)
340
+
341
+
342
+ def _column_sort_key(
343
+ members: list[int],
344
+ groups: list[str],
345
+ means: pd.Series | None,
346
+ ) -> tuple[float, int]:
347
+ lowest = min(members)
348
+ if means is not None:
349
+ highest_mean = max(float(means[groups[index]]) for index in members)
350
+ return (-highest_mean, lowest)
351
+ return (float(lowest), lowest)
cld_reducer/api.py ADDED
@@ -0,0 +1,97 @@
1
+ """Public API for CLD reduction."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping, Sequence
6
+ from typing import Any
7
+
8
+ import pandas as pd
9
+
10
+ from .algorithms.assignment_minimum import reduce_assignment_minimum
11
+ from .exceptions import InvalidInputError
12
+ from .result import CLDReductionResult
13
+ from .validation import (
14
+ adjacency_from_pairs,
15
+ groups_from_pairs,
16
+ normalize_means,
17
+ normalize_pairwise_frame,
18
+ validate_adjacency,
19
+ )
20
+
21
+
22
+ def reduce_letters(
23
+ post_hoc_results: pd.DataFrame | Sequence[Mapping[str, Any]],
24
+ means: Mapping[Any, float] | pd.Series | pd.DataFrame | None = None,
25
+ *,
26
+ method: str = "assignment_minimum",
27
+ group1: str = "group1",
28
+ group2: str = "group2",
29
+ significant: str = "significant",
30
+ time_limit: float | None = None,
31
+ max_cliques: int | None = 10_000,
32
+ ) -> CLDReductionResult:
33
+ """Reduce compact letter assignments from pairwise post-hoc results.
34
+
35
+ Parameters
36
+ ----------
37
+ post_hoc_results:
38
+ Pairwise comparison results. Each row must identify two groups and
39
+ whether the comparison is statistically significant.
40
+ means:
41
+ Optional group means used for stable display ordering.
42
+ method:
43
+ Reduction algorithm. Currently only `"assignment_minimum"` is supported.
44
+ group1, group2, significant:
45
+ Column names in `post_hoc_results`.
46
+ time_limit:
47
+ Optional solver time limit in seconds.
48
+ max_cliques:
49
+ Optional cap on maximal cliques to enumerate before failing with a
50
+ controlled solver error. Pass `None` to disable the cap.
51
+ """
52
+ frame = normalize_pairwise_frame(
53
+ post_hoc_results,
54
+ group1=group1,
55
+ group2=group2,
56
+ significant=significant,
57
+ )
58
+ groups = groups_from_pairs(frame, means)
59
+ normalized_means = normalize_means(means, groups)
60
+ adjacency = adjacency_from_pairs(frame, groups)
61
+ return reduce_from_adjacency(
62
+ adjacency,
63
+ groups=groups,
64
+ means=normalized_means,
65
+ method=method,
66
+ time_limit=time_limit,
67
+ max_cliques=max_cliques,
68
+ )
69
+
70
+
71
+ def reduce_from_adjacency(
72
+ adjacency: Any,
73
+ groups: Sequence[Any] | None = None,
74
+ means: Mapping[Any, float] | pd.Series | pd.DataFrame | None = None,
75
+ *,
76
+ method: str = "assignment_minimum",
77
+ time_limit: float | None = None,
78
+ max_cliques: int | None = 10_000,
79
+ ) -> CLDReductionResult:
80
+ """Reduce compact letters directly from a non-significance adjacency matrix.
81
+
82
+ `adjacency[i, j] == True` means groups `i` and `j` are not significantly
83
+ different and must share at least one letter in the returned CLD.
84
+ """
85
+ matrix, normalized_groups = validate_adjacency(adjacency, groups)
86
+ normalized_means = normalize_means(means, normalized_groups)
87
+ if method in {"assignment_minimum", "assignment-minimum"}:
88
+ return reduce_assignment_minimum(
89
+ matrix,
90
+ normalized_groups,
91
+ normalized_means,
92
+ method="assignment_minimum",
93
+ time_limit=time_limit,
94
+ max_cliques=max_cliques,
95
+ )
96
+ msg = f"unsupported CLD reduction method: {method!r}"
97
+ raise InvalidInputError(msg)