cld-reducer 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cld_reducer-0.2.0/.gitignore +30 -0
- cld_reducer-0.2.0/LICENSE +21 -0
- cld_reducer-0.2.0/NOTICE +35 -0
- cld_reducer-0.2.0/PKG-INFO +164 -0
- cld_reducer-0.2.0/README.md +130 -0
- cld_reducer-0.2.0/examples/piepho2004_wheat.py +40 -0
- cld_reducer-0.2.0/examples/piepho2004_wheat_pairs.csv +191 -0
- cld_reducer-0.2.0/examples/simple_abc_to_ac.py +27 -0
- cld_reducer-0.2.0/examples/simple_abc_to_ac_means.csv +6 -0
- cld_reducer-0.2.0/examples/simple_abc_to_ac_pairs.csv +11 -0
- cld_reducer-0.2.0/pyproject.toml +76 -0
- cld_reducer-0.2.0/src/cld_reducer/__init__.py +16 -0
- cld_reducer-0.2.0/src/cld_reducer/_solver.py +140 -0
- cld_reducer-0.2.0/src/cld_reducer/algorithms/__init__.py +1 -0
- cld_reducer-0.2.0/src/cld_reducer/algorithms/assignment_minimum.py +351 -0
- cld_reducer-0.2.0/src/cld_reducer/api.py +97 -0
- cld_reducer-0.2.0/src/cld_reducer/cli.py +79 -0
- cld_reducer-0.2.0/src/cld_reducer/cliques.py +53 -0
- cld_reducer-0.2.0/src/cld_reducer/exceptions.py +13 -0
- cld_reducer-0.2.0/src/cld_reducer/labels.py +33 -0
- cld_reducer-0.2.0/src/cld_reducer/result.py +34 -0
- cld_reducer-0.2.0/src/cld_reducer/validation.py +290 -0
- cld_reducer-0.2.0/tests/conftest.py +36 -0
- cld_reducer-0.2.0/tests/test_cli.py +127 -0
- cld_reducer-0.2.0/tests/test_cliques.py +82 -0
- cld_reducer-0.2.0/tests/test_conformance.py +120 -0
- cld_reducer-0.2.0/tests/test_input_rules.py +156 -0
- cld_reducer-0.2.0/tests/test_relationship_preservation.py +59 -0
- cld_reducer-0.2.0/tests/test_simple_example.py +57 -0
- cld_reducer-0.2.0/tests/test_solver.py +288 -0
- cld_reducer-0.2.0/tests/test_validation.py +132 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
.DS_Store
|
|
2
|
+
.venv/
|
|
3
|
+
__pycache__/
|
|
4
|
+
*.py[cod]
|
|
5
|
+
.pytest_cache/
|
|
6
|
+
.ruff_cache/
|
|
7
|
+
.mypy_cache/
|
|
8
|
+
build/
|
|
9
|
+
dist/
|
|
10
|
+
*.egg-info/
|
|
11
|
+
htmlcov/
|
|
12
|
+
.coverage
|
|
13
|
+
*.csv.tmp
|
|
14
|
+
|
|
15
|
+
# R
|
|
16
|
+
.Rproj.user
|
|
17
|
+
.Rhistory
|
|
18
|
+
.RData
|
|
19
|
+
.Ruserdata
|
|
20
|
+
*.tar.gz
|
|
21
|
+
*.Rcheck/
|
|
22
|
+
|
|
23
|
+
# Python
|
|
24
|
+
python/dist/
|
|
25
|
+
python/build/
|
|
26
|
+
|
|
27
|
+
# JavaScript
|
|
28
|
+
node_modules/
|
|
29
|
+
js/dist/
|
|
30
|
+
*.tgz
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Aigora
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
cld_reducer-0.2.0/NOTICE
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
Example data sources and reuse record
|
|
2
|
+
|
|
3
|
+
The MIT license covers the package code. The examples below contain public
|
|
4
|
+
summary data from the cited papers. On 2026-10-10, John Ennis confirmed that
|
|
5
|
+
the data are public and approved their inclusion in these distributions.
|
|
6
|
+
This record does not assign a new license to the source publications.
|
|
7
|
+
|
|
8
|
+
piepho2004_wheat_pairs.csv / piepho2004_wheat
|
|
9
|
+
Source: Piepho (2004), An Algorithm for a Letter-Based Representation of All-
|
|
10
|
+
Pairwise Comparisons, doi:10.1198/1061860043515; reproduced in Table 7 of
|
|
11
|
+
Ennis, Fayle, and Ennis (2012), Assignment-Minimum Clique Coverings,
|
|
12
|
+
doi:10.1145/2133803.2275596.
|
|
13
|
+
Content: significant/non-significant decisions for the 190 unordered pairs of
|
|
14
|
+
20 wheat treatments. The repository stores these decisions as CSV rows.
|
|
15
|
+
R data-raw/datasets.R reads labels as text and decisions as logical values,
|
|
16
|
+
then saves the same rows as piepho2004_wheat.rda. Python examples copy the CSV.
|
|
17
|
+
Reuse record: maintainer confirmation and approval on 2026-10-10.
|
|
18
|
+
|
|
19
|
+
simple_abc_to_ac_pairs.csv / simple_abc_pairs
|
|
20
|
+
simple_abc_to_ac_means.csv / simple_abc_means
|
|
21
|
+
Source: Ennis, Fayle, and Ennis (2012), Assignment-Minimum Clique Coverings,
|
|
22
|
+
doi:10.1145/2133803.2275596. The R help pages attribute the five-group
|
|
23
|
+
ABC-to-AC teaching example and its means to this paper.
|
|
24
|
+
Content: ten pairwise decisions and five numeric means. R data-raw/datasets.R
|
|
25
|
+
reads the CSV files and saves the corresponding data frames. Python examples
|
|
26
|
+
copy the CSV files.
|
|
27
|
+
Reuse record: maintainer confirmation and approval on 2026-10-10.
|
|
28
|
+
|
|
29
|
+
Retain this source record with the example data.
|
|
30
|
+
|
|
31
|
+
Distribution scope
|
|
32
|
+
The R archive contains both example data sets. The Python source archive contains
|
|
33
|
+
their CSV files. The Python wheel omits the CSV files but includes simple-example
|
|
34
|
+
output from the README in its metadata. The npm archive contains no wheat data,
|
|
35
|
+
but its README reproduces the simple five-group example.
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: cld-reducer
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Reduce compact letter displays while preserving pairwise statistical relationships
|
|
5
|
+
Project-URL: Homepage, https://github.com/aigorahub/cld-reducer
|
|
6
|
+
Project-URL: Repository, https://github.com/aigorahub/cld-reducer
|
|
7
|
+
Project-URL: Source, https://github.com/aigorahub/cld-reducer/tree/main/python
|
|
8
|
+
Project-URL: Issues, https://github.com/aigorahub/cld-reducer/issues
|
|
9
|
+
Author: John Ennis
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: compact-letter-display,mixed-integer-programming,post-hoc,sensometrics,statistics
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Requires-Dist: highspy<1.16,>=1.15.1
|
|
24
|
+
Requires-Dist: numpy>=1.24
|
|
25
|
+
Requires-Dist: pandas>=2.0
|
|
26
|
+
Provides-Extra: dev
|
|
27
|
+
Requires-Dist: build>=1.2; extra == 'dev'
|
|
28
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
29
|
+
Requires-Dist: pyyaml>=6.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: ruff>=0.8; extra == 'dev'
|
|
31
|
+
Requires-Dist: tomli>=2.0; (python_version < '3.11') and extra == 'dev'
|
|
32
|
+
Requires-Dist: twine>=6.0; extra == 'dev'
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+
# cld-reducer for Python
|
|
36
|
+
|
|
37
|
+
Reduce compact letter displays (CLDs) while preserving the pairwise statistical relationships they encode. This is the Python package of the [cld-reducer repository](https://github.com/aigorahub/cld-reducer), which also holds an R package and a JavaScript package. All three solve the assignment-minimum clique covering of Ennis, Fayle, and Ennis (2012), <https://doi.org/10.1145/2133803.2275596>, as a mixed-integer program with [HiGHS](https://highs.dev), and they return the same display for the same input. The rules they follow are in `docs/algorithm.md` in the repository.
|
|
38
|
+
|
|
39
|
+
The solver is HiGHS through `highspy`. The dependencies are `highspy`, NumPy, and pandas.
|
|
40
|
+
|
|
41
|
+
## Installation
|
|
42
|
+
|
|
43
|
+
Install from PyPI:
|
|
44
|
+
|
|
45
|
+
```sh
|
|
46
|
+
pip install cld-reducer
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
To install the development version from the repository:
|
|
50
|
+
|
|
51
|
+
```sh
|
|
52
|
+
pip install "git+https://github.com/aigorahub/cld-reducer.git#subdirectory=python"
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
The `#subdirectory=python` part is needed because the Python package lives in `python/`. Python 3.10 or later. The module is `cld_reducer` and the command line tool is `cld-reduce`.
|
|
56
|
+
|
|
57
|
+
For development, work from this folder:
|
|
58
|
+
|
|
59
|
+
```sh
|
|
60
|
+
cd python
|
|
61
|
+
python -m venv .venv
|
|
62
|
+
source .venv/bin/activate
|
|
63
|
+
python -m pip install -e ".[dev]"
|
|
64
|
+
ruff check . && ruff format --check . && pytest
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Usage
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
import pandas as pd
|
|
71
|
+
|
|
72
|
+
from cld_reducer import reduce_letters
|
|
73
|
+
|
|
74
|
+
# One row for each pair. True means the groups differ significantly.
|
|
75
|
+
pairs = pd.DataFrame(
|
|
76
|
+
[
|
|
77
|
+
("1", "2", False),
|
|
78
|
+
("1", "3", False),
|
|
79
|
+
("1", "4", True),
|
|
80
|
+
("1", "5", True),
|
|
81
|
+
("2", "3", False),
|
|
82
|
+
("2", "4", False),
|
|
83
|
+
("2", "5", True),
|
|
84
|
+
("3", "4", False),
|
|
85
|
+
("3", "5", False),
|
|
86
|
+
("4", "5", False),
|
|
87
|
+
],
|
|
88
|
+
columns=["group1", "group2", "significant"],
|
|
89
|
+
)
|
|
90
|
+
means = pd.DataFrame({"group": ["1", "2", "3", "4", "5"], "mean": [3.73, 3.57, 3.46, 3.33, 3.30]})
|
|
91
|
+
|
|
92
|
+
result = reduce_letters(pairs, means)
|
|
93
|
+
print(result.letters)
|
|
94
|
+
# {'1': 'A', '2': 'AB', '3': 'AC', '4': 'BC', '5': 'C'}
|
|
95
|
+
print(result.to_frame())
|
|
96
|
+
# group letters assignments
|
|
97
|
+
# 0 1 A A
|
|
98
|
+
# 1 2 AB A B
|
|
99
|
+
# 2 3 AC A C
|
|
100
|
+
# 3 4 BC B C
|
|
101
|
+
# 4 5 C C
|
|
102
|
+
print(result.stats)
|
|
103
|
+
# {'assignments_before': 9, 'assignments_after': 8, 'reduction_pct': 11.11111111111111, 'num_letters_before': 3, 'num_letters_after': 3, 'num_groups': 5, 'num_edges': 7, 'solver_status': 'Optimal', 'objective': 8}
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
This example works after package installation and needs no external files. `reduce_letters` takes complete pairwise results: one row for each pair of groups, with a boolean column that tells whether the pair differs significantly. Missing pairs are rejected. `means` is optional and orders the groups and the letters. If you already have the non-significance matrix, use `reduce_from_adjacency`; there, `True` means two groups are not significantly different and must share a letter.
|
|
107
|
+
|
|
108
|
+
The result, a `CLDReductionResult`, has:
|
|
109
|
+
|
|
110
|
+
- `letters`: the display of each group, for example `"AC"` (tokens are joined with spaces after `Z`, as in `"Z AA"`)
|
|
111
|
+
- `assignments`: the letters of each group as a tuple, the safe machine-readable form
|
|
112
|
+
- `stats`: the counts before and after the reduction and the solver status; `reduction_pct` is not rounded
|
|
113
|
+
- `relationship_preserved`: always `True` for a returned result
|
|
114
|
+
- `to_frame()`: a tidy `pandas.DataFrame`
|
|
115
|
+
|
|
116
|
+
The public names are `reduce_letters`, `reduce_from_adjacency`, `CLDReductionResult`, `CLDReducerError`, `InvalidInputError`, and `SolverError`.
|
|
117
|
+
|
|
118
|
+
When several displays have the same, smallest number of assignments, the package returns the canonical one defined in `docs/algorithm.md`, so R, Python, and JavaScript agree. `time_limit` is one time budget in seconds for all solves of a call, and `max_cliques` (10,000 by default; `None` removes it) bounds the number of maximal cliques.
|
|
119
|
+
|
|
120
|
+
## Command line
|
|
121
|
+
|
|
122
|
+
```sh
|
|
123
|
+
cld-reduce examples/simple_abc_to_ac_pairs.csv \
|
|
124
|
+
--means examples/simple_abc_to_ac_means.csv \
|
|
125
|
+
--time-limit 30 \
|
|
126
|
+
--out reduced.csv
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
The paths above refer to files in a checkout or an extracted source archive. For your own data, pass your CSV paths. The pairs file has the columns `group1`, `group2`, and `significant`; the optional means file has `group` and `mean`. The output CSV has the reduced letters and the summary statistics. Other flags: `--group1`, `--group2`, `--significant`, `--max-cliques`, `--no-max-cliques`, and `--method`.
|
|
130
|
+
|
|
131
|
+
## Examples
|
|
132
|
+
|
|
133
|
+
The following scripts and CSV files are included in a repository checkout or the source archive, but not in the installed wheel. Run them from `python/` in a checkout, or from the root of an extracted source archive:
|
|
134
|
+
|
|
135
|
+
```sh
|
|
136
|
+
python examples/simple_abc_to_ac.py
|
|
137
|
+
python examples/piepho2004_wheat.py
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
The CSV files in `examples/` are identical copies of the files in `conformance/data/` at the repository root.
|
|
141
|
+
|
|
142
|
+
## Changes in 0.2.0
|
|
143
|
+
|
|
144
|
+
The 0.2.0 release moved the package to `python/` and the solver to `highspy`, which can change the display for inputs with several minimal displays and can rename letters. See the root `NEWS.md` for the full list.
|
|
145
|
+
|
|
146
|
+
## License
|
|
147
|
+
|
|
148
|
+
MIT. See `LICENSE`.
|
|
149
|
+
|
|
150
|
+
## Input and release notes
|
|
151
|
+
|
|
152
|
+
CSV label columns preserve their exact text, including `001`, `NA`, `NaN`, and
|
|
153
|
+
empty strings. A blank CSV label is an empty-string label; missing values in API
|
|
154
|
+
objects still raise `InvalidInputError`. Means must remain finite numbers.
|
|
155
|
+
The means table uses `group` and `mean` when both exist, or its first two columns.
|
|
156
|
+
A list of pair rows preserves each numeric label before conversion to text.
|
|
157
|
+
Uneven adjacency rows raise the package's square-matrix error.
|
|
158
|
+
|
|
159
|
+
The source distribution contains example scripts and CSV files. The wheel
|
|
160
|
+
contains the importable package and its license and data notice. Shared
|
|
161
|
+
conformance fixtures remain in the repository, so source-distribution tests skip
|
|
162
|
+
those cases when the fixtures are absent. CI tests both installed distributions
|
|
163
|
+
outside the checkout. See `docs/releasing.md` in the repository for release steps.
|
|
164
|
+
Example data reuse remains subject to the conditions in `NOTICE`.
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# cld-reducer for Python
|
|
2
|
+
|
|
3
|
+
Reduce compact letter displays (CLDs) while preserving the pairwise statistical relationships they encode. This is the Python package of the [cld-reducer repository](https://github.com/aigorahub/cld-reducer), which also holds an R package and a JavaScript package. All three solve the assignment-minimum clique covering of Ennis, Fayle, and Ennis (2012), <https://doi.org/10.1145/2133803.2275596>, as a mixed-integer program with [HiGHS](https://highs.dev), and they return the same display for the same input. The rules they follow are in `docs/algorithm.md` in the repository.
|
|
4
|
+
|
|
5
|
+
The solver is HiGHS through `highspy`. The dependencies are `highspy`, NumPy, and pandas.
|
|
6
|
+
|
|
7
|
+
## Installation
|
|
8
|
+
|
|
9
|
+
Install from PyPI:
|
|
10
|
+
|
|
11
|
+
```sh
|
|
12
|
+
pip install cld-reducer
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
To install the development version from the repository:
|
|
16
|
+
|
|
17
|
+
```sh
|
|
18
|
+
pip install "git+https://github.com/aigorahub/cld-reducer.git#subdirectory=python"
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
The `#subdirectory=python` part is needed because the Python package lives in `python/`. Python 3.10 or later. The module is `cld_reducer` and the command line tool is `cld-reduce`.
|
|
22
|
+
|
|
23
|
+
For development, work from this folder:
|
|
24
|
+
|
|
25
|
+
```sh
|
|
26
|
+
cd python
|
|
27
|
+
python -m venv .venv
|
|
28
|
+
source .venv/bin/activate
|
|
29
|
+
python -m pip install -e ".[dev]"
|
|
30
|
+
ruff check . && ruff format --check . && pytest
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Usage
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
import pandas as pd
|
|
37
|
+
|
|
38
|
+
from cld_reducer import reduce_letters
|
|
39
|
+
|
|
40
|
+
# One row for each pair. True means the groups differ significantly.
|
|
41
|
+
pairs = pd.DataFrame(
|
|
42
|
+
[
|
|
43
|
+
("1", "2", False),
|
|
44
|
+
("1", "3", False),
|
|
45
|
+
("1", "4", True),
|
|
46
|
+
("1", "5", True),
|
|
47
|
+
("2", "3", False),
|
|
48
|
+
("2", "4", False),
|
|
49
|
+
("2", "5", True),
|
|
50
|
+
("3", "4", False),
|
|
51
|
+
("3", "5", False),
|
|
52
|
+
("4", "5", False),
|
|
53
|
+
],
|
|
54
|
+
columns=["group1", "group2", "significant"],
|
|
55
|
+
)
|
|
56
|
+
means = pd.DataFrame({"group": ["1", "2", "3", "4", "5"], "mean": [3.73, 3.57, 3.46, 3.33, 3.30]})
|
|
57
|
+
|
|
58
|
+
result = reduce_letters(pairs, means)
|
|
59
|
+
print(result.letters)
|
|
60
|
+
# {'1': 'A', '2': 'AB', '3': 'AC', '4': 'BC', '5': 'C'}
|
|
61
|
+
print(result.to_frame())
|
|
62
|
+
# group letters assignments
|
|
63
|
+
# 0 1 A A
|
|
64
|
+
# 1 2 AB A B
|
|
65
|
+
# 2 3 AC A C
|
|
66
|
+
# 3 4 BC B C
|
|
67
|
+
# 4 5 C C
|
|
68
|
+
print(result.stats)
|
|
69
|
+
# {'assignments_before': 9, 'assignments_after': 8, 'reduction_pct': 11.11111111111111, 'num_letters_before': 3, 'num_letters_after': 3, 'num_groups': 5, 'num_edges': 7, 'solver_status': 'Optimal', 'objective': 8}
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
This example works after package installation and needs no external files. `reduce_letters` takes complete pairwise results: one row for each pair of groups, with a boolean column that tells whether the pair differs significantly. Missing pairs are rejected. `means` is optional and orders the groups and the letters. If you already have the non-significance matrix, use `reduce_from_adjacency`; there, `True` means two groups are not significantly different and must share a letter.
|
|
73
|
+
|
|
74
|
+
The result, a `CLDReductionResult`, has:
|
|
75
|
+
|
|
76
|
+
- `letters`: the display of each group, for example `"AC"` (tokens are joined with spaces after `Z`, as in `"Z AA"`)
|
|
77
|
+
- `assignments`: the letters of each group as a tuple, the safe machine-readable form
|
|
78
|
+
- `stats`: the counts before and after the reduction and the solver status; `reduction_pct` is not rounded
|
|
79
|
+
- `relationship_preserved`: always `True` for a returned result
|
|
80
|
+
- `to_frame()`: a tidy `pandas.DataFrame`
|
|
81
|
+
|
|
82
|
+
The public names are `reduce_letters`, `reduce_from_adjacency`, `CLDReductionResult`, `CLDReducerError`, `InvalidInputError`, and `SolverError`.
|
|
83
|
+
|
|
84
|
+
When several displays have the same, smallest number of assignments, the package returns the canonical one defined in `docs/algorithm.md`, so R, Python, and JavaScript agree. `time_limit` is one time budget in seconds for all solves of a call, and `max_cliques` (10,000 by default; `None` removes it) bounds the number of maximal cliques.
|
|
85
|
+
|
|
86
|
+
## Command line
|
|
87
|
+
|
|
88
|
+
```sh
|
|
89
|
+
cld-reduce examples/simple_abc_to_ac_pairs.csv \
|
|
90
|
+
--means examples/simple_abc_to_ac_means.csv \
|
|
91
|
+
--time-limit 30 \
|
|
92
|
+
--out reduced.csv
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The paths above refer to files in a checkout or an extracted source archive. For your own data, pass your CSV paths. The pairs file has the columns `group1`, `group2`, and `significant`; the optional means file has `group` and `mean`. The output CSV has the reduced letters and the summary statistics. Other flags: `--group1`, `--group2`, `--significant`, `--max-cliques`, `--no-max-cliques`, and `--method`.
|
|
96
|
+
|
|
97
|
+
## Examples
|
|
98
|
+
|
|
99
|
+
The following scripts and CSV files are included in a repository checkout or the source archive, but not in the installed wheel. Run them from `python/` in a checkout, or from the root of an extracted source archive:
|
|
100
|
+
|
|
101
|
+
```sh
|
|
102
|
+
python examples/simple_abc_to_ac.py
|
|
103
|
+
python examples/piepho2004_wheat.py
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
The CSV files in `examples/` are identical copies of the files in `conformance/data/` at the repository root.
|
|
107
|
+
|
|
108
|
+
## Changes in 0.2.0
|
|
109
|
+
|
|
110
|
+
The 0.2.0 release moved the package to `python/` and the solver to `highspy`, which can change the display for inputs with several minimal displays and can rename letters. See the root `NEWS.md` for the full list.
|
|
111
|
+
|
|
112
|
+
## License
|
|
113
|
+
|
|
114
|
+
MIT. See `LICENSE`.
|
|
115
|
+
|
|
116
|
+
## Input and release notes
|
|
117
|
+
|
|
118
|
+
CSV label columns preserve their exact text, including `001`, `NA`, `NaN`, and
|
|
119
|
+
empty strings. A blank CSV label is an empty-string label; missing values in API
|
|
120
|
+
objects still raise `InvalidInputError`. Means must remain finite numbers.
|
|
121
|
+
The means table uses `group` and `mean` when both exist, or its first two columns.
|
|
122
|
+
A list of pair rows preserves each numeric label before conversion to text.
|
|
123
|
+
Uneven adjacency rows raise the package's square-matrix error.
|
|
124
|
+
|
|
125
|
+
The source distribution contains example scripts and CSV files. The wheel
|
|
126
|
+
contains the importable package and its license and data notice. Shared
|
|
127
|
+
conformance fixtures remain in the repository, so source-distribution tests skip
|
|
128
|
+
those cases when the fixtures are absent. CI tests both installed distributions
|
|
129
|
+
outside the checkout. See `docs/releasing.md` in the repository for release steps.
|
|
130
|
+
Example data reuse remains subject to the conditions in `NOTICE`.
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Run the Piepho (2004) CIMMYT wheat yield experiment example.
|
|
2
|
+
|
|
3
|
+
Source: 20-treatment multi-environment wheat yield trial reported by Piepho
|
|
4
|
+
(2004), reproduced in Table 7 of Ennis, Fayle, & Ennis (2012),
|
|
5
|
+
"Assignment-Minimum Clique Coverings", ACM JEA 17, Art. 1.5
|
|
6
|
+
(https://doi.org/10.1145/2133803.2275596). The maximal covering has 4 cliques
|
|
7
|
+
and 56 letter assignments; the assignment-minimum reduction has 4 cliques and
|
|
8
|
+
44 letter assignments, matching the result reported in the paper.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
from cld_reducer import reduce_letters
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def main() -> None:
|
|
21
|
+
example_dir = Path(__file__).resolve().parent
|
|
22
|
+
pairs = pd.read_csv(example_dir / "piepho2004_wheat_pairs.csv")
|
|
23
|
+
result = reduce_letters(pairs)
|
|
24
|
+
|
|
25
|
+
print(result.to_frame().to_string(index=False))
|
|
26
|
+
print()
|
|
27
|
+
print(f"Groups: {result.stats['num_groups']}")
|
|
28
|
+
print(f"Non-sig edges: {result.stats['num_edges']}")
|
|
29
|
+
print(
|
|
30
|
+
f"Letters: {result.stats['num_letters_before']} "
|
|
31
|
+
f"-> {result.stats['num_letters_after']}"
|
|
32
|
+
)
|
|
33
|
+
print(f"Assignments before: {result.stats['assignments_before']}")
|
|
34
|
+
print(f"Assignments after: {result.stats['assignments_after']}")
|
|
35
|
+
print(f"Reduction: {result.stats['reduction_pct']:.1f}%")
|
|
36
|
+
print(f"Preserved: {result.relationship_preserved}")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
if __name__ == "__main__":
|
|
40
|
+
main()
|
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
group1,group2,significant
|
|
2
|
+
1,2,false
|
|
3
|
+
1,3,false
|
|
4
|
+
1,4,false
|
|
5
|
+
1,5,false
|
|
6
|
+
1,6,false
|
|
7
|
+
1,7,false
|
|
8
|
+
1,8,false
|
|
9
|
+
1,9,false
|
|
10
|
+
1,10,false
|
|
11
|
+
1,11,false
|
|
12
|
+
1,12,false
|
|
13
|
+
1,13,false
|
|
14
|
+
1,14,false
|
|
15
|
+
1,15,false
|
|
16
|
+
1,16,false
|
|
17
|
+
1,17,false
|
|
18
|
+
1,18,false
|
|
19
|
+
1,19,false
|
|
20
|
+
1,20,false
|
|
21
|
+
2,3,false
|
|
22
|
+
2,4,false
|
|
23
|
+
2,5,false
|
|
24
|
+
2,6,false
|
|
25
|
+
2,7,false
|
|
26
|
+
2,8,false
|
|
27
|
+
2,9,false
|
|
28
|
+
2,10,false
|
|
29
|
+
2,11,false
|
|
30
|
+
2,12,false
|
|
31
|
+
2,13,false
|
|
32
|
+
2,14,false
|
|
33
|
+
2,15,false
|
|
34
|
+
2,16,false
|
|
35
|
+
2,17,false
|
|
36
|
+
2,18,false
|
|
37
|
+
2,19,false
|
|
38
|
+
2,20,true
|
|
39
|
+
3,4,false
|
|
40
|
+
3,5,false
|
|
41
|
+
3,6,false
|
|
42
|
+
3,7,false
|
|
43
|
+
3,8,false
|
|
44
|
+
3,9,false
|
|
45
|
+
3,10,false
|
|
46
|
+
3,11,false
|
|
47
|
+
3,12,false
|
|
48
|
+
3,13,false
|
|
49
|
+
3,14,false
|
|
50
|
+
3,15,false
|
|
51
|
+
3,16,false
|
|
52
|
+
3,17,false
|
|
53
|
+
3,18,true
|
|
54
|
+
3,19,false
|
|
55
|
+
3,20,true
|
|
56
|
+
4,5,false
|
|
57
|
+
4,6,false
|
|
58
|
+
4,7,false
|
|
59
|
+
4,8,false
|
|
60
|
+
4,9,false
|
|
61
|
+
4,10,false
|
|
62
|
+
4,11,false
|
|
63
|
+
4,12,false
|
|
64
|
+
4,13,false
|
|
65
|
+
4,14,false
|
|
66
|
+
4,15,false
|
|
67
|
+
4,16,false
|
|
68
|
+
4,17,false
|
|
69
|
+
4,18,false
|
|
70
|
+
4,19,false
|
|
71
|
+
4,20,true
|
|
72
|
+
5,6,false
|
|
73
|
+
5,7,false
|
|
74
|
+
5,8,false
|
|
75
|
+
5,9,false
|
|
76
|
+
5,10,false
|
|
77
|
+
5,11,false
|
|
78
|
+
5,12,true
|
|
79
|
+
5,13,false
|
|
80
|
+
5,14,false
|
|
81
|
+
5,15,false
|
|
82
|
+
5,16,false
|
|
83
|
+
5,17,false
|
|
84
|
+
5,18,false
|
|
85
|
+
5,19,false
|
|
86
|
+
5,20,true
|
|
87
|
+
6,7,false
|
|
88
|
+
6,8,false
|
|
89
|
+
6,9,false
|
|
90
|
+
6,10,false
|
|
91
|
+
6,11,false
|
|
92
|
+
6,12,false
|
|
93
|
+
6,13,false
|
|
94
|
+
6,14,false
|
|
95
|
+
6,15,false
|
|
96
|
+
6,16,false
|
|
97
|
+
6,17,false
|
|
98
|
+
6,18,false
|
|
99
|
+
6,19,false
|
|
100
|
+
6,20,true
|
|
101
|
+
7,8,false
|
|
102
|
+
7,9,false
|
|
103
|
+
7,10,false
|
|
104
|
+
7,11,false
|
|
105
|
+
7,12,true
|
|
106
|
+
7,13,false
|
|
107
|
+
7,14,false
|
|
108
|
+
7,15,false
|
|
109
|
+
7,16,false
|
|
110
|
+
7,17,false
|
|
111
|
+
7,18,false
|
|
112
|
+
7,19,false
|
|
113
|
+
7,20,true
|
|
114
|
+
8,9,false
|
|
115
|
+
8,10,false
|
|
116
|
+
8,11,false
|
|
117
|
+
8,12,false
|
|
118
|
+
8,13,false
|
|
119
|
+
8,14,false
|
|
120
|
+
8,15,false
|
|
121
|
+
8,16,false
|
|
122
|
+
8,17,false
|
|
123
|
+
8,18,false
|
|
124
|
+
8,19,false
|
|
125
|
+
8,20,false
|
|
126
|
+
9,10,false
|
|
127
|
+
9,11,false
|
|
128
|
+
9,12,false
|
|
129
|
+
9,13,false
|
|
130
|
+
9,14,false
|
|
131
|
+
9,15,false
|
|
132
|
+
9,16,false
|
|
133
|
+
9,17,false
|
|
134
|
+
9,18,false
|
|
135
|
+
9,19,false
|
|
136
|
+
9,20,true
|
|
137
|
+
10,11,false
|
|
138
|
+
10,12,true
|
|
139
|
+
10,13,false
|
|
140
|
+
10,14,false
|
|
141
|
+
10,15,false
|
|
142
|
+
10,16,false
|
|
143
|
+
10,17,false
|
|
144
|
+
10,18,false
|
|
145
|
+
10,19,false
|
|
146
|
+
10,20,true
|
|
147
|
+
11,12,false
|
|
148
|
+
11,13,false
|
|
149
|
+
11,14,false
|
|
150
|
+
11,15,false
|
|
151
|
+
11,16,false
|
|
152
|
+
11,17,false
|
|
153
|
+
11,18,false
|
|
154
|
+
11,19,false
|
|
155
|
+
11,20,true
|
|
156
|
+
12,13,false
|
|
157
|
+
12,14,false
|
|
158
|
+
12,15,false
|
|
159
|
+
12,16,true
|
|
160
|
+
12,17,false
|
|
161
|
+
12,18,true
|
|
162
|
+
12,19,false
|
|
163
|
+
12,20,false
|
|
164
|
+
13,14,false
|
|
165
|
+
13,15,false
|
|
166
|
+
13,16,false
|
|
167
|
+
13,17,false
|
|
168
|
+
13,18,false
|
|
169
|
+
13,19,false
|
|
170
|
+
13,20,true
|
|
171
|
+
14,15,false
|
|
172
|
+
14,16,false
|
|
173
|
+
14,17,false
|
|
174
|
+
14,18,false
|
|
175
|
+
14,19,false
|
|
176
|
+
14,20,false
|
|
177
|
+
15,16,false
|
|
178
|
+
15,17,false
|
|
179
|
+
15,18,false
|
|
180
|
+
15,19,false
|
|
181
|
+
15,20,false
|
|
182
|
+
16,17,false
|
|
183
|
+
16,18,false
|
|
184
|
+
16,19,false
|
|
185
|
+
16,20,true
|
|
186
|
+
17,18,false
|
|
187
|
+
17,19,false
|
|
188
|
+
17,20,false
|
|
189
|
+
18,19,false
|
|
190
|
+
18,20,true
|
|
191
|
+
19,20,false
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Run the simple CLD reduction example where ABC becomes AC."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
import pandas as pd
|
|
8
|
+
|
|
9
|
+
from cld_reducer import reduce_letters
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def main() -> None:
|
|
13
|
+
example_dir = Path(__file__).resolve().parent
|
|
14
|
+
pairs = pd.read_csv(example_dir / "simple_abc_to_ac_pairs.csv")
|
|
15
|
+
means = pd.read_csv(example_dir / "simple_abc_to_ac_means.csv")
|
|
16
|
+
result = reduce_letters(pairs, means)
|
|
17
|
+
|
|
18
|
+
print(result.to_frame().to_string(index=False))
|
|
19
|
+
print()
|
|
20
|
+
print(f"Assignments before: {result.stats['assignments_before']}")
|
|
21
|
+
print(f"Assignments after: {result.stats['assignments_after']}")
|
|
22
|
+
print(f"Reduction: {result.stats['reduction_pct']:.1f}%")
|
|
23
|
+
print(f"Preserved: {result.relationship_preserved}")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
if __name__ == "__main__":
|
|
27
|
+
main()
|