infoweight 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infoweight-0.0.1/LICENSE +28 -0
- infoweight-0.0.1/PKG-INFO +78 -0
- infoweight-0.0.1/README.rst +56 -0
- infoweight-0.0.1/infoweight/__init__.py +4 -0
- infoweight-0.0.1/infoweight/infoweight.py +375 -0
- infoweight-0.0.1/infoweight.egg-info/PKG-INFO +78 -0
- infoweight-0.0.1/infoweight.egg-info/SOURCES.txt +11 -0
- infoweight-0.0.1/infoweight.egg-info/dependency_links.txt +1 -0
- infoweight-0.0.1/infoweight.egg-info/requires.txt +4 -0
- infoweight-0.0.1/infoweight.egg-info/top_level.txt +1 -0
- infoweight-0.0.1/pyproject.toml +36 -0
- infoweight-0.0.1/setup.cfg +4 -0
- infoweight-0.0.1/tests/test_infoweight.py +72 -0
infoweight-0.0.1/LICENSE
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
Copyright (c) 2020, John Healy, Leland McInnes, Colin Weir and Vectorizers contributors
|
|
2
|
+
Copyright (c) 2026 Ryan DeWolfe
|
|
3
|
+
All rights reserved.
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
* Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
* Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
* Neither the name of project-template nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
22
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
23
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
24
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
25
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
26
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
27
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
28
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: infoweight
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: An information theoretic generalization of TF-IDF.
|
|
5
|
+
Maintainer-email: Ryan DeWolfe <ryan.dewolfe@uwaterloo.ca>
|
|
6
|
+
License-Expression: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/ryandewolfe33/infoweight
|
|
8
|
+
Project-URL: Repository, https://github.com/ryandewolfe33/infoweight
|
|
9
|
+
Keywords: tfidf,information theory,metric learning,unsupervised learning
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Programming Language :: Python
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/x-rst
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Requires-Dist: numpy>=2.0.0
|
|
18
|
+
Requires-Dist: numba>=0.51
|
|
19
|
+
Requires-Dist: scikit-learn>=0.22
|
|
20
|
+
Requires-Dist: scipy
|
|
21
|
+
Dynamic: license-file
|
|
22
|
+
|
|
23
|
+
===========
|
|
24
|
+
Infoweight
|
|
25
|
+
===========
|
|
26
|
+
|
|
27
|
+
This package provides an information theoretic generalization of the `TFIDFTransformer`.
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
This package is still under development,
|
|
31
|
+
The `InformationWeightTransformer` was originally part of the
|
|
32
|
+
`vectorizers <https://github.com/TutteInstitute/vectorizers>`_
|
|
33
|
+
package, but has been factored out to ease development. You can find an
|
|
34
|
+
overview and example of using the package in the
|
|
35
|
+
`vectorizers documentation <https://vectorizers.readthedocs.io/en/latest/>`_.
|
|
36
|
+
|
|
37
|
+
----------
|
|
38
|
+
Installing
|
|
39
|
+
----------
|
|
40
|
+
|
|
41
|
+
To install the package from PyPI:
|
|
42
|
+
|
|
43
|
+
.. code:: bash
|
|
44
|
+
|
|
45
|
+
pip install infoweight
|
|
46
|
+
|
|
47
|
+
Or to clone and install locally:
|
|
48
|
+
.. code:: bash
|
|
49
|
+
|
|
50
|
+
git clone https://github.com/ryandewolfe33/infoweight.git &&
|
|
51
|
+
cd infoweight &&
|
|
52
|
+
pip install .
|
|
53
|
+
|
|
54
|
+
To install the package from source:
|
|
55
|
+
|
|
56
|
+
.. code:: bash
|
|
57
|
+
|
|
58
|
+
pip install https://github.com/ryandewolfe33/infoweight/archive/master.zip
|
|
59
|
+
|
|
60
|
+
------------
|
|
61
|
+
Contributing
|
|
62
|
+
------------
|
|
63
|
+
|
|
64
|
+
Contributions are more than welcome! There are lots of opportunities
|
|
65
|
+
for potential projects, so please get in touch if you would like to
|
|
66
|
+
help out. Everything from code to notebooks to examples and documentation
|
|
67
|
+
are all *equally valuable* so please don't feel you can't contribute.
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
To contribute please `fork the project <https://github.com/ryandewolfe33/infoweight/issues#fork-destination-box>`_
|
|
71
|
+
make your changes and submit a pull request. We will do our best to work
|
|
72
|
+
through any issues with you and get your code merged into the main branch.
|
|
73
|
+
|
|
74
|
+
-------
|
|
75
|
+
License
|
|
76
|
+
-------
|
|
77
|
+
|
|
78
|
+
The infoweight package is 3-clause BSD licensed.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
===========
|
|
2
|
+
Infoweight
|
|
3
|
+
===========
|
|
4
|
+
|
|
5
|
+
This package provides an information theoretic generalization of the `TFIDFTransformer`.
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
This package is still under development,
|
|
9
|
+
The `InformationWeightTransformer` was originally part of the
|
|
10
|
+
`vectorizers <https://github.com/TutteInstitute/vectorizers>`_
|
|
11
|
+
package, but has been factored out to ease development. You can find an
|
|
12
|
+
overview and example of using the package in the
|
|
13
|
+
`vectorizers documentation <https://vectorizers.readthedocs.io/en/latest/>`_.
|
|
14
|
+
|
|
15
|
+
----------
|
|
16
|
+
Installing
|
|
17
|
+
----------
|
|
18
|
+
|
|
19
|
+
To install the package from PyPI:
|
|
20
|
+
|
|
21
|
+
.. code:: bash
|
|
22
|
+
|
|
23
|
+
pip install infoweight
|
|
24
|
+
|
|
25
|
+
Or to clone and install locally:
|
|
26
|
+
.. code:: bash
|
|
27
|
+
|
|
28
|
+
git clone https://github.com/ryandewolfe33/infoweight.git &&
|
|
29
|
+
cd infoweight &&
|
|
30
|
+
pip install .
|
|
31
|
+
|
|
32
|
+
To install the package from source:
|
|
33
|
+
|
|
34
|
+
.. code:: bash
|
|
35
|
+
|
|
36
|
+
pip install https://github.com/ryandewolfe33/infoweight/archive/master.zip
|
|
37
|
+
|
|
38
|
+
------------
|
|
39
|
+
Contributing
|
|
40
|
+
------------
|
|
41
|
+
|
|
42
|
+
Contributions are more than welcome! There are lots of opportunities
|
|
43
|
+
for potential projects, so please get in touch if you would like to
|
|
44
|
+
help out. Everything from code to notebooks to examples and documentation
|
|
45
|
+
are all *equally valuable* so please don't feel you can't contribute.
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
To contribute please `fork the project <https://github.com/ryandewolfe33/infoweight/issues#fork-destination-box>`_
|
|
49
|
+
make your changes and submit a pull request. We will do our best to work
|
|
50
|
+
through any issues with you and get your code merged into the main branch.
|
|
51
|
+
|
|
52
|
+
-------
|
|
53
|
+
License
|
|
54
|
+
-------
|
|
55
|
+
|
|
56
|
+
The infoweight package is 3-clause BSD licensed.
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
import numba
|
|
2
|
+
import numpy as np
|
|
3
|
+
from sklearn.base import BaseEstimator, TransformerMixin
|
|
4
|
+
import scipy.sparse
|
|
5
|
+
|
|
6
|
+
MOCK_TARGET = np.ones(1, dtype=np.int64)
|
|
7
|
+
MOCK_BOOL = np.ones(1, dtype=np.bool)
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@numba.njit(nogil=True)
|
|
11
|
+
def column_kl_divergence_exact_prior(
|
|
12
|
+
count_indices,
|
|
13
|
+
count_data,
|
|
14
|
+
baseline_probabilities,
|
|
15
|
+
prior_strength=0.1,
|
|
16
|
+
target=MOCK_TARGET,
|
|
17
|
+
):
|
|
18
|
+
observed_norm = count_data.sum() + prior_strength
|
|
19
|
+
observed_zero_constant = (prior_strength / observed_norm) * np.log(
|
|
20
|
+
prior_strength / observed_norm
|
|
21
|
+
)
|
|
22
|
+
result = 0.0
|
|
23
|
+
count_indices_set = set(count_indices)
|
|
24
|
+
for i in range(baseline_probabilities.shape[0]):
|
|
25
|
+
if i in count_indices_set:
|
|
26
|
+
idx = np.searchsorted(count_indices, i)
|
|
27
|
+
observed_probability = (
|
|
28
|
+
count_data[idx] + prior_strength * baseline_probabilities[i]
|
|
29
|
+
) / observed_norm
|
|
30
|
+
if observed_probability > 0.0:
|
|
31
|
+
result += observed_probability * np.log(
|
|
32
|
+
observed_probability / baseline_probabilities[i]
|
|
33
|
+
)
|
|
34
|
+
else:
|
|
35
|
+
result += baseline_probabilities[i] * observed_zero_constant
|
|
36
|
+
|
|
37
|
+
return result
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@numba.njit(nogil=True)
|
|
41
|
+
def column_kl_divergence_approx_prior(
|
|
42
|
+
count_indices,
|
|
43
|
+
count_data,
|
|
44
|
+
baseline_probabilities,
|
|
45
|
+
prior_strength=0.1,
|
|
46
|
+
target=MOCK_TARGET,
|
|
47
|
+
):
|
|
48
|
+
observed_norm = count_data.sum() + prior_strength
|
|
49
|
+
observed_zero_constant = (prior_strength / observed_norm) * np.log(
|
|
50
|
+
prior_strength / observed_norm
|
|
51
|
+
)
|
|
52
|
+
result = 0.0
|
|
53
|
+
zero_count_component_estimate = (
|
|
54
|
+
np.mean(baseline_probabilities)
|
|
55
|
+
* observed_zero_constant
|
|
56
|
+
* (baseline_probabilities.shape[0] - count_indices.shape[0])
|
|
57
|
+
)
|
|
58
|
+
result += zero_count_component_estimate
|
|
59
|
+
for i in range(count_indices.shape[0]):
|
|
60
|
+
idx = count_indices[i]
|
|
61
|
+
observed_probability = (
|
|
62
|
+
count_data[i] + prior_strength * baseline_probabilities[idx]
|
|
63
|
+
) / observed_norm
|
|
64
|
+
if observed_probability > 0.0 and baseline_probabilities[idx] > 0:
|
|
65
|
+
result += observed_probability * np.log(
|
|
66
|
+
observed_probability / baseline_probabilities[idx]
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
return result
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@numba.njit(nogil=True)
|
|
73
|
+
def supervised_column_kl(
|
|
74
|
+
count_indices,
|
|
75
|
+
count_data,
|
|
76
|
+
baseline_probabilities,
|
|
77
|
+
prior_strength=0.1,
|
|
78
|
+
target=MOCK_TARGET,
|
|
79
|
+
):
|
|
80
|
+
observed = np.zeros_like(baseline_probabilities)
|
|
81
|
+
for i in range(count_indices.shape[0]):
|
|
82
|
+
idx = count_indices[i]
|
|
83
|
+
label = target[idx]
|
|
84
|
+
if label >= 0:
|
|
85
|
+
observed[label] += count_data[i]
|
|
86
|
+
|
|
87
|
+
observed += prior_strength * baseline_probabilities
|
|
88
|
+
observed /= observed.sum()
|
|
89
|
+
|
|
90
|
+
# Zeros in baseline_probabilities may cause nans in the log
|
|
91
|
+
# But this can only happen when observed is also 0, so due
|
|
92
|
+
# to the multiplication it does not contribute to the sum
|
|
93
|
+
non_zero = observed > 0
|
|
94
|
+
result = np.sum(
|
|
95
|
+
observed[non_zero]
|
|
96
|
+
* np.log(observed[non_zero] / baseline_probabilities[non_zero])
|
|
97
|
+
)
|
|
98
|
+
return result
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@numba.njit(nogil=True, parallel=True)
|
|
102
|
+
def column_weights(
|
|
103
|
+
indptr,
|
|
104
|
+
indices,
|
|
105
|
+
data,
|
|
106
|
+
baseline_probabilities,
|
|
107
|
+
column_kl_divergence_func,
|
|
108
|
+
prior_strength=0.1,
|
|
109
|
+
target=MOCK_TARGET,
|
|
110
|
+
column_groups=None,
|
|
111
|
+
):
|
|
112
|
+
n_cols = indptr.shape[0] - 1
|
|
113
|
+
weights = np.ones(n_cols)
|
|
114
|
+
for i in numba.prange(n_cols):
|
|
115
|
+
group = 0
|
|
116
|
+
if column_groups is not None:
|
|
117
|
+
group = column_groups[i]
|
|
118
|
+
weights[i] = column_kl_divergence_func(
|
|
119
|
+
indices[indptr[i] : indptr[i + 1]],
|
|
120
|
+
data[indptr[i] : indptr[i + 1]],
|
|
121
|
+
baseline_probabilities[group, :],
|
|
122
|
+
prior_strength=prior_strength,
|
|
123
|
+
target=target,
|
|
124
|
+
)
|
|
125
|
+
return weights
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
@numba.njit(nogil=True)
|
|
129
|
+
def compute_baseline_probabilities(
|
|
130
|
+
indptr,
|
|
131
|
+
indices,
|
|
132
|
+
data,
|
|
133
|
+
target=None,
|
|
134
|
+
column_groups=None,
|
|
135
|
+
):
|
|
136
|
+
"""
|
|
137
|
+
Compute the marginals to compare each column to. Returns
|
|
138
|
+
an (n column groups) x (n samples) matrix (unsupervised) or an
|
|
139
|
+
(n column groups) x (n targets) matrix (supervised) where each
|
|
140
|
+
row is the marginal of the column group.
|
|
141
|
+
|
|
142
|
+
indptr, indices, and data arrays are from csr format.
|
|
143
|
+
"""
|
|
144
|
+
n_groups = 1
|
|
145
|
+
if column_groups is not None:
|
|
146
|
+
n_groups = column_groups.max() + 1
|
|
147
|
+
n_targets = indptr.shape[0] - 1
|
|
148
|
+
if target is not None:
|
|
149
|
+
n_targets = target.max() + 1
|
|
150
|
+
counts = np.zeros((n_groups, n_targets), dtype=np.int64)
|
|
151
|
+
for row in range(indptr.shape[0] - 1):
|
|
152
|
+
this_target = row
|
|
153
|
+
if target is not None:
|
|
154
|
+
if target[row] >= 0:
|
|
155
|
+
this_target = target[row]
|
|
156
|
+
else:
|
|
157
|
+
continue
|
|
158
|
+
for i in range(indptr[row], indptr[row + 1]):
|
|
159
|
+
group = 0
|
|
160
|
+
if column_groups is not None:
|
|
161
|
+
group = column_groups[indices[i]]
|
|
162
|
+
counts[group, this_target] += data[i]
|
|
163
|
+
probabilities = counts / np.sum(counts, axis=1).reshape(-1, 1)
|
|
164
|
+
return probabilities
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def information_weight(
|
|
168
|
+
data,
|
|
169
|
+
prior_strength=0.1,
|
|
170
|
+
approximate_prior=False,
|
|
171
|
+
target=None,
|
|
172
|
+
column_groups=None,
|
|
173
|
+
):
|
|
174
|
+
"""Compute information based weights for columns. The information weight
|
|
175
|
+
is estimated as the amount of information gained by moving from a baseline
|
|
176
|
+
model to a model derived from the observed counts. In practice this can be
|
|
177
|
+
computed as the KL-divergence between distributions. For the baseline model
|
|
178
|
+
we assume data will be distributed according to the row sums -- i.e.
|
|
179
|
+
proportional to the frequency of the row. For the observed counts we use
|
|
180
|
+
a background prior of pseudo counts equal to ``prior_strength`` times the
|
|
181
|
+
baseline prior distribution. The Bayesian prior can either be computed
|
|
182
|
+
exactly (the default) at some computational expense, or estimated for a much
|
|
183
|
+
fast computation, often suitable for large or very sparse datasets.
|
|
184
|
+
|
|
185
|
+
Parameters
|
|
186
|
+
----------
|
|
187
|
+
data: scipy sparse matrix (n_samples, n_features)
|
|
188
|
+
A matrix of count data where rows represent observations and
|
|
189
|
+
columns represent features. Column weightings will be learned
|
|
190
|
+
from this data.
|
|
191
|
+
|
|
192
|
+
prior_strength: float (optional, default=0.1)
|
|
193
|
+
How strongly to weight the prior when doing a Bayesian update to
|
|
194
|
+
derive a model based on observed counts of a column.
|
|
195
|
+
|
|
196
|
+
approximate_prior: bool (optional, default=False)
|
|
197
|
+
Whether to approximate weights based on the Bayesian prior or perform
|
|
198
|
+
exact computations. Approximations are much faster especially for very
|
|
199
|
+
large or very sparse datasets.
|
|
200
|
+
|
|
201
|
+
target: ndarray or None (optional, default=None)
|
|
202
|
+
If supervised target labels are available, these can be used to define distributions
|
|
203
|
+
over the target classes rather than over rows, allowing weights to be
|
|
204
|
+
supervised and target based. If None then unsupervised weighting is used.
|
|
205
|
+
|
|
206
|
+
column_groups: ndarray or None (optional, default=None)
|
|
207
|
+
If columns have a natural grouping, i.e. cols 10-15 are a one-hot-encoding of a single
|
|
208
|
+
categorical variable, we should compare the column distribution to the within group
|
|
209
|
+
marginal. If passed None then all columns have the same group.
|
|
210
|
+
|
|
211
|
+
Returns
|
|
212
|
+
-------
|
|
213
|
+
weights: ndarray of shape (n_features,)
|
|
214
|
+
The learned weights to be applied to columns based on the amount
|
|
215
|
+
of information provided by the column.
|
|
216
|
+
"""
|
|
217
|
+
if target is not None:
|
|
218
|
+
column_kl_divergence_func = supervised_column_kl
|
|
219
|
+
elif approximate_prior:
|
|
220
|
+
column_kl_divergence_func = column_kl_divergence_approx_prior
|
|
221
|
+
else:
|
|
222
|
+
column_kl_divergence_func = column_kl_divergence_exact_prior
|
|
223
|
+
|
|
224
|
+
csr_data = data.tocsr()
|
|
225
|
+
baseline_probabilities = compute_baseline_probabilities(
|
|
226
|
+
csr_data.indptr,
|
|
227
|
+
csr_data.indices,
|
|
228
|
+
csr_data.data,
|
|
229
|
+
target,
|
|
230
|
+
column_groups,
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
csc_data = data.tocsc()
|
|
234
|
+
csc_data.sort_indices()
|
|
235
|
+
weights = column_weights(
|
|
236
|
+
csc_data.indptr,
|
|
237
|
+
csc_data.indices,
|
|
238
|
+
csc_data.data,
|
|
239
|
+
baseline_probabilities,
|
|
240
|
+
column_kl_divergence_func,
|
|
241
|
+
prior_strength=prior_strength,
|
|
242
|
+
target=target,
|
|
243
|
+
column_groups=column_groups,
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
return weights
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
class InformationWeightTransformer(BaseEstimator, TransformerMixin):
|
|
250
|
+
"""A data transformer that re-weights columns of count data. Column weights
|
|
251
|
+
are computed as information based weights for columns. The information weight
|
|
252
|
+
is estimated as the amount of information gained by moving from a baseline
|
|
253
|
+
model to a model derived from the observed counts. In practice this can be
|
|
254
|
+
computed as the KL-divergence between distributions. For the baseline model
|
|
255
|
+
we assume data will be distributed according to the row sums -- i.e.
|
|
256
|
+
proportional to the frequency of the row. For the observed counts we use
|
|
257
|
+
a background prior of pseudo counts equal to ``prior_strength`` times the
|
|
258
|
+
baseline prior distribution. The Bayesian prior can either be computed
|
|
259
|
+
exactly (the default) at some computational expense, or estimated for a much
|
|
260
|
+
fast computation, often suitable for large or very sparse datasets.
|
|
261
|
+
|
|
262
|
+
Parameters
|
|
263
|
+
----------
|
|
264
|
+
prior_strength: float (optional, default=0.1)
|
|
265
|
+
How strongly to weight the prior when doing a Bayesian update to
|
|
266
|
+
derive a model based on observed counts of a column.
|
|
267
|
+
|
|
268
|
+
approximate_prior: bool (optional, default=False)
|
|
269
|
+
Whether to approximate weights based on the Bayesian prior or perform
|
|
270
|
+
exact computations. Approximations are much faster especially for very
|
|
271
|
+
large or very sparse datasets.
|
|
272
|
+
|
|
273
|
+
Attributes
|
|
274
|
+
----------
|
|
275
|
+
|
|
276
|
+
information_weights_: ndarray of shape (n_features,)
|
|
277
|
+
The learned weights to be applied to columns based on the amount
|
|
278
|
+
of information provided by the column.
|
|
279
|
+
"""
|
|
280
|
+
|
|
281
|
+
def __init__(
|
|
282
|
+
self,
|
|
283
|
+
prior_strength=1e-4,
|
|
284
|
+
approx_prior=True,
|
|
285
|
+
weight_power=2.0,
|
|
286
|
+
supervision_weight=0.95,
|
|
287
|
+
):
|
|
288
|
+
self.prior_strength = prior_strength
|
|
289
|
+
self.approx_prior = approx_prior
|
|
290
|
+
self.weight_power = weight_power
|
|
291
|
+
self.supervision_weight = supervision_weight
|
|
292
|
+
|
|
293
|
+
def fit(self, X, y=None, column_groups=None, **fit_kwds):
|
|
294
|
+
"""Learn the appropriate column weighting as information weights
|
|
295
|
+
from the observed count data ``X``.
|
|
296
|
+
|
|
297
|
+
Parameters
|
|
298
|
+
----------
|
|
299
|
+
X: ndarray of scipy sparse matrix of shape (n_samples, n_features)
|
|
300
|
+
The count data to be trained on. Note that, as count data all
|
|
301
|
+
entries should be positive or zero.
|
|
302
|
+
|
|
303
|
+
Returns
|
|
304
|
+
-------
|
|
305
|
+
self:
|
|
306
|
+
The trained model.
|
|
307
|
+
"""
|
|
308
|
+
if not scipy.sparse.isspmatrix(X):
|
|
309
|
+
X = scipy.sparse.csc_matrix(X)
|
|
310
|
+
|
|
311
|
+
self.information_weights_ = information_weight(
|
|
312
|
+
X,
|
|
313
|
+
self.prior_strength,
|
|
314
|
+
self.approx_prior,
|
|
315
|
+
column_groups=column_groups,
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
mean_weight = np.mean(self.information_weights_)
|
|
319
|
+
if mean_weight > 0:
|
|
320
|
+
self.information_weights_ /= mean_weight
|
|
321
|
+
# This should never happen
|
|
322
|
+
self.information_weights_ = np.maximum(self.information_weights_, 0.0)
|
|
323
|
+
self.information_weights_ = np.power(
|
|
324
|
+
self.information_weights_, self.weight_power
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
if y is not None:
|
|
328
|
+
unsupervised_power = (1.0 - self.supervision_weight) * self.weight_power
|
|
329
|
+
supervised_power = self.supervision_weight * self.weight_power
|
|
330
|
+
|
|
331
|
+
target_classes = np.unique(y)
|
|
332
|
+
target_dict = dict(
|
|
333
|
+
np.vstack((target_classes, np.arange(target_classes.shape[0]))).T
|
|
334
|
+
)
|
|
335
|
+
target = np.array(
|
|
336
|
+
[np.int64(target_dict[label]) for label in y], dtype=np.int64
|
|
337
|
+
)
|
|
338
|
+
self.supervised_weights_ = information_weight(
|
|
339
|
+
X,
|
|
340
|
+
self.prior_strength,
|
|
341
|
+
self.approx_prior,
|
|
342
|
+
target=target,
|
|
343
|
+
column_groups=column_groups,
|
|
344
|
+
)
|
|
345
|
+
mean_supervised_weight = np.mean(self.information_weights_)
|
|
346
|
+
if mean_supervised_weight > 0:
|
|
347
|
+
self.supervised_weights_ /= mean_supervised_weight
|
|
348
|
+
# This should never happen
|
|
349
|
+
self.supervised_weights_ = np.maximum(self.supervised_weights_, 0.0)
|
|
350
|
+
self.supervised_weights_ = np.power(
|
|
351
|
+
self.supervised_weights_, supervised_power
|
|
352
|
+
)
|
|
353
|
+
|
|
354
|
+
self.information_weights_ = (
|
|
355
|
+
self.information_weights_ * self.supervised_weights_
|
|
356
|
+
)
|
|
357
|
+
|
|
358
|
+
return self
|
|
359
|
+
|
|
360
|
+
def transform(self, X):
|
|
361
|
+
"""Reweight data ``X`` based on learned information weights of columns.
|
|
362
|
+
|
|
363
|
+
Parameters
|
|
364
|
+
----------
|
|
365
|
+
X: ndarray of scipy sparse matrix of shape (n_samples, n_features)
|
|
366
|
+
The count data to be transformed. Note that, as count data all
|
|
367
|
+
entries should be positive or zero.
|
|
368
|
+
|
|
369
|
+
Returns
|
|
370
|
+
-------
|
|
371
|
+
result: ndarray of scipy sparse matrix of shape (n_samples, n_features)
|
|
372
|
+
The reweighted data.
|
|
373
|
+
"""
|
|
374
|
+
result = X @ scipy.sparse.diags(self.information_weights_)
|
|
375
|
+
return result
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: infoweight
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: An information theoretic generalization of TF-IDF.
|
|
5
|
+
Maintainer-email: Ryan DeWolfe <ryan.dewolfe@uwaterloo.ca>
|
|
6
|
+
License-Expression: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/ryandewolfe33/infoweight
|
|
8
|
+
Project-URL: Repository, https://github.com/ryandewolfe33/infoweight
|
|
9
|
+
Keywords: tfidf,information theory,metric learning,unsupervised learning
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Programming Language :: Python
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Description-Content-Type: text/x-rst
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Requires-Dist: numpy>=2.0.0
|
|
18
|
+
Requires-Dist: numba>=0.51
|
|
19
|
+
Requires-Dist: scikit-learn>=0.22
|
|
20
|
+
Requires-Dist: scipy
|
|
21
|
+
Dynamic: license-file
|
|
22
|
+
|
|
23
|
+
===========
|
|
24
|
+
Infoweight
|
|
25
|
+
===========
|
|
26
|
+
|
|
27
|
+
This package provides an information theoretic generalization of the `TFIDFTransformer`.
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
This package is still under development,
|
|
31
|
+
The `InformationWeightTransformer` was originally part of the
|
|
32
|
+
`vectorizers <https://github.com/TutteInstitute/vectorizers>`_
|
|
33
|
+
package, but has been factored out to ease development. You can find an
|
|
34
|
+
overview and example of using the package in the
|
|
35
|
+
`vectorizers documentation <https://vectorizers.readthedocs.io/en/latest/>`_.
|
|
36
|
+
|
|
37
|
+
----------
|
|
38
|
+
Installing
|
|
39
|
+
----------
|
|
40
|
+
|
|
41
|
+
To install the package from PyPI:
|
|
42
|
+
|
|
43
|
+
.. code:: bash
|
|
44
|
+
|
|
45
|
+
pip install infoweight
|
|
46
|
+
|
|
47
|
+
Or to clone and install locally:
|
|
48
|
+
.. code:: bash
|
|
49
|
+
|
|
50
|
+
git clone https://github.com/ryandewolfe33/infoweight.git &&
|
|
51
|
+
cd infoweight &&
|
|
52
|
+
pip install .
|
|
53
|
+
|
|
54
|
+
To install the package from source:
|
|
55
|
+
|
|
56
|
+
.. code:: bash
|
|
57
|
+
|
|
58
|
+
pip install https://github.com/ryandewolfe33/infoweight/archive/master.zip
|
|
59
|
+
|
|
60
|
+
------------
|
|
61
|
+
Contributing
|
|
62
|
+
------------
|
|
63
|
+
|
|
64
|
+
Contributions are more than welcome! There are lots of opportunities
|
|
65
|
+
for potential projects, so please get in touch if you would like to
|
|
66
|
+
help out. Everything from code to notebooks to examples and documentation
|
|
67
|
+
are all *equally valuable* so please don't feel you can't contribute.
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
To contribute please `fork the project <https://github.com/ryandewolfe33/infoweight/issues#fork-destination-box>`_
|
|
71
|
+
make your changes and submit a pull request. We will do our best to work
|
|
72
|
+
through any issues with you and get your code merged into the main branch.
|
|
73
|
+
|
|
74
|
+
-------
|
|
75
|
+
License
|
|
76
|
+
-------
|
|
77
|
+
|
|
78
|
+
The infoweight package is 3-clause BSD licensed.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.rst
|
|
3
|
+
pyproject.toml
|
|
4
|
+
infoweight/__init__.py
|
|
5
|
+
infoweight/infoweight.py
|
|
6
|
+
infoweight.egg-info/PKG-INFO
|
|
7
|
+
infoweight.egg-info/SOURCES.txt
|
|
8
|
+
infoweight.egg-info/dependency_links.txt
|
|
9
|
+
infoweight.egg-info/requires.txt
|
|
10
|
+
infoweight.egg-info/top_level.txt
|
|
11
|
+
tests/test_infoweight.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
infoweight
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.2"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "infoweight"
|
|
7
|
+
version = "0.0.1"
|
|
8
|
+
readme = "README.rst"
|
|
9
|
+
description = "An information theoretic generalization of TF-IDF."
|
|
10
|
+
maintainers = [{name = "Ryan DeWolfe", email = "ryan.dewolfe@uwaterloo.ca"}]
|
|
11
|
+
keywords = ["tfidf", "information theory", "metric learning", "unsupervised learning"]
|
|
12
|
+
license = "BSD-3-Clause"
|
|
13
|
+
license-files = ["LICENSE"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Programming Language :: Python",
|
|
17
|
+
"Operating System :: OS Independent",
|
|
18
|
+
"Intended Audience :: Science/Research",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
requires-python = ">=3.10"
|
|
22
|
+
dependencies = [
|
|
23
|
+
"numpy>=2.0.0",
|
|
24
|
+
"numba>=0.51",
|
|
25
|
+
"scikit-learn>=0.22",
|
|
26
|
+
"scipy",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[dependency-groups]
|
|
30
|
+
dev = [
|
|
31
|
+
"pytest>=9.1.1",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://github.com/ryandewolfe33/infoweight"
|
|
36
|
+
Repository = "https://github.com/ryandewolfe33/infoweight"
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
import pytest
|
|
2
|
+
import numpy as np
|
|
3
|
+
import scipy.sparse
|
|
4
|
+
from infoweight import InformationWeightTransformer
|
|
5
|
+
|
|
6
|
+
test_matrix = scipy.sparse.csr_matrix([[1, 2, 3], [4, 5, 6], [7, 8, 9]])
|
|
7
|
+
test_matrix_zero_row = scipy.sparse.csr_matrix([[1, 2, 3], [4, 5, 6], [0, 0, 0]])
|
|
8
|
+
test_matrix_zero_row.eliminate_zeros()
|
|
9
|
+
test_matrix_zero_column = scipy.sparse.csr_matrix([[1, 2, 0], [4, 5, 0], [7, 8, 0]])
|
|
10
|
+
test_matrix_zero_column.eliminate_zeros()
|
|
11
|
+
|
|
12
|
+
@pytest.mark.parametrize("prior_strength", [0.1, 1.0])
|
|
13
|
+
@pytest.mark.parametrize("approx_prior", [True, False])
|
|
14
|
+
def test_iw_transformer(prior_strength, approx_prior):
|
|
15
|
+
IWT = InformationWeightTransformer(
|
|
16
|
+
prior_strength=prior_strength,
|
|
17
|
+
approx_prior=approx_prior,
|
|
18
|
+
)
|
|
19
|
+
result = IWT.fit_transform(test_matrix)
|
|
20
|
+
transform = IWT.transform(test_matrix)
|
|
21
|
+
assert np.allclose(result.toarray(), transform.toarray())
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@pytest.mark.parametrize("prior_strength", [0.1, 1.0])
|
|
25
|
+
@pytest.mark.parametrize("approx_prior", [True, False])
|
|
26
|
+
@pytest.mark.parametrize("target", [None, np.array([0, 1, 1])])
|
|
27
|
+
@pytest.mark.parametrize("column_groups", [None, np.array([0, 1, 1])])
|
|
28
|
+
def test_iw_transformer_fit_args(prior_strength, approx_prior, target, column_groups):
|
|
29
|
+
IWT = InformationWeightTransformer(
|
|
30
|
+
prior_strength=prior_strength,
|
|
31
|
+
approx_prior=approx_prior,
|
|
32
|
+
)
|
|
33
|
+
result = IWT.fit_transform(test_matrix, target, column_groups=column_groups)
|
|
34
|
+
transform = IWT.transform(test_matrix)
|
|
35
|
+
assert np.allclose(result.toarray(), transform.toarray())
|
|
36
|
+
assert np.all(IWT.information_weights_ >= 0)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@pytest.mark.parametrize("prior_strength", [0.1, 1.0])
|
|
40
|
+
@pytest.mark.parametrize("approx_prior", [True, False])
|
|
41
|
+
@pytest.mark.parametrize("target", [None, np.array([0, 1, 1])])
|
|
42
|
+
@pytest.mark.parametrize("column_groups", [None, np.array([0, 1, 1])])
|
|
43
|
+
def test_iw_transformer_zero_column(
|
|
44
|
+
prior_strength, approx_prior, target, column_groups
|
|
45
|
+
):
|
|
46
|
+
IWT = InformationWeightTransformer(
|
|
47
|
+
prior_strength=prior_strength,
|
|
48
|
+
approx_prior=approx_prior,
|
|
49
|
+
)
|
|
50
|
+
result = IWT.fit_transform(
|
|
51
|
+
test_matrix_zero_column, target, column_groups=column_groups
|
|
52
|
+
)
|
|
53
|
+
transform = IWT.transform(test_matrix_zero_column)
|
|
54
|
+
assert np.allclose(result.toarray(), transform.toarray())
|
|
55
|
+
assert np.all(IWT.information_weights_ >= 0)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@pytest.mark.parametrize("prior_strength", [0.1, 1.0])
|
|
59
|
+
@pytest.mark.parametrize("approx_prior", [True, False])
|
|
60
|
+
@pytest.mark.parametrize("target", [None, np.array([0, 1, 1])])
|
|
61
|
+
@pytest.mark.parametrize("column_groups", [None, np.array([0, 1, 1])])
|
|
62
|
+
def test_iw_transformer_zero_row(prior_strength, approx_prior, target, column_groups):
|
|
63
|
+
IWT = InformationWeightTransformer(
|
|
64
|
+
prior_strength=prior_strength,
|
|
65
|
+
approx_prior=approx_prior,
|
|
66
|
+
)
|
|
67
|
+
result = IWT.fit_transform(
|
|
68
|
+
test_matrix_zero_row, target, column_groups=column_groups
|
|
69
|
+
)
|
|
70
|
+
transform = IWT.transform(test_matrix_zero_row)
|
|
71
|
+
assert np.allclose(result.toarray(), transform.toarray())
|
|
72
|
+
assert np.all(IWT.information_weights_ >= 0)
|