infoweight 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,28 @@
1
+ Copyright (c) 2020, John Healy, Leland McInnes, Colin Weir and Vectorizers contributors
2
+ Copyright (c) 2026 Ryan DeWolfe
3
+ All rights reserved.
4
+
5
+ Redistribution and use in source and binary forms, with or without
6
+ modification, are permitted provided that the following conditions are met:
7
+
8
+ * Redistributions of source code must retain the above copyright notice, this
9
+ list of conditions and the following disclaimer.
10
+
11
+ * Redistributions in binary form must reproduce the above copyright notice,
12
+ this list of conditions and the following disclaimer in the documentation
13
+ and/or other materials provided with the distribution.
14
+
15
+ * Neither the name of project-template nor the names of its
16
+ contributors may be used to endorse or promote products derived from
17
+ this software without specific prior written permission.
18
+
19
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
20
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
21
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
22
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
23
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
24
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
25
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
26
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
27
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
28
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,78 @@
1
+ Metadata-Version: 2.4
2
+ Name: infoweight
3
+ Version: 0.0.1
4
+ Summary: An information theoretic generalization of TF-IDF.
5
+ Maintainer-email: Ryan DeWolfe <ryan.dewolfe@uwaterloo.ca>
6
+ License-Expression: BSD-3-Clause
7
+ Project-URL: Homepage, https://github.com/ryandewolfe33/infoweight
8
+ Project-URL: Repository, https://github.com/ryandewolfe33/infoweight
9
+ Keywords: tfidf,information theory,metric learning,unsupervised learning
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Programming Language :: Python
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Intended Audience :: Science/Research
14
+ Requires-Python: >=3.10
15
+ Description-Content-Type: text/x-rst
16
+ License-File: LICENSE
17
+ Requires-Dist: numpy>=2.0.0
18
+ Requires-Dist: numba>=0.51
19
+ Requires-Dist: scikit-learn>=0.22
20
+ Requires-Dist: scipy
21
+ Dynamic: license-file
22
+
23
+ ===========
24
+ Infoweight
25
+ ===========
26
+
27
+ This package provides an information theoretic generalization of the `TFIDFTransformer`.
28
+
29
+
30
+ This package is still under development,
31
+ The `InformationWeightTransformer` was originally part of the
32
+ `vectorizers <https://github.com/TutteInstitute/vectorizers>`_
33
+ package, but has been factored out to ease development. You can find an
34
+ overview and example of using the package in the
35
+ `vectorizers documentation <https://vectorizers.readthedocs.io/en/latest/>`_.
36
+
37
+ ----------
38
+ Installing
39
+ ----------
40
+
41
+ To install the package from PyPI:
42
+
43
+ .. code:: bash
44
+
45
+ pip install infoweight
46
+
47
+ Or to clone and install locally:
48
+ .. code:: bash
49
+
50
+ git clone https://github.com/ryandewolfe33/infoweight.git &&
51
+ cd infoweight &&
52
+ pip install .
53
+
54
+ To install the package from source:
55
+
56
+ .. code:: bash
57
+
58
+ pip install https://github.com/ryandewolfe33/infoweight/archive/master.zip
59
+
60
+ ------------
61
+ Contributing
62
+ ------------
63
+
64
+ Contributions are more than welcome! There are lots of opportunities
65
+ for potential projects, so please get in touch if you would like to
66
+ help out. Everything from code to notebooks to examples and documentation
67
+ are all *equally valuable* so please don't feel you can't contribute.
68
+
69
+
70
+ To contribute please `fork the project <https://github.com/ryandewolfe33/infoweight/issues#fork-destination-box>`_
71
+ make your changes and submit a pull request. We will do our best to work
72
+ through any issues with you and get your code merged into the main branch.
73
+
74
+ -------
75
+ License
76
+ -------
77
+
78
+ The infoweight package is 3-clause BSD licensed.
@@ -0,0 +1,56 @@
1
+ ===========
2
+ Infoweight
3
+ ===========
4
+
5
+ This package provides an information theoretic generalization of the `TFIDFTransformer`.
6
+
7
+
8
+ This package is still under development,
9
+ The `InformationWeightTransformer` was originally part of the
10
+ `vectorizers <https://github.com/TutteInstitute/vectorizers>`_
11
+ package, but has been factored out to ease development. You can find an
12
+ overview and example of using the package in the
13
+ `vectorizers documentation <https://vectorizers.readthedocs.io/en/latest/>`_.
14
+
15
+ ----------
16
+ Installing
17
+ ----------
18
+
19
+ To install the package from PyPI:
20
+
21
+ .. code:: bash
22
+
23
+ pip install infoweight
24
+
25
+ Or to clone and install locally:
26
+ .. code:: bash
27
+
28
+ git clone https://github.com/ryandewolfe33/infoweight.git &&
29
+ cd infoweight &&
30
+ pip install .
31
+
32
+ To install the package from source:
33
+
34
+ .. code:: bash
35
+
36
+ pip install https://github.com/ryandewolfe33/infoweight/archive/master.zip
37
+
38
+ ------------
39
+ Contributing
40
+ ------------
41
+
42
+ Contributions are more than welcome! There are lots of opportunities
43
+ for potential projects, so please get in touch if you would like to
44
+ help out. Everything from code to notebooks to examples and documentation
45
+ are all *equally valuable* so please don't feel you can't contribute.
46
+
47
+
48
+ To contribute please `fork the project <https://github.com/ryandewolfe33/infoweight/issues#fork-destination-box>`_
49
+ make your changes and submit a pull request. We will do our best to work
50
+ through any issues with you and get your code merged into the main branch.
51
+
52
+ -------
53
+ License
54
+ -------
55
+
56
+ The infoweight package is 3-clause BSD licensed.
@@ -0,0 +1,4 @@
1
+ from .infoweight import (
2
+ InformationWeightTransformer,
3
+ information_weight,
4
+ )
@@ -0,0 +1,375 @@
1
+ import numba
2
+ import numpy as np
3
+ from sklearn.base import BaseEstimator, TransformerMixin
4
+ import scipy.sparse
5
+
6
+ MOCK_TARGET = np.ones(1, dtype=np.int64)
7
+ MOCK_BOOL = np.ones(1, dtype=np.bool)
8
+
9
+
10
+ @numba.njit(nogil=True)
11
+ def column_kl_divergence_exact_prior(
12
+ count_indices,
13
+ count_data,
14
+ baseline_probabilities,
15
+ prior_strength=0.1,
16
+ target=MOCK_TARGET,
17
+ ):
18
+ observed_norm = count_data.sum() + prior_strength
19
+ observed_zero_constant = (prior_strength / observed_norm) * np.log(
20
+ prior_strength / observed_norm
21
+ )
22
+ result = 0.0
23
+ count_indices_set = set(count_indices)
24
+ for i in range(baseline_probabilities.shape[0]):
25
+ if i in count_indices_set:
26
+ idx = np.searchsorted(count_indices, i)
27
+ observed_probability = (
28
+ count_data[idx] + prior_strength * baseline_probabilities[i]
29
+ ) / observed_norm
30
+ if observed_probability > 0.0:
31
+ result += observed_probability * np.log(
32
+ observed_probability / baseline_probabilities[i]
33
+ )
34
+ else:
35
+ result += baseline_probabilities[i] * observed_zero_constant
36
+
37
+ return result
38
+
39
+
40
+ @numba.njit(nogil=True)
41
+ def column_kl_divergence_approx_prior(
42
+ count_indices,
43
+ count_data,
44
+ baseline_probabilities,
45
+ prior_strength=0.1,
46
+ target=MOCK_TARGET,
47
+ ):
48
+ observed_norm = count_data.sum() + prior_strength
49
+ observed_zero_constant = (prior_strength / observed_norm) * np.log(
50
+ prior_strength / observed_norm
51
+ )
52
+ result = 0.0
53
+ zero_count_component_estimate = (
54
+ np.mean(baseline_probabilities)
55
+ * observed_zero_constant
56
+ * (baseline_probabilities.shape[0] - count_indices.shape[0])
57
+ )
58
+ result += zero_count_component_estimate
59
+ for i in range(count_indices.shape[0]):
60
+ idx = count_indices[i]
61
+ observed_probability = (
62
+ count_data[i] + prior_strength * baseline_probabilities[idx]
63
+ ) / observed_norm
64
+ if observed_probability > 0.0 and baseline_probabilities[idx] > 0:
65
+ result += observed_probability * np.log(
66
+ observed_probability / baseline_probabilities[idx]
67
+ )
68
+
69
+ return result
70
+
71
+
72
+ @numba.njit(nogil=True)
73
+ def supervised_column_kl(
74
+ count_indices,
75
+ count_data,
76
+ baseline_probabilities,
77
+ prior_strength=0.1,
78
+ target=MOCK_TARGET,
79
+ ):
80
+ observed = np.zeros_like(baseline_probabilities)
81
+ for i in range(count_indices.shape[0]):
82
+ idx = count_indices[i]
83
+ label = target[idx]
84
+ if label >= 0:
85
+ observed[label] += count_data[i]
86
+
87
+ observed += prior_strength * baseline_probabilities
88
+ observed /= observed.sum()
89
+
90
+ # Zeros in baseline_probabilities may cause nans in the log
91
+ # But this can only happen when observed is also 0, so due
92
+ # to the multiplication it does not contribute to the sum
93
+ non_zero = observed > 0
94
+ result = np.sum(
95
+ observed[non_zero]
96
+ * np.log(observed[non_zero] / baseline_probabilities[non_zero])
97
+ )
98
+ return result
99
+
100
+
101
+ @numba.njit(nogil=True, parallel=True)
102
+ def column_weights(
103
+ indptr,
104
+ indices,
105
+ data,
106
+ baseline_probabilities,
107
+ column_kl_divergence_func,
108
+ prior_strength=0.1,
109
+ target=MOCK_TARGET,
110
+ column_groups=None,
111
+ ):
112
+ n_cols = indptr.shape[0] - 1
113
+ weights = np.ones(n_cols)
114
+ for i in numba.prange(n_cols):
115
+ group = 0
116
+ if column_groups is not None:
117
+ group = column_groups[i]
118
+ weights[i] = column_kl_divergence_func(
119
+ indices[indptr[i] : indptr[i + 1]],
120
+ data[indptr[i] : indptr[i + 1]],
121
+ baseline_probabilities[group, :],
122
+ prior_strength=prior_strength,
123
+ target=target,
124
+ )
125
+ return weights
126
+
127
+
128
+ @numba.njit(nogil=True)
129
+ def compute_baseline_probabilities(
130
+ indptr,
131
+ indices,
132
+ data,
133
+ target=None,
134
+ column_groups=None,
135
+ ):
136
+ """
137
+ Compute the marginals to compare each column to. Returns
138
+ an (n column groups) x (n samples) matrix (unsupervised) or an
139
+ (n column groups) x (n targets) matrix (supervised) where each
140
+ row is the marginal of the column group.
141
+
142
+ indptr, indices, and data arrays are from csr format.
143
+ """
144
+ n_groups = 1
145
+ if column_groups is not None:
146
+ n_groups = column_groups.max() + 1
147
+ n_targets = indptr.shape[0] - 1
148
+ if target is not None:
149
+ n_targets = target.max() + 1
150
+ counts = np.zeros((n_groups, n_targets), dtype=np.int64)
151
+ for row in range(indptr.shape[0] - 1):
152
+ this_target = row
153
+ if target is not None:
154
+ if target[row] >= 0:
155
+ this_target = target[row]
156
+ else:
157
+ continue
158
+ for i in range(indptr[row], indptr[row + 1]):
159
+ group = 0
160
+ if column_groups is not None:
161
+ group = column_groups[indices[i]]
162
+ counts[group, this_target] += data[i]
163
+ probabilities = counts / np.sum(counts, axis=1).reshape(-1, 1)
164
+ return probabilities
165
+
166
+
167
+ def information_weight(
168
+ data,
169
+ prior_strength=0.1,
170
+ approximate_prior=False,
171
+ target=None,
172
+ column_groups=None,
173
+ ):
174
+ """Compute information based weights for columns. The information weight
175
+ is estimated as the amount of information gained by moving from a baseline
176
+ model to a model derived from the observed counts. In practice this can be
177
+ computed as the KL-divergence between distributions. For the baseline model
178
+ we assume data will be distributed according to the row sums -- i.e.
179
+ proportional to the frequency of the row. For the observed counts we use
180
+ a background prior of pseudo counts equal to ``prior_strength`` times the
181
+ baseline prior distribution. The Bayesian prior can either be computed
182
+ exactly (the default) at some computational expense, or estimated for a much
183
+ fast computation, often suitable for large or very sparse datasets.
184
+
185
+ Parameters
186
+ ----------
187
+ data: scipy sparse matrix (n_samples, n_features)
188
+ A matrix of count data where rows represent observations and
189
+ columns represent features. Column weightings will be learned
190
+ from this data.
191
+
192
+ prior_strength: float (optional, default=0.1)
193
+ How strongly to weight the prior when doing a Bayesian update to
194
+ derive a model based on observed counts of a column.
195
+
196
+ approximate_prior: bool (optional, default=False)
197
+ Whether to approximate weights based on the Bayesian prior or perform
198
+ exact computations. Approximations are much faster especially for very
199
+ large or very sparse datasets.
200
+
201
+ target: ndarray or None (optional, default=None)
202
+ If supervised target labels are available, these can be used to define distributions
203
+ over the target classes rather than over rows, allowing weights to be
204
+ supervised and target based. If None then unsupervised weighting is used.
205
+
206
+ column_groups: ndarray or None (optional, default=None)
207
+ If columns have a natural grouping, i.e. cols 10-15 are a one-hot-encoding of a single
208
+ categorical variable, we should compare the column distribution to the within group
209
+ marginal. If passed None then all columns have the same group.
210
+
211
+ Returns
212
+ -------
213
+ weights: ndarray of shape (n_features,)
214
+ The learned weights to be applied to columns based on the amount
215
+ of information provided by the column.
216
+ """
217
+ if target is not None:
218
+ column_kl_divergence_func = supervised_column_kl
219
+ elif approximate_prior:
220
+ column_kl_divergence_func = column_kl_divergence_approx_prior
221
+ else:
222
+ column_kl_divergence_func = column_kl_divergence_exact_prior
223
+
224
+ csr_data = data.tocsr()
225
+ baseline_probabilities = compute_baseline_probabilities(
226
+ csr_data.indptr,
227
+ csr_data.indices,
228
+ csr_data.data,
229
+ target,
230
+ column_groups,
231
+ )
232
+
233
+ csc_data = data.tocsc()
234
+ csc_data.sort_indices()
235
+ weights = column_weights(
236
+ csc_data.indptr,
237
+ csc_data.indices,
238
+ csc_data.data,
239
+ baseline_probabilities,
240
+ column_kl_divergence_func,
241
+ prior_strength=prior_strength,
242
+ target=target,
243
+ column_groups=column_groups,
244
+ )
245
+
246
+ return weights
247
+
248
+
249
+ class InformationWeightTransformer(BaseEstimator, TransformerMixin):
250
+ """A data transformer that re-weights columns of count data. Column weights
251
+ are computed as information based weights for columns. The information weight
252
+ is estimated as the amount of information gained by moving from a baseline
253
+ model to a model derived from the observed counts. In practice this can be
254
+ computed as the KL-divergence between distributions. For the baseline model
255
+ we assume data will be distributed according to the row sums -- i.e.
256
+ proportional to the frequency of the row. For the observed counts we use
257
+ a background prior of pseudo counts equal to ``prior_strength`` times the
258
+ baseline prior distribution. The Bayesian prior can either be computed
259
+ exactly (the default) at some computational expense, or estimated for a much
260
+ fast computation, often suitable for large or very sparse datasets.
261
+
262
+ Parameters
263
+ ----------
264
+ prior_strength: float (optional, default=0.1)
265
+ How strongly to weight the prior when doing a Bayesian update to
266
+ derive a model based on observed counts of a column.
267
+
268
+ approximate_prior: bool (optional, default=False)
269
+ Whether to approximate weights based on the Bayesian prior or perform
270
+ exact computations. Approximations are much faster especially for very
271
+ large or very sparse datasets.
272
+
273
+ Attributes
274
+ ----------
275
+
276
+ information_weights_: ndarray of shape (n_features,)
277
+ The learned weights to be applied to columns based on the amount
278
+ of information provided by the column.
279
+ """
280
+
281
+ def __init__(
282
+ self,
283
+ prior_strength=1e-4,
284
+ approx_prior=True,
285
+ weight_power=2.0,
286
+ supervision_weight=0.95,
287
+ ):
288
+ self.prior_strength = prior_strength
289
+ self.approx_prior = approx_prior
290
+ self.weight_power = weight_power
291
+ self.supervision_weight = supervision_weight
292
+
293
+ def fit(self, X, y=None, column_groups=None, **fit_kwds):
294
+ """Learn the appropriate column weighting as information weights
295
+ from the observed count data ``X``.
296
+
297
+ Parameters
298
+ ----------
299
+ X: ndarray of scipy sparse matrix of shape (n_samples, n_features)
300
+ The count data to be trained on. Note that, as count data all
301
+ entries should be positive or zero.
302
+
303
+ Returns
304
+ -------
305
+ self:
306
+ The trained model.
307
+ """
308
+ if not scipy.sparse.isspmatrix(X):
309
+ X = scipy.sparse.csc_matrix(X)
310
+
311
+ self.information_weights_ = information_weight(
312
+ X,
313
+ self.prior_strength,
314
+ self.approx_prior,
315
+ column_groups=column_groups,
316
+ )
317
+
318
+ mean_weight = np.mean(self.information_weights_)
319
+ if mean_weight > 0:
320
+ self.information_weights_ /= mean_weight
321
+ # This should never happen
322
+ self.information_weights_ = np.maximum(self.information_weights_, 0.0)
323
+ self.information_weights_ = np.power(
324
+ self.information_weights_, self.weight_power
325
+ )
326
+
327
+ if y is not None:
328
+ unsupervised_power = (1.0 - self.supervision_weight) * self.weight_power
329
+ supervised_power = self.supervision_weight * self.weight_power
330
+
331
+ target_classes = np.unique(y)
332
+ target_dict = dict(
333
+ np.vstack((target_classes, np.arange(target_classes.shape[0]))).T
334
+ )
335
+ target = np.array(
336
+ [np.int64(target_dict[label]) for label in y], dtype=np.int64
337
+ )
338
+ self.supervised_weights_ = information_weight(
339
+ X,
340
+ self.prior_strength,
341
+ self.approx_prior,
342
+ target=target,
343
+ column_groups=column_groups,
344
+ )
345
+ mean_supervised_weight = np.mean(self.information_weights_)
346
+ if mean_supervised_weight > 0:
347
+ self.supervised_weights_ /= mean_supervised_weight
348
+ # This should never happen
349
+ self.supervised_weights_ = np.maximum(self.supervised_weights_, 0.0)
350
+ self.supervised_weights_ = np.power(
351
+ self.supervised_weights_, supervised_power
352
+ )
353
+
354
+ self.information_weights_ = (
355
+ self.information_weights_ * self.supervised_weights_
356
+ )
357
+
358
+ return self
359
+
360
+ def transform(self, X):
361
+ """Reweight data ``X`` based on learned information weights of columns.
362
+
363
+ Parameters
364
+ ----------
365
+ X: ndarray of scipy sparse matrix of shape (n_samples, n_features)
366
+ The count data to be transformed. Note that, as count data all
367
+ entries should be positive or zero.
368
+
369
+ Returns
370
+ -------
371
+ result: ndarray of scipy sparse matrix of shape (n_samples, n_features)
372
+ The reweighted data.
373
+ """
374
+ result = X @ scipy.sparse.diags(self.information_weights_)
375
+ return result
@@ -0,0 +1,78 @@
1
+ Metadata-Version: 2.4
2
+ Name: infoweight
3
+ Version: 0.0.1
4
+ Summary: An information theoretic generalization of TF-IDF.
5
+ Maintainer-email: Ryan DeWolfe <ryan.dewolfe@uwaterloo.ca>
6
+ License-Expression: BSD-3-Clause
7
+ Project-URL: Homepage, https://github.com/ryandewolfe33/infoweight
8
+ Project-URL: Repository, https://github.com/ryandewolfe33/infoweight
9
+ Keywords: tfidf,information theory,metric learning,unsupervised learning
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Programming Language :: Python
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Intended Audience :: Science/Research
14
+ Requires-Python: >=3.10
15
+ Description-Content-Type: text/x-rst
16
+ License-File: LICENSE
17
+ Requires-Dist: numpy>=2.0.0
18
+ Requires-Dist: numba>=0.51
19
+ Requires-Dist: scikit-learn>=0.22
20
+ Requires-Dist: scipy
21
+ Dynamic: license-file
22
+
23
+ ===========
24
+ Infoweight
25
+ ===========
26
+
27
+ This package provides an information theoretic generalization of the `TFIDFTransformer`.
28
+
29
+
30
+ This package is still under development,
31
+ The `InformationWeightTransformer` was originally part of the
32
+ `vectorizers <https://github.com/TutteInstitute/vectorizers>`_
33
+ package, but has been factored out to ease development. You can find an
34
+ overview and example of using the package in the
35
+ `vectorizers documentation <https://vectorizers.readthedocs.io/en/latest/>`_.
36
+
37
+ ----------
38
+ Installing
39
+ ----------
40
+
41
+ To install the package from PyPI:
42
+
43
+ .. code:: bash
44
+
45
+ pip install infoweight
46
+
47
+ Or to clone and install locally:
48
+ .. code:: bash
49
+
50
+ git clone https://github.com/ryandewolfe33/infoweight.git &&
51
+ cd infoweight &&
52
+ pip install .
53
+
54
+ To install the package from source:
55
+
56
+ .. code:: bash
57
+
58
+ pip install https://github.com/ryandewolfe33/infoweight/archive/master.zip
59
+
60
+ ------------
61
+ Contributing
62
+ ------------
63
+
64
+ Contributions are more than welcome! There are lots of opportunities
65
+ for potential projects, so please get in touch if you would like to
66
+ help out. Everything from code to notebooks to examples and documentation
67
+ are all *equally valuable* so please don't feel you can't contribute.
68
+
69
+
70
+ To contribute please `fork the project <https://github.com/ryandewolfe33/infoweight/issues#fork-destination-box>`_
71
+ make your changes and submit a pull request. We will do our best to work
72
+ through any issues with you and get your code merged into the main branch.
73
+
74
+ -------
75
+ License
76
+ -------
77
+
78
+ The infoweight package is 3-clause BSD licensed.
@@ -0,0 +1,11 @@
1
+ LICENSE
2
+ README.rst
3
+ pyproject.toml
4
+ infoweight/__init__.py
5
+ infoweight/infoweight.py
6
+ infoweight.egg-info/PKG-INFO
7
+ infoweight.egg-info/SOURCES.txt
8
+ infoweight.egg-info/dependency_links.txt
9
+ infoweight.egg-info/requires.txt
10
+ infoweight.egg-info/top_level.txt
11
+ tests/test_infoweight.py
@@ -0,0 +1,4 @@
1
+ numpy>=2.0.0
2
+ numba>=0.51
3
+ scikit-learn>=0.22
4
+ scipy
@@ -0,0 +1 @@
1
+ infoweight
@@ -0,0 +1,36 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.2"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "infoweight"
7
+ version = "0.0.1"
8
+ readme = "README.rst"
9
+ description = "An information theoretic generalization of TF-IDF."
10
+ maintainers = [{name = "Ryan DeWolfe", email = "ryan.dewolfe@uwaterloo.ca"}]
11
+ keywords = ["tfidf", "information theory", "metric learning", "unsupervised learning"]
12
+ license = "BSD-3-Clause"
13
+ license-files = ["LICENSE"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Programming Language :: Python",
17
+ "Operating System :: OS Independent",
18
+ "Intended Audience :: Science/Research",
19
+ ]
20
+
21
+ requires-python = ">=3.10"
22
+ dependencies = [
23
+ "numpy>=2.0.0",
24
+ "numba>=0.51",
25
+ "scikit-learn>=0.22",
26
+ "scipy",
27
+ ]
28
+
29
+ [dependency-groups]
30
+ dev = [
31
+ "pytest>=9.1.1",
32
+ ]
33
+
34
+ [project.urls]
35
+ Homepage = "https://github.com/ryandewolfe33/infoweight"
36
+ Repository = "https://github.com/ryandewolfe33/infoweight"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,72 @@
1
+ import pytest
2
+ import numpy as np
3
+ import scipy.sparse
4
+ from infoweight import InformationWeightTransformer
5
+
6
+ test_matrix = scipy.sparse.csr_matrix([[1, 2, 3], [4, 5, 6], [7, 8, 9]])
7
+ test_matrix_zero_row = scipy.sparse.csr_matrix([[1, 2, 3], [4, 5, 6], [0, 0, 0]])
8
+ test_matrix_zero_row.eliminate_zeros()
9
+ test_matrix_zero_column = scipy.sparse.csr_matrix([[1, 2, 0], [4, 5, 0], [7, 8, 0]])
10
+ test_matrix_zero_column.eliminate_zeros()
11
+
12
+ @pytest.mark.parametrize("prior_strength", [0.1, 1.0])
13
+ @pytest.mark.parametrize("approx_prior", [True, False])
14
+ def test_iw_transformer(prior_strength, approx_prior):
15
+ IWT = InformationWeightTransformer(
16
+ prior_strength=prior_strength,
17
+ approx_prior=approx_prior,
18
+ )
19
+ result = IWT.fit_transform(test_matrix)
20
+ transform = IWT.transform(test_matrix)
21
+ assert np.allclose(result.toarray(), transform.toarray())
22
+
23
+
24
+ @pytest.mark.parametrize("prior_strength", [0.1, 1.0])
25
+ @pytest.mark.parametrize("approx_prior", [True, False])
26
+ @pytest.mark.parametrize("target", [None, np.array([0, 1, 1])])
27
+ @pytest.mark.parametrize("column_groups", [None, np.array([0, 1, 1])])
28
+ def test_iw_transformer_fit_args(prior_strength, approx_prior, target, column_groups):
29
+ IWT = InformationWeightTransformer(
30
+ prior_strength=prior_strength,
31
+ approx_prior=approx_prior,
32
+ )
33
+ result = IWT.fit_transform(test_matrix, target, column_groups=column_groups)
34
+ transform = IWT.transform(test_matrix)
35
+ assert np.allclose(result.toarray(), transform.toarray())
36
+ assert np.all(IWT.information_weights_ >= 0)
37
+
38
+
39
+ @pytest.mark.parametrize("prior_strength", [0.1, 1.0])
40
+ @pytest.mark.parametrize("approx_prior", [True, False])
41
+ @pytest.mark.parametrize("target", [None, np.array([0, 1, 1])])
42
+ @pytest.mark.parametrize("column_groups", [None, np.array([0, 1, 1])])
43
+ def test_iw_transformer_zero_column(
44
+ prior_strength, approx_prior, target, column_groups
45
+ ):
46
+ IWT = InformationWeightTransformer(
47
+ prior_strength=prior_strength,
48
+ approx_prior=approx_prior,
49
+ )
50
+ result = IWT.fit_transform(
51
+ test_matrix_zero_column, target, column_groups=column_groups
52
+ )
53
+ transform = IWT.transform(test_matrix_zero_column)
54
+ assert np.allclose(result.toarray(), transform.toarray())
55
+ assert np.all(IWT.information_weights_ >= 0)
56
+
57
+
58
+ @pytest.mark.parametrize("prior_strength", [0.1, 1.0])
59
+ @pytest.mark.parametrize("approx_prior", [True, False])
60
+ @pytest.mark.parametrize("target", [None, np.array([0, 1, 1])])
61
+ @pytest.mark.parametrize("column_groups", [None, np.array([0, 1, 1])])
62
+ def test_iw_transformer_zero_row(prior_strength, approx_prior, target, column_groups):
63
+ IWT = InformationWeightTransformer(
64
+ prior_strength=prior_strength,
65
+ approx_prior=approx_prior,
66
+ )
67
+ result = IWT.fit_transform(
68
+ test_matrix_zero_row, target, column_groups=column_groups
69
+ )
70
+ transform = IWT.transform(test_matrix_zero_row)
71
+ assert np.allclose(result.toarray(), transform.toarray())
72
+ assert np.all(IWT.information_weights_ >= 0)