interrater 0.0.1.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,253 @@
1
+ """What a ratings array means, declared before the data is seen."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Sequence
6
+ from pathlib import Path
7
+
8
+ from ..contracts import MIN_RATERS, MIN_SUBJECTS
9
+ from .categories import Categories, build_categories
10
+
11
+
12
+ class RatingsSemantics:
13
+ """
14
+ I say what a ratings array means. A numpy array does not know what it is.
15
+
16
+ My biggest contribution is that I can be built from a YAML file: one
17
+ declaration that constructs me, is versioned with the analysis, and doubles
18
+ as the documentation of what the study was.
19
+
20
+ I name the columns, I own the categories that give the numbers meaning, and
21
+ I record what the data was expected to look like. I hold no ratings and do
22
+ no counting.
23
+
24
+ I am written before the data is read, not derived from it. That is the
25
+ point of me: if a spreadsheet arrives with five rater columns when four
26
+ reviewers were hired, the fifth is a mistake to be reported, not a rater to
27
+ be invented. Declaring the study up front is what turns a surprise into an
28
+ error message.
29
+
30
+ I am reusable. The same four reviewers screening a second review need the
31
+ same me, which is why I know nothing about any particular dataset -- no
32
+ subject ids, no counts, nothing a cleaner discovered. My ``n_subjects`` is
33
+ an expectation you may state and I may check, never a field anything fills
34
+ in later.
35
+
36
+ Attributes
37
+ ----------
38
+ raters:
39
+ Column labels, in column order, never sorted. Given a count instead of
40
+ names I generate ``rater0, rater1, ...``, so a label is file-relative:
41
+ ``rater0`` means "the first rater column of whatever was fed in", not a
42
+ person who carries across studies. Numbering starts at zero so a label
43
+ matches its column index.
44
+ categories:
45
+ What the integers in the array mean, and how far apart they are. See
46
+ ``Categories``.
47
+ n_subjects:
48
+ How many rows the data should have, or None for no expectation. Stated,
49
+ it is checked; absent, nothing is checked. Never populated after the
50
+ fact -- that would tie me to one dataset and end my reuse.
51
+ unique_subjects:
52
+ Whether repeated subject ids are an error. True by default, because a
53
+ repeated id can be frequently due to an export bug. Set False for a study that
54
+ deliberately re-shows subjects to measure intra-rater consistency.
55
+ Bootstrap draws repeat rows too, but they bypass validation entirely
56
+ and never consult me.
57
+
58
+ Examples
59
+ --------
60
+ >>> from interrater.semantics import UnorderedCategories
61
+ >>> sem = RatingsSemantics(
62
+ ... raters = ["ann", "bob", "cara", "dev"],
63
+ ... categories = UnorderedCategories(("include", "unsure", "exclude")),
64
+ ... )
65
+ >>> sem.n_raters
66
+ 4
67
+ >>> RatingsSemantics(raters=3, categories=sem.categories).raters
68
+ ('rater0', 'rater1', 'rater2')
69
+ """
70
+
71
+ # ----------------------------------------------------------- Attributes #
72
+
73
+ raters: tuple[str, ...]
74
+ # Always a tuple, even when a count was passed: __init__ generates the
75
+ # names immediately, so nothing downstream ever sees an int.
76
+ categories: Categories
77
+ n_subjects: int | None
78
+ unique_subjects: bool
79
+
80
+ # --------------------------------------------------------- Construction #
81
+
82
+ def __init__(
83
+ self,
84
+ raters: Sequence[str] | int,
85
+ categories: Categories,
86
+ n_subjects: int | None = None,
87
+ unique_subjects: bool = True,
88
+ ) -> None:
89
+ self.raters = self._as_rater_names(raters)
90
+ self.categories = categories
91
+ self.n_subjects = n_subjects
92
+ self.unique_subjects = unique_subjects
93
+ self._validate()
94
+
95
+ @staticmethod
96
+ def _as_rater_names(raters: Sequence[str] | int) -> tuple[str, ...]:
97
+ """A count becomes generated names; names are taken as given."""
98
+ if isinstance(raters, int):
99
+ if raters < MIN_RATERS:
100
+ raise ValueError(
101
+ f"RatingsSemantics needs at least {MIN_RATERS} raters, "
102
+ f"got {raters}. Agreement is undefined for a single rater."
103
+ )
104
+ return tuple(f"rater{i}" for i in range(raters))
105
+ return tuple(str(name) for name in raters)
106
+
107
+ def _validate(self) -> None:
108
+ name = type(self).__name__
109
+
110
+ if not isinstance(self.categories, Categories):
111
+ raise TypeError(
112
+ f"{name}.categories must be a Categories, got "
113
+ f"{type(self.categories).__name__}. Use UnorderedCategories "
114
+ f"for a nominal scheme, OrderedCategories for a ranked one, or "
115
+ f"WeightedCategories to supply distances directly."
116
+ )
117
+
118
+ if len(self.raters) < MIN_RATERS:
119
+ raise ValueError(
120
+ f"{name} needs at least {MIN_RATERS} raters, got "
121
+ f"{len(self.raters)}. Agreement is undefined for a single "
122
+ f"rater."
123
+ )
124
+
125
+ seen: set[str] = set()
126
+ dupes = [r for r in self.raters if r in seen or seen.add(r)]
127
+ if dupes:
128
+ raise ValueError(
129
+ f"{name}.raters contains {len(dupes)} duplicate name(s) "
130
+ f"(e.g. {dupes[:3]}). Two columns naming the same rater would "
131
+ f"be counted as independent by every coefficient."
132
+ )
133
+
134
+ blank = [i for i, r in enumerate(self.raters) if not r.strip()]
135
+ if blank:
136
+ raise ValueError(
137
+ f"{name}.raters has {len(blank)} blank name(s) at position(s) "
138
+ f"{blank[:3]}. Give every column a name, or pass a count and "
139
+ f"let me generate rater0, rater1, ..."
140
+ )
141
+
142
+ if self.n_subjects is not None and self.n_subjects < MIN_SUBJECTS:
143
+ raise ValueError(
144
+ f"{name}.n_subjects must be at least {MIN_SUBJECTS} if "
145
+ f"stated, got {self.n_subjects}. Pass None for no expectation."
146
+ )
147
+
148
+ # -------------------------------------------------------- Declarative #
149
+
150
+ @classmethod
151
+ def from_dict(cls, spec: dict) -> RatingsSemantics:
152
+ """
153
+ Build from a plain mapping, as parsed from YAML.
154
+
155
+ ``raters`` is either a list of names or an integer count.
156
+ ``categories`` is a mapping dispatched by ``kind``; see
157
+ ``build_categories``.
158
+ """
159
+ missing = {"raters", "categories"} - set(spec)
160
+ if missing:
161
+ raise KeyError(
162
+ f"RatingsSemantics config is missing {sorted(missing)}. A "
163
+ f"study must declare who rated and what they could say."
164
+ )
165
+ return cls(
166
+ raters = spec["raters"],
167
+ categories = build_categories(spec["categories"]),
168
+ n_subjects = spec.get("n_subjects"),
169
+ unique_subjects = spec.get("unique_subjects", True),
170
+ )
171
+
172
+ @classmethod
173
+ def build_from_yaml(cls, path: str | Path) -> RatingsSemantics:
174
+ """
175
+ Build from a YAML file.
176
+
177
+ The file is the study declaration, versioned alongside the analysis:
178
+
179
+ .. code-block:: yaml
180
+
181
+ raters: [ann, bob, cara, dev]
182
+ categories:
183
+ kind: ordered
184
+ names: [mild, moderate, severe]
185
+ weighting: quadratic
186
+ n_subjects: 1617
187
+ unique_subjects: true
188
+ """
189
+ import yaml
190
+
191
+ with open(path, "r", encoding="utf-8") as handle:
192
+ spec = yaml.safe_load(handle)
193
+ if not isinstance(spec, dict):
194
+ raise TypeError(
195
+ f"{path} must contain a YAML mapping, got "
196
+ f"{type(spec).__name__}."
197
+ )
198
+ return cls.from_dict(spec)
199
+
200
+ # ------------------------------------------------------------ Public API #
201
+
202
+ @property
203
+ def n_raters(self) -> int:
204
+ return len(self.raters)
205
+
206
+ @property
207
+ def n_categories(self) -> int:
208
+ """K, owned by my categories. Never inferred from any array."""
209
+ return self.categories.n_categories
210
+
211
+ def index_of_rater(self, rater: str) -> int:
212
+ """Which column holds ``rater``."""
213
+ try:
214
+ return self.raters.index(rater)
215
+ except ValueError:
216
+ raise KeyError(
217
+ f"{type(self).__name__} has no rater {rater!r}. Known: "
218
+ f"{list(self.raters)}."
219
+ ) from None
220
+
221
+ def __eq__(self, other: object) -> bool:
222
+ """
223
+ Two of me are equal when we declare the same study.
224
+
225
+ ``other`` is annotated ``object`` and returns ``NotImplemented`` rather
226
+ than raising, because that is the protocol: Python then tries the
227
+ reflected comparison and falls back to False. Raising here would break
228
+ ``sem in [a, b]``, dict comparison, and any assertion that compares me
229
+ with something else -- all of which are supposed to answer False, not
230
+ explode. A test helper is the place for a loud, diffing comparison.
231
+ """
232
+ if not isinstance(other, RatingsSemantics):
233
+ return NotImplemented
234
+ return (
235
+ self.raters == other.raters
236
+ and self.categories == other.categories
237
+ and self.n_subjects == other.n_subjects
238
+ and self.unique_subjects == other.unique_subjects
239
+ )
240
+
241
+ __hash__ = None # type: ignore[assignment]
242
+
243
+ def __repr__(self) -> str:
244
+ expected = "any" if self.n_subjects is None else f"{self.n_subjects:,}"
245
+ return (
246
+ f"RatingsSemantics(\n"
247
+ f" raters = {list(self.raters)}\n"
248
+ f" categories = {type(self.categories).__name__}"
249
+ f"{list(self.categories.names)}\n"
250
+ f" n_subjects = {expected}\n"
251
+ f" unique_subjects = {self.unique_subjects}\n"
252
+ f")"
253
+ )
@@ -0,0 +1,82 @@
1
+ Metadata-Version: 2.5
2
+ Name: interrater
3
+ Version: 0.0.1.dev0
4
+ Summary: Inter-rater agreement for systematic reviews: auditable cleaning, bootstrap intervals, and simulated ratings with known truth
5
+ Project-URL: Homepage, https://github.com/djarenas/Inter-Rater
6
+ Project-URL: Repository, https://github.com/djarenas/Inter-Rater
7
+ Project-URL: Issues, https://github.com/djarenas/Inter-Rater/issues
8
+ Project-URL: Prior version (arXiv), https://arxiv.org/abs/1809.05731
9
+ Project-URL: Prior version (Zenodo), https://doi.org/10.5281/zenodo.1227660
10
+ Author: Daniel J. Arenas
11
+ License-Expression: GPL-3.0-or-later
12
+ License-File: LICENSE
13
+ Keywords: bootstrap,cohens-kappa,fleiss-kappa,inter-rater-reliability,interrater-agreement,krippendorff-alpha,systematic-review
14
+ Classifier: Development Status :: 2 - Pre-Alpha
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
24
+ Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
25
+ Requires-Python: >=3.10
26
+ Requires-Dist: numpy>=1.24
27
+ Requires-Dist: pandas>=2.0
28
+ Requires-Dist: pyyaml>=6.0
29
+ Provides-Extra: dev
30
+ Requires-Dist: pytest>=8.0; extra == 'dev'
31
+ Description-Content-Type: text/markdown
32
+
33
+ # interrater
34
+
35
+ Inter-rater agreement for systematic reviews and other studies where not every
36
+ rater sees every subject.
37
+
38
+ **This release reserves the package name. It is not yet usable.** Version 2 is
39
+ under active development; follow
40
+ [the repository](https://github.com/djarenas/Inter-Rater) for progress.
41
+
42
+ ## Background
43
+
44
+ `interrater` is the successor to
45
+ [Inter-Rater](https://github.com/djarenas/Inter-Rater), described in
46
+ [arXiv:1809.05731](https://arxiv.org/abs/1809.05731) (2018) and archived at
47
+ [Zenodo](https://doi.org/10.5281/zenodo.1227660). Version 1 has been used in
48
+ published systematic reviews to compute Fleiss' kappa and pairwise Cohen's
49
+ kappa across multiple reviewers.
50
+
51
+ Version 2 is a complete rewrite with a different API. Version 1 remains
52
+ available at git tag `v1.5`.
53
+
54
+ ## What version 2 adds
55
+
56
+ - **Bootstrap confidence intervals.** Version 1 reported an analytic standard
57
+ error for the average of pairwise kappas, which treats those kappas as
58
+ independent. They are not: two pairwise coefficients sharing a rater are
59
+ correlated, and the resulting interval is too narrow. Resampling subjects
60
+ preserves that correlation without having to model it.
61
+
62
+ - **An auditable cleaning layer.** Which strings meant nothing, which were
63
+ repaired and into what, how thin the data was allowed to get: declared in a
64
+ versioned config file and reported afterwards, rather than reconstructed from
65
+ memory when a reviewer asks.
66
+
67
+ - **Simulated ratings with known truth.** Generate data from an explicit model
68
+ of rater accuracy, category prevalence and subject assignment, so an
69
+ estimator can be checked against an answer known in advance and a confidence
70
+ interval can be checked for actually covering it.
71
+
72
+ - **Support for unbalanced designs.** Where each subject is rated by a subset
73
+ of the reviewer pool rather than all of them, which is what most reviews can
74
+ actually afford.
75
+
76
+ ## Requirements
77
+
78
+ Python 3.10 or later.
79
+
80
+ ## License
81
+
82
+ GNU General Public License v3 or later. Same license as version 1.
@@ -0,0 +1,21 @@
1
+ interrater/__init__.py,sha256=PicnolOWJRpMczbpOTps44vc7o4SFLNODjd3IoP1IOg,69
2
+ interrater/contracts.py,sha256=MiKUQLtAw60mAZo3FxaxPYveNW1eFvLUfLGHRnAYGV8,4846
3
+ interrater/cleaners/__init__.py,sha256=ewgV1RO2se9wXbKXH63TMP1kaOBEiFy9TLbxqjJL8eE,544
4
+ interrater/cleaners/_string_utils.py,sha256=YY72TLUVtE1sH_g6fKZPN9n5B74wVsBmWYg_AKuR1xI,1333
5
+ interrater/cleaners/category_report.py,sha256=BkjvJt-qpoZTDQ3pogmsjQVSAv-5NUlqmOZWlBTtANA,3376
6
+ interrater/cleaners/cleaner_config.py,sha256=bbcqUCuX88KdaMT63mdDVYyp6mdmEeeHu1Au5_yCREk,12866
7
+ interrater/cleaners/cleaning_report.py,sha256=zGF1njuF28kBhfY_pLX_U-d_hlVXAMH8K1DCizQRkVE,4462
8
+ interrater/cleaners/drop_report.py,sha256=lDmCc5DEELUqyD5EA8rrWyEvrWXynb0jRercc0MOVUw,5599
9
+ interrater/cleaners/frame_layout.py,sha256=5p5F3ILUGmX_VFnsr54yz1bokpqJG7tAifeImmtiswU,3819
10
+ interrater/cleaners/insufficient_data_policy.py,sha256=MlwS1iz54FUpXgb5L-Yf9mZnF8CQ_NTGx3KYbS_w1Zs,4852
11
+ interrater/cleaners/matching_report.py,sha256=Hs8R0cH2zBgGzfdA598RJcDecYqkrX6EY7r1Yjf6i8U,6132
12
+ interrater/data_objects/__init__.py,sha256=f0Wo0vs-xVBPE06cCCd6OpF6t1jb18oLw76n0IduCi8,52
13
+ interrater/data_objects/_ratings_validation.py,sha256=IUgto3ltjHIw0i2cZGBfMnuyRbo1Y3H5rEQHiDwhqiE,5722
14
+ interrater/data_objects/ratings.py,sha256=H_PguBmhXwauDghLlA4L_cmhoNjLVDu8BxFjubY2wPI,7529
15
+ interrater/semantics/__init__.py,sha256=yOahD7VTbSa4Wp-EiNPpO2VYVLt-Ie8vABkBEjFM2QM,395
16
+ interrater/semantics/categories.py,sha256=6asxTX376oAFou6REvPk6lLKyhNGgdBlGZhRupnX1as,11659
17
+ interrater/semantics/ratings_semantics.py,sha256=WFM-79lISHk2lF3IHcJMrNf2o3EJJSby_HTBDSb7P6k,9943
18
+ interrater-0.0.1.dev0.dist-info/METADATA,sha256=FguP1Bfy2hhqK0CHERpoXBooNCLOdTElG-QQ9HjqcJ8,3598
19
+ interrater-0.0.1.dev0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
20
+ interrater-0.0.1.dev0.dist-info/licenses/LICENSE,sha256=jOtLnuWt7d5Hsx6XXB2QxzrSe2sWWh3NgMfFRetluQM,35147
21
+ interrater-0.0.1.dev0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any