interrater 0.0.1.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interrater/__init__.py +1 -0
- interrater/cleaners/__init__.py +19 -0
- interrater/cleaners/_string_utils.py +37 -0
- interrater/cleaners/category_report.py +96 -0
- interrater/cleaners/cleaner_config.py +308 -0
- interrater/cleaners/cleaning_report.py +125 -0
- interrater/cleaners/drop_report.py +159 -0
- interrater/cleaners/frame_layout.py +102 -0
- interrater/cleaners/insufficient_data_policy.py +123 -0
- interrater/cleaners/matching_report.py +146 -0
- interrater/contracts.py +120 -0
- interrater/data_objects/__init__.py +3 -0
- interrater/data_objects/_ratings_validation.py +160 -0
- interrater/data_objects/ratings.py +188 -0
- interrater/semantics/__init__.py +19 -0
- interrater/semantics/categories.py +327 -0
- interrater/semantics/ratings_semantics.py +253 -0
- interrater-0.0.1.dev0.dist-info/METADATA +82 -0
- interrater-0.0.1.dev0.dist-info/RECORD +21 -0
- interrater-0.0.1.dev0.dist-info/WHEEL +4 -0
- interrater-0.0.1.dev0.dist-info/licenses/LICENSE +674 -0
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
"""What a ratings array means, declared before the data is seen."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from ..contracts import MIN_RATERS, MIN_SUBJECTS
|
|
9
|
+
from .categories import Categories, build_categories
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class RatingsSemantics:
|
|
13
|
+
"""
|
|
14
|
+
I say what a ratings array means. A numpy array does not know what it is.
|
|
15
|
+
|
|
16
|
+
My biggest contribution is that I can be built from a YAML file: one
|
|
17
|
+
declaration that constructs me, is versioned with the analysis, and doubles
|
|
18
|
+
as the documentation of what the study was.
|
|
19
|
+
|
|
20
|
+
I name the columns, I own the categories that give the numbers meaning, and
|
|
21
|
+
I record what the data was expected to look like. I hold no ratings and do
|
|
22
|
+
no counting.
|
|
23
|
+
|
|
24
|
+
I am written before the data is read, not derived from it. That is the
|
|
25
|
+
point of me: if a spreadsheet arrives with five rater columns when four
|
|
26
|
+
reviewers were hired, the fifth is a mistake to be reported, not a rater to
|
|
27
|
+
be invented. Declaring the study up front is what turns a surprise into an
|
|
28
|
+
error message.
|
|
29
|
+
|
|
30
|
+
I am reusable. The same four reviewers screening a second review need the
|
|
31
|
+
same me, which is why I know nothing about any particular dataset -- no
|
|
32
|
+
subject ids, no counts, nothing a cleaner discovered. My ``n_subjects`` is
|
|
33
|
+
an expectation you may state and I may check, never a field anything fills
|
|
34
|
+
in later.
|
|
35
|
+
|
|
36
|
+
Attributes
|
|
37
|
+
----------
|
|
38
|
+
raters:
|
|
39
|
+
Column labels, in column order, never sorted. Given a count instead of
|
|
40
|
+
names I generate ``rater0, rater1, ...``, so a label is file-relative:
|
|
41
|
+
``rater0`` means "the first rater column of whatever was fed in", not a
|
|
42
|
+
person who carries across studies. Numbering starts at zero so a label
|
|
43
|
+
matches its column index.
|
|
44
|
+
categories:
|
|
45
|
+
What the integers in the array mean, and how far apart they are. See
|
|
46
|
+
``Categories``.
|
|
47
|
+
n_subjects:
|
|
48
|
+
How many rows the data should have, or None for no expectation. Stated,
|
|
49
|
+
it is checked; absent, nothing is checked. Never populated after the
|
|
50
|
+
fact -- that would tie me to one dataset and end my reuse.
|
|
51
|
+
unique_subjects:
|
|
52
|
+
Whether repeated subject ids are an error. True by default, because a
|
|
53
|
+
repeated id can be frequently due to an export bug. Set False for a study that
|
|
54
|
+
deliberately re-shows subjects to measure intra-rater consistency.
|
|
55
|
+
Bootstrap draws repeat rows too, but they bypass validation entirely
|
|
56
|
+
and never consult me.
|
|
57
|
+
|
|
58
|
+
Examples
|
|
59
|
+
--------
|
|
60
|
+
>>> from interrater.semantics import UnorderedCategories
|
|
61
|
+
>>> sem = RatingsSemantics(
|
|
62
|
+
... raters = ["ann", "bob", "cara", "dev"],
|
|
63
|
+
... categories = UnorderedCategories(("include", "unsure", "exclude")),
|
|
64
|
+
... )
|
|
65
|
+
>>> sem.n_raters
|
|
66
|
+
4
|
|
67
|
+
>>> RatingsSemantics(raters=3, categories=sem.categories).raters
|
|
68
|
+
('rater0', 'rater1', 'rater2')
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
# ----------------------------------------------------------- Attributes #
|
|
72
|
+
|
|
73
|
+
raters: tuple[str, ...]
|
|
74
|
+
# Always a tuple, even when a count was passed: __init__ generates the
|
|
75
|
+
# names immediately, so nothing downstream ever sees an int.
|
|
76
|
+
categories: Categories
|
|
77
|
+
n_subjects: int | None
|
|
78
|
+
unique_subjects: bool
|
|
79
|
+
|
|
80
|
+
# --------------------------------------------------------- Construction #
|
|
81
|
+
|
|
82
|
+
def __init__(
|
|
83
|
+
self,
|
|
84
|
+
raters: Sequence[str] | int,
|
|
85
|
+
categories: Categories,
|
|
86
|
+
n_subjects: int | None = None,
|
|
87
|
+
unique_subjects: bool = True,
|
|
88
|
+
) -> None:
|
|
89
|
+
self.raters = self._as_rater_names(raters)
|
|
90
|
+
self.categories = categories
|
|
91
|
+
self.n_subjects = n_subjects
|
|
92
|
+
self.unique_subjects = unique_subjects
|
|
93
|
+
self._validate()
|
|
94
|
+
|
|
95
|
+
@staticmethod
|
|
96
|
+
def _as_rater_names(raters: Sequence[str] | int) -> tuple[str, ...]:
|
|
97
|
+
"""A count becomes generated names; names are taken as given."""
|
|
98
|
+
if isinstance(raters, int):
|
|
99
|
+
if raters < MIN_RATERS:
|
|
100
|
+
raise ValueError(
|
|
101
|
+
f"RatingsSemantics needs at least {MIN_RATERS} raters, "
|
|
102
|
+
f"got {raters}. Agreement is undefined for a single rater."
|
|
103
|
+
)
|
|
104
|
+
return tuple(f"rater{i}" for i in range(raters))
|
|
105
|
+
return tuple(str(name) for name in raters)
|
|
106
|
+
|
|
107
|
+
def _validate(self) -> None:
|
|
108
|
+
name = type(self).__name__
|
|
109
|
+
|
|
110
|
+
if not isinstance(self.categories, Categories):
|
|
111
|
+
raise TypeError(
|
|
112
|
+
f"{name}.categories must be a Categories, got "
|
|
113
|
+
f"{type(self.categories).__name__}. Use UnorderedCategories "
|
|
114
|
+
f"for a nominal scheme, OrderedCategories for a ranked one, or "
|
|
115
|
+
f"WeightedCategories to supply distances directly."
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
if len(self.raters) < MIN_RATERS:
|
|
119
|
+
raise ValueError(
|
|
120
|
+
f"{name} needs at least {MIN_RATERS} raters, got "
|
|
121
|
+
f"{len(self.raters)}. Agreement is undefined for a single "
|
|
122
|
+
f"rater."
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
seen: set[str] = set()
|
|
126
|
+
dupes = [r for r in self.raters if r in seen or seen.add(r)]
|
|
127
|
+
if dupes:
|
|
128
|
+
raise ValueError(
|
|
129
|
+
f"{name}.raters contains {len(dupes)} duplicate name(s) "
|
|
130
|
+
f"(e.g. {dupes[:3]}). Two columns naming the same rater would "
|
|
131
|
+
f"be counted as independent by every coefficient."
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
blank = [i for i, r in enumerate(self.raters) if not r.strip()]
|
|
135
|
+
if blank:
|
|
136
|
+
raise ValueError(
|
|
137
|
+
f"{name}.raters has {len(blank)} blank name(s) at position(s) "
|
|
138
|
+
f"{blank[:3]}. Give every column a name, or pass a count and "
|
|
139
|
+
f"let me generate rater0, rater1, ..."
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
if self.n_subjects is not None and self.n_subjects < MIN_SUBJECTS:
|
|
143
|
+
raise ValueError(
|
|
144
|
+
f"{name}.n_subjects must be at least {MIN_SUBJECTS} if "
|
|
145
|
+
f"stated, got {self.n_subjects}. Pass None for no expectation."
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
# -------------------------------------------------------- Declarative #
|
|
149
|
+
|
|
150
|
+
@classmethod
|
|
151
|
+
def from_dict(cls, spec: dict) -> RatingsSemantics:
|
|
152
|
+
"""
|
|
153
|
+
Build from a plain mapping, as parsed from YAML.
|
|
154
|
+
|
|
155
|
+
``raters`` is either a list of names or an integer count.
|
|
156
|
+
``categories`` is a mapping dispatched by ``kind``; see
|
|
157
|
+
``build_categories``.
|
|
158
|
+
"""
|
|
159
|
+
missing = {"raters", "categories"} - set(spec)
|
|
160
|
+
if missing:
|
|
161
|
+
raise KeyError(
|
|
162
|
+
f"RatingsSemantics config is missing {sorted(missing)}. A "
|
|
163
|
+
f"study must declare who rated and what they could say."
|
|
164
|
+
)
|
|
165
|
+
return cls(
|
|
166
|
+
raters = spec["raters"],
|
|
167
|
+
categories = build_categories(spec["categories"]),
|
|
168
|
+
n_subjects = spec.get("n_subjects"),
|
|
169
|
+
unique_subjects = spec.get("unique_subjects", True),
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
@classmethod
|
|
173
|
+
def build_from_yaml(cls, path: str | Path) -> RatingsSemantics:
|
|
174
|
+
"""
|
|
175
|
+
Build from a YAML file.
|
|
176
|
+
|
|
177
|
+
The file is the study declaration, versioned alongside the analysis:
|
|
178
|
+
|
|
179
|
+
.. code-block:: yaml
|
|
180
|
+
|
|
181
|
+
raters: [ann, bob, cara, dev]
|
|
182
|
+
categories:
|
|
183
|
+
kind: ordered
|
|
184
|
+
names: [mild, moderate, severe]
|
|
185
|
+
weighting: quadratic
|
|
186
|
+
n_subjects: 1617
|
|
187
|
+
unique_subjects: true
|
|
188
|
+
"""
|
|
189
|
+
import yaml
|
|
190
|
+
|
|
191
|
+
with open(path, "r", encoding="utf-8") as handle:
|
|
192
|
+
spec = yaml.safe_load(handle)
|
|
193
|
+
if not isinstance(spec, dict):
|
|
194
|
+
raise TypeError(
|
|
195
|
+
f"{path} must contain a YAML mapping, got "
|
|
196
|
+
f"{type(spec).__name__}."
|
|
197
|
+
)
|
|
198
|
+
return cls.from_dict(spec)
|
|
199
|
+
|
|
200
|
+
# ------------------------------------------------------------ Public API #
|
|
201
|
+
|
|
202
|
+
@property
|
|
203
|
+
def n_raters(self) -> int:
|
|
204
|
+
return len(self.raters)
|
|
205
|
+
|
|
206
|
+
@property
|
|
207
|
+
def n_categories(self) -> int:
|
|
208
|
+
"""K, owned by my categories. Never inferred from any array."""
|
|
209
|
+
return self.categories.n_categories
|
|
210
|
+
|
|
211
|
+
def index_of_rater(self, rater: str) -> int:
|
|
212
|
+
"""Which column holds ``rater``."""
|
|
213
|
+
try:
|
|
214
|
+
return self.raters.index(rater)
|
|
215
|
+
except ValueError:
|
|
216
|
+
raise KeyError(
|
|
217
|
+
f"{type(self).__name__} has no rater {rater!r}. Known: "
|
|
218
|
+
f"{list(self.raters)}."
|
|
219
|
+
) from None
|
|
220
|
+
|
|
221
|
+
def __eq__(self, other: object) -> bool:
|
|
222
|
+
"""
|
|
223
|
+
Two of me are equal when we declare the same study.
|
|
224
|
+
|
|
225
|
+
``other`` is annotated ``object`` and returns ``NotImplemented`` rather
|
|
226
|
+
than raising, because that is the protocol: Python then tries the
|
|
227
|
+
reflected comparison and falls back to False. Raising here would break
|
|
228
|
+
``sem in [a, b]``, dict comparison, and any assertion that compares me
|
|
229
|
+
with something else -- all of which are supposed to answer False, not
|
|
230
|
+
explode. A test helper is the place for a loud, diffing comparison.
|
|
231
|
+
"""
|
|
232
|
+
if not isinstance(other, RatingsSemantics):
|
|
233
|
+
return NotImplemented
|
|
234
|
+
return (
|
|
235
|
+
self.raters == other.raters
|
|
236
|
+
and self.categories == other.categories
|
|
237
|
+
and self.n_subjects == other.n_subjects
|
|
238
|
+
and self.unique_subjects == other.unique_subjects
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
__hash__ = None # type: ignore[assignment]
|
|
242
|
+
|
|
243
|
+
def __repr__(self) -> str:
|
|
244
|
+
expected = "any" if self.n_subjects is None else f"{self.n_subjects:,}"
|
|
245
|
+
return (
|
|
246
|
+
f"RatingsSemantics(\n"
|
|
247
|
+
f" raters = {list(self.raters)}\n"
|
|
248
|
+
f" categories = {type(self.categories).__name__}"
|
|
249
|
+
f"{list(self.categories.names)}\n"
|
|
250
|
+
f" n_subjects = {expected}\n"
|
|
251
|
+
f" unique_subjects = {self.unique_subjects}\n"
|
|
252
|
+
f")"
|
|
253
|
+
)
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: interrater
|
|
3
|
+
Version: 0.0.1.dev0
|
|
4
|
+
Summary: Inter-rater agreement for systematic reviews: auditable cleaning, bootstrap intervals, and simulated ratings with known truth
|
|
5
|
+
Project-URL: Homepage, https://github.com/djarenas/Inter-Rater
|
|
6
|
+
Project-URL: Repository, https://github.com/djarenas/Inter-Rater
|
|
7
|
+
Project-URL: Issues, https://github.com/djarenas/Inter-Rater/issues
|
|
8
|
+
Project-URL: Prior version (arXiv), https://arxiv.org/abs/1809.05731
|
|
9
|
+
Project-URL: Prior version (Zenodo), https://doi.org/10.5281/zenodo.1227660
|
|
10
|
+
Author: Daniel J. Arenas
|
|
11
|
+
License-Expression: GPL-3.0-or-later
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Keywords: bootstrap,cohens-kappa,fleiss-kappa,inter-rater-reliability,interrater-agreement,krippendorff-alpha,systematic-review
|
|
14
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
25
|
+
Requires-Python: >=3.10
|
|
26
|
+
Requires-Dist: numpy>=1.24
|
|
27
|
+
Requires-Dist: pandas>=2.0
|
|
28
|
+
Requires-Dist: pyyaml>=6.0
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# interrater
|
|
34
|
+
|
|
35
|
+
Inter-rater agreement for systematic reviews and other studies where not every
|
|
36
|
+
rater sees every subject.
|
|
37
|
+
|
|
38
|
+
**This release reserves the package name. It is not yet usable.** Version 2 is
|
|
39
|
+
under active development; follow
|
|
40
|
+
[the repository](https://github.com/djarenas/Inter-Rater) for progress.
|
|
41
|
+
|
|
42
|
+
## Background
|
|
43
|
+
|
|
44
|
+
`interrater` is the successor to
|
|
45
|
+
[Inter-Rater](https://github.com/djarenas/Inter-Rater), described in
|
|
46
|
+
[arXiv:1809.05731](https://arxiv.org/abs/1809.05731) (2018) and archived at
|
|
47
|
+
[Zenodo](https://doi.org/10.5281/zenodo.1227660). Version 1 has been used in
|
|
48
|
+
published systematic reviews to compute Fleiss' kappa and pairwise Cohen's
|
|
49
|
+
kappa across multiple reviewers.
|
|
50
|
+
|
|
51
|
+
Version 2 is a complete rewrite with a different API. Version 1 remains
|
|
52
|
+
available at git tag `v1.5`.
|
|
53
|
+
|
|
54
|
+
## What version 2 adds
|
|
55
|
+
|
|
56
|
+
- **Bootstrap confidence intervals.** Version 1 reported an analytic standard
|
|
57
|
+
error for the average of pairwise kappas, which treats those kappas as
|
|
58
|
+
independent. They are not: two pairwise coefficients sharing a rater are
|
|
59
|
+
correlated, and the resulting interval is too narrow. Resampling subjects
|
|
60
|
+
preserves that correlation without having to model it.
|
|
61
|
+
|
|
62
|
+
- **An auditable cleaning layer.** Which strings meant nothing, which were
|
|
63
|
+
repaired and into what, how thin the data was allowed to get: declared in a
|
|
64
|
+
versioned config file and reported afterwards, rather than reconstructed from
|
|
65
|
+
memory when a reviewer asks.
|
|
66
|
+
|
|
67
|
+
- **Simulated ratings with known truth.** Generate data from an explicit model
|
|
68
|
+
of rater accuracy, category prevalence and subject assignment, so an
|
|
69
|
+
estimator can be checked against an answer known in advance and a confidence
|
|
70
|
+
interval can be checked for actually covering it.
|
|
71
|
+
|
|
72
|
+
- **Support for unbalanced designs.** Where each subject is rated by a subset
|
|
73
|
+
of the reviewer pool rather than all of them, which is what most reviews can
|
|
74
|
+
actually afford.
|
|
75
|
+
|
|
76
|
+
## Requirements
|
|
77
|
+
|
|
78
|
+
Python 3.10 or later.
|
|
79
|
+
|
|
80
|
+
## License
|
|
81
|
+
|
|
82
|
+
GNU General Public License v3 or later. Same license as version 1.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
interrater/__init__.py,sha256=PicnolOWJRpMczbpOTps44vc7o4SFLNODjd3IoP1IOg,69
|
|
2
|
+
interrater/contracts.py,sha256=MiKUQLtAw60mAZo3FxaxPYveNW1eFvLUfLGHRnAYGV8,4846
|
|
3
|
+
interrater/cleaners/__init__.py,sha256=ewgV1RO2se9wXbKXH63TMP1kaOBEiFy9TLbxqjJL8eE,544
|
|
4
|
+
interrater/cleaners/_string_utils.py,sha256=YY72TLUVtE1sH_g6fKZPN9n5B74wVsBmWYg_AKuR1xI,1333
|
|
5
|
+
interrater/cleaners/category_report.py,sha256=BkjvJt-qpoZTDQ3pogmsjQVSAv-5NUlqmOZWlBTtANA,3376
|
|
6
|
+
interrater/cleaners/cleaner_config.py,sha256=bbcqUCuX88KdaMT63mdDVYyp6mdmEeeHu1Au5_yCREk,12866
|
|
7
|
+
interrater/cleaners/cleaning_report.py,sha256=zGF1njuF28kBhfY_pLX_U-d_hlVXAMH8K1DCizQRkVE,4462
|
|
8
|
+
interrater/cleaners/drop_report.py,sha256=lDmCc5DEELUqyD5EA8rrWyEvrWXynb0jRercc0MOVUw,5599
|
|
9
|
+
interrater/cleaners/frame_layout.py,sha256=5p5F3ILUGmX_VFnsr54yz1bokpqJG7tAifeImmtiswU,3819
|
|
10
|
+
interrater/cleaners/insufficient_data_policy.py,sha256=MlwS1iz54FUpXgb5L-Yf9mZnF8CQ_NTGx3KYbS_w1Zs,4852
|
|
11
|
+
interrater/cleaners/matching_report.py,sha256=Hs8R0cH2zBgGzfdA598RJcDecYqkrX6EY7r1Yjf6i8U,6132
|
|
12
|
+
interrater/data_objects/__init__.py,sha256=f0Wo0vs-xVBPE06cCCd6OpF6t1jb18oLw76n0IduCi8,52
|
|
13
|
+
interrater/data_objects/_ratings_validation.py,sha256=IUgto3ltjHIw0i2cZGBfMnuyRbo1Y3H5rEQHiDwhqiE,5722
|
|
14
|
+
interrater/data_objects/ratings.py,sha256=H_PguBmhXwauDghLlA4L_cmhoNjLVDu8BxFjubY2wPI,7529
|
|
15
|
+
interrater/semantics/__init__.py,sha256=yOahD7VTbSa4Wp-EiNPpO2VYVLt-Ie8vABkBEjFM2QM,395
|
|
16
|
+
interrater/semantics/categories.py,sha256=6asxTX376oAFou6REvPk6lLKyhNGgdBlGZhRupnX1as,11659
|
|
17
|
+
interrater/semantics/ratings_semantics.py,sha256=WFM-79lISHk2lF3IHcJMrNf2o3EJJSby_HTBDSb7P6k,9943
|
|
18
|
+
interrater-0.0.1.dev0.dist-info/METADATA,sha256=FguP1Bfy2hhqK0CHERpoXBooNCLOdTElG-QQ9HjqcJ8,3598
|
|
19
|
+
interrater-0.0.1.dev0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
20
|
+
interrater-0.0.1.dev0.dist-info/licenses/LICENSE,sha256=jOtLnuWt7d5Hsx6XXB2QxzrSe2sWWh3NgMfFRetluQM,35147
|
|
21
|
+
interrater-0.0.1.dev0.dist-info/RECORD,,
|