interrater 0.0.1.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interrater/__init__.py +1 -0
- interrater/cleaners/__init__.py +19 -0
- interrater/cleaners/_string_utils.py +37 -0
- interrater/cleaners/category_report.py +96 -0
- interrater/cleaners/cleaner_config.py +308 -0
- interrater/cleaners/cleaning_report.py +125 -0
- interrater/cleaners/drop_report.py +159 -0
- interrater/cleaners/frame_layout.py +102 -0
- interrater/cleaners/insufficient_data_policy.py +123 -0
- interrater/cleaners/matching_report.py +146 -0
- interrater/contracts.py +120 -0
- interrater/data_objects/__init__.py +3 -0
- interrater/data_objects/_ratings_validation.py +160 -0
- interrater/data_objects/ratings.py +188 -0
- interrater/semantics/__init__.py +19 -0
- interrater/semantics/categories.py +327 -0
- interrater/semantics/ratings_semantics.py +253 -0
- interrater-0.0.1.dev0.dist-info/METADATA +82 -0
- interrater-0.0.1.dev0.dist-info/RECORD +21 -0
- interrater-0.0.1.dev0.dist-info/WHEEL +4 -0
- interrater-0.0.1.dev0.dist-info/licenses/LICENSE +674 -0
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Validation for Ratings.
|
|
3
|
+
|
|
4
|
+
Everything here raises. Semantics declares what a study should look like but
|
|
5
|
+
never sees an array; Ratings holds both, so this is where a declaration meets
|
|
6
|
+
the data that is supposed to satisfy it.
|
|
7
|
+
|
|
8
|
+
Kept apart from the arithmetic so that a bootstrapper can import the maths
|
|
9
|
+
without dragging error machinery along.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
from ..contracts import EMPTY, MIN_SUBJECTS
|
|
17
|
+
from ..semantics.ratings_semantics import RatingsSemantics
|
|
18
|
+
|
|
19
|
+
_P = "Ratings"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
# ---- The array on its own ---- #
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def validate_dtype(category_index: np.ndarray) -> None:
|
|
26
|
+
if category_index.dtype.kind == "f":
|
|
27
|
+
raise TypeError(
|
|
28
|
+
f"{_P}.category_index must be an integer array, got dtype "
|
|
29
|
+
f"{category_index.dtype}. Floats usually mean NaN was used for "
|
|
30
|
+
f"absent ratings; use EMPTY ({EMPTY}) instead, which keeps the "
|
|
31
|
+
f"array integral and indexable."
|
|
32
|
+
)
|
|
33
|
+
if category_index.dtype.kind not in "iu":
|
|
34
|
+
raise TypeError(
|
|
35
|
+
f"{_P}.category_index must be an integer array, got dtype "
|
|
36
|
+
f"{category_index.dtype}. Category names are resolved to positions "
|
|
37
|
+
f"by the cleaner; by the time an array reaches me it holds "
|
|
38
|
+
f"integers."
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def validate_shape(category_index: np.ndarray) -> None:
|
|
43
|
+
if category_index.ndim != 2:
|
|
44
|
+
raise ValueError(
|
|
45
|
+
f"{_P}.category_index must be 2-dimensional "
|
|
46
|
+
f"(n_subjects, n_raters), got shape {category_index.shape}."
|
|
47
|
+
)
|
|
48
|
+
if category_index.shape[0] < MIN_SUBJECTS:
|
|
49
|
+
raise ValueError(
|
|
50
|
+
f"{_P} needs at least {MIN_SUBJECTS} subjects, got "
|
|
51
|
+
f"{category_index.shape[0]}."
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# ---- The array against the declaration ---- #
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def validate_matches_raters(
|
|
59
|
+
category_index: np.ndarray, semantics: RatingsSemantics
|
|
60
|
+
) -> None:
|
|
61
|
+
"""The check that makes declaring a study worth the trouble."""
|
|
62
|
+
found = category_index.shape[1]
|
|
63
|
+
if found != semantics.n_raters:
|
|
64
|
+
raise ValueError(
|
|
65
|
+
f"{_P} was given {found} rater column(s) but its semantics declare "
|
|
66
|
+
f"{semantics.n_raters}: {list(semantics.raters)}. A column that "
|
|
67
|
+
f"was not declared is a mistake to look into, not a rater to "
|
|
68
|
+
f"invent; if the extra column is real, add it to the semantics."
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def validate_matches_n_subjects(
|
|
73
|
+
category_index: np.ndarray, semantics: RatingsSemantics
|
|
74
|
+
) -> None:
|
|
75
|
+
if semantics.n_subjects is None:
|
|
76
|
+
return
|
|
77
|
+
found = category_index.shape[0]
|
|
78
|
+
if found != semantics.n_subjects:
|
|
79
|
+
raise ValueError(
|
|
80
|
+
f"{_P} was given {found:,} subject row(s) but its semantics expect "
|
|
81
|
+
f"{semantics.n_subjects:,}. Either the file is not the one the "
|
|
82
|
+
f"study describes, or a cleaner dropped rows it should not have. "
|
|
83
|
+
f"Set n_subjects to None in the semantics for no expectation."
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def validate_values_in_categories(
|
|
88
|
+
category_index: np.ndarray, semantics: RatingsSemantics
|
|
89
|
+
) -> None:
|
|
90
|
+
k = semantics.n_categories
|
|
91
|
+
illegal = (category_index < EMPTY) | (category_index >= k)
|
|
92
|
+
n_bad = int(illegal.sum())
|
|
93
|
+
if n_bad:
|
|
94
|
+
examples = np.unique(category_index[illegal])[:3].tolist()
|
|
95
|
+
raise ValueError(
|
|
96
|
+
f"{_P}.category_index holds {n_bad:,} value(s) outside its "
|
|
97
|
+
f"categories (e.g. {examples}). Valid values are 0..{k - 1}, "
|
|
98
|
+
f"indexing {list(semantics.categories.names)}, or EMPTY ({EMPTY}). "
|
|
99
|
+
f"Use a cleaner to map raw responses onto the categories first."
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
# ---- Subject ids ---- #
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def validate_subject_ids_match_shape(
|
|
107
|
+
category_index: np.ndarray, subject_ids: np.ndarray
|
|
108
|
+
) -> None:
|
|
109
|
+
if len(subject_ids) != category_index.shape[0]:
|
|
110
|
+
raise ValueError(
|
|
111
|
+
f"{_P} has {category_index.shape[0]:,} subject row(s) but "
|
|
112
|
+
f"{len(subject_ids):,} subject id(s). Every row needs a label."
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def validate_subject_ids_unique(
|
|
117
|
+
subject_ids: np.ndarray, semantics: RatingsSemantics
|
|
118
|
+
) -> None:
|
|
119
|
+
if not semantics.unique_subjects:
|
|
120
|
+
return
|
|
121
|
+
uniques, counts = np.unique(subject_ids, return_counts=True)
|
|
122
|
+
repeated = uniques[counts > 1]
|
|
123
|
+
if repeated.size:
|
|
124
|
+
raise ValueError(
|
|
125
|
+
f"{_P} has {repeated.size:,} subject id(s) appearing more than "
|
|
126
|
+
f"once (e.g. {repeated[:3].tolist()}). That is usually an export "
|
|
127
|
+
f"that duplicated rows. If the study deliberately re-shows "
|
|
128
|
+
f"subjects, set unique_subjects=False in the semantics."
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# ---- Nothing may be entirely absent ---- #
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def validate_no_idle_raters(
|
|
136
|
+
category_index: np.ndarray, semantics: RatingsSemantics
|
|
137
|
+
) -> None:
|
|
138
|
+
rated = (category_index != EMPTY).sum(axis=0)
|
|
139
|
+
idle = np.flatnonzero(rated == 0)
|
|
140
|
+
if idle.size:
|
|
141
|
+
names = [semantics.raters[i] for i in idle[:3]]
|
|
142
|
+
raise ValueError(
|
|
143
|
+
f"{_P} has {idle.size} declared rater(s) who rated nothing at all "
|
|
144
|
+
f"(e.g. {names}). A rater with no ratings contributes to no "
|
|
145
|
+
f"coefficient and makes per-rater rates undefined. Drop them from "
|
|
146
|
+
f"the semantics, or find out why their column is empty."
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def validate_no_unrated_subjects(
|
|
151
|
+
category_index: np.ndarray, subject_ids: np.ndarray
|
|
152
|
+
) -> None:
|
|
153
|
+
rated = (category_index != EMPTY).sum(axis=1)
|
|
154
|
+
unrated = np.flatnonzero(rated == 0)
|
|
155
|
+
if unrated.size:
|
|
156
|
+
raise ValueError(
|
|
157
|
+
f"{_P} has {unrated.size:,} subject(s) rated by nobody (e.g. "
|
|
158
|
+
f"{subject_ids[unrated][:3].tolist()}). An all-empty row joins no "
|
|
159
|
+
f"comparison; drop these in the cleaner."
|
|
160
|
+
)
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""The data object: a validated array of ratings and what it means."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Sequence
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
|
|
9
|
+
from ..contracts import INDEX_DTYPE, LABEL_DTYPE
|
|
10
|
+
from ..semantics.ratings_semantics import RatingsSemantics
|
|
11
|
+
from . import _ratings_validation as _v
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class Ratings:
|
|
15
|
+
"""
|
|
16
|
+
I am a validated array of ratings. If I exist, I am complete.
|
|
17
|
+
|
|
18
|
+
I hold three things: a dense ``(n_subjects, n_raters)`` array of category
|
|
19
|
+
positions, a label for each row, and the semantics that say what all of it
|
|
20
|
+
means. That is the whole of me. I do not clean, analyse, transform, or
|
|
21
|
+
count -- a numpy array does not know what it is, and my job is to be the
|
|
22
|
+
array that does.
|
|
23
|
+
|
|
24
|
+
My constructor raises rather than repair. Anything holding me may assume my
|
|
25
|
+
values are in range, my shape matches what my semantics declare, and no
|
|
26
|
+
declared rater of mine sat idle.
|
|
27
|
+
|
|
28
|
+
I describe my contents, not my history. Nothing about what a cleaner
|
|
29
|
+
discarded to make me survives in me; that lives in its report, alongside
|
|
30
|
+
the config that produced it.
|
|
31
|
+
|
|
32
|
+
I compute nothing about myself either. How skewed my categories are,
|
|
33
|
+
whether every subject was rated the same number of times, which raters
|
|
34
|
+
overlap -- those are facts *about* me, and they belong to a
|
|
35
|
+
``RatingsProfile``. Keeping them out means I stay cheap to build, which
|
|
36
|
+
matters when a bootstrapper is making thousands of arrays a second.
|
|
37
|
+
|
|
38
|
+
I allow empties freely. Raters need not have rated the same subjects, which
|
|
39
|
+
is the ordinary case in a systematic review where each subject goes to a
|
|
40
|
+
subset of reviewers. An empty says only that no rating exists there, never
|
|
41
|
+
why: 'unsure' and 'abstain' are categories a study designer offered and a
|
|
42
|
+
rater chose, and they count like any other answer.
|
|
43
|
+
|
|
44
|
+
Attributes
|
|
45
|
+
----------
|
|
46
|
+
category_index:
|
|
47
|
+
(n_subjects, n_raters) int8, read-only. Each cell is a position in
|
|
48
|
+
``semantics.categories.names``, or EMPTY. Column order is rater order
|
|
49
|
+
as declared; I never sort.
|
|
50
|
+
subject_ids:
|
|
51
|
+
(n_subjects,) object. One label per row, in input order. Generated as
|
|
52
|
+
``subject0, subject1, ...`` when the data had none, since a row must be
|
|
53
|
+
nameable even if the file did not name it. May repeat only if the
|
|
54
|
+
semantics allow it.
|
|
55
|
+
semantics:
|
|
56
|
+
Who the raters are, what the categories mean, and what the data was
|
|
57
|
+
expected to look like.
|
|
58
|
+
|
|
59
|
+
Examples
|
|
60
|
+
--------
|
|
61
|
+
>>> import numpy as np
|
|
62
|
+
>>> from interrater.semantics import RatingsSemantics, UnorderedCategories
|
|
63
|
+
>>> sem = RatingsSemantics(
|
|
64
|
+
... raters=["ann", "bob"],
|
|
65
|
+
... categories=UnorderedCategories(("include", "exclude")),
|
|
66
|
+
... )
|
|
67
|
+
>>> r = Ratings(np.array([[0, 0], [1, 1], [0, 1]], dtype=np.int8), sem)
|
|
68
|
+
>>> r.n_subjects, r.n_raters
|
|
69
|
+
(3, 2)
|
|
70
|
+
>>> r.subject_ids[0]
|
|
71
|
+
'subject0'
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
# ----------------------------------------------------------- Attributes #
|
|
75
|
+
|
|
76
|
+
category_index: np.ndarray
|
|
77
|
+
# (n_subjects, n_raters) int8, C-contiguous, read-only after construction.
|
|
78
|
+
# A cell is a position in semantics.categories.names, or EMPTY (-1).
|
|
79
|
+
# K is owned by the semantics and is never inferred from these values:
|
|
80
|
+
# category_index.max() + 1 is only ever a lower bound on K.
|
|
81
|
+
|
|
82
|
+
subject_ids: np.ndarray
|
|
83
|
+
# (n_subjects,) object, holding real Python strings so long ids survive.
|
|
84
|
+
# Row order is meaningful -- it is what distinguishes a bootstrap draw from
|
|
85
|
+
# the sample it came from.
|
|
86
|
+
|
|
87
|
+
semantics: RatingsSemantics
|
|
88
|
+
# The declaration my contents were checked against.
|
|
89
|
+
|
|
90
|
+
# --------------------------------------------------------- Construction #
|
|
91
|
+
|
|
92
|
+
def __init__(
|
|
93
|
+
self,
|
|
94
|
+
category_index: np.ndarray,
|
|
95
|
+
semantics: RatingsSemantics,
|
|
96
|
+
subject_ids: Sequence[object] | None = None,
|
|
97
|
+
) -> None:
|
|
98
|
+
self.category_index = np.ascontiguousarray(category_index)
|
|
99
|
+
self.semantics = semantics
|
|
100
|
+
|
|
101
|
+
_v.validate_dtype(self.category_index)
|
|
102
|
+
self.category_index = self.category_index.astype(INDEX_DTYPE, copy=False)
|
|
103
|
+
_v.validate_shape(self.category_index)
|
|
104
|
+
|
|
105
|
+
self.subject_ids = self._as_subject_ids(
|
|
106
|
+
subject_ids, self.category_index.shape[0]
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
self._validate()
|
|
110
|
+
self._freeze()
|
|
111
|
+
|
|
112
|
+
@staticmethod
|
|
113
|
+
def _as_subject_ids(
|
|
114
|
+
subject_ids: Sequence[object] | None, n_rows: int
|
|
115
|
+
) -> np.ndarray:
|
|
116
|
+
"""Absent ids become positional ones; given ids are taken as given."""
|
|
117
|
+
if subject_ids is None:
|
|
118
|
+
return np.array(
|
|
119
|
+
[f"subject{i}" for i in range(n_rows)], dtype=LABEL_DTYPE
|
|
120
|
+
)
|
|
121
|
+
return np.asarray(subject_ids, dtype=LABEL_DTYPE)
|
|
122
|
+
|
|
123
|
+
def _validate(self) -> None:
|
|
124
|
+
"""Every check I make, in order."""
|
|
125
|
+
_v.validate_matches_raters(self.category_index, self.semantics)
|
|
126
|
+
_v.validate_matches_n_subjects(self.category_index, self.semantics)
|
|
127
|
+
_v.validate_values_in_categories(self.category_index, self.semantics)
|
|
128
|
+
_v.validate_subject_ids_match_shape(self.category_index, self.subject_ids)
|
|
129
|
+
_v.validate_subject_ids_unique(self.subject_ids, self.semantics)
|
|
130
|
+
_v.validate_no_idle_raters(self.category_index, self.semantics)
|
|
131
|
+
_v.validate_no_unrated_subjects(self.category_index, self.subject_ids)
|
|
132
|
+
|
|
133
|
+
def _freeze(self) -> None:
|
|
134
|
+
"""Read-only, so nothing computed from me can be invalidated later."""
|
|
135
|
+
self.category_index = self.category_index.view()
|
|
136
|
+
self.category_index.flags.writeable = False
|
|
137
|
+
self.subject_ids = self.subject_ids.view()
|
|
138
|
+
self.subject_ids.flags.writeable = False
|
|
139
|
+
|
|
140
|
+
# ---------------------------------------------------------------- Shape #
|
|
141
|
+
|
|
142
|
+
@property
|
|
143
|
+
def n_subjects(self) -> int:
|
|
144
|
+
return int(self.category_index.shape[0])
|
|
145
|
+
|
|
146
|
+
@property
|
|
147
|
+
def n_raters(self) -> int:
|
|
148
|
+
return int(self.category_index.shape[1])
|
|
149
|
+
|
|
150
|
+
@property
|
|
151
|
+
def n_categories(self) -> int:
|
|
152
|
+
"""K, from the semantics. Never inferred from my values."""
|
|
153
|
+
return self.semantics.n_categories
|
|
154
|
+
|
|
155
|
+
# ------------------------------------------------------------- Equality #
|
|
156
|
+
|
|
157
|
+
def __eq__(self, other: object) -> bool:
|
|
158
|
+
"""
|
|
159
|
+
Two of me are equal when we hold the same ratings and mean the same
|
|
160
|
+
thing.
|
|
161
|
+
|
|
162
|
+
Contents only. Where the data came from is not part of the comparison,
|
|
163
|
+
because I do not carry that. Positional on both axes: rater order is
|
|
164
|
+
declared order, and subject order distinguishes a bootstrap draw from
|
|
165
|
+
the sample it came from.
|
|
166
|
+
"""
|
|
167
|
+
if not isinstance(other, Ratings):
|
|
168
|
+
return NotImplemented
|
|
169
|
+
return (
|
|
170
|
+
self.semantics == other.semantics
|
|
171
|
+
and self.category_index.shape == other.category_index.shape
|
|
172
|
+
and np.array_equal(self.category_index, other.category_index)
|
|
173
|
+
and np.array_equal(self.subject_ids, other.subject_ids)
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
__hash__ = None # type: ignore[assignment]
|
|
177
|
+
|
|
178
|
+
def __repr__(self) -> str:
|
|
179
|
+
filled = int((self.category_index != -1).sum())
|
|
180
|
+
total = self.n_subjects * self.n_raters
|
|
181
|
+
return (
|
|
182
|
+
f"Ratings(\n"
|
|
183
|
+
f" n_subjects = {self.n_subjects:,}\n"
|
|
184
|
+
f" raters = {list(self.semantics.raters)}\n"
|
|
185
|
+
f" categories = {list(self.semantics.categories.names)}\n"
|
|
186
|
+
f" filled = {filled:,} / {total:,} ({filled / total:.1%})\n"
|
|
187
|
+
f")"
|
|
188
|
+
)
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
from .categories import (
|
|
2
|
+
Categories,
|
|
3
|
+
DistanceCategories,
|
|
4
|
+
OrderedCategories,
|
|
5
|
+
UnorderedCategories,
|
|
6
|
+
WeightedCategories,
|
|
7
|
+
build_categories,
|
|
8
|
+
)
|
|
9
|
+
from .ratings_semantics import RatingsSemantics
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"Categories",
|
|
13
|
+
"UnorderedCategories",
|
|
14
|
+
"DistanceCategories",
|
|
15
|
+
"OrderedCategories",
|
|
16
|
+
"WeightedCategories",
|
|
17
|
+
"build_categories",
|
|
18
|
+
"RatingsSemantics",
|
|
19
|
+
]
|
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
"""What a rater was allowed to say, and how far apart those answers are."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
|
|
9
|
+
_MIN_CATEGORIES = 2
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Categories(ABC):
|
|
13
|
+
"""
|
|
14
|
+
I am the set of responses a rater was allowed to give.
|
|
15
|
+
|
|
16
|
+
I am the study design, not a summary of the data. I list what was offered,
|
|
17
|
+
never what was chosen. If I name five grades and nobody earned a D, I still
|
|
18
|
+
have five categories, because the raters could have said D and did not --
|
|
19
|
+
and any coefficient whose chance term divides by the number of categories
|
|
20
|
+
would move if I quietly shrank.
|
|
21
|
+
|
|
22
|
+
I own K. Nothing else in the package may infer it: the largest code present
|
|
23
|
+
in a dataset is a lower bound on K and never K itself, and that gap is
|
|
24
|
+
exactly where a bootstrap draw goes wrong when a rare category happens to
|
|
25
|
+
miss a resample.
|
|
26
|
+
|
|
27
|
+
I hold no ratings and do no counting. My position in a list is the integer
|
|
28
|
+
a ratings array stores, so my order is my encoding.
|
|
29
|
+
|
|
30
|
+
Attributes
|
|
31
|
+
----------
|
|
32
|
+
names:
|
|
33
|
+
The allowed responses, in encoding order. Position i is code i.
|
|
34
|
+
labels:
|
|
35
|
+
Optional display names. Cosmetic; never used in arithmetic.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
# ----------------------------------------------------------- Attributes #
|
|
39
|
+
|
|
40
|
+
names: tuple[str, ...]
|
|
41
|
+
labels: dict[str, str]
|
|
42
|
+
|
|
43
|
+
# --------------------------------------------------------- Construction #
|
|
44
|
+
|
|
45
|
+
def __init__(
|
|
46
|
+
self,
|
|
47
|
+
names: object,
|
|
48
|
+
labels: dict[str, str] | None = None,
|
|
49
|
+
) -> None:
|
|
50
|
+
self.names = tuple(names)
|
|
51
|
+
self.labels = dict(labels or {})
|
|
52
|
+
self._validate()
|
|
53
|
+
|
|
54
|
+
def _validate(self) -> None:
|
|
55
|
+
"""Every check I make. Subclasses extend, never replace."""
|
|
56
|
+
name = type(self).__name__
|
|
57
|
+
|
|
58
|
+
if len(self.names) < _MIN_CATEGORIES:
|
|
59
|
+
raise ValueError(
|
|
60
|
+
f"{name} needs at least {_MIN_CATEGORIES} categories, got "
|
|
61
|
+
f"{len(self.names)}. A single category cannot produce "
|
|
62
|
+
f"agreement or disagreement."
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
seen: set[str] = set()
|
|
66
|
+
dupes = [n for n in self.names if n in seen or seen.add(n)]
|
|
67
|
+
if dupes:
|
|
68
|
+
raise ValueError(
|
|
69
|
+
f"{name}.names contains {len(dupes)} duplicate entr(ies) "
|
|
70
|
+
f"(e.g. {dupes[:3]}). Position is the integer encoding, so a "
|
|
71
|
+
f"duplicate would make two codes mean the same thing."
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
unknown = sorted(set(self.labels) - set(self.names))
|
|
75
|
+
if unknown:
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"{name}.labels names {len(unknown)} categor(ies) that are not "
|
|
78
|
+
f"in this set (e.g. {unknown[:3]}). Add them to names, or drop "
|
|
79
|
+
f"them from labels."
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
# ------------------------------------------------------------ Public API #
|
|
83
|
+
|
|
84
|
+
@property
|
|
85
|
+
def n_categories(self) -> int:
|
|
86
|
+
"""K. What the raters could have said, not what they did say."""
|
|
87
|
+
return len(self.names)
|
|
88
|
+
|
|
89
|
+
def index_of(self, category: str) -> int:
|
|
90
|
+
"""The integer a ratings array stores for ``category``."""
|
|
91
|
+
try:
|
|
92
|
+
return self.names.index(category)
|
|
93
|
+
except ValueError:
|
|
94
|
+
raise KeyError(
|
|
95
|
+
f"{type(self).__name__} has no category {category!r}. Known: "
|
|
96
|
+
f"{list(self.names)}."
|
|
97
|
+
) from None
|
|
98
|
+
|
|
99
|
+
def label_for(self, category: str) -> str:
|
|
100
|
+
"""Display name, falling back to the category itself."""
|
|
101
|
+
return self.labels.get(category, category)
|
|
102
|
+
|
|
103
|
+
def __eq__(self, other: object) -> bool:
|
|
104
|
+
if not isinstance(other, Categories):
|
|
105
|
+
return NotImplemented
|
|
106
|
+
return type(self) is type(other) and self.names == other.names
|
|
107
|
+
|
|
108
|
+
__hash__ = None # type: ignore[assignment]
|
|
109
|
+
|
|
110
|
+
def __repr__(self) -> str:
|
|
111
|
+
return (
|
|
112
|
+
f"{type(self).__name__}(\n"
|
|
113
|
+
f" n_categories = {self.n_categories}\n"
|
|
114
|
+
f" names = {list(self.names)}\n"
|
|
115
|
+
f")"
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class UnorderedCategories(Categories):
|
|
120
|
+
"""
|
|
121
|
+
I am categories with no ranking and no notion of distance.
|
|
122
|
+
|
|
123
|
+
Nominal. 'include', 'unsure' and 'exclude' are three different answers, and
|
|
124
|
+
nothing in me says 'unsure' sits between the other two -- even when a human
|
|
125
|
+
reading the words would assume it does. If you want that assumption in the
|
|
126
|
+
arithmetic, say so with OrderedCategories.
|
|
127
|
+
|
|
128
|
+
Every disagreement I describe is equally wrong, which is what unweighted
|
|
129
|
+
kappa already computes. A weighted coefficient handed me would reduce to
|
|
130
|
+
the unweighted one, so analysers that need distance should refuse me rather
|
|
131
|
+
than silently return the same number under a different name.
|
|
132
|
+
"""
|
|
133
|
+
|
|
134
|
+
names: tuple[str, ...]
|
|
135
|
+
labels: dict[str, str]
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
class DistanceCategories(Categories, ABC):
|
|
139
|
+
"""
|
|
140
|
+
I am categories that know how wrong each disagreement is.
|
|
141
|
+
|
|
142
|
+
My children disagree about where that knowledge comes from -- derived from
|
|
143
|
+
rank, or supplied outright -- but both can answer the same question, so a
|
|
144
|
+
weighted coefficient can take either and need not care which.
|
|
145
|
+
|
|
146
|
+
I exist because that question has exactly one form: a K x K matrix of
|
|
147
|
+
agreement weights, 1.0 on the diagonal, falling towards 0.0 as two
|
|
148
|
+
categories grow further apart.
|
|
149
|
+
"""
|
|
150
|
+
|
|
151
|
+
@abstractmethod
|
|
152
|
+
def weight_matrix(self) -> np.ndarray:
|
|
153
|
+
"""
|
|
154
|
+
(K, K) float. Agreement weight between every pair of categories.
|
|
155
|
+
|
|
156
|
+
1.0 means indistinguishable, 0.0 means maximally different. The
|
|
157
|
+
diagonal is 1.0 and the matrix is symmetric: how wrong a disagreement
|
|
158
|
+
is does not depend on which rater said which.
|
|
159
|
+
"""
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class OrderedCategories(DistanceCategories):
|
|
163
|
+
"""
|
|
164
|
+
I am categories on a ranked scale, and I derive distance from position.
|
|
165
|
+
|
|
166
|
+
'mild' < 'moderate' < 'severe'. My order is not merely an encoding, it is a
|
|
167
|
+
claim: adjacent categories are more alike than distant ones, so a reviewer
|
|
168
|
+
who says 'mild' where another said 'moderate' has disagreed less than one
|
|
169
|
+
who said 'severe'.
|
|
170
|
+
|
|
171
|
+
I compute weights from rank distance, either linearly or quadratically.
|
|
172
|
+
Quadratic is the common default in the medical literature and forgives
|
|
173
|
+
near-misses more readily; linear penalises distance evenly. Neither is more
|
|
174
|
+
correct, so I make you say which -- the choice changes the coefficient and
|
|
175
|
+
ought to appear in a methods section.
|
|
176
|
+
|
|
177
|
+
Attributes
|
|
178
|
+
----------
|
|
179
|
+
weighting:
|
|
180
|
+
'linear' or 'quadratic'. How rank distance becomes disagreement.
|
|
181
|
+
"""
|
|
182
|
+
|
|
183
|
+
names: tuple[str, ...]
|
|
184
|
+
labels: dict[str, str]
|
|
185
|
+
weighting: str
|
|
186
|
+
|
|
187
|
+
_ALLOWED = ("linear", "quadratic")
|
|
188
|
+
|
|
189
|
+
def __init__(
|
|
190
|
+
self,
|
|
191
|
+
names: object,
|
|
192
|
+
weighting: str = "quadratic",
|
|
193
|
+
labels: dict[str, str] | None = None,
|
|
194
|
+
) -> None:
|
|
195
|
+
self.weighting = weighting
|
|
196
|
+
super().__init__(names, labels)
|
|
197
|
+
|
|
198
|
+
def _validate(self) -> None:
|
|
199
|
+
super()._validate()
|
|
200
|
+
if self.weighting not in self._ALLOWED:
|
|
201
|
+
raise ValueError(
|
|
202
|
+
f"{type(self).__name__}.weighting must be one of "
|
|
203
|
+
f"{list(self._ALLOWED)}, got {self.weighting!r}. Linear "
|
|
204
|
+
f"penalises rank distance evenly; quadratic forgives "
|
|
205
|
+
f"near-misses. The choice changes the coefficient."
|
|
206
|
+
)
|
|
207
|
+
|
|
208
|
+
def weight_matrix(self) -> np.ndarray:
|
|
209
|
+
k = self.n_categories
|
|
210
|
+
rank = np.arange(k)
|
|
211
|
+
distance = np.abs(rank[:, None] - rank[None, :]).astype(np.float64)
|
|
212
|
+
if self.weighting == "quadratic":
|
|
213
|
+
distance = distance ** 2
|
|
214
|
+
return 1.0 - distance / distance.max()
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
class WeightedCategories(DistanceCategories):
|
|
218
|
+
"""
|
|
219
|
+
I am categories with distances supplied outright, not inferred.
|
|
220
|
+
|
|
221
|
+
I do not require a ranking. 'cat' and 'dog' may be closer to each other
|
|
222
|
+
than either is to 'car' without any of the three being greater or less than
|
|
223
|
+
another, and no ordered scale can express that.
|
|
224
|
+
|
|
225
|
+
Use me when the similarity you mean cannot be recovered from an order:
|
|
226
|
+
grouped diagnoses, a coding frame where two labels are near-synonyms, or a
|
|
227
|
+
scale whose steps are genuinely uneven. Otherwise OrderedCategories is
|
|
228
|
+
fewer numbers to get wrong.
|
|
229
|
+
|
|
230
|
+
Attributes
|
|
231
|
+
----------
|
|
232
|
+
weights:
|
|
233
|
+
(K, K) float. Symmetric, 1.0 on the diagonal, entries in [0, 1].
|
|
234
|
+
"""
|
|
235
|
+
|
|
236
|
+
names: tuple[str, ...]
|
|
237
|
+
labels: dict[str, str]
|
|
238
|
+
weights: np.ndarray
|
|
239
|
+
|
|
240
|
+
def __init__(
|
|
241
|
+
self,
|
|
242
|
+
names: object,
|
|
243
|
+
weights: np.ndarray,
|
|
244
|
+
labels: dict[str, str] | None = None,
|
|
245
|
+
) -> None:
|
|
246
|
+
self.weights = np.asarray(weights, dtype=np.float64)
|
|
247
|
+
super().__init__(names, labels)
|
|
248
|
+
|
|
249
|
+
def _validate(self) -> None:
|
|
250
|
+
super()._validate()
|
|
251
|
+
name = type(self).__name__
|
|
252
|
+
k = self.n_categories
|
|
253
|
+
|
|
254
|
+
if self.weights.shape != (k, k):
|
|
255
|
+
raise ValueError(
|
|
256
|
+
f"{name}.weights must be ({k}, {k}) to match its {k} "
|
|
257
|
+
f"categories, got {self.weights.shape}."
|
|
258
|
+
)
|
|
259
|
+
if not np.allclose(self.weights, self.weights.T):
|
|
260
|
+
raise ValueError(
|
|
261
|
+
f"{name}.weights must be symmetric. How wrong a disagreement "
|
|
262
|
+
f"is cannot depend on which rater said which."
|
|
263
|
+
)
|
|
264
|
+
if not np.allclose(np.diag(self.weights), 1.0):
|
|
265
|
+
raise ValueError(
|
|
266
|
+
f"{name}.weights must have 1.0 on the diagonal: a category "
|
|
267
|
+
f"agrees perfectly with itself. Got "
|
|
268
|
+
f"{np.diag(self.weights).round(3).tolist()}."
|
|
269
|
+
)
|
|
270
|
+
if self.weights.min() < 0.0 or self.weights.max() > 1.0:
|
|
271
|
+
raise ValueError(
|
|
272
|
+
f"{name}.weights must lie in [0, 1], where 1.0 is "
|
|
273
|
+
f"indistinguishable and 0.0 maximally different. Got range "
|
|
274
|
+
f"[{self.weights.min():.3f}, {self.weights.max():.3f}]. If you "
|
|
275
|
+
f"have distances rather than similarities, subtract them from "
|
|
276
|
+
f"1 after scaling to [0, 1]."
|
|
277
|
+
)
|
|
278
|
+
|
|
279
|
+
def weight_matrix(self) -> np.ndarray:
|
|
280
|
+
return self.weights
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
# ---------------------------------------------------------------- Factory #
|
|
284
|
+
|
|
285
|
+
_KINDS = {
|
|
286
|
+
"unordered": UnorderedCategories,
|
|
287
|
+
"ordered": OrderedCategories,
|
|
288
|
+
"weighted": WeightedCategories,
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def build_categories(spec: dict) -> Categories:
|
|
293
|
+
"""
|
|
294
|
+
Build the right Categories from a plain mapping, as parsed from YAML.
|
|
295
|
+
|
|
296
|
+
``kind`` selects the class; the remaining keys are its arguments. Kept here
|
|
297
|
+
rather than with the config that calls it, because this is the module that
|
|
298
|
+
knows what subclasses exist.
|
|
299
|
+
|
|
300
|
+
Examples
|
|
301
|
+
--------
|
|
302
|
+
>>> build_categories({"kind": "unordered", "names": ["yes", "no"]})
|
|
303
|
+
UnorderedCategories(
|
|
304
|
+
n_categories = 2
|
|
305
|
+
names = ['yes', 'no']
|
|
306
|
+
)
|
|
307
|
+
"""
|
|
308
|
+
spec = dict(spec)
|
|
309
|
+
kind = spec.pop("kind", None)
|
|
310
|
+
if kind not in _KINDS:
|
|
311
|
+
raise ValueError(
|
|
312
|
+
f"categories.kind must be one of {sorted(_KINDS)}, got {kind!r}. "
|
|
313
|
+
f"Use 'unordered' for a nominal scheme, 'ordered' for a ranked "
|
|
314
|
+
f"scale, or 'weighted' to supply a similarity matrix directly."
|
|
315
|
+
)
|
|
316
|
+
if "names" not in spec:
|
|
317
|
+
raise KeyError(
|
|
318
|
+
f"categories of kind {kind!r} needs 'names': the responses a rater "
|
|
319
|
+
f"was allowed to give, in encoding order."
|
|
320
|
+
)
|
|
321
|
+
try:
|
|
322
|
+
return _KINDS[kind](**spec)
|
|
323
|
+
except TypeError as error:
|
|
324
|
+
raise TypeError(
|
|
325
|
+
f"categories of kind {kind!r} got unexpected keys "
|
|
326
|
+
f"{sorted(set(spec) - {'names'})}. {error}"
|
|
327
|
+
) from None
|