interrater 0.0.1.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- interrater/__init__.py +1 -0
- interrater/cleaners/__init__.py +19 -0
- interrater/cleaners/_string_utils.py +37 -0
- interrater/cleaners/category_report.py +96 -0
- interrater/cleaners/cleaner_config.py +308 -0
- interrater/cleaners/cleaning_report.py +125 -0
- interrater/cleaners/drop_report.py +159 -0
- interrater/cleaners/frame_layout.py +102 -0
- interrater/cleaners/insufficient_data_policy.py +123 -0
- interrater/cleaners/matching_report.py +146 -0
- interrater/contracts.py +120 -0
- interrater/data_objects/__init__.py +3 -0
- interrater/data_objects/_ratings_validation.py +160 -0
- interrater/data_objects/ratings.py +188 -0
- interrater/semantics/__init__.py +19 -0
- interrater/semantics/categories.py +327 -0
- interrater/semantics/ratings_semantics.py +253 -0
- interrater-0.0.1.dev0.dist-info/METADATA +82 -0
- interrater-0.0.1.dev0.dist-info/RECORD +21 -0
- interrater-0.0.1.dev0.dist-info/WHEEL +4 -0
- interrater-0.0.1.dev0.dist-info/licenses/LICENSE +674 -0
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""What was removed for having too little data, and in what order."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True)
|
|
9
|
+
class DropPass:
|
|
10
|
+
"""One sweep of the drop loop."""
|
|
11
|
+
|
|
12
|
+
number: int
|
|
13
|
+
n_raters: int
|
|
14
|
+
n_subjects: int
|
|
15
|
+
|
|
16
|
+
@property
|
|
17
|
+
def dropped_nothing(self) -> bool:
|
|
18
|
+
return self.n_raters == 0 and self.n_subjects == 0
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class DropReport:
|
|
23
|
+
"""
|
|
24
|
+
I record what was removed for thinness, and how the removing went.
|
|
25
|
+
|
|
26
|
+
Dropping is not symmetric: taking out a rater can push a subject below its
|
|
27
|
+
threshold and taking out a subject can push a rater below its own. So the
|
|
28
|
+
order was a declared choice, and I record which one was made along with
|
|
29
|
+
what it cost.
|
|
30
|
+
|
|
31
|
+
Under ``iterate`` I keep every pass separately, because the sequence is the
|
|
32
|
+
diagnostic. A study going 1,617 to 1,580 to 1,204 to 340 has cascaded, and
|
|
33
|
+
that is a very different event from a single clean cut to 340 -- yet both
|
|
34
|
+
end at the same number and a total alone cannot tell them apart.
|
|
35
|
+
|
|
36
|
+
I also record how the loop ended. Converging means the data settled and the
|
|
37
|
+
result is what the policy asked for. Hitting the pass limit means the
|
|
38
|
+
cascade was still running when it was cut off, and what survived is
|
|
39
|
+
arbitrary rather than principled.
|
|
40
|
+
|
|
41
|
+
Attributes
|
|
42
|
+
----------
|
|
43
|
+
method:
|
|
44
|
+
Which order was used: raters first, subjects first, or iterate.
|
|
45
|
+
passes:
|
|
46
|
+
The sweeps that ran. One entry for the non-iterating methods.
|
|
47
|
+
converged:
|
|
48
|
+
Whether the loop stopped because a pass dropped nothing. False means it
|
|
49
|
+
hit the limit with work still to do, and the result should be treated
|
|
50
|
+
with suspicion.
|
|
51
|
+
dropped_raters:
|
|
52
|
+
Rater name to how many ratings that rater had when removed.
|
|
53
|
+
dropped_subject_ids:
|
|
54
|
+
Every subject removed, for the full rendering. Summarised as a
|
|
55
|
+
distribution otherwise.
|
|
56
|
+
dropped_subject_distribution:
|
|
57
|
+
How many subjects were removed holding how many ratings:
|
|
58
|
+
``{0: 12, 1: 380}``.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
method: str
|
|
62
|
+
converged: bool = True
|
|
63
|
+
passes: tuple[DropPass, ...] = ()
|
|
64
|
+
|
|
65
|
+
dropped_raters: dict[str, int] = field(default_factory=dict)
|
|
66
|
+
dropped_subject_ids: tuple[str, ...] = ()
|
|
67
|
+
dropped_subject_distribution: dict[int, int] = field(default_factory=dict)
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def n_passes(self) -> int:
|
|
71
|
+
return len(self.passes)
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def n_raters_dropped(self) -> int:
|
|
75
|
+
return len(self.dropped_raters)
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def n_subjects_dropped(self) -> int:
|
|
79
|
+
return len(self.dropped_subject_ids)
|
|
80
|
+
|
|
81
|
+
@property
|
|
82
|
+
def dropped_nothing(self) -> bool:
|
|
83
|
+
return not self.dropped_raters and not self.dropped_subject_ids
|
|
84
|
+
|
|
85
|
+
def summary(self) -> dict:
|
|
86
|
+
return {
|
|
87
|
+
"method": self.method,
|
|
88
|
+
"converged": self.converged,
|
|
89
|
+
"n_passes": self.n_passes,
|
|
90
|
+
"passes": [
|
|
91
|
+
{"pass": p.number, "raters": p.n_raters, "subjects": p.n_subjects}
|
|
92
|
+
for p in self.passes
|
|
93
|
+
],
|
|
94
|
+
"n_raters_dropped": self.n_raters_dropped,
|
|
95
|
+
"n_subjects_dropped": self.n_subjects_dropped,
|
|
96
|
+
"dropped_raters": dict(self.dropped_raters),
|
|
97
|
+
"dropped_subject_ids": list(self.dropped_subject_ids),
|
|
98
|
+
"dropped_subject_distribution": {
|
|
99
|
+
int(k): int(v)
|
|
100
|
+
for k, v in self.dropped_subject_distribution.items()
|
|
101
|
+
},
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
def render(self, verbosity: str = "summary") -> str:
|
|
105
|
+
pad = " "
|
|
106
|
+
|
|
107
|
+
if self.dropped_nothing:
|
|
108
|
+
return f"Drops\n{pad}nothing dropped ({self.method})"
|
|
109
|
+
|
|
110
|
+
ending = (
|
|
111
|
+
f"converged after {self.n_passes} pass(es)"
|
|
112
|
+
if self.converged
|
|
113
|
+
else f"STOPPED AT THE PASS LIMIT after {self.n_passes}; "
|
|
114
|
+
f"the cascade had not settled"
|
|
115
|
+
)
|
|
116
|
+
lines = [
|
|
117
|
+
"Drops",
|
|
118
|
+
f"{pad}method : {self.method} ({ending})",
|
|
119
|
+
f"{pad}raters : {self.n_raters_dropped} removed",
|
|
120
|
+
]
|
|
121
|
+
if self.dropped_raters:
|
|
122
|
+
shown = ", ".join(
|
|
123
|
+
f"{name} ({n:,} rating(s))"
|
|
124
|
+
for name, n in list(self.dropped_raters.items())[:3]
|
|
125
|
+
)
|
|
126
|
+
more = self.n_raters_dropped - min(3, self.n_raters_dropped)
|
|
127
|
+
lines.append(
|
|
128
|
+
f"{pad} {shown}"
|
|
129
|
+
+ (f", +{more} more" if more > 0 else "")
|
|
130
|
+
)
|
|
131
|
+
lines.append(f"{pad}subjects : {self.n_subjects_dropped:,} removed")
|
|
132
|
+
|
|
133
|
+
if self.dropped_subject_distribution:
|
|
134
|
+
shown = ", ".join(
|
|
135
|
+
f"{n:,} held {k}"
|
|
136
|
+
for k, n in sorted(self.dropped_subject_distribution.items())
|
|
137
|
+
)
|
|
138
|
+
lines.append(f"{pad} {shown}")
|
|
139
|
+
|
|
140
|
+
if verbosity == "full":
|
|
141
|
+
if len(self.passes) > 1:
|
|
142
|
+
for p in self.passes:
|
|
143
|
+
lines.append(
|
|
144
|
+
f"{pad} pass {p.number}: -{p.n_raters} rater(s), "
|
|
145
|
+
f"-{p.n_subjects:,} subject(s)"
|
|
146
|
+
)
|
|
147
|
+
if self.dropped_subject_ids:
|
|
148
|
+
ids = list(self.dropped_subject_ids)
|
|
149
|
+
shown = ", ".join(map(str, ids[:10]))
|
|
150
|
+
more = len(ids) - min(10, len(ids))
|
|
151
|
+
lines.append(
|
|
152
|
+
f"{pad} ids: {shown}"
|
|
153
|
+
+ (f", +{more:,} more" if more > 0 else "")
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
return "\n".join(lines)
|
|
157
|
+
|
|
158
|
+
def __repr__(self) -> str:
|
|
159
|
+
return self.render()
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""How to read the frame, as opposed to what the study means."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class FrameLayout:
|
|
7
|
+
"""
|
|
8
|
+
I describe the file, not the study.
|
|
9
|
+
|
|
10
|
+
Everything else a cleaner is configured with -- which strings mean nothing,
|
|
11
|
+
what to do with an undeclared response, how thin the data may get -- is a
|
|
12
|
+
decision about the research and travels unchanged to a second export. I am
|
|
13
|
+
the part that does not: which column holds the subject ids, whether the
|
|
14
|
+
header row can be trusted. A different spreadsheet of the same study needs
|
|
15
|
+
a different me.
|
|
16
|
+
|
|
17
|
+
I take a frame, never a path. Reading the file is the caller's job, and by
|
|
18
|
+
the time a frame reaches a cleaner the question of whether the file had a
|
|
19
|
+
header has already been answered -- pandas assigns integer labels when it
|
|
20
|
+
did not, so nothing downstream could tell the difference anyway. What I
|
|
21
|
+
decide is narrower: whether to trust the labels that are there.
|
|
22
|
+
|
|
23
|
+
Attributes
|
|
24
|
+
----------
|
|
25
|
+
use_column_names:
|
|
26
|
+
True means the frame's column labels must equal the declared raters, in
|
|
27
|
+
order, and anything else is an error. A column nobody declared is a
|
|
28
|
+
mistake to look into, not a rater to invent. False ignores the labels
|
|
29
|
+
entirely and takes rater names from the semantics by position.
|
|
30
|
+
subject_id_col:
|
|
31
|
+
Which column holds subject ids, or None to generate them positionally.
|
|
32
|
+
Requires ``use_column_names``: there is nothing to find a named column
|
|
33
|
+
by otherwise.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
# ----------------------------------------------------------- Attributes #
|
|
37
|
+
|
|
38
|
+
use_column_names: bool
|
|
39
|
+
subject_id_col: str | None
|
|
40
|
+
|
|
41
|
+
# --------------------------------------------------------- Construction #
|
|
42
|
+
|
|
43
|
+
def __init__(
|
|
44
|
+
self,
|
|
45
|
+
use_column_names: bool = True,
|
|
46
|
+
subject_id_col: str | None = None,
|
|
47
|
+
) -> None:
|
|
48
|
+
self.use_column_names = use_column_names
|
|
49
|
+
self.subject_id_col = subject_id_col
|
|
50
|
+
self._validate()
|
|
51
|
+
|
|
52
|
+
def _validate(self) -> None:
|
|
53
|
+
name = type(self).__name__
|
|
54
|
+
|
|
55
|
+
if not isinstance(self.use_column_names, bool):
|
|
56
|
+
raise TypeError(
|
|
57
|
+
f"{name}.use_column_names must be a boolean, got "
|
|
58
|
+
f"{type(self.use_column_names).__name__}."
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
if self.subject_id_col is None:
|
|
62
|
+
return
|
|
63
|
+
|
|
64
|
+
if not isinstance(self.subject_id_col, str):
|
|
65
|
+
raise TypeError(
|
|
66
|
+
f"{name}.subject_id_col must be a column name or None, got "
|
|
67
|
+
f"{type(self.subject_id_col).__name__}."
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
if not self.use_column_names:
|
|
71
|
+
raise ValueError(
|
|
72
|
+
f"{name} sets subject_id_col to {self.subject_id_col!r} but "
|
|
73
|
+
f"use_column_names is False, so there are no column names to "
|
|
74
|
+
f"find it by. Either trust the labels, or drop subject_id_col "
|
|
75
|
+
f"and let ids be generated."
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
@classmethod
|
|
79
|
+
def from_dict(cls, spec: dict) -> FrameLayout:
|
|
80
|
+
unknown = sorted(set(spec) - {"use_column_names", "subject_id_col"})
|
|
81
|
+
if unknown:
|
|
82
|
+
raise KeyError(
|
|
83
|
+
f"frame_layout has unknown key(s) {unknown}. Known: "
|
|
84
|
+
f"use_column_names, subject_id_col."
|
|
85
|
+
)
|
|
86
|
+
return cls(**spec)
|
|
87
|
+
|
|
88
|
+
def __eq__(self, other: object) -> bool:
|
|
89
|
+
if not isinstance(other, FrameLayout):
|
|
90
|
+
return NotImplemented
|
|
91
|
+
return (
|
|
92
|
+
self.use_column_names == other.use_column_names
|
|
93
|
+
and self.subject_id_col == other.subject_id_col
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
__hash__ = None # type: ignore[assignment]
|
|
97
|
+
|
|
98
|
+
def __repr__(self) -> str:
|
|
99
|
+
return (
|
|
100
|
+
f"FrameLayout(use_column_names={self.use_column_names}, "
|
|
101
|
+
f"subject_id_col={self.subject_id_col!r})"
|
|
102
|
+
)
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""How thin the data may get before rows and columns are dropped."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ..contracts import MIN_RATINGS_PER_RATER, MIN_RATINGS_PER_SUBJECT
|
|
6
|
+
|
|
7
|
+
MAX_DROP_PASSES: int = 100
|
|
8
|
+
"""Safety rail on ``method: iterate``.
|
|
9
|
+
|
|
10
|
+
Each pass only removes rows and columns, so iteration always converges and
|
|
11
|
+
normally stops after two or three. Reaching this limit means something
|
|
12
|
+
pathological is happening, and a result truncated mid-cascade is arbitrary --
|
|
13
|
+
which is why the report says which way the loop ended rather than only how many
|
|
14
|
+
passes it took.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class InsufficientDataPolicy:
|
|
19
|
+
"""
|
|
20
|
+
I say how much data is too little, and what to do about it.
|
|
21
|
+
|
|
22
|
+
Dropping is not symmetric. Removing a rater can push a subject below its
|
|
23
|
+
threshold, and removing a subject can push a rater below its own, so the
|
|
24
|
+
order changes the answer. I make that order a declared choice rather than
|
|
25
|
+
an accident of implementation.
|
|
26
|
+
|
|
27
|
+
My minimums may be raised but never lowered: the floors in ``contracts``
|
|
28
|
+
are what makes a row or column able to participate at all, and below them a
|
|
29
|
+
subject has nobody to disagree with. Raising a minimum is a study saying it
|
|
30
|
+
wants more evidence than the arithmetic strictly needs.
|
|
31
|
+
|
|
32
|
+
Attributes
|
|
33
|
+
----------
|
|
34
|
+
min_ratings_per_rater:
|
|
35
|
+
How many subjects a rater must have rated to be kept.
|
|
36
|
+
min_ratings_per_subject:
|
|
37
|
+
How many raters must have rated a subject for it to be kept.
|
|
38
|
+
method:
|
|
39
|
+
``raters_first`` drops raters, then subjects, once. ``subjects_first``
|
|
40
|
+
reverses that. ``iterate`` repeats until a pass drops nothing, which
|
|
41
|
+
can cascade a study down to very little -- the report shows each pass
|
|
42
|
+
so that is visible as a sequence rather than a single surprising total.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
# ----------------------------------------------------------- Attributes #
|
|
46
|
+
|
|
47
|
+
min_ratings_per_rater: int
|
|
48
|
+
min_ratings_per_subject: int
|
|
49
|
+
method: str
|
|
50
|
+
|
|
51
|
+
_METHODS = ("raters_first", "subjects_first", "iterate")
|
|
52
|
+
|
|
53
|
+
# --------------------------------------------------------- Construction #
|
|
54
|
+
|
|
55
|
+
def __init__(
|
|
56
|
+
self,
|
|
57
|
+
min_ratings_per_rater: int = MIN_RATINGS_PER_RATER,
|
|
58
|
+
min_ratings_per_subject: int = MIN_RATINGS_PER_SUBJECT,
|
|
59
|
+
method: str = "raters_first",
|
|
60
|
+
) -> None:
|
|
61
|
+
self.min_ratings_per_rater = min_ratings_per_rater
|
|
62
|
+
self.min_ratings_per_subject = min_ratings_per_subject
|
|
63
|
+
self.method = method
|
|
64
|
+
self._validate()
|
|
65
|
+
|
|
66
|
+
def _validate(self) -> None:
|
|
67
|
+
name = type(self).__name__
|
|
68
|
+
|
|
69
|
+
for field, value, floor in (
|
|
70
|
+
("min_ratings_per_rater", self.min_ratings_per_rater,
|
|
71
|
+
MIN_RATINGS_PER_RATER),
|
|
72
|
+
("min_ratings_per_subject", self.min_ratings_per_subject,
|
|
73
|
+
MIN_RATINGS_PER_SUBJECT),
|
|
74
|
+
):
|
|
75
|
+
if not isinstance(value, int) or isinstance(value, bool):
|
|
76
|
+
raise TypeError(
|
|
77
|
+
f"{name}.{field} must be an integer, got "
|
|
78
|
+
f"{type(value).__name__}."
|
|
79
|
+
)
|
|
80
|
+
if value < floor:
|
|
81
|
+
raise ValueError(
|
|
82
|
+
f"{name}.{field} is {value}, below the floor of {floor}. "
|
|
83
|
+
f"That floor is what makes a rater or subject able to take "
|
|
84
|
+
f"part in any comparison at all; it can be raised as a "
|
|
85
|
+
f"study choice, never lowered."
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
if self.method not in self._METHODS:
|
|
89
|
+
raise ValueError(
|
|
90
|
+
f"{name}.method must be one of {list(self._METHODS)}, got "
|
|
91
|
+
f"{self.method!r}. Dropping is not symmetric, so the order is "
|
|
92
|
+
f"a choice you have to make."
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
@classmethod
|
|
96
|
+
def from_dict(cls, spec: dict) -> InsufficientDataPolicy:
|
|
97
|
+
unknown = sorted(set(spec) - {
|
|
98
|
+
"min_ratings_per_rater", "min_ratings_per_subject", "method"
|
|
99
|
+
})
|
|
100
|
+
if unknown:
|
|
101
|
+
raise KeyError(
|
|
102
|
+
f"insufficient_data_drops has unknown key(s) {unknown}. Known: "
|
|
103
|
+
f"min_ratings_per_rater, min_ratings_per_subject, method."
|
|
104
|
+
)
|
|
105
|
+
return cls(**spec)
|
|
106
|
+
|
|
107
|
+
def __eq__(self, other: object) -> bool:
|
|
108
|
+
if not isinstance(other, InsufficientDataPolicy):
|
|
109
|
+
return NotImplemented
|
|
110
|
+
return (
|
|
111
|
+
self.min_ratings_per_rater == other.min_ratings_per_rater
|
|
112
|
+
and self.min_ratings_per_subject == other.min_ratings_per_subject
|
|
113
|
+
and self.method == other.method
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
__hash__ = None # type: ignore[assignment]
|
|
117
|
+
|
|
118
|
+
def __repr__(self) -> str:
|
|
119
|
+
return (
|
|
120
|
+
f"InsufficientDataPolicy(per_rater="
|
|
121
|
+
f"{self.min_ratings_per_rater}, per_subject="
|
|
122
|
+
f"{self.min_ratings_per_subject}, method={self.method!r})"
|
|
123
|
+
)
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""What happened to the strings, before anything became a number."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True)
|
|
9
|
+
class MatchingReport:
|
|
10
|
+
"""
|
|
11
|
+
I count what every cell turned into while it was still a string.
|
|
12
|
+
|
|
13
|
+
I am the first half of a cleaning report, covering everything that happens
|
|
14
|
+
before a value becomes a category position: cells that arrived null, cells
|
|
15
|
+
whose text meant nothing, cells that were repaired into a category, and
|
|
16
|
+
cells that were none of those and had to be dealt with by policy.
|
|
17
|
+
|
|
18
|
+
I count per rater as well as in total, because the totals hide the thing
|
|
19
|
+
most worth seeing. Four hundred missing cells spread evenly across four
|
|
20
|
+
reviewers is a study design; four hundred concentrated in one reviewer is
|
|
21
|
+
somebody who stopped halfway, and only the breakdown tells them apart.
|
|
22
|
+
|
|
23
|
+
For subjects I keep a distribution rather than a row per subject. With
|
|
24
|
+
sixteen hundred abstracts a per-subject listing is unreadable, and the
|
|
25
|
+
shape is the diagnostic anyway: a long tail of subjects each missing one
|
|
26
|
+
rating means something different from a handful missing all of them.
|
|
27
|
+
|
|
28
|
+
Attributes
|
|
29
|
+
----------
|
|
30
|
+
n_cells:
|
|
31
|
+
Rater cells examined, excluding any subject id column.
|
|
32
|
+
n_null:
|
|
33
|
+
Cells that arrived null. Always missing, whatever the config says --
|
|
34
|
+
an empty spreadsheet cell is not a string and could not be listed.
|
|
35
|
+
missing_matched:
|
|
36
|
+
Per entry in ``missing_values``, how many cells it caught. Entries that
|
|
37
|
+
caught nothing are still listed: a declared marker that never fired is
|
|
38
|
+
worth seeing.
|
|
39
|
+
repairs:
|
|
40
|
+
Per entry in ``value_map``, how many cells it fixed. Empty when no
|
|
41
|
+
repairs were configured, and then omitted from the rendering entirely
|
|
42
|
+
-- somebody who declared no repairs is not asking about them.
|
|
43
|
+
unknown_values:
|
|
44
|
+
Strings that were neither missing nor a category, with how often each
|
|
45
|
+
appeared. Non-empty only when ``unknown_policy`` is ``empty``; under
|
|
46
|
+
``error`` the clean stopped and there is no report to read.
|
|
47
|
+
per_rater_null, per_rater_missing, per_rater_unknown:
|
|
48
|
+
The same counts, keyed by rater name.
|
|
49
|
+
subject_missing_distribution:
|
|
50
|
+
How many subjects lost how many ratings: ``{0: 1204, 1: 380, 2: 33}``.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
n_cells: int
|
|
54
|
+
n_null: int
|
|
55
|
+
missing_matched: dict[str, int] = field(default_factory=dict)
|
|
56
|
+
repairs: dict[str, int] = field(default_factory=dict)
|
|
57
|
+
unknown_values: dict[str, int] = field(default_factory=dict)
|
|
58
|
+
|
|
59
|
+
per_rater_null: dict[str, int] = field(default_factory=dict)
|
|
60
|
+
per_rater_missing: dict[str, int] = field(default_factory=dict)
|
|
61
|
+
per_rater_unknown: dict[str, int] = field(default_factory=dict)
|
|
62
|
+
|
|
63
|
+
subject_missing_distribution: dict[int, int] = field(default_factory=dict)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def n_missing(self) -> int:
|
|
67
|
+
"""Cells with no rating: nulls plus every string that meant nothing."""
|
|
68
|
+
return self.n_null + sum(self.missing_matched.values())
|
|
69
|
+
|
|
70
|
+
@property
|
|
71
|
+
def n_repaired(self) -> int:
|
|
72
|
+
return sum(self.repairs.values())
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def n_unknown(self) -> int:
|
|
76
|
+
return sum(self.unknown_values.values())
|
|
77
|
+
|
|
78
|
+
def summary(self) -> dict:
|
|
79
|
+
"""Plain types only, so this can be written beside the config."""
|
|
80
|
+
return {
|
|
81
|
+
"n_cells": self.n_cells,
|
|
82
|
+
"n_null": self.n_null,
|
|
83
|
+
"n_missing": self.n_missing,
|
|
84
|
+
"n_repaired": self.n_repaired,
|
|
85
|
+
"n_unknown": self.n_unknown,
|
|
86
|
+
"missing_matched": dict(self.missing_matched),
|
|
87
|
+
"repairs": dict(self.repairs),
|
|
88
|
+
"unknown_values": dict(self.unknown_values),
|
|
89
|
+
"per_rater_null": dict(self.per_rater_null),
|
|
90
|
+
"per_rater_missing": dict(self.per_rater_missing),
|
|
91
|
+
"per_rater_unknown": dict(self.per_rater_unknown),
|
|
92
|
+
"subject_missing_distribution": {
|
|
93
|
+
int(k): int(v) for k, v in self.subject_missing_distribution.items()
|
|
94
|
+
},
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
def render(self, verbosity: str = "summary") -> str:
|
|
98
|
+
pad = " "
|
|
99
|
+
fill = self.n_missing / self.n_cells if self.n_cells else 0.0
|
|
100
|
+
lines = [
|
|
101
|
+
"Matching",
|
|
102
|
+
f"{pad}cells : {self.n_cells:,}",
|
|
103
|
+
f"{pad}no rating : {self.n_missing:,} ({fill:.1%}), "
|
|
104
|
+
f"of which {self.n_null:,} arrived null",
|
|
105
|
+
]
|
|
106
|
+
|
|
107
|
+
if self.missing_matched:
|
|
108
|
+
shown = ", ".join(
|
|
109
|
+
f"{k!r} {v:,}" for k, v in self.missing_matched.items()
|
|
110
|
+
)
|
|
111
|
+
lines.append(f"{pad}matched : {shown}")
|
|
112
|
+
|
|
113
|
+
# Nothing configured means nothing to report -- not "0 repairs".
|
|
114
|
+
if self.repairs:
|
|
115
|
+
shown = ", ".join(f"{k!r} {v:,}" for k, v in self.repairs.items())
|
|
116
|
+
lines.append(f"{pad}repaired : {self.n_repaired:,} ({shown})")
|
|
117
|
+
|
|
118
|
+
if self.unknown_values:
|
|
119
|
+
top = sorted(
|
|
120
|
+
self.unknown_values.items(), key=lambda kv: -kv[1]
|
|
121
|
+
)[:3]
|
|
122
|
+
shown = ", ".join(f"{k!r} {v:,}" for k, v in top)
|
|
123
|
+
more = len(self.unknown_values) - len(top)
|
|
124
|
+
lines.append(
|
|
125
|
+
f"{pad}unknown : {self.n_unknown:,} treated as empty "
|
|
126
|
+
f"({shown}{f', +{more} more' if more > 0 else ''})"
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
if verbosity == "full":
|
|
130
|
+
for rater, n in self.per_rater_missing.items():
|
|
131
|
+
extra = self.per_rater_unknown.get(rater, 0)
|
|
132
|
+
lines.append(
|
|
133
|
+
f"{pad} {rater}: {n:,} without a rating"
|
|
134
|
+
+ (f", {extra:,} unknown" if extra else "")
|
|
135
|
+
)
|
|
136
|
+
if self.subject_missing_distribution:
|
|
137
|
+
shown = ", ".join(
|
|
138
|
+
f"{n} subject(s) lost {k}"
|
|
139
|
+
for k, n in sorted(self.subject_missing_distribution.items())
|
|
140
|
+
)
|
|
141
|
+
lines.append(f"{pad} by subject: {shown}")
|
|
142
|
+
|
|
143
|
+
return "\n".join(lines)
|
|
144
|
+
|
|
145
|
+
def __repr__(self) -> str:
|
|
146
|
+
return self.render()
|
interrater/contracts.py
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The array contract.
|
|
3
|
+
|
|
4
|
+
Every workhorse function in this package operates on one array shape, described
|
|
5
|
+
here. Nothing else in the package may invent a second convention.
|
|
6
|
+
|
|
7
|
+
category_index : np.ndarray, (n_subjects, n_raters), int8, C-contiguous
|
|
8
|
+
|
|
9
|
+
Each cell is either a position in ``Categories.names`` or ``EMPTY``.
|
|
10
|
+
|
|
11
|
+
Why subjects are rows
|
|
12
|
+
---------------------
|
|
13
|
+
Resampling subjects is then ``category_index[idx]`` -- a contiguous row gather,
|
|
14
|
+
the fastest operation numpy offers. "Resample subjects" and "resample rows"
|
|
15
|
+
become the same instruction, so a bootstrapper never rebuilds an object.
|
|
16
|
+
|
|
17
|
+
Why int8 and not strings
|
|
18
|
+
------------------------
|
|
19
|
+
Integer positions let counts be a single ``bincount`` over a flattened index.
|
|
20
|
+
At int8, a 1,617 x 4 study is 6.5 KB, so thousands of draws stay in cache.
|
|
21
|
+
Names are resolved once, by the cleaner, and never enter the hot path.
|
|
22
|
+
|
|
23
|
+
Why -1 and not NaN
|
|
24
|
+
------------------
|
|
25
|
+
NaN forces float64, forbids integer indexing, and turns every count into a
|
|
26
|
+
masked operation. ``EMPTY`` keeps the array integral, and ``>= 0`` is a cheap
|
|
27
|
+
validity mask.
|
|
28
|
+
|
|
29
|
+
Where K comes from
|
|
30
|
+
------------------
|
|
31
|
+
K is ``Categories.n_categories`` and nothing else. It is never read off the
|
|
32
|
+
data: ``category_index.max() + 1`` is a lower bound on K and never K itself.
|
|
33
|
+
|
|
34
|
+
That gap is not academic. A bootstrap draw can miss a rare category entirely,
|
|
35
|
+
and a function inferring K from that draw would silently compute with a smaller
|
|
36
|
+
K than the study had. Kappa would survive it, since a category with no mass
|
|
37
|
+
contributes a zero term either way -- but Gwet's AC1 and Brennan-Prediger put
|
|
38
|
+
the number of categories in the chance term directly, so their values would
|
|
39
|
+
move draw to draw for no reason but the inference. The resulting interval would
|
|
40
|
+
look like sampling variability and would not be.
|
|
41
|
+
|
|
42
|
+
Only an analyser unpacks a Ratings. It reads K from the semantics and passes
|
|
43
|
+
it down to the workhorse functions and bootstrappers, which take plain arrays.
|
|
44
|
+
Keeping that unpacking in one layer is what stops K from ever being read off a
|
|
45
|
+
draw.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
import numpy as np
|
|
51
|
+
|
|
52
|
+
# ----------------------------------------------------------------- Sentinel #
|
|
53
|
+
|
|
54
|
+
EMPTY: int = -1
|
|
55
|
+
"""No rating exists in this cell.
|
|
56
|
+
|
|
57
|
+
One sentinel, one meaning: this cell takes no part in any calculation, but it
|
|
58
|
+
holds its place so pairwise overlaps and rater permutations stay correct.
|
|
59
|
+
|
|
60
|
+
I carry no reason. A subject never shown to a rater and a subject shown and
|
|
61
|
+
left blank are both simply empty here, because at this level the only fact that
|
|
62
|
+
matters is that there is no rating. Why a cell is empty is the cleaner's
|
|
63
|
+
business, is recorded in its report, and changes what you can say about a rater
|
|
64
|
+
-- never what any coefficient computes.
|
|
65
|
+
|
|
66
|
+
'Unsure', 'abstain', and anything else a study designer chose to offer are
|
|
67
|
+
*categories*, not empties. They were answers. They live in ``Categories.names``
|
|
68
|
+
and count like any other response.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
# ------------------------------------------------------------------- Dtypes #
|
|
72
|
+
|
|
73
|
+
INDEX_DTYPE: np.dtype = np.dtype(np.int8)
|
|
74
|
+
"""Storage type for the category index array. Caps a study at 127 categories."""
|
|
75
|
+
|
|
76
|
+
LABEL_DTYPE: np.dtype = np.dtype(object)
|
|
77
|
+
"""Storage type for subject ids.
|
|
78
|
+
|
|
79
|
+
``object`` holds real Python strings, so an id of any length survives intact.
|
|
80
|
+
A fixed-width dtype such as ``U32`` would be faster, but it truncates anything
|
|
81
|
+
longer without complaint, and a silently shortened identifier is a data
|
|
82
|
+
corruption nobody would notice.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
# ----------------------------------------------------------------- Minimums #
|
|
86
|
+
|
|
87
|
+
MIN_RATERS: int = 2
|
|
88
|
+
"""Fewer than two raters is not an agreement problem."""
|
|
89
|
+
|
|
90
|
+
MIN_SUBJECTS: int = 2
|
|
91
|
+
"""Fewer than two subjects leaves nothing to disagree about."""
|
|
92
|
+
|
|
93
|
+
MIN_RATINGS_PER_RATER: int = 2
|
|
94
|
+
"""Floor on how many subjects a rater must have rated.
|
|
95
|
+
|
|
96
|
+
A rater with one rating joins one comparison and describes nobody. A cleaner
|
|
97
|
+
may set its threshold higher as a study choice, never lower.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
MIN_RATINGS_PER_SUBJECT: int = 2
|
|
101
|
+
"""Floor on how many raters must have rated a subject.
|
|
102
|
+
|
|
103
|
+
A subject rated once has nobody to disagree with: it joins no pairwise
|
|
104
|
+
comparison and contributes nothing to any group coefficient. A cleaner may set
|
|
105
|
+
its threshold higher as a study choice, never lower.
|
|
106
|
+
"""
|
|
107
|
+
|
|
108
|
+
MIN_OVERLAP_HARD: int = 2
|
|
109
|
+
"""Mathematical floor on co-rated subjects for a pairwise coefficient.
|
|
110
|
+
|
|
111
|
+
Below two, expected agreement cannot be estimated and kappa is undefined. This
|
|
112
|
+
is a validity threshold, not a reliability one: a pair meeting it can still
|
|
113
|
+
produce a number nobody should trust.
|
|
114
|
+
|
|
115
|
+
The *practical* floor is higher and is not a constant. Under heavy skew,
|
|
116
|
+
expected agreement is large and kappa has little room left to work in, so more
|
|
117
|
+
co-rated subjects are needed for the same precision. That threshold depends on
|
|
118
|
+
prevalence and is not yet in this package; see the roadmap entry on overlap
|
|
119
|
+
adequacy.
|
|
120
|
+
"""
|