multiqc-pivot 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- multiqc_pivot/__init__.py +0 -0
- multiqc_pivot/config.py +107 -0
- multiqc_pivot/pivot.py +245 -0
- multiqc_pivot/plugin.py +89 -0
- multiqc_pivot/py.typed +0 -0
- multiqc_pivot-0.1.0.dist-info/METADATA +142 -0
- multiqc_pivot-0.1.0.dist-info/RECORD +10 -0
- multiqc_pivot-0.1.0.dist-info/WHEEL +4 -0
- multiqc_pivot-0.1.0.dist-info/entry_points.txt +3 -0
- multiqc_pivot-0.1.0.dist-info/licenses/LICENSE +21 -0
|
File without changes
|
multiqc_pivot/config.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Settings for the `sample_pivot` block of a MultiQC configuration."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import ClassVar
|
|
7
|
+
|
|
8
|
+
from pydantic import BaseModel
|
|
9
|
+
from pydantic import ConfigDict
|
|
10
|
+
from pydantic import Field
|
|
11
|
+
from pydantic import field_validator
|
|
12
|
+
from pydantic import model_validator
|
|
13
|
+
|
|
14
|
+
GROUP_CAPTURE = "group"
|
|
15
|
+
"""The named capture the group pattern must define; its value names the folded row."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _compile(pattern: str, where: str) -> re.Pattern[str]:
|
|
19
|
+
try:
|
|
20
|
+
return re.compile(pattern)
|
|
21
|
+
except re.error as exc:
|
|
22
|
+
raise ValueError(f"{where} is not a valid regular expression: {pattern!r} ({exc})") from exc
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class Level(BaseModel):
|
|
26
|
+
"""
|
|
27
|
+
One kind of sample name and what happens to the rows that carry it.
|
|
28
|
+
|
|
29
|
+
A level with neither `label` nor `table` folds its columns onto the group row as they are. A
|
|
30
|
+
level with a `label` renames its columns after the label, folds them onto the group row and
|
|
31
|
+
keeps the original row underneath. A level with a `table` moves its rows out of General
|
|
32
|
+
Statistics into a table of that name, keeping whatever grouping they already had.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
model_config: ClassVar[ConfigDict] = ConfigDict(extra="forbid")
|
|
36
|
+
|
|
37
|
+
match: str
|
|
38
|
+
label: str | None = None
|
|
39
|
+
table: str | None = None
|
|
40
|
+
|
|
41
|
+
@field_validator("match")
|
|
42
|
+
@classmethod
|
|
43
|
+
def _match_compiles(cls, value: str) -> str:
|
|
44
|
+
return _compile(value, "match").pattern
|
|
45
|
+
|
|
46
|
+
@model_validator(mode="after")
|
|
47
|
+
def _label_or_table(self) -> Level:
|
|
48
|
+
if self.label is not None and self.table is not None:
|
|
49
|
+
raise ValueError("a level may set label or table, not both")
|
|
50
|
+
if self.label is not None:
|
|
51
|
+
captures = dict.fromkeys(self.pattern.groupindex, "")
|
|
52
|
+
try:
|
|
53
|
+
_ = self.label.format(**captures)
|
|
54
|
+
except (KeyError, IndexError) as exc:
|
|
55
|
+
message = f"label {self.label!r} uses a capture that match {self.match!r} lacks"
|
|
56
|
+
raise ValueError(f"{message} ({exc})") from exc
|
|
57
|
+
return self
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def pattern(self) -> re.Pattern[str]:
|
|
61
|
+
"""The compiled `match` expression."""
|
|
62
|
+
return re.compile(self.match)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class TableSettings(BaseModel):
|
|
66
|
+
"""How a table that receives rows moved out of General Statistics presents itself."""
|
|
67
|
+
|
|
68
|
+
model_config: ClassVar[ConfigDict] = ConfigDict(extra="forbid")
|
|
69
|
+
|
|
70
|
+
description: str = ""
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class SamplePivotConfig(BaseModel):
|
|
74
|
+
"""The `sample_pivot` block of a MultiQC configuration."""
|
|
75
|
+
|
|
76
|
+
model_config: ClassVar[ConfigDict] = ConfigDict(extra="forbid")
|
|
77
|
+
|
|
78
|
+
group: str
|
|
79
|
+
levels: list[Level] = Field(min_length=1)
|
|
80
|
+
column_title: str = "{Label} {title}"
|
|
81
|
+
label_order: list[str] = Field(default_factory=list)
|
|
82
|
+
tables: dict[str, TableSettings] = Field(default_factory=dict)
|
|
83
|
+
|
|
84
|
+
@field_validator("group")
|
|
85
|
+
@classmethod
|
|
86
|
+
def _group_has_capture(cls, value: str) -> str:
|
|
87
|
+
if GROUP_CAPTURE not in _compile(value, "group").groupindex:
|
|
88
|
+
raise ValueError(
|
|
89
|
+
f"group must contain a named capture (?P<{GROUP_CAPTURE}>...): {value!r}"
|
|
90
|
+
)
|
|
91
|
+
return value
|
|
92
|
+
|
|
93
|
+
@field_validator("column_title")
|
|
94
|
+
@classmethod
|
|
95
|
+
def _column_title_placeholders(cls, value: str) -> str:
|
|
96
|
+
try:
|
|
97
|
+
_ = value.format(label="", Label="", title="")
|
|
98
|
+
except (KeyError, IndexError) as exc:
|
|
99
|
+
raise ValueError(
|
|
100
|
+
f"column_title may only use {{label}}, {{Label}} and {{title}}: {value!r} ({exc})"
|
|
101
|
+
) from exc
|
|
102
|
+
return value
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def group_pattern(self) -> re.Pattern[str]:
|
|
106
|
+
"""The compiled `group` expression."""
|
|
107
|
+
return re.compile(self.group)
|
multiqc_pivot/pivot.py
ADDED
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Fold related General Statistics rows into one row per group."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from dataclasses import field
|
|
9
|
+
|
|
10
|
+
from multiqc.plots.table_object import ColumnDict
|
|
11
|
+
from multiqc.plots.table_object import ColumnKeyT
|
|
12
|
+
from multiqc.plots.table_object import ExtValueT
|
|
13
|
+
from multiqc.plots.table_object import InputRow
|
|
14
|
+
from multiqc.types import ColumnKey
|
|
15
|
+
from multiqc.types import SampleGroup
|
|
16
|
+
from multiqc.types import SampleName
|
|
17
|
+
from multiqc.types import SectionKey
|
|
18
|
+
|
|
19
|
+
from multiqc_pivot.config import SamplePivotConfig
|
|
20
|
+
|
|
21
|
+
log = logging.getLogger("multiqc")
|
|
22
|
+
|
|
23
|
+
RowData = dict[ColumnKeyT, ExtValueT | None]
|
|
24
|
+
SectionRows = dict[SampleGroup, list[InputRow]]
|
|
25
|
+
SectionHeaders = dict[ColumnKey, ColumnDict]
|
|
26
|
+
Rows = dict[SectionKey, SectionRows]
|
|
27
|
+
Headers = dict[SectionKey, SectionHeaders]
|
|
28
|
+
|
|
29
|
+
PLACEMENT_BLOCK = 10_000
|
|
30
|
+
"""Pivoted columns for the n-th label are placed at n times this value plus a running index.
|
|
31
|
+
|
|
32
|
+
MultiQC orders General Statistics columns by their placement (1000 unless a module says otherwise),
|
|
33
|
+
so every label forms one contiguous block after all of the columns that were not pivoted.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class Table:
|
|
39
|
+
"""Rows moved out of General Statistics, keyed the same way General Statistics is."""
|
|
40
|
+
|
|
41
|
+
rows: Rows = field(default_factory=dict)
|
|
42
|
+
headers: Headers = field(default_factory=dict)
|
|
43
|
+
|
|
44
|
+
def add(
|
|
45
|
+
self, section: SectionKey, group: SampleGroup, row: InputRow, headers: SectionHeaders
|
|
46
|
+
) -> None:
|
|
47
|
+
"""Move a row in, keeping its section and group."""
|
|
48
|
+
self.rows.setdefault(section, {}).setdefault(group, []).append(row)
|
|
49
|
+
if section not in self.headers:
|
|
50
|
+
self.headers[section] = dict(headers)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass
|
|
54
|
+
class PivotResult:
|
|
55
|
+
"""What General Statistics becomes, plus any tables split off from it."""
|
|
56
|
+
|
|
57
|
+
rows: Rows
|
|
58
|
+
headers: Headers
|
|
59
|
+
tables: dict[str, Table]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass(frozen=True)
|
|
63
|
+
class Moved:
|
|
64
|
+
"""The row belongs in a secondary table."""
|
|
65
|
+
|
|
66
|
+
table: str
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass(frozen=True)
|
|
70
|
+
class Folded:
|
|
71
|
+
"""The row belongs on a group's row, under a label unless it is the group's own row."""
|
|
72
|
+
|
|
73
|
+
group: str
|
|
74
|
+
label: str | None = None
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
Route = Moved | Folded
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def classify(name: str, settings: SamplePivotConfig) -> Route | None:
|
|
81
|
+
"""
|
|
82
|
+
Decide where a sample's row goes from the first level that matches its name.
|
|
83
|
+
|
|
84
|
+
Returns None when no level matches, or when a level matches but the group pattern does not; the
|
|
85
|
+
latter is logged, since it usually means the group pattern is too narrow.
|
|
86
|
+
"""
|
|
87
|
+
for level in settings.levels:
|
|
88
|
+
match = level.pattern.search(name)
|
|
89
|
+
if match is None:
|
|
90
|
+
continue
|
|
91
|
+
if level.table is not None:
|
|
92
|
+
return Moved(level.table)
|
|
93
|
+
group = settings.group_pattern.search(name)
|
|
94
|
+
if group is None:
|
|
95
|
+
log.warning(
|
|
96
|
+
f"sample_pivot: {name!r} matches {level.match!r} but not the group; left as is"
|
|
97
|
+
)
|
|
98
|
+
return None
|
|
99
|
+
label = level.label.format(**match.groupdict()) if level.label is not None else None
|
|
100
|
+
return Folded(group.group("group"), label)
|
|
101
|
+
return None
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def slug(text: str) -> str:
|
|
105
|
+
"""Lower-case a label to letters, digits and single underscores."""
|
|
106
|
+
return re.sub(r"[^a-z0-9]+", "_", text.lower()).strip("_")
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def capitalise(text: str) -> str:
|
|
110
|
+
"""Upper-case the first character only."""
|
|
111
|
+
return text[:1].upper() + text[1:]
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
class Placement:
|
|
115
|
+
"""
|
|
116
|
+
Hands out column placements that cluster pivoted columns by label.
|
|
117
|
+
|
|
118
|
+
Labels follow the configured order; labels missing from it come afterwards in order of first
|
|
119
|
+
appearance.
|
|
120
|
+
"""
|
|
121
|
+
|
|
122
|
+
def __init__(self, order: list[str]) -> None:
|
|
123
|
+
"""Start from the configured label order."""
|
|
124
|
+
self._order: list[str] = list(order)
|
|
125
|
+
self._count: int = 0
|
|
126
|
+
|
|
127
|
+
def next(self, label: str) -> float:
|
|
128
|
+
"""The placement for the next pivoted column of a label."""
|
|
129
|
+
if label not in self._order:
|
|
130
|
+
self._order.append(label)
|
|
131
|
+
self._count += 1
|
|
132
|
+
return float(PLACEMENT_BLOCK * (self._order.index(label) + 1) + self._count)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def pivoted_header(
|
|
136
|
+
header: ColumnDict,
|
|
137
|
+
key: ColumnKey,
|
|
138
|
+
label: str,
|
|
139
|
+
settings: SamplePivotConfig,
|
|
140
|
+
placement: Placement,
|
|
141
|
+
) -> ColumnDict:
|
|
142
|
+
"""Copy a column header under a labelled title and a placement inside the label's block."""
|
|
143
|
+
copy = header.copy()
|
|
144
|
+
copy["title"] = settings.column_title.format(
|
|
145
|
+
label=label, Label=capitalise(label), title=header.get("title", key)
|
|
146
|
+
)
|
|
147
|
+
copy["placement"] = placement.next(label)
|
|
148
|
+
return copy
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@dataclass
|
|
152
|
+
class SectionPivot:
|
|
153
|
+
"""Rebuilds one General Statistics section a row at a time."""
|
|
154
|
+
|
|
155
|
+
headers: SectionHeaders
|
|
156
|
+
settings: SamplePivotConfig
|
|
157
|
+
placement: Placement
|
|
158
|
+
new_headers: SectionHeaders = field(init=False)
|
|
159
|
+
_kept: SectionRows = field(default_factory=dict)
|
|
160
|
+
_group_rows: dict[str, RowData] = field(default_factory=dict)
|
|
161
|
+
_sub_rows: dict[str, list[InputRow]] = field(default_factory=dict)
|
|
162
|
+
|
|
163
|
+
def __post_init__(self) -> None:
|
|
164
|
+
"""Start from the module's own headers; pivoted columns are added alongside them."""
|
|
165
|
+
self.new_headers = dict(self.headers)
|
|
166
|
+
|
|
167
|
+
def keep(self, group: SampleGroup, row: InputRow) -> None:
|
|
168
|
+
"""Leave a row where it was."""
|
|
169
|
+
self._kept.setdefault(group, []).append(row)
|
|
170
|
+
|
|
171
|
+
def fold(self, group: str, row: InputRow) -> None:
|
|
172
|
+
"""Put a row's declared columns onto the group row as they are; the row itself goes away."""
|
|
173
|
+
target = self._group_rows.setdefault(group, {})
|
|
174
|
+
for key, value in row.data.items():
|
|
175
|
+
if key in self.headers:
|
|
176
|
+
_fold(target, key, value, group)
|
|
177
|
+
|
|
178
|
+
def fold_labelled(self, group: str, label: str, row: InputRow) -> None:
|
|
179
|
+
"""
|
|
180
|
+
Rename a row's declared columns after the label and put them onto the group row.
|
|
181
|
+
|
|
182
|
+
The row itself stays beneath the group row, carrying the same renamed columns.
|
|
183
|
+
"""
|
|
184
|
+
target = self._group_rows.setdefault(group, {})
|
|
185
|
+
renamed: RowData = {}
|
|
186
|
+
for key, value in row.data.items():
|
|
187
|
+
if key not in self.headers:
|
|
188
|
+
continue
|
|
189
|
+
new_key = ColumnKey(f"{key}__{slug(label)}")
|
|
190
|
+
if new_key not in self.new_headers:
|
|
191
|
+
self.new_headers[new_key] = pivoted_header(
|
|
192
|
+
self.headers[key], key, label, self.settings, self.placement
|
|
193
|
+
)
|
|
194
|
+
renamed[new_key] = value
|
|
195
|
+
_fold(target, new_key, value, group)
|
|
196
|
+
self._sub_rows.setdefault(group, []).append(InputRow(sample=row.sample, data=renamed))
|
|
197
|
+
|
|
198
|
+
def finish(self) -> SectionRows:
|
|
199
|
+
"""The rebuilt rows, each group row first with its folded rows beneath."""
|
|
200
|
+
for name in dict.fromkeys([*self._group_rows, *self._sub_rows]):
|
|
201
|
+
group = SampleGroup(name)
|
|
202
|
+
first = InputRow(sample=SampleName(name), data=self._group_rows.get(name, {}))
|
|
203
|
+
self._kept[group] = [first, *self._sub_rows.get(name, []), *self._kept.get(group, [])]
|
|
204
|
+
return self._kept
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def pivot(rows: Rows, headers: Headers, settings: SamplePivotConfig) -> PivotResult:
|
|
208
|
+
"""
|
|
209
|
+
Rebuild General Statistics with one row per group.
|
|
210
|
+
|
|
211
|
+
Rows that match a labelled level have their declared columns renamed after the label and copied
|
|
212
|
+
onto the group's row; the original rows stay beneath it so the group can still be expanded.
|
|
213
|
+
Rows that match an unlabelled level are folded onto the group's row as they are. Rows that match
|
|
214
|
+
a level with a table are moved into that table with their grouping intact. Rows that match no
|
|
215
|
+
level, and columns a module did not declare a header for, are left alone.
|
|
216
|
+
"""
|
|
217
|
+
placement = Placement(settings.label_order)
|
|
218
|
+
out_rows: Rows = {}
|
|
219
|
+
out_headers: Headers = {}
|
|
220
|
+
tables: dict[str, Table] = {}
|
|
221
|
+
for section, rows_by_group in rows.items():
|
|
222
|
+
section_headers = headers.get(section, {})
|
|
223
|
+
pivoted = SectionPivot(section_headers, settings, placement)
|
|
224
|
+
for group, members in rows_by_group.items():
|
|
225
|
+
for row in members:
|
|
226
|
+
route = classify(str(row.sample), settings)
|
|
227
|
+
if route is None:
|
|
228
|
+
pivoted.keep(group, row)
|
|
229
|
+
elif isinstance(route, Moved):
|
|
230
|
+
table = tables.setdefault(route.table, Table())
|
|
231
|
+
table.add(section, group, row, section_headers)
|
|
232
|
+
elif route.label is None:
|
|
233
|
+
pivoted.fold(route.group, row)
|
|
234
|
+
else:
|
|
235
|
+
pivoted.fold_labelled(route.group, route.label, row)
|
|
236
|
+
out_rows[section] = pivoted.finish()
|
|
237
|
+
out_headers[section] = pivoted.new_headers
|
|
238
|
+
return PivotResult(out_rows, out_headers, tables)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _fold(target: RowData, key: ColumnKeyT, value: ExtValueT | None, group: str) -> None:
|
|
242
|
+
if key in target and target[key] != value:
|
|
243
|
+
log.warning(f"sample_pivot: group '{group}' has two values for '{key}'; keeping the first")
|
|
244
|
+
return
|
|
245
|
+
target[key] = value
|
multiqc_pivot/plugin.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""The functions MultiQC calls through the `multiqc.hooks.v1` entry points."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import importlib.metadata
|
|
6
|
+
import logging
|
|
7
|
+
from typing import TYPE_CHECKING
|
|
8
|
+
from typing import cast
|
|
9
|
+
|
|
10
|
+
from multiqc_pivot.config import SamplePivotConfig
|
|
11
|
+
from multiqc_pivot.config import TableSettings
|
|
12
|
+
|
|
13
|
+
if TYPE_CHECKING:
|
|
14
|
+
from multiqc.base_module import BaseMultiqcModule
|
|
15
|
+
|
|
16
|
+
from multiqc_pivot.pivot import Table
|
|
17
|
+
|
|
18
|
+
log = logging.getLogger("multiqc")
|
|
19
|
+
|
|
20
|
+
PACKAGE = "multiqc-pivot"
|
|
21
|
+
CONFIG_KEY = "sample_pivot"
|
|
22
|
+
"""The top-level key of a MultiQC configuration that holds this plugin's settings."""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def execution_start() -> None:
|
|
26
|
+
"""Record the plugin version among the report's software versions."""
|
|
27
|
+
from multiqc import config
|
|
28
|
+
|
|
29
|
+
version = importlib.metadata.version(PACKAGE)
|
|
30
|
+
config.software_versions.setdefault(PACKAGE, {})[PACKAGE] = [version]
|
|
31
|
+
log.info(f"Loaded {PACKAGE} {version}")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def after_modules() -> None:
|
|
35
|
+
"""Regroup General Statistics as the `sample_pivot` block asks, if the configuration has one."""
|
|
36
|
+
from multiqc import config
|
|
37
|
+
from multiqc import report
|
|
38
|
+
|
|
39
|
+
from multiqc_pivot.pivot import pivot
|
|
40
|
+
|
|
41
|
+
raw = getattr(config, CONFIG_KEY, None)
|
|
42
|
+
if not raw:
|
|
43
|
+
return
|
|
44
|
+
settings = SamplePivotConfig.model_validate(raw)
|
|
45
|
+
result = pivot(report.general_stats_data, report.general_stats_headers, settings)
|
|
46
|
+
report.general_stats_data.clear()
|
|
47
|
+
report.general_stats_data.update(result.rows)
|
|
48
|
+
report.general_stats_headers.clear()
|
|
49
|
+
report.general_stats_headers.update(result.headers)
|
|
50
|
+
for name in reversed(list(result.tables)):
|
|
51
|
+
module = table_module(name, settings.tables.get(name, TableSettings()), result.tables[name])
|
|
52
|
+
if module is not None:
|
|
53
|
+
report.modules.insert(0, module)
|
|
54
|
+
groups = {group for section in result.rows.values() for group in section}
|
|
55
|
+
tables = f" and {len(result.tables)} extra table(s)" if result.tables else ""
|
|
56
|
+
log.info(f"sample_pivot: General Statistics now has {len(groups)} rows{tables}")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def table_module(name: str, settings: TableSettings, table: Table) -> BaseMultiqcModule | None:
|
|
60
|
+
"""
|
|
61
|
+
Wrap rows moved out of General Statistics in a module holding a single table section.
|
|
62
|
+
|
|
63
|
+
The module carries the name and description so the report shows them once; the section itself
|
|
64
|
+
is untitled. Returns None when the table would be empty.
|
|
65
|
+
"""
|
|
66
|
+
from multiqc.base_module import BaseMultiqcModule
|
|
67
|
+
from multiqc.plots import table as table_plot
|
|
68
|
+
from multiqc.plots.table_object import SectionT
|
|
69
|
+
from multiqc.types import Anchor
|
|
70
|
+
from multiqc.types import SectionKey
|
|
71
|
+
|
|
72
|
+
from multiqc_pivot.pivot import slug
|
|
73
|
+
|
|
74
|
+
anchor = slug(name)
|
|
75
|
+
plot = table_plot.plot_with_sections(
|
|
76
|
+
data=cast(dict[SectionKey, SectionT], table.rows),
|
|
77
|
+
headers=table.headers,
|
|
78
|
+
pconfig={
|
|
79
|
+
"id": f"{anchor}_table",
|
|
80
|
+
"title": name,
|
|
81
|
+
"save_file": True,
|
|
82
|
+
"raw_data_fn": f"multiqc_{anchor}",
|
|
83
|
+
},
|
|
84
|
+
)
|
|
85
|
+
if plot is None:
|
|
86
|
+
return None
|
|
87
|
+
module = BaseMultiqcModule(name=name, anchor=Anchor(anchor), info=settings.description)
|
|
88
|
+
module.add_section(anchor=Anchor(f"{anchor}_section"), plot=plot)
|
|
89
|
+
return module
|
multiqc_pivot/py.typed
ADDED
|
File without changes
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: multiqc-pivot
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A MultiQC plugin that folds related samples into one General Statistics row per group with labelled metric columns.
|
|
5
|
+
Project-URL: homepage, https://github.com/clintval/multiqc-pivot
|
|
6
|
+
Project-URL: repository, https://github.com/clintval/multiqc-pivot
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/clintval/multiqc-pivot/issues
|
|
8
|
+
Author-email: Clint Valentine <valentine.clint@gmail.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Natural Language :: English
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
24
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
25
|
+
Classifier: Typing :: Typed
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Requires-Dist: multiqc<2,>=1.35
|
|
28
|
+
Requires-Dist: pydantic>=2
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# multiqc-pivot
|
|
32
|
+
|
|
33
|
+
[](https://github.com/clintval/multiqc-pivot/actions/workflows/tests.yml?query=branch%3Amain)
|
|
34
|
+
[](https://github.com/clintval/multiqc-pivot)
|
|
35
|
+
[](https://docs.basedpyright.com/latest/)
|
|
36
|
+
[](https://mypy-lang.org/)
|
|
37
|
+
[](https://docs.astral.sh/ruff/)
|
|
38
|
+
|
|
39
|
+
A [MultiQC](https://multiqc.info) plugin that folds related samples into one General Statistics row per group.
|
|
40
|
+
|
|
41
|
+
## Installation
|
|
42
|
+
|
|
43
|
+
```console
|
|
44
|
+
pip install multiqc-pivot
|
|
45
|
+
```
|
|
46
|
+
## Introduction
|
|
47
|
+
|
|
48
|
+
MultiQC gives every sample its own row.
|
|
49
|
+
When one subject yields several samples that are measured by different methods, say a tumour and a normal, or two tissues and a paired-genotype check, the General Statistics table ends up with a block of half-empty rows per subject.
|
|
50
|
+
MultiQC's own [sample grouping](https://docs.seqera.io/multiqc/reports/customisation#sample-grouping) only fills the group's row for the handful of modules that know how to merge their metrics.
|
|
51
|
+
|
|
52
|
+
This plugin runs after every module has reported and rebuilds the table:
|
|
53
|
+
|
|
54
|
+
1. Rows for one group fold into a single row, and every folded column is prefixed by which method it came from.
|
|
55
|
+
2. The original rows stay beneath the group row, so they still can be viewed.
|
|
56
|
+
3. Rows for a level that does not belong in the table, such as per-library read QC, move out into their own table under General Statistics with whatever grouping they already had.
|
|
57
|
+
4. Hover text, color scales, formats and hidden-by-default state carry over from the module that produced each column.
|
|
58
|
+
|
|
59
|
+
## Example
|
|
60
|
+
|
|
61
|
+
The report below comes from the [test fixtures](tests/data/report) and the [configuration](tests/data/multiqc_config.yml) shown in the usage section:
|
|
62
|
+
|
|
63
|
+

|
|
64
|
+
|
|
65
|
+
```yaml
|
|
66
|
+
sample_pivot:
|
|
67
|
+
group: '^(?P<group>[^. ]+)\.'
|
|
68
|
+
levels:
|
|
69
|
+
- match: '\.subject$'
|
|
70
|
+
- match: '\.(?P<analyte>tissueA|tissueB)$'
|
|
71
|
+
label: '{analyte}'
|
|
72
|
+
- match: '\.(?P<analyte>tissueA|tissueB) \(filtered\)$'
|
|
73
|
+
label: '{analyte} (filtered)'
|
|
74
|
+
- match: '\.library\.'
|
|
75
|
+
table: Library statistics
|
|
76
|
+
label_order: [tissueA, tissueB, tissueB (filtered)]
|
|
77
|
+
tables:
|
|
78
|
+
Library statistics:
|
|
79
|
+
description: Per-library read QC.
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Usage
|
|
83
|
+
|
|
84
|
+
Add a `sample_pivot` block to any MultiQC config, for example with `--config my_config.yml`.
|
|
85
|
+
|
|
86
|
+
So, with these sample names:
|
|
87
|
+
|
|
88
|
+
```text
|
|
89
|
+
101.subject
|
|
90
|
+
101.tissueA
|
|
91
|
+
101.tissueB
|
|
92
|
+
101.tissueB (filtered data, though)
|
|
93
|
+
101.tissueA.library.L1
|
|
94
|
+
101.tissueA.library.L2
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
This configuration produces one row named `101` carrying `Concordance`, `TissueA Median`, `TissueB Median`, `TissueB (filtered) % Aligned` and so on, and moves the library rows into a separate table:
|
|
98
|
+
|
|
99
|
+
```yaml
|
|
100
|
+
sample_pivot:
|
|
101
|
+
group: '^(?P<group>[^. ]+)\.'
|
|
102
|
+
levels:
|
|
103
|
+
- match: '\.subject$'
|
|
104
|
+
- match: '\.(?P<analyte>tissueA|tissueB)$'
|
|
105
|
+
label: '{analyte}'
|
|
106
|
+
- match: '\.(?P<analyte>tissueA|tissueB) \(filtered\)$'
|
|
107
|
+
label: '{analyte} (filtered)'
|
|
108
|
+
- match: '\.library\.'
|
|
109
|
+
table: Library statistics
|
|
110
|
+
label_order: [tissueA, tissueB, tissueB (filtered)]
|
|
111
|
+
tables:
|
|
112
|
+
Library statistics:
|
|
113
|
+
description: Per-library read QC; read pairs nest under their library.
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
### YAML Configuration Reference
|
|
117
|
+
|
|
118
|
+
| Key | Meaning |
|
|
119
|
+
| --- | --- |
|
|
120
|
+
| `group` | A regular expression searched in every matched sample name. Its `(?P<group>...)` capture names the folded row. Required. |
|
|
121
|
+
| `levels` | An ordered list; the first level whose `match` is found in a sample name wins. Required. |
|
|
122
|
+
| `levels[].match` | A regular expression searched in the sample name. Named captures are available to `label`. |
|
|
123
|
+
| `levels[].label` | A format string built from the captures of `match`. Columns of matching rows are renamed with it and folded onto the group row; the row itself stays beneath. Omit it, and omit `table`, to fold the row's columns onto the group row unchanged. |
|
|
124
|
+
| `levels[].table` | The name of a table that receives matching rows instead of General Statistics. Rows keep their grouping, so paired reads stay nested under their library. |
|
|
125
|
+
| `column_title` | How a pivoted column is titled. `{label}` is the label as written, `{Label}` has its first letter upper-cased, `{title}` is the module's title. Default `{Label} {title}`. |
|
|
126
|
+
| `label_order` | Labels in the order their column blocks should appear. Labels not listed follow in order of first appearance. |
|
|
127
|
+
| `tables` | Presentation of the tables named by `levels[].table`, currently a `description` each. |
|
|
128
|
+
|
|
129
|
+
> [!TIP]
|
|
130
|
+
> A sample that matches no level is left where it was.
|
|
131
|
+
> A sample that matches a level but not `group` is left alone as well, with a warning in the log.
|
|
132
|
+
> Columns that a module did not declare a header for are dropped from folded rows, as MultiQC would have dropped them anyway.
|
|
133
|
+
|
|
134
|
+
### Limitations
|
|
135
|
+
|
|
136
|
+
1. Sample names are matched after MultiQC has cleaned them, so you must write patterns against the names you see in an un-pivoted report.
|
|
137
|
+
2. Only one level of nesting exists in a MultiQC table. Rows moved into a secondary table keep the nesting they already had; rows folded into a group row become its children, and cannot nest further.
|
|
138
|
+
3. Two rows in the same group that resolve to the same label collide. The first value is kept and a warning is logged, so make labels specific enough to tell such rows apart.
|
|
139
|
+
|
|
140
|
+
## Development and Testing
|
|
141
|
+
|
|
142
|
+
See the [contributing guide](./CONTRIBUTING.md) for more information.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
multiqc_pivot/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
multiqc_pivot/config.py,sha256=WmUrKIAoyIQ1Nub0IwhgzXQMo26AT7dl9f5GQlfXrS4,3680
|
|
3
|
+
multiqc_pivot/pivot.py,sha256=87Ra1rr-6PBJ3iM2TE0ILB-gtO8smSaXbvLAZeMRzvk,9029
|
|
4
|
+
multiqc_pivot/plugin.py,sha256=m21v8ERzw-LCOsJnUruPpNYEVByBnEDjDpFWi7kLCMs,3215
|
|
5
|
+
multiqc_pivot/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
multiqc_pivot-0.1.0.dist-info/METADATA,sha256=Uj_ktuQ5M1BV_LxyDXpuD2ZtyDNrxtpmZnV8YhIqyhU,7130
|
|
7
|
+
multiqc_pivot-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
8
|
+
multiqc_pivot-0.1.0.dist-info/entry_points.txt,sha256=RmzBZvimOvwqgzyH3Mo3Lv0Fpcoyrx3tFBi3PqGP1yg,125
|
|
9
|
+
multiqc_pivot-0.1.0.dist-info/licenses/LICENSE,sha256=s5pKbH2BYYoyLu52QqVnqHytR4BVo3F_TOFbJQcRbzI,1071
|
|
10
|
+
multiqc_pivot-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright © 2026 Clint Valentine
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|