nucleus-cdk 0.3.0a1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,68 @@
1
+ Metadata-Version: 2.1
2
+ Name: nucleus-cdk
3
+ Version: 0.3.0a1
4
+ Summary: Cell Developer Kit for building synthetic cells and cytosols.
5
+ Home-page: https://nucleus.bnext.bio/
6
+ License: MIT
7
+ Keywords: synthetic biology,synthetic cell,syncell,cellfree,nucleus,cell developer kit
8
+ Author: Anton Jackson-Smith
9
+ Author-email: anton@bnext.bio
10
+ Maintainer: Anton Jackson-Smith
11
+ Maintainer-email: anton@bnext.bio
12
+ Requires-Python: >=3.12,<4.0
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
21
+ Requires-Dist: hvplot (>=0.11.1,<0.12.0)
22
+ Requires-Dist: jinja2 (>=3.1.4,<4.0.0)
23
+ Requires-Dist: jupyter-bokeh (>=4.0.5,<5.0.0)
24
+ Requires-Dist: matplotlib (>=3.9.2,<4.0.0)
25
+ Requires-Dist: numpy (>=2.1.2,<3.0.0)
26
+ Requires-Dist: openpyxl (>=3.1.5,<4.0.0)
27
+ Requires-Dist: pandas (>=2.2.3,<3.0.0)
28
+ Requires-Dist: rich (>=13.9.2,<14.0.0)
29
+ Requires-Dist: scipy (>=1.14.1,<2.0.0)
30
+ Requires-Dist: seaborn (>=0.13.2,<0.14.0)
31
+ Requires-Dist: timple (>=0.1.8,<0.2.0)
32
+ Requires-Dist: tqdm (>=4.66.5,<5.0.0)
33
+ Project-URL: Repository, https://github.com/bnext-bio/nucleus/
34
+ Description-Content-Type: text/markdown
35
+
36
+ # CDK: Nucleus Cell Developer Kit
37
+ The Nucleus Cell Developer Kit (CDK) is a set of tools and libraries for building synthetic cells, the stack for synthetic cell engineering. This package is the core library, modular code that can be used in specific experiments or in other applications.
38
+
39
+ Nucleus is an open-source project for building and working with synthetic cells. For more information, please check out the [Nucleus documentation](https://nucleus.bnext.bio/).
40
+
41
+ ## Features
42
+ Currently, the CDK contains our core analysis functionality: plate reader and liposome analysis. In particular, the plate reader code is designed to allow for easy analysis of kinetic timeseries of PURE experiments from Agilent/Biotek and Revvity Envision plate readers. The liposome analysis code is under heavy development, and the code included here is somewhat out of date---please contact us for more information.
43
+
44
+ ## Installation
45
+ `pip install nucleus-cdk`
46
+
47
+ ## Development
48
+ ### Install poetry
49
+ The CDK uses poetry for dependency control and packaging. Install poetry, and activate it to download the dependencies. You can use poetry to manage the development virtual environment (recommended), or create a new conda environment to develop in.
50
+
51
+ #### Linux
52
+ To install poetry on Linux, you can use the following command:
53
+
54
+ ```bash
55
+ curl -sSL https://install.python-poetry.org | python3 -
56
+ ```
57
+
58
+ #### Mac
59
+ *(untested)* Install poetry using homebrew:
60
+ ```bash
61
+ brew install poetry
62
+ ```
63
+
64
+ ### Activate poetry and download dependencies
65
+ ```bash
66
+ poetry install
67
+ ```
68
+
@@ -0,0 +1,32 @@
1
+ # CDK: Nucleus Cell Developer Kit
2
+ The Nucleus Cell Developer Kit (CDK) is a set of tools and libraries for building synthetic cells, the stack for synthetic cell engineering. This package is the core library, modular code that can be used in specific experiments or in other applications.
3
+
4
+ Nucleus is an open-source project for building and working with synthetic cells. For more information, please check out the [Nucleus documentation](https://nucleus.bnext.bio/).
5
+
6
+ ## Features
7
+ Currently, the CDK contains our core analysis functionality: plate reader and liposome analysis. In particular, the plate reader code is designed to allow for easy analysis of kinetic timeseries of PURE experiments from Agilent/Biotek and Revvity Envision plate readers. The liposome analysis code is under heavy development, and the code included here is somewhat out of date---please contact us for more information.
8
+
9
+ ## Installation
10
+ `pip install nucleus-cdk`
11
+
12
+ ## Development
13
+ ### Install poetry
14
+ The CDK uses poetry for dependency control and packaging. Install poetry, and activate it to download the dependencies. You can use poetry to manage the development virtual environment (recommended), or create a new conda environment to develop in.
15
+
16
+ #### Linux
17
+ To install poetry on Linux, you can use the following command:
18
+
19
+ ```bash
20
+ curl -sSL https://install.python-poetry.org | python3 -
21
+ ```
22
+
23
+ #### Mac
24
+ *(untested)* Install poetry using homebrew:
25
+ ```bash
26
+ brew install poetry
27
+ ```
28
+
29
+ ### Activate poetry and download dependencies
30
+ ```bash
31
+ poetry install
32
+ ```
@@ -0,0 +1,92 @@
1
+ [tool.poetry]
2
+ name = "nucleus-cdk"
3
+ version = "0.3.0a1"
4
+ description = "Cell Developer Kit for building synthetic cells and cytosols."
5
+ license = "MIT"
6
+ authors = [
7
+ "Anton Jackson-Smith <anton@bnext.bio>",
8
+ ]
9
+ maintainers = [
10
+ "Anton Jackson-Smith <anton@bnext.bio>",
11
+ ]
12
+ readme = "README.md"
13
+ homepage = "https://nucleus.bnext.bio/"
14
+ repository = "https://github.com/bnext-bio/nucleus/"
15
+ packages = [
16
+ { include = "cdk", from = "src" }
17
+ ]
18
+ classifiers = [
19
+ "Development Status :: 3 - Alpha",
20
+ "Intended Audience :: Science/Research",
21
+ "License :: OSI Approved :: MIT License",
22
+ "Programming Language :: Python",
23
+ "Topic :: Scientific/Engineering :: Bio-Informatics",
24
+ ]
25
+ keywords = [
26
+ "synthetic biology",
27
+ "synthetic cell",
28
+ "syncell",
29
+ "cellfree",
30
+ "nucleus",
31
+ "cell developer kit"
32
+ ]
33
+
34
+ [tool.poetry.dependencies]
35
+ python = "^3.12"
36
+ numpy = "^2.1.2"
37
+ pandas = "^2.2.3"
38
+ seaborn = "^0.13.2"
39
+ matplotlib = "^3.9.2"
40
+ tqdm = "^4.66.5"
41
+ rich = "^13.9.2"
42
+ openpyxl = "^3.1.5"
43
+ timple = "^0.1.8"
44
+ jinja2 = "^3.1.4"
45
+ scipy = "^1.14.1"
46
+ hvplot = "^0.11.1"
47
+ jupyter-bokeh = "^4.0.5"
48
+
49
+ [tool.poetry.group.dev.dependencies]
50
+ pytest = "^8.3.3"
51
+ flake8 = "^7.1.1"
52
+ black = "^24.10.0"
53
+ pre-commit = "^4.0.0"
54
+ ipykernel = "^6.29.5"
55
+ toml-cli = "^0.7.0"
56
+ flake8-pyproject = "^1.2.3"
57
+
58
+ [tool.flake8]
59
+ max-line-length = 120
60
+ extend-ignore = ["E203", "E501", "W503", ]
61
+ per-file-ignores = [
62
+ "__init__.py:F401"
63
+ ]
64
+
65
+ [tool.black]
66
+ line-length = 79
67
+ target-version = ['py311']
68
+ include = '\.pyi?$'
69
+ extend-exclude = '''
70
+ /(
71
+ # directories
72
+ \.eggs
73
+ | \.git
74
+ | \.hg
75
+ | \.mypy_cache
76
+ | \.tox
77
+ | \.venv
78
+ | build
79
+ | dist
80
+ )/
81
+ '''
82
+
83
+ [tool.pytest.ini_options]
84
+ minversion = "6.0"
85
+ addopts = "-ra"
86
+ testpaths = [
87
+ "tests",
88
+ ]
89
+
90
+ [build-system]
91
+ requires = ["poetry-core"]
92
+ build-backend = "poetry.core.masonry.api"
File without changes
@@ -0,0 +1,844 @@
1
+ """
2
+ Plate Reader Module
3
+
4
+ This module provides support for loading and analyzing data from various plate readers. Currently supported are:
5
+ + BioTek Cytation 5
6
+ + Revvity Envision Nexus
7
+ + Promega Glomax Discover
8
+
9
+ Usage:
10
+ *tbd*
11
+ """
12
+
13
+ import io
14
+ import re
15
+ import os.path
16
+ from pathlib import Path
17
+ import logging
18
+ from typing import Union, Optional, NamedTuple
19
+ from enum import Enum, auto
20
+
21
+ import numpy as np
22
+ import pandas as pd
23
+
24
+ # import matplotlib.pyplot as plt
25
+ import matplotlib as mpl
26
+ import seaborn as sns
27
+ import timple
28
+ import timple.timedelta
29
+ import scipy.optimize
30
+ import functools
31
+ import warnings
32
+
33
+
34
+ log = logging.getLogger(__name__)
35
+
36
+ DataFile = Union[str, Path, io.StringIO]
37
+
38
+
39
+ class SteadyStateMethod(Enum):
40
+ LOWEST_VELOCITY = (auto(),)
41
+ MAXIMUM_VALUE = (auto(),)
42
+ VELOCITY_INTERCEPT = auto()
43
+
44
+
45
+ class PlateReaderData(NamedTuple):
46
+ """
47
+ A named tuple representing plate reader data and associated plate map information.
48
+ """
49
+
50
+ data: pd.DataFrame
51
+ platemap: Optional[pd.DataFrame] = None
52
+
53
+
54
+ _timple = timple.Timple()
55
+
56
+
57
+ def load_platereader_data(
58
+ data_file: DataFile,
59
+ platemap_file: Optional[DataFile] = None,
60
+ platereader: Optional[str] = None,
61
+ ) -> Union[PlateReaderData, pd.DataFrame]:
62
+ """
63
+ Load plate reader data from a file and return a DataFrame.
64
+
65
+ This function loads platereader data from a CSV file, parsing it into a standardized format and labelling
66
+ it with a provided plate map.
67
+
68
+ Filenames should be formatted in a standard format: `[date]-[device]-[experiment].csv`. For
69
+ example, `20241004-envision-dna-concentration.csv`.
70
+
71
+ Data is loaded based on the device field in the filename, which is used to determine the appropriate reader-specific
72
+ data parser. Currently supported readers are:
73
+ - BioTek Cytation 5: `cytation`
74
+ - Revvity Envision Nexus: `envision`
75
+
76
+ Data is returned as a pandas DataFrame with the following mandatory columns:
77
+ - `Well`: Well identifier (e.g. `A1`)
78
+ - `Row`: Row identifier (e.g. `A`)
79
+ - `Column`: Column identifier (e.g. `1`)
80
+ - `Time`: Time of measurement
81
+ - `Seconds`: Time of measurement in seconds
82
+ - `Temperature (C)`: Temperature at time of measurement
83
+ - `Read`: A tag describing the type of measurement (e.g. `OD600`, `Fluorescence`). The format of this field is
84
+ currently device-specific.
85
+ - `Data`: The measured data value
86
+
87
+ In addition, the provided platemap will be merged to the loaded data on the `Well` column. All other columns within
88
+ the platemap will be present in the returned dataframe.
89
+
90
+ Args:
91
+ data_file (str): Path to the plate reader data file.
92
+
93
+ Returns:
94
+ If a platemap is provided, a PlateReaderData named tuple containing the data and platemap DataFrames. Otherwise,
95
+ just the data. If a platemap_file is provided, the returned platemap is guaranteed to be not None.
96
+
97
+ platemap_file is not None:
98
+ PlateReaderData: A named tuple containing the plate reader data and platemap DataFrames: (data, platemap)
99
+ platemap_file is None:
100
+ pd.DataFrame: DataFrame containing the plate reader data in a structured format.
101
+
102
+
103
+ """
104
+ if platereader is None:
105
+ platereader = os.path.basename(data_file).lower()
106
+
107
+ # TODO: Clean this up to use a proper platereader enum and not janky string parsing.
108
+ if "cytation" in platereader.lower():
109
+ data = read_cytation(data_file)
110
+ elif "biotek" in platereader.lower():
111
+ data = read_cytation(data_file)
112
+ elif "envision" in platereader.lower():
113
+ data = read_envision(data_file)
114
+ # elif filename_lower.startswith("glomax"):
115
+ # return read_glomax(os.path.dirname(data_file))
116
+ else:
117
+ raise ValueError(f"Unsupported plate reader data file: {data_file}")
118
+
119
+ platemap = None
120
+ if platemap_file is not None:
121
+ platemap = read_platemap(platemap_file)
122
+ data = data.merge(platemap, on="Well")
123
+ return PlateReaderData(data=data, platemap=platemap)
124
+
125
+ return data
126
+
127
+
128
+ def read_platemap(platemap_file: DataFile) -> pd.DataFrame:
129
+ if isinstance(platemap_file, io.StringIO):
130
+ platemap = pd.read_csv(platemap_file)
131
+ else:
132
+ extension = os.path.splitext(platemap_file)[1].lower()
133
+ if extension == ".csv":
134
+ platemap = pd.read_csv(platemap_file)
135
+ elif extension == ".tsv":
136
+ platemap = pd.read_table(platemap_file)
137
+ # TODO: create test for this
138
+ elif extension == ".xlsx":
139
+ platemap = pd.read_excel(platemap_file)
140
+ else:
141
+ raise ValueError(
142
+ f"Unsupported platemap file, use csv or xlsx: {platemap_file}"
143
+ )
144
+
145
+ # Remove unnamed columns from the plate map.
146
+ platemap = platemap[
147
+ [col for col in platemap.columns if not col.startswith("Unnamed:")]
148
+ ]
149
+
150
+ # Needed to make sure times are correctly converted, but we don't convert
151
+ # floats because they get upcast to a pandas Float64Dtype() class which
152
+ # messes up plotting.
153
+ # platemap = platemap.convert_dtypes(convert_floating=False)
154
+
155
+ platemap["Well"] = platemap["Well"].str.replace(
156
+ ":", ""
157
+ ) # Normalize well by removing : if it exists
158
+ return platemap
159
+
160
+
161
+ # def read_glomax(data_dir: str) -> pd.DataFrame:
162
+ # # glob over .csv files in dfpath; append to data; concatenate into one DataFrame
163
+ # data = list()
164
+ # for csv in glob.glob(f"{data_dir}/*.csv"):
165
+ # df = pd.read_csv(csv)
166
+ # df["File"] = os.path.basename(csv)
167
+ # df["Row"] = df["WellPosition"].str.split(":").str[0]
168
+ # df["Column"] = df["WellPosition"].str.split(":").str[1].astype(int)
169
+ # df["Time"] = pd.to_datetime(
170
+ # data["File"].str.replace(r".* ([0-9.]+ [0-9_]+).*", r"\1", regex=True), format="%Y.%m.%d %H_%M_%S"
171
+ # )
172
+ # df["WellTime"] = pd.to_timedelta(data["Timestamp(ms)"], "us")
173
+
174
+ # data.append(df)
175
+
176
+ # data = pd.concat(data, ignore_index=True)
177
+
178
+ # # label different wavelengths
179
+ # channel_map = dict(zip(data["ID"].unique(), ["A600", "Blue", "Green", "Red"]))
180
+ # data["Channel"] = data["ID"].map(channel_map)
181
+
182
+ # palette = dict(zip(dict.fromkeys(channel_map.values()), ["brown", "limegreen", "red", "firebrick"]))
183
+
184
+ # # massage Time
185
+ # data["TimeDelta"] = data["Time"] - data["Time"].min()
186
+ # data["TimeDeltaPretty"] = data["TimeDelta"].map(
187
+ # lambda x: "{:02d}:00".format(x.components.hours)
188
+ # ) # {:02d}".format(x.components.hours, x.components.minutes))
189
+
190
+ # # Get a generic data column
191
+ # data["Data"] = data["CalculatedFlux"]
192
+ # data["Data"].fillna(data[data["CalculatedFlux"].isna()]["OpticalDensity"], inplace=True)
193
+
194
+ # # Label replicates
195
+ # data["Replicate"] = data["File"].map(lambda x: re.sub(r".* OUT ([0-9]+).csv", r"\1", x))
196
+ # data.sort_values(by=["TimeDelta", "Row", "Column"], inplace=True)
197
+
198
+ # return data
199
+
200
+
201
+ def read_cytation(data_file: DataFile, sep="\t") -> pd.DataFrame:
202
+ log.debug(f"Reading Cytation data from {data_file}")
203
+ # read data file as long string
204
+ data = ""
205
+ with open(data_file, "r", encoding="latin1") as file:
206
+ data = file.read()
207
+
208
+ # extract indices for Proc Details, Layout
209
+ procidx = re.search(r"Procedure Details", data)
210
+ layoutidx = re.search(r"Layout", data)
211
+ readidx = re.search(r"^(Read\s)?\d+(/\d+)?,\d+(/\d+)?", data, re.MULTILINE)
212
+
213
+ # get header DataFrame
214
+ header = data[: procidx.start()]
215
+ header = pd.read_csv(
216
+ io.StringIO(header), delimiter=sep, header=0, names=["key", "value"]
217
+ )
218
+
219
+ # get procedure DataFrame
220
+ procedure = data[procidx.end() : layoutidx.start()]
221
+ procedure = pd.read_csv(
222
+ io.StringIO(procedure), skipinitialspace=True, names=range(4)
223
+ )
224
+ procedure = procedure.replace(np.nan, "")
225
+
226
+ # get Cytation plate map from data_file as DataFrame
227
+ layout = data[layoutidx.end() : readidx.start()]
228
+ layout = pd.read_csv(io.StringIO(layout), index_col=False)
229
+ layout = layout.set_index(layout.columns[0])
230
+ layout.index.name = "Row"
231
+
232
+ # iterate over data string to find individual reads
233
+ reads = dict()
234
+
235
+ sep = (
236
+ r"(?:Read\s\d+:)?(?:\s\d{3}(?:/\d+)?,\d{3}(?:/\d+)?(?:\[\d\])?)?" + sep
237
+ )
238
+
239
+ for readidx in re.finditer(
240
+ r"^(Read\s)?\d+(/\d+)?,\d+(/\d+)?.*\n", data, re.MULTILINE
241
+ ):
242
+ # for each iteration, extract string from start idx to end icx
243
+ read = data[readidx.end() :]
244
+ read = read[
245
+ : re.search(
246
+ r"(^(Read\s)?\d+,\d+|^Blank Read\s\d|Results|\Z)",
247
+ read[1:],
248
+ re.MULTILINE,
249
+ ).start()
250
+ ]
251
+ read = pd.read_csv(
252
+ io.StringIO(read), sep=sep, engine="python"
253
+ ).convert_dtypes(convert_floating=False)
254
+ reads[data[readidx.start() : readidx.end()].strip()] = read
255
+
256
+ # create a DataFrame for each read and process, then concatenate into a large DataFrame
257
+ # NOTE: JC 2024-05-21 - turns out, len(list(reads.items())) = 1 (one big mono table)
258
+ read_dataframes = list()
259
+ for name, r in reads.items():
260
+ # filter out Cytation calculated kinetic parameters, which are cool, but don't want rn
261
+ r = r[r.Time.str.contains(r"\d:\d{2}:\d{2}", regex=True)]
262
+
263
+ # extract meaningful parameters from really big string
264
+ r = r.melt(id_vars=["Time", "T°"], var_name="Well", value_name="Data")
265
+ r["Row"] = r["Well"].str.extract(r"([A-Z]+)")
266
+ r["Column"] = r["Well"].str.extract(r"(\d+)").astype(int)
267
+ r["Temperature (C)"] = r["T°"] # .str.extract(r"(\d+)").astype(float)
268
+ r["Data"] = r["Data"].replace("OVRFLW", np.inf)
269
+ r["Data"] = r["Data"].astype(float)
270
+ r["Read"] = name
271
+ r["Ex"] = r["Read"].str.extract(r"(\d+),\d+").astype(int)
272
+ r["Em"] = r["Read"].str.extract(r"\d+,(\d+)").astype(int)
273
+ read_dataframes.append(r)
274
+
275
+ data = pd.concat(read_dataframes)
276
+
277
+ # add time column to data DataFrame
278
+ data["Time"] = pd.to_timedelta(data["Time"])
279
+ data["Seconds"] = data["Time"].map(lambda x: x.total_seconds())
280
+
281
+ return data[
282
+ [
283
+ "Well",
284
+ "Row",
285
+ "Column",
286
+ "Time",
287
+ "Seconds",
288
+ "Temperature (C)",
289
+ "Read",
290
+ "Data",
291
+ ]
292
+ ]
293
+
294
+
295
+ def read_envision(data_file: DataFile) -> pd.DataFrame:
296
+ # load data
297
+ data = pd.read_csv(data_file).convert_dtypes()
298
+
299
+ # massage Row, Column, and Well information
300
+ data["Row"] = (
301
+ data["Well ID"].apply(lambda s: s[0]).astype(pd.StringDtype())
302
+ )
303
+ data["Column"] = data["Well ID"].apply(lambda s: str(int(s[1:])))
304
+ data["Well"] = data.apply(
305
+ lambda well: f"{well['Row']}{well['Column']}", axis=1
306
+ )
307
+
308
+ data["Time"] = pd.to_timedelta(data["Time [hhh:mm:ss.sss]"])
309
+ data["Seconds"] = data["Time"].map(lambda x: x.total_seconds())
310
+
311
+ data["Temperature (C)"] = data["Temperature current[°C]"]
312
+
313
+ data["Read"] = data["Operation"]
314
+
315
+ data["Data"] = data["Result Channel 1"]
316
+
317
+ data["Excitation (nm)"] = data["Exc WL[nm]"]
318
+ data["Emission (nm)"] = data["Ems WL Channel 1[nm]"]
319
+ data["Wavelength (nm)"] = (
320
+ data["Excitation (nm)"] + "," + data["Emission (nm)"]
321
+ )
322
+
323
+ return data[
324
+ [
325
+ "Well",
326
+ "Row",
327
+ "Column",
328
+ "Time",
329
+ "Seconds",
330
+ "Temperature (C)",
331
+ "Read",
332
+ "Data",
333
+ ]
334
+ ]
335
+
336
+
337
+ def blank_data(data: pd.DataFrame, blank_type="Blank"):
338
+ """
339
+ Blank data from plate reader measurements.
340
+
341
+ Adjusts plate reader data by subtracting the value of one or more blanks at each timepoint, for each read channel.
342
+ By default, the data will be blanked against the mean value of all wells of type "Blank".
343
+
344
+ This function adjusts the main "Data" column in the dataframe provided, so that blanked values can be easily
345
+ used in subsequent processing. The original (unblanked) data is available in a new 'Data_unblanked' column. The
346
+ blank value calculated for each row of the data is present in 'Data_blank'.
347
+
348
+ Args:
349
+ data (pd.DataFrame): Input DataFrame containing 'Well', 'Time', 'Data', and 'Type' columns.
350
+ blank_type (str, optional): Value in the 'Type' column to use as blank. Defaults to "Blank".
351
+
352
+ Returns:
353
+ pd.DataFrame: DataFrame with blanked 'Data' values and an additional 'Data_unblanked' column.
354
+
355
+ """
356
+ blank = (
357
+ data[data["Type"] == blank_type]
358
+ .groupby(["Time", "Read"])["Data"]
359
+ .mean()
360
+ )
361
+ data = data.merge(
362
+ blank, on=["Time", "Read"], suffixes=("", "_blank"), how="left"
363
+ )
364
+
365
+ # Check to make sure we don't have missing blanks for certain Time/Read combinations in the source data.
366
+ # The most likely way this could happen is if the platereader "Time" isn't aligned well-to-well.
367
+ if data["Data_blank"].isna().any():
368
+ log.warning(
369
+ "Not all data has a blank value; blanked data will contain NaNs."
370
+ )
371
+
372
+ data["Data_unblanked"] = data["Data"].copy()
373
+ data["Data"] = data["Data"] - data["Data_blank"]
374
+
375
+ return data
376
+
377
+
378
+ def plot_setup() -> None:
379
+ _timple.enable()
380
+ pd.set_option("display.float_format", "{:.2f}".format)
381
+
382
+
383
+ def _plot_timedelta(plot: sns.FacetGrid | mpl.axes.Axes) -> sns.FacetGrid:
384
+ axes = [plot]
385
+ if isinstance(plot, sns.FacetGrid):
386
+ axes = plot.axes.flatten()
387
+
388
+ for ax in axes:
389
+ # ax.xaxis.set_major_locator(timple.timedelta.AutoTimedeltaLocator(minticks=3))
390
+ ax.xaxis.set_major_formatter(
391
+ timple.timedelta.TimedeltaFormatter("%h:%m")
392
+ )
393
+ ax.set_xlabel("Time (hours)")
394
+
395
+ # g.set_xlabels("Time (hours)")
396
+ # g.figure.autofmt_xdate()
397
+
398
+
399
+ def plot_plate(data: pd.DataFrame) -> sns.FacetGrid:
400
+ g = sns.relplot(
401
+ data=data, x="Time", y="Data", row="Row", col="Column", kind="line"
402
+ )
403
+ _plot_timedelta(g)
404
+
405
+ g.set_ylabels("Fluorescence (RFU)")
406
+ g.set_titles("{row_name}{col_name}")
407
+
408
+ return g
409
+
410
+
411
+ def plot_curves_by_name(
412
+ data: pd.DataFrame, by_experiment=True
413
+ ) -> sns.FacetGrid:
414
+ """
415
+ Produce a basic plot of timeseries curves, coloring curves by the `Name` of the sample.
416
+
417
+ If there are multiple different `Read`s in the data (e.g., GFP, RFP), then a subplot will be
418
+ produced for each read. If there are multiple experiments, each experiment will be plotted separately.
419
+
420
+ Args:
421
+ data (pd.DataFrame): DataFrame containing plate reader data.
422
+ by_experiment (bool, optional): If True, then each experiment will be plotted in a separate subplot.
423
+
424
+ Returns:
425
+ sns.FacetGrid: Seaborn FacetGrid object containing the plot.
426
+ """
427
+ kwargs = {}
428
+ if "Experiment" in data.columns and by_experiment:
429
+ kwargs["col"] = "Experiment"
430
+
431
+ g = plot_curves(
432
+ data=data, x="Time", y="Data", hue="Name", row="Read", **kwargs
433
+ )
434
+
435
+ return g
436
+
437
+
438
+ def plot_curves(
439
+ data: pd.DataFrame,
440
+ x="Time",
441
+ y="Data",
442
+ hue="Name",
443
+ labels=(None, "Fluorescence (RFU)"),
444
+ **kwargs,
445
+ ) -> sns.FacetGrid:
446
+ """
447
+ Plot timeseries curves from a plate reader dataset, allowing selection of the parameters to
448
+ use for plotting and to divide the data into multiple subplots.
449
+
450
+ This function is a thin wrapper around Seaborn `relplot`, providing sensible defaults while
451
+ also allowing for the use of any `relplot` parameter.
452
+
453
+ Args:
454
+ data (pd.DataFrame): DataFrame containing plate reader data.
455
+ x (str, optional): Column name to use for x-axis. Defaults to "Time".
456
+ y (str, optional): Column name to use for y-axis. Defaults to "Data".
457
+ hue (str, optional): Column name to use for color coding. Defaults to "Name".
458
+ labels (tuple, optional): Labels for the x and y axes. Defaults to (None, "Fluorescence (RFU)").
459
+ If None, use the default label (the name of the field, or a formatted time label).
460
+ **kwargs: Additional keyword arguments passed to `sns.relplot`.
461
+
462
+ Returns:
463
+ sns.FacetGrid: A FacetGrid object containing the plotted data.
464
+
465
+ """
466
+ g = sns.relplot(data=data, x=x, y=y, hue=hue, kind="line", **kwargs)
467
+ _plot_timedelta(g)
468
+
469
+ x_label, y_label = labels
470
+ if x_label:
471
+ g.set_xlabels(x_label)
472
+ if y_label:
473
+ g.set_ylabels(y_label)
474
+
475
+ # Set simple row and column titles, if we're faceting on row or column.
476
+ # The join means the punctuation only gets added if we have both.
477
+ var_len = max(
478
+ [len(kwargs[var]) for var in ["row", "col"] if var in kwargs] + [0]
479
+ )
480
+ log.debug(f"{var_len=}")
481
+ row_title = (
482
+ f"{{row_var:>{var_len}}}: {{row_name}}" if "row" in kwargs else ""
483
+ )
484
+ col_title = (
485
+ f"{{col_var:>{var_len}}}: {{col_name}}" if "col" in kwargs else ""
486
+ )
487
+ g.set_titles("\n".join(filter(None, [row_title, col_title])))
488
+
489
+ return g
490
+
491
+
492
+ ###
493
+ # Kinetics Analysis
494
+ # TODO: Perhaps split this out into a submodule.
495
+ ###
496
+
497
+
498
+ def find_steady_state_for_well(well):
499
+ well = well.sort_values("Time")
500
+ pct_change = well["Data"].rolling(window=3).mean().pct_change()
501
+ idx_maxV = pct_change.idxmax()
502
+
503
+ ss_idx = pct_change.loc[idx_maxV:].abs().idxmin()
504
+ ss_time = well.loc[ss_idx, "Time"]
505
+ ss_level = well.loc[ss_idx, "Data"]
506
+
507
+ return pd.Series(
508
+ {"Time_steadystate": ss_time, "Data_steadystate": ss_level}
509
+ )
510
+
511
+
512
+ def find_steady_state(
513
+ data: pd.DataFrame, window_size=10, threshold=0.01
514
+ ) -> pd.DataFrame:
515
+ """
516
+ Find the steady state of the "Data" column in the provided data DataFrame.
517
+
518
+ Args:
519
+ data (pd.DataFrame): Input DataFrame containing 'Well', 'Time', and 'Data' columns.
520
+ window_size (int): Size of the rolling window for calculating the rate of change.
521
+ threshold (float): Threshold for determining steady state.
522
+
523
+ Returns:
524
+ pd.DataFrame: DataFrame with 'Well', 'SteadyStateTime', and 'SteadyStateLevel' columns.
525
+ """
526
+
527
+ result = data.groupby(["Well", "Read"]).apply(find_steady_state_for_well)
528
+ return result
529
+
530
+
531
+ def _sigmoid(x, L, k, x0):
532
+ return L / (1 + np.exp(-k * (x - x0)))
533
+
534
+
535
+ def kinetic_analysis_per_well(
536
+ data: pd.DataFrame, data_column="Data"
537
+ ) -> pd.DataFrame:
538
+
539
+ steadystate = find_steady_state_for_well(data)
540
+
541
+ data = data.loc[data["Time"] <= steadystate["Time_steadystate"]]
542
+ time = data["Time"].dt.total_seconds()
543
+
544
+ # make initial guesses for parameters
545
+ L_initial = np.max(data[data_column])
546
+ x0_initial = np.max(time) / 4
547
+ k_initial = (
548
+ np.log(L_initial * 1.1 / data[data_column] - 1) / (time - x0_initial)
549
+ ).dropna().mean() * -1.0
550
+ p0 = [L_initial, k_initial, x0_initial]
551
+
552
+ # attempt fitting
553
+ params = [0, 0, 0]
554
+ with warnings.catch_warnings():
555
+ warnings.simplefilter("error", scipy.optimize.OptimizeWarning)
556
+ try:
557
+ params, _ = scipy.optimize.curve_fit(
558
+ _sigmoid, time, data[data_column], p0=p0
559
+ )
560
+ except scipy.optimize.OptimizeWarning as w:
561
+ log.debug(f"Scipy optimize warning: {w}")
562
+ except Exception as e:
563
+ log.warning(f"Failed to solve: {e}")
564
+
565
+ return None
566
+ log.debug(f"{data['Well'].iloc[0]} Fitted params: {params}")
567
+
568
+ # calculate velocities and velocity params
569
+ v = data[data_column].diff() / data["Time"].dt.total_seconds().diff()
570
+
571
+ maxV = v.max()
572
+ maxV_d = data.loc[v.idxmax(), data_column]
573
+ maxV_time = data.loc[v.idxmax(), "Time"]
574
+
575
+ # calculate lag time
576
+ lag = -maxV_d / maxV + maxV_time.total_seconds()
577
+ lag_data = _sigmoid(lag, *params)
578
+
579
+ # decile_upper = data[data_column].quantile(0.95)
580
+ # decile_lower = data[data_column].quantile(0.05)
581
+
582
+ # growth_s = (decile_upper - maxV_d) / maxV + maxV_time.total_seconds()
583
+
584
+ # ss_time = data.loc[(data[data_column] > decile_upper).idxmax(), "Time"]
585
+ # ss_d = data.loc[
586
+ # (data[data_column] > decile_upper).idxmax() :, data_column
587
+ # ].mean()
588
+
589
+ # kinetics = {
590
+ # # f"{data_column}_fit_d": y_fit,
591
+ # f"{data_column}_maxV": maxV,
592
+ # f"{data_column}_t_maxV": t_maxV,
593
+ # f"{data_column}_maxV_d": maxV_d,
594
+ # f"{data_column}_lag_s": lag,
595
+ # f"{data_column}_growth_s": growth_s,
596
+ # f"{data_column}_ss_s": ss_s,
597
+ # f"{data_column}_ss_d": ss_d,
598
+ # f"{data_column}_low_d": decile_lower,
599
+ # f"{data_column}_high_d": decile_upper,
600
+ # }
601
+
602
+ kinetics = {
603
+ # f"{data_column}_fit_d": y_fit,
604
+ ("Velocity", "Time"): maxV_time,
605
+ ("Velocity", data_column): maxV_d,
606
+ ("Velocity", "Max"): maxV,
607
+ ("Lag", "Time"): pd.to_timedelta(lag, unit="s"),
608
+ ("Lag", "Data"): lag_data,
609
+ # f"{data_column}_growth_s": growth_s,
610
+ ("Steady State", "Time"): steadystate["Time_steadystate"],
611
+ ("Steady State", data_column): steadystate["Data_steadystate"],
612
+ ("Fit", "L"): params[0],
613
+ ("Fit", "k"): params[1],
614
+ ("Fit", "x0"): params[2],
615
+ }
616
+
617
+ return pd.Series(kinetics)
618
+ # return kinetics
619
+
620
+
621
+ def kinetic_analysis(data: pd.DataFrame, data_column="Data") -> pd.DataFrame:
622
+ kinetics = data.groupby(["Well", "Name", "Read"], sort=False).apply(
623
+ functools.partial(kinetic_analysis_per_well, data_column=data_column)
624
+ )
625
+ return kinetics
626
+
627
+
628
+ def kinetic_analysis_summary(
629
+ data: pd.DataFrame,
630
+ data_column="Data",
631
+ time_cutoff: int = 12000,
632
+ label_order: list[str] = None,
633
+ norm_label: str = None,
634
+ ):
635
+ def per_well_cleanup(df):
636
+ cols = df.columns
637
+ return df[["Well"] + list(cols[27:])].aggregate(lambda x: x.iloc[0])
638
+
639
+ tk = kinetic_analysis(
640
+ data=data, data_column=data_column, time_cutoff=time_cutoff
641
+ )
642
+ out = tk.groupby("Well").apply(per_well_cleanup).reset_index(drop=True)
643
+
644
+ if label_order:
645
+ out = out.set_index("Well").reindex(label_order).reset_index()
646
+
647
+ # normalize max value (calculated by kinetics) if norm_label given
648
+ if norm_label:
649
+ norm = out[out["Well"] == norm_label][f"{data_column}_high_d"].values
650
+ out["Normalized (%)"] = out[f"{data_column}_high_d"] / norm
651
+
652
+ return out
653
+
654
+
655
+ def plot_kinetics_by_well(
656
+ data: pd.DataFrame,
657
+ kinetics: pd.DataFrame,
658
+ x: str = "Time",
659
+ y: str = "Data",
660
+ show_fit: bool = False,
661
+ show_velocity: bool = False,
662
+ annotate: bool = False,
663
+ **kwargs,
664
+ ):
665
+ """
666
+ Typical usage:
667
+
668
+ > tk = kinetic_analysis(data=data, data_column="BackgroundSubtracted")
669
+ > g = sns.FacetGrid(tk, col="Well", col_wrap=2, sharey=False, height=4, aspect=1.5)
670
+ > g.map_dataframe(plot_kinetics, show_fit=True, show_velocity=True)
671
+ """
672
+ colors = sns.color_palette("Set2")
673
+
674
+ ax = sns.scatterplot(data=data, x=x, y=y, color=colors[2], alpha=0.5)
675
+
676
+ well = data["Well"].iloc[0]
677
+ name = data["Name"].iloc[0]
678
+ read = data["Read"].iloc[0]
679
+ kinetics = kinetics.loc[well, name, read]
680
+ if (kinetics.isna()).any():
681
+ log.info(f"Kinetics information not available for {well}.")
682
+ return
683
+
684
+ # ax_ylim = (
685
+ # ax.get_ylim()
686
+ # ) # Use this to run lines to bounds later, then restore them before returning.
687
+
688
+ if show_fit:
689
+ L = kinetics["Fit", "L"]
690
+ k = kinetics["Fit", "k"]
691
+ x0 = kinetics["Fit", "x0"]
692
+ sns.lineplot(
693
+ x=data["Time"],
694
+ y=_sigmoid(data["Time"].dt.total_seconds(), L, k, x0),
695
+ linestyle="--",
696
+ color=colors[3],
697
+ # alpha=0.5,
698
+ ax=ax,
699
+ )
700
+ # sns.lineplot(data=data, x=x, y=y, linestyle="--", c="red", alpha=0.5)
701
+
702
+ # Max Velocity
703
+ # maxV_x = np.linspace(ax.get_xlim()[0], ax.get_xlim()[1], 100)
704
+ maxV_y = (
705
+ kinetics["Velocity", "Max"]
706
+ * (data["Time"] - kinetics["Velocity", "Time"]).dt.total_seconds()
707
+ + kinetics["Velocity", "Data"]
708
+ )
709
+
710
+ sns.lineplot(
711
+ x=data["Time"].loc[(maxV_y > 0) & (maxV_y < data[y].max())],
712
+ y=maxV_y[(maxV_y > 0) & (maxV_y < data[y].max())],
713
+ linestyle="--",
714
+ color=colors[1],
715
+ ax=ax,
716
+ )
717
+
718
+ maxV = kinetics["Velocity", "Max"]
719
+ maxV_s = kinetics["Velocity", "Time"]
720
+ maxV_d = kinetics["Velocity", "Data"]
721
+
722
+ # decile_upper = summary[f"{y}_high_d"]
723
+ # decile_lower = summary[f"{y}_low_d"]
724
+ # ax.vlines(
725
+ # lag,
726
+ # ymin=ax_ylim[0],
727
+ # ymax=decile_lower,
728
+ # colors=colors[2],
729
+ # linestyle="--",
730
+ # )
731
+
732
+ # Time to Steady State
733
+ ss_s = kinetics["Steady State", "Time"]
734
+ ax.axvline(ss_s, c=colors[3], linestyle="--")
735
+
736
+ # # Range
737
+ # ax.axhline(decile_upper, c=colors[7], linestyle="--")
738
+ # ax.axhline(decile_lower, c=colors[7], linestyle="--")
739
+
740
+ if annotate:
741
+ # Plot the text annotations on the chart
742
+ ax.annotate(
743
+ f"$V_{{max}} =$ {maxV:.2f} u/s",
744
+ (maxV_s, maxV_d),
745
+ xytext=(24, 0),
746
+ textcoords="offset points",
747
+ arrowprops={"arrowstyle": "->"},
748
+ ha="left",
749
+ va="center",
750
+ c="black",
751
+ )
752
+
753
+ f = timple.timedelta.TimedeltaFormatter("%h:%m")
754
+ lag_label = f.format_data(
755
+ timple.timedelta.timedelta2num(kinetics["Lag", "Time"])
756
+ )
757
+ ax.annotate(
758
+ f"$t_{{lag}} =$ {lag_label}",
759
+ (kinetics["Lag", "Time"], kinetics["Lag", "Data"]),
760
+ xytext=(12, 0),
761
+ textcoords="offset points",
762
+ ha="left",
763
+ va="center",
764
+ )
765
+
766
+ ss_label = f.format_data(
767
+ timple.timedelta.timedelta2num(kinetics["Steady State", "Time"])
768
+ )
769
+ ax.annotate(
770
+ f"$t_{{steady state}} =$ {ss_label}",
771
+ (
772
+ kinetics["Steady State", "Time"],
773
+ kinetics["Steady State", "Data"],
774
+ ),
775
+ xytext=(0, -12),
776
+ textcoords="offset points",
777
+ ha="left",
778
+ va="top",
779
+ )
780
+
781
+ # Velocity
782
+ if show_velocity:
783
+ # TODO: This is currently broken due to rolling calculation and its effect on bounds.
784
+ # Show a velocity sparkline over the plot
785
+ velocity = (
786
+ data.transform({y: "diff", x: lambda x: x}).rolling(5).mean()
787
+ )
788
+ velocity[y] = velocity[y]
789
+ # velocity_ax = ax.secondary_yaxis(location="right",
790
+ # functions=(lambda x: pd.Series(x).rolling(5).mean().values, lambda x: x))
791
+ velocity_ax = ax.twinx()
792
+ sns.lineplot(data=velocity, x=x, y=y, alpha=0.5, ax=velocity_ax)
793
+ velocity_ax.set_ylabel("$V (u/s)$")
794
+ velocity_ax.set_ylim((0, velocity[y].max() * 2))
795
+
796
+ # ax.set_ylim(ax_ylim)
797
+
798
+ _plot_timedelta(ax)
799
+
800
+
801
+ def plot_kinetics(data: pd.DataFrame, kinetics: pd.DataFrame, **kwargs):
802
+ g = sns.FacetGrid(
803
+ data, col="Name", col_wrap=3, sharey=True, height=4, aspect=1.5
804
+ )
805
+ g.map_dataframe(
806
+ plot_kinetics_by_well,
807
+ kinetics=kinetics,
808
+ show_fit=True,
809
+ show_velocity=False,
810
+ annotate=True,
811
+ )
812
+ g.set_ylabels("Fluorescence (RFU)")
813
+
814
+
815
+ def plot_steadystate(data: pd.DataFrame, **kwargs):
816
+ steady_state = find_steady_state(data).reset_index()
817
+ data_with_steady_state = data.merge(steady_state, on="Well", how="left")
818
+
819
+ exp = "Experiment" if "Experiment" in data.columns else None
820
+ col = exp if exp is not None else None
821
+ col_wrap = 2 if col is not None else None
822
+
823
+ g = sns.catplot(
824
+ data=data_with_steady_state,
825
+ x="Name",
826
+ y="Data_steadystate",
827
+ hue=exp,
828
+ kind="bar",
829
+ col=col,
830
+ col_wrap=col_wrap,
831
+ height=4,
832
+ aspect=1.5,
833
+ sharex=False,
834
+ **kwargs,
835
+ )
836
+
837
+ g.set_xticklabels(rotation=90)
838
+ g.set_ylabels("Steady State Fluorescence (RFU)")
839
+ return g
840
+
841
+
842
+ def export():
843
+ # TODO: Write me
844
+ pass
@@ -0,0 +1,27 @@
1
+ import logging
2
+ import rich.console
3
+ import rich.logging
4
+
5
+
6
+ def setup_logging(log_level=logging.INFO):
7
+ # Set up logging
8
+ console = rich.console.Console(
9
+ force_jupyter=False,
10
+ # stderr=True,
11
+ theme=rich.theme.Theme(
12
+ {"logging.level.debug": "cyan", "logging.level.info": "green"}
13
+ ),
14
+ )
15
+
16
+ logging.basicConfig(
17
+ level=log_level,
18
+ format="%(message)s",
19
+ handlers=[rich.logging.RichHandler(
20
+ console=console,
21
+ enable_link_path=False)
22
+ ],
23
+ )
24
+
25
+ log = logging.getLogger(__name__)
26
+ log.info("Logging initialized")
27
+ return log