nucleus-cdk 0.3.0a1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: nucleus-cdk
|
|
3
|
+
Version: 0.3.0a1
|
|
4
|
+
Summary: Cell Developer Kit for building synthetic cells and cytosols.
|
|
5
|
+
Home-page: https://nucleus.bnext.bio/
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: synthetic biology,synthetic cell,syncell,cellfree,nucleus,cell developer kit
|
|
8
|
+
Author: Anton Jackson-Smith
|
|
9
|
+
Author-email: anton@bnext.bio
|
|
10
|
+
Maintainer: Anton Jackson-Smith
|
|
11
|
+
Maintainer-email: anton@bnext.bio
|
|
12
|
+
Requires-Python: >=3.12,<4.0
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
21
|
+
Requires-Dist: hvplot (>=0.11.1,<0.12.0)
|
|
22
|
+
Requires-Dist: jinja2 (>=3.1.4,<4.0.0)
|
|
23
|
+
Requires-Dist: jupyter-bokeh (>=4.0.5,<5.0.0)
|
|
24
|
+
Requires-Dist: matplotlib (>=3.9.2,<4.0.0)
|
|
25
|
+
Requires-Dist: numpy (>=2.1.2,<3.0.0)
|
|
26
|
+
Requires-Dist: openpyxl (>=3.1.5,<4.0.0)
|
|
27
|
+
Requires-Dist: pandas (>=2.2.3,<3.0.0)
|
|
28
|
+
Requires-Dist: rich (>=13.9.2,<14.0.0)
|
|
29
|
+
Requires-Dist: scipy (>=1.14.1,<2.0.0)
|
|
30
|
+
Requires-Dist: seaborn (>=0.13.2,<0.14.0)
|
|
31
|
+
Requires-Dist: timple (>=0.1.8,<0.2.0)
|
|
32
|
+
Requires-Dist: tqdm (>=4.66.5,<5.0.0)
|
|
33
|
+
Project-URL: Repository, https://github.com/bnext-bio/nucleus/
|
|
34
|
+
Description-Content-Type: text/markdown
|
|
35
|
+
|
|
36
|
+
# CDK: Nucleus Cell Developer Kit
|
|
37
|
+
The Nucleus Cell Developer Kit (CDK) is a set of tools and libraries for building synthetic cells, the stack for synthetic cell engineering. This package is the core library, modular code that can be used in specific experiments or in other applications.
|
|
38
|
+
|
|
39
|
+
Nucleus is an open-source project for building and working with synthetic cells. For more information, please check out the [Nucleus documentation](https://nucleus.bnext.bio/).
|
|
40
|
+
|
|
41
|
+
## Features
|
|
42
|
+
Currently, the CDK contains our core analysis functionality: plate reader and liposome analysis. In particular, the plate reader code is designed to allow for easy analysis of kinetic timeseries of PURE experiments from Agilent/Biotek and Revvity Envision plate readers. The liposome analysis code is under heavy development, and the code included here is somewhat out of date---please contact us for more information.
|
|
43
|
+
|
|
44
|
+
## Installation
|
|
45
|
+
`pip install nucleus-cdk`
|
|
46
|
+
|
|
47
|
+
## Development
|
|
48
|
+
### Install poetry
|
|
49
|
+
The CDK uses poetry for dependency control and packaging. Install poetry, and activate it to download the dependencies. You can use poetry to manage the development virtual environment (recommended), or create a new conda environment to develop in.
|
|
50
|
+
|
|
51
|
+
#### Linux
|
|
52
|
+
To install poetry on Linux, you can use the following command:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
curl -sSL https://install.python-poetry.org | python3 -
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
#### Mac
|
|
59
|
+
*(untested)* Install poetry using homebrew:
|
|
60
|
+
```bash
|
|
61
|
+
brew install poetry
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### Activate poetry and download dependencies
|
|
65
|
+
```bash
|
|
66
|
+
poetry install
|
|
67
|
+
```
|
|
68
|
+
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# CDK: Nucleus Cell Developer Kit
|
|
2
|
+
The Nucleus Cell Developer Kit (CDK) is a set of tools and libraries for building synthetic cells, the stack for synthetic cell engineering. This package is the core library, modular code that can be used in specific experiments or in other applications.
|
|
3
|
+
|
|
4
|
+
Nucleus is an open-source project for building and working with synthetic cells. For more information, please check out the [Nucleus documentation](https://nucleus.bnext.bio/).
|
|
5
|
+
|
|
6
|
+
## Features
|
|
7
|
+
Currently, the CDK contains our core analysis functionality: plate reader and liposome analysis. In particular, the plate reader code is designed to allow for easy analysis of kinetic timeseries of PURE experiments from Agilent/Biotek and Revvity Envision plate readers. The liposome analysis code is under heavy development, and the code included here is somewhat out of date---please contact us for more information.
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
`pip install nucleus-cdk`
|
|
11
|
+
|
|
12
|
+
## Development
|
|
13
|
+
### Install poetry
|
|
14
|
+
The CDK uses poetry for dependency control and packaging. Install poetry, and activate it to download the dependencies. You can use poetry to manage the development virtual environment (recommended), or create a new conda environment to develop in.
|
|
15
|
+
|
|
16
|
+
#### Linux
|
|
17
|
+
To install poetry on Linux, you can use the following command:
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
curl -sSL https://install.python-poetry.org | python3 -
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
#### Mac
|
|
24
|
+
*(untested)* Install poetry using homebrew:
|
|
25
|
+
```bash
|
|
26
|
+
brew install poetry
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
### Activate poetry and download dependencies
|
|
30
|
+
```bash
|
|
31
|
+
poetry install
|
|
32
|
+
```
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
[tool.poetry]
|
|
2
|
+
name = "nucleus-cdk"
|
|
3
|
+
version = "0.3.0a1"
|
|
4
|
+
description = "Cell Developer Kit for building synthetic cells and cytosols."
|
|
5
|
+
license = "MIT"
|
|
6
|
+
authors = [
|
|
7
|
+
"Anton Jackson-Smith <anton@bnext.bio>",
|
|
8
|
+
]
|
|
9
|
+
maintainers = [
|
|
10
|
+
"Anton Jackson-Smith <anton@bnext.bio>",
|
|
11
|
+
]
|
|
12
|
+
readme = "README.md"
|
|
13
|
+
homepage = "https://nucleus.bnext.bio/"
|
|
14
|
+
repository = "https://github.com/bnext-bio/nucleus/"
|
|
15
|
+
packages = [
|
|
16
|
+
{ include = "cdk", from = "src" }
|
|
17
|
+
]
|
|
18
|
+
classifiers = [
|
|
19
|
+
"Development Status :: 3 - Alpha",
|
|
20
|
+
"Intended Audience :: Science/Research",
|
|
21
|
+
"License :: OSI Approved :: MIT License",
|
|
22
|
+
"Programming Language :: Python",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
24
|
+
]
|
|
25
|
+
keywords = [
|
|
26
|
+
"synthetic biology",
|
|
27
|
+
"synthetic cell",
|
|
28
|
+
"syncell",
|
|
29
|
+
"cellfree",
|
|
30
|
+
"nucleus",
|
|
31
|
+
"cell developer kit"
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[tool.poetry.dependencies]
|
|
35
|
+
python = "^3.12"
|
|
36
|
+
numpy = "^2.1.2"
|
|
37
|
+
pandas = "^2.2.3"
|
|
38
|
+
seaborn = "^0.13.2"
|
|
39
|
+
matplotlib = "^3.9.2"
|
|
40
|
+
tqdm = "^4.66.5"
|
|
41
|
+
rich = "^13.9.2"
|
|
42
|
+
openpyxl = "^3.1.5"
|
|
43
|
+
timple = "^0.1.8"
|
|
44
|
+
jinja2 = "^3.1.4"
|
|
45
|
+
scipy = "^1.14.1"
|
|
46
|
+
hvplot = "^0.11.1"
|
|
47
|
+
jupyter-bokeh = "^4.0.5"
|
|
48
|
+
|
|
49
|
+
[tool.poetry.group.dev.dependencies]
|
|
50
|
+
pytest = "^8.3.3"
|
|
51
|
+
flake8 = "^7.1.1"
|
|
52
|
+
black = "^24.10.0"
|
|
53
|
+
pre-commit = "^4.0.0"
|
|
54
|
+
ipykernel = "^6.29.5"
|
|
55
|
+
toml-cli = "^0.7.0"
|
|
56
|
+
flake8-pyproject = "^1.2.3"
|
|
57
|
+
|
|
58
|
+
[tool.flake8]
|
|
59
|
+
max-line-length = 120
|
|
60
|
+
extend-ignore = ["E203", "E501", "W503", ]
|
|
61
|
+
per-file-ignores = [
|
|
62
|
+
"__init__.py:F401"
|
|
63
|
+
]
|
|
64
|
+
|
|
65
|
+
[tool.black]
|
|
66
|
+
line-length = 79
|
|
67
|
+
target-version = ['py311']
|
|
68
|
+
include = '\.pyi?$'
|
|
69
|
+
extend-exclude = '''
|
|
70
|
+
/(
|
|
71
|
+
# directories
|
|
72
|
+
\.eggs
|
|
73
|
+
| \.git
|
|
74
|
+
| \.hg
|
|
75
|
+
| \.mypy_cache
|
|
76
|
+
| \.tox
|
|
77
|
+
| \.venv
|
|
78
|
+
| build
|
|
79
|
+
| dist
|
|
80
|
+
)/
|
|
81
|
+
'''
|
|
82
|
+
|
|
83
|
+
[tool.pytest.ini_options]
|
|
84
|
+
minversion = "6.0"
|
|
85
|
+
addopts = "-ra"
|
|
86
|
+
testpaths = [
|
|
87
|
+
"tests",
|
|
88
|
+
]
|
|
89
|
+
|
|
90
|
+
[build-system]
|
|
91
|
+
requires = ["poetry-core"]
|
|
92
|
+
build-backend = "poetry.core.masonry.api"
|
|
File without changes
|
|
@@ -0,0 +1,844 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Plate Reader Module
|
|
3
|
+
|
|
4
|
+
This module provides support for loading and analyzing data from various plate readers. Currently supported are:
|
|
5
|
+
+ BioTek Cytation 5
|
|
6
|
+
+ Revvity Envision Nexus
|
|
7
|
+
+ Promega Glomax Discover
|
|
8
|
+
|
|
9
|
+
Usage:
|
|
10
|
+
*tbd*
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import io
|
|
14
|
+
import re
|
|
15
|
+
import os.path
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
import logging
|
|
18
|
+
from typing import Union, Optional, NamedTuple
|
|
19
|
+
from enum import Enum, auto
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
import pandas as pd
|
|
23
|
+
|
|
24
|
+
# import matplotlib.pyplot as plt
|
|
25
|
+
import matplotlib as mpl
|
|
26
|
+
import seaborn as sns
|
|
27
|
+
import timple
|
|
28
|
+
import timple.timedelta
|
|
29
|
+
import scipy.optimize
|
|
30
|
+
import functools
|
|
31
|
+
import warnings
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
log = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
DataFile = Union[str, Path, io.StringIO]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class SteadyStateMethod(Enum):
|
|
40
|
+
LOWEST_VELOCITY = (auto(),)
|
|
41
|
+
MAXIMUM_VALUE = (auto(),)
|
|
42
|
+
VELOCITY_INTERCEPT = auto()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class PlateReaderData(NamedTuple):
|
|
46
|
+
"""
|
|
47
|
+
A named tuple representing plate reader data and associated plate map information.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
data: pd.DataFrame
|
|
51
|
+
platemap: Optional[pd.DataFrame] = None
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
_timple = timple.Timple()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def load_platereader_data(
|
|
58
|
+
data_file: DataFile,
|
|
59
|
+
platemap_file: Optional[DataFile] = None,
|
|
60
|
+
platereader: Optional[str] = None,
|
|
61
|
+
) -> Union[PlateReaderData, pd.DataFrame]:
|
|
62
|
+
"""
|
|
63
|
+
Load plate reader data from a file and return a DataFrame.
|
|
64
|
+
|
|
65
|
+
This function loads platereader data from a CSV file, parsing it into a standardized format and labelling
|
|
66
|
+
it with a provided plate map.
|
|
67
|
+
|
|
68
|
+
Filenames should be formatted in a standard format: `[date]-[device]-[experiment].csv`. For
|
|
69
|
+
example, `20241004-envision-dna-concentration.csv`.
|
|
70
|
+
|
|
71
|
+
Data is loaded based on the device field in the filename, which is used to determine the appropriate reader-specific
|
|
72
|
+
data parser. Currently supported readers are:
|
|
73
|
+
- BioTek Cytation 5: `cytation`
|
|
74
|
+
- Revvity Envision Nexus: `envision`
|
|
75
|
+
|
|
76
|
+
Data is returned as a pandas DataFrame with the following mandatory columns:
|
|
77
|
+
- `Well`: Well identifier (e.g. `A1`)
|
|
78
|
+
- `Row`: Row identifier (e.g. `A`)
|
|
79
|
+
- `Column`: Column identifier (e.g. `1`)
|
|
80
|
+
- `Time`: Time of measurement
|
|
81
|
+
- `Seconds`: Time of measurement in seconds
|
|
82
|
+
- `Temperature (C)`: Temperature at time of measurement
|
|
83
|
+
- `Read`: A tag describing the type of measurement (e.g. `OD600`, `Fluorescence`). The format of this field is
|
|
84
|
+
currently device-specific.
|
|
85
|
+
- `Data`: The measured data value
|
|
86
|
+
|
|
87
|
+
In addition, the provided platemap will be merged to the loaded data on the `Well` column. All other columns within
|
|
88
|
+
the platemap will be present in the returned dataframe.
|
|
89
|
+
|
|
90
|
+
Args:
|
|
91
|
+
data_file (str): Path to the plate reader data file.
|
|
92
|
+
|
|
93
|
+
Returns:
|
|
94
|
+
If a platemap is provided, a PlateReaderData named tuple containing the data and platemap DataFrames. Otherwise,
|
|
95
|
+
just the data. If a platemap_file is provided, the returned platemap is guaranteed to be not None.
|
|
96
|
+
|
|
97
|
+
platemap_file is not None:
|
|
98
|
+
PlateReaderData: A named tuple containing the plate reader data and platemap DataFrames: (data, platemap)
|
|
99
|
+
platemap_file is None:
|
|
100
|
+
pd.DataFrame: DataFrame containing the plate reader data in a structured format.
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
"""
|
|
104
|
+
if platereader is None:
|
|
105
|
+
platereader = os.path.basename(data_file).lower()
|
|
106
|
+
|
|
107
|
+
# TODO: Clean this up to use a proper platereader enum and not janky string parsing.
|
|
108
|
+
if "cytation" in platereader.lower():
|
|
109
|
+
data = read_cytation(data_file)
|
|
110
|
+
elif "biotek" in platereader.lower():
|
|
111
|
+
data = read_cytation(data_file)
|
|
112
|
+
elif "envision" in platereader.lower():
|
|
113
|
+
data = read_envision(data_file)
|
|
114
|
+
# elif filename_lower.startswith("glomax"):
|
|
115
|
+
# return read_glomax(os.path.dirname(data_file))
|
|
116
|
+
else:
|
|
117
|
+
raise ValueError(f"Unsupported plate reader data file: {data_file}")
|
|
118
|
+
|
|
119
|
+
platemap = None
|
|
120
|
+
if platemap_file is not None:
|
|
121
|
+
platemap = read_platemap(platemap_file)
|
|
122
|
+
data = data.merge(platemap, on="Well")
|
|
123
|
+
return PlateReaderData(data=data, platemap=platemap)
|
|
124
|
+
|
|
125
|
+
return data
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def read_platemap(platemap_file: DataFile) -> pd.DataFrame:
|
|
129
|
+
if isinstance(platemap_file, io.StringIO):
|
|
130
|
+
platemap = pd.read_csv(platemap_file)
|
|
131
|
+
else:
|
|
132
|
+
extension = os.path.splitext(platemap_file)[1].lower()
|
|
133
|
+
if extension == ".csv":
|
|
134
|
+
platemap = pd.read_csv(platemap_file)
|
|
135
|
+
elif extension == ".tsv":
|
|
136
|
+
platemap = pd.read_table(platemap_file)
|
|
137
|
+
# TODO: create test for this
|
|
138
|
+
elif extension == ".xlsx":
|
|
139
|
+
platemap = pd.read_excel(platemap_file)
|
|
140
|
+
else:
|
|
141
|
+
raise ValueError(
|
|
142
|
+
f"Unsupported platemap file, use csv or xlsx: {platemap_file}"
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
# Remove unnamed columns from the plate map.
|
|
146
|
+
platemap = platemap[
|
|
147
|
+
[col for col in platemap.columns if not col.startswith("Unnamed:")]
|
|
148
|
+
]
|
|
149
|
+
|
|
150
|
+
# Needed to make sure times are correctly converted, but we don't convert
|
|
151
|
+
# floats because they get upcast to a pandas Float64Dtype() class which
|
|
152
|
+
# messes up plotting.
|
|
153
|
+
# platemap = platemap.convert_dtypes(convert_floating=False)
|
|
154
|
+
|
|
155
|
+
platemap["Well"] = platemap["Well"].str.replace(
|
|
156
|
+
":", ""
|
|
157
|
+
) # Normalize well by removing : if it exists
|
|
158
|
+
return platemap
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
# def read_glomax(data_dir: str) -> pd.DataFrame:
|
|
162
|
+
# # glob over .csv files in dfpath; append to data; concatenate into one DataFrame
|
|
163
|
+
# data = list()
|
|
164
|
+
# for csv in glob.glob(f"{data_dir}/*.csv"):
|
|
165
|
+
# df = pd.read_csv(csv)
|
|
166
|
+
# df["File"] = os.path.basename(csv)
|
|
167
|
+
# df["Row"] = df["WellPosition"].str.split(":").str[0]
|
|
168
|
+
# df["Column"] = df["WellPosition"].str.split(":").str[1].astype(int)
|
|
169
|
+
# df["Time"] = pd.to_datetime(
|
|
170
|
+
# data["File"].str.replace(r".* ([0-9.]+ [0-9_]+).*", r"\1", regex=True), format="%Y.%m.%d %H_%M_%S"
|
|
171
|
+
# )
|
|
172
|
+
# df["WellTime"] = pd.to_timedelta(data["Timestamp(ms)"], "us")
|
|
173
|
+
|
|
174
|
+
# data.append(df)
|
|
175
|
+
|
|
176
|
+
# data = pd.concat(data, ignore_index=True)
|
|
177
|
+
|
|
178
|
+
# # label different wavelengths
|
|
179
|
+
# channel_map = dict(zip(data["ID"].unique(), ["A600", "Blue", "Green", "Red"]))
|
|
180
|
+
# data["Channel"] = data["ID"].map(channel_map)
|
|
181
|
+
|
|
182
|
+
# palette = dict(zip(dict.fromkeys(channel_map.values()), ["brown", "limegreen", "red", "firebrick"]))
|
|
183
|
+
|
|
184
|
+
# # massage Time
|
|
185
|
+
# data["TimeDelta"] = data["Time"] - data["Time"].min()
|
|
186
|
+
# data["TimeDeltaPretty"] = data["TimeDelta"].map(
|
|
187
|
+
# lambda x: "{:02d}:00".format(x.components.hours)
|
|
188
|
+
# ) # {:02d}".format(x.components.hours, x.components.minutes))
|
|
189
|
+
|
|
190
|
+
# # Get a generic data column
|
|
191
|
+
# data["Data"] = data["CalculatedFlux"]
|
|
192
|
+
# data["Data"].fillna(data[data["CalculatedFlux"].isna()]["OpticalDensity"], inplace=True)
|
|
193
|
+
|
|
194
|
+
# # Label replicates
|
|
195
|
+
# data["Replicate"] = data["File"].map(lambda x: re.sub(r".* OUT ([0-9]+).csv", r"\1", x))
|
|
196
|
+
# data.sort_values(by=["TimeDelta", "Row", "Column"], inplace=True)
|
|
197
|
+
|
|
198
|
+
# return data
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def read_cytation(data_file: DataFile, sep="\t") -> pd.DataFrame:
|
|
202
|
+
log.debug(f"Reading Cytation data from {data_file}")
|
|
203
|
+
# read data file as long string
|
|
204
|
+
data = ""
|
|
205
|
+
with open(data_file, "r", encoding="latin1") as file:
|
|
206
|
+
data = file.read()
|
|
207
|
+
|
|
208
|
+
# extract indices for Proc Details, Layout
|
|
209
|
+
procidx = re.search(r"Procedure Details", data)
|
|
210
|
+
layoutidx = re.search(r"Layout", data)
|
|
211
|
+
readidx = re.search(r"^(Read\s)?\d+(/\d+)?,\d+(/\d+)?", data, re.MULTILINE)
|
|
212
|
+
|
|
213
|
+
# get header DataFrame
|
|
214
|
+
header = data[: procidx.start()]
|
|
215
|
+
header = pd.read_csv(
|
|
216
|
+
io.StringIO(header), delimiter=sep, header=0, names=["key", "value"]
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
# get procedure DataFrame
|
|
220
|
+
procedure = data[procidx.end() : layoutidx.start()]
|
|
221
|
+
procedure = pd.read_csv(
|
|
222
|
+
io.StringIO(procedure), skipinitialspace=True, names=range(4)
|
|
223
|
+
)
|
|
224
|
+
procedure = procedure.replace(np.nan, "")
|
|
225
|
+
|
|
226
|
+
# get Cytation plate map from data_file as DataFrame
|
|
227
|
+
layout = data[layoutidx.end() : readidx.start()]
|
|
228
|
+
layout = pd.read_csv(io.StringIO(layout), index_col=False)
|
|
229
|
+
layout = layout.set_index(layout.columns[0])
|
|
230
|
+
layout.index.name = "Row"
|
|
231
|
+
|
|
232
|
+
# iterate over data string to find individual reads
|
|
233
|
+
reads = dict()
|
|
234
|
+
|
|
235
|
+
sep = (
|
|
236
|
+
r"(?:Read\s\d+:)?(?:\s\d{3}(?:/\d+)?,\d{3}(?:/\d+)?(?:\[\d\])?)?" + sep
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
for readidx in re.finditer(
|
|
240
|
+
r"^(Read\s)?\d+(/\d+)?,\d+(/\d+)?.*\n", data, re.MULTILINE
|
|
241
|
+
):
|
|
242
|
+
# for each iteration, extract string from start idx to end icx
|
|
243
|
+
read = data[readidx.end() :]
|
|
244
|
+
read = read[
|
|
245
|
+
: re.search(
|
|
246
|
+
r"(^(Read\s)?\d+,\d+|^Blank Read\s\d|Results|\Z)",
|
|
247
|
+
read[1:],
|
|
248
|
+
re.MULTILINE,
|
|
249
|
+
).start()
|
|
250
|
+
]
|
|
251
|
+
read = pd.read_csv(
|
|
252
|
+
io.StringIO(read), sep=sep, engine="python"
|
|
253
|
+
).convert_dtypes(convert_floating=False)
|
|
254
|
+
reads[data[readidx.start() : readidx.end()].strip()] = read
|
|
255
|
+
|
|
256
|
+
# create a DataFrame for each read and process, then concatenate into a large DataFrame
|
|
257
|
+
# NOTE: JC 2024-05-21 - turns out, len(list(reads.items())) = 1 (one big mono table)
|
|
258
|
+
read_dataframes = list()
|
|
259
|
+
for name, r in reads.items():
|
|
260
|
+
# filter out Cytation calculated kinetic parameters, which are cool, but don't want rn
|
|
261
|
+
r = r[r.Time.str.contains(r"\d:\d{2}:\d{2}", regex=True)]
|
|
262
|
+
|
|
263
|
+
# extract meaningful parameters from really big string
|
|
264
|
+
r = r.melt(id_vars=["Time", "T°"], var_name="Well", value_name="Data")
|
|
265
|
+
r["Row"] = r["Well"].str.extract(r"([A-Z]+)")
|
|
266
|
+
r["Column"] = r["Well"].str.extract(r"(\d+)").astype(int)
|
|
267
|
+
r["Temperature (C)"] = r["T°"] # .str.extract(r"(\d+)").astype(float)
|
|
268
|
+
r["Data"] = r["Data"].replace("OVRFLW", np.inf)
|
|
269
|
+
r["Data"] = r["Data"].astype(float)
|
|
270
|
+
r["Read"] = name
|
|
271
|
+
r["Ex"] = r["Read"].str.extract(r"(\d+),\d+").astype(int)
|
|
272
|
+
r["Em"] = r["Read"].str.extract(r"\d+,(\d+)").astype(int)
|
|
273
|
+
read_dataframes.append(r)
|
|
274
|
+
|
|
275
|
+
data = pd.concat(read_dataframes)
|
|
276
|
+
|
|
277
|
+
# add time column to data DataFrame
|
|
278
|
+
data["Time"] = pd.to_timedelta(data["Time"])
|
|
279
|
+
data["Seconds"] = data["Time"].map(lambda x: x.total_seconds())
|
|
280
|
+
|
|
281
|
+
return data[
|
|
282
|
+
[
|
|
283
|
+
"Well",
|
|
284
|
+
"Row",
|
|
285
|
+
"Column",
|
|
286
|
+
"Time",
|
|
287
|
+
"Seconds",
|
|
288
|
+
"Temperature (C)",
|
|
289
|
+
"Read",
|
|
290
|
+
"Data",
|
|
291
|
+
]
|
|
292
|
+
]
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def read_envision(data_file: DataFile) -> pd.DataFrame:
|
|
296
|
+
# load data
|
|
297
|
+
data = pd.read_csv(data_file).convert_dtypes()
|
|
298
|
+
|
|
299
|
+
# massage Row, Column, and Well information
|
|
300
|
+
data["Row"] = (
|
|
301
|
+
data["Well ID"].apply(lambda s: s[0]).astype(pd.StringDtype())
|
|
302
|
+
)
|
|
303
|
+
data["Column"] = data["Well ID"].apply(lambda s: str(int(s[1:])))
|
|
304
|
+
data["Well"] = data.apply(
|
|
305
|
+
lambda well: f"{well['Row']}{well['Column']}", axis=1
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
data["Time"] = pd.to_timedelta(data["Time [hhh:mm:ss.sss]"])
|
|
309
|
+
data["Seconds"] = data["Time"].map(lambda x: x.total_seconds())
|
|
310
|
+
|
|
311
|
+
data["Temperature (C)"] = data["Temperature current[°C]"]
|
|
312
|
+
|
|
313
|
+
data["Read"] = data["Operation"]
|
|
314
|
+
|
|
315
|
+
data["Data"] = data["Result Channel 1"]
|
|
316
|
+
|
|
317
|
+
data["Excitation (nm)"] = data["Exc WL[nm]"]
|
|
318
|
+
data["Emission (nm)"] = data["Ems WL Channel 1[nm]"]
|
|
319
|
+
data["Wavelength (nm)"] = (
|
|
320
|
+
data["Excitation (nm)"] + "," + data["Emission (nm)"]
|
|
321
|
+
)
|
|
322
|
+
|
|
323
|
+
return data[
|
|
324
|
+
[
|
|
325
|
+
"Well",
|
|
326
|
+
"Row",
|
|
327
|
+
"Column",
|
|
328
|
+
"Time",
|
|
329
|
+
"Seconds",
|
|
330
|
+
"Temperature (C)",
|
|
331
|
+
"Read",
|
|
332
|
+
"Data",
|
|
333
|
+
]
|
|
334
|
+
]
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def blank_data(data: pd.DataFrame, blank_type="Blank"):
|
|
338
|
+
"""
|
|
339
|
+
Blank data from plate reader measurements.
|
|
340
|
+
|
|
341
|
+
Adjusts plate reader data by subtracting the value of one or more blanks at each timepoint, for each read channel.
|
|
342
|
+
By default, the data will be blanked against the mean value of all wells of type "Blank".
|
|
343
|
+
|
|
344
|
+
This function adjusts the main "Data" column in the dataframe provided, so that blanked values can be easily
|
|
345
|
+
used in subsequent processing. The original (unblanked) data is available in a new 'Data_unblanked' column. The
|
|
346
|
+
blank value calculated for each row of the data is present in 'Data_blank'.
|
|
347
|
+
|
|
348
|
+
Args:
|
|
349
|
+
data (pd.DataFrame): Input DataFrame containing 'Well', 'Time', 'Data', and 'Type' columns.
|
|
350
|
+
blank_type (str, optional): Value in the 'Type' column to use as blank. Defaults to "Blank".
|
|
351
|
+
|
|
352
|
+
Returns:
|
|
353
|
+
pd.DataFrame: DataFrame with blanked 'Data' values and an additional 'Data_unblanked' column.
|
|
354
|
+
|
|
355
|
+
"""
|
|
356
|
+
blank = (
|
|
357
|
+
data[data["Type"] == blank_type]
|
|
358
|
+
.groupby(["Time", "Read"])["Data"]
|
|
359
|
+
.mean()
|
|
360
|
+
)
|
|
361
|
+
data = data.merge(
|
|
362
|
+
blank, on=["Time", "Read"], suffixes=("", "_blank"), how="left"
|
|
363
|
+
)
|
|
364
|
+
|
|
365
|
+
# Check to make sure we don't have missing blanks for certain Time/Read combinations in the source data.
|
|
366
|
+
# The most likely way this could happen is if the platereader "Time" isn't aligned well-to-well.
|
|
367
|
+
if data["Data_blank"].isna().any():
|
|
368
|
+
log.warning(
|
|
369
|
+
"Not all data has a blank value; blanked data will contain NaNs."
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
data["Data_unblanked"] = data["Data"].copy()
|
|
373
|
+
data["Data"] = data["Data"] - data["Data_blank"]
|
|
374
|
+
|
|
375
|
+
return data
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def plot_setup() -> None:
|
|
379
|
+
_timple.enable()
|
|
380
|
+
pd.set_option("display.float_format", "{:.2f}".format)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _plot_timedelta(plot: sns.FacetGrid | mpl.axes.Axes) -> sns.FacetGrid:
|
|
384
|
+
axes = [plot]
|
|
385
|
+
if isinstance(plot, sns.FacetGrid):
|
|
386
|
+
axes = plot.axes.flatten()
|
|
387
|
+
|
|
388
|
+
for ax in axes:
|
|
389
|
+
# ax.xaxis.set_major_locator(timple.timedelta.AutoTimedeltaLocator(minticks=3))
|
|
390
|
+
ax.xaxis.set_major_formatter(
|
|
391
|
+
timple.timedelta.TimedeltaFormatter("%h:%m")
|
|
392
|
+
)
|
|
393
|
+
ax.set_xlabel("Time (hours)")
|
|
394
|
+
|
|
395
|
+
# g.set_xlabels("Time (hours)")
|
|
396
|
+
# g.figure.autofmt_xdate()
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def plot_plate(data: pd.DataFrame) -> sns.FacetGrid:
|
|
400
|
+
g = sns.relplot(
|
|
401
|
+
data=data, x="Time", y="Data", row="Row", col="Column", kind="line"
|
|
402
|
+
)
|
|
403
|
+
_plot_timedelta(g)
|
|
404
|
+
|
|
405
|
+
g.set_ylabels("Fluorescence (RFU)")
|
|
406
|
+
g.set_titles("{row_name}{col_name}")
|
|
407
|
+
|
|
408
|
+
return g
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def plot_curves_by_name(
|
|
412
|
+
data: pd.DataFrame, by_experiment=True
|
|
413
|
+
) -> sns.FacetGrid:
|
|
414
|
+
"""
|
|
415
|
+
Produce a basic plot of timeseries curves, coloring curves by the `Name` of the sample.
|
|
416
|
+
|
|
417
|
+
If there are multiple different `Read`s in the data (e.g., GFP, RFP), then a subplot will be
|
|
418
|
+
produced for each read. If there are multiple experiments, each experiment will be plotted separately.
|
|
419
|
+
|
|
420
|
+
Args:
|
|
421
|
+
data (pd.DataFrame): DataFrame containing plate reader data.
|
|
422
|
+
by_experiment (bool, optional): If True, then each experiment will be plotted in a separate subplot.
|
|
423
|
+
|
|
424
|
+
Returns:
|
|
425
|
+
sns.FacetGrid: Seaborn FacetGrid object containing the plot.
|
|
426
|
+
"""
|
|
427
|
+
kwargs = {}
|
|
428
|
+
if "Experiment" in data.columns and by_experiment:
|
|
429
|
+
kwargs["col"] = "Experiment"
|
|
430
|
+
|
|
431
|
+
g = plot_curves(
|
|
432
|
+
data=data, x="Time", y="Data", hue="Name", row="Read", **kwargs
|
|
433
|
+
)
|
|
434
|
+
|
|
435
|
+
return g
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def plot_curves(
|
|
439
|
+
data: pd.DataFrame,
|
|
440
|
+
x="Time",
|
|
441
|
+
y="Data",
|
|
442
|
+
hue="Name",
|
|
443
|
+
labels=(None, "Fluorescence (RFU)"),
|
|
444
|
+
**kwargs,
|
|
445
|
+
) -> sns.FacetGrid:
|
|
446
|
+
"""
|
|
447
|
+
Plot timeseries curves from a plate reader dataset, allowing selection of the parameters to
|
|
448
|
+
use for plotting and to divide the data into multiple subplots.
|
|
449
|
+
|
|
450
|
+
This function is a thin wrapper around Seaborn `relplot`, providing sensible defaults while
|
|
451
|
+
also allowing for the use of any `relplot` parameter.
|
|
452
|
+
|
|
453
|
+
Args:
|
|
454
|
+
data (pd.DataFrame): DataFrame containing plate reader data.
|
|
455
|
+
x (str, optional): Column name to use for x-axis. Defaults to "Time".
|
|
456
|
+
y (str, optional): Column name to use for y-axis. Defaults to "Data".
|
|
457
|
+
hue (str, optional): Column name to use for color coding. Defaults to "Name".
|
|
458
|
+
labels (tuple, optional): Labels for the x and y axes. Defaults to (None, "Fluorescence (RFU)").
|
|
459
|
+
If None, use the default label (the name of the field, or a formatted time label).
|
|
460
|
+
**kwargs: Additional keyword arguments passed to `sns.relplot`.
|
|
461
|
+
|
|
462
|
+
Returns:
|
|
463
|
+
sns.FacetGrid: A FacetGrid object containing the plotted data.
|
|
464
|
+
|
|
465
|
+
"""
|
|
466
|
+
g = sns.relplot(data=data, x=x, y=y, hue=hue, kind="line", **kwargs)
|
|
467
|
+
_plot_timedelta(g)
|
|
468
|
+
|
|
469
|
+
x_label, y_label = labels
|
|
470
|
+
if x_label:
|
|
471
|
+
g.set_xlabels(x_label)
|
|
472
|
+
if y_label:
|
|
473
|
+
g.set_ylabels(y_label)
|
|
474
|
+
|
|
475
|
+
# Set simple row and column titles, if we're faceting on row or column.
|
|
476
|
+
# The join means the punctuation only gets added if we have both.
|
|
477
|
+
var_len = max(
|
|
478
|
+
[len(kwargs[var]) for var in ["row", "col"] if var in kwargs] + [0]
|
|
479
|
+
)
|
|
480
|
+
log.debug(f"{var_len=}")
|
|
481
|
+
row_title = (
|
|
482
|
+
f"{{row_var:>{var_len}}}: {{row_name}}" if "row" in kwargs else ""
|
|
483
|
+
)
|
|
484
|
+
col_title = (
|
|
485
|
+
f"{{col_var:>{var_len}}}: {{col_name}}" if "col" in kwargs else ""
|
|
486
|
+
)
|
|
487
|
+
g.set_titles("\n".join(filter(None, [row_title, col_title])))
|
|
488
|
+
|
|
489
|
+
return g
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
###
|
|
493
|
+
# Kinetics Analysis
|
|
494
|
+
# TODO: Perhaps split this out into a submodule.
|
|
495
|
+
###
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def find_steady_state_for_well(well):
|
|
499
|
+
well = well.sort_values("Time")
|
|
500
|
+
pct_change = well["Data"].rolling(window=3).mean().pct_change()
|
|
501
|
+
idx_maxV = pct_change.idxmax()
|
|
502
|
+
|
|
503
|
+
ss_idx = pct_change.loc[idx_maxV:].abs().idxmin()
|
|
504
|
+
ss_time = well.loc[ss_idx, "Time"]
|
|
505
|
+
ss_level = well.loc[ss_idx, "Data"]
|
|
506
|
+
|
|
507
|
+
return pd.Series(
|
|
508
|
+
{"Time_steadystate": ss_time, "Data_steadystate": ss_level}
|
|
509
|
+
)
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def find_steady_state(
|
|
513
|
+
data: pd.DataFrame, window_size=10, threshold=0.01
|
|
514
|
+
) -> pd.DataFrame:
|
|
515
|
+
"""
|
|
516
|
+
Find the steady state of the "Data" column in the provided data DataFrame.
|
|
517
|
+
|
|
518
|
+
Args:
|
|
519
|
+
data (pd.DataFrame): Input DataFrame containing 'Well', 'Time', and 'Data' columns.
|
|
520
|
+
window_size (int): Size of the rolling window for calculating the rate of change.
|
|
521
|
+
threshold (float): Threshold for determining steady state.
|
|
522
|
+
|
|
523
|
+
Returns:
|
|
524
|
+
pd.DataFrame: DataFrame with 'Well', 'SteadyStateTime', and 'SteadyStateLevel' columns.
|
|
525
|
+
"""
|
|
526
|
+
|
|
527
|
+
result = data.groupby(["Well", "Read"]).apply(find_steady_state_for_well)
|
|
528
|
+
return result
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def _sigmoid(x, L, k, x0):
|
|
532
|
+
return L / (1 + np.exp(-k * (x - x0)))
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def kinetic_analysis_per_well(
|
|
536
|
+
data: pd.DataFrame, data_column="Data"
|
|
537
|
+
) -> pd.DataFrame:
|
|
538
|
+
|
|
539
|
+
steadystate = find_steady_state_for_well(data)
|
|
540
|
+
|
|
541
|
+
data = data.loc[data["Time"] <= steadystate["Time_steadystate"]]
|
|
542
|
+
time = data["Time"].dt.total_seconds()
|
|
543
|
+
|
|
544
|
+
# make initial guesses for parameters
|
|
545
|
+
L_initial = np.max(data[data_column])
|
|
546
|
+
x0_initial = np.max(time) / 4
|
|
547
|
+
k_initial = (
|
|
548
|
+
np.log(L_initial * 1.1 / data[data_column] - 1) / (time - x0_initial)
|
|
549
|
+
).dropna().mean() * -1.0
|
|
550
|
+
p0 = [L_initial, k_initial, x0_initial]
|
|
551
|
+
|
|
552
|
+
# attempt fitting
|
|
553
|
+
params = [0, 0, 0]
|
|
554
|
+
with warnings.catch_warnings():
|
|
555
|
+
warnings.simplefilter("error", scipy.optimize.OptimizeWarning)
|
|
556
|
+
try:
|
|
557
|
+
params, _ = scipy.optimize.curve_fit(
|
|
558
|
+
_sigmoid, time, data[data_column], p0=p0
|
|
559
|
+
)
|
|
560
|
+
except scipy.optimize.OptimizeWarning as w:
|
|
561
|
+
log.debug(f"Scipy optimize warning: {w}")
|
|
562
|
+
except Exception as e:
|
|
563
|
+
log.warning(f"Failed to solve: {e}")
|
|
564
|
+
|
|
565
|
+
return None
|
|
566
|
+
log.debug(f"{data['Well'].iloc[0]} Fitted params: {params}")
|
|
567
|
+
|
|
568
|
+
# calculate velocities and velocity params
|
|
569
|
+
v = data[data_column].diff() / data["Time"].dt.total_seconds().diff()
|
|
570
|
+
|
|
571
|
+
maxV = v.max()
|
|
572
|
+
maxV_d = data.loc[v.idxmax(), data_column]
|
|
573
|
+
maxV_time = data.loc[v.idxmax(), "Time"]
|
|
574
|
+
|
|
575
|
+
# calculate lag time
|
|
576
|
+
lag = -maxV_d / maxV + maxV_time.total_seconds()
|
|
577
|
+
lag_data = _sigmoid(lag, *params)
|
|
578
|
+
|
|
579
|
+
# decile_upper = data[data_column].quantile(0.95)
|
|
580
|
+
# decile_lower = data[data_column].quantile(0.05)
|
|
581
|
+
|
|
582
|
+
# growth_s = (decile_upper - maxV_d) / maxV + maxV_time.total_seconds()
|
|
583
|
+
|
|
584
|
+
# ss_time = data.loc[(data[data_column] > decile_upper).idxmax(), "Time"]
|
|
585
|
+
# ss_d = data.loc[
|
|
586
|
+
# (data[data_column] > decile_upper).idxmax() :, data_column
|
|
587
|
+
# ].mean()
|
|
588
|
+
|
|
589
|
+
# kinetics = {
|
|
590
|
+
# # f"{data_column}_fit_d": y_fit,
|
|
591
|
+
# f"{data_column}_maxV": maxV,
|
|
592
|
+
# f"{data_column}_t_maxV": t_maxV,
|
|
593
|
+
# f"{data_column}_maxV_d": maxV_d,
|
|
594
|
+
# f"{data_column}_lag_s": lag,
|
|
595
|
+
# f"{data_column}_growth_s": growth_s,
|
|
596
|
+
# f"{data_column}_ss_s": ss_s,
|
|
597
|
+
# f"{data_column}_ss_d": ss_d,
|
|
598
|
+
# f"{data_column}_low_d": decile_lower,
|
|
599
|
+
# f"{data_column}_high_d": decile_upper,
|
|
600
|
+
# }
|
|
601
|
+
|
|
602
|
+
kinetics = {
|
|
603
|
+
# f"{data_column}_fit_d": y_fit,
|
|
604
|
+
("Velocity", "Time"): maxV_time,
|
|
605
|
+
("Velocity", data_column): maxV_d,
|
|
606
|
+
("Velocity", "Max"): maxV,
|
|
607
|
+
("Lag", "Time"): pd.to_timedelta(lag, unit="s"),
|
|
608
|
+
("Lag", "Data"): lag_data,
|
|
609
|
+
# f"{data_column}_growth_s": growth_s,
|
|
610
|
+
("Steady State", "Time"): steadystate["Time_steadystate"],
|
|
611
|
+
("Steady State", data_column): steadystate["Data_steadystate"],
|
|
612
|
+
("Fit", "L"): params[0],
|
|
613
|
+
("Fit", "k"): params[1],
|
|
614
|
+
("Fit", "x0"): params[2],
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
return pd.Series(kinetics)
|
|
618
|
+
# return kinetics
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
def kinetic_analysis(data: pd.DataFrame, data_column="Data") -> pd.DataFrame:
|
|
622
|
+
kinetics = data.groupby(["Well", "Name", "Read"], sort=False).apply(
|
|
623
|
+
functools.partial(kinetic_analysis_per_well, data_column=data_column)
|
|
624
|
+
)
|
|
625
|
+
return kinetics
|
|
626
|
+
|
|
627
|
+
|
|
628
|
+
def kinetic_analysis_summary(
|
|
629
|
+
data: pd.DataFrame,
|
|
630
|
+
data_column="Data",
|
|
631
|
+
time_cutoff: int = 12000,
|
|
632
|
+
label_order: list[str] = None,
|
|
633
|
+
norm_label: str = None,
|
|
634
|
+
):
|
|
635
|
+
def per_well_cleanup(df):
|
|
636
|
+
cols = df.columns
|
|
637
|
+
return df[["Well"] + list(cols[27:])].aggregate(lambda x: x.iloc[0])
|
|
638
|
+
|
|
639
|
+
tk = kinetic_analysis(
|
|
640
|
+
data=data, data_column=data_column, time_cutoff=time_cutoff
|
|
641
|
+
)
|
|
642
|
+
out = tk.groupby("Well").apply(per_well_cleanup).reset_index(drop=True)
|
|
643
|
+
|
|
644
|
+
if label_order:
|
|
645
|
+
out = out.set_index("Well").reindex(label_order).reset_index()
|
|
646
|
+
|
|
647
|
+
# normalize max value (calculated by kinetics) if norm_label given
|
|
648
|
+
if norm_label:
|
|
649
|
+
norm = out[out["Well"] == norm_label][f"{data_column}_high_d"].values
|
|
650
|
+
out["Normalized (%)"] = out[f"{data_column}_high_d"] / norm
|
|
651
|
+
|
|
652
|
+
return out
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
def plot_kinetics_by_well(
|
|
656
|
+
data: pd.DataFrame,
|
|
657
|
+
kinetics: pd.DataFrame,
|
|
658
|
+
x: str = "Time",
|
|
659
|
+
y: str = "Data",
|
|
660
|
+
show_fit: bool = False,
|
|
661
|
+
show_velocity: bool = False,
|
|
662
|
+
annotate: bool = False,
|
|
663
|
+
**kwargs,
|
|
664
|
+
):
|
|
665
|
+
"""
|
|
666
|
+
Typical usage:
|
|
667
|
+
|
|
668
|
+
> tk = kinetic_analysis(data=data, data_column="BackgroundSubtracted")
|
|
669
|
+
> g = sns.FacetGrid(tk, col="Well", col_wrap=2, sharey=False, height=4, aspect=1.5)
|
|
670
|
+
> g.map_dataframe(plot_kinetics, show_fit=True, show_velocity=True)
|
|
671
|
+
"""
|
|
672
|
+
colors = sns.color_palette("Set2")
|
|
673
|
+
|
|
674
|
+
ax = sns.scatterplot(data=data, x=x, y=y, color=colors[2], alpha=0.5)
|
|
675
|
+
|
|
676
|
+
well = data["Well"].iloc[0]
|
|
677
|
+
name = data["Name"].iloc[0]
|
|
678
|
+
read = data["Read"].iloc[0]
|
|
679
|
+
kinetics = kinetics.loc[well, name, read]
|
|
680
|
+
if (kinetics.isna()).any():
|
|
681
|
+
log.info(f"Kinetics information not available for {well}.")
|
|
682
|
+
return
|
|
683
|
+
|
|
684
|
+
# ax_ylim = (
|
|
685
|
+
# ax.get_ylim()
|
|
686
|
+
# ) # Use this to run lines to bounds later, then restore them before returning.
|
|
687
|
+
|
|
688
|
+
if show_fit:
|
|
689
|
+
L = kinetics["Fit", "L"]
|
|
690
|
+
k = kinetics["Fit", "k"]
|
|
691
|
+
x0 = kinetics["Fit", "x0"]
|
|
692
|
+
sns.lineplot(
|
|
693
|
+
x=data["Time"],
|
|
694
|
+
y=_sigmoid(data["Time"].dt.total_seconds(), L, k, x0),
|
|
695
|
+
linestyle="--",
|
|
696
|
+
color=colors[3],
|
|
697
|
+
# alpha=0.5,
|
|
698
|
+
ax=ax,
|
|
699
|
+
)
|
|
700
|
+
# sns.lineplot(data=data, x=x, y=y, linestyle="--", c="red", alpha=0.5)
|
|
701
|
+
|
|
702
|
+
# Max Velocity
|
|
703
|
+
# maxV_x = np.linspace(ax.get_xlim()[0], ax.get_xlim()[1], 100)
|
|
704
|
+
maxV_y = (
|
|
705
|
+
kinetics["Velocity", "Max"]
|
|
706
|
+
* (data["Time"] - kinetics["Velocity", "Time"]).dt.total_seconds()
|
|
707
|
+
+ kinetics["Velocity", "Data"]
|
|
708
|
+
)
|
|
709
|
+
|
|
710
|
+
sns.lineplot(
|
|
711
|
+
x=data["Time"].loc[(maxV_y > 0) & (maxV_y < data[y].max())],
|
|
712
|
+
y=maxV_y[(maxV_y > 0) & (maxV_y < data[y].max())],
|
|
713
|
+
linestyle="--",
|
|
714
|
+
color=colors[1],
|
|
715
|
+
ax=ax,
|
|
716
|
+
)
|
|
717
|
+
|
|
718
|
+
maxV = kinetics["Velocity", "Max"]
|
|
719
|
+
maxV_s = kinetics["Velocity", "Time"]
|
|
720
|
+
maxV_d = kinetics["Velocity", "Data"]
|
|
721
|
+
|
|
722
|
+
# decile_upper = summary[f"{y}_high_d"]
|
|
723
|
+
# decile_lower = summary[f"{y}_low_d"]
|
|
724
|
+
# ax.vlines(
|
|
725
|
+
# lag,
|
|
726
|
+
# ymin=ax_ylim[0],
|
|
727
|
+
# ymax=decile_lower,
|
|
728
|
+
# colors=colors[2],
|
|
729
|
+
# linestyle="--",
|
|
730
|
+
# )
|
|
731
|
+
|
|
732
|
+
# Time to Steady State
|
|
733
|
+
ss_s = kinetics["Steady State", "Time"]
|
|
734
|
+
ax.axvline(ss_s, c=colors[3], linestyle="--")
|
|
735
|
+
|
|
736
|
+
# # Range
|
|
737
|
+
# ax.axhline(decile_upper, c=colors[7], linestyle="--")
|
|
738
|
+
# ax.axhline(decile_lower, c=colors[7], linestyle="--")
|
|
739
|
+
|
|
740
|
+
if annotate:
|
|
741
|
+
# Plot the text annotations on the chart
|
|
742
|
+
ax.annotate(
|
|
743
|
+
f"$V_{{max}} =$ {maxV:.2f} u/s",
|
|
744
|
+
(maxV_s, maxV_d),
|
|
745
|
+
xytext=(24, 0),
|
|
746
|
+
textcoords="offset points",
|
|
747
|
+
arrowprops={"arrowstyle": "->"},
|
|
748
|
+
ha="left",
|
|
749
|
+
va="center",
|
|
750
|
+
c="black",
|
|
751
|
+
)
|
|
752
|
+
|
|
753
|
+
f = timple.timedelta.TimedeltaFormatter("%h:%m")
|
|
754
|
+
lag_label = f.format_data(
|
|
755
|
+
timple.timedelta.timedelta2num(kinetics["Lag", "Time"])
|
|
756
|
+
)
|
|
757
|
+
ax.annotate(
|
|
758
|
+
f"$t_{{lag}} =$ {lag_label}",
|
|
759
|
+
(kinetics["Lag", "Time"], kinetics["Lag", "Data"]),
|
|
760
|
+
xytext=(12, 0),
|
|
761
|
+
textcoords="offset points",
|
|
762
|
+
ha="left",
|
|
763
|
+
va="center",
|
|
764
|
+
)
|
|
765
|
+
|
|
766
|
+
ss_label = f.format_data(
|
|
767
|
+
timple.timedelta.timedelta2num(kinetics["Steady State", "Time"])
|
|
768
|
+
)
|
|
769
|
+
ax.annotate(
|
|
770
|
+
f"$t_{{steady state}} =$ {ss_label}",
|
|
771
|
+
(
|
|
772
|
+
kinetics["Steady State", "Time"],
|
|
773
|
+
kinetics["Steady State", "Data"],
|
|
774
|
+
),
|
|
775
|
+
xytext=(0, -12),
|
|
776
|
+
textcoords="offset points",
|
|
777
|
+
ha="left",
|
|
778
|
+
va="top",
|
|
779
|
+
)
|
|
780
|
+
|
|
781
|
+
# Velocity
|
|
782
|
+
if show_velocity:
|
|
783
|
+
# TODO: This is currently broken due to rolling calculation and its effect on bounds.
|
|
784
|
+
# Show a velocity sparkline over the plot
|
|
785
|
+
velocity = (
|
|
786
|
+
data.transform({y: "diff", x: lambda x: x}).rolling(5).mean()
|
|
787
|
+
)
|
|
788
|
+
velocity[y] = velocity[y]
|
|
789
|
+
# velocity_ax = ax.secondary_yaxis(location="right",
|
|
790
|
+
# functions=(lambda x: pd.Series(x).rolling(5).mean().values, lambda x: x))
|
|
791
|
+
velocity_ax = ax.twinx()
|
|
792
|
+
sns.lineplot(data=velocity, x=x, y=y, alpha=0.5, ax=velocity_ax)
|
|
793
|
+
velocity_ax.set_ylabel("$V (u/s)$")
|
|
794
|
+
velocity_ax.set_ylim((0, velocity[y].max() * 2))
|
|
795
|
+
|
|
796
|
+
# ax.set_ylim(ax_ylim)
|
|
797
|
+
|
|
798
|
+
_plot_timedelta(ax)
|
|
799
|
+
|
|
800
|
+
|
|
801
|
+
def plot_kinetics(data: pd.DataFrame, kinetics: pd.DataFrame, **kwargs):
|
|
802
|
+
g = sns.FacetGrid(
|
|
803
|
+
data, col="Name", col_wrap=3, sharey=True, height=4, aspect=1.5
|
|
804
|
+
)
|
|
805
|
+
g.map_dataframe(
|
|
806
|
+
plot_kinetics_by_well,
|
|
807
|
+
kinetics=kinetics,
|
|
808
|
+
show_fit=True,
|
|
809
|
+
show_velocity=False,
|
|
810
|
+
annotate=True,
|
|
811
|
+
)
|
|
812
|
+
g.set_ylabels("Fluorescence (RFU)")
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
def plot_steadystate(data: pd.DataFrame, **kwargs):
|
|
816
|
+
steady_state = find_steady_state(data).reset_index()
|
|
817
|
+
data_with_steady_state = data.merge(steady_state, on="Well", how="left")
|
|
818
|
+
|
|
819
|
+
exp = "Experiment" if "Experiment" in data.columns else None
|
|
820
|
+
col = exp if exp is not None else None
|
|
821
|
+
col_wrap = 2 if col is not None else None
|
|
822
|
+
|
|
823
|
+
g = sns.catplot(
|
|
824
|
+
data=data_with_steady_state,
|
|
825
|
+
x="Name",
|
|
826
|
+
y="Data_steadystate",
|
|
827
|
+
hue=exp,
|
|
828
|
+
kind="bar",
|
|
829
|
+
col=col,
|
|
830
|
+
col_wrap=col_wrap,
|
|
831
|
+
height=4,
|
|
832
|
+
aspect=1.5,
|
|
833
|
+
sharex=False,
|
|
834
|
+
**kwargs,
|
|
835
|
+
)
|
|
836
|
+
|
|
837
|
+
g.set_xticklabels(rotation=90)
|
|
838
|
+
g.set_ylabels("Steady State Fluorescence (RFU)")
|
|
839
|
+
return g
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
def export():
|
|
843
|
+
# TODO: Write me
|
|
844
|
+
pass
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import rich.console
|
|
3
|
+
import rich.logging
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def setup_logging(log_level=logging.INFO):
|
|
7
|
+
# Set up logging
|
|
8
|
+
console = rich.console.Console(
|
|
9
|
+
force_jupyter=False,
|
|
10
|
+
# stderr=True,
|
|
11
|
+
theme=rich.theme.Theme(
|
|
12
|
+
{"logging.level.debug": "cyan", "logging.level.info": "green"}
|
|
13
|
+
),
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
logging.basicConfig(
|
|
17
|
+
level=log_level,
|
|
18
|
+
format="%(message)s",
|
|
19
|
+
handlers=[rich.logging.RichHandler(
|
|
20
|
+
console=console,
|
|
21
|
+
enable_link_path=False)
|
|
22
|
+
],
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
log = logging.getLogger(__name__)
|
|
26
|
+
log.info("Logging initialized")
|
|
27
|
+
return log
|