bindcurve 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bindcurve/__init__.py +16 -0
- bindcurve/calculate.py +639 -0
- bindcurve/data.py +829 -0
- bindcurve/models.py +369 -0
- bindcurve/models_other.py +100 -0
- bindcurve-0.1.0.dist-info/METADATA +93 -0
- bindcurve-0.1.0.dist-info/RECORD +10 -0
- bindcurve-0.1.0.dist-info/WHEEL +5 -0
- bindcurve-0.1.0.dist-info/licenses/LICENSE +21 -0
- bindcurve-0.1.0.dist-info/top_level.txt +1 -0
bindcurve/__init__.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
from bindcurve.data import load_csv, load_df, plot, plot_grid, plot_asymptotes, plot_traces, plot_value, report
|
|
2
|
+
from bindcurve.calculate import fit_50, fit_Kd_direct, fit_Kd_competition, convert
|
|
3
|
+
from bindcurve.models import IC50, logIC50,dir_simple, dir_specific, dir_total, comp_3st_specific, comp_3st_total, comp_4st_specific, comp_4st_total, cheng_prusoff, cheng_prusoff_corr, coleska
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"load_csv", "load_df", "plot", "plot_grid", "plot_asymptotes", "plot_traces", "plot_value", "report",
|
|
8
|
+
"fit_50", "fit_Kd_direct", "fit_Kd_competition", "convert", "IC50", "logIC50", "dir_simple", "dir_specific", "dir_total",
|
|
9
|
+
"comp_3st_specific", "comp_3st_total", "comp_4st_specific", "comp_4st_total", "cheng_prusoff", "cheng_prusoff_corr", "coleska",
|
|
10
|
+
"__version__"
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
|
bindcurve/calculate.py
ADDED
|
@@ -0,0 +1,639 @@
|
|
|
1
|
+
import pandas as pd
|
|
2
|
+
import numpy as np
|
|
3
|
+
#import matplotlib.pyplot as plt
|
|
4
|
+
import lmfit
|
|
5
|
+
import traceback
|
|
6
|
+
from bindcurve import data
|
|
7
|
+
from bindcurve import models
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def generate_guess(df, saturation=False):
|
|
13
|
+
|
|
14
|
+
# Sorting the df
|
|
15
|
+
if saturation:
|
|
16
|
+
df = df.sort_values(by=['c'], ascending=True)
|
|
17
|
+
else:
|
|
18
|
+
df = df.sort_values(by=['c'], ascending=False)
|
|
19
|
+
|
|
20
|
+
# Defining important points on "median response" axis
|
|
21
|
+
ymin_guess = min(df["median"])
|
|
22
|
+
ymax_guess = max(df["median"])
|
|
23
|
+
y_middle = ymin_guess+(ymax_guess-ymin_guess)/2
|
|
24
|
+
|
|
25
|
+
# Interpolating to obtain guess for concentration axis
|
|
26
|
+
IC50_guess = np.interp(y_middle, df["median"], df["c"])
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# This is plotting just for development purposes
|
|
30
|
+
#y_curve = np.linspace(ymin_guess, ymax_guess, 1000)
|
|
31
|
+
#plt.plot(df["median"], df["c"], "o")
|
|
32
|
+
#plt.plot(y_curve, np.interp(y_curve, df["median"], df["c"]))
|
|
33
|
+
#plt.yscale("log")
|
|
34
|
+
#plt.show()
|
|
35
|
+
|
|
36
|
+
return ymin_guess, ymax_guess, IC50_guess
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def define_pars(model, ymin_guess, ymax_guess, IC50_guess, RT=None, LsT=None, Kds=None, Ns=False, N=False, fix_ymin=False, fix_ymax=False, fix_slope=False):
|
|
41
|
+
|
|
42
|
+
# Initiating Parameters class in lmfit
|
|
43
|
+
pars = lmfit.Parameters()
|
|
44
|
+
|
|
45
|
+
# Setting ymin and ymax
|
|
46
|
+
if not fix_ymin:
|
|
47
|
+
pars.add('ymin', value = ymin_guess)
|
|
48
|
+
else:
|
|
49
|
+
pars.add('ymin', value = fix_ymin, vary=False)
|
|
50
|
+
|
|
51
|
+
if not fix_ymax:
|
|
52
|
+
pars.add('ymax', value = ymax_guess)
|
|
53
|
+
else:
|
|
54
|
+
pars.add('ymax', value = fix_ymax, vary=False)
|
|
55
|
+
|
|
56
|
+
# Setting parameters for the logistic models
|
|
57
|
+
if model in models.get_list_of_models("logistic"):
|
|
58
|
+
|
|
59
|
+
if not fix_slope:
|
|
60
|
+
pars.add('slope', value = 0)
|
|
61
|
+
else:
|
|
62
|
+
pars.add('slope', value = fix_slope, vary=False)
|
|
63
|
+
|
|
64
|
+
if model == "IC50":
|
|
65
|
+
pars.add('IC50', value = IC50_guess, min = 0)
|
|
66
|
+
if model == "logIC50":
|
|
67
|
+
pars.add('logIC50', value = np.log10(IC50_guess))
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# Setting parameters for the direct binding Kd models
|
|
71
|
+
if model in models.get_list_of_models("Kd_direct"):
|
|
72
|
+
# Experimental constants
|
|
73
|
+
pars.add('LsT', value = LsT, vary=False)
|
|
74
|
+
# Parameters to be fitted
|
|
75
|
+
pars.add('Kds', value = IC50_guess/2, min = 0)
|
|
76
|
+
|
|
77
|
+
if model == "dir_total":
|
|
78
|
+
pars.add('Ns', value = Ns, vary=False)
|
|
79
|
+
if model == "dir_simple":
|
|
80
|
+
pars.add('Kds', value = IC50_guess/2, min = 0)
|
|
81
|
+
|
|
82
|
+
# Setting parameters for the competitive binding Kd models
|
|
83
|
+
if model in models.get_list_of_models("Kd_competition"):
|
|
84
|
+
# Experimental constants
|
|
85
|
+
pars.add('RT', value = RT, vary=False)
|
|
86
|
+
pars.add('LsT', value = LsT, vary=False)
|
|
87
|
+
pars.add('Kds', value = Kds, vary=False)
|
|
88
|
+
|
|
89
|
+
# Parameters to be fitted
|
|
90
|
+
pars.add('Kd', value = IC50_guess/2, min=0)
|
|
91
|
+
|
|
92
|
+
if model in ["comp_3st_total", "comp_4st_total"]:
|
|
93
|
+
pars.add('N', value = N, vary=False)
|
|
94
|
+
|
|
95
|
+
if model in ["comp_4st_specific", "comp_4st_total"]:
|
|
96
|
+
pars.add('Kd3', value = (IC50_guess/2)*10, min=0)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
return pars
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def fit_50(input_df, model, compound_sel = False, fix_ymin = False, fix_ymax = False, fix_slope = False, ci=True, verbose = False):
|
|
105
|
+
"""Function for fitting the `IC50` and `logIC50` models.
|
|
106
|
+
|
|
107
|
+
Parameters
|
|
108
|
+
----------
|
|
109
|
+
input_df : DataFrame
|
|
110
|
+
Pandas DataFrame containing the input data.
|
|
111
|
+
model : str
|
|
112
|
+
Name of the model. Options: `IC50`, `logIC50`
|
|
113
|
+
compound_sel : list
|
|
114
|
+
List of compounds to execute the function on. If set to False, all compounds will be used.
|
|
115
|
+
fix_ymin : float or int
|
|
116
|
+
Lower asymptote of the model will be fixed at the provided value. If set to "False", it will be fitted freely.
|
|
117
|
+
fix_ymax : float or int
|
|
118
|
+
Upper asymptote of the model will be fixed at the provided value. If set to "False", it will be fitted freely.
|
|
119
|
+
fix_slope : float or int
|
|
120
|
+
Slope of the model will be fixed at the provided value. If set to "False", it will be fitted freely.
|
|
121
|
+
ci : bool
|
|
122
|
+
Whether to calculate 95% confidence intervals.
|
|
123
|
+
verbose : bool
|
|
124
|
+
If set to "True", more detailed output is printed. Intended mainly for troubleshooting.
|
|
125
|
+
|
|
126
|
+
Returns
|
|
127
|
+
-------
|
|
128
|
+
DataFrame
|
|
129
|
+
Pandas DataFrame containing the fit results.
|
|
130
|
+
"""
|
|
131
|
+
|
|
132
|
+
print("Fitting", model, "...")
|
|
133
|
+
|
|
134
|
+
# In compound selection is provided, than use it, otherwise calculate fit for all compounds
|
|
135
|
+
if not compound_sel:
|
|
136
|
+
compounds = input_df["compound"].unique()
|
|
137
|
+
else:
|
|
138
|
+
compounds = compound_sel
|
|
139
|
+
|
|
140
|
+
# Initiating empty output_df
|
|
141
|
+
if model == "IC50":
|
|
142
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'IC50', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'slope', 'Chi^2', 'R^2' ])
|
|
143
|
+
if model == "logIC50":
|
|
144
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'logIC50', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'slope', 'Chi^2', 'R^2' ])
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
for compound in compounds:
|
|
148
|
+
|
|
149
|
+
df_compound = input_df[input_df["compound"].isin([compound])]
|
|
150
|
+
df_compound_pooled = data.pool_data(df_compound)
|
|
151
|
+
|
|
152
|
+
# Generating initial guesses
|
|
153
|
+
ymin_guess, ymax_guess, IC50_guess = generate_guess(df_compound)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
# Defining x and y
|
|
157
|
+
if model == "IC50":
|
|
158
|
+
x = df_compound_pooled["c"]
|
|
159
|
+
if model == "logIC50":
|
|
160
|
+
x = df_compound_pooled["log c"]
|
|
161
|
+
|
|
162
|
+
y = df_compound_pooled["response"]
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
# Setting up the initial parameter values
|
|
166
|
+
pars = define_pars(model, ymin_guess, ymax_guess, IC50_guess, fix_ymin=fix_ymin, fix_ymax=fix_ymax, fix_slope=fix_slope)
|
|
167
|
+
|
|
168
|
+
try:
|
|
169
|
+
# Here is the actual fit in lmfit, the function is called from the "models" module
|
|
170
|
+
if model == "IC50":
|
|
171
|
+
fitter = lmfit.Minimizer(models.IC50_lmfit, pars, fcn_args=(x, y))
|
|
172
|
+
if model == "logIC50":
|
|
173
|
+
fitter = lmfit.Minimizer(models.logIC50_lmfit, pars, fcn_args=(x, y))
|
|
174
|
+
|
|
175
|
+
result = fitter.minimize()
|
|
176
|
+
|
|
177
|
+
# Getting Chi^2 from result container
|
|
178
|
+
Chi_squared = result.chisqr
|
|
179
|
+
# Calculating R^2
|
|
180
|
+
R_squared = 1 - result.residual.var() / np.var(y)
|
|
181
|
+
|
|
182
|
+
fitted_parameter = model
|
|
183
|
+
|
|
184
|
+
# Calculating confidence intervals at 2 sigmas (95%)
|
|
185
|
+
if ci:
|
|
186
|
+
ci = lmfit.conf_interval(fitter, result, p_names = [fitted_parameter], sigmas=[2])
|
|
187
|
+
ci_listoftuples = ci.get(fitted_parameter)
|
|
188
|
+
|
|
189
|
+
loCL = ci_listoftuples[0][1] # This is the lower confidence limit at 2 sigmas (95%)
|
|
190
|
+
upCL = ci_listoftuples[2][1] # This is the upper confidence limit at 2 sigmas (95%)
|
|
191
|
+
else:
|
|
192
|
+
loCL = "nd"
|
|
193
|
+
upCL = "nd"
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
# Printing verbose output if verbose=True
|
|
197
|
+
if verbose:
|
|
198
|
+
print()
|
|
199
|
+
print("===Compound:", compound)
|
|
200
|
+
print()
|
|
201
|
+
print("Data for compound:\n", df_compound_pooled)
|
|
202
|
+
print()
|
|
203
|
+
print("---Initial guesses:")
|
|
204
|
+
print("ymin_guess:", ymin_guess)
|
|
205
|
+
print("ymax_guess:", ymax_guess)
|
|
206
|
+
print("IC50_guess:", IC50_guess)
|
|
207
|
+
print()
|
|
208
|
+
print("---Fitting results:")
|
|
209
|
+
print(lmfit.fit_report(result))
|
|
210
|
+
print()
|
|
211
|
+
print("Chi_squared:", Chi_squared)
|
|
212
|
+
print("R_squared:", R_squared)
|
|
213
|
+
print()
|
|
214
|
+
if ci:
|
|
215
|
+
print("---Confidence intervals:")
|
|
216
|
+
lmfit.printfuncs.report_ci(ci)
|
|
217
|
+
print()
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
# Creating new row for the output dataframe
|
|
221
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
222
|
+
model, result.params['ymin'].value, result.params['ymax'].value, result.params['slope'].value,Chi_squared,R_squared]
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
# Adding new row to the output dataframe
|
|
226
|
+
output_df.loc[len(output_df)] = new_row
|
|
227
|
+
|
|
228
|
+
except Exception:
|
|
229
|
+
print("Calculation for compound " + compound + " failed.")
|
|
230
|
+
if verbose:
|
|
231
|
+
traceback.print_exc()
|
|
232
|
+
|
|
233
|
+
return output_df
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def fit_Kd_direct(input_df, model, LsT, Ns=None, compound_sel = False, fix_ymin = False, fix_ymax = False, ci=True, verbose = False):
|
|
238
|
+
"""Function for fitting the `dir_simple`, `dir_specific` and `dir_total` models.
|
|
239
|
+
|
|
240
|
+
Parameters
|
|
241
|
+
----------
|
|
242
|
+
input_df : DataFrame
|
|
243
|
+
Pandas DataFrame containing the input data.
|
|
244
|
+
model : str
|
|
245
|
+
Name of the model. Options: `dir_simple`, `dir_specific`, `dir_total`
|
|
246
|
+
LsT : float or int
|
|
247
|
+
Total concentration of the labeled ligand.
|
|
248
|
+
Ns : float or int
|
|
249
|
+
Parameter for nonspecific binding of the labeled ligand (needed only for `dir_total` model).
|
|
250
|
+
compound_sel : list
|
|
251
|
+
List of compounds to execute the function on. If set to False, all compounds will be used.
|
|
252
|
+
fix_ymin : float or int
|
|
253
|
+
Lower asymptote of the model will be fixed at the provided value. If set to "False", it will be fitted freely.
|
|
254
|
+
fix_ymax : float or int
|
|
255
|
+
Upper asymptote of the model will be fixed at the provided value. If set to "False", it will be fitted freely.
|
|
256
|
+
ci : bool
|
|
257
|
+
Whether to calculate 95% confidence intervals.
|
|
258
|
+
verbose : bool
|
|
259
|
+
If set to "True", more detailed output is printed. Intended mainly for troubleshooting.
|
|
260
|
+
|
|
261
|
+
Returns
|
|
262
|
+
-------
|
|
263
|
+
DataFrame
|
|
264
|
+
Pandas DataFrame containing the fit results.
|
|
265
|
+
"""
|
|
266
|
+
|
|
267
|
+
print("Fitting", model, "...")
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# Initial checks
|
|
271
|
+
if fix_ymin and fix_ymax:
|
|
272
|
+
ci=False
|
|
273
|
+
print("Only one parameter is fitted. Confidence intervals will not be calculated.")
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
# In compound selection is provided, than use it, otherwise calculate fit for all compounds
|
|
277
|
+
if not compound_sel:
|
|
278
|
+
compounds = input_df["compound"].unique()
|
|
279
|
+
else:
|
|
280
|
+
compounds = compound_sel
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
# Initiating empty output_df
|
|
284
|
+
if model == "dir_simple":
|
|
285
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kds', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'Chi^2', 'R^2'])
|
|
286
|
+
if model == "dir_specific":
|
|
287
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kds', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'LsT', 'Chi^2', 'R^2'])
|
|
288
|
+
if model == "dir_total":
|
|
289
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kds', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'LsT', 'Ns', 'Chi^2', 'R^2'])
|
|
290
|
+
|
|
291
|
+
for compound in compounds:
|
|
292
|
+
|
|
293
|
+
df_compound = input_df[input_df["compound"].isin([compound])]
|
|
294
|
+
df_compound_pooled = data.pool_data(df_compound)
|
|
295
|
+
|
|
296
|
+
# Generating initial guesses
|
|
297
|
+
ymin_guess, ymax_guess, IC50_guess = generate_guess(df_compound, saturation=True)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
# Defining x and y
|
|
301
|
+
x = df_compound_pooled["c"]
|
|
302
|
+
y = df_compound_pooled["response"]
|
|
303
|
+
|
|
304
|
+
if model == "dir_simple":
|
|
305
|
+
LsT=None
|
|
306
|
+
|
|
307
|
+
# Setting up the initial parameter values
|
|
308
|
+
pars = define_pars(model, ymin_guess, ymax_guess, IC50_guess, LsT=LsT, Ns=Ns, fix_ymin=fix_ymin, fix_ymax=fix_ymax)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
try:
|
|
312
|
+
# Here is the actual fit in lmfit, the function is called from the "models" module
|
|
313
|
+
if model == "dir_simple":
|
|
314
|
+
fitter = lmfit.Minimizer(models.dir_simple_lmfit, pars, fcn_args=(x, y))
|
|
315
|
+
if model == "dir_specific":
|
|
316
|
+
fitter = lmfit.Minimizer(models.dir_specific_lmfit, pars, fcn_args=(x, y))
|
|
317
|
+
if model == "dir_total":
|
|
318
|
+
fitter = lmfit.Minimizer(models.dir_total_lmfit, pars, fcn_args=(x, y))
|
|
319
|
+
|
|
320
|
+
result = fitter.minimize()
|
|
321
|
+
|
|
322
|
+
# Getting Chi^2 from result container
|
|
323
|
+
Chi_squared = result.chisqr
|
|
324
|
+
# Calculating R^2
|
|
325
|
+
R_squared = 1 - result.residual.var() / np.var(y)
|
|
326
|
+
|
|
327
|
+
fitted_parameter = "Kds"
|
|
328
|
+
|
|
329
|
+
# Calculating confidence intervals at 2 sigmas (95%)
|
|
330
|
+
if ci:
|
|
331
|
+
ci = lmfit.conf_interval(fitter, result, p_names = [fitted_parameter], sigmas=[2])
|
|
332
|
+
ci_listoftuples = ci.get(fitted_parameter)
|
|
333
|
+
|
|
334
|
+
loCL = ci_listoftuples[0][1] # This is the lower confidence limit at 2 sigmas (95%)
|
|
335
|
+
upCL = ci_listoftuples[2][1] # This is the upper confidence limit at 2 sigmas (95%)
|
|
336
|
+
else:
|
|
337
|
+
loCL = "nd"
|
|
338
|
+
upCL = "nd"
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
# Printing verbose output if verbose=True
|
|
342
|
+
if verbose:
|
|
343
|
+
print()
|
|
344
|
+
print("===Compound:", compound)
|
|
345
|
+
print()
|
|
346
|
+
print("Data for compound:\n", df_compound_pooled)
|
|
347
|
+
print()
|
|
348
|
+
print("---Initial guesses:")
|
|
349
|
+
print("ymin_guess:", ymin_guess)
|
|
350
|
+
print("ymax_guess:", ymax_guess)
|
|
351
|
+
print("IC50_guess:", IC50_guess)
|
|
352
|
+
print("Kds_guess:", IC50_guess/2)
|
|
353
|
+
print()
|
|
354
|
+
print("---Fitting results:")
|
|
355
|
+
print(lmfit.fit_report(result))
|
|
356
|
+
print()
|
|
357
|
+
print("Chi_squared:", Chi_squared)
|
|
358
|
+
print("R_squared:", R_squared)
|
|
359
|
+
print()
|
|
360
|
+
if ci:
|
|
361
|
+
print("---Confidence intervals:")
|
|
362
|
+
lmfit.printfuncs.report_ci(ci)
|
|
363
|
+
print()
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
# Creating new row for the output dataframe
|
|
367
|
+
if model == "dir_simple":
|
|
368
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
369
|
+
model, result.params['ymin'].value, result.params['ymax'].value, Chi_squared, R_squared]
|
|
370
|
+
if model == "dir_specific":
|
|
371
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
372
|
+
model, result.params['ymin'].value, result.params['ymax'].value, result.params['LsT'].value, Chi_squared, R_squared]
|
|
373
|
+
if model == "dir_total":
|
|
374
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
375
|
+
model, result.params['ymin'].value, result.params['ymax'].value, result.params['LsT'].value, result.params['Ns'].value, Chi_squared, R_squared]
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
# Adding new row to the output dataframe
|
|
379
|
+
output_df.loc[len(output_df)] = new_row
|
|
380
|
+
|
|
381
|
+
except Exception:
|
|
382
|
+
print("Calculation for compound " + compound + " failed.")
|
|
383
|
+
if verbose:
|
|
384
|
+
traceback.print_exc()
|
|
385
|
+
|
|
386
|
+
return output_df
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def fit_Kd_competition(input_df, model, RT, LsT, Kds, N=None, compound_sel = False, fix_ymin = False, fix_ymax = False, ci=True, verbose = False):
|
|
392
|
+
"""Function for fitting the `comp_3st_specific`, `comp_3st_total`, `comp_4st_specific` and `comp_4st_total` models.
|
|
393
|
+
|
|
394
|
+
Parameters
|
|
395
|
+
----------
|
|
396
|
+
input_df : DataFrame
|
|
397
|
+
Pandas DataFrame containing the input data.
|
|
398
|
+
model : str
|
|
399
|
+
Name of the model. Options: `comp_3st_specific`, `comp_3st_total`, `comp_4st_specific`, `comp_4st_total`
|
|
400
|
+
RT : float or int
|
|
401
|
+
Total concentration of the receptor.
|
|
402
|
+
LsT : float or int
|
|
403
|
+
Total concentration of the labeled ligand.
|
|
404
|
+
Kds : float or int
|
|
405
|
+
Dissociation constant of the labeled ligand.
|
|
406
|
+
N : float or int
|
|
407
|
+
Parameter for nonspecific binding of the unlabeled ligand (needed only for `comp_3st_total` and `comp_4st_total` models).
|
|
408
|
+
compound_sel : list
|
|
409
|
+
List of compounds to execute the function on. If set to False, all compounds will be used.
|
|
410
|
+
fix_ymin : float or int
|
|
411
|
+
Lower asymptote of the model will be fixed at the provided value. If set to "False", it will be fitted freely.
|
|
412
|
+
fix_ymax : float or int
|
|
413
|
+
Upper asymptote of the model will be fixed at the provided value. If set to "False", it will be fitted freely.
|
|
414
|
+
ci : bool
|
|
415
|
+
Whether to calculate 95% confidence intervals.
|
|
416
|
+
verbose : bool
|
|
417
|
+
If set to "True", more detailed output is printed. Intended mainly for troubleshooting.
|
|
418
|
+
|
|
419
|
+
Returns
|
|
420
|
+
-------
|
|
421
|
+
DataFrame
|
|
422
|
+
Pandas DataFrame containing the fit results.
|
|
423
|
+
"""
|
|
424
|
+
|
|
425
|
+
print("Fitting", model, "...")
|
|
426
|
+
|
|
427
|
+
# Initial checks
|
|
428
|
+
if fix_ymin and fix_ymax:
|
|
429
|
+
ci=False
|
|
430
|
+
print("Only one parameter is fitted. Confidence intervals will not be calculated.")
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
# In compound selection is provided, than use it, otherwise calculate fit for all compounds
|
|
434
|
+
if not compound_sel:
|
|
435
|
+
compounds = input_df["compound"].unique()
|
|
436
|
+
else:
|
|
437
|
+
compounds = compound_sel
|
|
438
|
+
|
|
439
|
+
# Initiating empty output_df
|
|
440
|
+
if model == "comp_3st_specific":
|
|
441
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kd', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'RT', 'LsT', 'Kds', 'Chi^2', 'R^2' ])
|
|
442
|
+
if model == "comp_3st_total":
|
|
443
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kd', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'RT', 'LsT', 'Kds', 'N', 'Chi^2', 'R^2' ])
|
|
444
|
+
if model == "comp_4st_specific":
|
|
445
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kd', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'RT', 'LsT', 'Kds', 'Kd3', 'Chi^2', 'R^2' ])
|
|
446
|
+
if model == "comp_4st_total":
|
|
447
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kd', 'loCL', 'upCL', 'SE', 'model', 'ymin', 'ymax', 'RT', 'LsT', 'Kds', 'Kd3', 'N', 'Chi^2', 'R^2' ])
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
for compound in compounds:
|
|
451
|
+
|
|
452
|
+
df_compound = input_df[input_df["compound"].isin([compound])]
|
|
453
|
+
df_compound_pooled = data.pool_data(df_compound)
|
|
454
|
+
|
|
455
|
+
# Generating initial guesses
|
|
456
|
+
ymin_guess, ymax_guess, IC50_guess = generate_guess(df_compound)
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
# Defining x and y
|
|
460
|
+
x = df_compound_pooled["c"]
|
|
461
|
+
y = df_compound_pooled["response"]
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
# Setting up the initial parameter values
|
|
465
|
+
pars = define_pars(model, ymin_guess, ymax_guess, IC50_guess, RT=RT, LsT=LsT, Kds=Kds, N=N, fix_ymin=fix_ymin, fix_ymax=fix_ymax)
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
try:
|
|
469
|
+
# Here is the actual fit in lmfit, the function is called from the "models" module
|
|
470
|
+
if model == "comp_3st_specific":
|
|
471
|
+
fitter = lmfit.Minimizer(models.comp_3st_specific_lmfit, pars, fcn_args=(x, y))
|
|
472
|
+
if model == "comp_3st_total":
|
|
473
|
+
fitter = lmfit.Minimizer(models.comp_3st_total_lmfit, pars, fcn_args=(x, y))
|
|
474
|
+
if model == "comp_4st_specific":
|
|
475
|
+
fitter = lmfit.Minimizer(models.comp_4st_specific_lmfit, pars, fcn_args=(x, y))
|
|
476
|
+
if model == "comp_4st_total":
|
|
477
|
+
fitter = lmfit.Minimizer(models.comp_4st_total_lmfit, pars, fcn_args=(x, y))
|
|
478
|
+
|
|
479
|
+
result = fitter.minimize()
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
# Getting Chi^2 from result container
|
|
483
|
+
Chi_squared = result.chisqr
|
|
484
|
+
# Calculating R^2
|
|
485
|
+
R_squared = 1 - result.residual.var() / np.var(y)
|
|
486
|
+
|
|
487
|
+
fitted_parameter = "Kd"
|
|
488
|
+
|
|
489
|
+
# Calculating confidence intervals at 2 sigmas (95%)
|
|
490
|
+
if ci:
|
|
491
|
+
ci = lmfit.conf_interval(fitter, result, p_names = [fitted_parameter], sigmas=[2])
|
|
492
|
+
ci_listoftuples = ci.get(fitted_parameter)
|
|
493
|
+
|
|
494
|
+
loCL = ci_listoftuples[0][1] # This is the lower confidence limit at 2 sigmas (95%)
|
|
495
|
+
upCL = ci_listoftuples[2][1] # This is the upper confidence limit at 2 sigmas (95%)
|
|
496
|
+
else:
|
|
497
|
+
loCL = "nd"
|
|
498
|
+
upCL = "nd"
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
# Printing verbose output if verbose=True
|
|
502
|
+
if verbose:
|
|
503
|
+
print()
|
|
504
|
+
print("===Compound:", compound)
|
|
505
|
+
print()
|
|
506
|
+
print("Data for compound:\n", df_compound_pooled)
|
|
507
|
+
print()
|
|
508
|
+
print("---Initial guesses:")
|
|
509
|
+
print("ymin_guess:", ymin_guess)
|
|
510
|
+
print("ymax_guess:", ymax_guess)
|
|
511
|
+
print("IC50_guess:", IC50_guess)
|
|
512
|
+
print("Kd_guess:", IC50_guess/2)
|
|
513
|
+
print()
|
|
514
|
+
print("---Fitting results:")
|
|
515
|
+
print(lmfit.fit_report(result))
|
|
516
|
+
print()
|
|
517
|
+
print("Chi_squared:", Chi_squared)
|
|
518
|
+
print("R_squared:", R_squared)
|
|
519
|
+
print()
|
|
520
|
+
if ci:
|
|
521
|
+
print("---Confidence intervals:")
|
|
522
|
+
lmfit.printfuncs.report_ci(ci)
|
|
523
|
+
print()
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
# Creating new row for the output dataframe
|
|
527
|
+
if model == "comp_3st_specific":
|
|
528
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
529
|
+
model, result.params['ymin'].value, result.params['ymax'].value, result.params['RT'].value, result.params['LsT'].value, result.params['Kds'].value, Chi_squared, R_squared]
|
|
530
|
+
if model == "comp_3st_total":
|
|
531
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
532
|
+
model, result.params['ymin'].value, result.params['ymax'].value, result.params['RT'].value, result.params['LsT'].value, result.params['Kds'].value, result.params['N'].value, Chi_squared, R_squared]
|
|
533
|
+
if model == "comp_4st_specific":
|
|
534
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
535
|
+
model, result.params['ymin'].value, result.params['ymax'].value, result.params['RT'].value, result.params['LsT'].value, result.params['Kds'].value, result.params['Kd3'].value, Chi_squared, R_squared]
|
|
536
|
+
if model == "comp_4st_total":
|
|
537
|
+
new_row = [compound, result.ndata, result.params[fitted_parameter].value, loCL, upCL, result.params[fitted_parameter].stderr,
|
|
538
|
+
model, result.params['ymin'].value, result.params['ymax'].value, result.params['RT'].value, result.params['LsT'].value, result.params['Kds'].value, result.params['Kd3'].value, result.params['N'].value, Chi_squared, R_squared]
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
# Adding new row to the output dataframe
|
|
542
|
+
output_df.loc[len(output_df)] = new_row
|
|
543
|
+
|
|
544
|
+
except Exception:
|
|
545
|
+
print("Calculation for compound " + compound + " failed.")
|
|
546
|
+
if verbose:
|
|
547
|
+
traceback.print_exc()
|
|
548
|
+
|
|
549
|
+
return output_df
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
def convert(IC50_df, model, RT=None, LsT=None, Kds=None, y0=None, compound_sel=False, ci=True, verbose=False):
|
|
555
|
+
"""Function for converting IC50 to Kd using `coleska`, `cheng_prusoff` and `cheng_prusoff_corr` models.
|
|
556
|
+
|
|
557
|
+
Parameters
|
|
558
|
+
----------
|
|
559
|
+
IC50_df : DataFrame
|
|
560
|
+
Pandas DataFrame containing the fitted IC50 values.
|
|
561
|
+
model : str
|
|
562
|
+
Name of the conversion model. Options: `coleska`, `cheng_prusoff`, `cheng_prusoff_corr`
|
|
563
|
+
RT : float or int
|
|
564
|
+
Total concentration of the receptor.
|
|
565
|
+
LsT : float or int
|
|
566
|
+
Total concentration of the labeled ligand.
|
|
567
|
+
Kds : float or int
|
|
568
|
+
Dissociation constant of the labeled ligand.
|
|
569
|
+
y0 : float or int
|
|
570
|
+
Parameter used in the corrected Cheng-Prusoff model.
|
|
571
|
+
compound_sel : list
|
|
572
|
+
List of compounds to execute the function on. If set to False, all compounds will be used.
|
|
573
|
+
ci : bool
|
|
574
|
+
Whether to calculate 95% confidence intervals.
|
|
575
|
+
verbose : bool
|
|
576
|
+
If set to "True", more detailed output is printed. Intended mainly for troubleshooting.
|
|
577
|
+
|
|
578
|
+
Returns
|
|
579
|
+
-------
|
|
580
|
+
DataFrame
|
|
581
|
+
Pandas DataFrame containing the conversion results.
|
|
582
|
+
"""
|
|
583
|
+
|
|
584
|
+
print("Converting IC50 to Kd using", model, "model...")
|
|
585
|
+
|
|
586
|
+
# In compound selection is provided, than use it, otherwise calculate fit for all compounds
|
|
587
|
+
if not compound_sel:
|
|
588
|
+
compounds = IC50_df["compound"].unique()
|
|
589
|
+
else:
|
|
590
|
+
compounds = compound_sel
|
|
591
|
+
|
|
592
|
+
if 'IC50' not in IC50_df.columns:
|
|
593
|
+
exit("Provided dataframe does not contain IC50 column. Aborting...")
|
|
594
|
+
|
|
595
|
+
# If the provided df contains no CL, than only convert means
|
|
596
|
+
if IC50_df["loCL"].iloc[0] == "nd" and IC50_df["upCL"].iloc[0] == "nd":
|
|
597
|
+
ci=False
|
|
598
|
+
print("Confidence limits not detected in the provided dataframe. Converting only mean values...")
|
|
599
|
+
if not ci:
|
|
600
|
+
loCL = "nd"
|
|
601
|
+
upCL = "nd"
|
|
602
|
+
|
|
603
|
+
# Initiating empty output_df
|
|
604
|
+
output_df = pd.DataFrame(columns=['compound', 'n_points', 'Kd', 'loCL', 'upCL', 'SE', 'model'])
|
|
605
|
+
|
|
606
|
+
for compound in compounds:
|
|
607
|
+
|
|
608
|
+
df_compound = IC50_df[IC50_df["compound"].isin([compound])]
|
|
609
|
+
|
|
610
|
+
try:
|
|
611
|
+
# Here are the actual conversions
|
|
612
|
+
if model == "cheng_prusoff":
|
|
613
|
+
Kd = models.cheng_prusoff(LsT, Kds, df_compound["IC50"].iloc[0])
|
|
614
|
+
if ci:
|
|
615
|
+
loCL = models.cheng_prusoff(LsT, Kds, df_compound["loCL"].iloc[0])
|
|
616
|
+
upCL = models.cheng_prusoff(LsT, Kds, df_compound["upCL"].iloc[0])
|
|
617
|
+
if model == "cheng_prusoff_corr":
|
|
618
|
+
Kd = models.cheng_prusoff_corr(LsT, Kds, y0, df_compound["IC50"].iloc[0])
|
|
619
|
+
if ci:
|
|
620
|
+
loCL = models.cheng_prusoff_corr(LsT, Kds, y0, df_compound["loCL"].iloc[0])
|
|
621
|
+
upCL = models.cheng_prusoff_corr(LsT, Kds, y0, df_compound["upCL"].iloc[0])
|
|
622
|
+
if model == "coleska":
|
|
623
|
+
Kd = models.coleska(RT, LsT, Kds, df_compound["IC50"].iloc[0])
|
|
624
|
+
if ci:
|
|
625
|
+
loCL = models.coleska(RT, LsT, Kds, df_compound["loCL"].iloc[0])
|
|
626
|
+
upCL = models.coleska(RT, LsT, Kds, df_compound["upCL"].iloc[0])
|
|
627
|
+
|
|
628
|
+
# Creating new row for the output dataframe
|
|
629
|
+
new_row = [compound, 1, Kd, loCL, upCL, 'nd', model]
|
|
630
|
+
|
|
631
|
+
# Adding new row to the output dataframe
|
|
632
|
+
output_df.loc[len(output_df)] = new_row
|
|
633
|
+
|
|
634
|
+
except Exception:
|
|
635
|
+
print("Calculation for compound " + compound + " failed.")
|
|
636
|
+
if verbose:
|
|
637
|
+
traceback.print_exc()
|
|
638
|
+
|
|
639
|
+
return output_df
|