PyTFBS 1.0.5__tar.gz → 1.0.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PKG-INFO +35 -4
- pytfbs-1.0.7/PyTFBS/clink.py +214 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS.egg-info/PKG-INFO +35 -4
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS.egg-info/SOURCES.txt +1 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS.egg-info/requires.txt +6 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/README.md +28 -3
- {pytfbs-1.0.5 → pytfbs-1.0.7}/pyproject.toml +7 -1
- {pytfbs-1.0.5 → pytfbs-1.0.7}/LICENSE +0 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS/__init__.py +0 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS/motif.py +0 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS/predict.py +0 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS.egg-info/dependency_links.txt +0 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS.egg-info/entry_points.txt +0 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/PyTFBS.egg-info/top_level.txt +0 -0
- {pytfbs-1.0.5 → pytfbs-1.0.7}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: PyTFBS
|
|
3
|
-
Version: 1.0.
|
|
3
|
+
Version: 1.0.7
|
|
4
4
|
Summary: PyTFBS: A Python Package for Transcription Factor Binding Site Prediction
|
|
5
5
|
Author-email: Tinghua Huang <thua45@126.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -21,6 +21,12 @@ Requires-Dist: torch
|
|
|
21
21
|
Requires-Dist: numpy==1.26; sys_platform == "darwin"
|
|
22
22
|
Requires-Dist: numpy>=2.0; sys_platform == "win32"
|
|
23
23
|
Requires-Dist: numpy>=2.0; sys_platform == "linux"
|
|
24
|
+
Requires-Dist: pandas==3.0; sys_platform == "darwin"
|
|
25
|
+
Requires-Dist: pandas>=3.0; sys_platform == "win32"
|
|
26
|
+
Requires-Dist: pandas>=3.0; sys_platform == "linux"
|
|
27
|
+
Requires-Dist: scipy==1.18; sys_platform == "darwin"
|
|
28
|
+
Requires-Dist: scipy==1.18; sys_platform == "win32"
|
|
29
|
+
Requires-Dist: scipy==1.18; sys_platform == "linux"
|
|
24
30
|
Provides-Extra: dev
|
|
25
31
|
Requires-Dist: matplotlib; extra == "dev"
|
|
26
32
|
Provides-Extra: test
|
|
@@ -29,18 +35,20 @@ Dynamic: license-file
|
|
|
29
35
|
|
|
30
36
|
# PyTFBS
|
|
31
37
|
|
|
32
|
-
A Python package for
|
|
38
|
+
A Python package for predicting transcription factor binding sites.
|
|
33
39
|
|
|
34
40
|
## Installation
|
|
35
41
|
|
|
36
42
|
```bash
|
|
37
|
-
pip install torch numpy PyTFBS
|
|
43
|
+
pip install torch numpy pandas scipy PyTFBS
|
|
38
44
|
```
|
|
39
45
|
|
|
40
46
|
## Usage
|
|
41
47
|
|
|
42
48
|
```python
|
|
43
|
-
from PyTFBS import motif, predict
|
|
49
|
+
from PyTFBS import motif, predict, clink
|
|
50
|
+
import random
|
|
51
|
+
random.seed(42)
|
|
44
52
|
|
|
45
53
|
# download PyTFBS data, only need to run once!!!
|
|
46
54
|
motif.download_data()
|
|
@@ -66,4 +74,27 @@ predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_fil
|
|
|
66
74
|
# the my_motif_dir should be organized as [[motif], [trace], [par]]
|
|
67
75
|
predict.script('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', data_dir='my_motif_dir')
|
|
68
76
|
predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', 64, data_dir='my_motif_dir')
|
|
77
|
+
|
|
78
|
+
# create the CLink TF-target set
|
|
79
|
+
|
|
80
|
+
# rexp_files = get_rexp_files()
|
|
81
|
+
# print(rexp_files)
|
|
82
|
+
|
|
83
|
+
df_rexps = clink.read_rexp_file()
|
|
84
|
+
tf_cnames = clink.read_tf_cname()
|
|
85
|
+
tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
|
|
86
|
+
tf_tars = clink.read_PyTFBS_output(tf_code + '_output.txt')
|
|
87
|
+
|
|
88
|
+
for tf_cname in tf_cnames[tf_code]:
|
|
89
|
+
if tf_cname[1] not in df_rexps.index:
|
|
90
|
+
continue
|
|
91
|
+
else:
|
|
92
|
+
tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
|
|
93
|
+
df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
|
|
94
|
+
tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
|
|
95
|
+
rcis = clink.rci(tar_exps, tf_exp)
|
|
96
|
+
for rci in rcis:
|
|
97
|
+
score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
|
|
98
|
+
print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
|
|
99
|
+
|
|
69
100
|
```
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import math
|
|
3
|
+
import pandas as pd
|
|
4
|
+
import numpy as np
|
|
5
|
+
np.random.seed(42)
|
|
6
|
+
from collections import defaultdict
|
|
7
|
+
import random
|
|
8
|
+
random.seed(42)
|
|
9
|
+
from scipy.stats import t
|
|
10
|
+
|
|
11
|
+
PyTFBS_data_dir = './PyTFBS_data'
|
|
12
|
+
|
|
13
|
+
def sign(x):
|
|
14
|
+
if x > 0:
|
|
15
|
+
return 1
|
|
16
|
+
elif x < 0:
|
|
17
|
+
return -1
|
|
18
|
+
else:
|
|
19
|
+
return 0
|
|
20
|
+
|
|
21
|
+
def weighted_corr(x1, y1, alpha):
|
|
22
|
+
"""
|
|
23
|
+
计算加权皮尔森相关系数及其p值。
|
|
24
|
+
|
|
25
|
+
权重由每个点的 (|x| + |y|) / 2 计算,并进一步按 alpha 次幂调整。
|
|
26
|
+
|
|
27
|
+
参数:
|
|
28
|
+
x, y : array-like
|
|
29
|
+
输入变量。
|
|
30
|
+
alpha : float
|
|
31
|
+
权重调整指数,控制权重对极端值的敏感度。
|
|
32
|
+
|
|
33
|
+
返回:
|
|
34
|
+
r : float
|
|
35
|
+
加权皮尔森相关系数。
|
|
36
|
+
p : float
|
|
37
|
+
双侧检验的p值,检验原假设为总体相关系数为0。
|
|
38
|
+
如果有效自由度 ≤ 0,则返回 p = np.nan。
|
|
39
|
+
"""
|
|
40
|
+
x = np.array(x1)
|
|
41
|
+
y = np.array(y1)
|
|
42
|
+
|
|
43
|
+
# 计算基础权重(基于数值大小)
|
|
44
|
+
w = (np.abs(x) + np.abs(y)) / 2.0
|
|
45
|
+
# 调整权重,使其最大值缩放至1后再取alpha次幂
|
|
46
|
+
w_adj = (w / np.max(w)) ** alpha
|
|
47
|
+
# 归一化权重(和为1)
|
|
48
|
+
w_norm = w_adj / np.sum(w_adj)
|
|
49
|
+
|
|
50
|
+
# 加权均值
|
|
51
|
+
x_bar = np.sum(w_norm * x)
|
|
52
|
+
y_bar = np.sum(w_norm * y)
|
|
53
|
+
|
|
54
|
+
# 加权协方差与标准差
|
|
55
|
+
dx = x - x_bar
|
|
56
|
+
dy = y - y_bar
|
|
57
|
+
cov = np.sum(w_norm * dx * dy)
|
|
58
|
+
var_x = np.sum(w_norm * dx**2)
|
|
59
|
+
var_y = np.sum(w_norm * dy**2)
|
|
60
|
+
std_x = np.sqrt(var_x)
|
|
61
|
+
std_y = np.sqrt(var_y)
|
|
62
|
+
|
|
63
|
+
if std_x * std_y == 0.0:
|
|
64
|
+
r = 0.0
|
|
65
|
+
else:
|
|
66
|
+
r = cov / (std_x * std_y)
|
|
67
|
+
|
|
68
|
+
# ---- 计算p值(基于t分布近似) ----
|
|
69
|
+
# 有效样本量(权重之和的平方除以权重平方和)
|
|
70
|
+
sum_w = np.sum(w_adj) # 注意:w_adj未经归一化,但比例不变
|
|
71
|
+
sum_w2 = np.sum(w_adj**2)
|
|
72
|
+
n_eff = (sum_w ** 2) / sum_w2 if sum_w2 > 0 else 0.0
|
|
73
|
+
|
|
74
|
+
df = n_eff - 2 # 自由度
|
|
75
|
+
if df > 0 and abs(r) < 1.0:
|
|
76
|
+
# t统计量:t = r * sqrt(df / (1 - r^2))
|
|
77
|
+
t_stat = r * np.sqrt(df / (1 - r**2))
|
|
78
|
+
p = 2 * (1 - t.cdf(abs(t_stat), df))
|
|
79
|
+
else:
|
|
80
|
+
p = np.nan # 无法计算或相关系数为±1时,p值无意义
|
|
81
|
+
|
|
82
|
+
return r, p
|
|
83
|
+
|
|
84
|
+
def norm_std(data, sample=True):
|
|
85
|
+
"""
|
|
86
|
+
手动计算标准差
|
|
87
|
+
sample=True: 样本标准差(除以 n-1)
|
|
88
|
+
sample=False: 总体标准差(除以 n)
|
|
89
|
+
"""
|
|
90
|
+
n = len(data)
|
|
91
|
+
if n < 2:
|
|
92
|
+
raise ValueError("need 2 obsvs")
|
|
93
|
+
|
|
94
|
+
# 步骤 1:计算平均值
|
|
95
|
+
# mean = sum(data) / n
|
|
96
|
+
|
|
97
|
+
# 步骤 2 & 3:计算每个数据与平均值的差的平方
|
|
98
|
+
# squared_diffs = [(x - mean) ** 2 for x in data]
|
|
99
|
+
squared_diffs = [x ** 2 for x in data]
|
|
100
|
+
|
|
101
|
+
# 步骤 4:计算方差
|
|
102
|
+
if sample:
|
|
103
|
+
variance = sum(squared_diffs) / (n - 1) # 样本方差
|
|
104
|
+
else:
|
|
105
|
+
variance = sum(squared_diffs) / n # 总体方差
|
|
106
|
+
|
|
107
|
+
# 步骤 5:开平方根得到标准差
|
|
108
|
+
std = variance ** 0.5
|
|
109
|
+
|
|
110
|
+
k = max(1, int(len(data) * 0.05)) # 至少取1个
|
|
111
|
+
sorted_lst = sorted([abs(i) for i in data], reverse=True)
|
|
112
|
+
max_obsv = sorted_lst[k - 1]
|
|
113
|
+
std_norm = std / max_obsv
|
|
114
|
+
|
|
115
|
+
return std_norm
|
|
116
|
+
|
|
117
|
+
def rci(tar_exps, tf_exp):
|
|
118
|
+
tar_rcis = []
|
|
119
|
+
for tar_name in tar_exps.keys():
|
|
120
|
+
tar_exp = tar_exps[tar_name]
|
|
121
|
+
corr_wt, p_value = weighted_corr(tar_exp, tf_exp, alpha=1.0)
|
|
122
|
+
tar_cv = norm_std(tar_exp)
|
|
123
|
+
if math.isnan(tar_cv):
|
|
124
|
+
tar_cv = 0.0
|
|
125
|
+
tf_cv = norm_std(tf_exp)
|
|
126
|
+
if math.isnan(tf_cv):
|
|
127
|
+
tf_cv = 0.0
|
|
128
|
+
if math.isnan(corr_wt) or p_value >= 0.05:
|
|
129
|
+
continue
|
|
130
|
+
else:
|
|
131
|
+
rci = sign(corr_wt) * (abs(corr_wt)**1.0 * tar_cv**0.5 * tf_cv**0.5) ** (1.0 / (1.0 + 0.5 + 0.5))
|
|
132
|
+
tar_rcis.append((tar_name, rci))
|
|
133
|
+
return tar_rcis
|
|
134
|
+
|
|
135
|
+
def read_tf_cname(file=None, data_dir=None):
|
|
136
|
+
if data_dir == None:
|
|
137
|
+
data_dir = PyTFBS_data_dir
|
|
138
|
+
if file == None:
|
|
139
|
+
file = data_dir + '/TF-code-name.txt'
|
|
140
|
+
else:
|
|
141
|
+
file = data_dir + '/' + file
|
|
142
|
+
if not os.path.exists(file):
|
|
143
|
+
print('TF-code-name.txt not exist, you may need to run download_data() first')
|
|
144
|
+
exit(1)
|
|
145
|
+
tf_cnames = defaultdict(list)
|
|
146
|
+
for line in open(file, 'r'):
|
|
147
|
+
lblocks = line.rstrip().split('\t')
|
|
148
|
+
tf_cnames[lblocks[1]].append((lblocks[2], lblocks[3]))
|
|
149
|
+
return tf_cnames
|
|
150
|
+
|
|
151
|
+
def read_rexp_file(file=None, data_dir=None):
|
|
152
|
+
if data_dir == None:
|
|
153
|
+
data_dir = PyTFBS_data_dir + '/Ref_Exps'
|
|
154
|
+
if file == None:
|
|
155
|
+
file = data_dir + '/Human_ED_matrix_1.5.txt'
|
|
156
|
+
else:
|
|
157
|
+
file = data_dir + '/' + file_name
|
|
158
|
+
if not os.path.exists(file):
|
|
159
|
+
print('Human_ED_matrix_1.5.txt not exist, you may need to run download_data() first')
|
|
160
|
+
exit(1)
|
|
161
|
+
df_rexps = pd.read_csv(file, sep='\t', header=0, index_col=0)
|
|
162
|
+
return df_rexps
|
|
163
|
+
|
|
164
|
+
def read_PyTFBS_output(file):
|
|
165
|
+
tf_tars = set()
|
|
166
|
+
for line in open(file):
|
|
167
|
+
if line[0] == '#':
|
|
168
|
+
continue
|
|
169
|
+
lblocks = line.rstrip().split('\t')
|
|
170
|
+
gene_name = lblocks[0].split('::')[0].split('|')[2]
|
|
171
|
+
score = float(lblocks[3])
|
|
172
|
+
jindex = float(lblocks[2])
|
|
173
|
+
if score > 0.8 and jindex > 0.3:
|
|
174
|
+
tf_tars.add(gene_name)
|
|
175
|
+
return tf_tars
|
|
176
|
+
|
|
177
|
+
def get_rexp_files(species=None, data_dir=None):
|
|
178
|
+
if data_dir == None:
|
|
179
|
+
data_dir = PyTFBS_data_dir
|
|
180
|
+
index_file = data_dir + '/RExps_index.txt'
|
|
181
|
+
if not os.path.exists(index_file):
|
|
182
|
+
print('RExps_index.txt not exist, you may need to run download_data() first')
|
|
183
|
+
exit(1)
|
|
184
|
+
header = ''
|
|
185
|
+
result = []
|
|
186
|
+
for line in open(index_file, 'r'):
|
|
187
|
+
if line[0] == "#":
|
|
188
|
+
header = line.rstrip()
|
|
189
|
+
continue
|
|
190
|
+
lblocks = line.rstrip().split('\t')
|
|
191
|
+
if species != None and lblocks[1] != species:
|
|
192
|
+
continue
|
|
193
|
+
result.append(lblocks[0])
|
|
194
|
+
return result
|
|
195
|
+
|
|
196
|
+
if __name__ == '__main__':
|
|
197
|
+
rexp_files = get_rexp_files()
|
|
198
|
+
print(rexp_files)
|
|
199
|
+
df_rexps = read_rexp_file()
|
|
200
|
+
tf_cnames = read_tf_cname()
|
|
201
|
+
tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
|
|
202
|
+
tf_tars = read_PyTFBS_output(tf_code + '_output.txt')
|
|
203
|
+
|
|
204
|
+
for tf_cname in tf_cnames[tf_code]:
|
|
205
|
+
if tf_cname[1] not in df_rexps.index:
|
|
206
|
+
continue
|
|
207
|
+
else:
|
|
208
|
+
tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
|
|
209
|
+
df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
|
|
210
|
+
tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
|
|
211
|
+
rcis = rci(tar_exps, tf_exp)
|
|
212
|
+
for rci in rcis:
|
|
213
|
+
score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
|
|
214
|
+
print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: PyTFBS
|
|
3
|
-
Version: 1.0.
|
|
3
|
+
Version: 1.0.7
|
|
4
4
|
Summary: PyTFBS: A Python Package for Transcription Factor Binding Site Prediction
|
|
5
5
|
Author-email: Tinghua Huang <thua45@126.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -21,6 +21,12 @@ Requires-Dist: torch
|
|
|
21
21
|
Requires-Dist: numpy==1.26; sys_platform == "darwin"
|
|
22
22
|
Requires-Dist: numpy>=2.0; sys_platform == "win32"
|
|
23
23
|
Requires-Dist: numpy>=2.0; sys_platform == "linux"
|
|
24
|
+
Requires-Dist: pandas==3.0; sys_platform == "darwin"
|
|
25
|
+
Requires-Dist: pandas>=3.0; sys_platform == "win32"
|
|
26
|
+
Requires-Dist: pandas>=3.0; sys_platform == "linux"
|
|
27
|
+
Requires-Dist: scipy==1.18; sys_platform == "darwin"
|
|
28
|
+
Requires-Dist: scipy==1.18; sys_platform == "win32"
|
|
29
|
+
Requires-Dist: scipy==1.18; sys_platform == "linux"
|
|
24
30
|
Provides-Extra: dev
|
|
25
31
|
Requires-Dist: matplotlib; extra == "dev"
|
|
26
32
|
Provides-Extra: test
|
|
@@ -29,18 +35,20 @@ Dynamic: license-file
|
|
|
29
35
|
|
|
30
36
|
# PyTFBS
|
|
31
37
|
|
|
32
|
-
A Python package for
|
|
38
|
+
A Python package for predicting transcription factor binding sites.
|
|
33
39
|
|
|
34
40
|
## Installation
|
|
35
41
|
|
|
36
42
|
```bash
|
|
37
|
-
pip install torch numpy PyTFBS
|
|
43
|
+
pip install torch numpy pandas scipy PyTFBS
|
|
38
44
|
```
|
|
39
45
|
|
|
40
46
|
## Usage
|
|
41
47
|
|
|
42
48
|
```python
|
|
43
|
-
from PyTFBS import motif, predict
|
|
49
|
+
from PyTFBS import motif, predict, clink
|
|
50
|
+
import random
|
|
51
|
+
random.seed(42)
|
|
44
52
|
|
|
45
53
|
# download PyTFBS data, only need to run once!!!
|
|
46
54
|
motif.download_data()
|
|
@@ -66,4 +74,27 @@ predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_fil
|
|
|
66
74
|
# the my_motif_dir should be organized as [[motif], [trace], [par]]
|
|
67
75
|
predict.script('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', data_dir='my_motif_dir')
|
|
68
76
|
predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', 64, data_dir='my_motif_dir')
|
|
77
|
+
|
|
78
|
+
# create the CLink TF-target set
|
|
79
|
+
|
|
80
|
+
# rexp_files = get_rexp_files()
|
|
81
|
+
# print(rexp_files)
|
|
82
|
+
|
|
83
|
+
df_rexps = clink.read_rexp_file()
|
|
84
|
+
tf_cnames = clink.read_tf_cname()
|
|
85
|
+
tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
|
|
86
|
+
tf_tars = clink.read_PyTFBS_output(tf_code + '_output.txt')
|
|
87
|
+
|
|
88
|
+
for tf_cname in tf_cnames[tf_code]:
|
|
89
|
+
if tf_cname[1] not in df_rexps.index:
|
|
90
|
+
continue
|
|
91
|
+
else:
|
|
92
|
+
tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
|
|
93
|
+
df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
|
|
94
|
+
tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
|
|
95
|
+
rcis = clink.rci(tar_exps, tf_exp)
|
|
96
|
+
for rci in rcis:
|
|
97
|
+
score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
|
|
98
|
+
print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
|
|
99
|
+
|
|
69
100
|
```
|
|
@@ -2,12 +2,18 @@ torch
|
|
|
2
2
|
|
|
3
3
|
[:sys_platform == "darwin"]
|
|
4
4
|
numpy==1.26
|
|
5
|
+
pandas==3.0
|
|
6
|
+
scipy==1.18
|
|
5
7
|
|
|
6
8
|
[:sys_platform == "linux"]
|
|
7
9
|
numpy>=2.0
|
|
10
|
+
pandas>=3.0
|
|
11
|
+
scipy==1.18
|
|
8
12
|
|
|
9
13
|
[:sys_platform == "win32"]
|
|
10
14
|
numpy>=2.0
|
|
15
|
+
pandas>=3.0
|
|
16
|
+
scipy==1.18
|
|
11
17
|
|
|
12
18
|
[dev]
|
|
13
19
|
matplotlib
|
|
@@ -1,17 +1,19 @@
|
|
|
1
1
|
# PyTFBS
|
|
2
2
|
|
|
3
|
-
A Python package for
|
|
3
|
+
A Python package for predicting transcription factor binding sites.
|
|
4
4
|
|
|
5
5
|
## Installation
|
|
6
6
|
|
|
7
7
|
```bash
|
|
8
|
-
pip install torch numpy PyTFBS
|
|
8
|
+
pip install torch numpy pandas scipy PyTFBS
|
|
9
9
|
```
|
|
10
10
|
|
|
11
11
|
## Usage
|
|
12
12
|
|
|
13
13
|
```python
|
|
14
|
-
from PyTFBS import motif, predict
|
|
14
|
+
from PyTFBS import motif, predict, clink
|
|
15
|
+
import random
|
|
16
|
+
random.seed(42)
|
|
15
17
|
|
|
16
18
|
# download PyTFBS data, only need to run once!!!
|
|
17
19
|
motif.download_data()
|
|
@@ -37,4 +39,27 @@ predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_fil
|
|
|
37
39
|
# the my_motif_dir should be organized as [[motif], [trace], [par]]
|
|
38
40
|
predict.script('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', data_dir='my_motif_dir')
|
|
39
41
|
predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', 64, data_dir='my_motif_dir')
|
|
42
|
+
|
|
43
|
+
# create the CLink TF-target set
|
|
44
|
+
|
|
45
|
+
# rexp_files = get_rexp_files()
|
|
46
|
+
# print(rexp_files)
|
|
47
|
+
|
|
48
|
+
df_rexps = clink.read_rexp_file()
|
|
49
|
+
tf_cnames = clink.read_tf_cname()
|
|
50
|
+
tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
|
|
51
|
+
tf_tars = clink.read_PyTFBS_output(tf_code + '_output.txt')
|
|
52
|
+
|
|
53
|
+
for tf_cname in tf_cnames[tf_code]:
|
|
54
|
+
if tf_cname[1] not in df_rexps.index:
|
|
55
|
+
continue
|
|
56
|
+
else:
|
|
57
|
+
tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
|
|
58
|
+
df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
|
|
59
|
+
tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
|
|
60
|
+
rcis = clink.rci(tar_exps, tf_exp)
|
|
61
|
+
for rci in rcis:
|
|
62
|
+
score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
|
|
63
|
+
print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
|
|
64
|
+
|
|
40
65
|
```
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "PyTFBS"
|
|
7
|
-
version = "1.0.
|
|
7
|
+
version = "1.0.7"
|
|
8
8
|
license = "MIT" # SPDX expression
|
|
9
9
|
description = "PyTFBS: A Python Package for Transcription Factor Binding Site Prediction"
|
|
10
10
|
readme = "README.md"
|
|
@@ -25,6 +25,12 @@ dependencies = [
|
|
|
25
25
|
"numpy==1.26; sys_platform == 'darwin'",
|
|
26
26
|
"numpy>=2.0; sys_platform == 'win32'",
|
|
27
27
|
"numpy>=2.0; sys_platform == 'linux'",
|
|
28
|
+
"pandas==3.0; sys_platform == 'darwin'",
|
|
29
|
+
"pandas>=3.0; sys_platform == 'win32'",
|
|
30
|
+
"pandas>=3.0; sys_platform == 'linux'",
|
|
31
|
+
"scipy==1.18; sys_platform == 'darwin'",
|
|
32
|
+
"scipy==1.18; sys_platform == 'win32'",
|
|
33
|
+
"scipy==1.18; sys_platform == 'linux'",
|
|
28
34
|
]
|
|
29
35
|
|
|
30
36
|
[project.urls]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|