PyTFBS 1.0.5__tar.gz → 1.0.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyTFBS
3
- Version: 1.0.5
3
+ Version: 1.0.7
4
4
  Summary: PyTFBS: A Python Package for Transcription Factor Binding Site Prediction
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -21,6 +21,12 @@ Requires-Dist: torch
21
21
  Requires-Dist: numpy==1.26; sys_platform == "darwin"
22
22
  Requires-Dist: numpy>=2.0; sys_platform == "win32"
23
23
  Requires-Dist: numpy>=2.0; sys_platform == "linux"
24
+ Requires-Dist: pandas==3.0; sys_platform == "darwin"
25
+ Requires-Dist: pandas>=3.0; sys_platform == "win32"
26
+ Requires-Dist: pandas>=3.0; sys_platform == "linux"
27
+ Requires-Dist: scipy==1.18; sys_platform == "darwin"
28
+ Requires-Dist: scipy==1.18; sys_platform == "win32"
29
+ Requires-Dist: scipy==1.18; sys_platform == "linux"
24
30
  Provides-Extra: dev
25
31
  Requires-Dist: matplotlib; extra == "dev"
26
32
  Provides-Extra: test
@@ -29,18 +35,20 @@ Dynamic: license-file
29
35
 
30
36
  # PyTFBS
31
37
 
32
- A Python package for dpredicting transcription factor binding sites.
38
+ A Python package for predicting transcription factor binding sites.
33
39
 
34
40
  ## Installation
35
41
 
36
42
  ```bash
37
- pip install torch numpy PyTFBS
43
+ pip install torch numpy pandas scipy PyTFBS
38
44
  ```
39
45
 
40
46
  ## Usage
41
47
 
42
48
  ```python
43
- from PyTFBS import motif, predict
49
+ from PyTFBS import motif, predict, clink
50
+ import random
51
+ random.seed(42)
44
52
 
45
53
  # download PyTFBS data, only need to run once!!!
46
54
  motif.download_data()
@@ -66,4 +74,27 @@ predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_fil
66
74
  # the my_motif_dir should be organized as [[motif], [trace], [par]]
67
75
  predict.script('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', data_dir='my_motif_dir')
68
76
  predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', 64, data_dir='my_motif_dir')
77
+
78
+ # create the CLink TF-target set
79
+
80
+ # rexp_files = get_rexp_files()
81
+ # print(rexp_files)
82
+
83
+ df_rexps = clink.read_rexp_file()
84
+ tf_cnames = clink.read_tf_cname()
85
+ tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
86
+ tf_tars = clink.read_PyTFBS_output(tf_code + '_output.txt')
87
+
88
+ for tf_cname in tf_cnames[tf_code]:
89
+ if tf_cname[1] not in df_rexps.index:
90
+ continue
91
+ else:
92
+ tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
93
+ df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
94
+ tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
95
+ rcis = clink.rci(tar_exps, tf_exp)
96
+ for rci in rcis:
97
+ score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
98
+ print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
99
+
69
100
  ```
@@ -0,0 +1,214 @@
1
+ import os
2
+ import math
3
+ import pandas as pd
4
+ import numpy as np
5
+ np.random.seed(42)
6
+ from collections import defaultdict
7
+ import random
8
+ random.seed(42)
9
+ from scipy.stats import t
10
+
11
+ PyTFBS_data_dir = './PyTFBS_data'
12
+
13
+ def sign(x):
14
+ if x > 0:
15
+ return 1
16
+ elif x < 0:
17
+ return -1
18
+ else:
19
+ return 0
20
+
21
+ def weighted_corr(x1, y1, alpha):
22
+ """
23
+ 计算加权皮尔森相关系数及其p值。
24
+
25
+ 权重由每个点的 (|x| + |y|) / 2 计算,并进一步按 alpha 次幂调整。
26
+
27
+ 参数:
28
+ x, y : array-like
29
+ 输入变量。
30
+ alpha : float
31
+ 权重调整指数,控制权重对极端值的敏感度。
32
+
33
+ 返回:
34
+ r : float
35
+ 加权皮尔森相关系数。
36
+ p : float
37
+ 双侧检验的p值,检验原假设为总体相关系数为0。
38
+ 如果有效自由度 ≤ 0,则返回 p = np.nan。
39
+ """
40
+ x = np.array(x1)
41
+ y = np.array(y1)
42
+
43
+ # 计算基础权重(基于数值大小)
44
+ w = (np.abs(x) + np.abs(y)) / 2.0
45
+ # 调整权重,使其最大值缩放至1后再取alpha次幂
46
+ w_adj = (w / np.max(w)) ** alpha
47
+ # 归一化权重(和为1)
48
+ w_norm = w_adj / np.sum(w_adj)
49
+
50
+ # 加权均值
51
+ x_bar = np.sum(w_norm * x)
52
+ y_bar = np.sum(w_norm * y)
53
+
54
+ # 加权协方差与标准差
55
+ dx = x - x_bar
56
+ dy = y - y_bar
57
+ cov = np.sum(w_norm * dx * dy)
58
+ var_x = np.sum(w_norm * dx**2)
59
+ var_y = np.sum(w_norm * dy**2)
60
+ std_x = np.sqrt(var_x)
61
+ std_y = np.sqrt(var_y)
62
+
63
+ if std_x * std_y == 0.0:
64
+ r = 0.0
65
+ else:
66
+ r = cov / (std_x * std_y)
67
+
68
+ # ---- 计算p值(基于t分布近似) ----
69
+ # 有效样本量(权重之和的平方除以权重平方和)
70
+ sum_w = np.sum(w_adj) # 注意:w_adj未经归一化,但比例不变
71
+ sum_w2 = np.sum(w_adj**2)
72
+ n_eff = (sum_w ** 2) / sum_w2 if sum_w2 > 0 else 0.0
73
+
74
+ df = n_eff - 2 # 自由度
75
+ if df > 0 and abs(r) < 1.0:
76
+ # t统计量:t = r * sqrt(df / (1 - r^2))
77
+ t_stat = r * np.sqrt(df / (1 - r**2))
78
+ p = 2 * (1 - t.cdf(abs(t_stat), df))
79
+ else:
80
+ p = np.nan # 无法计算或相关系数为±1时,p值无意义
81
+
82
+ return r, p
83
+
84
+ def norm_std(data, sample=True):
85
+ """
86
+ 手动计算标准差
87
+ sample=True: 样本标准差(除以 n-1)
88
+ sample=False: 总体标准差(除以 n)
89
+ """
90
+ n = len(data)
91
+ if n < 2:
92
+ raise ValueError("need 2 obsvs")
93
+
94
+ # 步骤 1:计算平均值
95
+ # mean = sum(data) / n
96
+
97
+ # 步骤 2 & 3:计算每个数据与平均值的差的平方
98
+ # squared_diffs = [(x - mean) ** 2 for x in data]
99
+ squared_diffs = [x ** 2 for x in data]
100
+
101
+ # 步骤 4:计算方差
102
+ if sample:
103
+ variance = sum(squared_diffs) / (n - 1) # 样本方差
104
+ else:
105
+ variance = sum(squared_diffs) / n # 总体方差
106
+
107
+ # 步骤 5:开平方根得到标准差
108
+ std = variance ** 0.5
109
+
110
+ k = max(1, int(len(data) * 0.05)) # 至少取1个
111
+ sorted_lst = sorted([abs(i) for i in data], reverse=True)
112
+ max_obsv = sorted_lst[k - 1]
113
+ std_norm = std / max_obsv
114
+
115
+ return std_norm
116
+
117
+ def rci(tar_exps, tf_exp):
118
+ tar_rcis = []
119
+ for tar_name in tar_exps.keys():
120
+ tar_exp = tar_exps[tar_name]
121
+ corr_wt, p_value = weighted_corr(tar_exp, tf_exp, alpha=1.0)
122
+ tar_cv = norm_std(tar_exp)
123
+ if math.isnan(tar_cv):
124
+ tar_cv = 0.0
125
+ tf_cv = norm_std(tf_exp)
126
+ if math.isnan(tf_cv):
127
+ tf_cv = 0.0
128
+ if math.isnan(corr_wt) or p_value >= 0.05:
129
+ continue
130
+ else:
131
+ rci = sign(corr_wt) * (abs(corr_wt)**1.0 * tar_cv**0.5 * tf_cv**0.5) ** (1.0 / (1.0 + 0.5 + 0.5))
132
+ tar_rcis.append((tar_name, rci))
133
+ return tar_rcis
134
+
135
+ def read_tf_cname(file=None, data_dir=None):
136
+ if data_dir == None:
137
+ data_dir = PyTFBS_data_dir
138
+ if file == None:
139
+ file = data_dir + '/TF-code-name.txt'
140
+ else:
141
+ file = data_dir + '/' + file
142
+ if not os.path.exists(file):
143
+ print('TF-code-name.txt not exist, you may need to run download_data() first')
144
+ exit(1)
145
+ tf_cnames = defaultdict(list)
146
+ for line in open(file, 'r'):
147
+ lblocks = line.rstrip().split('\t')
148
+ tf_cnames[lblocks[1]].append((lblocks[2], lblocks[3]))
149
+ return tf_cnames
150
+
151
+ def read_rexp_file(file=None, data_dir=None):
152
+ if data_dir == None:
153
+ data_dir = PyTFBS_data_dir + '/Ref_Exps'
154
+ if file == None:
155
+ file = data_dir + '/Human_ED_matrix_1.5.txt'
156
+ else:
157
+ file = data_dir + '/' + file_name
158
+ if not os.path.exists(file):
159
+ print('Human_ED_matrix_1.5.txt not exist, you may need to run download_data() first')
160
+ exit(1)
161
+ df_rexps = pd.read_csv(file, sep='\t', header=0, index_col=0)
162
+ return df_rexps
163
+
164
+ def read_PyTFBS_output(file):
165
+ tf_tars = set()
166
+ for line in open(file):
167
+ if line[0] == '#':
168
+ continue
169
+ lblocks = line.rstrip().split('\t')
170
+ gene_name = lblocks[0].split('::')[0].split('|')[2]
171
+ score = float(lblocks[3])
172
+ jindex = float(lblocks[2])
173
+ if score > 0.8 and jindex > 0.3:
174
+ tf_tars.add(gene_name)
175
+ return tf_tars
176
+
177
+ def get_rexp_files(species=None, data_dir=None):
178
+ if data_dir == None:
179
+ data_dir = PyTFBS_data_dir
180
+ index_file = data_dir + '/RExps_index.txt'
181
+ if not os.path.exists(index_file):
182
+ print('RExps_index.txt not exist, you may need to run download_data() first')
183
+ exit(1)
184
+ header = ''
185
+ result = []
186
+ for line in open(index_file, 'r'):
187
+ if line[0] == "#":
188
+ header = line.rstrip()
189
+ continue
190
+ lblocks = line.rstrip().split('\t')
191
+ if species != None and lblocks[1] != species:
192
+ continue
193
+ result.append(lblocks[0])
194
+ return result
195
+
196
+ if __name__ == '__main__':
197
+ rexp_files = get_rexp_files()
198
+ print(rexp_files)
199
+ df_rexps = read_rexp_file()
200
+ tf_cnames = read_tf_cname()
201
+ tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
202
+ tf_tars = read_PyTFBS_output(tf_code + '_output.txt')
203
+
204
+ for tf_cname in tf_cnames[tf_code]:
205
+ if tf_cname[1] not in df_rexps.index:
206
+ continue
207
+ else:
208
+ tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
209
+ df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
210
+ tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
211
+ rcis = rci(tar_exps, tf_exp)
212
+ for rci in rcis:
213
+ score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
214
+ print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyTFBS
3
- Version: 1.0.5
3
+ Version: 1.0.7
4
4
  Summary: PyTFBS: A Python Package for Transcription Factor Binding Site Prediction
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -21,6 +21,12 @@ Requires-Dist: torch
21
21
  Requires-Dist: numpy==1.26; sys_platform == "darwin"
22
22
  Requires-Dist: numpy>=2.0; sys_platform == "win32"
23
23
  Requires-Dist: numpy>=2.0; sys_platform == "linux"
24
+ Requires-Dist: pandas==3.0; sys_platform == "darwin"
25
+ Requires-Dist: pandas>=3.0; sys_platform == "win32"
26
+ Requires-Dist: pandas>=3.0; sys_platform == "linux"
27
+ Requires-Dist: scipy==1.18; sys_platform == "darwin"
28
+ Requires-Dist: scipy==1.18; sys_platform == "win32"
29
+ Requires-Dist: scipy==1.18; sys_platform == "linux"
24
30
  Provides-Extra: dev
25
31
  Requires-Dist: matplotlib; extra == "dev"
26
32
  Provides-Extra: test
@@ -29,18 +35,20 @@ Dynamic: license-file
29
35
 
30
36
  # PyTFBS
31
37
 
32
- A Python package for dpredicting transcription factor binding sites.
38
+ A Python package for predicting transcription factor binding sites.
33
39
 
34
40
  ## Installation
35
41
 
36
42
  ```bash
37
- pip install torch numpy PyTFBS
43
+ pip install torch numpy pandas scipy PyTFBS
38
44
  ```
39
45
 
40
46
  ## Usage
41
47
 
42
48
  ```python
43
- from PyTFBS import motif, predict
49
+ from PyTFBS import motif, predict, clink
50
+ import random
51
+ random.seed(42)
44
52
 
45
53
  # download PyTFBS data, only need to run once!!!
46
54
  motif.download_data()
@@ -66,4 +74,27 @@ predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_fil
66
74
  # the my_motif_dir should be organized as [[motif], [trace], [par]]
67
75
  predict.script('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', data_dir='my_motif_dir')
68
76
  predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', 64, data_dir='my_motif_dir')
77
+
78
+ # create the CLink TF-target set
79
+
80
+ # rexp_files = get_rexp_files()
81
+ # print(rexp_files)
82
+
83
+ df_rexps = clink.read_rexp_file()
84
+ tf_cnames = clink.read_tf_cname()
85
+ tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
86
+ tf_tars = clink.read_PyTFBS_output(tf_code + '_output.txt')
87
+
88
+ for tf_cname in tf_cnames[tf_code]:
89
+ if tf_cname[1] not in df_rexps.index:
90
+ continue
91
+ else:
92
+ tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
93
+ df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
94
+ tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
95
+ rcis = clink.rci(tar_exps, tf_exp)
96
+ for rci in rcis:
97
+ score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
98
+ print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
99
+
69
100
  ```
@@ -2,6 +2,7 @@ LICENSE
2
2
  README.md
3
3
  pyproject.toml
4
4
  PyTFBS/__init__.py
5
+ PyTFBS/clink.py
5
6
  PyTFBS/motif.py
6
7
  PyTFBS/predict.py
7
8
  PyTFBS.egg-info/PKG-INFO
@@ -2,12 +2,18 @@ torch
2
2
 
3
3
  [:sys_platform == "darwin"]
4
4
  numpy==1.26
5
+ pandas==3.0
6
+ scipy==1.18
5
7
 
6
8
  [:sys_platform == "linux"]
7
9
  numpy>=2.0
10
+ pandas>=3.0
11
+ scipy==1.18
8
12
 
9
13
  [:sys_platform == "win32"]
10
14
  numpy>=2.0
15
+ pandas>=3.0
16
+ scipy==1.18
11
17
 
12
18
  [dev]
13
19
  matplotlib
@@ -1,17 +1,19 @@
1
1
  # PyTFBS
2
2
 
3
- A Python package for dpredicting transcription factor binding sites.
3
+ A Python package for predicting transcription factor binding sites.
4
4
 
5
5
  ## Installation
6
6
 
7
7
  ```bash
8
- pip install torch numpy PyTFBS
8
+ pip install torch numpy pandas scipy PyTFBS
9
9
  ```
10
10
 
11
11
  ## Usage
12
12
 
13
13
  ```python
14
- from PyTFBS import motif, predict
14
+ from PyTFBS import motif, predict, clink
15
+ import random
16
+ random.seed(42)
15
17
 
16
18
  # download PyTFBS data, only need to run once!!!
17
19
  motif.download_data()
@@ -37,4 +39,27 @@ predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_fil
37
39
  # the my_motif_dir should be organized as [[motif], [trace], [par]]
38
40
  predict.script('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', data_dir='my_motif_dir')
39
41
  predict.win_bin('CEBPB_HUMAN.H11MO.0.A', 'CEBPB_HUMAN.H11MO.0.A', 'input_seq_file.fasta', 64, data_dir='my_motif_dir')
42
+
43
+ # create the CLink TF-target set
44
+
45
+ # rexp_files = get_rexp_files()
46
+ # print(rexp_files)
47
+
48
+ df_rexps = clink.read_rexp_file()
49
+ tf_cnames = clink.read_tf_cname()
50
+ tf_code = 'CREB1_HUMAN.H11MO.0.A_RC'
51
+ tf_tars = clink.read_PyTFBS_output(tf_code + '_output.txt')
52
+
53
+ for tf_cname in tf_cnames[tf_code]:
54
+ if tf_cname[1] not in df_rexps.index:
55
+ continue
56
+ else:
57
+ tf_exp = df_rexps.loc[tf_cname[1]].values.tolist()
58
+ df_rexps_tars = df_rexps[df_rexps.index.isin(tf_tars)]
59
+ tar_exps = {idx: row.tolist() for idx, row in df_rexps_tars.iterrows()}
60
+ rcis = clink.rci(tar_exps, tf_exp)
61
+ for rci in rcis:
62
+ score = rci[1] * (1 + (random.uniform(1E-5, 1.0) / 1E10))
63
+ print(tf_cname[0] + '\t' + rci[0] + '\t' + str(score))
64
+
40
65
  ```
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "PyTFBS"
7
- version = "1.0.5"
7
+ version = "1.0.7"
8
8
  license = "MIT" # SPDX expression
9
9
  description = "PyTFBS: A Python Package for Transcription Factor Binding Site Prediction"
10
10
  readme = "README.md"
@@ -25,6 +25,12 @@ dependencies = [
25
25
  "numpy==1.26; sys_platform == 'darwin'",
26
26
  "numpy>=2.0; sys_platform == 'win32'",
27
27
  "numpy>=2.0; sys_platform == 'linux'",
28
+ "pandas==3.0; sys_platform == 'darwin'",
29
+ "pandas>=3.0; sys_platform == 'win32'",
30
+ "pandas>=3.0; sys_platform == 'linux'",
31
+ "scipy==1.18; sys_platform == 'darwin'",
32
+ "scipy==1.18; sys_platform == 'win32'",
33
+ "scipy==1.18; sys_platform == 'linux'",
28
34
  ]
29
35
 
30
36
  [project.urls]
File without changes
File without changes
File without changes
File without changes
File without changes