PyOVERCAST 1.0.8__tar.gz → 1.0.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyOVERCAST
3
- Version: 1.0.8
3
+ Version: 1.0.9
4
4
  Summary: A Python package for mining key transcription factors from transcriptome data.
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -78,6 +78,9 @@ if __name__ == '__main__':
78
78
  # or predict one DEG-list with bootstrap
79
79
  result = predict.olcr_bootstrap(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
80
80
 
81
+ # or predict one DEG-list with bootstrap and permutation
82
+ result = predict.olcr_bootstrap_permutation(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
83
+
81
84
  # save result to text file
82
85
  result.to_csv('output.txt', sep='\t', index=False, encoding='utf-8')
83
86
 
@@ -13,8 +13,8 @@ plt.rcParams['font.sans-serif'] = ['SimHei', 'Microsoft YaHei', 'PingFang SC', '
13
13
  plt.rcParams['axes.unicode_minus'] = False # 解决负号显示问题
14
14
  import multiprocessing as mp
15
15
  import seaborn as sns
16
-
17
16
  from scipy import stats
17
+ import random
18
18
 
19
19
  overcast_data_dir = './OVERCAST_data'
20
20
 
@@ -520,6 +520,36 @@ def rrho_coor_bs(df1, df2, win, bs_n):
520
520
 
521
521
  return beta_xy, pval_xy, DTCI, bs_pval
522
522
 
523
+ def permutation(gene_set, gene_list):
524
+ random.seed(42)
525
+ gene_set0 = gene_set[gene_set.index.isin(gene_list.index)]
526
+ gene_list0 = gene_list[gene_list.index.isin(gene_set0.index)]
527
+
528
+ diff_idx = list(gene_list.index.difference(gene_list0.index))
529
+ if len(diff_idx) == 0:
530
+ gene_sample = []
531
+ else:
532
+ gene_sample = random.sample(diff_idx, gene_set0.shape[0])
533
+
534
+ set1_values = []
535
+ set0_max, set0_min = gene_set0[gene_set0.columns[1]].max(), gene_set0[gene_set0.columns[1]].min()
536
+ set0_range = set0_max - set0_min
537
+ if set0_range < 0.0:
538
+ set0_range = set0_min
539
+ for gene in gene_sample:
540
+ set1_values.append(random.uniform(set0_min - set0_range, set0_min))
541
+
542
+ if len(gene_sample) == 0:
543
+ gene_set_pt = gene_set0
544
+ gene_list_pt = gene_list0
545
+ else:
546
+ gene_set1 = pd.DataFrame({"gene": gene_sample, "logFC": set1_values}, index=gene_sample)
547
+ gene_set_pt = pd.concat([gene_set0, gene_set1], axis=0)
548
+ gene_list1 = gene_list[gene_list.index.isin(gene_sample)]
549
+ gene_list_pt = pd.concat([gene_list0, gene_list1], axis=0)
550
+
551
+ return gene_set_pt, gene_list_pt
552
+
523
553
  def thread_one(tfbs_array, deg_list, win, shared_list, nn, lock):
524
554
  tf_n = len(tfbs_array)
525
555
  epsilon = 1e-15 # 常用值
@@ -546,6 +576,19 @@ def thread_one_bs(tfbs_array, deg_list, win, bs_n, shared_list, nn, lock):
546
576
  nn[1] += 1
547
577
  print(str(nn[1]) + ' / ' + str(nn[0]))
548
578
 
579
+ def thread_one_bs_pt(tfbs_array, deg_list, win, bs_n, shared_list, nn, lock):
580
+ tf_n = len(tfbs_array)
581
+ epsilon = 1e-15 # 常用值
582
+ for i in range(tf_n):
583
+ # fin_n += 1
584
+ gene_set_pt, gene_list_pt = permutation(tfbs_array[i][1], deg_list)
585
+ beta_xy, pval_xy, DTCI, bs_pval = rrho_coor_bs(gene_list_pt, gene_set_pt, win, bs_n)
586
+ rank_score = (DTCI * min(1.0, -np.log10(max(bs_pval, 1e-5)) / 5.0)) ** 0.5
587
+ with lock:
588
+ shared_list.append([tfbs_array[i][0], beta_xy, pval_xy, DTCI, bs_pval, rank_score])
589
+ nn[1] += 1
590
+ print(str(nn[1]) + ' / ' + str(nn[0]))
591
+
549
592
  def split_into_n_even(lst, n):
550
593
  nn = len(lst)
551
594
  base = nn // n # 每份基础大小
@@ -670,6 +713,62 @@ def olcr_bootstrap(set_names=None, list_file=None, win=30, thread_n=16, bs_n=200
670
713
 
671
714
  return df
672
715
 
716
+ def olcr_bootstrap_permutation(set_names=None, list_file=None, win=30, thread_n=16, bs_n=2000, data_dir=None):
717
+ if data_dir == None:
718
+ data_dir = overcast_data_dir
719
+ if not (os.path.exists(data_dir) and os.path.isdir(data_dir)):
720
+ print("OVERCAST_data folder can not found!")
721
+ exit(1)
722
+ sets_files = []
723
+ for set1 in set_names:
724
+ sets_file = data_dir + '/TF-target_sets/' + set1 + '.txt'
725
+ if not os.path.exists(sets_file):
726
+ print(sets_file, 'not exist!')
727
+ exit(1)
728
+ else:
729
+ sets_files.append(sets_file)
730
+ degl_file = list_file
731
+ if not os.path.exists(degl_file):
732
+ print(degl_file, 'not exist!')
733
+ exit(1)
734
+
735
+ tfbs_array = read_gene_sets(sets_files)
736
+ deg_list = read_DEG_list(degl_file)
737
+
738
+ manager = mp.Manager()
739
+ shared_list = manager.list([]) # 支持任意类型
740
+ nn = manager.list([len(tfbs_array), 0]) # 支持任意类型
741
+ lock = manager.Lock()
742
+
743
+ tfbs_array_splits = split_into_n_even(tfbs_array, thread_n)
744
+ task_args = []
745
+ for ti in range(thread_n):
746
+ task_args.append((tfbs_array_splits[ti], deg_list, win, bs_n, shared_list, nn, lock))
747
+
748
+ # 创建进程池
749
+ with mp.Pool(processes=thread_n) as pool:
750
+ print(f"thrend_n:", thread_n)
751
+ print(f"total {len(task_args)} tasks")
752
+ # starmap 会自动分配任务到进程池
753
+ start_time = time.time()
754
+ results = pool.starmap(thread_one_bs_pt, task_args)
755
+ elapsed = time.time() - start_time
756
+ # 打印结果
757
+ print("time consumed:", elapsed)
758
+
759
+ #sorted_results = sorted(shared_list, key=lambda x: (x[-1], -x[-2]), reverse=False) # lambda x: (x[4], -x[1]), reverse=False
760
+ sorted_results = sorted(shared_list, key=lambda x: x[-1], reverse=True) # lambda x: (x[4], -x[1]), reverse=False
761
+ results = []
762
+ ln = len(sorted_results)
763
+ for li in range(len(sorted_results)):
764
+ ratio = (li + 1) / ln
765
+ rline = sorted_results[li] + [ratio]
766
+ results.append(rline)
767
+ col_names = ['TF_motif', 'Beta3', 'Beta3_Pvalue', 'DTCI', 'DTCI_Pvalue', 'Rank_Score', 'Rank / N']
768
+ df = pd.DataFrame(results, columns=col_names)
769
+
770
+ return df
771
+
673
772
  def olcr_bootstrap_1cpu(set_names=None, list_file=None, win=30, thread_n=16, bs_n=2000, data_dir=None):
674
773
  if data_dir == None:
675
774
  data_dir = overcast_data_dir
@@ -866,6 +965,37 @@ def plot_contour(set_names=['human_hocomoco_CLink_wtcoor_1w_0.8'], list_file='in
866
965
  plot_contour_plot(deg_list, tfbs_array[i][1], win)
867
966
  break
868
967
 
968
+ def plot_contour_permutation(set_names=['human_hocomoco_CLink_wtcoor_1w_0.8'], list_file='input_deg-list.txt', tf='NFKB1_HUMAN.H11MO.1.B', win=30, data_dir=None):
969
+ if data_dir == None:
970
+ data_dir = overcast_data_dir
971
+ if not (os.path.exists(data_dir) and os.path.isdir(data_dir)):
972
+ print("OVERCAST_data folder can not found!")
973
+ exit(1)
974
+ sets_files = []
975
+ for set1 in set_names:
976
+ sets_file = data_dir + '/TF-target_sets/' + set1 + '.txt'
977
+ if not os.path.exists(sets_file):
978
+ print(sets_file, 'not exist!')
979
+ exit(1)
980
+ else:
981
+ sets_files.append(sets_file)
982
+ degl_file = list_file
983
+ if not os.path.exists(degl_file):
984
+ print(degl_file, 'not exist!')
985
+ exit(1)
986
+
987
+ tfbs_array = read_gene_sets(sets_files)
988
+ deg_list = read_DEG_list(degl_file)
989
+
990
+ tf_n = len(tfbs_array)
991
+ epsilon = 1e-15 # 常用值
992
+ for i in range(tf_n):
993
+ if tfbs_array[i][0] == tf:
994
+ gene_set_pt, gene_list_pt = permutation(tfbs_array[i][1], deg_list)
995
+ plot_contour_plot(gene_list_pt, gene_set_pt, win)
996
+ # plot_contour_plot(deg_list, tfbs_array[i][1], win)
997
+ break
998
+
869
999
  if __name__ == '__main__':
870
1000
  '''
871
1001
  result = olcr(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='OVERCAST_data/input_deg-list.txt', win=30, thread_n=16)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyOVERCAST
3
- Version: 1.0.8
3
+ Version: 1.0.9
4
4
  Summary: A Python package for mining key transcription factors from transcriptome data.
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -78,6 +78,9 @@ if __name__ == '__main__':
78
78
  # or predict one DEG-list with bootstrap
79
79
  result = predict.olcr_bootstrap(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
80
80
 
81
+ # or predict one DEG-list with bootstrap and permutation
82
+ result = predict.olcr_bootstrap_permutation(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
83
+
81
84
  # save result to text file
82
85
  result.to_csv('output.txt', sep='\t', index=False, encoding='utf-8')
83
86
 
@@ -35,6 +35,9 @@ if __name__ == '__main__':
35
35
  # or predict one DEG-list with bootstrap
36
36
  result = predict.olcr_bootstrap(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
37
37
 
38
+ # or predict one DEG-list with bootstrap and permutation
39
+ result = predict.olcr_bootstrap_permutation(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
40
+
38
41
  # save result to text file
39
42
  result.to_csv('output.txt', sep='\t', index=False, encoding='utf-8')
40
43
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "PyOVERCAST"
7
- version = "1.0.8"
7
+ version = "1.0.9"
8
8
  license = "MIT" # SPDX expression
9
9
  description = "A Python package for mining key transcription factors from transcriptome data."
10
10
  readme = "README.md"
File without changes
File without changes