PyOVERCAST 1.0.8__tar.gz → 1.0.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyOVERCAST
3
- Version: 1.0.8
3
+ Version: 1.0.10
4
4
  Summary: A Python package for mining key transcription factors from transcriptome data.
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -78,6 +78,9 @@ if __name__ == '__main__':
78
78
  # or predict one DEG-list with bootstrap
79
79
  result = predict.olcr_bootstrap(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
80
80
 
81
+ # or predict one DEG-list with bootstrap and permutation
82
+ result = predict.olcr_bootstrap_permutation(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
83
+
81
84
  # save result to text file
82
85
  result.to_csv('output.txt', sep='\t', index=False, encoding='utf-8')
83
86
 
@@ -13,8 +13,8 @@ plt.rcParams['font.sans-serif'] = ['SimHei', 'Microsoft YaHei', 'PingFang SC', '
13
13
  plt.rcParams['axes.unicode_minus'] = False # 解决负号显示问题
14
14
  import multiprocessing as mp
15
15
  import seaborn as sns
16
-
17
16
  from scipy import stats
17
+ import random
18
18
 
19
19
  overcast_data_dir = './OVERCAST_data'
20
20
 
@@ -520,6 +520,29 @@ def rrho_coor_bs(df1, df2, win, bs_n):
520
520
 
521
521
  return beta_xy, pval_xy, DTCI, bs_pval
522
522
 
523
+ def permutation(gene_set, gene_list):
524
+ random.seed(42)
525
+ gene_set0 = gene_set[gene_set.index.isin(gene_list.index)]
526
+ gene_list0 = gene_list[gene_list.index.isin(gene_set0.index)]
527
+
528
+ diff_idx = list(gene_list.index.difference(gene_list0.index))
529
+ gene_sample = random.sample(diff_idx, min(gene_set0.shape[0], len(diff_idx)))
530
+
531
+ set1_values = []
532
+ set0_max, set0_min = gene_set0[gene_set0.columns[1]].max(), gene_set0[gene_set0.columns[1]].min()
533
+ set0_range = set0_max - set0_min
534
+ if set0_range < 0.0:
535
+ set0_range = set0_min
536
+ for gene in gene_sample:
537
+ set1_values.append(random.uniform(set0_min - set0_range, set0_min))
538
+
539
+ gene_set1 = pd.DataFrame({"gene": gene_sample, "logFC": set1_values}, index=gene_sample)
540
+ gene_set_pt = pd.concat([gene_set0, gene_set1], axis=0)
541
+ gene_list1 = gene_list[gene_list.index.isin(gene_sample)]
542
+ gene_list_pt = pd.concat([gene_list0, gene_list1], axis=0)
543
+
544
+ return gene_set_pt, gene_list_pt
545
+
523
546
  def thread_one(tfbs_array, deg_list, win, shared_list, nn, lock):
524
547
  tf_n = len(tfbs_array)
525
548
  epsilon = 1e-15 # 常用值
@@ -546,6 +569,19 @@ def thread_one_bs(tfbs_array, deg_list, win, bs_n, shared_list, nn, lock):
546
569
  nn[1] += 1
547
570
  print(str(nn[1]) + ' / ' + str(nn[0]))
548
571
 
572
+ def thread_one_bs_pt(tfbs_array, deg_list, win, bs_n, shared_list, nn, lock):
573
+ tf_n = len(tfbs_array)
574
+ epsilon = 1e-15 # 常用值
575
+ for i in range(tf_n):
576
+ # fin_n += 1
577
+ gene_set_pt, gene_list_pt = permutation(tfbs_array[i][1], deg_list)
578
+ beta_xy, pval_xy, DTCI, bs_pval = rrho_coor_bs(gene_list_pt, gene_set_pt, win, bs_n)
579
+ rank_score = (DTCI * min(1.0, -np.log10(max(bs_pval, 1e-5)) / 5.0)) ** 0.5
580
+ with lock:
581
+ shared_list.append([tfbs_array[i][0], beta_xy, pval_xy, DTCI, bs_pval, rank_score])
582
+ nn[1] += 1
583
+ print(str(nn[1]) + ' / ' + str(nn[0]))
584
+
549
585
  def split_into_n_even(lst, n):
550
586
  nn = len(lst)
551
587
  base = nn // n # 每份基础大小
@@ -670,6 +706,62 @@ def olcr_bootstrap(set_names=None, list_file=None, win=30, thread_n=16, bs_n=200
670
706
 
671
707
  return df
672
708
 
709
+ def olcr_bootstrap_permutation(set_names=None, list_file=None, win=30, thread_n=16, bs_n=2000, data_dir=None):
710
+ if data_dir == None:
711
+ data_dir = overcast_data_dir
712
+ if not (os.path.exists(data_dir) and os.path.isdir(data_dir)):
713
+ print("OVERCAST_data folder can not found!")
714
+ exit(1)
715
+ sets_files = []
716
+ for set1 in set_names:
717
+ sets_file = data_dir + '/TF-target_sets/' + set1 + '.txt'
718
+ if not os.path.exists(sets_file):
719
+ print(sets_file, 'not exist!')
720
+ exit(1)
721
+ else:
722
+ sets_files.append(sets_file)
723
+ degl_file = list_file
724
+ if not os.path.exists(degl_file):
725
+ print(degl_file, 'not exist!')
726
+ exit(1)
727
+
728
+ tfbs_array = read_gene_sets(sets_files)
729
+ deg_list = read_DEG_list(degl_file)
730
+
731
+ manager = mp.Manager()
732
+ shared_list = manager.list([]) # 支持任意类型
733
+ nn = manager.list([len(tfbs_array), 0]) # 支持任意类型
734
+ lock = manager.Lock()
735
+
736
+ tfbs_array_splits = split_into_n_even(tfbs_array, thread_n)
737
+ task_args = []
738
+ for ti in range(thread_n):
739
+ task_args.append((tfbs_array_splits[ti], deg_list, win, bs_n, shared_list, nn, lock))
740
+
741
+ # 创建进程池
742
+ with mp.Pool(processes=thread_n) as pool:
743
+ print(f"thrend_n:", thread_n)
744
+ print(f"total {len(task_args)} tasks")
745
+ # starmap 会自动分配任务到进程池
746
+ start_time = time.time()
747
+ results = pool.starmap(thread_one_bs_pt, task_args)
748
+ elapsed = time.time() - start_time
749
+ # 打印结果
750
+ print("time consumed:", elapsed)
751
+
752
+ #sorted_results = sorted(shared_list, key=lambda x: (x[-1], -x[-2]), reverse=False) # lambda x: (x[4], -x[1]), reverse=False
753
+ sorted_results = sorted(shared_list, key=lambda x: x[-1], reverse=True) # lambda x: (x[4], -x[1]), reverse=False
754
+ results = []
755
+ ln = len(sorted_results)
756
+ for li in range(len(sorted_results)):
757
+ ratio = (li + 1) / ln
758
+ rline = sorted_results[li] + [ratio]
759
+ results.append(rline)
760
+ col_names = ['TF_motif', 'Beta3', 'Beta3_Pvalue', 'DTCI', 'DTCI_Pvalue', 'Rank_Score', 'Rank / N']
761
+ df = pd.DataFrame(results, columns=col_names)
762
+
763
+ return df
764
+
673
765
  def olcr_bootstrap_1cpu(set_names=None, list_file=None, win=30, thread_n=16, bs_n=2000, data_dir=None):
674
766
  if data_dir == None:
675
767
  data_dir = overcast_data_dir
@@ -866,6 +958,37 @@ def plot_contour(set_names=['human_hocomoco_CLink_wtcoor_1w_0.8'], list_file='in
866
958
  plot_contour_plot(deg_list, tfbs_array[i][1], win)
867
959
  break
868
960
 
961
+ def plot_contour_permutation(set_names=['human_hocomoco_CLink_wtcoor_1w_0.8'], list_file='input_deg-list.txt', tf='NFKB1_HUMAN.H11MO.1.B', win=30, data_dir=None):
962
+ if data_dir == None:
963
+ data_dir = overcast_data_dir
964
+ if not (os.path.exists(data_dir) and os.path.isdir(data_dir)):
965
+ print("OVERCAST_data folder can not found!")
966
+ exit(1)
967
+ sets_files = []
968
+ for set1 in set_names:
969
+ sets_file = data_dir + '/TF-target_sets/' + set1 + '.txt'
970
+ if not os.path.exists(sets_file):
971
+ print(sets_file, 'not exist!')
972
+ exit(1)
973
+ else:
974
+ sets_files.append(sets_file)
975
+ degl_file = list_file
976
+ if not os.path.exists(degl_file):
977
+ print(degl_file, 'not exist!')
978
+ exit(1)
979
+
980
+ tfbs_array = read_gene_sets(sets_files)
981
+ deg_list = read_DEG_list(degl_file)
982
+
983
+ tf_n = len(tfbs_array)
984
+ epsilon = 1e-15 # 常用值
985
+ for i in range(tf_n):
986
+ if tfbs_array[i][0] == tf:
987
+ gene_set_pt, gene_list_pt = permutation(tfbs_array[i][1], deg_list)
988
+ plot_contour_plot(gene_list_pt, gene_set_pt, win)
989
+ # plot_contour_plot(deg_list, tfbs_array[i][1], win)
990
+ break
991
+
869
992
  if __name__ == '__main__':
870
993
  '''
871
994
  result = olcr(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='OVERCAST_data/input_deg-list.txt', win=30, thread_n=16)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyOVERCAST
3
- Version: 1.0.8
3
+ Version: 1.0.10
4
4
  Summary: A Python package for mining key transcription factors from transcriptome data.
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -78,6 +78,9 @@ if __name__ == '__main__':
78
78
  # or predict one DEG-list with bootstrap
79
79
  result = predict.olcr_bootstrap(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
80
80
 
81
+ # or predict one DEG-list with bootstrap and permutation
82
+ result = predict.olcr_bootstrap_permutation(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
83
+
81
84
  # save result to text file
82
85
  result.to_csv('output.txt', sep='\t', index=False, encoding='utf-8')
83
86
 
@@ -35,6 +35,9 @@ if __name__ == '__main__':
35
35
  # or predict one DEG-list with bootstrap
36
36
  result = predict.olcr_bootstrap(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
37
37
 
38
+ # or predict one DEG-list with bootstrap and permutation
39
+ result = predict.olcr_bootstrap_permutation(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./OVERCAST_data/input_deg-list.txt', win=30, bs_n=1000, thread_n=32)
40
+
38
41
  # save result to text file
39
42
  result.to_csv('output.txt', sep='\t', index=False, encoding='utf-8')
40
43
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "PyOVERCAST"
7
- version = "1.0.8"
7
+ version = "1.0.10"
8
8
  license = "MIT" # SPDX expression
9
9
  description = "A Python package for mining key transcription factors from transcriptome data."
10
10
  readme = "README.md"
File without changes
File without changes