PyOVERCAST 1.0.4__tar.gz → 1.0.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyOVERCAST
3
- Version: 1.0.4
3
+ Version: 1.0.5
4
4
  Summary: A Python package for mining key transcription factors from transcriptome data.
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -84,6 +84,9 @@ if __name__ == '__main__':
84
84
  # plot OLC matrix
85
85
  predict.plot_olc(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
86
86
 
87
+ # plot OLC matrix
88
+ predict.plot_olc(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
89
+
87
90
  # plot fitted 3D U-surface
88
- predict.plot_fit3D(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
91
+ predict.plot_contour(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
89
92
  ```
@@ -353,7 +353,7 @@ def plot_rrho(rrho_df, out_file=None, title="RRLO Matrix", cmap="magma"):
353
353
  """
354
354
  Plot RRHO heatmap.
355
355
  """
356
- plt.figure(figsize=(6, 5), dpi=150)
356
+ plt.figure(figsize=(8, 6))
357
357
  ax = sns.heatmap(
358
358
  rrho_df,
359
359
  cmap=cmap,
@@ -363,8 +363,8 @@ def plot_rrho(rrho_df, out_file=None, title="RRLO Matrix", cmap="magma"):
363
363
  yticklabels=max(1, len(rrho_df.index) // 10),
364
364
  )
365
365
  ax.invert_yaxis() # 添加这一行,让 [0,0] 位于左下角
366
- ax.set_xlabel("Prefix overlap in TF-target set")
367
- ax.set_ylabel("Prefix overlap in DEG list")
366
+ ax.set_xlabel("Overlap in TF-target set")
367
+ ax.set_ylabel("Overlap in DEG list")
368
368
  ax.set_title(title, fontsize=10)
369
369
  plt.tight_layout()
370
370
 
@@ -762,6 +762,110 @@ def plot_fit3D(set_names=['human_hocomoco_CLink_wtcoor_1w_0.8'], list_file='inpu
762
762
  plot_fit3D_plot(deg_list, tfbs_array[i][1], win)
763
763
  break
764
764
 
765
+ def plot_contour_graph(rrho_df, out_file=None):
766
+ Z_data = rrho_df.to_numpy()
767
+ row_n, col_n = Z_data.shape
768
+ #x = np.linspace(-1, 1, col_n)
769
+ #y = np.linspace(-1, 1, row_n)
770
+ y = rrho_df.index
771
+ x = rrho_df.columns
772
+ X, Y = np.meshgrid(x, y)
773
+
774
+ #x = np.linspace(-5, 5, 400)
775
+ #y = np.linspace(-5, 5, 400)
776
+ #X, Y = np.meshgrid(x, y)
777
+ #Z = np.sin(X) * np.cos(Y) * np.exp(-(X**2 + Y**2) / 10)
778
+
779
+ def fit_surface(Z, X, Y):
780
+ df = pd.DataFrame({
781
+ 'x': X.ravel(),
782
+ 'y': Y.ravel(),
783
+ 'z': Z.ravel()
784
+ })
785
+ df['xy'] = df['x'] * df['y'] # 交互项
786
+ X_design = sm.add_constant(df[['x', 'y', 'xy']])
787
+ y_dep = df['z']
788
+ model = sm.OLS(y_dep, X_design).fit()
789
+ beta = model.params
790
+ # print(beta)
791
+
792
+ Z_fit = beta['const'] + beta['x']*X + beta['y']*Y + beta['xy']*X*Y
793
+ return beta, Z_fit
794
+
795
+ beta, Z = fit_surface(Z_data, X, Y)
796
+
797
+ fig, ax = plt.subplots(figsize=(8, 6))
798
+ levels = np.linspace(Z.min(), Z.max(), 16)
799
+ cs = ax.contour(X, Y, Z, levels=levels, cmap='viridis')
800
+ ax.clabel(cs, inline=True, fontsize=8, fmt='%.2f')
801
+ cf = ax.contourf(X, Y, Z, levels=levels, cmap='viridis', alpha=0.35)
802
+ fig.colorbar(cf, ax=ax, label='z value')
803
+ ax.set_xlabel('Rank of Gene Set'); ax.set_ylabel('Rank of DEG List')
804
+ ax.set_title('Contour Graph of RRLO Matrix')
805
+ # ax.set_aspect('equal')
806
+ plt.tight_layout()
807
+ plt.savefig('contour_graph_invF.png', dpi=150)
808
+ plt.show()
809
+
810
+
811
+ def plot_contour_plot(df1, df2, win):
812
+ gene_col1 = df1.columns[0]
813
+ score_col1 = df1.columns[1]
814
+ df1 = df1.sort_values(score_col1, ascending=False)
815
+ genes1 = df1[gene_col1].astype(str).tolist()
816
+
817
+ # remove duplicates while preserving order
818
+ genes1 = list(dict.fromkeys(genes1))
819
+
820
+ gene_col2 = df1.columns[0]
821
+ score_col2 = df2.columns[1]
822
+ df2 = df2.sort_values(score_col2, ascending=False)
823
+ genes2 = df2[gene_col2].astype(str).tolist()
824
+ # remove duplicates while preserving order
825
+ genes2 = list(dict.fromkeys(genes2))
826
+
827
+ window = win
828
+ if (len(genes1) < window*3 or len(genes2) < window*3):
829
+ return 0.0, 1.0, 0.0
830
+ rrho_df = rrho_matrix_str(genes1, genes2, window=window)
831
+
832
+ if rrho_df.shape[0] == 0 or rrho_df.shape[1] == 0 or (rrho_df > 0.0).sum().sum() == 0:
833
+ print('no overlaps found')
834
+ exit(1)
835
+
836
+ #beta_xy, pval_xy = td_coor(rrho_df.to_numpy())
837
+ # df = Z_to_frame(rrho_df.to_numpy())
838
+ plot_contour_graph(rrho_df)
839
+
840
+ def plot_contour(set_names=['human_hocomoco_CLink_wtcoor_1w_0.8'], list_file='input_deg-list.txt', tf='NFKB1_HUMAN.H11MO.1.B', win=30, data_dir=None):
841
+ if data_dir == None:
842
+ data_dir = overcast_data_dir
843
+ if not (os.path.exists(data_dir) and os.path.isdir(data_dir)):
844
+ print("OVERCAST_data folder can not found!")
845
+ exit(1)
846
+ sets_files = []
847
+ for set1 in set_names:
848
+ sets_file = data_dir + '/TF-target_sets/' + set1 + '.txt'
849
+ if not os.path.exists(sets_file):
850
+ print(sets_file, 'not exist!')
851
+ exit(1)
852
+ else:
853
+ sets_files.append(sets_file)
854
+ degl_file = list_file
855
+ if not os.path.exists(degl_file):
856
+ print(degl_file, 'not exist!')
857
+ exit(1)
858
+
859
+ tfbs_array = read_gene_sets(sets_files)
860
+ deg_list = read_DEG_list(degl_file)
861
+
862
+ tf_n = len(tfbs_array)
863
+ epsilon = 1e-15 # 常用值
864
+ for i in range(tf_n):
865
+ if tfbs_array[i][0] == tf:
866
+ plot_contour_plot(deg_list, tfbs_array[i][1], win)
867
+ break
868
+
765
869
  if __name__ == '__main__':
766
870
  '''
767
871
  result = olcr(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='OVERCAST_data/input_deg-list.txt', win=30, thread_n=16)
@@ -773,4 +877,6 @@ if __name__ == '__main__':
773
877
 
774
878
  #plot_olc(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='OVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
775
879
 
880
+ #plot_contour(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='OVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
881
+
776
882
  #plot_fit3D(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='OVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: PyOVERCAST
3
- Version: 1.0.4
3
+ Version: 1.0.5
4
4
  Summary: A Python package for mining key transcription factors from transcriptome data.
5
5
  Author-email: Tinghua Huang <thua45@126.com>
6
6
  License-Expression: MIT
@@ -84,6 +84,9 @@ if __name__ == '__main__':
84
84
  # plot OLC matrix
85
85
  predict.plot_olc(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
86
86
 
87
+ # plot OLC matrix
88
+ predict.plot_olc(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
89
+
87
90
  # plot fitted 3D U-surface
88
- predict.plot_fit3D(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
91
+ predict.plot_contour(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
89
92
  ```
@@ -41,6 +41,9 @@ if __name__ == '__main__':
41
41
  # plot OLC matrix
42
42
  predict.plot_olc(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
43
43
 
44
+ # plot OLC matrix
45
+ predict.plot_olc(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
46
+
44
47
  # plot fitted 3D U-surface
45
- predict.plot_fit3D(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
48
+ predict.plot_contour(set_names=['human_jaspar_CLink_rci4_1w_0.8', 'human_hocomoco_CLink_rci4_1w_0.8'], list_file='./PyOVERCAST_data/input_deg-list.txt', tf='MA0844.2_XBP1', win=30)
46
49
  ```
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "PyOVERCAST"
7
- version = "1.0.4"
7
+ version = "1.0.5"
8
8
  license = "MIT" # SPDX expression
9
9
  description = "A Python package for mining key transcription factors from transcriptome data."
10
10
  readme = "README.md"
File without changes
File without changes