FScanpy 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- FScanpy/__init__.py +206 -0
- FScanpy/data/__init__.py +122 -0
- FScanpy/data/test_data/blastx_example.xlsx +0 -0
- FScanpy/data/test_data/full_seq.xlsx +0 -0
- FScanpy/data/test_data/mrna_example.fasta +2584 -0
- FScanpy/data/test_data/region_example.csv +4 -0
- FScanpy/features/__init__.py +4 -0
- FScanpy/features/cnn_input.py +79 -0
- FScanpy/features/sequence.py +283 -0
- FScanpy/predictor.py +616 -0
- FScanpy/pretrained/long.pth +4 -0
- FScanpy/pretrained/short.pkl +0 -0
- FScanpy/utils.py +203 -0
- fscanpy-1.0.0.dist-info/METADATA +262 -0
- fscanpy-1.0.0.dist-info/RECORD +18 -0
- fscanpy-1.0.0.dist-info/WHEEL +5 -0
- fscanpy-1.0.0.dist-info/licenses/LICENSE +21 -0
- fscanpy-1.0.0.dist-info/top_level.txt +1 -0
FScanpy/utils.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import pandas as pd
|
|
3
|
+
from typing import Tuple, Optional
|
|
4
|
+
from Bio import SeqIO
|
|
5
|
+
from Bio.Seq import Seq
|
|
6
|
+
|
|
7
|
+
def fscanr(blastx_output: pd.DataFrame,
|
|
8
|
+
mismatch_cutoff: float = 10,
|
|
9
|
+
evalue_cutoff: float = 1e-5,
|
|
10
|
+
frameDist_cutoff: float = 10) -> pd.DataFrame:
|
|
11
|
+
"""
|
|
12
|
+
identify PRF sites from BLASTX output
|
|
13
|
+
|
|
14
|
+
Args:
|
|
15
|
+
blastx_output: BLASTX output DataFrame
|
|
16
|
+
mismatch_cutoff: mismatch threshold
|
|
17
|
+
evalue_cutoff: E-value threshold
|
|
18
|
+
frameDist_cutoff: frame distance threshold
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
pd.DataFrame: DataFrame containing PRF site information
|
|
22
|
+
"""
|
|
23
|
+
blastx = blastx_output.copy()
|
|
24
|
+
|
|
25
|
+
blastx.columns = ["qseqid", "sseqid", "pident", "length", "mismatch",
|
|
26
|
+
"gapopen", "qstart", "qend", "sstart", "send",
|
|
27
|
+
"evalue", "bitscore", "qframe", "sframe"]
|
|
28
|
+
|
|
29
|
+
blastx = blastx[
|
|
30
|
+
(blastx['evalue'] <= evalue_cutoff) &
|
|
31
|
+
(blastx['mismatch'] <= mismatch_cutoff)
|
|
32
|
+
].dropna()
|
|
33
|
+
|
|
34
|
+
freq = blastx['qseqid'].value_counts()
|
|
35
|
+
multi_hits = freq[freq > 1].index
|
|
36
|
+
blastx = blastx[blastx['qseqid'].isin(multi_hits)]
|
|
37
|
+
|
|
38
|
+
blastx = blastx.sort_values(['qseqid', 'sseqid', 'qstart'])
|
|
39
|
+
|
|
40
|
+
prf_list = []
|
|
41
|
+
for i in range(1, len(blastx)):
|
|
42
|
+
curr = blastx.iloc[i]
|
|
43
|
+
prev = blastx.iloc[i-1]
|
|
44
|
+
|
|
45
|
+
if (curr['qseqid'] == prev['qseqid'] and
|
|
46
|
+
curr['sseqid'] == prev['sseqid'] and
|
|
47
|
+
curr['qframe'] != prev['qframe'] and
|
|
48
|
+
curr['qframe'] * prev['qframe'] > 0):
|
|
49
|
+
|
|
50
|
+
if curr['qframe'] > 0 and prev['qframe'] > 0:
|
|
51
|
+
frame_start = prev['qend']
|
|
52
|
+
frame_end = curr['qstart']
|
|
53
|
+
pep_start = prev['send']
|
|
54
|
+
pep_end = curr['sstart']
|
|
55
|
+
strand = "+"
|
|
56
|
+
elif curr['qframe'] < 0 and prev['qframe'] < 0:
|
|
57
|
+
frame_start = prev['qstart']
|
|
58
|
+
frame_end = curr['qend']
|
|
59
|
+
pep_start = curr['send']
|
|
60
|
+
pep_end = prev['sstart']
|
|
61
|
+
strand = "-"
|
|
62
|
+
else:
|
|
63
|
+
continue
|
|
64
|
+
|
|
65
|
+
q_dist = frame_end - frame_start - 1
|
|
66
|
+
s_dist = pep_end - pep_start
|
|
67
|
+
fs_type = q_dist + (1 - s_dist) * 3
|
|
68
|
+
|
|
69
|
+
if (abs(q_dist) <= frameDist_cutoff and
|
|
70
|
+
abs(s_dist) <= frameDist_cutoff // 3 and
|
|
71
|
+
-3 < fs_type < 3):
|
|
72
|
+
|
|
73
|
+
prf_list.append({
|
|
74
|
+
'DNA_seqid': curr['qseqid'],
|
|
75
|
+
'FS_start': frame_start,
|
|
76
|
+
'FS_end': frame_end,
|
|
77
|
+
'Pep_seqid': curr['sseqid'],
|
|
78
|
+
'Pep_FS_start': prev['send'] + 1,
|
|
79
|
+
'Pep_FS_end': curr['sstart'],
|
|
80
|
+
'FS_type': fs_type,
|
|
81
|
+
'Strand': strand
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
if not prf_list:
|
|
85
|
+
print("No PRF events detected!")
|
|
86
|
+
return pd.DataFrame()
|
|
87
|
+
|
|
88
|
+
prf = pd.DataFrame(prf_list)
|
|
89
|
+
|
|
90
|
+
for col in ['DNA_seqid', 'Pep_seqid']:
|
|
91
|
+
for pos in ['FS_start', 'FS_end']:
|
|
92
|
+
loci = prf[col] + '_' + prf[pos].astype(str)
|
|
93
|
+
prf = prf[~loci.duplicated()]
|
|
94
|
+
|
|
95
|
+
return prf
|
|
96
|
+
|
|
97
|
+
def extract_prf_regions(mrna_file: str, prf_data: pd.DataFrame) -> pd.DataFrame:
|
|
98
|
+
"""
|
|
99
|
+
从mRNA序列中提取PRF位点周围的序列
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
mrna_file: mRNA序列文件路径 (FASTA格式)
|
|
103
|
+
prf_data: FScanR输出的PRF位点数据
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
pd.DataFrame: 包含399bp序列的DataFrame
|
|
107
|
+
"""
|
|
108
|
+
mrna_dict = {rec.id: str(rec.seq)
|
|
109
|
+
for rec in SeqIO.parse(mrna_file, "fasta")}
|
|
110
|
+
|
|
111
|
+
results = []
|
|
112
|
+
for _, row in prf_data.iterrows():
|
|
113
|
+
seq_id = row['DNA_seqid']
|
|
114
|
+
if seq_id not in mrna_dict:
|
|
115
|
+
print(f"警告: {seq_id} 未在mRNA文件中找到")
|
|
116
|
+
continue
|
|
117
|
+
|
|
118
|
+
sequence = mrna_dict[seq_id]
|
|
119
|
+
strand = row['Strand']
|
|
120
|
+
fs_start = int(row['FS_start'])
|
|
121
|
+
|
|
122
|
+
try:
|
|
123
|
+
if strand == '-':
|
|
124
|
+
sequence = str(Seq(sequence).reverse_complement())
|
|
125
|
+
|
|
126
|
+
# 只提取399bp序列,33bp由predictor内部截取
|
|
127
|
+
full_seq = extract_window_sequences(sequence, fs_start)[1]
|
|
128
|
+
|
|
129
|
+
results.append({
|
|
130
|
+
'DNA_seqid': seq_id,
|
|
131
|
+
'FS_start': fs_start,
|
|
132
|
+
'FS_end': int(row['FS_end']),
|
|
133
|
+
'Strand': strand,
|
|
134
|
+
'399bp': full_seq,
|
|
135
|
+
'FS_type': row['FS_type']
|
|
136
|
+
})
|
|
137
|
+
|
|
138
|
+
except Exception as e:
|
|
139
|
+
print(f"处理 {seq_id} 时出错: {str(e)}")
|
|
140
|
+
continue
|
|
141
|
+
|
|
142
|
+
return pd.DataFrame(results)
|
|
143
|
+
|
|
144
|
+
def extract_window_sequences(seq: str, position: int) -> Tuple[Optional[str], Optional[str]]:
|
|
145
|
+
"""
|
|
146
|
+
从指定位置提取分析窗口序列
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
seq: 输入DNA序列
|
|
150
|
+
position: 当前分析位置 (FS_start)
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
Tuple[str, str]: (33bp序列, 399bp序列) - 已调整为与训练模型匹配的长度
|
|
154
|
+
"""
|
|
155
|
+
# 确保位置在密码子边界上(整数倍的3)
|
|
156
|
+
frame_position = position - (position % 3)
|
|
157
|
+
|
|
158
|
+
# 计算33bp窗口的起止位置 (GB模型)
|
|
159
|
+
half_size_small = 33 // 2
|
|
160
|
+
start_small = frame_position - half_size_small
|
|
161
|
+
end_small = frame_position + half_size_small + (33 % 2) # 添加余数以处理奇数长度
|
|
162
|
+
|
|
163
|
+
# 计算399bp窗口的起止位置 (CNN模型)
|
|
164
|
+
half_size_large = 399 // 2
|
|
165
|
+
start_large = frame_position - half_size_large
|
|
166
|
+
end_large = frame_position + half_size_large + (399 % 2) # 添加余数以处理奇数长度
|
|
167
|
+
|
|
168
|
+
# 提取序列并填充
|
|
169
|
+
seq_small = _extract_and_pad(seq, start_small, end_small, 33)
|
|
170
|
+
seq_large = _extract_and_pad(seq, start_large, end_large, 399)
|
|
171
|
+
|
|
172
|
+
return seq_small, seq_large
|
|
173
|
+
|
|
174
|
+
def _extract_and_pad(seq: str, start: int, end: int, target_length: int) -> str:
|
|
175
|
+
"""提取序列并用N填充"""
|
|
176
|
+
if start < 0:
|
|
177
|
+
prefix = 'N' * abs(start)
|
|
178
|
+
extracted = prefix + seq[:end]
|
|
179
|
+
elif end > len(seq):
|
|
180
|
+
suffix = 'N' * (end - len(seq))
|
|
181
|
+
extracted = seq[start:] + suffix
|
|
182
|
+
else:
|
|
183
|
+
extracted = seq[start:end]
|
|
184
|
+
|
|
185
|
+
# 确保序列长度正确
|
|
186
|
+
if len(extracted) < target_length:
|
|
187
|
+
# 从中心填充
|
|
188
|
+
pad_left = (target_length - len(extracted)) // 2
|
|
189
|
+
pad_right = target_length - len(extracted) - pad_left
|
|
190
|
+
extracted = 'N' * pad_left + extracted + 'N' * pad_right
|
|
191
|
+
elif len(extracted) > target_length:
|
|
192
|
+
# 从序列两端等量截取
|
|
193
|
+
excess = len(extracted) - target_length
|
|
194
|
+
trim_each_side = excess // 2
|
|
195
|
+
extracted = extracted[trim_each_side:len(extracted)-trim_each_side]
|
|
196
|
+
|
|
197
|
+
return extracted
|
|
198
|
+
|
|
199
|
+
def prepare_cnn_input(sequence: str) -> np.ndarray:
|
|
200
|
+
"""prepare CNN model input"""
|
|
201
|
+
base_to_num = {'A': 1, 'T': 2, 'G': 3, 'C': 4, 'N': 0}
|
|
202
|
+
seq_numeric = [base_to_num.get(base, 0) for base in sequence.upper()]
|
|
203
|
+
return np.array(seq_numeric).reshape(1, len(sequence), 1)
|
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: FScanpy
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Machine learning-based prediction of programmed ribosomal frameshifting sites
|
|
5
|
+
Author-email: Yang Yuhao <ykongxiang@qq.com>
|
|
6
|
+
Maintainer-email: Yang Yuhao <ykongxiang@qq.com>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/ykongxiang/FScanpy-package
|
|
9
|
+
Project-URL: Documentation, https://github.com/ykongxiang/FScanpy-package/blob/master/tutorial/tutorial.md
|
|
10
|
+
Project-URL: Repository, https://github.com/ykongxiang/FScanpy-package
|
|
11
|
+
Project-URL: Issues, https://github.com/ykongxiang/FScanpy-package/issues
|
|
12
|
+
Project-URL: Changelog, https://github.com/ykongxiang/FScanpy-package/blob/master/CHANGELOG.md
|
|
13
|
+
Project-URL: Publication, https://doi.org/10.24272/j.issn.2095-8137.2025.648
|
|
14
|
+
Keywords: bioinformatics,ribosomal-frameshifting,PRF,machine-learning,FScanR
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: numpy>=2.3
|
|
22
|
+
Requires-Dist: pandas>=2.3
|
|
23
|
+
Requires-Dist: scikit-learn==1.7.2
|
|
24
|
+
Requires-Dist: matplotlib>=3.10
|
|
25
|
+
Requires-Dist: joblib>=1.5
|
|
26
|
+
Requires-Dist: biopython>=1.85
|
|
27
|
+
Requires-Dist: torch>=2.5
|
|
28
|
+
Requires-Dist: openpyxl
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
# FScanpy
|
|
32
|
+
## A Machine Learning-Based Framework for Programmed Ribosomal Frameshifting Prediction
|
|
33
|
+
|
|
34
|
+
[](https://github.com/ykongxiang/FScanpy-package/blob/master/README_zh.md)
|
|
35
|
+
[](https://www.python.org/)
|
|
36
|
+
[](https://github.com/ykongxiang/FScanpy-package/blob/master/LICENSE)
|
|
37
|
+
|
|
38
|
+
FScanpy is a comprehensive Python package designed for the prediction of [Programmed Ribosomal Frameshifting (PRF)](https://en.wikipedia.org/wiki/Ribosomal_frameshift) sites in nucleotide sequences. By integrating advanced machine learning approaches (HistGradientBoosting and BiLSTM-CNN) with the established [FScanR](https://github.com/seanchen607/FScanR.git) framework, FScanpy provides robust and accurate PRF site predictions.
|
|
39
|
+
|
|
40
|
+

|
|
41
|
+
|
|
42
|
+
## 🔧 Installation
|
|
43
|
+
|
|
44
|
+
### Prerequisites
|
|
45
|
+
- Python ≥ 3.11
|
|
46
|
+
- All dependencies are automatically installed
|
|
47
|
+
|
|
48
|
+
### Install via pip (Recommended)
|
|
49
|
+
```bash
|
|
50
|
+
pip install FScanpy
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
### Install from Source
|
|
54
|
+
```bash
|
|
55
|
+
git clone https://github.com/ykongxiang/FScanpy-package.git
|
|
56
|
+
cd FScanpy-package
|
|
57
|
+
pip install -e .
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
### Jupyter Notebook
|
|
61
|
+
|
|
62
|
+
From the cloned package directory, install into the Python environment you will use for Jupyter:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
python -m pip install . notebook ipykernel
|
|
66
|
+
python -m ipykernel install --sys-prefix --name fscanpy --display-name "Python (FScanpy)"
|
|
67
|
+
python -m notebook
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Open `FScanpy_Demo.ipynb` or `tutorial/predict_sample.ipynb`, select **Python (FScanpy)**, then restart the kernel and run all cells. The API overview in the demo is explanatory Markdown; the runnable examples use the bundled data. If installing into an already open notebook, use `%pip install /path/to/FScanpy-package` and restart the kernel.
|
|
71
|
+
|
|
72
|
+
## 🚀 Quick Start
|
|
73
|
+
|
|
74
|
+
### Basic Usage
|
|
75
|
+
```python
|
|
76
|
+
from FScanpy import predict_prf
|
|
77
|
+
|
|
78
|
+
# Simple sequence prediction
|
|
79
|
+
sequence = "ATGCGTACGTTAGC"*100 # Your DNA sequence
|
|
80
|
+
results = predict_prf(sequence=sequence)
|
|
81
|
+
|
|
82
|
+
# View top predictions
|
|
83
|
+
print(results[['Position', 'Ensemble_Probability', 'Short_Probability', 'Long_Probability']].head(10))
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
### Visualization
|
|
87
|
+
```python
|
|
88
|
+
from FScanpy import plot_prf_prediction
|
|
89
|
+
|
|
90
|
+
# Generate prediction plot
|
|
91
|
+
results, fig = plot_prf_prediction(
|
|
92
|
+
sequence=sequence,
|
|
93
|
+
short_threshold=0.65, # HistGB threshold
|
|
94
|
+
long_threshold=0.8, # BiLSTM-CNN threshold
|
|
95
|
+
ensemble_weight=0.4, # 40% Short, 60% Long
|
|
96
|
+
title="PRF Prediction Results"
|
|
97
|
+
)
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
### Advanced Usage
|
|
101
|
+
```python
|
|
102
|
+
from FScanpy import PRFPredictor
|
|
103
|
+
import pandas as pd
|
|
104
|
+
|
|
105
|
+
# Create predictor instance
|
|
106
|
+
predictor = PRFPredictor()
|
|
107
|
+
|
|
108
|
+
# Batch prediction on pre-extracted regions
|
|
109
|
+
data = pd.DataFrame({
|
|
110
|
+
'Long_Sequence': ['ATG' * 133, 'GCT' * 133] # 399bp sequences
|
|
111
|
+
})
|
|
112
|
+
results = predictor.predict_regions(data, ensemble_weight=0.4)
|
|
113
|
+
|
|
114
|
+
# Sequence-level prediction with custom parameters
|
|
115
|
+
results = predictor.predict_sequence(
|
|
116
|
+
sequence=sequence,
|
|
117
|
+
window_size=1, # Step size for sliding window
|
|
118
|
+
ensemble_weight=0.3, # Model weighting
|
|
119
|
+
short_threshold=0.5 # Filtering threshold
|
|
120
|
+
)
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## 🎛️ Ensemble Weight Configuration
|
|
124
|
+
|
|
125
|
+
The `ensemble_weight` parameter controls the weight ratio between HistGB and BiLSTM-CNN models:
|
|
126
|
+
|
|
127
|
+
| ensemble_weight | HistGB Model | BiLSTM-CNN Model | Characteristics | Best For |
|
|
128
|
+
|----------------|-------------|------------------|-----------------|----------|
|
|
129
|
+
| **0.2-0.3** | 20-30% | 70-80% | **High specificity**, reduces false positives | Precise validation, clinical applications |
|
|
130
|
+
| **0.4** | 40% | 60% | **Optimal balance**, highest AUC | Standard analysis (recommended) |
|
|
131
|
+
| **0.6-0.8** | 60-80% | 20-40% | **High sensitivity**, captures more sites | High-throughput screening, exploratory research |
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
### Weight Selection Examples
|
|
135
|
+
```python
|
|
136
|
+
# High specificity configuration (favoring HistGB)
|
|
137
|
+
precise_results = predict_prf(sequence, ensemble_weight=0.25)
|
|
138
|
+
|
|
139
|
+
# Optimal balance configuration (4:6 ratio)
|
|
140
|
+
balanced_results = predict_prf(sequence, ensemble_weight=0.4)
|
|
141
|
+
|
|
142
|
+
# High sensitivity configuration (favoring BiLSTM-CNN)
|
|
143
|
+
sensitive_results = predict_prf(sequence, ensemble_weight=0.7)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
## 📊 Core Functions
|
|
147
|
+
|
|
148
|
+
### Main Prediction Interface
|
|
149
|
+
```python
|
|
150
|
+
predict_prf(
|
|
151
|
+
sequence=None, # Single/multiple sequences or None
|
|
152
|
+
data=None, # DataFrame with 399bp sequences or None
|
|
153
|
+
window_size=3, # Sliding window step size
|
|
154
|
+
short_threshold=0.1, # Short model filtering threshold
|
|
155
|
+
ensemble_weight=0.4, # Short model weight (0.0-1.0)
|
|
156
|
+
model_dir=None # Custom model directory
|
|
157
|
+
)
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
### Visualization Function
|
|
161
|
+
```python
|
|
162
|
+
plot_prf_prediction(
|
|
163
|
+
sequence, # Input DNA sequence
|
|
164
|
+
window_size=3, # Scanning step size
|
|
165
|
+
short_threshold=0.65, # Short model threshold for plotting
|
|
166
|
+
long_threshold=0.8, # Long model threshold for plotting
|
|
167
|
+
ensemble_weight=0.4, # Model weighting
|
|
168
|
+
title=None, # Plot title
|
|
169
|
+
save_path=None, # Save file path
|
|
170
|
+
figsize=(12,8), # Figure size
|
|
171
|
+
dpi=300 # Resolution for saved plots
|
|
172
|
+
)
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
### PRFPredictor Class Methods
|
|
176
|
+
```python
|
|
177
|
+
predictor = PRFPredictor()
|
|
178
|
+
|
|
179
|
+
# Sequence prediction (sliding window)
|
|
180
|
+
predictor.predict_sequence(sequence, ensemble_weight=0.4)
|
|
181
|
+
|
|
182
|
+
# Region prediction (batch processing)
|
|
183
|
+
predictor.predict_regions(dataframe, ensemble_weight=0.4)
|
|
184
|
+
|
|
185
|
+
# Feature extraction
|
|
186
|
+
predictor.extract_features(sequences)
|
|
187
|
+
|
|
188
|
+
# Model information
|
|
189
|
+
predictor.get_model_info()
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
## 📈 Output Fields
|
|
193
|
+
|
|
194
|
+
### Prediction Results
|
|
195
|
+
- **`Position`**: Position in the original sequence
|
|
196
|
+
- **`Ensemble_Probability`**: Final ensemble prediction (main result)
|
|
197
|
+
- **`Short_Probability`**: HistGradientBoosting prediction (0-1)
|
|
198
|
+
- **`Long_Probability`**: BiLSTM-CNN prediction (0-1)
|
|
199
|
+
- **`Ensemble_Weights`**: Model weight configuration used
|
|
200
|
+
|
|
201
|
+
### Sequence Information
|
|
202
|
+
- **`Short_Sequence`**: 33bp sequence for Short model
|
|
203
|
+
- **`Long_Sequence`**: 399bp sequence for Long model
|
|
204
|
+
- **`Codon`**: 3bp codon at the prediction position
|
|
205
|
+
- **`Sequence_ID`**: Identifier for multi-sequence inputs
|
|
206
|
+
|
|
207
|
+
## 🔬 Integration with FScanR
|
|
208
|
+
|
|
209
|
+
FScanpy works seamlessly with the FScanR pipeline for comprehensive PRF analysis:
|
|
210
|
+
|
|
211
|
+
```python
|
|
212
|
+
from FScanpy import fscanr, extract_prf_regions, predict_prf
|
|
213
|
+
|
|
214
|
+
# Step 1: BLASTX analysis with FScanR
|
|
215
|
+
blastx_results = fscanr(
|
|
216
|
+
blastx_data,
|
|
217
|
+
mismatch_cutoff=10,
|
|
218
|
+
evalue_cutoff=1e-5,
|
|
219
|
+
frameDist_cutoff=10
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
# Step 2: Extract PRF candidate regions
|
|
223
|
+
prf_regions = extract_prf_regions(original_sequence, blastx_results)
|
|
224
|
+
|
|
225
|
+
# Step 3: Predict with FScanpy
|
|
226
|
+
final_predictions = predict_prf(data=prf_regions, ensemble_weight=0.4)
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
## 📚 Documentation
|
|
230
|
+
|
|
231
|
+
- **[Complete Tutorial](https://github.com/ykongxiang/FScanpy-package/blob/master/tutorial/tutorial.md)**: Comprehensive usage guide with examples
|
|
232
|
+
- **[Demo Notebook](https://github.com/ykongxiang/FScanpy-package/blob/master/FScanpy_Demo.ipynb)**: Practical usage of each function in the library and demonstration of analysis workflow results
|
|
233
|
+
- **[Predict Sample Interpretation](https://github.com/ykongxiang/FScanpy-package/blob/master/tutorial/predict_sample.ipynb)**: Detailed interpretation of FScanpy's plotting results and signal analysis
|
|
234
|
+
|
|
235
|
+
## 📝 Citation
|
|
236
|
+
|
|
237
|
+
If you use FScanpy in your research, please cite:
|
|
238
|
+
|
|
239
|
+
```bibtex
|
|
240
|
+
@article{yang2026deciphering,
|
|
241
|
+
|
|
242
|
+
author = {Yang, Yu-Hao and Yang, Juan and Liu, Zi-Jia and Li, Yuan and Song, Weibo and Stover, Naomi and Chen, Xiao},
|
|
243
|
+
|
|
244
|
+
title = {Deciphering ribosomal frameshifting determinants across species with a semi-supervised hybrid learning framework},
|
|
245
|
+
|
|
246
|
+
journal = {Zoological Research},
|
|
247
|
+
|
|
248
|
+
doi = {10.24272/j.issn.2095-8137.2025.648},
|
|
249
|
+
|
|
250
|
+
url = {https://doi.org/10.24272/j.issn.2095-8137.2025.648}
|
|
251
|
+
|
|
252
|
+
}
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
**FScanpy** - Advancing programmed ribosomal frameshifting research through machine learning 🧬
|
|
257
|
+
|
|
258
|
+
## Maintainer and License
|
|
259
|
+
|
|
260
|
+
Primary maintainer: **Yang Yuhao** ([ykongxiang@qq.com](mailto:ykongxiang@qq.com)).
|
|
261
|
+
|
|
262
|
+
FScanpy is distributed under the [MIT License](https://github.com/ykongxiang/FScanpy-package/blob/master/LICENSE). See the [changelog](https://github.com/ykongxiang/FScanpy-package/blob/master/CHANGELOG.md) for version 1.0.0 changes.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
FScanpy/__init__.py,sha256=mXZ6goG_0fW4Sbfx_D73KsnEdAWe1lbdl_nq-3SVpTI,8241
|
|
2
|
+
FScanpy/predictor.py,sha256=KucUuuUW5TvPL34iPf-plfFiAIy9wcWp0Sk6HCwtZqc,28270
|
|
3
|
+
FScanpy/utils.py,sha256=fidcKoTWNovG91jX8_cWrpgPyoVy3gwSKU6JrP3oFGU,7268
|
|
4
|
+
FScanpy/data/__init__.py,sha256=In_VmR8joAx1cgiw4nwLjxLa-_4Yz6N3W71hei2FVbk,4046
|
|
5
|
+
FScanpy/data/test_data/blastx_example.xlsx,sha256=pA5rfQzIAHq2P6EsxekDswmgbobkM0ofVzr7AEkW6Cs,98992
|
|
6
|
+
FScanpy/data/test_data/full_seq.xlsx,sha256=6iiQPjdlFeTRR2pP1k5-fC8Y_HbznoTOKs32NeKjXg8,12835
|
|
7
|
+
FScanpy/data/test_data/mrna_example.fasta,sha256=KA8WgICvYoGGyf9d2LvP6troa86gmC__sgAluNGg8ZQ,2992403
|
|
8
|
+
FScanpy/data/test_data/region_example.csv,sha256=jZm-wu1JHStRdxFhKiLEkkAykbTeFJglfPfbgdZ0cBE,1514
|
|
9
|
+
FScanpy/features/__init__.py,sha256=lTr7l2vWZWSgzrK0s0dIQ4OPyfXX54rKbGR3xmCPHy0,152
|
|
10
|
+
FScanpy/features/cnn_input.py,sha256=0GL4InzOUmmQ_CwrgfZ_GxsvtoeQdhZBedEr7gsQWUo,2855
|
|
11
|
+
FScanpy/features/sequence.py,sha256=2GtLLiHMZ1yW_KHv9Kk13fVNpzBcO6p8dTN_JYVO7oA,10939
|
|
12
|
+
FScanpy/pretrained/long.pth,sha256=Egf6tqgMn4-Iezv_IGLZ6yDfljh9cCBxQoaPvmXE_Uo,39531146
|
|
13
|
+
FScanpy/pretrained/short.pkl,sha256=AGVwsi6lIdbDBrXMkQ7QZ-n0Zt55HLWB3AOKclHAiZY,283657
|
|
14
|
+
fscanpy-1.0.0.dist-info/licenses/LICENSE,sha256=Q2KsPV2-icUm7EvGt_GUkE5OmSgcnyktBPIFk-8LPQY,1092
|
|
15
|
+
fscanpy-1.0.0.dist-info/METADATA,sha256=981uXrQpb-quEN3-JB60O72aW91ZTMqCtt4Wuj6E6lA,9912
|
|
16
|
+
fscanpy-1.0.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
17
|
+
fscanpy-1.0.0.dist-info/top_level.txt,sha256=5qIFIGC8pZHgM5MEkYh2EtQpBU0wUhRzbVktnX7hNjY,8
|
|
18
|
+
fscanpy-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Yang Yuhao and FScanpy contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
FScanpy
|