isoends 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. isoends-1.0.0/LICENSE +21 -0
  2. isoends-1.0.0/MANIFEST.in +1 -0
  3. isoends-1.0.0/PKG-INFO +317 -0
  4. isoends-1.0.0/README.md +284 -0
  5. isoends-1.0.0/pyproject.toml +53 -0
  6. isoends-1.0.0/setup.cfg +4 -0
  7. isoends-1.0.0/src/isoends/R/limma_moderated_t.R +30 -0
  8. isoends-1.0.0/src/isoends/__init__.py +12 -0
  9. isoends-1.0.0/src/isoends/annotation.py +95 -0
  10. isoends-1.0.0/src/isoends/apa.py +267 -0
  11. isoends-1.0.0/src/isoends/apa_run.py +366 -0
  12. isoends-1.0.0/src/isoends/api.py +453 -0
  13. isoends-1.0.0/src/isoends/cli.py +115 -0
  14. isoends-1.0.0/src/isoends/config.py +408 -0
  15. isoends-1.0.0/src/isoends/figures/__init__.py +73 -0
  16. isoends-1.0.0/src/isoends/figures/apa.py +478 -0
  17. isoends-1.0.0/src/isoends/figures/isotracks.py +107 -0
  18. isoends-1.0.0/src/isoends/figures/pubfig.py +5206 -0
  19. isoends-1.0.0/src/isoends/figures/style.py +203 -0
  20. isoends-1.0.0/src/isoends/figures/tails.py +348 -0
  21. isoends-1.0.0/src/isoends/gtf.py +200 -0
  22. isoends-1.0.0/src/isoends/hist.py +81 -0
  23. isoends-1.0.0/src/isoends/ingest.py +377 -0
  24. isoends-1.0.0/src/isoends/pipeline.py +397 -0
  25. isoends-1.0.0/src/isoends/report.py +356 -0
  26. isoends-1.0.0/src/isoends/sites.py +328 -0
  27. isoends-1.0.0/src/isoends/stats/__init__.py +11 -0
  28. isoends-1.0.0/src/isoends/stats/basic.py +128 -0
  29. isoends-1.0.0/src/isoends/stats/cmh.py +43 -0
  30. isoends-1.0.0/src/isoends/stats/limma.py +629 -0
  31. isoends-1.0.0/src/isoends/stats/readlevel.py +51 -0
  32. isoends-1.0.0/src/isoends/synthetic.py +447 -0
  33. isoends-1.0.0/src/isoends/tails.py +646 -0
  34. isoends-1.0.0/src/isoends/tailsource.py +238 -0
  35. isoends-1.0.0/src/isoends/templates.py +82 -0
  36. isoends-1.0.0/src/isoends/tss_run.py +150 -0
  37. isoends-1.0.0/src/isoends/util.py +115 -0
  38. isoends-1.0.0/src/isoends.egg-info/PKG-INFO +317 -0
  39. isoends-1.0.0/src/isoends.egg-info/SOURCES.txt +47 -0
  40. isoends-1.0.0/src/isoends.egg-info/dependency_links.txt +1 -0
  41. isoends-1.0.0/src/isoends.egg-info/entry_points.txt +2 -0
  42. isoends-1.0.0/src/isoends.egg-info/requires.txt +13 -0
  43. isoends-1.0.0/src/isoends.egg-info/top_level.txt +1 -0
  44. isoends-1.0.0/tests/__init__.py +0 -0
  45. isoends-1.0.0/tests/conftest.py +55 -0
  46. isoends-1.0.0/tests/test_api.py +97 -0
  47. isoends-1.0.0/tests/test_pipeline.py +364 -0
  48. isoends-1.0.0/tests/test_sites.py +148 -0
  49. isoends-1.0.0/tests/test_stats.py +165 -0
isoends-1.0.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Mustafa Elshani
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ include tests/*.py
isoends-1.0.0/PKG-INFO ADDED
@@ -0,0 +1,317 @@
1
+ Metadata-Version: 2.4
2
+ Name: isoends
3
+ Version: 1.0.0
4
+ Summary: Companion tool for IsoForge: replicate-aware statistics for poly(A) tail length, alternative polyadenylation and transcription start sites
5
+ Author: Mustafa Elshani
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/MustafaElshani/IsoEnds
8
+ Project-URL: Issues, https://github.com/MustafaElshani/IsoEnds/issues
9
+ Keywords: nanopore,long-read,RNA-seq,poly(A),alternative polyadenylation,transcription start site,isoforge
10
+ Classifier: Development Status :: 5 - Production/Stable
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Programming Language :: Python :: 3.12
15
+ Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Programming Language :: Python :: 3.14
17
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
18
+ Requires-Python: >=3.11
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: numpy>=1.24
22
+ Requires-Dist: pandas>=2.2
23
+ Requires-Dist: scipy>=1.11
24
+ Requires-Dist: pyarrow>=14
25
+ Requires-Dist: matplotlib>=3.8
26
+ Requires-Dist: pyyaml>=6
27
+ Provides-Extra: bam
28
+ Requires-Dist: pysam>=0.22; extra == "bam"
29
+ Provides-Extra: test
30
+ Requires-Dist: pytest>=8; extra == "test"
31
+ Requires-Dist: statsmodels>=0.14; extra == "test"
32
+ Dynamic: license-file
33
+
34
+ <p align="center">
35
+ <picture>
36
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/isoends_logo_dark.svg">
37
+ <img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/isoends_logo.svg" alt="IsoEnds: start sites, poly(A) sites and poly(A) tails" width="380">
38
+ </picture>
39
+ </p>
40
+
41
+ # IsoEnds
42
+
43
+ **Poly(A) tail length, alternative polyadenylation and transcription start sites from
44
+ [IsoForge](https://github.com/MustafaElshani/IsoForge) long-read RNA-seq results, tested over biological replicates.**
45
+
46
+ IsoEnds reads an IsoForge run as it is. The per-read table of IsoForge holds each read's transcript, start site,
47
+ poly(A) site and tail length, so IsoEnds needs no alignments. It compares two conditions and reports, per gene:
48
+
49
+ | Analysis | What is tested |
50
+ |---|---|
51
+ | Poly(A) tails | Change in the median tail length of genes, transcripts, transcript categories and gene sets |
52
+ | Alternative polyadenylation | Tandem 3′UTR sites (PDUI), intronic sites on composite and skipped exons, internal exonic sites, 5′UTR sites, and tandem sites of non-coding genes |
53
+ | Transcription start sites | Promoter switches (DTUI), upstream promoters, alternative first exons and internal exonic starts |
54
+
55
+ Every test uses replicates as its unit, never reads. The figures on this page are IsoEnds output for a simulated
56
+ experiment of 10 units with three replicate pairs each.
57
+
58
+ ```bash
59
+ isoends init --out config.yaml # write a template, then edit it
60
+ isoends all --config config.yaml -j 4 # tails, APA, TSS and the report
61
+ ```
62
+
63
+ ```python
64
+ import isoends as ie
65
+
66
+ exp = ie.Experiment.from_isoforge(
67
+ runs={"my_run": "isoforge/results"}, samples="samples.tsv", control="control", treatment="treated",
68
+ reference_gtf="gencode.primary_assembly.annotation.gtf", outdir="isoends_out")
69
+ res = exp.run(jobs=4)
70
+ res.volcano("tandem")
71
+ ```
72
+
73
+ ## Contents
74
+
75
+ - [Install](#install)
76
+ - [Input](#input)
77
+ - [Poly(A) tails](#polya-tails)
78
+ - [Alternative polyadenylation](#alternative-polyadenylation)
79
+ - [Transcription start sites](#transcription-start-sites)
80
+ - [Statistics](#statistics)
81
+ - [Output](#output)
82
+ - [Python API](#python-api)
83
+ - [Command line](#command-line)
84
+ - [Development](#development)
85
+
86
+ ## Install
87
+
88
+ ```bash
89
+ pip install isoends
90
+ ```
91
+
92
+ Python 3.11 or later. `pip install "isoends[bam]"` adds pysam, for tails read from BAM files. R with limma is used
93
+ for the moderated t-test when present; without it the built-in Python implementation gives the same results.
94
+ `isoends demo` makes a synthetic experiment and analyses it.
95
+
96
+ ## Input
97
+
98
+ | Input | Where | Needed for |
99
+ |---|---|---|
100
+ | IsoForge results folder | `runs: {name: {results: ...}}` | everything. IsoEnds reads `<prefix>_isoforge_read_annot.parquet` and `<prefix>_isoforge.gtf`, so run IsoForge with `--read_annot_format parquet` or `both` |
101
+ | Sample sheet | `samples:` | everything |
102
+ | Reference GTF with `Ensembl_canonical` tags (GENCODE) | `reference: {gtf: ...}` | APA and TSS |
103
+ | Characterised GTF of `isoforge-characterize` | found in the results folder or a sibling `SQANTI3` folder | categories of novel transcripts (NMD, retained intron) |
104
+ | Genome BAMs | `runs: {name: {bams: "{dataset}.bam"}}` | IsoTracks views of chosen genes |
105
+
106
+ IsoEnds uses the supported reads of the Parquet read table: reads marked `cluster_supported` whose transcript is in
107
+ the IsoForge GTF. These are the reads of IsoForge's filtered read TSV. Tail lengths come from their `polya_length`:
108
+ a read without a call has -1 and is left out, a length of 0 is a call (`tails.min_tail_nt: 1` leaves it out). A
109
+ per-read tail table or BAMs with the `pt:i` tag can be given instead (`tails:`, `tail_bams:`).
110
+
111
+ ### Sample sheet
112
+
113
+ A tab- or comma-separated table with one row per dataset of the read table:
114
+
115
+ ```
116
+ dataset unit condition pair
117
+ A_control_1 line_A control 1
118
+ A_treated_1 line_A treated 1
119
+ A_control_2 line_A control 2
120
+ A_treated_2 line_A treated 2
121
+ B_control_1 line_B control 1
122
+ ...
123
+ ```
124
+
125
+ | Column | Meaning |
126
+ |---|---|
127
+ | `dataset` | The dataset name in the IsoForge run |
128
+ | `unit` | The experiment a sample belongs to: a cell line, a donor, a time point. One unit is enough; with several, IsoEnds also tests across them |
129
+ | `condition` | The two conditions compared are named by `control` and `treatment` in the config. Datasets with any other condition are left out, so one sheet can hold several treatments |
130
+ | `pair` | Optional. Give it when each control sample has a matched treated sample; the tests within the unit are then paired. Leave the column out for independent replicates |
131
+ | `run` | Optional. The IsoForge run of the dataset, when several runs are configured and units are not named after them |
132
+
133
+ A unit needs two or more replicates of each condition to be tested on its own. To compare another pair of
134
+ conditions from the same sheet, run `isoends all --control A --treatment C --outdir out_C`.
135
+
136
+ ## Poly(A) tails
137
+
138
+ Per sample, the tail of a gene or transcript is the median over its reads, and the overall tail of the sample is
139
+ the median over its transcripts with more than 20 reads. The change between conditions is tested within each unit
140
+ and across units.
141
+
142
+ ```python
143
+ res.volcano("tail_genes") # genes; "tail_transcripts" for transcripts
144
+ res.get("tail_genes", level="within") # the table behind it, per unit
145
+ exp.plot_tails("GENE", unit="line_A") # the tail distribution of one gene
146
+ ```
147
+
148
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_tails.png" alt="Poly(A) tail output" width="760"></p>
149
+
150
+ (A) Per-transcript median tails of each unit by condition, with their medians. (B) Change in the median tail of
151
+ each gene, tested across units.
152
+
153
+ Also reported: the overall tail of each sample, tails by transcript category (reference type, or SQANTI3 class for
154
+ novel models), NMD and retained-intron isoforms against the coding isoform of the same gene, and the share of short
155
+ tails per gene set. Gene sets are ribosomal proteins, histones, snoRNA hosts and mitochondrial genes, or your own
156
+ (`gene_classes:`).
157
+
158
+ ## Alternative polyadenylation
159
+
160
+ The 3′ ends of the reads are clustered per gene and each cluster is placed on the gene's Ensembl-canonical
161
+ transcript. Genes are Ensembl gene IDs, so two genes that share a name are never pooled.
162
+
163
+ | Class | Site | Measure |
164
+ |---|---|---|
165
+ | Tandem 3′UTR | the two most used sites that IsoForge labels `three_prime_utr` | PDUI = distal / (proximal + distal) reads |
166
+ | Composite exon | in an intron, on transcripts that read through the upstream exon | share of the gene's reads, against terminal sites |
167
+ | Skipped exon | in an intron, on a terminal exon of its own | share |
168
+ | Internal exonic | inside an exon other than the last | share |
169
+ | 5′UTR | upstream of the start codon | share |
170
+ | Intronic (all) | any intronic site | share |
171
+ | Non-coding | tandem sites of lncRNAs and novel genes | PDUI, with q over the non-coding genes |
172
+
173
+ ```python
174
+ res.volcano("tandem") # also "composite", "skipped", "internal", "five_utr", "intronic",
175
+ res.get("noncoding_tandem") # "noncoding_tandem"
176
+ res.plot_sites("GENE", unit="line_A")
177
+ ```
178
+
179
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_apa.png" alt="Alternative polyadenylation output" width="760"></p>
180
+
181
+ Each point is a gene tested across units. Under each panel are the two isoforms compared, coloured as the points:
182
+ orange points are genes that move towards the orange isoform, teal points towards the teal one.
183
+
184
+ For each gene the tables give both sites of the tandem pair, their distance from the stop codon, the 3′UTR length
185
+ gained or lost and the transcripts ending at each site. Where two sites are equally used, the one with the lower
186
+ coordinate is taken. An intronic site is composite or skipped by the majority of the transcripts ending there, of
187
+ any gene on that strand.
188
+
189
+ | Option | Default | Alternative |
190
+ |---|---|---|
191
+ | `apa.tandem_sites` | `three_prime_utr`: sites with IsoForge's 3′UTR label | `terminal`: sites in the canonical last exon, or up to 5 kb past it |
192
+ | `apa.same_exon_check` | `false`: every pair is tested; `same_exon` and `share_distal_models_split` flag the pairs an intron splits | `true`: a pair is tested only when one annotated exon holds both sites and the transcripts ending at the distal site do not splice over the proximal one |
193
+ | `apa.across_units_same_pair` | `false`: across units, every unit's own pair | `true`: only the units whose pair matches the most common pair (`same_pair`) |
194
+ | `apa.intronic_models` | `any`: every transcript ending at the site | `gene`: only the transcripts of the site's gene |
195
+ | `apa.novel_gene_models` | `false`: genes outside the reference are not placed | `true`: placed on their best-supported IsoForge model |
196
+
197
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_gene.png" alt="One gene" width="760"></p>
198
+
199
+ (A) Usage of the poly(A) sites of one gene (`plot_sites`). (B) Its tail distribution, with the sample medians that
200
+ are tested as ticks (`plot_tails`).
201
+
202
+ ## Transcription start sites
203
+
204
+ The 5′ ends IsoForge assigns to a start site are clustered and placed on the canonical transcript in the same way.
205
+
206
+ | Class | Site | Measure |
207
+ |---|---|---|
208
+ | Promoter switch | the two most used start sites of the gene | DTUI = downstream / (upstream + downstream) reads |
209
+ | Upstream TSS | more than 1 kb upstream of the first exon | share, against the canonical first exon |
210
+ | Alternative first exon | in an intron, on transcripts with a first exon of their own | share |
211
+ | Internal exonic TSS | inside an exon other than the first | share |
212
+
213
+ ```python
214
+ res.volcano("tss_switch") # also "tss_upstream", "tss_first_exon", "tss_internal"
215
+ res.plot_sites("GENE", unit="line_A", kind="tss")
216
+ ```
217
+
218
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_tss.png" alt="Transcription start site output" width="620"></p>
219
+
220
+ Each point is a gene tested across units, with the two isoforms compared under each panel, coloured as the
221
+ points.
222
+
223
+ A 5′ end inside an internal exon can be a promoter or a truncated read, so confirm internal exonic starts with an
224
+ independent method.
225
+
226
+ ## Statistics
227
+
228
+ Reads from one sample are not independent, so a test with reads as its unit calls a difference between any two
229
+ samples. IsoEnds summarises each sample first and tests over replicates.
230
+
231
+ | | Paired replicates | Unpaired replicates |
232
+ |---|---|---|
233
+ | Overall tail of a sample, within a unit | paired t-test of the sample medians | Welch t-test of the sample medians |
234
+ | Tail of a gene or transcript, within a unit | moderated one-sample t-test of the paired differences | moderated two-sample t-test of the sample medians |
235
+ | Site usage, within a unit | Cochran-Mantel-Haenszel test over the pairs | moderated two-sample t-test of each sample's usage |
236
+ | Any change, across units | moderated one-sample t-test of the per-unit changes | the same |
237
+
238
+ The moderated t-test is limma's (empirical Bayes, robust). q is the Benjamini-Hochberg-adjusted P over the genes of
239
+ a unit, or over the genes tested across units. Across units a gene must be measured in two thirds of the units.
240
+
241
+ The read-level test is still reported next to the replicate-level one, as a description only
242
+ (`q_reads_pooled` in `tails_gene_within_unit`):
243
+
244
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_reads_vs_replicates.png" alt="Replicates and reads as units" width="760"></p>
245
+
246
+ The genes of one unit tested over its three replicate pairs (A) and with every read as a unit (B).
247
+
248
+ ## Output
249
+
250
+ ```
251
+ isoends_out/
252
+ report.md results, checks, parameters, tests, figures and table descriptions
253
+ tables/ every table as Parquet and TSV
254
+ figures/ PDF, SVG and PNG with captions
255
+ provenance.json inputs with SHA-256, versions, parameters
256
+ logs/ counts at each step
257
+ ```
258
+
259
+ | Table | Content |
260
+ |---|---|
261
+ | `tests_summary` | one row per family of tests: the test, n tested, n at q < 0.05 each way |
262
+ | `tails_gene_within_unit`, `tails_gene_across_units` | tail change per gene |
263
+ | `tails_transcript_within_unit`, `tails_transcript_across_units` | tail change per transcript, and per intron chain across units |
264
+ | `apa_tandem_*`, `apa_class_*`, `apa_ipa_*` | tandem 3′UTR, the site classes, and all intronic sites |
265
+ | `tss_switch_*`, `tss_class_*` | promoter switches and the start-site classes |
266
+ | `sites`, `tss_sites` | every poly(A) and start-site cluster with its class and reads |
267
+ | `qc_samples`, `validation` | per-sample counts and the input checks |
268
+
269
+ `report.md` describes every table.
270
+
271
+ ## Python API
272
+
273
+ | | |
274
+ |---|---|
275
+ | `ie.Experiment.from_config(path)` / `.from_isoforge(runs, samples, control, treatment, reference_gtf)` | set up an analysis |
276
+ | `exp.validate()` | the input checks as a table |
277
+ | `exp.run(jobs)` / `exp.tails()` / `exp.apa()` / `exp.tss()` / `exp.report()` | run everything or one stage |
278
+ | `ie.load(outdir)` | open a finished run |
279
+ | `res.analyses()` | the analyses with the genes tested and significant |
280
+ | `res.get(analysis, level, unit)` | one analysis as a table with `effect` and `q` |
281
+ | `res.volcano(analysis)` / `res.volcano_grid(analysis)` | a volcano, or one per unit |
282
+ | `res.plot_sites(gene, unit, kind)` / `exp.plot_tails(gene, unit)` | one gene |
283
+ | `res.gene(gene)` / `res.table(name)` | every row of a gene, or any table |
284
+ | `exp.tails_of(gene, unit)` | the tail of a gene per sample |
285
+ | `exp.isotracks(gene, unit)` | the reads of a gene, drawn with IsoTracks |
286
+
287
+ Tables are pandas DataFrames and plots are Matplotlib axes. `docs/make_readme_figures.py` draws the figures of this
288
+ page, from a simulated experiment (`--simulate FOLDER`) or from a finished run of your own.
289
+
290
+ ## Command line
291
+
292
+ ```
293
+ isoends init [--out isoends.yaml] write a configuration template
294
+ isoends validate --config config.yaml check the inputs and the sample sheet
295
+ isoends all --config config.yaml tails, apa, tss and the report
296
+ isoends tails | apa | tss | report one stage
297
+ isoends isotracks --config config.yaml --gene NAME [--unit UNIT]
298
+ isoends demo [--out DIR] [--unpaired] a synthetic experiment, analysed
299
+ ```
300
+
301
+ `-j N` summarises N runs in parallel; summaries are cached, so later stages and reruns start from them.
302
+ `--control`, `--treatment` and `--outdir` override the config. Every parameter and its default is in the template
303
+ that `isoends init` writes.
304
+
305
+ ## Development
306
+
307
+ ```bash
308
+ git clone https://github.com/MustafaElshani/IsoEnds
309
+ cd IsoEnds
310
+ pip install -e ".[test]"
311
+ pytest
312
+ ```
313
+
314
+ The tests check the statistics against R limma, statsmodels and SciPy, the site classes on hand-built gene models,
315
+ and the whole pipeline on synthetic experiments with known effects.
316
+
317
+ IsoEnds is released under the MIT licence.
@@ -0,0 +1,284 @@
1
+ <p align="center">
2
+ <picture>
3
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/isoends_logo_dark.svg">
4
+ <img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/isoends_logo.svg" alt="IsoEnds: start sites, poly(A) sites and poly(A) tails" width="380">
5
+ </picture>
6
+ </p>
7
+
8
+ # IsoEnds
9
+
10
+ **Poly(A) tail length, alternative polyadenylation and transcription start sites from
11
+ [IsoForge](https://github.com/MustafaElshani/IsoForge) long-read RNA-seq results, tested over biological replicates.**
12
+
13
+ IsoEnds reads an IsoForge run as it is. The per-read table of IsoForge holds each read's transcript, start site,
14
+ poly(A) site and tail length, so IsoEnds needs no alignments. It compares two conditions and reports, per gene:
15
+
16
+ | Analysis | What is tested |
17
+ |---|---|
18
+ | Poly(A) tails | Change in the median tail length of genes, transcripts, transcript categories and gene sets |
19
+ | Alternative polyadenylation | Tandem 3′UTR sites (PDUI), intronic sites on composite and skipped exons, internal exonic sites, 5′UTR sites, and tandem sites of non-coding genes |
20
+ | Transcription start sites | Promoter switches (DTUI), upstream promoters, alternative first exons and internal exonic starts |
21
+
22
+ Every test uses replicates as its unit, never reads. The figures on this page are IsoEnds output for a simulated
23
+ experiment of 10 units with three replicate pairs each.
24
+
25
+ ```bash
26
+ isoends init --out config.yaml # write a template, then edit it
27
+ isoends all --config config.yaml -j 4 # tails, APA, TSS and the report
28
+ ```
29
+
30
+ ```python
31
+ import isoends as ie
32
+
33
+ exp = ie.Experiment.from_isoforge(
34
+ runs={"my_run": "isoforge/results"}, samples="samples.tsv", control="control", treatment="treated",
35
+ reference_gtf="gencode.primary_assembly.annotation.gtf", outdir="isoends_out")
36
+ res = exp.run(jobs=4)
37
+ res.volcano("tandem")
38
+ ```
39
+
40
+ ## Contents
41
+
42
+ - [Install](#install)
43
+ - [Input](#input)
44
+ - [Poly(A) tails](#polya-tails)
45
+ - [Alternative polyadenylation](#alternative-polyadenylation)
46
+ - [Transcription start sites](#transcription-start-sites)
47
+ - [Statistics](#statistics)
48
+ - [Output](#output)
49
+ - [Python API](#python-api)
50
+ - [Command line](#command-line)
51
+ - [Development](#development)
52
+
53
+ ## Install
54
+
55
+ ```bash
56
+ pip install isoends
57
+ ```
58
+
59
+ Python 3.11 or later. `pip install "isoends[bam]"` adds pysam, for tails read from BAM files. R with limma is used
60
+ for the moderated t-test when present; without it the built-in Python implementation gives the same results.
61
+ `isoends demo` makes a synthetic experiment and analyses it.
62
+
63
+ ## Input
64
+
65
+ | Input | Where | Needed for |
66
+ |---|---|---|
67
+ | IsoForge results folder | `runs: {name: {results: ...}}` | everything. IsoEnds reads `<prefix>_isoforge_read_annot.parquet` and `<prefix>_isoforge.gtf`, so run IsoForge with `--read_annot_format parquet` or `both` |
68
+ | Sample sheet | `samples:` | everything |
69
+ | Reference GTF with `Ensembl_canonical` tags (GENCODE) | `reference: {gtf: ...}` | APA and TSS |
70
+ | Characterised GTF of `isoforge-characterize` | found in the results folder or a sibling `SQANTI3` folder | categories of novel transcripts (NMD, retained intron) |
71
+ | Genome BAMs | `runs: {name: {bams: "{dataset}.bam"}}` | IsoTracks views of chosen genes |
72
+
73
+ IsoEnds uses the supported reads of the Parquet read table: reads marked `cluster_supported` whose transcript is in
74
+ the IsoForge GTF. These are the reads of IsoForge's filtered read TSV. Tail lengths come from their `polya_length`:
75
+ a read without a call has -1 and is left out, a length of 0 is a call (`tails.min_tail_nt: 1` leaves it out). A
76
+ per-read tail table or BAMs with the `pt:i` tag can be given instead (`tails:`, `tail_bams:`).
77
+
78
+ ### Sample sheet
79
+
80
+ A tab- or comma-separated table with one row per dataset of the read table:
81
+
82
+ ```
83
+ dataset unit condition pair
84
+ A_control_1 line_A control 1
85
+ A_treated_1 line_A treated 1
86
+ A_control_2 line_A control 2
87
+ A_treated_2 line_A treated 2
88
+ B_control_1 line_B control 1
89
+ ...
90
+ ```
91
+
92
+ | Column | Meaning |
93
+ |---|---|
94
+ | `dataset` | The dataset name in the IsoForge run |
95
+ | `unit` | The experiment a sample belongs to: a cell line, a donor, a time point. One unit is enough; with several, IsoEnds also tests across them |
96
+ | `condition` | The two conditions compared are named by `control` and `treatment` in the config. Datasets with any other condition are left out, so one sheet can hold several treatments |
97
+ | `pair` | Optional. Give it when each control sample has a matched treated sample; the tests within the unit are then paired. Leave the column out for independent replicates |
98
+ | `run` | Optional. The IsoForge run of the dataset, when several runs are configured and units are not named after them |
99
+
100
+ A unit needs two or more replicates of each condition to be tested on its own. To compare another pair of
101
+ conditions from the same sheet, run `isoends all --control A --treatment C --outdir out_C`.
102
+
103
+ ## Poly(A) tails
104
+
105
+ Per sample, the tail of a gene or transcript is the median over its reads, and the overall tail of the sample is
106
+ the median over its transcripts with more than 20 reads. The change between conditions is tested within each unit
107
+ and across units.
108
+
109
+ ```python
110
+ res.volcano("tail_genes") # genes; "tail_transcripts" for transcripts
111
+ res.get("tail_genes", level="within") # the table behind it, per unit
112
+ exp.plot_tails("GENE", unit="line_A") # the tail distribution of one gene
113
+ ```
114
+
115
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_tails.png" alt="Poly(A) tail output" width="760"></p>
116
+
117
+ (A) Per-transcript median tails of each unit by condition, with their medians. (B) Change in the median tail of
118
+ each gene, tested across units.
119
+
120
+ Also reported: the overall tail of each sample, tails by transcript category (reference type, or SQANTI3 class for
121
+ novel models), NMD and retained-intron isoforms against the coding isoform of the same gene, and the share of short
122
+ tails per gene set. Gene sets are ribosomal proteins, histones, snoRNA hosts and mitochondrial genes, or your own
123
+ (`gene_classes:`).
124
+
125
+ ## Alternative polyadenylation
126
+
127
+ The 3′ ends of the reads are clustered per gene and each cluster is placed on the gene's Ensembl-canonical
128
+ transcript. Genes are Ensembl gene IDs, so two genes that share a name are never pooled.
129
+
130
+ | Class | Site | Measure |
131
+ |---|---|---|
132
+ | Tandem 3′UTR | the two most used sites that IsoForge labels `three_prime_utr` | PDUI = distal / (proximal + distal) reads |
133
+ | Composite exon | in an intron, on transcripts that read through the upstream exon | share of the gene's reads, against terminal sites |
134
+ | Skipped exon | in an intron, on a terminal exon of its own | share |
135
+ | Internal exonic | inside an exon other than the last | share |
136
+ | 5′UTR | upstream of the start codon | share |
137
+ | Intronic (all) | any intronic site | share |
138
+ | Non-coding | tandem sites of lncRNAs and novel genes | PDUI, with q over the non-coding genes |
139
+
140
+ ```python
141
+ res.volcano("tandem") # also "composite", "skipped", "internal", "five_utr", "intronic",
142
+ res.get("noncoding_tandem") # "noncoding_tandem"
143
+ res.plot_sites("GENE", unit="line_A")
144
+ ```
145
+
146
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_apa.png" alt="Alternative polyadenylation output" width="760"></p>
147
+
148
+ Each point is a gene tested across units. Under each panel are the two isoforms compared, coloured as the points:
149
+ orange points are genes that move towards the orange isoform, teal points towards the teal one.
150
+
151
+ For each gene the tables give both sites of the tandem pair, their distance from the stop codon, the 3′UTR length
152
+ gained or lost and the transcripts ending at each site. Where two sites are equally used, the one with the lower
153
+ coordinate is taken. An intronic site is composite or skipped by the majority of the transcripts ending there, of
154
+ any gene on that strand.
155
+
156
+ | Option | Default | Alternative |
157
+ |---|---|---|
158
+ | `apa.tandem_sites` | `three_prime_utr`: sites with IsoForge's 3′UTR label | `terminal`: sites in the canonical last exon, or up to 5 kb past it |
159
+ | `apa.same_exon_check` | `false`: every pair is tested; `same_exon` and `share_distal_models_split` flag the pairs an intron splits | `true`: a pair is tested only when one annotated exon holds both sites and the transcripts ending at the distal site do not splice over the proximal one |
160
+ | `apa.across_units_same_pair` | `false`: across units, every unit's own pair | `true`: only the units whose pair matches the most common pair (`same_pair`) |
161
+ | `apa.intronic_models` | `any`: every transcript ending at the site | `gene`: only the transcripts of the site's gene |
162
+ | `apa.novel_gene_models` | `false`: genes outside the reference are not placed | `true`: placed on their best-supported IsoForge model |
163
+
164
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_gene.png" alt="One gene" width="760"></p>
165
+
166
+ (A) Usage of the poly(A) sites of one gene (`plot_sites`). (B) Its tail distribution, with the sample medians that
167
+ are tested as ticks (`plot_tails`).
168
+
169
+ ## Transcription start sites
170
+
171
+ The 5′ ends IsoForge assigns to a start site are clustered and placed on the canonical transcript in the same way.
172
+
173
+ | Class | Site | Measure |
174
+ |---|---|---|
175
+ | Promoter switch | the two most used start sites of the gene | DTUI = downstream / (upstream + downstream) reads |
176
+ | Upstream TSS | more than 1 kb upstream of the first exon | share, against the canonical first exon |
177
+ | Alternative first exon | in an intron, on transcripts with a first exon of their own | share |
178
+ | Internal exonic TSS | inside an exon other than the first | share |
179
+
180
+ ```python
181
+ res.volcano("tss_switch") # also "tss_upstream", "tss_first_exon", "tss_internal"
182
+ res.plot_sites("GENE", unit="line_A", kind="tss")
183
+ ```
184
+
185
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_tss.png" alt="Transcription start site output" width="620"></p>
186
+
187
+ Each point is a gene tested across units, with the two isoforms compared under each panel, coloured as the
188
+ points.
189
+
190
+ A 5′ end inside an internal exon can be a promoter or a truncated read, so confirm internal exonic starts with an
191
+ independent method.
192
+
193
+ ## Statistics
194
+
195
+ Reads from one sample are not independent, so a test with reads as its unit calls a difference between any two
196
+ samples. IsoEnds summarises each sample first and tests over replicates.
197
+
198
+ | | Paired replicates | Unpaired replicates |
199
+ |---|---|---|
200
+ | Overall tail of a sample, within a unit | paired t-test of the sample medians | Welch t-test of the sample medians |
201
+ | Tail of a gene or transcript, within a unit | moderated one-sample t-test of the paired differences | moderated two-sample t-test of the sample medians |
202
+ | Site usage, within a unit | Cochran-Mantel-Haenszel test over the pairs | moderated two-sample t-test of each sample's usage |
203
+ | Any change, across units | moderated one-sample t-test of the per-unit changes | the same |
204
+
205
+ The moderated t-test is limma's (empirical Bayes, robust). q is the Benjamini-Hochberg-adjusted P over the genes of
206
+ a unit, or over the genes tested across units. Across units a gene must be measured in two thirds of the units.
207
+
208
+ The read-level test is still reported next to the replicate-level one, as a description only
209
+ (`q_reads_pooled` in `tails_gene_within_unit`):
210
+
211
+ <p align="center"><img src="https://raw.githubusercontent.com/MustafaElshani/IsoEnds/v1.0.0/docs/img/example_reads_vs_replicates.png" alt="Replicates and reads as units" width="760"></p>
212
+
213
+ The genes of one unit tested over its three replicate pairs (A) and with every read as a unit (B).
214
+
215
+ ## Output
216
+
217
+ ```
218
+ isoends_out/
219
+ report.md results, checks, parameters, tests, figures and table descriptions
220
+ tables/ every table as Parquet and TSV
221
+ figures/ PDF, SVG and PNG with captions
222
+ provenance.json inputs with SHA-256, versions, parameters
223
+ logs/ counts at each step
224
+ ```
225
+
226
+ | Table | Content |
227
+ |---|---|
228
+ | `tests_summary` | one row per family of tests: the test, n tested, n at q < 0.05 each way |
229
+ | `tails_gene_within_unit`, `tails_gene_across_units` | tail change per gene |
230
+ | `tails_transcript_within_unit`, `tails_transcript_across_units` | tail change per transcript, and per intron chain across units |
231
+ | `apa_tandem_*`, `apa_class_*`, `apa_ipa_*` | tandem 3′UTR, the site classes, and all intronic sites |
232
+ | `tss_switch_*`, `tss_class_*` | promoter switches and the start-site classes |
233
+ | `sites`, `tss_sites` | every poly(A) and start-site cluster with its class and reads |
234
+ | `qc_samples`, `validation` | per-sample counts and the input checks |
235
+
236
+ `report.md` describes every table.
237
+
238
+ ## Python API
239
+
240
+ | | |
241
+ |---|---|
242
+ | `ie.Experiment.from_config(path)` / `.from_isoforge(runs, samples, control, treatment, reference_gtf)` | set up an analysis |
243
+ | `exp.validate()` | the input checks as a table |
244
+ | `exp.run(jobs)` / `exp.tails()` / `exp.apa()` / `exp.tss()` / `exp.report()` | run everything or one stage |
245
+ | `ie.load(outdir)` | open a finished run |
246
+ | `res.analyses()` | the analyses with the genes tested and significant |
247
+ | `res.get(analysis, level, unit)` | one analysis as a table with `effect` and `q` |
248
+ | `res.volcano(analysis)` / `res.volcano_grid(analysis)` | a volcano, or one per unit |
249
+ | `res.plot_sites(gene, unit, kind)` / `exp.plot_tails(gene, unit)` | one gene |
250
+ | `res.gene(gene)` / `res.table(name)` | every row of a gene, or any table |
251
+ | `exp.tails_of(gene, unit)` | the tail of a gene per sample |
252
+ | `exp.isotracks(gene, unit)` | the reads of a gene, drawn with IsoTracks |
253
+
254
+ Tables are pandas DataFrames and plots are Matplotlib axes. `docs/make_readme_figures.py` draws the figures of this
255
+ page, from a simulated experiment (`--simulate FOLDER`) or from a finished run of your own.
256
+
257
+ ## Command line
258
+
259
+ ```
260
+ isoends init [--out isoends.yaml] write a configuration template
261
+ isoends validate --config config.yaml check the inputs and the sample sheet
262
+ isoends all --config config.yaml tails, apa, tss and the report
263
+ isoends tails | apa | tss | report one stage
264
+ isoends isotracks --config config.yaml --gene NAME [--unit UNIT]
265
+ isoends demo [--out DIR] [--unpaired] a synthetic experiment, analysed
266
+ ```
267
+
268
+ `-j N` summarises N runs in parallel; summaries are cached, so later stages and reruns start from them.
269
+ `--control`, `--treatment` and `--outdir` override the config. Every parameter and its default is in the template
270
+ that `isoends init` writes.
271
+
272
+ ## Development
273
+
274
+ ```bash
275
+ git clone https://github.com/MustafaElshani/IsoEnds
276
+ cd IsoEnds
277
+ pip install -e ".[test]"
278
+ pytest
279
+ ```
280
+
281
+ The tests check the statistics against R limma, statsmodels and SciPy, the site classes on hand-built gene models,
282
+ and the whole pipeline on synthetic experiments with known effects.
283
+
284
+ IsoEnds is released under the MIT licence.