gene-viewer 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,53 @@
1
+ name: Publish Package to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+
7
+ jobs:
8
+ build:
9
+ name: Build distribution 📦
10
+ runs-on: ubuntu-latest
11
+
12
+ steps:
13
+ - uses: actions/checkout@v6
14
+ with:
15
+ persist-credentials: false
16
+ - name: Set up Python
17
+ uses: actions/setup-python@v6
18
+ with:
19
+ python-version: "3.12"
20
+ - name: Install pypa/build
21
+ run: >-
22
+ python3 -m
23
+ pip install
24
+ build
25
+ --user
26
+ - name: Build a binary wheel and a source tarball
27
+ run: python3 -m build
28
+ - name: Store the distribution packages
29
+ uses: actions/upload-artifact@v5
30
+ with:
31
+ name: python-package-distributions
32
+ path: dist/
33
+
34
+ pypi-publish:
35
+ name: >-
36
+ Publish Python 🐍 distribution 📦 to PyPI
37
+ needs:
38
+ - build
39
+ runs-on: ubuntu-latest
40
+ environment:
41
+ name: pypi
42
+ url: https://pypi.org/p/gene-viewer
43
+ permissions:
44
+ id-token: write # IMPORTANT: this permission is mandatory for trusted publishing
45
+
46
+ steps:
47
+ - name: Download all the dists
48
+ uses: actions/download-artifact@v6
49
+ with:
50
+ name: python-package-distributions
51
+ path: dist/
52
+ - name: Publish distribution 📦 to PyPI
53
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,6 @@
1
+ data
2
+ reference
3
+ __pycache__
4
+ .vscode
5
+ visualizations
6
+ *.egg-info/
@@ -0,0 +1,10 @@
1
+ Metadata-Version: 2.5
2
+ Name: gene_viewer
3
+ Version: 0.1.0
4
+ Summary: A lightweight and embeddable per-gene visualization of genomic regions, probes, and custom tracks.
5
+ Author: Simon Ament
6
+ Requires-Python: <3.13,>=3.10
7
+ Requires-Dist: biopython
8
+ Requires-Dist: oligo-designer-toolsuite
9
+ Requires-Dist: pybedtools
10
+ Requires-Dist: zstandard
@@ -0,0 +1,31 @@
1
+ import random
2
+
3
+ from src.gene_viewer.index import FastaFileIndex
4
+ from src.gene_viewer.types import GeneLocation
5
+
6
+ if __name__ == "__main__":
7
+ gene_location = GeneLocation(
8
+ id="Dmel_CG3082",
9
+ seq_id="NT_033778.4",
10
+ start=23071552,
11
+ end=23071555,
12
+ strand="+",
13
+ )
14
+ gene_locations = {gene_location.id: gene_location}
15
+ fastaIndex = FastaFileIndex(
16
+ "data/GCF_000001215.4_Release_6_plus_ISO1_MT_genomic.fna",
17
+ gene_locations=gene_locations,
18
+ )
19
+
20
+ fasta_keys = fastaIndex.keys()
21
+ random.shuffle(fasta_keys)
22
+ for key in fasta_keys:
23
+ print(f"Fetching sequence for gene {key}...")
24
+ print(fastaIndex.get(key))
25
+ fastaIndex.get(key)
26
+
27
+ # gtfIndex = GTFFileIndex("data/GCF_009729015.1_ASM972901v1_genomic.gtf")
28
+ # gtf_keys = gtfIndex.keys()
29
+ # random.shuffle(gtf_keys)
30
+ # for key in gtf_keys:
31
+ # gtfIndex.get(key)
@@ -0,0 +1,18 @@
1
+ [project]
2
+ name="gene_viewer"
3
+ version="0.1.0"
4
+ requires-python = ">=3.10,<3.13"
5
+ authors = [
6
+ { name="Simon Ament" }
7
+ ]
8
+ description = "A lightweight and embeddable per-gene visualization of genomic regions, probes, and custom tracks."
9
+ dependencies = [
10
+ "zstandard",
11
+ "oligo_designer_toolsuite",
12
+ "pybedtools",
13
+ "biopython",
14
+ ]
15
+
16
+ [build-system]
17
+ requires = ["hatchling >= 1.26"]
18
+ build-backend = "hatchling.build"
@@ -0,0 +1,2 @@
1
+ from gene_viewer.gene_viewer import GeneViewer as GeneViewer
2
+ from gene_viewer.gene_viewer_server import GeneViewerServer as GeneViewerServer
@@ -0,0 +1,190 @@
1
+ import hashlib
2
+ import json
3
+ from collections import defaultdict
4
+ from pathlib import Path
5
+
6
+ import zstandard as zstd
7
+
8
+ from gene_viewer.processor import Processor
9
+ from gene_viewer.types import GeneLocation
10
+
11
+
12
+ def hash(string: str) -> str:
13
+ return hashlib.sha256(string.encode()).hexdigest()
14
+
15
+
16
+ class ProcessorNode:
17
+ # can process data from children
18
+ _last_gene_id: str = None
19
+ _last_gene_data: dict[str, any] = None
20
+
21
+ def __init__(self, processor: Processor, children: list["DataNode | LoaderNode"]):
22
+ self._processor = processor
23
+ self._children = sorted(children, key=lambda x: x.type)
24
+
25
+ self.computation_path = (
26
+ self._processor.id
27
+ + "("
28
+ + ",".join([child.computation_path for child in self._children])
29
+ + ")"
30
+ )
31
+ self.cache_id = hash(self.computation_path)
32
+
33
+ def load_gene(self, gene: GeneLocation, loaders: dict[str, list]) -> dict[str, any]:
34
+ if self._last_gene_id == gene.id:
35
+ return self._last_gene_data
36
+ # Load the gene data from the processor
37
+ gene_data = {}
38
+ for child in self._children:
39
+ gene_data[child.type] = child.load_gene(gene, loaders)
40
+ self._last_gene_id = gene.id
41
+ self._last_gene_data = self._processor.process(gene_data)
42
+ return self._last_gene_data
43
+
44
+
45
+ class CachedNode:
46
+ cache_id: str = ""
47
+
48
+ _last_index_id: str = None
49
+ _last_index_data: any = None
50
+
51
+ def __init__(self, dir_path: Path, type: str):
52
+ self.dir_path = dir_path
53
+ self.type = type
54
+
55
+ def _load_cached_gene_ids(self) -> set[str]:
56
+ cache_metadata_path = (
57
+ self.dir_path / f"{self.type}_cache" / self.cache_id / "_metadata.json"
58
+ )
59
+ if cache_metadata_path.exists():
60
+ with open(cache_metadata_path, "r") as f:
61
+ metadata = json.load(f)
62
+ return set(metadata.get("genes_cached", []))
63
+ else:
64
+ return set()
65
+
66
+ def _load_cached_gene_data(
67
+ self, gene_id: str, return_ref: bool = False
68
+ ) -> dict[str, any]:
69
+ if return_ref:
70
+ return {"_ref": self.cache_id}
71
+
72
+ ref_dir_path = self.dir_path / f"{self.type}_cache" / self.cache_id
73
+ if ref_dir_path.exists():
74
+ if self._last_index_id == self.cache_id:
75
+ index = self._last_index_data
76
+ else:
77
+ with open(ref_dir_path / "_index.json", "r") as index_file:
78
+ index = json.load(index_file)
79
+ self._last_index_id = self.cache_id
80
+ self._last_index_data = index
81
+ gene_info = index.get(gene_id)
82
+ gene_offset = gene_info["offset"]
83
+ gene_length = gene_info["length"]
84
+ with open(ref_dir_path / "data.blob", "rb") as blob_file:
85
+ blob_file.seek(gene_offset)
86
+ cache_data = zstd.ZstdDecompressor().decompress(
87
+ blob_file.read(gene_length)
88
+ )
89
+
90
+ return json.loads(cache_data)
91
+ else:
92
+ return {}
93
+
94
+
95
+ class DataNode(CachedNode):
96
+ # can be cached, otheriwse loads data from a child processor node
97
+ _last_gene_id: str = None
98
+ _last_gene_data: any = None
99
+
100
+ def __init__(self, type: str, dir_path: Path, child: ProcessorNode):
101
+ super().__init__(dir_path, type)
102
+ self._child = child
103
+
104
+ self.computation_path = self.type + ":" + self._child.computation_path
105
+ self.cache_id = hash(self.computation_path)
106
+
107
+ # Load the cached gene IDs from the metadata file if it exists
108
+ self._cached_gene_ids = self._load_cached_gene_ids()
109
+
110
+ def load_gene(
111
+ self, gene: GeneLocation, loaders: dict[str, list], return_ref: bool = False
112
+ ) -> any:
113
+ if self._last_gene_id == gene.id:
114
+ return self._last_gene_data
115
+
116
+ if gene.id in self._cached_gene_ids:
117
+ # Load the gene data from the cache
118
+ self._last_gene_data = self._load_cached_gene_data(gene.id, return_ref)
119
+ else:
120
+ # Load the gene data from the child processor node
121
+ self._last_gene_data = self._child.load_gene(gene, loaders)[self.type]
122
+
123
+ self._last_gene_id = gene.id
124
+ return self._last_gene_data
125
+
126
+
127
+ class LoaderNode(CachedNode):
128
+ # can be cached, otheriwse loads data from loaders
129
+ _last_gene_id: str = None
130
+ _last_gene_data: any = None
131
+
132
+ def __init__(
133
+ self,
134
+ type: str,
135
+ dir_path: Path,
136
+ loader_cache_ids: list[str],
137
+ region_loader_cache_ids: list[str],
138
+ ):
139
+ super().__init__(dir_path, type)
140
+ self._loader_cache_ids = sorted(loader_cache_ids)
141
+ self._region_loader_cache_ids = sorted(region_loader_cache_ids)
142
+
143
+ self.computation_path = (
144
+ self.type
145
+ + "_loaders("
146
+ + ",".join(self._loader_cache_ids)
147
+ + "|"
148
+ + ",".join(self._region_loader_cache_ids)
149
+ + ")"
150
+ )
151
+ self.cache_id = hash(self.computation_path)
152
+
153
+ # Load the cached gene IDs from the metadata file if it exists
154
+ self._cached_gene_ids = self._load_cached_gene_ids()
155
+
156
+ def load_gene(
157
+ self, gene: GeneLocation, loaders: dict[str, list], return_ref: bool = False
158
+ ) -> any:
159
+ if self._last_gene_id == gene.id:
160
+ return self._last_gene_data
161
+
162
+ if gene.id in self._cached_gene_ids:
163
+ # Load the gene data from the cache
164
+ self._last_gene_id = gene.id
165
+ self._last_gene_data = self._load_cached_gene_data(gene.id, return_ref)
166
+ return self._last_gene_data
167
+
168
+ # Load the gene data from the loaders
169
+ if self.type == "sequences":
170
+ gene_data = [
171
+ item
172
+ for loader in loaders["sequences"]
173
+ for item in loader.load_gene(gene)
174
+ ]
175
+ elif self.type == "tracks":
176
+ gene_data = defaultdict(list)
177
+ for track_name, track_loaders in loaders["tracks"].items():
178
+ for loader in track_loaders:
179
+ loaded_data = loader.load_gene(gene)
180
+ for item in loaded_data:
181
+ gene_data[track_name].append(item)
182
+ else:
183
+ gene_data = defaultdict(list)
184
+ for loader in loaders[self.type]:
185
+ loaded_data = loader.load_gene(gene)
186
+ for key, value in loaded_data.items():
187
+ gene_data[key].extend(value)
188
+ self._last_gene_id = gene.id
189
+ self._last_gene_data = gene_data
190
+ return self._last_gene_data