scigantic-facebase 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scigantic_facebase-0.1.0/LICENSE +16 -0
- scigantic_facebase-0.1.0/PKG-INFO +88 -0
- scigantic_facebase-0.1.0/README.md +54 -0
- scigantic_facebase-0.1.0/pyproject.toml +58 -0
- scigantic_facebase-0.1.0/setup.cfg +4 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/__init__.py +42 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/_client.py +107 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/_version.py +1 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/catalog.py +149 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/microct.py +147 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/models.py +66 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/py.typed +0 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase/remote.py +171 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase.egg-info/PKG-INFO +88 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase.egg-info/SOURCES.txt +18 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase.egg-info/dependency_links.txt +1 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase.egg-info/requires.txt +9 -0
- scigantic_facebase-0.1.0/src/scigantic_facebase.egg-info/top_level.txt +1 -0
- scigantic_facebase-0.1.0/tests/test_live.py +45 -0
- scigantic_facebase-0.1.0/tests/test_remote.py +103 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
MIT No Attribution
|
|
2
|
+
|
|
3
|
+
Copyright 2026 Scigantic
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of this
|
|
6
|
+
software and associated documentation files (the "Software"), to deal in the Software
|
|
7
|
+
without restriction, including without limitation the rights to use, copy, modify,
|
|
8
|
+
merge, publish, distribute, sublicense, and/or sell copies of the Software, and to
|
|
9
|
+
permit persons to whom the Software is furnished to do so.
|
|
10
|
+
|
|
11
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED,
|
|
12
|
+
INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A
|
|
13
|
+
PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
|
14
|
+
HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF
|
|
15
|
+
CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE
|
|
16
|
+
OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scigantic-facebase
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Search FaceBase, the craniofacial research data hub, and read its open-access micro-CT, imaging and sequencing data from Python: a typed catalog over FaceBase's open metadata API, citations in the form its Terms of Use ask for, ranged reads of remote files, and single-slice reads from zipped micro-CT stacks without downloading them.
|
|
5
|
+
Author: Scigantic
|
|
6
|
+
License: MIT-0
|
|
7
|
+
Project-URL: Homepage, https://scigantic.com
|
|
8
|
+
Project-URL: Repository, https://github.com/Scigantic/scigantic-facebase
|
|
9
|
+
Project-URL: Issues, https://github.com/Scigantic/scigantic-facebase/issues
|
|
10
|
+
Project-URL: FaceBase, https://www.facebase.org
|
|
11
|
+
Keywords: facebase,craniofacial,micro-ct,zebrafish,deriva,ermrest,bioinformatics
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT No Attribution License (MIT-0)
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: requests<3,>=2.28
|
|
26
|
+
Requires-Dist: numpy<3,>=1.24
|
|
27
|
+
Requires-Dist: tifffile<2027,>=2023.7
|
|
28
|
+
Requires-Dist: imagecodecs>=2023.9
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
31
|
+
Requires-Dist: mypy>=1.10; extra == "dev"
|
|
32
|
+
Requires-Dist: types-requests; extra == "dev"
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# scigantic-facebase
|
|
36
|
+
|
|
37
|
+
Search [FaceBase](https://www.facebase.org), the craniofacial research data hub, and read its open-access data from Python.
|
|
38
|
+
|
|
39
|
+
FaceBase holds about 1,070 open datasets: micro-CT and light-sheet imaging, ChIP-seq and RNA-seq tracks, expression tables and more, from mouse, zebrafish, human and other models. This package is a typed layer over FaceBase's open metadata API and file store. No key or login is needed.
|
|
40
|
+
|
|
41
|
+
```
|
|
42
|
+
pip install scigantic-facebase
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Use
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import scigantic_facebase as fb
|
|
49
|
+
|
|
50
|
+
hits = fb.search("zebrafish thyroid microCT") # newest first
|
|
51
|
+
ds = hits[0]
|
|
52
|
+
print(ds.title, ds.doi_url)
|
|
53
|
+
print(fb.cite(ds))
|
|
54
|
+
|
|
55
|
+
files = fb.files(ds) # size, MD5, download URL
|
|
56
|
+
fb.download(files[0], "data/") # MD5-checked
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Search matches every word against the whole record, and treats "zebrafish" and "Danio rerio", "mouse" and "Mus musculus" as the same thing.
|
|
60
|
+
|
|
61
|
+
## Read a micro-CT scan without downloading it
|
|
62
|
+
|
|
63
|
+
Many micro-CT datasets are a zip of TIFF slices. A zip keeps its directory at the end, so one slice costs one HTTP range request.
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
stack = fb.microct.open_stack(files[0])
|
|
67
|
+
len(stack) # 596 slices
|
|
68
|
+
stack.voxel_size_um # 10.5, from the scan log
|
|
69
|
+
img = stack.read_slice(300)
|
|
70
|
+
vol = stack.volume(step=10) # every 10th slice, (z, y, x)
|
|
71
|
+
stack.bytes_fetched # a few MB of a 49 MB file
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
`fb.microct.thyroid_atlas()` lists the zebrafish thyroid-hormone atlas: 81 head scans from 12 to 25 mm standard length, raised hypothyroid, euthyroid or hyperthyroid.
|
|
75
|
+
|
|
76
|
+
`fb.open_remote(file)` gives any FaceBase file as a seekable file object backed by range requests, which `zipfile`, `tifffile` and `h5py` accept.
|
|
77
|
+
|
|
78
|
+
## What is not here
|
|
79
|
+
|
|
80
|
+
Protected human-subjects datasets sit behind FaceBase's data access request. Their metadata is public and `fb.search(include_protected=True)` lists them, but reading their files raises `FacebaseAccessError`.
|
|
81
|
+
|
|
82
|
+
## Terms of use
|
|
83
|
+
|
|
84
|
+
Copyright to FaceBase data belongs to the contributing investigators. FaceBase's [Terms of Use](https://www.facebase.org/policies/tou/) ask users to cite the dataset, acknowledge FaceBase, and send other people to FaceBase instead of circulating downloaded files. `fb.cite()` produces the citation. This package fetches from FaceBase on request and does not mirror anything.
|
|
85
|
+
|
|
86
|
+
## License
|
|
87
|
+
|
|
88
|
+
MIT-0 for this code. The data carries its own terms.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# scigantic-facebase
|
|
2
|
+
|
|
3
|
+
Search [FaceBase](https://www.facebase.org), the craniofacial research data hub, and read its open-access data from Python.
|
|
4
|
+
|
|
5
|
+
FaceBase holds about 1,070 open datasets: micro-CT and light-sheet imaging, ChIP-seq and RNA-seq tracks, expression tables and more, from mouse, zebrafish, human and other models. This package is a typed layer over FaceBase's open metadata API and file store. No key or login is needed.
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
pip install scigantic-facebase
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Use
|
|
12
|
+
|
|
13
|
+
```python
|
|
14
|
+
import scigantic_facebase as fb
|
|
15
|
+
|
|
16
|
+
hits = fb.search("zebrafish thyroid microCT") # newest first
|
|
17
|
+
ds = hits[0]
|
|
18
|
+
print(ds.title, ds.doi_url)
|
|
19
|
+
print(fb.cite(ds))
|
|
20
|
+
|
|
21
|
+
files = fb.files(ds) # size, MD5, download URL
|
|
22
|
+
fb.download(files[0], "data/") # MD5-checked
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
Search matches every word against the whole record, and treats "zebrafish" and "Danio rerio", "mouse" and "Mus musculus" as the same thing.
|
|
26
|
+
|
|
27
|
+
## Read a micro-CT scan without downloading it
|
|
28
|
+
|
|
29
|
+
Many micro-CT datasets are a zip of TIFF slices. A zip keeps its directory at the end, so one slice costs one HTTP range request.
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
stack = fb.microct.open_stack(files[0])
|
|
33
|
+
len(stack) # 596 slices
|
|
34
|
+
stack.voxel_size_um # 10.5, from the scan log
|
|
35
|
+
img = stack.read_slice(300)
|
|
36
|
+
vol = stack.volume(step=10) # every 10th slice, (z, y, x)
|
|
37
|
+
stack.bytes_fetched # a few MB of a 49 MB file
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
`fb.microct.thyroid_atlas()` lists the zebrafish thyroid-hormone atlas: 81 head scans from 12 to 25 mm standard length, raised hypothyroid, euthyroid or hyperthyroid.
|
|
41
|
+
|
|
42
|
+
`fb.open_remote(file)` gives any FaceBase file as a seekable file object backed by range requests, which `zipfile`, `tifffile` and `h5py` accept.
|
|
43
|
+
|
|
44
|
+
## What is not here
|
|
45
|
+
|
|
46
|
+
Protected human-subjects datasets sit behind FaceBase's data access request. Their metadata is public and `fb.search(include_protected=True)` lists them, but reading their files raises `FacebaseAccessError`.
|
|
47
|
+
|
|
48
|
+
## Terms of use
|
|
49
|
+
|
|
50
|
+
Copyright to FaceBase data belongs to the contributing investigators. FaceBase's [Terms of Use](https://www.facebase.org/policies/tou/) ask users to cite the dataset, acknowledge FaceBase, and send other people to FaceBase instead of circulating downloaded files. `fb.cite()` produces the citation. This package fetches from FaceBase on request and does not mirror anything.
|
|
51
|
+
|
|
52
|
+
## License
|
|
53
|
+
|
|
54
|
+
MIT-0 for this code. The data carries its own terms.
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "scigantic-facebase"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Search FaceBase, the craniofacial research data hub, and read its open-access micro-CT, imaging and sequencing data from Python: a typed catalog over FaceBase's open metadata API, citations in the form its Terms of Use ask for, ranged reads of remote files, and single-slice reads from zipped micro-CT stacks without downloading them."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT-0" }
|
|
12
|
+
authors = [{ name = "Scigantic" }]
|
|
13
|
+
keywords = ["facebase", "craniofacial", "micro-ct", "zebrafish", "deriva", "ermrest", "bioinformatics"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"License :: OSI Approved :: MIT No Attribution License (MIT-0)",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Programming Language :: Python :: 3.10",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"Programming Language :: Python :: 3.14",
|
|
22
|
+
"Typing :: Typed",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Bio-Informatics",
|
|
24
|
+
"Topic :: Scientific/Engineering :: Image Processing",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
dependencies = [
|
|
28
|
+
"requests>=2.28,<3",
|
|
29
|
+
"numpy>=1.24,<3",
|
|
30
|
+
"tifffile>=2023.7,<2027",
|
|
31
|
+
"imagecodecs>=2023.9",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[project.optional-dependencies]
|
|
35
|
+
dev = ["pytest>=7", "mypy>=1.10", "types-requests"]
|
|
36
|
+
|
|
37
|
+
[project.urls]
|
|
38
|
+
Homepage = "https://scigantic.com"
|
|
39
|
+
Repository = "https://github.com/Scigantic/scigantic-facebase"
|
|
40
|
+
Issues = "https://github.com/Scigantic/scigantic-facebase/issues"
|
|
41
|
+
FaceBase = "https://www.facebase.org"
|
|
42
|
+
|
|
43
|
+
[tool.setuptools.packages.find]
|
|
44
|
+
where = ["src"]
|
|
45
|
+
|
|
46
|
+
[tool.setuptools.package-data]
|
|
47
|
+
scigantic_facebase = ["py.typed"]
|
|
48
|
+
|
|
49
|
+
[tool.pytest.ini_options]
|
|
50
|
+
testpaths = ["tests"]
|
|
51
|
+
markers = ["live: hits facebase.org"]
|
|
52
|
+
|
|
53
|
+
[tool.mypy]
|
|
54
|
+
strict = true
|
|
55
|
+
|
|
56
|
+
[[tool.mypy.overrides]]
|
|
57
|
+
module = ["tifffile", "tifffile.*"]
|
|
58
|
+
ignore_missing_imports = true
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Search FaceBase and read its open-access craniofacial data from Python.
|
|
2
|
+
|
|
3
|
+
import scigantic_facebase as fb
|
|
4
|
+
|
|
5
|
+
hits = fb.search("zebrafish thyroid microCT")
|
|
6
|
+
fb.cite(hits[0])
|
|
7
|
+
stack = fb.microct.open_stack(fb.files(hits[0])[0])
|
|
8
|
+
stack.read_slice(300)
|
|
9
|
+
|
|
10
|
+
Metadata comes from FaceBase's open ERMrest API and files from its open-access
|
|
11
|
+
file store, both anonymous. Copyright for the data stays with the contributing
|
|
12
|
+
investigators and FaceBase's Terms of Use apply: cite the dataset DOI
|
|
13
|
+
(`cite()`), and send other people to FaceBase rather than redistributing files.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from . import microct
|
|
17
|
+
from ._client import FacebaseAccessError, FacebaseError, FacebaseNotFoundError
|
|
18
|
+
from ._version import __version__
|
|
19
|
+
from .catalog import cite, contributors, count, files, get, project, search
|
|
20
|
+
from .models import Dataset, File, Project
|
|
21
|
+
from .remote import RangeFile, download, open_remote
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"Dataset",
|
|
25
|
+
"FacebaseAccessError",
|
|
26
|
+
"FacebaseError",
|
|
27
|
+
"FacebaseNotFoundError",
|
|
28
|
+
"File",
|
|
29
|
+
"Project",
|
|
30
|
+
"RangeFile",
|
|
31
|
+
"__version__",
|
|
32
|
+
"cite",
|
|
33
|
+
"contributors",
|
|
34
|
+
"count",
|
|
35
|
+
"download",
|
|
36
|
+
"files",
|
|
37
|
+
"get",
|
|
38
|
+
"microct",
|
|
39
|
+
"open_remote",
|
|
40
|
+
"project",
|
|
41
|
+
"search",
|
|
42
|
+
]
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Shared HTTP plumbing for FaceBase: one lazily-built requests.Session,
|
|
2
|
+
retry with backoff on transient failures, and the two FaceBase endpoints this
|
|
3
|
+
package talks to.
|
|
4
|
+
|
|
5
|
+
FaceBase publishes no rate limit. Its metadata (ERMrest) and open-access file
|
|
6
|
+
store (Hatrac) answer anonymous requests, so there is no key or login here.
|
|
7
|
+
Retries cover 429/5xx and connection errors only. A 404 is a real answer and
|
|
8
|
+
raises FacebaseNotFoundError at once. A 401 or 403 means the data is behind
|
|
9
|
+
FaceBase's data access request process and raises FacebaseAccessError.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import threading
|
|
15
|
+
import time
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
import requests
|
|
19
|
+
|
|
20
|
+
from ._version import __version__
|
|
21
|
+
|
|
22
|
+
SITE = "https://www.facebase.org"
|
|
23
|
+
ERMREST = f"{SITE}/ermrest/catalog/1"
|
|
24
|
+
|
|
25
|
+
_USER_AGENT = f"scigantic-facebase/{__version__} (+https://scigantic.com; mailto:support@scigantic.com)"
|
|
26
|
+
|
|
27
|
+
_MAX_RETRIES = 4
|
|
28
|
+
_RETRY_STATUS_CODES = {429, 500, 502, 503, 504}
|
|
29
|
+
|
|
30
|
+
_session: requests.Session | None = None
|
|
31
|
+
_session_lock = threading.Lock()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class FacebaseError(Exception):
|
|
35
|
+
"""An HTTP error after retries were exhausted, or a malformed response."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class FacebaseNotFoundError(FacebaseError):
|
|
39
|
+
"""A dataset, file or path that does not exist (HTTP 404)."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class FacebaseAccessError(FacebaseError):
|
|
43
|
+
"""The data needs authentication (HTTP 401 or 403). FaceBase holds
|
|
44
|
+
protected human-subjects data behind a data access request, and this
|
|
45
|
+
package only reads the open-access part."""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def get_session() -> requests.Session:
|
|
49
|
+
global _session
|
|
50
|
+
if _session is None:
|
|
51
|
+
with _session_lock:
|
|
52
|
+
if _session is None:
|
|
53
|
+
_session = requests.Session()
|
|
54
|
+
_session.headers["User-Agent"] = _USER_AGENT
|
|
55
|
+
return _session
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def send(
|
|
59
|
+
method: str,
|
|
60
|
+
url: str,
|
|
61
|
+
params: dict[str, Any] | None = None,
|
|
62
|
+
headers: dict[str, str] | None = None,
|
|
63
|
+
stream: bool = False,
|
|
64
|
+
timeout: float = 60.0,
|
|
65
|
+
) -> requests.Response:
|
|
66
|
+
"""Issue one request with retry on 429/5xx and connection errors."""
|
|
67
|
+
session = get_session()
|
|
68
|
+
last_error: Exception | None = None
|
|
69
|
+
for attempt in range(_MAX_RETRIES + 1):
|
|
70
|
+
try:
|
|
71
|
+
resp = session.request(
|
|
72
|
+
method, url, params=params, headers=headers, stream=stream, timeout=timeout
|
|
73
|
+
)
|
|
74
|
+
except requests.RequestException as exc:
|
|
75
|
+
last_error = exc
|
|
76
|
+
if attempt == _MAX_RETRIES:
|
|
77
|
+
break
|
|
78
|
+
time.sleep(1.5 * (2**attempt))
|
|
79
|
+
continue
|
|
80
|
+
if resp.status_code == 404:
|
|
81
|
+
resp.close()
|
|
82
|
+
raise FacebaseNotFoundError(f"404 for {resp.url}")
|
|
83
|
+
if resp.status_code in (401, 403):
|
|
84
|
+
resp.close()
|
|
85
|
+
raise FacebaseAccessError(
|
|
86
|
+
f"HTTP {resp.status_code} for {resp.url}: this data is not open access. "
|
|
87
|
+
"FaceBase releases protected human-subjects data only through a data access request."
|
|
88
|
+
)
|
|
89
|
+
if resp.status_code in _RETRY_STATUS_CODES and attempt < _MAX_RETRIES:
|
|
90
|
+
resp.close()
|
|
91
|
+
time.sleep(1.5 * (2**attempt))
|
|
92
|
+
continue
|
|
93
|
+
if resp.status_code >= 400:
|
|
94
|
+
body = resp.text[:300]
|
|
95
|
+
resp.close()
|
|
96
|
+
raise FacebaseError(f"HTTP {resp.status_code} for {resp.url}: {body}")
|
|
97
|
+
return resp
|
|
98
|
+
raise FacebaseError(f"request to {url} failed after {_MAX_RETRIES + 1} attempts: {last_error}")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def ermrest(path: str) -> Any:
|
|
102
|
+
"""GET an ERMrest path (everything after /catalog/1) and return parsed JSON."""
|
|
103
|
+
resp = send("GET", f"{ERMREST}/{path.lstrip('/')}")
|
|
104
|
+
try:
|
|
105
|
+
return resp.json()
|
|
106
|
+
except ValueError as exc:
|
|
107
|
+
raise FacebaseError(f"non-JSON response from {resp.url}") from exc
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Search FaceBase datasets and list their files through the open ERMrest
|
|
2
|
+
metadata API. Everything here is a metadata call; no file bytes move."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import re
|
|
7
|
+
from typing import Any
|
|
8
|
+
from functools import lru_cache
|
|
9
|
+
from urllib.parse import quote
|
|
10
|
+
|
|
11
|
+
from ._client import SITE, FacebaseNotFoundError, ermrest
|
|
12
|
+
from .models import Dataset, File, Project
|
|
13
|
+
|
|
14
|
+
_OPEN = "released=true&protected_human_subjects=false"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _q(value: str) -> str:
|
|
18
|
+
return quote(value, safe="")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# FaceBase records name species inconsistently ("zebrafish" in some titles,
|
|
22
|
+
# "Danio rerio" in most). A search word that is a common name also matches
|
|
23
|
+
# the Latin name, and the other way round.
|
|
24
|
+
_SYNONYMS = {
|
|
25
|
+
"zebrafish": ("zebrafish", "danio rerio"),
|
|
26
|
+
"danio": ("zebrafish", "danio rerio"),
|
|
27
|
+
"mouse": ("mouse", "mice", "mus musculus"),
|
|
28
|
+
"mice": ("mouse", "mice", "mus musculus"),
|
|
29
|
+
"human": ("human", "homo sapiens"),
|
|
30
|
+
"chick": ("chick", "gallus gallus"),
|
|
31
|
+
"microct": ("microct", "micro-ct", "micro ct", "\u00b5ct"),
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _filter(text: str | None, include_protected: bool) -> str:
|
|
36
|
+
parts = [_OPEN if not include_protected else "released=true"]
|
|
37
|
+
for word in (text or "").split():
|
|
38
|
+
options = _SYNONYMS.get(word.lower(), (word,))
|
|
39
|
+
pattern = "|".join(re.escape(o) for o in options)
|
|
40
|
+
parts.append(f"*::ciregexp::{_q(pattern)}")
|
|
41
|
+
return "&".join(parts)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _str(row: dict[str, Any], key: str) -> str | None:
|
|
45
|
+
val = row.get(key)
|
|
46
|
+
return val if isinstance(val, str) else None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _dataset(row: dict[str, Any]) -> Dataset:
|
|
50
|
+
project_id = row.get("project")
|
|
51
|
+
return Dataset(
|
|
52
|
+
rid=str(row["RID"]),
|
|
53
|
+
accession=_str(row, "accession"),
|
|
54
|
+
title=str(row.get("title") or ""),
|
|
55
|
+
description=str(row.get("description") or ""),
|
|
56
|
+
release_date=_str(row, "release_date"),
|
|
57
|
+
doi=_str(row, "DOI"),
|
|
58
|
+
project_id=project_id if isinstance(project_id, int) else None,
|
|
59
|
+
protected=bool(row.get("protected_human_subjects")),
|
|
60
|
+
internal_id=int(row["id"]),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def search(text: str | None = None, *, limit: int = 50, include_protected: bool = False) -> list[Dataset]:
|
|
65
|
+
"""Find datasets, newest first. Every word in `text` must appear somewhere
|
|
66
|
+
in the dataset's title, description or keywords (case-insensitive).
|
|
67
|
+
Protected human-subjects datasets are left out unless include_protected=True;
|
|
68
|
+
their metadata is public but their files need a data access request."""
|
|
69
|
+
path = f"entity/isa:dataset/{_filter(text, include_protected)}@sort(release_date::desc::,RID)"
|
|
70
|
+
rows = ermrest(f"{path}?limit={int(limit)}")
|
|
71
|
+
return [_dataset(r) for r in rows]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def count(text: str | None = None, *, include_protected: bool = False) -> int:
|
|
75
|
+
"""How many datasets match `text` (same rules as search)."""
|
|
76
|
+
rows = ermrest(f"aggregate/isa:dataset/{_filter(text, include_protected)}/n:=cnt(RID)")
|
|
77
|
+
return int(rows[0]["n"])
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def get(rid: str) -> Dataset:
|
|
81
|
+
"""One dataset by FaceBase record id (e.g. '2E-YW20')."""
|
|
82
|
+
rows = ermrest(f"entity/isa:dataset/RID={_q(rid)}")
|
|
83
|
+
if not rows:
|
|
84
|
+
raise FacebaseNotFoundError(f"no FaceBase dataset {rid!r}")
|
|
85
|
+
return _dataset(rows[0])
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@lru_cache(maxsize=1)
|
|
89
|
+
def _formats() -> dict[str, str]:
|
|
90
|
+
out: dict[str, str] = {}
|
|
91
|
+
for r in ermrest("entity/vocab:file_format?limit=500"):
|
|
92
|
+
key = r.get("id") or r.get("ID")
|
|
93
|
+
name = r.get("name") or r.get("Name")
|
|
94
|
+
if key and name:
|
|
95
|
+
out[str(key)] = str(name)
|
|
96
|
+
return out
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def files(dataset: Dataset | str) -> list[File]:
|
|
100
|
+
"""Every file in a dataset, with size, MD5 and an absolute download URL."""
|
|
101
|
+
ds = get(dataset) if isinstance(dataset, str) else dataset
|
|
102
|
+
rows = ermrest(f"entity/isa:file/dataset={_q(ds.rid)}@sort(filename)?limit=20000")
|
|
103
|
+
fmt = _formats()
|
|
104
|
+
return [
|
|
105
|
+
File(
|
|
106
|
+
rid=str(r["RID"]),
|
|
107
|
+
dataset_rid=ds.rid,
|
|
108
|
+
filename=str(r["filename"]),
|
|
109
|
+
size=int(r.get("byte_count") or 0),
|
|
110
|
+
md5=_str(r, "md5"),
|
|
111
|
+
url=SITE + str(r["url"]),
|
|
112
|
+
format=fmt.get(str(r["file_format"])) if r.get("file_format") else None,
|
|
113
|
+
relative_path=_str(r, "relative_path"),
|
|
114
|
+
description=_str(r, "description"),
|
|
115
|
+
protected=ds.protected,
|
|
116
|
+
)
|
|
117
|
+
for r in rows
|
|
118
|
+
]
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def project(dataset: Dataset) -> Project | None:
|
|
122
|
+
"""The project (grant and investigator group) a dataset belongs to."""
|
|
123
|
+
if dataset.project_id is None:
|
|
124
|
+
return None
|
|
125
|
+
rows = ermrest(f"entity/isa:project/id={dataset.project_id}")
|
|
126
|
+
if not rows:
|
|
127
|
+
return None
|
|
128
|
+
r = rows[0]
|
|
129
|
+
return Project(
|
|
130
|
+
id=int(r["id"]),
|
|
131
|
+
name=str(r.get("name") or ""),
|
|
132
|
+
doi=_str(r, "DOI"),
|
|
133
|
+
funding=_str(r, "funding"),
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def contributors(dataset: Dataset) -> list[str]:
|
|
138
|
+
rows = ermrest(f"attribute/isa:dataset_contributor/dataset_id={dataset.internal_id}/order,full_name@sort(order)")
|
|
139
|
+
return [str(r["full_name"]) for r in rows if r.get("full_name")]
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def cite(dataset: Dataset | str) -> str:
|
|
143
|
+
"""The citation FaceBase's Terms of Use ask for: authors, title,
|
|
144
|
+
FaceBase Consortium, DOI link and release year."""
|
|
145
|
+
ds = get(dataset) if isinstance(dataset, str) else dataset
|
|
146
|
+
authors = ", ".join(contributors(ds)) or "FaceBase Consortium contributors"
|
|
147
|
+
year = (ds.release_date or "")[:4]
|
|
148
|
+
link = ds.doi_url or ds.url
|
|
149
|
+
return f"{authors}. {ds.title}. FaceBase Consortium {link}" + (f" ({year})." if year else ".")
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Zip-of-TIFF micro-CT stacks, read in place.
|
|
2
|
+
|
|
3
|
+
Many FaceBase micro-CT datasets ship each scan as one .zip holding a stack of
|
|
4
|
+
reconstructed TIFF slices plus a SkyScan scan log. A zip keeps a directory at
|
|
5
|
+
its end, so a single slice can be read with one HTTP range request; a 50 MB
|
|
6
|
+
scan is never downloaded just to look at one cross-section.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import io
|
|
12
|
+
import re
|
|
13
|
+
import threading
|
|
14
|
+
import zipfile
|
|
15
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
import numpy.typing as npt
|
|
21
|
+
|
|
22
|
+
from . import catalog
|
|
23
|
+
from .models import Dataset, File
|
|
24
|
+
from .remote import RangeFile, open_remote
|
|
25
|
+
|
|
26
|
+
_TIF = re.compile(r"\.tiff?$", re.IGNORECASE)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Stack:
|
|
30
|
+
"""A remote zip of TIFF slices. Create with open_stack()."""
|
|
31
|
+
|
|
32
|
+
def __init__(self, source: RangeFile, name: str = "") -> None:
|
|
33
|
+
self.name = name
|
|
34
|
+
self._source = source
|
|
35
|
+
self._zip = zipfile.ZipFile(source)
|
|
36
|
+
self._lock = threading.Lock()
|
|
37
|
+
self.slice_names: list[str] = sorted(
|
|
38
|
+
i.filename for i in self._zip.infolist() if _TIF.search(i.filename) and not i.is_dir()
|
|
39
|
+
)
|
|
40
|
+
self._logs = [i.filename for i in self._zip.infolist() if i.filename.lower().endswith(".log")]
|
|
41
|
+
|
|
42
|
+
def __len__(self) -> int:
|
|
43
|
+
return len(self.slice_names)
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def requests(self) -> int:
|
|
47
|
+
"""HTTP requests made so far, to see how little was fetched."""
|
|
48
|
+
return self._source.requests
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def bytes_fetched(self) -> int:
|
|
52
|
+
return self._source.bytes_fetched
|
|
53
|
+
|
|
54
|
+
def read_slice(self, index: int) -> npt.NDArray[Any]:
|
|
55
|
+
"""One slice as a 2-D array (decoded with tifffile)."""
|
|
56
|
+
import tifffile
|
|
57
|
+
|
|
58
|
+
with self._lock:
|
|
59
|
+
raw = self._zip.read(self.slice_names[index])
|
|
60
|
+
return np.asarray(tifffile.imread(io.BytesIO(raw)))
|
|
61
|
+
|
|
62
|
+
def volume(self, *, step: int = 1, start: int = 0, stop: int | None = None, workers: int = 4) -> npt.NDArray[Any]:
|
|
63
|
+
"""Slices start:stop:step stacked into a (z, y, x) array. Slices are
|
|
64
|
+
read in parallel; each worker reuses the same cached connection."""
|
|
65
|
+
idx = list(range(start, len(self) if stop is None else stop, step))
|
|
66
|
+
if not idx:
|
|
67
|
+
raise ValueError("no slices selected")
|
|
68
|
+
with ThreadPoolExecutor(max_workers=max(1, workers)) as pool:
|
|
69
|
+
slices = list(pool.map(self.read_slice, idx))
|
|
70
|
+
return np.stack(slices)
|
|
71
|
+
|
|
72
|
+
def scan_log(self) -> dict[str, dict[str, str]]:
|
|
73
|
+
"""The SkyScan scan log as {section: {key: value}}; empty when the
|
|
74
|
+
zip has none."""
|
|
75
|
+
if not self._logs:
|
|
76
|
+
return {}
|
|
77
|
+
with self._lock:
|
|
78
|
+
text = self._zip.read(self._logs[0]).decode("latin-1")
|
|
79
|
+
out: dict[str, dict[str, str]] = {}
|
|
80
|
+
section = ""
|
|
81
|
+
for line in text.splitlines():
|
|
82
|
+
line = line.strip()
|
|
83
|
+
if line.startswith("[") and line.endswith("]"):
|
|
84
|
+
section = line[1:-1]
|
|
85
|
+
out.setdefault(section, {})
|
|
86
|
+
elif "=" in line and section:
|
|
87
|
+
key, _, val = line.partition("=")
|
|
88
|
+
out[section][key.strip()] = val.strip()
|
|
89
|
+
return out
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def voxel_size_um(self) -> float | None:
|
|
93
|
+
"""Isotropic voxel edge in micrometres, from the scan log."""
|
|
94
|
+
log = self.scan_log()
|
|
95
|
+
for section in ("Reconstruction", "Acquisition"):
|
|
96
|
+
for key in ("Pixel Size (um)", "Image Pixel Size (um)", "Scaled Image Pixel Size (um)"):
|
|
97
|
+
val = log.get(section, {}).get(key)
|
|
98
|
+
if val:
|
|
99
|
+
try:
|
|
100
|
+
return float(val)
|
|
101
|
+
except ValueError:
|
|
102
|
+
continue
|
|
103
|
+
return None
|
|
104
|
+
|
|
105
|
+
def close(self) -> None:
|
|
106
|
+
self._zip.close()
|
|
107
|
+
|
|
108
|
+
def __enter__(self) -> Stack:
|
|
109
|
+
return self
|
|
110
|
+
|
|
111
|
+
def __exit__(self, *exc: object) -> None:
|
|
112
|
+
self.close()
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def open_stack(file: File) -> Stack:
|
|
116
|
+
"""Open a zip-of-TIFFs file from FaceBase without downloading it."""
|
|
117
|
+
return Stack(open_remote(file), name=file.filename)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclass(frozen=True)
|
|
121
|
+
class ThyroidScan:
|
|
122
|
+
"""One scan in the zebrafish thyroid-hormone craniofacial atlas."""
|
|
123
|
+
|
|
124
|
+
dataset: Dataset
|
|
125
|
+
condition: str # hypothyroid, euthyroid or hyperthyroid
|
|
126
|
+
standard_length_mm: float
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
_TITLE = re.compile(r"microCT scan of ([\d.]+) mm (hypothyroid|euthyroid|hyperthyroid) Danio rerio head", re.IGNORECASE)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def parse_thyroid_title(title: str) -> tuple[float, str] | None:
|
|
133
|
+
"""(standard length in mm, condition) from an atlas dataset title."""
|
|
134
|
+
m = _TITLE.search(title)
|
|
135
|
+
return (float(m.group(1)), m.group(2).lower()) if m else None
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def thyroid_atlas() -> list[ThyroidScan]:
|
|
139
|
+
"""The zebrafish head micro-CT series: fish from 12 to 25 mm standard
|
|
140
|
+
length raised hypothyroid, euthyroid (normal) or hyperthyroid, one dataset
|
|
141
|
+
per fish. Sorted by condition then length."""
|
|
142
|
+
scans = []
|
|
143
|
+
for ds in catalog.search("microCT Danio rerio head thyroid", limit=300):
|
|
144
|
+
parsed = parse_thyroid_title(ds.title)
|
|
145
|
+
if parsed:
|
|
146
|
+
scans.append(ThyroidScan(ds, parsed[1], parsed[0]))
|
|
147
|
+
return sorted(scans, key=lambda s: (s.condition, s.standard_length_mm, s.dataset.rid))
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Typed records for FaceBase datasets and files."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
|
|
7
|
+
from ._client import SITE
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True)
|
|
11
|
+
class Dataset:
|
|
12
|
+
"""One FaceBase dataset. `rid` is FaceBase's record id, the same string
|
|
13
|
+
that appears in the dataset's landing page and DOI."""
|
|
14
|
+
|
|
15
|
+
rid: str
|
|
16
|
+
accession: str | None
|
|
17
|
+
title: str
|
|
18
|
+
description: str
|
|
19
|
+
release_date: str | None
|
|
20
|
+
doi: str | None
|
|
21
|
+
project_id: int | None
|
|
22
|
+
protected: bool
|
|
23
|
+
internal_id: int
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def url(self) -> str:
|
|
27
|
+
return f"{SITE}/id/{self.rid}"
|
|
28
|
+
|
|
29
|
+
@property
|
|
30
|
+
def doi_url(self) -> str | None:
|
|
31
|
+
return f"https://doi.org/{self.doi}" if self.doi else None
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def open_access(self) -> bool:
|
|
35
|
+
"""False for datasets behind FaceBase's data access request."""
|
|
36
|
+
return not self.protected
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True)
|
|
40
|
+
class Project:
|
|
41
|
+
id: int
|
|
42
|
+
name: str
|
|
43
|
+
doi: str | None
|
|
44
|
+
funding: str | None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True)
|
|
48
|
+
class File:
|
|
49
|
+
"""One file in a FaceBase dataset. `url` is absolute and anonymous for
|
|
50
|
+
open-access datasets; byte-range requests are honoured."""
|
|
51
|
+
|
|
52
|
+
rid: str
|
|
53
|
+
dataset_rid: str
|
|
54
|
+
filename: str
|
|
55
|
+
size: int
|
|
56
|
+
md5: str | None
|
|
57
|
+
url: str
|
|
58
|
+
format: str | None = None
|
|
59
|
+
relative_path: str | None = None
|
|
60
|
+
description: str | None = None
|
|
61
|
+
protected: bool = field(default=False, compare=False)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def path(self) -> str:
|
|
65
|
+
"""Path within the dataset, folders included."""
|
|
66
|
+
return f"{self.relative_path or ''}{self.filename}"
|
|
File without changes
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""Read FaceBase files over HTTP without downloading them first.
|
|
2
|
+
|
|
3
|
+
`RangeFile` is a seekable, read-only file object backed by HTTP Range
|
|
4
|
+
requests with a small block cache. Hand it to zipfile.ZipFile or tifffile and
|
|
5
|
+
only the bytes they actually touch cross the network. For a 50 MB zip of
|
|
6
|
+
600 TIFF slices, opening the archive reads the central directory (one small
|
|
7
|
+
request) and each slice costs one request of its own size.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import hashlib
|
|
13
|
+
import io
|
|
14
|
+
import os
|
|
15
|
+
import threading
|
|
16
|
+
from collections import OrderedDict
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from ._client import FacebaseError, send
|
|
21
|
+
from .models import File
|
|
22
|
+
|
|
23
|
+
_BLOCK = 1 << 20 # 1 MiB
|
|
24
|
+
_CACHE_BLOCKS = 32
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class RangeFile(io.RawIOBase):
|
|
28
|
+
"""Seekable read-only view of one remote file."""
|
|
29
|
+
|
|
30
|
+
def __init__(self, url: str, size: int | None = None) -> None:
|
|
31
|
+
super().__init__()
|
|
32
|
+
self.url = url
|
|
33
|
+
self._pos = 0
|
|
34
|
+
self._lock = threading.Lock()
|
|
35
|
+
self._cache: OrderedDict[int, bytes] = OrderedDict()
|
|
36
|
+
self.requests = 0
|
|
37
|
+
self.bytes_fetched = 0
|
|
38
|
+
self._size = size if size is not None else self._probe_size()
|
|
39
|
+
|
|
40
|
+
def _probe_size(self) -> int:
|
|
41
|
+
resp = send("GET", self.url, headers={"Range": "bytes=0-0"}, stream=True)
|
|
42
|
+
try:
|
|
43
|
+
cr = resp.headers.get("Content-Range", "")
|
|
44
|
+
if "/" in cr and cr.rsplit("/", 1)[1].isdigit():
|
|
45
|
+
return int(cr.rsplit("/", 1)[1])
|
|
46
|
+
length = resp.headers.get("Content-Length")
|
|
47
|
+
if resp.status_code == 200 and length and length.isdigit():
|
|
48
|
+
return int(length)
|
|
49
|
+
finally:
|
|
50
|
+
resp.close()
|
|
51
|
+
raise FacebaseError(f"cannot determine size of {self.url}")
|
|
52
|
+
|
|
53
|
+
def __len__(self) -> int:
|
|
54
|
+
return self._size
|
|
55
|
+
|
|
56
|
+
def readable(self) -> bool:
|
|
57
|
+
return True
|
|
58
|
+
|
|
59
|
+
def seekable(self) -> bool:
|
|
60
|
+
return True
|
|
61
|
+
|
|
62
|
+
def tell(self) -> int:
|
|
63
|
+
return self._pos
|
|
64
|
+
|
|
65
|
+
def seek(self, offset: int, whence: int = os.SEEK_SET) -> int:
|
|
66
|
+
if whence == os.SEEK_SET:
|
|
67
|
+
self._pos = offset
|
|
68
|
+
elif whence == os.SEEK_CUR:
|
|
69
|
+
self._pos += offset
|
|
70
|
+
elif whence == os.SEEK_END:
|
|
71
|
+
self._pos = self._size + offset
|
|
72
|
+
else:
|
|
73
|
+
raise ValueError(f"invalid whence: {whence}")
|
|
74
|
+
self._pos = max(self._pos, 0)
|
|
75
|
+
return self._pos
|
|
76
|
+
|
|
77
|
+
def _block(self, index: int) -> bytes:
|
|
78
|
+
with self._lock:
|
|
79
|
+
hit = self._cache.get(index)
|
|
80
|
+
if hit is not None:
|
|
81
|
+
self._cache.move_to_end(index)
|
|
82
|
+
return hit
|
|
83
|
+
start = index * _BLOCK
|
|
84
|
+
end = min(start + _BLOCK, self._size) - 1
|
|
85
|
+
resp = send("GET", self.url, headers={"Range": f"bytes={start}-{end}"})
|
|
86
|
+
data = resp.content
|
|
87
|
+
if resp.status_code == 200 and len(data) != end - start + 1:
|
|
88
|
+
raise FacebaseError(f"{self.url} ignored the Range header")
|
|
89
|
+
with self._lock:
|
|
90
|
+
self.requests += 1
|
|
91
|
+
self.bytes_fetched += len(data)
|
|
92
|
+
self._cache[index] = data
|
|
93
|
+
while len(self._cache) > _CACHE_BLOCKS:
|
|
94
|
+
self._cache.popitem(last=False)
|
|
95
|
+
return data
|
|
96
|
+
|
|
97
|
+
def read(self, size: int = -1) -> bytes:
|
|
98
|
+
if self._pos >= self._size:
|
|
99
|
+
return b""
|
|
100
|
+
if size is None or size < 0:
|
|
101
|
+
size = self._size - self._pos
|
|
102
|
+
stop = min(self._pos + size, self._size)
|
|
103
|
+
# A request larger than the cache would thrash it; fetch it in one go.
|
|
104
|
+
if stop - self._pos > _BLOCK * 4:
|
|
105
|
+
resp = send("GET", self.url, headers={"Range": f"bytes={self._pos}-{stop - 1}"})
|
|
106
|
+
data = resp.content
|
|
107
|
+
with self._lock:
|
|
108
|
+
self.requests += 1
|
|
109
|
+
self.bytes_fetched += len(data)
|
|
110
|
+
self._pos += len(data)
|
|
111
|
+
return data
|
|
112
|
+
out = bytearray()
|
|
113
|
+
pos = self._pos
|
|
114
|
+
while pos < stop:
|
|
115
|
+
block = self._block(pos // _BLOCK)
|
|
116
|
+
off = pos % _BLOCK
|
|
117
|
+
take = block[off : off + (stop - pos)]
|
|
118
|
+
if not take:
|
|
119
|
+
break
|
|
120
|
+
out += take
|
|
121
|
+
pos += len(take)
|
|
122
|
+
self._pos = pos
|
|
123
|
+
return bytes(out)
|
|
124
|
+
|
|
125
|
+
def readinto(self, b: Any) -> int:
|
|
126
|
+
data = self.read(len(b))
|
|
127
|
+
b[: len(data)] = data
|
|
128
|
+
return len(data)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def open_remote(file: File) -> RangeFile:
|
|
132
|
+
"""Open a FaceBase file as a seekable file object (nothing downloaded)."""
|
|
133
|
+
return RangeFile(file.url, size=file.size or None)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def download(file: File, dest: str | os.PathLike[str], *, verify: bool = True, chunk: int = 1 << 20) -> Path:
|
|
137
|
+
"""Download a file to `dest` (a directory or a file path) and return the
|
|
138
|
+
path. Checks size and MD5 against FaceBase's record unless verify=False.
|
|
139
|
+
An existing file with the right size and MD5 is kept, so reruns are free."""
|
|
140
|
+
target = Path(dest)
|
|
141
|
+
if target.is_dir():
|
|
142
|
+
target = target / file.filename
|
|
143
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
144
|
+
if target.exists() and target.stat().st_size == file.size and (not verify or not file.md5 or _md5(target) == file.md5):
|
|
145
|
+
return target
|
|
146
|
+
tmp = target.with_suffix(target.suffix + ".part")
|
|
147
|
+
md5 = hashlib.md5()
|
|
148
|
+
resp = send("GET", file.url, stream=True, timeout=120.0)
|
|
149
|
+
try:
|
|
150
|
+
with open(tmp, "wb") as fh:
|
|
151
|
+
for part in resp.iter_content(chunk):
|
|
152
|
+
fh.write(part)
|
|
153
|
+
md5.update(part)
|
|
154
|
+
finally:
|
|
155
|
+
resp.close()
|
|
156
|
+
if file.size and tmp.stat().st_size != file.size:
|
|
157
|
+
tmp.unlink(missing_ok=True)
|
|
158
|
+
raise FacebaseError(f"{file.filename}: got {tmp.stat().st_size if tmp.exists() else 0} bytes, expected {file.size}")
|
|
159
|
+
if verify and file.md5 and md5.hexdigest() != file.md5:
|
|
160
|
+
tmp.unlink(missing_ok=True)
|
|
161
|
+
raise FacebaseError(f"{file.filename}: MD5 mismatch")
|
|
162
|
+
tmp.replace(target)
|
|
163
|
+
return target
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _md5(path: Path) -> str:
|
|
167
|
+
h = hashlib.md5()
|
|
168
|
+
with open(path, "rb") as fh:
|
|
169
|
+
for part in iter(lambda: fh.read(1 << 20), b""):
|
|
170
|
+
h.update(part)
|
|
171
|
+
return h.hexdigest()
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scigantic-facebase
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Search FaceBase, the craniofacial research data hub, and read its open-access micro-CT, imaging and sequencing data from Python: a typed catalog over FaceBase's open metadata API, citations in the form its Terms of Use ask for, ranged reads of remote files, and single-slice reads from zipped micro-CT stacks without downloading them.
|
|
5
|
+
Author: Scigantic
|
|
6
|
+
License: MIT-0
|
|
7
|
+
Project-URL: Homepage, https://scigantic.com
|
|
8
|
+
Project-URL: Repository, https://github.com/Scigantic/scigantic-facebase
|
|
9
|
+
Project-URL: Issues, https://github.com/Scigantic/scigantic-facebase/issues
|
|
10
|
+
Project-URL: FaceBase, https://www.facebase.org
|
|
11
|
+
Keywords: facebase,craniofacial,micro-ct,zebrafish,deriva,ermrest,bioinformatics
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT No Attribution License (MIT-0)
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: requests<3,>=2.28
|
|
26
|
+
Requires-Dist: numpy<3,>=1.24
|
|
27
|
+
Requires-Dist: tifffile<2027,>=2023.7
|
|
28
|
+
Requires-Dist: imagecodecs>=2023.9
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
31
|
+
Requires-Dist: mypy>=1.10; extra == "dev"
|
|
32
|
+
Requires-Dist: types-requests; extra == "dev"
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# scigantic-facebase
|
|
36
|
+
|
|
37
|
+
Search [FaceBase](https://www.facebase.org), the craniofacial research data hub, and read its open-access data from Python.
|
|
38
|
+
|
|
39
|
+
FaceBase holds about 1,070 open datasets: micro-CT and light-sheet imaging, ChIP-seq and RNA-seq tracks, expression tables and more, from mouse, zebrafish, human and other models. This package is a typed layer over FaceBase's open metadata API and file store. No key or login is needed.
|
|
40
|
+
|
|
41
|
+
```
|
|
42
|
+
pip install scigantic-facebase
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Use
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import scigantic_facebase as fb
|
|
49
|
+
|
|
50
|
+
hits = fb.search("zebrafish thyroid microCT") # newest first
|
|
51
|
+
ds = hits[0]
|
|
52
|
+
print(ds.title, ds.doi_url)
|
|
53
|
+
print(fb.cite(ds))
|
|
54
|
+
|
|
55
|
+
files = fb.files(ds) # size, MD5, download URL
|
|
56
|
+
fb.download(files[0], "data/") # MD5-checked
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Search matches every word against the whole record, and treats "zebrafish" and "Danio rerio", "mouse" and "Mus musculus" as the same thing.
|
|
60
|
+
|
|
61
|
+
## Read a micro-CT scan without downloading it
|
|
62
|
+
|
|
63
|
+
Many micro-CT datasets are a zip of TIFF slices. A zip keeps its directory at the end, so one slice costs one HTTP range request.
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
stack = fb.microct.open_stack(files[0])
|
|
67
|
+
len(stack) # 596 slices
|
|
68
|
+
stack.voxel_size_um # 10.5, from the scan log
|
|
69
|
+
img = stack.read_slice(300)
|
|
70
|
+
vol = stack.volume(step=10) # every 10th slice, (z, y, x)
|
|
71
|
+
stack.bytes_fetched # a few MB of a 49 MB file
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
`fb.microct.thyroid_atlas()` lists the zebrafish thyroid-hormone atlas: 81 head scans from 12 to 25 mm standard length, raised hypothyroid, euthyroid or hyperthyroid.
|
|
75
|
+
|
|
76
|
+
`fb.open_remote(file)` gives any FaceBase file as a seekable file object backed by range requests, which `zipfile`, `tifffile` and `h5py` accept.
|
|
77
|
+
|
|
78
|
+
## What is not here
|
|
79
|
+
|
|
80
|
+
Protected human-subjects datasets sit behind FaceBase's data access request. Their metadata is public and `fb.search(include_protected=True)` lists them, but reading their files raises `FacebaseAccessError`.
|
|
81
|
+
|
|
82
|
+
## Terms of use
|
|
83
|
+
|
|
84
|
+
Copyright to FaceBase data belongs to the contributing investigators. FaceBase's [Terms of Use](https://www.facebase.org/policies/tou/) ask users to cite the dataset, acknowledge FaceBase, and send other people to FaceBase instead of circulating downloaded files. `fb.cite()` produces the citation. This package fetches from FaceBase on request and does not mirror anything.
|
|
85
|
+
|
|
86
|
+
## License
|
|
87
|
+
|
|
88
|
+
MIT-0 for this code. The data carries its own terms.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
src/scigantic_facebase/__init__.py
|
|
5
|
+
src/scigantic_facebase/_client.py
|
|
6
|
+
src/scigantic_facebase/_version.py
|
|
7
|
+
src/scigantic_facebase/catalog.py
|
|
8
|
+
src/scigantic_facebase/microct.py
|
|
9
|
+
src/scigantic_facebase/models.py
|
|
10
|
+
src/scigantic_facebase/py.typed
|
|
11
|
+
src/scigantic_facebase/remote.py
|
|
12
|
+
src/scigantic_facebase.egg-info/PKG-INFO
|
|
13
|
+
src/scigantic_facebase.egg-info/SOURCES.txt
|
|
14
|
+
src/scigantic_facebase.egg-info/dependency_links.txt
|
|
15
|
+
src/scigantic_facebase.egg-info/requires.txt
|
|
16
|
+
src/scigantic_facebase.egg-info/top_level.txt
|
|
17
|
+
tests/test_live.py
|
|
18
|
+
tests/test_remote.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
scigantic_facebase
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Live checks against facebase.org. Skipped with `-m 'not live'`."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
import scigantic_facebase as fb
|
|
8
|
+
|
|
9
|
+
pytestmark = pytest.mark.live
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_search_and_get() -> None:
|
|
13
|
+
hits = fb.search("zebrafish thyroid microCT", limit=5)
|
|
14
|
+
assert hits and all(h.open_access for h in hits)
|
|
15
|
+
assert fb.count("zebrafish thyroid microCT") >= 80
|
|
16
|
+
assert fb.get("2E-YW20").doi == "10.25550/2E-YW20"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_citation_has_doi_and_year() -> None:
|
|
20
|
+
c = fb.cite("2E-YW20")
|
|
21
|
+
assert "https://doi.org/10.25550/2E-YW20" in c and "(2024)" in c and "FaceBase Consortium" in c
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def test_stack_read_in_place() -> None:
|
|
25
|
+
f = fb.files("2E-YW20")[0]
|
|
26
|
+
with fb.microct.open_stack(f) as stack:
|
|
27
|
+
assert len(stack) > 500
|
|
28
|
+
assert stack.voxel_size_um and 10 < stack.voxel_size_um < 11
|
|
29
|
+
img = stack.read_slice(len(stack) // 2)
|
|
30
|
+
assert img.ndim == 2
|
|
31
|
+
assert stack.bytes_fetched < f.size // 4
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_atlas_is_three_conditions() -> None:
|
|
35
|
+
atlas = fb.microct.thyroid_atlas()
|
|
36
|
+
assert {s.condition for s in atlas} == {"hypothyroid", "euthyroid", "hyperthyroid"}
|
|
37
|
+
assert len(atlas) >= 80
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_protected_dataset_files_raise() -> None:
|
|
41
|
+
ds = next(d for d in fb.search(include_protected=True, limit=400) if d.protected)
|
|
42
|
+
files = fb.files(ds)
|
|
43
|
+
assert files, "protected dataset lists its files (metadata is public)"
|
|
44
|
+
with pytest.raises(fb.FacebaseAccessError):
|
|
45
|
+
fb.open_remote(files[0]).read(1)
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""RangeFile and download against a local Range-capable server (offline)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import http.server
|
|
7
|
+
import io
|
|
8
|
+
import threading
|
|
9
|
+
import zipfile
|
|
10
|
+
from collections.abc import Iterator
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import pytest
|
|
14
|
+
|
|
15
|
+
from scigantic_facebase import File, RangeFile, download
|
|
16
|
+
from scigantic_facebase.microct import Stack
|
|
17
|
+
|
|
18
|
+
PAYLOAD = bytes(range(256)) * 20000 # ~5 MB
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class _Handler(http.server.BaseHTTPRequestHandler):
|
|
22
|
+
body = PAYLOAD
|
|
23
|
+
hits: list[str] = []
|
|
24
|
+
|
|
25
|
+
def log_message(self, *a: object) -> None: # silence
|
|
26
|
+
pass
|
|
27
|
+
|
|
28
|
+
def do_GET(self) -> None:
|
|
29
|
+
rng = self.headers.get("Range")
|
|
30
|
+
data = self.body
|
|
31
|
+
if rng:
|
|
32
|
+
a, _, b = rng.removeprefix("bytes=").partition("-")
|
|
33
|
+
lo, hi = int(a), min(int(b), len(data) - 1)
|
|
34
|
+
self.hits.append(rng)
|
|
35
|
+
self.send_response(206)
|
|
36
|
+
self.send_header("Content-Range", f"bytes {lo}-{hi}/{len(data)}")
|
|
37
|
+
chunk = data[lo : hi + 1]
|
|
38
|
+
else:
|
|
39
|
+
self.send_response(200)
|
|
40
|
+
chunk = data
|
|
41
|
+
self.send_header("Content-Length", str(len(chunk)))
|
|
42
|
+
self.end_headers()
|
|
43
|
+
self.wfile.write(chunk)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@pytest.fixture()
|
|
47
|
+
def server() -> Iterator[str]:
|
|
48
|
+
_Handler.hits = []
|
|
49
|
+
srv = http.server.ThreadingHTTPServer(("127.0.0.1", 0), _Handler)
|
|
50
|
+
threading.Thread(target=srv.serve_forever, daemon=True).start()
|
|
51
|
+
yield f"http://127.0.0.1:{srv.server_port}/f.bin"
|
|
52
|
+
srv.shutdown()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def test_random_access_reads_only_needed_blocks(server: str) -> None:
|
|
56
|
+
f = RangeFile(server)
|
|
57
|
+
assert len(f) == len(PAYLOAD)
|
|
58
|
+
f.seek(3_000_000)
|
|
59
|
+
assert f.read(10) == PAYLOAD[3_000_000:3_000_010]
|
|
60
|
+
assert f.requests == 1 # the size probe is not counted; one block read here
|
|
61
|
+
assert f.bytes_fetched <= 1 << 20
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def test_read_past_end_and_seek_end(server: str) -> None:
|
|
65
|
+
f = RangeFile(server)
|
|
66
|
+
f.seek(-5, 2)
|
|
67
|
+
assert f.read() == PAYLOAD[-5:]
|
|
68
|
+
assert f.read(10) == b""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_download_verifies_md5(server: str, tmp_path: Path) -> None:
|
|
72
|
+
good = File("r", "d", "f.bin", len(PAYLOAD), hashlib.md5(PAYLOAD).hexdigest(), server)
|
|
73
|
+
out = download(good, tmp_path)
|
|
74
|
+
assert out.read_bytes() == PAYLOAD
|
|
75
|
+
bad = File("r", "d", "g.bin", len(PAYLOAD), "0" * 32, server)
|
|
76
|
+
with pytest.raises(Exception, match="MD5"):
|
|
77
|
+
download(bad, tmp_path)
|
|
78
|
+
assert not (tmp_path / "g.bin").exists()
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def test_stack_reads_one_slice_from_a_remote_zip(tmp_path: Path) -> None:
|
|
82
|
+
import numpy as np
|
|
83
|
+
import tifffile
|
|
84
|
+
|
|
85
|
+
buf = io.BytesIO()
|
|
86
|
+
with zipfile.ZipFile(buf, "w", zipfile.ZIP_STORED) as z:
|
|
87
|
+
for i in range(5):
|
|
88
|
+
t = io.BytesIO()
|
|
89
|
+
tifffile.imwrite(t, np.full((8, 9), i, dtype=np.uint16))
|
|
90
|
+
z.writestr(f"scan/s{i:04d}.tif", t.getvalue())
|
|
91
|
+
z.writestr("scan/scan.log", "[Acquisition]\nImage Pixel Size (um)=12.5\n")
|
|
92
|
+
_Handler.body = buf.getvalue()
|
|
93
|
+
srv = http.server.ThreadingHTTPServer(("127.0.0.1", 0), _Handler)
|
|
94
|
+
threading.Thread(target=srv.serve_forever, daemon=True).start()
|
|
95
|
+
try:
|
|
96
|
+
stack = Stack(RangeFile(f"http://127.0.0.1:{srv.server_port}/z.zip"))
|
|
97
|
+
assert len(stack) == 5
|
|
98
|
+
assert int(stack.read_slice(3).mean()) == 3
|
|
99
|
+
assert stack.volume(step=2).shape == (3, 8, 9)
|
|
100
|
+
assert stack.voxel_size_um == 12.5
|
|
101
|
+
finally:
|
|
102
|
+
srv.shutdown()
|
|
103
|
+
_Handler.body = PAYLOAD
|