scigantic-facebase 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,42 @@
1
+ """Search FaceBase and read its open-access craniofacial data from Python.
2
+
3
+ import scigantic_facebase as fb
4
+
5
+ hits = fb.search("zebrafish thyroid microCT")
6
+ fb.cite(hits[0])
7
+ stack = fb.microct.open_stack(fb.files(hits[0])[0])
8
+ stack.read_slice(300)
9
+
10
+ Metadata comes from FaceBase's open ERMrest API and files from its open-access
11
+ file store, both anonymous. Copyright for the data stays with the contributing
12
+ investigators and FaceBase's Terms of Use apply: cite the dataset DOI
13
+ (`cite()`), and send other people to FaceBase rather than redistributing files.
14
+ """
15
+
16
+ from . import microct
17
+ from ._client import FacebaseAccessError, FacebaseError, FacebaseNotFoundError
18
+ from ._version import __version__
19
+ from .catalog import cite, contributors, count, files, get, project, search
20
+ from .models import Dataset, File, Project
21
+ from .remote import RangeFile, download, open_remote
22
+
23
+ __all__ = [
24
+ "Dataset",
25
+ "FacebaseAccessError",
26
+ "FacebaseError",
27
+ "FacebaseNotFoundError",
28
+ "File",
29
+ "Project",
30
+ "RangeFile",
31
+ "__version__",
32
+ "cite",
33
+ "contributors",
34
+ "count",
35
+ "download",
36
+ "files",
37
+ "get",
38
+ "microct",
39
+ "open_remote",
40
+ "project",
41
+ "search",
42
+ ]
@@ -0,0 +1,107 @@
1
+ """Shared HTTP plumbing for FaceBase: one lazily-built requests.Session,
2
+ retry with backoff on transient failures, and the two FaceBase endpoints this
3
+ package talks to.
4
+
5
+ FaceBase publishes no rate limit. Its metadata (ERMrest) and open-access file
6
+ store (Hatrac) answer anonymous requests, so there is no key or login here.
7
+ Retries cover 429/5xx and connection errors only. A 404 is a real answer and
8
+ raises FacebaseNotFoundError at once. A 401 or 403 means the data is behind
9
+ FaceBase's data access request process and raises FacebaseAccessError.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import threading
15
+ import time
16
+ from typing import Any
17
+
18
+ import requests
19
+
20
+ from ._version import __version__
21
+
22
+ SITE = "https://www.facebase.org"
23
+ ERMREST = f"{SITE}/ermrest/catalog/1"
24
+
25
+ _USER_AGENT = f"scigantic-facebase/{__version__} (+https://scigantic.com; mailto:support@scigantic.com)"
26
+
27
+ _MAX_RETRIES = 4
28
+ _RETRY_STATUS_CODES = {429, 500, 502, 503, 504}
29
+
30
+ _session: requests.Session | None = None
31
+ _session_lock = threading.Lock()
32
+
33
+
34
+ class FacebaseError(Exception):
35
+ """An HTTP error after retries were exhausted, or a malformed response."""
36
+
37
+
38
+ class FacebaseNotFoundError(FacebaseError):
39
+ """A dataset, file or path that does not exist (HTTP 404)."""
40
+
41
+
42
+ class FacebaseAccessError(FacebaseError):
43
+ """The data needs authentication (HTTP 401 or 403). FaceBase holds
44
+ protected human-subjects data behind a data access request, and this
45
+ package only reads the open-access part."""
46
+
47
+
48
+ def get_session() -> requests.Session:
49
+ global _session
50
+ if _session is None:
51
+ with _session_lock:
52
+ if _session is None:
53
+ _session = requests.Session()
54
+ _session.headers["User-Agent"] = _USER_AGENT
55
+ return _session
56
+
57
+
58
+ def send(
59
+ method: str,
60
+ url: str,
61
+ params: dict[str, Any] | None = None,
62
+ headers: dict[str, str] | None = None,
63
+ stream: bool = False,
64
+ timeout: float = 60.0,
65
+ ) -> requests.Response:
66
+ """Issue one request with retry on 429/5xx and connection errors."""
67
+ session = get_session()
68
+ last_error: Exception | None = None
69
+ for attempt in range(_MAX_RETRIES + 1):
70
+ try:
71
+ resp = session.request(
72
+ method, url, params=params, headers=headers, stream=stream, timeout=timeout
73
+ )
74
+ except requests.RequestException as exc:
75
+ last_error = exc
76
+ if attempt == _MAX_RETRIES:
77
+ break
78
+ time.sleep(1.5 * (2**attempt))
79
+ continue
80
+ if resp.status_code == 404:
81
+ resp.close()
82
+ raise FacebaseNotFoundError(f"404 for {resp.url}")
83
+ if resp.status_code in (401, 403):
84
+ resp.close()
85
+ raise FacebaseAccessError(
86
+ f"HTTP {resp.status_code} for {resp.url}: this data is not open access. "
87
+ "FaceBase releases protected human-subjects data only through a data access request."
88
+ )
89
+ if resp.status_code in _RETRY_STATUS_CODES and attempt < _MAX_RETRIES:
90
+ resp.close()
91
+ time.sleep(1.5 * (2**attempt))
92
+ continue
93
+ if resp.status_code >= 400:
94
+ body = resp.text[:300]
95
+ resp.close()
96
+ raise FacebaseError(f"HTTP {resp.status_code} for {resp.url}: {body}")
97
+ return resp
98
+ raise FacebaseError(f"request to {url} failed after {_MAX_RETRIES + 1} attempts: {last_error}")
99
+
100
+
101
+ def ermrest(path: str) -> Any:
102
+ """GET an ERMrest path (everything after /catalog/1) and return parsed JSON."""
103
+ resp = send("GET", f"{ERMREST}/{path.lstrip('/')}")
104
+ try:
105
+ return resp.json()
106
+ except ValueError as exc:
107
+ raise FacebaseError(f"non-JSON response from {resp.url}") from exc
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"
@@ -0,0 +1,149 @@
1
+ """Search FaceBase datasets and list their files through the open ERMrest
2
+ metadata API. Everything here is a metadata call; no file bytes move."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import re
7
+ from typing import Any
8
+ from functools import lru_cache
9
+ from urllib.parse import quote
10
+
11
+ from ._client import SITE, FacebaseNotFoundError, ermrest
12
+ from .models import Dataset, File, Project
13
+
14
+ _OPEN = "released=true&protected_human_subjects=false"
15
+
16
+
17
+ def _q(value: str) -> str:
18
+ return quote(value, safe="")
19
+
20
+
21
+ # FaceBase records name species inconsistently ("zebrafish" in some titles,
22
+ # "Danio rerio" in most). A search word that is a common name also matches
23
+ # the Latin name, and the other way round.
24
+ _SYNONYMS = {
25
+ "zebrafish": ("zebrafish", "danio rerio"),
26
+ "danio": ("zebrafish", "danio rerio"),
27
+ "mouse": ("mouse", "mice", "mus musculus"),
28
+ "mice": ("mouse", "mice", "mus musculus"),
29
+ "human": ("human", "homo sapiens"),
30
+ "chick": ("chick", "gallus gallus"),
31
+ "microct": ("microct", "micro-ct", "micro ct", "\u00b5ct"),
32
+ }
33
+
34
+
35
+ def _filter(text: str | None, include_protected: bool) -> str:
36
+ parts = [_OPEN if not include_protected else "released=true"]
37
+ for word in (text or "").split():
38
+ options = _SYNONYMS.get(word.lower(), (word,))
39
+ pattern = "|".join(re.escape(o) for o in options)
40
+ parts.append(f"*::ciregexp::{_q(pattern)}")
41
+ return "&".join(parts)
42
+
43
+
44
+ def _str(row: dict[str, Any], key: str) -> str | None:
45
+ val = row.get(key)
46
+ return val if isinstance(val, str) else None
47
+
48
+
49
+ def _dataset(row: dict[str, Any]) -> Dataset:
50
+ project_id = row.get("project")
51
+ return Dataset(
52
+ rid=str(row["RID"]),
53
+ accession=_str(row, "accession"),
54
+ title=str(row.get("title") or ""),
55
+ description=str(row.get("description") or ""),
56
+ release_date=_str(row, "release_date"),
57
+ doi=_str(row, "DOI"),
58
+ project_id=project_id if isinstance(project_id, int) else None,
59
+ protected=bool(row.get("protected_human_subjects")),
60
+ internal_id=int(row["id"]),
61
+ )
62
+
63
+
64
+ def search(text: str | None = None, *, limit: int = 50, include_protected: bool = False) -> list[Dataset]:
65
+ """Find datasets, newest first. Every word in `text` must appear somewhere
66
+ in the dataset's title, description or keywords (case-insensitive).
67
+ Protected human-subjects datasets are left out unless include_protected=True;
68
+ their metadata is public but their files need a data access request."""
69
+ path = f"entity/isa:dataset/{_filter(text, include_protected)}@sort(release_date::desc::,RID)"
70
+ rows = ermrest(f"{path}?limit={int(limit)}")
71
+ return [_dataset(r) for r in rows]
72
+
73
+
74
+ def count(text: str | None = None, *, include_protected: bool = False) -> int:
75
+ """How many datasets match `text` (same rules as search)."""
76
+ rows = ermrest(f"aggregate/isa:dataset/{_filter(text, include_protected)}/n:=cnt(RID)")
77
+ return int(rows[0]["n"])
78
+
79
+
80
+ def get(rid: str) -> Dataset:
81
+ """One dataset by FaceBase record id (e.g. '2E-YW20')."""
82
+ rows = ermrest(f"entity/isa:dataset/RID={_q(rid)}")
83
+ if not rows:
84
+ raise FacebaseNotFoundError(f"no FaceBase dataset {rid!r}")
85
+ return _dataset(rows[0])
86
+
87
+
88
+ @lru_cache(maxsize=1)
89
+ def _formats() -> dict[str, str]:
90
+ out: dict[str, str] = {}
91
+ for r in ermrest("entity/vocab:file_format?limit=500"):
92
+ key = r.get("id") or r.get("ID")
93
+ name = r.get("name") or r.get("Name")
94
+ if key and name:
95
+ out[str(key)] = str(name)
96
+ return out
97
+
98
+
99
+ def files(dataset: Dataset | str) -> list[File]:
100
+ """Every file in a dataset, with size, MD5 and an absolute download URL."""
101
+ ds = get(dataset) if isinstance(dataset, str) else dataset
102
+ rows = ermrest(f"entity/isa:file/dataset={_q(ds.rid)}@sort(filename)?limit=20000")
103
+ fmt = _formats()
104
+ return [
105
+ File(
106
+ rid=str(r["RID"]),
107
+ dataset_rid=ds.rid,
108
+ filename=str(r["filename"]),
109
+ size=int(r.get("byte_count") or 0),
110
+ md5=_str(r, "md5"),
111
+ url=SITE + str(r["url"]),
112
+ format=fmt.get(str(r["file_format"])) if r.get("file_format") else None,
113
+ relative_path=_str(r, "relative_path"),
114
+ description=_str(r, "description"),
115
+ protected=ds.protected,
116
+ )
117
+ for r in rows
118
+ ]
119
+
120
+
121
+ def project(dataset: Dataset) -> Project | None:
122
+ """The project (grant and investigator group) a dataset belongs to."""
123
+ if dataset.project_id is None:
124
+ return None
125
+ rows = ermrest(f"entity/isa:project/id={dataset.project_id}")
126
+ if not rows:
127
+ return None
128
+ r = rows[0]
129
+ return Project(
130
+ id=int(r["id"]),
131
+ name=str(r.get("name") or ""),
132
+ doi=_str(r, "DOI"),
133
+ funding=_str(r, "funding"),
134
+ )
135
+
136
+
137
+ def contributors(dataset: Dataset) -> list[str]:
138
+ rows = ermrest(f"attribute/isa:dataset_contributor/dataset_id={dataset.internal_id}/order,full_name@sort(order)")
139
+ return [str(r["full_name"]) for r in rows if r.get("full_name")]
140
+
141
+
142
+ def cite(dataset: Dataset | str) -> str:
143
+ """The citation FaceBase's Terms of Use ask for: authors, title,
144
+ FaceBase Consortium, DOI link and release year."""
145
+ ds = get(dataset) if isinstance(dataset, str) else dataset
146
+ authors = ", ".join(contributors(ds)) or "FaceBase Consortium contributors"
147
+ year = (ds.release_date or "")[:4]
148
+ link = ds.doi_url or ds.url
149
+ return f"{authors}. {ds.title}. FaceBase Consortium {link}" + (f" ({year})." if year else ".")
@@ -0,0 +1,147 @@
1
+ """Zip-of-TIFF micro-CT stacks, read in place.
2
+
3
+ Many FaceBase micro-CT datasets ship each scan as one .zip holding a stack of
4
+ reconstructed TIFF slices plus a SkyScan scan log. A zip keeps a directory at
5
+ its end, so a single slice can be read with one HTTP range request; a 50 MB
6
+ scan is never downloaded just to look at one cross-section.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import io
12
+ import re
13
+ import threading
14
+ import zipfile
15
+ from concurrent.futures import ThreadPoolExecutor
16
+ from dataclasses import dataclass
17
+ from typing import Any
18
+
19
+ import numpy as np
20
+ import numpy.typing as npt
21
+
22
+ from . import catalog
23
+ from .models import Dataset, File
24
+ from .remote import RangeFile, open_remote
25
+
26
+ _TIF = re.compile(r"\.tiff?$", re.IGNORECASE)
27
+
28
+
29
+ class Stack:
30
+ """A remote zip of TIFF slices. Create with open_stack()."""
31
+
32
+ def __init__(self, source: RangeFile, name: str = "") -> None:
33
+ self.name = name
34
+ self._source = source
35
+ self._zip = zipfile.ZipFile(source)
36
+ self._lock = threading.Lock()
37
+ self.slice_names: list[str] = sorted(
38
+ i.filename for i in self._zip.infolist() if _TIF.search(i.filename) and not i.is_dir()
39
+ )
40
+ self._logs = [i.filename for i in self._zip.infolist() if i.filename.lower().endswith(".log")]
41
+
42
+ def __len__(self) -> int:
43
+ return len(self.slice_names)
44
+
45
+ @property
46
+ def requests(self) -> int:
47
+ """HTTP requests made so far, to see how little was fetched."""
48
+ return self._source.requests
49
+
50
+ @property
51
+ def bytes_fetched(self) -> int:
52
+ return self._source.bytes_fetched
53
+
54
+ def read_slice(self, index: int) -> npt.NDArray[Any]:
55
+ """One slice as a 2-D array (decoded with tifffile)."""
56
+ import tifffile
57
+
58
+ with self._lock:
59
+ raw = self._zip.read(self.slice_names[index])
60
+ return np.asarray(tifffile.imread(io.BytesIO(raw)))
61
+
62
+ def volume(self, *, step: int = 1, start: int = 0, stop: int | None = None, workers: int = 4) -> npt.NDArray[Any]:
63
+ """Slices start:stop:step stacked into a (z, y, x) array. Slices are
64
+ read in parallel; each worker reuses the same cached connection."""
65
+ idx = list(range(start, len(self) if stop is None else stop, step))
66
+ if not idx:
67
+ raise ValueError("no slices selected")
68
+ with ThreadPoolExecutor(max_workers=max(1, workers)) as pool:
69
+ slices = list(pool.map(self.read_slice, idx))
70
+ return np.stack(slices)
71
+
72
+ def scan_log(self) -> dict[str, dict[str, str]]:
73
+ """The SkyScan scan log as {section: {key: value}}; empty when the
74
+ zip has none."""
75
+ if not self._logs:
76
+ return {}
77
+ with self._lock:
78
+ text = self._zip.read(self._logs[0]).decode("latin-1")
79
+ out: dict[str, dict[str, str]] = {}
80
+ section = ""
81
+ for line in text.splitlines():
82
+ line = line.strip()
83
+ if line.startswith("[") and line.endswith("]"):
84
+ section = line[1:-1]
85
+ out.setdefault(section, {})
86
+ elif "=" in line and section:
87
+ key, _, val = line.partition("=")
88
+ out[section][key.strip()] = val.strip()
89
+ return out
90
+
91
+ @property
92
+ def voxel_size_um(self) -> float | None:
93
+ """Isotropic voxel edge in micrometres, from the scan log."""
94
+ log = self.scan_log()
95
+ for section in ("Reconstruction", "Acquisition"):
96
+ for key in ("Pixel Size (um)", "Image Pixel Size (um)", "Scaled Image Pixel Size (um)"):
97
+ val = log.get(section, {}).get(key)
98
+ if val:
99
+ try:
100
+ return float(val)
101
+ except ValueError:
102
+ continue
103
+ return None
104
+
105
+ def close(self) -> None:
106
+ self._zip.close()
107
+
108
+ def __enter__(self) -> Stack:
109
+ return self
110
+
111
+ def __exit__(self, *exc: object) -> None:
112
+ self.close()
113
+
114
+
115
+ def open_stack(file: File) -> Stack:
116
+ """Open a zip-of-TIFFs file from FaceBase without downloading it."""
117
+ return Stack(open_remote(file), name=file.filename)
118
+
119
+
120
+ @dataclass(frozen=True)
121
+ class ThyroidScan:
122
+ """One scan in the zebrafish thyroid-hormone craniofacial atlas."""
123
+
124
+ dataset: Dataset
125
+ condition: str # hypothyroid, euthyroid or hyperthyroid
126
+ standard_length_mm: float
127
+
128
+
129
+ _TITLE = re.compile(r"microCT scan of ([\d.]+) mm (hypothyroid|euthyroid|hyperthyroid) Danio rerio head", re.IGNORECASE)
130
+
131
+
132
+ def parse_thyroid_title(title: str) -> tuple[float, str] | None:
133
+ """(standard length in mm, condition) from an atlas dataset title."""
134
+ m = _TITLE.search(title)
135
+ return (float(m.group(1)), m.group(2).lower()) if m else None
136
+
137
+
138
+ def thyroid_atlas() -> list[ThyroidScan]:
139
+ """The zebrafish head micro-CT series: fish from 12 to 25 mm standard
140
+ length raised hypothyroid, euthyroid (normal) or hyperthyroid, one dataset
141
+ per fish. Sorted by condition then length."""
142
+ scans = []
143
+ for ds in catalog.search("microCT Danio rerio head thyroid", limit=300):
144
+ parsed = parse_thyroid_title(ds.title)
145
+ if parsed:
146
+ scans.append(ThyroidScan(ds, parsed[1], parsed[0]))
147
+ return sorted(scans, key=lambda s: (s.condition, s.standard_length_mm, s.dataset.rid))
@@ -0,0 +1,66 @@
1
+ """Typed records for FaceBase datasets and files."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, field
6
+
7
+ from ._client import SITE
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class Dataset:
12
+ """One FaceBase dataset. `rid` is FaceBase's record id, the same string
13
+ that appears in the dataset's landing page and DOI."""
14
+
15
+ rid: str
16
+ accession: str | None
17
+ title: str
18
+ description: str
19
+ release_date: str | None
20
+ doi: str | None
21
+ project_id: int | None
22
+ protected: bool
23
+ internal_id: int
24
+
25
+ @property
26
+ def url(self) -> str:
27
+ return f"{SITE}/id/{self.rid}"
28
+
29
+ @property
30
+ def doi_url(self) -> str | None:
31
+ return f"https://doi.org/{self.doi}" if self.doi else None
32
+
33
+ @property
34
+ def open_access(self) -> bool:
35
+ """False for datasets behind FaceBase's data access request."""
36
+ return not self.protected
37
+
38
+
39
+ @dataclass(frozen=True)
40
+ class Project:
41
+ id: int
42
+ name: str
43
+ doi: str | None
44
+ funding: str | None
45
+
46
+
47
+ @dataclass(frozen=True)
48
+ class File:
49
+ """One file in a FaceBase dataset. `url` is absolute and anonymous for
50
+ open-access datasets; byte-range requests are honoured."""
51
+
52
+ rid: str
53
+ dataset_rid: str
54
+ filename: str
55
+ size: int
56
+ md5: str | None
57
+ url: str
58
+ format: str | None = None
59
+ relative_path: str | None = None
60
+ description: str | None = None
61
+ protected: bool = field(default=False, compare=False)
62
+
63
+ @property
64
+ def path(self) -> str:
65
+ """Path within the dataset, folders included."""
66
+ return f"{self.relative_path or ''}{self.filename}"
File without changes
@@ -0,0 +1,171 @@
1
+ """Read FaceBase files over HTTP without downloading them first.
2
+
3
+ `RangeFile` is a seekable, read-only file object backed by HTTP Range
4
+ requests with a small block cache. Hand it to zipfile.ZipFile or tifffile and
5
+ only the bytes they actually touch cross the network. For a 50 MB zip of
6
+ 600 TIFF slices, opening the archive reads the central directory (one small
7
+ request) and each slice costs one request of its own size.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import hashlib
13
+ import io
14
+ import os
15
+ import threading
16
+ from collections import OrderedDict
17
+ from pathlib import Path
18
+ from typing import Any
19
+
20
+ from ._client import FacebaseError, send
21
+ from .models import File
22
+
23
+ _BLOCK = 1 << 20 # 1 MiB
24
+ _CACHE_BLOCKS = 32
25
+
26
+
27
+ class RangeFile(io.RawIOBase):
28
+ """Seekable read-only view of one remote file."""
29
+
30
+ def __init__(self, url: str, size: int | None = None) -> None:
31
+ super().__init__()
32
+ self.url = url
33
+ self._pos = 0
34
+ self._lock = threading.Lock()
35
+ self._cache: OrderedDict[int, bytes] = OrderedDict()
36
+ self.requests = 0
37
+ self.bytes_fetched = 0
38
+ self._size = size if size is not None else self._probe_size()
39
+
40
+ def _probe_size(self) -> int:
41
+ resp = send("GET", self.url, headers={"Range": "bytes=0-0"}, stream=True)
42
+ try:
43
+ cr = resp.headers.get("Content-Range", "")
44
+ if "/" in cr and cr.rsplit("/", 1)[1].isdigit():
45
+ return int(cr.rsplit("/", 1)[1])
46
+ length = resp.headers.get("Content-Length")
47
+ if resp.status_code == 200 and length and length.isdigit():
48
+ return int(length)
49
+ finally:
50
+ resp.close()
51
+ raise FacebaseError(f"cannot determine size of {self.url}")
52
+
53
+ def __len__(self) -> int:
54
+ return self._size
55
+
56
+ def readable(self) -> bool:
57
+ return True
58
+
59
+ def seekable(self) -> bool:
60
+ return True
61
+
62
+ def tell(self) -> int:
63
+ return self._pos
64
+
65
+ def seek(self, offset: int, whence: int = os.SEEK_SET) -> int:
66
+ if whence == os.SEEK_SET:
67
+ self._pos = offset
68
+ elif whence == os.SEEK_CUR:
69
+ self._pos += offset
70
+ elif whence == os.SEEK_END:
71
+ self._pos = self._size + offset
72
+ else:
73
+ raise ValueError(f"invalid whence: {whence}")
74
+ self._pos = max(self._pos, 0)
75
+ return self._pos
76
+
77
+ def _block(self, index: int) -> bytes:
78
+ with self._lock:
79
+ hit = self._cache.get(index)
80
+ if hit is not None:
81
+ self._cache.move_to_end(index)
82
+ return hit
83
+ start = index * _BLOCK
84
+ end = min(start + _BLOCK, self._size) - 1
85
+ resp = send("GET", self.url, headers={"Range": f"bytes={start}-{end}"})
86
+ data = resp.content
87
+ if resp.status_code == 200 and len(data) != end - start + 1:
88
+ raise FacebaseError(f"{self.url} ignored the Range header")
89
+ with self._lock:
90
+ self.requests += 1
91
+ self.bytes_fetched += len(data)
92
+ self._cache[index] = data
93
+ while len(self._cache) > _CACHE_BLOCKS:
94
+ self._cache.popitem(last=False)
95
+ return data
96
+
97
+ def read(self, size: int = -1) -> bytes:
98
+ if self._pos >= self._size:
99
+ return b""
100
+ if size is None or size < 0:
101
+ size = self._size - self._pos
102
+ stop = min(self._pos + size, self._size)
103
+ # A request larger than the cache would thrash it; fetch it in one go.
104
+ if stop - self._pos > _BLOCK * 4:
105
+ resp = send("GET", self.url, headers={"Range": f"bytes={self._pos}-{stop - 1}"})
106
+ data = resp.content
107
+ with self._lock:
108
+ self.requests += 1
109
+ self.bytes_fetched += len(data)
110
+ self._pos += len(data)
111
+ return data
112
+ out = bytearray()
113
+ pos = self._pos
114
+ while pos < stop:
115
+ block = self._block(pos // _BLOCK)
116
+ off = pos % _BLOCK
117
+ take = block[off : off + (stop - pos)]
118
+ if not take:
119
+ break
120
+ out += take
121
+ pos += len(take)
122
+ self._pos = pos
123
+ return bytes(out)
124
+
125
+ def readinto(self, b: Any) -> int:
126
+ data = self.read(len(b))
127
+ b[: len(data)] = data
128
+ return len(data)
129
+
130
+
131
+ def open_remote(file: File) -> RangeFile:
132
+ """Open a FaceBase file as a seekable file object (nothing downloaded)."""
133
+ return RangeFile(file.url, size=file.size or None)
134
+
135
+
136
+ def download(file: File, dest: str | os.PathLike[str], *, verify: bool = True, chunk: int = 1 << 20) -> Path:
137
+ """Download a file to `dest` (a directory or a file path) and return the
138
+ path. Checks size and MD5 against FaceBase's record unless verify=False.
139
+ An existing file with the right size and MD5 is kept, so reruns are free."""
140
+ target = Path(dest)
141
+ if target.is_dir():
142
+ target = target / file.filename
143
+ target.parent.mkdir(parents=True, exist_ok=True)
144
+ if target.exists() and target.stat().st_size == file.size and (not verify or not file.md5 or _md5(target) == file.md5):
145
+ return target
146
+ tmp = target.with_suffix(target.suffix + ".part")
147
+ md5 = hashlib.md5()
148
+ resp = send("GET", file.url, stream=True, timeout=120.0)
149
+ try:
150
+ with open(tmp, "wb") as fh:
151
+ for part in resp.iter_content(chunk):
152
+ fh.write(part)
153
+ md5.update(part)
154
+ finally:
155
+ resp.close()
156
+ if file.size and tmp.stat().st_size != file.size:
157
+ tmp.unlink(missing_ok=True)
158
+ raise FacebaseError(f"{file.filename}: got {tmp.stat().st_size if tmp.exists() else 0} bytes, expected {file.size}")
159
+ if verify and file.md5 and md5.hexdigest() != file.md5:
160
+ tmp.unlink(missing_ok=True)
161
+ raise FacebaseError(f"{file.filename}: MD5 mismatch")
162
+ tmp.replace(target)
163
+ return target
164
+
165
+
166
+ def _md5(path: Path) -> str:
167
+ h = hashlib.md5()
168
+ with open(path, "rb") as fh:
169
+ for part in iter(lambda: fh.read(1 << 20), b""):
170
+ h.update(part)
171
+ return h.hexdigest()
@@ -0,0 +1,88 @@
1
+ Metadata-Version: 2.4
2
+ Name: scigantic-facebase
3
+ Version: 0.1.0
4
+ Summary: Search FaceBase, the craniofacial research data hub, and read its open-access micro-CT, imaging and sequencing data from Python: a typed catalog over FaceBase's open metadata API, citations in the form its Terms of Use ask for, ranged reads of remote files, and single-slice reads from zipped micro-CT stacks without downloading them.
5
+ Author: Scigantic
6
+ License: MIT-0
7
+ Project-URL: Homepage, https://scigantic.com
8
+ Project-URL: Repository, https://github.com/Scigantic/scigantic-facebase
9
+ Project-URL: Issues, https://github.com/Scigantic/scigantic-facebase/issues
10
+ Project-URL: FaceBase, https://www.facebase.org
11
+ Keywords: facebase,craniofacial,micro-ct,zebrafish,deriva,ermrest,bioinformatics
12
+ Classifier: License :: OSI Approved :: MIT No Attribution License (MIT-0)
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.10
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
19
+ Classifier: Typing :: Typed
20
+ Classifier: Topic :: Scientific/Engineering :: Bio-Informatics
21
+ Classifier: Topic :: Scientific/Engineering :: Image Processing
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: requests<3,>=2.28
26
+ Requires-Dist: numpy<3,>=1.24
27
+ Requires-Dist: tifffile<2027,>=2023.7
28
+ Requires-Dist: imagecodecs>=2023.9
29
+ Provides-Extra: dev
30
+ Requires-Dist: pytest>=7; extra == "dev"
31
+ Requires-Dist: mypy>=1.10; extra == "dev"
32
+ Requires-Dist: types-requests; extra == "dev"
33
+ Dynamic: license-file
34
+
35
+ # scigantic-facebase
36
+
37
+ Search [FaceBase](https://www.facebase.org), the craniofacial research data hub, and read its open-access data from Python.
38
+
39
+ FaceBase holds about 1,070 open datasets: micro-CT and light-sheet imaging, ChIP-seq and RNA-seq tracks, expression tables and more, from mouse, zebrafish, human and other models. This package is a typed layer over FaceBase's open metadata API and file store. No key or login is needed.
40
+
41
+ ```
42
+ pip install scigantic-facebase
43
+ ```
44
+
45
+ ## Use
46
+
47
+ ```python
48
+ import scigantic_facebase as fb
49
+
50
+ hits = fb.search("zebrafish thyroid microCT") # newest first
51
+ ds = hits[0]
52
+ print(ds.title, ds.doi_url)
53
+ print(fb.cite(ds))
54
+
55
+ files = fb.files(ds) # size, MD5, download URL
56
+ fb.download(files[0], "data/") # MD5-checked
57
+ ```
58
+
59
+ Search matches every word against the whole record, and treats "zebrafish" and "Danio rerio", "mouse" and "Mus musculus" as the same thing.
60
+
61
+ ## Read a micro-CT scan without downloading it
62
+
63
+ Many micro-CT datasets are a zip of TIFF slices. A zip keeps its directory at the end, so one slice costs one HTTP range request.
64
+
65
+ ```python
66
+ stack = fb.microct.open_stack(files[0])
67
+ len(stack) # 596 slices
68
+ stack.voxel_size_um # 10.5, from the scan log
69
+ img = stack.read_slice(300)
70
+ vol = stack.volume(step=10) # every 10th slice, (z, y, x)
71
+ stack.bytes_fetched # a few MB of a 49 MB file
72
+ ```
73
+
74
+ `fb.microct.thyroid_atlas()` lists the zebrafish thyroid-hormone atlas: 81 head scans from 12 to 25 mm standard length, raised hypothyroid, euthyroid or hyperthyroid.
75
+
76
+ `fb.open_remote(file)` gives any FaceBase file as a seekable file object backed by range requests, which `zipfile`, `tifffile` and `h5py` accept.
77
+
78
+ ## What is not here
79
+
80
+ Protected human-subjects datasets sit behind FaceBase's data access request. Their metadata is public and `fb.search(include_protected=True)` lists them, but reading their files raises `FacebaseAccessError`.
81
+
82
+ ## Terms of use
83
+
84
+ Copyright to FaceBase data belongs to the contributing investigators. FaceBase's [Terms of Use](https://www.facebase.org/policies/tou/) ask users to cite the dataset, acknowledge FaceBase, and send other people to FaceBase instead of circulating downloaded files. `fb.cite()` produces the citation. This package fetches from FaceBase on request and does not mirror anything.
85
+
86
+ ## License
87
+
88
+ MIT-0 for this code. The data carries its own terms.
@@ -0,0 +1,13 @@
1
+ scigantic_facebase/__init__.py,sha256=Fvj_kQ_pzgbojz1RVx0AYLfnoRCs3ITyAr8vPSXcSjU,1205
2
+ scigantic_facebase/_client.py,sha256=NjyCv8iVdAzMI6ANsiaKNzbaw6uLy7d2n8ye6s6xIU8,3715
3
+ scigantic_facebase/_version.py,sha256=kUR5RAFc7HCeiqdlX36dZOHkUI5wI6V_43RpEcD8b-0,22
4
+ scigantic_facebase/catalog.py,sha256=MDuczuSTnEq5ZLu5-KxwSFZeZruvFYfjacKzmDnNrT8,5488
5
+ scigantic_facebase/microct.py,sha256=BtY0AmjHdrQMXTtV2adSgsuQwFDaMRhaYy80RcB4f3U,5253
6
+ scigantic_facebase/models.py,sha256=k3AgeraNKYhA9smIwFNx0_LnzDkgRDbxTyPEdLhHfAc,1588
7
+ scigantic_facebase/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
8
+ scigantic_facebase/remote.py,sha256=S6opLgZZpM_npBbuQ10kAkv4rrpEhsslt5PFzrWwVvY,6056
9
+ scigantic_facebase-0.1.0.dist-info/licenses/LICENSE,sha256=RiBhEtE0qZ65CeSf05VqMHWAhMh0ETETZAj4TEtscWE,904
10
+ scigantic_facebase-0.1.0.dist-info/METADATA,sha256=-KmSeI57mOf1IRgzTUM7mN_kN2aZSMqh-CowwBs4foU,4101
11
+ scigantic_facebase-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
12
+ scigantic_facebase-0.1.0.dist-info/top_level.txt,sha256=azMHiR2cPMJb0hNbkrRcjYNZHjbVhl5Bqx5dx9GqF48,19
13
+ scigantic_facebase-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,16 @@
1
+ MIT No Attribution
2
+
3
+ Copyright 2026 Scigantic
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy of this
6
+ software and associated documentation files (the "Software"), to deal in the Software
7
+ without restriction, including without limitation the rights to use, copy, modify,
8
+ merge, publish, distribute, sublicense, and/or sell copies of the Software, and to
9
+ permit persons to whom the Software is furnished to do so.
10
+
11
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED,
12
+ INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A
13
+ PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
14
+ HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF
15
+ CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE
16
+ OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
@@ -0,0 +1 @@
1
+ scigantic_facebase